parse-stack-next 5.7.3 → 5.7.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +190 -0
- data/README.md +3 -0
- data/docs/atlas_vector_search_guide.md +125 -10
- data/docs/mcp_guide.md +81 -0
- data/lib/parse/agent/log_levels.rb +11 -0
- data/lib/parse/agent/mcp_dispatcher.rb +250 -13
- data/lib/parse/agent/mcp_rack_app.rb +120 -17
- data/lib/parse/agent.rb +40 -0
- data/lib/parse/atlas_search.rb +5 -1
- data/lib/parse/client.rb +85 -12
- data/lib/parse/embeddings/cache.rb +17 -5
- data/lib/parse/embeddings/provider.rb +26 -0
- data/lib/parse/embeddings/voyage.rb +214 -15
- data/lib/parse/embeddings.rb +3 -1
- data/lib/parse/model/core/embed_managed.rb +8 -2
- data/lib/parse/mongodb.rb +49 -2
- data/lib/parse/query.rb +5 -5
- data/lib/parse/retrieval/reranker/voyage.rb +282 -0
- data/lib/parse/retrieval/reranker.rb +2 -0
- data/lib/parse/stack/version.rb +1 -1
- data/parse-stack-next.gemspec +1 -0
- metadata +3 -1
data/lib/parse/agent.rb
CHANGED
|
@@ -21,6 +21,7 @@ require_relative "agent/result_formatter"
|
|
|
21
21
|
require_relative "agent/pipeline_validator"
|
|
22
22
|
require_relative "agent/rate_limiter"
|
|
23
23
|
require_relative "agent/cancellation_token"
|
|
24
|
+
require_relative "agent/log_levels"
|
|
24
25
|
require_relative "agent/approval_gate"
|
|
25
26
|
require_relative "agent/prompt_hardening"
|
|
26
27
|
require_relative "agent/describe"
|
|
@@ -1002,6 +1003,15 @@ module Parse
|
|
|
1002
1003
|
# Numeric. `total` and `message` are optional.
|
|
1003
1004
|
attr_accessor :progress_callback
|
|
1004
1005
|
|
|
1006
|
+
# @return [#call, nil] callback that emits MCP `notifications/message`
|
|
1007
|
+
# log events. Installed per request by Parse::Agent::MCPDispatcher on
|
|
1008
|
+
# streaming transports, which drop messages below the level the
|
|
1009
|
+
# client set with `logging/setLevel`. When nil, {#log} is a no-op.
|
|
1010
|
+
# Application code should log through {#log}, not this accessor.
|
|
1011
|
+
#
|
|
1012
|
+
# The callback signature is `call(level:, data:, logger:)`.
|
|
1013
|
+
attr_accessor :log_callback
|
|
1014
|
+
|
|
1005
1015
|
# @return [Parse::Agent::CancellationToken, nil] cooperative
|
|
1006
1016
|
# cancellation token installed by Parse::Agent::MCPDispatcher around
|
|
1007
1017
|
# tool dispatch when the transport supports cancellation
|
|
@@ -1319,6 +1329,35 @@ module Parse
|
|
|
1319
1329
|
nil
|
|
1320
1330
|
end
|
|
1321
1331
|
|
|
1332
|
+
# Emit an MCP log message to the connected client. Tools (built-in or
|
|
1333
|
+
# registered through `Parse::Agent::Tools.register`) call this to
|
|
1334
|
+
# surface diagnostics without failing the call. The message reaches
|
|
1335
|
+
# the client only on a streaming transport and only when the client
|
|
1336
|
+
# asked for this level or a lower one via `logging/setLevel`;
|
|
1337
|
+
# otherwise this is a no-op.
|
|
1338
|
+
#
|
|
1339
|
+
# @param level [Symbol, String] an RFC 5424 severity: `:debug`,
|
|
1340
|
+
# `:info`, `:notice`, `:warning`, `:error`, `:critical`, `:alert`,
|
|
1341
|
+
# or `:emergency`.
|
|
1342
|
+
# @param data [Object] any JSON-serializable value. Do not include
|
|
1343
|
+
# secrets or data the agent's scope may not see.
|
|
1344
|
+
# @param logger [String, nil] optional logger name shown by the client.
|
|
1345
|
+
# @return [void]
|
|
1346
|
+
# @raise [ArgumentError] for an unknown level.
|
|
1347
|
+
def log(level, data, logger: nil)
|
|
1348
|
+
# Validate before the no-callback return so a bad level fails the
|
|
1349
|
+
# same way on every transport, not only on a streaming request.
|
|
1350
|
+
level = level.to_s
|
|
1351
|
+
unless LOG_LEVELS.include?(level)
|
|
1352
|
+
raise ArgumentError, "log level must be one of #{LOG_LEVELS.join(", ")} (got #{level.inspect})"
|
|
1353
|
+
end
|
|
1354
|
+
cb = @log_callback
|
|
1355
|
+
return if cb.nil?
|
|
1356
|
+
|
|
1357
|
+
cb.call(level: level, data: data, logger: logger)
|
|
1358
|
+
nil
|
|
1359
|
+
end
|
|
1360
|
+
|
|
1322
1361
|
# @return [Integer] total prompt tokens used across all requests
|
|
1323
1362
|
attr_reader :total_prompt_tokens
|
|
1324
1363
|
|
|
@@ -1719,6 +1758,7 @@ module Parse
|
|
|
1719
1758
|
# client is observing.
|
|
1720
1759
|
@cancellation_token = parent.cancellation_token
|
|
1721
1760
|
@progress_callback = parent.progress_callback
|
|
1761
|
+
@log_callback = parent.log_callback
|
|
1722
1762
|
|
|
1723
1763
|
# Clamp the sub-agent's permission tier at the parent's. The
|
|
1724
1764
|
# default :readonly is always ≤ any parent tier, so this fires
|
data/lib/parse/atlas_search.rb
CHANGED
|
@@ -1067,7 +1067,11 @@ module Parse
|
|
|
1067
1067
|
if (mode = Parse::MongoDB.send(:normalize_read_preference, read_preference))
|
|
1068
1068
|
coll = coll.with(read: { mode: mode })
|
|
1069
1069
|
end
|
|
1070
|
-
|
|
1070
|
+
# Same QueryPlanKilled re-run as Parse::MongoDB.aggregate (private
|
|
1071
|
+
# there too, hence `send`).
|
|
1072
|
+
Parse::MongoDB.send(:with_query_killed_retry, collection_name) do
|
|
1073
|
+
coll.aggregate(pipeline, agg_opts).to_a
|
|
1074
|
+
end
|
|
1071
1075
|
rescue => e
|
|
1072
1076
|
# `raise_if_timeout!` is module-private on Parse::MongoDB; use
|
|
1073
1077
|
# `send` so we can reuse the timeout-translation logic without
|
data/lib/parse/client.rb
CHANGED
|
@@ -1424,6 +1424,30 @@ module Parse
|
|
|
1424
1424
|
retry
|
|
1425
1425
|
end
|
|
1426
1426
|
raise
|
|
1427
|
+
rescue Faraday::ConnectionFailed => e
|
|
1428
|
+
# `Faraday::ConnectionFailed` covers two very different failures under
|
|
1429
|
+
# one class, so it is split on the wrapped cause (see
|
|
1430
|
+
# #connection_reset_error?):
|
|
1431
|
+
#
|
|
1432
|
+
# - RESET mid-flight (`Errno::ECONNRESET` / `Errno::EPIPE` /
|
|
1433
|
+
# `Errno::ECONNABORTED` / `EOFError`): the classic stale
|
|
1434
|
+
# keep-alive failure. A pooled persistent connection idled past
|
|
1435
|
+
# the server's (or an LB's) keep-alive window and was closed
|
|
1436
|
+
# remotely, and the next request on it dies at the socket.
|
|
1437
|
+
# Transient by nature (a fresh connection succeeds immediately),
|
|
1438
|
+
# so it retries under the same idempotency rules as a read
|
|
1439
|
+
# timeout: the outcome is unknown, so only idempotent requests
|
|
1440
|
+
# are re-sent.
|
|
1441
|
+
#
|
|
1442
|
+
# - REFUSED (and DNS failure): the server is down or misconfigured.
|
|
1443
|
+
# Retrying only adds backoff latency and `[Parse:Retry]` noise
|
|
1444
|
+
# before the inevitable error, so it propagates raw and fast.
|
|
1445
|
+
raise unless connection_reset_error?(e)
|
|
1446
|
+
if _retry_count > 0 && idempotent_retry?(method, body, headers)
|
|
1447
|
+
_retry_count = consume_retry_with_backoff(_retry_count, _retry_max, _request)
|
|
1448
|
+
retry
|
|
1449
|
+
end
|
|
1450
|
+
raise Parse::Error::ConnectionError, "#{_request} : #{e.class} - #{e.message}"
|
|
1427
1451
|
rescue Faraday::ClientError, Faraday::TimeoutError, Net::OpenTimeout => e
|
|
1428
1452
|
# Request timed out mid-flight: the outcome is unknown (the server may
|
|
1429
1453
|
# have received and applied the write but never answered), so only
|
|
@@ -1431,25 +1455,74 @@ module Parse
|
|
|
1431
1455
|
#
|
|
1432
1456
|
# Faraday 2.x raises `Faraday::TimeoutError` for a read timeout
|
|
1433
1457
|
# (`Timeout::Error` / `Errno::ETIMEDOUT`); it subclasses `Faraday::Error`,
|
|
1434
|
-
# not `ClientError`, so it must be listed explicitly to be caught.
|
|
1435
|
-
#
|
|
1436
|
-
#
|
|
1437
|
-
# non-transient "server down / misconfigured" failure, and auto-retrying
|
|
1438
|
-
# it only adds backoff latency before the inevitable error. Broadening to
|
|
1439
|
-
# reset connections safely (retry reset, fail fast on refused) is tracked
|
|
1440
|
-
# as a follow-up.
|
|
1458
|
+
# not `ClientError`, so it must be listed explicitly to be caught.
|
|
1459
|
+
# `Faraday::ConnectionFailed` is handled in its own rescue above,
|
|
1460
|
+
# split into retry-reset / fail-fast-refused.
|
|
1441
1461
|
if _retry_count > 0 && idempotent_retry?(method, body, headers)
|
|
1442
|
-
|
|
1443
|
-
_retry_count -= 1
|
|
1444
|
-
backoff_delay = RETRY_DELAY * (_retry_max - _retry_count)
|
|
1445
|
-
_retry_delay = backoff_delay * (0.75 + rand * 0.5)
|
|
1446
|
-
sleep _retry_delay if _retry_delay > 0
|
|
1462
|
+
_retry_count = consume_retry_with_backoff(_retry_count, _retry_max, _request)
|
|
1447
1463
|
retry
|
|
1448
1464
|
end
|
|
1449
1465
|
raise Parse::Error::ConnectionError, "#{_request} : #{e.class} - #{e.message}"
|
|
1450
1466
|
end
|
|
1451
1467
|
end
|
|
1452
1468
|
|
|
1469
|
+
# Consumes one attempt from the retry budget: logs the remaining count,
|
|
1470
|
+
# sleeps the linear backoff (RETRY_DELAY x attempt number, +/-25% jitter,
|
|
1471
|
+
# never zero), and returns the decremented budget. Shared by the
|
|
1472
|
+
# connection-reset and timeout rescue branches in {#request} so their
|
|
1473
|
+
# backoff behavior cannot drift apart; the `retry` keyword itself must
|
|
1474
|
+
# stay lexically inside each rescue clause, so it remains at the call
|
|
1475
|
+
# sites. The 429/503 branch keeps its own inline version because it also
|
|
1476
|
+
# honors a server-supplied Retry-After header.
|
|
1477
|
+
# @param retry_count [Integer] the remaining retry budget (must be > 0).
|
|
1478
|
+
# @param retry_max [Integer] the effective starting budget.
|
|
1479
|
+
# @param request [Parse::Request] the request being retried (for logging).
|
|
1480
|
+
# @return [Integer] the decremented retry budget.
|
|
1481
|
+
def consume_retry_with_backoff(retry_count, retry_max, request)
|
|
1482
|
+
warn "[Parse:Retry] Retries remaining #{retry_count} : #{request}"
|
|
1483
|
+
retry_count -= 1
|
|
1484
|
+
backoff_delay = RETRY_DELAY * (retry_max - retry_count)
|
|
1485
|
+
retry_delay = backoff_delay * (0.75 + rand * 0.5)
|
|
1486
|
+
sleep retry_delay if retry_delay > 0
|
|
1487
|
+
retry_count
|
|
1488
|
+
end
|
|
1489
|
+
|
|
1490
|
+
# The wrapped causes that mark a `Faraday::ConnectionFailed` as a RESET
|
|
1491
|
+
# connection (transient, retry-safe for idempotent requests) rather than a
|
|
1492
|
+
# REFUSED one (server down, fail fast). `EOFError` is what net/http raises
|
|
1493
|
+
# when the remote end closes a keep-alive socket cleanly between requests;
|
|
1494
|
+
# ECONNRESET/EPIPE/ECONNABORTED are the unclean variants.
|
|
1495
|
+
# @!visibility private
|
|
1496
|
+
CONNECTION_RESET_CAUSES = [
|
|
1497
|
+
Errno::ECONNRESET, Errno::EPIPE, Errno::ECONNABORTED, EOFError,
|
|
1498
|
+
].freeze
|
|
1499
|
+
|
|
1500
|
+
# Message fallback for adapters that raise `Faraday::ConnectionFailed`
|
|
1501
|
+
# with the cause flattened into the message instead of wrapped.
|
|
1502
|
+
# @!visibility private
|
|
1503
|
+
CONNECTION_RESET_MESSAGE = /connection reset|broken pipe|end of file reached/i
|
|
1504
|
+
|
|
1505
|
+
# Whether a `Faraday::ConnectionFailed` was caused by a reset/dropped
|
|
1506
|
+
# connection (retryable) as opposed to connection-refused or a DNS
|
|
1507
|
+
# failure (fail fast). Walks the wrapped exception and the `#cause`
|
|
1508
|
+
# chain looking for a reset-class error.
|
|
1509
|
+
# @param error [Exception] the rescued `Faraday::ConnectionFailed`.
|
|
1510
|
+
# @return [Boolean]
|
|
1511
|
+
def connection_reset_error?(error)
|
|
1512
|
+
inner = error.respond_to?(:wrapped_exception) ? error.wrapped_exception : nil
|
|
1513
|
+
inner ||= error.cause
|
|
1514
|
+
seen = 0
|
|
1515
|
+
while inner && seen < 8
|
|
1516
|
+
return true if CONNECTION_RESET_CAUSES.any? { |klass| inner.is_a?(klass) }
|
|
1517
|
+
inner = inner.cause
|
|
1518
|
+
seen += 1
|
|
1519
|
+
end
|
|
1520
|
+
CONNECTION_RESET_MESSAGE.match?(error.message.to_s)
|
|
1521
|
+
end
|
|
1522
|
+
|
|
1523
|
+
private :consume_retry_with_backoff, :connection_reset_error?
|
|
1524
|
+
private_constant :CONNECTION_RESET_CAUSES, :CONNECTION_RESET_MESSAGE
|
|
1525
|
+
|
|
1453
1526
|
# Whether a request whose outcome is UNKNOWN (a 500/503 or a dropped
|
|
1454
1527
|
# connection) is safe to transparently re-send.
|
|
1455
1528
|
#
|
|
@@ -8,7 +8,7 @@ require "json"
|
|
|
8
8
|
module Parse
|
|
9
9
|
module Embeddings
|
|
10
10
|
# Process-local embedding cache keyed by
|
|
11
|
-
# `(provider, model, input_type, input_hash)`.
|
|
11
|
+
# `(provider, model, dimensions, input_type, deployment, input_hash)`.
|
|
12
12
|
#
|
|
13
13
|
# Query-side embedding is the hot repeat path: the same natural-
|
|
14
14
|
# language query (an agent retrying a tool call, a user paging
|
|
@@ -32,7 +32,10 @@ module Parse
|
|
|
32
32
|
#
|
|
33
33
|
# == Key derivation
|
|
34
34
|
#
|
|
35
|
-
# `provider.class.name | model_name | input_type |
|
|
35
|
+
# `provider.class.name | model_name | dimensions | input_type |
|
|
36
|
+
# cache_identity | SHA-256(input)`. `cache_identity` (the provider's
|
|
37
|
+
# endpoint without credentials, see {Provider#cache_identity}) is
|
|
38
|
+
# omitted for providers that have none.
|
|
36
39
|
# The full input text never becomes part of the key, so a shared
|
|
37
40
|
# external store does not accumulate plaintext queries.
|
|
38
41
|
#
|
|
@@ -305,8 +308,8 @@ module Parse
|
|
|
305
308
|
# @!visibility private
|
|
306
309
|
# Composite cache key. The input is hashed so plaintext never
|
|
307
310
|
# lands in a shared store; provider identity + model + dimensions
|
|
308
|
-
# + input_type namespace the hash (two models'
|
|
309
|
-
# confused). Dimensions matter independently of the model name:
|
|
311
|
+
# + input_type + deployment namespace the hash (two models' or two
|
|
312
|
+
# deployments' vectors are never confused). Dimensions matter independently of the model name:
|
|
310
313
|
# Matryoshka-capable providers (OpenAI text-embedding-3-*, Cohere
|
|
311
314
|
# embed-v4, Voyage, Jina, Qwen) can register the same model at
|
|
312
315
|
# different output widths, and serving one width's cached vector
|
|
@@ -322,7 +325,16 @@ module Parse
|
|
|
322
325
|
rescue NotImplementedError
|
|
323
326
|
"unknown"
|
|
324
327
|
end
|
|
325
|
-
|
|
328
|
+
# Deployment identity (endpoint, never credentials) separates two
|
|
329
|
+
# deployments of the same model. Omitted when the provider has
|
|
330
|
+
# none, so such providers keep their existing keys.
|
|
331
|
+
identity = begin
|
|
332
|
+
provider.respond_to?(:cache_identity) ? provider.cache_identity : nil
|
|
333
|
+
rescue StandardError
|
|
334
|
+
nil
|
|
335
|
+
end
|
|
336
|
+
deployment = identity.nil? || identity.to_s.empty? ? "" : "#{identity}|"
|
|
337
|
+
"#{provider.class.name}|#{model}|#{dims}|#{input_type}|#{deployment}#{Digest::SHA256.hexdigest(input.to_s)}"
|
|
326
338
|
end
|
|
327
339
|
|
|
328
340
|
# @!visibility private
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
# encoding: UTF-8
|
|
2
2
|
# frozen_string_literal: true
|
|
3
3
|
|
|
4
|
+
require "uri"
|
|
5
|
+
|
|
4
6
|
module Parse
|
|
5
7
|
module Embeddings
|
|
6
8
|
# Abstract base class for embedding providers. Concrete subclasses
|
|
@@ -242,6 +244,30 @@ module Parse
|
|
|
242
244
|
|
|
243
245
|
# @return [Hash] attributes safe to surface in {#inspect}. Override
|
|
244
246
|
# in subclasses to add fields; never add credentials.
|
|
247
|
+
# Identity of the deployment this provider talks to, for use in cache
|
|
248
|
+
# keys: the endpoint's scheme, host, non-default port, and path. Never
|
|
249
|
+
# includes credentials, userinfo, or a query string. Two providers of
|
|
250
|
+
# the same class and model pointed at different deployments (two
|
|
251
|
+
# self-hosted servers, or a provider behind a proxy) can return
|
|
252
|
+
# different vectors for the same input, so their cache entries must
|
|
253
|
+
# not be shared.
|
|
254
|
+
#
|
|
255
|
+
# The default reads `@base_url`, which every built-in HTTP provider
|
|
256
|
+
# sets. A provider with no configurable endpoint returns nil, which
|
|
257
|
+
# leaves its cache key unchanged. Override to supply another identity.
|
|
258
|
+
#
|
|
259
|
+
# @return [String, nil]
|
|
260
|
+
def cache_identity
|
|
261
|
+
url = instance_variable_defined?(:@base_url) ? @base_url : nil
|
|
262
|
+
return nil if url.nil? || url.to_s.empty?
|
|
263
|
+
uri = URI.parse(url.to_s)
|
|
264
|
+
return nil if uri.host.nil? || uri.host.empty?
|
|
265
|
+
port = uri.port && uri.port != uri.default_port ? ":#{uri.port}" : ""
|
|
266
|
+
"#{uri.scheme}://#{uri.host.downcase}#{port}#{uri.path.to_s.chomp("/")}"
|
|
267
|
+
rescue URI::InvalidURIError
|
|
268
|
+
nil
|
|
269
|
+
end
|
|
270
|
+
|
|
245
271
|
def inspect_attrs
|
|
246
272
|
out = {}
|
|
247
273
|
out[:model] = safe_call(:model_name)
|
|
@@ -9,9 +9,11 @@ require_relative "provider"
|
|
|
9
9
|
module Parse
|
|
10
10
|
module Embeddings
|
|
11
11
|
# Voyage AI embeddings provider. Wraps `POST /v1/embeddings` for
|
|
12
|
-
# text-only models
|
|
12
|
+
# text-only models, `POST /v1/multimodalembeddings` for the
|
|
13
13
|
# multimodal text+image models (text via {#embed_text}, images via
|
|
14
|
-
# {#embed_image})
|
|
14
|
+
# {#embed_image}), and `POST /v1/contextualizedembeddings` for the
|
|
15
|
+
# contextualized chunk models (single texts via {#embed_text},
|
|
16
|
+
# whole chunked documents via {#embed_chunks}).
|
|
15
17
|
#
|
|
16
18
|
# Supported models:
|
|
17
19
|
#
|
|
@@ -21,7 +23,8 @@ module Parse
|
|
|
21
23
|
# llama.cpp).
|
|
22
24
|
# * **v3 family** — `voyage-3-large`, `voyage-3.5`,
|
|
23
25
|
# `voyage-3.5-lite`, `voyage-3`, `voyage-3-lite`.
|
|
24
|
-
# * **code models
|
|
26
|
+
# * **code models**: `voyage-code-4`, `voyage-code-3`, and
|
|
27
|
+
# `voyage-code-2` (the only 1536-dim one).
|
|
25
28
|
# * **domain models** — `voyage-finance-2`, `voyage-law-2`.
|
|
26
29
|
# * **multimodal** — `voyage-multimodal-3` (text+image) and
|
|
27
30
|
# `voyage-multimodal-3.5` (text+image+video). Unified vector
|
|
@@ -31,6 +34,14 @@ module Parse
|
|
|
31
34
|
# {#embed_image}, video through {#embed_video}. All three share
|
|
32
35
|
# the same space, so stored text vectors are comparable against
|
|
33
36
|
# image and video vectors without re-embedding.
|
|
37
|
+
# * **contextualized chunk**: `voyage-context-4` and
|
|
38
|
+
# `voyage-context-3`. Each chunk's vector also encodes the
|
|
39
|
+
# document it came from, so chunks are embedded a document at a
|
|
40
|
+
# time through {#embed_chunks}. {#embed_text} sends every string
|
|
41
|
+
# as a one-chunk document. That is the right shape for queries, and it
|
|
42
|
+
# is also what the `embed` macro and {BatchEmbedder} send for stored
|
|
43
|
+
# fields, so those vectors carry no surrounding-document context;
|
|
44
|
+
# call {#embed_chunks} directly for chunk-level context.
|
|
34
45
|
#
|
|
35
46
|
# Audio is not offered by any Voyage model, and neither PDF nor
|
|
36
47
|
# DOCX is accepted as a content type — render document pages to
|
|
@@ -127,6 +138,13 @@ module Parse
|
|
|
127
138
|
DEFAULT_BATCH_SIZE = 128
|
|
128
139
|
MAX_RESPONSE_BYTES = 16 * 1024 * 1024
|
|
129
140
|
|
|
141
|
+
# Upper bounds on the JSON size of one returned vector, used to plan
|
|
142
|
+
# contextualized requests: a float such as `-1.2345678901234567e-05,`
|
|
143
|
+
# is under 32 bytes, and the `{"object":"embedding","embedding":[],
|
|
144
|
+
# "index":N}` wrapper is under 256.
|
|
145
|
+
RESPONSE_BYTES_PER_VALUE = 32
|
|
146
|
+
RESPONSE_BYTES_PER_VECTOR_ENVELOPE = 256
|
|
147
|
+
|
|
130
148
|
# Default (native) vector width per model — the width returned
|
|
131
149
|
# when `output_dimension` is omitted from the request.
|
|
132
150
|
#
|
|
@@ -147,12 +165,15 @@ module Parse
|
|
|
147
165
|
"voyage-3.5-lite" => 1024,
|
|
148
166
|
"voyage-3" => 1024,
|
|
149
167
|
"voyage-3-lite" => 512,
|
|
168
|
+
"voyage-code-4" => 1024,
|
|
150
169
|
"voyage-code-3" => 1024,
|
|
151
170
|
"voyage-code-2" => 1536,
|
|
152
171
|
"voyage-finance-2" => 1024,
|
|
153
172
|
"voyage-law-2" => 1024,
|
|
154
173
|
"voyage-multimodal-3" => 1024,
|
|
155
174
|
"voyage-multimodal-3.5" => 1024,
|
|
175
|
+
"voyage-context-4" => 1024,
|
|
176
|
+
"voyage-context-3" => 1024,
|
|
156
177
|
}.freeze
|
|
157
178
|
|
|
158
179
|
# Every width a model's Matryoshka head will actually return.
|
|
@@ -174,12 +195,15 @@ module Parse
|
|
|
174
195
|
"voyage-3.5-lite" => [256, 512, 1024, 2048],
|
|
175
196
|
"voyage-3" => [1024],
|
|
176
197
|
"voyage-3-lite" => [512],
|
|
198
|
+
"voyage-code-4" => [256, 512, 1024, 2048],
|
|
177
199
|
"voyage-code-3" => [256, 512, 1024, 2048],
|
|
178
200
|
"voyage-code-2" => [1536],
|
|
179
201
|
"voyage-finance-2" => [1024],
|
|
180
202
|
"voyage-law-2" => [1024],
|
|
181
203
|
"voyage-multimodal-3" => [1024],
|
|
182
204
|
"voyage-multimodal-3.5" => [256, 512, 1024, 2048],
|
|
205
|
+
"voyage-context-4" => [256, 512, 1024, 2048],
|
|
206
|
+
"voyage-context-3" => [256, 512, 1024, 2048],
|
|
183
207
|
}.freeze
|
|
184
208
|
|
|
185
209
|
# Back-compat alias: the set of models accepting any
|
|
@@ -198,12 +222,17 @@ module Parse
|
|
|
198
222
|
"voyage-3.5-lite" => 32_000,
|
|
199
223
|
"voyage-3" => 32_000,
|
|
200
224
|
"voyage-3-lite" => 32_000,
|
|
225
|
+
"voyage-code-4" => 32_000,
|
|
201
226
|
"voyage-code-3" => 32_000,
|
|
202
227
|
"voyage-code-2" => 16_000,
|
|
203
228
|
"voyage-finance-2" => 32_000,
|
|
204
229
|
"voyage-law-2" => 16_000,
|
|
205
230
|
"voyage-multimodal-3" => 32_000,
|
|
206
231
|
"voyage-multimodal-3.5" => 32_000,
|
|
232
|
+
# Per document (one inner list of chunks). The request as a
|
|
233
|
+
# whole is capped at 120k tokens across every document.
|
|
234
|
+
"voyage-context-4" => 32_000,
|
|
235
|
+
"voyage-context-3" => 32_000,
|
|
207
236
|
}.freeze
|
|
208
237
|
|
|
209
238
|
# Models that route to `/v1/multimodalembeddings` with the
|
|
@@ -226,6 +255,27 @@ module Parse
|
|
|
226
255
|
# "does not support video inputs" 400.
|
|
227
256
|
VIDEO_MODELS = %w[voyage-multimodal-3.5].freeze
|
|
228
257
|
|
|
258
|
+
# Models that route to `/v1/contextualizedembeddings` with the
|
|
259
|
+
# `{ inputs: [[chunk, ...], ...] }` envelope: one inner list per
|
|
260
|
+
# document, each chunk embedded with the rest of its document as
|
|
261
|
+
# context. The endpoint has no `truncation` field, so it is never
|
|
262
|
+
# sent for these models.
|
|
263
|
+
CONTEXTUALIZED_MODELS = %w[voyage-context-4 voyage-context-3].freeze
|
|
264
|
+
|
|
265
|
+
# Voyage's per-request limits for the contextualized endpoint:
|
|
266
|
+
# at most this many documents, and this many chunks summed
|
|
267
|
+
# across them.
|
|
268
|
+
# The endpoint also caps a request at 120k tokens summed across
|
|
269
|
+
# every document, which the SDK cannot check without a tokenizer.
|
|
270
|
+
MAX_CONTEXT_DOCUMENTS = 1_000
|
|
271
|
+
MAX_CONTEXT_CHUNKS = 16_000
|
|
272
|
+
|
|
273
|
+
# Default `embed_batch_size` for {CONTEXTUALIZED_MODELS}. Each
|
|
274
|
+
# string is a whole document there, so 128 paragraph-sized inputs
|
|
275
|
+
# would overrun the 120k-token request cap; 32 leaves room for
|
|
276
|
+
# documents averaging under about 3,700 tokens.
|
|
277
|
+
CONTEXT_DEFAULT_BATCH_SIZE = 32
|
|
278
|
+
|
|
229
279
|
# Models Voyage's hosted API serves but the Atlas Embedding and
|
|
230
280
|
# Reranking API does not. Verified against both endpoints.
|
|
231
281
|
ATLAS_UNAVAILABLE_MODELS = %w[voyage-3 voyage-3-lite].freeze
|
|
@@ -271,11 +321,13 @@ module Parse
|
|
|
271
321
|
# @param timeout [Integer] read timeout, seconds.
|
|
272
322
|
# @param open_timeout [Integer] connect timeout, seconds.
|
|
273
323
|
# @param max_retries [Integer] retry attempts on 429/5xx/timeouts.
|
|
274
|
-
# @param embed_batch_size [Integer] inputs per request (max 128).
|
|
324
|
+
# @param embed_batch_size [Integer, nil] inputs per request (max 128).
|
|
325
|
+
# Defaults to {DEFAULT_BATCH_SIZE}, or {CONTEXT_DEFAULT_BATCH_SIZE}
|
|
326
|
+
# for a contextualized model.
|
|
275
327
|
# @param dimensions [Integer, nil] override output width via
|
|
276
|
-
# Voyage's `output_dimension` Matryoshka parameter.
|
|
277
|
-
#
|
|
278
|
-
#
|
|
328
|
+
# Voyage's `output_dimension` Matryoshka parameter. Must be one of
|
|
329
|
+
# the model's {MODEL_SUPPORTED_DIMENSIONS}; a model with a single
|
|
330
|
+
# supported width accepts only that width (or nil).
|
|
279
331
|
# @param truncation [Boolean] forward Voyage's `truncation:` field.
|
|
280
332
|
# Defaults `true` to match Voyage's API default. Set `false` to
|
|
281
333
|
# force the API to reject over-length inputs rather than silently
|
|
@@ -292,7 +344,7 @@ module Parse
|
|
|
292
344
|
timeout: DEFAULT_TIMEOUT,
|
|
293
345
|
open_timeout: DEFAULT_OPEN_TIMEOUT,
|
|
294
346
|
max_retries: DEFAULT_MAX_RETRIES,
|
|
295
|
-
embed_batch_size:
|
|
347
|
+
embed_batch_size: nil,
|
|
296
348
|
dimensions: nil,
|
|
297
349
|
truncation: true,
|
|
298
350
|
allow_faraday_proxy: false,
|
|
@@ -308,6 +360,7 @@ module Parse
|
|
|
308
360
|
validate_positive_integer!(:timeout, timeout)
|
|
309
361
|
validate_positive_integer!(:open_timeout, open_timeout)
|
|
310
362
|
validate_non_negative_integer!(:max_retries, max_retries)
|
|
363
|
+
embed_batch_size ||= CONTEXTUALIZED_MODELS.include?(model) ? CONTEXT_DEFAULT_BATCH_SIZE : DEFAULT_BATCH_SIZE
|
|
311
364
|
validate_positive_integer!(:embed_batch_size, embed_batch_size)
|
|
312
365
|
if embed_batch_size > 128
|
|
313
366
|
raise ArgumentError,
|
|
@@ -396,10 +449,22 @@ module Parse
|
|
|
396
449
|
end
|
|
397
450
|
wire_input_type = INPUT_TYPE_WIRE_VALUES[input_type]
|
|
398
451
|
|
|
452
|
+
if CONTEXTUALIZED_MODELS.include?(@model)
|
|
453
|
+
if strings.length > MAX_CONTEXT_DOCUMENTS
|
|
454
|
+
raise ArgumentError,
|
|
455
|
+
"Parse::Embeddings::Voyage#embed_text: #{strings.length} inputs exceeds Voyage's " \
|
|
456
|
+
"per-request cap for #{@model} (#{MAX_CONTEXT_DOCUMENTS}). Split the input."
|
|
457
|
+
end
|
|
458
|
+
# Each string is its own one-chunk document, so the response
|
|
459
|
+
# carries exactly one vector per document.
|
|
460
|
+
return embed_contextualized(strings.map { |s| [s] }, input_type, wire_input_type).map(&:first)
|
|
461
|
+
end
|
|
462
|
+
|
|
399
463
|
# Multimodal models route to a different endpoint with a
|
|
400
464
|
# different request envelope. The response envelope shape is
|
|
401
465
|
# the same (`{ data: [{ embedding, index }], usage: {...} }`)
|
|
402
466
|
# so `extract_vectors!` is reused as-is.
|
|
467
|
+
|
|
403
468
|
body = if MULTIMODAL_MODELS.include?(@model)
|
|
404
469
|
build_multimodal_body(strings, wire_input_type)
|
|
405
470
|
else
|
|
@@ -500,6 +565,72 @@ module Parse
|
|
|
500
565
|
allow_insecure: allow_insecure)
|
|
501
566
|
end
|
|
502
567
|
|
|
568
|
+
# Embed chunked documents through Voyage's
|
|
569
|
+
# `/v1/contextualizedembeddings` endpoint. Every chunk's vector
|
|
570
|
+
# encodes the surrounding document as well as the chunk itself,
|
|
571
|
+
# so pass the chunks of one document together, in order, rather
|
|
572
|
+
# than one call per chunk.
|
|
573
|
+
#
|
|
574
|
+
# **Contextualized model required.** Only {CONTEXTUALIZED_MODELS}
|
|
575
|
+
# accept this shape; any other model raises {BadRequestError}
|
|
576
|
+
# before any network call.
|
|
577
|
+
#
|
|
578
|
+
# @param documents [Array<Array<String>>] one inner Array per
|
|
579
|
+
# document, holding that document's chunks in order. At most
|
|
580
|
+
# {MAX_CONTEXT_DOCUMENTS} documents and {MAX_CONTEXT_CHUNKS}
|
|
581
|
+
# chunks in total.
|
|
582
|
+
# @param input_type [Symbol] one of {INPUT_TYPE_WIRE_VALUES}'s keys.
|
|
583
|
+
# @return [Array<Array<Array<Float>>>] one Array of chunk vectors
|
|
584
|
+
# per document, aligned 1:1 with `documents` and with each
|
|
585
|
+
# document's chunks.
|
|
586
|
+
def embed_chunks(documents, input_type: :search_document)
|
|
587
|
+
unless CONTEXTUALIZED_MODELS.include?(@model)
|
|
588
|
+
raise BadRequestError,
|
|
589
|
+
"Parse::Embeddings::Voyage#embed_chunks: model #{@model.inspect} does not " \
|
|
590
|
+
"accept chunked documents. Configure the provider with a contextualized model " \
|
|
591
|
+
"(supported: #{CONTEXTUALIZED_MODELS.inspect})."
|
|
592
|
+
end
|
|
593
|
+
unless documents.is_a?(Array)
|
|
594
|
+
raise ArgumentError,
|
|
595
|
+
"Parse::Embeddings::Voyage#embed_chunks expects Array<Array<String>> " \
|
|
596
|
+
"(got #{documents.class})."
|
|
597
|
+
end
|
|
598
|
+
return [] if documents.empty?
|
|
599
|
+
|
|
600
|
+
documents.each_with_index do |chunks, i|
|
|
601
|
+
unless chunks.is_a?(Array) && !chunks.empty?
|
|
602
|
+
raise ArgumentError,
|
|
603
|
+
"Parse::Embeddings::Voyage#embed_chunks documents[#{i}] must be a non-empty " \
|
|
604
|
+
"Array of chunk Strings."
|
|
605
|
+
end
|
|
606
|
+
chunks.each_with_index do |c, j|
|
|
607
|
+
unless c.is_a?(String) && !c.empty?
|
|
608
|
+
raise ArgumentError,
|
|
609
|
+
"Parse::Embeddings::Voyage#embed_chunks documents[#{i}][#{j}] must be a " \
|
|
610
|
+
"non-empty String."
|
|
611
|
+
end
|
|
612
|
+
end
|
|
613
|
+
end
|
|
614
|
+
if documents.length > MAX_CONTEXT_DOCUMENTS
|
|
615
|
+
raise ArgumentError,
|
|
616
|
+
"Parse::Embeddings::Voyage#embed_chunks: #{documents.length} documents exceeds " \
|
|
617
|
+
"Voyage's per-request cap (#{MAX_CONTEXT_DOCUMENTS}). Split the input."
|
|
618
|
+
end
|
|
619
|
+
chunk_total = documents.sum(&:length)
|
|
620
|
+
if chunk_total > MAX_CONTEXT_CHUNKS
|
|
621
|
+
raise ArgumentError,
|
|
622
|
+
"Parse::Embeddings::Voyage#embed_chunks: #{chunk_total} chunks exceeds " \
|
|
623
|
+
"Voyage's per-request cap (#{MAX_CONTEXT_CHUNKS}). Split the input."
|
|
624
|
+
end
|
|
625
|
+
unless INPUT_TYPE_WIRE_VALUES.key?(input_type)
|
|
626
|
+
raise ArgumentError,
|
|
627
|
+
"Parse::Embeddings::Voyage#embed_chunks input_type #{input_type.inspect} not in " \
|
|
628
|
+
"#{INPUT_TYPE_WIRE_VALUES.keys.inspect}."
|
|
629
|
+
end
|
|
630
|
+
|
|
631
|
+
embed_contextualized(documents, input_type, INPUT_TYPE_WIRE_VALUES[input_type])
|
|
632
|
+
end
|
|
633
|
+
|
|
503
634
|
def inspect_attrs
|
|
504
635
|
super.merge(base: safe_base_host, endpoint: @endpoint, retries: @max_retries)
|
|
505
636
|
end
|
|
@@ -561,6 +692,68 @@ module Parse
|
|
|
561
692
|
body
|
|
562
693
|
end
|
|
563
694
|
|
|
695
|
+
# Build the wire body for `/v1/contextualizedembeddings`. The
|
|
696
|
+
# endpoint documents no `truncation` field, so none is sent.
|
|
697
|
+
def build_contextualized_body(documents, wire_input_type)
|
|
698
|
+
body = { inputs: documents, model: @model }
|
|
699
|
+
body[:input_type] = wire_input_type if wire_input_type
|
|
700
|
+
apply_output_dimension!(body)
|
|
701
|
+
body
|
|
702
|
+
end
|
|
703
|
+
|
|
704
|
+
# Embed documents through the contextualized endpoint and return one
|
|
705
|
+
# Array of chunk vectors per document, in input order.
|
|
706
|
+
#
|
|
707
|
+
# Voyage accepts up to {MAX_CONTEXT_CHUNKS} chunks per request, but
|
|
708
|
+
# that many vectors serialize to far more than {MAX_RESPONSE_BYTES}.
|
|
709
|
+
# Whole documents are therefore grouped so each response's estimated
|
|
710
|
+
# size stays within the cap, and the groups are sent in turn. A
|
|
711
|
+
# single document too large for the cap on its own is sent alone
|
|
712
|
+
# with a response allowance sized to its chunk count.
|
|
713
|
+
def embed_contextualized(documents, input_type, wire_input_type)
|
|
714
|
+
per_vector = response_bytes_per_vector
|
|
715
|
+
chunks_per_request = [MAX_RESPONSE_BYTES / per_vector, 1].max
|
|
716
|
+
groups = []
|
|
717
|
+
documents.each do |doc|
|
|
718
|
+
last = groups.last
|
|
719
|
+
if last && last.sum(&:length) + doc.length <= chunks_per_request
|
|
720
|
+
last << doc
|
|
721
|
+
else
|
|
722
|
+
groups << [doc]
|
|
723
|
+
end
|
|
724
|
+
end
|
|
725
|
+
groups.flat_map { |group| embed_contextualized_request(group, input_type, wire_input_type) }
|
|
726
|
+
end
|
|
727
|
+
|
|
728
|
+
# @return [Integer] the planning estimate for one vector's JSON size.
|
|
729
|
+
def response_bytes_per_vector
|
|
730
|
+
(@dimensions * RESPONSE_BYTES_PER_VALUE) + RESPONSE_BYTES_PER_VECTOR_ENVELOPE
|
|
731
|
+
end
|
|
732
|
+
|
|
733
|
+
# Issue one contextualized request and return one Array of chunk
|
|
734
|
+
# vectors per document.
|
|
735
|
+
def embed_contextualized_request(documents, input_type, wire_input_type)
|
|
736
|
+
body = build_contextualized_body(documents, wire_input_type)
|
|
737
|
+
chunk_count = documents.sum(&:length)
|
|
738
|
+
allowance = [MAX_RESPONSE_BYTES, (chunk_count * response_bytes_per_vector) + 65_536].max
|
|
739
|
+
|
|
740
|
+
instrument_embed(chunk_count, input_type) do |emit_payload|
|
|
741
|
+
payload = post_embeddings(body, path: "contextualizedembeddings", max_response_bytes: allowance)
|
|
742
|
+
if payload.is_a?(Hash) && payload["usage"].is_a?(Hash)
|
|
743
|
+
tt = payload["usage"]["total_tokens"]
|
|
744
|
+
emit_payload[:total_tokens] = tt if tt.is_a?(Integer) && tt >= 0
|
|
745
|
+
end
|
|
746
|
+
# The response nests the standard envelope: the outer `data`
|
|
747
|
+
# holds one `{ data: [...], index: }` list per document, and
|
|
748
|
+
# each inner list is shaped like a `/v1/embeddings` response.
|
|
749
|
+
per_document = extract_vectors!(payload, documents.length, value_key: "data")
|
|
750
|
+
per_document.each_with_index.map do |entry, i|
|
|
751
|
+
vectors = extract_vectors!({ "data" => entry }, documents[i].length)
|
|
752
|
+
validate_response!(documents[i].length, vectors)
|
|
753
|
+
end
|
|
754
|
+
end
|
|
755
|
+
end
|
|
756
|
+
|
|
564
757
|
# Forward `output_dimension` only when the configured width
|
|
565
758
|
# differs from the model's native default. Sending it to a model
|
|
566
759
|
# with a single supported width is a 400, and sending the native
|
|
@@ -792,7 +985,9 @@ module Parse
|
|
|
792
985
|
encoded[0...-1]
|
|
793
986
|
end
|
|
794
987
|
|
|
795
|
-
|
|
988
|
+
# @param max_response_bytes [Integer] refuse a success body larger than
|
|
989
|
+
# this. Defaults to {MAX_RESPONSE_BYTES}.
|
|
990
|
+
def post_embeddings(body, path: "embeddings", max_response_bytes: MAX_RESPONSE_BYTES)
|
|
796
991
|
attempts = 0
|
|
797
992
|
loop do
|
|
798
993
|
attempts += 1
|
|
@@ -820,7 +1015,7 @@ module Parse
|
|
|
820
1015
|
end
|
|
821
1016
|
|
|
822
1017
|
status = response.status
|
|
823
|
-
return parse_json_body!(response.body) if status >= 200 && status < 300
|
|
1018
|
+
return parse_json_body!(response.body, max_response_bytes) if status >= 200 && status < 300
|
|
824
1019
|
|
|
825
1020
|
if status == 401
|
|
826
1021
|
raise AuthenticationError,
|
|
@@ -847,11 +1042,11 @@ module Parse
|
|
|
847
1042
|
end
|
|
848
1043
|
end
|
|
849
1044
|
|
|
850
|
-
def parse_json_body!(body)
|
|
1045
|
+
def parse_json_body!(body, max_bytes = MAX_RESPONSE_BYTES)
|
|
851
1046
|
s = body.to_s
|
|
852
|
-
if s.bytesize >
|
|
1047
|
+
if s.bytesize > max_bytes
|
|
853
1048
|
raise InvalidResponseError,
|
|
854
|
-
"Parse::Embeddings::Voyage: response body exceeds #{
|
|
1049
|
+
"Parse::Embeddings::Voyage: response body exceeds #{max_bytes} bytes " \
|
|
855
1050
|
"(#{s.bytesize}). Refusing to parse."
|
|
856
1051
|
end
|
|
857
1052
|
JSON.parse(s, max_nesting: 32)
|
|
@@ -871,7 +1066,11 @@ module Parse
|
|
|
871
1066
|
# "model": "voyage-3",
|
|
872
1067
|
# "usage": { "total_tokens": N }
|
|
873
1068
|
# }
|
|
874
|
-
|
|
1069
|
+
#
|
|
1070
|
+
# `value_key` names the field read from each entry: `"embedding"`
|
|
1071
|
+
# for the flat envelope, `"data"` for the outer list of the
|
|
1072
|
+
# contextualized envelope.
|
|
1073
|
+
def extract_vectors!(payload, input_count, value_key: "embedding")
|
|
875
1074
|
unless payload.is_a?(Hash)
|
|
876
1075
|
raise InvalidResponseError,
|
|
877
1076
|
"Parse::Embeddings::Voyage: response body is not a JSON object."
|
|
@@ -895,7 +1094,7 @@ module Parse
|
|
|
895
1094
|
raise InvalidResponseError,
|
|
896
1095
|
"Parse::Embeddings::Voyage: response.data[#{i}].index #{idx.inspect} out of range."
|
|
897
1096
|
end
|
|
898
|
-
[idx, entry[
|
|
1097
|
+
[idx, entry[value_key]]
|
|
899
1098
|
end
|
|
900
1099
|
indices = sorted.map(&:first)
|
|
901
1100
|
if indices.uniq.length != indices.length
|