parse-stack-next 5.7.3 → 5.7.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/lib/parse/agent.rb CHANGED
@@ -21,6 +21,7 @@ require_relative "agent/result_formatter"
21
21
  require_relative "agent/pipeline_validator"
22
22
  require_relative "agent/rate_limiter"
23
23
  require_relative "agent/cancellation_token"
24
+ require_relative "agent/log_levels"
24
25
  require_relative "agent/approval_gate"
25
26
  require_relative "agent/prompt_hardening"
26
27
  require_relative "agent/describe"
@@ -1002,6 +1003,15 @@ module Parse
1002
1003
  # Numeric. `total` and `message` are optional.
1003
1004
  attr_accessor :progress_callback
1004
1005
 
1006
+ # @return [#call, nil] callback that emits MCP `notifications/message`
1007
+ # log events. Installed per request by Parse::Agent::MCPDispatcher on
1008
+ # streaming transports, which drop messages below the level the
1009
+ # client set with `logging/setLevel`. When nil, {#log} is a no-op.
1010
+ # Application code should log through {#log}, not this accessor.
1011
+ #
1012
+ # The callback signature is `call(level:, data:, logger:)`.
1013
+ attr_accessor :log_callback
1014
+
1005
1015
  # @return [Parse::Agent::CancellationToken, nil] cooperative
1006
1016
  # cancellation token installed by Parse::Agent::MCPDispatcher around
1007
1017
  # tool dispatch when the transport supports cancellation
@@ -1319,6 +1329,35 @@ module Parse
1319
1329
  nil
1320
1330
  end
1321
1331
 
1332
+ # Emit an MCP log message to the connected client. Tools (built-in or
1333
+ # registered through `Parse::Agent::Tools.register`) call this to
1334
+ # surface diagnostics without failing the call. The message reaches
1335
+ # the client only on a streaming transport and only when the client
1336
+ # asked for this level or a lower one via `logging/setLevel`;
1337
+ # otherwise this is a no-op.
1338
+ #
1339
+ # @param level [Symbol, String] an RFC 5424 severity: `:debug`,
1340
+ # `:info`, `:notice`, `:warning`, `:error`, `:critical`, `:alert`,
1341
+ # or `:emergency`.
1342
+ # @param data [Object] any JSON-serializable value. Do not include
1343
+ # secrets or data the agent's scope may not see.
1344
+ # @param logger [String, nil] optional logger name shown by the client.
1345
+ # @return [void]
1346
+ # @raise [ArgumentError] for an unknown level.
1347
+ def log(level, data, logger: nil)
1348
+ # Validate before the no-callback return so a bad level fails the
1349
+ # same way on every transport, not only on a streaming request.
1350
+ level = level.to_s
1351
+ unless LOG_LEVELS.include?(level)
1352
+ raise ArgumentError, "log level must be one of #{LOG_LEVELS.join(", ")} (got #{level.inspect})"
1353
+ end
1354
+ cb = @log_callback
1355
+ return if cb.nil?
1356
+
1357
+ cb.call(level: level, data: data, logger: logger)
1358
+ nil
1359
+ end
1360
+
1322
1361
  # @return [Integer] total prompt tokens used across all requests
1323
1362
  attr_reader :total_prompt_tokens
1324
1363
 
@@ -1719,6 +1758,7 @@ module Parse
1719
1758
  # client is observing.
1720
1759
  @cancellation_token = parent.cancellation_token
1721
1760
  @progress_callback = parent.progress_callback
1761
+ @log_callback = parent.log_callback
1722
1762
 
1723
1763
  # Clamp the sub-agent's permission tier at the parent's. The
1724
1764
  # default :readonly is always ≤ any parent tier, so this fires
@@ -1067,7 +1067,11 @@ module Parse
1067
1067
  if (mode = Parse::MongoDB.send(:normalize_read_preference, read_preference))
1068
1068
  coll = coll.with(read: { mode: mode })
1069
1069
  end
1070
- coll.aggregate(pipeline, agg_opts).to_a
1070
+ # Same QueryPlanKilled re-run as Parse::MongoDB.aggregate (private
1071
+ # there too, hence `send`).
1072
+ Parse::MongoDB.send(:with_query_killed_retry, collection_name) do
1073
+ coll.aggregate(pipeline, agg_opts).to_a
1074
+ end
1071
1075
  rescue => e
1072
1076
  # `raise_if_timeout!` is module-private on Parse::MongoDB; use
1073
1077
  # `send` so we can reuse the timeout-translation logic without
data/lib/parse/client.rb CHANGED
@@ -1424,6 +1424,30 @@ module Parse
1424
1424
  retry
1425
1425
  end
1426
1426
  raise
1427
+ rescue Faraday::ConnectionFailed => e
1428
+ # `Faraday::ConnectionFailed` covers two very different failures under
1429
+ # one class, so it is split on the wrapped cause (see
1430
+ # #connection_reset_error?):
1431
+ #
1432
+ # - RESET mid-flight (`Errno::ECONNRESET` / `Errno::EPIPE` /
1433
+ # `Errno::ECONNABORTED` / `EOFError`): the classic stale
1434
+ # keep-alive failure. A pooled persistent connection idled past
1435
+ # the server's (or an LB's) keep-alive window and was closed
1436
+ # remotely, and the next request on it dies at the socket.
1437
+ # Transient by nature (a fresh connection succeeds immediately),
1438
+ # so it retries under the same idempotency rules as a read
1439
+ # timeout: the outcome is unknown, so only idempotent requests
1440
+ # are re-sent.
1441
+ #
1442
+ # - REFUSED (and DNS failure): the server is down or misconfigured.
1443
+ # Retrying only adds backoff latency and `[Parse:Retry]` noise
1444
+ # before the inevitable error, so it propagates raw and fast.
1445
+ raise unless connection_reset_error?(e)
1446
+ if _retry_count > 0 && idempotent_retry?(method, body, headers)
1447
+ _retry_count = consume_retry_with_backoff(_retry_count, _retry_max, _request)
1448
+ retry
1449
+ end
1450
+ raise Parse::Error::ConnectionError, "#{_request} : #{e.class} - #{e.message}"
1427
1451
  rescue Faraday::ClientError, Faraday::TimeoutError, Net::OpenTimeout => e
1428
1452
  # Request timed out mid-flight: the outcome is unknown (the server may
1429
1453
  # have received and applied the write but never answered), so only
@@ -1431,25 +1455,74 @@ module Parse
1431
1455
  #
1432
1456
  # Faraday 2.x raises `Faraday::TimeoutError` for a read timeout
1433
1457
  # (`Timeout::Error` / `Errno::ETIMEDOUT`); it subclasses `Faraday::Error`,
1434
- # not `ClientError`, so it must be listed explicitly to be caught. We
1435
- # deliberately do NOT catch `Faraday::ConnectionFailed` (connection
1436
- # refused/reset, plus the wrapped connect-timeout): refused is a
1437
- # non-transient "server down / misconfigured" failure, and auto-retrying
1438
- # it only adds backoff latency before the inevitable error. Broadening to
1439
- # reset connections safely (retry reset, fail fast on refused) is tracked
1440
- # as a follow-up.
1458
+ # not `ClientError`, so it must be listed explicitly to be caught.
1459
+ # `Faraday::ConnectionFailed` is handled in its own rescue above,
1460
+ # split into retry-reset / fail-fast-refused.
1441
1461
  if _retry_count > 0 && idempotent_retry?(method, body, headers)
1442
- warn "[Parse:Retry] Retries remaining #{_retry_count} : #{_request}"
1443
- _retry_count -= 1
1444
- backoff_delay = RETRY_DELAY * (_retry_max - _retry_count)
1445
- _retry_delay = backoff_delay * (0.75 + rand * 0.5)
1446
- sleep _retry_delay if _retry_delay > 0
1462
+ _retry_count = consume_retry_with_backoff(_retry_count, _retry_max, _request)
1447
1463
  retry
1448
1464
  end
1449
1465
  raise Parse::Error::ConnectionError, "#{_request} : #{e.class} - #{e.message}"
1450
1466
  end
1451
1467
  end
1452
1468
 
1469
+ # Consumes one attempt from the retry budget: logs the remaining count,
1470
+ # sleeps the linear backoff (RETRY_DELAY x attempt number, +/-25% jitter,
1471
+ # never zero), and returns the decremented budget. Shared by the
1472
+ # connection-reset and timeout rescue branches in {#request} so their
1473
+ # backoff behavior cannot drift apart; the `retry` keyword itself must
1474
+ # stay lexically inside each rescue clause, so it remains at the call
1475
+ # sites. The 429/503 branch keeps its own inline version because it also
1476
+ # honors a server-supplied Retry-After header.
1477
+ # @param retry_count [Integer] the remaining retry budget (must be > 0).
1478
+ # @param retry_max [Integer] the effective starting budget.
1479
+ # @param request [Parse::Request] the request being retried (for logging).
1480
+ # @return [Integer] the decremented retry budget.
1481
+ def consume_retry_with_backoff(retry_count, retry_max, request)
1482
+ warn "[Parse:Retry] Retries remaining #{retry_count} : #{request}"
1483
+ retry_count -= 1
1484
+ backoff_delay = RETRY_DELAY * (retry_max - retry_count)
1485
+ retry_delay = backoff_delay * (0.75 + rand * 0.5)
1486
+ sleep retry_delay if retry_delay > 0
1487
+ retry_count
1488
+ end
1489
+
1490
+ # The wrapped causes that mark a `Faraday::ConnectionFailed` as a RESET
1491
+ # connection (transient, retry-safe for idempotent requests) rather than a
1492
+ # REFUSED one (server down, fail fast). `EOFError` is what net/http raises
1493
+ # when the remote end closes a keep-alive socket cleanly between requests;
1494
+ # ECONNRESET/EPIPE/ECONNABORTED are the unclean variants.
1495
+ # @!visibility private
1496
+ CONNECTION_RESET_CAUSES = [
1497
+ Errno::ECONNRESET, Errno::EPIPE, Errno::ECONNABORTED, EOFError,
1498
+ ].freeze
1499
+
1500
+ # Message fallback for adapters that raise `Faraday::ConnectionFailed`
1501
+ # with the cause flattened into the message instead of wrapped.
1502
+ # @!visibility private
1503
+ CONNECTION_RESET_MESSAGE = /connection reset|broken pipe|end of file reached/i
1504
+
1505
+ # Whether a `Faraday::ConnectionFailed` was caused by a reset/dropped
1506
+ # connection (retryable) as opposed to connection-refused or a DNS
1507
+ # failure (fail fast). Walks the wrapped exception and the `#cause`
1508
+ # chain looking for a reset-class error.
1509
+ # @param error [Exception] the rescued `Faraday::ConnectionFailed`.
1510
+ # @return [Boolean]
1511
+ def connection_reset_error?(error)
1512
+ inner = error.respond_to?(:wrapped_exception) ? error.wrapped_exception : nil
1513
+ inner ||= error.cause
1514
+ seen = 0
1515
+ while inner && seen < 8
1516
+ return true if CONNECTION_RESET_CAUSES.any? { |klass| inner.is_a?(klass) }
1517
+ inner = inner.cause
1518
+ seen += 1
1519
+ end
1520
+ CONNECTION_RESET_MESSAGE.match?(error.message.to_s)
1521
+ end
1522
+
1523
+ private :consume_retry_with_backoff, :connection_reset_error?
1524
+ private_constant :CONNECTION_RESET_CAUSES, :CONNECTION_RESET_MESSAGE
1525
+
1453
1526
  # Whether a request whose outcome is UNKNOWN (a 500/503 or a dropped
1454
1527
  # connection) is safe to transparently re-send.
1455
1528
  #
@@ -8,7 +8,7 @@ require "json"
8
8
  module Parse
9
9
  module Embeddings
10
10
  # Process-local embedding cache keyed by
11
- # `(provider, model, input_type, input_hash)`.
11
+ # `(provider, model, dimensions, input_type, deployment, input_hash)`.
12
12
  #
13
13
  # Query-side embedding is the hot repeat path: the same natural-
14
14
  # language query (an agent retrying a tool call, a user paging
@@ -32,7 +32,10 @@ module Parse
32
32
  #
33
33
  # == Key derivation
34
34
  #
35
- # `provider.class.name | model_name | input_type | SHA-256(input)`.
35
+ # `provider.class.name | model_name | dimensions | input_type |
36
+ # cache_identity | SHA-256(input)`. `cache_identity` (the provider's
37
+ # endpoint without credentials, see {Provider#cache_identity}) is
38
+ # omitted for providers that have none.
36
39
  # The full input text never becomes part of the key, so a shared
37
40
  # external store does not accumulate plaintext queries.
38
41
  #
@@ -305,8 +308,8 @@ module Parse
305
308
  # @!visibility private
306
309
  # Composite cache key. The input is hashed so plaintext never
307
310
  # lands in a shared store; provider identity + model + dimensions
308
- # + input_type namespace the hash (two models' vectors are never
309
- # confused). Dimensions matter independently of the model name:
311
+ # + input_type + deployment namespace the hash (two models' or two
312
+ # deployments' vectors are never confused). Dimensions matter independently of the model name:
310
313
  # Matryoshka-capable providers (OpenAI text-embedding-3-*, Cohere
311
314
  # embed-v4, Voyage, Jina, Qwen) can register the same model at
312
315
  # different output widths, and serving one width's cached vector
@@ -322,7 +325,16 @@ module Parse
322
325
  rescue NotImplementedError
323
326
  "unknown"
324
327
  end
325
- "#{provider.class.name}|#{model}|#{dims}|#{input_type}|#{Digest::SHA256.hexdigest(input.to_s)}"
328
+ # Deployment identity (endpoint, never credentials) separates two
329
+ # deployments of the same model. Omitted when the provider has
330
+ # none, so such providers keep their existing keys.
331
+ identity = begin
332
+ provider.respond_to?(:cache_identity) ? provider.cache_identity : nil
333
+ rescue StandardError
334
+ nil
335
+ end
336
+ deployment = identity.nil? || identity.to_s.empty? ? "" : "#{identity}|"
337
+ "#{provider.class.name}|#{model}|#{dims}|#{input_type}|#{deployment}#{Digest::SHA256.hexdigest(input.to_s)}"
326
338
  end
327
339
 
328
340
  # @!visibility private
@@ -1,6 +1,8 @@
1
1
  # encoding: UTF-8
2
2
  # frozen_string_literal: true
3
3
 
4
+ require "uri"
5
+
4
6
  module Parse
5
7
  module Embeddings
6
8
  # Abstract base class for embedding providers. Concrete subclasses
@@ -242,6 +244,30 @@ module Parse
242
244
 
243
245
  # @return [Hash] attributes safe to surface in {#inspect}. Override
244
246
  # in subclasses to add fields; never add credentials.
247
+ # Identity of the deployment this provider talks to, for use in cache
248
+ # keys: the endpoint's scheme, host, non-default port, and path. Never
249
+ # includes credentials, userinfo, or a query string. Two providers of
250
+ # the same class and model pointed at different deployments (two
251
+ # self-hosted servers, or a provider behind a proxy) can return
252
+ # different vectors for the same input, so their cache entries must
253
+ # not be shared.
254
+ #
255
+ # The default reads `@base_url`, which every built-in HTTP provider
256
+ # sets. A provider with no configurable endpoint returns nil, which
257
+ # leaves its cache key unchanged. Override to supply another identity.
258
+ #
259
+ # @return [String, nil]
260
+ def cache_identity
261
+ url = instance_variable_defined?(:@base_url) ? @base_url : nil
262
+ return nil if url.nil? || url.to_s.empty?
263
+ uri = URI.parse(url.to_s)
264
+ return nil if uri.host.nil? || uri.host.empty?
265
+ port = uri.port && uri.port != uri.default_port ? ":#{uri.port}" : ""
266
+ "#{uri.scheme}://#{uri.host.downcase}#{port}#{uri.path.to_s.chomp("/")}"
267
+ rescue URI::InvalidURIError
268
+ nil
269
+ end
270
+
245
271
  def inspect_attrs
246
272
  out = {}
247
273
  out[:model] = safe_call(:model_name)
@@ -9,9 +9,11 @@ require_relative "provider"
9
9
  module Parse
10
10
  module Embeddings
11
11
  # Voyage AI embeddings provider. Wraps `POST /v1/embeddings` for
12
- # text-only models and `POST /v1/multimodalembeddings` for the
12
+ # text-only models, `POST /v1/multimodalembeddings` for the
13
13
  # multimodal text+image models (text via {#embed_text}, images via
14
- # {#embed_image}).
14
+ # {#embed_image}), and `POST /v1/contextualizedembeddings` for the
15
+ # contextualized chunk models (single texts via {#embed_text},
16
+ # whole chunked documents via {#embed_chunks}).
15
17
  #
16
18
  # Supported models:
17
19
  #
@@ -21,7 +23,8 @@ module Parse
21
23
  # llama.cpp).
22
24
  # * **v3 family** — `voyage-3-large`, `voyage-3.5`,
23
25
  # `voyage-3.5-lite`, `voyage-3`, `voyage-3-lite`.
24
- # * **code models** — `voyage-code-3`, `voyage-code-2` (1536-dim).
26
+ # * **code models**: `voyage-code-4`, `voyage-code-3`, and
27
+ # `voyage-code-2` (the only 1536-dim one).
25
28
  # * **domain models** — `voyage-finance-2`, `voyage-law-2`.
26
29
  # * **multimodal** — `voyage-multimodal-3` (text+image) and
27
30
  # `voyage-multimodal-3.5` (text+image+video). Unified vector
@@ -31,6 +34,14 @@ module Parse
31
34
  # {#embed_image}, video through {#embed_video}. All three share
32
35
  # the same space, so stored text vectors are comparable against
33
36
  # image and video vectors without re-embedding.
37
+ # * **contextualized chunk**: `voyage-context-4` and
38
+ # `voyage-context-3`. Each chunk's vector also encodes the
39
+ # document it came from, so chunks are embedded a document at a
40
+ # time through {#embed_chunks}. {#embed_text} sends every string
41
+ # as a one-chunk document. That is the right shape for queries, and it
42
+ # is also what the `embed` macro and {BatchEmbedder} send for stored
43
+ # fields, so those vectors carry no surrounding-document context;
44
+ # call {#embed_chunks} directly for chunk-level context.
34
45
  #
35
46
  # Audio is not offered by any Voyage model, and neither PDF nor
36
47
  # DOCX is accepted as a content type — render document pages to
@@ -127,6 +138,13 @@ module Parse
127
138
  DEFAULT_BATCH_SIZE = 128
128
139
  MAX_RESPONSE_BYTES = 16 * 1024 * 1024
129
140
 
141
+ # Upper bounds on the JSON size of one returned vector, used to plan
142
+ # contextualized requests: a float such as `-1.2345678901234567e-05,`
143
+ # is under 32 bytes, and the `{"object":"embedding","embedding":[],
144
+ # "index":N}` wrapper is under 256.
145
+ RESPONSE_BYTES_PER_VALUE = 32
146
+ RESPONSE_BYTES_PER_VECTOR_ENVELOPE = 256
147
+
130
148
  # Default (native) vector width per model — the width returned
131
149
  # when `output_dimension` is omitted from the request.
132
150
  #
@@ -147,12 +165,15 @@ module Parse
147
165
  "voyage-3.5-lite" => 1024,
148
166
  "voyage-3" => 1024,
149
167
  "voyage-3-lite" => 512,
168
+ "voyage-code-4" => 1024,
150
169
  "voyage-code-3" => 1024,
151
170
  "voyage-code-2" => 1536,
152
171
  "voyage-finance-2" => 1024,
153
172
  "voyage-law-2" => 1024,
154
173
  "voyage-multimodal-3" => 1024,
155
174
  "voyage-multimodal-3.5" => 1024,
175
+ "voyage-context-4" => 1024,
176
+ "voyage-context-3" => 1024,
156
177
  }.freeze
157
178
 
158
179
  # Every width a model's Matryoshka head will actually return.
@@ -174,12 +195,15 @@ module Parse
174
195
  "voyage-3.5-lite" => [256, 512, 1024, 2048],
175
196
  "voyage-3" => [1024],
176
197
  "voyage-3-lite" => [512],
198
+ "voyage-code-4" => [256, 512, 1024, 2048],
177
199
  "voyage-code-3" => [256, 512, 1024, 2048],
178
200
  "voyage-code-2" => [1536],
179
201
  "voyage-finance-2" => [1024],
180
202
  "voyage-law-2" => [1024],
181
203
  "voyage-multimodal-3" => [1024],
182
204
  "voyage-multimodal-3.5" => [256, 512, 1024, 2048],
205
+ "voyage-context-4" => [256, 512, 1024, 2048],
206
+ "voyage-context-3" => [256, 512, 1024, 2048],
183
207
  }.freeze
184
208
 
185
209
  # Back-compat alias: the set of models accepting any
@@ -198,12 +222,17 @@ module Parse
198
222
  "voyage-3.5-lite" => 32_000,
199
223
  "voyage-3" => 32_000,
200
224
  "voyage-3-lite" => 32_000,
225
+ "voyage-code-4" => 32_000,
201
226
  "voyage-code-3" => 32_000,
202
227
  "voyage-code-2" => 16_000,
203
228
  "voyage-finance-2" => 32_000,
204
229
  "voyage-law-2" => 16_000,
205
230
  "voyage-multimodal-3" => 32_000,
206
231
  "voyage-multimodal-3.5" => 32_000,
232
+ # Per document (one inner list of chunks). The request as a
233
+ # whole is capped at 120k tokens across every document.
234
+ "voyage-context-4" => 32_000,
235
+ "voyage-context-3" => 32_000,
207
236
  }.freeze
208
237
 
209
238
  # Models that route to `/v1/multimodalembeddings` with the
@@ -226,6 +255,27 @@ module Parse
226
255
  # "does not support video inputs" 400.
227
256
  VIDEO_MODELS = %w[voyage-multimodal-3.5].freeze
228
257
 
258
+ # Models that route to `/v1/contextualizedembeddings` with the
259
+ # `{ inputs: [[chunk, ...], ...] }` envelope: one inner list per
260
+ # document, each chunk embedded with the rest of its document as
261
+ # context. The endpoint has no `truncation` field, so it is never
262
+ # sent for these models.
263
+ CONTEXTUALIZED_MODELS = %w[voyage-context-4 voyage-context-3].freeze
264
+
265
+ # Voyage's per-request limits for the contextualized endpoint:
266
+ # at most this many documents, and this many chunks summed
267
+ # across them.
268
+ # The endpoint also caps a request at 120k tokens summed across
269
+ # every document, which the SDK cannot check without a tokenizer.
270
+ MAX_CONTEXT_DOCUMENTS = 1_000
271
+ MAX_CONTEXT_CHUNKS = 16_000
272
+
273
+ # Default `embed_batch_size` for {CONTEXTUALIZED_MODELS}. Each
274
+ # string is a whole document there, so 128 paragraph-sized inputs
275
+ # would overrun the 120k-token request cap; 32 leaves room for
276
+ # documents averaging under about 3,700 tokens.
277
+ CONTEXT_DEFAULT_BATCH_SIZE = 32
278
+
229
279
  # Models Voyage's hosted API serves but the Atlas Embedding and
230
280
  # Reranking API does not. Verified against both endpoints.
231
281
  ATLAS_UNAVAILABLE_MODELS = %w[voyage-3 voyage-3-lite].freeze
@@ -271,11 +321,13 @@ module Parse
271
321
  # @param timeout [Integer] read timeout, seconds.
272
322
  # @param open_timeout [Integer] connect timeout, seconds.
273
323
  # @param max_retries [Integer] retry attempts on 429/5xx/timeouts.
274
- # @param embed_batch_size [Integer] inputs per request (max 128).
324
+ # @param embed_batch_size [Integer, nil] inputs per request (max 128).
325
+ # Defaults to {DEFAULT_BATCH_SIZE}, or {CONTEXT_DEFAULT_BATCH_SIZE}
326
+ # for a contextualized model.
275
327
  # @param dimensions [Integer, nil] override output width via
276
- # Voyage's `output_dimension` Matryoshka parameter. Only
277
- # `voyage-4-large` accepts the field; for every other model the
278
- # override must equal the native width or be omitted.
328
+ # Voyage's `output_dimension` Matryoshka parameter. Must be one of
329
+ # the model's {MODEL_SUPPORTED_DIMENSIONS}; a model with a single
330
+ # supported width accepts only that width (or nil).
279
331
  # @param truncation [Boolean] forward Voyage's `truncation:` field.
280
332
  # Defaults `true` to match Voyage's API default. Set `false` to
281
333
  # force the API to reject over-length inputs rather than silently
@@ -292,7 +344,7 @@ module Parse
292
344
  timeout: DEFAULT_TIMEOUT,
293
345
  open_timeout: DEFAULT_OPEN_TIMEOUT,
294
346
  max_retries: DEFAULT_MAX_RETRIES,
295
- embed_batch_size: DEFAULT_BATCH_SIZE,
347
+ embed_batch_size: nil,
296
348
  dimensions: nil,
297
349
  truncation: true,
298
350
  allow_faraday_proxy: false,
@@ -308,6 +360,7 @@ module Parse
308
360
  validate_positive_integer!(:timeout, timeout)
309
361
  validate_positive_integer!(:open_timeout, open_timeout)
310
362
  validate_non_negative_integer!(:max_retries, max_retries)
363
+ embed_batch_size ||= CONTEXTUALIZED_MODELS.include?(model) ? CONTEXT_DEFAULT_BATCH_SIZE : DEFAULT_BATCH_SIZE
311
364
  validate_positive_integer!(:embed_batch_size, embed_batch_size)
312
365
  if embed_batch_size > 128
313
366
  raise ArgumentError,
@@ -396,10 +449,22 @@ module Parse
396
449
  end
397
450
  wire_input_type = INPUT_TYPE_WIRE_VALUES[input_type]
398
451
 
452
+ if CONTEXTUALIZED_MODELS.include?(@model)
453
+ if strings.length > MAX_CONTEXT_DOCUMENTS
454
+ raise ArgumentError,
455
+ "Parse::Embeddings::Voyage#embed_text: #{strings.length} inputs exceeds Voyage's " \
456
+ "per-request cap for #{@model} (#{MAX_CONTEXT_DOCUMENTS}). Split the input."
457
+ end
458
+ # Each string is its own one-chunk document, so the response
459
+ # carries exactly one vector per document.
460
+ return embed_contextualized(strings.map { |s| [s] }, input_type, wire_input_type).map(&:first)
461
+ end
462
+
399
463
  # Multimodal models route to a different endpoint with a
400
464
  # different request envelope. The response envelope shape is
401
465
  # the same (`{ data: [{ embedding, index }], usage: {...} }`)
402
466
  # so `extract_vectors!` is reused as-is.
467
+
403
468
  body = if MULTIMODAL_MODELS.include?(@model)
404
469
  build_multimodal_body(strings, wire_input_type)
405
470
  else
@@ -500,6 +565,72 @@ module Parse
500
565
  allow_insecure: allow_insecure)
501
566
  end
502
567
 
568
+ # Embed chunked documents through Voyage's
569
+ # `/v1/contextualizedembeddings` endpoint. Every chunk's vector
570
+ # encodes the surrounding document as well as the chunk itself,
571
+ # so pass the chunks of one document together, in order, rather
572
+ # than one call per chunk.
573
+ #
574
+ # **Contextualized model required.** Only {CONTEXTUALIZED_MODELS}
575
+ # accept this shape; any other model raises {BadRequestError}
576
+ # before any network call.
577
+ #
578
+ # @param documents [Array<Array<String>>] one inner Array per
579
+ # document, holding that document's chunks in order. At most
580
+ # {MAX_CONTEXT_DOCUMENTS} documents and {MAX_CONTEXT_CHUNKS}
581
+ # chunks in total.
582
+ # @param input_type [Symbol] one of {INPUT_TYPE_WIRE_VALUES}'s keys.
583
+ # @return [Array<Array<Array<Float>>>] one Array of chunk vectors
584
+ # per document, aligned 1:1 with `documents` and with each
585
+ # document's chunks.
586
+ def embed_chunks(documents, input_type: :search_document)
587
+ unless CONTEXTUALIZED_MODELS.include?(@model)
588
+ raise BadRequestError,
589
+ "Parse::Embeddings::Voyage#embed_chunks: model #{@model.inspect} does not " \
590
+ "accept chunked documents. Configure the provider with a contextualized model " \
591
+ "(supported: #{CONTEXTUALIZED_MODELS.inspect})."
592
+ end
593
+ unless documents.is_a?(Array)
594
+ raise ArgumentError,
595
+ "Parse::Embeddings::Voyage#embed_chunks expects Array<Array<String>> " \
596
+ "(got #{documents.class})."
597
+ end
598
+ return [] if documents.empty?
599
+
600
+ documents.each_with_index do |chunks, i|
601
+ unless chunks.is_a?(Array) && !chunks.empty?
602
+ raise ArgumentError,
603
+ "Parse::Embeddings::Voyage#embed_chunks documents[#{i}] must be a non-empty " \
604
+ "Array of chunk Strings."
605
+ end
606
+ chunks.each_with_index do |c, j|
607
+ unless c.is_a?(String) && !c.empty?
608
+ raise ArgumentError,
609
+ "Parse::Embeddings::Voyage#embed_chunks documents[#{i}][#{j}] must be a " \
610
+ "non-empty String."
611
+ end
612
+ end
613
+ end
614
+ if documents.length > MAX_CONTEXT_DOCUMENTS
615
+ raise ArgumentError,
616
+ "Parse::Embeddings::Voyage#embed_chunks: #{documents.length} documents exceeds " \
617
+ "Voyage's per-request cap (#{MAX_CONTEXT_DOCUMENTS}). Split the input."
618
+ end
619
+ chunk_total = documents.sum(&:length)
620
+ if chunk_total > MAX_CONTEXT_CHUNKS
621
+ raise ArgumentError,
622
+ "Parse::Embeddings::Voyage#embed_chunks: #{chunk_total} chunks exceeds " \
623
+ "Voyage's per-request cap (#{MAX_CONTEXT_CHUNKS}). Split the input."
624
+ end
625
+ unless INPUT_TYPE_WIRE_VALUES.key?(input_type)
626
+ raise ArgumentError,
627
+ "Parse::Embeddings::Voyage#embed_chunks input_type #{input_type.inspect} not in " \
628
+ "#{INPUT_TYPE_WIRE_VALUES.keys.inspect}."
629
+ end
630
+
631
+ embed_contextualized(documents, input_type, INPUT_TYPE_WIRE_VALUES[input_type])
632
+ end
633
+
503
634
  def inspect_attrs
504
635
  super.merge(base: safe_base_host, endpoint: @endpoint, retries: @max_retries)
505
636
  end
@@ -561,6 +692,68 @@ module Parse
561
692
  body
562
693
  end
563
694
 
695
+ # Build the wire body for `/v1/contextualizedembeddings`. The
696
+ # endpoint documents no `truncation` field, so none is sent.
697
+ def build_contextualized_body(documents, wire_input_type)
698
+ body = { inputs: documents, model: @model }
699
+ body[:input_type] = wire_input_type if wire_input_type
700
+ apply_output_dimension!(body)
701
+ body
702
+ end
703
+
704
+ # Embed documents through the contextualized endpoint and return one
705
+ # Array of chunk vectors per document, in input order.
706
+ #
707
+ # Voyage accepts up to {MAX_CONTEXT_CHUNKS} chunks per request, but
708
+ # that many vectors serialize to far more than {MAX_RESPONSE_BYTES}.
709
+ # Whole documents are therefore grouped so each response's estimated
710
+ # size stays within the cap, and the groups are sent in turn. A
711
+ # single document too large for the cap on its own is sent alone
712
+ # with a response allowance sized to its chunk count.
713
+ def embed_contextualized(documents, input_type, wire_input_type)
714
+ per_vector = response_bytes_per_vector
715
+ chunks_per_request = [MAX_RESPONSE_BYTES / per_vector, 1].max
716
+ groups = []
717
+ documents.each do |doc|
718
+ last = groups.last
719
+ if last && last.sum(&:length) + doc.length <= chunks_per_request
720
+ last << doc
721
+ else
722
+ groups << [doc]
723
+ end
724
+ end
725
+ groups.flat_map { |group| embed_contextualized_request(group, input_type, wire_input_type) }
726
+ end
727
+
728
+ # @return [Integer] the planning estimate for one vector's JSON size.
729
+ def response_bytes_per_vector
730
+ (@dimensions * RESPONSE_BYTES_PER_VALUE) + RESPONSE_BYTES_PER_VECTOR_ENVELOPE
731
+ end
732
+
733
+ # Issue one contextualized request and return one Array of chunk
734
+ # vectors per document.
735
+ def embed_contextualized_request(documents, input_type, wire_input_type)
736
+ body = build_contextualized_body(documents, wire_input_type)
737
+ chunk_count = documents.sum(&:length)
738
+ allowance = [MAX_RESPONSE_BYTES, (chunk_count * response_bytes_per_vector) + 65_536].max
739
+
740
+ instrument_embed(chunk_count, input_type) do |emit_payload|
741
+ payload = post_embeddings(body, path: "contextualizedembeddings", max_response_bytes: allowance)
742
+ if payload.is_a?(Hash) && payload["usage"].is_a?(Hash)
743
+ tt = payload["usage"]["total_tokens"]
744
+ emit_payload[:total_tokens] = tt if tt.is_a?(Integer) && tt >= 0
745
+ end
746
+ # The response nests the standard envelope: the outer `data`
747
+ # holds one `{ data: [...], index: }` list per document, and
748
+ # each inner list is shaped like a `/v1/embeddings` response.
749
+ per_document = extract_vectors!(payload, documents.length, value_key: "data")
750
+ per_document.each_with_index.map do |entry, i|
751
+ vectors = extract_vectors!({ "data" => entry }, documents[i].length)
752
+ validate_response!(documents[i].length, vectors)
753
+ end
754
+ end
755
+ end
756
+
564
757
  # Forward `output_dimension` only when the configured width
565
758
  # differs from the model's native default. Sending it to a model
566
759
  # with a single supported width is a 400, and sending the native
@@ -792,7 +985,9 @@ module Parse
792
985
  encoded[0...-1]
793
986
  end
794
987
 
795
- def post_embeddings(body, path: "embeddings")
988
+ # @param max_response_bytes [Integer] refuse a success body larger than
989
+ # this. Defaults to {MAX_RESPONSE_BYTES}.
990
+ def post_embeddings(body, path: "embeddings", max_response_bytes: MAX_RESPONSE_BYTES)
796
991
  attempts = 0
797
992
  loop do
798
993
  attempts += 1
@@ -820,7 +1015,7 @@ module Parse
820
1015
  end
821
1016
 
822
1017
  status = response.status
823
- return parse_json_body!(response.body) if status >= 200 && status < 300
1018
+ return parse_json_body!(response.body, max_response_bytes) if status >= 200 && status < 300
824
1019
 
825
1020
  if status == 401
826
1021
  raise AuthenticationError,
@@ -847,11 +1042,11 @@ module Parse
847
1042
  end
848
1043
  end
849
1044
 
850
- def parse_json_body!(body)
1045
+ def parse_json_body!(body, max_bytes = MAX_RESPONSE_BYTES)
851
1046
  s = body.to_s
852
- if s.bytesize > MAX_RESPONSE_BYTES
1047
+ if s.bytesize > max_bytes
853
1048
  raise InvalidResponseError,
854
- "Parse::Embeddings::Voyage: response body exceeds #{MAX_RESPONSE_BYTES} bytes " \
1049
+ "Parse::Embeddings::Voyage: response body exceeds #{max_bytes} bytes " \
855
1050
  "(#{s.bytesize}). Refusing to parse."
856
1051
  end
857
1052
  JSON.parse(s, max_nesting: 32)
@@ -871,7 +1066,11 @@ module Parse
871
1066
  # "model": "voyage-3",
872
1067
  # "usage": { "total_tokens": N }
873
1068
  # }
874
- def extract_vectors!(payload, input_count)
1069
+ #
1070
+ # `value_key` names the field read from each entry: `"embedding"`
1071
+ # for the flat envelope, `"data"` for the outer list of the
1072
+ # contextualized envelope.
1073
+ def extract_vectors!(payload, input_count, value_key: "embedding")
875
1074
  unless payload.is_a?(Hash)
876
1075
  raise InvalidResponseError,
877
1076
  "Parse::Embeddings::Voyage: response body is not a JSON object."
@@ -895,7 +1094,7 @@ module Parse
895
1094
  raise InvalidResponseError,
896
1095
  "Parse::Embeddings::Voyage: response.data[#{i}].index #{idx.inspect} out of range."
897
1096
  end
898
- [idx, entry["embedding"]]
1097
+ [idx, entry[value_key]]
899
1098
  end
900
1099
  indices = sorted.map(&:first)
901
1100
  if indices.uniq.length != indices.length