parse-stack-next 5.7.6 → 5.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +830 -0
- data/README.md +14 -4
- data/docs/TEST_SERVER.md +2 -2
- data/docs/acl_clp_guide.md +7 -0
- data/docs/atlas_vector_search_guide.md +181 -13
- data/docs/client_sdk_guide.md +11 -0
- data/docs/mcp_guide.md +317 -6
- data/docs/mongodb_direct_guide.md +27 -0
- data/docs/usage_guide.md +38 -0
- data/docs/webhooks_guide.md +74 -17
- data/lib/parse/acl_scope.rb +159 -41
- data/lib/parse/agent/approval_gate.rb +0 -0
- data/lib/parse/agent/constraint_translator.rb +42 -15
- data/lib/parse/agent/describe.rb +3 -1
- data/lib/parse/agent/field_names.rb +53 -0
- data/lib/parse/agent/field_policy.rb +74 -0
- data/lib/parse/agent/mcp_deployments.rb +426 -0
- data/lib/parse/agent/mcp_rack_app.rb +424 -45
- data/lib/parse/agent/mcp_server.rb +23 -1
- data/lib/parse/agent/mcp_subscriptions.rb +124 -6
- data/lib/parse/agent/metadata_registry.rb +67 -8
- data/lib/parse/agent/prompt_hardening.rb +9 -3
- data/lib/parse/agent/tools.rb +378 -29
- data/lib/parse/agent.rb +93 -1
- data/lib/parse/api/batch.rb +10 -1
- data/lib/parse/api/schema.rb +23 -4
- data/lib/parse/api/sessions.rb +6 -2
- data/lib/parse/api/users.rb +88 -14
- data/lib/parse/atlas_search/protected_paths.rb +236 -0
- data/lib/parse/atlas_search.rb +95 -23
- data/lib/parse/authorization.rb +54 -1
- data/lib/parse/client/batch.rb +231 -35
- data/lib/parse/client/body_builder.rb +21 -0
- data/lib/parse/client/caching.rb +371 -27
- data/lib/parse/client/request.rb +26 -14
- data/lib/parse/client/response.rb +49 -6
- data/lib/parse/client.rb +201 -38
- data/lib/parse/clp_scope.rb +281 -23
- data/lib/parse/console.rb +2 -2
- data/lib/parse/embeddings/voyage.rb +181 -17
- data/lib/parse/graphql/type_generator.rb +3 -0
- data/lib/parse/model/acl.rb +119 -21
- data/lib/parse/model/associations/belongs_to.rb +25 -3
- data/lib/parse/model/associations/collection_proxy.rb +138 -17
- data/lib/parse/model/associations/has_many.rb +38 -9
- data/lib/parse/model/associations/has_one.rb +3 -1
- data/lib/parse/model/associations/pointer_collection_proxy.rb +109 -17
- data/lib/parse/model/associations/relation_collection_proxy.rb +134 -28
- data/lib/parse/model/bytes.rb +13 -5
- data/lib/parse/model/classes/role.rb +72 -0
- data/lib/parse/model/classes/session.rb +43 -0
- data/lib/parse/model/classes/user.rb +78 -3
- data/lib/parse/model/core/actions.rb +269 -67
- data/lib/parse/model/core/builder.rb +100 -8
- data/lib/parse/model/core/create_lock.rb +27 -2
- data/lib/parse/model/core/describe.rb +2 -0
- data/lib/parse/model/core/fetching.rb +21 -3
- data/lib/parse/model/core/pluralized_aliases.rb +8 -4
- data/lib/parse/model/core/properties.rb +488 -39
- data/lib/parse/model/core/querying.rb +7 -0
- data/lib/parse/model/core/schema.rb +5 -3
- data/lib/parse/model/core/search_indexing.rb +63 -0
- data/lib/parse/model/core/vector_searchable.rb +35 -6
- data/lib/parse/model/file.rb +9 -2
- data/lib/parse/model/geopoint.rb +61 -13
- data/lib/parse/model/model.rb +160 -9
- data/lib/parse/model/object.rb +265 -17
- data/lib/parse/model/phone.rb +54 -5
- data/lib/parse/model/pointer.rb +40 -6
- data/lib/parse/mongodb.rb +170 -60
- data/lib/parse/pipeline_security.rb +415 -26
- data/lib/parse/query/constraint.rb +30 -0
- data/lib/parse/query/constraints.rb +58 -32
- data/lib/parse/query/cursor.rb +3 -1
- data/lib/parse/query/operation.rb +62 -8
- data/lib/parse/query/ordering.rb +34 -6
- data/lib/parse/query.rb +1100 -134
- data/lib/parse/retrieval/agent_tool.rb +225 -8
- data/lib/parse/retrieval/benchmark.rb +149 -0
- data/lib/parse/retrieval/profiles.rb +320 -0
- data/lib/parse/retrieval/retriever.rb +10 -1
- data/lib/parse/retrieval.rb +2 -0
- data/lib/parse/schema/search_index_migrator.rb +23 -5
- data/lib/parse/schema.rb +74 -18
- data/lib/parse/stack/tasks.rb +6 -4
- data/lib/parse/stack/version.rb +1 -1
- data/lib/parse/stack.rb +72 -14
- data/lib/parse/two_factor_auth/user_extension.rb +14 -2
- data/lib/parse/two_factor_auth.rb +11 -0
- data/lib/parse/vector_search/hybrid.rb +36 -18
- data/lib/parse/vector_search/index_definition.rb +237 -0
- data/lib/parse/vector_search.rb +46 -17
- data/lib/parse/webhooks/payload.rb +93 -6
- data/lib/parse/webhooks/replay_protection.rb +58 -20
- data/lib/parse/webhooks.rb +412 -40
- metadata +8 -1
data/lib/parse/clp_scope.rb
CHANGED
|
@@ -21,6 +21,10 @@ module Parse
|
|
|
21
21
|
EMPTY_SET = Set.new.freeze
|
|
22
22
|
private_constant :EMPTY_SET
|
|
23
23
|
|
|
24
|
+
# Parse Server's default server-config `protectedFields` option. See
|
|
25
|
+
# {.default_protected_fields}.
|
|
26
|
+
DEFAULT_PROTECTED_FIELDS = { "_User" => { "*" => ["email"].freeze }.freeze }.freeze
|
|
27
|
+
|
|
24
28
|
# Cache-entry shape. `kind:` is the disposition of the most recent
|
|
25
29
|
# schema-fetch attempt:
|
|
26
30
|
#
|
|
@@ -288,31 +292,185 @@ module Parse
|
|
|
288
292
|
# `permits?` already refused the query, so this branch is only
|
|
289
293
|
# reached when callers ask for the protected-fields set directly
|
|
290
294
|
# (e.g. for documentation or audit tooling).
|
|
291
|
-
return EMPTY_SET if entry.kind == :
|
|
292
|
-
|
|
295
|
+
return EMPTY_SET if entry.kind == :unresolvable
|
|
296
|
+
stored_map = if entry.kind == :no_clp
|
|
297
|
+
nil
|
|
298
|
+
else
|
|
299
|
+
entry.clp["protectedFields"] || entry.clp[:protectedFields]
|
|
300
|
+
end
|
|
301
|
+
protected_map = merge_default_protected_fields(class_name, stored_map)
|
|
293
302
|
return EMPTY_SET if protected_map.nil? || protected_map.empty?
|
|
294
303
|
|
|
295
|
-
strip = Set.new(Array(protected_map["*"] || protected_map[:"*"]).map(&:to_s))
|
|
296
|
-
|
|
297
304
|
claim_set = permission_strings.is_a?(Set) ? permission_strings : permission_strings.to_set
|
|
298
|
-
claim_set.
|
|
299
|
-
|
|
300
|
-
override = protected_map[claim.to_s] || protected_map[claim.to_sym]
|
|
301
|
-
next if override.nil?
|
|
302
|
-
override_set = Set.new(Array(override).map(&:to_s))
|
|
303
|
-
strip &= override_set
|
|
304
|
-
end
|
|
305
|
+
protected_sets_for(protected_map, claim_set).reduce { |acc, set| acc & set }&.freeze || EMPTY_SET
|
|
306
|
+
end
|
|
305
307
|
|
|
306
|
-
|
|
308
|
+
# Parse Server's server-config `protectedFields` default, merged into
|
|
309
|
+
# every class's CLP `protectedFields` the way Parse Server's
|
|
310
|
+
# `SchemaData` merges them (union per group key).
|
|
311
|
+
#
|
|
312
|
+
# Parse Server ships `{ _User: { "*": ["email"] } }` as the default
|
|
313
|
+
# and does NOT expose the server-config value through
|
|
314
|
+
# `GET /schemas/<Class>`, so the SDK cannot discover it. Without this
|
|
315
|
+
# default a scoped mongo-direct read returned every user's email,
|
|
316
|
+
# and an `email` filter matched, where REST hides the field and
|
|
317
|
+
# refuses the filter. Set this to the same value as the server's
|
|
318
|
+
# `protectedFields` option when it is customized; `{}` disables the
|
|
319
|
+
# merge.
|
|
320
|
+
#
|
|
321
|
+
# @return [Hash{String => Hash{String => Array<String>}}]
|
|
322
|
+
def default_protected_fields
|
|
323
|
+
@default_protected_fields ||= DEFAULT_PROTECTED_FIELDS
|
|
324
|
+
end
|
|
325
|
+
|
|
326
|
+
# @param value [Hash, nil] class name => protectedFields map; nil
|
|
327
|
+
# restores Parse Server's default.
|
|
328
|
+
def default_protected_fields=(value)
|
|
329
|
+
@default_protected_fields = value.nil? ? DEFAULT_PROTECTED_FIELDS : value
|
|
307
330
|
end
|
|
308
331
|
|
|
309
|
-
|
|
332
|
+
# Strip protected fields from result rows, top-level keys only.
|
|
333
|
+
#
|
|
334
|
+
# Parse Server deletes `object[key]` for each protected key on the
|
|
335
|
+
# returned object and does not descend into nested values, so an
|
|
336
|
+
# `:object` column holding `{ "secret" => ... }` keeps that key even
|
|
337
|
+
# when a top-level `secret` column is protected. Included objects are
|
|
338
|
+
# separate rows of another class; strip them with that class's own set
|
|
339
|
+
# (see {Parse::MongoDB.aggregate}).
|
|
340
|
+
#
|
|
341
|
+
# Parse Server also skips the strip for a `_User` row that IS the
|
|
342
|
+
# requesting user, so a user always sees their own `email`. Pass
|
|
343
|
+
# `class_name:` and `user_id:` to apply that exemption.
|
|
344
|
+
#
|
|
345
|
+
# The exemption reads the row's `_id` (or `objectId`) as it comes out
|
|
346
|
+
# of the pipeline. A caller stage can rewrite that value, so pass
|
|
347
|
+
# `user_id:` only when ownership was settled before any caller stage
|
|
348
|
+
# ran: either no caller stage can change the identity (see
|
|
349
|
+
# {Parse::PipelineSecurity.identity_preserving?}), or the rows went
|
|
350
|
+
# through {.protected_strip_stage} at the head of their pipeline.
|
|
351
|
+
#
|
|
352
|
+
# @param documents [Array<Hash>] rows (Mongo storage or Parse form).
|
|
353
|
+
# @param strip_set [Set<String>] protected field names.
|
|
354
|
+
# @param class_name [String, nil] the rows' class.
|
|
355
|
+
# @param user_id [String, nil] the requesting user's objectId.
|
|
356
|
+
# @return [Array<Hash>] the same array, modified in place.
|
|
357
|
+
def redact_protected_fields!(documents, strip_set, class_name: nil, user_id: nil)
|
|
310
358
|
return documents if documents.nil? || documents.empty?
|
|
311
359
|
return documents if strip_set.nil? || strip_set.empty?
|
|
312
|
-
|
|
360
|
+
self_exempt = class_name.to_s == Parse::Model::CLASS_USER && !user_id.to_s.empty?
|
|
361
|
+
documents.each do |doc|
|
|
362
|
+
next unless doc.is_a?(Hash)
|
|
363
|
+
if self_exempt
|
|
364
|
+
row_id = doc["_id"] || doc[:_id] || doc["objectId"] || doc[:objectId]
|
|
365
|
+
next if row_id.to_s == user_id.to_s
|
|
366
|
+
end
|
|
367
|
+
strip_top_level!(doc, strip_set)
|
|
368
|
+
end
|
|
313
369
|
documents
|
|
314
370
|
end
|
|
315
371
|
|
|
372
|
+
# A pipeline stage that removes protected fields at the database,
|
|
373
|
+
# placed at the head of a (sub-)pipeline before any caller stage runs.
|
|
374
|
+
#
|
|
375
|
+
# Ownership of a `_User` row is decided here, on the stored `_id`, so
|
|
376
|
+
# the self exemption cannot be claimed by a later stage that rewrites
|
|
377
|
+
# `_id`. For a `_User` read by a known user the fields are kept only on
|
|
378
|
+
# that user's own row; everywhere else they are unset. The `_p_<field>`
|
|
379
|
+
# storage column of a protected pointer is removed with the field.
|
|
380
|
+
#
|
|
381
|
+
# @param strip_set [Set<String>] protected field names for the scope.
|
|
382
|
+
# @param class_name [String, nil] the class the stage reads.
|
|
383
|
+
# @param user_id [String, nil] the requesting user's objectId.
|
|
384
|
+
# @return [Hash, nil] a `$unset` or `$set` stage, or nil when nothing
|
|
385
|
+
# is protected.
|
|
386
|
+
def protected_strip_stage(strip_set, class_name: nil, user_id: nil)
|
|
387
|
+
return nil if strip_set.nil? || strip_set.empty?
|
|
388
|
+
columns = strip_set.flat_map { |f| [f.to_s, "_p_#{f}"] }.uniq
|
|
389
|
+
if class_name.to_s == Parse::Model::CLASS_USER && !user_id.to_s.empty?
|
|
390
|
+
own_row = { "$eq" => ["$_id", { "$literal" => user_id.to_s }] }
|
|
391
|
+
{ "$set" => columns.to_h { |c| [c, { "$cond" => [own_row, "$#{c}", "$$REMOVE"] }] } }
|
|
392
|
+
else
|
|
393
|
+
{ "$unset" => columns }
|
|
394
|
+
end
|
|
395
|
+
end
|
|
396
|
+
|
|
397
|
+
# Evaluate the `op` CLP for a resolved mongo-direct scope with the same
|
|
398
|
+
# mutually exclusive branches Parse Server uses, and return the row
|
|
399
|
+
# constraint the caller must apply.
|
|
400
|
+
#
|
|
401
|
+
# A public, user, or role grant permits every row, even when the CLP
|
|
402
|
+
# also lists `pointerFields` / `readUserFields` (Parse Server only
|
|
403
|
+
# consults those when no other branch grants). When the only grant is
|
|
404
|
+
# a pointer branch, the caller must keep just the rows whose named
|
|
405
|
+
# pointer fields reference the requesting user. `readUserFields`
|
|
406
|
+
# counts as a pointer branch here; the older {permits?} plus
|
|
407
|
+
# {pointer_fields_for} pair ignored it, so a `readUserFields` class
|
|
408
|
+
# let every authenticated user read every row on the direct path.
|
|
409
|
+
#
|
|
410
|
+
# @param class_name [String]
|
|
411
|
+
# @param op [Symbol] one of {OPERATIONS}.
|
|
412
|
+
# @param resolution [Parse::ACLScope::Resolution, nil]
|
|
413
|
+
# @param client [Parse::Client, nil] application whose schema owns the CLP.
|
|
414
|
+
# @param label [String, nil] scope label used in the error message.
|
|
415
|
+
# @return [Array<String>, nil] pointer field names that must reference
|
|
416
|
+
# the requesting user, or nil when every row is permitted. Always nil
|
|
417
|
+
# for master and nil resolutions.
|
|
418
|
+
# @raise [Denied] when the scope cannot perform `op` at all, including
|
|
419
|
+
# when the CLP is unresolvable (fail closed).
|
|
420
|
+
def row_constraint_for!(class_name, op, resolution, client: nil, label: nil)
|
|
421
|
+
return nil if resolution.nil?
|
|
422
|
+
return nil if resolution.respond_to?(:master?) && resolution.master?
|
|
423
|
+
perms = resolution.respond_to?(:permission_strings) ? resolution.permission_strings : nil
|
|
424
|
+
return nil if perms.nil?
|
|
425
|
+
|
|
426
|
+
user_id = resolution.respond_to?(:user_id) ? resolution.user_id.to_s : ""
|
|
427
|
+
evaluation = evaluate_access(
|
|
428
|
+
class_name, op,
|
|
429
|
+
claims: perms,
|
|
430
|
+
authenticated: !user_id.empty?,
|
|
431
|
+
user_id: user_id.empty? ? nil : user_id,
|
|
432
|
+
client: client,
|
|
433
|
+
)
|
|
434
|
+
if evaluation.allowed?
|
|
435
|
+
fields = Array(evaluation.pointer_fields)
|
|
436
|
+
return fields.empty? ? nil : fields
|
|
437
|
+
end
|
|
438
|
+
|
|
439
|
+
warn_unresolvable_once!(class_name) if evaluation.reason == :clp_unresolvable
|
|
440
|
+
scope = label ? "the current #{label} scope" : "the current scope"
|
|
441
|
+
message = if %i[concrete_user_required authentication_required].include?(evaluation.reason) &&
|
|
442
|
+
!pointer_fields_from_entry(class_name, op, client).empty?
|
|
443
|
+
"CLP requires user identity (pointerFields=#{pointer_fields_from_entry(class_name, op, client).inspect}) " \
|
|
444
|
+
"but #{scope} has no user_id."
|
|
445
|
+
else
|
|
446
|
+
"CLP refuses #{op} on '#{class_name}' for #{scope}."
|
|
447
|
+
end
|
|
448
|
+
raise Denied.new(class_name, op, message)
|
|
449
|
+
end
|
|
450
|
+
|
|
451
|
+
# A `$match` predicate keeping only rows whose pointer fields reference
|
|
452
|
+
# `user_id`, in Mongo storage form. A single pointer is stored as
|
|
453
|
+
# `_p_<field>: "_User$<id>"`; an array of pointers is stored under the
|
|
454
|
+
# bare field name as `{ __type, className, objectId }` elements. Both
|
|
455
|
+
# forms are matched, ORed across fields, as Parse Server's
|
|
456
|
+
# `addPointerPermissions` does.
|
|
457
|
+
#
|
|
458
|
+
# @param pointer_fields [Array<String>]
|
|
459
|
+
# @param user_id [String]
|
|
460
|
+
# @return [Hash] a `$match` predicate.
|
|
461
|
+
def pointer_fields_predicate(pointer_fields, user_id)
|
|
462
|
+
uid = user_id.to_s
|
|
463
|
+
storage = "#{Parse::Model::CLASS_USER}$#{uid}"
|
|
464
|
+
clauses = Array(pointer_fields).flat_map do |field|
|
|
465
|
+
f = field.to_s
|
|
466
|
+
[
|
|
467
|
+
{ "_p_#{f}" => storage },
|
|
468
|
+
{ f => { "$elemMatch" => { "className" => Parse::Model::CLASS_USER, "objectId" => uid } } },
|
|
469
|
+
]
|
|
470
|
+
end
|
|
471
|
+
{ "$or" => clauses }
|
|
472
|
+
end
|
|
473
|
+
|
|
316
474
|
def filter_by_pointer_fields(documents, pointer_fields, user_id)
|
|
317
475
|
return documents if pointer_fields.nil? || pointer_fields.empty?
|
|
318
476
|
return [] if user_id.nil? || user_id.to_s.empty?
|
|
@@ -373,6 +531,54 @@ module Parse
|
|
|
373
531
|
|
|
374
532
|
private
|
|
375
533
|
|
|
534
|
+
# Collect one field set per protectedFields group that applies to the
|
|
535
|
+
# caller, mirroring Parse Server's `DatabaseController.addProtectedFields`.
|
|
536
|
+
# Only groups PRESENT in the map participate. A missing `"*"` key does
|
|
537
|
+
# not mean "nothing protected": a map of `{ "role:Restricted" => ["secret"] }`
|
|
538
|
+
# still strips `secret` for a member of `Restricted`. The final strip set
|
|
539
|
+
# is the intersection of the applicable sets, so membership in a group
|
|
540
|
+
# with a narrower (or empty) list relaxes protection.
|
|
541
|
+
#
|
|
542
|
+
# Groups:
|
|
543
|
+
# - `"*"` applies to everyone.
|
|
544
|
+
# - `"authenticated"` applies when the claim set carries a concrete user id.
|
|
545
|
+
# - `"role:<name>"` applies for each role claim.
|
|
546
|
+
# - `"<userId>"` applies for the caller's own user id.
|
|
547
|
+
# - `"userField:<field>"` entries are per-object pointer rules that
|
|
548
|
+
# Parse Server evaluates against each row. They are skipped here,
|
|
549
|
+
# which can only over-protect, never under-protect.
|
|
550
|
+
#
|
|
551
|
+
# Role entries apply even when the claim set has no user id (an
|
|
552
|
+
# `acl_role:` scope). Parse Server has no REST equivalent of a
|
|
553
|
+
# role-only caller, and the SDK treats a granted role claim as held.
|
|
554
|
+
#
|
|
555
|
+
# @return [Array<Set<String>>]
|
|
556
|
+
def protected_sets_for(protected_map, claim_set)
|
|
557
|
+
lookup = lambda do |key|
|
|
558
|
+
value = protected_map[key]
|
|
559
|
+
value = protected_map[key.to_sym] if value.nil?
|
|
560
|
+
value.nil? ? nil : Set.new(Array(value).map(&:to_s))
|
|
561
|
+
end
|
|
562
|
+
|
|
563
|
+
sets = []
|
|
564
|
+
public_set = lookup.call("*")
|
|
565
|
+
sets << public_set if public_set
|
|
566
|
+
|
|
567
|
+
user_claims = claim_set.map(&:to_s).reject { |c| c == "*" || c.start_with?("role:") }
|
|
568
|
+
if user_claims.any?
|
|
569
|
+
auth_set = lookup.call("authenticated")
|
|
570
|
+
sets << auth_set if auth_set
|
|
571
|
+
end
|
|
572
|
+
|
|
573
|
+
claim_set.each do |claim|
|
|
574
|
+
key = claim.to_s
|
|
575
|
+
next if key == "*" || key == "authenticated" || key.start_with?("userField:")
|
|
576
|
+
set = lookup.call(key)
|
|
577
|
+
sets << set if set
|
|
578
|
+
end
|
|
579
|
+
sets
|
|
580
|
+
end
|
|
581
|
+
|
|
376
582
|
# Always returns a {CacheEntry}. On schema-fetch failure (network
|
|
377
583
|
# error, unsuccessful response, raised exception, missing client)
|
|
378
584
|
# the entry has `kind: :unresolvable` and is held for
|
|
@@ -400,7 +606,7 @@ module Parse
|
|
|
400
606
|
unresolvable_entry
|
|
401
607
|
else
|
|
402
608
|
begin
|
|
403
|
-
response = resolved_client
|
|
609
|
+
response = fetch_schema_response(resolved_client, class_key)
|
|
404
610
|
if response&.success?
|
|
405
611
|
schema = response.result || {}
|
|
406
612
|
clp = schema["classLevelPermissions"] || {}
|
|
@@ -418,6 +624,37 @@ module Parse
|
|
|
418
624
|
entry
|
|
419
625
|
end
|
|
420
626
|
|
|
627
|
+
# Fetch a class schema with the master key, explicitly.
|
|
628
|
+
#
|
|
629
|
+
# `GET /schemas/<Class>` is master-key-only on Parse Server. A plain
|
|
630
|
+
# `client.schema(name)` call inherits the request layer's auth
|
|
631
|
+
# resolution, so inside `Parse.with_session(token)` (or on a client in
|
|
632
|
+
# `client_mode`) it sends the user's session token instead of the master
|
|
633
|
+
# key. Parse Server refuses that, the entry is recorded as
|
|
634
|
+
# `:unresolvable`, and every scoped mongo-direct read in the block is
|
|
635
|
+
# denied by CLP (and the denial is negatively cached). Passing
|
|
636
|
+
# `use_master_key: true` makes the request layer skip both the ambient
|
|
637
|
+
# and the client-bound session token. A client that holds no master key
|
|
638
|
+
# still cannot read schemas, which stays fail-closed.
|
|
639
|
+
#
|
|
640
|
+
# Objects that only implement `#schema` (test doubles, custom schema
|
|
641
|
+
# sources installed via {.schema_client}), and clients whose `#schema`
|
|
642
|
+
# was overridden on the instance, keep the old call.
|
|
643
|
+
def fetch_schema_response(resolved_client, class_key)
|
|
644
|
+
stock_schema = begin
|
|
645
|
+
resolved_client.method(:schema).owner == Parse::API::Schema
|
|
646
|
+
rescue NameError
|
|
647
|
+
false
|
|
648
|
+
end
|
|
649
|
+
if resolved_client.is_a?(Parse::Client) && stock_schema
|
|
650
|
+
safe = Parse::API::PathSegment.identifier!(class_key, kind: "class name")
|
|
651
|
+
resolved_client.request(:get, "schemas/#{safe}",
|
|
652
|
+
opts: { cache: false, use_master_key: true })
|
|
653
|
+
else
|
|
654
|
+
resolved_client.schema(class_key)
|
|
655
|
+
end
|
|
656
|
+
end
|
|
657
|
+
|
|
421
658
|
def unresolvable_entry
|
|
422
659
|
CacheEntry.new(kind: :unresolvable, clp: nil, fetched_at: monotonic_now)
|
|
423
660
|
end
|
|
@@ -514,15 +751,36 @@ module Parse
|
|
|
514
751
|
s != "*" && !s.start_with?("role:")
|
|
515
752
|
end
|
|
516
753
|
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
754
|
+
# Delete protected keys from one row. Mongo stores a pointer column as
|
|
755
|
+
# `_p_<field>`, so the storage form of a protected pointer is removed
|
|
756
|
+
# along with the Parse-form key.
|
|
757
|
+
def strip_top_level!(doc, strip_set)
|
|
758
|
+
strip_set.each do |k|
|
|
759
|
+
key = k.to_s
|
|
760
|
+
doc.delete(key)
|
|
761
|
+
doc.delete(key.to_sym)
|
|
762
|
+
doc.delete("_p_#{key}")
|
|
763
|
+
end
|
|
764
|
+
doc
|
|
765
|
+
end
|
|
766
|
+
|
|
767
|
+
def merge_default_protected_fields(class_name, stored_map)
|
|
768
|
+
defaults = default_protected_fields
|
|
769
|
+
extra = defaults.is_a?(Hash) ? (defaults[class_name.to_s] || defaults[class_name.to_s.to_sym]) : nil
|
|
770
|
+
return stored_map if extra.nil? || extra.empty?
|
|
771
|
+
merged = {}
|
|
772
|
+
(stored_map || {}).each { |k, v| merged[k.to_s] = Array(v).map(&:to_s) }
|
|
773
|
+
extra.each do |k, v|
|
|
774
|
+
key = k.to_s
|
|
775
|
+
merged[key] = (Array(merged[key]) + Array(v).map(&:to_s)).uniq
|
|
524
776
|
end
|
|
525
|
-
|
|
777
|
+
merged
|
|
778
|
+
end
|
|
779
|
+
|
|
780
|
+
def pointer_fields_from_entry(class_name, op, client)
|
|
781
|
+
entry = fetch(class_name, client: client)
|
|
782
|
+
return [] unless entry.kind == :cached_clp
|
|
783
|
+
pointer_fields_from(entry.clp, op)
|
|
526
784
|
end
|
|
527
785
|
|
|
528
786
|
def any_pointer_matches?(doc, pointer_fields, user_id)
|
data/lib/parse/console.rb
CHANGED
|
@@ -46,8 +46,8 @@ module Parse
|
|
|
46
46
|
# puts "[#{event}] #{obj.title} (#{obj.id})"
|
|
47
47
|
# end
|
|
48
48
|
#
|
|
49
|
-
# @example
|
|
50
|
-
# Parse.
|
|
49
|
+
# @example Outside any session block (master key on a server client)
|
|
50
|
+
# Parse.watch(JobRun)
|
|
51
51
|
#
|
|
52
52
|
# @param klass [Class] a Parse::Object subclass.
|
|
53
53
|
# @param where [Hash] optional query constraints.
|
|
@@ -116,7 +116,64 @@ module Parse
|
|
|
116
116
|
# {Parse::Middleware::BodyBuilder::REDACTED_HEADERS}.
|
|
117
117
|
class Voyage < Provider
|
|
118
118
|
class AuthenticationError < Error; end
|
|
119
|
-
|
|
119
|
+
# A 4xx the provider refused outright. Carries the HTTP status and
|
|
120
|
+
# the provider's own error text (bounded and terminal-sanitized) so
|
|
121
|
+
# callers can tell a request that was merely too large from one that
|
|
122
|
+
# is malformed.
|
|
123
|
+
class BadRequestError < Error
|
|
124
|
+
# Phrases that positively identify a request rejected for its
|
|
125
|
+
# SIZE (too many inputs or too many tokens summed across the
|
|
126
|
+
# batch). Matched case-insensitively against the provider's error
|
|
127
|
+
# text. Kept deliberately narrow: an unrelated 400 (bad parameter,
|
|
128
|
+
# invalid JSON) must never be mistaken for a size error, because
|
|
129
|
+
# the response to a size error is to split and resend.
|
|
130
|
+
#
|
|
131
|
+
# Voyage documents these 400 causes as "Batch size is too large"
|
|
132
|
+
# and "Total number of tokens in the batch exceeds the limit"; its
|
|
133
|
+
# messages read like "The max allowed tokens per submitted batch
|
|
134
|
+
# is 120000" and "The batch size limit is 128".
|
|
135
|
+
REQUEST_TOO_LARGE_PATTERNS = [
|
|
136
|
+
/batch size is too large/i,
|
|
137
|
+
/batch size limit/i,
|
|
138
|
+
/total number of tokens in the batch exceeds/i,
|
|
139
|
+
/max(?:imum)? allowed tokens per (?:submitted )?batch/i,
|
|
140
|
+
].freeze
|
|
141
|
+
|
|
142
|
+
# Phrases identifying ONE input that is longer than the model's
|
|
143
|
+
# context window ("Number of tokens in an example exceeds the
|
|
144
|
+
# context length"). Splitting the batch cannot fix this.
|
|
145
|
+
INPUT_TOO_LONG_PATTERNS = [
|
|
146
|
+
/tokens in an example exceeds the context length/i,
|
|
147
|
+
].freeze
|
|
148
|
+
|
|
149
|
+
# @return [Integer, nil] the HTTP status the provider returned.
|
|
150
|
+
attr_reader :status
|
|
151
|
+
# @return [String, nil] the provider's error text, truncated and
|
|
152
|
+
# sanitized for terminals and logs.
|
|
153
|
+
attr_reader :detail
|
|
154
|
+
|
|
155
|
+
def initialize(message = nil, status: nil, detail: nil)
|
|
156
|
+
super(message)
|
|
157
|
+
@status = status
|
|
158
|
+
@detail = detail
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
# @return [Boolean] true when the provider rejected the request
|
|
162
|
+
# because the batch (input count or summed tokens) is too large,
|
|
163
|
+
# so resending it in smaller pieces can succeed.
|
|
164
|
+
def request_too_large?
|
|
165
|
+
return true if @status == 413
|
|
166
|
+
return false if @detail.nil? || input_too_long?
|
|
167
|
+
REQUEST_TOO_LARGE_PATTERNS.any? { |re| re.match?(@detail) }
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
# @return [Boolean] true when a single input exceeds the model's
|
|
171
|
+
# context length.
|
|
172
|
+
def input_too_long?
|
|
173
|
+
return false if @detail.nil?
|
|
174
|
+
INPUT_TOO_LONG_PATTERNS.any? { |re| re.match?(@detail) }
|
|
175
|
+
end
|
|
176
|
+
end
|
|
120
177
|
class RateLimitError < Error; end
|
|
121
178
|
class TransientError < Error; end
|
|
122
179
|
|
|
@@ -266,10 +323,31 @@ module Parse
|
|
|
266
323
|
# at most this many documents, and this many chunks summed
|
|
267
324
|
# across them.
|
|
268
325
|
# The endpoint also caps a request at 120k tokens summed across
|
|
269
|
-
# every document, which the SDK
|
|
326
|
+
# every document ({CONTEXT_MAX_REQUEST_TOKENS}), which the SDK can
|
|
327
|
+
# only estimate: it has no tokenizer.
|
|
270
328
|
MAX_CONTEXT_DOCUMENTS = 1_000
|
|
271
329
|
MAX_CONTEXT_CHUNKS = 16_000
|
|
272
330
|
|
|
331
|
+
# Voyage's cap on input tokens summed across every document in one
|
|
332
|
+
# contextualized request.
|
|
333
|
+
CONTEXT_MAX_REQUEST_TOKENS = 120_000
|
|
334
|
+
|
|
335
|
+
# Bytes per token assumed when ESTIMATING a request's input tokens
|
|
336
|
+
# for packing. Real tokenizers average closer to four bytes per token
|
|
337
|
+
# for English text, so dividing by three over-estimates and packs
|
|
338
|
+
# conservatively. This is an estimate, not a token count: the SDK has
|
|
339
|
+
# no tokenizer. When the estimate still lets an over-limit request
|
|
340
|
+
# through, the provider's size rejection is handled by splitting the
|
|
341
|
+
# request (see {CONTEXT_MAX_SPLIT_DEPTH}).
|
|
342
|
+
CONTEXT_ESTIMATED_BYTES_PER_TOKEN = 3
|
|
343
|
+
|
|
344
|
+
# How many times one contextualized request may be halved after the
|
|
345
|
+
# provider rejects it as too large. Ten halvings take the largest
|
|
346
|
+
# allowed request ({MAX_CONTEXT_DOCUMENTS} documents) down to single
|
|
347
|
+
# documents, so this bounds the retries without stopping short of a
|
|
348
|
+
# one-document request.
|
|
349
|
+
CONTEXT_MAX_SPLIT_DEPTH = 10
|
|
350
|
+
|
|
273
351
|
# Default `embed_batch_size` for {CONTEXTUALIZED_MODELS}. Each
|
|
274
352
|
# string is a whole document there, so 128 paragraph-sized inputs
|
|
275
353
|
# would overrun the 120k-token request cap; 32 leaves room for
|
|
@@ -704,25 +782,83 @@ module Parse
|
|
|
704
782
|
# Embed documents through the contextualized endpoint and return one
|
|
705
783
|
# Array of chunk vectors per document, in input order.
|
|
706
784
|
#
|
|
707
|
-
#
|
|
708
|
-
#
|
|
709
|
-
#
|
|
710
|
-
#
|
|
711
|
-
#
|
|
712
|
-
#
|
|
785
|
+
# This is robust adaptation, not exact token counting. Whole
|
|
786
|
+
# documents are packed into requests that stay within every limit
|
|
787
|
+
# the SDK can check or estimate:
|
|
788
|
+
#
|
|
789
|
+
# * at most {MAX_CONTEXT_DOCUMENTS} documents;
|
|
790
|
+
# * a chunk count whose response stays within {MAX_RESPONSE_BYTES}
|
|
791
|
+
# (sized by the vectors coming back);
|
|
792
|
+
# * an ESTIMATED input-token total within
|
|
793
|
+
# {CONTEXT_MAX_REQUEST_TOKENS} (see
|
|
794
|
+
# {CONTEXT_ESTIMATED_BYTES_PER_TOKEN}).
|
|
795
|
+
#
|
|
796
|
+
# A document is never split across requests. A document that alone
|
|
797
|
+
# exceeds a budget is sent on its own (its response allowance is
|
|
798
|
+
# sized to its chunk count). If the provider still rejects a request
|
|
799
|
+
# as too large, that request is halved by document and each half
|
|
800
|
+
# resent (see {#embed_contextualized_group}), so a mis-estimate costs
|
|
801
|
+
# extra requests rather than failing the batch.
|
|
713
802
|
def embed_contextualized(documents, input_type, wire_input_type)
|
|
714
|
-
|
|
715
|
-
chunks_per_request = [MAX_RESPONSE_BYTES / per_vector, 1].max
|
|
803
|
+
chunks_per_request = [MAX_RESPONSE_BYTES / response_bytes_per_vector, 1].max
|
|
716
804
|
groups = []
|
|
717
|
-
documents.
|
|
805
|
+
documents.each_with_index do |doc, i|
|
|
718
806
|
last = groups.last
|
|
719
|
-
|
|
720
|
-
|
|
807
|
+
tokens = estimate_input_tokens(doc)
|
|
808
|
+
if last &&
|
|
809
|
+
last[:docs].length < MAX_CONTEXT_DOCUMENTS &&
|
|
810
|
+
last[:chunks] + doc.length <= chunks_per_request &&
|
|
811
|
+
last[:tokens] + tokens <= CONTEXT_MAX_REQUEST_TOKENS
|
|
812
|
+
last[:docs] << doc
|
|
813
|
+
last[:chunks] += doc.length
|
|
814
|
+
last[:tokens] += tokens
|
|
721
815
|
else
|
|
722
|
-
groups << [doc]
|
|
816
|
+
groups << { docs: [doc], offset: i, chunks: doc.length, tokens: tokens }
|
|
723
817
|
end
|
|
724
818
|
end
|
|
725
|
-
groups.flat_map
|
|
819
|
+
groups.flat_map do |group|
|
|
820
|
+
embed_contextualized_group(group[:docs], group[:offset], input_type, wire_input_type, 0)
|
|
821
|
+
end
|
|
822
|
+
end
|
|
823
|
+
|
|
824
|
+
# Send one packed group, halving it by document and resending each
|
|
825
|
+
# half when the provider rejects it as too large. Only an error that
|
|
826
|
+
# {BadRequestError#request_too_large?} positively identifies is split;
|
|
827
|
+
# any other error propagates unchanged. Results are concatenated in
|
|
828
|
+
# input order, so alignment with the caller's documents is preserved
|
|
829
|
+
# however the group was divided.
|
|
830
|
+
#
|
|
831
|
+
# @param offset [Integer] index of `documents.first` in the caller's
|
|
832
|
+
# original input, for error messages.
|
|
833
|
+
# @param depth [Integer] halvings so far, bounded by
|
|
834
|
+
# {CONTEXT_MAX_SPLIT_DEPTH}.
|
|
835
|
+
def embed_contextualized_group(documents, offset, input_type, wire_input_type, depth)
|
|
836
|
+
embed_contextualized_request(documents, input_type, wire_input_type)
|
|
837
|
+
rescue BadRequestError => e
|
|
838
|
+
raise unless e.request_too_large?
|
|
839
|
+
|
|
840
|
+
if documents.length == 1
|
|
841
|
+
raise BadRequestError.new(
|
|
842
|
+
"Parse::Embeddings::Voyage: document #{offset} (#{documents.first.length} chunk(s), " \
|
|
843
|
+
"about #{estimate_input_tokens(documents.first)} estimated tokens) is too large for a " \
|
|
844
|
+
"single #{@model} request even on its own. Split that document into fewer or shorter " \
|
|
845
|
+
"chunks, or into several documents. Provider said: #{e.detail || e.message}",
|
|
846
|
+
status: e.status, detail: e.detail,
|
|
847
|
+
)
|
|
848
|
+
end
|
|
849
|
+
raise if depth >= CONTEXT_MAX_SPLIT_DEPTH
|
|
850
|
+
|
|
851
|
+
mid = documents.length / 2
|
|
852
|
+
embed_contextualized_group(documents[0...mid], offset, input_type, wire_input_type, depth + 1) +
|
|
853
|
+
embed_contextualized_group(documents[mid..], offset + mid, input_type, wire_input_type, depth + 1)
|
|
854
|
+
end
|
|
855
|
+
|
|
856
|
+
# Conservative ESTIMATE of a document's input tokens: its chunks'
|
|
857
|
+
# bytes divided by {CONTEXT_ESTIMATED_BYTES_PER_TOKEN}, rounded up.
|
|
858
|
+
# Used only for packing, never to refuse a document.
|
|
859
|
+
def estimate_input_tokens(chunks)
|
|
860
|
+
bytes = chunks.sum(&:bytesize)
|
|
861
|
+
(bytes + CONTEXT_ESTIMATED_BYTES_PER_TOKEN - 1) / CONTEXT_ESTIMATED_BYTES_PER_TOKEN
|
|
726
862
|
end
|
|
727
863
|
|
|
728
864
|
# @return [Integer] the planning estimate for one vector's JSON size.
|
|
@@ -1037,11 +1173,39 @@ module Parse
|
|
|
1037
1173
|
sleep(backoff_seconds(attempts))
|
|
1038
1174
|
next
|
|
1039
1175
|
end
|
|
1040
|
-
|
|
1041
|
-
|
|
1176
|
+
detail = provider_error_detail(response.body)
|
|
1177
|
+
message = +"Parse::Embeddings::Voyage: #{status} from POST /#{path}."
|
|
1178
|
+
message << " #{detail}" if detail
|
|
1179
|
+
raise BadRequestError.new(message, status: status, detail: detail)
|
|
1042
1180
|
end
|
|
1043
1181
|
end
|
|
1044
1182
|
|
|
1183
|
+
# Longest provider error text kept on a {BadRequestError}.
|
|
1184
|
+
MAX_ERROR_DETAIL_CHARS = 500
|
|
1185
|
+
|
|
1186
|
+
# Extract the provider's error text from a non-2xx body, bounded and
|
|
1187
|
+
# made safe to print. Voyage returns `{ "detail": "..." }`; a
|
|
1188
|
+
# `message`, a string `error`, or a nested `error.message` is accepted
|
|
1189
|
+
# too. Returns nil for an empty, oversized, or unparseable body rather
|
|
1190
|
+
# than raising, since the status code alone still describes the
|
|
1191
|
+
# failure.
|
|
1192
|
+
def provider_error_detail(body)
|
|
1193
|
+
s = body.to_s
|
|
1194
|
+
return nil if s.empty? || s.bytesize > 64 * 1024
|
|
1195
|
+
parsed = JSON.parse(s, max_nesting: 8)
|
|
1196
|
+
return nil unless parsed.is_a?(Hash)
|
|
1197
|
+
text = parsed["detail"] || parsed["message"]
|
|
1198
|
+
err = parsed["error"]
|
|
1199
|
+
text ||= err["message"] if err.is_a?(Hash)
|
|
1200
|
+
text ||= err if err.is_a?(String)
|
|
1201
|
+
text = text.to_json unless text.nil? || text.is_a?(String)
|
|
1202
|
+
return nil if text.nil? || text.strip.empty?
|
|
1203
|
+
text = text[0, MAX_ERROR_DETAIL_CHARS]
|
|
1204
|
+
defined?(Parse::TerminalSafe) ? Parse::TerminalSafe.sanitize_line(text) : text
|
|
1205
|
+
rescue JSON::ParserError
|
|
1206
|
+
nil
|
|
1207
|
+
end
|
|
1208
|
+
|
|
1045
1209
|
def parse_json_body!(body, max_bytes = MAX_RESPONSE_BYTES)
|
|
1046
1210
|
s = body.to_s
|
|
1047
1211
|
if s.bytesize > max_bytes
|
|
@@ -29,6 +29,9 @@ module Parse
|
|
|
29
29
|
string: ::GraphQL::Types::String,
|
|
30
30
|
integer: ::GraphQL::Types::Int,
|
|
31
31
|
float: ::GraphQL::Types::Float,
|
|
32
|
+
# A :number property may hold fractional values, which GraphQL's
|
|
33
|
+
# 32-bit Int cannot represent, so it is exposed as Float.
|
|
34
|
+
number: ::GraphQL::Types::Float,
|
|
32
35
|
boolean: ::GraphQL::Types::Boolean,
|
|
33
36
|
date: ::GraphQL::Types::ISO8601DateTime,
|
|
34
37
|
timezone: ::GraphQL::Types::String,
|