parse-stack-next 5.7.5 → 5.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +856 -0
- data/README.md +15 -4
- data/docs/TEST_SERVER.md +2 -2
- data/docs/acl_clp_guide.md +7 -0
- data/docs/atlas_vector_search_guide.md +190 -14
- data/docs/client_sdk_guide.md +11 -0
- data/docs/mcp_guide.md +318 -6
- data/docs/mongodb_direct_guide.md +27 -0
- data/docs/usage_guide.md +38 -0
- data/docs/webhooks_guide.md +74 -17
- data/lib/parse/acl_scope.rb +159 -41
- data/lib/parse/agent/approval_gate.rb +0 -0
- data/lib/parse/agent/constraint_translator.rb +42 -15
- data/lib/parse/agent/describe.rb +3 -1
- data/lib/parse/agent/field_names.rb +53 -0
- data/lib/parse/agent/field_policy.rb +74 -0
- data/lib/parse/agent/mcp_deployments.rb +426 -0
- data/lib/parse/agent/mcp_rack_app.rb +424 -45
- data/lib/parse/agent/mcp_server.rb +23 -1
- data/lib/parse/agent/mcp_subscriptions.rb +124 -6
- data/lib/parse/agent/metadata_registry.rb +67 -8
- data/lib/parse/agent/prompt_hardening.rb +9 -3
- data/lib/parse/agent/tools.rb +378 -29
- data/lib/parse/agent.rb +93 -1
- data/lib/parse/api/batch.rb +10 -1
- data/lib/parse/api/schema.rb +23 -4
- data/lib/parse/api/sessions.rb +6 -2
- data/lib/parse/api/users.rb +88 -14
- data/lib/parse/atlas_search/protected_paths.rb +236 -0
- data/lib/parse/atlas_search.rb +95 -23
- data/lib/parse/authorization.rb +54 -1
- data/lib/parse/client/batch.rb +231 -35
- data/lib/parse/client/body_builder.rb +21 -0
- data/lib/parse/client/caching.rb +371 -27
- data/lib/parse/client/request.rb +26 -14
- data/lib/parse/client/response.rb +49 -6
- data/lib/parse/client.rb +201 -38
- data/lib/parse/clp_scope.rb +281 -23
- data/lib/parse/console.rb +2 -2
- data/lib/parse/embeddings/voyage.rb +181 -17
- data/lib/parse/graphql/type_generator.rb +3 -0
- data/lib/parse/model/acl.rb +119 -21
- data/lib/parse/model/associations/belongs_to.rb +25 -3
- data/lib/parse/model/associations/collection_proxy.rb +138 -17
- data/lib/parse/model/associations/has_many.rb +38 -9
- data/lib/parse/model/associations/has_one.rb +3 -1
- data/lib/parse/model/associations/pointer_collection_proxy.rb +109 -17
- data/lib/parse/model/associations/relation_collection_proxy.rb +134 -28
- data/lib/parse/model/bytes.rb +13 -5
- data/lib/parse/model/classes/role.rb +72 -0
- data/lib/parse/model/classes/session.rb +43 -0
- data/lib/parse/model/classes/user.rb +78 -3
- data/lib/parse/model/core/actions.rb +269 -67
- data/lib/parse/model/core/builder.rb +100 -8
- data/lib/parse/model/core/create_lock.rb +27 -2
- data/lib/parse/model/core/describe.rb +2 -0
- data/lib/parse/model/core/fetching.rb +21 -3
- data/lib/parse/model/core/pluralized_aliases.rb +8 -4
- data/lib/parse/model/core/properties.rb +488 -39
- data/lib/parse/model/core/querying.rb +7 -0
- data/lib/parse/model/core/schema.rb +5 -3
- data/lib/parse/model/core/search_indexing.rb +63 -0
- data/lib/parse/model/core/vector_searchable.rb +35 -6
- data/lib/parse/model/file.rb +9 -2
- data/lib/parse/model/geopoint.rb +61 -13
- data/lib/parse/model/model.rb +160 -9
- data/lib/parse/model/object.rb +265 -17
- data/lib/parse/model/phone.rb +54 -5
- data/lib/parse/model/pointer.rb +40 -6
- data/lib/parse/mongodb.rb +170 -60
- data/lib/parse/pipeline_security.rb +415 -26
- data/lib/parse/query/constraint.rb +30 -0
- data/lib/parse/query/constraints.rb +58 -32
- data/lib/parse/query/cursor.rb +3 -1
- data/lib/parse/query/operation.rb +62 -8
- data/lib/parse/query/ordering.rb +34 -6
- data/lib/parse/query.rb +1100 -134
- data/lib/parse/retrieval/agent_tool.rb +290 -17
- data/lib/parse/retrieval/benchmark.rb +149 -0
- data/lib/parse/retrieval/profiles.rb +320 -0
- data/lib/parse/retrieval/retriever.rb +10 -1
- data/lib/parse/retrieval.rb +2 -0
- data/lib/parse/schema/search_index_migrator.rb +23 -5
- data/lib/parse/schema.rb +74 -18
- data/lib/parse/stack/tasks.rb +6 -4
- data/lib/parse/stack/version.rb +1 -1
- data/lib/parse/stack.rb +72 -14
- data/lib/parse/two_factor_auth/user_extension.rb +14 -2
- data/lib/parse/two_factor_auth.rb +11 -0
- data/lib/parse/vector_search/hybrid.rb +36 -18
- data/lib/parse/vector_search/index_definition.rb +237 -0
- data/lib/parse/vector_search.rb +46 -17
- data/lib/parse/webhooks/payload.rb +93 -6
- data/lib/parse/webhooks/replay_protection.rb +58 -20
- data/lib/parse/webhooks.rb +412 -40
- metadata +8 -1
|
@@ -57,12 +57,20 @@ module Parse
|
|
|
57
57
|
# truncation is never silent. Pass `max_total_tokens: 0` to disable.
|
|
58
58
|
DEFAULT_MAX_TOTAL_TOKENS = 20_000
|
|
59
59
|
|
|
60
|
+
# Longest `query` accepted. A search query is a short natural-language
|
|
61
|
+
# request; the bound keeps one call from sending a body-sized string to
|
|
62
|
+
# the embedding provider and, under a reranking profile, pairing it
|
|
63
|
+
# with every candidate document.
|
|
64
|
+
MAX_QUERY_CHARS = 4_000
|
|
65
|
+
|
|
60
66
|
# @param agent [Parse::Agent]
|
|
61
67
|
# @param text_field [String, Symbol, nil] which embedded text source to
|
|
62
|
-
# chunk and return as `content`.
|
|
63
|
-
#
|
|
64
|
-
#
|
|
65
|
-
#
|
|
68
|
+
# chunk and return as `content`. Must name one of the class's declared
|
|
69
|
+
# embed sources that is also inside its `agent_fields` allowlist: an
|
|
70
|
+
# arbitrary field is refused so chunk `content` can't disclose a
|
|
71
|
+
# non-embedded field, and an embedded-but-hidden field is refused with
|
|
72
|
+
# `:field_denied` so it can't disclose a field the agent may not read.
|
|
73
|
+
# When omitted, it is inferred from the readable embed sources.
|
|
66
74
|
# @param max_chunks_per_document [Integer, nil] cap on chunks emitted per
|
|
67
75
|
# matched document (forwarded to the chunker).
|
|
68
76
|
# @param max_total_tokens [Integer, nil] ceiling on total returned
|
|
@@ -73,10 +81,36 @@ module Parse
|
|
|
73
81
|
# by objectId) instead of being duplicated on every chunk. When the
|
|
74
82
|
# token budget trims the result, `budget_truncated: true` and
|
|
75
83
|
# `budget_dropped: <n>` are added.
|
|
76
|
-
def semantic_search(agent,
|
|
84
|
+
def semantic_search(agent, **args)
|
|
85
|
+
started = monotonic_now
|
|
86
|
+
semantic_search_unobserved(agent, **args)
|
|
87
|
+
rescue StandardError => e
|
|
88
|
+
# Failures are observable too: one sanitized event naming the error
|
|
89
|
+
# class (never its message, which can echo input).
|
|
90
|
+
emit_failure_event(args, e, started)
|
|
91
|
+
raise
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
# @!visibility private
|
|
95
|
+
def emit_failure_event(args, error, started)
|
|
96
|
+
return unless defined?(ActiveSupport::Notifications)
|
|
97
|
+
payload = {
|
|
98
|
+
class_name: (args[:class_name] || args[:klass]).to_s,
|
|
99
|
+
profile: args[:profile]&.to_s,
|
|
100
|
+
error: error.class.name,
|
|
101
|
+
duration_ms: ((monotonic_now - started) * 1000).round(1),
|
|
102
|
+
}
|
|
103
|
+
ActiveSupport::Notifications.instrument("parse.retrieval.search", payload)
|
|
104
|
+
rescue StandardError
|
|
105
|
+
nil
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
# @!visibility private
|
|
109
|
+
def semantic_search_unobserved(agent, class_name: nil, query: nil, k: nil,
|
|
77
110
|
filter: nil, vector_filter: nil, text_field: nil,
|
|
78
111
|
chunk_size: nil, chunk_overlap: nil, chunk_by: nil,
|
|
79
112
|
max_chunks_per_document: nil, max_total_tokens: nil,
|
|
113
|
+
profile: nil,
|
|
80
114
|
# Back-compat / ergonomic aliases for direct callers:
|
|
81
115
|
# `klass:`/`class:` for class_name, and the chunker's
|
|
82
116
|
# own `size:`/`overlap:`/`by:` names.
|
|
@@ -93,14 +127,32 @@ module Parse
|
|
|
93
127
|
unless query.is_a?(String) && !query.strip.empty?
|
|
94
128
|
raise Parse::Agent::ValidationError, "semantic_search: `query` must be a non-empty String."
|
|
95
129
|
end
|
|
130
|
+
if query.length > MAX_QUERY_CHARS
|
|
131
|
+
raise Parse::Agent::ValidationError,
|
|
132
|
+
"semantic_search: `query` is #{query.length} characters; the limit is #{MAX_QUERY_CHARS}."
|
|
133
|
+
end
|
|
96
134
|
|
|
97
135
|
resolved_text_field = normalize_text_field!(text_field, klass)
|
|
136
|
+
# A named, server-configured retrieval profile (Parse::Retrieval::Profiles).
|
|
137
|
+
# Unknown names fail here, before any provider call.
|
|
138
|
+
prof = profile.nil? || profile.to_s.strip.empty? ? nil : Parse::Retrieval::Profiles.fetch!(profile)
|
|
98
139
|
|
|
99
140
|
# Reject reserved underscore keys at any depth, then enforce the
|
|
100
141
|
# per-class filter-field allowlist on top-level keys.
|
|
101
142
|
Parse::Retrieval.assert_no_underscore_keys!(filter) unless filter.nil?
|
|
102
143
|
Parse::Retrieval.assert_no_underscore_keys!(vector_filter) unless vector_filter.nil?
|
|
103
144
|
allowed = Parse::Agent::MetadataRegistry.searchable_filter_fields(cname).map(&:to_s)
|
|
145
|
+
# Filterable fields are also bounded by what the agent may read (the
|
|
146
|
+
# class `agent_fields` ceiling, narrowed by any per-agent `fields:`
|
|
147
|
+
# policy): filtering on a field the agent cannot read would reveal
|
|
148
|
+
# its value through which rows match. A `filter_fields` entry outside
|
|
149
|
+
# `agent_fields` is therefore never usable.
|
|
150
|
+
readable = Parse::Agent::MetadataRegistry.field_allowlist(cname)&.map(&:to_s)
|
|
151
|
+
if readable && !readable.empty?
|
|
152
|
+
allowed = allowed.select do |f|
|
|
153
|
+
readable.include?(Parse::Agent::MetadataRegistry.wire_field_names(cname, [f]).first)
|
|
154
|
+
end
|
|
155
|
+
end
|
|
104
156
|
assert_filter_fields_allowed!(filter, allowed)
|
|
105
157
|
assert_filter_fields_allowed!(vector_filter, allowed)
|
|
106
158
|
|
|
@@ -122,6 +174,39 @@ module Parse
|
|
|
122
174
|
score_quantize = (agent.permissions != :admin)
|
|
123
175
|
vector_field = Parse::Agent::MetadataRegistry.searchable_field(cname)
|
|
124
176
|
|
|
177
|
+
# Profile resolution: k is bounded by the profile's max_k; a reranking
|
|
178
|
+
# profile retrieves `rerank_candidates` and keeps `rerank_top_n` (or
|
|
179
|
+
# the effective k); hybrid settings come only from the profile.
|
|
180
|
+
effective_k = if prof
|
|
181
|
+
requested = k.to_i.positive? ? k.to_i : prof.k
|
|
182
|
+
clamp_k([requested, prof.max_k].min)
|
|
183
|
+
else
|
|
184
|
+
clamp_k(k)
|
|
185
|
+
end
|
|
186
|
+
reranker = nil
|
|
187
|
+
retrieve_k = effective_k
|
|
188
|
+
rerank_top_n = nil
|
|
189
|
+
if prof&.rerank?
|
|
190
|
+
reranker = Parse::Retrieval::BudgetedReranker.new(
|
|
191
|
+
Parse::Retrieval.reranker(prof.reranker), prof,
|
|
192
|
+
charge: ->(tokens) { charge_rerank_tokens!(agent, scope, tokens) },
|
|
193
|
+
)
|
|
194
|
+
# rerank_candidates is a hard budget: the caller's k can never
|
|
195
|
+
# raise how many documents are retrieved and sent to the
|
|
196
|
+
# reranker, so k is capped at it.
|
|
197
|
+
retrieve_k = prof.rerank_candidates
|
|
198
|
+
effective_k = [effective_k, retrieve_k].min
|
|
199
|
+
rerank_top_n = [prof.rerank_top_n || effective_k, effective_k].min
|
|
200
|
+
end
|
|
201
|
+
if prof
|
|
202
|
+
# Under a profile the response budget is mandatory: the caller can
|
|
203
|
+
# lower it but never raise or disable it (0 does not switch it off).
|
|
204
|
+
ceiling = prof.max_total_tokens || DEFAULT_MAX_TOTAL_TOKENS
|
|
205
|
+
requested = max_total_tokens.to_i
|
|
206
|
+
max_total_tokens = requested.positive? ? [requested, ceiling].min : ceiling
|
|
207
|
+
end
|
|
208
|
+
started = monotonic_now
|
|
209
|
+
|
|
125
210
|
# with_precharged: the cap was charged above with per-tenant
|
|
126
211
|
# identity (or deliberately skipped for trusted admin agents) —
|
|
127
212
|
# suppress the generic query-embed charge inside
|
|
@@ -133,7 +218,10 @@ module Parse
|
|
|
133
218
|
klass: klass,
|
|
134
219
|
field: vector_field,
|
|
135
220
|
text_field: resolved_text_field,
|
|
136
|
-
k:
|
|
221
|
+
k: retrieve_k,
|
|
222
|
+
hybrid: prof&.hybrid ? hybrid_config_for(prof, klass) : nil,
|
|
223
|
+
rerank: reranker,
|
|
224
|
+
rerank_top_n: rerank_top_n,
|
|
137
225
|
filter: filter,
|
|
138
226
|
vector_filter: vector_filter,
|
|
139
227
|
chunker: build_chunker(chunk_size, chunk_overlap, chunk_by, max_chunks_per_document),
|
|
@@ -147,7 +235,7 @@ module Parse
|
|
|
147
235
|
# Token budget (B4): trim the score-ordered chunk list before
|
|
148
236
|
# building the envelope so `documents` only carries parents whose
|
|
149
237
|
# chunks survived.
|
|
150
|
-
kept, dropped = apply_token_budget(chunks, resolve_token_budget(max_total_tokens))
|
|
238
|
+
kept, dropped = apply_token_budget(chunks, resolve_token_budget(max_total_tokens), strict: !prof.nil?)
|
|
151
239
|
|
|
152
240
|
# Source dedup (A3): a document's (projected) source record is
|
|
153
241
|
# identical across all its chunks. Hoist it into a `documents` map
|
|
@@ -170,9 +258,113 @@ module Parse
|
|
|
170
258
|
envelope[:budget_truncated] = true
|
|
171
259
|
envelope[:budget_dropped] = dropped
|
|
172
260
|
end
|
|
261
|
+
if prof
|
|
262
|
+
envelope[:profile] = prof.name
|
|
263
|
+
if reranker&.stats&.dig(:fallback)
|
|
264
|
+
# Observable fallback: the result is in retrieval order, not
|
|
265
|
+
# reranked, and the caller is told why.
|
|
266
|
+
envelope[:rerank_fallback] = true
|
|
267
|
+
envelope[:rerank_fallback_reason] = reranker.stats[:fallback_reason]
|
|
268
|
+
end
|
|
269
|
+
end
|
|
270
|
+
emit_search_event(cname, prof, effective_k, retrieve_k, reranker, envelope, dropped, started)
|
|
173
271
|
envelope
|
|
174
272
|
end
|
|
175
273
|
|
|
274
|
+
# @!visibility private
|
|
275
|
+
# A profile's hybrid settings with the lexical branch restricted to the
|
|
276
|
+
# text sources the agent may read. Without this the lexical search runs
|
|
277
|
+
# over every field (`wildcard: "*"`), so which documents match, and
|
|
278
|
+
# their rank, could depend on a hidden field. Refused when the class
|
|
279
|
+
# has an allowlist and no readable text source.
|
|
280
|
+
def hybrid_config_for(prof, klass)
|
|
281
|
+
cfg = Marshal.load(Marshal.dump(prof.hybrid.to_h))
|
|
282
|
+
allowlist = Parse::Agent::MetadataRegistry.field_allowlist(klass.parse_class)
|
|
283
|
+
if allowlist.nil? || allowlist.empty?
|
|
284
|
+
# No allowlist: never fall back to `wildcard: "*"`, which would
|
|
285
|
+
# let every column (including CLP protectedFields) decide matches.
|
|
286
|
+
# Search the embedded text sources unless the profile names fields;
|
|
287
|
+
# Atlas search then refuses any named field protected for the caller.
|
|
288
|
+
lexical = (cfg[:lexical] || {}).dup
|
|
289
|
+
if Array(lexical[:fields]).empty?
|
|
290
|
+
lexical[:fields] = searchable_text_fields(klass).map { |f| Parse::Retrieval.send(:wire_name, klass, f) }
|
|
291
|
+
cfg[:lexical] = lexical
|
|
292
|
+
end
|
|
293
|
+
return cfg
|
|
294
|
+
end
|
|
295
|
+
lexical = (cfg[:lexical] || {}).dup
|
|
296
|
+
if lexical[:fields]
|
|
297
|
+
# Server-configured lexical fields are kept when the agent may read
|
|
298
|
+
# them (any readable field, not only embedding sources), and
|
|
299
|
+
# translated to their stored names.
|
|
300
|
+
readable_wire = allowlist.map(&:to_s) - Parse::Agent::MetadataRegistry::ALWAYS_KEEP_FIELDS
|
|
301
|
+
configured = Array(lexical[:fields]).map { |f| Parse::Retrieval.send(:wire_name, klass, f) }
|
|
302
|
+
lexical[:fields] = configured & readable_wire
|
|
303
|
+
else
|
|
304
|
+
# Unconfigured: search the readable embedded text sources.
|
|
305
|
+
readable = readable_text_fields(klass) || []
|
|
306
|
+
lexical[:fields] = readable.map { |f| Parse::Retrieval.send(:wire_name, klass, f) }
|
|
307
|
+
end
|
|
308
|
+
if lexical[:fields].empty?
|
|
309
|
+
# An empty list would mean `wildcard: "*"`, letting hidden fields
|
|
310
|
+
# decide matches; refuse instead.
|
|
311
|
+
raise text_field_denied(klass, Array(cfg.dig(:lexical, :fields)).first || searchable_text_fields(klass).first)
|
|
312
|
+
end
|
|
313
|
+
cfg[:lexical] = lexical
|
|
314
|
+
cfg
|
|
315
|
+
end
|
|
316
|
+
|
|
317
|
+
# @!visibility private
|
|
318
|
+
def monotonic_now
|
|
319
|
+
Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
320
|
+
end
|
|
321
|
+
|
|
322
|
+
# @!visibility private
|
|
323
|
+
# One sanitized `parse.retrieval.search` event per semantic_search
|
|
324
|
+
# call: profile, budgets, stage counts and timings, and estimated
|
|
325
|
+
# rerank usage. Never document text, field values, URLs, or
|
|
326
|
+
# credentials. Rerank tokens are the SDK's estimate
|
|
327
|
+
# (`tokens_estimated`), not provider-reported usage.
|
|
328
|
+
def emit_search_event(cname, prof, k, retrieve_k, reranker, envelope, dropped, started)
|
|
329
|
+
return unless defined?(ActiveSupport::Notifications)
|
|
330
|
+
total_ms = ((monotonic_now - started) * 1000).round(1)
|
|
331
|
+
rerank = reranker ? reranker.stats.dup : { used: false }
|
|
332
|
+
payload = {
|
|
333
|
+
class_name: cname,
|
|
334
|
+
profile: prof&.name,
|
|
335
|
+
hybrid: !prof&.hybrid.nil?,
|
|
336
|
+
k: k,
|
|
337
|
+
candidates: retrieve_k,
|
|
338
|
+
rerank: rerank,
|
|
339
|
+
chunks_returned: envelope[:count],
|
|
340
|
+
documents_returned: envelope[:documents].size,
|
|
341
|
+
budget_dropped: dropped,
|
|
342
|
+
duration_ms: total_ms,
|
|
343
|
+
retrieve_ms: (total_ms - (rerank[:duration_ms] || 0)).round(1),
|
|
344
|
+
}
|
|
345
|
+
ActiveSupport::Notifications.instrument("parse.retrieval.search", payload)
|
|
346
|
+
rescue StandardError
|
|
347
|
+
nil
|
|
348
|
+
end
|
|
349
|
+
|
|
350
|
+
# @!visibility private
|
|
351
|
+
# Charge estimated reranker tokens to the same per-tenant spend cap
|
|
352
|
+
# the query embedding uses (admin agents are exempt, as there). A
|
|
353
|
+
# transient cap hit surfaces as RateLimitExceeded; an impossible one
|
|
354
|
+
# as ValidationError, mirroring {#charge_spend_cap!}.
|
|
355
|
+
def charge_rerank_tokens!(agent, scope, tokens)
|
|
356
|
+
return if agent.permissions == :admin
|
|
357
|
+
tenant_id = scope && (scope[:value] || scope["value"])
|
|
358
|
+
Parse::Embeddings::SpendCap.charge!(tenant_id: tenant_id, tokens: tokens)
|
|
359
|
+
rescue Parse::Embeddings::SpendCap::Exceeded => e
|
|
360
|
+
if e.retry_after.nil?
|
|
361
|
+
raise Parse::Agent::ValidationError,
|
|
362
|
+
"semantic_search: reranking exceeds the spend cap " \
|
|
363
|
+
"(#{e.requested} tokens requested, limit #{e.limit}/#{e.window}s)."
|
|
364
|
+
end
|
|
365
|
+
raise Parse::Agent::RateLimitExceeded.new(retry_after: e.retry_after, limit: e.limit, window: e.window)
|
|
366
|
+
end
|
|
367
|
+
|
|
176
368
|
# @!visibility private
|
|
177
369
|
# Charge the estimated query-embedding token cost against the
|
|
178
370
|
# tenant's spend cap. The tenant key is the resolved tenant-scope
|
|
@@ -228,14 +420,40 @@ module Parse
|
|
|
228
420
|
# least the first chunk so a single oversize chunk still returns
|
|
229
421
|
# something (flagged truncated).
|
|
230
422
|
# @return [Array(Array<Chunk>, Integer)] [kept, dropped_count]
|
|
231
|
-
|
|
423
|
+
#
|
|
424
|
+
# The estimate covers the whole response, not only chunk text: each
|
|
425
|
+
# chunk's content plus, the first time a parent document appears, that
|
|
426
|
+
# document's serialized source record (it is hoisted into `documents`).
|
|
427
|
+
#
|
|
428
|
+
# `strict:` (a profile's mandatory budget) drops even the first chunk
|
|
429
|
+
# when it alone exceeds the budget; otherwise the first chunk is always
|
|
430
|
+
# kept so an oversized single result still returns something.
|
|
431
|
+
# Characters each returned chunk adds beyond its content and metadata:
|
|
432
|
+
# its key names, score, and `_source` provenance stamp.
|
|
433
|
+
CHUNK_OVERHEAD_CHARS = 160
|
|
434
|
+
# Characters the response envelope adds once (counts, profile,
|
|
435
|
+
# truncation flags, the `documents` map wrapper).
|
|
436
|
+
ENVELOPE_OVERHEAD_CHARS = 400
|
|
437
|
+
|
|
438
|
+
def apply_token_budget(chunks, budget, strict: false)
|
|
232
439
|
return [chunks, 0] if budget.nil? || chunks.empty?
|
|
233
|
-
|
|
440
|
+
# Every chunk carries metadata and per-chunk keys alongside its text,
|
|
441
|
+
# so a response of many tiny chunks is mostly overhead; count it.
|
|
442
|
+
total = (ENVELOPE_OVERHEAD_CHARS / 4.0).ceil
|
|
234
443
|
kept = []
|
|
444
|
+
seen_docs = {}
|
|
235
445
|
chunks.each do |chunk|
|
|
236
|
-
|
|
237
|
-
|
|
446
|
+
meta = chunk.respond_to?(:metadata) ? chunk.metadata : nil
|
|
447
|
+
meta_chars = meta.is_a?(Hash) ? (JSON.generate(meta).length rescue 0) : 0
|
|
448
|
+
est = ((chunk.content.to_s.length + meta_chars + CHUNK_OVERHEAD_CHARS) / 4.0).ceil
|
|
449
|
+
oid = chunk.respond_to?(:metadata) && chunk.metadata.is_a?(Hash) ? chunk.metadata[:object_id] : nil
|
|
450
|
+
if oid && !seen_docs.key?(oid) && chunk.respond_to?(:source) && chunk.source
|
|
451
|
+
doc_est = (JSON.generate(chunk.source).length / 4.0).ceil rescue 0
|
|
452
|
+
est += doc_est
|
|
453
|
+
end
|
|
454
|
+
break unless (kept.empty? && !strict) || total + est <= budget
|
|
238
455
|
kept << chunk
|
|
456
|
+
seen_docs[oid] = true if oid
|
|
239
457
|
total += est
|
|
240
458
|
end
|
|
241
459
|
[kept, chunks.length - kept.length]
|
|
@@ -329,22 +547,76 @@ module Parse
|
|
|
329
547
|
.flat_map(&:sources).map(&:to_s).uniq
|
|
330
548
|
end
|
|
331
549
|
|
|
550
|
+
# @!visibility private
|
|
551
|
+
# The embed text sources the agent may also READ: those whose wire
|
|
552
|
+
# column is inside the class's `agent_fields` allowlist. Chunk
|
|
553
|
+
# `content` (and any reranker input) is built from the chosen source's
|
|
554
|
+
# raw value BEFORE the per-record projection strips disallowed fields,
|
|
555
|
+
# so the source itself must be readable. With no allowlist every
|
|
556
|
+
# source is readable, as before.
|
|
557
|
+
#
|
|
558
|
+
# @return [Array<String>, nil] readable sources, or nil when the class
|
|
559
|
+
# declares no `agent_fields` allowlist.
|
|
560
|
+
def readable_text_fields(klass)
|
|
561
|
+
allowlist = Parse::Agent::MetadataRegistry.field_allowlist(klass.parse_class)
|
|
562
|
+
return nil if allowlist.nil? || allowlist.empty?
|
|
563
|
+
permitted = allowlist.map(&:to_s)
|
|
564
|
+
searchable_text_fields(klass).select do |source|
|
|
565
|
+
permitted.include?(Parse::Retrieval.send(:wire_name, klass, source))
|
|
566
|
+
end
|
|
567
|
+
end
|
|
568
|
+
|
|
332
569
|
# @!visibility private
|
|
333
570
|
# Validate a caller-supplied text_field against the embedded-source
|
|
334
|
-
#
|
|
335
|
-
#
|
|
336
|
-
#
|
|
571
|
+
# list and the `agent_fields` allowlist, or infer one.
|
|
572
|
+
#
|
|
573
|
+
# * Explicit field that is not an embed source: ValidationError.
|
|
574
|
+
# * Explicit embed source outside `agent_fields`: AccessDenied
|
|
575
|
+
# (`:field_denied`), raised before any search runs.
|
|
576
|
+
# * Omitted, class has no `agent_fields`: nil, so retrieve infers as
|
|
577
|
+
# before (single source) or raises AmbiguousTextField (several).
|
|
578
|
+
# * Omitted, class has `agent_fields`: the sole readable source; a
|
|
579
|
+
# `:field_denied` refusal when none is readable; a ValidationError
|
|
580
|
+
# asking for `text_field` when several are.
|
|
337
581
|
def normalize_text_field!(text_field, klass)
|
|
338
|
-
|
|
582
|
+
readable = readable_text_fields(klass)
|
|
583
|
+
|
|
584
|
+
if text_field.nil? || text_field.to_s.strip.empty?
|
|
585
|
+
return nil if readable.nil?
|
|
586
|
+
return readable.first.to_sym if readable.length == 1
|
|
587
|
+
if readable.empty?
|
|
588
|
+
raise text_field_denied(klass, searchable_text_fields(klass).first)
|
|
589
|
+
end
|
|
590
|
+
raise Parse::Agent::ValidationError,
|
|
591
|
+
"semantic_search: this class embeds several readable text sources; pass " \
|
|
592
|
+
"text_field (allowed: #{readable.inspect})."
|
|
593
|
+
end
|
|
594
|
+
|
|
339
595
|
allowed = searchable_text_fields(klass)
|
|
340
596
|
unless allowed.include?(text_field.to_s)
|
|
341
597
|
raise Parse::Agent::ValidationError,
|
|
342
598
|
"semantic_search: text_field #{text_field.to_s.inspect} is not an embedded " \
|
|
343
|
-
"text source for this class (allowed: #{allowed.inspect})."
|
|
599
|
+
"text source for this class (allowed: #{(readable || allowed).inspect})."
|
|
600
|
+
end
|
|
601
|
+
if readable && !readable.include?(text_field.to_s)
|
|
602
|
+
raise text_field_denied(klass, text_field.to_s)
|
|
344
603
|
end
|
|
345
604
|
text_field.to_sym
|
|
346
605
|
end
|
|
347
606
|
|
|
607
|
+
# @!visibility private
|
|
608
|
+
def text_field_denied(klass, source)
|
|
609
|
+
allowlist = Parse::Agent::MetadataRegistry.field_allowlist(klass.parse_class)
|
|
610
|
+
Parse::Agent::AccessDenied.new(
|
|
611
|
+
klass.parse_class,
|
|
612
|
+
"semantic_search: text source #{source.to_s.inspect} is outside the agent_fields " \
|
|
613
|
+
"allowlist for class '#{klass.parse_class}', so it cannot be returned as chunk content.",
|
|
614
|
+
kind: :field_denied,
|
|
615
|
+
denied_field: source.to_s,
|
|
616
|
+
allowed_fields: allowlist&.map(&:to_s),
|
|
617
|
+
)
|
|
618
|
+
end
|
|
619
|
+
|
|
348
620
|
# @!visibility private
|
|
349
621
|
# Refuse any top-level filter key not in the class's declared
|
|
350
622
|
# `filter_fields` allowlist (compound operators included — the
|
|
@@ -367,11 +639,12 @@ module Parse
|
|
|
367
639
|
"type" => "object",
|
|
368
640
|
"properties" => {
|
|
369
641
|
"class_name" => { "type" => "string", "description" => "Parse class name (must be agent_searchable)." },
|
|
370
|
-
"query" => { "type" => "string", "description" => "Natural-language query." },
|
|
642
|
+
"query" => { "type" => "string", "description" => "Natural-language query.", "maxLength" => MAX_QUERY_CHARS },
|
|
371
643
|
"k" => { "type" => "integer", "default" => DEFAULT_K, "minimum" => 1, "maximum" => MAX_K },
|
|
372
644
|
"filter" => { "type" => "object", "description" => "Post-search field filter (allowlisted fields only)." },
|
|
373
645
|
"vector_filter" => { "type" => "object", "description" => "Atlas pre-search filter (allowlisted fields only)." },
|
|
374
646
|
"text_field" => { "type" => "string", "description" => "Which embedded text source to chunk and return as content. Required only when the class embeds more than one text field; must name one of those sources." },
|
|
647
|
+
"profile" => { "type" => "string", "description" => "Optional server-configured retrieval profile name (for example fast, balanced, precise). Profiles set result counts, hybrid search, and reranking; omit for the default search. An unknown name is refused with the list of available profiles." },
|
|
375
648
|
"chunk_size" => { "type" => "integer", "description" => "Override chunk window size." },
|
|
376
649
|
"chunk_overlap" => { "type" => "integer", "description" => "Override chunk overlap." },
|
|
377
650
|
"chunk_by" => { "type" => "string", "enum" => %w[chars tokens], "description" => "Chunk unit." },
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
# encoding: UTF-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require "json"
|
|
5
|
+
|
|
6
|
+
module Parse
|
|
7
|
+
module Retrieval
|
|
8
|
+
# A small evaluation harness for comparing retrieval profiles.
|
|
9
|
+
#
|
|
10
|
+
# Run a labeled case set through each profile and report quality
|
|
11
|
+
# (recall@k, MRR, hit rate), latency (mean and p95), and estimated rerank
|
|
12
|
+
# usage, overall and per tag. Tags group the cases a deployment cares
|
|
13
|
+
# about: exact names, semantic questions, long documents, restrictive
|
|
14
|
+
# ACLs, tenant boundaries. For ACL and tenant cases, `relevant` lists only
|
|
15
|
+
# what the caller is ALLOWED to retrieve, and `forbidden` lists ids that
|
|
16
|
+
# must never appear; any forbidden hit is reported as a violation.
|
|
17
|
+
#
|
|
18
|
+
# The harness is runner-agnostic. {.semantic_search_runner} drives the
|
|
19
|
+
# real `semantic_search` tool through an agent, so measurements reflect
|
|
20
|
+
# the access policy, budgets, and fallbacks that production uses.
|
|
21
|
+
#
|
|
22
|
+
# @example
|
|
23
|
+
# cases = Parse::Retrieval::Benchmark.load_cases("eval/cases.json")
|
|
24
|
+
# runner = Parse::Retrieval::Benchmark.semantic_search_runner(agent, class_name: "Article")
|
|
25
|
+
# report = Parse::Retrieval::Benchmark.run(cases: cases, profiles: %w[fast precise], runner: runner)
|
|
26
|
+
# report["precise"][:recall_at_k] # => 0.83
|
|
27
|
+
module Benchmark
|
|
28
|
+
# One labeled query. `relevant` and `forbidden` are object ids.
|
|
29
|
+
Case = Struct.new(:id, :query, :relevant, :forbidden, :tags, keyword_init: true)
|
|
30
|
+
|
|
31
|
+
module_function
|
|
32
|
+
|
|
33
|
+
# @param path [String] JSON file: an Array of
|
|
34
|
+
# `{ "id", "query", "relevant": [...], "forbidden": [...], "tags": [...] }`.
|
|
35
|
+
# @return [Array<Case>]
|
|
36
|
+
def load_cases(path)
|
|
37
|
+
Array(JSON.parse(::File.read(path))).map { |h| case_from(h) }
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
# @param hash [Hash]
|
|
41
|
+
# @return [Case]
|
|
42
|
+
def case_from(hash)
|
|
43
|
+
h = hash.transform_keys(&:to_s)
|
|
44
|
+
Case.new(
|
|
45
|
+
id: h.fetch("id").to_s, query: h.fetch("query").to_s,
|
|
46
|
+
relevant: Array(h["relevant"]).map(&:to_s), forbidden: Array(h["forbidden"]).map(&:to_s),
|
|
47
|
+
tags: Array(h["tags"]).map(&:to_s),
|
|
48
|
+
)
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
# Run every case through every profile.
|
|
52
|
+
#
|
|
53
|
+
# @param cases [Array<Case>]
|
|
54
|
+
# @param profiles [Array<String, nil>] profile names (nil = default search).
|
|
55
|
+
# @param runner [#call] `runner.call(case, profile)` returning
|
|
56
|
+
# `{ ids: Array<String> ranked best-first, tokens_estimated: Integer }`.
|
|
57
|
+
# @param k [Integer] cutoff for recall and hits.
|
|
58
|
+
# @return [Hash{String => Hash}] per-profile report.
|
|
59
|
+
def run(cases:, profiles:, runner:, k: 10)
|
|
60
|
+
profiles.each_with_object({}) do |profile, report|
|
|
61
|
+
rows = cases.map { |c| score_case(c, profile, runner, k) }
|
|
62
|
+
report[profile.nil? ? "default" : profile.to_s] = summarize(rows).merge(
|
|
63
|
+
by_tag: rows.flat_map { |r| r[:tags].map { |t| [t, r] } }
|
|
64
|
+
.group_by(&:first)
|
|
65
|
+
.transform_values { |pairs| summarize(pairs.map(&:last)) },
|
|
66
|
+
)
|
|
67
|
+
end
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# A runner that executes the `semantic_search` tool through `agent`
|
|
71
|
+
# and ranks parent documents by their first chunk.
|
|
72
|
+
#
|
|
73
|
+
# @param agent [Parse::Agent]
|
|
74
|
+
# @param class_name [String]
|
|
75
|
+
# @param options [Hash] extra semantic_search arguments.
|
|
76
|
+
# @return [Proc]
|
|
77
|
+
def semantic_search_runner(agent, class_name:, **options)
|
|
78
|
+
lambda do |bench_case, profile|
|
|
79
|
+
tokens = 0
|
|
80
|
+
# Notifications are delivered on the instrumenting thread, so only
|
|
81
|
+
# events from THIS thread belong to this case; other threads'
|
|
82
|
+
# searches are ignored.
|
|
83
|
+
runner_thread = Thread.current
|
|
84
|
+
sub = if defined?(ActiveSupport::Notifications)
|
|
85
|
+
ActiveSupport::Notifications.subscribe("parse.retrieval.search") do |*args|
|
|
86
|
+
next unless Thread.current.equal?(runner_thread)
|
|
87
|
+
payload = args.last
|
|
88
|
+
tokens += payload.dig(:rerank, :tokens_estimated).to_i if payload.is_a?(Hash)
|
|
89
|
+
end
|
|
90
|
+
end
|
|
91
|
+
begin
|
|
92
|
+
args = { class_name: class_name, query: bench_case.query }.merge(options)
|
|
93
|
+
args[:profile] = profile unless profile.nil?
|
|
94
|
+
result = agent.execute(:semantic_search, **args)
|
|
95
|
+
data = result[:success] ? (result[:data] || {}) : {}
|
|
96
|
+
chunks = data[:chunks] || data["chunks"] || []
|
|
97
|
+
ids = chunks.map { |c| (c[:metadata] || c["metadata"] || {})[:object_id] || c.dig("metadata", "object_id") }
|
|
98
|
+
.compact.map(&:to_s).uniq
|
|
99
|
+
{ ids: ids, tokens_estimated: tokens, error: result[:success] ? nil : result[:error_code] }
|
|
100
|
+
ensure
|
|
101
|
+
ActiveSupport::Notifications.unsubscribe(sub) if sub
|
|
102
|
+
end
|
|
103
|
+
end
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
# @!visibility private
|
|
107
|
+
def score_case(bench_case, profile, runner, k)
|
|
108
|
+
started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
109
|
+
out = runner.call(bench_case, profile) || {}
|
|
110
|
+
ms = (Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) * 1000
|
|
111
|
+
ids = Array(out[:ids]).map(&:to_s)
|
|
112
|
+
top = ids.first(k)
|
|
113
|
+
relevant = bench_case.relevant
|
|
114
|
+
found = relevant & top
|
|
115
|
+
# MRR is cut off at k, like recall: a relevant hit below the cutoff
|
|
116
|
+
# scores 0.
|
|
117
|
+
first_rank = top.index { |id| relevant.include?(id) }
|
|
118
|
+
{
|
|
119
|
+
tags: bench_case.tags,
|
|
120
|
+
recall: relevant.empty? ? (top.empty? ? 1.0 : 0.0) : found.size.to_f / relevant.size,
|
|
121
|
+
reciprocal_rank: first_rank ? 1.0 / (first_rank + 1) : 0.0,
|
|
122
|
+
hit: !found.empty? || (relevant.empty? && top.empty?),
|
|
123
|
+
violations: (bench_case.forbidden & ids).size,
|
|
124
|
+
ms: ms,
|
|
125
|
+
tokens: out[:tokens_estimated].to_i,
|
|
126
|
+
error: out[:error],
|
|
127
|
+
}
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
# @!visibility private
|
|
131
|
+
def summarize(rows)
|
|
132
|
+
n = rows.size
|
|
133
|
+
return { cases: 0 } if n.zero?
|
|
134
|
+
latencies = rows.map { |r| r[:ms] }.sort
|
|
135
|
+
{
|
|
136
|
+
cases: n,
|
|
137
|
+
recall_at_k: (rows.sum { |r| r[:recall] } / n).round(4),
|
|
138
|
+
mrr: (rows.sum { |r| r[:reciprocal_rank] } / n).round(4),
|
|
139
|
+
hit_rate: (rows.count { |r| r[:hit] }.to_f / n).round(4),
|
|
140
|
+
violations: rows.sum { |r| r[:violations] },
|
|
141
|
+
errors: rows.count { |r| r[:error] },
|
|
142
|
+
mean_ms: (latencies.sum / n).round(1),
|
|
143
|
+
p95_ms: latencies[[(n * 0.95).ceil - 1, 0].max].round(1),
|
|
144
|
+
tokens_estimated: rows.sum { |r| r[:tokens] },
|
|
145
|
+
}
|
|
146
|
+
end
|
|
147
|
+
end
|
|
148
|
+
end
|
|
149
|
+
end
|