parse-stack-next 5.7.6 → 5.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +830 -0
  3. data/README.md +14 -4
  4. data/docs/TEST_SERVER.md +2 -2
  5. data/docs/acl_clp_guide.md +7 -0
  6. data/docs/atlas_vector_search_guide.md +181 -13
  7. data/docs/client_sdk_guide.md +11 -0
  8. data/docs/mcp_guide.md +317 -6
  9. data/docs/mongodb_direct_guide.md +27 -0
  10. data/docs/usage_guide.md +38 -0
  11. data/docs/webhooks_guide.md +74 -17
  12. data/lib/parse/acl_scope.rb +159 -41
  13. data/lib/parse/agent/approval_gate.rb +0 -0
  14. data/lib/parse/agent/constraint_translator.rb +42 -15
  15. data/lib/parse/agent/describe.rb +3 -1
  16. data/lib/parse/agent/field_names.rb +53 -0
  17. data/lib/parse/agent/field_policy.rb +74 -0
  18. data/lib/parse/agent/mcp_deployments.rb +426 -0
  19. data/lib/parse/agent/mcp_rack_app.rb +424 -45
  20. data/lib/parse/agent/mcp_server.rb +23 -1
  21. data/lib/parse/agent/mcp_subscriptions.rb +124 -6
  22. data/lib/parse/agent/metadata_registry.rb +67 -8
  23. data/lib/parse/agent/prompt_hardening.rb +9 -3
  24. data/lib/parse/agent/tools.rb +378 -29
  25. data/lib/parse/agent.rb +93 -1
  26. data/lib/parse/api/batch.rb +10 -1
  27. data/lib/parse/api/schema.rb +23 -4
  28. data/lib/parse/api/sessions.rb +6 -2
  29. data/lib/parse/api/users.rb +88 -14
  30. data/lib/parse/atlas_search/protected_paths.rb +236 -0
  31. data/lib/parse/atlas_search.rb +95 -23
  32. data/lib/parse/authorization.rb +54 -1
  33. data/lib/parse/client/batch.rb +231 -35
  34. data/lib/parse/client/body_builder.rb +21 -0
  35. data/lib/parse/client/caching.rb +371 -27
  36. data/lib/parse/client/request.rb +26 -14
  37. data/lib/parse/client/response.rb +49 -6
  38. data/lib/parse/client.rb +201 -38
  39. data/lib/parse/clp_scope.rb +281 -23
  40. data/lib/parse/console.rb +2 -2
  41. data/lib/parse/embeddings/voyage.rb +181 -17
  42. data/lib/parse/graphql/type_generator.rb +3 -0
  43. data/lib/parse/model/acl.rb +119 -21
  44. data/lib/parse/model/associations/belongs_to.rb +25 -3
  45. data/lib/parse/model/associations/collection_proxy.rb +138 -17
  46. data/lib/parse/model/associations/has_many.rb +38 -9
  47. data/lib/parse/model/associations/has_one.rb +3 -1
  48. data/lib/parse/model/associations/pointer_collection_proxy.rb +109 -17
  49. data/lib/parse/model/associations/relation_collection_proxy.rb +134 -28
  50. data/lib/parse/model/bytes.rb +13 -5
  51. data/lib/parse/model/classes/role.rb +72 -0
  52. data/lib/parse/model/classes/session.rb +43 -0
  53. data/lib/parse/model/classes/user.rb +78 -3
  54. data/lib/parse/model/core/actions.rb +269 -67
  55. data/lib/parse/model/core/builder.rb +100 -8
  56. data/lib/parse/model/core/create_lock.rb +27 -2
  57. data/lib/parse/model/core/describe.rb +2 -0
  58. data/lib/parse/model/core/fetching.rb +21 -3
  59. data/lib/parse/model/core/pluralized_aliases.rb +8 -4
  60. data/lib/parse/model/core/properties.rb +488 -39
  61. data/lib/parse/model/core/querying.rb +7 -0
  62. data/lib/parse/model/core/schema.rb +5 -3
  63. data/lib/parse/model/core/search_indexing.rb +63 -0
  64. data/lib/parse/model/core/vector_searchable.rb +35 -6
  65. data/lib/parse/model/file.rb +9 -2
  66. data/lib/parse/model/geopoint.rb +61 -13
  67. data/lib/parse/model/model.rb +160 -9
  68. data/lib/parse/model/object.rb +265 -17
  69. data/lib/parse/model/phone.rb +54 -5
  70. data/lib/parse/model/pointer.rb +40 -6
  71. data/lib/parse/mongodb.rb +170 -60
  72. data/lib/parse/pipeline_security.rb +415 -26
  73. data/lib/parse/query/constraint.rb +30 -0
  74. data/lib/parse/query/constraints.rb +58 -32
  75. data/lib/parse/query/cursor.rb +3 -1
  76. data/lib/parse/query/operation.rb +62 -8
  77. data/lib/parse/query/ordering.rb +34 -6
  78. data/lib/parse/query.rb +1100 -134
  79. data/lib/parse/retrieval/agent_tool.rb +225 -8
  80. data/lib/parse/retrieval/benchmark.rb +149 -0
  81. data/lib/parse/retrieval/profiles.rb +320 -0
  82. data/lib/parse/retrieval/retriever.rb +10 -1
  83. data/lib/parse/retrieval.rb +2 -0
  84. data/lib/parse/schema/search_index_migrator.rb +23 -5
  85. data/lib/parse/schema.rb +74 -18
  86. data/lib/parse/stack/tasks.rb +6 -4
  87. data/lib/parse/stack/version.rb +1 -1
  88. data/lib/parse/stack.rb +72 -14
  89. data/lib/parse/two_factor_auth/user_extension.rb +14 -2
  90. data/lib/parse/two_factor_auth.rb +11 -0
  91. data/lib/parse/vector_search/hybrid.rb +36 -18
  92. data/lib/parse/vector_search/index_definition.rb +237 -0
  93. data/lib/parse/vector_search.rb +46 -17
  94. data/lib/parse/webhooks/payload.rb +93 -6
  95. data/lib/parse/webhooks/replay_protection.rb +58 -20
  96. data/lib/parse/webhooks.rb +412 -40
  97. metadata +8 -1
@@ -57,6 +57,12 @@ module Parse
57
57
  # truncation is never silent. Pass `max_total_tokens: 0` to disable.
58
58
  DEFAULT_MAX_TOTAL_TOKENS = 20_000
59
59
 
60
+ # Longest `query` accepted. A search query is a short natural-language
61
+ # request; the bound keeps one call from sending a body-sized string to
62
+ # the embedding provider and, under a reranking profile, pairing it
63
+ # with every candidate document.
64
+ MAX_QUERY_CHARS = 4_000
65
+
60
66
  # @param agent [Parse::Agent]
61
67
  # @param text_field [String, Symbol, nil] which embedded text source to
62
68
  # chunk and return as `content`. Must name one of the class's declared
@@ -75,10 +81,36 @@ module Parse
75
81
  # by objectId) instead of being duplicated on every chunk. When the
76
82
  # token budget trims the result, `budget_truncated: true` and
77
83
  # `budget_dropped: <n>` are added.
78
- def semantic_search(agent, class_name: nil, query: nil, k: DEFAULT_K,
84
+ def semantic_search(agent, **args)
85
+ started = monotonic_now
86
+ semantic_search_unobserved(agent, **args)
87
+ rescue StandardError => e
88
+ # Failures are observable too: one sanitized event naming the error
89
+ # class (never its message, which can echo input).
90
+ emit_failure_event(args, e, started)
91
+ raise
92
+ end
93
+
94
+ # @!visibility private
95
+ def emit_failure_event(args, error, started)
96
+ return unless defined?(ActiveSupport::Notifications)
97
+ payload = {
98
+ class_name: (args[:class_name] || args[:klass]).to_s,
99
+ profile: args[:profile]&.to_s,
100
+ error: error.class.name,
101
+ duration_ms: ((monotonic_now - started) * 1000).round(1),
102
+ }
103
+ ActiveSupport::Notifications.instrument("parse.retrieval.search", payload)
104
+ rescue StandardError
105
+ nil
106
+ end
107
+
108
+ # @!visibility private
109
+ def semantic_search_unobserved(agent, class_name: nil, query: nil, k: nil,
79
110
  filter: nil, vector_filter: nil, text_field: nil,
80
111
  chunk_size: nil, chunk_overlap: nil, chunk_by: nil,
81
112
  max_chunks_per_document: nil, max_total_tokens: nil,
113
+ profile: nil,
82
114
  # Back-compat / ergonomic aliases for direct callers:
83
115
  # `klass:`/`class:` for class_name, and the chunker's
84
116
  # own `size:`/`overlap:`/`by:` names.
@@ -95,14 +127,32 @@ module Parse
95
127
  unless query.is_a?(String) && !query.strip.empty?
96
128
  raise Parse::Agent::ValidationError, "semantic_search: `query` must be a non-empty String."
97
129
  end
130
+ if query.length > MAX_QUERY_CHARS
131
+ raise Parse::Agent::ValidationError,
132
+ "semantic_search: `query` is #{query.length} characters; the limit is #{MAX_QUERY_CHARS}."
133
+ end
98
134
 
99
135
  resolved_text_field = normalize_text_field!(text_field, klass)
136
+ # A named, server-configured retrieval profile (Parse::Retrieval::Profiles).
137
+ # Unknown names fail here, before any provider call.
138
+ prof = profile.nil? || profile.to_s.strip.empty? ? nil : Parse::Retrieval::Profiles.fetch!(profile)
100
139
 
101
140
  # Reject reserved underscore keys at any depth, then enforce the
102
141
  # per-class filter-field allowlist on top-level keys.
103
142
  Parse::Retrieval.assert_no_underscore_keys!(filter) unless filter.nil?
104
143
  Parse::Retrieval.assert_no_underscore_keys!(vector_filter) unless vector_filter.nil?
105
144
  allowed = Parse::Agent::MetadataRegistry.searchable_filter_fields(cname).map(&:to_s)
145
+ # Filterable fields are also bounded by what the agent may read (the
146
+ # class `agent_fields` ceiling, narrowed by any per-agent `fields:`
147
+ # policy): filtering on a field the agent cannot read would reveal
148
+ # its value through which rows match. A `filter_fields` entry outside
149
+ # `agent_fields` is therefore never usable.
150
+ readable = Parse::Agent::MetadataRegistry.field_allowlist(cname)&.map(&:to_s)
151
+ if readable && !readable.empty?
152
+ allowed = allowed.select do |f|
153
+ readable.include?(Parse::Agent::MetadataRegistry.wire_field_names(cname, [f]).first)
154
+ end
155
+ end
106
156
  assert_filter_fields_allowed!(filter, allowed)
107
157
  assert_filter_fields_allowed!(vector_filter, allowed)
108
158
 
@@ -124,6 +174,39 @@ module Parse
124
174
  score_quantize = (agent.permissions != :admin)
125
175
  vector_field = Parse::Agent::MetadataRegistry.searchable_field(cname)
126
176
 
177
+ # Profile resolution: k is bounded by the profile's max_k; a reranking
178
+ # profile retrieves `rerank_candidates` and keeps `rerank_top_n` (or
179
+ # the effective k); hybrid settings come only from the profile.
180
+ effective_k = if prof
181
+ requested = k.to_i.positive? ? k.to_i : prof.k
182
+ clamp_k([requested, prof.max_k].min)
183
+ else
184
+ clamp_k(k)
185
+ end
186
+ reranker = nil
187
+ retrieve_k = effective_k
188
+ rerank_top_n = nil
189
+ if prof&.rerank?
190
+ reranker = Parse::Retrieval::BudgetedReranker.new(
191
+ Parse::Retrieval.reranker(prof.reranker), prof,
192
+ charge: ->(tokens) { charge_rerank_tokens!(agent, scope, tokens) },
193
+ )
194
+ # rerank_candidates is a hard budget: the caller's k can never
195
+ # raise how many documents are retrieved and sent to the
196
+ # reranker, so k is capped at it.
197
+ retrieve_k = prof.rerank_candidates
198
+ effective_k = [effective_k, retrieve_k].min
199
+ rerank_top_n = [prof.rerank_top_n || effective_k, effective_k].min
200
+ end
201
+ if prof
202
+ # Under a profile the response budget is mandatory: the caller can
203
+ # lower it but never raise or disable it (0 does not switch it off).
204
+ ceiling = prof.max_total_tokens || DEFAULT_MAX_TOTAL_TOKENS
205
+ requested = max_total_tokens.to_i
206
+ max_total_tokens = requested.positive? ? [requested, ceiling].min : ceiling
207
+ end
208
+ started = monotonic_now
209
+
127
210
  # with_precharged: the cap was charged above with per-tenant
128
211
  # identity (or deliberately skipped for trusted admin agents) —
129
212
  # suppress the generic query-embed charge inside
@@ -135,7 +218,10 @@ module Parse
135
218
  klass: klass,
136
219
  field: vector_field,
137
220
  text_field: resolved_text_field,
138
- k: clamp_k(k),
221
+ k: retrieve_k,
222
+ hybrid: prof&.hybrid ? hybrid_config_for(prof, klass) : nil,
223
+ rerank: reranker,
224
+ rerank_top_n: rerank_top_n,
139
225
  filter: filter,
140
226
  vector_filter: vector_filter,
141
227
  chunker: build_chunker(chunk_size, chunk_overlap, chunk_by, max_chunks_per_document),
@@ -149,7 +235,7 @@ module Parse
149
235
  # Token budget (B4): trim the score-ordered chunk list before
150
236
  # building the envelope so `documents` only carries parents whose
151
237
  # chunks survived.
152
- kept, dropped = apply_token_budget(chunks, resolve_token_budget(max_total_tokens))
238
+ kept, dropped = apply_token_budget(chunks, resolve_token_budget(max_total_tokens), strict: !prof.nil?)
153
239
 
154
240
  # Source dedup (A3): a document's (projected) source record is
155
241
  # identical across all its chunks. Hoist it into a `documents` map
@@ -172,9 +258,113 @@ module Parse
172
258
  envelope[:budget_truncated] = true
173
259
  envelope[:budget_dropped] = dropped
174
260
  end
261
+ if prof
262
+ envelope[:profile] = prof.name
263
+ if reranker&.stats&.dig(:fallback)
264
+ # Observable fallback: the result is in retrieval order, not
265
+ # reranked, and the caller is told why.
266
+ envelope[:rerank_fallback] = true
267
+ envelope[:rerank_fallback_reason] = reranker.stats[:fallback_reason]
268
+ end
269
+ end
270
+ emit_search_event(cname, prof, effective_k, retrieve_k, reranker, envelope, dropped, started)
175
271
  envelope
176
272
  end
177
273
 
274
+ # @!visibility private
275
+ # A profile's hybrid settings with the lexical branch restricted to the
276
+ # text sources the agent may read. Without this the lexical search runs
277
+ # over every field (`wildcard: "*"`), so which documents match, and
278
+ # their rank, could depend on a hidden field. Refused when the class
279
+ # has an allowlist and no readable text source.
280
+ def hybrid_config_for(prof, klass)
281
+ cfg = Marshal.load(Marshal.dump(prof.hybrid.to_h))
282
+ allowlist = Parse::Agent::MetadataRegistry.field_allowlist(klass.parse_class)
283
+ if allowlist.nil? || allowlist.empty?
284
+ # No allowlist: never fall back to `wildcard: "*"`, which would
285
+ # let every column (including CLP protectedFields) decide matches.
286
+ # Search the embedded text sources unless the profile names fields;
287
+ # Atlas search then refuses any named field protected for the caller.
288
+ lexical = (cfg[:lexical] || {}).dup
289
+ if Array(lexical[:fields]).empty?
290
+ lexical[:fields] = searchable_text_fields(klass).map { |f| Parse::Retrieval.send(:wire_name, klass, f) }
291
+ cfg[:lexical] = lexical
292
+ end
293
+ return cfg
294
+ end
295
+ lexical = (cfg[:lexical] || {}).dup
296
+ if lexical[:fields]
297
+ # Server-configured lexical fields are kept when the agent may read
298
+ # them (any readable field, not only embedding sources), and
299
+ # translated to their stored names.
300
+ readable_wire = allowlist.map(&:to_s) - Parse::Agent::MetadataRegistry::ALWAYS_KEEP_FIELDS
301
+ configured = Array(lexical[:fields]).map { |f| Parse::Retrieval.send(:wire_name, klass, f) }
302
+ lexical[:fields] = configured & readable_wire
303
+ else
304
+ # Unconfigured: search the readable embedded text sources.
305
+ readable = readable_text_fields(klass) || []
306
+ lexical[:fields] = readable.map { |f| Parse::Retrieval.send(:wire_name, klass, f) }
307
+ end
308
+ if lexical[:fields].empty?
309
+ # An empty list would mean `wildcard: "*"`, letting hidden fields
310
+ # decide matches; refuse instead.
311
+ raise text_field_denied(klass, Array(cfg.dig(:lexical, :fields)).first || searchable_text_fields(klass).first)
312
+ end
313
+ cfg[:lexical] = lexical
314
+ cfg
315
+ end
316
+
317
+ # @!visibility private
318
+ def monotonic_now
319
+ Process.clock_gettime(Process::CLOCK_MONOTONIC)
320
+ end
321
+
322
+ # @!visibility private
323
+ # One sanitized `parse.retrieval.search` event per semantic_search
324
+ # call: profile, budgets, stage counts and timings, and estimated
325
+ # rerank usage. Never document text, field values, URLs, or
326
+ # credentials. Rerank tokens are the SDK's estimate
327
+ # (`tokens_estimated`), not provider-reported usage.
328
+ def emit_search_event(cname, prof, k, retrieve_k, reranker, envelope, dropped, started)
329
+ return unless defined?(ActiveSupport::Notifications)
330
+ total_ms = ((monotonic_now - started) * 1000).round(1)
331
+ rerank = reranker ? reranker.stats.dup : { used: false }
332
+ payload = {
333
+ class_name: cname,
334
+ profile: prof&.name,
335
+ hybrid: !prof&.hybrid.nil?,
336
+ k: k,
337
+ candidates: retrieve_k,
338
+ rerank: rerank,
339
+ chunks_returned: envelope[:count],
340
+ documents_returned: envelope[:documents].size,
341
+ budget_dropped: dropped,
342
+ duration_ms: total_ms,
343
+ retrieve_ms: (total_ms - (rerank[:duration_ms] || 0)).round(1),
344
+ }
345
+ ActiveSupport::Notifications.instrument("parse.retrieval.search", payload)
346
+ rescue StandardError
347
+ nil
348
+ end
349
+
350
+ # @!visibility private
351
+ # Charge estimated reranker tokens to the same per-tenant spend cap
352
+ # the query embedding uses (admin agents are exempt, as there). A
353
+ # transient cap hit surfaces as RateLimitExceeded; an impossible one
354
+ # as ValidationError, mirroring {#charge_spend_cap!}.
355
+ def charge_rerank_tokens!(agent, scope, tokens)
356
+ return if agent.permissions == :admin
357
+ tenant_id = scope && (scope[:value] || scope["value"])
358
+ Parse::Embeddings::SpendCap.charge!(tenant_id: tenant_id, tokens: tokens)
359
+ rescue Parse::Embeddings::SpendCap::Exceeded => e
360
+ if e.retry_after.nil?
361
+ raise Parse::Agent::ValidationError,
362
+ "semantic_search: reranking exceeds the spend cap " \
363
+ "(#{e.requested} tokens requested, limit #{e.limit}/#{e.window}s)."
364
+ end
365
+ raise Parse::Agent::RateLimitExceeded.new(retry_after: e.retry_after, limit: e.limit, window: e.window)
366
+ end
367
+
178
368
  # @!visibility private
179
369
  # Charge the estimated query-embedding token cost against the
180
370
  # tenant's spend cap. The tenant key is the resolved tenant-scope
@@ -230,14 +420,40 @@ module Parse
230
420
  # least the first chunk so a single oversize chunk still returns
231
421
  # something (flagged truncated).
232
422
  # @return [Array(Array<Chunk>, Integer)] [kept, dropped_count]
233
- def apply_token_budget(chunks, budget)
423
+ #
424
+ # The estimate covers the whole response, not only chunk text: each
425
+ # chunk's content plus, the first time a parent document appears, that
426
+ # document's serialized source record (it is hoisted into `documents`).
427
+ #
428
+ # `strict:` (a profile's mandatory budget) drops even the first chunk
429
+ # when it alone exceeds the budget; otherwise the first chunk is always
430
+ # kept so an oversized single result still returns something.
431
+ # Characters each returned chunk adds beyond its content and metadata:
432
+ # its key names, score, and `_source` provenance stamp.
433
+ CHUNK_OVERHEAD_CHARS = 160
434
+ # Characters the response envelope adds once (counts, profile,
435
+ # truncation flags, the `documents` map wrapper).
436
+ ENVELOPE_OVERHEAD_CHARS = 400
437
+
438
+ def apply_token_budget(chunks, budget, strict: false)
234
439
  return [chunks, 0] if budget.nil? || chunks.empty?
235
- total = 0
440
+ # Every chunk carries metadata and per-chunk keys alongside its text,
441
+ # so a response of many tiny chunks is mostly overhead; count it.
442
+ total = (ENVELOPE_OVERHEAD_CHARS / 4.0).ceil
236
443
  kept = []
444
+ seen_docs = {}
237
445
  chunks.each do |chunk|
238
- est = (chunk.content.to_s.length / 4.0).ceil
239
- break unless kept.empty? || total + est <= budget
446
+ meta = chunk.respond_to?(:metadata) ? chunk.metadata : nil
447
+ meta_chars = meta.is_a?(Hash) ? (JSON.generate(meta).length rescue 0) : 0
448
+ est = ((chunk.content.to_s.length + meta_chars + CHUNK_OVERHEAD_CHARS) / 4.0).ceil
449
+ oid = chunk.respond_to?(:metadata) && chunk.metadata.is_a?(Hash) ? chunk.metadata[:object_id] : nil
450
+ if oid && !seen_docs.key?(oid) && chunk.respond_to?(:source) && chunk.source
451
+ doc_est = (JSON.generate(chunk.source).length / 4.0).ceil rescue 0
452
+ est += doc_est
453
+ end
454
+ break unless (kept.empty? && !strict) || total + est <= budget
240
455
  kept << chunk
456
+ seen_docs[oid] = true if oid
241
457
  total += est
242
458
  end
243
459
  [kept, chunks.length - kept.length]
@@ -423,11 +639,12 @@ module Parse
423
639
  "type" => "object",
424
640
  "properties" => {
425
641
  "class_name" => { "type" => "string", "description" => "Parse class name (must be agent_searchable)." },
426
- "query" => { "type" => "string", "description" => "Natural-language query." },
642
+ "query" => { "type" => "string", "description" => "Natural-language query.", "maxLength" => MAX_QUERY_CHARS },
427
643
  "k" => { "type" => "integer", "default" => DEFAULT_K, "minimum" => 1, "maximum" => MAX_K },
428
644
  "filter" => { "type" => "object", "description" => "Post-search field filter (allowlisted fields only)." },
429
645
  "vector_filter" => { "type" => "object", "description" => "Atlas pre-search filter (allowlisted fields only)." },
430
646
  "text_field" => { "type" => "string", "description" => "Which embedded text source to chunk and return as content. Required only when the class embeds more than one text field; must name one of those sources." },
647
+ "profile" => { "type" => "string", "description" => "Optional server-configured retrieval profile name (for example fast, balanced, precise). Profiles set result counts, hybrid search, and reranking; omit for the default search. An unknown name is refused with the list of available profiles." },
431
648
  "chunk_size" => { "type" => "integer", "description" => "Override chunk window size." },
432
649
  "chunk_overlap" => { "type" => "integer", "description" => "Override chunk overlap." },
433
650
  "chunk_by" => { "type" => "string", "enum" => %w[chars tokens], "description" => "Chunk unit." },
@@ -0,0 +1,149 @@
1
+ # encoding: UTF-8
2
+ # frozen_string_literal: true
3
+
4
+ require "json"
5
+
6
+ module Parse
7
+ module Retrieval
8
+ # A small evaluation harness for comparing retrieval profiles.
9
+ #
10
+ # Run a labeled case set through each profile and report quality
11
+ # (recall@k, MRR, hit rate), latency (mean and p95), and estimated rerank
12
+ # usage, overall and per tag. Tags group the cases a deployment cares
13
+ # about: exact names, semantic questions, long documents, restrictive
14
+ # ACLs, tenant boundaries. For ACL and tenant cases, `relevant` lists only
15
+ # what the caller is ALLOWED to retrieve, and `forbidden` lists ids that
16
+ # must never appear; any forbidden hit is reported as a violation.
17
+ #
18
+ # The harness is runner-agnostic. {.semantic_search_runner} drives the
19
+ # real `semantic_search` tool through an agent, so measurements reflect
20
+ # the access policy, budgets, and fallbacks that production uses.
21
+ #
22
+ # @example
23
+ # cases = Parse::Retrieval::Benchmark.load_cases("eval/cases.json")
24
+ # runner = Parse::Retrieval::Benchmark.semantic_search_runner(agent, class_name: "Article")
25
+ # report = Parse::Retrieval::Benchmark.run(cases: cases, profiles: %w[fast precise], runner: runner)
26
+ # report["precise"][:recall_at_k] # => 0.83
27
+ module Benchmark
28
+ # One labeled query. `relevant` and `forbidden` are object ids.
29
+ Case = Struct.new(:id, :query, :relevant, :forbidden, :tags, keyword_init: true)
30
+
31
+ module_function
32
+
33
+ # @param path [String] JSON file: an Array of
34
+ # `{ "id", "query", "relevant": [...], "forbidden": [...], "tags": [...] }`.
35
+ # @return [Array<Case>]
36
+ def load_cases(path)
37
+ Array(JSON.parse(::File.read(path))).map { |h| case_from(h) }
38
+ end
39
+
40
+ # @param hash [Hash]
41
+ # @return [Case]
42
+ def case_from(hash)
43
+ h = hash.transform_keys(&:to_s)
44
+ Case.new(
45
+ id: h.fetch("id").to_s, query: h.fetch("query").to_s,
46
+ relevant: Array(h["relevant"]).map(&:to_s), forbidden: Array(h["forbidden"]).map(&:to_s),
47
+ tags: Array(h["tags"]).map(&:to_s),
48
+ )
49
+ end
50
+
51
+ # Run every case through every profile.
52
+ #
53
+ # @param cases [Array<Case>]
54
+ # @param profiles [Array<String, nil>] profile names (nil = default search).
55
+ # @param runner [#call] `runner.call(case, profile)` returning
56
+ # `{ ids: Array<String> ranked best-first, tokens_estimated: Integer }`.
57
+ # @param k [Integer] cutoff for recall and hits.
58
+ # @return [Hash{String => Hash}] per-profile report.
59
+ def run(cases:, profiles:, runner:, k: 10)
60
+ profiles.each_with_object({}) do |profile, report|
61
+ rows = cases.map { |c| score_case(c, profile, runner, k) }
62
+ report[profile.nil? ? "default" : profile.to_s] = summarize(rows).merge(
63
+ by_tag: rows.flat_map { |r| r[:tags].map { |t| [t, r] } }
64
+ .group_by(&:first)
65
+ .transform_values { |pairs| summarize(pairs.map(&:last)) },
66
+ )
67
+ end
68
+ end
69
+
70
+ # A runner that executes the `semantic_search` tool through `agent`
71
+ # and ranks parent documents by their first chunk.
72
+ #
73
+ # @param agent [Parse::Agent]
74
+ # @param class_name [String]
75
+ # @param options [Hash] extra semantic_search arguments.
76
+ # @return [Proc]
77
+ def semantic_search_runner(agent, class_name:, **options)
78
+ lambda do |bench_case, profile|
79
+ tokens = 0
80
+ # Notifications are delivered on the instrumenting thread, so only
81
+ # events from THIS thread belong to this case; other threads'
82
+ # searches are ignored.
83
+ runner_thread = Thread.current
84
+ sub = if defined?(ActiveSupport::Notifications)
85
+ ActiveSupport::Notifications.subscribe("parse.retrieval.search") do |*args|
86
+ next unless Thread.current.equal?(runner_thread)
87
+ payload = args.last
88
+ tokens += payload.dig(:rerank, :tokens_estimated).to_i if payload.is_a?(Hash)
89
+ end
90
+ end
91
+ begin
92
+ args = { class_name: class_name, query: bench_case.query }.merge(options)
93
+ args[:profile] = profile unless profile.nil?
94
+ result = agent.execute(:semantic_search, **args)
95
+ data = result[:success] ? (result[:data] || {}) : {}
96
+ chunks = data[:chunks] || data["chunks"] || []
97
+ ids = chunks.map { |c| (c[:metadata] || c["metadata"] || {})[:object_id] || c.dig("metadata", "object_id") }
98
+ .compact.map(&:to_s).uniq
99
+ { ids: ids, tokens_estimated: tokens, error: result[:success] ? nil : result[:error_code] }
100
+ ensure
101
+ ActiveSupport::Notifications.unsubscribe(sub) if sub
102
+ end
103
+ end
104
+ end
105
+
106
+ # @!visibility private
107
+ def score_case(bench_case, profile, runner, k)
108
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
109
+ out = runner.call(bench_case, profile) || {}
110
+ ms = (Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) * 1000
111
+ ids = Array(out[:ids]).map(&:to_s)
112
+ top = ids.first(k)
113
+ relevant = bench_case.relevant
114
+ found = relevant & top
115
+ # MRR is cut off at k, like recall: a relevant hit below the cutoff
116
+ # scores 0.
117
+ first_rank = top.index { |id| relevant.include?(id) }
118
+ {
119
+ tags: bench_case.tags,
120
+ recall: relevant.empty? ? (top.empty? ? 1.0 : 0.0) : found.size.to_f / relevant.size,
121
+ reciprocal_rank: first_rank ? 1.0 / (first_rank + 1) : 0.0,
122
+ hit: !found.empty? || (relevant.empty? && top.empty?),
123
+ violations: (bench_case.forbidden & ids).size,
124
+ ms: ms,
125
+ tokens: out[:tokens_estimated].to_i,
126
+ error: out[:error],
127
+ }
128
+ end
129
+
130
+ # @!visibility private
131
+ def summarize(rows)
132
+ n = rows.size
133
+ return { cases: 0 } if n.zero?
134
+ latencies = rows.map { |r| r[:ms] }.sort
135
+ {
136
+ cases: n,
137
+ recall_at_k: (rows.sum { |r| r[:recall] } / n).round(4),
138
+ mrr: (rows.sum { |r| r[:reciprocal_rank] } / n).round(4),
139
+ hit_rate: (rows.count { |r| r[:hit] }.to_f / n).round(4),
140
+ violations: rows.sum { |r| r[:violations] },
141
+ errors: rows.count { |r| r[:error] },
142
+ mean_ms: (latencies.sum / n).round(1),
143
+ p95_ms: latencies[[(n * 0.95).ceil - 1, 0].max].round(1),
144
+ tokens_estimated: rows.sum { |r| r[:tokens] },
145
+ }
146
+ end
147
+ end
148
+ end
149
+ end