ruby-mcp-client 2.1.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. checksums.yaml +4 -4
  2. data/OAUTH.md +555 -0
  3. data/README.md +825 -48
  4. data/lib/mcp_client/audio_content.rb +1 -1
  5. data/lib/mcp_client/auth/browser_oauth.rb +131 -21
  6. data/lib/mcp_client/auth/oauth_provider/challenge_handling.rb +532 -0
  7. data/lib/mcp_client/auth/oauth_provider/client_authentication.rb +121 -0
  8. data/lib/mcp_client/auth/oauth_provider/pending_requests.rb +51 -0
  9. data/lib/mcp_client/auth/oauth_provider/registration_store.rb +486 -0
  10. data/lib/mcp_client/auth/oauth_provider/response_validation.rb +441 -0
  11. data/lib/mcp_client/auth/oauth_provider/scope_selection.rb +134 -0
  12. data/lib/mcp_client/auth/oauth_provider/token_store.rb +419 -0
  13. data/lib/mcp_client/auth/oauth_provider.rb +1354 -386
  14. data/lib/mcp_client/auth/peer_text.rb +174 -0
  15. data/lib/mcp_client/auth.rb +298 -32
  16. data/lib/mcp_client/cached_result.rb +145 -0
  17. data/lib/mcp_client/called_tool_definition.rb +138 -0
  18. data/lib/mcp_client/client/cache_slices.rb +195 -0
  19. data/lib/mcp_client/client/list_aggregation.rb +243 -0
  20. data/lib/mcp_client/client/notification_routing.rb +155 -0
  21. data/lib/mcp_client/client/sampling_validation.rb +200 -0
  22. data/lib/mcp_client/client/task_api.rb +531 -0
  23. data/lib/mcp_client/client/task_lifetimes.rb +269 -0
  24. data/lib/mcp_client/client/task_registry.rb +254 -0
  25. data/lib/mcp_client/client/task_shape.rb +102 -0
  26. data/lib/mcp_client/client/task_support.rb +1166 -0
  27. data/lib/mcp_client/client/task_updates.rb +457 -0
  28. data/lib/mcp_client/client/task_wait_boundaries.rb +198 -0
  29. data/lib/mcp_client/client/task_workers.rb +63 -0
  30. data/lib/mcp_client/client.rb +796 -518
  31. data/lib/mcp_client/deep_copy.rb +49 -0
  32. data/lib/mcp_client/deprecation_notices.rb +94 -0
  33. data/lib/mcp_client/deprecations.rb +419 -0
  34. data/lib/mcp_client/errors.rb +474 -7
  35. data/lib/mcp_client/header_params.rb +320 -0
  36. data/lib/mcp_client/http_transport_base/bounded_inflate.rb +41 -0
  37. data/lib/mcp_client/http_transport_base/cache_support.rb +694 -0
  38. data/lib/mcp_client/http_transport_base/era_detection.rb +134 -0
  39. data/lib/mcp_client/http_transport_base/listen_stream.rb +763 -0
  40. data/lib/mcp_client/http_transport_base/param_headers.rb +35 -0
  41. data/lib/mcp_client/http_transport_base/request_recovery.rb +156 -0
  42. data/lib/mcp_client/http_transport_base/session_recovery.rb +113 -0
  43. data/lib/mcp_client/http_transport_base/sse_event_scanner.rb +145 -0
  44. data/lib/mcp_client/http_transport_base/stream_capture.rb +160 -0
  45. data/lib/mcp_client/http_transport_base/stream_recovery.rb +318 -0
  46. data/lib/mcp_client/http_transport_base/tool_listing.rb +277 -0
  47. data/lib/mcp_client/http_transport_base.rb +666 -120
  48. data/lib/mcp_client/input_round_trips.rb +128 -0
  49. data/lib/mcp_client/json_rpc_common/envelopes.rb +32 -0
  50. data/lib/mcp_client/json_rpc_common/error_bodies.rb +105 -0
  51. data/lib/mcp_client/json_rpc_common/input_waits.rb +167 -0
  52. data/lib/mcp_client/json_rpc_common.rb +900 -13
  53. data/lib/mcp_client/oauth_client.rb +14 -5
  54. data/lib/mcp_client/prompt.rb +4 -0
  55. data/lib/mcp_client/request_authorization.rb +128 -0
  56. data/lib/mcp_client/request_meta_scope.rb +77 -0
  57. data/lib/mcp_client/request_metadata.rb +287 -0
  58. data/lib/mcp_client/resource.rb +4 -0
  59. data/lib/mcp_client/resource_content.rb +20 -0
  60. data/lib/mcp_client/resource_template.rb +4 -0
  61. data/lib/mcp_client/result_caching.rb +999 -0
  62. data/lib/mcp_client/result_completeness.rb +34 -0
  63. data/lib/mcp_client/root.rb +6 -0
  64. data/lib/mcp_client/round_trip_marker.rb +28 -0
  65. data/lib/mcp_client/schema_validator/annotations.rb +82 -0
  66. data/lib/mcp_client/schema_validator/composition.rb +86 -0
  67. data/lib/mcp_client/schema_validator/dialects.rb +66 -0
  68. data/lib/mcp_client/schema_validator/ecma_patterns.rb +567 -0
  69. data/lib/mcp_client/schema_validator/evaluation.rb +517 -0
  70. data/lib/mcp_client/schema_validator/input_requirements.rb +84 -0
  71. data/lib/mcp_client/schema_validator/instances.rb +449 -0
  72. data/lib/mcp_client/schema_validator/keyword_scan.rb +121 -0
  73. data/lib/mcp_client/schema_validator/normalization.rb +104 -0
  74. data/lib/mcp_client/schema_validator/references.rb +610 -0
  75. data/lib/mcp_client/schema_validator/scalars.rb +126 -0
  76. data/lib/mcp_client/schema_validator/shapes.rb +319 -0
  77. data/lib/mcp_client/schema_validator/uri_references.rb +153 -0
  78. data/lib/mcp_client/schema_validator.rb +882 -208
  79. data/lib/mcp_client/server_base.rb +233 -5
  80. data/lib/mcp_client/server_factory.rb +9 -3
  81. data/lib/mcp_client/server_http/json_rpc_transport.rb +219 -4
  82. data/lib/mcp_client/server_http.rb +307 -90
  83. data/lib/mcp_client/server_sse/json_rpc_transport.rb +113 -25
  84. data/lib/mcp_client/server_sse/sse_parser.rb +39 -6
  85. data/lib/mcp_client/server_sse.rb +227 -62
  86. data/lib/mcp_client/server_stdio/child_session.rb +98 -0
  87. data/lib/mcp_client/server_stdio/json_rpc_transport.rb +1003 -28
  88. data/lib/mcp_client/server_stdio.rb +772 -183
  89. data/lib/mcp_client/server_streamable_http/json_rpc_transport.rb +189 -25
  90. data/lib/mcp_client/server_streamable_http.rb +302 -115
  91. data/lib/mcp_client/session_pin.rb +119 -0
  92. data/lib/mcp_client/subscription/notification_dispatcher.rb +354 -0
  93. data/lib/mcp_client/subscription.rb +852 -0
  94. data/lib/mcp_client/subscription_support.rb +715 -0
  95. data/lib/mcp_client/task.rb +286 -14
  96. data/lib/mcp_client/tool.rb +31 -3
  97. data/lib/mcp_client/version.rb +21 -6
  98. data/lib/mcp_client.rb +108 -19
  99. metadata +68 -2
@@ -0,0 +1,610 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'uri'
4
+
5
+ module MCPClient
6
+ module SchemaValidator
7
+ # Resolution of local `$ref` values: JSON pointer fragments (RFC 6901)
8
+ # and plain-name fragments naming an anchor. Nothing here ever fetches:
9
+ # a reference outside the document is reported as external. Extended
10
+ # into SchemaValidator, so the methods are its own.
11
+ #
12
+ # Plain names are scoped to their schema resource (JSON Schema 2020-12
13
+ # Core Sections 8.2.1 and 8.2.2): a subschema whose `$id` is a URI
14
+ # starts a new resource, and `#name` names an anchor of the resource the
15
+ # referencing schema belongs to — never one of an embedded resource, and
16
+ # never one of the enclosing document from inside an embedded resource.
17
+ module References
18
+ # A reference that does not point inside this document, so using it
19
+ # would need a retrieval that never happens. A bare fragment is always
20
+ # local; anything else is resolved against the base URI of the
21
+ # resource holding it (RFC 3986 Section 5.2) and is local when the
22
+ # document bundles a resource whose `$id` is that URI (JSON Schema
23
+ # 2020-12 Core Section 9.3.1) -- the empty reference, which names the
24
+ # base itself, included. Without the document and its index only the
25
+ # syntactic answer is available, and a reference outside the fragment
26
+ # space is external.
27
+ # @param ref [String]
28
+ # @param root [Hash, nil] the normalized root schema
29
+ # @param dialect [String, nil] the canonical dialect
30
+ # @param resolver [Hash, Context, nil] holder of the memoized index
31
+ # @param from [Hash, nil] the schema object holding the reference
32
+ # @return [Boolean]
33
+ def external_ref?(ref, root = nil, dialect = nil, resolver = nil, from: nil)
34
+ return false if ref.start_with?('#')
35
+ return true unless root.is_a?(Hash) && resolver
36
+
37
+ index = (resolver[:anchors] ||= anchor_index(root, dialect))
38
+ return true if from.is_a?(Hash) && !index[:resources].key?(from)
39
+
40
+ retarget_reference(index, (from && index[:resources][from]) || root, ref).first.nil?
41
+ end
42
+
43
+ # The bundled resource a reference outside the fragment space names,
44
+ # with the bare fragment left to resolve inside it.
45
+ # @param index [Hash] the anchor index
46
+ # @param resource [Hash] the resource the reference is written in
47
+ # @param ref [String] the `$ref` value
48
+ # @return [Array(Hash, String), Array(nil, nil)]
49
+ def retarget_reference(index, resource, ref)
50
+ uri, fragment = ref.split('#', 2)
51
+ base = merge_uri(index[:bases][resource] || '', uri.to_s)
52
+ target = base && index[:by_base][base]
53
+ target ? [target, "##{fragment}"] : [nil, nil]
54
+ end
55
+
56
+ # Resolve a local reference: a JSON pointer fragment, or a plain-name
57
+ # fragment naming an anchor. Both are relative to the schema resource
58
+ # the referencing schema belongs to (JSON Schema 2020-12 Core Section
59
+ # 8.2.1): inside an embedded resource `#` is that resource and
60
+ # `#/$defs/x` its own definitions, never the enclosing document's.
61
+ # @param root [Hash] the root schema (string keys)
62
+ # @param ref [String] the `$ref` value
63
+ # @param dialect [String, nil] the canonical dialect
64
+ # @param resolver [Hash, Context] holder of the memoized anchor index
65
+ # @param from [Hash, nil] the schema object holding the reference
66
+ # @return [Object] the referenced value, or UNRESOLVED
67
+ def resolve_reference(root, ref, dialect, resolver, from: nil)
68
+ index = (resolver[:anchors] ||= anchor_index(root, dialect))
69
+ # A referring schema the index never reached has no known resource:
70
+ # resolving against the document root would apply the wrong `#`.
71
+ return UNRESOLVED if from.is_a?(Hash) && !index[:resources].key?(from)
72
+
73
+ resource = (from && index[:resources][from]) || root
74
+ # A reference outside the fragment space names a resource by URI:
75
+ # the document may bundle it, and then the fragment applies there.
76
+ unless ref.start_with?('#')
77
+ resource, ref = retarget_reference(index, resource, ref)
78
+ return UNRESOLVED unless resource
79
+ end
80
+ raw = ref.delete_prefix('#')
81
+ return resolve_adopted_pointer(resource, ref, index) if raw.empty? || raw.start_with?('/', '%2F', '%2f')
82
+
83
+ # A plain-name fragment is percent-decoded like a pointer fragment
84
+ # (RFC 3986 Section 2.1): "#foo%2Dbar" names the anchor "foo-bar".
85
+ # What counts as a name is the target resource's dialect's business.
86
+ fragment = decoded_fragment(ref)
87
+ return UNRESOLVED unless anchor_name?(fragment, index[:dialects][resource] || dialect)
88
+
89
+ index[:anchors].fetch(resource, {}).fetch(fragment, UNRESOLVED)
90
+ end
91
+
92
+ # The decoded fragment of a reference (RFC 3986 Section 2.1), or nil
93
+ # when what the peer wrote does not decode to readable text: a
94
+ # malformed escape ("a%ZZ") and escapes that are not valid UTF-8 name
95
+ # nothing in this document, and reading them must never raise out of
96
+ # the validation. The undecoded text is never substituted -- it would
97
+ # make "#/$defs/a%ZZ" resolve onto a literal "a%ZZ" member and pass a
98
+ # reference the peer never wrote off as valid.
99
+ # @param ref [String] the `$ref` value
100
+ # @return [String, nil]
101
+ def decoded_fragment(ref)
102
+ decoded = decode_component(ref.delete_prefix('#'))
103
+ decoded if decoded&.valid_encoding?
104
+ end
105
+
106
+ # Percent-decode one component.
107
+ # @param component [String]
108
+ # @return [String, nil] nil when an escape is malformed ("a%ZZ", "a%")
109
+ def decode_component(component)
110
+ URI.decode_uri_component(component)
111
+ rescue ArgumentError
112
+ nil
113
+ end
114
+
115
+ # Resolve a JSON pointer within a schema resource, adopting on the way
116
+ # whatever subtree the pointer enters through a data keyword (`default`
117
+ # and the rest): that subtree is normalized and indexed once — memoized
118
+ # by the identity of the object the document holds — and the remaining
119
+ # tokens are walked through the copy. So every pointer into it lands on
120
+ # the objects the index already knows, with the resource, dialect and
121
+ # subschema charge it gave them: a nested pointer is not a second copy
122
+ # attributed to the referrer (JSON Schema 2020-12 Core Sections 8.1.1
123
+ # and 8.2.1: a schema belongs to the resource of its nearest `$id`
124
+ # ancestor, wherever a reference reached it from).
125
+ # @param resource [Hash] the resource root the pointer starts at
126
+ # @param ref [String] the `$ref` value
127
+ # @param index [Hash] the anchor index
128
+ # @return [Object] the referenced value, or UNRESOLVED
129
+ def resolve_adopted_pointer(resource, ref, index)
130
+ fragment = decoded_fragment(ref)
131
+ return UNRESOLVED unless fragment
132
+ return adopt_reached_target(resource, resource, index) if fragment.empty?
133
+ return UNRESOLVED unless fragment.start_with?('/')
134
+
135
+ node = resource
136
+ mode = :schema
137
+ # RFC 6901 Section 5: the pointer "/" is the member named "", so the
138
+ # leading separator is dropped rather than split off ("" splits to no
139
+ # tokens at all, which would read "#/" as the whole document).
140
+ fragment.split('/', -1).drop(1).each do |token|
141
+ # RFC 6901 Section 3: "~" is only ever followed by "0" or "1".
142
+ return UNRESOLVED if token.match?(/~(?![01])/)
143
+
144
+ node, resource, mode = adopt_step(node, resource, index, mode)
145
+ token = token.gsub('~1', '/').gsub('~0', '~')
146
+ child = pointer_child(node, token)
147
+ return UNRESOLVED if child.equal?(UNRESOLVED)
148
+
149
+ mode = member_mode(token, child, mode)
150
+ node = child
151
+ end
152
+ adopt_reached_target(node, resource, index, raw: mode == :data)
153
+ end
154
+
155
+ # Adopt the object a pointer is about to step through when the document
156
+ # holds it as data, and follow the resource the index attributes it to.
157
+ # @return [Array(Object, Hash, Symbol)] the node to step through, the
158
+ # resource it belongs to, and the mode it is read in
159
+ def adopt_step(node, resource, index, mode)
160
+ return [node, resource, mode] unless node.is_a?(Hash)
161
+
162
+ if mode == :data
163
+ node = adopt_reached_target(node, resource, index, raw: true)
164
+ mode = :schema
165
+ end
166
+ [node, index[:resources][node] || resource, mode]
167
+ end
168
+
169
+ # A schema a pointer reaches outside every indexed position (inside
170
+ # `default`, another data keyword or a vendor member) is still part of
171
+ # the resource it was reached from: it (and what it holds) is indexed
172
+ # on arrival, so a `$ref` written anywhere in it resolves within that
173
+ # resource and its nested schemas follow the dialect it adopts, and
174
+ # what the document holds as data is normalized like everywhere else
175
+ # (values are kept as given for equality; a subtree resolved as a
176
+ # schema gets one string-keyed copy, memoized by identity).
177
+ # @param raw [Boolean] whether the document holds the target as data,
178
+ # so it still needs its string-keyed copy
179
+ # @return [Object] the target (or its normalized copy)
180
+ def adopt_reached_target(target, resource, index, raw: false)
181
+ return target unless target.is_a?(Hash)
182
+
183
+ # Copied under the same structural budget as the root document,
184
+ # whatever key form it arrived in (over the wire every key is a
185
+ # string): a data keyword may not hide an unbounded map behind a
186
+ # pointer.
187
+ target = normalized_copy(target, index) if raw && !index[:resources].key?(target)
188
+ # An indexed position was already walked, charged and attributed.
189
+ return target if index[:resources].key?(target)
190
+
191
+ # The adopted target is indexed like any schema position, so its own
192
+ # `$id` / `$schema` and the positions below it are seen: within the
193
+ # bounds the index already runs under. It declares no names of its
194
+ # own — `resource_root?` turns naming on where an `$id` really starts
195
+ # a resource — since a data keyword is not a schema position and an
196
+ # `$anchor` written inside one names nothing in the document (Core
197
+ # Sections 8.2.2 and 4.3.1).
198
+ index_positions(index, [[target, resource, 0, index[:dialects][resource], false]])
199
+ target
200
+ end
201
+
202
+ # The one string-keyed copy of a subtree the document holds as data,
203
+ # memoized by the identity of the object it holds.
204
+ # @return [Hash]
205
+ def normalized_copy(target, index)
206
+ copies = (index[:normalized] ||= {}.compare_by_identity)
207
+ index[:budget] ||= { objects: 0, deadline: nil }
208
+ copies[target] ||= deep_stringify(target, 0, index[:budget])
209
+ end
210
+
211
+ # Every plain-name anchor in the document, per schema resource:
212
+ # `$anchor` (and `$dynamicAnchor`) in 2019-09 / 2020-12, `$id: "#name"`
213
+ # in draft-07. The walk is bounded like the preflight walk and follows
214
+ # the dialect of each resource (an embedded resource may declare its
215
+ # own `$schema`). Identifiers are taken only from schema positions the
216
+ # dialect walks (JSON Schema 2020-12 Core Sections 8.2.2 and 4.3.1: an
217
+ # identifier belongs to a schema object, and the value of a keyword
218
+ # the dialect does not define is not a schema): a definition bag the
219
+ # dialect does not define stays reachable through JSON pointers, and
220
+ # its objects are attributed to their lexical resource, but it is
221
+ # never a source of names — nor is anything beside a draft-07 `$ref`
222
+ # apart from its `definitions`.
223
+ # @return [Hash] :resources (schema object => its resource root, by
224
+ # identity), :anchors (resource root => name => subschema, first
225
+ # occurrence, by identity), :dialects (resource root => dialect) and
226
+ # :duplicates (names declared more than once within one resource)
227
+ def anchor_index(root, dialect)
228
+ index = { resources: {}.compare_by_identity, anchors: {}.compare_by_identity,
229
+ dialects: {}.compare_by_identity, bases: {}.compare_by_identity, by_base: {},
230
+ duplicates: [], duplicate_ids: [], visited: 0, truncated: false }
231
+ index[:dialects][root] = dialect
232
+ # The document's own base: the root's `$id` where it declares one,
233
+ # else the empty reference. Relative references resolve against each
234
+ # other either way, which is what a bundled document needs.
235
+ register_resource_base(index, root, '', declared: root.is_a?(Hash) && resource_root?(root, dialect))
236
+ index_positions(index, [[root, root, 0, dialect, true]])
237
+ index
238
+ end
239
+
240
+ # Record the base URI a schema resource establishes (RFC 3986 Section
241
+ # 5.1.1: an `$id` is resolved against the base in force where it is
242
+ # written), and the resource that base names. The first declaration of
243
+ # a base wins, as the first declaration of an anchor name does.
244
+ # @param declared [Boolean] whether the resource declares an `$id`
245
+ # @return [void]
246
+ def register_resource_base(index, resource, parent_base, declared: true)
247
+ id = declared && resource.is_a?(Hash) ? resource['$id'] : nil
248
+ base = (id.is_a?(String) ? merge_uri(parent_base, id) : parent_base) || parent_base
249
+ index[:bases][resource] = base
250
+ known = index[:by_base][base]
251
+ return index[:by_base][base] = resource if known.nil?
252
+
253
+ # Two resources answering to one URI: which one a reference lands on
254
+ # would depend on the order the walk met them in, so the document is
255
+ # unusable rather than resolved by luck (JSON Schema 2020-12 Core
256
+ # Section 9.1.2). Only a declared `$id` can collide; the base a
257
+ # resource merely inherits is its parent's, already registered.
258
+ index[:duplicate_ids] << base if id.is_a?(String) && !known.equal?(resource)
259
+ end
260
+
261
+ # Index every schema position reachable from the pending seeds, within
262
+ # the visit and depth bounds the whole index runs under (an adopted
263
+ # pointer target is seeded here too, so it shares them).
264
+ # @param index [Hash] the index being built
265
+ # @param pending [Array<Array>] seeds: schema, resource, depth,
266
+ # dialect, whether the position may declare names
267
+ # @return [void]
268
+ def index_positions(index, pending)
269
+ until pending.empty? || index[:visited] >= MAX_SUBSCHEMAS
270
+ schema, resource, depth, dialect, named = pending.shift
271
+ next unless schema.is_a?(Hash)
272
+ next if index[:resources].key?(schema)
273
+ # A schema below the depth bound is left unindexed just like one
274
+ # beyond the visit bound: a reference from it (or an `$id` there)
275
+ # would otherwise resolve under guesses.
276
+ (index[:truncated] = true) && next if depth > MAX_SCHEMA_DEPTH
277
+
278
+ index[:visited] += 1
279
+ resource, dialect, named = enter_resource(index, schema, resource, dialect, named)
280
+ index[:resources][schema] = resource
281
+ if named && !(dialect == DRAFT_07 && schema.key?('$ref'))
282
+ record_anchor_names(index, resource, schema,
283
+ dialect)
284
+ end
285
+ each_walked_position(schema, dialect) { |sub| pending << [sub, resource, depth + 1, dialect, named] }
286
+ each_foreign_definition(schema, dialect) { |sub| pending << [sub, resource, depth + 1, dialect, false] }
287
+ end
288
+ # Objects left unindexed at the bound would resolve and validate
289
+ # under guesses; the index says so and the schema is unusable.
290
+ index[:truncated] ||= pending.any? { |schema, *| schema.is_a?(Hash) && !index[:resources].key?(schema) }
291
+ end
292
+
293
+ # Enter the schema resource a position starts, if it starts one: a
294
+ # resource is a schema wherever it sits (one reached through a bag the
295
+ # dialect does not walk names its own anchors, though nothing outside
296
+ # it can see them), it may declare its own dialect, and its `$id`
297
+ # establishes the base its references resolve against.
298
+ # @return [Array(Hash, String, Boolean)] the resource in force, its
299
+ # dialect, and whether the position may declare names
300
+ def enter_resource(index, schema, resource, dialect, named)
301
+ return [resource, dialect, named] unless resource_root?(schema, dialect)
302
+
303
+ parent_base = index[:bases][resource] || ''
304
+ dialect = embedded_dialect(schema, dialect) || dialect
305
+ index[:dialects][schema] = dialect
306
+ register_resource_base(index, schema, parent_base)
307
+ [schema, dialect, true]
308
+ end
309
+
310
+ # Record the plain names a schema object declares for its resource; a
311
+ # name already bound to another object of the same resource is a
312
+ # duplicate (anchor names are unique within a resource).
313
+ # @return [void]
314
+ def record_anchor_names(index, resource, schema, dialect)
315
+ names = (index[:anchors][resource] ||= {})
316
+ anchor_names(schema, dialect).each do |name|
317
+ index[:duplicates] << name if names.key?(name) && !names[name].equal?(schema)
318
+ names[name] ||= schema
319
+ end
320
+ end
321
+
322
+ # Yield the schema positions the dialect walks under a schema object:
323
+ # under a draft-07 `$ref` only the `definitions` bag, else every
324
+ # subschema (the dialect's definition bag included).
325
+ # @return [void]
326
+ def each_walked_position(schema, dialect, &)
327
+ if dialect == DRAFT_07 && schema.key?('$ref')
328
+ each_definition(schema, dialect, &)
329
+ else
330
+ each_subschema(schema, dialect, &)
331
+ end
332
+ end
333
+
334
+ # The dialect the memoized anchor index recorded for a schema object's
335
+ # resource, or nil when the object was not indexed.
336
+ # @param schema [Hash]
337
+ # @param resolver [Hash, Context] holder of the memoized anchor index
338
+ # @return [String, nil]
339
+ def indexed_dialect(schema, resolver)
340
+ index = resolver[:anchors]
341
+ return nil unless index
342
+
343
+ resource = index[:resources][schema]
344
+ resource && index[:dialects][resource]
345
+ end
346
+
347
+ # Whether a schema object starts a new schema resource: its `$id` is a
348
+ # URI rather than a bare fragment (a draft-07 `$id: "#name"` is a
349
+ # plain-name identifier, not a base).
350
+ # @param schema [Hash]
351
+ # @return [Boolean]
352
+ def resource_start?(schema)
353
+ id = schema['$id']
354
+ id.is_a?(String) && !id.empty? && !id.start_with?('#')
355
+ end
356
+
357
+ # {#resource_start?} in the dialect in force: under draft-07 a `$ref`
358
+ # replaces its whole schema object, `$id` and `$schema` included, so
359
+ # nothing beside it starts a resource.
360
+ # @param schema [Hash]
361
+ # @param dialect [String, nil]
362
+ # @return [Boolean]
363
+ def resource_root?(schema, dialect)
364
+ return false if dialect == DRAFT_07 && schema.key?('$ref')
365
+
366
+ resource_start?(schema)
367
+ end
368
+
369
+ # How a dynamic reference binds (JSON Schema 2020-12 Core Section
370
+ # 8.2.3.2; 2019-09 Section 8.2.4.2.2). A reference whose initial
371
+ # target declares no matching dynamic anchor is the plain reference it
372
+ # resolves to (`:plain`). One that does re-binds to the declaration in
373
+ # the OUTERMOST resource of the dynamic scope — the resources the
374
+ # evaluation entered on its way here, which {SchemaValidator.entered_scope?}
375
+ # records as they are entered (`:bound`, with the target). What the
376
+ # document holds elsewhere decides nothing: a resource the instance
377
+ # never entered is not in the scope, so a duplicate anchor there is
378
+ # neither ambiguity nor a reason to leave the reference unevaluated.
379
+ # Outside a validation there is no scope to read, and every reference
380
+ # is the plain one it resolves to — which is what the preflight, whose
381
+ # job is that the reference resolves at all, needs.
382
+ # @param schema [Hash] the schema object holding the reference
383
+ # @param keyword [String] "$dynamicRef" or "$recursiveRef"
384
+ # @param root [Hash] the root schema
385
+ # @param dialect [String, nil] the canonical root dialect
386
+ # @param resolver [Hash, Context] holder of the memoized anchor index,
387
+ # and — during a validation — of the dynamic scope
388
+ # @return [Array] [:plain] or [:bound, target]
389
+ def dynamic_binding(schema, keyword, root, dialect, resolver)
390
+ ref = schema[keyword]
391
+ return [:plain] unless ref.is_a?(String) && !external_ref?(ref, root, dialect, resolver, from: schema)
392
+
393
+ target = resolve_reference(root, ref, dialect, resolver, from: schema)
394
+ return [:plain] if target.equal?(UNRESOLVED)
395
+ return [:plain] unless target.is_a?(Hash)
396
+
397
+ index = resolver[:anchors]
398
+ scope = resolver[:scope]
399
+ bound = if keyword == '$recursiveRef'
400
+ return [:plain] unless target['$recursiveAnchor'] == true
401
+
402
+ outermost_recursive_anchor(index, scope, dialect)
403
+ else
404
+ fragment = ref.include?('#') ? decoded_fragment(ref[ref.index('#')..]) : nil
405
+ return [:plain] if fragment.nil? || fragment.empty? || fragment.start_with?('/')
406
+ return [:plain] unless target['$dynamicAnchor'] == fragment
407
+
408
+ outermost_dynamic_anchor(index, scope, fragment)
409
+ end
410
+ bound ? [:bound, bound] : [:plain]
411
+ end
412
+
413
+ # The schema declaring `$dynamicAnchor: name` in the outermost resource
414
+ # of the dynamic scope that declares it — the scope being the resources
415
+ # the evaluation actually entered, outermost first. A resource the
416
+ # instance never entered declares nothing for this reference, however
417
+ # many of them the document holds; and where no entered resource
418
+ # declares the name, the caller keeps the target the reference resolved
419
+ # to on its own, which is what the specification's "otherwise behave as
420
+ # $ref" says.
421
+ # @param index [Hash] the anchor index
422
+ # @param scope [Array<Hash>, nil] the dynamic scope, outermost first
423
+ # @param name [String] the anchor name
424
+ # @return [Hash, nil]
425
+ def outermost_dynamic_anchor(index, scope, name)
426
+ Array(scope).each do |resource|
427
+ declaring = index[:anchors][resource]&.[](name)
428
+ return declaring if declaring.is_a?(Hash) && declaring['$dynamicAnchor'] == name
429
+ end
430
+ nil
431
+ end
432
+
433
+ # The outermost resource of the dynamic scope whose `$recursiveAnchor`
434
+ # is true (2019-09 Core Section 8.2.4.2.2); the target of a
435
+ # `$recursiveRef` is the resource root itself.
436
+ # @return [Hash, nil]
437
+ def outermost_recursive_anchor(index, scope, dialect)
438
+ Array(scope).find do |resource|
439
+ resource.is_a?(Hash) && resource['$recursiveAnchor'] == true &&
440
+ (index[:dialects][resource] || dialect) != DRAFT_07
441
+ end
442
+ end
443
+
444
+ # @return [Array<String>] the plain names a schema object declares
445
+ def anchor_names(schema, dialect)
446
+ names = if dialect == DRAFT_07
447
+ # draft-07 Core Section 8.2.3: only an $id that is exactly a
448
+ # fragment is a plain-name identifier. An $id is a URI
449
+ # reference, so its fragment is percent-decoded (RFC 3986
450
+ # Section 2.1) exactly as a $ref's is: `$id: "#foo%2Dbar"`
451
+ # declares the name "foo-bar", which is what
452
+ # `$ref: "#foo%2Dbar"` — decoded the same way — looks for.
453
+ id = schema['$id']
454
+ [id.is_a?(String) && id.start_with?('#') ? decoded_fragment(id) : nil]
455
+ else
456
+ %w[$anchor $dynamicAnchor].map { |k| schema[k] if keyword_known?(k, dialect) }
457
+ end
458
+ names.select { |name| anchor_name?(name, dialect) }
459
+ end
460
+
461
+ # The lexical nesting depth of every schema object reachable from the
462
+ # root (subschema positions, definition bags of any dialect), so a
463
+ # referenced target is bounded by where it is written, not by where it
464
+ # is referenced from: neither member order nor reference fan-out can
465
+ # change the verdict.
466
+ # @return [Hash{Hash => Integer}] identity-keyed
467
+ def lexical_depths(root, dialect)
468
+ depths = {}.compare_by_identity
469
+ pending = [[root, 0, dialect]]
470
+ while (schema, depth, current = pending.shift)
471
+ next unless schema.is_a?(Hash) && !depths.key?(schema)
472
+
473
+ depths[schema] = depth
474
+ break if depths.size > MAX_SUBSCHEMAS * 2
475
+
476
+ # An embedded resource's positions follow its own dialect.
477
+ current = embedded_dialect(schema, current) || current
478
+ each_subschema(schema, current) { |sub| pending << [sub, depth + 1, current] }
479
+ each_foreign_definition(schema, current) { |sub| pending << [sub, depth + 1, current] }
480
+ end
481
+ depths
482
+ end
483
+
484
+ # The lexical depth of the value a pointer reference reaches, counted
485
+ # in schema steps along the (percent-decoded) pointer from its resource
486
+ # root: a keyword holding one subschema is one step, a map or array of
487
+ # subschemas is one step per member (`#/properties/b` and `#/allOf/0`
488
+ # are both one below the enclosing schema), and every token under a
489
+ # data or unknown keyword is a step, so a document hidden inside
490
+ # `default`, `enum`, `const`, `examples` or a vendor keyword obeys the
491
+ # same bound as one written in a schema position. Keywords are
492
+ # classified by the dialect in force at each node (an embedded
493
+ # resource's own `$schema` takes over when the pointer enters it). A
494
+ # schema object whose depth the lexical index knows resets the count
495
+ # to that depth — unless the pointer already passed through an opaque
496
+ # keyword, after which every token counts (the index placed such an
497
+ # object under another dialect's grammar).
498
+ # @return [Integer, nil] nil when the pointer cannot be followed
499
+ def referenced_position_depth(ref, root, dialect, counter, from)
500
+ pointer_position(ref, root, dialect, counter, from)&.first
501
+ end
502
+
503
+ # @return [Array(Integer, Boolean), nil] the depth and whether the
504
+ # pointer crossed an opaque keyword; nil when it cannot be followed
505
+ def pointer_position(ref, root, dialect, counter, from)
506
+ index = (counter[:anchors] ||= anchor_index(root, dialect))
507
+ node, ref = pointer_origin(index, root, ref, from)
508
+ return nil unless node
509
+
510
+ tokens = pointer_tokens(ref)
511
+ return nil unless tokens
512
+
513
+ depths = counter[:depths] || {}
514
+ walk = { index: index, depths: depths, dialect: index[:dialects][node] || dialect,
515
+ depth: depths[node] || 0, mode: :schema, opaque: false, visited: true }
516
+ tokens.each do |token|
517
+ child = pointer_child(node, token)
518
+ return nil if child.equal?(UNRESOLVED)
519
+
520
+ pointer_step(walk, node, token)
521
+ node = child
522
+ end
523
+ [walk[:depth], walk[:opaque], walk[:visited]]
524
+ end
525
+
526
+ # Advance one pointer token: in schema mode the node's own dialect
527
+ # and known depth apply and the keyword decides how the next token
528
+ # counts; inside a map or array keyword the member is the step;
529
+ # under an opaque keyword every token is a step.
530
+ # @return [void]
531
+ def pointer_step(walk, node, token)
532
+ unless walk[:mode] == :schema
533
+ walk[:depth] += 1
534
+ walk[:mode] = :schema if %i[map array].include?(walk[:mode])
535
+ return
536
+ end
537
+ if node.is_a?(Hash)
538
+ walk[:depth] = walk[:depths][node] if !walk[:opaque] && walk[:depths].key?(node)
539
+ walk[:dialect] = walk[:index][:dialects][node] || embedded_dialect(node, walk[:dialect]) || walk[:dialect]
540
+ end
541
+ walk[:mode] = pointer_step_mode(node, token, walk[:dialect])
542
+ walk[:opaque] ||= walk[:mode] == :opaque
543
+ # The preflight walk does not descend into an opaque keyword, nor
544
+ # into anything beside a draft-07 $ref but its definitions.
545
+ walk[:visited] &&= walk[:mode] != :opaque &&
546
+ !(walk[:dialect] == DRAFT_07 && node.is_a?(Hash) && node.key?('$ref') &&
547
+ token != 'definitions')
548
+ walk[:depth] += 1 unless %i[map array].include?(walk[:mode])
549
+ end
550
+
551
+ # The resource a reference's pointer is read inside, and the bare
552
+ # fragment that applies there. A reference written as an absolute URI
553
+ # into the bundled document (`urn:root#/x/y`) addresses the very
554
+ # position its bare spelling (`#/x/y`) does, so the pointer is followed
555
+ # — and the target accounted for — the same way whichever the peer
556
+ # wrote (JSON Schema 2020-12 Core Section 9.3.1).
557
+ # @return [Array(Hash, String), Array(nil, nil)]
558
+ def pointer_origin(index, root, ref, from)
559
+ resource = (from && index[:resources][from]) || root
560
+ return [resource, ref] if ref.start_with?('#')
561
+
562
+ retarget_reference(index, resource, ref)
563
+ end
564
+
565
+ # The decoded RFC 6901 tokens of a fragment pointer.
566
+ # @return [Array<String>, nil] nil unless the reference is a pointer
567
+ def pointer_tokens(ref)
568
+ fragment = decoded_fragment(ref)
569
+ return nil unless fragment&.start_with?('/')
570
+
571
+ fragment.split('/', -1).drop(1).map { |token| token.gsub('~1', '/').gsub('~0', '~') }
572
+ end
573
+
574
+ # @return [Object] the member a pointer token selects, or UNRESOLVED
575
+ def pointer_child(node, token)
576
+ case node
577
+ when Hash then node.key?(token) ? node[token] : UNRESOLVED
578
+ when Array then token.match?(/\A(0|[1-9]\d*)\z/) && token.to_i < node.length ? node[token.to_i] : UNRESOLVED
579
+ else UNRESOLVED
580
+ end
581
+ end
582
+
583
+ # How a keyword of a schema object holds what its pointer token reaches.
584
+ # @return [Symbol] :schema (one subschema), :map, :array, or :opaque
585
+ # (data or unknown keyword: every token below is a step)
586
+ # A keyword the dialect in force does not define (`prefixItems` under
587
+ # draft-07, `additionalItems` under 2020-12) is opaque data there.
588
+ def pointer_step_mode(node, token, dialect)
589
+ return :opaque unless node.is_a?(Hash) && keyword_known?(token, dialect)
590
+ return :map if SUBSCHEMA_MAP_KEYWORDS.include?(token)
591
+ return :array if SUBSCHEMA_ARRAY_KEYWORDS.include?(token) || (token == 'items' && node[token].is_a?(Array))
592
+ return :schema if SUBSCHEMA_KEYWORDS.include?(token)
593
+
594
+ :opaque
595
+ end
596
+
597
+ # Yield the definitions held in the bag the dialect does not define
598
+ # (`$defs` under draft-07, which predates it): unknown to the dialect,
599
+ # but pointer-addressable all the same.
600
+ # @return [void]
601
+ def each_foreign_definition(schema, dialect, &block)
602
+ %w[$defs definitions].each do |keyword|
603
+ next if keyword_known?(keyword, dialect)
604
+
605
+ schema[keyword].each_value(&block) if schema[keyword].is_a?(Hash)
606
+ end
607
+ end
608
+ end
609
+ end
610
+ end