ruby-mcp-client 2.1.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/OAUTH.md +555 -0
- data/README.md +825 -48
- data/lib/mcp_client/audio_content.rb +1 -1
- data/lib/mcp_client/auth/browser_oauth.rb +131 -21
- data/lib/mcp_client/auth/oauth_provider/challenge_handling.rb +532 -0
- data/lib/mcp_client/auth/oauth_provider/client_authentication.rb +121 -0
- data/lib/mcp_client/auth/oauth_provider/pending_requests.rb +51 -0
- data/lib/mcp_client/auth/oauth_provider/registration_store.rb +486 -0
- data/lib/mcp_client/auth/oauth_provider/response_validation.rb +441 -0
- data/lib/mcp_client/auth/oauth_provider/scope_selection.rb +134 -0
- data/lib/mcp_client/auth/oauth_provider/token_store.rb +419 -0
- data/lib/mcp_client/auth/oauth_provider.rb +1354 -386
- data/lib/mcp_client/auth/peer_text.rb +174 -0
- data/lib/mcp_client/auth.rb +298 -32
- data/lib/mcp_client/cached_result.rb +145 -0
- data/lib/mcp_client/called_tool_definition.rb +138 -0
- data/lib/mcp_client/client/cache_slices.rb +195 -0
- data/lib/mcp_client/client/list_aggregation.rb +243 -0
- data/lib/mcp_client/client/notification_routing.rb +155 -0
- data/lib/mcp_client/client/sampling_validation.rb +200 -0
- data/lib/mcp_client/client/task_api.rb +531 -0
- data/lib/mcp_client/client/task_lifetimes.rb +269 -0
- data/lib/mcp_client/client/task_registry.rb +254 -0
- data/lib/mcp_client/client/task_shape.rb +102 -0
- data/lib/mcp_client/client/task_support.rb +1166 -0
- data/lib/mcp_client/client/task_updates.rb +457 -0
- data/lib/mcp_client/client/task_wait_boundaries.rb +198 -0
- data/lib/mcp_client/client/task_workers.rb +63 -0
- data/lib/mcp_client/client.rb +796 -518
- data/lib/mcp_client/deep_copy.rb +49 -0
- data/lib/mcp_client/deprecation_notices.rb +94 -0
- data/lib/mcp_client/deprecations.rb +419 -0
- data/lib/mcp_client/errors.rb +474 -7
- data/lib/mcp_client/header_params.rb +320 -0
- data/lib/mcp_client/http_transport_base/bounded_inflate.rb +41 -0
- data/lib/mcp_client/http_transport_base/cache_support.rb +694 -0
- data/lib/mcp_client/http_transport_base/era_detection.rb +134 -0
- data/lib/mcp_client/http_transport_base/listen_stream.rb +763 -0
- data/lib/mcp_client/http_transport_base/param_headers.rb +35 -0
- data/lib/mcp_client/http_transport_base/request_recovery.rb +156 -0
- data/lib/mcp_client/http_transport_base/session_recovery.rb +113 -0
- data/lib/mcp_client/http_transport_base/sse_event_scanner.rb +145 -0
- data/lib/mcp_client/http_transport_base/stream_capture.rb +160 -0
- data/lib/mcp_client/http_transport_base/stream_recovery.rb +318 -0
- data/lib/mcp_client/http_transport_base/tool_listing.rb +277 -0
- data/lib/mcp_client/http_transport_base.rb +666 -120
- data/lib/mcp_client/input_round_trips.rb +128 -0
- data/lib/mcp_client/json_rpc_common/envelopes.rb +32 -0
- data/lib/mcp_client/json_rpc_common/error_bodies.rb +105 -0
- data/lib/mcp_client/json_rpc_common/input_waits.rb +167 -0
- data/lib/mcp_client/json_rpc_common.rb +900 -13
- data/lib/mcp_client/oauth_client.rb +14 -5
- data/lib/mcp_client/prompt.rb +4 -0
- data/lib/mcp_client/request_authorization.rb +128 -0
- data/lib/mcp_client/request_meta_scope.rb +77 -0
- data/lib/mcp_client/request_metadata.rb +287 -0
- data/lib/mcp_client/resource.rb +4 -0
- data/lib/mcp_client/resource_content.rb +20 -0
- data/lib/mcp_client/resource_template.rb +4 -0
- data/lib/mcp_client/result_caching.rb +999 -0
- data/lib/mcp_client/result_completeness.rb +34 -0
- data/lib/mcp_client/root.rb +6 -0
- data/lib/mcp_client/round_trip_marker.rb +28 -0
- data/lib/mcp_client/schema_validator/annotations.rb +82 -0
- data/lib/mcp_client/schema_validator/composition.rb +86 -0
- data/lib/mcp_client/schema_validator/dialects.rb +66 -0
- data/lib/mcp_client/schema_validator/ecma_patterns.rb +567 -0
- data/lib/mcp_client/schema_validator/evaluation.rb +517 -0
- data/lib/mcp_client/schema_validator/input_requirements.rb +84 -0
- data/lib/mcp_client/schema_validator/instances.rb +449 -0
- data/lib/mcp_client/schema_validator/keyword_scan.rb +121 -0
- data/lib/mcp_client/schema_validator/normalization.rb +104 -0
- data/lib/mcp_client/schema_validator/references.rb +610 -0
- data/lib/mcp_client/schema_validator/scalars.rb +126 -0
- data/lib/mcp_client/schema_validator/shapes.rb +319 -0
- data/lib/mcp_client/schema_validator/uri_references.rb +153 -0
- data/lib/mcp_client/schema_validator.rb +882 -208
- data/lib/mcp_client/server_base.rb +233 -5
- data/lib/mcp_client/server_factory.rb +9 -3
- data/lib/mcp_client/server_http/json_rpc_transport.rb +219 -4
- data/lib/mcp_client/server_http.rb +307 -90
- data/lib/mcp_client/server_sse/json_rpc_transport.rb +113 -25
- data/lib/mcp_client/server_sse/sse_parser.rb +39 -6
- data/lib/mcp_client/server_sse.rb +227 -62
- data/lib/mcp_client/server_stdio/child_session.rb +98 -0
- data/lib/mcp_client/server_stdio/json_rpc_transport.rb +1003 -28
- data/lib/mcp_client/server_stdio.rb +772 -183
- data/lib/mcp_client/server_streamable_http/json_rpc_transport.rb +189 -25
- data/lib/mcp_client/server_streamable_http.rb +302 -115
- data/lib/mcp_client/session_pin.rb +119 -0
- data/lib/mcp_client/subscription/notification_dispatcher.rb +354 -0
- data/lib/mcp_client/subscription.rb +852 -0
- data/lib/mcp_client/subscription_support.rb +715 -0
- data/lib/mcp_client/task.rb +286 -14
- data/lib/mcp_client/tool.rb +31 -3
- data/lib/mcp_client/version.rb +21 -6
- data/lib/mcp_client.rb +108 -19
- metadata +68 -2
|
@@ -0,0 +1,610 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'uri'
|
|
4
|
+
|
|
5
|
+
module MCPClient
|
|
6
|
+
module SchemaValidator
|
|
7
|
+
# Resolution of local `$ref` values: JSON pointer fragments (RFC 6901)
|
|
8
|
+
# and plain-name fragments naming an anchor. Nothing here ever fetches:
|
|
9
|
+
# a reference outside the document is reported as external. Extended
|
|
10
|
+
# into SchemaValidator, so the methods are its own.
|
|
11
|
+
#
|
|
12
|
+
# Plain names are scoped to their schema resource (JSON Schema 2020-12
|
|
13
|
+
# Core Sections 8.2.1 and 8.2.2): a subschema whose `$id` is a URI
|
|
14
|
+
# starts a new resource, and `#name` names an anchor of the resource the
|
|
15
|
+
# referencing schema belongs to — never one of an embedded resource, and
|
|
16
|
+
# never one of the enclosing document from inside an embedded resource.
|
|
17
|
+
module References
|
|
18
|
+
# A reference that does not point inside this document, so using it
|
|
19
|
+
# would need a retrieval that never happens. A bare fragment is always
|
|
20
|
+
# local; anything else is resolved against the base URI of the
|
|
21
|
+
# resource holding it (RFC 3986 Section 5.2) and is local when the
|
|
22
|
+
# document bundles a resource whose `$id` is that URI (JSON Schema
|
|
23
|
+
# 2020-12 Core Section 9.3.1) -- the empty reference, which names the
|
|
24
|
+
# base itself, included. Without the document and its index only the
|
|
25
|
+
# syntactic answer is available, and a reference outside the fragment
|
|
26
|
+
# space is external.
|
|
27
|
+
# @param ref [String]
|
|
28
|
+
# @param root [Hash, nil] the normalized root schema
|
|
29
|
+
# @param dialect [String, nil] the canonical dialect
|
|
30
|
+
# @param resolver [Hash, Context, nil] holder of the memoized index
|
|
31
|
+
# @param from [Hash, nil] the schema object holding the reference
|
|
32
|
+
# @return [Boolean]
|
|
33
|
+
def external_ref?(ref, root = nil, dialect = nil, resolver = nil, from: nil)
|
|
34
|
+
return false if ref.start_with?('#')
|
|
35
|
+
return true unless root.is_a?(Hash) && resolver
|
|
36
|
+
|
|
37
|
+
index = (resolver[:anchors] ||= anchor_index(root, dialect))
|
|
38
|
+
return true if from.is_a?(Hash) && !index[:resources].key?(from)
|
|
39
|
+
|
|
40
|
+
retarget_reference(index, (from && index[:resources][from]) || root, ref).first.nil?
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# The bundled resource a reference outside the fragment space names,
|
|
44
|
+
# with the bare fragment left to resolve inside it.
|
|
45
|
+
# @param index [Hash] the anchor index
|
|
46
|
+
# @param resource [Hash] the resource the reference is written in
|
|
47
|
+
# @param ref [String] the `$ref` value
|
|
48
|
+
# @return [Array(Hash, String), Array(nil, nil)]
|
|
49
|
+
def retarget_reference(index, resource, ref)
|
|
50
|
+
uri, fragment = ref.split('#', 2)
|
|
51
|
+
base = merge_uri(index[:bases][resource] || '', uri.to_s)
|
|
52
|
+
target = base && index[:by_base][base]
|
|
53
|
+
target ? [target, "##{fragment}"] : [nil, nil]
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# Resolve a local reference: a JSON pointer fragment, or a plain-name
|
|
57
|
+
# fragment naming an anchor. Both are relative to the schema resource
|
|
58
|
+
# the referencing schema belongs to (JSON Schema 2020-12 Core Section
|
|
59
|
+
# 8.2.1): inside an embedded resource `#` is that resource and
|
|
60
|
+
# `#/$defs/x` its own definitions, never the enclosing document's.
|
|
61
|
+
# @param root [Hash] the root schema (string keys)
|
|
62
|
+
# @param ref [String] the `$ref` value
|
|
63
|
+
# @param dialect [String, nil] the canonical dialect
|
|
64
|
+
# @param resolver [Hash, Context] holder of the memoized anchor index
|
|
65
|
+
# @param from [Hash, nil] the schema object holding the reference
|
|
66
|
+
# @return [Object] the referenced value, or UNRESOLVED
|
|
67
|
+
def resolve_reference(root, ref, dialect, resolver, from: nil)
|
|
68
|
+
index = (resolver[:anchors] ||= anchor_index(root, dialect))
|
|
69
|
+
# A referring schema the index never reached has no known resource:
|
|
70
|
+
# resolving against the document root would apply the wrong `#`.
|
|
71
|
+
return UNRESOLVED if from.is_a?(Hash) && !index[:resources].key?(from)
|
|
72
|
+
|
|
73
|
+
resource = (from && index[:resources][from]) || root
|
|
74
|
+
# A reference outside the fragment space names a resource by URI:
|
|
75
|
+
# the document may bundle it, and then the fragment applies there.
|
|
76
|
+
unless ref.start_with?('#')
|
|
77
|
+
resource, ref = retarget_reference(index, resource, ref)
|
|
78
|
+
return UNRESOLVED unless resource
|
|
79
|
+
end
|
|
80
|
+
raw = ref.delete_prefix('#')
|
|
81
|
+
return resolve_adopted_pointer(resource, ref, index) if raw.empty? || raw.start_with?('/', '%2F', '%2f')
|
|
82
|
+
|
|
83
|
+
# A plain-name fragment is percent-decoded like a pointer fragment
|
|
84
|
+
# (RFC 3986 Section 2.1): "#foo%2Dbar" names the anchor "foo-bar".
|
|
85
|
+
# What counts as a name is the target resource's dialect's business.
|
|
86
|
+
fragment = decoded_fragment(ref)
|
|
87
|
+
return UNRESOLVED unless anchor_name?(fragment, index[:dialects][resource] || dialect)
|
|
88
|
+
|
|
89
|
+
index[:anchors].fetch(resource, {}).fetch(fragment, UNRESOLVED)
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
# The decoded fragment of a reference (RFC 3986 Section 2.1), or nil
|
|
93
|
+
# when what the peer wrote does not decode to readable text: a
|
|
94
|
+
# malformed escape ("a%ZZ") and escapes that are not valid UTF-8 name
|
|
95
|
+
# nothing in this document, and reading them must never raise out of
|
|
96
|
+
# the validation. The undecoded text is never substituted -- it would
|
|
97
|
+
# make "#/$defs/a%ZZ" resolve onto a literal "a%ZZ" member and pass a
|
|
98
|
+
# reference the peer never wrote off as valid.
|
|
99
|
+
# @param ref [String] the `$ref` value
|
|
100
|
+
# @return [String, nil]
|
|
101
|
+
def decoded_fragment(ref)
|
|
102
|
+
decoded = decode_component(ref.delete_prefix('#'))
|
|
103
|
+
decoded if decoded&.valid_encoding?
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
# Percent-decode one component.
|
|
107
|
+
# @param component [String]
|
|
108
|
+
# @return [String, nil] nil when an escape is malformed ("a%ZZ", "a%")
|
|
109
|
+
def decode_component(component)
|
|
110
|
+
URI.decode_uri_component(component)
|
|
111
|
+
rescue ArgumentError
|
|
112
|
+
nil
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
# Resolve a JSON pointer within a schema resource, adopting on the way
|
|
116
|
+
# whatever subtree the pointer enters through a data keyword (`default`
|
|
117
|
+
# and the rest): that subtree is normalized and indexed once — memoized
|
|
118
|
+
# by the identity of the object the document holds — and the remaining
|
|
119
|
+
# tokens are walked through the copy. So every pointer into it lands on
|
|
120
|
+
# the objects the index already knows, with the resource, dialect and
|
|
121
|
+
# subschema charge it gave them: a nested pointer is not a second copy
|
|
122
|
+
# attributed to the referrer (JSON Schema 2020-12 Core Sections 8.1.1
|
|
123
|
+
# and 8.2.1: a schema belongs to the resource of its nearest `$id`
|
|
124
|
+
# ancestor, wherever a reference reached it from).
|
|
125
|
+
# @param resource [Hash] the resource root the pointer starts at
|
|
126
|
+
# @param ref [String] the `$ref` value
|
|
127
|
+
# @param index [Hash] the anchor index
|
|
128
|
+
# @return [Object] the referenced value, or UNRESOLVED
|
|
129
|
+
def resolve_adopted_pointer(resource, ref, index)
|
|
130
|
+
fragment = decoded_fragment(ref)
|
|
131
|
+
return UNRESOLVED unless fragment
|
|
132
|
+
return adopt_reached_target(resource, resource, index) if fragment.empty?
|
|
133
|
+
return UNRESOLVED unless fragment.start_with?('/')
|
|
134
|
+
|
|
135
|
+
node = resource
|
|
136
|
+
mode = :schema
|
|
137
|
+
# RFC 6901 Section 5: the pointer "/" is the member named "", so the
|
|
138
|
+
# leading separator is dropped rather than split off ("" splits to no
|
|
139
|
+
# tokens at all, which would read "#/" as the whole document).
|
|
140
|
+
fragment.split('/', -1).drop(1).each do |token|
|
|
141
|
+
# RFC 6901 Section 3: "~" is only ever followed by "0" or "1".
|
|
142
|
+
return UNRESOLVED if token.match?(/~(?![01])/)
|
|
143
|
+
|
|
144
|
+
node, resource, mode = adopt_step(node, resource, index, mode)
|
|
145
|
+
token = token.gsub('~1', '/').gsub('~0', '~')
|
|
146
|
+
child = pointer_child(node, token)
|
|
147
|
+
return UNRESOLVED if child.equal?(UNRESOLVED)
|
|
148
|
+
|
|
149
|
+
mode = member_mode(token, child, mode)
|
|
150
|
+
node = child
|
|
151
|
+
end
|
|
152
|
+
adopt_reached_target(node, resource, index, raw: mode == :data)
|
|
153
|
+
end
|
|
154
|
+
|
|
155
|
+
# Adopt the object a pointer is about to step through when the document
|
|
156
|
+
# holds it as data, and follow the resource the index attributes it to.
|
|
157
|
+
# @return [Array(Object, Hash, Symbol)] the node to step through, the
|
|
158
|
+
# resource it belongs to, and the mode it is read in
|
|
159
|
+
def adopt_step(node, resource, index, mode)
|
|
160
|
+
return [node, resource, mode] unless node.is_a?(Hash)
|
|
161
|
+
|
|
162
|
+
if mode == :data
|
|
163
|
+
node = adopt_reached_target(node, resource, index, raw: true)
|
|
164
|
+
mode = :schema
|
|
165
|
+
end
|
|
166
|
+
[node, index[:resources][node] || resource, mode]
|
|
167
|
+
end
|
|
168
|
+
|
|
169
|
+
# A schema a pointer reaches outside every indexed position (inside
|
|
170
|
+
# `default`, another data keyword or a vendor member) is still part of
|
|
171
|
+
# the resource it was reached from: it (and what it holds) is indexed
|
|
172
|
+
# on arrival, so a `$ref` written anywhere in it resolves within that
|
|
173
|
+
# resource and its nested schemas follow the dialect it adopts, and
|
|
174
|
+
# what the document holds as data is normalized like everywhere else
|
|
175
|
+
# (values are kept as given for equality; a subtree resolved as a
|
|
176
|
+
# schema gets one string-keyed copy, memoized by identity).
|
|
177
|
+
# @param raw [Boolean] whether the document holds the target as data,
|
|
178
|
+
# so it still needs its string-keyed copy
|
|
179
|
+
# @return [Object] the target (or its normalized copy)
|
|
180
|
+
def adopt_reached_target(target, resource, index, raw: false)
|
|
181
|
+
return target unless target.is_a?(Hash)
|
|
182
|
+
|
|
183
|
+
# Copied under the same structural budget as the root document,
|
|
184
|
+
# whatever key form it arrived in (over the wire every key is a
|
|
185
|
+
# string): a data keyword may not hide an unbounded map behind a
|
|
186
|
+
# pointer.
|
|
187
|
+
target = normalized_copy(target, index) if raw && !index[:resources].key?(target)
|
|
188
|
+
# An indexed position was already walked, charged and attributed.
|
|
189
|
+
return target if index[:resources].key?(target)
|
|
190
|
+
|
|
191
|
+
# The adopted target is indexed like any schema position, so its own
|
|
192
|
+
# `$id` / `$schema` and the positions below it are seen: within the
|
|
193
|
+
# bounds the index already runs under. It declares no names of its
|
|
194
|
+
# own — `resource_root?` turns naming on where an `$id` really starts
|
|
195
|
+
# a resource — since a data keyword is not a schema position and an
|
|
196
|
+
# `$anchor` written inside one names nothing in the document (Core
|
|
197
|
+
# Sections 8.2.2 and 4.3.1).
|
|
198
|
+
index_positions(index, [[target, resource, 0, index[:dialects][resource], false]])
|
|
199
|
+
target
|
|
200
|
+
end
|
|
201
|
+
|
|
202
|
+
# The one string-keyed copy of a subtree the document holds as data,
|
|
203
|
+
# memoized by the identity of the object it holds.
|
|
204
|
+
# @return [Hash]
|
|
205
|
+
def normalized_copy(target, index)
|
|
206
|
+
copies = (index[:normalized] ||= {}.compare_by_identity)
|
|
207
|
+
index[:budget] ||= { objects: 0, deadline: nil }
|
|
208
|
+
copies[target] ||= deep_stringify(target, 0, index[:budget])
|
|
209
|
+
end
|
|
210
|
+
|
|
211
|
+
# Every plain-name anchor in the document, per schema resource:
|
|
212
|
+
# `$anchor` (and `$dynamicAnchor`) in 2019-09 / 2020-12, `$id: "#name"`
|
|
213
|
+
# in draft-07. The walk is bounded like the preflight walk and follows
|
|
214
|
+
# the dialect of each resource (an embedded resource may declare its
|
|
215
|
+
# own `$schema`). Identifiers are taken only from schema positions the
|
|
216
|
+
# dialect walks (JSON Schema 2020-12 Core Sections 8.2.2 and 4.3.1: an
|
|
217
|
+
# identifier belongs to a schema object, and the value of a keyword
|
|
218
|
+
# the dialect does not define is not a schema): a definition bag the
|
|
219
|
+
# dialect does not define stays reachable through JSON pointers, and
|
|
220
|
+
# its objects are attributed to their lexical resource, but it is
|
|
221
|
+
# never a source of names — nor is anything beside a draft-07 `$ref`
|
|
222
|
+
# apart from its `definitions`.
|
|
223
|
+
# @return [Hash] :resources (schema object => its resource root, by
|
|
224
|
+
# identity), :anchors (resource root => name => subschema, first
|
|
225
|
+
# occurrence, by identity), :dialects (resource root => dialect) and
|
|
226
|
+
# :duplicates (names declared more than once within one resource)
|
|
227
|
+
def anchor_index(root, dialect)
|
|
228
|
+
index = { resources: {}.compare_by_identity, anchors: {}.compare_by_identity,
|
|
229
|
+
dialects: {}.compare_by_identity, bases: {}.compare_by_identity, by_base: {},
|
|
230
|
+
duplicates: [], duplicate_ids: [], visited: 0, truncated: false }
|
|
231
|
+
index[:dialects][root] = dialect
|
|
232
|
+
# The document's own base: the root's `$id` where it declares one,
|
|
233
|
+
# else the empty reference. Relative references resolve against each
|
|
234
|
+
# other either way, which is what a bundled document needs.
|
|
235
|
+
register_resource_base(index, root, '', declared: root.is_a?(Hash) && resource_root?(root, dialect))
|
|
236
|
+
index_positions(index, [[root, root, 0, dialect, true]])
|
|
237
|
+
index
|
|
238
|
+
end
|
|
239
|
+
|
|
240
|
+
# Record the base URI a schema resource establishes (RFC 3986 Section
|
|
241
|
+
# 5.1.1: an `$id` is resolved against the base in force where it is
|
|
242
|
+
# written), and the resource that base names. The first declaration of
|
|
243
|
+
# a base wins, as the first declaration of an anchor name does.
|
|
244
|
+
# @param declared [Boolean] whether the resource declares an `$id`
|
|
245
|
+
# @return [void]
|
|
246
|
+
def register_resource_base(index, resource, parent_base, declared: true)
|
|
247
|
+
id = declared && resource.is_a?(Hash) ? resource['$id'] : nil
|
|
248
|
+
base = (id.is_a?(String) ? merge_uri(parent_base, id) : parent_base) || parent_base
|
|
249
|
+
index[:bases][resource] = base
|
|
250
|
+
known = index[:by_base][base]
|
|
251
|
+
return index[:by_base][base] = resource if known.nil?
|
|
252
|
+
|
|
253
|
+
# Two resources answering to one URI: which one a reference lands on
|
|
254
|
+
# would depend on the order the walk met them in, so the document is
|
|
255
|
+
# unusable rather than resolved by luck (JSON Schema 2020-12 Core
|
|
256
|
+
# Section 9.1.2). Only a declared `$id` can collide; the base a
|
|
257
|
+
# resource merely inherits is its parent's, already registered.
|
|
258
|
+
index[:duplicate_ids] << base if id.is_a?(String) && !known.equal?(resource)
|
|
259
|
+
end
|
|
260
|
+
|
|
261
|
+
# Index every schema position reachable from the pending seeds, within
|
|
262
|
+
# the visit and depth bounds the whole index runs under (an adopted
|
|
263
|
+
# pointer target is seeded here too, so it shares them).
|
|
264
|
+
# @param index [Hash] the index being built
|
|
265
|
+
# @param pending [Array<Array>] seeds: schema, resource, depth,
|
|
266
|
+
# dialect, whether the position may declare names
|
|
267
|
+
# @return [void]
|
|
268
|
+
def index_positions(index, pending)
|
|
269
|
+
until pending.empty? || index[:visited] >= MAX_SUBSCHEMAS
|
|
270
|
+
schema, resource, depth, dialect, named = pending.shift
|
|
271
|
+
next unless schema.is_a?(Hash)
|
|
272
|
+
next if index[:resources].key?(schema)
|
|
273
|
+
# A schema below the depth bound is left unindexed just like one
|
|
274
|
+
# beyond the visit bound: a reference from it (or an `$id` there)
|
|
275
|
+
# would otherwise resolve under guesses.
|
|
276
|
+
(index[:truncated] = true) && next if depth > MAX_SCHEMA_DEPTH
|
|
277
|
+
|
|
278
|
+
index[:visited] += 1
|
|
279
|
+
resource, dialect, named = enter_resource(index, schema, resource, dialect, named)
|
|
280
|
+
index[:resources][schema] = resource
|
|
281
|
+
if named && !(dialect == DRAFT_07 && schema.key?('$ref'))
|
|
282
|
+
record_anchor_names(index, resource, schema,
|
|
283
|
+
dialect)
|
|
284
|
+
end
|
|
285
|
+
each_walked_position(schema, dialect) { |sub| pending << [sub, resource, depth + 1, dialect, named] }
|
|
286
|
+
each_foreign_definition(schema, dialect) { |sub| pending << [sub, resource, depth + 1, dialect, false] }
|
|
287
|
+
end
|
|
288
|
+
# Objects left unindexed at the bound would resolve and validate
|
|
289
|
+
# under guesses; the index says so and the schema is unusable.
|
|
290
|
+
index[:truncated] ||= pending.any? { |schema, *| schema.is_a?(Hash) && !index[:resources].key?(schema) }
|
|
291
|
+
end
|
|
292
|
+
|
|
293
|
+
# Enter the schema resource a position starts, if it starts one: a
|
|
294
|
+
# resource is a schema wherever it sits (one reached through a bag the
|
|
295
|
+
# dialect does not walk names its own anchors, though nothing outside
|
|
296
|
+
# it can see them), it may declare its own dialect, and its `$id`
|
|
297
|
+
# establishes the base its references resolve against.
|
|
298
|
+
# @return [Array(Hash, String, Boolean)] the resource in force, its
|
|
299
|
+
# dialect, and whether the position may declare names
|
|
300
|
+
def enter_resource(index, schema, resource, dialect, named)
|
|
301
|
+
return [resource, dialect, named] unless resource_root?(schema, dialect)
|
|
302
|
+
|
|
303
|
+
parent_base = index[:bases][resource] || ''
|
|
304
|
+
dialect = embedded_dialect(schema, dialect) || dialect
|
|
305
|
+
index[:dialects][schema] = dialect
|
|
306
|
+
register_resource_base(index, schema, parent_base)
|
|
307
|
+
[schema, dialect, true]
|
|
308
|
+
end
|
|
309
|
+
|
|
310
|
+
# Record the plain names a schema object declares for its resource; a
|
|
311
|
+
# name already bound to another object of the same resource is a
|
|
312
|
+
# duplicate (anchor names are unique within a resource).
|
|
313
|
+
# @return [void]
|
|
314
|
+
def record_anchor_names(index, resource, schema, dialect)
|
|
315
|
+
names = (index[:anchors][resource] ||= {})
|
|
316
|
+
anchor_names(schema, dialect).each do |name|
|
|
317
|
+
index[:duplicates] << name if names.key?(name) && !names[name].equal?(schema)
|
|
318
|
+
names[name] ||= schema
|
|
319
|
+
end
|
|
320
|
+
end
|
|
321
|
+
|
|
322
|
+
# Yield the schema positions the dialect walks under a schema object:
|
|
323
|
+
# under a draft-07 `$ref` only the `definitions` bag, else every
|
|
324
|
+
# subschema (the dialect's definition bag included).
|
|
325
|
+
# @return [void]
|
|
326
|
+
def each_walked_position(schema, dialect, &)
|
|
327
|
+
if dialect == DRAFT_07 && schema.key?('$ref')
|
|
328
|
+
each_definition(schema, dialect, &)
|
|
329
|
+
else
|
|
330
|
+
each_subschema(schema, dialect, &)
|
|
331
|
+
end
|
|
332
|
+
end
|
|
333
|
+
|
|
334
|
+
# The dialect the memoized anchor index recorded for a schema object's
|
|
335
|
+
# resource, or nil when the object was not indexed.
|
|
336
|
+
# @param schema [Hash]
|
|
337
|
+
# @param resolver [Hash, Context] holder of the memoized anchor index
|
|
338
|
+
# @return [String, nil]
|
|
339
|
+
def indexed_dialect(schema, resolver)
|
|
340
|
+
index = resolver[:anchors]
|
|
341
|
+
return nil unless index
|
|
342
|
+
|
|
343
|
+
resource = index[:resources][schema]
|
|
344
|
+
resource && index[:dialects][resource]
|
|
345
|
+
end
|
|
346
|
+
|
|
347
|
+
# Whether a schema object starts a new schema resource: its `$id` is a
|
|
348
|
+
# URI rather than a bare fragment (a draft-07 `$id: "#name"` is a
|
|
349
|
+
# plain-name identifier, not a base).
|
|
350
|
+
# @param schema [Hash]
|
|
351
|
+
# @return [Boolean]
|
|
352
|
+
def resource_start?(schema)
|
|
353
|
+
id = schema['$id']
|
|
354
|
+
id.is_a?(String) && !id.empty? && !id.start_with?('#')
|
|
355
|
+
end
|
|
356
|
+
|
|
357
|
+
# {#resource_start?} in the dialect in force: under draft-07 a `$ref`
|
|
358
|
+
# replaces its whole schema object, `$id` and `$schema` included, so
|
|
359
|
+
# nothing beside it starts a resource.
|
|
360
|
+
# @param schema [Hash]
|
|
361
|
+
# @param dialect [String, nil]
|
|
362
|
+
# @return [Boolean]
|
|
363
|
+
def resource_root?(schema, dialect)
|
|
364
|
+
return false if dialect == DRAFT_07 && schema.key?('$ref')
|
|
365
|
+
|
|
366
|
+
resource_start?(schema)
|
|
367
|
+
end
|
|
368
|
+
|
|
369
|
+
# How a dynamic reference binds (JSON Schema 2020-12 Core Section
|
|
370
|
+
# 8.2.3.2; 2019-09 Section 8.2.4.2.2). A reference whose initial
|
|
371
|
+
# target declares no matching dynamic anchor is the plain reference it
|
|
372
|
+
# resolves to (`:plain`). One that does re-binds to the declaration in
|
|
373
|
+
# the OUTERMOST resource of the dynamic scope — the resources the
|
|
374
|
+
# evaluation entered on its way here, which {SchemaValidator.entered_scope?}
|
|
375
|
+
# records as they are entered (`:bound`, with the target). What the
|
|
376
|
+
# document holds elsewhere decides nothing: a resource the instance
|
|
377
|
+
# never entered is not in the scope, so a duplicate anchor there is
|
|
378
|
+
# neither ambiguity nor a reason to leave the reference unevaluated.
|
|
379
|
+
# Outside a validation there is no scope to read, and every reference
|
|
380
|
+
# is the plain one it resolves to — which is what the preflight, whose
|
|
381
|
+
# job is that the reference resolves at all, needs.
|
|
382
|
+
# @param schema [Hash] the schema object holding the reference
|
|
383
|
+
# @param keyword [String] "$dynamicRef" or "$recursiveRef"
|
|
384
|
+
# @param root [Hash] the root schema
|
|
385
|
+
# @param dialect [String, nil] the canonical root dialect
|
|
386
|
+
# @param resolver [Hash, Context] holder of the memoized anchor index,
|
|
387
|
+
# and — during a validation — of the dynamic scope
|
|
388
|
+
# @return [Array] [:plain] or [:bound, target]
|
|
389
|
+
def dynamic_binding(schema, keyword, root, dialect, resolver)
|
|
390
|
+
ref = schema[keyword]
|
|
391
|
+
return [:plain] unless ref.is_a?(String) && !external_ref?(ref, root, dialect, resolver, from: schema)
|
|
392
|
+
|
|
393
|
+
target = resolve_reference(root, ref, dialect, resolver, from: schema)
|
|
394
|
+
return [:plain] if target.equal?(UNRESOLVED)
|
|
395
|
+
return [:plain] unless target.is_a?(Hash)
|
|
396
|
+
|
|
397
|
+
index = resolver[:anchors]
|
|
398
|
+
scope = resolver[:scope]
|
|
399
|
+
bound = if keyword == '$recursiveRef'
|
|
400
|
+
return [:plain] unless target['$recursiveAnchor'] == true
|
|
401
|
+
|
|
402
|
+
outermost_recursive_anchor(index, scope, dialect)
|
|
403
|
+
else
|
|
404
|
+
fragment = ref.include?('#') ? decoded_fragment(ref[ref.index('#')..]) : nil
|
|
405
|
+
return [:plain] if fragment.nil? || fragment.empty? || fragment.start_with?('/')
|
|
406
|
+
return [:plain] unless target['$dynamicAnchor'] == fragment
|
|
407
|
+
|
|
408
|
+
outermost_dynamic_anchor(index, scope, fragment)
|
|
409
|
+
end
|
|
410
|
+
bound ? [:bound, bound] : [:plain]
|
|
411
|
+
end
|
|
412
|
+
|
|
413
|
+
# The schema declaring `$dynamicAnchor: name` in the outermost resource
|
|
414
|
+
# of the dynamic scope that declares it — the scope being the resources
|
|
415
|
+
# the evaluation actually entered, outermost first. A resource the
|
|
416
|
+
# instance never entered declares nothing for this reference, however
|
|
417
|
+
# many of them the document holds; and where no entered resource
|
|
418
|
+
# declares the name, the caller keeps the target the reference resolved
|
|
419
|
+
# to on its own, which is what the specification's "otherwise behave as
|
|
420
|
+
# $ref" says.
|
|
421
|
+
# @param index [Hash] the anchor index
|
|
422
|
+
# @param scope [Array<Hash>, nil] the dynamic scope, outermost first
|
|
423
|
+
# @param name [String] the anchor name
|
|
424
|
+
# @return [Hash, nil]
|
|
425
|
+
def outermost_dynamic_anchor(index, scope, name)
|
|
426
|
+
Array(scope).each do |resource|
|
|
427
|
+
declaring = index[:anchors][resource]&.[](name)
|
|
428
|
+
return declaring if declaring.is_a?(Hash) && declaring['$dynamicAnchor'] == name
|
|
429
|
+
end
|
|
430
|
+
nil
|
|
431
|
+
end
|
|
432
|
+
|
|
433
|
+
# The outermost resource of the dynamic scope whose `$recursiveAnchor`
|
|
434
|
+
# is true (2019-09 Core Section 8.2.4.2.2); the target of a
|
|
435
|
+
# `$recursiveRef` is the resource root itself.
|
|
436
|
+
# @return [Hash, nil]
|
|
437
|
+
def outermost_recursive_anchor(index, scope, dialect)
|
|
438
|
+
Array(scope).find do |resource|
|
|
439
|
+
resource.is_a?(Hash) && resource['$recursiveAnchor'] == true &&
|
|
440
|
+
(index[:dialects][resource] || dialect) != DRAFT_07
|
|
441
|
+
end
|
|
442
|
+
end
|
|
443
|
+
|
|
444
|
+
# @return [Array<String>] the plain names a schema object declares
|
|
445
|
+
def anchor_names(schema, dialect)
|
|
446
|
+
names = if dialect == DRAFT_07
|
|
447
|
+
# draft-07 Core Section 8.2.3: only an $id that is exactly a
|
|
448
|
+
# fragment is a plain-name identifier. An $id is a URI
|
|
449
|
+
# reference, so its fragment is percent-decoded (RFC 3986
|
|
450
|
+
# Section 2.1) exactly as a $ref's is: `$id: "#foo%2Dbar"`
|
|
451
|
+
# declares the name "foo-bar", which is what
|
|
452
|
+
# `$ref: "#foo%2Dbar"` — decoded the same way — looks for.
|
|
453
|
+
id = schema['$id']
|
|
454
|
+
[id.is_a?(String) && id.start_with?('#') ? decoded_fragment(id) : nil]
|
|
455
|
+
else
|
|
456
|
+
%w[$anchor $dynamicAnchor].map { |k| schema[k] if keyword_known?(k, dialect) }
|
|
457
|
+
end
|
|
458
|
+
names.select { |name| anchor_name?(name, dialect) }
|
|
459
|
+
end
|
|
460
|
+
|
|
461
|
+
# The lexical nesting depth of every schema object reachable from the
|
|
462
|
+
# root (subschema positions, definition bags of any dialect), so a
|
|
463
|
+
# referenced target is bounded by where it is written, not by where it
|
|
464
|
+
# is referenced from: neither member order nor reference fan-out can
|
|
465
|
+
# change the verdict.
|
|
466
|
+
# @return [Hash{Hash => Integer}] identity-keyed
|
|
467
|
+
def lexical_depths(root, dialect)
|
|
468
|
+
depths = {}.compare_by_identity
|
|
469
|
+
pending = [[root, 0, dialect]]
|
|
470
|
+
while (schema, depth, current = pending.shift)
|
|
471
|
+
next unless schema.is_a?(Hash) && !depths.key?(schema)
|
|
472
|
+
|
|
473
|
+
depths[schema] = depth
|
|
474
|
+
break if depths.size > MAX_SUBSCHEMAS * 2
|
|
475
|
+
|
|
476
|
+
# An embedded resource's positions follow its own dialect.
|
|
477
|
+
current = embedded_dialect(schema, current) || current
|
|
478
|
+
each_subschema(schema, current) { |sub| pending << [sub, depth + 1, current] }
|
|
479
|
+
each_foreign_definition(schema, current) { |sub| pending << [sub, depth + 1, current] }
|
|
480
|
+
end
|
|
481
|
+
depths
|
|
482
|
+
end
|
|
483
|
+
|
|
484
|
+
# The lexical depth of the value a pointer reference reaches, counted
|
|
485
|
+
# in schema steps along the (percent-decoded) pointer from its resource
|
|
486
|
+
# root: a keyword holding one subschema is one step, a map or array of
|
|
487
|
+
# subschemas is one step per member (`#/properties/b` and `#/allOf/0`
|
|
488
|
+
# are both one below the enclosing schema), and every token under a
|
|
489
|
+
# data or unknown keyword is a step, so a document hidden inside
|
|
490
|
+
# `default`, `enum`, `const`, `examples` or a vendor keyword obeys the
|
|
491
|
+
# same bound as one written in a schema position. Keywords are
|
|
492
|
+
# classified by the dialect in force at each node (an embedded
|
|
493
|
+
# resource's own `$schema` takes over when the pointer enters it). A
|
|
494
|
+
# schema object whose depth the lexical index knows resets the count
|
|
495
|
+
# to that depth — unless the pointer already passed through an opaque
|
|
496
|
+
# keyword, after which every token counts (the index placed such an
|
|
497
|
+
# object under another dialect's grammar).
|
|
498
|
+
# @return [Integer, nil] nil when the pointer cannot be followed
|
|
499
|
+
def referenced_position_depth(ref, root, dialect, counter, from)
|
|
500
|
+
pointer_position(ref, root, dialect, counter, from)&.first
|
|
501
|
+
end
|
|
502
|
+
|
|
503
|
+
# @return [Array(Integer, Boolean), nil] the depth and whether the
|
|
504
|
+
# pointer crossed an opaque keyword; nil when it cannot be followed
|
|
505
|
+
def pointer_position(ref, root, dialect, counter, from)
|
|
506
|
+
index = (counter[:anchors] ||= anchor_index(root, dialect))
|
|
507
|
+
node, ref = pointer_origin(index, root, ref, from)
|
|
508
|
+
return nil unless node
|
|
509
|
+
|
|
510
|
+
tokens = pointer_tokens(ref)
|
|
511
|
+
return nil unless tokens
|
|
512
|
+
|
|
513
|
+
depths = counter[:depths] || {}
|
|
514
|
+
walk = { index: index, depths: depths, dialect: index[:dialects][node] || dialect,
|
|
515
|
+
depth: depths[node] || 0, mode: :schema, opaque: false, visited: true }
|
|
516
|
+
tokens.each do |token|
|
|
517
|
+
child = pointer_child(node, token)
|
|
518
|
+
return nil if child.equal?(UNRESOLVED)
|
|
519
|
+
|
|
520
|
+
pointer_step(walk, node, token)
|
|
521
|
+
node = child
|
|
522
|
+
end
|
|
523
|
+
[walk[:depth], walk[:opaque], walk[:visited]]
|
|
524
|
+
end
|
|
525
|
+
|
|
526
|
+
# Advance one pointer token: in schema mode the node's own dialect
|
|
527
|
+
# and known depth apply and the keyword decides how the next token
|
|
528
|
+
# counts; inside a map or array keyword the member is the step;
|
|
529
|
+
# under an opaque keyword every token is a step.
|
|
530
|
+
# @return [void]
|
|
531
|
+
def pointer_step(walk, node, token)
|
|
532
|
+
unless walk[:mode] == :schema
|
|
533
|
+
walk[:depth] += 1
|
|
534
|
+
walk[:mode] = :schema if %i[map array].include?(walk[:mode])
|
|
535
|
+
return
|
|
536
|
+
end
|
|
537
|
+
if node.is_a?(Hash)
|
|
538
|
+
walk[:depth] = walk[:depths][node] if !walk[:opaque] && walk[:depths].key?(node)
|
|
539
|
+
walk[:dialect] = walk[:index][:dialects][node] || embedded_dialect(node, walk[:dialect]) || walk[:dialect]
|
|
540
|
+
end
|
|
541
|
+
walk[:mode] = pointer_step_mode(node, token, walk[:dialect])
|
|
542
|
+
walk[:opaque] ||= walk[:mode] == :opaque
|
|
543
|
+
# The preflight walk does not descend into an opaque keyword, nor
|
|
544
|
+
# into anything beside a draft-07 $ref but its definitions.
|
|
545
|
+
walk[:visited] &&= walk[:mode] != :opaque &&
|
|
546
|
+
!(walk[:dialect] == DRAFT_07 && node.is_a?(Hash) && node.key?('$ref') &&
|
|
547
|
+
token != 'definitions')
|
|
548
|
+
walk[:depth] += 1 unless %i[map array].include?(walk[:mode])
|
|
549
|
+
end
|
|
550
|
+
|
|
551
|
+
# The resource a reference's pointer is read inside, and the bare
|
|
552
|
+
# fragment that applies there. A reference written as an absolute URI
|
|
553
|
+
# into the bundled document (`urn:root#/x/y`) addresses the very
|
|
554
|
+
# position its bare spelling (`#/x/y`) does, so the pointer is followed
|
|
555
|
+
# — and the target accounted for — the same way whichever the peer
|
|
556
|
+
# wrote (JSON Schema 2020-12 Core Section 9.3.1).
|
|
557
|
+
# @return [Array(Hash, String), Array(nil, nil)]
|
|
558
|
+
def pointer_origin(index, root, ref, from)
|
|
559
|
+
resource = (from && index[:resources][from]) || root
|
|
560
|
+
return [resource, ref] if ref.start_with?('#')
|
|
561
|
+
|
|
562
|
+
retarget_reference(index, resource, ref)
|
|
563
|
+
end
|
|
564
|
+
|
|
565
|
+
# The decoded RFC 6901 tokens of a fragment pointer.
|
|
566
|
+
# @return [Array<String>, nil] nil unless the reference is a pointer
|
|
567
|
+
def pointer_tokens(ref)
|
|
568
|
+
fragment = decoded_fragment(ref)
|
|
569
|
+
return nil unless fragment&.start_with?('/')
|
|
570
|
+
|
|
571
|
+
fragment.split('/', -1).drop(1).map { |token| token.gsub('~1', '/').gsub('~0', '~') }
|
|
572
|
+
end
|
|
573
|
+
|
|
574
|
+
# @return [Object] the member a pointer token selects, or UNRESOLVED
|
|
575
|
+
def pointer_child(node, token)
|
|
576
|
+
case node
|
|
577
|
+
when Hash then node.key?(token) ? node[token] : UNRESOLVED
|
|
578
|
+
when Array then token.match?(/\A(0|[1-9]\d*)\z/) && token.to_i < node.length ? node[token.to_i] : UNRESOLVED
|
|
579
|
+
else UNRESOLVED
|
|
580
|
+
end
|
|
581
|
+
end
|
|
582
|
+
|
|
583
|
+
# How a keyword of a schema object holds what its pointer token reaches.
|
|
584
|
+
# @return [Symbol] :schema (one subschema), :map, :array, or :opaque
|
|
585
|
+
# (data or unknown keyword: every token below is a step)
|
|
586
|
+
# A keyword the dialect in force does not define (`prefixItems` under
|
|
587
|
+
# draft-07, `additionalItems` under 2020-12) is opaque data there.
|
|
588
|
+
def pointer_step_mode(node, token, dialect)
|
|
589
|
+
return :opaque unless node.is_a?(Hash) && keyword_known?(token, dialect)
|
|
590
|
+
return :map if SUBSCHEMA_MAP_KEYWORDS.include?(token)
|
|
591
|
+
return :array if SUBSCHEMA_ARRAY_KEYWORDS.include?(token) || (token == 'items' && node[token].is_a?(Array))
|
|
592
|
+
return :schema if SUBSCHEMA_KEYWORDS.include?(token)
|
|
593
|
+
|
|
594
|
+
:opaque
|
|
595
|
+
end
|
|
596
|
+
|
|
597
|
+
# Yield the definitions held in the bag the dialect does not define
|
|
598
|
+
# (`$defs` under draft-07, which predates it): unknown to the dialect,
|
|
599
|
+
# but pointer-addressable all the same.
|
|
600
|
+
# @return [void]
|
|
601
|
+
def each_foreign_definition(schema, dialect, &block)
|
|
602
|
+
%w[$defs definitions].each do |keyword|
|
|
603
|
+
next if keyword_known?(keyword, dialect)
|
|
604
|
+
|
|
605
|
+
schema[keyword].each_value(&block) if schema[keyword].is_a?(Hash)
|
|
606
|
+
end
|
|
607
|
+
end
|
|
608
|
+
end
|
|
609
|
+
end
|
|
610
|
+
end
|