ruby-mcp-client 2.1.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/OAUTH.md +555 -0
- data/README.md +825 -48
- data/lib/mcp_client/audio_content.rb +1 -1
- data/lib/mcp_client/auth/browser_oauth.rb +131 -21
- data/lib/mcp_client/auth/oauth_provider/challenge_handling.rb +532 -0
- data/lib/mcp_client/auth/oauth_provider/client_authentication.rb +121 -0
- data/lib/mcp_client/auth/oauth_provider/pending_requests.rb +51 -0
- data/lib/mcp_client/auth/oauth_provider/registration_store.rb +486 -0
- data/lib/mcp_client/auth/oauth_provider/response_validation.rb +441 -0
- data/lib/mcp_client/auth/oauth_provider/scope_selection.rb +134 -0
- data/lib/mcp_client/auth/oauth_provider/token_store.rb +419 -0
- data/lib/mcp_client/auth/oauth_provider.rb +1354 -386
- data/lib/mcp_client/auth/peer_text.rb +174 -0
- data/lib/mcp_client/auth.rb +298 -32
- data/lib/mcp_client/cached_result.rb +145 -0
- data/lib/mcp_client/called_tool_definition.rb +138 -0
- data/lib/mcp_client/client/cache_slices.rb +195 -0
- data/lib/mcp_client/client/list_aggregation.rb +243 -0
- data/lib/mcp_client/client/notification_routing.rb +155 -0
- data/lib/mcp_client/client/sampling_validation.rb +200 -0
- data/lib/mcp_client/client/task_api.rb +531 -0
- data/lib/mcp_client/client/task_lifetimes.rb +269 -0
- data/lib/mcp_client/client/task_registry.rb +254 -0
- data/lib/mcp_client/client/task_shape.rb +102 -0
- data/lib/mcp_client/client/task_support.rb +1166 -0
- data/lib/mcp_client/client/task_updates.rb +457 -0
- data/lib/mcp_client/client/task_wait_boundaries.rb +198 -0
- data/lib/mcp_client/client/task_workers.rb +63 -0
- data/lib/mcp_client/client.rb +796 -518
- data/lib/mcp_client/deep_copy.rb +49 -0
- data/lib/mcp_client/deprecation_notices.rb +94 -0
- data/lib/mcp_client/deprecations.rb +419 -0
- data/lib/mcp_client/errors.rb +474 -7
- data/lib/mcp_client/header_params.rb +320 -0
- data/lib/mcp_client/http_transport_base/bounded_inflate.rb +41 -0
- data/lib/mcp_client/http_transport_base/cache_support.rb +694 -0
- data/lib/mcp_client/http_transport_base/era_detection.rb +134 -0
- data/lib/mcp_client/http_transport_base/listen_stream.rb +763 -0
- data/lib/mcp_client/http_transport_base/param_headers.rb +35 -0
- data/lib/mcp_client/http_transport_base/request_recovery.rb +156 -0
- data/lib/mcp_client/http_transport_base/session_recovery.rb +113 -0
- data/lib/mcp_client/http_transport_base/sse_event_scanner.rb +145 -0
- data/lib/mcp_client/http_transport_base/stream_capture.rb +160 -0
- data/lib/mcp_client/http_transport_base/stream_recovery.rb +318 -0
- data/lib/mcp_client/http_transport_base/tool_listing.rb +277 -0
- data/lib/mcp_client/http_transport_base.rb +666 -120
- data/lib/mcp_client/input_round_trips.rb +128 -0
- data/lib/mcp_client/json_rpc_common/envelopes.rb +32 -0
- data/lib/mcp_client/json_rpc_common/error_bodies.rb +105 -0
- data/lib/mcp_client/json_rpc_common/input_waits.rb +167 -0
- data/lib/mcp_client/json_rpc_common.rb +900 -13
- data/lib/mcp_client/oauth_client.rb +14 -5
- data/lib/mcp_client/prompt.rb +4 -0
- data/lib/mcp_client/request_authorization.rb +128 -0
- data/lib/mcp_client/request_meta_scope.rb +77 -0
- data/lib/mcp_client/request_metadata.rb +287 -0
- data/lib/mcp_client/resource.rb +4 -0
- data/lib/mcp_client/resource_content.rb +20 -0
- data/lib/mcp_client/resource_template.rb +4 -0
- data/lib/mcp_client/result_caching.rb +999 -0
- data/lib/mcp_client/result_completeness.rb +34 -0
- data/lib/mcp_client/root.rb +6 -0
- data/lib/mcp_client/round_trip_marker.rb +28 -0
- data/lib/mcp_client/schema_validator/annotations.rb +82 -0
- data/lib/mcp_client/schema_validator/composition.rb +86 -0
- data/lib/mcp_client/schema_validator/dialects.rb +66 -0
- data/lib/mcp_client/schema_validator/ecma_patterns.rb +567 -0
- data/lib/mcp_client/schema_validator/evaluation.rb +517 -0
- data/lib/mcp_client/schema_validator/input_requirements.rb +84 -0
- data/lib/mcp_client/schema_validator/instances.rb +449 -0
- data/lib/mcp_client/schema_validator/keyword_scan.rb +121 -0
- data/lib/mcp_client/schema_validator/normalization.rb +104 -0
- data/lib/mcp_client/schema_validator/references.rb +610 -0
- data/lib/mcp_client/schema_validator/scalars.rb +126 -0
- data/lib/mcp_client/schema_validator/shapes.rb +319 -0
- data/lib/mcp_client/schema_validator/uri_references.rb +153 -0
- data/lib/mcp_client/schema_validator.rb +882 -208
- data/lib/mcp_client/server_base.rb +233 -5
- data/lib/mcp_client/server_factory.rb +9 -3
- data/lib/mcp_client/server_http/json_rpc_transport.rb +219 -4
- data/lib/mcp_client/server_http.rb +307 -90
- data/lib/mcp_client/server_sse/json_rpc_transport.rb +113 -25
- data/lib/mcp_client/server_sse/sse_parser.rb +39 -6
- data/lib/mcp_client/server_sse.rb +227 -62
- data/lib/mcp_client/server_stdio/child_session.rb +98 -0
- data/lib/mcp_client/server_stdio/json_rpc_transport.rb +1003 -28
- data/lib/mcp_client/server_stdio.rb +772 -183
- data/lib/mcp_client/server_streamable_http/json_rpc_transport.rb +189 -25
- data/lib/mcp_client/server_streamable_http.rb +302 -115
- data/lib/mcp_client/session_pin.rb +119 -0
- data/lib/mcp_client/subscription/notification_dispatcher.rb +354 -0
- data/lib/mcp_client/subscription.rb +852 -0
- data/lib/mcp_client/subscription_support.rb +715 -0
- data/lib/mcp_client/task.rb +286 -14
- data/lib/mcp_client/tool.rb +31 -3
- data/lib/mcp_client/version.rb +21 -6
- data/lib/mcp_client.rb +108 -19
- metadata +68 -2
|
@@ -0,0 +1,567 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module MCPClient
|
|
4
|
+
module SchemaValidator
|
|
5
|
+
# The rewrite of an ECMA-262 pattern as the Ruby expression that means
|
|
6
|
+
# the same thing. JSON Schema 2020-12 Core Section 4.3 requires patterns
|
|
7
|
+
# to be interpreted as ECMA-262 regular expressions, and Ruby's differ
|
|
8
|
+
# from them in both directions: what Ruby accepts that ECMA-262 rejects
|
|
9
|
+
# makes the validator accept a value the schema refuses, and the
|
|
10
|
+
# converse rejects a conforming one (and, through `not` or
|
|
11
|
+
# `additionalProperties: false`, flips both). Extended into
|
|
12
|
+
# SchemaValidator, so the methods are its own.
|
|
13
|
+
#
|
|
14
|
+
# A pattern comes from the remote peer and is as long as the peer made
|
|
15
|
+
# it, so the translation is one linear pass over its characters that
|
|
16
|
+
# consults the validation-wide deadline as it goes, and a pattern past
|
|
17
|
+
# MAX_PATTERN_LENGTH is refused before it is read at all.
|
|
18
|
+
module EcmaPatterns
|
|
19
|
+
# A pattern that is no ECMA-262 expression: syntax ECMA-262 does not
|
|
20
|
+
# define (Ruby's inline flags, possessive quantifiers, atomic groups,
|
|
21
|
+
# comments) or an expression neither dialect accepts. A RegexpError,
|
|
22
|
+
# so every caller that rescues an unreadable expression sees it.
|
|
23
|
+
class SyntaxError < RegexpError; end
|
|
24
|
+
|
|
25
|
+
# A pattern that IS an ECMA-262 expression but whose meaning Ruby's
|
|
26
|
+
# engine cannot be made to reproduce, so translating it would answer
|
|
27
|
+
# some instances wrongly. The two engines differ in two ways no
|
|
28
|
+
# rewriting bridges:
|
|
29
|
+
#
|
|
30
|
+
# - ECMA-262 clears the captures inside a quantified group at the start
|
|
31
|
+
# of every iteration (a back-reference to a group the last iteration
|
|
32
|
+
# did not enter matches the empty string); Ruby keeps whatever the
|
|
33
|
+
# last iteration that entered it captured. `^(a|(b))*\2$` accepts
|
|
34
|
+
# "aba" and refuses "abab" there, and exactly the opposite here.
|
|
35
|
+
# - ECMA-262 lookbehind is variable-length (ES2018); Ruby's is not.
|
|
36
|
+
#
|
|
37
|
+
# Refusing the schema is the only honest answer left: a verdict from
|
|
38
|
+
# the other engine's rules would accept structured content the schema
|
|
39
|
+
# forbids as readily as it would refuse conforming content.
|
|
40
|
+
class Untranslatable < RegexpError; end
|
|
41
|
+
|
|
42
|
+
# What each ECMA-262 anchor means in Ruby: the ends of the subject,
|
|
43
|
+
# never a line boundary.
|
|
44
|
+
ECMA_ANCHORS = { '^' => '\\A', '$' => '\\z' }.freeze
|
|
45
|
+
|
|
46
|
+
# ECMA-262 `.` matches every character except the four line
|
|
47
|
+
# terminators; Ruby's excludes only "\n".
|
|
48
|
+
ECMA_DOT = '[^\\n\\r\\u2028\\u2029]'
|
|
49
|
+
|
|
50
|
+
# The members of ECMA-262's `\s` (WhiteSpace plus LineTerminator):
|
|
51
|
+
# Ruby's is `[ \t\r\n\f\v]` and knows none of the Unicode spaces, so a
|
|
52
|
+
# non-breaking space failed a pattern ECMAScript satisfies.
|
|
53
|
+
ECMA_SPACE_MEMBERS = '\\t\\n\\v\\f\\r \\u00a0\\u1680\\u2000-\\u200a\\u2028\\u2029\\u202f\\u205f\\u3000\\ufeff'
|
|
54
|
+
|
|
55
|
+
# A character class matching nothing, which is what ECMA-262 makes of
|
|
56
|
+
# `[]` — Ruby cannot compile that at all, so the pattern used to be
|
|
57
|
+
# dropped and every string satisfied it.
|
|
58
|
+
ECMA_EMPTY_CLASS = '[^\\s\\S]'
|
|
59
|
+
|
|
60
|
+
# Its complement: ECMA-262 `[^]` matches any character, line
|
|
61
|
+
# terminators included.
|
|
62
|
+
ECMA_ANY_CLASS = '[\\s\\S]'
|
|
63
|
+
|
|
64
|
+
# The escapes ECMA-262 defines, which Ruby reads the same way: the
|
|
65
|
+
# class escapes, the control escapes, a hex or control-letter escape
|
|
66
|
+
# and a Unicode property. `\s` / `\S` are defined by both but over
|
|
67
|
+
# different sets, `\b` / `\B` over different word characters, the
|
|
68
|
+
# digits are back-references or legacy octal escapes, `\u` may spell
|
|
69
|
+
# a surrogate pair and `\k` a named back-reference, so those are
|
|
70
|
+
# rewritten rather than kept.
|
|
71
|
+
ECMA_KEPT_ESCAPES = 'dDwWfnrtvxcpP'
|
|
72
|
+
|
|
73
|
+
# ECMA-262's word characters: `\w` is [A-Za-z0-9_] there, and its
|
|
74
|
+
# word-boundary assertions are defined over exactly those, while Ruby's
|
|
75
|
+
# `\b` knows every Unicode letter — so "é" has a boundary in Ruby and
|
|
76
|
+
# none in ECMA-262.
|
|
77
|
+
ECMA_WORD = '[A-Za-z0-9_]'
|
|
78
|
+
|
|
79
|
+
# `\b`: a word character on exactly one side.
|
|
80
|
+
ECMA_WORD_BOUNDARY = "(?:(?<=#{ECMA_WORD})(?!#{ECMA_WORD})|(?<!#{ECMA_WORD})(?=#{ECMA_WORD}))".freeze
|
|
81
|
+
|
|
82
|
+
# `\B`: word characters on both sides, or on neither.
|
|
83
|
+
ECMA_NON_BOUNDARY = "(?:(?<=#{ECMA_WORD})(?=#{ECMA_WORD})|(?<!#{ECMA_WORD})(?!#{ECMA_WORD}))".freeze
|
|
84
|
+
|
|
85
|
+
# The characters that may follow `(?` in ECMA-262: a non-capturing
|
|
86
|
+
# group, a lookahead, and (after `<`) a lookbehind or a named group.
|
|
87
|
+
# Anything else Ruby reads as an inline option, an atomic group, a
|
|
88
|
+
# comment or its own named-group syntax, none of which ECMA-262 has.
|
|
89
|
+
ECMA_GROUP_OPENERS = ':=!<'
|
|
90
|
+
|
|
91
|
+
# How many characters the translation reads between two looks at the
|
|
92
|
+
# deadline.
|
|
93
|
+
TRANSLATION_CHECK_INTERVAL = 256
|
|
94
|
+
|
|
95
|
+
# The Ruby expression an ECMA-262 pattern means.
|
|
96
|
+
# @param pattern [String] the peer's pattern
|
|
97
|
+
# @param timeout [Float] seconds the match may take
|
|
98
|
+
# @param deadline [Float, nil] monotonic deadline the translation runs under
|
|
99
|
+
# @return [Regexp]
|
|
100
|
+
# @raise [RegexpError] when the pattern is not a usable expression
|
|
101
|
+
# @raise [Untranslatable] when it is one Ruby cannot reproduce
|
|
102
|
+
# @raise [Aborted] when the deadline passes during the translation
|
|
103
|
+
def ecma_regexp(pattern, timeout, deadline = nil)
|
|
104
|
+
Regexp.new(ecma_source(pattern, deadline), timeout: timeout)
|
|
105
|
+
rescue RegexpError => e
|
|
106
|
+
# ECMA-262 lookbehind has been variable-length since ES2018 and
|
|
107
|
+
# Ruby's never has been: the pattern is a good expression this
|
|
108
|
+
# engine cannot be given, not a bad one.
|
|
109
|
+
raise Untranslatable, "variable-length lookbehind cannot be evaluated faithfully (#{e.message})" if
|
|
110
|
+
e.message.include?('look-behind')
|
|
111
|
+
|
|
112
|
+
raise
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
# Rewrite an ECMA-262 pattern as Ruby regexp source. Everything
|
|
116
|
+
# ECMA-262 defines is kept; what only Ruby defines is either read the
|
|
117
|
+
# way ECMA-262 reads it — an escape ECMA-262 does not define is an
|
|
118
|
+
# identity escape there (Annex B.1.2), so `\A` is a literal "A" and not
|
|
119
|
+
# the start of the subject — or refused where ECMA-262 refuses it.
|
|
120
|
+
# @param pattern [String] the peer's pattern
|
|
121
|
+
# @param deadline [Float, nil] monotonic deadline the translation runs under
|
|
122
|
+
# @return [String] Ruby regexp source
|
|
123
|
+
# @raise [SyntaxError] for syntax ECMA-262 does not define
|
|
124
|
+
# @raise [Aborted] when the deadline passes during the translation
|
|
125
|
+
def ecma_source(pattern, deadline = nil)
|
|
126
|
+
raise SyntaxError, "pattern is longer than #{MAX_PATTERN_LENGTH} characters" if
|
|
127
|
+
pattern.length > MAX_PATTERN_LENGTH
|
|
128
|
+
|
|
129
|
+
chars = pattern.chars
|
|
130
|
+
scan = { chars: chars, index: 0, out: +'', deadline: deadline, read: 0, last: :none, opened: 0 }
|
|
131
|
+
scan[:order] = count_capture_groups(chars)
|
|
132
|
+
scan[:groups] = scan[:order].length
|
|
133
|
+
scan[:names] = scan[:order].compact
|
|
134
|
+
scan[:repeated] = repeated_capture_groups(chars)
|
|
135
|
+
scan[:generated] = generated_name_prefix(scan[:names])
|
|
136
|
+
while scan[:index] < chars.length
|
|
137
|
+
note_translation_progress(scan)
|
|
138
|
+
chars[scan[:index]] == '[' ? copy_character_class(scan) : copy_ecma_token(scan)
|
|
139
|
+
end
|
|
140
|
+
scan[:out]
|
|
141
|
+
end
|
|
142
|
+
|
|
143
|
+
# Consult the deadline every TRANSLATION_CHECK_INTERVAL characters.
|
|
144
|
+
# @raise [Aborted]
|
|
145
|
+
def note_translation_progress(scan)
|
|
146
|
+
scan[:read] += 1
|
|
147
|
+
return unless (scan[:read] % TRANSLATION_CHECK_INTERVAL).zero?
|
|
148
|
+
|
|
149
|
+
raise Aborted, 'validation time budget exhausted while translating a pattern' if
|
|
150
|
+
budget_exhausted?(scan[:deadline])
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
# The capturing groups a pattern declares, in order and in one pass:
|
|
154
|
+
# what a numeric escape refers to depends on how many there are (Annex
|
|
155
|
+
# B.1.4: a number past the count is a legacy octal escape) and on which
|
|
156
|
+
# one it names — ECMA-262 numbers named and unnamed groups alike,
|
|
157
|
+
# left to right, while Ruby stops capturing unnamed groups once a
|
|
158
|
+
# named one exists, so the numbering is kept here and every group is
|
|
159
|
+
# written as a named one there ({#copy_group_opener}).
|
|
160
|
+
# @return [Array<String, nil>] each group's name, nil for an unnamed one
|
|
161
|
+
def count_capture_groups(chars)
|
|
162
|
+
order = []
|
|
163
|
+
index = 0
|
|
164
|
+
in_class = false
|
|
165
|
+
while index < chars.length
|
|
166
|
+
char = chars[index]
|
|
167
|
+
if char == '\\'
|
|
168
|
+
index += 2
|
|
169
|
+
next
|
|
170
|
+
end
|
|
171
|
+
in_class = true if char == '['
|
|
172
|
+
in_class = false if char == ']' && in_class
|
|
173
|
+
if char == '(' && !in_class
|
|
174
|
+
order << nil if chars[index + 1] != '?'
|
|
175
|
+
name = group_name_at(chars, index + 2)
|
|
176
|
+
order << name if name
|
|
177
|
+
end
|
|
178
|
+
index += 1
|
|
179
|
+
end
|
|
180
|
+
order
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
# The Ruby name of a capturing group, by its ECMA-262 number: its own
|
|
184
|
+
# where it has one, a generated one otherwise.
|
|
185
|
+
# @param number [Integer] the group's number, from 1
|
|
186
|
+
# @return [String]
|
|
187
|
+
def group_name_for(scan, number)
|
|
188
|
+
scan[:order][number - 1] || "#{scan[:generated]}#{number}"
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
# A prefix for the generated names that none of the pattern's own
|
|
192
|
+
# names begins with, so a written `(?<__mcp_g1>` can never be the
|
|
193
|
+
# group a numeric back-reference is rewritten to name.
|
|
194
|
+
# @param names [Array<String>] the names the pattern wrote
|
|
195
|
+
# @return [String]
|
|
196
|
+
def generated_name_prefix(names)
|
|
197
|
+
prefix = +'__mcp_g'
|
|
198
|
+
prefix << '_' while names.any? { |name| name.start_with?(prefix) }
|
|
199
|
+
prefix
|
|
200
|
+
end
|
|
201
|
+
|
|
202
|
+
# The name of a `(?<name>` group opening at the index of its `<`.
|
|
203
|
+
# @return [String, nil]
|
|
204
|
+
def group_name_at(chars, index)
|
|
205
|
+
return nil unless chars[index] == '<' && !'=!'.include?(chars[index + 1].to_s)
|
|
206
|
+
|
|
207
|
+
close = index + 1
|
|
208
|
+
close += 1 while close < chars.length && chars[close] != '>'
|
|
209
|
+
chars[(index + 1)...close].join if close < chars.length
|
|
210
|
+
end
|
|
211
|
+
|
|
212
|
+
# Copy one token from outside a character class.
|
|
213
|
+
# @return [void]
|
|
214
|
+
def copy_ecma_token(scan)
|
|
215
|
+
char = scan[:chars][scan[:index]]
|
|
216
|
+
case char
|
|
217
|
+
when '\\' then copy_escape(scan, in_class: false)
|
|
218
|
+
when '(' then copy_group_opener(scan)
|
|
219
|
+
when '*', '+', '?' then copy_quantifier(scan, char)
|
|
220
|
+
when '{' then copy_brace(scan)
|
|
221
|
+
else copy_plain_token(scan, char)
|
|
222
|
+
end
|
|
223
|
+
end
|
|
224
|
+
|
|
225
|
+
# A character that is neither an escape, a group opener nor a
|
|
226
|
+
# quantifier.
|
|
227
|
+
# @return [void]
|
|
228
|
+
def copy_plain_token(scan, char)
|
|
229
|
+
scan[:index] += 1
|
|
230
|
+
case char
|
|
231
|
+
when '^', '$'
|
|
232
|
+
emit(scan, ECMA_ANCHORS[char], :none)
|
|
233
|
+
when '.' then emit(scan, ECMA_DOT, :atom)
|
|
234
|
+
when '|' then emit(scan, char, :none)
|
|
235
|
+
# A `]` or `}` that closes nothing is a literal in ECMA-262 (Annex
|
|
236
|
+
# B.1.4); Ruby reads a bare `}` the same way but is spared the guess.
|
|
237
|
+
when ']', '}' then emit(scan, "\\#{char}", :atom)
|
|
238
|
+
else emit(scan, char, :atom)
|
|
239
|
+
end
|
|
240
|
+
end
|
|
241
|
+
|
|
242
|
+
# Append translated source and record what kind of token it was.
|
|
243
|
+
# @return [void]
|
|
244
|
+
def emit(scan, source, kind)
|
|
245
|
+
scan[:out] << source
|
|
246
|
+
scan[:last] = kind
|
|
247
|
+
end
|
|
248
|
+
|
|
249
|
+
# A `(`: a capturing group, or `(?` followed by one of the openers
|
|
250
|
+
# ECMA-262 defines. Ruby's other `(?` forms are refused.
|
|
251
|
+
# @return [void]
|
|
252
|
+
def copy_group_opener(scan)
|
|
253
|
+
chars = scan[:chars]
|
|
254
|
+
index = scan[:index]
|
|
255
|
+
if chars[index + 1] != '?'
|
|
256
|
+
scan[:index] += 1
|
|
257
|
+
scan[:opened] += 1
|
|
258
|
+
# Beside a named group Ruby would not capture this one at all.
|
|
259
|
+
return emit(scan, scan[:names].empty? ? '(' : "(?<#{group_name_for(scan, scan[:opened])}>", :none)
|
|
260
|
+
end
|
|
261
|
+
|
|
262
|
+
opener = chars[index + 2].to_s
|
|
263
|
+
unless ECMA_GROUP_OPENERS.include?(opener) && !opener.empty?
|
|
264
|
+
raise SyntaxError,
|
|
265
|
+
"invalid group at index #{index}"
|
|
266
|
+
end
|
|
267
|
+
|
|
268
|
+
if opener == '<' && !'=!'.include?(chars[index + 3].to_s)
|
|
269
|
+
raise SyntaxError, "invalid group name at index #{index}" unless group_name_at(chars, index + 2)
|
|
270
|
+
|
|
271
|
+
scan[:opened] += 1
|
|
272
|
+
end
|
|
273
|
+
|
|
274
|
+
scan[:index] += 3
|
|
275
|
+
emit(scan, "(?#{opener}", :none)
|
|
276
|
+
end
|
|
277
|
+
|
|
278
|
+
# A `*`, `+` or `?`: a quantifier on the preceding atom, or the lazy
|
|
279
|
+
# marker on the preceding quantifier. ECMA-262 has nothing else for
|
|
280
|
+
# them to be: on nothing, or on a quantifier (Ruby's possessive `++`,
|
|
281
|
+
# nested `+*`), they are a syntax error ("Nothing to repeat").
|
|
282
|
+
# @return [void]
|
|
283
|
+
def copy_quantifier(scan, char)
|
|
284
|
+
scan[:index] += 1
|
|
285
|
+
case scan[:last]
|
|
286
|
+
when :atom then emit(scan, char, :quantifier)
|
|
287
|
+
when :quantifier
|
|
288
|
+
raise SyntaxError, "nothing to repeat at index #{scan[:index] - 1}" unless char == '?'
|
|
289
|
+
|
|
290
|
+
emit(scan, char, :lazy)
|
|
291
|
+
else
|
|
292
|
+
raise SyntaxError, "nothing to repeat at index #{scan[:index] - 1}"
|
|
293
|
+
end
|
|
294
|
+
end
|
|
295
|
+
|
|
296
|
+
# A `{`: a counted quantifier when it spells one (`{n}`, `{n,}`,
|
|
297
|
+
# `{n,m}`), else a literal brace (Annex B.1.4). Ruby also reads `{,m}`
|
|
298
|
+
# as a quantifier, so a literal is escaped rather than copied.
|
|
299
|
+
# @return [void]
|
|
300
|
+
def copy_brace(scan)
|
|
301
|
+
chars = scan[:chars]
|
|
302
|
+
index = scan[:index]
|
|
303
|
+
close = index + 1
|
|
304
|
+
close += 1 while close < chars.length && chars[close] != '}' && close - index <= 32
|
|
305
|
+
body = chars[(index + 1)...close].join
|
|
306
|
+
unless chars[close] == '}' && body.match?(/\A\d+(,\d*)?\z/)
|
|
307
|
+
scan[:index] += 1
|
|
308
|
+
return emit(scan, '\\{', :atom)
|
|
309
|
+
end
|
|
310
|
+
|
|
311
|
+
raise SyntaxError, "nothing to repeat at index #{index}" unless scan[:last] == :atom
|
|
312
|
+
|
|
313
|
+
scan[:index] = close + 1
|
|
314
|
+
emit(scan, "{#{body}}", :quantifier)
|
|
315
|
+
end
|
|
316
|
+
|
|
317
|
+
# One escape sequence, outside or inside a character class.
|
|
318
|
+
# @return [void]
|
|
319
|
+
def copy_escape(scan, in_class:)
|
|
320
|
+
chars = scan[:chars]
|
|
321
|
+
char = chars[scan[:index] + 1]
|
|
322
|
+
# A trailing backslash is no expression in either dialect; Ruby says so.
|
|
323
|
+
if char.nil?
|
|
324
|
+
scan[:index] += 1
|
|
325
|
+
return emit(scan, '\\', :atom)
|
|
326
|
+
end
|
|
327
|
+
|
|
328
|
+
scan[:index] += 2
|
|
329
|
+
case char
|
|
330
|
+
when '0'..'9' then copy_numeric_escape(scan, char, in_class: in_class)
|
|
331
|
+
when 'u' then copy_unicode_escape(scan)
|
|
332
|
+
when 'k' then copy_named_reference(scan, in_class: in_class)
|
|
333
|
+
else emit(scan, ecma_escape(char, in_class: in_class), 'bB'.include?(char) && !in_class ? :none : :atom)
|
|
334
|
+
end
|
|
335
|
+
end
|
|
336
|
+
|
|
337
|
+
# One escape's Ruby spelling. An escape ECMA-262 leaves undefined is
|
|
338
|
+
# an identity escape: the character itself.
|
|
339
|
+
# @param char [String] what followed the backslash
|
|
340
|
+
# @param in_class [Boolean] whether the escape sits in a character class
|
|
341
|
+
# @return [String]
|
|
342
|
+
def ecma_escape(char, in_class:)
|
|
343
|
+
return in_class ? ECMA_SPACE_MEMBERS : "[#{ECMA_SPACE_MEMBERS}]" if char == 's'
|
|
344
|
+
# Inside a class Ruby reads the nested one as a union, which is
|
|
345
|
+
# what a member set complement means there.
|
|
346
|
+
return "[^#{ECMA_SPACE_MEMBERS}]" if char == 'S'
|
|
347
|
+
# `\b` is a backspace inside a class and a word boundary outside it;
|
|
348
|
+
# `\B` asserts outside a class and is an identity escape inside one.
|
|
349
|
+
return in_class ? '\\b' : ECMA_WORD_BOUNDARY if char == 'b'
|
|
350
|
+
return in_class ? 'B' : ECMA_NON_BOUNDARY if char == 'B'
|
|
351
|
+
return "\\#{char}" if ECMA_KEPT_ESCAPES.include?(char)
|
|
352
|
+
|
|
353
|
+
# An identity escape: a letter stands for itself, and punctuation
|
|
354
|
+
# keeps the backslash (which means the same in both dialects).
|
|
355
|
+
char.match?(/[A-Za-z]/) ? char : "\\#{char}"
|
|
356
|
+
end
|
|
357
|
+
|
|
358
|
+
# A `\` followed by digits: a back-reference to a group the pattern
|
|
359
|
+
# declares, else (Annex B.1.4) a legacy octal escape of up to three
|
|
360
|
+
# octal digits, or the identity escapes `8` and `9`. A back-reference
|
|
361
|
+
# to a group that did not participate matches the empty string in
|
|
362
|
+
# ECMA-262 and fails in Ruby, so it is written as the conditional Ruby
|
|
363
|
+
# reads that way.
|
|
364
|
+
# @return [void]
|
|
365
|
+
def copy_numeric_escape(scan, first, in_class:)
|
|
366
|
+
chars = scan[:chars]
|
|
367
|
+
digits = +first
|
|
368
|
+
(digits << chars[scan[:index]]) && scan[:index] += 1 while chars[scan[:index]].to_s.match?(/\d/)
|
|
369
|
+
number = digits.to_i
|
|
370
|
+
if number.positive? && number <= scan[:groups] && !in_class
|
|
371
|
+
reject_repeated_reference(scan, number)
|
|
372
|
+
return emit(scan, "(?(#{number})\\#{number}|)", :atom) if scan[:names].empty?
|
|
373
|
+
|
|
374
|
+
name = group_name_for(scan, number)
|
|
375
|
+
return emit(scan, "(?(<#{name}>)\\k<#{name}>|)", :atom)
|
|
376
|
+
end
|
|
377
|
+
|
|
378
|
+
octal = digits.match(/\A[0-7]{1,3}/)&.to_s
|
|
379
|
+
octal = octal[0, 2] if octal && octal.length == 3 && octal.to_i(8) > 255
|
|
380
|
+
if octal.nil?
|
|
381
|
+
# `8` and `9` stand for themselves; the digits after them too.
|
|
382
|
+
return emit(scan, digits, :atom)
|
|
383
|
+
end
|
|
384
|
+
|
|
385
|
+
emit(scan, format('\\x%02X', octal.to_i(8)) + digits[octal.length..], :atom)
|
|
386
|
+
end
|
|
387
|
+
|
|
388
|
+
# `\uXXXX`: a code unit, which Ruby reads as a code point. A surrogate
|
|
389
|
+
# pair (ECMA-262 without the `u` flag matches the two units of one
|
|
390
|
+
# character) is joined into the character it encodes; a lone
|
|
391
|
+
# surrogate is no character any string carries, so it matches
|
|
392
|
+
# nothing. A `\u` that spells no code unit is an identity escape.
|
|
393
|
+
# @return [void]
|
|
394
|
+
def copy_unicode_escape(scan)
|
|
395
|
+
high = hex_code_unit(scan, scan[:index])
|
|
396
|
+
return emit(scan, 'u', :atom) unless high
|
|
397
|
+
|
|
398
|
+
scan[:index] += 4
|
|
399
|
+
low = nil
|
|
400
|
+
if high.between?(0xD800, 0xDBFF) && scan[:chars][scan[:index]] == '\\' && scan[:chars][scan[:index] + 1] == 'u'
|
|
401
|
+
low = hex_code_unit(scan, scan[:index] + 2)
|
|
402
|
+
low = nil unless low&.between?(0xDC00, 0xDFFF)
|
|
403
|
+
end
|
|
404
|
+
if low
|
|
405
|
+
scan[:index] += 6
|
|
406
|
+
return emit(scan, format('\\u{%X}', 0x10000 + ((high - 0xD800) << 10) + (low - 0xDC00)), :atom)
|
|
407
|
+
end
|
|
408
|
+
return emit(scan, ECMA_EMPTY_CLASS, :atom) if high.between?(0xD800, 0xDFFF)
|
|
409
|
+
|
|
410
|
+
emit(scan, format('\\u%04X', high), :atom)
|
|
411
|
+
end
|
|
412
|
+
|
|
413
|
+
# @return [Integer, nil] the four hex digits at an index, as a number
|
|
414
|
+
def hex_code_unit(scan, index)
|
|
415
|
+
text = scan[:chars][index, 4]&.join.to_s
|
|
416
|
+
text.match?(/\A\h{4}\z/) ? text.to_i(16) : nil
|
|
417
|
+
end
|
|
418
|
+
|
|
419
|
+
# `\k<name>`: a named back-reference where the pattern declares named
|
|
420
|
+
# groups (with the same empty-match rule as a numbered one); with none
|
|
421
|
+
# declared it is an identity escape for "k" (Annex B.1.2).
|
|
422
|
+
# @return [void]
|
|
423
|
+
def copy_named_reference(scan, in_class:)
|
|
424
|
+
return emit(scan, 'k', :atom) if scan[:names].empty? || in_class
|
|
425
|
+
|
|
426
|
+
name = group_name_at(scan[:chars], scan[:index])
|
|
427
|
+
unless name && scan[:names].include?(name)
|
|
428
|
+
raise SyntaxError,
|
|
429
|
+
"invalid named reference at index #{scan[:index] - 2}"
|
|
430
|
+
end
|
|
431
|
+
|
|
432
|
+
scan[:index] += name.length + 2
|
|
433
|
+
reject_repeated_reference(scan, scan[:order].index(name) + 1)
|
|
434
|
+
emit(scan, "(?(<#{name}>)\\k<#{name}>|)", :atom)
|
|
435
|
+
end
|
|
436
|
+
|
|
437
|
+
# A back-reference to a group a quantifier may repeat reads one way in
|
|
438
|
+
# ECMA-262 (cleared at each iteration) and another in Ruby (kept), and
|
|
439
|
+
# the difference decides instances either way round, so the pattern is
|
|
440
|
+
# refused rather than answered.
|
|
441
|
+
# @param number [Integer] the group the reference names
|
|
442
|
+
# @return [void]
|
|
443
|
+
# @raise [Untranslatable]
|
|
444
|
+
def reject_repeated_reference(scan, number)
|
|
445
|
+
return unless scan[:repeated].include?(number)
|
|
446
|
+
|
|
447
|
+
raise Untranslatable,
|
|
448
|
+
"a back-reference to group #{number}, which a quantifier repeats, cannot be evaluated faithfully " \
|
|
449
|
+
"(ECMA-262 clears the group's capture at each repetition and Ruby keeps it)"
|
|
450
|
+
end
|
|
451
|
+
|
|
452
|
+
# The capturing groups a quantifier may repeat: those inside (or being)
|
|
453
|
+
# a group followed by a quantifier that allows a second iteration.
|
|
454
|
+
# `?` and `{0,1}` allow only one, so nothing is ever cleared between
|
|
455
|
+
# iterations there and the existing "did the group participate"
|
|
456
|
+
# conditional already reads the way ECMA-262 does. Read in one pass,
|
|
457
|
+
# skipping escapes and character classes, so a `(` written as a
|
|
458
|
+
# literal opens nothing.
|
|
459
|
+
# @param chars [Array<String>] the pattern
|
|
460
|
+
# @return [Array<Integer>] the group numbers
|
|
461
|
+
def repeated_capture_groups(chars)
|
|
462
|
+
state = { open: [], repeated: [], number: 0, index: 0, in_class: false }
|
|
463
|
+
while state[:index] < chars.length
|
|
464
|
+
char = chars[state[:index]]
|
|
465
|
+
state[:index] += 1
|
|
466
|
+
next state[:index] += 1 if char == '\\'
|
|
467
|
+
next state[:in_class] = true if char == '[' && !state[:in_class]
|
|
468
|
+
next state[:in_class] = false if char == ']' && state[:in_class]
|
|
469
|
+
next if state[:in_class]
|
|
470
|
+
|
|
471
|
+
open_repeat_group(state, chars) if char == '('
|
|
472
|
+
close_repeat_group(state, chars) if char == ')'
|
|
473
|
+
end
|
|
474
|
+
state[:repeated]
|
|
475
|
+
end
|
|
476
|
+
|
|
477
|
+
# @return [void]
|
|
478
|
+
def open_repeat_group(state, chars)
|
|
479
|
+
capturing = chars[state[:index]] != '?' || group_name_at(chars, state[:index] + 1).to_s != ''
|
|
480
|
+
number = capturing ? (state[:number] += 1) : nil
|
|
481
|
+
state[:open] << { number: number, inner: [] }
|
|
482
|
+
end
|
|
483
|
+
|
|
484
|
+
# @return [void]
|
|
485
|
+
def close_repeat_group(state, chars)
|
|
486
|
+
group = state[:open].pop
|
|
487
|
+
return unless group
|
|
488
|
+
|
|
489
|
+
members = group[:inner] + [group[:number]].compact
|
|
490
|
+
state[:repeated].concat(members) if quantifier_at?(chars, state[:index])
|
|
491
|
+
parent = state[:open].last
|
|
492
|
+
parent ? parent[:inner].concat(members) : nil
|
|
493
|
+
end
|
|
494
|
+
|
|
495
|
+
# @return [Boolean] whether the quantifier at an index admits a second
|
|
496
|
+
# iteration, which is when ECMA-262's per-iteration clearing of the
|
|
497
|
+
# captures inside it can be seen at all
|
|
498
|
+
def quantifier_at?(chars, index)
|
|
499
|
+
char = chars[index]
|
|
500
|
+
return true if ['*', '+'].include?(char)
|
|
501
|
+
return false unless char == '{'
|
|
502
|
+
|
|
503
|
+
bounds = chars[index..].join[/\A\{(\d+)(,(\d*))?\}/, 0]
|
|
504
|
+
return false unless bounds
|
|
505
|
+
|
|
506
|
+
repeated_bounds?(Regexp.last_match(1).to_i, Regexp.last_match(2), Regexp.last_match(3))
|
|
507
|
+
end
|
|
508
|
+
|
|
509
|
+
# @param least [Integer] the `{n` of the quantifier
|
|
510
|
+
# @param comma [String, nil] its `,`, when it has one
|
|
511
|
+
# @param most [String, nil] its `m`, when it has one
|
|
512
|
+
# @return [Boolean] whether it admits two iterations
|
|
513
|
+
def repeated_bounds?(least, comma, most)
|
|
514
|
+
return least >= 2 if comma.nil?
|
|
515
|
+
return true if most.nil? || most.empty?
|
|
516
|
+
|
|
517
|
+
most.to_i >= 2
|
|
518
|
+
end
|
|
519
|
+
|
|
520
|
+
# Copy a character class, which ECMA-262 and Ruby read differently: an
|
|
521
|
+
# empty class is legal there and matches nothing, and `[` and `&`
|
|
522
|
+
# inside a class are literals rather than the openers of Ruby's nested
|
|
523
|
+
# classes and set intersection.
|
|
524
|
+
# @return [void]
|
|
525
|
+
def copy_character_class(scan)
|
|
526
|
+
chars = scan[:chars]
|
|
527
|
+
opening = scan[:index]
|
|
528
|
+
negated = chars[opening + 1] == '^'
|
|
529
|
+
scan[:index] = opening + (negated ? 2 : 1)
|
|
530
|
+
outer = scan[:out]
|
|
531
|
+
scan[:out] = +''
|
|
532
|
+
while scan[:index] < chars.length && chars[scan[:index]] != ']'
|
|
533
|
+
note_translation_progress(scan)
|
|
534
|
+
if chars[scan[:index]] == '\\'
|
|
535
|
+
copy_escape(scan, in_class: true)
|
|
536
|
+
else
|
|
537
|
+
emit(scan, class_member(chars[scan[:index]]), :atom)
|
|
538
|
+
scan[:index] += 1
|
|
539
|
+
end
|
|
540
|
+
end
|
|
541
|
+
body = scan[:out]
|
|
542
|
+
scan[:out] = outer
|
|
543
|
+
# Unterminated: no expression in either dialect.
|
|
544
|
+
raise SyntaxError, "unterminated character class at index #{opening}" if scan[:index] >= chars.length
|
|
545
|
+
|
|
546
|
+
scan[:index] += 1
|
|
547
|
+
emit(scan, class_source(body, negated), :atom)
|
|
548
|
+
end
|
|
549
|
+
|
|
550
|
+
# @return [String] the Ruby spelling of one unescaped class member
|
|
551
|
+
def class_member(char)
|
|
552
|
+
# Ruby reads a nested `[` as another class and `&&` as intersection;
|
|
553
|
+
# ECMA-262 has neither, so both are literals there.
|
|
554
|
+
['[', '&'].include?(char) ? "\\#{char}" : char
|
|
555
|
+
end
|
|
556
|
+
|
|
557
|
+
# @param body [String] the translated members
|
|
558
|
+
# @param negated [Boolean] whether the class was written with a `^`
|
|
559
|
+
# @return [String] the Ruby character class
|
|
560
|
+
def class_source(body, negated)
|
|
561
|
+
return negated ? ECMA_ANY_CLASS : ECMA_EMPTY_CLASS if body.empty?
|
|
562
|
+
|
|
563
|
+
negated ? "[^#{body}]" : "[#{body}]"
|
|
564
|
+
end
|
|
565
|
+
end
|
|
566
|
+
end
|
|
567
|
+
end
|