ruby-mcp-client 2.1.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. checksums.yaml +4 -4
  2. data/OAUTH.md +555 -0
  3. data/README.md +825 -48
  4. data/lib/mcp_client/audio_content.rb +1 -1
  5. data/lib/mcp_client/auth/browser_oauth.rb +131 -21
  6. data/lib/mcp_client/auth/oauth_provider/challenge_handling.rb +532 -0
  7. data/lib/mcp_client/auth/oauth_provider/client_authentication.rb +121 -0
  8. data/lib/mcp_client/auth/oauth_provider/pending_requests.rb +51 -0
  9. data/lib/mcp_client/auth/oauth_provider/registration_store.rb +486 -0
  10. data/lib/mcp_client/auth/oauth_provider/response_validation.rb +441 -0
  11. data/lib/mcp_client/auth/oauth_provider/scope_selection.rb +134 -0
  12. data/lib/mcp_client/auth/oauth_provider/token_store.rb +419 -0
  13. data/lib/mcp_client/auth/oauth_provider.rb +1354 -386
  14. data/lib/mcp_client/auth/peer_text.rb +174 -0
  15. data/lib/mcp_client/auth.rb +298 -32
  16. data/lib/mcp_client/cached_result.rb +145 -0
  17. data/lib/mcp_client/called_tool_definition.rb +138 -0
  18. data/lib/mcp_client/client/cache_slices.rb +195 -0
  19. data/lib/mcp_client/client/list_aggregation.rb +243 -0
  20. data/lib/mcp_client/client/notification_routing.rb +155 -0
  21. data/lib/mcp_client/client/sampling_validation.rb +200 -0
  22. data/lib/mcp_client/client/task_api.rb +531 -0
  23. data/lib/mcp_client/client/task_lifetimes.rb +269 -0
  24. data/lib/mcp_client/client/task_registry.rb +254 -0
  25. data/lib/mcp_client/client/task_shape.rb +102 -0
  26. data/lib/mcp_client/client/task_support.rb +1166 -0
  27. data/lib/mcp_client/client/task_updates.rb +457 -0
  28. data/lib/mcp_client/client/task_wait_boundaries.rb +198 -0
  29. data/lib/mcp_client/client/task_workers.rb +63 -0
  30. data/lib/mcp_client/client.rb +796 -518
  31. data/lib/mcp_client/deep_copy.rb +49 -0
  32. data/lib/mcp_client/deprecation_notices.rb +94 -0
  33. data/lib/mcp_client/deprecations.rb +419 -0
  34. data/lib/mcp_client/errors.rb +474 -7
  35. data/lib/mcp_client/header_params.rb +320 -0
  36. data/lib/mcp_client/http_transport_base/bounded_inflate.rb +41 -0
  37. data/lib/mcp_client/http_transport_base/cache_support.rb +694 -0
  38. data/lib/mcp_client/http_transport_base/era_detection.rb +134 -0
  39. data/lib/mcp_client/http_transport_base/listen_stream.rb +763 -0
  40. data/lib/mcp_client/http_transport_base/param_headers.rb +35 -0
  41. data/lib/mcp_client/http_transport_base/request_recovery.rb +156 -0
  42. data/lib/mcp_client/http_transport_base/session_recovery.rb +113 -0
  43. data/lib/mcp_client/http_transport_base/sse_event_scanner.rb +145 -0
  44. data/lib/mcp_client/http_transport_base/stream_capture.rb +160 -0
  45. data/lib/mcp_client/http_transport_base/stream_recovery.rb +318 -0
  46. data/lib/mcp_client/http_transport_base/tool_listing.rb +277 -0
  47. data/lib/mcp_client/http_transport_base.rb +666 -120
  48. data/lib/mcp_client/input_round_trips.rb +128 -0
  49. data/lib/mcp_client/json_rpc_common/envelopes.rb +32 -0
  50. data/lib/mcp_client/json_rpc_common/error_bodies.rb +105 -0
  51. data/lib/mcp_client/json_rpc_common/input_waits.rb +167 -0
  52. data/lib/mcp_client/json_rpc_common.rb +900 -13
  53. data/lib/mcp_client/oauth_client.rb +14 -5
  54. data/lib/mcp_client/prompt.rb +4 -0
  55. data/lib/mcp_client/request_authorization.rb +128 -0
  56. data/lib/mcp_client/request_meta_scope.rb +77 -0
  57. data/lib/mcp_client/request_metadata.rb +287 -0
  58. data/lib/mcp_client/resource.rb +4 -0
  59. data/lib/mcp_client/resource_content.rb +20 -0
  60. data/lib/mcp_client/resource_template.rb +4 -0
  61. data/lib/mcp_client/result_caching.rb +999 -0
  62. data/lib/mcp_client/result_completeness.rb +34 -0
  63. data/lib/mcp_client/root.rb +6 -0
  64. data/lib/mcp_client/round_trip_marker.rb +28 -0
  65. data/lib/mcp_client/schema_validator/annotations.rb +82 -0
  66. data/lib/mcp_client/schema_validator/composition.rb +86 -0
  67. data/lib/mcp_client/schema_validator/dialects.rb +66 -0
  68. data/lib/mcp_client/schema_validator/ecma_patterns.rb +567 -0
  69. data/lib/mcp_client/schema_validator/evaluation.rb +517 -0
  70. data/lib/mcp_client/schema_validator/input_requirements.rb +84 -0
  71. data/lib/mcp_client/schema_validator/instances.rb +449 -0
  72. data/lib/mcp_client/schema_validator/keyword_scan.rb +121 -0
  73. data/lib/mcp_client/schema_validator/normalization.rb +104 -0
  74. data/lib/mcp_client/schema_validator/references.rb +610 -0
  75. data/lib/mcp_client/schema_validator/scalars.rb +126 -0
  76. data/lib/mcp_client/schema_validator/shapes.rb +319 -0
  77. data/lib/mcp_client/schema_validator/uri_references.rb +153 -0
  78. data/lib/mcp_client/schema_validator.rb +882 -208
  79. data/lib/mcp_client/server_base.rb +233 -5
  80. data/lib/mcp_client/server_factory.rb +9 -3
  81. data/lib/mcp_client/server_http/json_rpc_transport.rb +219 -4
  82. data/lib/mcp_client/server_http.rb +307 -90
  83. data/lib/mcp_client/server_sse/json_rpc_transport.rb +113 -25
  84. data/lib/mcp_client/server_sse/sse_parser.rb +39 -6
  85. data/lib/mcp_client/server_sse.rb +227 -62
  86. data/lib/mcp_client/server_stdio/child_session.rb +98 -0
  87. data/lib/mcp_client/server_stdio/json_rpc_transport.rb +1003 -28
  88. data/lib/mcp_client/server_stdio.rb +772 -183
  89. data/lib/mcp_client/server_streamable_http/json_rpc_transport.rb +189 -25
  90. data/lib/mcp_client/server_streamable_http.rb +302 -115
  91. data/lib/mcp_client/session_pin.rb +119 -0
  92. data/lib/mcp_client/subscription/notification_dispatcher.rb +354 -0
  93. data/lib/mcp_client/subscription.rb +852 -0
  94. data/lib/mcp_client/subscription_support.rb +715 -0
  95. data/lib/mcp_client/task.rb +286 -14
  96. data/lib/mcp_client/tool.rb +31 -3
  97. data/lib/mcp_client/version.rb +21 -6
  98. data/lib/mcp_client.rb +108 -19
  99. metadata +68 -2
@@ -0,0 +1,567 @@
1
+ # frozen_string_literal: true
2
+
3
+ module MCPClient
4
+ module SchemaValidator
5
+ # The rewrite of an ECMA-262 pattern as the Ruby expression that means
6
+ # the same thing. JSON Schema 2020-12 Core Section 4.3 requires patterns
7
+ # to be interpreted as ECMA-262 regular expressions, and Ruby's differ
8
+ # from them in both directions: what Ruby accepts that ECMA-262 rejects
9
+ # makes the validator accept a value the schema refuses, and the
10
+ # converse rejects a conforming one (and, through `not` or
11
+ # `additionalProperties: false`, flips both). Extended into
12
+ # SchemaValidator, so the methods are its own.
13
+ #
14
+ # A pattern comes from the remote peer and is as long as the peer made
15
+ # it, so the translation is one linear pass over its characters that
16
+ # consults the validation-wide deadline as it goes, and a pattern past
17
+ # MAX_PATTERN_LENGTH is refused before it is read at all.
18
+ module EcmaPatterns
19
+ # A pattern that is no ECMA-262 expression: syntax ECMA-262 does not
20
+ # define (Ruby's inline flags, possessive quantifiers, atomic groups,
21
+ # comments) or an expression neither dialect accepts. A RegexpError,
22
+ # so every caller that rescues an unreadable expression sees it.
23
+ class SyntaxError < RegexpError; end
24
+
25
+ # A pattern that IS an ECMA-262 expression but whose meaning Ruby's
26
+ # engine cannot be made to reproduce, so translating it would answer
27
+ # some instances wrongly. The two engines differ in two ways no
28
+ # rewriting bridges:
29
+ #
30
+ # - ECMA-262 clears the captures inside a quantified group at the start
31
+ # of every iteration (a back-reference to a group the last iteration
32
+ # did not enter matches the empty string); Ruby keeps whatever the
33
+ # last iteration that entered it captured. `^(a|(b))*\2$` accepts
34
+ # "aba" and refuses "abab" there, and exactly the opposite here.
35
+ # - ECMA-262 lookbehind is variable-length (ES2018); Ruby's is not.
36
+ #
37
+ # Refusing the schema is the only honest answer left: a verdict from
38
+ # the other engine's rules would accept structured content the schema
39
+ # forbids as readily as it would refuse conforming content.
40
+ class Untranslatable < RegexpError; end
41
+
42
+ # What each ECMA-262 anchor means in Ruby: the ends of the subject,
43
+ # never a line boundary.
44
+ ECMA_ANCHORS = { '^' => '\\A', '$' => '\\z' }.freeze
45
+
46
+ # ECMA-262 `.` matches every character except the four line
47
+ # terminators; Ruby's excludes only "\n".
48
+ ECMA_DOT = '[^\\n\\r\\u2028\\u2029]'
49
+
50
+ # The members of ECMA-262's `\s` (WhiteSpace plus LineTerminator):
51
+ # Ruby's is `[ \t\r\n\f\v]` and knows none of the Unicode spaces, so a
52
+ # non-breaking space failed a pattern ECMAScript satisfies.
53
+ ECMA_SPACE_MEMBERS = '\\t\\n\\v\\f\\r \\u00a0\\u1680\\u2000-\\u200a\\u2028\\u2029\\u202f\\u205f\\u3000\\ufeff'
54
+
55
+ # A character class matching nothing, which is what ECMA-262 makes of
56
+ # `[]` — Ruby cannot compile that at all, so the pattern used to be
57
+ # dropped and every string satisfied it.
58
+ ECMA_EMPTY_CLASS = '[^\\s\\S]'
59
+
60
+ # Its complement: ECMA-262 `[^]` matches any character, line
61
+ # terminators included.
62
+ ECMA_ANY_CLASS = '[\\s\\S]'
63
+
64
+ # The escapes ECMA-262 defines, which Ruby reads the same way: the
65
+ # class escapes, the control escapes, a hex or control-letter escape
66
+ # and a Unicode property. `\s` / `\S` are defined by both but over
67
+ # different sets, `\b` / `\B` over different word characters, the
68
+ # digits are back-references or legacy octal escapes, `\u` may spell
69
+ # a surrogate pair and `\k` a named back-reference, so those are
70
+ # rewritten rather than kept.
71
+ ECMA_KEPT_ESCAPES = 'dDwWfnrtvxcpP'
72
+
73
+ # ECMA-262's word characters: `\w` is [A-Za-z0-9_] there, and its
74
+ # word-boundary assertions are defined over exactly those, while Ruby's
75
+ # `\b` knows every Unicode letter — so "é" has a boundary in Ruby and
76
+ # none in ECMA-262.
77
+ ECMA_WORD = '[A-Za-z0-9_]'
78
+
79
+ # `\b`: a word character on exactly one side.
80
+ ECMA_WORD_BOUNDARY = "(?:(?<=#{ECMA_WORD})(?!#{ECMA_WORD})|(?<!#{ECMA_WORD})(?=#{ECMA_WORD}))".freeze
81
+
82
+ # `\B`: word characters on both sides, or on neither.
83
+ ECMA_NON_BOUNDARY = "(?:(?<=#{ECMA_WORD})(?=#{ECMA_WORD})|(?<!#{ECMA_WORD})(?!#{ECMA_WORD}))".freeze
84
+
85
+ # The characters that may follow `(?` in ECMA-262: a non-capturing
86
+ # group, a lookahead, and (after `<`) a lookbehind or a named group.
87
+ # Anything else Ruby reads as an inline option, an atomic group, a
88
+ # comment or its own named-group syntax, none of which ECMA-262 has.
89
+ ECMA_GROUP_OPENERS = ':=!<'
90
+
91
+ # How many characters the translation reads between two looks at the
92
+ # deadline.
93
+ TRANSLATION_CHECK_INTERVAL = 256
94
+
95
+ # The Ruby expression an ECMA-262 pattern means.
96
+ # @param pattern [String] the peer's pattern
97
+ # @param timeout [Float] seconds the match may take
98
+ # @param deadline [Float, nil] monotonic deadline the translation runs under
99
+ # @return [Regexp]
100
+ # @raise [RegexpError] when the pattern is not a usable expression
101
+ # @raise [Untranslatable] when it is one Ruby cannot reproduce
102
+ # @raise [Aborted] when the deadline passes during the translation
103
+ def ecma_regexp(pattern, timeout, deadline = nil)
104
+ Regexp.new(ecma_source(pattern, deadline), timeout: timeout)
105
+ rescue RegexpError => e
106
+ # ECMA-262 lookbehind has been variable-length since ES2018 and
107
+ # Ruby's never has been: the pattern is a good expression this
108
+ # engine cannot be given, not a bad one.
109
+ raise Untranslatable, "variable-length lookbehind cannot be evaluated faithfully (#{e.message})" if
110
+ e.message.include?('look-behind')
111
+
112
+ raise
113
+ end
114
+
115
+ # Rewrite an ECMA-262 pattern as Ruby regexp source. Everything
116
+ # ECMA-262 defines is kept; what only Ruby defines is either read the
117
+ # way ECMA-262 reads it — an escape ECMA-262 does not define is an
118
+ # identity escape there (Annex B.1.2), so `\A` is a literal "A" and not
119
+ # the start of the subject — or refused where ECMA-262 refuses it.
120
+ # @param pattern [String] the peer's pattern
121
+ # @param deadline [Float, nil] monotonic deadline the translation runs under
122
+ # @return [String] Ruby regexp source
123
+ # @raise [SyntaxError] for syntax ECMA-262 does not define
124
+ # @raise [Aborted] when the deadline passes during the translation
125
+ def ecma_source(pattern, deadline = nil)
126
+ raise SyntaxError, "pattern is longer than #{MAX_PATTERN_LENGTH} characters" if
127
+ pattern.length > MAX_PATTERN_LENGTH
128
+
129
+ chars = pattern.chars
130
+ scan = { chars: chars, index: 0, out: +'', deadline: deadline, read: 0, last: :none, opened: 0 }
131
+ scan[:order] = count_capture_groups(chars)
132
+ scan[:groups] = scan[:order].length
133
+ scan[:names] = scan[:order].compact
134
+ scan[:repeated] = repeated_capture_groups(chars)
135
+ scan[:generated] = generated_name_prefix(scan[:names])
136
+ while scan[:index] < chars.length
137
+ note_translation_progress(scan)
138
+ chars[scan[:index]] == '[' ? copy_character_class(scan) : copy_ecma_token(scan)
139
+ end
140
+ scan[:out]
141
+ end
142
+
143
+ # Consult the deadline every TRANSLATION_CHECK_INTERVAL characters.
144
+ # @raise [Aborted]
145
+ def note_translation_progress(scan)
146
+ scan[:read] += 1
147
+ return unless (scan[:read] % TRANSLATION_CHECK_INTERVAL).zero?
148
+
149
+ raise Aborted, 'validation time budget exhausted while translating a pattern' if
150
+ budget_exhausted?(scan[:deadline])
151
+ end
152
+
153
+ # The capturing groups a pattern declares, in order and in one pass:
154
+ # what a numeric escape refers to depends on how many there are (Annex
155
+ # B.1.4: a number past the count is a legacy octal escape) and on which
156
+ # one it names — ECMA-262 numbers named and unnamed groups alike,
157
+ # left to right, while Ruby stops capturing unnamed groups once a
158
+ # named one exists, so the numbering is kept here and every group is
159
+ # written as a named one there ({#copy_group_opener}).
160
+ # @return [Array<String, nil>] each group's name, nil for an unnamed one
161
+ def count_capture_groups(chars)
162
+ order = []
163
+ index = 0
164
+ in_class = false
165
+ while index < chars.length
166
+ char = chars[index]
167
+ if char == '\\'
168
+ index += 2
169
+ next
170
+ end
171
+ in_class = true if char == '['
172
+ in_class = false if char == ']' && in_class
173
+ if char == '(' && !in_class
174
+ order << nil if chars[index + 1] != '?'
175
+ name = group_name_at(chars, index + 2)
176
+ order << name if name
177
+ end
178
+ index += 1
179
+ end
180
+ order
181
+ end
182
+
183
+ # The Ruby name of a capturing group, by its ECMA-262 number: its own
184
+ # where it has one, a generated one otherwise.
185
+ # @param number [Integer] the group's number, from 1
186
+ # @return [String]
187
+ def group_name_for(scan, number)
188
+ scan[:order][number - 1] || "#{scan[:generated]}#{number}"
189
+ end
190
+
191
+ # A prefix for the generated names that none of the pattern's own
192
+ # names begins with, so a written `(?<__mcp_g1>` can never be the
193
+ # group a numeric back-reference is rewritten to name.
194
+ # @param names [Array<String>] the names the pattern wrote
195
+ # @return [String]
196
+ def generated_name_prefix(names)
197
+ prefix = +'__mcp_g'
198
+ prefix << '_' while names.any? { |name| name.start_with?(prefix) }
199
+ prefix
200
+ end
201
+
202
+ # The name of a `(?<name>` group opening at the index of its `<`.
203
+ # @return [String, nil]
204
+ def group_name_at(chars, index)
205
+ return nil unless chars[index] == '<' && !'=!'.include?(chars[index + 1].to_s)
206
+
207
+ close = index + 1
208
+ close += 1 while close < chars.length && chars[close] != '>'
209
+ chars[(index + 1)...close].join if close < chars.length
210
+ end
211
+
212
+ # Copy one token from outside a character class.
213
+ # @return [void]
214
+ def copy_ecma_token(scan)
215
+ char = scan[:chars][scan[:index]]
216
+ case char
217
+ when '\\' then copy_escape(scan, in_class: false)
218
+ when '(' then copy_group_opener(scan)
219
+ when '*', '+', '?' then copy_quantifier(scan, char)
220
+ when '{' then copy_brace(scan)
221
+ else copy_plain_token(scan, char)
222
+ end
223
+ end
224
+
225
+ # A character that is neither an escape, a group opener nor a
226
+ # quantifier.
227
+ # @return [void]
228
+ def copy_plain_token(scan, char)
229
+ scan[:index] += 1
230
+ case char
231
+ when '^', '$'
232
+ emit(scan, ECMA_ANCHORS[char], :none)
233
+ when '.' then emit(scan, ECMA_DOT, :atom)
234
+ when '|' then emit(scan, char, :none)
235
+ # A `]` or `}` that closes nothing is a literal in ECMA-262 (Annex
236
+ # B.1.4); Ruby reads a bare `}` the same way but is spared the guess.
237
+ when ']', '}' then emit(scan, "\\#{char}", :atom)
238
+ else emit(scan, char, :atom)
239
+ end
240
+ end
241
+
242
+ # Append translated source and record what kind of token it was.
243
+ # @return [void]
244
+ def emit(scan, source, kind)
245
+ scan[:out] << source
246
+ scan[:last] = kind
247
+ end
248
+
249
+ # A `(`: a capturing group, or `(?` followed by one of the openers
250
+ # ECMA-262 defines. Ruby's other `(?` forms are refused.
251
+ # @return [void]
252
+ def copy_group_opener(scan)
253
+ chars = scan[:chars]
254
+ index = scan[:index]
255
+ if chars[index + 1] != '?'
256
+ scan[:index] += 1
257
+ scan[:opened] += 1
258
+ # Beside a named group Ruby would not capture this one at all.
259
+ return emit(scan, scan[:names].empty? ? '(' : "(?<#{group_name_for(scan, scan[:opened])}>", :none)
260
+ end
261
+
262
+ opener = chars[index + 2].to_s
263
+ unless ECMA_GROUP_OPENERS.include?(opener) && !opener.empty?
264
+ raise SyntaxError,
265
+ "invalid group at index #{index}"
266
+ end
267
+
268
+ if opener == '<' && !'=!'.include?(chars[index + 3].to_s)
269
+ raise SyntaxError, "invalid group name at index #{index}" unless group_name_at(chars, index + 2)
270
+
271
+ scan[:opened] += 1
272
+ end
273
+
274
+ scan[:index] += 3
275
+ emit(scan, "(?#{opener}", :none)
276
+ end
277
+
278
+ # A `*`, `+` or `?`: a quantifier on the preceding atom, or the lazy
279
+ # marker on the preceding quantifier. ECMA-262 has nothing else for
280
+ # them to be: on nothing, or on a quantifier (Ruby's possessive `++`,
281
+ # nested `+*`), they are a syntax error ("Nothing to repeat").
282
+ # @return [void]
283
+ def copy_quantifier(scan, char)
284
+ scan[:index] += 1
285
+ case scan[:last]
286
+ when :atom then emit(scan, char, :quantifier)
287
+ when :quantifier
288
+ raise SyntaxError, "nothing to repeat at index #{scan[:index] - 1}" unless char == '?'
289
+
290
+ emit(scan, char, :lazy)
291
+ else
292
+ raise SyntaxError, "nothing to repeat at index #{scan[:index] - 1}"
293
+ end
294
+ end
295
+
296
+ # A `{`: a counted quantifier when it spells one (`{n}`, `{n,}`,
297
+ # `{n,m}`), else a literal brace (Annex B.1.4). Ruby also reads `{,m}`
298
+ # as a quantifier, so a literal is escaped rather than copied.
299
+ # @return [void]
300
+ def copy_brace(scan)
301
+ chars = scan[:chars]
302
+ index = scan[:index]
303
+ close = index + 1
304
+ close += 1 while close < chars.length && chars[close] != '}' && close - index <= 32
305
+ body = chars[(index + 1)...close].join
306
+ unless chars[close] == '}' && body.match?(/\A\d+(,\d*)?\z/)
307
+ scan[:index] += 1
308
+ return emit(scan, '\\{', :atom)
309
+ end
310
+
311
+ raise SyntaxError, "nothing to repeat at index #{index}" unless scan[:last] == :atom
312
+
313
+ scan[:index] = close + 1
314
+ emit(scan, "{#{body}}", :quantifier)
315
+ end
316
+
317
+ # One escape sequence, outside or inside a character class.
318
+ # @return [void]
319
+ def copy_escape(scan, in_class:)
320
+ chars = scan[:chars]
321
+ char = chars[scan[:index] + 1]
322
+ # A trailing backslash is no expression in either dialect; Ruby says so.
323
+ if char.nil?
324
+ scan[:index] += 1
325
+ return emit(scan, '\\', :atom)
326
+ end
327
+
328
+ scan[:index] += 2
329
+ case char
330
+ when '0'..'9' then copy_numeric_escape(scan, char, in_class: in_class)
331
+ when 'u' then copy_unicode_escape(scan)
332
+ when 'k' then copy_named_reference(scan, in_class: in_class)
333
+ else emit(scan, ecma_escape(char, in_class: in_class), 'bB'.include?(char) && !in_class ? :none : :atom)
334
+ end
335
+ end
336
+
337
+ # One escape's Ruby spelling. An escape ECMA-262 leaves undefined is
338
+ # an identity escape: the character itself.
339
+ # @param char [String] what followed the backslash
340
+ # @param in_class [Boolean] whether the escape sits in a character class
341
+ # @return [String]
342
+ def ecma_escape(char, in_class:)
343
+ return in_class ? ECMA_SPACE_MEMBERS : "[#{ECMA_SPACE_MEMBERS}]" if char == 's'
344
+ # Inside a class Ruby reads the nested one as a union, which is
345
+ # what a member set complement means there.
346
+ return "[^#{ECMA_SPACE_MEMBERS}]" if char == 'S'
347
+ # `\b` is a backspace inside a class and a word boundary outside it;
348
+ # `\B` asserts outside a class and is an identity escape inside one.
349
+ return in_class ? '\\b' : ECMA_WORD_BOUNDARY if char == 'b'
350
+ return in_class ? 'B' : ECMA_NON_BOUNDARY if char == 'B'
351
+ return "\\#{char}" if ECMA_KEPT_ESCAPES.include?(char)
352
+
353
+ # An identity escape: a letter stands for itself, and punctuation
354
+ # keeps the backslash (which means the same in both dialects).
355
+ char.match?(/[A-Za-z]/) ? char : "\\#{char}"
356
+ end
357
+
358
+ # A `\` followed by digits: a back-reference to a group the pattern
359
+ # declares, else (Annex B.1.4) a legacy octal escape of up to three
360
+ # octal digits, or the identity escapes `8` and `9`. A back-reference
361
+ # to a group that did not participate matches the empty string in
362
+ # ECMA-262 and fails in Ruby, so it is written as the conditional Ruby
363
+ # reads that way.
364
+ # @return [void]
365
+ def copy_numeric_escape(scan, first, in_class:)
366
+ chars = scan[:chars]
367
+ digits = +first
368
+ (digits << chars[scan[:index]]) && scan[:index] += 1 while chars[scan[:index]].to_s.match?(/\d/)
369
+ number = digits.to_i
370
+ if number.positive? && number <= scan[:groups] && !in_class
371
+ reject_repeated_reference(scan, number)
372
+ return emit(scan, "(?(#{number})\\#{number}|)", :atom) if scan[:names].empty?
373
+
374
+ name = group_name_for(scan, number)
375
+ return emit(scan, "(?(<#{name}>)\\k<#{name}>|)", :atom)
376
+ end
377
+
378
+ octal = digits.match(/\A[0-7]{1,3}/)&.to_s
379
+ octal = octal[0, 2] if octal && octal.length == 3 && octal.to_i(8) > 255
380
+ if octal.nil?
381
+ # `8` and `9` stand for themselves; the digits after them too.
382
+ return emit(scan, digits, :atom)
383
+ end
384
+
385
+ emit(scan, format('\\x%02X', octal.to_i(8)) + digits[octal.length..], :atom)
386
+ end
387
+
388
+ # `\uXXXX`: a code unit, which Ruby reads as a code point. A surrogate
389
+ # pair (ECMA-262 without the `u` flag matches the two units of one
390
+ # character) is joined into the character it encodes; a lone
391
+ # surrogate is no character any string carries, so it matches
392
+ # nothing. A `\u` that spells no code unit is an identity escape.
393
+ # @return [void]
394
+ def copy_unicode_escape(scan)
395
+ high = hex_code_unit(scan, scan[:index])
396
+ return emit(scan, 'u', :atom) unless high
397
+
398
+ scan[:index] += 4
399
+ low = nil
400
+ if high.between?(0xD800, 0xDBFF) && scan[:chars][scan[:index]] == '\\' && scan[:chars][scan[:index] + 1] == 'u'
401
+ low = hex_code_unit(scan, scan[:index] + 2)
402
+ low = nil unless low&.between?(0xDC00, 0xDFFF)
403
+ end
404
+ if low
405
+ scan[:index] += 6
406
+ return emit(scan, format('\\u{%X}', 0x10000 + ((high - 0xD800) << 10) + (low - 0xDC00)), :atom)
407
+ end
408
+ return emit(scan, ECMA_EMPTY_CLASS, :atom) if high.between?(0xD800, 0xDFFF)
409
+
410
+ emit(scan, format('\\u%04X', high), :atom)
411
+ end
412
+
413
+ # @return [Integer, nil] the four hex digits at an index, as a number
414
+ def hex_code_unit(scan, index)
415
+ text = scan[:chars][index, 4]&.join.to_s
416
+ text.match?(/\A\h{4}\z/) ? text.to_i(16) : nil
417
+ end
418
+
419
+ # `\k<name>`: a named back-reference where the pattern declares named
420
+ # groups (with the same empty-match rule as a numbered one); with none
421
+ # declared it is an identity escape for "k" (Annex B.1.2).
422
+ # @return [void]
423
+ def copy_named_reference(scan, in_class:)
424
+ return emit(scan, 'k', :atom) if scan[:names].empty? || in_class
425
+
426
+ name = group_name_at(scan[:chars], scan[:index])
427
+ unless name && scan[:names].include?(name)
428
+ raise SyntaxError,
429
+ "invalid named reference at index #{scan[:index] - 2}"
430
+ end
431
+
432
+ scan[:index] += name.length + 2
433
+ reject_repeated_reference(scan, scan[:order].index(name) + 1)
434
+ emit(scan, "(?(<#{name}>)\\k<#{name}>|)", :atom)
435
+ end
436
+
437
+ # A back-reference to a group a quantifier may repeat reads one way in
438
+ # ECMA-262 (cleared at each iteration) and another in Ruby (kept), and
439
+ # the difference decides instances either way round, so the pattern is
440
+ # refused rather than answered.
441
+ # @param number [Integer] the group the reference names
442
+ # @return [void]
443
+ # @raise [Untranslatable]
444
+ def reject_repeated_reference(scan, number)
445
+ return unless scan[:repeated].include?(number)
446
+
447
+ raise Untranslatable,
448
+ "a back-reference to group #{number}, which a quantifier repeats, cannot be evaluated faithfully " \
449
+ "(ECMA-262 clears the group's capture at each repetition and Ruby keeps it)"
450
+ end
451
+
452
+ # The capturing groups a quantifier may repeat: those inside (or being)
453
+ # a group followed by a quantifier that allows a second iteration.
454
+ # `?` and `{0,1}` allow only one, so nothing is ever cleared between
455
+ # iterations there and the existing "did the group participate"
456
+ # conditional already reads the way ECMA-262 does. Read in one pass,
457
+ # skipping escapes and character classes, so a `(` written as a
458
+ # literal opens nothing.
459
+ # @param chars [Array<String>] the pattern
460
+ # @return [Array<Integer>] the group numbers
461
+ def repeated_capture_groups(chars)
462
+ state = { open: [], repeated: [], number: 0, index: 0, in_class: false }
463
+ while state[:index] < chars.length
464
+ char = chars[state[:index]]
465
+ state[:index] += 1
466
+ next state[:index] += 1 if char == '\\'
467
+ next state[:in_class] = true if char == '[' && !state[:in_class]
468
+ next state[:in_class] = false if char == ']' && state[:in_class]
469
+ next if state[:in_class]
470
+
471
+ open_repeat_group(state, chars) if char == '('
472
+ close_repeat_group(state, chars) if char == ')'
473
+ end
474
+ state[:repeated]
475
+ end
476
+
477
+ # @return [void]
478
+ def open_repeat_group(state, chars)
479
+ capturing = chars[state[:index]] != '?' || group_name_at(chars, state[:index] + 1).to_s != ''
480
+ number = capturing ? (state[:number] += 1) : nil
481
+ state[:open] << { number: number, inner: [] }
482
+ end
483
+
484
+ # @return [void]
485
+ def close_repeat_group(state, chars)
486
+ group = state[:open].pop
487
+ return unless group
488
+
489
+ members = group[:inner] + [group[:number]].compact
490
+ state[:repeated].concat(members) if quantifier_at?(chars, state[:index])
491
+ parent = state[:open].last
492
+ parent ? parent[:inner].concat(members) : nil
493
+ end
494
+
495
+ # @return [Boolean] whether the quantifier at an index admits a second
496
+ # iteration, which is when ECMA-262's per-iteration clearing of the
497
+ # captures inside it can be seen at all
498
+ def quantifier_at?(chars, index)
499
+ char = chars[index]
500
+ return true if ['*', '+'].include?(char)
501
+ return false unless char == '{'
502
+
503
+ bounds = chars[index..].join[/\A\{(\d+)(,(\d*))?\}/, 0]
504
+ return false unless bounds
505
+
506
+ repeated_bounds?(Regexp.last_match(1).to_i, Regexp.last_match(2), Regexp.last_match(3))
507
+ end
508
+
509
+ # @param least [Integer] the `{n` of the quantifier
510
+ # @param comma [String, nil] its `,`, when it has one
511
+ # @param most [String, nil] its `m`, when it has one
512
+ # @return [Boolean] whether it admits two iterations
513
+ def repeated_bounds?(least, comma, most)
514
+ return least >= 2 if comma.nil?
515
+ return true if most.nil? || most.empty?
516
+
517
+ most.to_i >= 2
518
+ end
519
+
520
+ # Copy a character class, which ECMA-262 and Ruby read differently: an
521
+ # empty class is legal there and matches nothing, and `[` and `&`
522
+ # inside a class are literals rather than the openers of Ruby's nested
523
+ # classes and set intersection.
524
+ # @return [void]
525
+ def copy_character_class(scan)
526
+ chars = scan[:chars]
527
+ opening = scan[:index]
528
+ negated = chars[opening + 1] == '^'
529
+ scan[:index] = opening + (negated ? 2 : 1)
530
+ outer = scan[:out]
531
+ scan[:out] = +''
532
+ while scan[:index] < chars.length && chars[scan[:index]] != ']'
533
+ note_translation_progress(scan)
534
+ if chars[scan[:index]] == '\\'
535
+ copy_escape(scan, in_class: true)
536
+ else
537
+ emit(scan, class_member(chars[scan[:index]]), :atom)
538
+ scan[:index] += 1
539
+ end
540
+ end
541
+ body = scan[:out]
542
+ scan[:out] = outer
543
+ # Unterminated: no expression in either dialect.
544
+ raise SyntaxError, "unterminated character class at index #{opening}" if scan[:index] >= chars.length
545
+
546
+ scan[:index] += 1
547
+ emit(scan, class_source(body, negated), :atom)
548
+ end
549
+
550
+ # @return [String] the Ruby spelling of one unescaped class member
551
+ def class_member(char)
552
+ # Ruby reads a nested `[` as another class and `&&` as intersection;
553
+ # ECMA-262 has neither, so both are literals there.
554
+ ['[', '&'].include?(char) ? "\\#{char}" : char
555
+ end
556
+
557
+ # @param body [String] the translated members
558
+ # @param negated [Boolean] whether the class was written with a `^`
559
+ # @return [String] the Ruby character class
560
+ def class_source(body, negated)
561
+ return negated ? ECMA_ANY_CLASS : ECMA_EMPTY_CLASS if body.empty?
562
+
563
+ negated ? "[^#{body}]" : "[#{body}]"
564
+ end
565
+ end
566
+ end
567
+ end