regexp_parser 2.11.3 → 2.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/LICENSE +1 -1
- data/lib/regexp_parser/expression/base.rb +50 -14
- data/lib/regexp_parser/expression/classes/character_set/range.rb +1 -1
- data/lib/regexp_parser/expression/classes/character_set.rb +1 -1
- data/lib/regexp_parser/expression/classes/unicode_property.rb +1 -1
- data/lib/regexp_parser/expression/methods/match_length.rb +15 -7
- data/lib/regexp_parser/expression/methods/tests.rb +20 -4
- data/lib/regexp_parser/expression/methods/traverse.rb +51 -13
- data/lib/regexp_parser/expression/quantifier.rb +1 -1
- data/lib/regexp_parser/expression/shared.rb +24 -22
- data/lib/regexp_parser/expression/subexpression.rb +2 -2
- data/lib/regexp_parser/lexer.rb +21 -22
- data/lib/regexp_parser/parser.rb +12 -36
- data/lib/regexp_parser/scanner/properties/long.csv +13 -0
- data/lib/regexp_parser/scanner/properties/short.csv +4 -0
- data/lib/regexp_parser/scanner.rb +184 -236
- data/lib/regexp_parser/syntax/token/unicode_property.rb +71 -26
- data/lib/regexp_parser/syntax/version_lookup.rb +25 -10
- data/lib/regexp_parser/syntax/versions/4.0.0.rb +4 -0
- data/lib/regexp_parser/version.rb +1 -1
- data/regexp_parser.gemspec +3 -9
- metadata +3 -9
- data/Gemfile +0 -17
- data/Rakefile +0 -25
- data/lib/regexp_parser/scanner/char_type.rl +0 -28
- data/lib/regexp_parser/scanner/property.rl +0 -30
- data/lib/regexp_parser/scanner/scanner.rl +0 -864
|
@@ -3,7 +3,10 @@
|
|
|
3
3
|
module Regexp::Syntax
|
|
4
4
|
module Token
|
|
5
5
|
module UnicodeProperty
|
|
6
|
-
|
|
6
|
+
def self.all(name)
|
|
7
|
+
constants.grep(/#{name}/).flat_map(&method(:const_get)).freeze
|
|
8
|
+
end
|
|
9
|
+
private_class_method :all
|
|
7
10
|
|
|
8
11
|
CharType_V1_9_0 = %i[alnum alpha ascii blank cntrl digit graph
|
|
9
12
|
lower print punct space upper word xdigit].freeze
|
|
@@ -63,9 +66,11 @@ module Regexp::Syntax
|
|
|
63
66
|
|
|
64
67
|
Age_V3_2_0 = %i[age=14.0 age=15.0].freeze
|
|
65
68
|
|
|
66
|
-
Age_V3_5_0 = %i[age=15.1]
|
|
69
|
+
Age_V3_5_0 = %i[age=15.1].freeze
|
|
70
|
+
|
|
71
|
+
Age_V4_0_0 = %i[age=16.0 age=17.0].freeze
|
|
67
72
|
|
|
68
|
-
Age = all
|
|
73
|
+
Age = all(:Age_V)
|
|
69
74
|
|
|
70
75
|
Derived_V1_9_0 = %i[
|
|
71
76
|
ascii_hex_digit
|
|
@@ -138,9 +143,13 @@ module Regexp::Syntax
|
|
|
138
143
|
id_compat_math_continue
|
|
139
144
|
id_compat_math_start
|
|
140
145
|
ids_unary_operator
|
|
146
|
+
].freeze
|
|
147
|
+
|
|
148
|
+
Derived_V4_0_0 = %i[
|
|
149
|
+
modifier_combining_mark
|
|
141
150
|
]
|
|
142
151
|
|
|
143
|
-
Derived = all
|
|
152
|
+
Derived = all(:Derived_V)
|
|
144
153
|
|
|
145
154
|
Script_V1_9_0 = %i[
|
|
146
155
|
arabic
|
|
@@ -339,7 +348,21 @@ module Regexp::Syntax
|
|
|
339
348
|
vithkuqi
|
|
340
349
|
].freeze
|
|
341
350
|
|
|
342
|
-
|
|
351
|
+
Script_V4_0_0 = %i[
|
|
352
|
+
beria_erfe
|
|
353
|
+
garay
|
|
354
|
+
gurung_khema
|
|
355
|
+
kirat_rai
|
|
356
|
+
ol_onal
|
|
357
|
+
sidetic
|
|
358
|
+
sunuwar
|
|
359
|
+
tai_yo
|
|
360
|
+
todhri
|
|
361
|
+
tolong_siki
|
|
362
|
+
tulu_tigalari
|
|
363
|
+
].freeze
|
|
364
|
+
|
|
365
|
+
Script = all(:Script_V)
|
|
343
366
|
|
|
344
367
|
UnicodeBlock_V1_9_0 = %i[
|
|
345
368
|
in_alphabetic_presentation_forms
|
|
@@ -701,9 +724,30 @@ module Regexp::Syntax
|
|
|
701
724
|
|
|
702
725
|
UnicodeBlock_V3_5_0 = %i[
|
|
703
726
|
in_cjk_unified_ideographs_extension_i
|
|
704
|
-
]
|
|
727
|
+
].freeze
|
|
728
|
+
|
|
729
|
+
UnicodeBlock_V4_0_0 = %i[
|
|
730
|
+
in_beria_erfe
|
|
731
|
+
in_cjk_unified_ideographs_extension_j
|
|
732
|
+
in_egyptian_hieroglyphs_extended_a
|
|
733
|
+
in_garay
|
|
734
|
+
in_gurung_khema
|
|
735
|
+
in_kirat_rai
|
|
736
|
+
in_miscellaneous_symbols_supplement
|
|
737
|
+
in_myanmar_extended_c
|
|
738
|
+
in_ol_onal
|
|
739
|
+
in_sharada_supplement
|
|
740
|
+
in_sidetic
|
|
741
|
+
in_sunuwar
|
|
742
|
+
in_symbols_for_legacy_computing_supplement
|
|
743
|
+
in_tai_yo
|
|
744
|
+
in_tangut_components_supplement
|
|
745
|
+
in_todhri
|
|
746
|
+
in_tolong_siki
|
|
747
|
+
in_tulu_tigalari
|
|
748
|
+
].freeze
|
|
705
749
|
|
|
706
|
-
UnicodeBlock = all
|
|
750
|
+
UnicodeBlock = all(:UnicodeBlock_V)
|
|
707
751
|
|
|
708
752
|
Emoji_V2_5_0 = %i[
|
|
709
753
|
emoji
|
|
@@ -733,25 +777,26 @@ module Regexp::Syntax
|
|
|
733
777
|
grapheme_cluster_break=zwj
|
|
734
778
|
].freeze
|
|
735
779
|
|
|
736
|
-
Enumerated = all
|
|
737
|
-
|
|
738
|
-
Emoji = all
|
|
739
|
-
|
|
740
|
-
V1_9_0 = Category::All + POSIX + all
|
|
741
|
-
V1_9_3 = all
|
|
742
|
-
V2_0_0 = all
|
|
743
|
-
V2_2_0 = all
|
|
744
|
-
V2_3_0 = all
|
|
745
|
-
V2_4_0 = all
|
|
746
|
-
V2_5_0 = all
|
|
747
|
-
V2_6_0 = all
|
|
748
|
-
V2_6_2 = all
|
|
749
|
-
V2_6_3 = all
|
|
750
|
-
V3_1_0 = all
|
|
751
|
-
V3_2_0 = all
|
|
752
|
-
V3_5_0 = all
|
|
753
|
-
|
|
754
|
-
|
|
780
|
+
Enumerated = all(:Enumerated_V)
|
|
781
|
+
|
|
782
|
+
Emoji = all(:Emoji_V)
|
|
783
|
+
|
|
784
|
+
V1_9_0 = Category::All + POSIX + all(:V1_9_0)
|
|
785
|
+
V1_9_3 = all(:V1_9_3)
|
|
786
|
+
V2_0_0 = all(:V2_0_0)
|
|
787
|
+
V2_2_0 = all(:V2_2_0)
|
|
788
|
+
V2_3_0 = all(:V2_3_0)
|
|
789
|
+
V2_4_0 = all(:V2_4_0)
|
|
790
|
+
V2_5_0 = all(:V2_5_0)
|
|
791
|
+
V2_6_0 = all(:V2_6_0)
|
|
792
|
+
V2_6_2 = all(:V2_6_2)
|
|
793
|
+
V2_6_3 = all(:V2_6_3)
|
|
794
|
+
V3_1_0 = all(:V3_1_0)
|
|
795
|
+
V3_2_0 = all(:V3_2_0)
|
|
796
|
+
V3_5_0 = all(:V3_5_0)
|
|
797
|
+
V4_0_0 = all(:V4_0_0)
|
|
798
|
+
|
|
799
|
+
All = all(/^V\d+_\d+_\d+$/)
|
|
755
800
|
|
|
756
801
|
Type = :property
|
|
757
802
|
NonType = :nonproperty
|
|
@@ -22,7 +22,11 @@ module Regexp::Syntax
|
|
|
22
22
|
# Returns the syntax specification class for the given syntax
|
|
23
23
|
# version name. The special names 'any' and '*' return Syntax::Any.
|
|
24
24
|
def for(name)
|
|
25
|
-
|
|
25
|
+
return Regexp::Syntax::Any if ['*', 'any'].include?(name.to_s)
|
|
26
|
+
|
|
27
|
+
name =~ VERSION_REGEXP || raise(InvalidVersionNameError, name)
|
|
28
|
+
version_const_name = "V#{name.to_s.scan(/\d+/).join('_')}"
|
|
29
|
+
const_get(version_const_name)
|
|
26
30
|
end
|
|
27
31
|
|
|
28
32
|
def new(name)
|
|
@@ -31,25 +35,36 @@ module Regexp::Syntax
|
|
|
31
35
|
self.for(name)
|
|
32
36
|
end
|
|
33
37
|
|
|
34
|
-
def
|
|
35
|
-
|
|
38
|
+
def version_class(name)
|
|
39
|
+
warn 'Regexp::Syntax.version_class is deprecated in favor of Regexp::Syntax.for. '\
|
|
40
|
+
'It will be removed in regexp_parser v3.0.0.'
|
|
41
|
+
self.for(name)
|
|
36
42
|
end
|
|
37
43
|
|
|
38
|
-
def
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
version =~ VERSION_REGEXP || raise(InvalidVersionNameError, version)
|
|
42
|
-
version_const_name = "V#{version.to_s.scan(/\d+/).join('_')}"
|
|
43
|
-
const_get(version_const_name) || raise(UnknownSyntaxNameError, version)
|
|
44
|
+
def supported?(name)
|
|
45
|
+
name =~ VERSION_REGEXP && comparable(name) >= comparable('1.8.6')
|
|
44
46
|
end
|
|
45
47
|
|
|
46
48
|
def const_missing(const_name)
|
|
47
49
|
if const_name =~ VERSION_CONST_REGEXP
|
|
48
|
-
return
|
|
50
|
+
return set_fallback_version_class(const_name) ||
|
|
51
|
+
raise(UnknownSyntaxNameError, const_name)
|
|
49
52
|
end
|
|
50
53
|
super
|
|
51
54
|
end
|
|
52
55
|
|
|
56
|
+
def set_fallback_version_class(const_name)
|
|
57
|
+
return unless (klass = fallback_version_class(const_name))
|
|
58
|
+
|
|
59
|
+
if constants.count { |c| c =~ VERSION_REGEXP } > 1000
|
|
60
|
+
raise Regexp::Syntax::SyntaxError,
|
|
61
|
+
"Unexpected high number of syntax versions defined. "\
|
|
62
|
+
"Do not accept unfiltered user input as Regexp::Syntax version."
|
|
63
|
+
end
|
|
64
|
+
const_set(const_name, klass)
|
|
65
|
+
klass
|
|
66
|
+
end
|
|
67
|
+
|
|
53
68
|
def fallback_version_class(version)
|
|
54
69
|
sorted = (specified_versions + [version]).sort_by { |ver| comparable(ver) }
|
|
55
70
|
index = sorted.index(version)
|
data/regexp_parser.gemspec
CHANGED
|
@@ -1,8 +1,6 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
require 'regexp_parser/version'
|
|
3
|
+
require_relative 'lib/regexp_parser/version'
|
|
6
4
|
|
|
7
5
|
Gem::Specification.new do |spec|
|
|
8
6
|
spec.name = 'regexp_parser'
|
|
@@ -14,8 +12,6 @@ Gem::Specification.new do |spec|
|
|
|
14
12
|
|
|
15
13
|
spec.metadata['bug_tracker_uri'] = "#{spec.homepage}/issues"
|
|
16
14
|
spec.metadata['changelog_uri'] = "#{spec.homepage}/blob/master/CHANGELOG.md"
|
|
17
|
-
spec.metadata['homepage_uri'] = spec.homepage
|
|
18
|
-
spec.metadata['source_code_uri'] = spec.homepage
|
|
19
15
|
spec.metadata['wiki_uri'] = "#{spec.homepage}/wiki"
|
|
20
16
|
|
|
21
17
|
spec.metadata['rubygems_mfa_required'] = 'true'
|
|
@@ -27,10 +23,8 @@ Gem::Specification.new do |spec|
|
|
|
27
23
|
|
|
28
24
|
spec.require_paths = ['lib']
|
|
29
25
|
|
|
30
|
-
spec.files = Dir.glob('lib/**/*.{csv,rb
|
|
31
|
-
%w[
|
|
32
|
-
|
|
33
|
-
spec.platform = Gem::Platform::RUBY
|
|
26
|
+
spec.files = Dir.glob('lib/**/*.{csv,rb}') +
|
|
27
|
+
%w[LICENSE regexp_parser.gemspec]
|
|
34
28
|
|
|
35
29
|
spec.required_ruby_version = '>= 2.0.0'
|
|
36
30
|
end
|
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: regexp_parser
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 2.
|
|
4
|
+
version: 2.13.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Ammar Ali
|
|
@@ -18,9 +18,7 @@ executables: []
|
|
|
18
18
|
extensions: []
|
|
19
19
|
extra_rdoc_files: []
|
|
20
20
|
files:
|
|
21
|
-
- Gemfile
|
|
22
21
|
- LICENSE
|
|
23
|
-
- Rakefile
|
|
24
22
|
- lib/regexp_parser.rb
|
|
25
23
|
- lib/regexp_parser/error.rb
|
|
26
24
|
- lib/regexp_parser/expression.rb
|
|
@@ -63,14 +61,11 @@ files:
|
|
|
63
61
|
- lib/regexp_parser/lexer.rb
|
|
64
62
|
- lib/regexp_parser/parser.rb
|
|
65
63
|
- lib/regexp_parser/scanner.rb
|
|
66
|
-
- lib/regexp_parser/scanner/char_type.rl
|
|
67
64
|
- lib/regexp_parser/scanner/errors/premature_end_error.rb
|
|
68
65
|
- lib/regexp_parser/scanner/errors/scanner_error.rb
|
|
69
66
|
- lib/regexp_parser/scanner/errors/validation_error.rb
|
|
70
67
|
- lib/regexp_parser/scanner/properties/long.csv
|
|
71
68
|
- lib/regexp_parser/scanner/properties/short.csv
|
|
72
|
-
- lib/regexp_parser/scanner/property.rl
|
|
73
|
-
- lib/regexp_parser/scanner/scanner.rl
|
|
74
69
|
- lib/regexp_parser/syntax.rb
|
|
75
70
|
- lib/regexp_parser/syntax/any.rb
|
|
76
71
|
- lib/regexp_parser/syntax/base.rb
|
|
@@ -106,6 +101,7 @@ files:
|
|
|
106
101
|
- lib/regexp_parser/syntax/versions/3.1.0.rb
|
|
107
102
|
- lib/regexp_parser/syntax/versions/3.2.0.rb
|
|
108
103
|
- lib/regexp_parser/syntax/versions/3.5.0.rb
|
|
104
|
+
- lib/regexp_parser/syntax/versions/4.0.0.rb
|
|
109
105
|
- lib/regexp_parser/token.rb
|
|
110
106
|
- lib/regexp_parser/version.rb
|
|
111
107
|
- regexp_parser.gemspec
|
|
@@ -115,8 +111,6 @@ licenses:
|
|
|
115
111
|
metadata:
|
|
116
112
|
bug_tracker_uri: https://github.com/ammar/regexp_parser/issues
|
|
117
113
|
changelog_uri: https://github.com/ammar/regexp_parser/blob/master/CHANGELOG.md
|
|
118
|
-
homepage_uri: https://github.com/ammar/regexp_parser
|
|
119
|
-
source_code_uri: https://github.com/ammar/regexp_parser
|
|
120
114
|
wiki_uri: https://github.com/ammar/regexp_parser/wiki
|
|
121
115
|
rubygems_mfa_required: 'true'
|
|
122
116
|
rdoc_options: []
|
|
@@ -133,7 +127,7 @@ required_rubygems_version: !ruby/object:Gem::Requirement
|
|
|
133
127
|
- !ruby/object:Gem::Version
|
|
134
128
|
version: '0'
|
|
135
129
|
requirements: []
|
|
136
|
-
rubygems_version:
|
|
130
|
+
rubygems_version: 4.0.20
|
|
137
131
|
specification_version: 4
|
|
138
132
|
summary: Scanner, lexer, parser for ruby's regular expressions
|
|
139
133
|
test_files: []
|
data/Gemfile
DELETED
|
@@ -1,17 +0,0 @@
|
|
|
1
|
-
# frozen_string_literal: true
|
|
2
|
-
|
|
3
|
-
source 'https://rubygems.org'
|
|
4
|
-
|
|
5
|
-
gemspec
|
|
6
|
-
|
|
7
|
-
group :development, :test do
|
|
8
|
-
gem 'leto', '~> 2.1'
|
|
9
|
-
gem 'rake', '~> 13.1'
|
|
10
|
-
gem 'regexp_property_values', '~> 1.5'
|
|
11
|
-
gem 'rspec', '~> 3.10'
|
|
12
|
-
if RUBY_VERSION.to_f >= 2.7
|
|
13
|
-
gem 'benchmark-ips', '~> 2.1'
|
|
14
|
-
gem 'gouteur', '~> 1.1'
|
|
15
|
-
gem 'rubocop', '>= 1.80.2'
|
|
16
|
-
end
|
|
17
|
-
end
|
data/Rakefile
DELETED
|
@@ -1,25 +0,0 @@
|
|
|
1
|
-
# frozen_string_literal: true
|
|
2
|
-
|
|
3
|
-
require 'bundler'
|
|
4
|
-
require 'rubygems'
|
|
5
|
-
require 'rubygems/package_task'
|
|
6
|
-
require 'rake'
|
|
7
|
-
require 'rake/testtask'
|
|
8
|
-
require 'rspec/core/rake_task'
|
|
9
|
-
|
|
10
|
-
Dir['tasks/**/*.rake'].each { |file| load(file) }
|
|
11
|
-
|
|
12
|
-
Bundler::GemHelper.install_tasks
|
|
13
|
-
|
|
14
|
-
RSpec::Core::RakeTask.new(:spec)
|
|
15
|
-
|
|
16
|
-
task :default => [:'test:full']
|
|
17
|
-
|
|
18
|
-
namespace :test do
|
|
19
|
-
task full: [:ragel, :spec]
|
|
20
|
-
end
|
|
21
|
-
|
|
22
|
-
# Add ragel task as a prerequisite for building the gem to ensure that the
|
|
23
|
-
# latest scanner code is generated and included in the build.
|
|
24
|
-
desc "Runs ragel before building the gem"
|
|
25
|
-
task build: :ragel
|
|
@@ -1,28 +0,0 @@
|
|
|
1
|
-
%%{
|
|
2
|
-
machine re_char_type;
|
|
3
|
-
|
|
4
|
-
single_codepoint_char_type = [dDhHsSwW];
|
|
5
|
-
multi_codepoint_char_type = [RX];
|
|
6
|
-
|
|
7
|
-
char_type_char = single_codepoint_char_type | multi_codepoint_char_type;
|
|
8
|
-
|
|
9
|
-
# Char types scanner
|
|
10
|
-
# --------------------------------------------------------------------------
|
|
11
|
-
char_type := |*
|
|
12
|
-
char_type_char {
|
|
13
|
-
case text = copy(data, ts-1, te)
|
|
14
|
-
when '\d'; emit(:type, :digit, text)
|
|
15
|
-
when '\D'; emit(:type, :nondigit, text)
|
|
16
|
-
when '\h'; emit(:type, :hex, text)
|
|
17
|
-
when '\H'; emit(:type, :nonhex, text)
|
|
18
|
-
when '\s'; emit(:type, :space, text)
|
|
19
|
-
when '\S'; emit(:type, :nonspace, text)
|
|
20
|
-
when '\w'; emit(:type, :word, text)
|
|
21
|
-
when '\W'; emit(:type, :nonword, text)
|
|
22
|
-
when '\R'; emit(:type, :linebreak, text)
|
|
23
|
-
when '\X'; emit(:type, :xgrapheme, text)
|
|
24
|
-
end
|
|
25
|
-
fret;
|
|
26
|
-
};
|
|
27
|
-
*|;
|
|
28
|
-
}%%
|
|
@@ -1,30 +0,0 @@
|
|
|
1
|
-
%%{
|
|
2
|
-
machine re_property;
|
|
3
|
-
|
|
4
|
-
property_char = [pP];
|
|
5
|
-
|
|
6
|
-
property_sequence = property_char . '{' . '^'? (alnum|space|[_\-\.=])+ '}';
|
|
7
|
-
|
|
8
|
-
action premature_property_end {
|
|
9
|
-
raise PrematureEndError.new('unicode property')
|
|
10
|
-
}
|
|
11
|
-
|
|
12
|
-
# Unicode properties scanner
|
|
13
|
-
# --------------------------------------------------------------------------
|
|
14
|
-
unicode_property := |*
|
|
15
|
-
|
|
16
|
-
property_sequence < eof(premature_property_end) {
|
|
17
|
-
text = copy(data, ts-1, te)
|
|
18
|
-
type = (text[1] == 'P') ^ (text[3] == '^') ? :nonproperty : :property
|
|
19
|
-
|
|
20
|
-
name = text[3..-2].gsub(/[\^\s_\-]/, '').downcase
|
|
21
|
-
|
|
22
|
-
token = self.class.short_prop_map[name] || self.class.long_prop_map[name]
|
|
23
|
-
raise ValidationError.for(:property, name) unless token
|
|
24
|
-
|
|
25
|
-
self.emit(type, token.to_sym, text)
|
|
26
|
-
|
|
27
|
-
fret;
|
|
28
|
-
};
|
|
29
|
-
*|;
|
|
30
|
-
}%%
|