interscript 2.4.5 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +15 -0
- data/docs/demo/20191118-interscript-demo-cast.gif +0 -0
- data/exe/codemod-imp-to-isc +5 -0
- data/exe/diagnose_parse_failures +47 -0
- data/exe/interscript +2 -1
- data/exe/verify_isc_deep +211 -0
- data/exe/verify_isc_equivalence +128 -0
- data/interscript.gemspec +27 -20
- data/lib/interscript/command.rb +21 -12
- data/lib/interscript/compiler/javascript.rb +34 -35
- data/lib/interscript/compiler/json_ir.rb +214 -0
- data/lib/interscript/compiler/python.rb +354 -0
- data/lib/interscript/compiler/ruby.rb +23 -22
- data/lib/interscript/compiler.rb +20 -3
- data/lib/interscript/detector.rb +14 -7
- data/lib/interscript/dsl/aliases.rb +1 -1
- data/lib/interscript/dsl/document.rb +6 -3
- data/lib/interscript/dsl/group/parallel.rb +1 -1
- data/lib/interscript/dsl/group.rb +18 -9
- data/lib/interscript/dsl/items.rb +28 -16
- data/lib/interscript/dsl/metadata.rb +19 -18
- data/lib/interscript/dsl/stage.rb +1 -1
- data/lib/interscript/dsl/symbol_mm.rb +4 -2
- data/lib/interscript/dsl/tests.rb +1 -1
- data/lib/interscript/dsl.rb +21 -13
- data/lib/interscript/interpreter.rb +32 -21
- data/lib/interscript/isc/codemod.rb +791 -0
- data/lib/interscript/isc/document_builder.rb +354 -0
- data/lib/interscript/isc/generator.rb +191 -0
- data/lib/interscript/isc/grammar/concerns/aliases.rb +29 -0
- data/lib/interscript/isc/grammar/concerns/dependencies.rb +23 -0
- data/lib/interscript/isc/grammar/concerns/items.rb +176 -0
- data/lib/interscript/isc/grammar/concerns/metadata.rb +142 -0
- data/lib/interscript/isc/grammar/concerns/primitives.rb +112 -0
- data/lib/interscript/isc/grammar/concerns/stages.rb +129 -0
- data/lib/interscript/isc/grammar/concerns/system.rb +37 -0
- data/lib/interscript/isc/grammar/concerns/tests.rb +31 -0
- data/lib/interscript/isc/grammar/concerns.rb +18 -0
- data/lib/interscript/isc/grammar/core.rb +23 -0
- data/lib/interscript/isc/grammar/isc.artifact.json +1 -0
- data/lib/interscript/isc/grammar/isc.parg +196 -0
- data/lib/interscript/isc/grammar.rb +12 -0
- data/lib/interscript/isc/items.rb +185 -0
- data/lib/interscript/isc/model/alias.rb +20 -0
- data/lib/interscript/isc/model/constraint.rb +20 -0
- data/lib/interscript/isc/model/dependency.rb +19 -0
- data/lib/interscript/isc/model/document.rb +31 -0
- data/lib/interscript/isc/model/item.rb +50 -0
- data/lib/interscript/isc/model/rule.rb +23 -0
- data/lib/interscript/isc/model/stage.rb +20 -0
- data/lib/interscript/isc/model/stage_item.rb +33 -0
- data/lib/interscript/isc/model/test.rb +21 -0
- data/lib/interscript/isc/model.rb +19 -0
- data/lib/interscript/isc/node_adapter.rb +310 -0
- data/lib/interscript/isc/normalizer.rb +127 -0
- data/lib/interscript/isc/parser.rb +78 -0
- data/lib/interscript/isc/serializer.rb +248 -0
- data/lib/interscript/isc/transform.rb +138 -0
- data/lib/interscript/isc/yaml_bridge.rb +236 -0
- data/lib/interscript/isc.rb +30 -0
- data/lib/interscript/ml/byt5_onnx.rb +102 -0
- data/lib/interscript/ml/imf.rb +182 -0
- data/lib/interscript/ml/model.rb +52 -0
- data/lib/interscript/ml/provisioning.rb +189 -0
- data/lib/interscript/ml/translator.rb +38 -0
- data/lib/interscript/ml/vocab.rb +48 -0
- data/lib/interscript/ml.rb +25 -0
- data/lib/interscript/node/alias_def.rb +5 -5
- data/lib/interscript/node/dependency.rb +7 -7
- data/lib/interscript/node/document.rb +10 -10
- data/lib/interscript/node/group.rb +11 -9
- data/lib/interscript/node/item/alias.rb +14 -10
- data/lib/interscript/node/item/any.rb +14 -7
- data/lib/interscript/node/item/capture.rb +14 -9
- data/lib/interscript/node/item/group.rb +22 -17
- data/lib/interscript/node/item/repeat.rb +3 -3
- data/lib/interscript/node/item/stage.rb +4 -5
- data/lib/interscript/node/item/string.rb +18 -13
- data/lib/interscript/node/item.rb +27 -17
- data/lib/interscript/node/metadata.rb +6 -5
- data/lib/interscript/node/rule/funcall.rb +4 -5
- data/lib/interscript/node/rule/run.rb +4 -5
- data/lib/interscript/node/rule/sub.rb +55 -55
- data/lib/interscript/node/rule.rb +6 -5
- data/lib/interscript/node/stage.rb +11 -14
- data/lib/interscript/node/tests.rb +6 -6
- data/lib/interscript/node.rb +14 -17
- data/lib/interscript/stdlib/functions/rababa_adapter.rb +56 -0
- data/lib/interscript/stdlib/functions/secryst_adapter.rb +37 -0
- data/lib/interscript/stdlib/functions.rb +49 -0
- data/lib/interscript/stdlib.rb +25 -95
- data/lib/interscript/utils/helpers.rb +8 -5
- data/lib/interscript/utils/regexp_converter.rb +143 -144
- data/lib/interscript/version.rb +1 -1
- data/lib/interscript/visualize/json.rb +12 -13
- data/lib/interscript/visualize/nodes.rb +13 -13
- data/lib/interscript/visualize.rb +18 -18
- data/lib/interscript.rb +59 -45
- metadata +88 -31
- data/.github/workflows/assets.yml +0 -91
- data/.github/workflows/rake.yml +0 -52
- data/.github/workflows/release.yml +0 -68
- data/.gitignore +0 -79
- data/.rspec +0 -3
- data/Gemfile +0 -38
- data/Rakefile +0 -139
- data/bin/console +0 -10
- data/bin/interscript +0 -5
- data/bin/maps_analyze_staging +0 -168
- data/bin/maps_debug_compilers +0 -58
- data/bin/maps_debug_ordering +0 -88
- data/bin/maps_debug_ruby_compile +0 -24
- data/bin/maps_debug_step_by_step +0 -44
- data/bin/maps_optimize_order +0 -112
- data/bin/maps_v1_analyze_regexps +0 -45
- data/bin/maps_v1_to_v2 +0 -426
- data/bin/set_version +0 -16
- data/bin/setup +0 -8
- data/requirements.txt +0 -1
|
@@ -1,6 +1,8 @@
|
|
|
1
|
+
# rubocop:disable Style/GlobalVars
|
|
2
|
+
# spec-support globals; deliberate process-level state
|
|
1
3
|
module Interscript::Utils
|
|
2
4
|
module Helpers
|
|
3
|
-
def document name=nil, &block
|
|
5
|
+
def document name = nil, &block
|
|
4
6
|
$example_id ||= 0
|
|
5
7
|
$example_id += 1
|
|
6
8
|
name ||= "example-#{$example_id}"
|
|
@@ -20,14 +22,14 @@ module Interscript::Utils
|
|
|
20
22
|
end
|
|
21
23
|
|
|
22
24
|
class Interscript::Node::Document
|
|
23
|
-
def call(str, stage
|
|
24
|
-
compiler.(self).(str, stage, **kwargs)
|
|
25
|
+
def call(str, stage = :main, compiler = $compiler || Interscript::Interpreter, **kwargs)
|
|
26
|
+
compiler.call(self).call(str, stage, **kwargs)
|
|
25
27
|
end
|
|
26
28
|
end
|
|
27
29
|
|
|
28
30
|
module Interscript::DSL
|
|
29
31
|
class << self
|
|
30
|
-
|
|
32
|
+
alias_method :original_parse, :parse
|
|
31
33
|
def parse(map_name, **kwargs)
|
|
32
34
|
if $documents && $documents[map_name]
|
|
33
35
|
$documents[map_name]
|
|
@@ -36,4 +38,5 @@ module Interscript::DSL
|
|
|
36
38
|
end
|
|
37
39
|
end
|
|
38
40
|
end
|
|
39
|
-
end
|
|
41
|
+
end
|
|
42
|
+
# rubocop:enable Style/GlobalVars
|
|
@@ -1,81 +1,80 @@
|
|
|
1
|
-
require
|
|
2
|
-
|
|
1
|
+
require "regexp_parser"
|
|
3
2
|
|
|
4
3
|
def process(node)
|
|
5
4
|
children = if node.respond_to?(:expressions) && node.expressions
|
|
6
|
-
|
|
7
|
-
|
|
5
|
+
node.expressions.map.each { |expr| process(expr) }
|
|
6
|
+
end
|
|
8
7
|
# puts node.inspect
|
|
9
8
|
out = case node
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
9
|
+
when Regexp::Expression::Root
|
|
10
|
+
children
|
|
11
|
+
when Regexp::Expression::Assertion::Lookbehind
|
|
12
|
+
[:lookbehind_start, children, :lookbehind_stop]
|
|
13
|
+
when Regexp::Expression::Assertion::NegativeLookbehind
|
|
14
|
+
[:negative_lookbehind_start, children, :negative_lookbehind_stop]
|
|
15
|
+
when Regexp::Expression::Assertion::Lookahead
|
|
16
|
+
[:lookahead_start, children, :lookahead_stop]
|
|
17
|
+
when Regexp::Expression::Assertion::NegativeLookahead
|
|
18
|
+
[:negative_lookahead_start, children, :negative_lookahead_stop]
|
|
19
|
+
when Regexp::Expression::Group::Capture
|
|
20
|
+
[:capture_start, children, :capture_stop]
|
|
21
|
+
when Regexp::Expression::CharacterSet
|
|
22
|
+
# puts children.inspect
|
|
23
|
+
if children.flatten.include?(:range_start) # or children.size > 1
|
|
24
|
+
[:characterset_start, :array_start, children, :array_stop, :characterset_stop]
|
|
25
|
+
else
|
|
26
|
+
[:characterset_start, children, :characterset_stop]
|
|
27
|
+
end
|
|
28
|
+
when Regexp::Expression::Alternation
|
|
29
|
+
[:alternation_start, children, :alternation_stop]
|
|
30
|
+
when Regexp::Expression::Alternative
|
|
31
|
+
[:alternative_start, children, :alternative_stop]
|
|
32
|
+
when Regexp::Expression::CharacterSet::Range
|
|
33
|
+
lit1 = node.expressions[0].text
|
|
34
|
+
lit2 = node.expressions[1].text
|
|
35
|
+
[:range_start, lit1, :range_mid, lit2, :range_stop]
|
|
36
|
+
when Regexp::Expression::Anchor::WordBoundary
|
|
37
|
+
:boundary
|
|
38
|
+
when Regexp::Expression::Anchor::NonWordBoundary
|
|
39
|
+
:non_word_boundary
|
|
40
|
+
when Regexp::Expression::EscapeSequence::Backspace
|
|
41
|
+
:boundary # most probably boundary
|
|
42
|
+
when Regexp::Expression::CharacterType::Space
|
|
43
|
+
:space
|
|
44
|
+
when Regexp::Expression::Anchor::BeginningOfLine
|
|
45
|
+
:line_start
|
|
46
|
+
when Regexp::Expression::Anchor::EndOfLine
|
|
47
|
+
:line_end
|
|
48
|
+
when Regexp::Expression::CharacterType::Any
|
|
49
|
+
:any_character
|
|
50
|
+
when Regexp::Expression::Literal
|
|
51
|
+
node.text
|
|
52
|
+
when Regexp::Expression::EscapeSequence::Literal
|
|
53
|
+
node.text
|
|
54
|
+
when Regexp::Expression::EscapeSequence::Codepoint
|
|
55
|
+
node.text
|
|
56
|
+
when Regexp::Expression::PosixClass
|
|
57
|
+
"[" + node.text + "]"
|
|
58
|
+
when Regexp::Expression::UnicodeProperty::Script
|
|
59
|
+
node.text
|
|
60
|
+
when Regexp::Expression::Backreference::Number # why is there a space before after node.number?
|
|
61
|
+
[:backref_num_start, node.number, :backref_num_stop]
|
|
62
|
+
else
|
|
63
|
+
out = [:missing, node.class]
|
|
65
64
|
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
65
|
+
out << children if node.respond_to? :expressions
|
|
66
|
+
if node.respond_to?(:quantifier) && node.quantifier
|
|
67
|
+
# TODO add quantifier support
|
|
68
|
+
pp node
|
|
69
|
+
# out << process(node.quantifier)
|
|
70
|
+
end
|
|
71
|
+
out
|
|
72
|
+
end
|
|
74
73
|
if node.respond_to?(:quantifier) && node.quantifier&.token.to_s == "interval" && node.quantifier.max == node.quantifier.min
|
|
75
74
|
out = [out] * node.quantifier.max
|
|
76
75
|
elsif node.respond_to?(:quantifier) && node.quantifier
|
|
77
76
|
qname = node.quantifier.token.to_s
|
|
78
|
-
out = ["#{qname}_start"
|
|
77
|
+
out = [:"#{qname}_start", [out], :"#{qname}_stop"]
|
|
79
78
|
end
|
|
80
79
|
out
|
|
81
80
|
end
|
|
@@ -83,76 +82,76 @@ end
|
|
|
83
82
|
def process_root(node)
|
|
84
83
|
node2 = node.dup
|
|
85
84
|
root = {}
|
|
86
|
-
if before = node.select { |x| x[0] == :lookbehind_start }
|
|
85
|
+
if (before = node.select { |x| x[0] == :lookbehind_start })
|
|
87
86
|
# root[:before] = before[1]
|
|
88
87
|
# node2.delete(before)
|
|
89
88
|
if before.size == 1
|
|
90
89
|
root[:before] = before[0][1]
|
|
91
90
|
node2.delete(before[0])
|
|
92
|
-
elsif before.size >1
|
|
91
|
+
elsif before.size > 1
|
|
93
92
|
# pp not_before
|
|
94
93
|
|
|
95
94
|
a = [:alternation_start]
|
|
96
|
-
a << before.map{|x| [:alternative_start, x[1], :alternative_stop] }
|
|
95
|
+
a << before.map { |x| [:alternative_start, x[1], :alternative_stop] }
|
|
97
96
|
a << [:alternation_stop]
|
|
98
97
|
root[:before] = a
|
|
99
98
|
# pp root[:not_before]
|
|
100
|
-
before.each{|n| node2.delete(n)}
|
|
99
|
+
before.each { |n| node2.delete(n) }
|
|
101
100
|
end
|
|
102
101
|
|
|
103
102
|
end
|
|
104
|
-
if not_before = node.select { |x| x[0] == :negative_lookbehind_start }
|
|
103
|
+
if (not_before = node.select { |x| x[0] == :negative_lookbehind_start })
|
|
105
104
|
# root[:not_before] = not_before[1]
|
|
106
105
|
# node2.delete(not_before)
|
|
107
106
|
|
|
108
107
|
if not_before.size == 1
|
|
109
108
|
root[:not_before] = not_before[0][1]
|
|
110
109
|
node2.delete(not_before[0])
|
|
111
|
-
elsif not_before.size >1
|
|
110
|
+
elsif not_before.size > 1
|
|
112
111
|
# pp not_before
|
|
113
112
|
|
|
114
113
|
a = [:alternation_start]
|
|
115
|
-
a << not_before.map{|x| [:alternative_start, x[1], :alternative_stop] }
|
|
114
|
+
a << not_before.map { |x| [:alternative_start, x[1], :alternative_stop] }
|
|
116
115
|
a << [:alternation_stop]
|
|
117
116
|
root[:not_before] = a
|
|
118
117
|
# pp root[:not_before]
|
|
119
|
-
not_before.each{|n| node2.delete(n)}
|
|
118
|
+
not_before.each { |n| node2.delete(n) }
|
|
120
119
|
end
|
|
121
120
|
end
|
|
122
|
-
if after = node.select { |x| x[0] == :lookahead_start }
|
|
121
|
+
if (after = node.select { |x| x[0] == :lookahead_start })
|
|
123
122
|
# root[:after] = after[1]
|
|
124
123
|
# node2.delete(after)
|
|
125
124
|
|
|
126
125
|
if after.size == 1
|
|
127
126
|
root[:after] = after[0][1]
|
|
128
127
|
node2.delete(after[0])
|
|
129
|
-
elsif after.size >1
|
|
128
|
+
elsif after.size > 1
|
|
130
129
|
# pp not_before
|
|
131
130
|
|
|
132
131
|
a = [:alternation_start]
|
|
133
|
-
a << after.map{|x| [:alternative_start, x[1], :alternative_stop] }
|
|
132
|
+
a << after.map { |x| [:alternative_start, x[1], :alternative_stop] }
|
|
134
133
|
a << [:alternation_stop]
|
|
135
134
|
root[:after] = a
|
|
136
135
|
# pp root[:not_before]
|
|
137
|
-
after.each{|n| node2.delete(n)}
|
|
136
|
+
after.each { |n| node2.delete(n) }
|
|
138
137
|
end
|
|
139
138
|
|
|
140
139
|
end
|
|
141
|
-
if not_after = node.select { |x| x[0] == :negative_lookahead_start }
|
|
140
|
+
if (not_after = node.select { |x| x[0] == :negative_lookahead_start })
|
|
142
141
|
# root[:not_after] = not_after[1]
|
|
143
142
|
# node2.delete(not_after)
|
|
144
143
|
if not_after.size == 1
|
|
145
144
|
root[:not_after] = not_after[0][1]
|
|
146
145
|
node2.delete(not_after[0])
|
|
147
|
-
elsif not_after.size >1
|
|
146
|
+
elsif not_after.size > 1
|
|
148
147
|
# pp not_after
|
|
149
148
|
|
|
150
149
|
a = [:alternation_start]
|
|
151
|
-
a << not_after.map{|x| [:alternative_start, x[1], :alternative_stop] }
|
|
150
|
+
a << not_after.map { |x| [:alternative_start, x[1], :alternative_stop] }
|
|
152
151
|
a << [:alternation_stop]
|
|
153
152
|
root[:not_after] = a
|
|
154
153
|
# pp root[:not_after]
|
|
155
|
-
not_after.each{|n| node2.delete(n)}
|
|
154
|
+
not_after.each { |n| node2.delete(n) }
|
|
156
155
|
end
|
|
157
156
|
|
|
158
157
|
end
|
|
@@ -163,57 +162,57 @@ end
|
|
|
163
162
|
def stringify(node)
|
|
164
163
|
tokens = node.flatten
|
|
165
164
|
subs = {
|
|
166
|
-
characterset_start:
|
|
167
|
-
characterset_stop:
|
|
168
|
-
array_start:
|
|
169
|
-
array_stop:
|
|
170
|
-
capture_start:
|
|
171
|
-
capture_stop:
|
|
172
|
-
zero_or_one_start:
|
|
173
|
-
zero_or_one_stop:
|
|
174
|
-
zero_or_more_start:
|
|
175
|
-
zero_or_more_stop:
|
|
176
|
-
one_or_more_start:
|
|
177
|
-
one_or_more_stop:
|
|
178
|
-
alternation_start:
|
|
179
|
-
alternation_stop:
|
|
180
|
-
alternative_start:
|
|
181
|
-
alternative_stop:
|
|
182
|
-
boundary:
|
|
183
|
-
non_word_boundary:
|
|
184
|
-
space:
|
|
185
|
-
line_start:
|
|
186
|
-
line_end:
|
|
187
|
-
any_character:
|
|
188
|
-
range_start:
|
|
189
|
-
range_mid:
|
|
190
|
-
range_stop:
|
|
191
|
-
backref_num_start:
|
|
192
|
-
backref_num_stop:
|
|
165
|
+
characterset_start: "any(",
|
|
166
|
+
characterset_stop: ")",
|
|
167
|
+
array_start: "[",
|
|
168
|
+
array_stop: "]",
|
|
169
|
+
capture_start: "capture(",
|
|
170
|
+
capture_stop: ")",
|
|
171
|
+
zero_or_one_start: "maybe(",
|
|
172
|
+
zero_or_one_stop: ")",
|
|
173
|
+
zero_or_more_start: "maybe_some(",
|
|
174
|
+
zero_or_more_stop: ")",
|
|
175
|
+
one_or_more_start: "some(",
|
|
176
|
+
one_or_more_stop: ")",
|
|
177
|
+
alternation_start: "any([",
|
|
178
|
+
alternation_stop: "])",
|
|
179
|
+
alternative_start: "",
|
|
180
|
+
alternative_stop: "",
|
|
181
|
+
boundary: "boundary",
|
|
182
|
+
non_word_boundary: "non_word_boundary",
|
|
183
|
+
space: "space",
|
|
184
|
+
line_start: "line_start",
|
|
185
|
+
line_end: "line_end",
|
|
186
|
+
any_character: "any_character",
|
|
187
|
+
range_start: "any(",
|
|
188
|
+
range_mid: "..",
|
|
189
|
+
range_stop: ")",
|
|
190
|
+
backref_num_start: "ref(",
|
|
191
|
+
backref_num_stop: ")"
|
|
193
192
|
}
|
|
194
193
|
|
|
195
194
|
str = []
|
|
196
195
|
tokens.each_with_index do |token, idx|
|
|
197
196
|
prev = tokens[idx - 1] if idx > 0
|
|
198
197
|
left_side = %i[characterset_stop capture_stop
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
198
|
+
zero_or_one_stop zero_or_more_stop one_or_more_stop
|
|
199
|
+
boundary non_word_boundary
|
|
200
|
+
line_start any_character range_stop space
|
|
201
|
+
backref_num_stop]
|
|
203
202
|
right_side = %i[characterset_start capture_start
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
#if prev==:range_stop and token==:range_start
|
|
203
|
+
zero_or_one_start zero_or_more_start one_or_more_start
|
|
204
|
+
boundary non_word_boundary
|
|
205
|
+
line_end any_character range_start space
|
|
206
|
+
backref_num_start]
|
|
207
|
+
# if prev==:range_stop and token==:range_start
|
|
209
208
|
# str << ' :adding_ranges '
|
|
210
|
-
#end
|
|
211
|
-
if (prev.instance_of?(String) && right_side.include?(token))
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
str <<
|
|
209
|
+
# end
|
|
210
|
+
if (prev.instance_of?(String) && right_side.include?(token)) ||
|
|
211
|
+
(left_side.include?(prev) && token.instance_of?(String)) ||
|
|
212
|
+
(left_side.include?(prev) && right_side.include?(token))
|
|
213
|
+
str << " + "
|
|
215
214
|
end
|
|
216
|
-
str <<
|
|
215
|
+
str << ", " if prev == :alternative_stop && token == :alternative_start
|
|
217
216
|
# str << '[' if prev == :characterset_start and token == :range_start
|
|
218
217
|
# str << ']' if prev == :range_stop and token ==:characterset_stop
|
|
219
218
|
if subs.include? token
|
|
@@ -234,48 +233,48 @@ def stringify(node)
|
|
|
234
233
|
end
|
|
235
234
|
|
|
236
235
|
def stringify_root(root, indent: 0)
|
|
237
|
-
warning =
|
|
236
|
+
warning = ""
|
|
238
237
|
root[:from] = [""] if root[:from] == []
|
|
239
|
-
str = " "*indent+"sub #{stringify(root[:from])}, #{root[:to]}"
|
|
238
|
+
str = " " * indent + "sub #{stringify(root[:from])}, #{root[:to]}"
|
|
240
239
|
[:before, :not_before, :after, :not_after].each do |look|
|
|
241
240
|
# puts "#{look.inspect} = #{root[look]}"
|
|
242
241
|
next unless root[look]
|
|
243
242
|
str_look = stringify(root[look])
|
|
244
|
-
str_look = "\"\"" if root[look] == [] || root[look]
|
|
245
|
-
#if str_look.empty? #apparently it is empty sometimes. iso-mal-Mlym-Latn for example
|
|
243
|
+
str_look = "\"\"" if root[look] == [] || root[look].nil?
|
|
244
|
+
# if str_look.empty? #apparently it is empty sometimes. iso-mal-Mlym-Latn for example
|
|
246
245
|
# warning << "warning: #{look} is empty string;"
|
|
247
|
-
#else
|
|
248
|
-
|
|
249
|
-
#end
|
|
246
|
+
# else
|
|
247
|
+
str << ", #{look}: #{str_look}"
|
|
248
|
+
# end
|
|
250
249
|
end
|
|
251
|
-
str = " "*indent+"# #{str} # warning: :" if
|
|
252
|
-
str = " "*indent+"# #{str} # #{warning}" if !warning.empty?
|
|
250
|
+
str = " " * indent + "# #{str} # warning: :" if /[^\[]:[^ \]]/.match?(str)
|
|
251
|
+
str = " " * indent + "# #{str} # #{warning}" if !warning.empty?
|
|
253
252
|
|
|
254
|
-
str = " "*indent+"# #{str} # warning: :missing unimplemented" if str.include?(
|
|
255
|
-
str = " "*indent+"# #{str} # warning: :interval unimplemented" if str.include?(
|
|
256
|
-
str = " "*indent+"# #{str} # warning: :adding_ranges unimplemented" if str.include?(
|
|
257
|
-
if str.include?(
|
|
258
|
-
str = " "*indent+"# #{str} # warning: zero_or_one"
|
|
253
|
+
str = " " * indent + "# #{str} # warning: :missing unimplemented" if str.include?(":missing")
|
|
254
|
+
str = " " * indent + "# #{str} # warning: :interval unimplemented" if str.include?(":interval")
|
|
255
|
+
str = " " * indent + "# #{str} # warning: :adding_ranges unimplemented" if str.include?(":adding_ranges")
|
|
256
|
+
if str.include?("zero_or_one")
|
|
257
|
+
str = " " * indent + "# #{str} # warning: zero_or_one"
|
|
259
258
|
puts "str.includes 'zero_or_one'"
|
|
260
259
|
pp root
|
|
261
260
|
end
|
|
262
261
|
# str = " "*indent+"# #{str} # warning: one_or_more" if str.include?('one_or_more')
|
|
263
|
-
|
|
262
|
+
str = " " * indent + "# #{str} # warning: :lookahead_start" if str.include?(":lookahead_start")
|
|
264
263
|
# str += " # original: #{root[:from]}"
|
|
265
264
|
str
|
|
266
265
|
end
|
|
267
266
|
|
|
268
267
|
if __FILE__ == $0
|
|
269
|
-
rs = File.
|
|
268
|
+
rs = File.read(__dir__ + "/../../docs/utils/regexp_examples.txt").gsub(/([^\\^])\\u/, '\\1\\\\u').gsub("\\\\b", '\b')
|
|
270
269
|
rs = rs.split("\n")
|
|
271
270
|
rs.each do |r|
|
|
272
271
|
puts r
|
|
273
|
-
tree = Regexp::Parser.parse(r,
|
|
272
|
+
tree = Regexp::Parser.parse(r, "ruby/2.1")
|
|
274
273
|
conv = process(tree)
|
|
275
274
|
pp conv
|
|
276
275
|
root = process_root(conv)
|
|
277
276
|
pp root
|
|
278
|
-
root[:to] = [
|
|
277
|
+
root[:to] = ["X"]
|
|
279
278
|
str = stringify_root(root)
|
|
280
279
|
puts str
|
|
281
280
|
puts "\n\n"
|
data/lib/interscript/version.rb
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
class Interscript::Node::Group
|
|
2
|
-
def to_visualization_array(map=self)
|
|
2
|
+
def to_visualization_array(map = self)
|
|
3
3
|
out = []
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
children.each do |rule|
|
|
6
6
|
case rule
|
|
7
7
|
when Interscript::Node::Rule::Sub
|
|
8
8
|
more = []
|
|
@@ -16,7 +16,7 @@ class Interscript::Node::Group
|
|
|
16
16
|
out << {
|
|
17
17
|
type: "Replace",
|
|
18
18
|
from: rule.from.to_html(map),
|
|
19
|
-
to: Symbol === rule.to ? rule.to : rule.to.to_html(map),
|
|
19
|
+
to: (Symbol === rule.to) ? rule.to : rule.to.to_html(map),
|
|
20
20
|
more: more
|
|
21
21
|
}
|
|
22
22
|
when Interscript::Node::Group::Parallel
|
|
@@ -25,22 +25,21 @@ class Interscript::Node::Group
|
|
|
25
25
|
children: rule.to_visualization_array(map)
|
|
26
26
|
}
|
|
27
27
|
when Interscript::Node::Rule::Funcall
|
|
28
|
-
more = rule.kwargs.map do |k,v|
|
|
29
|
-
"#{k.to_s.
|
|
28
|
+
more = rule.kwargs.map do |k, v|
|
|
29
|
+
"#{k.to_s.tr("_", " ")}: #{v}"
|
|
30
30
|
end
|
|
31
31
|
more << "<nobr>reverse run:</nobr> #{rule.reverse_run}" unless rule.reverse_run.nil?
|
|
32
32
|
|
|
33
33
|
out << {
|
|
34
|
-
type: rule.name.to_s.
|
|
34
|
+
type: rule.name.to_s.tr("_", " ").gsub(/^(.)/, &:upcase),
|
|
35
35
|
more: more.join(", ")
|
|
36
36
|
}
|
|
37
37
|
when Interscript::Node::Rule::Run
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
38
|
+
stage = rule.stage.name
|
|
39
|
+
doc = if rule.stage.map
|
|
40
|
+
map.dep_aliases[rule.stage.map].document
|
|
41
41
|
else
|
|
42
|
-
|
|
43
|
-
stage = rule.stage.name
|
|
42
|
+
map
|
|
44
43
|
end
|
|
45
44
|
|
|
46
45
|
more = []
|
|
@@ -50,7 +49,7 @@ class Interscript::Node::Group
|
|
|
50
49
|
type: "Run",
|
|
51
50
|
doc: doc.name,
|
|
52
51
|
stage: stage,
|
|
53
|
-
more: more.join(", ")
|
|
52
|
+
more: more.join(", ")
|
|
54
53
|
}
|
|
55
54
|
else
|
|
56
55
|
out << {
|
|
@@ -62,4 +61,4 @@ class Interscript::Node::Group
|
|
|
62
61
|
|
|
63
62
|
out
|
|
64
63
|
end
|
|
65
|
-
end
|
|
64
|
+
end
|
|
@@ -3,9 +3,9 @@ class Interscript::Node::Item
|
|
|
3
3
|
def to_html(doc)
|
|
4
4
|
if map
|
|
5
5
|
n = doc.dep_aliases[map].full_name
|
|
6
|
-
"#{name.
|
|
6
|
+
"#{name.tr("_", " ")} from map #{n}"
|
|
7
7
|
else
|
|
8
|
-
|
|
8
|
+
name.to_s.tr("_", " ")
|
|
9
9
|
end
|
|
10
10
|
end
|
|
11
11
|
end
|
|
@@ -14,9 +14,9 @@ class Interscript::Node::Item
|
|
|
14
14
|
def to_html(doc)
|
|
15
15
|
if map
|
|
16
16
|
n = doc.dep_aliases[map].full_name
|
|
17
|
-
"stage #{name.
|
|
17
|
+
"stage #{name.tr("_", " ")} from map #{n}"
|
|
18
18
|
else
|
|
19
|
-
|
|
19
|
+
name.to_s.tr("_", " ")
|
|
20
20
|
end
|
|
21
21
|
end
|
|
22
22
|
end
|
|
@@ -26,15 +26,15 @@ class Interscript::Node::Item
|
|
|
26
26
|
"<nobr>any (</nobr>" +
|
|
27
27
|
case @value
|
|
28
28
|
when Array
|
|
29
|
-
value.map(&Interscript::Node::Item.method(:try_convert)).map{|i|i.to_html(doc)}.join(", ")
|
|
29
|
+
value.map(&Interscript::Node::Item.method(:try_convert)).map { |i| i.to_html(doc) }.join(", ")
|
|
30
30
|
when ::String
|
|
31
|
-
value.
|
|
31
|
+
value.chars.map(&Interscript::Node::Item.method(:try_convert)).map { |i| i.to_html(doc) }.join(", ")
|
|
32
32
|
when Range
|
|
33
|
-
[value.begin, value.end].map(&Interscript::Node::Item.method(:try_convert)).map{|i|i.to_html(doc)}.join(" to ")
|
|
33
|
+
[value.begin, value.end].map(&Interscript::Node::Item.method(:try_convert)).map { |i| i.to_html(doc) }.join(" to ")
|
|
34
34
|
else
|
|
35
35
|
h(value.inspect)
|
|
36
36
|
end +
|
|
37
|
-
|
|
37
|
+
")"
|
|
38
38
|
end
|
|
39
39
|
end
|
|
40
40
|
|
|
@@ -42,7 +42,7 @@ class Interscript::Node::Item
|
|
|
42
42
|
def to_html(doc)
|
|
43
43
|
"<nobr>capture group (</nobr>" +
|
|
44
44
|
data.to_html(doc) +
|
|
45
|
-
|
|
45
|
+
")"
|
|
46
46
|
end
|
|
47
47
|
end
|
|
48
48
|
|
|
@@ -50,13 +50,13 @@ class Interscript::Node::Item
|
|
|
50
50
|
def to_html(_)
|
|
51
51
|
"<nobr>capture reference (</nobr>" +
|
|
52
52
|
id.to_s +
|
|
53
|
-
|
|
53
|
+
")"
|
|
54
54
|
end
|
|
55
55
|
end
|
|
56
56
|
|
|
57
57
|
class Group < self
|
|
58
58
|
def to_html(doc)
|
|
59
|
-
@children.map{|i|i.to_html(doc)}.join(" + ")
|
|
59
|
+
@children.map { |i| i.to_html(doc) }.join(" + ")
|
|
60
60
|
end
|
|
61
61
|
end
|
|
62
62
|
|
|
@@ -77,7 +77,7 @@ class Interscript::Node::Item
|
|
|
77
77
|
class String < self
|
|
78
78
|
def to_html(_)
|
|
79
79
|
out = ""
|
|
80
|
-
|
|
80
|
+
data.each_char do |i|
|
|
81
81
|
out << "<ruby>"
|
|
82
82
|
out << "<kbd>#{h i}</kbd>"
|
|
83
83
|
out << "<rt>#{"%04x" % i.ord}</rt>"
|
|
@@ -86,4 +86,4 @@ class Interscript::Node::Item
|
|
|
86
86
|
out
|
|
87
87
|
end
|
|
88
88
|
end
|
|
89
|
-
end
|
|
89
|
+
end
|