parsanol 1.3.56-arm-linux → 1.3.57-arm-linux
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/exe/parsanol +7 -0
- data/lib/parsanol/3.2/parsanol_native.so +0 -0
- data/lib/parsanol/3.3/parsanol_native.so +0 -0
- data/lib/parsanol/3.4/parsanol_native.so +0 -0
- data/lib/parsanol/4.0/parsanol_native.so +0 -0
- data/lib/parsanol/native/libparsanol.so +0 -0
- data/lib/parsanol/native/serializer.rb +5 -2
- data/lib/parsanol/parg/artifact.rb +213 -0
- data/lib/parsanol/parg/authoring.rb +110 -0
- data/lib/parsanol/parg/bindings.rb +129 -0
- data/lib/parsanol/parg/cli.rb +251 -0
- data/lib/parsanol/parg/compiler.rb +465 -0
- data/lib/parsanol/parg/derive.rb +28 -0
- data/lib/parsanol/parg/document.rb +69 -0
- data/lib/parsanol/parg/error.rb +18 -0
- data/lib/parsanol/parg/frontend.rb +173 -0
- data/lib/parsanol/parg/import.rb +34 -0
- data/lib/parsanol/parg/importers/abnf.rb +299 -0
- data/lib/parsanol/parg/importers/ebnf.rb +201 -0
- data/lib/parsanol/parg/importers/pest.rb +314 -0
- data/lib/parsanol/parg/imports.rb +87 -0
- data/lib/parsanol/parg/lexer.rb +72 -0
- data/lib/parsanol/parg/lints.rb +171 -0
- data/lib/parsanol/parg/lsp.rb +208 -0
- data/lib/parsanol/parg/lutaml.rb +67 -0
- data/lib/parsanol/parg/node.rb +20 -0
- data/lib/parsanol/parg/parser.rb +458 -0
- data/lib/parsanol/parg/preprocess.rb +36 -0
- data/lib/parsanol/parg/render.rb +44 -0
- data/lib/parsanol/parg/selfhost.rb +49 -0
- data/lib/parsanol/parg/visitor.rb +64 -0
- data/lib/parsanol/parg.rb +38 -0
- data/lib/parsanol/version.rb +1 -1
- data/lib/parsanol/vm.rb +7 -5
- data/lib/parsanol.rb +3 -0
- data/parsanol.gemspec +3 -1
- metadata +31 -4
|
@@ -0,0 +1,314 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Parsanol
|
|
4
|
+
module PARG
|
|
5
|
+
module Import
|
|
6
|
+
# pest (Rust PEG) importer — the common subset, mapped to PARG.
|
|
7
|
+
#
|
|
8
|
+
# Conversions:
|
|
9
|
+
# - ordered choice "|", predicates "!"/"&", postfix "*"/"+"/"?" map
|
|
10
|
+
# directly (PARG: *x / 1*x / [ x ])
|
|
11
|
+
# - "lit" exact; ^"lit" case-insensitive -> %i"..."
|
|
12
|
+
# - 'a'..'z' ranges -> %x61-7A
|
|
13
|
+
# - builtin character classes (ASCII_DIGIT, ...) -> %x ranges
|
|
14
|
+
# - "a" ~ "b": pest inserts implicit WHITESPACE at ~; PARG emits plain
|
|
15
|
+
# sequence — whitespace must be explicit (recorded in the header)
|
|
16
|
+
# - rule modifiers _/@/$ are accepted and noted (PARG captures stay
|
|
17
|
+
# enabled inside them); silent name_ rules keep their name
|
|
18
|
+
# - PUSH/POP/PEEK/EOI/SOI and the implicit WHITESPACE/Comment rules
|
|
19
|
+
# are rejected with an explanatory error
|
|
20
|
+
class Pest
|
|
21
|
+
TOKEN = /
|
|
22
|
+
(?<ws>\s+)
|
|
23
|
+
| (?<comment>\/\/[^\n]*)
|
|
24
|
+
| (?<insens>\^(?:"(?:[^"\\]|\\.)*"|i"(?:[^"\\]|\\.)*"))
|
|
25
|
+
| (?<str>"(?:[^"\\]|\\.)*")
|
|
26
|
+
| (?<range>'(?:[^'\\]|\\.)'\.\.'(?:[^'\\]|\\.)')
|
|
27
|
+
| (?<ident>[A-Za-z_][A-Za-z0-9_]*)
|
|
28
|
+
| (?<punct>[|~!&*+?(){}=@$])
|
|
29
|
+
/x
|
|
30
|
+
|
|
31
|
+
BUILTINS = {
|
|
32
|
+
"ANY" => "%x00-FF",
|
|
33
|
+
"ASCII_DIGIT" => "%x30-39",
|
|
34
|
+
"ASCII_NONZERO_DIGIT" => "%x31-39",
|
|
35
|
+
"ASCII_BIN_DIGIT" => "%x30-31",
|
|
36
|
+
"ASCII_OCT_DIGIT" => "%x30-37",
|
|
37
|
+
"ASCII_HEX_DIGIT" => "%x30-39 / %x41-46 / %x61-66",
|
|
38
|
+
"ASCII_ALPHA" => "%x41-5A / %x61-7A",
|
|
39
|
+
"ASCII_ALPHA_LOWER" => "%x61-7A",
|
|
40
|
+
"ASCII_ALPHA_UPPER" => "%x41-5A",
|
|
41
|
+
"ASCII_ALPHANUMERIC" => "%x30-39 / %x41-5A / %x61-7A",
|
|
42
|
+
"NEWLINE" => '"\n" / "\r\n"',
|
|
43
|
+
}.freeze
|
|
44
|
+
|
|
45
|
+
SKIPPED_NAMES = %w[ws comment].freeze
|
|
46
|
+
MODIFIER_CHARS = %w[_ @ $].freeze
|
|
47
|
+
|
|
48
|
+
REJECTED = %w[SOI EOI WHITESPACE COMMENT PUSH POP PEEK PEEK_ALL DROP
|
|
49
|
+
RESET].freeze
|
|
50
|
+
|
|
51
|
+
class << self
|
|
52
|
+
def call(text)
|
|
53
|
+
new(text).call
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
def initialize(text)
|
|
58
|
+
@tokens = []
|
|
59
|
+
@pos = 0
|
|
60
|
+
@rules = {}
|
|
61
|
+
@notes = []
|
|
62
|
+
scan(text)
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
def call
|
|
66
|
+
parse_rules
|
|
67
|
+
emit
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
private
|
|
71
|
+
|
|
72
|
+
def scan(text)
|
|
73
|
+
pos = 0
|
|
74
|
+
until pos >= text.length
|
|
75
|
+
match = TOKEN.match(text, pos)
|
|
76
|
+
if match.nil? || match.begin(0) != pos
|
|
77
|
+
raise Error,
|
|
78
|
+
"pest: unexpected #{text[pos].inspect} at offset #{pos}"
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
name = TOKEN.names.find { |n| match[n] }
|
|
82
|
+
@tokens << [name.to_sym, match[0], pos] unless SKIPPED_NAMES.include?(name)
|
|
83
|
+
pos = match.end(0)
|
|
84
|
+
end
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
def peek(offset = 0)
|
|
88
|
+
@tokens[@pos + offset]
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
def advance
|
|
92
|
+
token = @tokens[@pos]
|
|
93
|
+
@pos += 1
|
|
94
|
+
token
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
def parse_rules
|
|
98
|
+
until peek.nil?
|
|
99
|
+
token = advance
|
|
100
|
+
unless token[0] == :ident
|
|
101
|
+
raise Error, "pest: expected rule name, got #{token[1].inspect}"
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
name = token[1]
|
|
105
|
+
if REJECTED.include?(name)
|
|
106
|
+
raise Error,
|
|
107
|
+
"pest: #{name} is not importable — " \
|
|
108
|
+
"#{name == 'WHITESPACE' ? 'pest applies it implicitly; define whitespace explicitly in PARG' : 'express it in PARG syntax'}"
|
|
109
|
+
end
|
|
110
|
+
@notes << "silent rule #{name.inspect} imported as a normal rule" if name.end_with?("_")
|
|
111
|
+
equals = advance
|
|
112
|
+
unless equals && equals[0] == :punct && equals[1] == "="
|
|
113
|
+
raise Error, "pest: expected '=' after #{name.inspect}"
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
nil
|
|
117
|
+
if peek&.[](0) == :punct && MODIFIER_CHARS.include?(peek[1])
|
|
118
|
+
modifier = advance[1]
|
|
119
|
+
@notes << "rule #{name.inspect}: modifier #{modifier.inspect} " \
|
|
120
|
+
"accepted; PARG captures stay enabled inside it"
|
|
121
|
+
end
|
|
122
|
+
expect_punct("{")
|
|
123
|
+
if @rules.key?(name)
|
|
124
|
+
raise Error, "pest: duplicate rule #{name.inspect}"
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
@rules[name] = parse_alt
|
|
128
|
+
expect_punct("}")
|
|
129
|
+
end
|
|
130
|
+
end
|
|
131
|
+
|
|
132
|
+
def parse_alt
|
|
133
|
+
branches = [parse_seq]
|
|
134
|
+
while peek&.[](0) == :punct && peek[1] == "|"
|
|
135
|
+
advance
|
|
136
|
+
branches << parse_seq
|
|
137
|
+
end
|
|
138
|
+
branches.length == 1 ? branches.first : Node.new(:alt, branches)
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
def parse_seq
|
|
142
|
+
items = [parse_prefixed]
|
|
143
|
+
until seq_stops?
|
|
144
|
+
items << parse_prefixed
|
|
145
|
+
end
|
|
146
|
+
items.length == 1 ? items.first : Node.new(:seq, items)
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
def seq_stops?
|
|
150
|
+
token = peek
|
|
151
|
+
return true if token.nil?
|
|
152
|
+
|
|
153
|
+
token[0] == :punct && %w[| ) }].include?(token[1])
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
def parse_prefixed
|
|
157
|
+
sign = nil
|
|
158
|
+
if peek&.[](0) == :punct && %w[! &].include?(peek[1])
|
|
159
|
+
sign = advance[1] == "&"
|
|
160
|
+
end
|
|
161
|
+
node = parse_suffixed
|
|
162
|
+
sign.nil? ? node : Node.new(:pred, sign, node)
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
def parse_suffixed
|
|
166
|
+
node = parse_primary
|
|
167
|
+
op = peek
|
|
168
|
+
return node unless op&.[](0) == :punct && %w[* + ?].include?(op[1])
|
|
169
|
+
|
|
170
|
+
advance
|
|
171
|
+
case op[1]
|
|
172
|
+
when "*" then Node.new(:rep, node, 0, nil)
|
|
173
|
+
when "+" then Node.new(:rep, node, 1, nil)
|
|
174
|
+
else Node.new(:opt, node)
|
|
175
|
+
end
|
|
176
|
+
end
|
|
177
|
+
|
|
178
|
+
def parse_primary
|
|
179
|
+
token = advance
|
|
180
|
+
raise Error, "pest: unexpected end of rule body" if token.nil?
|
|
181
|
+
|
|
182
|
+
case token[0]
|
|
183
|
+
when :str then Node.new(:lit, unescape(token[1][1..-2]), false)
|
|
184
|
+
when :insens
|
|
185
|
+
body = token[1][1..]
|
|
186
|
+
body = body[1..] if body.start_with?("i")
|
|
187
|
+
Node.new(:lit, unescape(body[1..-2]), true)
|
|
188
|
+
when :range
|
|
189
|
+
lo = token[1][1]
|
|
190
|
+
hi = token[1][-2]
|
|
191
|
+
Node.new(:class, [[lo.ord, hi.ord]])
|
|
192
|
+
when :ident
|
|
193
|
+
if BUILTINS.key?(token[1])
|
|
194
|
+
parse_builtin(token[1])
|
|
195
|
+
elsif REJECTED.include?(token[1])
|
|
196
|
+
raise Error,
|
|
197
|
+
"pest: #{token[1]} is not importable — " \
|
|
198
|
+
"#{token[1] == 'WHITESPACE' ? 'pest applies it implicitly; define whitespace explicitly in PARG' : 'express it in PARG syntax'}"
|
|
199
|
+
else
|
|
200
|
+
Node.new(:ref, token[1])
|
|
201
|
+
end
|
|
202
|
+
when :punct
|
|
203
|
+
if token[1] == "("
|
|
204
|
+
node = parse_alt
|
|
205
|
+
expect_punct(")")
|
|
206
|
+
node
|
|
207
|
+
elsif token[1] == "~"
|
|
208
|
+
unless @noted_tilde
|
|
209
|
+
@notes << "~ treated as plain sequence (pest inserts implicit " \
|
|
210
|
+
"WHITESPACE; PARG does not)"
|
|
211
|
+
end
|
|
212
|
+
@noted_tilde = true
|
|
213
|
+
Node.new(:lit, " ", false)
|
|
214
|
+
else
|
|
215
|
+
raise Error, "pest: unexpected #{token[1].inspect}"
|
|
216
|
+
end
|
|
217
|
+
else
|
|
218
|
+
raise Error, "pest: unexpected token #{token[0]}"
|
|
219
|
+
end
|
|
220
|
+
end
|
|
221
|
+
|
|
222
|
+
def parse_builtin(name)
|
|
223
|
+
spec = BUILTINS.fetch(name)
|
|
224
|
+
branches = spec.split(" / ").map do |part|
|
|
225
|
+
if part.start_with?("%x")
|
|
226
|
+
parse_hex_node(part)
|
|
227
|
+
else
|
|
228
|
+
Node.new(:lit, unescape(part[1..-2]), false)
|
|
229
|
+
end
|
|
230
|
+
end
|
|
231
|
+
branches.length == 1 ? branches.first : Node.new(:alt, branches)
|
|
232
|
+
end
|
|
233
|
+
|
|
234
|
+
def parse_hex_node(spec)
|
|
235
|
+
body = spec[2..]
|
|
236
|
+
if body.include?("-")
|
|
237
|
+
lo, hi = body.split("-")
|
|
238
|
+
Node.new(:class, [[lo.to_i(16), hi.to_i(16)]])
|
|
239
|
+
else
|
|
240
|
+
byte = body.to_i(16)
|
|
241
|
+
Node.new(:class, [[byte, byte]])
|
|
242
|
+
end
|
|
243
|
+
end
|
|
244
|
+
|
|
245
|
+
def expect_punct(char)
|
|
246
|
+
token = advance
|
|
247
|
+
unless token && token[0] == :punct && token[1] == char
|
|
248
|
+
raise Error, "pest: expected #{char.inspect}"
|
|
249
|
+
end
|
|
250
|
+
end
|
|
251
|
+
|
|
252
|
+
def unescape(body)
|
|
253
|
+
body.gsub(/\\(.)/) do
|
|
254
|
+
case Regexp.last_match(1)
|
|
255
|
+
when "n" then "\n"
|
|
256
|
+
when "t" then "\t"
|
|
257
|
+
when "r" then "\r"
|
|
258
|
+
when "0" then "\0"
|
|
259
|
+
else Regexp.last_match(1)
|
|
260
|
+
end
|
|
261
|
+
end
|
|
262
|
+
end
|
|
263
|
+
|
|
264
|
+
# ---- emission ------------------------------------------------------
|
|
265
|
+
|
|
266
|
+
def emit
|
|
267
|
+
lines = [
|
|
268
|
+
"# Imported from pest (Rust PEG).",
|
|
269
|
+
"# Notes:",
|
|
270
|
+
"# - ordered choice and predicates map directly",
|
|
271
|
+
]
|
|
272
|
+
@notes.each { |note| lines << "# - #{note}" }
|
|
273
|
+
lines << "grammar imported_pest version \"0.0.0\" {"
|
|
274
|
+
lines.concat(@rules.map { |name, node| " #{name} = #{emit_node(node, :top)}" })
|
|
275
|
+
lines << "}"
|
|
276
|
+
"#{lines.join("\n")}\n"
|
|
277
|
+
end
|
|
278
|
+
|
|
279
|
+
def emit_node(node, context)
|
|
280
|
+
case node.kind
|
|
281
|
+
when :lit then emit_string(node.a, fold: node.b)
|
|
282
|
+
when :class
|
|
283
|
+
node.a.map { |lo, hi| "%x#{lo.to_s(16)}-#{hi.to_s(16)}" }.join(" ")
|
|
284
|
+
when :ref then node.a
|
|
285
|
+
when :alt
|
|
286
|
+
text = node.a.map { |branch| emit_node(branch, :branch) }.join(" / ")
|
|
287
|
+
context == :branch ? "(#{text})" : text
|
|
288
|
+
when :seq
|
|
289
|
+
text = node.a.map { |item| emit_node(item, :seq_item) }.join(" ")
|
|
290
|
+
context == :branch ? "(#{text})" : text
|
|
291
|
+
when :rep
|
|
292
|
+
prefix = node.b.zero? ? "*" : "#{node.b}*"
|
|
293
|
+
"#{prefix}#{emit_operand(node.a)}"
|
|
294
|
+
when :opt then "[ #{emit_node(node.a, :top)} ]"
|
|
295
|
+
when :pred
|
|
296
|
+
"#{node.a ? '&' : '!'}#{emit_operand(node.b)}"
|
|
297
|
+
end
|
|
298
|
+
end
|
|
299
|
+
|
|
300
|
+
def emit_operand(node)
|
|
301
|
+
case node.kind
|
|
302
|
+
when :lit, :class, :ref then emit_node(node, :top)
|
|
303
|
+
else "(#{emit_node(node, :top)})"
|
|
304
|
+
end
|
|
305
|
+
end
|
|
306
|
+
|
|
307
|
+
def emit_string(body, fold:)
|
|
308
|
+
escaped = body.gsub("\\", "\\\\").gsub('"', '\\"')
|
|
309
|
+
fold ? "%i\"#{escaped}\"" : "\"#{escaped}\""
|
|
310
|
+
end
|
|
311
|
+
end
|
|
312
|
+
end
|
|
313
|
+
end
|
|
314
|
+
end
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Parsanol
|
|
4
|
+
module PARG
|
|
5
|
+
# Cross-file imports (PN 1): `use <name>` at the top of a .parg file
|
|
6
|
+
# loads <name>.parg from an import directory, prefixes every rule with
|
|
7
|
+
# "<name>." (capture names are preserved — embedded grammars keep
|
|
8
|
+
# their capture contract), and merges entries, bindings, preprocess
|
|
9
|
+
# steps, tests, and docs. Cycles and missing files are compile
|
|
10
|
+
# errors. The artifact bakes the expansion: consumers never see
|
|
11
|
+
# imports.
|
|
12
|
+
module Imports
|
|
13
|
+
module_function
|
|
14
|
+
|
|
15
|
+
def merge!(document, import_dirs, merging = [])
|
|
16
|
+
return document if document.uses.empty?
|
|
17
|
+
|
|
18
|
+
document.uses.each do |name|
|
|
19
|
+
raise CompileError, "import cycle: #{(merging + [name]).join(' -> ')}" if merging.include?(name)
|
|
20
|
+
|
|
21
|
+
source = find_file(name, import_dirs)
|
|
22
|
+
used = Parser.new(File.read(source)).parse
|
|
23
|
+
merge!(used, import_dirs, merging + [name])
|
|
24
|
+
merge_used(document, used, "#{name}.")
|
|
25
|
+
end
|
|
26
|
+
document.uses = []
|
|
27
|
+
document
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
IMPORT_EXTENSIONS = %w[.yaml .parg .json].freeze
|
|
31
|
+
|
|
32
|
+
def find_file(name, import_dirs)
|
|
33
|
+
dirs = Array(import_dirs)
|
|
34
|
+
raise CompileError, "no import directories configured for `use #{name}`" if dirs.empty?
|
|
35
|
+
|
|
36
|
+
path = dirs.flat_map { |dir| IMPORT_EXTENSIONS.map { |ext| File.join(dir, "#{name}#{ext}") } }
|
|
37
|
+
.select { |candidate| candidate.end_with?(".parg") }
|
|
38
|
+
.find { |candidate| File.file?(candidate) }
|
|
39
|
+
path || raise(CompileError, "import `use #{name}` not found in #{dirs.inspect}")
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
def merge_used(document, used, prefix)
|
|
43
|
+
used.rules.each do |name, node|
|
|
44
|
+
qualified = "#{prefix}#{name}"
|
|
45
|
+
raise CompileError, "imported rule #{qualified.inspect} collides" if document.rules.key?(qualified)
|
|
46
|
+
|
|
47
|
+
document.rules[qualified] = copy(node, prefix)
|
|
48
|
+
end
|
|
49
|
+
used.entries.each do |entry, rule|
|
|
50
|
+
document.entries["#{prefix}#{entry}"] = "#{prefix}#{rule}"
|
|
51
|
+
end
|
|
52
|
+
used.bindings.each do |rule, list|
|
|
53
|
+
document.bindings["#{prefix}#{rule}"] = list
|
|
54
|
+
end
|
|
55
|
+
used.preprocess.each do |name, steps|
|
|
56
|
+
document.preprocess["#{prefix}#{name}"] = steps
|
|
57
|
+
end
|
|
58
|
+
used.docs.each do |rule, text|
|
|
59
|
+
document.docs["#{prefix}#{rule}"] = text
|
|
60
|
+
end
|
|
61
|
+
default_used_entry = used.entries.size == 1 ? used.entries.keys.first : nil
|
|
62
|
+
used.tests.each do |test|
|
|
63
|
+
entry = test.entry || default_used_entry
|
|
64
|
+
document.tests << Document::Test.new(
|
|
65
|
+
entry ? "#{prefix}#{entry}" : test.entry,
|
|
66
|
+
test.kind, test.input, test.expect
|
|
67
|
+
)
|
|
68
|
+
end
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
def copy(node, prefix)
|
|
72
|
+
return node unless node.is_a?(Node)
|
|
73
|
+
|
|
74
|
+
case node.kind
|
|
75
|
+
when :seq then Node.new(:seq, node.a.map { |child| copy(child, prefix) })
|
|
76
|
+
when :alt then Node.new(:alt, node.a.map { |child| copy(child, prefix) })
|
|
77
|
+
when :rep then Node.new(:rep, copy(node.a, prefix), node.b, node.c)
|
|
78
|
+
when :opt then Node.new(:opt, copy(node.a, prefix))
|
|
79
|
+
when :pred then Node.new(:pred, node.a, copy(node.b, prefix))
|
|
80
|
+
when :cap then Node.new(:cap, node.a, copy(node.b, prefix))
|
|
81
|
+
when :ref then Node.new(:ref, "#{prefix}#{node.a}")
|
|
82
|
+
else node
|
|
83
|
+
end
|
|
84
|
+
end
|
|
85
|
+
end
|
|
86
|
+
end
|
|
87
|
+
end
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Parsanol
|
|
4
|
+
module PARG
|
|
5
|
+
# Tokenizer for PARG source text.
|
|
6
|
+
#
|
|
7
|
+
# Emits Token structs. Comments (# to end of line) and whitespace are
|
|
8
|
+
# dropped. Punctuation arrives as :punct tokens carrying the character.
|
|
9
|
+
class Lexer
|
|
10
|
+
Token = Struct.new(:type, :value, :offset)
|
|
11
|
+
|
|
12
|
+
DROPPED = %i[ws comment].freeze
|
|
13
|
+
|
|
14
|
+
TOKEN = /
|
|
15
|
+
(?<newline>\r?\n)
|
|
16
|
+
| (?<ws>[ \t]+)
|
|
17
|
+
| (?<doc>\#\#[^\n]*)
|
|
18
|
+
| (?<comment>\#[^\n]*)
|
|
19
|
+
| (?<hex>%x[0-9A-Fa-f]{2,6}(?:-[0-9A-Fa-f]{2,6}|(?:\.[0-9A-Fa-f]{2,6})+)?)
|
|
20
|
+
| (?<istr>%i"(?:[^"\\]|\\.)*")
|
|
21
|
+
| (?<sstr>%s"(?:[^"\\]|\\.)*")
|
|
22
|
+
| (?<str>"(?:[^"\\]|\\.)*")
|
|
23
|
+
| (?<arrow>->)
|
|
24
|
+
| (?<num>\d+)
|
|
25
|
+
| (?<ident>[A-Za-z_][A-Za-z0-9_]*)
|
|
26
|
+
| (?<punct>[=\/()\[\]{}*!&:,.])
|
|
27
|
+
/x
|
|
28
|
+
|
|
29
|
+
def initialize(text)
|
|
30
|
+
@text = text
|
|
31
|
+
@tokens = []
|
|
32
|
+
scan
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
attr_reader :tokens
|
|
36
|
+
|
|
37
|
+
private
|
|
38
|
+
|
|
39
|
+
def scan
|
|
40
|
+
pos = 0
|
|
41
|
+
until pos >= @text.length
|
|
42
|
+
match = TOKEN.match(@text, pos)
|
|
43
|
+
if match.nil? || match.begin(0) != pos
|
|
44
|
+
raise ParseError,
|
|
45
|
+
"unexpected character #{@text[pos].inspect} at offset #{pos}"
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
type = matched_type(match)
|
|
49
|
+
if type && !DROPPED.include?(type)
|
|
50
|
+
value = token_value(type, match)
|
|
51
|
+
@tokens << Token.new(type, value, pos)
|
|
52
|
+
end
|
|
53
|
+
pos = match.end(0)
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
def matched_type(match)
|
|
58
|
+
TOKEN.names.each do |name|
|
|
59
|
+
return name.to_sym if match[name]
|
|
60
|
+
end
|
|
61
|
+
nil
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
def token_value(type, match)
|
|
65
|
+
case type
|
|
66
|
+
when :hex, :istr, :sstr then match[0][2..]
|
|
67
|
+
else match[0]
|
|
68
|
+
end
|
|
69
|
+
end
|
|
70
|
+
end
|
|
71
|
+
end
|
|
72
|
+
end
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Parsanol
|
|
4
|
+
module PARG
|
|
5
|
+
# Registry of compile-time lints (OCP): each lint is a class with
|
|
6
|
+
# `errors(document, compiler)` and `warnings(document, compiler)` —
|
|
7
|
+
# both pure (data in, data out; no side effects). The compiler runs
|
|
8
|
+
# every registered lint; adding a lint never touches the pipeline.
|
|
9
|
+
# Lints use the compiler only as an analysis facade (find_left_cycle,
|
|
10
|
+
# first_set, nullable?).
|
|
11
|
+
module Lints
|
|
12
|
+
class << self
|
|
13
|
+
def register(name, strategy)
|
|
14
|
+
strategies[name.to_sym] = strategy
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
def errors(document, compiler)
|
|
18
|
+
strategies.values.flat_map { |strategy| strategy.errors(document, compiler) }
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
def warnings(document, compiler)
|
|
22
|
+
strategies.values.flat_map { |strategy| strategy.warnings(document, compiler) }
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def strategies
|
|
26
|
+
@strategies ||= {}
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
# Left recursion (direct + indirect, leftmost positions) rejects
|
|
31
|
+
# the compile: every PEG engine would loop forever.
|
|
32
|
+
class LeftRecursion
|
|
33
|
+
def errors(document, compiler)
|
|
34
|
+
document.rules.filter_map do |name, _node|
|
|
35
|
+
cycle = compiler.find_left_cycle(name)
|
|
36
|
+
"left recursion detected: #{cycle.join(' -> ')}" if cycle
|
|
37
|
+
end
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
def warnings(_document, _compiler) = []
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# Ordered-choice hazards: the class of silent failures PEGs are
|
|
44
|
+
# known for, turned into build failures plus recorded warnings.
|
|
45
|
+
class Alternatives
|
|
46
|
+
def errors(document, compiler)
|
|
47
|
+
checked(document, compiler).flat_map { |result| result[:errors] }
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
def warnings(document, compiler)
|
|
51
|
+
checked(document, compiler).flat_map { |result| result[:warnings] }
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
private
|
|
55
|
+
|
|
56
|
+
def checked(document, compiler)
|
|
57
|
+
@checked ||= {}
|
|
58
|
+
@checked[[document.object_id, compiler.object_id]] ||= begin
|
|
59
|
+
results = []
|
|
60
|
+
document.rules.each do |name, node|
|
|
61
|
+
next unless node.kind == :alt
|
|
62
|
+
|
|
63
|
+
results.concat(check_alternatives(name, node, compiler))
|
|
64
|
+
end
|
|
65
|
+
results
|
|
66
|
+
end
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
def check_alternatives(name, node, compiler)
|
|
70
|
+
results = []
|
|
71
|
+
branches = node.a
|
|
72
|
+
branches.each_with_index do |branch, index|
|
|
73
|
+
if compiler.nullable?(branch) && index < branches.length - 1
|
|
74
|
+
results << { errors: [empty_shadow_error(name, index + 1)], warnings: [] }
|
|
75
|
+
end
|
|
76
|
+
branches[(index + 1)..].each_with_index do |later, j|
|
|
77
|
+
results << compare_branches(name, branch, index + 1, later, index + j + 2, compiler)
|
|
78
|
+
end
|
|
79
|
+
end
|
|
80
|
+
results
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
def empty_shadow_error(name, branch_number)
|
|
84
|
+
"rule #{name}: branch #{branch_number} can match empty input and " \
|
|
85
|
+
"shadows all later branches"
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
def compare_branches(name, earlier, earlier_n, later, later_n, compiler)
|
|
89
|
+
earlier_lit = literal_string(earlier)
|
|
90
|
+
later_lit = literal_string(later)
|
|
91
|
+
if earlier_lit && later_lit
|
|
92
|
+
if earlier_lit == later_lit
|
|
93
|
+
return { errors: ["rule #{name}: branches #{earlier_n} and #{later_n} are identical"], warnings: [] }
|
|
94
|
+
elsif later_lit.start_with?(earlier_lit)
|
|
95
|
+
return { errors: [shadow_error(name, earlier_n, earlier_lit, later_n, later_lit)], warnings: [] }
|
|
96
|
+
end
|
|
97
|
+
end
|
|
98
|
+
first_earlier = compiler.first_set(earlier) - [Compiler::EPS]
|
|
99
|
+
first_later = compiler.first_set(later) - [Compiler::EPS]
|
|
100
|
+
if first_earlier.include?(Compiler::ANY) || first_later.include?(Compiler::ANY)
|
|
101
|
+
return { errors: [], warnings: [order_warning(name, earlier_n, later_n, "first set not statically known")] }
|
|
102
|
+
end
|
|
103
|
+
return { errors: [], warnings: [] } if !first_earlier.intersect?(first_later)
|
|
104
|
+
|
|
105
|
+
{ errors: [],
|
|
106
|
+
warnings: [order_warning(name, earlier_n, later_n, "shared first bytes; ordered choice is decisive")] }
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def shadow_error(name, earlier_n, earlier_lit, later_n, later_lit)
|
|
110
|
+
"rule #{name}: branch #{earlier_n} (#{earlier_lit.inspect}) shadows " \
|
|
111
|
+
"branch #{later_n} (#{later_lit.inspect}) — reorder longest-first or " \
|
|
112
|
+
"the shorter always wins"
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
def order_warning(name, earlier_n, later_n, reason)
|
|
116
|
+
"rule #{name}: branches #{earlier_n} and #{later_n} are order-dependent (#{reason})"
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
def literal_string(node)
|
|
120
|
+
case node.kind
|
|
121
|
+
when :lit then node.a
|
|
122
|
+
when :seq
|
|
123
|
+
node.a.map { |child| literal_string(child) }.join if node.a.all? { |child| child.kind == :lit }
|
|
124
|
+
end
|
|
125
|
+
end
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
register :left_recursion, LeftRecursion.new
|
|
129
|
+
register :alternatives, Alternatives.new
|
|
130
|
+
|
|
131
|
+
# An unbounded repetition whose body can match empty input is an
|
|
132
|
+
# INVALID GRAMMAR: the parse can never terminate, and any VM that
|
|
133
|
+
# tolerates it allocates without bound. Rejected at compile time —
|
|
134
|
+
# unboundedness is a grammar-validity property, not a runtime
|
|
135
|
+
# hazard to be hardened against.
|
|
136
|
+
class Repetitions
|
|
137
|
+
def errors(document, compiler)
|
|
138
|
+
document.rules.flat_map do |name, node|
|
|
139
|
+
walk(name, node, compiler)
|
|
140
|
+
end
|
|
141
|
+
end
|
|
142
|
+
|
|
143
|
+
def warnings(_document, _compiler) = []
|
|
144
|
+
|
|
145
|
+
private
|
|
146
|
+
|
|
147
|
+
def walk(name, node, compiler)
|
|
148
|
+
return [] if node.nil?
|
|
149
|
+
|
|
150
|
+
kind = node.kind
|
|
151
|
+
errors =
|
|
152
|
+
if kind == :rep && node.c.nil? && compiler.nullable?(node.a)
|
|
153
|
+
["rule #{name}: repetition body can match empty input — " \
|
|
154
|
+
"unbounded repetition is an invalid grammar"]
|
|
155
|
+
else
|
|
156
|
+
[]
|
|
157
|
+
end
|
|
158
|
+
children =
|
|
159
|
+
case kind
|
|
160
|
+
when :rep, :opt then [node.a]
|
|
161
|
+
when :pred, :cap then [node.b]
|
|
162
|
+
when :seq, :alt then node.a
|
|
163
|
+
else []
|
|
164
|
+
end
|
|
165
|
+
errors + children.flat_map { |child| walk(name, child, compiler) }
|
|
166
|
+
end
|
|
167
|
+
end
|
|
168
|
+
register :repetitions, Repetitions.new
|
|
169
|
+
end
|
|
170
|
+
end
|
|
171
|
+
end
|