parsanol 1.3.56-arm-linux → 1.3.57-arm-linux
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/exe/parsanol +7 -0
- data/lib/parsanol/3.2/parsanol_native.so +0 -0
- data/lib/parsanol/3.3/parsanol_native.so +0 -0
- data/lib/parsanol/3.4/parsanol_native.so +0 -0
- data/lib/parsanol/4.0/parsanol_native.so +0 -0
- data/lib/parsanol/native/libparsanol.so +0 -0
- data/lib/parsanol/native/serializer.rb +5 -2
- data/lib/parsanol/parg/artifact.rb +213 -0
- data/lib/parsanol/parg/authoring.rb +110 -0
- data/lib/parsanol/parg/bindings.rb +129 -0
- data/lib/parsanol/parg/cli.rb +251 -0
- data/lib/parsanol/parg/compiler.rb +465 -0
- data/lib/parsanol/parg/derive.rb +28 -0
- data/lib/parsanol/parg/document.rb +69 -0
- data/lib/parsanol/parg/error.rb +18 -0
- data/lib/parsanol/parg/frontend.rb +173 -0
- data/lib/parsanol/parg/import.rb +34 -0
- data/lib/parsanol/parg/importers/abnf.rb +299 -0
- data/lib/parsanol/parg/importers/ebnf.rb +201 -0
- data/lib/parsanol/parg/importers/pest.rb +314 -0
- data/lib/parsanol/parg/imports.rb +87 -0
- data/lib/parsanol/parg/lexer.rb +72 -0
- data/lib/parsanol/parg/lints.rb +171 -0
- data/lib/parsanol/parg/lsp.rb +208 -0
- data/lib/parsanol/parg/lutaml.rb +67 -0
- data/lib/parsanol/parg/node.rb +20 -0
- data/lib/parsanol/parg/parser.rb +458 -0
- data/lib/parsanol/parg/preprocess.rb +36 -0
- data/lib/parsanol/parg/render.rb +44 -0
- data/lib/parsanol/parg/selfhost.rb +49 -0
- data/lib/parsanol/parg/visitor.rb +64 -0
- data/lib/parsanol/parg.rb +38 -0
- data/lib/parsanol/version.rb +1 -1
- data/lib/parsanol/vm.rb +7 -5
- data/lib/parsanol.rb +3 -0
- data/parsanol.gemspec +3 -1
- metadata +31 -4
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Parsanol
|
|
4
|
+
module PARG
|
|
5
|
+
# F10 phase 2: the artifact-driven front end.
|
|
6
|
+
#
|
|
7
|
+
# The parg artifact is the parser of record: SelfHost.validate accepts
|
|
8
|
+
# or rejects the source with the native engine before anything else.
|
|
9
|
+
# Document construction then routes each top-level item —
|
|
10
|
+
# line-classified, since PARG's document grammar is line-oriented
|
|
11
|
+
# (PN 1) — through the reference semantics. Acceptance gate: the
|
|
12
|
+
# compiled envelope checksum must equal the reference compiler's for
|
|
13
|
+
# every grammar in the corpus.
|
|
14
|
+
module Frontend
|
|
15
|
+
NAMED_SECTION = /\A(render|bindings)\s+(\w+)\s*\{/
|
|
16
|
+
PREPROCESS_SECTION = /\Apreprocess\s+(\w+)\s*\{/
|
|
17
|
+
DERIVE_LINE = /\Aderive\s+(\w+)\s+"((?:[^"\\]|\\.)*)"/
|
|
18
|
+
TEST_SECTION = /\Atest\s*\{/
|
|
19
|
+
RULE_LINE = /\A([a-z_][a-z0-9_]*)\s*=\s*/
|
|
20
|
+
ENTRY_LINE = /\Aentry\s+(\w+)\s*:\s*(\w+)/
|
|
21
|
+
USE_LINE = /\Ause\s+(\w+)/
|
|
22
|
+
GRAMMAR_LINE = /\Agrammar\s+(\w+)\s+version\s+"([^"]+)"/
|
|
23
|
+
|
|
24
|
+
module_function
|
|
25
|
+
|
|
26
|
+
def parse(source)
|
|
27
|
+
SelfHost.validate(source) unless SelfHost.available? && !SelfHost.valid?(source)
|
|
28
|
+
|
|
29
|
+
document = Document.new
|
|
30
|
+
document.source = source
|
|
31
|
+
state = { section: nil, section_name: nil, section_lines: [],
|
|
32
|
+
depth: 0, doc_comments: [], rule_lines: {},
|
|
33
|
+
entry_lines: [] }
|
|
34
|
+
deferred = []
|
|
35
|
+
preprocess_texts = []
|
|
36
|
+
document.own_entries ||= []
|
|
37
|
+
|
|
38
|
+
source.each_line do |line|
|
|
39
|
+
stripped = line.strip
|
|
40
|
+
if stripped.start_with?("##")
|
|
41
|
+
state[:doc_comments] << stripped.sub(/\A##\s?/, "")
|
|
42
|
+
next
|
|
43
|
+
end
|
|
44
|
+
next if stripped.empty? || stripped.start_with?("#")
|
|
45
|
+
|
|
46
|
+
if state[:section]
|
|
47
|
+
advance_section(document, state, line, deferred, preprocess_texts)
|
|
48
|
+
else
|
|
49
|
+
route_top_level(document, stripped, line, state,
|
|
50
|
+
preprocess_texts)
|
|
51
|
+
end
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
replay_deferred(document, deferred, state[:rule_lines],
|
|
55
|
+
state[:entry_lines], preprocess_texts)
|
|
56
|
+
document
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
# Consumes one line inside a deferred/closed-by-brace section.
|
|
60
|
+
def advance_section(document, state, line, deferred, preprocess_texts)
|
|
61
|
+
section = state[:section]
|
|
62
|
+
state[:depth] += line.scan("{").count - line.scan("}").count
|
|
63
|
+
if state[:depth].positive?
|
|
64
|
+
preprocess_texts.last << line if section == "preprocess"
|
|
65
|
+
state[:section_lines] << line
|
|
66
|
+
return
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
case section
|
|
70
|
+
when "test"
|
|
71
|
+
deferred << ["test {", state[:section_lines]]
|
|
72
|
+
when "bindings"
|
|
73
|
+
deferred << ["bindings #{state[:section_name]} {", state[:section_lines]]
|
|
74
|
+
when "preprocess"
|
|
75
|
+
preprocess_texts.last << "}\n"
|
|
76
|
+
close_section(document, section, state[:section_name],
|
|
77
|
+
state[:section_lines])
|
|
78
|
+
else
|
|
79
|
+
close_section(document, section, state[:section_name],
|
|
80
|
+
state[:section_lines])
|
|
81
|
+
end
|
|
82
|
+
state[:section] = nil
|
|
83
|
+
state[:section_name] = nil
|
|
84
|
+
state[:section_lines] = []
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
def route_top_level(document, stripped, line, state, preprocess_texts)
|
|
88
|
+
doc_comments = state[:doc_comments]
|
|
89
|
+
|
|
90
|
+
case stripped
|
|
91
|
+
when USE_LINE
|
|
92
|
+
document.uses << Regexp.last_match(1)
|
|
93
|
+
doc_comments = []
|
|
94
|
+
when GRAMMAR_LINE
|
|
95
|
+
document.grammar_name = Regexp.last_match(1)
|
|
96
|
+
document.version = Regexp.last_match(2)
|
|
97
|
+
doc_comments = []
|
|
98
|
+
when NAMED_SECTION
|
|
99
|
+
state[:section] = Regexp.last_match(1)
|
|
100
|
+
state[:section_name] = Regexp.last_match(2)
|
|
101
|
+
state[:section_lines] = []
|
|
102
|
+
state[:depth] = 1
|
|
103
|
+
doc_comments = []
|
|
104
|
+
when PREPROCESS_SECTION
|
|
105
|
+
state[:section] = "preprocess"
|
|
106
|
+
state[:section_name] = Regexp.last_match(1)
|
|
107
|
+
state[:section_lines] = []
|
|
108
|
+
preprocess_texts << "preprocess #{state[:section_name]} {\n"
|
|
109
|
+
state[:depth] = 1
|
|
110
|
+
doc_comments = []
|
|
111
|
+
when TEST_SECTION
|
|
112
|
+
state[:section] = "test"
|
|
113
|
+
state[:section_lines] = []
|
|
114
|
+
state[:depth] = 1
|
|
115
|
+
doc_comments = []
|
|
116
|
+
when ENTRY_LINE
|
|
117
|
+
entry_name = Regexp.last_match(1)
|
|
118
|
+
document.entries[entry_name] = Regexp.last_match(2)
|
|
119
|
+
document.own_entries << entry_name
|
|
120
|
+
state[:entry_lines] << line
|
|
121
|
+
doc_comments = []
|
|
122
|
+
when DERIVE_LINE
|
|
123
|
+
document.derive[Regexp.last_match(1)] =
|
|
124
|
+
Regexp.last_match(2).gsub(/\\(.)/, '\1')
|
|
125
|
+
doc_comments = []
|
|
126
|
+
when RULE_LINE
|
|
127
|
+
rule_name = Regexp.last_match(1)
|
|
128
|
+
document.docs[rule_name] = doc_comments.join("\n") unless doc_comments.empty?
|
|
129
|
+
doc_comments = []
|
|
130
|
+
state[:rule_lines][rule_name] = line
|
|
131
|
+
merge_rule(document, line)
|
|
132
|
+
else
|
|
133
|
+
doc_comments = []
|
|
134
|
+
end
|
|
135
|
+
state[:doc_comments] = doc_comments
|
|
136
|
+
end
|
|
137
|
+
|
|
138
|
+
# Deferred test/bindings sections re-parse against a mini preamble
|
|
139
|
+
# built from the collected rule/entry/preprocess text.
|
|
140
|
+
def replay_deferred(document, deferred, rule_lines, entry_lines,
|
|
141
|
+
preprocess_texts)
|
|
142
|
+
return if deferred.empty?
|
|
143
|
+
|
|
144
|
+
preamble = "grammar Mini version \"1\" {\n" \
|
|
145
|
+
"#{rule_lines.values.join}#{entry_lines.join}" \
|
|
146
|
+
"#{preprocess_texts.join}}\n"
|
|
147
|
+
deferred.each do |(opener, lines)|
|
|
148
|
+
text = "#{preamble[0..-2]}#{opener}\n#{lines.join}}\n"
|
|
149
|
+
mini = Parser.new(text).parse
|
|
150
|
+
if opener.start_with?("test")
|
|
151
|
+
document.tests.concat(mini.tests)
|
|
152
|
+
else
|
|
153
|
+
document.bindings.merge!(mini.bindings)
|
|
154
|
+
end
|
|
155
|
+
end
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
def merge_rule(document, line)
|
|
159
|
+
mini = Parser.new("grammar Mini version \"1\" {\n#{line}}\n").parse
|
|
160
|
+
document.rules.merge!(mini.rules)
|
|
161
|
+
mini.docs.each { |rule, text| document.docs[rule] = text }
|
|
162
|
+
end
|
|
163
|
+
|
|
164
|
+
def close_section(document, kind, name, lines)
|
|
165
|
+
mini = Parser.new("grammar Mini version \"1\" {\nb = \"q\"\n}#{kind} #{name} {\n#{lines.join}}\n").parse
|
|
166
|
+
document.bindings.merge!(mini.bindings)
|
|
167
|
+
document.preprocess.merge!(mini.preprocess)
|
|
168
|
+
document.render.merge!(mini.render)
|
|
169
|
+
document.derive.merge!(mini.derive)
|
|
170
|
+
end
|
|
171
|
+
end
|
|
172
|
+
end
|
|
173
|
+
end
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Parsanol
|
|
4
|
+
module PARG
|
|
5
|
+
# Foreign-grammar importers. Each converts a foreign grammar notation to
|
|
6
|
+
# PARG source text — the .parg file is committed as the single source of
|
|
7
|
+
# truth, then compiled like any hand-written grammar. Import results are
|
|
8
|
+
# self-checked: the emitted source must round-trip through PARG::Parser.
|
|
9
|
+
module Import
|
|
10
|
+
class Error < PARG::Error; end
|
|
11
|
+
|
|
12
|
+
autoload :Abnf, "parsanol/parg/importers/abnf"
|
|
13
|
+
autoload :Ebnf, "parsanol/parg/importers/ebnf"
|
|
14
|
+
autoload :Pest, "parsanol/parg/importers/pest"
|
|
15
|
+
|
|
16
|
+
KINDS = %i[abnf ebnf pest].freeze
|
|
17
|
+
|
|
18
|
+
module_function
|
|
19
|
+
|
|
20
|
+
def import(kind, text)
|
|
21
|
+
source = case kind
|
|
22
|
+
when :abnf then Abnf.call(text)
|
|
23
|
+
when :ebnf then Ebnf.call(text)
|
|
24
|
+
when :pest then Pest.call(text)
|
|
25
|
+
else
|
|
26
|
+
raise Error,
|
|
27
|
+
"unknown import kind #{kind.inspect} (supported: #{KINDS.map(&:inspect).join(', ')})"
|
|
28
|
+
end
|
|
29
|
+
Parser.new(source).parse
|
|
30
|
+
source
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
end
|
|
34
|
+
end
|
|
@@ -0,0 +1,299 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Parsanol
|
|
4
|
+
module PARG
|
|
5
|
+
module Import
|
|
6
|
+
# RFC 5234 (+ RFC 7405 case prefixes) ABNF importer.
|
|
7
|
+
#
|
|
8
|
+
# Semantic conversions (recorded in the emitted header):
|
|
9
|
+
# - bare ABNF strings are CASE-INSENSITIVE -> emitted as %i"..."
|
|
10
|
+
# - %s"..." (case-sensitive) -> emitted as plain "..."
|
|
11
|
+
# - ABNF alternation is unordered; PARG's is ordered. The PARG compiler's
|
|
12
|
+
# first-set lint flags order-dependent branches after import.
|
|
13
|
+
# - rule names are case-insensitive in ABNF -> normalized to snake_case
|
|
14
|
+
# - prose-vals (<...>) are rejected: they are not machine-parseable
|
|
15
|
+
class Abnf
|
|
16
|
+
TOKEN = /
|
|
17
|
+
(?<ws>[ \t]+)
|
|
18
|
+
| (?<comment>;[^\n]*)
|
|
19
|
+
| (?<crlf>\r?\n)
|
|
20
|
+
| (?<sstr>%s"(?:[^"\\]|\\.)*")
|
|
21
|
+
| (?<cistr>"(?:[^"\\]|\\.)*")
|
|
22
|
+
| (?<numval>%[xbdo][0-9A-Za-z]+(?:-[0-9A-Za-z]+|(?:\.[0-9A-Za-z]+)+)?)
|
|
23
|
+
| (?<prose><[^>\n]*>)
|
|
24
|
+
| (?<definedas>=\/|=)
|
|
25
|
+
| (?<name>[A-Za-z][A-Za-z0-9-]*)
|
|
26
|
+
| (?<num>\d+)
|
|
27
|
+
| (?<punct>[*\/()\[\]])
|
|
28
|
+
/x
|
|
29
|
+
|
|
30
|
+
INLINE_WS = %i[ws comment].freeze
|
|
31
|
+
LEADING = %i[ws comment crlf].freeze
|
|
32
|
+
ELEMENT_START_TYPES = %i[name cistr sstr numval num].freeze
|
|
33
|
+
ELEMENT_START_PUNCT = ["*", "(", "["].freeze
|
|
34
|
+
|
|
35
|
+
CORE_RULES = {
|
|
36
|
+
"ALPHA" => "%x41-5A / %x61-7A",
|
|
37
|
+
"BIT" => '"0" / "1"',
|
|
38
|
+
"CHAR" => "%x01-7F",
|
|
39
|
+
"CR" => "%x0D",
|
|
40
|
+
"CRLF" => "%x0D %x0A",
|
|
41
|
+
"CTL" => "%x00-1F / %x7F",
|
|
42
|
+
"DIGIT" => "%x30-39",
|
|
43
|
+
"DQUOTE" => "%x22",
|
|
44
|
+
"HEXDIG" => '%x30-39 / %i"a" / %i"b" / %i"c" / %i"d" / %i"e" / %i"f"',
|
|
45
|
+
"HTAB" => "%x09",
|
|
46
|
+
"LF" => "%x0A",
|
|
47
|
+
"OCTET" => "%x00-FF",
|
|
48
|
+
"SP" => "%x20",
|
|
49
|
+
"VCHAR" => "%x21-7E",
|
|
50
|
+
"WSP" => "%x20 / %x09",
|
|
51
|
+
}.freeze
|
|
52
|
+
|
|
53
|
+
class << self
|
|
54
|
+
def call(text)
|
|
55
|
+
new(text).call
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def initialize(text)
|
|
60
|
+
@text = text
|
|
61
|
+
@tokens = []
|
|
62
|
+
@pos = 0
|
|
63
|
+
@rules = {}
|
|
64
|
+
scan
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
def call
|
|
68
|
+
parse_rules
|
|
69
|
+
emit
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
private
|
|
73
|
+
|
|
74
|
+
def scan
|
|
75
|
+
pos = 0
|
|
76
|
+
until pos >= @text.length
|
|
77
|
+
match = TOKEN.match(@text, pos)
|
|
78
|
+
if match.nil? || match.begin(0) != pos
|
|
79
|
+
raise Error,
|
|
80
|
+
"ABNF: unexpected #{@text[pos].inspect} at offset #{pos}"
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
name = TOKEN.names.find { |n| match[n] }
|
|
84
|
+
@tokens << [name.to_sym, match[0], pos]
|
|
85
|
+
pos = match.end(0)
|
|
86
|
+
end
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
def peek(offset = 0)
|
|
90
|
+
@tokens[@pos + offset]
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
def advance
|
|
94
|
+
token = @tokens[@pos]
|
|
95
|
+
@pos += 1
|
|
96
|
+
token
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
def skip_inline_ws
|
|
100
|
+
advance while peek && INLINE_WS.include?(peek[0])
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
def parse_rules
|
|
104
|
+
loop do
|
|
105
|
+
skip_leading
|
|
106
|
+
break if peek.nil?
|
|
107
|
+
|
|
108
|
+
parse_rule
|
|
109
|
+
end
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
def skip_leading
|
|
113
|
+
loop do
|
|
114
|
+
token = peek
|
|
115
|
+
break unless token && LEADING.include?(token[0])
|
|
116
|
+
|
|
117
|
+
advance
|
|
118
|
+
end
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
def parse_rule
|
|
122
|
+
token = advance
|
|
123
|
+
unless token[0] == :name
|
|
124
|
+
raise Error, "ABNF: expected rule name, got #{token[1].inspect}"
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
name = normalize_name(token[1])
|
|
128
|
+
skip_inline_ws
|
|
129
|
+
defined = advance
|
|
130
|
+
unless defined && defined[0] == :definedas
|
|
131
|
+
raise Error, "ABNF: expected '=' after #{token[1].inspect}"
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
body = parse_alternation
|
|
135
|
+
if defined[1] == "=/"
|
|
136
|
+
@rules[name] = @rules[name] ? "#{@rules[name]} / #{body}" : body
|
|
137
|
+
elsif @rules.key?(name)
|
|
138
|
+
raise Error, "ABNF: duplicate rule #{name.inspect}"
|
|
139
|
+
else
|
|
140
|
+
@rules[name] = body
|
|
141
|
+
end
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
def parse_alternation
|
|
145
|
+
parts = [parse_concatenation]
|
|
146
|
+
while peek&.[](0) == :punct && peek[1] == "/"
|
|
147
|
+
advance
|
|
148
|
+
parts << parse_concatenation
|
|
149
|
+
end
|
|
150
|
+
parts.join(" / ")
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
def parse_concatenation
|
|
154
|
+
parts = [parse_repetition]
|
|
155
|
+
while continuation?
|
|
156
|
+
parts << parse_repetition
|
|
157
|
+
end
|
|
158
|
+
parts.join(" ")
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
def continuation?
|
|
162
|
+
skip_inline_ws
|
|
163
|
+
token = peek
|
|
164
|
+
return false if token.nil?
|
|
165
|
+
|
|
166
|
+
if token[0] == :crlf
|
|
167
|
+
unless peek(1)&.[](0) == :ws
|
|
168
|
+
return false
|
|
169
|
+
end
|
|
170
|
+
|
|
171
|
+
advance
|
|
172
|
+
advance
|
|
173
|
+
return true
|
|
174
|
+
end
|
|
175
|
+
element_start?(token)
|
|
176
|
+
end
|
|
177
|
+
|
|
178
|
+
def element_start?(token)
|
|
179
|
+
ELEMENT_START_TYPES.include?(token[0]) ||
|
|
180
|
+
(token[0] == :punct && ELEMENT_START_PUNCT.include?(token[1]))
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
def parse_repetition
|
|
184
|
+
skip_inline_ws
|
|
185
|
+
if peek&.[](0) == :num
|
|
186
|
+
prefix = advance[1]
|
|
187
|
+
if peek&.[](0) == :punct && peek[1] == "*"
|
|
188
|
+
advance
|
|
189
|
+
prefix += "*"
|
|
190
|
+
prefix += advance[1] if peek&.[](0) == :num
|
|
191
|
+
end
|
|
192
|
+
return "#{prefix}#{parse_element}"
|
|
193
|
+
elsif peek&.[](0) == :punct && peek[1] == "*"
|
|
194
|
+
advance
|
|
195
|
+
prefix = "*"
|
|
196
|
+
prefix += advance[1] if peek&.[](0) == :num
|
|
197
|
+
return "#{prefix}#{parse_element}"
|
|
198
|
+
end
|
|
199
|
+
parse_element
|
|
200
|
+
end
|
|
201
|
+
|
|
202
|
+
def parse_element
|
|
203
|
+
skip_inline_ws
|
|
204
|
+
token = advance
|
|
205
|
+
raise Error, "ABNF: unexpected end of rule body" if token.nil?
|
|
206
|
+
|
|
207
|
+
case token[0]
|
|
208
|
+
when :cistr then emit_string(token[1][1..-2], fold: true)
|
|
209
|
+
when :sstr then emit_string(token[1][3..-2], fold: false)
|
|
210
|
+
when :numval then emit_numval(token[1])
|
|
211
|
+
when :name then normalize_name(token[1])
|
|
212
|
+
when :prose
|
|
213
|
+
raise Error,
|
|
214
|
+
"ABNF: prose-val #{token[1].inspect} cannot be imported — " \
|
|
215
|
+
"replace it with a machine-parseable rule"
|
|
216
|
+
when :punct
|
|
217
|
+
case token[1]
|
|
218
|
+
when "("
|
|
219
|
+
body = parse_alternation
|
|
220
|
+
expect_punct(")")
|
|
221
|
+
"(#{body})"
|
|
222
|
+
when "["
|
|
223
|
+
body = parse_alternation
|
|
224
|
+
expect_punct("]")
|
|
225
|
+
"[#{body}]"
|
|
226
|
+
else
|
|
227
|
+
raise Error, "ABNF: unexpected #{token[1].inspect}"
|
|
228
|
+
end
|
|
229
|
+
else
|
|
230
|
+
raise Error, "ABNF: unexpected token #{token[0]}"
|
|
231
|
+
end
|
|
232
|
+
end
|
|
233
|
+
|
|
234
|
+
def expect_punct(char)
|
|
235
|
+
skip_inline_ws
|
|
236
|
+
token = advance
|
|
237
|
+
unless token && token[0] == :punct && token[1] == char
|
|
238
|
+
raise Error, "ABNF: expected #{char.inspect}"
|
|
239
|
+
end
|
|
240
|
+
end
|
|
241
|
+
|
|
242
|
+
def emit_string(body, fold:)
|
|
243
|
+
escaped = body.gsub("\\", "\\\\").gsub('"', '\\"')
|
|
244
|
+
fold ? "%i\"#{escaped}\"" : "\"#{escaped}\""
|
|
245
|
+
end
|
|
246
|
+
|
|
247
|
+
def emit_numval(spec)
|
|
248
|
+
kind = spec[1]
|
|
249
|
+
body = spec[2..]
|
|
250
|
+
if body.include?("-")
|
|
251
|
+
lo, hi = body.split("-")
|
|
252
|
+
"%x#{convert_value(kind, lo)}-#{convert_value(kind, hi)}"
|
|
253
|
+
elsif body.include?(".")
|
|
254
|
+
body.split(".").map { |value| "%x#{convert_value(kind, value)}" }.join(" ")
|
|
255
|
+
else
|
|
256
|
+
"%x#{convert_value(kind, body)}"
|
|
257
|
+
end
|
|
258
|
+
end
|
|
259
|
+
|
|
260
|
+
def convert_value(kind, value)
|
|
261
|
+
case kind
|
|
262
|
+
when "x" then format("%02x", value.to_i(16))
|
|
263
|
+
when "d" then format("%02x", value.to_i)
|
|
264
|
+
when "b" then format("%02x", value.to_i(2))
|
|
265
|
+
when "o" then format("%02x", value.to_i(8))
|
|
266
|
+
else raise Error, "ABNF: unknown numeric value kind #{kind.inspect}"
|
|
267
|
+
end
|
|
268
|
+
end
|
|
269
|
+
|
|
270
|
+
def normalize_name(name)
|
|
271
|
+
name.downcase.tr("-", "_")
|
|
272
|
+
end
|
|
273
|
+
|
|
274
|
+
def emit
|
|
275
|
+
preamble = CORE_RULES.filter_map do |name, body|
|
|
276
|
+
normalized = normalize_name(name)
|
|
277
|
+
next if @rules.key?(normalized)
|
|
278
|
+
|
|
279
|
+
" #{normalized} = #{body}"
|
|
280
|
+
end
|
|
281
|
+
rules = @rules.map { |name, body| " #{name} = #{body}" }
|
|
282
|
+
lines = [
|
|
283
|
+
"# Imported from ABNF (RFC 5234/7405).",
|
|
284
|
+
"# Notes:",
|
|
285
|
+
"# - bare ABNF strings are case-insensitive; imported as %i\"...\"",
|
|
286
|
+
"# - ABNF alternation is unordered; PARG's is ordered — the compiler",
|
|
287
|
+
"# lints order-dependent branches",
|
|
288
|
+
"grammar imported_abnf version \"0.0.0\" {",
|
|
289
|
+
]
|
|
290
|
+
lines.concat(preamble)
|
|
291
|
+
lines << "" unless preamble.empty?
|
|
292
|
+
lines.concat(rules)
|
|
293
|
+
lines << "}"
|
|
294
|
+
"#{lines.join("\n")}\n"
|
|
295
|
+
end
|
|
296
|
+
end
|
|
297
|
+
end
|
|
298
|
+
end
|
|
299
|
+
end
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Parsanol
|
|
4
|
+
module PARG
|
|
5
|
+
module Import
|
|
6
|
+
# ISO 14977 EBNF importer.
|
|
7
|
+
#
|
|
8
|
+
# Semantic conversions (recorded in the emitted header):
|
|
9
|
+
# - meta-identifiers are case-insensitive in ISO EBNF -> downcased
|
|
10
|
+
# - terminals are exact -> emitted as plain PARG strings
|
|
11
|
+
# - sequence "," -> juxtaposition; "|" -> ordered "/"
|
|
12
|
+
# - { x } (zero-or-more) -> *( x )
|
|
13
|
+
# - syntactic exceptions "term - exception" are approximated as
|
|
14
|
+
# !( exception ) term — exact whenever the exception matches a
|
|
15
|
+
# prefix of the term's match; each occurrence is commented
|
|
16
|
+
# - special sequences ? ... ? are rejected
|
|
17
|
+
class Ebnf
|
|
18
|
+
COMMENT = /\(\*(?:[^*]|\*(?!\)))*\*\)/m
|
|
19
|
+
|
|
20
|
+
TOKEN = /
|
|
21
|
+
(?<ws>\s+)
|
|
22
|
+
| (?<squote>'[^']*')
|
|
23
|
+
| (?<dquote>"[^"]*")
|
|
24
|
+
| (?<special>\?[^?\n]*\?)
|
|
25
|
+
| (?<name>[A-Za-z][A-Za-z0-9]*)
|
|
26
|
+
| (?<punct>[=;,|()\[\]{}-])
|
|
27
|
+
/x
|
|
28
|
+
|
|
29
|
+
class << self
|
|
30
|
+
def call(text)
|
|
31
|
+
new(text).call
|
|
32
|
+
end
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
def initialize(text)
|
|
36
|
+
@text = text.gsub(COMMENT, "")
|
|
37
|
+
@tokens = []
|
|
38
|
+
@pos = 0
|
|
39
|
+
@rules = {}
|
|
40
|
+
@notes = []
|
|
41
|
+
@current_name = nil
|
|
42
|
+
scan
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
def call
|
|
46
|
+
parse_rules
|
|
47
|
+
emit
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
private
|
|
51
|
+
|
|
52
|
+
def scan
|
|
53
|
+
pos = 0
|
|
54
|
+
until pos >= @text.length
|
|
55
|
+
match = TOKEN.match(@text, pos)
|
|
56
|
+
if match.nil? || match.begin(0) != pos
|
|
57
|
+
raise Error,
|
|
58
|
+
"EBNF: unexpected #{@text[pos].inspect} at offset #{pos}"
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
name = TOKEN.names.find { |n| match[n] }
|
|
62
|
+
unless name == "ws"
|
|
63
|
+
if name == "special"
|
|
64
|
+
raise Error,
|
|
65
|
+
"EBNF: special sequence #{match[0].inspect} cannot be " \
|
|
66
|
+
"imported — express it with PARG syntax"
|
|
67
|
+
end
|
|
68
|
+
@tokens << [name.to_sym, match[0], pos]
|
|
69
|
+
end
|
|
70
|
+
pos = match.end(0)
|
|
71
|
+
end
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
def peek(offset = 0)
|
|
75
|
+
@tokens[@pos + offset]
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
def advance
|
|
79
|
+
token = @tokens[@pos]
|
|
80
|
+
@pos += 1
|
|
81
|
+
token
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def parse_rules
|
|
85
|
+
until peek.nil?
|
|
86
|
+
token = advance
|
|
87
|
+
unless token[0] == :name
|
|
88
|
+
raise Error, "EBNF: expected rule name, got #{token[1].inspect}"
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
name = normalize_name(token[1])
|
|
92
|
+
equals = advance
|
|
93
|
+
unless equals && equals[0] == :punct && equals[1] == "="
|
|
94
|
+
raise Error, "EBNF: expected '=' after #{token[1].inspect}"
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
if @rules.key?(name)
|
|
98
|
+
raise Error, "EBNF: duplicate rule #{name.inspect}"
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
@current_name = name
|
|
102
|
+
@rules[name] = parse_alternation
|
|
103
|
+
terminator = advance
|
|
104
|
+
unless terminator && terminator[0] == :punct && terminator[1] == ";"
|
|
105
|
+
raise Error, "EBNF: expected ';' after rule #{name.inspect}"
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
def parse_alternation
|
|
111
|
+
parts = [parse_sequence]
|
|
112
|
+
while peek&.[](0) == :punct && peek[1] == "|"
|
|
113
|
+
advance
|
|
114
|
+
parts << parse_sequence
|
|
115
|
+
end
|
|
116
|
+
parts.join(" / ")
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
def parse_sequence
|
|
120
|
+
parts = [parse_term]
|
|
121
|
+
while peek&.[](0) == :punct && peek[1] == ","
|
|
122
|
+
advance
|
|
123
|
+
parts << parse_term
|
|
124
|
+
end
|
|
125
|
+
parts.join(" ")
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
def parse_term
|
|
129
|
+
factor = parse_factor
|
|
130
|
+
if peek&.[](0) == :punct && peek[1] == "-"
|
|
131
|
+
advance
|
|
132
|
+
exception = parse_factor
|
|
133
|
+
@notes << "rule '#{@current_name}': syntactic exception " \
|
|
134
|
+
"'#{factor} - #{exception}' approximated as " \
|
|
135
|
+
"'!(#{exception}) #{factor}'"
|
|
136
|
+
"!(#{exception}) #{factor}"
|
|
137
|
+
else
|
|
138
|
+
factor
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
def parse_factor
|
|
143
|
+
token = advance
|
|
144
|
+
raise Error, "EBNF: unexpected end of rule body" if token.nil?
|
|
145
|
+
|
|
146
|
+
case token[0]
|
|
147
|
+
when :squote, :dquote then emit_string(token[1][1..-2])
|
|
148
|
+
when :name then normalize_name(token[1])
|
|
149
|
+
when :punct
|
|
150
|
+
case token[1]
|
|
151
|
+
when "("
|
|
152
|
+
body = parse_alternation
|
|
153
|
+
expect_punct(")")
|
|
154
|
+
"(#{body})"
|
|
155
|
+
when "["
|
|
156
|
+
body = parse_alternation
|
|
157
|
+
expect_punct("]")
|
|
158
|
+
"[#{body}]"
|
|
159
|
+
when "{"
|
|
160
|
+
body = parse_alternation
|
|
161
|
+
expect_punct("}")
|
|
162
|
+
"*(#{body})"
|
|
163
|
+
else
|
|
164
|
+
raise Error, "EBNF: unexpected #{token[1].inspect}"
|
|
165
|
+
end
|
|
166
|
+
else
|
|
167
|
+
raise Error, "EBNF: unexpected token #{token[0]}"
|
|
168
|
+
end
|
|
169
|
+
end
|
|
170
|
+
|
|
171
|
+
def expect_punct(char)
|
|
172
|
+
token = advance
|
|
173
|
+
unless token && token[0] == :punct && token[1] == char
|
|
174
|
+
raise Error, "EBNF: expected #{char.inspect}"
|
|
175
|
+
end
|
|
176
|
+
end
|
|
177
|
+
|
|
178
|
+
def emit_string(body)
|
|
179
|
+
body.gsub("\\", "\\\\").gsub('"', '\\"').then { |escaped| "\"#{escaped}\"" }
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
def normalize_name(name)
|
|
183
|
+
name.downcase
|
|
184
|
+
end
|
|
185
|
+
|
|
186
|
+
def emit
|
|
187
|
+
lines = [
|
|
188
|
+
"# Imported from ISO 14977 EBNF.",
|
|
189
|
+
"# Notes:",
|
|
190
|
+
"# - meta-identifiers were case-insensitive; downcased",
|
|
191
|
+
]
|
|
192
|
+
@notes.each { |note| lines << "# - #{note}" }
|
|
193
|
+
lines << "grammar imported_ebnf version \"0.0.0\" {"
|
|
194
|
+
lines.concat(@rules.map { |name, body| " #{name} = #{body}" })
|
|
195
|
+
lines << "}"
|
|
196
|
+
"#{lines.join("\n")}\n"
|
|
197
|
+
end
|
|
198
|
+
end
|
|
199
|
+
end
|
|
200
|
+
end
|
|
201
|
+
end
|