parsanol 1.3.56-arm-linux → 1.3.57-arm-linux

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. checksums.yaml +4 -4
  2. data/exe/parsanol +7 -0
  3. data/lib/parsanol/3.2/parsanol_native.so +0 -0
  4. data/lib/parsanol/3.3/parsanol_native.so +0 -0
  5. data/lib/parsanol/3.4/parsanol_native.so +0 -0
  6. data/lib/parsanol/4.0/parsanol_native.so +0 -0
  7. data/lib/parsanol/native/libparsanol.so +0 -0
  8. data/lib/parsanol/native/serializer.rb +5 -2
  9. data/lib/parsanol/parg/artifact.rb +213 -0
  10. data/lib/parsanol/parg/authoring.rb +110 -0
  11. data/lib/parsanol/parg/bindings.rb +129 -0
  12. data/lib/parsanol/parg/cli.rb +251 -0
  13. data/lib/parsanol/parg/compiler.rb +465 -0
  14. data/lib/parsanol/parg/derive.rb +28 -0
  15. data/lib/parsanol/parg/document.rb +69 -0
  16. data/lib/parsanol/parg/error.rb +18 -0
  17. data/lib/parsanol/parg/frontend.rb +173 -0
  18. data/lib/parsanol/parg/import.rb +34 -0
  19. data/lib/parsanol/parg/importers/abnf.rb +299 -0
  20. data/lib/parsanol/parg/importers/ebnf.rb +201 -0
  21. data/lib/parsanol/parg/importers/pest.rb +314 -0
  22. data/lib/parsanol/parg/imports.rb +87 -0
  23. data/lib/parsanol/parg/lexer.rb +72 -0
  24. data/lib/parsanol/parg/lints.rb +171 -0
  25. data/lib/parsanol/parg/lsp.rb +208 -0
  26. data/lib/parsanol/parg/lutaml.rb +67 -0
  27. data/lib/parsanol/parg/node.rb +20 -0
  28. data/lib/parsanol/parg/parser.rb +458 -0
  29. data/lib/parsanol/parg/preprocess.rb +36 -0
  30. data/lib/parsanol/parg/render.rb +44 -0
  31. data/lib/parsanol/parg/selfhost.rb +49 -0
  32. data/lib/parsanol/parg/visitor.rb +64 -0
  33. data/lib/parsanol/parg.rb +38 -0
  34. data/lib/parsanol/version.rb +1 -1
  35. data/lib/parsanol/vm.rb +7 -5
  36. data/lib/parsanol.rb +3 -0
  37. data/parsanol.gemspec +3 -1
  38. metadata +31 -4
@@ -0,0 +1,173 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Parsanol
4
+ module PARG
5
+ # F10 phase 2: the artifact-driven front end.
6
+ #
7
+ # The parg artifact is the parser of record: SelfHost.validate accepts
8
+ # or rejects the source with the native engine before anything else.
9
+ # Document construction then routes each top-level item —
10
+ # line-classified, since PARG's document grammar is line-oriented
11
+ # (PN 1) — through the reference semantics. Acceptance gate: the
12
+ # compiled envelope checksum must equal the reference compiler's for
13
+ # every grammar in the corpus.
14
+ module Frontend
15
+ NAMED_SECTION = /\A(render|bindings)\s+(\w+)\s*\{/
16
+ PREPROCESS_SECTION = /\Apreprocess\s+(\w+)\s*\{/
17
+ DERIVE_LINE = /\Aderive\s+(\w+)\s+"((?:[^"\\]|\\.)*)"/
18
+ TEST_SECTION = /\Atest\s*\{/
19
+ RULE_LINE = /\A([a-z_][a-z0-9_]*)\s*=\s*/
20
+ ENTRY_LINE = /\Aentry\s+(\w+)\s*:\s*(\w+)/
21
+ USE_LINE = /\Ause\s+(\w+)/
22
+ GRAMMAR_LINE = /\Agrammar\s+(\w+)\s+version\s+"([^"]+)"/
23
+
24
+ module_function
25
+
26
+ def parse(source)
27
+ SelfHost.validate(source) unless SelfHost.available? && !SelfHost.valid?(source)
28
+
29
+ document = Document.new
30
+ document.source = source
31
+ state = { section: nil, section_name: nil, section_lines: [],
32
+ depth: 0, doc_comments: [], rule_lines: {},
33
+ entry_lines: [] }
34
+ deferred = []
35
+ preprocess_texts = []
36
+ document.own_entries ||= []
37
+
38
+ source.each_line do |line|
39
+ stripped = line.strip
40
+ if stripped.start_with?("##")
41
+ state[:doc_comments] << stripped.sub(/\A##\s?/, "")
42
+ next
43
+ end
44
+ next if stripped.empty? || stripped.start_with?("#")
45
+
46
+ if state[:section]
47
+ advance_section(document, state, line, deferred, preprocess_texts)
48
+ else
49
+ route_top_level(document, stripped, line, state,
50
+ preprocess_texts)
51
+ end
52
+ end
53
+
54
+ replay_deferred(document, deferred, state[:rule_lines],
55
+ state[:entry_lines], preprocess_texts)
56
+ document
57
+ end
58
+
59
+ # Consumes one line inside a deferred/closed-by-brace section.
60
+ def advance_section(document, state, line, deferred, preprocess_texts)
61
+ section = state[:section]
62
+ state[:depth] += line.scan("{").count - line.scan("}").count
63
+ if state[:depth].positive?
64
+ preprocess_texts.last << line if section == "preprocess"
65
+ state[:section_lines] << line
66
+ return
67
+ end
68
+
69
+ case section
70
+ when "test"
71
+ deferred << ["test {", state[:section_lines]]
72
+ when "bindings"
73
+ deferred << ["bindings #{state[:section_name]} {", state[:section_lines]]
74
+ when "preprocess"
75
+ preprocess_texts.last << "}\n"
76
+ close_section(document, section, state[:section_name],
77
+ state[:section_lines])
78
+ else
79
+ close_section(document, section, state[:section_name],
80
+ state[:section_lines])
81
+ end
82
+ state[:section] = nil
83
+ state[:section_name] = nil
84
+ state[:section_lines] = []
85
+ end
86
+
87
+ def route_top_level(document, stripped, line, state, preprocess_texts)
88
+ doc_comments = state[:doc_comments]
89
+
90
+ case stripped
91
+ when USE_LINE
92
+ document.uses << Regexp.last_match(1)
93
+ doc_comments = []
94
+ when GRAMMAR_LINE
95
+ document.grammar_name = Regexp.last_match(1)
96
+ document.version = Regexp.last_match(2)
97
+ doc_comments = []
98
+ when NAMED_SECTION
99
+ state[:section] = Regexp.last_match(1)
100
+ state[:section_name] = Regexp.last_match(2)
101
+ state[:section_lines] = []
102
+ state[:depth] = 1
103
+ doc_comments = []
104
+ when PREPROCESS_SECTION
105
+ state[:section] = "preprocess"
106
+ state[:section_name] = Regexp.last_match(1)
107
+ state[:section_lines] = []
108
+ preprocess_texts << "preprocess #{state[:section_name]} {\n"
109
+ state[:depth] = 1
110
+ doc_comments = []
111
+ when TEST_SECTION
112
+ state[:section] = "test"
113
+ state[:section_lines] = []
114
+ state[:depth] = 1
115
+ doc_comments = []
116
+ when ENTRY_LINE
117
+ entry_name = Regexp.last_match(1)
118
+ document.entries[entry_name] = Regexp.last_match(2)
119
+ document.own_entries << entry_name
120
+ state[:entry_lines] << line
121
+ doc_comments = []
122
+ when DERIVE_LINE
123
+ document.derive[Regexp.last_match(1)] =
124
+ Regexp.last_match(2).gsub(/\\(.)/, '\1')
125
+ doc_comments = []
126
+ when RULE_LINE
127
+ rule_name = Regexp.last_match(1)
128
+ document.docs[rule_name] = doc_comments.join("\n") unless doc_comments.empty?
129
+ doc_comments = []
130
+ state[:rule_lines][rule_name] = line
131
+ merge_rule(document, line)
132
+ else
133
+ doc_comments = []
134
+ end
135
+ state[:doc_comments] = doc_comments
136
+ end
137
+
138
+ # Deferred test/bindings sections re-parse against a mini preamble
139
+ # built from the collected rule/entry/preprocess text.
140
+ def replay_deferred(document, deferred, rule_lines, entry_lines,
141
+ preprocess_texts)
142
+ return if deferred.empty?
143
+
144
+ preamble = "grammar Mini version \"1\" {\n" \
145
+ "#{rule_lines.values.join}#{entry_lines.join}" \
146
+ "#{preprocess_texts.join}}\n"
147
+ deferred.each do |(opener, lines)|
148
+ text = "#{preamble[0..-2]}#{opener}\n#{lines.join}}\n"
149
+ mini = Parser.new(text).parse
150
+ if opener.start_with?("test")
151
+ document.tests.concat(mini.tests)
152
+ else
153
+ document.bindings.merge!(mini.bindings)
154
+ end
155
+ end
156
+ end
157
+
158
+ def merge_rule(document, line)
159
+ mini = Parser.new("grammar Mini version \"1\" {\n#{line}}\n").parse
160
+ document.rules.merge!(mini.rules)
161
+ mini.docs.each { |rule, text| document.docs[rule] = text }
162
+ end
163
+
164
+ def close_section(document, kind, name, lines)
165
+ mini = Parser.new("grammar Mini version \"1\" {\nb = \"q\"\n}#{kind} #{name} {\n#{lines.join}}\n").parse
166
+ document.bindings.merge!(mini.bindings)
167
+ document.preprocess.merge!(mini.preprocess)
168
+ document.render.merge!(mini.render)
169
+ document.derive.merge!(mini.derive)
170
+ end
171
+ end
172
+ end
173
+ end
@@ -0,0 +1,34 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Parsanol
4
+ module PARG
5
+ # Foreign-grammar importers. Each converts a foreign grammar notation to
6
+ # PARG source text — the .parg file is committed as the single source of
7
+ # truth, then compiled like any hand-written grammar. Import results are
8
+ # self-checked: the emitted source must round-trip through PARG::Parser.
9
+ module Import
10
+ class Error < PARG::Error; end
11
+
12
+ autoload :Abnf, "parsanol/parg/importers/abnf"
13
+ autoload :Ebnf, "parsanol/parg/importers/ebnf"
14
+ autoload :Pest, "parsanol/parg/importers/pest"
15
+
16
+ KINDS = %i[abnf ebnf pest].freeze
17
+
18
+ module_function
19
+
20
+ def import(kind, text)
21
+ source = case kind
22
+ when :abnf then Abnf.call(text)
23
+ when :ebnf then Ebnf.call(text)
24
+ when :pest then Pest.call(text)
25
+ else
26
+ raise Error,
27
+ "unknown import kind #{kind.inspect} (supported: #{KINDS.map(&:inspect).join(', ')})"
28
+ end
29
+ Parser.new(source).parse
30
+ source
31
+ end
32
+ end
33
+ end
34
+ end
@@ -0,0 +1,299 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Parsanol
4
+ module PARG
5
+ module Import
6
+ # RFC 5234 (+ RFC 7405 case prefixes) ABNF importer.
7
+ #
8
+ # Semantic conversions (recorded in the emitted header):
9
+ # - bare ABNF strings are CASE-INSENSITIVE -> emitted as %i"..."
10
+ # - %s"..." (case-sensitive) -> emitted as plain "..."
11
+ # - ABNF alternation is unordered; PARG's is ordered. The PARG compiler's
12
+ # first-set lint flags order-dependent branches after import.
13
+ # - rule names are case-insensitive in ABNF -> normalized to snake_case
14
+ # - prose-vals (<...>) are rejected: they are not machine-parseable
15
+ class Abnf
16
+ TOKEN = /
17
+ (?<ws>[ \t]+)
18
+ | (?<comment>;[^\n]*)
19
+ | (?<crlf>\r?\n)
20
+ | (?<sstr>%s"(?:[^"\\]|\\.)*")
21
+ | (?<cistr>"(?:[^"\\]|\\.)*")
22
+ | (?<numval>%[xbdo][0-9A-Za-z]+(?:-[0-9A-Za-z]+|(?:\.[0-9A-Za-z]+)+)?)
23
+ | (?<prose><[^>\n]*>)
24
+ | (?<definedas>=\/|=)
25
+ | (?<name>[A-Za-z][A-Za-z0-9-]*)
26
+ | (?<num>\d+)
27
+ | (?<punct>[*\/()\[\]])
28
+ /x
29
+
30
+ INLINE_WS = %i[ws comment].freeze
31
+ LEADING = %i[ws comment crlf].freeze
32
+ ELEMENT_START_TYPES = %i[name cistr sstr numval num].freeze
33
+ ELEMENT_START_PUNCT = ["*", "(", "["].freeze
34
+
35
+ CORE_RULES = {
36
+ "ALPHA" => "%x41-5A / %x61-7A",
37
+ "BIT" => '"0" / "1"',
38
+ "CHAR" => "%x01-7F",
39
+ "CR" => "%x0D",
40
+ "CRLF" => "%x0D %x0A",
41
+ "CTL" => "%x00-1F / %x7F",
42
+ "DIGIT" => "%x30-39",
43
+ "DQUOTE" => "%x22",
44
+ "HEXDIG" => '%x30-39 / %i"a" / %i"b" / %i"c" / %i"d" / %i"e" / %i"f"',
45
+ "HTAB" => "%x09",
46
+ "LF" => "%x0A",
47
+ "OCTET" => "%x00-FF",
48
+ "SP" => "%x20",
49
+ "VCHAR" => "%x21-7E",
50
+ "WSP" => "%x20 / %x09",
51
+ }.freeze
52
+
53
+ class << self
54
+ def call(text)
55
+ new(text).call
56
+ end
57
+ end
58
+
59
+ def initialize(text)
60
+ @text = text
61
+ @tokens = []
62
+ @pos = 0
63
+ @rules = {}
64
+ scan
65
+ end
66
+
67
+ def call
68
+ parse_rules
69
+ emit
70
+ end
71
+
72
+ private
73
+
74
+ def scan
75
+ pos = 0
76
+ until pos >= @text.length
77
+ match = TOKEN.match(@text, pos)
78
+ if match.nil? || match.begin(0) != pos
79
+ raise Error,
80
+ "ABNF: unexpected #{@text[pos].inspect} at offset #{pos}"
81
+ end
82
+
83
+ name = TOKEN.names.find { |n| match[n] }
84
+ @tokens << [name.to_sym, match[0], pos]
85
+ pos = match.end(0)
86
+ end
87
+ end
88
+
89
+ def peek(offset = 0)
90
+ @tokens[@pos + offset]
91
+ end
92
+
93
+ def advance
94
+ token = @tokens[@pos]
95
+ @pos += 1
96
+ token
97
+ end
98
+
99
+ def skip_inline_ws
100
+ advance while peek && INLINE_WS.include?(peek[0])
101
+ end
102
+
103
+ def parse_rules
104
+ loop do
105
+ skip_leading
106
+ break if peek.nil?
107
+
108
+ parse_rule
109
+ end
110
+ end
111
+
112
+ def skip_leading
113
+ loop do
114
+ token = peek
115
+ break unless token && LEADING.include?(token[0])
116
+
117
+ advance
118
+ end
119
+ end
120
+
121
+ def parse_rule
122
+ token = advance
123
+ unless token[0] == :name
124
+ raise Error, "ABNF: expected rule name, got #{token[1].inspect}"
125
+ end
126
+
127
+ name = normalize_name(token[1])
128
+ skip_inline_ws
129
+ defined = advance
130
+ unless defined && defined[0] == :definedas
131
+ raise Error, "ABNF: expected '=' after #{token[1].inspect}"
132
+ end
133
+
134
+ body = parse_alternation
135
+ if defined[1] == "=/"
136
+ @rules[name] = @rules[name] ? "#{@rules[name]} / #{body}" : body
137
+ elsif @rules.key?(name)
138
+ raise Error, "ABNF: duplicate rule #{name.inspect}"
139
+ else
140
+ @rules[name] = body
141
+ end
142
+ end
143
+
144
+ def parse_alternation
145
+ parts = [parse_concatenation]
146
+ while peek&.[](0) == :punct && peek[1] == "/"
147
+ advance
148
+ parts << parse_concatenation
149
+ end
150
+ parts.join(" / ")
151
+ end
152
+
153
+ def parse_concatenation
154
+ parts = [parse_repetition]
155
+ while continuation?
156
+ parts << parse_repetition
157
+ end
158
+ parts.join(" ")
159
+ end
160
+
161
+ def continuation?
162
+ skip_inline_ws
163
+ token = peek
164
+ return false if token.nil?
165
+
166
+ if token[0] == :crlf
167
+ unless peek(1)&.[](0) == :ws
168
+ return false
169
+ end
170
+
171
+ advance
172
+ advance
173
+ return true
174
+ end
175
+ element_start?(token)
176
+ end
177
+
178
+ def element_start?(token)
179
+ ELEMENT_START_TYPES.include?(token[0]) ||
180
+ (token[0] == :punct && ELEMENT_START_PUNCT.include?(token[1]))
181
+ end
182
+
183
+ def parse_repetition
184
+ skip_inline_ws
185
+ if peek&.[](0) == :num
186
+ prefix = advance[1]
187
+ if peek&.[](0) == :punct && peek[1] == "*"
188
+ advance
189
+ prefix += "*"
190
+ prefix += advance[1] if peek&.[](0) == :num
191
+ end
192
+ return "#{prefix}#{parse_element}"
193
+ elsif peek&.[](0) == :punct && peek[1] == "*"
194
+ advance
195
+ prefix = "*"
196
+ prefix += advance[1] if peek&.[](0) == :num
197
+ return "#{prefix}#{parse_element}"
198
+ end
199
+ parse_element
200
+ end
201
+
202
+ def parse_element
203
+ skip_inline_ws
204
+ token = advance
205
+ raise Error, "ABNF: unexpected end of rule body" if token.nil?
206
+
207
+ case token[0]
208
+ when :cistr then emit_string(token[1][1..-2], fold: true)
209
+ when :sstr then emit_string(token[1][3..-2], fold: false)
210
+ when :numval then emit_numval(token[1])
211
+ when :name then normalize_name(token[1])
212
+ when :prose
213
+ raise Error,
214
+ "ABNF: prose-val #{token[1].inspect} cannot be imported — " \
215
+ "replace it with a machine-parseable rule"
216
+ when :punct
217
+ case token[1]
218
+ when "("
219
+ body = parse_alternation
220
+ expect_punct(")")
221
+ "(#{body})"
222
+ when "["
223
+ body = parse_alternation
224
+ expect_punct("]")
225
+ "[#{body}]"
226
+ else
227
+ raise Error, "ABNF: unexpected #{token[1].inspect}"
228
+ end
229
+ else
230
+ raise Error, "ABNF: unexpected token #{token[0]}"
231
+ end
232
+ end
233
+
234
+ def expect_punct(char)
235
+ skip_inline_ws
236
+ token = advance
237
+ unless token && token[0] == :punct && token[1] == char
238
+ raise Error, "ABNF: expected #{char.inspect}"
239
+ end
240
+ end
241
+
242
+ def emit_string(body, fold:)
243
+ escaped = body.gsub("\\", "\\\\").gsub('"', '\\"')
244
+ fold ? "%i\"#{escaped}\"" : "\"#{escaped}\""
245
+ end
246
+
247
+ def emit_numval(spec)
248
+ kind = spec[1]
249
+ body = spec[2..]
250
+ if body.include?("-")
251
+ lo, hi = body.split("-")
252
+ "%x#{convert_value(kind, lo)}-#{convert_value(kind, hi)}"
253
+ elsif body.include?(".")
254
+ body.split(".").map { |value| "%x#{convert_value(kind, value)}" }.join(" ")
255
+ else
256
+ "%x#{convert_value(kind, body)}"
257
+ end
258
+ end
259
+
260
+ def convert_value(kind, value)
261
+ case kind
262
+ when "x" then format("%02x", value.to_i(16))
263
+ when "d" then format("%02x", value.to_i)
264
+ when "b" then format("%02x", value.to_i(2))
265
+ when "o" then format("%02x", value.to_i(8))
266
+ else raise Error, "ABNF: unknown numeric value kind #{kind.inspect}"
267
+ end
268
+ end
269
+
270
+ def normalize_name(name)
271
+ name.downcase.tr("-", "_")
272
+ end
273
+
274
+ def emit
275
+ preamble = CORE_RULES.filter_map do |name, body|
276
+ normalized = normalize_name(name)
277
+ next if @rules.key?(normalized)
278
+
279
+ " #{normalized} = #{body}"
280
+ end
281
+ rules = @rules.map { |name, body| " #{name} = #{body}" }
282
+ lines = [
283
+ "# Imported from ABNF (RFC 5234/7405).",
284
+ "# Notes:",
285
+ "# - bare ABNF strings are case-insensitive; imported as %i\"...\"",
286
+ "# - ABNF alternation is unordered; PARG's is ordered — the compiler",
287
+ "# lints order-dependent branches",
288
+ "grammar imported_abnf version \"0.0.0\" {",
289
+ ]
290
+ lines.concat(preamble)
291
+ lines << "" unless preamble.empty?
292
+ lines.concat(rules)
293
+ lines << "}"
294
+ "#{lines.join("\n")}\n"
295
+ end
296
+ end
297
+ end
298
+ end
299
+ end
@@ -0,0 +1,201 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Parsanol
4
+ module PARG
5
+ module Import
6
+ # ISO 14977 EBNF importer.
7
+ #
8
+ # Semantic conversions (recorded in the emitted header):
9
+ # - meta-identifiers are case-insensitive in ISO EBNF -> downcased
10
+ # - terminals are exact -> emitted as plain PARG strings
11
+ # - sequence "," -> juxtaposition; "|" -> ordered "/"
12
+ # - { x } (zero-or-more) -> *( x )
13
+ # - syntactic exceptions "term - exception" are approximated as
14
+ # !( exception ) term — exact whenever the exception matches a
15
+ # prefix of the term's match; each occurrence is commented
16
+ # - special sequences ? ... ? are rejected
17
+ class Ebnf
18
+ COMMENT = /\(\*(?:[^*]|\*(?!\)))*\*\)/m
19
+
20
+ TOKEN = /
21
+ (?<ws>\s+)
22
+ | (?<squote>'[^']*')
23
+ | (?<dquote>"[^"]*")
24
+ | (?<special>\?[^?\n]*\?)
25
+ | (?<name>[A-Za-z][A-Za-z0-9]*)
26
+ | (?<punct>[=;,|()\[\]{}-])
27
+ /x
28
+
29
+ class << self
30
+ def call(text)
31
+ new(text).call
32
+ end
33
+ end
34
+
35
+ def initialize(text)
36
+ @text = text.gsub(COMMENT, "")
37
+ @tokens = []
38
+ @pos = 0
39
+ @rules = {}
40
+ @notes = []
41
+ @current_name = nil
42
+ scan
43
+ end
44
+
45
+ def call
46
+ parse_rules
47
+ emit
48
+ end
49
+
50
+ private
51
+
52
+ def scan
53
+ pos = 0
54
+ until pos >= @text.length
55
+ match = TOKEN.match(@text, pos)
56
+ if match.nil? || match.begin(0) != pos
57
+ raise Error,
58
+ "EBNF: unexpected #{@text[pos].inspect} at offset #{pos}"
59
+ end
60
+
61
+ name = TOKEN.names.find { |n| match[n] }
62
+ unless name == "ws"
63
+ if name == "special"
64
+ raise Error,
65
+ "EBNF: special sequence #{match[0].inspect} cannot be " \
66
+ "imported — express it with PARG syntax"
67
+ end
68
+ @tokens << [name.to_sym, match[0], pos]
69
+ end
70
+ pos = match.end(0)
71
+ end
72
+ end
73
+
74
+ def peek(offset = 0)
75
+ @tokens[@pos + offset]
76
+ end
77
+
78
+ def advance
79
+ token = @tokens[@pos]
80
+ @pos += 1
81
+ token
82
+ end
83
+
84
+ def parse_rules
85
+ until peek.nil?
86
+ token = advance
87
+ unless token[0] == :name
88
+ raise Error, "EBNF: expected rule name, got #{token[1].inspect}"
89
+ end
90
+
91
+ name = normalize_name(token[1])
92
+ equals = advance
93
+ unless equals && equals[0] == :punct && equals[1] == "="
94
+ raise Error, "EBNF: expected '=' after #{token[1].inspect}"
95
+ end
96
+
97
+ if @rules.key?(name)
98
+ raise Error, "EBNF: duplicate rule #{name.inspect}"
99
+ end
100
+
101
+ @current_name = name
102
+ @rules[name] = parse_alternation
103
+ terminator = advance
104
+ unless terminator && terminator[0] == :punct && terminator[1] == ";"
105
+ raise Error, "EBNF: expected ';' after rule #{name.inspect}"
106
+ end
107
+ end
108
+ end
109
+
110
+ def parse_alternation
111
+ parts = [parse_sequence]
112
+ while peek&.[](0) == :punct && peek[1] == "|"
113
+ advance
114
+ parts << parse_sequence
115
+ end
116
+ parts.join(" / ")
117
+ end
118
+
119
+ def parse_sequence
120
+ parts = [parse_term]
121
+ while peek&.[](0) == :punct && peek[1] == ","
122
+ advance
123
+ parts << parse_term
124
+ end
125
+ parts.join(" ")
126
+ end
127
+
128
+ def parse_term
129
+ factor = parse_factor
130
+ if peek&.[](0) == :punct && peek[1] == "-"
131
+ advance
132
+ exception = parse_factor
133
+ @notes << "rule '#{@current_name}': syntactic exception " \
134
+ "'#{factor} - #{exception}' approximated as " \
135
+ "'!(#{exception}) #{factor}'"
136
+ "!(#{exception}) #{factor}"
137
+ else
138
+ factor
139
+ end
140
+ end
141
+
142
+ def parse_factor
143
+ token = advance
144
+ raise Error, "EBNF: unexpected end of rule body" if token.nil?
145
+
146
+ case token[0]
147
+ when :squote, :dquote then emit_string(token[1][1..-2])
148
+ when :name then normalize_name(token[1])
149
+ when :punct
150
+ case token[1]
151
+ when "("
152
+ body = parse_alternation
153
+ expect_punct(")")
154
+ "(#{body})"
155
+ when "["
156
+ body = parse_alternation
157
+ expect_punct("]")
158
+ "[#{body}]"
159
+ when "{"
160
+ body = parse_alternation
161
+ expect_punct("}")
162
+ "*(#{body})"
163
+ else
164
+ raise Error, "EBNF: unexpected #{token[1].inspect}"
165
+ end
166
+ else
167
+ raise Error, "EBNF: unexpected token #{token[0]}"
168
+ end
169
+ end
170
+
171
+ def expect_punct(char)
172
+ token = advance
173
+ unless token && token[0] == :punct && token[1] == char
174
+ raise Error, "EBNF: expected #{char.inspect}"
175
+ end
176
+ end
177
+
178
+ def emit_string(body)
179
+ body.gsub("\\", "\\\\").gsub('"', '\\"').then { |escaped| "\"#{escaped}\"" }
180
+ end
181
+
182
+ def normalize_name(name)
183
+ name.downcase
184
+ end
185
+
186
+ def emit
187
+ lines = [
188
+ "# Imported from ISO 14977 EBNF.",
189
+ "# Notes:",
190
+ "# - meta-identifiers were case-insensitive; downcased",
191
+ ]
192
+ @notes.each { |note| lines << "# - #{note}" }
193
+ lines << "grammar imported_ebnf version \"0.0.0\" {"
194
+ lines.concat(@rules.map { |name, body| " #{name} = #{body}" })
195
+ lines << "}"
196
+ "#{lines.join("\n")}\n"
197
+ end
198
+ end
199
+ end
200
+ end
201
+ end