@tabnas/abnf 0.4.15 → 0.4.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/abnf.d.ts +1 -1
- package/dist/abnf.js +1 -1
- package/package.json +5 -3
- package/src/abnf.ts +92 -0
- package/src/bin/tabnas-abnf-cli.ts +222 -0
- package/src/compile.ts +62 -0
- package/src/converter.ts +1155 -0
- package/src/tsconfig.json +16 -0
package/src/converter.ts
ADDED
|
@@ -0,0 +1,1155 @@
|
|
|
1
|
+
/* Copyright (c) 2025-2026 Richard Rodger and other contributors, MIT License */
|
|
2
|
+
|
|
3
|
+
/* converter.ts
|
|
4
|
+
* ABNF -> tabnas grammar spec converter: the RFC 5234 FRONT-END.
|
|
5
|
+
*
|
|
6
|
+
* This file parses ABNF text into the notation-neutral grammar IR that
|
|
7
|
+
* `@tabnas/bnf` compiles. Everything downstream of that IR — desugaring,
|
|
8
|
+
* left-recursion elimination, tail repeats, probe dispatch, literal
|
|
9
|
+
* lifting, token allocation, first-set analysis, chain emission — lives
|
|
10
|
+
* in `@tabnas/bnf` and is shared with the GBNF and EBNF front-ends.
|
|
11
|
+
*
|
|
12
|
+
* ABNF text ──parseAbnf──▶ Grammar ──bnf.emitGrammarSpec──▶ GrammarSpec
|
|
13
|
+
*
|
|
14
|
+
* What stays here is what is genuinely ABNF:
|
|
15
|
+
*
|
|
16
|
+
* - `abnfRules`, the tabnas grammar that reads ABNF syntax itself,
|
|
17
|
+
* and `getAbnfParser`, which installs it on a fresh instance;
|
|
18
|
+
* - the RFC 5234 Appendix B.1 core rules (ALPHA, DIGIT, CRLF, …);
|
|
19
|
+
* - incremental alternatives (`name =/ alt`);
|
|
20
|
+
* - numeric values (`%d65`, `%x41-5A`, `%b1010`, dotted sequences);
|
|
21
|
+
* - case-insensitive quoted strings, the RFC 5234 default.
|
|
22
|
+
*
|
|
23
|
+
* Prose values (`NR = <number>`) are parsed here into the IR's `prose`
|
|
24
|
+
* element; `@tabnas/bnf` resolves them, since a prose terminal naming a
|
|
25
|
+
* built-in lexer token is useful to any notation.
|
|
26
|
+
*/
|
|
27
|
+
|
|
28
|
+
import type { GrammarSpec, Rule } from '@tabnas/parser'
|
|
29
|
+
import { util as engineUtil } from '@tabnas/parser'
|
|
30
|
+
|
|
31
|
+
import {
|
|
32
|
+
emitGrammarSpec as bnfEmitGrammarSpec,
|
|
33
|
+
eliminateLeftRecursion,
|
|
34
|
+
refsIn,
|
|
35
|
+
} from '@tabnas/bnf'
|
|
36
|
+
|
|
37
|
+
import type {
|
|
38
|
+
ConvertOptions,
|
|
39
|
+
SrcSpan,
|
|
40
|
+
Element,
|
|
41
|
+
Sequence,
|
|
42
|
+
Production,
|
|
43
|
+
Grammar,
|
|
44
|
+
} from '@tabnas/bnf'
|
|
45
|
+
|
|
46
|
+
// The IR types keep their historical names in this package's public API.
|
|
47
|
+
type AbnfConvertOptions = ConvertOptions
|
|
48
|
+
type AbnfElement = Element
|
|
49
|
+
type AbnfSequence = Sequence
|
|
50
|
+
type AbnfProduction = Production
|
|
51
|
+
type AbnfGrammar = Grammar
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
// Source span of a token, for the IR (`@tabnas/bnf` SrcSpan). Every
|
|
55
|
+
// field is copied straight off the token: the compiler stores whatever
|
|
56
|
+
// units the front-end's own engine tokens use, precisely so that no
|
|
57
|
+
// arithmetic — and so no off-by-one — happens at this boundary.
|
|
58
|
+
function spanOf(tkn: any): SrcSpan | undefined {
|
|
59
|
+
if (null == tkn || null == tkn.sI) return undefined
|
|
60
|
+
const len = null != tkn.len ? tkn.len : (tkn.src ? String(tkn.src).length : 0)
|
|
61
|
+
return { s: tkn.sI, e: tkn.sI + len, r: tkn.rI, c: tkn.cI }
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
// Deep-copy a production so a caller cannot reach the shared original.
|
|
66
|
+
// Elements are copied too: they are what a consumer would most likely
|
|
67
|
+
// annotate, and a shallow copy would leave them aliased.
|
|
68
|
+
function cloneProduction(prod: AbnfProduction): AbnfProduction {
|
|
69
|
+
const el = (e: any): any => {
|
|
70
|
+
const o: any = { ...e }
|
|
71
|
+
if (e.inner) o.inner = el(e.inner)
|
|
72
|
+
if (e.alts) o.alts = e.alts.map((alt: any[]) => alt.map(el))
|
|
73
|
+
return o
|
|
74
|
+
}
|
|
75
|
+
return { ...prod, alts: prod.alts.map((alt) => alt.map(el)) }
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
// Remove every span from a production and everything under it. Used for
|
|
80
|
+
// the RFC 5234 core rules, which are parsed from a string in this file
|
|
81
|
+
// rather than from the user's grammar.
|
|
82
|
+
function stripSpans(prod: AbnfProduction): void {
|
|
83
|
+
delete (prod as any).sp
|
|
84
|
+
const walk = (el: any): void => {
|
|
85
|
+
if (null == el || 'object' !== typeof el) return
|
|
86
|
+
delete el.sp
|
|
87
|
+
if (el.inner) walk(el.inner)
|
|
88
|
+
if (el.alts) for (const alt of el.alts) for (const e of alt) walk(e)
|
|
89
|
+
}
|
|
90
|
+
for (const alt of prod.alts) for (const el of alt) walk(el)
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
// One span covering two tokens — a group runs from its `(` to its `)`,
|
|
95
|
+
// a bracketed optional from `[` to `]`. Falls back to whichever end is
|
|
96
|
+
// known when the other is not.
|
|
97
|
+
function spanTo(from: any, to: any): SrcSpan | undefined {
|
|
98
|
+
const a = spanOf(from)
|
|
99
|
+
const b = spanOf(to)
|
|
100
|
+
if (null == a) return b
|
|
101
|
+
if (null == b) return a
|
|
102
|
+
return { s: a.s, e: b.e, r: a.r, c: a.c }
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
// Declarative definition of the ABNF grammar itself, expressed as
|
|
107
|
+
// tabnas rules. Each rule names its `open`/`close` alt list and, where
|
|
108
|
+
// necessary, a `bo`/`bc` state hook for AST assembly.
|
|
109
|
+
//
|
|
110
|
+
// Stage 8: incremental alternatives via `name =/ alt` now fold
|
|
111
|
+
// into the earlier production with the same name. Quoted strings
|
|
112
|
+
// default to case-insensitive (ABNF semantics), `%s` / `%i` force
|
|
113
|
+
// sensitivity explicitly, numeric values and repetition prefixes
|
|
114
|
+
// work as in previous stages.
|
|
115
|
+
//
|
|
116
|
+
// Token vocabulary:
|
|
117
|
+
// #DEF `=` (rule-definition operator)
|
|
118
|
+
// #DEFA `=/` (incremental-alternatives operator)
|
|
119
|
+
// #ALT `/` (alternation)
|
|
120
|
+
// #STAR `*` (repetition separator)
|
|
121
|
+
// #NUM decimal repetition count (matched via match.token)
|
|
122
|
+
// #NV `%[xdb]NN[(-NN|(.NN)*)]` numeric value (match.token)
|
|
123
|
+
// #SS `%s` (case-sensitive string prefix)
|
|
124
|
+
// #SI `%i` (case-insensitive string prefix — same as default)
|
|
125
|
+
// #LP `(`
|
|
126
|
+
// #RP `)`
|
|
127
|
+
// #OB `[` (optional-group open)
|
|
128
|
+
// #CB `]` (optional-group close)
|
|
129
|
+
// #TX bare identifier (tabnas default text token)
|
|
130
|
+
// #ST quoted string literal (tabnas default string token)
|
|
131
|
+
// #ZZ end-of-source
|
|
132
|
+
//
|
|
133
|
+
// Grammar:
|
|
134
|
+
// abnf = production*
|
|
135
|
+
// production = IDENT ('=' / '=/') alts
|
|
136
|
+
// alts = seq ('/' seq)*
|
|
137
|
+
// seq = element*
|
|
138
|
+
// element = repetition? atom
|
|
139
|
+
// repetition = NUM '*' NUM / NUM '*' / '*' NUM / '*' / NUM
|
|
140
|
+
// atom = IDENT | STRING | ['%s' | '%i'] STRING | NUMVAL
|
|
141
|
+
// | '(' alts ')' | '[' alts ']' | PROSE
|
|
142
|
+
// prose = '<' *(%x20-3D / %x3F-7E) '>'
|
|
143
|
+
// numval = '%' ('x' / 'd' / 'b') DIGITS [ '-' DIGITS | ('.' DIGITS)* ]
|
|
144
|
+
const abnfRules: Record<
|
|
145
|
+
string,
|
|
146
|
+
{
|
|
147
|
+
bo?: (r: Rule) => void
|
|
148
|
+
bc?: (r: Rule) => void
|
|
149
|
+
open?: any[]
|
|
150
|
+
close?: any[]
|
|
151
|
+
}
|
|
152
|
+
> = {
|
|
153
|
+
// Top-level: accumulates productions into r.node.
|
|
154
|
+
abnf: {
|
|
155
|
+
bo: (r) => { r.node = [] },
|
|
156
|
+
open: [
|
|
157
|
+
{ s: '#ZZ', g: 'empty' },
|
|
158
|
+
{ p: 'prod' },
|
|
159
|
+
],
|
|
160
|
+
close: [{ s: '#ZZ' }],
|
|
161
|
+
},
|
|
162
|
+
|
|
163
|
+
// One production per invocation; tail-recurses (r:'prod') for the
|
|
164
|
+
// next. Inherits its parent's node (the productions array) and
|
|
165
|
+
// appends to it in `bc` once its `alts` child has returned.
|
|
166
|
+
// Production header is `IDENT =` — a bareword rule name followed
|
|
167
|
+
// by the `=` definition operator.
|
|
168
|
+
prod: {
|
|
169
|
+
open: [
|
|
170
|
+
// Standalone definition: name = alts
|
|
171
|
+
{
|
|
172
|
+
s: '#TX #DEF',
|
|
173
|
+
a: (r: Rule) => {
|
|
174
|
+
r.u.name = r.o[0].val
|
|
175
|
+
r.u.nameTkn = r.o[0]
|
|
176
|
+
r.u.incremental = false
|
|
177
|
+
},
|
|
178
|
+
p: 'alts',
|
|
179
|
+
},
|
|
180
|
+
// `<all> = <remove>` — the whole-grammar reset. Prose lexes as its
|
|
181
|
+
// own token (#PV), so it cannot be confused with a rule name or
|
|
182
|
+
// with the `*` repetition operator. The name is kept with its
|
|
183
|
+
// angle brackets, which no #TX rulename can produce, so it can
|
|
184
|
+
// never collide with a production actually called `all`.
|
|
185
|
+
{
|
|
186
|
+
s: '#PV #DEF',
|
|
187
|
+
a: (r: Rule) => {
|
|
188
|
+
r.u.name = (r.o[0].src as string)
|
|
189
|
+
r.u.nameTkn = r.o[0]
|
|
190
|
+
},
|
|
191
|
+
p: 'alts',
|
|
192
|
+
},
|
|
193
|
+
|
|
194
|
+
// Incremental alternatives: name =/ alts
|
|
195
|
+
{
|
|
196
|
+
s: '#TX #DEFA',
|
|
197
|
+
a: (r: Rule) => {
|
|
198
|
+
r.u.name = r.o[0].val
|
|
199
|
+
r.u.nameTkn = r.o[0]
|
|
200
|
+
r.u.incremental = true
|
|
201
|
+
},
|
|
202
|
+
p: 'alts',
|
|
203
|
+
},
|
|
204
|
+
],
|
|
205
|
+
close: [
|
|
206
|
+
// A TX followed by `=` or `=/` means the next production has
|
|
207
|
+
// begun — back up 2 tokens so a fresh `prod` invocation sees
|
|
208
|
+
// them.
|
|
209
|
+
{ s: '#TX #DEF', b: 2, r: 'prod' },
|
|
210
|
+
{ s: '#TX #DEFA', b: 2, r: 'prod' },
|
|
211
|
+
{ s: '#PV #DEF', b: 2, r: 'prod' },
|
|
212
|
+
{ b: 1 },
|
|
213
|
+
],
|
|
214
|
+
bc: (r) => {
|
|
215
|
+
if (r.child && r.child.node !== undefined) {
|
|
216
|
+
const prod: any = {
|
|
217
|
+
name: r.u.name,
|
|
218
|
+
alts: r.child.node,
|
|
219
|
+
// The name, not the body: that is what an outline entry,
|
|
220
|
+
// go-to-definition and a whole-rule diagnostic want, and an
|
|
221
|
+
// ABNF body can run over many folded lines.
|
|
222
|
+
sp: spanOf(r.u.nameTkn),
|
|
223
|
+
}
|
|
224
|
+
if (r.u.incremental) prod.incremental = true
|
|
225
|
+
r.node.push(prod)
|
|
226
|
+
}
|
|
227
|
+
},
|
|
228
|
+
},
|
|
229
|
+
|
|
230
|
+
// A list of alternative sequences separated by `/` (ABNF
|
|
231
|
+
// alternation). Owns its own array (`bo` resets it) and pushes
|
|
232
|
+
// each seq result in `bc`.
|
|
233
|
+
alts: {
|
|
234
|
+
bo: (r) => { r.node = [] },
|
|
235
|
+
open: [{ p: 'seq' }],
|
|
236
|
+
close: [
|
|
237
|
+
{ s: '#ALT', p: 'seq' },
|
|
238
|
+
{ b: 1 },
|
|
239
|
+
],
|
|
240
|
+
bc: (r) => {
|
|
241
|
+
if (r.child && r.child.node !== undefined) {
|
|
242
|
+
r.node.push(r.child.node)
|
|
243
|
+
}
|
|
244
|
+
},
|
|
245
|
+
},
|
|
246
|
+
|
|
247
|
+
// A (possibly empty) sequence of elements. The 2-token lookahead
|
|
248
|
+
// `#TX #DEF` detects a following production boundary and bails
|
|
249
|
+
// out without consuming the tokens; a plain `#TX` at the leading
|
|
250
|
+
// position (tried later so the longer alt wins) is a rule
|
|
251
|
+
// reference inside the current sequence.
|
|
252
|
+
seq: {
|
|
253
|
+
bo: (r) => { r.node = [] },
|
|
254
|
+
open: [
|
|
255
|
+
{ s: '#TX #DEF', b: 2, g: 'end' },
|
|
256
|
+
{ s: '#TX #DEFA', b: 2, g: 'end' },
|
|
257
|
+
// A rulename followed by a repetition count — `a 1*b`, `a 2b`,
|
|
258
|
+
// `simple-key 1*( dot-sep simple-key )`.
|
|
259
|
+
//
|
|
260
|
+
// This alternative exists for its `s:` pattern, not its action:
|
|
261
|
+
// it widens the tcol the lexer uses for the token AFTER a
|
|
262
|
+
// rulename. The `#TX #DEF` / `#TX #DEFA` lookaheads above are
|
|
263
|
+
// the only other two-token patterns starting with `#TX`, so
|
|
264
|
+
// without this one that tcol is just {#DEF, #DEFA} — the `#NUM`
|
|
265
|
+
// match.token matcher is never offered the position, and the
|
|
266
|
+
// digits fall through to the engine's default matchers. `1*b`
|
|
267
|
+
// then arrives as #NR and fails to parse at all, and `2b`
|
|
268
|
+
// arrives as a #TX bareword, silently misparsing as a reference
|
|
269
|
+
// to a rule named `2b`. Same reason `elem.open` spells its
|
|
270
|
+
// atom position `#ATOM` rather than leaving it implicit.
|
|
271
|
+
{ s: '#TX #NUM', b: 2, p: 'elem' },
|
|
272
|
+
// Same problem, same fix, for a rulename followed by another
|
|
273
|
+
// atom. `#ATOM` covers every atom-starter, so the `%xNN` (#NV),
|
|
274
|
+
// `%s"…"` / `%i"…"` (#SS/#SI) and `<prose>` (#PV) matchers are
|
|
275
|
+
// all offered the position. Without it `ws %x5B ws` silently
|
|
276
|
+
// became a reference to a rule named `%x5B`, and `a <foo>` a
|
|
277
|
+
// reference to `<foo>` instead of the prose error it should
|
|
278
|
+
// raise. (The Go port gets these right via `match.tokenEager`,
|
|
279
|
+
// which is what exposed the discrepancy.)
|
|
280
|
+
{ s: '#TX #ATOM', b: 2, p: 'elem' },
|
|
281
|
+
// `<all> =` starts the next production; without this the prose
|
|
282
|
+
// would be taken as another element of this sequence.
|
|
283
|
+
{ s: '#PV #DEF', b: 2, g: 'end' },
|
|
284
|
+
{ s: '#ALT', b: 1, g: 'end' },
|
|
285
|
+
{ s: '#ZZ', b: 1, g: 'end' },
|
|
286
|
+
{ s: '#RP', b: 1, g: 'end' },
|
|
287
|
+
{ s: '#CB', b: 1, g: 'end' },
|
|
288
|
+
// Listing element-starter tokens in `s:` here ensures the
|
|
289
|
+
// tcol-driven matcher considers each one when lexing.
|
|
290
|
+
{ s: '#ST', b: 1, p: 'elem' },
|
|
291
|
+
{ s: '#NV', b: 1, p: 'elem' },
|
|
292
|
+
{ s: '#SS', b: 1, p: 'elem' },
|
|
293
|
+
{ s: '#SI', b: 1, p: 'elem' },
|
|
294
|
+
{ s: '#PV', b: 1, p: 'elem' },
|
|
295
|
+
{ s: '#TX', b: 1, p: 'elem' },
|
|
296
|
+
{ s: '#LP', b: 1, p: 'elem' },
|
|
297
|
+
{ s: '#OB', b: 1, p: 'elem' },
|
|
298
|
+
{ s: '#STAR', b: 1, p: 'elem' },
|
|
299
|
+
{ s: '#NUM', b: 1, p: 'elem' },
|
|
300
|
+
{ p: 'elem' },
|
|
301
|
+
],
|
|
302
|
+
close: [
|
|
303
|
+
{ s: '#TX #DEF', b: 2, g: 'end' },
|
|
304
|
+
{ s: '#TX #DEFA', b: 2, g: 'end' },
|
|
305
|
+
// `<all> = …` starts the next production, exactly as in `open`.
|
|
306
|
+
// `open` has always carried this alternative; `close` did not,
|
|
307
|
+
// and only got away with it because the prose was mis-lexed as a
|
|
308
|
+
// bareword here (so the `#TX #DEF` boundary above caught it by
|
|
309
|
+
// accident). Now that `#TX #ATOM` lets the #PV matcher see the
|
|
310
|
+
// position, the boundary has to be checked properly — and before
|
|
311
|
+
// the `{ s: '#PV', p: 'elem' }` alternative further down, which
|
|
312
|
+
// would otherwise take `<all>` as a prose element of this
|
|
313
|
+
// sequence.
|
|
314
|
+
{ s: '#PV #DEF', b: 2, g: 'end' },
|
|
315
|
+
// See the matching alternatives in `open` — these widen the tcol
|
|
316
|
+
// for the token after a rulename, so `a 1*b` / `a 2b` lex as
|
|
317
|
+
// #NUM and `ws %x5B` / `a %s"Q"` / `a <foo>` reach their own
|
|
318
|
+
// matchers instead of falling through to a #TX bareword.
|
|
319
|
+
{ s: '#TX #NUM', b: 2, p: 'elem' },
|
|
320
|
+
{ s: '#TX #ATOM', b: 2, p: 'elem' },
|
|
321
|
+
{ s: '#ALT', b: 1, g: 'end' },
|
|
322
|
+
{ s: '#ZZ', b: 1, g: 'end' },
|
|
323
|
+
{ s: '#RP', b: 1, g: 'end' },
|
|
324
|
+
{ s: '#CB', b: 1, g: 'end' },
|
|
325
|
+
{ s: '#ST', b: 1, p: 'elem' },
|
|
326
|
+
{ s: '#NV', b: 1, p: 'elem' },
|
|
327
|
+
{ s: '#SS', b: 1, p: 'elem' },
|
|
328
|
+
{ s: '#SI', b: 1, p: 'elem' },
|
|
329
|
+
{ s: '#PV', b: 1, p: 'elem' },
|
|
330
|
+
{ s: '#TX', b: 1, p: 'elem' },
|
|
331
|
+
{ s: '#LP', b: 1, p: 'elem' },
|
|
332
|
+
{ s: '#OB', b: 1, p: 'elem' },
|
|
333
|
+
{ s: '#STAR', b: 1, p: 'elem' },
|
|
334
|
+
{ s: '#NUM', b: 1, p: 'elem' },
|
|
335
|
+
{ b: 1 },
|
|
336
|
+
],
|
|
337
|
+
},
|
|
338
|
+
|
|
339
|
+
// One element: an optional ABNF repetition prefix (`*A`, `1*A`,
|
|
340
|
+
// `m*nA`, `*nA`, `m*A`, `nA`) followed by an atom. The prefix is
|
|
341
|
+
// matched up front, stored on `r.u.min`/`r.u.max`; then `atom` is
|
|
342
|
+
// pushed to parse the actual element body, whose result is wrapped
|
|
343
|
+
// into an AST node and appended to the parent seq's array in close.
|
|
344
|
+
elem: {
|
|
345
|
+
bo: (r) => { r.u.min = 1; r.u.max = 1 },
|
|
346
|
+
open: [
|
|
347
|
+
// NUM '*' NUM — bounded repetition, followed by the atom
|
|
348
|
+
// itself (listed via the ATOM tokenset so every atom-starter
|
|
349
|
+
// tin — including `#NV` — is in tcol for this position).
|
|
350
|
+
{
|
|
351
|
+
s: '#NUM #STAR #NUM #ATOM',
|
|
352
|
+
b: 1,
|
|
353
|
+
a: (r: Rule) => {
|
|
354
|
+
r.u.min = parseInt(r.o[0].src, 10)
|
|
355
|
+
r.u.max = parseInt(r.o[2].src, 10)
|
|
356
|
+
},
|
|
357
|
+
p: 'atom',
|
|
358
|
+
},
|
|
359
|
+
// NUM '*' — at-least-NUM repetition followed by an atom.
|
|
360
|
+
{
|
|
361
|
+
s: '#NUM #STAR #ATOM',
|
|
362
|
+
b: 1,
|
|
363
|
+
a: (r: Rule) => {
|
|
364
|
+
r.u.min = parseInt(r.o[0].src, 10)
|
|
365
|
+
r.u.max = Infinity
|
|
366
|
+
},
|
|
367
|
+
p: 'atom',
|
|
368
|
+
},
|
|
369
|
+
// '*' NUM — at-most-NUM repetition.
|
|
370
|
+
{
|
|
371
|
+
s: '#STAR #NUM #ATOM',
|
|
372
|
+
b: 1,
|
|
373
|
+
a: (r: Rule) => {
|
|
374
|
+
r.u.min = 0
|
|
375
|
+
r.u.max = parseInt(r.o[1].src, 10)
|
|
376
|
+
},
|
|
377
|
+
p: 'atom',
|
|
378
|
+
},
|
|
379
|
+
// '*' — zero-or-more.
|
|
380
|
+
{
|
|
381
|
+
s: '#STAR #ATOM',
|
|
382
|
+
b: 1,
|
|
383
|
+
a: (r: Rule) => { r.u.min = 0; r.u.max = Infinity },
|
|
384
|
+
p: 'atom',
|
|
385
|
+
},
|
|
386
|
+
// NUM — exact repetition count.
|
|
387
|
+
{
|
|
388
|
+
s: '#NUM #ATOM',
|
|
389
|
+
b: 1,
|
|
390
|
+
a: (r: Rule) => {
|
|
391
|
+
const n = parseInt(r.o[0].src, 10)
|
|
392
|
+
r.u.min = n
|
|
393
|
+
r.u.max = n
|
|
394
|
+
},
|
|
395
|
+
p: 'atom',
|
|
396
|
+
},
|
|
397
|
+
// No prefix — push atom directly (min = max = 1).
|
|
398
|
+
{ p: 'atom' },
|
|
399
|
+
],
|
|
400
|
+
close: [{
|
|
401
|
+
// Wrap the returned atom (r.child.node) based on r.u.min/max
|
|
402
|
+
// and append to the parent seq's array.
|
|
403
|
+
a: (r: Rule) => {
|
|
404
|
+
const item = r.child.node
|
|
405
|
+
const { min, max } = r.u
|
|
406
|
+
if (min === 1 && max === 1) {
|
|
407
|
+
r.node.push(item)
|
|
408
|
+
} else if (min === 0 && max === Infinity) {
|
|
409
|
+
r.node.push({ kind: 'star', inner: item })
|
|
410
|
+
} else if (min === 1 && max === Infinity) {
|
|
411
|
+
r.node.push({ kind: 'plus', inner: item })
|
|
412
|
+
} else if (min === 0 && max === 1) {
|
|
413
|
+
r.node.push({ kind: 'opt', inner: item })
|
|
414
|
+
} else {
|
|
415
|
+
r.node.push({ kind: 'rep', min, max, inner: item })
|
|
416
|
+
}
|
|
417
|
+
},
|
|
418
|
+
}],
|
|
419
|
+
},
|
|
420
|
+
|
|
421
|
+
// The atom body — a bareword ref, quoted-string terminal,
|
|
422
|
+
// parenthesised group, or bracketed optional. Sets its OWN r.node
|
|
423
|
+
// to the AST element so the enclosing `elem` rule can read it
|
|
424
|
+
// from `r.child.node` in its close state.
|
|
425
|
+
atom: {
|
|
426
|
+
bo: (r) => { r.node = undefined },
|
|
427
|
+
open: [
|
|
428
|
+
// Case-sensitive string: %s"foo"
|
|
429
|
+
{
|
|
430
|
+
s: '#SS #ST',
|
|
431
|
+
a: (r: Rule) => {
|
|
432
|
+
r.node = {
|
|
433
|
+
kind: 'term',
|
|
434
|
+
literal: r.o[1].val,
|
|
435
|
+
caseSensitive: true,
|
|
436
|
+
sp: spanTo(r.o[0], r.o[1]),
|
|
437
|
+
}
|
|
438
|
+
},
|
|
439
|
+
},
|
|
440
|
+
// Case-insensitive string: %i"foo" (same as bare "foo" below,
|
|
441
|
+
// but spelled explicitly).
|
|
442
|
+
{
|
|
443
|
+
s: '#SI #ST',
|
|
444
|
+
a: (r: Rule) => {
|
|
445
|
+
r.node = {
|
|
446
|
+
kind: 'term', literal: r.o[1].val, sp: spanTo(r.o[0], r.o[1]),
|
|
447
|
+
}
|
|
448
|
+
},
|
|
449
|
+
},
|
|
450
|
+
// Bare quoted string — case-insensitive per ABNF default.
|
|
451
|
+
{
|
|
452
|
+
s: '#ST',
|
|
453
|
+
a: (r: Rule) => {
|
|
454
|
+
r.node = { kind: 'term', literal: r.o[0].val, sp: spanOf(r.o[0]) }
|
|
455
|
+
},
|
|
456
|
+
},
|
|
457
|
+
{
|
|
458
|
+
s: '#NV',
|
|
459
|
+
a: (r: Rule) => {
|
|
460
|
+
r.node = parseNumericValue(r.o[0].src as string, r.o[0])
|
|
461
|
+
},
|
|
462
|
+
},
|
|
463
|
+
// Prose terminal `<free text>` — carried through as-is; the
|
|
464
|
+
// `resolveProseTerminals` pass decides what it means.
|
|
465
|
+
{
|
|
466
|
+
s: '#PV',
|
|
467
|
+
a: (r: Rule) => {
|
|
468
|
+
const src = r.o[0].src as string
|
|
469
|
+
r.node = {
|
|
470
|
+
kind: 'prose', text: src.slice(1, -1), sp: spanOf(r.o[0]),
|
|
471
|
+
}
|
|
472
|
+
},
|
|
473
|
+
},
|
|
474
|
+
{
|
|
475
|
+
s: '#TX',
|
|
476
|
+
a: (r: Rule) => {
|
|
477
|
+
r.node = { kind: 'ref', name: r.o[0].val, sp: spanOf(r.o[0]) }
|
|
478
|
+
},
|
|
479
|
+
},
|
|
480
|
+
{
|
|
481
|
+
s: '#LP',
|
|
482
|
+
a: (r: Rule) => { r.u.groupKind = 'group'; r.u.open = r.o[0] },
|
|
483
|
+
p: 'alts',
|
|
484
|
+
},
|
|
485
|
+
{
|
|
486
|
+
s: '#OB',
|
|
487
|
+
a: (r: Rule) => { r.u.groupKind = 'opt'; r.u.open = r.o[0] },
|
|
488
|
+
p: 'alts',
|
|
489
|
+
},
|
|
490
|
+
],
|
|
491
|
+
close: [
|
|
492
|
+
{
|
|
493
|
+
s: '#RP',
|
|
494
|
+
c: (r: Rule) => r.u.groupKind === 'group',
|
|
495
|
+
a: (r: Rule) => {
|
|
496
|
+
r.node = {
|
|
497
|
+
kind: 'group', alts: r.child.node, sp: spanTo(r.u.open, r.c0),
|
|
498
|
+
}
|
|
499
|
+
},
|
|
500
|
+
},
|
|
501
|
+
{
|
|
502
|
+
s: '#CB',
|
|
503
|
+
c: (r: Rule) => r.u.groupKind === 'opt',
|
|
504
|
+
a: (r: Rule) => {
|
|
505
|
+
const bracket = spanTo(r.u.open, r.c0)
|
|
506
|
+
r.node = {
|
|
507
|
+
kind: 'opt',
|
|
508
|
+
inner: { kind: 'group', alts: r.child.node, sp: bracket },
|
|
509
|
+
sp: bracket,
|
|
510
|
+
}
|
|
511
|
+
},
|
|
512
|
+
},
|
|
513
|
+
// For simple atoms (string/ref), r.node is already set by
|
|
514
|
+
// open; we want to pop without consuming the next token.
|
|
515
|
+
// List every token that can legitimately follow an atom so
|
|
516
|
+
// the lexer's tcol-driven match-matcher emits #NUM, #STAR,
|
|
517
|
+
// and friends as their proper types here — otherwise the
|
|
518
|
+
// default number-matcher would lex `1` as #NR and the
|
|
519
|
+
// enclosing seq.close wouldn't recognise the digit as the
|
|
520
|
+
// start of a repetition prefix.
|
|
521
|
+
{ s: '#TX', b: 1 },
|
|
522
|
+
{ s: '#ST', b: 1 },
|
|
523
|
+
{ s: '#NV', b: 1 },
|
|
524
|
+
{ s: '#SS', b: 1 },
|
|
525
|
+
{ s: '#SI', b: 1 },
|
|
526
|
+
{ s: '#PV', b: 1 },
|
|
527
|
+
{ s: '#NUM', b: 1 },
|
|
528
|
+
{ s: '#STAR', b: 1 },
|
|
529
|
+
{ s: '#LP', b: 1 },
|
|
530
|
+
{ s: '#OB', b: 1 },
|
|
531
|
+
{ s: '#RP', b: 1 },
|
|
532
|
+
{ s: '#CB', b: 1 },
|
|
533
|
+
{ s: '#ALT', b: 1 },
|
|
534
|
+
{ s: '#DEF', b: 1 },
|
|
535
|
+
{ s: '#ZZ', b: 1 },
|
|
536
|
+
{ b: 1 },
|
|
537
|
+
],
|
|
538
|
+
},
|
|
539
|
+
}
|
|
540
|
+
|
|
541
|
+
// Cached tabnas instance for the ABNF grammar above; built on first use.
|
|
542
|
+
let _abnfParser: ((src: string) => Production[]) | null = null
|
|
543
|
+
|
|
544
|
+
function getAbnfParser(): (src: string) => AbnfProduction[] {
|
|
545
|
+
if (_abnfParser) return _abnfParser
|
|
546
|
+
|
|
547
|
+
const { Tabnas } = require('@tabnas/parser')
|
|
548
|
+
|
|
549
|
+
// ABNF defines its own grammar from scratch, so we don't load any
|
|
550
|
+
// grammar plugin — just use the bare engine with default tokens.
|
|
551
|
+
const j = new Tabnas({
|
|
552
|
+
rule: { start: 'abnf' },
|
|
553
|
+
fixed: {
|
|
554
|
+
token: {
|
|
555
|
+
// Clear JSON-oriented defaults we're not using so `:`, `,`
|
|
556
|
+
// and `{` have no special meaning inside ABNF source.
|
|
557
|
+
'#OS': null,
|
|
558
|
+
'#CS': null,
|
|
559
|
+
'#CL': null,
|
|
560
|
+
'#CA': null,
|
|
561
|
+
// Re-map `#OB` / `#CB` from JSON's `{` / `}` to ABNF's
|
|
562
|
+
// `[` / `]` optional-group brackets.
|
|
563
|
+
'#OB': '[',
|
|
564
|
+
'#CB': ']',
|
|
565
|
+
'#DEF': '=',
|
|
566
|
+
// `=/` — ABNF's incremental-alternatives operator. Longer
|
|
567
|
+
// than `=`, so tabnas's longest-match-wins fixed matcher
|
|
568
|
+
// tries it first.
|
|
569
|
+
'#DEFA': '=/',
|
|
570
|
+
'#ALT': '/',
|
|
571
|
+
'#STAR': '*',
|
|
572
|
+
'#LP': '(',
|
|
573
|
+
'#RP': ')',
|
|
574
|
+
},
|
|
575
|
+
},
|
|
576
|
+
match: {
|
|
577
|
+
token: {
|
|
578
|
+
// ABNF repetition counts: decimal integers.
|
|
579
|
+
'#NUM': /^[0-9]+/,
|
|
580
|
+
// ABNF numeric value notation:
|
|
581
|
+
// %xNN single hex code point
|
|
582
|
+
// %dNN single decimal code point
|
|
583
|
+
// %bNN single binary code point
|
|
584
|
+
// %xNN-NN hex range
|
|
585
|
+
// %xNN.NN.NN concatenated hex code points (= string)
|
|
586
|
+
// Digits are permissive (hex covers the decimal / binary
|
|
587
|
+
// subsets); `parseNumericValue` re-validates against the
|
|
588
|
+
// actual base.
|
|
589
|
+
'#NV': /^%[xdbXDB][0-9a-fA-F]+(?:[-.][0-9a-fA-F]+)*/,
|
|
590
|
+
// `%s` / `%i` prefixes on a quoted string. The lookahead
|
|
591
|
+
// requires `"` so they don't steal the `%` of `%xNN`.
|
|
592
|
+
'#SS': /^%[sS](?=")/,
|
|
593
|
+
'#SI': /^%[iI](?=")/,
|
|
594
|
+
// RFC 5234 prose-val: `<` free text `>`. The body is every
|
|
595
|
+
// printable char except `>` itself (%x20-3D / %x3F-7E).
|
|
596
|
+
'#PV': /^<[\x20-\x3D\x3F-\x7E]*>/,
|
|
597
|
+
},
|
|
598
|
+
},
|
|
599
|
+
value: {
|
|
600
|
+
// RFC 5234 rulename is `ALPHA *(ALPHA / DIGIT / "-")` — nothing is
|
|
601
|
+
// reserved, so `true`, `false` and `null` are ordinary rule names.
|
|
602
|
+
// JSON's grammar uses all three (`value = false / null / true /
|
|
603
|
+
// object / array / number / string`), and with the engine's default
|
|
604
|
+
// keyword-value lexing they arrived as `#VL` value tokens instead of
|
|
605
|
+
// `#TX` barewords, so no ABNF rendering of JSON would compile.
|
|
606
|
+
// The ABNF meta-grammar has no use for `#VL` at all — this switch
|
|
607
|
+
// only affects the parser that reads ABNF source, not the grammars
|
|
608
|
+
// it emits, where `VL` remains a built-in token name.
|
|
609
|
+
lex: false,
|
|
610
|
+
},
|
|
611
|
+
string: {
|
|
612
|
+
// RFC 5234 char-val has NO escape sequences at all:
|
|
613
|
+
// char-val = DQUOTE *(%x20-21 / %x23-7E) DQUOTE
|
|
614
|
+
// A backslash is just %x5C, an ordinary member of that range, so
|
|
615
|
+
// `"\"` is a one-character literal — and it is a common one, since
|
|
616
|
+
// every RFC that defines `quoted-pair` writes it that way
|
|
617
|
+
// (RFC 5322, RFC 3261, RFC 8259, …). With the engine's default
|
|
618
|
+
// JSON-style escaping the backslash swallowed the closing quote
|
|
619
|
+
// and the grammar died with `unterminated_string`, while `"a\b"`
|
|
620
|
+
// silently became `a<BS>b` instead of the three characters
|
|
621
|
+
// `a`, `\`, `b`.
|
|
622
|
+
//
|
|
623
|
+
// The engine offers no "escaping off" switch that both runtimes
|
|
624
|
+
// share (TS takes `escapeChar: null`, Go falls back to `\` on an
|
|
625
|
+
// empty string), so instead point the escape character at DEL
|
|
626
|
+
// (%x7F) — outside char-val's %x20-21 / %x23-7E body, hence
|
|
627
|
+
// unreachable in any legal ABNF literal.
|
|
628
|
+
escapeChar: '\x7F',
|
|
629
|
+
},
|
|
630
|
+
tokenSet: {
|
|
631
|
+
// Tokens that can legitimately open an atom. Declaring this
|
|
632
|
+
// as a set lets elem.open use `#ATOM` inside its `s:` patterns
|
|
633
|
+
// — that way the tcol at the atom-starter position includes
|
|
634
|
+
// every matcher tin (notably #NV), so the lexer doesn't fall
|
|
635
|
+
// through to #TX when the actual atom is `%xNN`.
|
|
636
|
+
ATOM: ['#ST', '#NV', '#TX', '#LP', '#OB', '#SS', '#SI', '#PV'],
|
|
637
|
+
},
|
|
638
|
+
comment: {
|
|
639
|
+
// ABNF uses `;` to start a line comment. Override tabnas's
|
|
640
|
+
// default `hash` definition (which used `#`) and disable the
|
|
641
|
+
// other comment styles so `//` and `/* */` aren't confused
|
|
642
|
+
// with the alternation operator.
|
|
643
|
+
def: {
|
|
644
|
+
hash: { line: true, start: ';', lex: true, eatline: false },
|
|
645
|
+
slash: null as any,
|
|
646
|
+
multi: null as any,
|
|
647
|
+
},
|
|
648
|
+
},
|
|
649
|
+
})
|
|
650
|
+
|
|
651
|
+
// Drop the default JSON rules — they would otherwise compete with
|
|
652
|
+
// ours for the starting token set.
|
|
653
|
+
const existing = j.rule()
|
|
654
|
+
for (const name of Object.keys(existing)) {
|
|
655
|
+
j.rule(name, null)
|
|
656
|
+
}
|
|
657
|
+
|
|
658
|
+
for (const name of Object.keys(abnfRules)) {
|
|
659
|
+
const spec = abnfRules[name]
|
|
660
|
+
j.rule(name, (rs: any) => {
|
|
661
|
+
if (spec.bo) rs.bo(spec.bo)
|
|
662
|
+
if (spec.bc) rs.bc(spec.bc)
|
|
663
|
+
if (spec.open) rs.open(spec.open)
|
|
664
|
+
if (spec.close) rs.close(spec.close)
|
|
665
|
+
})
|
|
666
|
+
}
|
|
667
|
+
|
|
668
|
+
_abnfParser = (src: string) => j.parse(src) as AbnfProduction[]
|
|
669
|
+
return _abnfParser
|
|
670
|
+
}
|
|
671
|
+
|
|
672
|
+
|
|
673
|
+
// Error raised when the ABNF source itself can't be parsed. Surfaces
|
|
674
|
+
// line and column from the underlying tabnas error so the caller can
|
|
675
|
+
// report them directly. The original error is kept on `.cause`.
|
|
676
|
+
class AbnfParseError extends Error {
|
|
677
|
+
readonly line?: number
|
|
678
|
+
readonly column?: number
|
|
679
|
+
readonly cause?: unknown
|
|
680
|
+
constructor(message: string, location?: { line?: number; column?: number }, cause?: unknown) {
|
|
681
|
+
super(message)
|
|
682
|
+
this.name = 'AbnfParseError'
|
|
683
|
+
this.line = location?.line
|
|
684
|
+
this.column = location?.column
|
|
685
|
+
this.cause = cause
|
|
686
|
+
}
|
|
687
|
+
}
|
|
688
|
+
|
|
689
|
+
|
|
690
|
+
// Parse ABNF source into a grammar AST via the tabnas-based parser.
|
|
691
|
+
function parseAbnf(src: string): AbnfGrammar {
|
|
692
|
+
const parser = getAbnfParser()
|
|
693
|
+
let productions: AbnfProduction[]
|
|
694
|
+
try {
|
|
695
|
+
productions = parser(src) ?? []
|
|
696
|
+
} catch (e: any) {
|
|
697
|
+
// TabnasError carries `lineNumber` / `columnNumber`; fall back to
|
|
698
|
+
// ad-hoc extraction from the error message otherwise.
|
|
699
|
+
const line = e?.lineNumber ?? e?.row
|
|
700
|
+
const column = e?.columnNumber ?? e?.col
|
|
701
|
+
const loc = (line != null && column != null)
|
|
702
|
+
? ` at line ${line}, column ${column}`
|
|
703
|
+
: ''
|
|
704
|
+
const raw = e?.message ? String(e.message).split('\n')[0] : String(e)
|
|
705
|
+
throw new AbnfParseError(
|
|
706
|
+
`abnf: parse error${loc}: ${raw}`,
|
|
707
|
+
{ line, column },
|
|
708
|
+
e,
|
|
709
|
+
)
|
|
710
|
+
}
|
|
711
|
+
if (!Array.isArray(productions) || productions.length === 0) {
|
|
712
|
+
throw new AbnfParseError('abnf: no productions found')
|
|
713
|
+
}
|
|
714
|
+
// BEFORE merging, not after. `mergeIncrementals` drops each `=/`
|
|
715
|
+
// production, keeping only the base's span — so an annotation on an
|
|
716
|
+
// incremental line was resolved against a production list that no
|
|
717
|
+
// longer contained the line it followed, and attached to whatever rule
|
|
718
|
+
// happened to be declared before it instead. `a = "a"`, `b = 1*DIGIT`,
|
|
719
|
+
// `a =/ "c" ; @object` silently annotated **b**.
|
|
720
|
+
attachValueAnnotations(src, productions)
|
|
721
|
+
const merged = mergeIncrementals(productions)
|
|
722
|
+
return { productions: withCoreRules(merged) }
|
|
723
|
+
}
|
|
724
|
+
|
|
725
|
+
|
|
726
|
+
// A trailing comment claiming a value annotation:
|
|
727
|
+
//
|
|
728
|
+
// ver = maj "." min "." pat ; @object maj min pat
|
|
729
|
+
// tags = tag *("," tag) ; @array
|
|
730
|
+
//
|
|
731
|
+
// RFC 5234 has nowhere else to put this. A comment is the only place in
|
|
732
|
+
// the notation that carries no meaning of its own, which is exactly why
|
|
733
|
+
// it can carry one here without changing what the grammar accepts: strip
|
|
734
|
+
// every annotation and the same language parses, just into a tree
|
|
735
|
+
// instead of a value.
|
|
736
|
+
//
|
|
737
|
+
// ONLY `@object` and `@array` are claimed. Any other `; @…` comment is
|
|
738
|
+
// left alone — the notation has no directive namespace, so this must not
|
|
739
|
+
// assume one, and a reader's own `; @deprecated` has to keep meaning
|
|
740
|
+
// nothing.
|
|
741
|
+
const ANNOTATION = /^@(object|array)\b\s*(.*)$/
|
|
742
|
+
const MEMBER_NAME = /^[A-Za-z][A-Za-z0-9-]*$/
|
|
743
|
+
|
|
744
|
+
|
|
745
|
+
// Every `;` comment in the source that claims an annotation, with the
|
|
746
|
+
// offset it starts at.
|
|
747
|
+
//
|
|
748
|
+
// Quoted strings and prose are skipped: a `;` inside `"a;b"` or
|
|
749
|
+
// `<a;b>` is CONTENT, not a comment, and treating it as one would
|
|
750
|
+
// silently attach an annotation that the author did not write.
|
|
751
|
+
function* annotationComments(
|
|
752
|
+
src: string,
|
|
753
|
+
): Generator<{ at: number; body: string }> {
|
|
754
|
+
for (let i = 0; i < src.length; i++) {
|
|
755
|
+
const ch = src[i]
|
|
756
|
+
if ('"' === ch || '<' === ch) {
|
|
757
|
+
// RFC 5234 char-val and prose-val have no escapes, so the next
|
|
758
|
+
// closing mark ends them.
|
|
759
|
+
const close = src.indexOf('"' === ch ? '"' : '>', i + 1)
|
|
760
|
+
i = close < 0 ? src.length : close
|
|
761
|
+
continue
|
|
762
|
+
}
|
|
763
|
+
if (';' !== ch) continue
|
|
764
|
+
let end = src.indexOf('\n', i)
|
|
765
|
+
if (end < 0) end = src.length
|
|
766
|
+
const body = src.slice(i + 1, end).trim()
|
|
767
|
+
if (body.startsWith('@')) yield { at: i, body }
|
|
768
|
+
i = end
|
|
769
|
+
}
|
|
770
|
+
}
|
|
771
|
+
|
|
772
|
+
|
|
773
|
+
// Attach each annotation to the production it FOLLOWS — the last one
|
|
774
|
+
// that begins before it.
|
|
775
|
+
//
|
|
776
|
+
// Not "the production on the same line": a rule may be written across
|
|
777
|
+
// several lines, and an author putting the annotation on the last of
|
|
778
|
+
// them means the same thing. Following the definition is the rule that
|
|
779
|
+
// reads the same either way.
|
|
780
|
+
function attachValueAnnotations(
|
|
781
|
+
src: string,
|
|
782
|
+
prods: AbnfProduction[],
|
|
783
|
+
): void {
|
|
784
|
+
const ordered = prods
|
|
785
|
+
.filter((p) => null != p.sp)
|
|
786
|
+
.slice()
|
|
787
|
+
.sort((a, b) => (a.sp as SrcSpan).s - (b.sp as SrcSpan).s)
|
|
788
|
+
if (0 === ordered.length) return
|
|
789
|
+
|
|
790
|
+
let idx = 0
|
|
791
|
+
for (const { at, body } of annotationComments(src)) {
|
|
792
|
+
const m = ANNOTATION.exec(body)
|
|
793
|
+
if (null == m) continue
|
|
794
|
+
|
|
795
|
+
// Both `ordered` and the comments are in source order, so the search
|
|
796
|
+
// only ever moves FORWARD — `idx` is not reset per comment. Restarting
|
|
797
|
+
// it made attachment quadratic in the number of annotated rules, which
|
|
798
|
+
// a generated grammar can make expensive for nothing.
|
|
799
|
+
while (idx < ordered.length && (ordered[idx].sp as SrcSpan).s < at) idx++
|
|
800
|
+
const owner: AbnfProduction | undefined =
|
|
801
|
+
0 < idx ? ordered[idx - 1] : undefined
|
|
802
|
+
if (null == owner) {
|
|
803
|
+
throw new AbnfParseError(
|
|
804
|
+
`abnf: '; ${body}' appears before any rule, so there is nothing ` +
|
|
805
|
+
`for it to annotate. A value annotation goes after the rule it ` +
|
|
806
|
+
`describes.`,
|
|
807
|
+
)
|
|
808
|
+
}
|
|
809
|
+
|
|
810
|
+
const kind = m[1] as 'object' | 'array'
|
|
811
|
+
const members = m[2].split(/[\s,]+/).filter((w) => '' !== w)
|
|
812
|
+
|
|
813
|
+
if ('array' === kind) {
|
|
814
|
+
if (0 < members.length) {
|
|
815
|
+
throw new AbnfParseError(
|
|
816
|
+
`abnf: rule '${owner.name}': '@array' names no members — every ` +
|
|
817
|
+
`part that produces a value becomes an element, in order. Got ` +
|
|
818
|
+
`'${members.join(' ')}'.`,
|
|
819
|
+
)
|
|
820
|
+
}
|
|
821
|
+
} else {
|
|
822
|
+
const seen = new Set<string>()
|
|
823
|
+
for (const name of members) {
|
|
824
|
+
if (!MEMBER_NAME.test(name)) {
|
|
825
|
+
throw new AbnfParseError(
|
|
826
|
+
`abnf: rule '${owner.name}': '${name}' is not a rule name, so ` +
|
|
827
|
+
`it cannot name a member of '@object'.`,
|
|
828
|
+
)
|
|
829
|
+
}
|
|
830
|
+
// Each member is a separate KEY. Two parts named the same thing
|
|
831
|
+
// both write to it, so the second silently overwrites the first
|
|
832
|
+
// and that much of the input is gone from the result.
|
|
833
|
+
if (seen.has(name)) {
|
|
834
|
+
throw new AbnfParseError(
|
|
835
|
+
`abnf: rule '${owner.name}': '@object' names '${name}' twice. ` +
|
|
836
|
+
`Each member is a separate key, so the second part would ` +
|
|
837
|
+
`overwrite the first. Give them different names.`,
|
|
838
|
+
)
|
|
839
|
+
}
|
|
840
|
+
seen.add(name)
|
|
841
|
+
}
|
|
842
|
+
}
|
|
843
|
+
|
|
844
|
+
if (null != owner.value) {
|
|
845
|
+
throw new AbnfParseError(
|
|
846
|
+
`abnf: rule '${owner.name}' has more than one value annotation. A ` +
|
|
847
|
+
`rule builds one thing.`,
|
|
848
|
+
)
|
|
849
|
+
}
|
|
850
|
+
owner.value = 'array' === kind ? { kind } : { kind, members }
|
|
851
|
+
}
|
|
852
|
+
}
|
|
853
|
+
|
|
854
|
+
|
|
855
|
+
// RFC 5234 Appendix B.1 core rules. Parsed lazily on first use
|
|
856
|
+
// and spliced into any user grammar that references them but
|
|
857
|
+
// doesn't define them locally.
|
|
858
|
+
const CORE_RULES_ABNF = `
|
|
859
|
+
ALPHA = %x41-5A / %x61-7A
|
|
860
|
+
BIT = "0" / "1"
|
|
861
|
+
CHAR = %x01-7F
|
|
862
|
+
CR = %x0D
|
|
863
|
+
LF = %x0A
|
|
864
|
+
CRLF = CR LF
|
|
865
|
+
CTL = %x00-1F / %x7F
|
|
866
|
+
DIGIT = %x30-39
|
|
867
|
+
DQUOTE = %x22
|
|
868
|
+
HEXDIG = DIGIT / "A" / "B" / "C" / "D" / "E" / "F"
|
|
869
|
+
HTAB = %x09
|
|
870
|
+
OCTET = %x00-FF
|
|
871
|
+
SP = %x20
|
|
872
|
+
VCHAR = %x21-7E
|
|
873
|
+
WSP = SP / HTAB
|
|
874
|
+
LWSP = *( WSP / CRLF WSP )
|
|
875
|
+
`
|
|
876
|
+
|
|
877
|
+
let _coreRules: Map<string, AbnfProduction> | null = null
|
|
878
|
+
|
|
879
|
+
function getCoreRules(): Map<string, AbnfProduction> {
|
|
880
|
+
if (_coreRules) return _coreRules
|
|
881
|
+
const parser = getAbnfParser()
|
|
882
|
+
const raw = parser(CORE_RULES_ABNF) as AbnfProduction[]
|
|
883
|
+
// Core rules flatten to `src` in the output AST — they're
|
|
884
|
+
// character-class bricks, not structural nodes users want to see
|
|
885
|
+
// one-per-matched-character.
|
|
886
|
+
for (const p of raw) p.nodeKind = 'core'
|
|
887
|
+
|
|
888
|
+
// Strip source spans from the core rules. They are parsed from
|
|
889
|
+
// CORE_RULES_ABNF, a string in THIS FILE, so their spans are offsets
|
|
890
|
+
// into a document the user never wrote — an editor asked to reveal
|
|
891
|
+
// one would jump to a position in the user's grammar that has nothing
|
|
892
|
+
// to do with ALPHA or DIGIT. A missing span means "nowhere to point",
|
|
893
|
+
// which is exactly right for a rule the library supplied; a wrong one
|
|
894
|
+
// is worse than none.
|
|
895
|
+
//
|
|
896
|
+
// It also removes an aliasing hazard: this map is module-cached and
|
|
897
|
+
// handed out BY REFERENCE to every grammar compiled in the process,
|
|
898
|
+
// so any position on it would be shared across unrelated documents.
|
|
899
|
+
//
|
|
900
|
+
// A reference TO a core rule still carries a span — that reference is
|
|
901
|
+
// in the user's source, and it is what a diagnostic points at.
|
|
902
|
+
for (const p of raw) stripSpans(p)
|
|
903
|
+
_coreRules = new Map(raw.map((p) => [p.name, p]))
|
|
904
|
+
return _coreRules
|
|
905
|
+
}
|
|
906
|
+
|
|
907
|
+
|
|
908
|
+
// Add each RFC 5234 core rule that the user's grammar references
|
|
909
|
+
// but doesn't define locally. Resolution is transitive: if the
|
|
910
|
+
// user mentions HEXDIG, DIGIT is pulled in too. User definitions
|
|
911
|
+
// always win — a local `DIGIT = …` is left untouched.
|
|
912
|
+
function withCoreRules(user: AbnfProduction[]): AbnfProduction[] {
|
|
913
|
+
const core = getCoreRules()
|
|
914
|
+
const defined = new Set(user.map((p) => p.name))
|
|
915
|
+
const needed = new Set<string>()
|
|
916
|
+
|
|
917
|
+
// A malformed element reaches here as a hole in an alt — an unclosed group
|
|
918
|
+
// (`( "a" / "b"` with no `)`) pops without ever building its node, leaving
|
|
919
|
+
// `undefined` in the sequence. bnf's refsIn then reads `.kind` off it and
|
|
920
|
+
// throws a TypeError, which is a CRASH, not a rejection: the conformance
|
|
921
|
+
// harness scored it as a correct rejection for every base grammar in the
|
|
922
|
+
// corpus, and abnf's own record says such input is rejected.
|
|
923
|
+
//
|
|
924
|
+
// Reject it here, as the parse error it is. The deeper repair — having the
|
|
925
|
+
// `elem` rule refuse to close a group on anything but its own `)` — is a
|
|
926
|
+
// grammar change and is deliberately not attempted in the same commit as the
|
|
927
|
+
// harness fix that exposed this.
|
|
928
|
+
//
|
|
929
|
+
// The walk MIRRORS bnf's `refsIn` exactly, because "where does refsIn
|
|
930
|
+
// dereference?" is the definition of where a hole crashes. A top-level scan
|
|
931
|
+
// of the sequence is not enough: when a repetition wraps the unclosed group
|
|
932
|
+
// — `bad = *( "a"` — `elem.close` builds a perfectly real `star` whose
|
|
933
|
+
// `inner` is the hole, so the sequence entry is non-null and refsIn walks
|
|
934
|
+
// into `[undefined]` one level down. `*[`, `1*2(` and a group's `alts` all
|
|
935
|
+
// reach it the same way.
|
|
936
|
+
const holeIn = (alt: readonly (AbnfElement | undefined)[] | undefined): boolean => {
|
|
937
|
+
if (null == alt) return true
|
|
938
|
+
for (const el of alt) {
|
|
939
|
+
if (null == el) return true
|
|
940
|
+
const k = (el as any).kind
|
|
941
|
+
if ('opt' === k || 'star' === k || 'plus' === k || 'rep' === k) {
|
|
942
|
+
if (holeIn([(el as any).inner])) return true
|
|
943
|
+
} else if ('group' === k) {
|
|
944
|
+
const alts = (el as any).alts
|
|
945
|
+
if (null == alts) return true
|
|
946
|
+
for (const a of alts) if (holeIn(a)) return true
|
|
947
|
+
}
|
|
948
|
+
}
|
|
949
|
+
return false
|
|
950
|
+
}
|
|
951
|
+
|
|
952
|
+
const rejectHoles = (prods: AbnfProduction[]) => {
|
|
953
|
+
for (const p of prods) {
|
|
954
|
+
if (null == p.alts) {
|
|
955
|
+
throw new AbnfParseError(
|
|
956
|
+
`abnf: rule '${p.name}' is malformed — no alternatives were built.`)
|
|
957
|
+
}
|
|
958
|
+
for (const alt of p.alts) {
|
|
959
|
+
if (holeIn(alt)) {
|
|
960
|
+
throw new AbnfParseError(
|
|
961
|
+
`abnf: rule '${p.name}' is malformed — an element could not be ` +
|
|
962
|
+
`built. The usual cause is an unclosed group or option.`)
|
|
963
|
+
}
|
|
964
|
+
}
|
|
965
|
+
}
|
|
966
|
+
}
|
|
967
|
+
rejectHoles(user)
|
|
968
|
+
|
|
969
|
+
const scan = (prods: AbnfProduction[]) => {
|
|
970
|
+
for (const p of prods) {
|
|
971
|
+
for (const alt of p.alts) refsIn(alt, needed)
|
|
972
|
+
}
|
|
973
|
+
}
|
|
974
|
+
|
|
975
|
+
scan(user)
|
|
976
|
+
const out: AbnfProduction[] = []
|
|
977
|
+
// Transitively add core rules, in declaration order.
|
|
978
|
+
let added = true
|
|
979
|
+
while (added) {
|
|
980
|
+
added = false
|
|
981
|
+
for (const [name, prod] of core) {
|
|
982
|
+
if (defined.has(name)) continue
|
|
983
|
+
if (!needed.has(name)) continue
|
|
984
|
+
defined.add(name)
|
|
985
|
+
// A COPY, not the cached production. `getCoreRules` hands back a
|
|
986
|
+
// module-level map, so pushing `prod` itself would put the same
|
|
987
|
+
// object into every grammar parsed in this process — and
|
|
988
|
+
// `parseAbnf` returns it to the caller. A consumer that annotated
|
|
989
|
+
// an ALPHA node (writing a span onto it, say) would then see that
|
|
990
|
+
// annotation on unrelated documents, and the "core rules carry no
|
|
991
|
+
// span" guarantee would hold only until someone broke it for
|
|
992
|
+
// everyone. Cloning is cheap: these are a dozen small
|
|
993
|
+
// character-class rules, and only the referenced ones are added.
|
|
994
|
+
const copy = cloneProduction(prod)
|
|
995
|
+
out.push(copy)
|
|
996
|
+
scan([copy])
|
|
997
|
+
added = true
|
|
998
|
+
}
|
|
999
|
+
}
|
|
1000
|
+
return [...user, ...out]
|
|
1001
|
+
}
|
|
1002
|
+
|
|
1003
|
+
|
|
1004
|
+
// Fold every `name =/ alt` production into the earlier production
|
|
1005
|
+
// with the same name by appending its alternatives. Throws if an
|
|
1006
|
+
// incremental references a name that hasn't been defined yet — ABNF
|
|
1007
|
+
// requires the base production to appear first.
|
|
1008
|
+
function mergeIncrementals(prods: AbnfProduction[]): AbnfProduction[] {
|
|
1009
|
+
const out: AbnfProduction[] = []
|
|
1010
|
+
const byName = new Map<string, AbnfProduction>()
|
|
1011
|
+
for (const p of prods) {
|
|
1012
|
+
if (p.incremental) {
|
|
1013
|
+
const base = byName.get(p.name)
|
|
1014
|
+
if (!base) {
|
|
1015
|
+
throw new AbnfParseError(
|
|
1016
|
+
`abnf: '${p.name} =/ …' has no earlier '${p.name} = …' to extend`,
|
|
1017
|
+
)
|
|
1018
|
+
}
|
|
1019
|
+
base.alts.push(...p.alts)
|
|
1020
|
+
// This production is about to be dropped, and annotations are now
|
|
1021
|
+
// attached before that happens — so an annotation on the `=/` line
|
|
1022
|
+
// has to move to the base, which IS the rule it describes.
|
|
1023
|
+
if (p.value) {
|
|
1024
|
+
if (base.value) {
|
|
1025
|
+
throw new AbnfParseError(
|
|
1026
|
+
`abnf: rule '${p.name}' has more than one value annotation. A ` +
|
|
1027
|
+
`rule builds one thing.`,
|
|
1028
|
+
)
|
|
1029
|
+
}
|
|
1030
|
+
base.value = p.value
|
|
1031
|
+
}
|
|
1032
|
+
continue
|
|
1033
|
+
}
|
|
1034
|
+
// Strip the (absent) flag on a cleanly-written production so
|
|
1035
|
+
// downstream code never sees it.
|
|
1036
|
+
// Rebuilt field by field, so every field carried on a production
|
|
1037
|
+
// has to be listed here or it is silently dropped — `sp` included.
|
|
1038
|
+
const clean: AbnfProduction = { name: p.name, alts: p.alts, sp: p.sp }
|
|
1039
|
+
if (p.nodeKind) clean.nodeKind = p.nodeKind
|
|
1040
|
+
// Annotations are attached BEFORE this runs, so this carry is live:
|
|
1041
|
+
// without it every annotation in the grammar would vanish here
|
|
1042
|
+
// without a word. The compiler downstream had six rebuilds like this
|
|
1043
|
+
// one and shipped with all six dropping it.
|
|
1044
|
+
if (p.value) clean.value = p.value
|
|
1045
|
+
out.push(clean)
|
|
1046
|
+
byName.set(p.name, clean)
|
|
1047
|
+
}
|
|
1048
|
+
return out
|
|
1049
|
+
}
|
|
1050
|
+
|
|
1051
|
+
|
|
1052
|
+
// Decode an ABNF numeric value (`%xNN`, `%dNN`, `%bNN`, or one of
|
|
1053
|
+
// the range/concatenation forms) into a `AbnfElement`.
|
|
1054
|
+
//
|
|
1055
|
+
// %x61 => single-char term "a"
|
|
1056
|
+
// %x66.6f.6f => concatenated term "foo"
|
|
1057
|
+
// %x30-39 => regex character class [\u0030-\u0039]
|
|
1058
|
+
//
|
|
1059
|
+
// Hex is case-insensitive; decimal and binary accept only digits
|
|
1060
|
+
// in their respective ranges. Range endpoints must be the same
|
|
1061
|
+
// base as the prefix (RFC 5234 doesn't allow mixing).
|
|
1062
|
+
function parseNumericValue(src: string, tkn?: any): AbnfElement {
|
|
1063
|
+
const sp = spanOf(tkn)
|
|
1064
|
+
const base = src[1].toLowerCase()
|
|
1065
|
+
const radix = base === 'x' ? 16 : base === 'd' ? 10 : 2
|
|
1066
|
+
const body = src.slice(2)
|
|
1067
|
+
|
|
1068
|
+
// RFC 5234 puts no ceiling on a numeric value, but Unicode does:
|
|
1069
|
+
// nothing above U+10FFFF is a code point. Check it here so an
|
|
1070
|
+
// out-of-range grammar gets an ABNF diagnostic naming the offending
|
|
1071
|
+
// value, rather than a bare `RangeError: Invalid code point` from
|
|
1072
|
+
// String.fromCodePoint (or, as before, a silently truncated
|
|
1073
|
+
// character from String.fromCharCode).
|
|
1074
|
+
const codePoint = (text: string): number => {
|
|
1075
|
+
const n = parseInt(text, radix)
|
|
1076
|
+
if (!Number.isFinite(n) || n < 0 || 0x10FFFF < n) {
|
|
1077
|
+
throw new Error(
|
|
1078
|
+
`numeric value '%${src[1]}${text}' is ${n}, which is not a ` +
|
|
1079
|
+
`Unicode code point (the maximum is %x10FFFF).`)
|
|
1080
|
+
}
|
|
1081
|
+
return n
|
|
1082
|
+
}
|
|
1083
|
+
|
|
1084
|
+
if (body.includes('-')) {
|
|
1085
|
+
const [loStr, hiStr] = body.split('-')
|
|
1086
|
+
const lo = codePoint(loStr)
|
|
1087
|
+
const hi = codePoint(hiStr)
|
|
1088
|
+
if (lo === hi) {
|
|
1089
|
+
return { kind: 'term', literal: String.fromCodePoint(lo), sp }
|
|
1090
|
+
}
|
|
1091
|
+
// `\uXXXX` only reaches U+FFFF: `%xE000-10FFFF` (JSONPath, TOML)
|
|
1092
|
+
// became `[-ჿff]`, which JS reads as the range E000–10FF
|
|
1093
|
+
// plus a literal `ff` — "Range out of order". Above the BMP the
|
|
1094
|
+
// escape has to be `\u{…}`, which in turn requires the `u` flag.
|
|
1095
|
+
// Stay on the plain form below U+FFFF so existing output is
|
|
1096
|
+
// unchanged. (The Go port emits `\x{…}`, whose length is already
|
|
1097
|
+
// variable, and never had the bug.)
|
|
1098
|
+
const astral = 0xFFFF < lo || 0xFFFF < hi
|
|
1099
|
+
const toEsc = (n: number) =>
|
|
1100
|
+
astral
|
|
1101
|
+
? '\\u{' + n.toString(16) + '}'
|
|
1102
|
+
: '\\u' + n.toString(16).padStart(4, '0')
|
|
1103
|
+
return {
|
|
1104
|
+
kind: 'regex',
|
|
1105
|
+
pattern: '[' + toEsc(lo) + '-' + toEsc(hi) + ']',
|
|
1106
|
+
flags: astral ? 'u' : '',
|
|
1107
|
+
sp,
|
|
1108
|
+
}
|
|
1109
|
+
}
|
|
1110
|
+
|
|
1111
|
+
const parts = body.split('.')
|
|
1112
|
+
const chars = parts.map((n) => String.fromCodePoint(codePoint(n)))
|
|
1113
|
+
return { kind: 'term', literal: chars.join(''), sp }
|
|
1114
|
+
}
|
|
1115
|
+
|
|
1116
|
+
|
|
1117
|
+
// Public entry point: take ABNF source and return a tabnas GrammarSpec.
|
|
1118
|
+
|
|
1119
|
+
|
|
1120
|
+
// Convert ABNF source into a tabnas grammar spec: parse this notation,
|
|
1121
|
+
// then hand the IR to the shared compiler. `tag` defaults to 'abnf' so
|
|
1122
|
+
// the emitted alts keep their historical group tag.
|
|
1123
|
+
// Emit a spec from an already-parsed ABNF grammar. Wraps the shared
|
|
1124
|
+
// emitter to keep this package's historical `tag: 'abnf'` default, which
|
|
1125
|
+
// consumers use to group and inspect the emitted alts. An explicit tag
|
|
1126
|
+
// still wins.
|
|
1127
|
+
function emitGrammarSpec(
|
|
1128
|
+
grammar: AbnfGrammar,
|
|
1129
|
+
opts?: AbnfConvertOptions,
|
|
1130
|
+
): GrammarSpec {
|
|
1131
|
+
return bnfEmitGrammarSpec(grammar, { ...opts, tag: opts?.tag ?? 'abnf' })
|
|
1132
|
+
}
|
|
1133
|
+
|
|
1134
|
+
|
|
1135
|
+
function abnf(src: string, opts?: AbnfConvertOptions): GrammarSpec {
|
|
1136
|
+
return emitGrammarSpec(parseAbnf(src), opts)
|
|
1137
|
+
}
|
|
1138
|
+
|
|
1139
|
+
|
|
1140
|
+
export {
|
|
1141
|
+
abnf,
|
|
1142
|
+
parseAbnf,
|
|
1143
|
+
emitGrammarSpec,
|
|
1144
|
+
eliminateLeftRecursion,
|
|
1145
|
+
abnfRules,
|
|
1146
|
+
AbnfParseError,
|
|
1147
|
+
}
|
|
1148
|
+
|
|
1149
|
+
export type {
|
|
1150
|
+
AbnfConvertOptions,
|
|
1151
|
+
AbnfElement,
|
|
1152
|
+
AbnfSequence,
|
|
1153
|
+
AbnfProduction,
|
|
1154
|
+
AbnfGrammar,
|
|
1155
|
+
}
|