@tabnas/abnf 0.4.15 → 0.4.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1155 @@
1
+ /* Copyright (c) 2025-2026 Richard Rodger and other contributors, MIT License */
2
+
3
+ /* converter.ts
4
+ * ABNF -> tabnas grammar spec converter: the RFC 5234 FRONT-END.
5
+ *
6
+ * This file parses ABNF text into the notation-neutral grammar IR that
7
+ * `@tabnas/bnf` compiles. Everything downstream of that IR — desugaring,
8
+ * left-recursion elimination, tail repeats, probe dispatch, literal
9
+ * lifting, token allocation, first-set analysis, chain emission — lives
10
+ * in `@tabnas/bnf` and is shared with the GBNF and EBNF front-ends.
11
+ *
12
+ * ABNF text ──parseAbnf──▶ Grammar ──bnf.emitGrammarSpec──▶ GrammarSpec
13
+ *
14
+ * What stays here is what is genuinely ABNF:
15
+ *
16
+ * - `abnfRules`, the tabnas grammar that reads ABNF syntax itself,
17
+ * and `getAbnfParser`, which installs it on a fresh instance;
18
+ * - the RFC 5234 Appendix B.1 core rules (ALPHA, DIGIT, CRLF, …);
19
+ * - incremental alternatives (`name =/ alt`);
20
+ * - numeric values (`%d65`, `%x41-5A`, `%b1010`, dotted sequences);
21
+ * - case-insensitive quoted strings, the RFC 5234 default.
22
+ *
23
+ * Prose values (`NR = <number>`) are parsed here into the IR's `prose`
24
+ * element; `@tabnas/bnf` resolves them, since a prose terminal naming a
25
+ * built-in lexer token is useful to any notation.
26
+ */
27
+
28
+ import type { GrammarSpec, Rule } from '@tabnas/parser'
29
+ import { util as engineUtil } from '@tabnas/parser'
30
+
31
+ import {
32
+ emitGrammarSpec as bnfEmitGrammarSpec,
33
+ eliminateLeftRecursion,
34
+ refsIn,
35
+ } from '@tabnas/bnf'
36
+
37
+ import type {
38
+ ConvertOptions,
39
+ SrcSpan,
40
+ Element,
41
+ Sequence,
42
+ Production,
43
+ Grammar,
44
+ } from '@tabnas/bnf'
45
+
46
+ // The IR types keep their historical names in this package's public API.
47
+ type AbnfConvertOptions = ConvertOptions
48
+ type AbnfElement = Element
49
+ type AbnfSequence = Sequence
50
+ type AbnfProduction = Production
51
+ type AbnfGrammar = Grammar
52
+
53
+
54
+ // Source span of a token, for the IR (`@tabnas/bnf` SrcSpan). Every
55
+ // field is copied straight off the token: the compiler stores whatever
56
+ // units the front-end's own engine tokens use, precisely so that no
57
+ // arithmetic — and so no off-by-one — happens at this boundary.
58
+ function spanOf(tkn: any): SrcSpan | undefined {
59
+ if (null == tkn || null == tkn.sI) return undefined
60
+ const len = null != tkn.len ? tkn.len : (tkn.src ? String(tkn.src).length : 0)
61
+ return { s: tkn.sI, e: tkn.sI + len, r: tkn.rI, c: tkn.cI }
62
+ }
63
+
64
+
65
+ // Deep-copy a production so a caller cannot reach the shared original.
66
+ // Elements are copied too: they are what a consumer would most likely
67
+ // annotate, and a shallow copy would leave them aliased.
68
+ function cloneProduction(prod: AbnfProduction): AbnfProduction {
69
+ const el = (e: any): any => {
70
+ const o: any = { ...e }
71
+ if (e.inner) o.inner = el(e.inner)
72
+ if (e.alts) o.alts = e.alts.map((alt: any[]) => alt.map(el))
73
+ return o
74
+ }
75
+ return { ...prod, alts: prod.alts.map((alt) => alt.map(el)) }
76
+ }
77
+
78
+
79
+ // Remove every span from a production and everything under it. Used for
80
+ // the RFC 5234 core rules, which are parsed from a string in this file
81
+ // rather than from the user's grammar.
82
+ function stripSpans(prod: AbnfProduction): void {
83
+ delete (prod as any).sp
84
+ const walk = (el: any): void => {
85
+ if (null == el || 'object' !== typeof el) return
86
+ delete el.sp
87
+ if (el.inner) walk(el.inner)
88
+ if (el.alts) for (const alt of el.alts) for (const e of alt) walk(e)
89
+ }
90
+ for (const alt of prod.alts) for (const el of alt) walk(el)
91
+ }
92
+
93
+
94
+ // One span covering two tokens — a group runs from its `(` to its `)`,
95
+ // a bracketed optional from `[` to `]`. Falls back to whichever end is
96
+ // known when the other is not.
97
+ function spanTo(from: any, to: any): SrcSpan | undefined {
98
+ const a = spanOf(from)
99
+ const b = spanOf(to)
100
+ if (null == a) return b
101
+ if (null == b) return a
102
+ return { s: a.s, e: b.e, r: a.r, c: a.c }
103
+ }
104
+
105
+
106
+ // Declarative definition of the ABNF grammar itself, expressed as
107
+ // tabnas rules. Each rule names its `open`/`close` alt list and, where
108
+ // necessary, a `bo`/`bc` state hook for AST assembly.
109
+ //
110
+ // Stage 8: incremental alternatives via `name =/ alt` now fold
111
+ // into the earlier production with the same name. Quoted strings
112
+ // default to case-insensitive (ABNF semantics), `%s` / `%i` force
113
+ // sensitivity explicitly, numeric values and repetition prefixes
114
+ // work as in previous stages.
115
+ //
116
+ // Token vocabulary:
117
+ // #DEF `=` (rule-definition operator)
118
+ // #DEFA `=/` (incremental-alternatives operator)
119
+ // #ALT `/` (alternation)
120
+ // #STAR `*` (repetition separator)
121
+ // #NUM decimal repetition count (matched via match.token)
122
+ // #NV `%[xdb]NN[(-NN|(.NN)*)]` numeric value (match.token)
123
+ // #SS `%s` (case-sensitive string prefix)
124
+ // #SI `%i` (case-insensitive string prefix — same as default)
125
+ // #LP `(`
126
+ // #RP `)`
127
+ // #OB `[` (optional-group open)
128
+ // #CB `]` (optional-group close)
129
+ // #TX bare identifier (tabnas default text token)
130
+ // #ST quoted string literal (tabnas default string token)
131
+ // #ZZ end-of-source
132
+ //
133
+ // Grammar:
134
+ // abnf = production*
135
+ // production = IDENT ('=' / '=/') alts
136
+ // alts = seq ('/' seq)*
137
+ // seq = element*
138
+ // element = repetition? atom
139
+ // repetition = NUM '*' NUM / NUM '*' / '*' NUM / '*' / NUM
140
+ // atom = IDENT | STRING | ['%s' | '%i'] STRING | NUMVAL
141
+ // | '(' alts ')' | '[' alts ']' | PROSE
142
+ // prose = '<' *(%x20-3D / %x3F-7E) '>'
143
+ // numval = '%' ('x' / 'd' / 'b') DIGITS [ '-' DIGITS | ('.' DIGITS)* ]
144
+ const abnfRules: Record<
145
+ string,
146
+ {
147
+ bo?: (r: Rule) => void
148
+ bc?: (r: Rule) => void
149
+ open?: any[]
150
+ close?: any[]
151
+ }
152
+ > = {
153
+ // Top-level: accumulates productions into r.node.
154
+ abnf: {
155
+ bo: (r) => { r.node = [] },
156
+ open: [
157
+ { s: '#ZZ', g: 'empty' },
158
+ { p: 'prod' },
159
+ ],
160
+ close: [{ s: '#ZZ' }],
161
+ },
162
+
163
+ // One production per invocation; tail-recurses (r:'prod') for the
164
+ // next. Inherits its parent's node (the productions array) and
165
+ // appends to it in `bc` once its `alts` child has returned.
166
+ // Production header is `IDENT =` — a bareword rule name followed
167
+ // by the `=` definition operator.
168
+ prod: {
169
+ open: [
170
+ // Standalone definition: name = alts
171
+ {
172
+ s: '#TX #DEF',
173
+ a: (r: Rule) => {
174
+ r.u.name = r.o[0].val
175
+ r.u.nameTkn = r.o[0]
176
+ r.u.incremental = false
177
+ },
178
+ p: 'alts',
179
+ },
180
+ // `<all> = <remove>` — the whole-grammar reset. Prose lexes as its
181
+ // own token (#PV), so it cannot be confused with a rule name or
182
+ // with the `*` repetition operator. The name is kept with its
183
+ // angle brackets, which no #TX rulename can produce, so it can
184
+ // never collide with a production actually called `all`.
185
+ {
186
+ s: '#PV #DEF',
187
+ a: (r: Rule) => {
188
+ r.u.name = (r.o[0].src as string)
189
+ r.u.nameTkn = r.o[0]
190
+ },
191
+ p: 'alts',
192
+ },
193
+
194
+ // Incremental alternatives: name =/ alts
195
+ {
196
+ s: '#TX #DEFA',
197
+ a: (r: Rule) => {
198
+ r.u.name = r.o[0].val
199
+ r.u.nameTkn = r.o[0]
200
+ r.u.incremental = true
201
+ },
202
+ p: 'alts',
203
+ },
204
+ ],
205
+ close: [
206
+ // A TX followed by `=` or `=/` means the next production has
207
+ // begun — back up 2 tokens so a fresh `prod` invocation sees
208
+ // them.
209
+ { s: '#TX #DEF', b: 2, r: 'prod' },
210
+ { s: '#TX #DEFA', b: 2, r: 'prod' },
211
+ { s: '#PV #DEF', b: 2, r: 'prod' },
212
+ { b: 1 },
213
+ ],
214
+ bc: (r) => {
215
+ if (r.child && r.child.node !== undefined) {
216
+ const prod: any = {
217
+ name: r.u.name,
218
+ alts: r.child.node,
219
+ // The name, not the body: that is what an outline entry,
220
+ // go-to-definition and a whole-rule diagnostic want, and an
221
+ // ABNF body can run over many folded lines.
222
+ sp: spanOf(r.u.nameTkn),
223
+ }
224
+ if (r.u.incremental) prod.incremental = true
225
+ r.node.push(prod)
226
+ }
227
+ },
228
+ },
229
+
230
+ // A list of alternative sequences separated by `/` (ABNF
231
+ // alternation). Owns its own array (`bo` resets it) and pushes
232
+ // each seq result in `bc`.
233
+ alts: {
234
+ bo: (r) => { r.node = [] },
235
+ open: [{ p: 'seq' }],
236
+ close: [
237
+ { s: '#ALT', p: 'seq' },
238
+ { b: 1 },
239
+ ],
240
+ bc: (r) => {
241
+ if (r.child && r.child.node !== undefined) {
242
+ r.node.push(r.child.node)
243
+ }
244
+ },
245
+ },
246
+
247
+ // A (possibly empty) sequence of elements. The 2-token lookahead
248
+ // `#TX #DEF` detects a following production boundary and bails
249
+ // out without consuming the tokens; a plain `#TX` at the leading
250
+ // position (tried later so the longer alt wins) is a rule
251
+ // reference inside the current sequence.
252
+ seq: {
253
+ bo: (r) => { r.node = [] },
254
+ open: [
255
+ { s: '#TX #DEF', b: 2, g: 'end' },
256
+ { s: '#TX #DEFA', b: 2, g: 'end' },
257
+ // A rulename followed by a repetition count — `a 1*b`, `a 2b`,
258
+ // `simple-key 1*( dot-sep simple-key )`.
259
+ //
260
+ // This alternative exists for its `s:` pattern, not its action:
261
+ // it widens the tcol the lexer uses for the token AFTER a
262
+ // rulename. The `#TX #DEF` / `#TX #DEFA` lookaheads above are
263
+ // the only other two-token patterns starting with `#TX`, so
264
+ // without this one that tcol is just {#DEF, #DEFA} — the `#NUM`
265
+ // match.token matcher is never offered the position, and the
266
+ // digits fall through to the engine's default matchers. `1*b`
267
+ // then arrives as #NR and fails to parse at all, and `2b`
268
+ // arrives as a #TX bareword, silently misparsing as a reference
269
+ // to a rule named `2b`. Same reason `elem.open` spells its
270
+ // atom position `#ATOM` rather than leaving it implicit.
271
+ { s: '#TX #NUM', b: 2, p: 'elem' },
272
+ // Same problem, same fix, for a rulename followed by another
273
+ // atom. `#ATOM` covers every atom-starter, so the `%xNN` (#NV),
274
+ // `%s"…"` / `%i"…"` (#SS/#SI) and `<prose>` (#PV) matchers are
275
+ // all offered the position. Without it `ws %x5B ws` silently
276
+ // became a reference to a rule named `%x5B`, and `a <foo>` a
277
+ // reference to `<foo>` instead of the prose error it should
278
+ // raise. (The Go port gets these right via `match.tokenEager`,
279
+ // which is what exposed the discrepancy.)
280
+ { s: '#TX #ATOM', b: 2, p: 'elem' },
281
+ // `<all> =` starts the next production; without this the prose
282
+ // would be taken as another element of this sequence.
283
+ { s: '#PV #DEF', b: 2, g: 'end' },
284
+ { s: '#ALT', b: 1, g: 'end' },
285
+ { s: '#ZZ', b: 1, g: 'end' },
286
+ { s: '#RP', b: 1, g: 'end' },
287
+ { s: '#CB', b: 1, g: 'end' },
288
+ // Listing element-starter tokens in `s:` here ensures the
289
+ // tcol-driven matcher considers each one when lexing.
290
+ { s: '#ST', b: 1, p: 'elem' },
291
+ { s: '#NV', b: 1, p: 'elem' },
292
+ { s: '#SS', b: 1, p: 'elem' },
293
+ { s: '#SI', b: 1, p: 'elem' },
294
+ { s: '#PV', b: 1, p: 'elem' },
295
+ { s: '#TX', b: 1, p: 'elem' },
296
+ { s: '#LP', b: 1, p: 'elem' },
297
+ { s: '#OB', b: 1, p: 'elem' },
298
+ { s: '#STAR', b: 1, p: 'elem' },
299
+ { s: '#NUM', b: 1, p: 'elem' },
300
+ { p: 'elem' },
301
+ ],
302
+ close: [
303
+ { s: '#TX #DEF', b: 2, g: 'end' },
304
+ { s: '#TX #DEFA', b: 2, g: 'end' },
305
+ // `<all> = …` starts the next production, exactly as in `open`.
306
+ // `open` has always carried this alternative; `close` did not,
307
+ // and only got away with it because the prose was mis-lexed as a
308
+ // bareword here (so the `#TX #DEF` boundary above caught it by
309
+ // accident). Now that `#TX #ATOM` lets the #PV matcher see the
310
+ // position, the boundary has to be checked properly — and before
311
+ // the `{ s: '#PV', p: 'elem' }` alternative further down, which
312
+ // would otherwise take `<all>` as a prose element of this
313
+ // sequence.
314
+ { s: '#PV #DEF', b: 2, g: 'end' },
315
+ // See the matching alternatives in `open` — these widen the tcol
316
+ // for the token after a rulename, so `a 1*b` / `a 2b` lex as
317
+ // #NUM and `ws %x5B` / `a %s"Q"` / `a <foo>` reach their own
318
+ // matchers instead of falling through to a #TX bareword.
319
+ { s: '#TX #NUM', b: 2, p: 'elem' },
320
+ { s: '#TX #ATOM', b: 2, p: 'elem' },
321
+ { s: '#ALT', b: 1, g: 'end' },
322
+ { s: '#ZZ', b: 1, g: 'end' },
323
+ { s: '#RP', b: 1, g: 'end' },
324
+ { s: '#CB', b: 1, g: 'end' },
325
+ { s: '#ST', b: 1, p: 'elem' },
326
+ { s: '#NV', b: 1, p: 'elem' },
327
+ { s: '#SS', b: 1, p: 'elem' },
328
+ { s: '#SI', b: 1, p: 'elem' },
329
+ { s: '#PV', b: 1, p: 'elem' },
330
+ { s: '#TX', b: 1, p: 'elem' },
331
+ { s: '#LP', b: 1, p: 'elem' },
332
+ { s: '#OB', b: 1, p: 'elem' },
333
+ { s: '#STAR', b: 1, p: 'elem' },
334
+ { s: '#NUM', b: 1, p: 'elem' },
335
+ { b: 1 },
336
+ ],
337
+ },
338
+
339
+ // One element: an optional ABNF repetition prefix (`*A`, `1*A`,
340
+ // `m*nA`, `*nA`, `m*A`, `nA`) followed by an atom. The prefix is
341
+ // matched up front, stored on `r.u.min`/`r.u.max`; then `atom` is
342
+ // pushed to parse the actual element body, whose result is wrapped
343
+ // into an AST node and appended to the parent seq's array in close.
344
+ elem: {
345
+ bo: (r) => { r.u.min = 1; r.u.max = 1 },
346
+ open: [
347
+ // NUM '*' NUM — bounded repetition, followed by the atom
348
+ // itself (listed via the ATOM tokenset so every atom-starter
349
+ // tin — including `#NV` — is in tcol for this position).
350
+ {
351
+ s: '#NUM #STAR #NUM #ATOM',
352
+ b: 1,
353
+ a: (r: Rule) => {
354
+ r.u.min = parseInt(r.o[0].src, 10)
355
+ r.u.max = parseInt(r.o[2].src, 10)
356
+ },
357
+ p: 'atom',
358
+ },
359
+ // NUM '*' — at-least-NUM repetition followed by an atom.
360
+ {
361
+ s: '#NUM #STAR #ATOM',
362
+ b: 1,
363
+ a: (r: Rule) => {
364
+ r.u.min = parseInt(r.o[0].src, 10)
365
+ r.u.max = Infinity
366
+ },
367
+ p: 'atom',
368
+ },
369
+ // '*' NUM — at-most-NUM repetition.
370
+ {
371
+ s: '#STAR #NUM #ATOM',
372
+ b: 1,
373
+ a: (r: Rule) => {
374
+ r.u.min = 0
375
+ r.u.max = parseInt(r.o[1].src, 10)
376
+ },
377
+ p: 'atom',
378
+ },
379
+ // '*' — zero-or-more.
380
+ {
381
+ s: '#STAR #ATOM',
382
+ b: 1,
383
+ a: (r: Rule) => { r.u.min = 0; r.u.max = Infinity },
384
+ p: 'atom',
385
+ },
386
+ // NUM — exact repetition count.
387
+ {
388
+ s: '#NUM #ATOM',
389
+ b: 1,
390
+ a: (r: Rule) => {
391
+ const n = parseInt(r.o[0].src, 10)
392
+ r.u.min = n
393
+ r.u.max = n
394
+ },
395
+ p: 'atom',
396
+ },
397
+ // No prefix — push atom directly (min = max = 1).
398
+ { p: 'atom' },
399
+ ],
400
+ close: [{
401
+ // Wrap the returned atom (r.child.node) based on r.u.min/max
402
+ // and append to the parent seq's array.
403
+ a: (r: Rule) => {
404
+ const item = r.child.node
405
+ const { min, max } = r.u
406
+ if (min === 1 && max === 1) {
407
+ r.node.push(item)
408
+ } else if (min === 0 && max === Infinity) {
409
+ r.node.push({ kind: 'star', inner: item })
410
+ } else if (min === 1 && max === Infinity) {
411
+ r.node.push({ kind: 'plus', inner: item })
412
+ } else if (min === 0 && max === 1) {
413
+ r.node.push({ kind: 'opt', inner: item })
414
+ } else {
415
+ r.node.push({ kind: 'rep', min, max, inner: item })
416
+ }
417
+ },
418
+ }],
419
+ },
420
+
421
+ // The atom body — a bareword ref, quoted-string terminal,
422
+ // parenthesised group, or bracketed optional. Sets its OWN r.node
423
+ // to the AST element so the enclosing `elem` rule can read it
424
+ // from `r.child.node` in its close state.
425
+ atom: {
426
+ bo: (r) => { r.node = undefined },
427
+ open: [
428
+ // Case-sensitive string: %s"foo"
429
+ {
430
+ s: '#SS #ST',
431
+ a: (r: Rule) => {
432
+ r.node = {
433
+ kind: 'term',
434
+ literal: r.o[1].val,
435
+ caseSensitive: true,
436
+ sp: spanTo(r.o[0], r.o[1]),
437
+ }
438
+ },
439
+ },
440
+ // Case-insensitive string: %i"foo" (same as bare "foo" below,
441
+ // but spelled explicitly).
442
+ {
443
+ s: '#SI #ST',
444
+ a: (r: Rule) => {
445
+ r.node = {
446
+ kind: 'term', literal: r.o[1].val, sp: spanTo(r.o[0], r.o[1]),
447
+ }
448
+ },
449
+ },
450
+ // Bare quoted string — case-insensitive per ABNF default.
451
+ {
452
+ s: '#ST',
453
+ a: (r: Rule) => {
454
+ r.node = { kind: 'term', literal: r.o[0].val, sp: spanOf(r.o[0]) }
455
+ },
456
+ },
457
+ {
458
+ s: '#NV',
459
+ a: (r: Rule) => {
460
+ r.node = parseNumericValue(r.o[0].src as string, r.o[0])
461
+ },
462
+ },
463
+ // Prose terminal `<free text>` — carried through as-is; the
464
+ // `resolveProseTerminals` pass decides what it means.
465
+ {
466
+ s: '#PV',
467
+ a: (r: Rule) => {
468
+ const src = r.o[0].src as string
469
+ r.node = {
470
+ kind: 'prose', text: src.slice(1, -1), sp: spanOf(r.o[0]),
471
+ }
472
+ },
473
+ },
474
+ {
475
+ s: '#TX',
476
+ a: (r: Rule) => {
477
+ r.node = { kind: 'ref', name: r.o[0].val, sp: spanOf(r.o[0]) }
478
+ },
479
+ },
480
+ {
481
+ s: '#LP',
482
+ a: (r: Rule) => { r.u.groupKind = 'group'; r.u.open = r.o[0] },
483
+ p: 'alts',
484
+ },
485
+ {
486
+ s: '#OB',
487
+ a: (r: Rule) => { r.u.groupKind = 'opt'; r.u.open = r.o[0] },
488
+ p: 'alts',
489
+ },
490
+ ],
491
+ close: [
492
+ {
493
+ s: '#RP',
494
+ c: (r: Rule) => r.u.groupKind === 'group',
495
+ a: (r: Rule) => {
496
+ r.node = {
497
+ kind: 'group', alts: r.child.node, sp: spanTo(r.u.open, r.c0),
498
+ }
499
+ },
500
+ },
501
+ {
502
+ s: '#CB',
503
+ c: (r: Rule) => r.u.groupKind === 'opt',
504
+ a: (r: Rule) => {
505
+ const bracket = spanTo(r.u.open, r.c0)
506
+ r.node = {
507
+ kind: 'opt',
508
+ inner: { kind: 'group', alts: r.child.node, sp: bracket },
509
+ sp: bracket,
510
+ }
511
+ },
512
+ },
513
+ // For simple atoms (string/ref), r.node is already set by
514
+ // open; we want to pop without consuming the next token.
515
+ // List every token that can legitimately follow an atom so
516
+ // the lexer's tcol-driven match-matcher emits #NUM, #STAR,
517
+ // and friends as their proper types here — otherwise the
518
+ // default number-matcher would lex `1` as #NR and the
519
+ // enclosing seq.close wouldn't recognise the digit as the
520
+ // start of a repetition prefix.
521
+ { s: '#TX', b: 1 },
522
+ { s: '#ST', b: 1 },
523
+ { s: '#NV', b: 1 },
524
+ { s: '#SS', b: 1 },
525
+ { s: '#SI', b: 1 },
526
+ { s: '#PV', b: 1 },
527
+ { s: '#NUM', b: 1 },
528
+ { s: '#STAR', b: 1 },
529
+ { s: '#LP', b: 1 },
530
+ { s: '#OB', b: 1 },
531
+ { s: '#RP', b: 1 },
532
+ { s: '#CB', b: 1 },
533
+ { s: '#ALT', b: 1 },
534
+ { s: '#DEF', b: 1 },
535
+ { s: '#ZZ', b: 1 },
536
+ { b: 1 },
537
+ ],
538
+ },
539
+ }
540
+
541
+ // Cached tabnas instance for the ABNF grammar above; built on first use.
542
+ let _abnfParser: ((src: string) => Production[]) | null = null
543
+
544
+ function getAbnfParser(): (src: string) => AbnfProduction[] {
545
+ if (_abnfParser) return _abnfParser
546
+
547
+ const { Tabnas } = require('@tabnas/parser')
548
+
549
+ // ABNF defines its own grammar from scratch, so we don't load any
550
+ // grammar plugin — just use the bare engine with default tokens.
551
+ const j = new Tabnas({
552
+ rule: { start: 'abnf' },
553
+ fixed: {
554
+ token: {
555
+ // Clear JSON-oriented defaults we're not using so `:`, `,`
556
+ // and `{` have no special meaning inside ABNF source.
557
+ '#OS': null,
558
+ '#CS': null,
559
+ '#CL': null,
560
+ '#CA': null,
561
+ // Re-map `#OB` / `#CB` from JSON's `{` / `}` to ABNF's
562
+ // `[` / `]` optional-group brackets.
563
+ '#OB': '[',
564
+ '#CB': ']',
565
+ '#DEF': '=',
566
+ // `=/` — ABNF's incremental-alternatives operator. Longer
567
+ // than `=`, so tabnas's longest-match-wins fixed matcher
568
+ // tries it first.
569
+ '#DEFA': '=/',
570
+ '#ALT': '/',
571
+ '#STAR': '*',
572
+ '#LP': '(',
573
+ '#RP': ')',
574
+ },
575
+ },
576
+ match: {
577
+ token: {
578
+ // ABNF repetition counts: decimal integers.
579
+ '#NUM': /^[0-9]+/,
580
+ // ABNF numeric value notation:
581
+ // %xNN single hex code point
582
+ // %dNN single decimal code point
583
+ // %bNN single binary code point
584
+ // %xNN-NN hex range
585
+ // %xNN.NN.NN concatenated hex code points (= string)
586
+ // Digits are permissive (hex covers the decimal / binary
587
+ // subsets); `parseNumericValue` re-validates against the
588
+ // actual base.
589
+ '#NV': /^%[xdbXDB][0-9a-fA-F]+(?:[-.][0-9a-fA-F]+)*/,
590
+ // `%s` / `%i` prefixes on a quoted string. The lookahead
591
+ // requires `"` so they don't steal the `%` of `%xNN`.
592
+ '#SS': /^%[sS](?=")/,
593
+ '#SI': /^%[iI](?=")/,
594
+ // RFC 5234 prose-val: `<` free text `>`. The body is every
595
+ // printable char except `>` itself (%x20-3D / %x3F-7E).
596
+ '#PV': /^<[\x20-\x3D\x3F-\x7E]*>/,
597
+ },
598
+ },
599
+ value: {
600
+ // RFC 5234 rulename is `ALPHA *(ALPHA / DIGIT / "-")` — nothing is
601
+ // reserved, so `true`, `false` and `null` are ordinary rule names.
602
+ // JSON's grammar uses all three (`value = false / null / true /
603
+ // object / array / number / string`), and with the engine's default
604
+ // keyword-value lexing they arrived as `#VL` value tokens instead of
605
+ // `#TX` barewords, so no ABNF rendering of JSON would compile.
606
+ // The ABNF meta-grammar has no use for `#VL` at all — this switch
607
+ // only affects the parser that reads ABNF source, not the grammars
608
+ // it emits, where `VL` remains a built-in token name.
609
+ lex: false,
610
+ },
611
+ string: {
612
+ // RFC 5234 char-val has NO escape sequences at all:
613
+ // char-val = DQUOTE *(%x20-21 / %x23-7E) DQUOTE
614
+ // A backslash is just %x5C, an ordinary member of that range, so
615
+ // `"\"` is a one-character literal — and it is a common one, since
616
+ // every RFC that defines `quoted-pair` writes it that way
617
+ // (RFC 5322, RFC 3261, RFC 8259, …). With the engine's default
618
+ // JSON-style escaping the backslash swallowed the closing quote
619
+ // and the grammar died with `unterminated_string`, while `"a\b"`
620
+ // silently became `a<BS>b` instead of the three characters
621
+ // `a`, `\`, `b`.
622
+ //
623
+ // The engine offers no "escaping off" switch that both runtimes
624
+ // share (TS takes `escapeChar: null`, Go falls back to `\` on an
625
+ // empty string), so instead point the escape character at DEL
626
+ // (%x7F) — outside char-val's %x20-21 / %x23-7E body, hence
627
+ // unreachable in any legal ABNF literal.
628
+ escapeChar: '\x7F',
629
+ },
630
+ tokenSet: {
631
+ // Tokens that can legitimately open an atom. Declaring this
632
+ // as a set lets elem.open use `#ATOM` inside its `s:` patterns
633
+ // — that way the tcol at the atom-starter position includes
634
+ // every matcher tin (notably #NV), so the lexer doesn't fall
635
+ // through to #TX when the actual atom is `%xNN`.
636
+ ATOM: ['#ST', '#NV', '#TX', '#LP', '#OB', '#SS', '#SI', '#PV'],
637
+ },
638
+ comment: {
639
+ // ABNF uses `;` to start a line comment. Override tabnas's
640
+ // default `hash` definition (which used `#`) and disable the
641
+ // other comment styles so `//` and `/* */` aren't confused
642
+ // with the alternation operator.
643
+ def: {
644
+ hash: { line: true, start: ';', lex: true, eatline: false },
645
+ slash: null as any,
646
+ multi: null as any,
647
+ },
648
+ },
649
+ })
650
+
651
+ // Drop the default JSON rules — they would otherwise compete with
652
+ // ours for the starting token set.
653
+ const existing = j.rule()
654
+ for (const name of Object.keys(existing)) {
655
+ j.rule(name, null)
656
+ }
657
+
658
+ for (const name of Object.keys(abnfRules)) {
659
+ const spec = abnfRules[name]
660
+ j.rule(name, (rs: any) => {
661
+ if (spec.bo) rs.bo(spec.bo)
662
+ if (spec.bc) rs.bc(spec.bc)
663
+ if (spec.open) rs.open(spec.open)
664
+ if (spec.close) rs.close(spec.close)
665
+ })
666
+ }
667
+
668
+ _abnfParser = (src: string) => j.parse(src) as AbnfProduction[]
669
+ return _abnfParser
670
+ }
671
+
672
+
673
+ // Error raised when the ABNF source itself can't be parsed. Surfaces
674
+ // line and column from the underlying tabnas error so the caller can
675
+ // report them directly. The original error is kept on `.cause`.
676
+ class AbnfParseError extends Error {
677
+ readonly line?: number
678
+ readonly column?: number
679
+ readonly cause?: unknown
680
+ constructor(message: string, location?: { line?: number; column?: number }, cause?: unknown) {
681
+ super(message)
682
+ this.name = 'AbnfParseError'
683
+ this.line = location?.line
684
+ this.column = location?.column
685
+ this.cause = cause
686
+ }
687
+ }
688
+
689
+
690
+ // Parse ABNF source into a grammar AST via the tabnas-based parser.
691
+ function parseAbnf(src: string): AbnfGrammar {
692
+ const parser = getAbnfParser()
693
+ let productions: AbnfProduction[]
694
+ try {
695
+ productions = parser(src) ?? []
696
+ } catch (e: any) {
697
+ // TabnasError carries `lineNumber` / `columnNumber`; fall back to
698
+ // ad-hoc extraction from the error message otherwise.
699
+ const line = e?.lineNumber ?? e?.row
700
+ const column = e?.columnNumber ?? e?.col
701
+ const loc = (line != null && column != null)
702
+ ? ` at line ${line}, column ${column}`
703
+ : ''
704
+ const raw = e?.message ? String(e.message).split('\n')[0] : String(e)
705
+ throw new AbnfParseError(
706
+ `abnf: parse error${loc}: ${raw}`,
707
+ { line, column },
708
+ e,
709
+ )
710
+ }
711
+ if (!Array.isArray(productions) || productions.length === 0) {
712
+ throw new AbnfParseError('abnf: no productions found')
713
+ }
714
+ // BEFORE merging, not after. `mergeIncrementals` drops each `=/`
715
+ // production, keeping only the base's span — so an annotation on an
716
+ // incremental line was resolved against a production list that no
717
+ // longer contained the line it followed, and attached to whatever rule
718
+ // happened to be declared before it instead. `a = "a"`, `b = 1*DIGIT`,
719
+ // `a =/ "c" ; @object` silently annotated **b**.
720
+ attachValueAnnotations(src, productions)
721
+ const merged = mergeIncrementals(productions)
722
+ return { productions: withCoreRules(merged) }
723
+ }
724
+
725
+
726
+ // A trailing comment claiming a value annotation:
727
+ //
728
+ // ver = maj "." min "." pat ; @object maj min pat
729
+ // tags = tag *("," tag) ; @array
730
+ //
731
+ // RFC 5234 has nowhere else to put this. A comment is the only place in
732
+ // the notation that carries no meaning of its own, which is exactly why
733
+ // it can carry one here without changing what the grammar accepts: strip
734
+ // every annotation and the same language parses, just into a tree
735
+ // instead of a value.
736
+ //
737
+ // ONLY `@object` and `@array` are claimed. Any other `; @…` comment is
738
+ // left alone — the notation has no directive namespace, so this must not
739
+ // assume one, and a reader's own `; @deprecated` has to keep meaning
740
+ // nothing.
741
+ const ANNOTATION = /^@(object|array)\b\s*(.*)$/
742
+ const MEMBER_NAME = /^[A-Za-z][A-Za-z0-9-]*$/
743
+
744
+
745
+ // Every `;` comment in the source that claims an annotation, with the
746
+ // offset it starts at.
747
+ //
748
+ // Quoted strings and prose are skipped: a `;` inside `"a;b"` or
749
+ // `<a;b>` is CONTENT, not a comment, and treating it as one would
750
+ // silently attach an annotation that the author did not write.
751
+ function* annotationComments(
752
+ src: string,
753
+ ): Generator<{ at: number; body: string }> {
754
+ for (let i = 0; i < src.length; i++) {
755
+ const ch = src[i]
756
+ if ('"' === ch || '<' === ch) {
757
+ // RFC 5234 char-val and prose-val have no escapes, so the next
758
+ // closing mark ends them.
759
+ const close = src.indexOf('"' === ch ? '"' : '>', i + 1)
760
+ i = close < 0 ? src.length : close
761
+ continue
762
+ }
763
+ if (';' !== ch) continue
764
+ let end = src.indexOf('\n', i)
765
+ if (end < 0) end = src.length
766
+ const body = src.slice(i + 1, end).trim()
767
+ if (body.startsWith('@')) yield { at: i, body }
768
+ i = end
769
+ }
770
+ }
771
+
772
+
773
+ // Attach each annotation to the production it FOLLOWS — the last one
774
+ // that begins before it.
775
+ //
776
+ // Not "the production on the same line": a rule may be written across
777
+ // several lines, and an author putting the annotation on the last of
778
+ // them means the same thing. Following the definition is the rule that
779
+ // reads the same either way.
780
+ function attachValueAnnotations(
781
+ src: string,
782
+ prods: AbnfProduction[],
783
+ ): void {
784
+ const ordered = prods
785
+ .filter((p) => null != p.sp)
786
+ .slice()
787
+ .sort((a, b) => (a.sp as SrcSpan).s - (b.sp as SrcSpan).s)
788
+ if (0 === ordered.length) return
789
+
790
+ let idx = 0
791
+ for (const { at, body } of annotationComments(src)) {
792
+ const m = ANNOTATION.exec(body)
793
+ if (null == m) continue
794
+
795
+ // Both `ordered` and the comments are in source order, so the search
796
+ // only ever moves FORWARD — `idx` is not reset per comment. Restarting
797
+ // it made attachment quadratic in the number of annotated rules, which
798
+ // a generated grammar can make expensive for nothing.
799
+ while (idx < ordered.length && (ordered[idx].sp as SrcSpan).s < at) idx++
800
+ const owner: AbnfProduction | undefined =
801
+ 0 < idx ? ordered[idx - 1] : undefined
802
+ if (null == owner) {
803
+ throw new AbnfParseError(
804
+ `abnf: '; ${body}' appears before any rule, so there is nothing ` +
805
+ `for it to annotate. A value annotation goes after the rule it ` +
806
+ `describes.`,
807
+ )
808
+ }
809
+
810
+ const kind = m[1] as 'object' | 'array'
811
+ const members = m[2].split(/[\s,]+/).filter((w) => '' !== w)
812
+
813
+ if ('array' === kind) {
814
+ if (0 < members.length) {
815
+ throw new AbnfParseError(
816
+ `abnf: rule '${owner.name}': '@array' names no members — every ` +
817
+ `part that produces a value becomes an element, in order. Got ` +
818
+ `'${members.join(' ')}'.`,
819
+ )
820
+ }
821
+ } else {
822
+ const seen = new Set<string>()
823
+ for (const name of members) {
824
+ if (!MEMBER_NAME.test(name)) {
825
+ throw new AbnfParseError(
826
+ `abnf: rule '${owner.name}': '${name}' is not a rule name, so ` +
827
+ `it cannot name a member of '@object'.`,
828
+ )
829
+ }
830
+ // Each member is a separate KEY. Two parts named the same thing
831
+ // both write to it, so the second silently overwrites the first
832
+ // and that much of the input is gone from the result.
833
+ if (seen.has(name)) {
834
+ throw new AbnfParseError(
835
+ `abnf: rule '${owner.name}': '@object' names '${name}' twice. ` +
836
+ `Each member is a separate key, so the second part would ` +
837
+ `overwrite the first. Give them different names.`,
838
+ )
839
+ }
840
+ seen.add(name)
841
+ }
842
+ }
843
+
844
+ if (null != owner.value) {
845
+ throw new AbnfParseError(
846
+ `abnf: rule '${owner.name}' has more than one value annotation. A ` +
847
+ `rule builds one thing.`,
848
+ )
849
+ }
850
+ owner.value = 'array' === kind ? { kind } : { kind, members }
851
+ }
852
+ }
853
+
854
+
855
+ // RFC 5234 Appendix B.1 core rules. Parsed lazily on first use
856
+ // and spliced into any user grammar that references them but
857
+ // doesn't define them locally.
858
+ const CORE_RULES_ABNF = `
859
+ ALPHA = %x41-5A / %x61-7A
860
+ BIT = "0" / "1"
861
+ CHAR = %x01-7F
862
+ CR = %x0D
863
+ LF = %x0A
864
+ CRLF = CR LF
865
+ CTL = %x00-1F / %x7F
866
+ DIGIT = %x30-39
867
+ DQUOTE = %x22
868
+ HEXDIG = DIGIT / "A" / "B" / "C" / "D" / "E" / "F"
869
+ HTAB = %x09
870
+ OCTET = %x00-FF
871
+ SP = %x20
872
+ VCHAR = %x21-7E
873
+ WSP = SP / HTAB
874
+ LWSP = *( WSP / CRLF WSP )
875
+ `
876
+
877
+ let _coreRules: Map<string, AbnfProduction> | null = null
878
+
879
+ function getCoreRules(): Map<string, AbnfProduction> {
880
+ if (_coreRules) return _coreRules
881
+ const parser = getAbnfParser()
882
+ const raw = parser(CORE_RULES_ABNF) as AbnfProduction[]
883
+ // Core rules flatten to `src` in the output AST — they're
884
+ // character-class bricks, not structural nodes users want to see
885
+ // one-per-matched-character.
886
+ for (const p of raw) p.nodeKind = 'core'
887
+
888
+ // Strip source spans from the core rules. They are parsed from
889
+ // CORE_RULES_ABNF, a string in THIS FILE, so their spans are offsets
890
+ // into a document the user never wrote — an editor asked to reveal
891
+ // one would jump to a position in the user's grammar that has nothing
892
+ // to do with ALPHA or DIGIT. A missing span means "nowhere to point",
893
+ // which is exactly right for a rule the library supplied; a wrong one
894
+ // is worse than none.
895
+ //
896
+ // It also removes an aliasing hazard: this map is module-cached and
897
+ // handed out BY REFERENCE to every grammar compiled in the process,
898
+ // so any position on it would be shared across unrelated documents.
899
+ //
900
+ // A reference TO a core rule still carries a span — that reference is
901
+ // in the user's source, and it is what a diagnostic points at.
902
+ for (const p of raw) stripSpans(p)
903
+ _coreRules = new Map(raw.map((p) => [p.name, p]))
904
+ return _coreRules
905
+ }
906
+
907
+
908
+ // Add each RFC 5234 core rule that the user's grammar references
909
+ // but doesn't define locally. Resolution is transitive: if the
910
+ // user mentions HEXDIG, DIGIT is pulled in too. User definitions
911
+ // always win — a local `DIGIT = …` is left untouched.
912
+ function withCoreRules(user: AbnfProduction[]): AbnfProduction[] {
913
+ const core = getCoreRules()
914
+ const defined = new Set(user.map((p) => p.name))
915
+ const needed = new Set<string>()
916
+
917
+ // A malformed element reaches here as a hole in an alt — an unclosed group
918
+ // (`( "a" / "b"` with no `)`) pops without ever building its node, leaving
919
+ // `undefined` in the sequence. bnf's refsIn then reads `.kind` off it and
920
+ // throws a TypeError, which is a CRASH, not a rejection: the conformance
921
+ // harness scored it as a correct rejection for every base grammar in the
922
+ // corpus, and abnf's own record says such input is rejected.
923
+ //
924
+ // Reject it here, as the parse error it is. The deeper repair — having the
925
+ // `elem` rule refuse to close a group on anything but its own `)` — is a
926
+ // grammar change and is deliberately not attempted in the same commit as the
927
+ // harness fix that exposed this.
928
+ //
929
+ // The walk MIRRORS bnf's `refsIn` exactly, because "where does refsIn
930
+ // dereference?" is the definition of where a hole crashes. A top-level scan
931
+ // of the sequence is not enough: when a repetition wraps the unclosed group
932
+ // — `bad = *( "a"` — `elem.close` builds a perfectly real `star` whose
933
+ // `inner` is the hole, so the sequence entry is non-null and refsIn walks
934
+ // into `[undefined]` one level down. `*[`, `1*2(` and a group's `alts` all
935
+ // reach it the same way.
936
+ const holeIn = (alt: readonly (AbnfElement | undefined)[] | undefined): boolean => {
937
+ if (null == alt) return true
938
+ for (const el of alt) {
939
+ if (null == el) return true
940
+ const k = (el as any).kind
941
+ if ('opt' === k || 'star' === k || 'plus' === k || 'rep' === k) {
942
+ if (holeIn([(el as any).inner])) return true
943
+ } else if ('group' === k) {
944
+ const alts = (el as any).alts
945
+ if (null == alts) return true
946
+ for (const a of alts) if (holeIn(a)) return true
947
+ }
948
+ }
949
+ return false
950
+ }
951
+
952
+ const rejectHoles = (prods: AbnfProduction[]) => {
953
+ for (const p of prods) {
954
+ if (null == p.alts) {
955
+ throw new AbnfParseError(
956
+ `abnf: rule '${p.name}' is malformed — no alternatives were built.`)
957
+ }
958
+ for (const alt of p.alts) {
959
+ if (holeIn(alt)) {
960
+ throw new AbnfParseError(
961
+ `abnf: rule '${p.name}' is malformed — an element could not be ` +
962
+ `built. The usual cause is an unclosed group or option.`)
963
+ }
964
+ }
965
+ }
966
+ }
967
+ rejectHoles(user)
968
+
969
+ const scan = (prods: AbnfProduction[]) => {
970
+ for (const p of prods) {
971
+ for (const alt of p.alts) refsIn(alt, needed)
972
+ }
973
+ }
974
+
975
+ scan(user)
976
+ const out: AbnfProduction[] = []
977
+ // Transitively add core rules, in declaration order.
978
+ let added = true
979
+ while (added) {
980
+ added = false
981
+ for (const [name, prod] of core) {
982
+ if (defined.has(name)) continue
983
+ if (!needed.has(name)) continue
984
+ defined.add(name)
985
+ // A COPY, not the cached production. `getCoreRules` hands back a
986
+ // module-level map, so pushing `prod` itself would put the same
987
+ // object into every grammar parsed in this process — and
988
+ // `parseAbnf` returns it to the caller. A consumer that annotated
989
+ // an ALPHA node (writing a span onto it, say) would then see that
990
+ // annotation on unrelated documents, and the "core rules carry no
991
+ // span" guarantee would hold only until someone broke it for
992
+ // everyone. Cloning is cheap: these are a dozen small
993
+ // character-class rules, and only the referenced ones are added.
994
+ const copy = cloneProduction(prod)
995
+ out.push(copy)
996
+ scan([copy])
997
+ added = true
998
+ }
999
+ }
1000
+ return [...user, ...out]
1001
+ }
1002
+
1003
+
1004
+ // Fold every `name =/ alt` production into the earlier production
1005
+ // with the same name by appending its alternatives. Throws if an
1006
+ // incremental references a name that hasn't been defined yet — ABNF
1007
+ // requires the base production to appear first.
1008
+ function mergeIncrementals(prods: AbnfProduction[]): AbnfProduction[] {
1009
+ const out: AbnfProduction[] = []
1010
+ const byName = new Map<string, AbnfProduction>()
1011
+ for (const p of prods) {
1012
+ if (p.incremental) {
1013
+ const base = byName.get(p.name)
1014
+ if (!base) {
1015
+ throw new AbnfParseError(
1016
+ `abnf: '${p.name} =/ …' has no earlier '${p.name} = …' to extend`,
1017
+ )
1018
+ }
1019
+ base.alts.push(...p.alts)
1020
+ // This production is about to be dropped, and annotations are now
1021
+ // attached before that happens — so an annotation on the `=/` line
1022
+ // has to move to the base, which IS the rule it describes.
1023
+ if (p.value) {
1024
+ if (base.value) {
1025
+ throw new AbnfParseError(
1026
+ `abnf: rule '${p.name}' has more than one value annotation. A ` +
1027
+ `rule builds one thing.`,
1028
+ )
1029
+ }
1030
+ base.value = p.value
1031
+ }
1032
+ continue
1033
+ }
1034
+ // Strip the (absent) flag on a cleanly-written production so
1035
+ // downstream code never sees it.
1036
+ // Rebuilt field by field, so every field carried on a production
1037
+ // has to be listed here or it is silently dropped — `sp` included.
1038
+ const clean: AbnfProduction = { name: p.name, alts: p.alts, sp: p.sp }
1039
+ if (p.nodeKind) clean.nodeKind = p.nodeKind
1040
+ // Annotations are attached BEFORE this runs, so this carry is live:
1041
+ // without it every annotation in the grammar would vanish here
1042
+ // without a word. The compiler downstream had six rebuilds like this
1043
+ // one and shipped with all six dropping it.
1044
+ if (p.value) clean.value = p.value
1045
+ out.push(clean)
1046
+ byName.set(p.name, clean)
1047
+ }
1048
+ return out
1049
+ }
1050
+
1051
+
1052
+ // Decode an ABNF numeric value (`%xNN`, `%dNN`, `%bNN`, or one of
1053
+ // the range/concatenation forms) into a `AbnfElement`.
1054
+ //
1055
+ // %x61 => single-char term "a"
1056
+ // %x66.6f.6f => concatenated term "foo"
1057
+ // %x30-39 => regex character class [\u0030-\u0039]
1058
+ //
1059
+ // Hex is case-insensitive; decimal and binary accept only digits
1060
+ // in their respective ranges. Range endpoints must be the same
1061
+ // base as the prefix (RFC 5234 doesn't allow mixing).
1062
+ function parseNumericValue(src: string, tkn?: any): AbnfElement {
1063
+ const sp = spanOf(tkn)
1064
+ const base = src[1].toLowerCase()
1065
+ const radix = base === 'x' ? 16 : base === 'd' ? 10 : 2
1066
+ const body = src.slice(2)
1067
+
1068
+ // RFC 5234 puts no ceiling on a numeric value, but Unicode does:
1069
+ // nothing above U+10FFFF is a code point. Check it here so an
1070
+ // out-of-range grammar gets an ABNF diagnostic naming the offending
1071
+ // value, rather than a bare `RangeError: Invalid code point` from
1072
+ // String.fromCodePoint (or, as before, a silently truncated
1073
+ // character from String.fromCharCode).
1074
+ const codePoint = (text: string): number => {
1075
+ const n = parseInt(text, radix)
1076
+ if (!Number.isFinite(n) || n < 0 || 0x10FFFF < n) {
1077
+ throw new Error(
1078
+ `numeric value '%${src[1]}${text}' is ${n}, which is not a ` +
1079
+ `Unicode code point (the maximum is %x10FFFF).`)
1080
+ }
1081
+ return n
1082
+ }
1083
+
1084
+ if (body.includes('-')) {
1085
+ const [loStr, hiStr] = body.split('-')
1086
+ const lo = codePoint(loStr)
1087
+ const hi = codePoint(hiStr)
1088
+ if (lo === hi) {
1089
+ return { kind: 'term', literal: String.fromCodePoint(lo), sp }
1090
+ }
1091
+ // `\uXXXX` only reaches U+FFFF: `%xE000-10FFFF` (JSONPath, TOML)
1092
+ // became `[-ჿff]`, which JS reads as the range E000–10FF
1093
+ // plus a literal `ff` — "Range out of order". Above the BMP the
1094
+ // escape has to be `\u{…}`, which in turn requires the `u` flag.
1095
+ // Stay on the plain form below U+FFFF so existing output is
1096
+ // unchanged. (The Go port emits `\x{…}`, whose length is already
1097
+ // variable, and never had the bug.)
1098
+ const astral = 0xFFFF < lo || 0xFFFF < hi
1099
+ const toEsc = (n: number) =>
1100
+ astral
1101
+ ? '\\u{' + n.toString(16) + '}'
1102
+ : '\\u' + n.toString(16).padStart(4, '0')
1103
+ return {
1104
+ kind: 'regex',
1105
+ pattern: '[' + toEsc(lo) + '-' + toEsc(hi) + ']',
1106
+ flags: astral ? 'u' : '',
1107
+ sp,
1108
+ }
1109
+ }
1110
+
1111
+ const parts = body.split('.')
1112
+ const chars = parts.map((n) => String.fromCodePoint(codePoint(n)))
1113
+ return { kind: 'term', literal: chars.join(''), sp }
1114
+ }
1115
+
1116
+
1117
+ // Public entry point: take ABNF source and return a tabnas GrammarSpec.
1118
+
1119
+
1120
+ // Convert ABNF source into a tabnas grammar spec: parse this notation,
1121
+ // then hand the IR to the shared compiler. `tag` defaults to 'abnf' so
1122
+ // the emitted alts keep their historical group tag.
1123
+ // Emit a spec from an already-parsed ABNF grammar. Wraps the shared
1124
+ // emitter to keep this package's historical `tag: 'abnf'` default, which
1125
+ // consumers use to group and inspect the emitted alts. An explicit tag
1126
+ // still wins.
1127
+ function emitGrammarSpec(
1128
+ grammar: AbnfGrammar,
1129
+ opts?: AbnfConvertOptions,
1130
+ ): GrammarSpec {
1131
+ return bnfEmitGrammarSpec(grammar, { ...opts, tag: opts?.tag ?? 'abnf' })
1132
+ }
1133
+
1134
+
1135
+ function abnf(src: string, opts?: AbnfConvertOptions): GrammarSpec {
1136
+ return emitGrammarSpec(parseAbnf(src), opts)
1137
+ }
1138
+
1139
+
1140
+ export {
1141
+ abnf,
1142
+ parseAbnf,
1143
+ emitGrammarSpec,
1144
+ eliminateLeftRecursion,
1145
+ abnfRules,
1146
+ AbnfParseError,
1147
+ }
1148
+
1149
+ export type {
1150
+ AbnfConvertOptions,
1151
+ AbnfElement,
1152
+ AbnfSequence,
1153
+ AbnfProduction,
1154
+ AbnfGrammar,
1155
+ }