@tabnas/parser 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/error.js +5 -0
- package/dist/error.js.map +1 -1
- package/dist/lexer.d.ts +6 -3
- package/dist/lexer.js +189 -89
- package/dist/lexer.js.map +1 -1
- package/dist/merge.d.ts +3 -0
- package/dist/merge.js +482 -0
- package/dist/merge.js.map +1 -0
- package/dist/parser.d.ts +1 -0
- package/dist/parser.js +35 -6
- package/dist/parser.js.map +1 -1
- package/dist/rules.d.ts +10 -3
- package/dist/rules.js +121 -52
- package/dist/rules.js.map +1 -1
- package/dist/tabnas.d.ts +1 -0
- package/dist/tabnas.js +49 -51
- package/dist/tabnas.js.map +1 -1
- package/dist/tsconfig.tsbuildinfo +1 -1
- package/dist/types.d.ts +1 -0
- package/dist/types.js.map +1 -1
- package/dist/utility.js +80 -1
- package/dist/utility.js.map +1 -1
- package/package.json +1 -1
package/dist/lexer.js
CHANGED
|
@@ -42,27 +42,50 @@ exports.Point = Point;
|
|
|
42
42
|
const makePoint = (...params) => new Point(...params);
|
|
43
43
|
exports.makePoint = makePoint;
|
|
44
44
|
// A single lexed token: numeric token id, JS-typed value, raw source text, and match position.
|
|
45
|
+
//
|
|
46
|
+
// NOTE: `src` is a prototype accessor over a lazily-materialized backing
|
|
47
|
+
// field — matchers that only know the token's span (space, line,
|
|
48
|
+
// comment, quoted-string raw text) defer the substring until someone
|
|
49
|
+
// actually reads it, so ignored tokens never allocate one. Reading
|
|
50
|
+
// tkn.src always yields the correct string; the one observable
|
|
51
|
+
// difference from a plain data property is that `src` is not an OWN
|
|
52
|
+
// enumerable property, so Object.keys(tkn), {...tkn}, and
|
|
53
|
+
// JSON.stringify(tkn) do not include it.
|
|
45
54
|
class Token {
|
|
46
|
-
|
|
55
|
+
#src; // Materialized source text (undefined = not yet).
|
|
56
|
+
#ref; // Full source backing the [sI, sI+len) span.
|
|
57
|
+
constructor(name, tin, val, src, pnt, use, why, ref, len) {
|
|
47
58
|
this.isToken = true; // Marker discriminating Tokens from other values.
|
|
48
59
|
this.name = types_1.EMPTY; // Token name (e.g. '#NR', '#ST').
|
|
49
60
|
this.tin = -1; // Numeric token id corresponding to name.
|
|
50
61
|
this.val = undefined; // JS-typed value (e.g. a number for #NR).
|
|
51
|
-
this.src = types_1.EMPTY; // Raw matching source text.
|
|
52
62
|
this.sI = -1; // Source index where the match started.
|
|
53
63
|
this.rI = -1; // Row where the match started.
|
|
54
64
|
this.cI = -1; // Column where the match started.
|
|
55
65
|
this.len = -1; // Length of src.
|
|
56
66
|
this.name = name;
|
|
57
67
|
this.tin = tin;
|
|
58
|
-
this
|
|
68
|
+
this.#src = src;
|
|
69
|
+
this.#ref = ref;
|
|
59
70
|
this.val = val;
|
|
60
71
|
this.sI = pnt.sI;
|
|
61
72
|
this.rI = pnt.rI;
|
|
62
73
|
this.cI = pnt.cI;
|
|
63
74
|
this.use = use;
|
|
64
75
|
this.why = why;
|
|
65
|
-
this.len = null == src ? 0 : src.length;
|
|
76
|
+
this.len = null != len ? len : null == src ? 0 : src.length;
|
|
77
|
+
}
|
|
78
|
+
get src() {
|
|
79
|
+
let s = this.#src;
|
|
80
|
+
if (undefined === s) {
|
|
81
|
+
const ref = this.#ref;
|
|
82
|
+
s = this.#src =
|
|
83
|
+
undefined === ref ? types_1.EMPTY : ref.substring(this.sI, this.sI + this.len);
|
|
84
|
+
}
|
|
85
|
+
return s;
|
|
86
|
+
}
|
|
87
|
+
set src(s) {
|
|
88
|
+
this.#src = s;
|
|
66
89
|
}
|
|
67
90
|
resolveVal(rule, ctx) {
|
|
68
91
|
let out = 'function' === typeof this.val ? this.val(rule, ctx) : this.val;
|
|
@@ -116,6 +139,8 @@ function guardedMatcher(mcfg, body) {
|
|
|
116
139
|
if (!mcfg.lex)
|
|
117
140
|
return undefined;
|
|
118
141
|
if (mcfg.check) {
|
|
142
|
+
// Check hooks are user code and may read lex.fwd directly.
|
|
143
|
+
lex.refwd();
|
|
119
144
|
const r = mcfg.check(lex);
|
|
120
145
|
if (r && r.done)
|
|
121
146
|
return r.token;
|
|
@@ -329,23 +354,48 @@ const STRING_BODY_TABLE = new Int32Array([
|
|
|
329
354
|
CONSUME | IS_ROW,
|
|
330
355
|
]);
|
|
331
356
|
let makeFixedMatcher = (cfg, _opts) => {
|
|
332
|
-
|
|
357
|
+
// First-char dispatch table replacing the old anchored alternation
|
|
358
|
+
// regex: match cost per position becomes one charCode index plus (for
|
|
359
|
+
// multi-char tokens) a startsWith verify, with no match-array or
|
|
360
|
+
// capture substring allocation. Candidates sharing a first char are
|
|
361
|
+
// ordered longest-first, preserving the regex's longest-match-wins
|
|
362
|
+
// ordering (utility.ts sorts the alternation the same way).
|
|
363
|
+
const table = new Array(256);
|
|
364
|
+
const wide = []; // Tokens whose first char is >= U+0100.
|
|
365
|
+
const byLenDesc = (a, b) => b.len - a.len || (a.src < b.src ? -1 : a.src > b.src ? 1 : 0);
|
|
366
|
+
for (const fsrc of (0, utility_1.keys)(cfg.fixed.token)) {
|
|
367
|
+
const tin = cfg.fixed.token[fsrc];
|
|
368
|
+
if (null == tin || 0 === fsrc.length)
|
|
369
|
+
continue;
|
|
370
|
+
const cand = { src: fsrc, len: fsrc.length, tin };
|
|
371
|
+
const cc = fsrc.charCodeAt(0);
|
|
372
|
+
if (cc < 256) {
|
|
373
|
+
;
|
|
374
|
+
(table[cc] = table[cc] || []).push(cand);
|
|
375
|
+
}
|
|
376
|
+
else {
|
|
377
|
+
wide.push(cand);
|
|
378
|
+
}
|
|
379
|
+
}
|
|
380
|
+
for (const cands of table) {
|
|
381
|
+
if (cands)
|
|
382
|
+
cands.sort(byLenDesc);
|
|
383
|
+
}
|
|
384
|
+
wide.sort(byLenDesc);
|
|
333
385
|
return guardedMatcher(cfg.fixed, function fixedBody(lex) {
|
|
334
|
-
const
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
if (
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
pnt.cI += mlen;
|
|
348
|
-
}
|
|
386
|
+
const pnt = lex.pnt;
|
|
387
|
+
const src = lex.src;
|
|
388
|
+
const cc = src.charCodeAt(pnt.sI);
|
|
389
|
+
const cands = cc < 256 ? table[cc] : wide;
|
|
390
|
+
if (undefined === cands)
|
|
391
|
+
return undefined;
|
|
392
|
+
for (const cand of cands) {
|
|
393
|
+
// Single-char Latin-1 candidates already matched via the table
|
|
394
|
+
// index; longer (or wide-char) candidates verify in place.
|
|
395
|
+
if ((1 === cand.len && cc < 256) || src.startsWith(cand.src, pnt.sI)) {
|
|
396
|
+
const tkn = lex.token(cand.tin, undefined, cand.src, pnt);
|
|
397
|
+
pnt.sI += cand.len;
|
|
398
|
+
pnt.cI += cand.len;
|
|
349
399
|
return tkn;
|
|
350
400
|
}
|
|
351
401
|
}
|
|
@@ -367,7 +417,9 @@ let makeMatchMatcher = (cfg, _opts) => {
|
|
|
367
417
|
}
|
|
368
418
|
return guardedMatcher(cfg.match, function matchBody(lex, rule, tI = 0) {
|
|
369
419
|
let pnt = lex.pnt;
|
|
370
|
-
|
|
420
|
+
// Value/token matcher regexes are documented to run against the
|
|
421
|
+
// remainder string, so materialize it (memoized per position).
|
|
422
|
+
let fwd = lex.refwd();
|
|
371
423
|
let oc = 'o' === rule.state ? 0 : 1;
|
|
372
424
|
for (let valueMatcher of valueMatchers) {
|
|
373
425
|
if (valueMatcher.match instanceof RegExp) {
|
|
@@ -474,9 +526,11 @@ let makeCommentMatcher = (cfg, opts) => {
|
|
|
474
526
|
// so the table and class array are reused across comment matches.
|
|
475
527
|
const lineRunSpec = buildLineRunSpec(cfg.line);
|
|
476
528
|
const scanOut = { sI: 0, rI: 0, cI: 0 };
|
|
529
|
+
// The body walks lex.src with absolute indices (aI) — no remainder
|
|
530
|
+
// slice is materialized per position.
|
|
477
531
|
return guardedMatcher(cfg.comment, function commentBody(lex) {
|
|
478
532
|
let pnt = lex.pnt;
|
|
479
|
-
let
|
|
533
|
+
let src = lex.src;
|
|
480
534
|
let rI = pnt.rI;
|
|
481
535
|
let cI = pnt.cI;
|
|
482
536
|
// Single line comment.
|
|
@@ -485,32 +539,32 @@ let makeCommentMatcher = (cfg, opts) => {
|
|
|
485
539
|
const lineChars = cfg.line.chars;
|
|
486
540
|
const rowChars = cfg.line.rowChars;
|
|
487
541
|
for (let mc of lineComments) {
|
|
488
|
-
if (
|
|
489
|
-
let
|
|
490
|
-
let
|
|
542
|
+
if (src.startsWith(mc.start, pnt.sI)) {
|
|
543
|
+
let srclen = src.length;
|
|
544
|
+
let aI = pnt.sI + mc.start.length;
|
|
491
545
|
cI += mc.start.length;
|
|
492
546
|
let suffixLen = 0;
|
|
493
547
|
let cc;
|
|
494
|
-
while (
|
|
495
|
-
!((cc =
|
|
548
|
+
while (aI < srclen &&
|
|
549
|
+
!((cc = src.charCodeAt(aI)) < 256
|
|
496
550
|
? lineBM[cc]
|
|
497
|
-
: lineChars[
|
|
498
|
-
let n = commentSuffixMatch(
|
|
551
|
+
: lineChars[src[aI]])) {
|
|
552
|
+
let n = commentSuffixMatch(src, aI, mc.suffixes);
|
|
499
553
|
if (n > 0) {
|
|
500
554
|
suffixLen = n;
|
|
501
555
|
break;
|
|
502
556
|
}
|
|
503
|
-
n = commentSuffixFnMatch(lex,
|
|
557
|
+
n = commentSuffixFnMatch(lex, aI, mc.suffixFn);
|
|
504
558
|
if (n > 0) {
|
|
505
559
|
suffixLen = n;
|
|
506
560
|
break;
|
|
507
561
|
}
|
|
508
562
|
cI++;
|
|
509
|
-
|
|
563
|
+
aI++;
|
|
510
564
|
}
|
|
511
565
|
if (suffixLen > 0) {
|
|
512
566
|
// Consume the suffix as the tail of the comment body.
|
|
513
|
-
|
|
567
|
+
aI += suffixLen;
|
|
514
568
|
cI += suffixLen;
|
|
515
569
|
}
|
|
516
570
|
else if (mc.eatline) {
|
|
@@ -518,13 +572,12 @@ let makeCommentMatcher = (cfg, opts) => {
|
|
|
518
572
|
// a line char (not from a suffix match). cI is intentionally
|
|
519
573
|
// NOT pulled from scanOut — current semantics leave the
|
|
520
574
|
// pnt.cI at end-of-comment-body even after eating newlines.
|
|
521
|
-
scan(
|
|
575
|
+
scan(src, aI, rI, cI, lineRunSpec, scanOut);
|
|
522
576
|
rI = scanOut.rI;
|
|
523
|
-
|
|
577
|
+
aI = scanOut.sI;
|
|
524
578
|
}
|
|
525
|
-
let
|
|
526
|
-
|
|
527
|
-
pnt.sI += csrc.length;
|
|
579
|
+
let tkn = lex.token('#CM', undefined, undefined, pnt, undefined, undefined, aI - pnt.sI);
|
|
580
|
+
pnt.sI = aI;
|
|
528
581
|
pnt.cI = cI;
|
|
529
582
|
pnt.rI = rI;
|
|
530
583
|
return tkn;
|
|
@@ -532,59 +585,57 @@ let makeCommentMatcher = (cfg, opts) => {
|
|
|
532
585
|
}
|
|
533
586
|
// Multiline comment.
|
|
534
587
|
for (let mc of blockComments) {
|
|
535
|
-
if (
|
|
536
|
-
let
|
|
537
|
-
let
|
|
588
|
+
if (src.startsWith(mc.start, pnt.sI)) {
|
|
589
|
+
let srclen = src.length;
|
|
590
|
+
let aI = pnt.sI + mc.start.length;
|
|
538
591
|
let end = mc.end;
|
|
539
592
|
cI += mc.start.length;
|
|
540
593
|
let suffixLen = 0;
|
|
541
594
|
let cc;
|
|
542
|
-
while (
|
|
543
|
-
let n = commentSuffixMatch(
|
|
595
|
+
while (aI < srclen && !src.startsWith(end, aI)) {
|
|
596
|
+
let n = commentSuffixMatch(src, aI, mc.suffixes);
|
|
544
597
|
if (n > 0) {
|
|
545
598
|
suffixLen = n;
|
|
546
599
|
break;
|
|
547
600
|
}
|
|
548
|
-
n = commentSuffixFnMatch(lex,
|
|
601
|
+
n = commentSuffixFnMatch(lex, aI, mc.suffixFn);
|
|
549
602
|
if (n > 0) {
|
|
550
603
|
suffixLen = n;
|
|
551
604
|
break;
|
|
552
605
|
}
|
|
553
|
-
cc =
|
|
554
|
-
if (cc < 256 ? rowBM[cc] : rowChars[
|
|
606
|
+
cc = src.charCodeAt(aI);
|
|
607
|
+
if (cc < 256 ? rowBM[cc] : rowChars[src[aI]]) {
|
|
555
608
|
rI++;
|
|
556
609
|
cI = 0;
|
|
557
610
|
}
|
|
558
611
|
cI++;
|
|
559
|
-
|
|
612
|
+
aI++;
|
|
560
613
|
}
|
|
561
614
|
if (suffixLen > 0) {
|
|
562
615
|
// Advance through the consumed suffix, tracking newlines.
|
|
563
616
|
for (let k = 0; k < suffixLen; k++) {
|
|
564
|
-
cc =
|
|
565
|
-
if (cc < 256 ? rowBM[cc] : rowChars[
|
|
617
|
+
cc = src.charCodeAt(aI + k);
|
|
618
|
+
if (cc < 256 ? rowBM[cc] : rowChars[src[aI + k]]) {
|
|
566
619
|
rI++;
|
|
567
620
|
cI = 0;
|
|
568
621
|
}
|
|
569
622
|
cI++;
|
|
570
623
|
}
|
|
571
|
-
let
|
|
572
|
-
|
|
573
|
-
pnt.sI += csrc.length;
|
|
624
|
+
let tkn = lex.token('#CM', undefined, undefined, pnt, undefined, undefined, aI + suffixLen - pnt.sI);
|
|
625
|
+
pnt.sI = aI + suffixLen;
|
|
574
626
|
pnt.rI = rI;
|
|
575
627
|
pnt.cI = cI;
|
|
576
628
|
return tkn;
|
|
577
629
|
}
|
|
578
|
-
if (
|
|
630
|
+
if (src.startsWith(end, aI)) {
|
|
579
631
|
cI += end.length;
|
|
580
632
|
if (mc.eatline) {
|
|
581
|
-
scan(
|
|
633
|
+
scan(src, aI, rI, cI, lineRunSpec, scanOut);
|
|
582
634
|
rI = scanOut.rI;
|
|
583
|
-
|
|
635
|
+
aI = scanOut.sI;
|
|
584
636
|
}
|
|
585
|
-
let
|
|
586
|
-
|
|
587
|
-
pnt.sI += csrc.length;
|
|
637
|
+
let tkn = lex.token('#CM', undefined, undefined, pnt, undefined, undefined, aI + end.length - pnt.sI);
|
|
638
|
+
pnt.sI = aI + end.length;
|
|
588
639
|
pnt.rI = rI;
|
|
589
640
|
pnt.cI = cI;
|
|
590
641
|
return tkn;
|
|
@@ -627,29 +678,30 @@ function normalizeCommentSuffix(raw) {
|
|
|
627
678
|
};
|
|
628
679
|
}
|
|
629
680
|
// commentSuffixMatch returns the length of the best suffix match at
|
|
630
|
-
//
|
|
681
|
+
// src[aI:] or 0 if none matches. Suffixes are pre-sorted longest-first,
|
|
631
682
|
// so the first match is the best match.
|
|
632
|
-
function commentSuffixMatch(
|
|
683
|
+
function commentSuffixMatch(src, aI, suffixes) {
|
|
633
684
|
if (!suffixes || 0 === suffixes.length)
|
|
634
685
|
return 0;
|
|
635
686
|
for (let s of suffixes) {
|
|
636
|
-
if (
|
|
687
|
+
if (src.startsWith(s, aI))
|
|
637
688
|
return s.length;
|
|
638
689
|
}
|
|
639
690
|
return 0;
|
|
640
691
|
}
|
|
641
692
|
// commentSuffixFnMatch probes the LexMatcher-form suffix terminator at
|
|
642
|
-
// offset
|
|
643
|
-
// consumed) or 0 if no termination. The lex point is
|
|
644
|
-
// restored so a misbehaving matcher can't advance the
|
|
645
|
-
|
|
693
|
+
// absolute source offset aI. Returns the length of the returned token's
|
|
694
|
+
// src (to be consumed) or 0 if no termination. The lex point is
|
|
695
|
+
// snapshotted and restored so a misbehaving matcher can't advance the
|
|
696
|
+
// stream itself.
|
|
697
|
+
function commentSuffixFnMatch(lex, aI, fn) {
|
|
646
698
|
if (!fn)
|
|
647
699
|
return 0;
|
|
648
700
|
let pnt = lex.pnt;
|
|
649
701
|
let savedSI = pnt.sI;
|
|
650
702
|
let savedRI = pnt.rI;
|
|
651
703
|
let savedCI = pnt.cI;
|
|
652
|
-
pnt.sI =
|
|
704
|
+
pnt.sI = aI;
|
|
653
705
|
let tkn;
|
|
654
706
|
try {
|
|
655
707
|
tkn = fn(lex, undefined);
|
|
@@ -666,9 +718,13 @@ function commentSuffixFnMatch(lex, fI, fn) {
|
|
|
666
718
|
// Match text, checking for literal values, optionally followed by a fixed token.
|
|
667
719
|
// Text strings are terminated by end markers.
|
|
668
720
|
let makeTextMatcher = (cfg, opts) => {
|
|
669
|
-
|
|
721
|
+
// Sticky ('y') so the regex runs against the full source at an offset
|
|
722
|
+
// (ender.lastIndex = pnt.sI) instead of an allocated remainder slice.
|
|
723
|
+
let ender = (0, utility_1.regexp)(cfg.line.lex ? 'y' : 'ys', '(.*?)', ...cfg.rePart.ender);
|
|
670
724
|
return function textMatcher(lex) {
|
|
671
725
|
if (cfg.text.check) {
|
|
726
|
+
// Check hooks are user code and may read lex.fwd directly.
|
|
727
|
+
lex.refwd();
|
|
672
728
|
let check = cfg.text.check(lex);
|
|
673
729
|
if (check && check.done) {
|
|
674
730
|
return check.token;
|
|
@@ -676,10 +732,10 @@ let makeTextMatcher = (cfg, opts) => {
|
|
|
676
732
|
}
|
|
677
733
|
let mcfg = cfg.text;
|
|
678
734
|
let pnt = lex.pnt;
|
|
679
|
-
let fwd = lex.fwd;
|
|
680
735
|
let def = cfg.value.def;
|
|
681
736
|
let defre = cfg.value.defre;
|
|
682
|
-
|
|
737
|
+
ender.lastIndex = pnt.sI;
|
|
738
|
+
let m = ender.exec(lex.src);
|
|
683
739
|
if (m) {
|
|
684
740
|
let msrc = m[1];
|
|
685
741
|
let tsrc = m[2];
|
|
@@ -702,8 +758,10 @@ let makeTextMatcher = (cfg, opts) => {
|
|
|
702
758
|
// utility.ts) — iteration order is deterministic.
|
|
703
759
|
for (let vspec of defre) {
|
|
704
760
|
if (vspec.match) {
|
|
705
|
-
// If consume, assume regexp starts with ^.
|
|
706
|
-
|
|
761
|
+
// If consume, assume regexp starts with ^. Consuming
|
|
762
|
+
// value regexes are user-supplied and match against the
|
|
763
|
+
// remainder string (materialized lazily, memoized).
|
|
764
|
+
let res = vspec.match.exec(vspec.consume ? lex.refwd() : msrc);
|
|
707
765
|
// Must match entire text.
|
|
708
766
|
if (res && (vspec.consume || res[0].length === msrc.length)) {
|
|
709
767
|
let remsrc = res[0];
|
|
@@ -748,8 +806,10 @@ let makeTextMatcher = (cfg, opts) => {
|
|
|
748
806
|
exports.makeTextMatcher = makeTextMatcher;
|
|
749
807
|
let makeNumberMatcher = (cfg, _opts) => {
|
|
750
808
|
let mcfg = cfg.number;
|
|
751
|
-
|
|
752
|
-
|
|
809
|
+
// Sticky ('y') so the regex runs against the full source at an offset
|
|
810
|
+
// (ender.lastIndex = pnt.sI) instead of an allocated remainder slice.
|
|
811
|
+
let ender = (0, utility_1.regexp)('y', [
|
|
812
|
+
'([-+]?(0(',
|
|
753
813
|
[
|
|
754
814
|
mcfg.hex ? 'x[0-9a-fA-F_]+' : null,
|
|
755
815
|
mcfg.oct ? 'o[0-7_]+' : null,
|
|
@@ -767,12 +827,13 @@ let makeNumberMatcher = (cfg, _opts) => {
|
|
|
767
827
|
let numberSep = mcfg.sep
|
|
768
828
|
? (0, utility_1.regexp)('g', (0, utility_1.escre)(mcfg.sepChar))
|
|
769
829
|
: undefined;
|
|
830
|
+
let numberSepChar = mcfg.sep ? mcfg.sepChar : undefined;
|
|
770
831
|
return guardedMatcher(cfg.number, function numberBody(lex) {
|
|
771
832
|
mcfg = cfg.number;
|
|
772
833
|
let pnt = lex.pnt;
|
|
773
|
-
let fwd = lex.fwd;
|
|
774
834
|
let valdef = cfg.value.def;
|
|
775
|
-
|
|
835
|
+
ender.lastIndex = pnt.sI;
|
|
836
|
+
let m = ender.exec(lex.src);
|
|
776
837
|
if (m) {
|
|
777
838
|
let msrc = m[1];
|
|
778
839
|
let tsrc = m[9]; // NOTE: count parens in numberEnder!
|
|
@@ -787,7 +848,11 @@ let makeNumberMatcher = (cfg, _opts) => {
|
|
|
787
848
|
out = lex.token('#VL', vs.val, msrc, pnt);
|
|
788
849
|
}
|
|
789
850
|
else {
|
|
790
|
-
|
|
851
|
+
// Strip separators only when one is actually present — the
|
|
852
|
+
// common separator-free number skips the regex replace.
|
|
853
|
+
let nstr = numberSep && numberSepChar && -1 < msrc.indexOf(numberSepChar)
|
|
854
|
+
? msrc.replace(numberSep, '')
|
|
855
|
+
: msrc;
|
|
791
856
|
let num = +nstr;
|
|
792
857
|
// Special case: +- prefix of 0x... format
|
|
793
858
|
if (isNaN(num)) {
|
|
@@ -881,7 +946,11 @@ let makeStringMatcher = (cfg, opts) => {
|
|
|
881
946
|
let sI = startSI + 1;
|
|
882
947
|
let rI = startRI;
|
|
883
948
|
let cI = pnt.cI + 1;
|
|
884
|
-
|
|
949
|
+
// Escape-free fast path: until the first escape or replace is
|
|
950
|
+
// processed the value is the contiguous source run after the opening
|
|
951
|
+
// quote, so no segment buffer is needed — a clean quoted string costs
|
|
952
|
+
// one substring instead of a buffer, per-segment pushes, and a join.
|
|
953
|
+
let buf = undefined;
|
|
885
954
|
while (sI < srclen) {
|
|
886
955
|
// Body scan: consume body chars (and multi-line newlines)
|
|
887
956
|
// until something interesting (quote, escape, control, replace).
|
|
@@ -890,15 +959,16 @@ let makeStringMatcher = (cfg, opts) => {
|
|
|
890
959
|
sI = scanOut.sI;
|
|
891
960
|
rI = scanOut.rI;
|
|
892
961
|
cI = scanOut.cI;
|
|
893
|
-
if (bodyStart < sI)
|
|
962
|
+
if (undefined !== buf && bodyStart < sI)
|
|
894
963
|
buf.push(src.substring(bodyStart, sI));
|
|
895
964
|
if (sI >= srclen)
|
|
896
965
|
break;
|
|
897
966
|
const cc = src.charCodeAt(sI);
|
|
898
967
|
// Closing quote — string done.
|
|
899
968
|
if (cc === qcc) {
|
|
969
|
+
const val = undefined === buf ? src.substring(startSI + 1, sI) : buf.join(types_1.EMPTY);
|
|
900
970
|
sI++;
|
|
901
|
-
const tkn = lex.token('#ST',
|
|
971
|
+
const tkn = lex.token('#ST', val, undefined, pnt, undefined, undefined, sI - startSI);
|
|
902
972
|
pnt.sI = sI;
|
|
903
973
|
pnt.rI = rI;
|
|
904
974
|
pnt.cI = cI + 1;
|
|
@@ -908,6 +978,9 @@ let makeStringMatcher = (cfg, opts) => {
|
|
|
908
978
|
if (hasReplace) {
|
|
909
979
|
const rs = replaceCodeMap[cc];
|
|
910
980
|
if (rs !== undefined) {
|
|
981
|
+
if (undefined === buf) {
|
|
982
|
+
buf = startSI + 1 < sI ? [src.substring(startSI + 1, sI)] : [];
|
|
983
|
+
}
|
|
911
984
|
buf.push(rs);
|
|
912
985
|
sI++;
|
|
913
986
|
cI++;
|
|
@@ -916,6 +989,9 @@ let makeStringMatcher = (cfg, opts) => {
|
|
|
916
989
|
}
|
|
917
990
|
// Escape sequence.
|
|
918
991
|
if (cc === escCharCode) {
|
|
992
|
+
if (undefined === buf) {
|
|
993
|
+
buf = startSI + 1 < sI ? [src.substring(startSI + 1, sI)] : [];
|
|
994
|
+
}
|
|
919
995
|
sI++;
|
|
920
996
|
cI++;
|
|
921
997
|
if (sI >= srclen)
|
|
@@ -1048,8 +1124,7 @@ let makeLineMatcher = (cfg, _opts) => {
|
|
|
1048
1124
|
sI++;
|
|
1049
1125
|
}
|
|
1050
1126
|
if (pnt.sI < sI) {
|
|
1051
|
-
const
|
|
1052
|
-
const tkn = lex.token('#LN', undefined, msrc, pnt);
|
|
1127
|
+
const tkn = lex.token('#LN', undefined, undefined, pnt, undefined, undefined, sI - pnt.sI);
|
|
1053
1128
|
pnt.sI = sI;
|
|
1054
1129
|
pnt.rI = rI;
|
|
1055
1130
|
pnt.cI = 1;
|
|
@@ -1058,8 +1133,7 @@ let makeLineMatcher = (cfg, _opts) => {
|
|
|
1058
1133
|
return;
|
|
1059
1134
|
}
|
|
1060
1135
|
if (scan(src, pnt.sI, pnt.rI, pnt.cI, spec, out)) {
|
|
1061
|
-
const
|
|
1062
|
-
const tkn = lex.token('#LN', undefined, msrc, pnt);
|
|
1136
|
+
const tkn = lex.token('#LN', undefined, undefined, pnt, undefined, undefined, out.sI - pnt.sI);
|
|
1063
1137
|
pnt.sI = out.sI;
|
|
1064
1138
|
pnt.rI = out.rI;
|
|
1065
1139
|
pnt.cI = 1;
|
|
@@ -1078,8 +1152,7 @@ let makeSpaceMatcher = (cfg, _opts) => {
|
|
|
1078
1152
|
return guardedMatcher(cfg.space, function spaceBody(lex) {
|
|
1079
1153
|
const { pnt, src } = lex;
|
|
1080
1154
|
if (scan(src, pnt.sI, pnt.rI, pnt.cI, spec, out)) {
|
|
1081
|
-
const
|
|
1082
|
-
const tkn = lex.token('#SP', undefined, msrc, pnt);
|
|
1155
|
+
const tkn = lex.token('#SP', undefined, undefined, pnt, undefined, undefined, out.sI - pnt.sI);
|
|
1083
1156
|
pnt.sI = out.sI;
|
|
1084
1157
|
pnt.rI = out.rI;
|
|
1085
1158
|
pnt.cI = out.cI;
|
|
@@ -1113,10 +1186,23 @@ function subMatchFixed(lex, first, tsrc) {
|
|
|
1113
1186
|
}
|
|
1114
1187
|
return out;
|
|
1115
1188
|
}
|
|
1189
|
+
// Built-in matcher names (the `matcher` annotation set in configure()).
|
|
1190
|
+
// These matchers scan lex.src at pnt offsets or call refwd() themselves;
|
|
1191
|
+
// anything else in the matcher list gets a fresh lex.fwd before running.
|
|
1192
|
+
const BUILTIN_MATCHER = {
|
|
1193
|
+
match: 1, fixed: 1, space: 1, line: 1,
|
|
1194
|
+
string: 1, comment: 1, number: 1, text: 1,
|
|
1195
|
+
};
|
|
1116
1196
|
// Lexer driver: holds the scan Point and runs the configured matchers in order via next().
|
|
1117
1197
|
class Lex {
|
|
1198
|
+
// Slice the remainder lazily and memoize on the position — the slice
|
|
1199
|
+
// is only materialized for consumers that genuinely need a remainder
|
|
1200
|
+
// string (custom matchers, value.defre regexes), never once per token.
|
|
1118
1201
|
refwd() {
|
|
1119
|
-
this.
|
|
1202
|
+
if (this.fwdSI !== this.pnt.sI) {
|
|
1203
|
+
this.fwd = this.src.substring(this.pnt.sI);
|
|
1204
|
+
this.fwdSI = this.pnt.sI;
|
|
1205
|
+
}
|
|
1120
1206
|
return this.fwd;
|
|
1121
1207
|
}
|
|
1122
1208
|
constructor(ctx) {
|
|
@@ -1125,12 +1211,16 @@ class Lex {
|
|
|
1125
1211
|
this.cfg = {}; // Resolved configuration.
|
|
1126
1212
|
this.pnt = makePoint(-1); // Current scan position.
|
|
1127
1213
|
this.fwd = types_1.EMPTY; // Source from pnt.sI onward (the unconsumed remainder).
|
|
1214
|
+
this.fwdSI = -1; // pnt.sI the current fwd slice was taken at (memo key).
|
|
1128
1215
|
this.ctx = ctx;
|
|
1129
1216
|
this.src = ctx.src();
|
|
1130
1217
|
this.cfg = ctx.cfg;
|
|
1131
1218
|
this.pnt = makePoint(this.src.length);
|
|
1132
1219
|
}
|
|
1133
|
-
token
|
|
1220
|
+
// Create a token. Pass `src` when the matcher already holds the
|
|
1221
|
+
// matched text; pass src=undefined with an explicit `len` to defer
|
|
1222
|
+
// the substring to first read (a span over this lexer's source).
|
|
1223
|
+
token(ref, val, src, pnt, use, why, len) {
|
|
1134
1224
|
let tin;
|
|
1135
1225
|
let name;
|
|
1136
1226
|
if ('string' === typeof ref) {
|
|
@@ -1141,7 +1231,7 @@ class Lex {
|
|
|
1141
1231
|
tin = ref;
|
|
1142
1232
|
name = (0, utility_1.tokenize)(ref, this.cfg);
|
|
1143
1233
|
}
|
|
1144
|
-
let tkn = makeToken(name, tin, val, src, pnt || this.pnt, use, why);
|
|
1234
|
+
let tkn = makeToken(name, tin, val, src, pnt || this.pnt, use, why, this.src, len);
|
|
1145
1235
|
return tkn;
|
|
1146
1236
|
}
|
|
1147
1237
|
next(rule, alt, altI, tI) {
|
|
@@ -1160,9 +1250,19 @@ class Lex {
|
|
|
1160
1250
|
tkn = pnt.end;
|
|
1161
1251
|
}
|
|
1162
1252
|
else {
|
|
1163
|
-
|
|
1253
|
+
// First-char dispatch: run only the matchers that could produce a
|
|
1254
|
+
// token starting with this char (non-Latin-1 chars and configs
|
|
1255
|
+
// without a table run the full pipeline).
|
|
1256
|
+
const dispatch = this.cfg.lex.dispatch;
|
|
1257
|
+
const cc = this.src.charCodeAt(pnt.sI);
|
|
1258
|
+
const mats = undefined === dispatch
|
|
1259
|
+
? this.cfg.lex.match
|
|
1260
|
+
: dispatch[cc < 256 ? cc : 256];
|
|
1164
1261
|
try {
|
|
1165
|
-
for (let mat of
|
|
1262
|
+
for (let mat of mats) {
|
|
1263
|
+
if (undefined === BUILTIN_MATCHER[mat.matcher]) {
|
|
1264
|
+
this.refwd();
|
|
1265
|
+
}
|
|
1166
1266
|
if ((tkn = mat(this, rule, tI))) {
|
|
1167
1267
|
match = mat;
|
|
1168
1268
|
break;
|