@tabnas/parser 0.2.0 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/lexer.js CHANGED
@@ -42,27 +42,50 @@ exports.Point = Point;
42
42
  const makePoint = (...params) => new Point(...params);
43
43
  exports.makePoint = makePoint;
44
44
  // A single lexed token: numeric token id, JS-typed value, raw source text, and match position.
45
+ //
46
+ // NOTE: `src` is a prototype accessor over a lazily-materialized backing
47
+ // field — matchers that only know the token's span (space, line,
48
+ // comment, quoted-string raw text) defer the substring until someone
49
+ // actually reads it, so ignored tokens never allocate one. Reading
50
+ // tkn.src always yields the correct string; the one observable
51
+ // difference from a plain data property is that `src` is not an OWN
52
+ // enumerable property, so Object.keys(tkn), {...tkn}, and
53
+ // JSON.stringify(tkn) do not include it.
45
54
  class Token {
46
- constructor(name, tin, val, src, pnt, use, why) {
55
+ #src; // Materialized source text (undefined = not yet).
56
+ #ref; // Full source backing the [sI, sI+len) span.
57
+ constructor(name, tin, val, src, pnt, use, why, ref, len) {
47
58
  this.isToken = true; // Marker discriminating Tokens from other values.
48
59
  this.name = types_1.EMPTY; // Token name (e.g. '#NR', '#ST').
49
60
  this.tin = -1; // Numeric token id corresponding to name.
50
61
  this.val = undefined; // JS-typed value (e.g. a number for #NR).
51
- this.src = types_1.EMPTY; // Raw matching source text.
52
62
  this.sI = -1; // Source index where the match started.
53
63
  this.rI = -1; // Row where the match started.
54
64
  this.cI = -1; // Column where the match started.
55
65
  this.len = -1; // Length of src.
56
66
  this.name = name;
57
67
  this.tin = tin;
58
- this.src = src;
68
+ this.#src = src;
69
+ this.#ref = ref;
59
70
  this.val = val;
60
71
  this.sI = pnt.sI;
61
72
  this.rI = pnt.rI;
62
73
  this.cI = pnt.cI;
63
74
  this.use = use;
64
75
  this.why = why;
65
- this.len = null == src ? 0 : src.length;
76
+ this.len = null != len ? len : null == src ? 0 : src.length;
77
+ }
78
+ get src() {
79
+ let s = this.#src;
80
+ if (undefined === s) {
81
+ const ref = this.#ref;
82
+ s = this.#src =
83
+ undefined === ref ? types_1.EMPTY : ref.substring(this.sI, this.sI + this.len);
84
+ }
85
+ return s;
86
+ }
87
+ set src(s) {
88
+ this.#src = s;
66
89
  }
67
90
  resolveVal(rule, ctx) {
68
91
  let out = 'function' === typeof this.val ? this.val(rule, ctx) : this.val;
@@ -116,6 +139,8 @@ function guardedMatcher(mcfg, body) {
116
139
  if (!mcfg.lex)
117
140
  return undefined;
118
141
  if (mcfg.check) {
142
+ // Check hooks are user code and may read lex.fwd directly.
143
+ lex.refwd();
119
144
  const r = mcfg.check(lex);
120
145
  if (r && r.done)
121
146
  return r.token;
@@ -329,23 +354,48 @@ const STRING_BODY_TABLE = new Int32Array([
329
354
  CONSUME | IS_ROW,
330
355
  ]);
331
356
  let makeFixedMatcher = (cfg, _opts) => {
332
- let fixed = (0, utility_1.regexp)(null, '^(', cfg.rePart.fixed, ')');
357
+ // First-char dispatch table replacing the old anchored alternation
358
+ // regex: match cost per position becomes one charCode index plus (for
359
+ // multi-char tokens) a startsWith verify, with no match-array or
360
+ // capture substring allocation. Candidates sharing a first char are
361
+ // ordered longest-first, preserving the regex's longest-match-wins
362
+ // ordering (utility.ts sorts the alternation the same way).
363
+ const table = new Array(256);
364
+ const wide = []; // Tokens whose first char is >= U+0100.
365
+ const byLenDesc = (a, b) => b.len - a.len || (a.src < b.src ? -1 : a.src > b.src ? 1 : 0);
366
+ for (const fsrc of (0, utility_1.keys)(cfg.fixed.token)) {
367
+ const tin = cfg.fixed.token[fsrc];
368
+ if (null == tin || 0 === fsrc.length)
369
+ continue;
370
+ const cand = { src: fsrc, len: fsrc.length, tin };
371
+ const cc = fsrc.charCodeAt(0);
372
+ if (cc < 256) {
373
+ ;
374
+ (table[cc] = table[cc] || []).push(cand);
375
+ }
376
+ else {
377
+ wide.push(cand);
378
+ }
379
+ }
380
+ for (const cands of table) {
381
+ if (cands)
382
+ cands.sort(byLenDesc);
383
+ }
384
+ wide.sort(byLenDesc);
333
385
  return guardedMatcher(cfg.fixed, function fixedBody(lex) {
334
- const mcfg = cfg.fixed;
335
- let pnt = lex.pnt;
336
- let fwd = lex.fwd;
337
- let m = fwd.match(fixed);
338
- if (m) {
339
- let msrc = m[1];
340
- let mlen = msrc.length;
341
- if (0 < mlen) {
342
- let tkn = undefined;
343
- let tin = mcfg.token[msrc];
344
- if (null != tin) {
345
- tkn = lex.token(tin, undefined, msrc, pnt);
346
- pnt.sI += mlen;
347
- pnt.cI += mlen;
348
- }
386
+ const pnt = lex.pnt;
387
+ const src = lex.src;
388
+ const cc = src.charCodeAt(pnt.sI);
389
+ const cands = cc < 256 ? table[cc] : wide;
390
+ if (undefined === cands)
391
+ return undefined;
392
+ for (const cand of cands) {
393
+ // Single-char Latin-1 candidates already matched via the table
394
+ // index; longer (or wide-char) candidates verify in place.
395
+ if ((1 === cand.len && cc < 256) || src.startsWith(cand.src, pnt.sI)) {
396
+ const tkn = lex.token(cand.tin, undefined, cand.src, pnt);
397
+ pnt.sI += cand.len;
398
+ pnt.cI += cand.len;
349
399
  return tkn;
350
400
  }
351
401
  }
@@ -367,7 +417,9 @@ let makeMatchMatcher = (cfg, _opts) => {
367
417
  }
368
418
  return guardedMatcher(cfg.match, function matchBody(lex, rule, tI = 0) {
369
419
  let pnt = lex.pnt;
370
- let fwd = lex.fwd;
420
+ // Value/token matcher regexes are documented to run against the
421
+ // remainder string, so materialize it (memoized per position).
422
+ let fwd = lex.refwd();
371
423
  let oc = 'o' === rule.state ? 0 : 1;
372
424
  for (let valueMatcher of valueMatchers) {
373
425
  if (valueMatcher.match instanceof RegExp) {
@@ -474,9 +526,11 @@ let makeCommentMatcher = (cfg, opts) => {
474
526
  // so the table and class array are reused across comment matches.
475
527
  const lineRunSpec = buildLineRunSpec(cfg.line);
476
528
  const scanOut = { sI: 0, rI: 0, cI: 0 };
529
+ // The body walks lex.src with absolute indices (aI) — no remainder
530
+ // slice is materialized per position.
477
531
  return guardedMatcher(cfg.comment, function commentBody(lex) {
478
532
  let pnt = lex.pnt;
479
- let fwd = lex.fwd;
533
+ let src = lex.src;
480
534
  let rI = pnt.rI;
481
535
  let cI = pnt.cI;
482
536
  // Single line comment.
@@ -485,32 +539,32 @@ let makeCommentMatcher = (cfg, opts) => {
485
539
  const lineChars = cfg.line.chars;
486
540
  const rowChars = cfg.line.rowChars;
487
541
  for (let mc of lineComments) {
488
- if (fwd.startsWith(mc.start)) {
489
- let fwdlen = fwd.length;
490
- let fI = mc.start.length;
542
+ if (src.startsWith(mc.start, pnt.sI)) {
543
+ let srclen = src.length;
544
+ let aI = pnt.sI + mc.start.length;
491
545
  cI += mc.start.length;
492
546
  let suffixLen = 0;
493
547
  let cc;
494
- while (fI < fwdlen &&
495
- !((cc = fwd.charCodeAt(fI)) < 256
548
+ while (aI < srclen &&
549
+ !((cc = src.charCodeAt(aI)) < 256
496
550
  ? lineBM[cc]
497
- : lineChars[fwd[fI]])) {
498
- let n = commentSuffixMatch(fwd, fI, mc.suffixes);
551
+ : lineChars[src[aI]])) {
552
+ let n = commentSuffixMatch(src, aI, mc.suffixes);
499
553
  if (n > 0) {
500
554
  suffixLen = n;
501
555
  break;
502
556
  }
503
- n = commentSuffixFnMatch(lex, fI, mc.suffixFn);
557
+ n = commentSuffixFnMatch(lex, aI, mc.suffixFn);
504
558
  if (n > 0) {
505
559
  suffixLen = n;
506
560
  break;
507
561
  }
508
562
  cI++;
509
- fI++;
563
+ aI++;
510
564
  }
511
565
  if (suffixLen > 0) {
512
566
  // Consume the suffix as the tail of the comment body.
513
- fI += suffixLen;
567
+ aI += suffixLen;
514
568
  cI += suffixLen;
515
569
  }
516
570
  else if (mc.eatline) {
@@ -518,13 +572,12 @@ let makeCommentMatcher = (cfg, opts) => {
518
572
  // a line char (not from a suffix match). cI is intentionally
519
573
  // NOT pulled from scanOut — current semantics leave the
520
574
  // pnt.cI at end-of-comment-body even after eating newlines.
521
- scan(fwd, fI, rI, cI, lineRunSpec, scanOut);
575
+ scan(src, aI, rI, cI, lineRunSpec, scanOut);
522
576
  rI = scanOut.rI;
523
- fI = scanOut.sI;
577
+ aI = scanOut.sI;
524
578
  }
525
- let csrc = fwd.substring(0, fI);
526
- let tkn = lex.token('#CM', undefined, csrc, pnt);
527
- pnt.sI += csrc.length;
579
+ let tkn = lex.token('#CM', undefined, undefined, pnt, undefined, undefined, aI - pnt.sI);
580
+ pnt.sI = aI;
528
581
  pnt.cI = cI;
529
582
  pnt.rI = rI;
530
583
  return tkn;
@@ -532,59 +585,57 @@ let makeCommentMatcher = (cfg, opts) => {
532
585
  }
533
586
  // Multiline comment.
534
587
  for (let mc of blockComments) {
535
- if (fwd.startsWith(mc.start)) {
536
- let fwdlen = fwd.length;
537
- let fI = mc.start.length;
588
+ if (src.startsWith(mc.start, pnt.sI)) {
589
+ let srclen = src.length;
590
+ let aI = pnt.sI + mc.start.length;
538
591
  let end = mc.end;
539
592
  cI += mc.start.length;
540
593
  let suffixLen = 0;
541
594
  let cc;
542
- while (fI < fwdlen && !fwd.startsWith(end, fI)) {
543
- let n = commentSuffixMatch(fwd, fI, mc.suffixes);
595
+ while (aI < srclen && !src.startsWith(end, aI)) {
596
+ let n = commentSuffixMatch(src, aI, mc.suffixes);
544
597
  if (n > 0) {
545
598
  suffixLen = n;
546
599
  break;
547
600
  }
548
- n = commentSuffixFnMatch(lex, fI, mc.suffixFn);
601
+ n = commentSuffixFnMatch(lex, aI, mc.suffixFn);
549
602
  if (n > 0) {
550
603
  suffixLen = n;
551
604
  break;
552
605
  }
553
- cc = fwd.charCodeAt(fI);
554
- if (cc < 256 ? rowBM[cc] : rowChars[fwd[fI]]) {
606
+ cc = src.charCodeAt(aI);
607
+ if (cc < 256 ? rowBM[cc] : rowChars[src[aI]]) {
555
608
  rI++;
556
609
  cI = 0;
557
610
  }
558
611
  cI++;
559
- fI++;
612
+ aI++;
560
613
  }
561
614
  if (suffixLen > 0) {
562
615
  // Advance through the consumed suffix, tracking newlines.
563
616
  for (let k = 0; k < suffixLen; k++) {
564
- cc = fwd.charCodeAt(fI + k);
565
- if (cc < 256 ? rowBM[cc] : rowChars[fwd[fI + k]]) {
617
+ cc = src.charCodeAt(aI + k);
618
+ if (cc < 256 ? rowBM[cc] : rowChars[src[aI + k]]) {
566
619
  rI++;
567
620
  cI = 0;
568
621
  }
569
622
  cI++;
570
623
  }
571
- let csrc = fwd.substring(0, fI + suffixLen);
572
- let tkn = lex.token('#CM', undefined, csrc, pnt);
573
- pnt.sI += csrc.length;
624
+ let tkn = lex.token('#CM', undefined, undefined, pnt, undefined, undefined, aI + suffixLen - pnt.sI);
625
+ pnt.sI = aI + suffixLen;
574
626
  pnt.rI = rI;
575
627
  pnt.cI = cI;
576
628
  return tkn;
577
629
  }
578
- if (fwd.startsWith(end, fI)) {
630
+ if (src.startsWith(end, aI)) {
579
631
  cI += end.length;
580
632
  if (mc.eatline) {
581
- scan(fwd, fI, rI, cI, lineRunSpec, scanOut);
633
+ scan(src, aI, rI, cI, lineRunSpec, scanOut);
582
634
  rI = scanOut.rI;
583
- fI = scanOut.sI;
635
+ aI = scanOut.sI;
584
636
  }
585
- let csrc = fwd.substring(0, fI + end.length);
586
- let tkn = lex.token('#CM', undefined, csrc, pnt);
587
- pnt.sI += csrc.length;
637
+ let tkn = lex.token('#CM', undefined, undefined, pnt, undefined, undefined, aI + end.length - pnt.sI);
638
+ pnt.sI = aI + end.length;
588
639
  pnt.rI = rI;
589
640
  pnt.cI = cI;
590
641
  return tkn;
@@ -627,29 +678,30 @@ function normalizeCommentSuffix(raw) {
627
678
  };
628
679
  }
629
680
  // commentSuffixMatch returns the length of the best suffix match at
630
- // fwd[fI:] or 0 if none matches. Suffixes are pre-sorted longest-first,
681
+ // src[aI:] or 0 if none matches. Suffixes are pre-sorted longest-first,
631
682
  // so the first match is the best match.
632
- function commentSuffixMatch(fwd, fI, suffixes) {
683
+ function commentSuffixMatch(src, aI, suffixes) {
633
684
  if (!suffixes || 0 === suffixes.length)
634
685
  return 0;
635
686
  for (let s of suffixes) {
636
- if (fwd.substring(fI, fI + s.length) === s)
687
+ if (src.startsWith(s, aI))
637
688
  return s.length;
638
689
  }
639
690
  return 0;
640
691
  }
641
692
  // commentSuffixFnMatch probes the LexMatcher-form suffix terminator at
642
- // offset fI. Returns the length of the returned token's src (to be
643
- // consumed) or 0 if no termination. The lex point is snapshotted and
644
- // restored so a misbehaving matcher can't advance the stream itself.
645
- function commentSuffixFnMatch(lex, fI, fn) {
693
+ // absolute source offset aI. Returns the length of the returned token's
694
+ // src (to be consumed) or 0 if no termination. The lex point is
695
+ // snapshotted and restored so a misbehaving matcher can't advance the
696
+ // stream itself.
697
+ function commentSuffixFnMatch(lex, aI, fn) {
646
698
  if (!fn)
647
699
  return 0;
648
700
  let pnt = lex.pnt;
649
701
  let savedSI = pnt.sI;
650
702
  let savedRI = pnt.rI;
651
703
  let savedCI = pnt.cI;
652
- pnt.sI = savedSI + fI;
704
+ pnt.sI = aI;
653
705
  let tkn;
654
706
  try {
655
707
  tkn = fn(lex, undefined);
@@ -666,9 +718,13 @@ function commentSuffixFnMatch(lex, fI, fn) {
666
718
  // Match text, checking for literal values, optionally followed by a fixed token.
667
719
  // Text strings are terminated by end markers.
668
720
  let makeTextMatcher = (cfg, opts) => {
669
- let ender = (0, utility_1.regexp)(cfg.line.lex ? null : 's', '^(.*?)', ...cfg.rePart.ender);
721
+ // Sticky ('y') so the regex runs against the full source at an offset
722
+ // (ender.lastIndex = pnt.sI) instead of an allocated remainder slice.
723
+ let ender = (0, utility_1.regexp)(cfg.line.lex ? 'y' : 'ys', '(.*?)', ...cfg.rePart.ender);
670
724
  return function textMatcher(lex) {
671
725
  if (cfg.text.check) {
726
+ // Check hooks are user code and may read lex.fwd directly.
727
+ lex.refwd();
672
728
  let check = cfg.text.check(lex);
673
729
  if (check && check.done) {
674
730
  return check.token;
@@ -676,10 +732,10 @@ let makeTextMatcher = (cfg, opts) => {
676
732
  }
677
733
  let mcfg = cfg.text;
678
734
  let pnt = lex.pnt;
679
- let fwd = lex.fwd;
680
735
  let def = cfg.value.def;
681
736
  let defre = cfg.value.defre;
682
- let m = fwd.match(ender);
737
+ ender.lastIndex = pnt.sI;
738
+ let m = ender.exec(lex.src);
683
739
  if (m) {
684
740
  let msrc = m[1];
685
741
  let tsrc = m[2];
@@ -702,8 +758,10 @@ let makeTextMatcher = (cfg, opts) => {
702
758
  // utility.ts) — iteration order is deterministic.
703
759
  for (let vspec of defre) {
704
760
  if (vspec.match) {
705
- // If consume, assume regexp starts with ^.
706
- let res = vspec.match.exec(vspec.consume ? fwd : msrc);
761
+ // If consume, assume regexp starts with ^. Consuming
762
+ // value regexes are user-supplied and match against the
763
+ // remainder string (materialized lazily, memoized).
764
+ let res = vspec.match.exec(vspec.consume ? lex.refwd() : msrc);
707
765
  // Must match entire text.
708
766
  if (res && (vspec.consume || res[0].length === msrc.length)) {
709
767
  let remsrc = res[0];
@@ -748,8 +806,10 @@ let makeTextMatcher = (cfg, opts) => {
748
806
  exports.makeTextMatcher = makeTextMatcher;
749
807
  let makeNumberMatcher = (cfg, _opts) => {
750
808
  let mcfg = cfg.number;
751
- let ender = (0, utility_1.regexp)(null, [
752
- '^([-+]?(0(',
809
+ // Sticky ('y') so the regex runs against the full source at an offset
810
+ // (ender.lastIndex = pnt.sI) instead of an allocated remainder slice.
811
+ let ender = (0, utility_1.regexp)('y', [
812
+ '([-+]?(0(',
753
813
  [
754
814
  mcfg.hex ? 'x[0-9a-fA-F_]+' : null,
755
815
  mcfg.oct ? 'o[0-7_]+' : null,
@@ -767,12 +827,13 @@ let makeNumberMatcher = (cfg, _opts) => {
767
827
  let numberSep = mcfg.sep
768
828
  ? (0, utility_1.regexp)('g', (0, utility_1.escre)(mcfg.sepChar))
769
829
  : undefined;
830
+ let numberSepChar = mcfg.sep ? mcfg.sepChar : undefined;
770
831
  return guardedMatcher(cfg.number, function numberBody(lex) {
771
832
  mcfg = cfg.number;
772
833
  let pnt = lex.pnt;
773
- let fwd = lex.fwd;
774
834
  let valdef = cfg.value.def;
775
- let m = fwd.match(ender);
835
+ ender.lastIndex = pnt.sI;
836
+ let m = ender.exec(lex.src);
776
837
  if (m) {
777
838
  let msrc = m[1];
778
839
  let tsrc = m[9]; // NOTE: count parens in numberEnder!
@@ -787,7 +848,11 @@ let makeNumberMatcher = (cfg, _opts) => {
787
848
  out = lex.token('#VL', vs.val, msrc, pnt);
788
849
  }
789
850
  else {
790
- let nstr = numberSep ? msrc.replace(numberSep, '') : msrc;
851
+ // Strip separators only when one is actually present — the
852
+ // common separator-free number skips the regex replace.
853
+ let nstr = numberSep && numberSepChar && -1 < msrc.indexOf(numberSepChar)
854
+ ? msrc.replace(numberSep, '')
855
+ : msrc;
791
856
  let num = +nstr;
792
857
  // Special case: +- prefix of 0x... format
793
858
  if (isNaN(num)) {
@@ -881,7 +946,11 @@ let makeStringMatcher = (cfg, opts) => {
881
946
  let sI = startSI + 1;
882
947
  let rI = startRI;
883
948
  let cI = pnt.cI + 1;
884
- const buf = [];
949
+ // Escape-free fast path: until the first escape or replace is
950
+ // processed the value is the contiguous source run after the opening
951
+ // quote, so no segment buffer is needed — a clean quoted string costs
952
+ // one substring instead of a buffer, per-segment pushes, and a join.
953
+ let buf = undefined;
885
954
  while (sI < srclen) {
886
955
  // Body scan: consume body chars (and multi-line newlines)
887
956
  // until something interesting (quote, escape, control, replace).
@@ -890,15 +959,16 @@ let makeStringMatcher = (cfg, opts) => {
890
959
  sI = scanOut.sI;
891
960
  rI = scanOut.rI;
892
961
  cI = scanOut.cI;
893
- if (bodyStart < sI)
962
+ if (undefined !== buf && bodyStart < sI)
894
963
  buf.push(src.substring(bodyStart, sI));
895
964
  if (sI >= srclen)
896
965
  break;
897
966
  const cc = src.charCodeAt(sI);
898
967
  // Closing quote — string done.
899
968
  if (cc === qcc) {
969
+ const val = undefined === buf ? src.substring(startSI + 1, sI) : buf.join(types_1.EMPTY);
900
970
  sI++;
901
- const tkn = lex.token('#ST', buf.join(types_1.EMPTY), src.substring(startSI, sI), pnt);
971
+ const tkn = lex.token('#ST', val, undefined, pnt, undefined, undefined, sI - startSI);
902
972
  pnt.sI = sI;
903
973
  pnt.rI = rI;
904
974
  pnt.cI = cI + 1;
@@ -908,6 +978,9 @@ let makeStringMatcher = (cfg, opts) => {
908
978
  if (hasReplace) {
909
979
  const rs = replaceCodeMap[cc];
910
980
  if (rs !== undefined) {
981
+ if (undefined === buf) {
982
+ buf = startSI + 1 < sI ? [src.substring(startSI + 1, sI)] : [];
983
+ }
911
984
  buf.push(rs);
912
985
  sI++;
913
986
  cI++;
@@ -916,6 +989,9 @@ let makeStringMatcher = (cfg, opts) => {
916
989
  }
917
990
  // Escape sequence.
918
991
  if (cc === escCharCode) {
992
+ if (undefined === buf) {
993
+ buf = startSI + 1 < sI ? [src.substring(startSI + 1, sI)] : [];
994
+ }
919
995
  sI++;
920
996
  cI++;
921
997
  if (sI >= srclen)
@@ -1048,8 +1124,7 @@ let makeLineMatcher = (cfg, _opts) => {
1048
1124
  sI++;
1049
1125
  }
1050
1126
  if (pnt.sI < sI) {
1051
- const msrc = src.substring(pnt.sI, sI);
1052
- const tkn = lex.token('#LN', undefined, msrc, pnt);
1127
+ const tkn = lex.token('#LN', undefined, undefined, pnt, undefined, undefined, sI - pnt.sI);
1053
1128
  pnt.sI = sI;
1054
1129
  pnt.rI = rI;
1055
1130
  pnt.cI = 1;
@@ -1058,8 +1133,7 @@ let makeLineMatcher = (cfg, _opts) => {
1058
1133
  return;
1059
1134
  }
1060
1135
  if (scan(src, pnt.sI, pnt.rI, pnt.cI, spec, out)) {
1061
- const msrc = src.substring(pnt.sI, out.sI);
1062
- const tkn = lex.token('#LN', undefined, msrc, pnt);
1136
+ const tkn = lex.token('#LN', undefined, undefined, pnt, undefined, undefined, out.sI - pnt.sI);
1063
1137
  pnt.sI = out.sI;
1064
1138
  pnt.rI = out.rI;
1065
1139
  pnt.cI = 1;
@@ -1078,8 +1152,7 @@ let makeSpaceMatcher = (cfg, _opts) => {
1078
1152
  return guardedMatcher(cfg.space, function spaceBody(lex) {
1079
1153
  const { pnt, src } = lex;
1080
1154
  if (scan(src, pnt.sI, pnt.rI, pnt.cI, spec, out)) {
1081
- const msrc = src.substring(pnt.sI, out.sI);
1082
- const tkn = lex.token('#SP', undefined, msrc, pnt);
1155
+ const tkn = lex.token('#SP', undefined, undefined, pnt, undefined, undefined, out.sI - pnt.sI);
1083
1156
  pnt.sI = out.sI;
1084
1157
  pnt.rI = out.rI;
1085
1158
  pnt.cI = out.cI;
@@ -1113,10 +1186,23 @@ function subMatchFixed(lex, first, tsrc) {
1113
1186
  }
1114
1187
  return out;
1115
1188
  }
1189
+ // Built-in matcher names (the `matcher` annotation set in configure()).
1190
+ // These matchers scan lex.src at pnt offsets or call refwd() themselves;
1191
+ // anything else in the matcher list gets a fresh lex.fwd before running.
1192
+ const BUILTIN_MATCHER = {
1193
+ match: 1, fixed: 1, space: 1, line: 1,
1194
+ string: 1, comment: 1, number: 1, text: 1,
1195
+ };
1116
1196
  // Lexer driver: holds the scan Point and runs the configured matchers in order via next().
1117
1197
  class Lex {
1198
+ // Slice the remainder lazily and memoize on the position — the slice
1199
+ // is only materialized for consumers that genuinely need a remainder
1200
+ // string (custom matchers, value.defre regexes), never once per token.
1118
1201
  refwd() {
1119
- this.fwd = this.src.substring(this.pnt.sI);
1202
+ if (this.fwdSI !== this.pnt.sI) {
1203
+ this.fwd = this.src.substring(this.pnt.sI);
1204
+ this.fwdSI = this.pnt.sI;
1205
+ }
1120
1206
  return this.fwd;
1121
1207
  }
1122
1208
  constructor(ctx) {
@@ -1125,12 +1211,16 @@ class Lex {
1125
1211
  this.cfg = {}; // Resolved configuration.
1126
1212
  this.pnt = makePoint(-1); // Current scan position.
1127
1213
  this.fwd = types_1.EMPTY; // Source from pnt.sI onward (the unconsumed remainder).
1214
+ this.fwdSI = -1; // pnt.sI the current fwd slice was taken at (memo key).
1128
1215
  this.ctx = ctx;
1129
1216
  this.src = ctx.src();
1130
1217
  this.cfg = ctx.cfg;
1131
1218
  this.pnt = makePoint(this.src.length);
1132
1219
  }
1133
- token(ref, val, src, pnt, use, why) {
1220
+ // Create a token. Pass `src` when the matcher already holds the
1221
+ // matched text; pass src=undefined with an explicit `len` to defer
1222
+ // the substring to first read (a span over this lexer's source).
1223
+ token(ref, val, src, pnt, use, why, len) {
1134
1224
  let tin;
1135
1225
  let name;
1136
1226
  if ('string' === typeof ref) {
@@ -1141,7 +1231,7 @@ class Lex {
1141
1231
  tin = ref;
1142
1232
  name = (0, utility_1.tokenize)(ref, this.cfg);
1143
1233
  }
1144
- let tkn = makeToken(name, tin, val, src, pnt || this.pnt, use, why);
1234
+ let tkn = makeToken(name, tin, val, src, pnt || this.pnt, use, why, this.src, len);
1145
1235
  return tkn;
1146
1236
  }
1147
1237
  next(rule, alt, altI, tI) {
@@ -1160,9 +1250,19 @@ class Lex {
1160
1250
  tkn = pnt.end;
1161
1251
  }
1162
1252
  else {
1163
- this.fwd = this.src.substring(pnt.sI);
1253
+ // First-char dispatch: run only the matchers that could produce a
1254
+ // token starting with this char (non-Latin-1 chars and configs
1255
+ // without a table run the full pipeline).
1256
+ const dispatch = this.cfg.lex.dispatch;
1257
+ const cc = this.src.charCodeAt(pnt.sI);
1258
+ const mats = undefined === dispatch
1259
+ ? this.cfg.lex.match
1260
+ : dispatch[cc < 256 ? cc : 256];
1164
1261
  try {
1165
- for (let mat of this.cfg.lex.match) {
1262
+ for (let mat of mats) {
1263
+ if (undefined === BUILTIN_MATCHER[mat.matcher]) {
1264
+ this.refwd();
1265
+ }
1166
1266
  if ((tkn = mat(this, rule, tI))) {
1167
1267
  match = mat;
1168
1268
  break;