@tabnas/parser 0.8.7 → 0.8.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/context.d.ts +5 -1
- package/dist/context.js +6 -0
- package/dist/context.js.map +1 -1
- package/dist/defaults.js +42 -1
- package/dist/defaults.js.map +1 -1
- package/dist/error.d.ts +33 -0
- package/dist/error.js +167 -0
- package/dist/error.js.map +1 -1
- package/dist/merge.js +1 -1
- package/dist/merge.js.map +1 -1
- package/dist/parser.js +180 -51
- package/dist/parser.js.map +1 -1
- package/dist/rules.d.ts +2 -1
- package/dist/rules.js +522 -4
- package/dist/rules.js.map +1 -1
- package/dist/tabnas.d.ts +10 -3
- package/dist/tabnas.js +163 -4
- package/dist/tabnas.js.map +1 -1
- package/dist/types.d.ts +42 -0
- package/dist/types.js.map +1 -1
- package/dist/utility.js +43 -4
- package/dist/utility.js.map +1 -1
- package/package.json +4 -3
package/dist/rules.js
CHANGED
|
@@ -4,6 +4,7 @@ Object.defineProperty(exports, "__esModule", { value: true });
|
|
|
4
4
|
exports.makeRuleSpec = exports.makeNoRule = exports.makeRule = exports.AltMatch = exports.RuleSpec = exports.Rule = void 0;
|
|
5
5
|
exports.validateAlt = validateAlt;
|
|
6
6
|
exports.validateAlts = validateAlts;
|
|
7
|
+
exports.continuationTins = continuationTins;
|
|
7
8
|
const types_1 = require("./types");
|
|
8
9
|
const utility_1 = require("./utility");
|
|
9
10
|
const error_1 = require("./error");
|
|
@@ -377,6 +378,13 @@ class RuleSpec {
|
|
|
377
378
|
let alts = (is_open ? def.open : def.close);
|
|
378
379
|
// Handle "before" call.
|
|
379
380
|
let befores = is_open ? (rule.bo ? def.bo : null) : rule.bc ? def.bc : null;
|
|
381
|
+
// Error recovery resumed this rule on a pass whose before actions
|
|
382
|
+
// already ran — do not replay them (see attemptRecover).
|
|
383
|
+
if (rule._skipBefores) {
|
|
384
|
+
;
|
|
385
|
+
rule._skipBefores = false;
|
|
386
|
+
befores = null;
|
|
387
|
+
}
|
|
380
388
|
if (befores) {
|
|
381
389
|
let bout = undefined;
|
|
382
390
|
for (let bI = 0; bI < befores.length; bI++) {
|
|
@@ -394,6 +402,12 @@ class RuleSpec {
|
|
|
394
402
|
if (logging)
|
|
395
403
|
why += 'H';
|
|
396
404
|
}
|
|
405
|
+
// Expose the alternate this pass resolved to (post-modifier, so a
|
|
406
|
+
// replacement from alt.h is what consumers see), for the parser's
|
|
407
|
+
// post-process ruleDone event. A pass with no alternates exposes
|
|
408
|
+
// null, matching the public RuleDone contract.
|
|
409
|
+
;
|
|
410
|
+
ctx._dalt = 0 < alts.length ? alt : null;
|
|
397
411
|
// Unconditional error.
|
|
398
412
|
if (alt.e) {
|
|
399
413
|
return this.bad(alt.e, rule, ctx, { is_open });
|
|
@@ -560,6 +574,20 @@ class RuleSpec {
|
|
|
560
574
|
return next;
|
|
561
575
|
}
|
|
562
576
|
bad(tkn, rule, ctx, parse) {
|
|
577
|
+
// Opt-in recovery: record the error and continue from a sync point
|
|
578
|
+
// instead of throwing (options.parse.recover).
|
|
579
|
+
const rec = ctx.cfg?.parse?.recover;
|
|
580
|
+
if (rec?.enabled) {
|
|
581
|
+
const next = attemptRecover(tkn, rule, ctx, parse);
|
|
582
|
+
if (null != next)
|
|
583
|
+
return next;
|
|
584
|
+
// Recovery gave up (caps or exhausted stack). The error is
|
|
585
|
+
// already recorded; throw the last recorded one so parser.start
|
|
586
|
+
// can convert the parse to { value, errors }.
|
|
587
|
+
const lastErr = ctx.errs?.[ctx.errs.length - 1];
|
|
588
|
+
if (null != lastErr)
|
|
589
|
+
throw lastErr;
|
|
590
|
+
}
|
|
563
591
|
throw new error_1.TabnasError(tkn.err || utility_1.S.unexpected, {
|
|
564
592
|
...tkn.use,
|
|
565
593
|
state: parse.is_open ? utility_1.S.open : utility_1.S.close,
|
|
@@ -575,14 +603,401 @@ class RuleSpec {
|
|
|
575
603
|
exports.RuleSpec = RuleSpec;
|
|
576
604
|
const makeRuleSpec = (...params) => new RuleSpec(...params);
|
|
577
605
|
exports.makeRuleSpec = makeRuleSpec;
|
|
578
|
-
|
|
579
|
-
//
|
|
606
|
+
const closeInfoCache = new WeakMap();
|
|
607
|
+
// A bad token is produced without advancing the lex point (the
|
|
608
|
+
// matchers declined it) — skipping one forward requires moving the
|
|
609
|
+
// point past its span by hand, tracking rows and columns.
|
|
610
|
+
// Bad-token codes raised from INSIDE a compound construct (string
|
|
611
|
+
// body escapes and control characters): resuming the lexer just past
|
|
612
|
+
// the offending character would restart lexing mid-construct and
|
|
613
|
+
// mis-tokenize the remainder (a stray quote opens a "new" string).
|
|
614
|
+
// For these, recovery advances to a lexical boundary — the next line
|
|
615
|
+
// end — instead.
|
|
616
|
+
const MID_CONSTRUCT = {
|
|
617
|
+
unprintable: true,
|
|
618
|
+
invalid_unicode: true,
|
|
619
|
+
invalid_ascii: true,
|
|
620
|
+
};
|
|
621
|
+
function advanceLexPast(lex, t, toLineEnd) {
|
|
622
|
+
const pnt = lex.pnt;
|
|
623
|
+
const src = lex.src;
|
|
624
|
+
const cfg = lex.ctx?.cfg;
|
|
625
|
+
// Config.line.rowChars is a Chars lookup object; fall back to '\n'
|
|
626
|
+
// when line lexing is not configured.
|
|
627
|
+
const rowsObj = cfg?.line?.rowChars;
|
|
628
|
+
const isRow = (ch) => (null != rowsObj ? !!rowsObj[ch] : '\n' === ch);
|
|
629
|
+
let target = Math.max(pnt.sI, t.sI + Math.max(1, t.len | 0));
|
|
630
|
+
if (toLineEnd) {
|
|
631
|
+
let e = target;
|
|
632
|
+
while (e < src.length && !isRow(src[e]))
|
|
633
|
+
e++;
|
|
634
|
+
// Include the line end itself so lexing resumes on a fresh row.
|
|
635
|
+
target = Math.max(target, Math.min(src.length, e + 1));
|
|
636
|
+
}
|
|
637
|
+
while (pnt.sI < target && pnt.sI < src.length) {
|
|
638
|
+
if (isRow(src[pnt.sI])) {
|
|
639
|
+
pnt.rI++;
|
|
640
|
+
pnt.cI = 1;
|
|
641
|
+
}
|
|
642
|
+
else {
|
|
643
|
+
pnt.cI++;
|
|
644
|
+
}
|
|
645
|
+
pnt.sI++;
|
|
646
|
+
}
|
|
647
|
+
}
|
|
648
|
+
function closeInfo(spec, groups, sig) {
|
|
649
|
+
const close = (spec.def?.close ?? []);
|
|
650
|
+
let info = closeInfoCache.get(close);
|
|
651
|
+
if (null == info || info.len !== close.length || info.sig !== sig) {
|
|
652
|
+
const sync = new Set();
|
|
653
|
+
const all = new Set();
|
|
654
|
+
let any = false;
|
|
655
|
+
for (const alt of close) {
|
|
656
|
+
if (0 === (alt.sN | 0)) {
|
|
657
|
+
any = true;
|
|
658
|
+
continue;
|
|
659
|
+
}
|
|
660
|
+
const lead = alt.t?.[0] ?? [];
|
|
661
|
+
for (const tin of lead)
|
|
662
|
+
all.add(tin);
|
|
663
|
+
const g = alt.g ?? [];
|
|
664
|
+
for (const tag of g) {
|
|
665
|
+
if (groups.includes(tag)) {
|
|
666
|
+
for (const tin of lead)
|
|
667
|
+
sync.add(tin);
|
|
668
|
+
break;
|
|
669
|
+
}
|
|
670
|
+
}
|
|
671
|
+
}
|
|
672
|
+
info = { len: close.length, sig, sync, all, any };
|
|
673
|
+
closeInfoCache.set(close, info);
|
|
674
|
+
}
|
|
675
|
+
return info;
|
|
676
|
+
}
|
|
677
|
+
// The sync token set for the current error, computed from the live
|
|
678
|
+
// rule stack: leading tins of close alternates whose g tags intersect
|
|
679
|
+
// parse.recover.syncGroups, plus explicit syncTins. When the grammar
|
|
680
|
+
// is untagged (sync set empty) the normative structural fallback
|
|
681
|
+
// applies: every close-leading tin anywhere in the stack.
|
|
682
|
+
function computeSyncTins(ctx, rule) {
|
|
683
|
+
const rec = ctx.cfg.parse.recover;
|
|
684
|
+
const sig = rec.syncGroups.join(',');
|
|
685
|
+
const out = new Set();
|
|
686
|
+
const addSync = (r) => {
|
|
687
|
+
if (null != r && r !== ctx.NORULE) {
|
|
688
|
+
closeInfo(r.spec, rec.syncGroups, sig).sync.forEach((t) => out.add(t));
|
|
689
|
+
}
|
|
690
|
+
};
|
|
691
|
+
addSync(rule);
|
|
692
|
+
for (let d = ctx.rsI - 1; 0 <= d; d--)
|
|
693
|
+
addSync(ctx.rs[d]);
|
|
694
|
+
// Structural fallback for untagged grammars is decided on the
|
|
695
|
+
// grammar-derived set alone — explicit syncTokens are extras, and
|
|
696
|
+
// must not disable the fallback by their mere presence.
|
|
697
|
+
if (0 === out.size) {
|
|
698
|
+
const addAll = (r) => {
|
|
699
|
+
if (null != r && r !== ctx.NORULE) {
|
|
700
|
+
closeInfo(r.spec, rec.syncGroups, sig).all.forEach((t) => out.add(t));
|
|
701
|
+
}
|
|
702
|
+
};
|
|
703
|
+
addAll(rule);
|
|
704
|
+
for (let d = ctx.rsI - 1; 0 <= d; d--)
|
|
705
|
+
addAll(ctx.rs[d]);
|
|
706
|
+
}
|
|
707
|
+
for (const tin of rec.syncTins)
|
|
708
|
+
out.add(tin);
|
|
709
|
+
return out;
|
|
710
|
+
}
|
|
711
|
+
// Does this rule's close state accept the token? An empty-s close
|
|
712
|
+
// alternate accepts anything (over-approximation: conditions and
|
|
713
|
+
// counters may still reject — cascade suppression absorbs that).
|
|
714
|
+
function acceptsClose(spec, tin, groups, sig) {
|
|
715
|
+
const info = closeInfo(spec, groups, sig);
|
|
716
|
+
return info.any || info.all.has(tin);
|
|
717
|
+
}
|
|
718
|
+
// Legal-continuation tokens at an error point: the failing rule's
|
|
719
|
+
// collated lookahead tins at the deepest position any alternate
|
|
720
|
+
// matched (ctx._eMax), widened by a pop-closure — while a rule's close
|
|
721
|
+
// state has an empty-s catch-all alternate (it can close on anything),
|
|
722
|
+
// the parent's close continuations are legal here too. Powers
|
|
723
|
+
// tn.continuations() (completion in the unified-LSP design).
|
|
724
|
+
// How many leading positions of an alternate match the currently
|
|
725
|
+
// buffered lookahead. Used by the continuation computation to decide
|
|
726
|
+
// whether an alternate is FULLY matched (and so its push/replace
|
|
727
|
+
// target's opening tokens are legal at this position).
|
|
728
|
+
function altMatchDepth(alt, ctx) {
|
|
729
|
+
const aN = alt.sN | 0;
|
|
730
|
+
const tbuf = ctx.t;
|
|
731
|
+
const NOTOKEN = ctx.NOTOKEN;
|
|
732
|
+
const bitAA = 1 << (ctx.cfg.t.AA - 1);
|
|
733
|
+
let k = 0;
|
|
734
|
+
while (k < aN) {
|
|
735
|
+
const tk = tbuf[k];
|
|
736
|
+
if (null == tk || NOTOKEN === tk)
|
|
737
|
+
break;
|
|
738
|
+
const Sk = alt.S ? alt.S[k] : null;
|
|
739
|
+
if (null != Sk) {
|
|
740
|
+
const tin = tk.tin;
|
|
741
|
+
const part = (tin / 31) | 0;
|
|
742
|
+
const aaBit = 0 === part ? bitAA : 0;
|
|
743
|
+
if (0 === (Sk[part] & ((1 << ((tin % 31) - 1)) | aaBit)))
|
|
744
|
+
break;
|
|
745
|
+
}
|
|
746
|
+
k++;
|
|
747
|
+
}
|
|
748
|
+
return k;
|
|
749
|
+
}
|
|
750
|
+
function continuationTins(ctx, rule, atPos) {
|
|
751
|
+
const out = new Set();
|
|
752
|
+
if (null == rule || null == ctx || rule === ctx.NORULE)
|
|
753
|
+
return [];
|
|
754
|
+
// The buffer position being asked about: a token offered here must
|
|
755
|
+
// FOLLOW everything already buffered, never replace part of it.
|
|
756
|
+
const queryPos = null == atPos ? 0 : atPos;
|
|
757
|
+
// Path-aware tins recorded by parse_alts at the failure point; the
|
|
758
|
+
// collated tcol is the fallback (exact at position 0, where no
|
|
759
|
+
// prefix exists to disambiguate).
|
|
760
|
+
const cont = ctx._contTins;
|
|
761
|
+
if (null != cont && 0 < cont.length) {
|
|
762
|
+
for (const t of cont)
|
|
763
|
+
out.add(t);
|
|
764
|
+
}
|
|
765
|
+
else {
|
|
766
|
+
// No failure record (the prefix parsed): compute the same
|
|
767
|
+
// path-aware set directly from the buffer. Per alternate, only the
|
|
768
|
+
// position it is actually waiting on contributes — the collated
|
|
769
|
+
// tcol is path-blind and would offer a sibling alternate's later
|
|
770
|
+
// tokens whose own prefix never matched.
|
|
771
|
+
const stateAltsNow = (types_1.OPEN === rule.state
|
|
772
|
+
? rule.spec.def?.open
|
|
773
|
+
: rule.spec.def?.close);
|
|
774
|
+
for (const a of stateAltsNow ?? []) {
|
|
775
|
+
const aN = a.sN | 0;
|
|
776
|
+
const k = altMatchDepth(a, ctx);
|
|
777
|
+
// An alternate speaks only for the position it actually reached.
|
|
778
|
+
// One that stopped short of the query (k < queryPos) mismatched a
|
|
779
|
+
// token that is already in the buffer, so the tokens it wants at
|
|
780
|
+
// k would have to REPLACE that token rather than follow it —
|
|
781
|
+
// offering them puts a sibling branch's opener in a completion
|
|
782
|
+
// list where it cannot legally appear.
|
|
783
|
+
if (k === queryPos && k < aN) {
|
|
784
|
+
for (const t of a.t?.[k] ?? [])
|
|
785
|
+
out.add(t);
|
|
786
|
+
}
|
|
787
|
+
}
|
|
788
|
+
}
|
|
789
|
+
// Pop-closure over empty-close ancestors: bounded by the finite rule
|
|
790
|
+
// stack itself (d decrements to exhaustion), no arbitrary cap.
|
|
791
|
+
let r = rule;
|
|
792
|
+
let d = ctx.rsI - 1;
|
|
793
|
+
for (;;) {
|
|
794
|
+
const info = closeInfo(r.spec, [], '');
|
|
795
|
+
if (!info.any)
|
|
796
|
+
break;
|
|
797
|
+
const parent = 0 <= d ? ctx.rs[d--] : undefined;
|
|
798
|
+
if (null == parent || parent === ctx.NORULE)
|
|
799
|
+
break;
|
|
800
|
+
const ptcol = parent.spec.def?.tcol;
|
|
801
|
+
const pAt = ptcol?.[1]?.[0] ?? [];
|
|
802
|
+
for (const t of pAt)
|
|
803
|
+
out.add(t);
|
|
804
|
+
r = parent;
|
|
805
|
+
}
|
|
806
|
+
// Push/replace closure. An alternate whose token sequence is FULLY
|
|
807
|
+
// matched at this position is about to push (or become) another
|
|
808
|
+
// rule, so that rule's opening tokens are legal here too — this is
|
|
809
|
+
// what makes `[1,` offer the next element's value starters rather
|
|
810
|
+
// than only the closers its own close alternates name. Alternates
|
|
811
|
+
// that are only partially matched are deliberately excluded: they
|
|
812
|
+
// still require their own next token first (`{"a"` wants `:`, not
|
|
813
|
+
// a value).
|
|
814
|
+
const opened = new Set();
|
|
815
|
+
const addOpeners = (name) => {
|
|
816
|
+
if (null == name || false === name || 0 === name)
|
|
817
|
+
return;
|
|
818
|
+
const rn = String(name);
|
|
819
|
+
if ('' === rn || opened.has(rn))
|
|
820
|
+
return;
|
|
821
|
+
opened.add(rn);
|
|
822
|
+
const spec = ctx.rsm[rn];
|
|
823
|
+
if (null == spec)
|
|
824
|
+
return;
|
|
825
|
+
const openTins = spec.def?.tcol?.[0]?.[0] ?? [];
|
|
826
|
+
for (const t of openTins)
|
|
827
|
+
out.add(t);
|
|
828
|
+
// A rule that opens with an empty-sequence alternate immediately
|
|
829
|
+
// hands over to its own target: follow that through.
|
|
830
|
+
for (const a of (spec.def?.open ?? [])) {
|
|
831
|
+
if (0 === (a.sN | 0)) {
|
|
832
|
+
addOpeners(a.p);
|
|
833
|
+
addOpeners(a.r);
|
|
834
|
+
}
|
|
835
|
+
}
|
|
836
|
+
};
|
|
837
|
+
const stateAlts = (types_1.OPEN === rule.state
|
|
838
|
+
? rule.spec.def?.open
|
|
839
|
+
: rule.spec.def?.close);
|
|
840
|
+
for (const a of stateAlts ?? []) {
|
|
841
|
+
const aN = a.sN | 0;
|
|
842
|
+
if (altMatchDepth(a, ctx) !== aN)
|
|
843
|
+
continue;
|
|
844
|
+
// Backtracking moves the handover point: an alternate that matches
|
|
845
|
+
// sN tokens and pushes back b of them leaves the target rule
|
|
846
|
+
// starting at sN-b, so its openers are legal at the QUERY position
|
|
847
|
+
// only when those coincide. Without this, `[A] {b:1, p:child}`
|
|
848
|
+
// would offer child's openers at the position after A even though
|
|
849
|
+
// child re-consumes A itself.
|
|
850
|
+
// The functional form (`b: (rule, ctx, alt) => n`) resolves only
|
|
851
|
+
// while the alternate is being matched, and this runs outside that
|
|
852
|
+
// — there is no resolved count to read, so the handover point is
|
|
853
|
+
// unknown. Skip: a missing completion beats a wrong one.
|
|
854
|
+
const ab = a.b;
|
|
855
|
+
if (null != ab && false !== ab && 'number' !== typeof ab)
|
|
856
|
+
continue;
|
|
857
|
+
const bN = 'number' === typeof ab ? ab : 0;
|
|
858
|
+
if (aN - bN !== queryPos)
|
|
859
|
+
continue;
|
|
860
|
+
addOpeners(a.p);
|
|
861
|
+
addOpeners(a.r);
|
|
862
|
+
}
|
|
863
|
+
return [...out].sort((a, b) => a - b);
|
|
864
|
+
}
|
|
865
|
+
// Panic-mode recovery: record the error, skip forward to a sync token,
|
|
866
|
+
// pop the rule stack to a rule that can consume it, and return that
|
|
867
|
+
// rule so the main loop continues. Returns undefined to give up (the
|
|
868
|
+
// caller throws; parser.start converts to { value, errors } in
|
|
869
|
+
// recovery mode).
|
|
870
|
+
function attemptRecover(tkn, rule, ctx, parse) {
|
|
871
|
+
const rec = ctx.cfg.parse.recover;
|
|
872
|
+
const lex = ctx.lex;
|
|
873
|
+
if (null == lex)
|
|
874
|
+
return undefined;
|
|
875
|
+
// Record the error (the TabnasError constructor pushes to ctx.errs).
|
|
876
|
+
const err = new error_1.TabnasError(tkn.err || tkn.why || utility_1.S.unexpected, { ...tkn.use, state: parse.is_open ? utility_1.S.open : utility_1.S.close }, tkn, rule, ctx);
|
|
877
|
+
// Cascade suppression: an error within `suppress` consumed tokens of
|
|
878
|
+
// the previous recovery is dropped as a follow-on of the same fault.
|
|
879
|
+
const lastAbs = ctx._recoverAt;
|
|
880
|
+
if (null != lastAbs && ctx.vAbs - lastAbs < rec.suppress) {
|
|
881
|
+
if (ctx.errs[ctx.errs.length - 1] === err)
|
|
882
|
+
ctx.errs.pop();
|
|
883
|
+
}
|
|
884
|
+
if (rec.maxRecoveries < ctx.errs.length)
|
|
885
|
+
return undefined;
|
|
886
|
+
// Strict-progress guard: when nothing was consumed since the last
|
|
887
|
+
// recovery, the previous resume point failed to advance the parse
|
|
888
|
+
// (e.g. every accepting alternate's condition rejected). Requiring
|
|
889
|
+
// the next sync candidate to sit strictly beyond the previous one
|
|
890
|
+
// bounds total recoveries by source length, so a recovery can never
|
|
891
|
+
// loop in place.
|
|
892
|
+
const lastSI = ctx._recoverSI ?? -1;
|
|
893
|
+
const noProgress = null != lastAbs && lastAbs === ctx.vAbs;
|
|
894
|
+
ctx._recoverAt = ctx.vAbs;
|
|
895
|
+
const sync = computeSyncTins(ctx, rule);
|
|
896
|
+
const ZZ = ctx.cfg.t.ZZ;
|
|
897
|
+
const BD = ctx.cfg.t.BD;
|
|
898
|
+
const IGNORE = ctx.cfg.tokenSetTins.IGNORE;
|
|
899
|
+
const NOTOKEN = ctx.NOTOKEN;
|
|
900
|
+
const tbuf = ctx.t;
|
|
901
|
+
const advancePast = (t, toLineEnd) => advanceLexPast(lex, t, toLineEnd);
|
|
902
|
+
// Skip forward: drain already-fetched lookahead first (those tokens
|
|
903
|
+
// advanced the lexer and must not be lost), then pull fresh tokens.
|
|
904
|
+
// Bad tokens are skipped without recording — they are part of the
|
|
905
|
+
// same error region.
|
|
906
|
+
const pending = [];
|
|
907
|
+
for (let i = 0; i < tbuf.length; i++) {
|
|
908
|
+
const t = tbuf[i];
|
|
909
|
+
if (null != t && NOTOKEN !== t)
|
|
910
|
+
pending.push(t);
|
|
911
|
+
tbuf[i] = NOTOKEN;
|
|
912
|
+
}
|
|
913
|
+
const fetch = () => {
|
|
914
|
+
let t;
|
|
915
|
+
do {
|
|
916
|
+
t = lex.next(rule);
|
|
917
|
+
} while (IGNORE[t.tin]);
|
|
918
|
+
return t;
|
|
919
|
+
};
|
|
920
|
+
let cand = 0 < pending.length ? pending.shift() : fetch();
|
|
921
|
+
let skipped = 0;
|
|
922
|
+
while (ZZ !== cand.tin &&
|
|
923
|
+
(!sync.has(cand.tin) || (noProgress && cand.sI <= lastSI))) {
|
|
924
|
+
if (BD === cand.tin) {
|
|
925
|
+
advancePast(cand, true === MID_CONSTRUCT[cand.why]);
|
|
926
|
+
}
|
|
927
|
+
if (rec.maxSkip <= skipped++)
|
|
928
|
+
return undefined;
|
|
929
|
+
cand = 0 < pending.length ? pending.shift() : fetch();
|
|
930
|
+
}
|
|
931
|
+
// End of source with no progress since the last recovery: the
|
|
932
|
+
// grammar has already had its chance to close on the end token —
|
|
933
|
+
// give up rather than spin on the pinned end token.
|
|
934
|
+
if (ZZ === cand.tin && noProgress && lastSI >= cand.sI) {
|
|
935
|
+
return undefined;
|
|
936
|
+
}
|
|
937
|
+
;
|
|
938
|
+
ctx._recoverSI = cand.sI;
|
|
939
|
+
// Restore the buffer: the sync token first, then any remaining
|
|
940
|
+
// pre-fetched lookahead in original order.
|
|
941
|
+
tbuf[0] = cand;
|
|
942
|
+
for (let i = 0; i < pending.length && i + 1 < tbuf.length; i++) {
|
|
943
|
+
tbuf[i + 1] = pending[i];
|
|
944
|
+
}
|
|
945
|
+
// The skipped region, for diagnostics consumers (not part of the
|
|
946
|
+
// structured diagnostic JSON shape).
|
|
947
|
+
;
|
|
948
|
+
err.recovered = { skipped, sync: cand.tin };
|
|
949
|
+
const sig = rec.syncGroups.join(',');
|
|
950
|
+
if (rec.popUntilValid) {
|
|
951
|
+
// Resume with the erroring rule itself if its close state accepts
|
|
952
|
+
// the sync token, else pop ancestors until one does.
|
|
953
|
+
if (rule !== ctx.NORULE && acceptsClose(rule.spec, cand.tin, rec.syncGroups, sig)) {
|
|
954
|
+
if (types_1.OPEN === rule.state) {
|
|
955
|
+
rule.state = types_1.CLOSE;
|
|
956
|
+
}
|
|
957
|
+
else if (!parse.is_open) {
|
|
958
|
+
// The failed pass was this rule's OWN close: its before-close
|
|
959
|
+
// actions already ran, and re-processing the close pass would
|
|
960
|
+
// run them again — corrupting non-idempotent actions (e.g. an
|
|
961
|
+
// element append). Skip them on the resumed pass.
|
|
962
|
+
;
|
|
963
|
+
rule._skipBefores = true;
|
|
964
|
+
}
|
|
965
|
+
return rule;
|
|
966
|
+
}
|
|
967
|
+
// The erroring rule itself is being abandoned (its close cannot
|
|
968
|
+
// accept the sync token, and it is not on ctx.rs): synthesize its
|
|
969
|
+
// close notification first so the structural stream stays balanced.
|
|
970
|
+
if (rule !== ctx.NORULE && ctx.sub.ruleDone) {
|
|
971
|
+
const done = { state: types_1.CLOSE, alt: null, forced: true };
|
|
972
|
+
ctx.sub.ruleDone.map((s) => s(rule, ctx, done));
|
|
973
|
+
}
|
|
974
|
+
while (0 < ctx.rsI) {
|
|
975
|
+
const r = ctx.rs[--ctx.rsI];
|
|
976
|
+
if (null != r && acceptsClose(r.spec, cand.tin, rec.syncGroups, sig)) {
|
|
977
|
+
return r;
|
|
978
|
+
}
|
|
979
|
+
// Force-popped without a close pass: synthesize the close
|
|
980
|
+
// notification so structural consumers (outline/folding) see a
|
|
981
|
+
// balanced event stream even through recovery.
|
|
982
|
+
if (null != r && ctx.sub.ruleDone) {
|
|
983
|
+
const done = { state: types_1.CLOSE, alt: null, forced: true };
|
|
984
|
+
ctx.sub.ruleDone.map((s) => s(r, ctx, done));
|
|
985
|
+
}
|
|
986
|
+
}
|
|
987
|
+
return undefined;
|
|
988
|
+
}
|
|
989
|
+
// Fixed-depth pop: one rule.
|
|
990
|
+
if (0 < ctx.rsI)
|
|
991
|
+
return ctx.rs[--ctx.rsI];
|
|
992
|
+
return undefined;
|
|
993
|
+
}
|
|
580
994
|
function parse_alts(is_open, alts, lex, rule, ctx) {
|
|
581
995
|
// One reusable scratch AltMatch per parse Context (allocated lazily on
|
|
582
996
|
// first use). Scoping it to the Context — rather than a module global —
|
|
583
997
|
// keeps nested parses (a plugin action parsing with another instance)
|
|
584
998
|
// from clobbering each other's in-flight match state.
|
|
585
999
|
let out = ctx._palt || (ctx._palt = makeAltMatch());
|
|
1000
|
+
ctx._contTins = undefined;
|
|
586
1001
|
out.b = 0; // Backtrack n tokens.
|
|
587
1002
|
out.p = types_1.EMPTY; // Push named rule onto stack.
|
|
588
1003
|
out.r = types_1.EMPTY; // Replace current rule with named rule.
|
|
@@ -627,6 +1042,7 @@ function parse_alts(is_open, alts, lex, rule, ctx) {
|
|
|
627
1042
|
let unCI = 0;
|
|
628
1043
|
let unQueue = null;
|
|
629
1044
|
let unEnd = undefined;
|
|
1045
|
+
let deepest = 0;
|
|
630
1046
|
for (altI = 0; altI < len; altI++) {
|
|
631
1047
|
alt = alts[altI];
|
|
632
1048
|
// Number of positions that matched in this alt. Tracked so the
|
|
@@ -650,7 +1066,9 @@ function parse_alts(is_open, alts, lex, rule, ctx) {
|
|
|
650
1066
|
if (null == tkn || NOTOKEN === tkn) {
|
|
651
1067
|
// Fetch (skipping IGNORE tokens) inline — a nested function here
|
|
652
1068
|
// would allocate a closure on every parse_alts call.
|
|
1069
|
+
let refetch = false;
|
|
653
1070
|
do {
|
|
1071
|
+
refetch = false;
|
|
654
1072
|
tkn = lex.next(rule, alt, altI, i);
|
|
655
1073
|
ctx.tC++;
|
|
656
1074
|
// Bad tokens abort the parse with their own error code
|
|
@@ -669,9 +1087,61 @@ function parse_alts(is_open, alts, lex, rule, ctx) {
|
|
|
669
1087
|
if (null != tkn.use) {
|
|
670
1088
|
details.use = tkn.use;
|
|
671
1089
|
}
|
|
1090
|
+
// Opt-in recovery: lexer soft mode. Record the bad token's
|
|
1091
|
+
// own error and skip it, keeping the fetch going; beyond
|
|
1092
|
+
// the recovery cap the parse gives up as usual.
|
|
1093
|
+
const rec = ctx.cfg.parse.recover;
|
|
1094
|
+
if (rec.enabled) {
|
|
1095
|
+
const bderr = new error_1.TabnasError(tkn.why || UNEXPECTED, details, tkn, rule, ctx);
|
|
1096
|
+
// Coalesce a contiguous run of bad tokens (e.g. each
|
|
1097
|
+
// character of an unlexable word) into one recorded
|
|
1098
|
+
// error whose region metadata grows with the run, and
|
|
1099
|
+
// apply the same cascade-suppression window recoveries
|
|
1100
|
+
// use: a fresh bad region with no consumed tokens since
|
|
1101
|
+
// the previous recovery is a follow-on of the same fault.
|
|
1102
|
+
const runEnd = ctx._badTo;
|
|
1103
|
+
const runErr = ctx._badErr;
|
|
1104
|
+
const lastAbs = ctx._recoverAt;
|
|
1105
|
+
if (null != runEnd &&
|
|
1106
|
+
tkn.sI <= runEnd &&
|
|
1107
|
+
null != runErr &&
|
|
1108
|
+
ctx.errs[ctx.errs.length - 1] === bderr) {
|
|
1109
|
+
ctx.errs.pop();
|
|
1110
|
+
runErr.recovered.skipped++;
|
|
1111
|
+
// A run longer than maxSkip gives up like any other
|
|
1112
|
+
// over-long recovery.
|
|
1113
|
+
if (rec.maxSkip < runErr.recovered.skipped) {
|
|
1114
|
+
throw bderr;
|
|
1115
|
+
}
|
|
1116
|
+
}
|
|
1117
|
+
else if (null != lastAbs &&
|
|
1118
|
+
ctx.vAbs - lastAbs < rec.suppress &&
|
|
1119
|
+
ctx.errs[ctx.errs.length - 1] === bderr) {
|
|
1120
|
+
ctx.errs.pop();
|
|
1121
|
+
ctx._badErr = null;
|
|
1122
|
+
}
|
|
1123
|
+
else {
|
|
1124
|
+
;
|
|
1125
|
+
bderr.recovered = { skipped: 1, bad: true };
|
|
1126
|
+
ctx._badErr = bderr;
|
|
1127
|
+
ctx._recoverAt = ctx.vAbs;
|
|
1128
|
+
}
|
|
1129
|
+
;
|
|
1130
|
+
ctx._badTo = tkn.sI + Math.max(1, tkn.len | 0);
|
|
1131
|
+
if (rec.maxRecoveries < ctx.errs.length) {
|
|
1132
|
+
throw bderr;
|
|
1133
|
+
}
|
|
1134
|
+
// The bad token did not advance the lex point — move
|
|
1135
|
+
// past it or the refetch would loop in place. Bad tokens
|
|
1136
|
+
// raised mid-construct (string escapes/control chars)
|
|
1137
|
+
// resynchronize at the next line end instead.
|
|
1138
|
+
advanceLexPast(lex, tkn, true === MID_CONSTRUCT[tkn.why]);
|
|
1139
|
+
refetch = true;
|
|
1140
|
+
continue;
|
|
1141
|
+
}
|
|
672
1142
|
throw new error_1.TabnasError(tkn.why || UNEXPECTED, details, tkn, rule, ctx);
|
|
673
1143
|
}
|
|
674
|
-
} while (IGNORE[tkn.tin]);
|
|
1144
|
+
} while (refetch || IGNORE[tkn.tin]);
|
|
675
1145
|
tbuf[i] = tkn;
|
|
676
1146
|
}
|
|
677
1147
|
const Si = S ? S[i] : null;
|
|
@@ -765,6 +1235,8 @@ function parse_alts(is_open, alts, lex, rule, ctx) {
|
|
|
765
1235
|
break;
|
|
766
1236
|
}
|
|
767
1237
|
else {
|
|
1238
|
+
if (matched > deepest)
|
|
1239
|
+
deepest = matched;
|
|
768
1240
|
alt = null;
|
|
769
1241
|
// This alternate renegotiated a token and then failed anyway —
|
|
770
1242
|
// put the cut back, so the alternates and rules that follow see
|
|
@@ -775,23 +1247,69 @@ function parse_alts(is_open, alts, lex, rule, ctx) {
|
|
|
775
1247
|
for (let j = unI + 1; j < tbuf.length; j++) {
|
|
776
1248
|
tbuf[j] = NOTOKEN;
|
|
777
1249
|
}
|
|
1250
|
+
// Re-announce the RESTORED token to lex subscribers: the recut
|
|
1251
|
+
// fired an event when it was cut, and without a matching event
|
|
1252
|
+
// for the undo, a position-keyed consumer's last-write-wins
|
|
1253
|
+
// reconstruction would keep the abandoned recut. After this,
|
|
1254
|
+
// "latest event per source position" is always the token the
|
|
1255
|
+
// parse actually proceeded with.
|
|
1256
|
+
if (ctx.sub.lex) {
|
|
1257
|
+
ctx.sub.lex.map((sub) => sub(unTkn, rule, ctx));
|
|
1258
|
+
}
|
|
778
1259
|
unI = -1;
|
|
779
1260
|
}
|
|
780
1261
|
}
|
|
781
1262
|
}
|
|
782
1263
|
if (!cond) {
|
|
783
1264
|
const bad = tbuf[0];
|
|
1265
|
+
ctx._eMax = deepest;
|
|
1266
|
+
// Path-aware continuation tins: for each alternate, count how many
|
|
1267
|
+
// leading positions actually match the buffered lookahead, and
|
|
1268
|
+
// collect the alternate's OWN next-position tins — tcol collation
|
|
1269
|
+
// is path-blind (after matching A of [A,B], a sibling [C,D] must
|
|
1270
|
+
// not contribute D). Error path only; no hot-path cost.
|
|
1271
|
+
{
|
|
1272
|
+
const cont = new Set();
|
|
1273
|
+
for (let aI = 0; aI < len; aI++) {
|
|
1274
|
+
const a = alts[aI];
|
|
1275
|
+
const aN = a.sN | 0;
|
|
1276
|
+
let k = 0;
|
|
1277
|
+
while (k < aN) {
|
|
1278
|
+
const tk = tbuf[k];
|
|
1279
|
+
if (null == tk || NOTOKEN === tk)
|
|
1280
|
+
break;
|
|
1281
|
+
const Sk = a.S ? a.S[k] : null;
|
|
1282
|
+
if (null != Sk) {
|
|
1283
|
+
const tin = tk.tin;
|
|
1284
|
+
const part = (tin / 31) | 0;
|
|
1285
|
+
const aaBit = 0 === part ? bitAA : 0;
|
|
1286
|
+
if (0 === (Sk[part] & ((1 << ((tin % 31) - 1)) | aaBit)))
|
|
1287
|
+
break;
|
|
1288
|
+
}
|
|
1289
|
+
k++;
|
|
1290
|
+
}
|
|
1291
|
+
if (k < aN) {
|
|
1292
|
+
const next = a.t?.[k] ?? [];
|
|
1293
|
+
for (const tin of next)
|
|
1294
|
+
cont.add(tin);
|
|
1295
|
+
}
|
|
1296
|
+
}
|
|
1297
|
+
;
|
|
1298
|
+
ctx._contTins = [...cont];
|
|
1299
|
+
}
|
|
784
1300
|
// No alternate could use the token and it is a bad one: raise the
|
|
785
1301
|
// lexer's own error, exactly as the non-negotiated path does at
|
|
786
1302
|
// fetch time. Deferring that throw is what let the alternates try to
|
|
787
1303
|
// re-cut it; now that all of them have declined, the specific
|
|
788
1304
|
// diagnostic is the useful one.
|
|
789
|
-
if (RELEX && null != bad && BD === bad.tin) {
|
|
1305
|
+
if (RELEX && null != bad && BD === bad.tin && !ctx.cfg.parse.recover.enabled) {
|
|
790
1306
|
const details = {};
|
|
791
1307
|
if (null != bad.use) {
|
|
792
1308
|
details.use = bad.use;
|
|
793
1309
|
}
|
|
794
1310
|
throw new error_1.TabnasError(bad.why || UNEXPECTED, details, bad, rule, ctx);
|
|
1311
|
+
// In recovery mode the bad token flows through out.e below into
|
|
1312
|
+
// RuleSpec.bad, which routes it into attemptRecover.
|
|
795
1313
|
}
|
|
796
1314
|
out.e = tbuf[0] ?? NOTOKEN;
|
|
797
1315
|
}
|