@tabnas/parser 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +20 -0
- package/README.md +75 -0
- package/dist/builtins.d.ts +4 -0
- package/dist/builtins.js +202 -0
- package/dist/builtins.js.map +1 -0
- package/dist/context.d.ts +64 -0
- package/dist/context.js +138 -0
- package/dist/context.js.map +1 -0
- package/dist/defaults.d.ts +3 -0
- package/dist/defaults.js +301 -0
- package/dist/defaults.js.map +1 -0
- package/dist/error.d.ts +51 -0
- package/dist/error.js +362 -0
- package/dist/error.js.map +1 -0
- package/dist/lexer.d.ts +85 -0
- package/dist/lexer.js +1202 -0
- package/dist/lexer.js.map +1 -0
- package/dist/parser.d.ts +15 -0
- package/dist/parser.js +141 -0
- package/dist/parser.js.map +1 -0
- package/dist/rules.d.ts +100 -0
- package/dist/rules.js +923 -0
- package/dist/rules.js.map +1 -0
- package/dist/tabnas.d.ts +103 -0
- package/dist/tabnas.js +387 -0
- package/dist/tabnas.js.map +1 -0
- package/dist/tsconfig.tsbuildinfo +1 -0
- package/dist/types.d.ts +527 -0
- package/dist/types.js +24 -0
- package/dist/types.js.map +1 -0
- package/dist/utility.d.ts +80 -0
- package/dist/utility.js +868 -0
- package/dist/utility.js.map +1 -0
- package/package.json +59 -0
package/dist/lexer.js
ADDED
|
@@ -0,0 +1,1202 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/* Copyright (c) 2013-2026 Richard Rodger, MIT License */
|
|
3
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
4
|
+
exports.STATE_MASK = exports.STOP = exports.CI_RESET = exports.IS_ROW = exports.CONSUME = exports.makeTextMatcher = exports.makeNumberMatcher = exports.makeCommentMatcher = exports.makeStringMatcher = exports.makeLineMatcher = exports.makeSpaceMatcher = exports.makeFixedMatcher = exports.makeMatchMatcher = exports.makeToken = exports.makePoint = exports.makeLex = exports.makeNoToken = exports.Token = exports.Point = exports.Lex = void 0;
|
|
5
|
+
exports.guardedMatcher = guardedMatcher;
|
|
6
|
+
exports.scan = scan;
|
|
7
|
+
exports.buildCharRunSpec = buildCharRunSpec;
|
|
8
|
+
exports.buildLineRunSpec = buildLineRunSpec;
|
|
9
|
+
exports.buildStringBodySpec = buildStringBodySpec;
|
|
10
|
+
const types_1 = require("./types");
|
|
11
|
+
const utility_1 = require("./utility");
|
|
12
|
+
// Scan position threaded through the parse: source index, row/column, and pending-token queue.
|
|
13
|
+
class Point {
|
|
14
|
+
constructor(len, sI, rI, cI) {
|
|
15
|
+
this.len = -1; // Total source length.
|
|
16
|
+
this.sI = 0; // Source index (chars consumed so far).
|
|
17
|
+
this.rI = 1; // Row (1-based, for error messages).
|
|
18
|
+
this.cI = 1; // Column (1-based, for error messages).
|
|
19
|
+
this.token = []; // Pending-token queue (lookahead / rewind feed it).
|
|
20
|
+
this.len = len;
|
|
21
|
+
if (null != sI) {
|
|
22
|
+
this.sI = sI;
|
|
23
|
+
}
|
|
24
|
+
if (null != rI) {
|
|
25
|
+
this.rI = rI;
|
|
26
|
+
}
|
|
27
|
+
if (null != cI) {
|
|
28
|
+
this.cI = cI;
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
toString() {
|
|
32
|
+
return ('Point[' +
|
|
33
|
+
[this.sI + '/' + this.len, this.rI, this.cI] +
|
|
34
|
+
(0 < this.token.length ? ' ' + this.token : '') +
|
|
35
|
+
']');
|
|
36
|
+
}
|
|
37
|
+
[types_1.INSPECT]() {
|
|
38
|
+
return this.toString();
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
exports.Point = Point;
|
|
42
|
+
const makePoint = (...params) => new Point(...params);
|
|
43
|
+
exports.makePoint = makePoint;
|
|
44
|
+
// A single lexed token: numeric token id, JS-typed value, raw source text, and match position.
|
|
45
|
+
class Token {
|
|
46
|
+
constructor(name, tin, val, src, pnt, use, why) {
|
|
47
|
+
this.isToken = true; // Marker discriminating Tokens from other values.
|
|
48
|
+
this.name = types_1.EMPTY; // Token name (e.g. '#NR', '#ST').
|
|
49
|
+
this.tin = -1; // Numeric token id corresponding to name.
|
|
50
|
+
this.val = undefined; // JS-typed value (e.g. a number for #NR).
|
|
51
|
+
this.src = types_1.EMPTY; // Raw matching source text.
|
|
52
|
+
this.sI = -1; // Source index where the match started.
|
|
53
|
+
this.rI = -1; // Row where the match started.
|
|
54
|
+
this.cI = -1; // Column where the match started.
|
|
55
|
+
this.len = -1; // Length of src.
|
|
56
|
+
this.name = name;
|
|
57
|
+
this.tin = tin;
|
|
58
|
+
this.src = src;
|
|
59
|
+
this.val = val;
|
|
60
|
+
this.sI = pnt.sI;
|
|
61
|
+
this.rI = pnt.rI;
|
|
62
|
+
this.cI = pnt.cI;
|
|
63
|
+
this.use = use;
|
|
64
|
+
this.why = why;
|
|
65
|
+
this.len = null == src ? 0 : src.length;
|
|
66
|
+
}
|
|
67
|
+
resolveVal(rule, ctx) {
|
|
68
|
+
let out = 'function' === typeof this.val ? this.val(rule, ctx) : this.val;
|
|
69
|
+
return out;
|
|
70
|
+
}
|
|
71
|
+
bad(err, details) {
|
|
72
|
+
this.err = err;
|
|
73
|
+
if (null != details) {
|
|
74
|
+
this.use = (0, utility_1.deep)(this.use || {}, details);
|
|
75
|
+
}
|
|
76
|
+
return this;
|
|
77
|
+
}
|
|
78
|
+
toString() {
|
|
79
|
+
return ('Token[' +
|
|
80
|
+
this.name +
|
|
81
|
+
'=' +
|
|
82
|
+
this.tin +
|
|
83
|
+
' ' +
|
|
84
|
+
(0, utility_1.snip)(this.src) +
|
|
85
|
+
(undefined === this.val || '#ST' === this.name || '#TX' === this.name
|
|
86
|
+
? ''
|
|
87
|
+
: '=' + (0, utility_1.snip)(this.val)) +
|
|
88
|
+
' ' +
|
|
89
|
+
[this.sI, this.rI, this.cI] +
|
|
90
|
+
(null == this.use
|
|
91
|
+
? ''
|
|
92
|
+
: ' ' + (0, utility_1.snip)('' + JSON.stringify(this.use).replace(/"/g, ''), 22)) +
|
|
93
|
+
(null == this.err ? '' : ' ' + this.err) +
|
|
94
|
+
(null == this.why ? '' : ' ' + (0, utility_1.snip)('' + this.why, 22)) +
|
|
95
|
+
']');
|
|
96
|
+
}
|
|
97
|
+
[types_1.INSPECT]() {
|
|
98
|
+
return this.toString();
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
exports.Token = Token;
|
|
102
|
+
const makeToken = (...params) => new Token(...params);
|
|
103
|
+
exports.makeToken = makeToken;
|
|
104
|
+
const makeNoToken = () => makeToken('', -1, undefined, types_1.EMPTY, makePoint(-1));
|
|
105
|
+
exports.makeNoToken = makeNoToken;
|
|
106
|
+
// Wrap a matcher body in the standard entry guards: skip when
|
|
107
|
+
// `mcfg.lex` is false, and consult an optional `check` hook that
|
|
108
|
+
// may short-circuit by returning `{ done: true, token }`.
|
|
109
|
+
//
|
|
110
|
+
// `mcfg` is captured once at matcher-build time. The matcher
|
|
111
|
+
// factories are re-invoked on every `tn.make()` clone (via
|
|
112
|
+
// `configure()`), so a stale closure can never outlive the cfg
|
|
113
|
+
// snapshot it was built from.
|
|
114
|
+
function guardedMatcher(mcfg, body) {
|
|
115
|
+
return function guarded(lex, rule, tI) {
|
|
116
|
+
if (!mcfg.lex)
|
|
117
|
+
return undefined;
|
|
118
|
+
if (mcfg.check) {
|
|
119
|
+
const r = mcfg.check(lex);
|
|
120
|
+
if (r && r.done)
|
|
121
|
+
return r.token;
|
|
122
|
+
}
|
|
123
|
+
return body(lex, rule, tI);
|
|
124
|
+
};
|
|
125
|
+
}
|
|
126
|
+
// ---------------------------------------------------------------------------
|
|
127
|
+
// Declarative single-character state machine driver.
|
|
128
|
+
//
|
|
129
|
+
// The simpler matchers (space, line, comment-eatline tails) all
|
|
130
|
+
// have the shape "walk bytes, dispatch on (state, char-class), emit
|
|
131
|
+
// position-tracking actions, stop when told". The driver below
|
|
132
|
+
// centralises that shape.
|
|
133
|
+
//
|
|
134
|
+
// Each spec declares:
|
|
135
|
+
// - `initialState` which state the walk starts in
|
|
136
|
+
// - `nclasses` how many byte-classes the spec uses
|
|
137
|
+
// - `classOf` (Uint8Array) per-byte class index (ASCII fast path)
|
|
138
|
+
// - `fallback` class for non-ASCII bytes
|
|
139
|
+
// - `table` (Int32Array) action keyed on `state * nclasses + class`
|
|
140
|
+
//
|
|
141
|
+
// An action is a packed Int32 — `STATE_MASK` bits hold the next
|
|
142
|
+
// state, plus three single-bit flags below. The driver applies
|
|
143
|
+
// CONSUME / IS_ROW first, then transitions, then STOP. That ordering
|
|
144
|
+
// makes "consume the char that ends the match" express as
|
|
145
|
+
// `CONSUME | STOP`, while "stop without consuming" is just `STOP`.
|
|
146
|
+
//
|
|
147
|
+
// Performance-wise the loop is uniform: one Uint8Array index, one
|
|
148
|
+
// Int32Array index, three bit tests, no function calls per byte.
|
|
149
|
+
// ---------------------------------------------------------------------------
|
|
150
|
+
const CONSUME = 1 << 16;
|
|
151
|
+
exports.CONSUME = CONSUME;
|
|
152
|
+
const IS_ROW = 1 << 17;
|
|
153
|
+
exports.IS_ROW = IS_ROW;
|
|
154
|
+
const CI_RESET = 1 << 18; // cI = 1 without rI++ (line chars in multi-line strings)
|
|
155
|
+
exports.CI_RESET = CI_RESET;
|
|
156
|
+
const STOP = 1 << 19;
|
|
157
|
+
exports.STOP = STOP;
|
|
158
|
+
const STATE_MASK = 0xffff;
|
|
159
|
+
exports.STATE_MASK = STATE_MASK;
|
|
160
|
+
// Walk `src` from `(startSI, startRI, startCI)` according to `spec`.
|
|
161
|
+
// Position fields are written into `out` (a caller-owned scratch
|
|
162
|
+
// object — no allocation per call). Returns true if any char was
|
|
163
|
+
// consumed.
|
|
164
|
+
//
|
|
165
|
+
// Takes raw position numbers rather than a Point because some
|
|
166
|
+
// callers (notably the comment matcher) track positions as locals
|
|
167
|
+
// against a sliced `fwd` string rather than on the lex's pnt.
|
|
168
|
+
function scan(src, startSI, startRI, startCI, spec, out) {
|
|
169
|
+
let sI = startSI;
|
|
170
|
+
let rI = startRI;
|
|
171
|
+
let cI = startCI;
|
|
172
|
+
const len = src.length;
|
|
173
|
+
const ncls = spec.nclasses;
|
|
174
|
+
const classOf = spec.classOf;
|
|
175
|
+
const table = spec.table;
|
|
176
|
+
let state = spec.initialState;
|
|
177
|
+
while (sI < len) {
|
|
178
|
+
const cc = src.charCodeAt(sI);
|
|
179
|
+
const cls = cc < 256 ? classOf[cc] : spec.fallback(src[sI]);
|
|
180
|
+
const action = table[state * ncls + cls];
|
|
181
|
+
if (action & CONSUME) {
|
|
182
|
+
sI++;
|
|
183
|
+
if (action & IS_ROW) {
|
|
184
|
+
rI++;
|
|
185
|
+
cI = 1;
|
|
186
|
+
}
|
|
187
|
+
else if (action & CI_RESET) {
|
|
188
|
+
cI = 1;
|
|
189
|
+
}
|
|
190
|
+
else {
|
|
191
|
+
cI++;
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
state = action & STATE_MASK;
|
|
195
|
+
if (action & STOP)
|
|
196
|
+
break;
|
|
197
|
+
}
|
|
198
|
+
out.sI = sI;
|
|
199
|
+
out.rI = rI;
|
|
200
|
+
out.cI = cI;
|
|
201
|
+
return startSI < sI;
|
|
202
|
+
}
|
|
203
|
+
// Build a 3-class line-run spec from cfg.line. Class 0 = not a line
|
|
204
|
+
// char, class 1 = line char, class 2 = line char that also advances
|
|
205
|
+
// the row counter. Used by the line matcher (when not in `single`
|
|
206
|
+
// mode) and by the comment matcher's `eatline` tails.
|
|
207
|
+
function buildLineRunSpec(cfgLine) {
|
|
208
|
+
const classOf = new Uint8Array(256);
|
|
209
|
+
for (let cc = 0; cc < 256; cc++) {
|
|
210
|
+
if (cfgLine.charsBitmap[cc]) {
|
|
211
|
+
classOf[cc] = cfgLine.rowCharsBitmap[cc] ? 2 : 1;
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
const lineChars = cfgLine.chars;
|
|
215
|
+
const rowChars = cfgLine.rowChars;
|
|
216
|
+
const fallback = (c) => {
|
|
217
|
+
if (lineChars[c])
|
|
218
|
+
return rowChars[c] ? 2 : 1;
|
|
219
|
+
return 0;
|
|
220
|
+
};
|
|
221
|
+
return {
|
|
222
|
+
initialState: 0,
|
|
223
|
+
nclasses: 3,
|
|
224
|
+
classOf,
|
|
225
|
+
fallback,
|
|
226
|
+
table: LINE_RUN_TABLE,
|
|
227
|
+
};
|
|
228
|
+
}
|
|
229
|
+
// (state=0, class=NOT_LINE) -> stop
|
|
230
|
+
// (state=0, class=LINE) -> consume, stay in 0
|
|
231
|
+
// (state=0, class=LINE+ROW) -> consume + row, stay in 0
|
|
232
|
+
const LINE_RUN_TABLE = new Int32Array([
|
|
233
|
+
STOP,
|
|
234
|
+
CONSUME,
|
|
235
|
+
CONSUME | IS_ROW,
|
|
236
|
+
]);
|
|
237
|
+
// Build a 2-class run spec from a chars / charsBitmap pair. Class 0
|
|
238
|
+
// = not in set, class 1 = in set. Used by the space matcher.
|
|
239
|
+
function buildCharRunSpec(charsBitmap, chars) {
|
|
240
|
+
const fallback = (c) => (chars[c] ? 1 : 0);
|
|
241
|
+
return {
|
|
242
|
+
initialState: 0,
|
|
243
|
+
nclasses: 2,
|
|
244
|
+
classOf: charsBitmap, // already 0/1
|
|
245
|
+
fallback,
|
|
246
|
+
table: CHAR_RUN_TABLE,
|
|
247
|
+
};
|
|
248
|
+
}
|
|
249
|
+
// (state=0, class=OUT) -> stop
|
|
250
|
+
// (state=0, class=IN) -> consume col, stay in 0
|
|
251
|
+
const CHAR_RUN_TABLE = new Int32Array([
|
|
252
|
+
STOP,
|
|
253
|
+
CONSUME,
|
|
254
|
+
]);
|
|
255
|
+
// Build a string-body scan spec for one quote character. Class 0 =
|
|
256
|
+
// BODY (consume, advance col); class 1 = STOP (caller decides what
|
|
257
|
+
// to do); class 2 = LINE (multi-line strings only — consume, reset
|
|
258
|
+
// col); class 3 = LINE+ROW (multi-line — consume, reset col,
|
|
259
|
+
// advance row). The opening / closing quote, the escape char, the
|
|
260
|
+
// replace chars and any control char that can't be consumed in
|
|
261
|
+
// the current quote context all map to class 1.
|
|
262
|
+
//
|
|
263
|
+
// One spec per quote char because the quote char is encoded in the
|
|
264
|
+
// class table. For a typical config (1-3 quote chars) this is
|
|
265
|
+
// cheap; the matcher caches them per make.
|
|
266
|
+
function buildStringBodySpec(cfg, qchar) {
|
|
267
|
+
const qcc = qchar.charCodeAt(0);
|
|
268
|
+
const escCharCode = cfg.string.escCharCode;
|
|
269
|
+
const replaceCodeMap = cfg.string.replaceCodeMap;
|
|
270
|
+
const hasReplace = cfg.string.hasReplace;
|
|
271
|
+
const isMultiLine = !!cfg.string.multiBitmap[qcc];
|
|
272
|
+
const lineBM = cfg.line.charsBitmap;
|
|
273
|
+
const rowBM = cfg.line.rowCharsBitmap;
|
|
274
|
+
const classOf = new Uint8Array(256);
|
|
275
|
+
for (let cc = 0; cc < 256; cc++) {
|
|
276
|
+
if (cc === qcc) {
|
|
277
|
+
classOf[cc] = 1;
|
|
278
|
+
}
|
|
279
|
+
else if (cc === escCharCode) {
|
|
280
|
+
classOf[cc] = 1;
|
|
281
|
+
}
|
|
282
|
+
else if (hasReplace && replaceCodeMap[cc] !== undefined) {
|
|
283
|
+
classOf[cc] = 1;
|
|
284
|
+
}
|
|
285
|
+
else if (cc < 32) {
|
|
286
|
+
if (isMultiLine && lineBM[cc]) {
|
|
287
|
+
classOf[cc] = rowBM[cc] ? 3 : 2;
|
|
288
|
+
}
|
|
289
|
+
else {
|
|
290
|
+
classOf[cc] = 1;
|
|
291
|
+
}
|
|
292
|
+
}
|
|
293
|
+
// else BODY (class 0)
|
|
294
|
+
}
|
|
295
|
+
// Char codes >= 256 classify like the table: the quote itself, the
|
|
296
|
+
// escape char, replace chars, and (multi-line) line chars are special;
|
|
297
|
+
// everything else is plain body. Without this, non-Latin-1 quote chars
|
|
298
|
+
// could open a string but never close it.
|
|
299
|
+
const lineChars = cfg.line.chars;
|
|
300
|
+
const rowChars = cfg.line.rowChars;
|
|
301
|
+
const fallback = (c) => {
|
|
302
|
+
const cc = c.charCodeAt(0);
|
|
303
|
+
if (c === qchar)
|
|
304
|
+
return 1;
|
|
305
|
+
if (cc === escCharCode)
|
|
306
|
+
return 1;
|
|
307
|
+
if (hasReplace && replaceCodeMap[cc] !== undefined)
|
|
308
|
+
return 1;
|
|
309
|
+
if (isMultiLine && lineChars[c])
|
|
310
|
+
return rowChars[c] ? 3 : 2;
|
|
311
|
+
return 0;
|
|
312
|
+
};
|
|
313
|
+
return {
|
|
314
|
+
initialState: 0,
|
|
315
|
+
nclasses: 4,
|
|
316
|
+
classOf,
|
|
317
|
+
fallback,
|
|
318
|
+
table: STRING_BODY_TABLE,
|
|
319
|
+
};
|
|
320
|
+
}
|
|
321
|
+
// (s=0, BODY) -> consume + col
|
|
322
|
+
// (s=0, STOP) -> stop, caller dispatches on src[sI]
|
|
323
|
+
// (s=0, LINE_NONROW) -> consume + cI=1 (multi-line)
|
|
324
|
+
// (s=0, LINE_ROW) -> consume + rI++; cI=1 (multi-line)
|
|
325
|
+
const STRING_BODY_TABLE = new Int32Array([
|
|
326
|
+
CONSUME,
|
|
327
|
+
STOP,
|
|
328
|
+
CONSUME | CI_RESET,
|
|
329
|
+
CONSUME | IS_ROW,
|
|
330
|
+
]);
|
|
331
|
+
let makeFixedMatcher = (cfg, _opts) => {
|
|
332
|
+
let fixed = (0, utility_1.regexp)(null, '^(', cfg.rePart.fixed, ')');
|
|
333
|
+
return guardedMatcher(cfg.fixed, function fixedBody(lex) {
|
|
334
|
+
const mcfg = cfg.fixed;
|
|
335
|
+
let pnt = lex.pnt;
|
|
336
|
+
let fwd = lex.fwd;
|
|
337
|
+
let m = fwd.match(fixed);
|
|
338
|
+
if (m) {
|
|
339
|
+
let msrc = m[1];
|
|
340
|
+
let mlen = msrc.length;
|
|
341
|
+
if (0 < mlen) {
|
|
342
|
+
let tkn = undefined;
|
|
343
|
+
let tin = mcfg.token[msrc];
|
|
344
|
+
if (null != tin) {
|
|
345
|
+
tkn = lex.token(tin, undefined, msrc, pnt);
|
|
346
|
+
pnt.sI += mlen;
|
|
347
|
+
pnt.cI += mlen;
|
|
348
|
+
}
|
|
349
|
+
return tkn;
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
});
|
|
353
|
+
};
|
|
354
|
+
exports.makeFixedMatcher = makeFixedMatcher;
|
|
355
|
+
let makeMatchMatcher = (cfg, _opts) => {
|
|
356
|
+
// Pre-sort both matcher lists at configure time so lexing iterates in
|
|
357
|
+
// a deterministic order regardless of how the config object was built.
|
|
358
|
+
// Value matchers: sort by user-supplied name (ascending).
|
|
359
|
+
// Token matchers: sort by attached tin$ (ascending), set in utility.ts.
|
|
360
|
+
let valueMatchers = (0, utility_1.entries)(cfg.match.value)
|
|
361
|
+
.sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0))
|
|
362
|
+
.map(([, spec]) => spec);
|
|
363
|
+
let tokenMatchers = (0, utility_1.values)(cfg.match.token).sort((a, b) => (a.tin$ || 0) - (b.tin$ || 0));
|
|
364
|
+
// Don't add a matcher if there's nothing to do.
|
|
365
|
+
if (0 === valueMatchers.length && 0 === tokenMatchers.length) {
|
|
366
|
+
return null;
|
|
367
|
+
}
|
|
368
|
+
return guardedMatcher(cfg.match, function matchBody(lex, rule, tI = 0) {
|
|
369
|
+
let pnt = lex.pnt;
|
|
370
|
+
let fwd = lex.fwd;
|
|
371
|
+
let oc = 'o' === rule.state ? 0 : 1;
|
|
372
|
+
for (let valueMatcher of valueMatchers) {
|
|
373
|
+
if (valueMatcher.match instanceof RegExp) {
|
|
374
|
+
// TODO: only match VL if present in rule
|
|
375
|
+
let m = fwd.match(valueMatcher.match);
|
|
376
|
+
if (m) {
|
|
377
|
+
let msrc = m[0];
|
|
378
|
+
let mlen = msrc.length;
|
|
379
|
+
if (0 < mlen) {
|
|
380
|
+
let tkn = undefined;
|
|
381
|
+
let val = valueMatcher.val ? valueMatcher.val(m) : msrc;
|
|
382
|
+
tkn = lex.token('#VL', val, msrc, pnt);
|
|
383
|
+
pnt.sI += mlen;
|
|
384
|
+
pnt.cI += mlen;
|
|
385
|
+
return tkn;
|
|
386
|
+
}
|
|
387
|
+
}
|
|
388
|
+
}
|
|
389
|
+
else {
|
|
390
|
+
let tkn = valueMatcher.match(lex, rule);
|
|
391
|
+
if (null != tkn) {
|
|
392
|
+
return tkn;
|
|
393
|
+
}
|
|
394
|
+
}
|
|
395
|
+
}
|
|
396
|
+
for (let tokenMatcher of tokenMatchers) {
|
|
397
|
+
// Only match Token if present in Rule sequence.
|
|
398
|
+
// Exception: an `eager$` flag on the matcher opts out of
|
|
399
|
+
// tcol gating — the matcher fires whenever its regex matches
|
|
400
|
+
// and the downstream parser rejects tokens it doesn't expect
|
|
401
|
+
// at the current position. This is what ABNF's
|
|
402
|
+
// case-insensitive literals need: the lexer has to emit the
|
|
403
|
+
// literal's own tin even when the current rule's tcol is
|
|
404
|
+
// narrower, so the next rule up the stack can see the token
|
|
405
|
+
// as its proper type rather than falling through to #TX.
|
|
406
|
+
if (tokenMatcher.tin$ &&
|
|
407
|
+
!tokenMatcher.eager$ &&
|
|
408
|
+
!rule.spec.def.tcol[oc][tI].includes(tokenMatcher.tin$)) {
|
|
409
|
+
continue;
|
|
410
|
+
}
|
|
411
|
+
if (tokenMatcher instanceof RegExp) {
|
|
412
|
+
let m = fwd.match(tokenMatcher);
|
|
413
|
+
if (m) {
|
|
414
|
+
let msrc = m[0];
|
|
415
|
+
let mlen = msrc.length;
|
|
416
|
+
if (0 < mlen) {
|
|
417
|
+
let tkn = undefined;
|
|
418
|
+
let tin = tokenMatcher.tin$;
|
|
419
|
+
tkn = lex.token(tin, msrc, msrc, pnt);
|
|
420
|
+
pnt.sI += mlen;
|
|
421
|
+
pnt.cI += mlen;
|
|
422
|
+
return tkn;
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
else {
|
|
427
|
+
let tkn = tokenMatcher(lex, rule);
|
|
428
|
+
if (null != tkn) {
|
|
429
|
+
return tkn;
|
|
430
|
+
}
|
|
431
|
+
}
|
|
432
|
+
}
|
|
433
|
+
});
|
|
434
|
+
};
|
|
435
|
+
exports.makeMatchMatcher = makeMatchMatcher;
|
|
436
|
+
let makeCommentMatcher = (cfg, opts) => {
|
|
437
|
+
let oc = opts.comment;
|
|
438
|
+
cfg.comment = {
|
|
439
|
+
lex: oc ? !!oc.lex : false,
|
|
440
|
+
def: (oc?.def ? (0, utility_1.entries)(oc.def) : []).reduce((def, [name, om]) => {
|
|
441
|
+
// Set comment marker to null to remove
|
|
442
|
+
if (null == om || false === om) {
|
|
443
|
+
return def;
|
|
444
|
+
}
|
|
445
|
+
let { suffixes, suffixFn } = normalizeCommentSuffix(om.suffix);
|
|
446
|
+
let cm = {
|
|
447
|
+
name,
|
|
448
|
+
start: om.start,
|
|
449
|
+
end: om.end,
|
|
450
|
+
line: !!om.line,
|
|
451
|
+
lex: !!om.lex,
|
|
452
|
+
eatline: !!om.eatline,
|
|
453
|
+
suffixes,
|
|
454
|
+
suffixFn,
|
|
455
|
+
};
|
|
456
|
+
def[name] = cm;
|
|
457
|
+
return def;
|
|
458
|
+
}, {}),
|
|
459
|
+
};
|
|
460
|
+
// Pre-sort by start length (longest first) so that a longer marker
|
|
461
|
+
// shadows any shorter marker it contains (e.g. '##' wins over '#'),
|
|
462
|
+
// regardless of the insertion order of cfg.comment.def. Ties break by
|
|
463
|
+
// name for deterministic iteration across runtimes.
|
|
464
|
+
let byStartLenDesc = (a, b) => b.start.length - a.start.length ||
|
|
465
|
+
(a.name < b.name ? -1 : a.name > b.name ? 1 : 0);
|
|
466
|
+
let lineComments = cfg.comment.lex
|
|
467
|
+
? (0, utility_1.values)(cfg.comment.def).filter((c) => c.lex && c.line).sort(byStartLenDesc)
|
|
468
|
+
: [];
|
|
469
|
+
let blockComments = cfg.comment.lex
|
|
470
|
+
? (0, utility_1.values)(cfg.comment.def).filter((c) => c.lex && !c.line).sort(byStartLenDesc)
|
|
471
|
+
: [];
|
|
472
|
+
// Eatline tail: a 3-class line-run state machine, same shape as
|
|
473
|
+
// the line matcher's no-single path. Built once per matcher build
|
|
474
|
+
// so the table and class array are reused across comment matches.
|
|
475
|
+
const lineRunSpec = buildLineRunSpec(cfg.line);
|
|
476
|
+
const scanOut = { sI: 0, rI: 0, cI: 0 };
|
|
477
|
+
return guardedMatcher(cfg.comment, function commentBody(lex) {
|
|
478
|
+
let pnt = lex.pnt;
|
|
479
|
+
let fwd = lex.fwd;
|
|
480
|
+
let rI = pnt.rI;
|
|
481
|
+
let cI = pnt.cI;
|
|
482
|
+
// Single line comment.
|
|
483
|
+
const lineBM = cfg.line.charsBitmap;
|
|
484
|
+
const rowBM = cfg.line.rowCharsBitmap;
|
|
485
|
+
const lineChars = cfg.line.chars;
|
|
486
|
+
const rowChars = cfg.line.rowChars;
|
|
487
|
+
for (let mc of lineComments) {
|
|
488
|
+
if (fwd.startsWith(mc.start)) {
|
|
489
|
+
let fwdlen = fwd.length;
|
|
490
|
+
let fI = mc.start.length;
|
|
491
|
+
cI += mc.start.length;
|
|
492
|
+
let suffixLen = 0;
|
|
493
|
+
let cc;
|
|
494
|
+
while (fI < fwdlen &&
|
|
495
|
+
!((cc = fwd.charCodeAt(fI)) < 256
|
|
496
|
+
? lineBM[cc]
|
|
497
|
+
: lineChars[fwd[fI]])) {
|
|
498
|
+
let n = commentSuffixMatch(fwd, fI, mc.suffixes);
|
|
499
|
+
if (n > 0) {
|
|
500
|
+
suffixLen = n;
|
|
501
|
+
break;
|
|
502
|
+
}
|
|
503
|
+
n = commentSuffixFnMatch(lex, fI, mc.suffixFn);
|
|
504
|
+
if (n > 0) {
|
|
505
|
+
suffixLen = n;
|
|
506
|
+
break;
|
|
507
|
+
}
|
|
508
|
+
cI++;
|
|
509
|
+
fI++;
|
|
510
|
+
}
|
|
511
|
+
if (suffixLen > 0) {
|
|
512
|
+
// Consume the suffix as the tail of the comment body.
|
|
513
|
+
fI += suffixLen;
|
|
514
|
+
cI += suffixLen;
|
|
515
|
+
}
|
|
516
|
+
else if (mc.eatline) {
|
|
517
|
+
// Only absorb trailing line chars when termination came from
|
|
518
|
+
// a line char (not from a suffix match). cI is intentionally
|
|
519
|
+
// NOT pulled from scanOut — current semantics leave the
|
|
520
|
+
// pnt.cI at end-of-comment-body even after eating newlines.
|
|
521
|
+
scan(fwd, fI, rI, cI, lineRunSpec, scanOut);
|
|
522
|
+
rI = scanOut.rI;
|
|
523
|
+
fI = scanOut.sI;
|
|
524
|
+
}
|
|
525
|
+
let csrc = fwd.substring(0, fI);
|
|
526
|
+
let tkn = lex.token('#CM', undefined, csrc, pnt);
|
|
527
|
+
pnt.sI += csrc.length;
|
|
528
|
+
pnt.cI = cI;
|
|
529
|
+
pnt.rI = rI;
|
|
530
|
+
return tkn;
|
|
531
|
+
}
|
|
532
|
+
}
|
|
533
|
+
// Multiline comment.
|
|
534
|
+
for (let mc of blockComments) {
|
|
535
|
+
if (fwd.startsWith(mc.start)) {
|
|
536
|
+
let fwdlen = fwd.length;
|
|
537
|
+
let fI = mc.start.length;
|
|
538
|
+
let end = mc.end;
|
|
539
|
+
cI += mc.start.length;
|
|
540
|
+
let suffixLen = 0;
|
|
541
|
+
let cc;
|
|
542
|
+
while (fI < fwdlen && !fwd.startsWith(end, fI)) {
|
|
543
|
+
let n = commentSuffixMatch(fwd, fI, mc.suffixes);
|
|
544
|
+
if (n > 0) {
|
|
545
|
+
suffixLen = n;
|
|
546
|
+
break;
|
|
547
|
+
}
|
|
548
|
+
n = commentSuffixFnMatch(lex, fI, mc.suffixFn);
|
|
549
|
+
if (n > 0) {
|
|
550
|
+
suffixLen = n;
|
|
551
|
+
break;
|
|
552
|
+
}
|
|
553
|
+
cc = fwd.charCodeAt(fI);
|
|
554
|
+
if (cc < 256 ? rowBM[cc] : rowChars[fwd[fI]]) {
|
|
555
|
+
rI++;
|
|
556
|
+
cI = 0;
|
|
557
|
+
}
|
|
558
|
+
cI++;
|
|
559
|
+
fI++;
|
|
560
|
+
}
|
|
561
|
+
if (suffixLen > 0) {
|
|
562
|
+
// Advance through the consumed suffix, tracking newlines.
|
|
563
|
+
for (let k = 0; k < suffixLen; k++) {
|
|
564
|
+
cc = fwd.charCodeAt(fI + k);
|
|
565
|
+
if (cc < 256 ? rowBM[cc] : rowChars[fwd[fI + k]]) {
|
|
566
|
+
rI++;
|
|
567
|
+
cI = 0;
|
|
568
|
+
}
|
|
569
|
+
cI++;
|
|
570
|
+
}
|
|
571
|
+
let csrc = fwd.substring(0, fI + suffixLen);
|
|
572
|
+
let tkn = lex.token('#CM', undefined, csrc, pnt);
|
|
573
|
+
pnt.sI += csrc.length;
|
|
574
|
+
pnt.rI = rI;
|
|
575
|
+
pnt.cI = cI;
|
|
576
|
+
return tkn;
|
|
577
|
+
}
|
|
578
|
+
if (fwd.startsWith(end, fI)) {
|
|
579
|
+
cI += end.length;
|
|
580
|
+
if (mc.eatline) {
|
|
581
|
+
scan(fwd, fI, rI, cI, lineRunSpec, scanOut);
|
|
582
|
+
rI = scanOut.rI;
|
|
583
|
+
fI = scanOut.sI;
|
|
584
|
+
}
|
|
585
|
+
let csrc = fwd.substring(0, fI + end.length);
|
|
586
|
+
let tkn = lex.token('#CM', undefined, csrc, pnt);
|
|
587
|
+
pnt.sI += csrc.length;
|
|
588
|
+
pnt.rI = rI;
|
|
589
|
+
pnt.cI = cI;
|
|
590
|
+
return tkn;
|
|
591
|
+
}
|
|
592
|
+
else {
|
|
593
|
+
return lex.bad(utility_1.S.unterminated_comment, pnt.sI, pnt.sI + 9 * mc.start.length);
|
|
594
|
+
}
|
|
595
|
+
}
|
|
596
|
+
}
|
|
597
|
+
});
|
|
598
|
+
};
|
|
599
|
+
exports.makeCommentMatcher = makeCommentMatcher;
|
|
600
|
+
// normalizeCommentSuffix splits the polymorphic suffix option value
|
|
601
|
+
// (string | string[] | LexMatcher) into a length-sorted string list and
|
|
602
|
+
// an optional LexMatcher probe. Empty/absent input yields empty outputs.
|
|
603
|
+
function normalizeCommentSuffix(raw) {
|
|
604
|
+
if (null == raw) {
|
|
605
|
+
return { suffixes: undefined, suffixFn: undefined };
|
|
606
|
+
}
|
|
607
|
+
if ('function' === typeof raw) {
|
|
608
|
+
return { suffixes: undefined, suffixFn: raw };
|
|
609
|
+
}
|
|
610
|
+
let list = [];
|
|
611
|
+
if ('string' === typeof raw) {
|
|
612
|
+
if ('' !== raw)
|
|
613
|
+
list.push(raw);
|
|
614
|
+
}
|
|
615
|
+
else if (Array.isArray(raw)) {
|
|
616
|
+
for (let s of raw) {
|
|
617
|
+
if ('string' === typeof s && '' !== s)
|
|
618
|
+
list.push(s);
|
|
619
|
+
}
|
|
620
|
+
}
|
|
621
|
+
if (list.length > 1) {
|
|
622
|
+
list.sort((a, b) => b.length - a.length || (a < b ? -1 : a > b ? 1 : 0));
|
|
623
|
+
}
|
|
624
|
+
return {
|
|
625
|
+
suffixes: 0 === list.length ? undefined : list,
|
|
626
|
+
suffixFn: undefined,
|
|
627
|
+
};
|
|
628
|
+
}
|
|
629
|
+
// commentSuffixMatch returns the length of the best suffix match at
|
|
630
|
+
// fwd[fI:] or 0 if none matches. Suffixes are pre-sorted longest-first,
|
|
631
|
+
// so the first match is the best match.
|
|
632
|
+
function commentSuffixMatch(fwd, fI, suffixes) {
|
|
633
|
+
if (!suffixes || 0 === suffixes.length)
|
|
634
|
+
return 0;
|
|
635
|
+
for (let s of suffixes) {
|
|
636
|
+
if (fwd.substring(fI, fI + s.length) === s)
|
|
637
|
+
return s.length;
|
|
638
|
+
}
|
|
639
|
+
return 0;
|
|
640
|
+
}
|
|
641
|
+
// commentSuffixFnMatch probes the LexMatcher-form suffix terminator at
|
|
642
|
+
// offset fI. Returns the length of the returned token's src (to be
|
|
643
|
+
// consumed) or 0 if no termination. The lex point is snapshotted and
|
|
644
|
+
// restored so a misbehaving matcher can't advance the stream itself.
|
|
645
|
+
function commentSuffixFnMatch(lex, fI, fn) {
|
|
646
|
+
if (!fn)
|
|
647
|
+
return 0;
|
|
648
|
+
let pnt = lex.pnt;
|
|
649
|
+
let savedSI = pnt.sI;
|
|
650
|
+
let savedRI = pnt.rI;
|
|
651
|
+
let savedCI = pnt.cI;
|
|
652
|
+
pnt.sI = savedSI + fI;
|
|
653
|
+
let tkn;
|
|
654
|
+
try {
|
|
655
|
+
tkn = fn(lex, undefined);
|
|
656
|
+
}
|
|
657
|
+
finally {
|
|
658
|
+
pnt.sI = savedSI;
|
|
659
|
+
pnt.rI = savedRI;
|
|
660
|
+
pnt.cI = savedCI;
|
|
661
|
+
}
|
|
662
|
+
if (null == tkn)
|
|
663
|
+
return 0;
|
|
664
|
+
return ('string' === typeof tkn.src) ? tkn.src.length : 0;
|
|
665
|
+
}
|
|
666
|
+
// Match text, checking for literal values, optionally followed by a fixed token.
|
|
667
|
+
// Text strings are terminated by end markers.
|
|
668
|
+
let makeTextMatcher = (cfg, opts) => {
|
|
669
|
+
let ender = (0, utility_1.regexp)(cfg.line.lex ? null : 's', '^(.*?)', ...cfg.rePart.ender);
|
|
670
|
+
return function textMatcher(lex) {
|
|
671
|
+
if (cfg.text.check) {
|
|
672
|
+
let check = cfg.text.check(lex);
|
|
673
|
+
if (check && check.done) {
|
|
674
|
+
return check.token;
|
|
675
|
+
}
|
|
676
|
+
}
|
|
677
|
+
let mcfg = cfg.text;
|
|
678
|
+
let pnt = lex.pnt;
|
|
679
|
+
let fwd = lex.fwd;
|
|
680
|
+
let def = cfg.value.def;
|
|
681
|
+
let defre = cfg.value.defre;
|
|
682
|
+
let m = fwd.match(ender);
|
|
683
|
+
if (m) {
|
|
684
|
+
let msrc = m[1];
|
|
685
|
+
let tsrc = m[2];
|
|
686
|
+
let out = undefined;
|
|
687
|
+
if (null != msrc) {
|
|
688
|
+
let mlen = msrc.length;
|
|
689
|
+
if (0 < mlen) {
|
|
690
|
+
// Check for values first.
|
|
691
|
+
let vs = undefined;
|
|
692
|
+
if (cfg.value.lex) {
|
|
693
|
+
// Fixed values (e.g true, false, null).
|
|
694
|
+
if (undefined !== (vs = def[msrc])) {
|
|
695
|
+
out = lex.token('#VL', vs.val, msrc, pnt);
|
|
696
|
+
pnt.sI += mlen;
|
|
697
|
+
pnt.cI += mlen;
|
|
698
|
+
}
|
|
699
|
+
// Regexp processed values.
|
|
700
|
+
else {
|
|
701
|
+
// defre is a name-sorted array (see cfg.value.defre build in
|
|
702
|
+
// utility.ts) — iteration order is deterministic.
|
|
703
|
+
for (let vspec of defre) {
|
|
704
|
+
if (vspec.match) {
|
|
705
|
+
// If consume, assume regexp starts with ^.
|
|
706
|
+
let res = vspec.match.exec(vspec.consume ? fwd : msrc);
|
|
707
|
+
// Must match entire text.
|
|
708
|
+
if (res && (vspec.consume || res[0].length === msrc.length)) {
|
|
709
|
+
let remsrc = res[0];
|
|
710
|
+
if (null == vspec.val) {
|
|
711
|
+
out = lex.token('#VL', remsrc, remsrc, pnt);
|
|
712
|
+
}
|
|
713
|
+
else {
|
|
714
|
+
let val = vspec.val(res);
|
|
715
|
+
out = lex.token('#VL', val, remsrc, pnt);
|
|
716
|
+
}
|
|
717
|
+
pnt.sI += remsrc.length;
|
|
718
|
+
pnt.cI += remsrc.length;
|
|
719
|
+
}
|
|
720
|
+
}
|
|
721
|
+
}
|
|
722
|
+
}
|
|
723
|
+
}
|
|
724
|
+
// Not a value, so plain text.
|
|
725
|
+
// NOTEL if !text.lex then only values are matched.
|
|
726
|
+
if (null == out && mcfg.lex) {
|
|
727
|
+
out = lex.token('#TX', msrc, msrc, pnt);
|
|
728
|
+
pnt.sI += mlen;
|
|
729
|
+
pnt.cI += mlen;
|
|
730
|
+
}
|
|
731
|
+
}
|
|
732
|
+
}
|
|
733
|
+
// A following fixed token can only match if there was already a
|
|
734
|
+
// valid text or value match.
|
|
735
|
+
if (out) {
|
|
736
|
+
out = subMatchFixed(lex, out, tsrc);
|
|
737
|
+
}
|
|
738
|
+
if (out && 0 < cfg.text.modify.length) {
|
|
739
|
+
const modify = cfg.text.modify;
|
|
740
|
+
for (let mI = 0; mI < modify.length; mI++) {
|
|
741
|
+
out.val = modify[mI](out.val, lex, cfg, opts);
|
|
742
|
+
}
|
|
743
|
+
}
|
|
744
|
+
return out;
|
|
745
|
+
}
|
|
746
|
+
};
|
|
747
|
+
};
|
|
748
|
+
exports.makeTextMatcher = makeTextMatcher;
|
|
749
|
+
let makeNumberMatcher = (cfg, _opts) => {
|
|
750
|
+
let mcfg = cfg.number;
|
|
751
|
+
let ender = (0, utility_1.regexp)(null, [
|
|
752
|
+
'^([-+]?(0(',
|
|
753
|
+
[
|
|
754
|
+
mcfg.hex ? 'x[0-9a-fA-F_]+' : null,
|
|
755
|
+
mcfg.oct ? 'o[0-7_]+' : null,
|
|
756
|
+
mcfg.bin ? 'b[01_]+' : null,
|
|
757
|
+
]
|
|
758
|
+
.filter((s) => null != s)
|
|
759
|
+
.join('|'),
|
|
760
|
+
// ')|[.0-9]+([0-9_]*[0-9])?)',
|
|
761
|
+
')|\\.?[0-9]+([0-9_]*[0-9])?)',
|
|
762
|
+
'(\\.[0-9]?([0-9_]*[0-9])?)?',
|
|
763
|
+
'([eE][-+]?[0-9]+([0-9_]*[0-9])?)?',
|
|
764
|
+
]
|
|
765
|
+
.join('')
|
|
766
|
+
.replace(/_/g, mcfg.sep ? (0, utility_1.escre)(mcfg.sepChar) : ''), ')', ...cfg.rePart.ender);
|
|
767
|
+
let numberSep = mcfg.sep
|
|
768
|
+
? (0, utility_1.regexp)('g', (0, utility_1.escre)(mcfg.sepChar))
|
|
769
|
+
: undefined;
|
|
770
|
+
return guardedMatcher(cfg.number, function numberBody(lex) {
|
|
771
|
+
mcfg = cfg.number;
|
|
772
|
+
let pnt = lex.pnt;
|
|
773
|
+
let fwd = lex.fwd;
|
|
774
|
+
let valdef = cfg.value.def;
|
|
775
|
+
let m = fwd.match(ender);
|
|
776
|
+
if (m) {
|
|
777
|
+
let msrc = m[1];
|
|
778
|
+
let tsrc = m[9]; // NOTE: count parens in numberEnder!
|
|
779
|
+
let out = undefined;
|
|
780
|
+
let included = true;
|
|
781
|
+
if (null != msrc &&
|
|
782
|
+
(included = !cfg.number.exclude || !msrc.match(cfg.number.exclude))) {
|
|
783
|
+
let mlen = msrc.length;
|
|
784
|
+
if (0 < mlen) {
|
|
785
|
+
let vs = undefined;
|
|
786
|
+
if (cfg.value.lex && undefined !== (vs = valdef[msrc])) {
|
|
787
|
+
out = lex.token('#VL', vs.val, msrc, pnt);
|
|
788
|
+
}
|
|
789
|
+
else {
|
|
790
|
+
let nstr = numberSep ? msrc.replace(numberSep, '') : msrc;
|
|
791
|
+
let num = +nstr;
|
|
792
|
+
// Special case: +- prefix of 0x... format
|
|
793
|
+
if (isNaN(num)) {
|
|
794
|
+
let first = nstr[0];
|
|
795
|
+
if ('-' === first || '+' === first) {
|
|
796
|
+
num = ('-' === first ? -1 : 1) * +nstr.substring(1);
|
|
797
|
+
}
|
|
798
|
+
}
|
|
799
|
+
if (!isNaN(num)) {
|
|
800
|
+
out = lex.token('#NR', num, msrc, pnt);
|
|
801
|
+
pnt.sI += mlen;
|
|
802
|
+
pnt.cI += mlen;
|
|
803
|
+
}
|
|
804
|
+
// Else let later matchers try.
|
|
805
|
+
}
|
|
806
|
+
}
|
|
807
|
+
}
|
|
808
|
+
if (included) {
|
|
809
|
+
out = subMatchFixed(lex, out, tsrc);
|
|
810
|
+
}
|
|
811
|
+
return out;
|
|
812
|
+
}
|
|
813
|
+
});
|
|
814
|
+
};
|
|
815
|
+
exports.makeNumberMatcher = makeNumberMatcher;
|
|
816
|
+
let makeStringMatcher = (cfg, opts) => {
|
|
817
|
+
// TODO: does `clean` make sense here?
|
|
818
|
+
let os = opts.string || {};
|
|
819
|
+
cfg.string = cfg.string || {};
|
|
820
|
+
// Replace map-shaped fields outright rather than deep-merging — when
|
|
821
|
+
// an instance reconfigures with stricter options (e.g. JSON's
|
|
822
|
+
// `chars: '"'`), the old quote/multi-char/escape entries from the
|
|
823
|
+
// permissive default would otherwise linger.
|
|
824
|
+
cfg.string.quoteMap = (0, utility_1.charset)(os.chars);
|
|
825
|
+
cfg.string.quoteBitmap = (0, utility_1.charsBitmap)(os.chars);
|
|
826
|
+
cfg.string.multiChars = (0, utility_1.charset)(os.multiChars);
|
|
827
|
+
cfg.string.multiBitmap = (0, utility_1.charsBitmap)(os.multiChars);
|
|
828
|
+
cfg.string.escMap = { ...os.escape };
|
|
829
|
+
cfg.string.replaceCodeMap = (0, utility_1.omap)((0, utility_1.clean)({ ...os.replace }), ([c, r]) => [c.charCodeAt(0), r]);
|
|
830
|
+
cfg.string = (0, utility_1.deep)(cfg.string, {
|
|
831
|
+
lex: !!os?.lex,
|
|
832
|
+
escChar: os.escapeChar,
|
|
833
|
+
escCharCode: null == os.escapeChar ? undefined : os.escapeChar.charCodeAt(0),
|
|
834
|
+
allowUnknown: !!os.allowUnknown,
|
|
835
|
+
// Strict escapes: disable the non-standard structural escapes \xHH
|
|
836
|
+
// and \u{...} (plain \uXXXX stays). Combined with escape-map removals
|
|
837
|
+
// and allowUnknown:false this yields JSON.parse-conformant handling.
|
|
838
|
+
escapeStrict: !!os.escapeStrict,
|
|
839
|
+
hasReplace: false,
|
|
840
|
+
abandon: !!os.abandon,
|
|
841
|
+
});
|
|
842
|
+
cfg.string.escMap = (0, utility_1.clean)(cfg.string.escMap);
|
|
843
|
+
// An escape mapped to '' (or null/undefined, removed by clean above)
|
|
844
|
+
// is treated as removed, so a built-in escape such as \v can be
|
|
845
|
+
// dropped via `string.escape: { v: '' }` — parity with the Go runtime.
|
|
846
|
+
for (const k of (0, utility_1.keys)(cfg.string.escMap)) {
|
|
847
|
+
if ('' === cfg.string.escMap[k])
|
|
848
|
+
delete cfg.string.escMap[k];
|
|
849
|
+
}
|
|
850
|
+
cfg.string.escBitmap = (0, utility_1.charsBitmap)(cfg.string.escMap);
|
|
851
|
+
cfg.string.hasReplace = 0 < (0, utility_1.keys)(cfg.string.replaceCodeMap).length;
|
|
852
|
+
// Pre-build a body-scan ScanSpec per quote character. The spec
|
|
853
|
+
// classifies every byte 0..255 against THAT quote's role table
|
|
854
|
+
// (the quote char, the escape char, replace chars, and control /
|
|
855
|
+
// line chars depending on whether the quote allows multiline).
|
|
856
|
+
// The body-scan loop below is then just a scan + dispatch.
|
|
857
|
+
const bodySpecs = new Map();
|
|
858
|
+
for (const qc of Object.keys(cfg.string.quoteMap)) {
|
|
859
|
+
bodySpecs.set(qc.charCodeAt(0), buildStringBodySpec(cfg, qc));
|
|
860
|
+
}
|
|
861
|
+
const scanOut = { sI: 0, rI: 0, cI: 0 };
|
|
862
|
+
return guardedMatcher(cfg.string, function stringBody(lex) {
|
|
863
|
+
const mcfg = cfg.string;
|
|
864
|
+
const { quoteMap, quoteBitmap, escMap, escCharCode, multiChars, multiBitmap, allowUnknown, replaceCodeMap, hasReplace, escapeStrict, } = mcfg;
|
|
865
|
+
const { pnt, src } = lex;
|
|
866
|
+
const startSI = pnt.sI;
|
|
867
|
+
const startRI = pnt.rI;
|
|
868
|
+
const srclen = src.length;
|
|
869
|
+
const qcc = src.charCodeAt(startSI);
|
|
870
|
+
// Is this byte an opening quote?
|
|
871
|
+
if (!(qcc < 256 ? quoteBitmap[qcc] : quoteMap[src[startSI]]))
|
|
872
|
+
return undefined;
|
|
873
|
+
const q = src[startSI];
|
|
874
|
+
const isMultiLine = qcc < 256 ? !!multiBitmap[qcc] : !!multiChars[q];
|
|
875
|
+
const bodySpec = bodySpecs.get(qcc) ||
|
|
876
|
+
(() => {
|
|
877
|
+
const spec = buildStringBodySpec(cfg, q);
|
|
878
|
+
bodySpecs.set(qcc, spec);
|
|
879
|
+
return spec;
|
|
880
|
+
})();
|
|
881
|
+
let sI = startSI + 1;
|
|
882
|
+
let rI = startRI;
|
|
883
|
+
let cI = pnt.cI + 1;
|
|
884
|
+
const buf = [];
|
|
885
|
+
while (sI < srclen) {
|
|
886
|
+
// Body scan: consume body chars (and multi-line newlines)
|
|
887
|
+
// until something interesting (quote, escape, control, replace).
|
|
888
|
+
const bodyStart = sI;
|
|
889
|
+
scan(src, sI, rI, cI, bodySpec, scanOut);
|
|
890
|
+
sI = scanOut.sI;
|
|
891
|
+
rI = scanOut.rI;
|
|
892
|
+
cI = scanOut.cI;
|
|
893
|
+
if (bodyStart < sI)
|
|
894
|
+
buf.push(src.substring(bodyStart, sI));
|
|
895
|
+
if (sI >= srclen)
|
|
896
|
+
break;
|
|
897
|
+
const cc = src.charCodeAt(sI);
|
|
898
|
+
// Closing quote — string done.
|
|
899
|
+
if (cc === qcc) {
|
|
900
|
+
sI++;
|
|
901
|
+
const tkn = lex.token('#ST', buf.join(types_1.EMPTY), src.substring(startSI, sI), pnt);
|
|
902
|
+
pnt.sI = sI;
|
|
903
|
+
pnt.rI = rI;
|
|
904
|
+
pnt.cI = cI + 1;
|
|
905
|
+
return tkn;
|
|
906
|
+
}
|
|
907
|
+
// Replace map override.
|
|
908
|
+
if (hasReplace) {
|
|
909
|
+
const rs = replaceCodeMap[cc];
|
|
910
|
+
if (rs !== undefined) {
|
|
911
|
+
buf.push(rs);
|
|
912
|
+
sI++;
|
|
913
|
+
cI++;
|
|
914
|
+
continue;
|
|
915
|
+
}
|
|
916
|
+
}
|
|
917
|
+
// Escape sequence.
|
|
918
|
+
if (cc === escCharCode) {
|
|
919
|
+
sI++;
|
|
920
|
+
cI++;
|
|
921
|
+
if (sI >= srclen)
|
|
922
|
+
break; // unterminated
|
|
923
|
+
const ec = src[sI];
|
|
924
|
+
const es = escMap[ec];
|
|
925
|
+
if (es != null) {
|
|
926
|
+
buf.push(es);
|
|
927
|
+
sI++;
|
|
928
|
+
cI++;
|
|
929
|
+
}
|
|
930
|
+
else if ('x' === ec && !escapeStrict) {
|
|
931
|
+
sI++; // past 'x'
|
|
932
|
+
const xx = parseInt(src.substring(sI, sI + 2), 16);
|
|
933
|
+
if (isNaN(xx)) {
|
|
934
|
+
if (mcfg.abandon)
|
|
935
|
+
return undefined;
|
|
936
|
+
sI -= 2;
|
|
937
|
+
cI -= 1;
|
|
938
|
+
pnt.sI = sI;
|
|
939
|
+
pnt.cI = cI;
|
|
940
|
+
return lex.bad(utility_1.S.invalid_ascii, sI, sI + 4);
|
|
941
|
+
}
|
|
942
|
+
buf.push(String.fromCharCode(xx));
|
|
943
|
+
sI += 2;
|
|
944
|
+
cI += 3;
|
|
945
|
+
}
|
|
946
|
+
else if ('u' === ec) {
|
|
947
|
+
sI++; // past 'u'
|
|
948
|
+
if ('{' === src[sI] && !escapeStrict) {
|
|
949
|
+
// Braced form \u{H...H}: 1-6 hex digits, any code point.
|
|
950
|
+
const endI = src.indexOf('}', sI + 1);
|
|
951
|
+
const digits = -1 === endI ? '' : src.substring(sI + 1, endI);
|
|
952
|
+
const uu = 0 < digits.length && digits.length <= 6 && /^[0-9a-fA-F]+$/.test(digits)
|
|
953
|
+
? parseInt(digits, 16)
|
|
954
|
+
: NaN;
|
|
955
|
+
if (isNaN(uu) || 0x10ffff < uu) {
|
|
956
|
+
if (mcfg.abandon)
|
|
957
|
+
return undefined;
|
|
958
|
+
sI = sI - 2;
|
|
959
|
+
pnt.sI = sI;
|
|
960
|
+
pnt.cI = cI - 1;
|
|
961
|
+
return lex.bad(utility_1.S.invalid_unicode, sI, -1 === endI ? srclen : endI + 1);
|
|
962
|
+
}
|
|
963
|
+
buf.push(String.fromCodePoint(uu));
|
|
964
|
+
cI += endI + 1 - sI + 1;
|
|
965
|
+
sI = endI + 1;
|
|
966
|
+
}
|
|
967
|
+
else {
|
|
968
|
+
const uu = parseInt(src.substring(sI, sI + 4), 16);
|
|
969
|
+
if (isNaN(uu)) {
|
|
970
|
+
if (mcfg.abandon)
|
|
971
|
+
return undefined;
|
|
972
|
+
sI = sI - 2;
|
|
973
|
+
cI -= 1;
|
|
974
|
+
pnt.sI = sI;
|
|
975
|
+
pnt.cI = cI;
|
|
976
|
+
return lex.bad(utility_1.S.invalid_unicode, sI, sI + 6);
|
|
977
|
+
}
|
|
978
|
+
buf.push(String.fromCharCode(uu));
|
|
979
|
+
sI += 4;
|
|
980
|
+
cI += 5;
|
|
981
|
+
}
|
|
982
|
+
}
|
|
983
|
+
else if (allowUnknown) {
|
|
984
|
+
buf.push(ec);
|
|
985
|
+
sI++;
|
|
986
|
+
cI++;
|
|
987
|
+
}
|
|
988
|
+
else {
|
|
989
|
+
if (mcfg.abandon)
|
|
990
|
+
return undefined;
|
|
991
|
+
pnt.sI = sI;
|
|
992
|
+
pnt.cI = cI;
|
|
993
|
+
return lex.bad(utility_1.S.unexpected, sI, sI + 1);
|
|
994
|
+
}
|
|
995
|
+
continue;
|
|
996
|
+
}
|
|
997
|
+
// Stopped on a control char (cc < 32) that wasn't a multi-line
|
|
998
|
+
// newline (those are consumed by the spec). Either an embedded
|
|
999
|
+
// line char in a non-multi-line string or a real unprintable.
|
|
1000
|
+
// Both are errors.
|
|
1001
|
+
if (cc < 32) {
|
|
1002
|
+
if (mcfg.abandon)
|
|
1003
|
+
return undefined;
|
|
1004
|
+
pnt.sI = sI;
|
|
1005
|
+
pnt.cI = cI;
|
|
1006
|
+
return lex.bad(utility_1.S.unprintable, sI, sI + 1);
|
|
1007
|
+
}
|
|
1008
|
+
// Unreachable — the spec only stops on classes handled above.
|
|
1009
|
+
break;
|
|
1010
|
+
}
|
|
1011
|
+
// Hit EOF without closing quote.
|
|
1012
|
+
if (mcfg.abandon)
|
|
1013
|
+
return undefined;
|
|
1014
|
+
pnt.rI = startRI;
|
|
1015
|
+
return lex.bad(utility_1.S.unterminated_string, startSI, sI);
|
|
1016
|
+
});
|
|
1017
|
+
};
|
|
1018
|
+
exports.makeStringMatcher = makeStringMatcher;
|
|
1019
|
+
// Line ending matcher.
|
|
1020
|
+
//
|
|
1021
|
+
// The non-single path is the 3-class line-run state machine
|
|
1022
|
+
// (LINE_RUN_TABLE). `single` mode tracks "has this exact char been
|
|
1023
|
+
// seen yet" — that's not bounded by a small state count so it stays
|
|
1024
|
+
// inline.
|
|
1025
|
+
let makeLineMatcher = (cfg, _opts) => {
|
|
1026
|
+
const spec = buildLineRunSpec(cfg.line);
|
|
1027
|
+
const out = { sI: 0, rI: 0, cI: 0 };
|
|
1028
|
+
return guardedMatcher(cfg.line, function lineBody(lex) {
|
|
1029
|
+
const { pnt, src } = lex;
|
|
1030
|
+
if (cfg.line.single) {
|
|
1031
|
+
// Inline: tracks per-char counts to stop at the first repeat.
|
|
1032
|
+
const bm = cfg.line.charsBitmap;
|
|
1033
|
+
const rbm = cfg.line.rowCharsBitmap;
|
|
1034
|
+
const chars = cfg.line.chars;
|
|
1035
|
+
const rowChars = cfg.line.rowChars;
|
|
1036
|
+
let sI = pnt.sI;
|
|
1037
|
+
let rI = pnt.rI;
|
|
1038
|
+
let cc;
|
|
1039
|
+
const counts = {};
|
|
1040
|
+
while ((cc = src.charCodeAt(sI)) < 256 ? bm[cc] : chars[src[sI]]) {
|
|
1041
|
+
const c = src[sI];
|
|
1042
|
+
const n = (counts[c] || 0) + 1;
|
|
1043
|
+
counts[c] = n;
|
|
1044
|
+
if (n > 1)
|
|
1045
|
+
break;
|
|
1046
|
+
if (cc < 256 ? rbm[cc] : rowChars[src[sI]])
|
|
1047
|
+
rI++;
|
|
1048
|
+
sI++;
|
|
1049
|
+
}
|
|
1050
|
+
if (pnt.sI < sI) {
|
|
1051
|
+
const msrc = src.substring(pnt.sI, sI);
|
|
1052
|
+
const tkn = lex.token('#LN', undefined, msrc, pnt);
|
|
1053
|
+
pnt.sI = sI;
|
|
1054
|
+
pnt.rI = rI;
|
|
1055
|
+
pnt.cI = 1;
|
|
1056
|
+
return tkn;
|
|
1057
|
+
}
|
|
1058
|
+
return;
|
|
1059
|
+
}
|
|
1060
|
+
if (scan(src, pnt.sI, pnt.rI, pnt.cI, spec, out)) {
|
|
1061
|
+
const msrc = src.substring(pnt.sI, out.sI);
|
|
1062
|
+
const tkn = lex.token('#LN', undefined, msrc, pnt);
|
|
1063
|
+
pnt.sI = out.sI;
|
|
1064
|
+
pnt.rI = out.rI;
|
|
1065
|
+
pnt.cI = 1;
|
|
1066
|
+
return tkn;
|
|
1067
|
+
}
|
|
1068
|
+
});
|
|
1069
|
+
};
|
|
1070
|
+
exports.makeLineMatcher = makeLineMatcher;
|
|
1071
|
+
// Space matcher.
|
|
1072
|
+
//
|
|
1073
|
+
// Spec: walk a run of `cfg.space.chars`. Class 0 = not a space,
|
|
1074
|
+
// class 1 = a space. The driver does the rest.
|
|
1075
|
+
let makeSpaceMatcher = (cfg, _opts) => {
|
|
1076
|
+
const spec = buildCharRunSpec(cfg.space.charsBitmap, cfg.space.chars);
|
|
1077
|
+
const out = { sI: 0, rI: 0, cI: 0 };
|
|
1078
|
+
return guardedMatcher(cfg.space, function spaceBody(lex) {
|
|
1079
|
+
const { pnt, src } = lex;
|
|
1080
|
+
if (scan(src, pnt.sI, pnt.rI, pnt.cI, spec, out)) {
|
|
1081
|
+
const msrc = src.substring(pnt.sI, out.sI);
|
|
1082
|
+
const tkn = lex.token('#SP', undefined, msrc, pnt);
|
|
1083
|
+
pnt.sI = out.sI;
|
|
1084
|
+
pnt.rI = out.rI;
|
|
1085
|
+
pnt.cI = out.cI;
|
|
1086
|
+
return tkn;
|
|
1087
|
+
}
|
|
1088
|
+
});
|
|
1089
|
+
};
|
|
1090
|
+
exports.makeSpaceMatcher = makeSpaceMatcher;
|
|
1091
|
+
function subMatchFixed(lex, first, tsrc) {
|
|
1092
|
+
let pnt = lex.pnt;
|
|
1093
|
+
let out = first;
|
|
1094
|
+
if (lex.cfg.fixed.lex && null != tsrc) {
|
|
1095
|
+
let tknlen = tsrc.length;
|
|
1096
|
+
if (0 < tknlen) {
|
|
1097
|
+
let tkn = undefined;
|
|
1098
|
+
let tin = lex.cfg.fixed.token[tsrc];
|
|
1099
|
+
if (null != tin) {
|
|
1100
|
+
tkn = lex.token(tin, undefined, tsrc, pnt);
|
|
1101
|
+
}
|
|
1102
|
+
if (null != tkn) {
|
|
1103
|
+
pnt.sI += tkn.src.length;
|
|
1104
|
+
pnt.cI += tkn.src.length;
|
|
1105
|
+
if (null == first) {
|
|
1106
|
+
out = tkn;
|
|
1107
|
+
}
|
|
1108
|
+
else {
|
|
1109
|
+
pnt.token.push(tkn);
|
|
1110
|
+
}
|
|
1111
|
+
}
|
|
1112
|
+
}
|
|
1113
|
+
}
|
|
1114
|
+
return out;
|
|
1115
|
+
}
|
|
1116
|
+
// Lexer driver: holds the scan Point and runs the configured matchers in order via next().
|
|
1117
|
+
class Lex {
|
|
1118
|
+
refwd() {
|
|
1119
|
+
this.fwd = this.src.substring(this.pnt.sI);
|
|
1120
|
+
return this.fwd;
|
|
1121
|
+
}
|
|
1122
|
+
constructor(ctx) {
|
|
1123
|
+
this.src = types_1.EMPTY; // Full source text being lexed.
|
|
1124
|
+
this.ctx = {}; // Parse context (config, logging, subscribers).
|
|
1125
|
+
this.cfg = {}; // Resolved configuration.
|
|
1126
|
+
this.pnt = makePoint(-1); // Current scan position.
|
|
1127
|
+
this.fwd = types_1.EMPTY; // Source from pnt.sI onward (the unconsumed remainder).
|
|
1128
|
+
this.ctx = ctx;
|
|
1129
|
+
this.src = ctx.src();
|
|
1130
|
+
this.cfg = ctx.cfg;
|
|
1131
|
+
this.pnt = makePoint(this.src.length);
|
|
1132
|
+
}
|
|
1133
|
+
token(ref, val, src, pnt, use, why) {
|
|
1134
|
+
let tin;
|
|
1135
|
+
let name;
|
|
1136
|
+
if ('string' === typeof ref) {
|
|
1137
|
+
name = ref;
|
|
1138
|
+
tin = (0, utility_1.tokenize)(name, this.cfg);
|
|
1139
|
+
}
|
|
1140
|
+
else {
|
|
1141
|
+
tin = ref;
|
|
1142
|
+
name = (0, utility_1.tokenize)(ref, this.cfg);
|
|
1143
|
+
}
|
|
1144
|
+
let tkn = makeToken(name, tin, val, src, pnt || this.pnt, use, why);
|
|
1145
|
+
return tkn;
|
|
1146
|
+
}
|
|
1147
|
+
next(rule, alt, altI, tI) {
|
|
1148
|
+
let tkn;
|
|
1149
|
+
let pnt = this.pnt;
|
|
1150
|
+
let sI = pnt.sI;
|
|
1151
|
+
let match = undefined;
|
|
1152
|
+
if (pnt.end) {
|
|
1153
|
+
tkn = pnt.end;
|
|
1154
|
+
}
|
|
1155
|
+
else if (0 < pnt.token.length) {
|
|
1156
|
+
tkn = pnt.token.shift();
|
|
1157
|
+
}
|
|
1158
|
+
else if (pnt.len <= pnt.sI) {
|
|
1159
|
+
pnt.end = this.token('#ZZ', undefined, '', pnt);
|
|
1160
|
+
tkn = pnt.end;
|
|
1161
|
+
}
|
|
1162
|
+
else {
|
|
1163
|
+
this.fwd = this.src.substring(pnt.sI);
|
|
1164
|
+
try {
|
|
1165
|
+
for (let mat of this.cfg.lex.match) {
|
|
1166
|
+
if ((tkn = mat(this, rule, tI))) {
|
|
1167
|
+
match = mat;
|
|
1168
|
+
break;
|
|
1169
|
+
}
|
|
1170
|
+
}
|
|
1171
|
+
}
|
|
1172
|
+
catch (err) {
|
|
1173
|
+
tkn =
|
|
1174
|
+
tkn ||
|
|
1175
|
+
this.token('#BD', undefined, this.src[pnt.sI], pnt, { err }, err.code || utility_1.S.unexpected);
|
|
1176
|
+
}
|
|
1177
|
+
tkn =
|
|
1178
|
+
tkn ||
|
|
1179
|
+
this.token('#BD', undefined, this.src[pnt.sI], pnt, undefined, utility_1.S.unexpected);
|
|
1180
|
+
}
|
|
1181
|
+
this.ctx.log &&
|
|
1182
|
+
this.ctx.log(utility_1.S.lex, this.ctx, rule, this, pnt, sI, match, tkn, alt, altI, tI);
|
|
1183
|
+
// this.ctx.log &&
|
|
1184
|
+
// log_lex(this.ctx, rule, this, pnt, sI, match, tkn, alt, altI, tI)
|
|
1185
|
+
if (this.ctx.sub.lex) {
|
|
1186
|
+
this.ctx.sub.lex.map((sub) => sub(tkn, rule, this.ctx));
|
|
1187
|
+
}
|
|
1188
|
+
return tkn;
|
|
1189
|
+
}
|
|
1190
|
+
tokenize(ref) {
|
|
1191
|
+
return (0, utility_1.tokenize)(ref, this.cfg);
|
|
1192
|
+
}
|
|
1193
|
+
bad(why, pstart, pend) {
|
|
1194
|
+
return this.token('#BD', undefined, 0 <= pstart && pstart <= pend
|
|
1195
|
+
? this.src.substring(pstart, pend)
|
|
1196
|
+
: this.src[this.pnt.sI], undefined, undefined, why);
|
|
1197
|
+
}
|
|
1198
|
+
}
|
|
1199
|
+
exports.Lex = Lex;
|
|
1200
|
+
const makeLex = (...params) => new Lex(...params);
|
|
1201
|
+
exports.makeLex = makeLex;
|
|
1202
|
+
//# sourceMappingURL=lexer.js.map
|