grammar-composer 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,3 +1,4 @@
1
+ import { isNumber } from '../utilities/Utilities.js'
1
2
  import { Grammar, GrammarElement, Terminal } from './Grammar.js'
2
3
 
3
4
  export function parse(inputString: string, grammar: Grammar<any>) {
@@ -18,37 +19,30 @@ export function parse(inputString: string, grammar: Grammar<any>) {
18
19
  }
19
20
 
20
21
  function tryParse(grammarElement: GrammarElement, startOffset: number): ParseResult | null {
21
- if (grammarElement.cached === true) {
22
+ if (isNumber(grammarElement.cacheId)) {
22
23
  return tryParseCached(grammarElement, startOffset)
23
24
  } else {
24
25
  return tryParseUncached(grammarElement, startOffset)
25
26
  }
26
27
  }
27
28
 
28
- type Slot = Map<GrammarElement, ParseResult | null> | undefined
29
+ const offsetMultiplier = grammar.maxCacheId + 1
29
30
 
30
- const cachedParseResults: Slot[] = new Array(inputLength)
31
+ const parseResultsCache = new Map<number, ParseResult | null>()
31
32
 
32
33
  function tryParseCached(grammarElement: GrammarElement, startOffset: number): ParseResult | null {
33
- let slot = cachedParseResults[startOffset]
34
+ const cacheId = grammarElement.cacheId!
35
+ const cacheKey = (startOffset * offsetMultiplier) + cacheId
34
36
 
35
- if (slot === undefined) {
36
- slot = new Map<GrammarElement, ParseResult | null>()
37
-
38
- cachedParseResults[startOffset] = slot
37
+ if (parseResultsCache.has(cacheKey)) {
38
+ return parseResultsCache.get(cacheKey)!
39
39
  } else {
40
- const cachedResult = slot.get(grammarElement)
41
-
42
- if (cachedResult !== undefined) {
43
- return cachedResult
44
- }
45
- }
46
-
47
- const parseResult = tryParseUncached(grammarElement, startOffset)
40
+ const parseResult = tryParseUncached(grammarElement, startOffset)
48
41
 
49
- slot.set(grammarElement, parseResult)
42
+ parseResultsCache.set(cacheKey, parseResult)
50
43
 
51
- return parseResult
44
+ return parseResult
45
+ }
52
46
  }
53
47
 
54
48
  function tryParseUncached(grammarElement: GrammarElement, startOffset: number): ParseResult | null {
package/src/tests/Test.ts CHANGED
@@ -4,6 +4,8 @@ import { anyOf, buildGrammar } from '../exports/Exports.js'
4
4
  import { JsonGrammar } from './test-grammars/JsonGrammar.js'
5
5
  import { XmlGrammar } from './test-grammars/XmlGrammar.js'
6
6
  import { RegExpGrammar } from './test-grammars/RegExpGrammar.js'
7
+ import { writeFile } from 'fs/promises'
8
+ import { SimpleTestGrammar1 } from './test-grammars/SimpleTestGrammar1.js'
7
9
 
8
10
  const log = console.log
9
11
 
@@ -83,9 +85,11 @@ function testXmlParser() {
83
85
 
84
86
  async function testRegExpParser() {
85
87
  const regExpString = /^([+]?[1]?(1 )?[-.+]?\(?\d{1}[- .+]*\d{1}[- .+]*\d{1}\)?[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1})$/.source
86
- //const regExpString = /^asdf{$/.source
88
+ //const regExpString = /^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$/.source
89
+ //const regExpString = /(?=.*[!@#$%^&*])^mongodb:\/\/(?<user>[a-zA-Z0-9]+):(?<pass>[a-zA-Z0-9!@#$%^&*]{8,})@(?<host>[a-z0-9.-]+):(?<port>\d{2,5})$/.source
90
+ //const regExpString = /(abcd)*ef+g/.source
87
91
 
88
- const grammar = buildGrammar(RegExpGrammar, 'disjunction')
92
+ const grammar = buildGrammar(RegExpGrammar, 'root')
89
93
 
90
94
  const parseTree = grammar.parse(regExpString)
91
95
 
@@ -93,13 +97,10 @@ async function testRegExpParser() {
93
97
 
94
98
  log(parseTreeJson)
95
99
 
96
- const { writeFile } = await import('fs/promises')
97
-
98
100
  await writeFile('out/out.json', parseTreeJson)
99
101
  }
100
102
 
101
-
102
- async function testParserError1() {
103
+ function testParserError1() {
103
104
  const xmlData = `<hello> wo rld <!!! `
104
105
 
105
106
  const grammar = buildGrammar(XmlGrammar, 'document')
@@ -109,7 +110,7 @@ async function testParserError1() {
109
110
  console.log(JSON.stringify(result, undefined, 4))
110
111
  }
111
112
 
112
- async function testParserError2() {
113
+ function testParserError2() {
113
114
  const jsonData = `{ "asdf": 12.5 `
114
115
 
115
116
  const grammar = buildGrammar(JsonGrammar, 'expression')
@@ -119,6 +120,19 @@ async function testParserError2() {
119
120
  console.log(JSON.stringify(result, undefined, 4))
120
121
  }
121
122
 
123
+ function test1() {
124
+ const input = `abcdefg`
125
+
126
+ const grammar = buildGrammar(SimpleTestGrammar1, 'root')
127
+
128
+ const result = grammar.parse(input)
129
+
130
+ console.log(JSON.stringify(result, undefined, 4))
131
+ }
132
+
133
+
122
134
  //testJsonParser()
123
135
 
124
136
  testRegExpParser()
137
+
138
+ //test1()
@@ -3,101 +3,132 @@ import * as R from 'regexp-composer'
3
3
 
4
4
  export class RegExpGrammar {
5
5
  //////////////////////////////////////////////////////////////////////////////////////////////
6
- // High level productions
6
+ // High-level productions
7
7
  //////////////////////////////////////////////////////////////////////////////////////////////
8
+ root = () => this.sequenceOrDisjunction
9
+
8
10
  disjunction = () => [
9
11
  this.sequence,
10
12
 
11
- G.possibly([
12
- this.disjunctionSeparator,
13
- this.disjunction,
13
+ G.oneOrMore([
14
+ '|',
15
+ this.sequence,
14
16
  ])
15
17
  ]
16
18
 
17
- sequence = () => G.oneOrMore(this.element)
18
-
19
- element = () => G.anyOf(
20
- this.anchor,
21
- this.lookaround,
22
- this.possiblyQuantifiedExpression,
19
+ sequence = () => G.cached(
20
+ G.zeroOrMore(this.sequenceElement)
23
21
  )
24
22
 
23
+ sequenceOrDisjunction = [
24
+ G.anyOf(
25
+ this.disjunction,
26
+ this.sequence,
27
+ )
28
+ ]
29
+
25
30
  //////////////////////////////////////////////////////////////////////////////////////////////
26
- // Quantifed expressions
31
+ // Quantifier expressions
27
32
  //////////////////////////////////////////////////////////////////////////////////////////////
28
- possiblyQuantifiedExpression = () => [
29
- G.anyOf(
30
- this.group,
31
- this.backReference,
32
- this.singleCharExpression
33
- ),
33
+ starQuantifier = () => [
34
+ '*',
35
+ G.possibly(this.nongreedyQuantifier)
36
+ ]
34
37
 
35
- G.possibly(G.anyOf(
36
- this.starQuantifier,
37
- this.plusQuantifier,
38
- this.exactCountQuantifier,
39
- this.countRangeQuantifier,
40
- )),
38
+ plusQuantifier = () => [
39
+ '+',
40
+ G.possibly(this.nongreedyQuantifier)
41
+ ]
41
42
 
42
- G.possibly(this.nongreedyQuantifier), // Nongreedy quantifier
43
+ optionalQuantifier = () => [
44
+ '?',
45
+ G.possibly(this.nongreedyQuantifier)
43
46
  ]
44
47
 
45
- exactCountQuantifier = () => G.pattern([
46
- '{',
47
- R.captureAs('count', R.oneOrMore(digit)),
48
- '}',
49
- ])
48
+ exactCountQuantifier = () => [
49
+ G.pattern([
50
+ '{',
51
+ R.captureAs('count',
52
+ R.oneOrMore(digit)
53
+ ),
54
+ '}',
55
+ ]),
56
+ G.possibly(this.nongreedyQuantifier)
57
+ ]
50
58
 
51
- countRangeQuantifier = () => G.pattern([
52
- '{',
53
- R.captureAs('start', R.oneOrMore(digit)),
54
- ',',
55
- R.possibly(R.captureAs('end', R.oneOrMore(digit))),
56
- '}',
57
- ])
59
+ countRangeQuantifier = () => [
60
+ G.pattern([
61
+ '{',
62
+ R.captureAs('start',
63
+ R.oneOrMore(digit)
64
+ ),
65
+ ',',
66
+ R.possibly(R.captureAs('end',
67
+ R.oneOrMore(digit))
68
+ ),
69
+ '}',
70
+ ]),
71
+ G.possibly(this.nongreedyQuantifier)
72
+ ]
73
+
74
+ nongreedyQuantifier = () => '?'
75
+
76
+ quantifier = () => G.anyOf(
77
+ this.starQuantifier,
78
+ this.plusQuantifier,
79
+ this.optionalQuantifier,
80
+ this.exactCountQuantifier,
81
+ this.countRangeQuantifier,
82
+ )
58
83
 
59
84
  //////////////////////////////////////////////////////////////////////////////////////////////
60
85
  // Group expressions
61
86
  //////////////////////////////////////////////////////////////////////////////////////////////
62
- group = () => G.anyOf(
63
- this.uncapturedGroup,
64
- this.namedCaptureGroup,
65
- this.unnamedCaptureGroup,
66
- )
67
-
68
87
  uncapturedGroup = () => [
69
- '(:',
70
- this.disjunction,
88
+ '(?:',
89
+ this.sequenceOrDisjunction,
71
90
  ')',
72
91
  ]
73
92
 
74
93
  namedCaptureGroup = () => [
75
94
  '(?<',
76
- G.pattern(R.captureAs('name', R.oneOrMore(letter))),
95
+ G.pattern(R.captureAs('name',
96
+ R.oneOrMore(identifierChar))
97
+ ),
77
98
  '>',
78
- this.disjunction,
99
+ this.sequenceOrDisjunction,
79
100
  ')',
80
101
  ]
81
102
 
82
103
  unnamedCaptureGroup = () => [
83
104
  '(',
84
- this.disjunction,
105
+ this.sequenceOrDisjunction,
85
106
  ')',
86
107
  ]
87
108
 
109
+ group = G.anyOf(
110
+ this.uncapturedGroup,
111
+ this.namedCaptureGroup,
112
+ this.unnamedCaptureGroup,
113
+ )
114
+
88
115
  //////////////////////////////////////////////////////////////////////////////////////////////
89
116
  // Backreferences
90
117
  //////////////////////////////////////////////////////////////////////////////////////////////
91
118
  unnamedBackreference = () =>
92
119
  G.pattern([
93
120
  '\\',
94
- R.captureAs('index', digit),
121
+ R.captureAs('index',
122
+ digit
123
+ ),
95
124
  ])
96
125
 
97
126
  namedBackreference = () =>
98
127
  G.pattern([
99
128
  '\\k<',
100
- R.captureAs('name', digit),
129
+ R.captureAs('name',
130
+ R.oneOrMore(identifierChar)
131
+ ),
101
132
  '>',
102
133
  ])
103
134
 
@@ -111,25 +142,25 @@ export class RegExpGrammar {
111
142
  //////////////////////////////////////////////////////////////////////////////////////////////
112
143
  positiveLookahead = () => [
113
144
  '(?=',
114
- this.disjunction,
145
+ this.sequenceOrDisjunction,
115
146
  ')',
116
147
  ]
117
148
 
118
149
  negativeLookahead = () => [
119
150
  '(?!',
120
- this.disjunction,
151
+ this.sequenceOrDisjunction,
121
152
  ')',
122
153
  ]
123
154
 
124
155
  positiveLookbehind = () => [
125
156
  '(?<=',
126
- this.disjunction,
157
+ this.sequenceOrDisjunction,
127
158
  ')',
128
159
  ]
129
160
 
130
161
  negativeLookbehind = () => [
131
162
  '(?<!',
132
- this.disjunction,
163
+ this.sequenceOrDisjunction,
133
164
  ')',
134
165
  ]
135
166
 
@@ -148,44 +179,68 @@ export class RegExpGrammar {
148
179
  this.charClass,
149
180
  this.unicodeProperty,
150
181
  this.notUnicodeProperty,
182
+ this.escapedCharacterClass,
151
183
  this.charcodeOrEscapedChar,
152
184
  this.unreservedCharLiteral
153
185
  )
154
186
 
155
187
  unreservedCharLiteral = () => G.pattern(
156
- R.notAnyOfChars('[', '.', '*', '+', '?', '^', '$', '{', '}', '(', ')', '|', ']', '\\')
188
+ R.captureAs('char',
189
+ R.notAnyOfChars('[', '.', '*', '+', '?', '^', '$', '{', '}', '(', ')', '|', ']', '\\')
190
+ )
157
191
  )
158
192
 
159
193
  escapedCharLiteral = () => G.pattern([
160
194
  '\\',
161
- R.anyChar
195
+ R.captureAs('char',
196
+ R.anyChar
197
+ )
162
198
  ])
163
199
 
164
200
  hexCharcode = () =>
165
201
  G.pattern([
166
202
  '\\x',
167
- hexDigit,
168
- hexDigit
203
+ R.captureAs('value',
204
+ R.repeated(2, hexDigit)
205
+ )
169
206
  ])
170
207
 
171
208
  controlCharcode = () => G.pattern([
172
209
  '\\c',
173
- R.charRange('A', 'Z'),
210
+ R.captureAs('value',
211
+ R.anyOf(R.charRange('A', 'Z'), R.charRange('a', 'z'))
212
+ )
174
213
  ])
175
214
 
176
- codepoint = () => G.pattern([
177
- '\\u',
178
- '{',
179
- R.captureAs('value', R.oneOrMore(hexDigit)),
180
- '}',
181
- ])
215
+ nullCharcode = () => G.pattern('\\0')
216
+
217
+ codepoint = () => G.anyOf(
218
+ G.pattern([
219
+ '\\u',
220
+ '{',
221
+ R.captureAs('value',
222
+ R.repeated([1, 6], hexDigit)
223
+ ),
224
+ '}',
225
+ ]),
226
+ G.pattern([
227
+ '\\u',
228
+ R.captureAs('value',
229
+ R.repeated(4, hexDigit)
230
+ )
231
+ ])
232
+ )
182
233
 
183
234
  codepointRange = () => G.pattern([
184
235
  '\\u',
185
236
  '{',
186
- R.captureAs('start', R.oneOrMore(hexDigit)),
237
+ R.captureAs('start',
238
+ R.oneOrMore(hexDigit)
239
+ ),
187
240
  '-',
188
- R.captureAs('end', R.oneOrMore(hexDigit)),
241
+ R.captureAs('end',
242
+ R.oneOrMore(hexDigit)
243
+ ),
189
244
  '}',
190
245
  ])
191
246
 
@@ -201,27 +256,29 @@ export class RegExpGrammar {
201
256
 
202
257
  unicodePropertyBodyPattern: R.Pattern = [
203
258
  '{',
204
- R.captureAs('property', R.oneOrMore(letterOrDigit)),
259
+ R.captureAs('property',
260
+ R.oneOrMore(R.anyOf(letter, digit, '_', '-'))
261
+ ),
205
262
  R.possibly([
206
263
  '=',
207
- R.captureAs('value', R.oneOrMore(letterOrDigit)),
264
+ R.captureAs('value',
265
+ R.oneOrMore(R.anyOf(letter, digit, '_', '-'))
266
+ ),
208
267
  ]),
209
268
  '}',
210
269
  ]
211
270
 
212
- charClassChar = () => G.pattern(R.notAnyOfChars(']'))
213
-
214
271
  charClass = () => [
215
272
  '[',
216
273
  G.possibly(this.charClassNegator),
217
274
  G.oneOrMore(G.anyOf(
218
275
  this.charRange,
219
276
  this.codepointRange,
220
- this.charClass,
221
277
  this.unicodeProperty,
222
278
  this.notUnicodeProperty,
279
+ this.escapedCharacterClass,
223
280
  this.charcodeOrEscapedChar,
224
- this.charClassChar
281
+ this.charClassLiteral
225
282
  )),
226
283
  ']',
227
284
  ]
@@ -230,27 +287,33 @@ export class RegExpGrammar {
230
287
  this.codepoint,
231
288
  this.hexCharcode,
232
289
  this.controlCharcode,
290
+ this.nullCharcode,
233
291
  this.escapedCharLiteral,
234
292
  )
235
293
 
294
+ charClassLiteral = () => G.pattern(
295
+ R.captureAs('char',
296
+ R.notAnyOfChars(']')
297
+ )
298
+ )
299
+
236
300
  charRangeElement = G.anyOf(
237
301
  this.charcodeOrEscapedChar,
238
- G.pattern(R.anyChar),
302
+ this.charClassLiteral,
239
303
  )
240
304
 
305
+ charRangeStart = () => this.charRangeElement
306
+ charRangeEnd = () => this.charRangeElement
307
+
241
308
  charRange = () => [
242
- this.charRangeElement,
309
+ this.charRangeStart,
243
310
  '-',
244
- this.charRangeElement,
311
+ this.charRangeEnd,
245
312
  ]
246
313
 
247
314
  //////////////////////////////////////////////////////////////////////////////////////////////
248
315
  // Special symbols
249
316
  //////////////////////////////////////////////////////////////////////////////////////////////
250
- starQuantifier = () => '*'
251
- plusQuantifier = () => '+'
252
- nongreedyQuantifier = () => '?'
253
-
254
317
  charClassNegator = () => '^'
255
318
 
256
319
  inputStartAnchor = () => '^'
@@ -258,38 +321,67 @@ export class RegExpGrammar {
258
321
 
259
322
  anyCharWildcard = () => '.'
260
323
 
261
- disjunctionSeparator = () => '|'
262
-
263
324
  anchor = G.anyOf(this.inputStartAnchor, this.inputEndAnchor)
264
325
 
265
326
  escapedCharacterClass = () => G.pattern(R.anyOf(
266
- R.captureAs('whitespace', escapedCharString.whitespace),
267
- R.captureAs('nonWhitespace', escapedCharString.nonWhitespace),
268
- R.captureAs('digit', escapedCharString.digit),
269
- R.captureAs('nonDigit', escapedCharString.nonDigit),
270
- R.captureAs('wordBoundary', escapedCharString.wordBoundary),
271
- R.captureAs('nonWordBoundary', escapedCharString.nonWordBoundary),
272
- R.captureAs('formFeed', escapedCharString.formFeed),
273
- R.captureAs('carriageReturn', escapedCharString.carriageReturn),
274
- R.captureAs('lineFeed', escapedCharString.lineFeed),
275
- R.captureAs('tab', escapedCharString.tab),
276
- R.captureAs('verticalTab', escapedCharString.verticalTab),
277
- R.captureAs('backwardSlash', escapedCharString.backwardSlash),
327
+ R.captureAs('whitespace', escapedChars.whitespace),
328
+ R.captureAs('nonWhitespace', escapedChars.nonWhitespace),
329
+ R.captureAs('digit', escapedChars.digit),
330
+ R.captureAs('nonDigit', escapedChars.nonDigit),
331
+ R.captureAs('word', escapedChars.word),
332
+ R.captureAs('nonWord', escapedChars.nonWord),
333
+ R.captureAs('wordBoundary', escapedChars.wordBoundary),
334
+ R.captureAs('nonWordBoundary', escapedChars.nonWordBoundary),
335
+ R.captureAs('formFeed', escapedChars.formFeed),
336
+ R.captureAs('carriageReturn', escapedChars.carriageReturn),
337
+ R.captureAs('lineFeed', escapedChars.lineFeed),
338
+ R.captureAs('tab', escapedChars.tab),
339
+ R.captureAs('verticalTab', escapedChars.verticalTab),
340
+ R.captureAs('backwardSlash', escapedChars.backwardSlash),
278
341
  ))
342
+
343
+ //////////////////////////////////////////////////////////////////////////////////////////////
344
+ // Sequence elements (positioned here due to TypeScript class member ordering requirements)
345
+ //////////////////////////////////////////////////////////////////////////////////////////////
346
+ quantifiableExpression = G.cached(
347
+ G.anyOf(
348
+ this.group,
349
+ this.backReference,
350
+ this.singleCharExpression
351
+ )
352
+ )
353
+
354
+ quantifiedExpression = () => [
355
+ this.quantifiableExpression,
356
+ this.quantifier,
357
+ ]
358
+
359
+ sequenceElement = G.anyOf(
360
+ this.anchor,
361
+ this.lookaround,
362
+ this.quantifiedExpression,
363
+ this.quantifiableExpression,
364
+ )
279
365
  }
280
366
 
367
+ //////////////////////////////////////////////////////////////////////////////////////////////
368
+ // Shared regular expressions
369
+ //////////////////////////////////////////////////////////////////////////////////////////////
281
370
  const digit = R.charRange('0', '9')
282
371
  const letter = R.anyOf(R.charRange('a', 'z'), R.charRange('A', 'Z'))
283
- const letterOrDigit = R.anyOf(letter, digit)
284
372
  const hexDigit = R.anyOf(digit, R.charRange('a', 'f'), R.charRange('A', 'F'))
373
+ const identifierChar = R.anyOf(letter, digit, '_', '$')
285
374
 
286
- const escapedCharString = {
375
+ //////////////////////////////////////////////////////////////////////////////////////////////
376
+ // Escaped characters lookup
377
+ //////////////////////////////////////////////////////////////////////////////////////////////
378
+ const escapedChars = {
287
379
  whitespace: '\\s',
288
380
  nonWhitespace: '\\S',
289
381
  digit: '\\d',
290
382
  nonDigit: '\\D',
291
- word: '\\d',
292
- nonWord: '\\D',
383
+ word: '\\w',
384
+ nonWord: '\\W',
293
385
  wordBoundary: '\\b',
294
386
  nonWordBoundary: '\\B',
295
387
  formFeed: '\\f',
@@ -0,0 +1,10 @@
1
+ import * as G from '../../exports/Exports.js'
2
+
3
+ export class SimpleTestGrammar1 {
4
+ root = () => [
5
+ G.anyOf(
6
+ 'abc',
7
+ 'abcdefg',
8
+ )
9
+ ]
10
+ }
@@ -4,30 +4,30 @@ export function roundToDigits(val: number, digits = 3) {
4
4
  return Math.round(val * multiplier) / multiplier
5
5
  }
6
6
 
7
- export function isNumber(value: any): value is number {
7
+ export function isNumber(value: unknown): value is number {
8
8
  return typeof value === 'number'
9
9
  }
10
10
 
11
- export function isString(value: any): value is string {
11
+ export function isString(value: unknown): value is string {
12
12
  return typeof value === 'string'
13
13
  }
14
14
 
15
- export function isBigInt(value: any): value is bigint {
15
+ export function isBigInt(value: unknown): value is bigint {
16
16
  return typeof value === 'bigint'
17
17
  }
18
18
 
19
- export function isObject(value: any): value is object {
19
+ export function isObject(value: unknown): value is object {
20
20
  return typeof value === 'object' && !Array.isArray(value)
21
21
  }
22
22
 
23
- export function isArray(value: any): value is any[] {
23
+ export function isArray(value: unknown): value is unknown[] {
24
24
  return Array.isArray(value)
25
25
  }
26
26
 
27
- export function isBoolean(value: any): value is boolean {
27
+ export function isBoolean(value: unknown): value is boolean {
28
28
  return typeof value === 'boolean'
29
29
  }
30
30
 
31
- export function isFunction(value: any): value is Function {
31
+ export function isFunction(value: unknown): value is Function {
32
32
  return typeof value === 'function'
33
33
  }