grammar-composer 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +144 -89
- package/dist/exports/Exports.d.ts +2 -0
- package/dist/exports/Exports.d.ts.map +1 -1
- package/dist/exports/Exports.js +2 -0
- package/dist/exports/Exports.js.map +1 -1
- package/dist/parser-generator/Grammar.d.ts +8 -1
- package/dist/parser-generator/Grammar.d.ts.map +1 -1
- package/dist/parser-generator/Grammar.js +30 -225
- package/dist/parser-generator/Grammar.js.map +1 -1
- package/dist/parser-generator/ParseError.d.ts +37 -0
- package/dist/parser-generator/ParseError.d.ts.map +1 -0
- package/dist/parser-generator/ParseError.js +226 -0
- package/dist/parser-generator/ParseError.js.map +1 -0
- package/dist/parser-generator/StaticAnalysis.d.ts +6 -0
- package/dist/parser-generator/StaticAnalysis.d.ts.map +1 -0
- package/dist/parser-generator/StaticAnalysis.js +283 -0
- package/dist/parser-generator/StaticAnalysis.js.map +1 -0
- package/dist/parser-generator/TerminalToText.d.ts +15 -0
- package/dist/parser-generator/TerminalToText.d.ts.map +1 -0
- package/dist/parser-generator/TerminalToText.js +158 -0
- package/dist/parser-generator/TerminalToText.js.map +1 -0
- package/dist/parser-generator/TopDownParser.d.ts +4 -2
- package/dist/parser-generator/TopDownParser.d.ts.map +1 -1
- package/dist/parser-generator/TopDownParser.js +44 -42
- package/dist/parser-generator/TopDownParser.js.map +1 -1
- package/dist/tests/Test.js +23 -23
- package/dist/tests/Test.js.map +1 -1
- package/dist/tests/test-grammars/JsonGrammar.d.ts +12 -11
- package/dist/tests/test-grammars/JsonGrammar.d.ts.map +1 -1
- package/dist/tests/test-grammars/JsonGrammar.js +25 -10
- package/dist/tests/test-grammars/JsonGrammar.js.map +1 -1
- package/dist/tests/test-grammars/RegExpGrammar.d.ts +25 -25
- package/dist/tests/test-grammars/RegExpGrammar.d.ts.map +1 -1
- package/dist/tests/test-grammars/RegExpGrammar.js +48 -31
- package/dist/tests/test-grammars/RegExpGrammar.js.map +1 -1
- package/dist/tests/test-grammars/XmlGrammar.d.ts +11 -10
- package/dist/tests/test-grammars/XmlGrammar.d.ts.map +1 -1
- package/dist/tests/test-grammars/XmlGrammar.js +19 -8
- package/dist/tests/test-grammars/XmlGrammar.js.map +1 -1
- package/dist/utilities/LineAndColumn.d.ts +7 -0
- package/dist/utilities/LineAndColumn.d.ts.map +1 -0
- package/dist/utilities/LineAndColumn.js +52 -0
- package/dist/utilities/LineAndColumn.js.map +1 -0
- package/dist/utilities/Timer.d.ts +7 -3
- package/dist/utilities/Timer.d.ts.map +1 -1
- package/dist/utilities/Timer.js +41 -39
- package/dist/utilities/Timer.js.map +1 -1
- package/package.json +3 -3
- package/src/exports/Exports.ts +2 -0
- package/src/parser-generator/Grammar.ts +59 -265
- package/src/parser-generator/ParseError.ts +337 -0
- package/src/parser-generator/StaticAnalysis.ts +340 -0
- package/src/parser-generator/TerminalToText.ts +196 -0
- package/src/parser-generator/TopDownParser.ts +67 -49
- package/src/tests/Test.ts +26 -24
- package/src/tests/test-grammars/JsonGrammar.ts +28 -10
- package/src/tests/test-grammars/RegExpGrammar.ts +65 -46
- package/src/tests/test-grammars/XmlGrammar.ts +20 -8
- package/src/utilities/LineAndColumn.ts +61 -0
- package/src/utilities/Timer.ts +50 -48
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
import type { Pattern, SpecialToken } from 'regexp-composer'
|
|
2
|
+
import type { PatternTerminal, StringTerminal, Terminal } from './Grammar.js'
|
|
3
|
+
|
|
4
|
+
/////////////////////////////////////////////////////////////////////////////////////////////////
|
|
5
|
+
// Terminal to text conversion
|
|
6
|
+
/////////////////////////////////////////////////////////////////////////////////////////////////
|
|
7
|
+
export class TerminalToText {
|
|
8
|
+
constructor(private stringifiedLengthLimit: number) {
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
stringifyTerminal(terminal: Terminal): string {
|
|
12
|
+
if (terminal.type === 'StringTerminal') {
|
|
13
|
+
return this.stringifyStringTerminal(terminal)
|
|
14
|
+
} else {
|
|
15
|
+
return this.stringifyPatternTerminal(terminal)
|
|
16
|
+
}
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
stringifyStringTerminal(terminal: StringTerminal): string {
|
|
20
|
+
return this.stringifyPatternStringLiteral(terminal.content)
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
stringifyPatternTerminal(terminal: PatternTerminal) {
|
|
24
|
+
let pattern = terminal.pattern
|
|
25
|
+
|
|
26
|
+
// Pattern terminals are automatically anchored at the start of the current parse
|
|
27
|
+
// position. This anchor is an implementation detail, so remove it.
|
|
28
|
+
if (Array.isArray(pattern) && pattern.length > 1) {
|
|
29
|
+
const firstElement = pattern[0]
|
|
30
|
+
|
|
31
|
+
if (!Array.isArray(firstElement) && typeof firstElement !== 'string' &&
|
|
32
|
+
firstElement.type === 'specialToken' && firstElement.name === 'inputStart') {
|
|
33
|
+
|
|
34
|
+
pattern = pattern.slice(1)
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
const stringifiedPattern = this.stringifyPattern(pattern)
|
|
39
|
+
const truncatedStringifiedPattern = this.truncateString(stringifiedPattern, this.stringifiedLengthLimit)
|
|
40
|
+
|
|
41
|
+
return truncatedStringifiedPattern
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
stringifyPattern(pattern: Pattern): string {
|
|
45
|
+
if (typeof pattern === 'string') {
|
|
46
|
+
return this.stringifyPatternStringLiteral(pattern)
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
if (Array.isArray(pattern)) {
|
|
50
|
+
return `[${pattern.map((pattern) => this.stringifyPattern(pattern)).join(', ')}]`
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
switch (pattern.type) {
|
|
54
|
+
case 'specialToken': {
|
|
55
|
+
return this.stringifySpecialToken(pattern)
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
case 'possibly': {
|
|
59
|
+
return `possibly(${this.stringifyPattern(pattern.content)})`
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
case 'zeroOrMore': {
|
|
63
|
+
return `zeroOrMore(${this.stringifyPattern(pattern.content)})`
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
case 'oneOrMore': {
|
|
67
|
+
return `oneOrMore(${this.stringifyPattern(pattern.content)})`
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
case 'repeated': {
|
|
71
|
+
const { minCount, maxCount } = pattern
|
|
72
|
+
const stringifiedContent = this.stringifyPattern(pattern.content)
|
|
73
|
+
|
|
74
|
+
if (minCount === maxCount) {
|
|
75
|
+
return `repeated(${minCount}, ${stringifiedContent})`
|
|
76
|
+
} else if (maxCount === Number.POSITIVE_INFINITY) {
|
|
77
|
+
return `repeated([${minCount}, ...], ${stringifiedContent})`
|
|
78
|
+
} else {
|
|
79
|
+
return `repeated([${minCount}, ${maxCount}], ${stringifiedContent})`
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
case 'capture': {
|
|
84
|
+
if (pattern.name !== undefined) {
|
|
85
|
+
return pattern.name
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
return `(${this.stringifyPattern(pattern.content)})`
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
case 'anyOf': {
|
|
92
|
+
const stringifiedMembers = pattern.members.map((pattern) => this.stringifyPattern(pattern))
|
|
93
|
+
|
|
94
|
+
let stringified = `anyOf(${stringifiedMembers.join(', ')})`
|
|
95
|
+
|
|
96
|
+
if (stringified.length > this.stringifiedLengthLimit && stringifiedMembers.length > 2) {
|
|
97
|
+
stringified = `anyOf(${stringifiedMembers[0]}, ${stringifiedMembers[1]}, …, ${stringifiedMembers[stringifiedMembers.length - 1]})`
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
return stringified
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
case 'notAnyOfChars': {
|
|
104
|
+
const stringifiedMembers = pattern.members.map((pattern) => this.stringifyPattern(pattern))
|
|
105
|
+
|
|
106
|
+
return `notAnyOfChars(${stringifiedMembers.join(', ')})`
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
case 'followedBy': {
|
|
110
|
+
return `followedBy(${this.stringifyPattern(pattern.content)})`
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
case 'notFollowedBy': {
|
|
114
|
+
return `notFollowedBy(${this.stringifyPattern(pattern.content)})`
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
case 'precededBy': {
|
|
118
|
+
return `precededBy(${this.stringifyPattern(pattern.content)})`
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
case 'notPrecededBy': {
|
|
122
|
+
return `notPrecededBy(${this.stringifyPattern(pattern.content)})`
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
case 'sameAs': {
|
|
126
|
+
const captureGroupNameOrIndex = pattern.captureGroupNameOrIndex
|
|
127
|
+
|
|
128
|
+
if (typeof captureGroupNameOrIndex === 'string') {
|
|
129
|
+
return `sameAs('${captureGroupNameOrIndex}')`
|
|
130
|
+
} else {
|
|
131
|
+
return `sameAs(${captureGroupNameOrIndex})`
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
default: {
|
|
136
|
+
throw new Error(`Unrecognized pattern type: ${(pattern as any).type}.`)
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
stringifyPatternStringLiteral(text: string): string {
|
|
142
|
+
let result = ''
|
|
143
|
+
|
|
144
|
+
for (const char of text) {
|
|
145
|
+
const escapedChar = escapeChar(char)
|
|
146
|
+
|
|
147
|
+
result += escapedChar
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
return `'${result}'`
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
stringifySpecialToken(token: SpecialToken): string {
|
|
154
|
+
if (token.name === 'charRange') {
|
|
155
|
+
const dashIndex = token.rawRegExp.indexOf('-')
|
|
156
|
+
|
|
157
|
+
if (dashIndex > 0 && dashIndex < token.rawRegExp.length - 1) {
|
|
158
|
+
const startChar = token.rawRegExp.substring(0, dashIndex).replace(/^\\/, '')
|
|
159
|
+
|
|
160
|
+
if (startChar.length > 0) {
|
|
161
|
+
return `charRange('${startChar}', '${token.rawRegExp.substring(dashIndex + 1)}')`
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
} else if (token.name !== undefined) {
|
|
165
|
+
return token.name
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
return token.rawRegExp
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
truncateString(str: string, lengthLimit: number): string {
|
|
172
|
+
if (str.length <= lengthLimit) {
|
|
173
|
+
return str
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
return `${str.substring(0, lengthLimit - 1)}…`
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
export function escapeChar(char: string): string {
|
|
181
|
+
const charMapping: Record<string, string> = {
|
|
182
|
+
//'\\': '\\\\',
|
|
183
|
+
"'": "\\'",
|
|
184
|
+
'\n': '\\n',
|
|
185
|
+
'\r': '\\r',
|
|
186
|
+
'\t': '\\t',
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
const mappedChar = charMapping[char]
|
|
190
|
+
|
|
191
|
+
if (mappedChar !== undefined) {
|
|
192
|
+
return mappedChar
|
|
193
|
+
} else {
|
|
194
|
+
return char
|
|
195
|
+
}
|
|
196
|
+
}
|
|
@@ -1,19 +1,35 @@
|
|
|
1
1
|
import { isNumber } from '../utilities/Utilities.js'
|
|
2
|
-
import { Grammar, GrammarElement, Terminal } from './Grammar.js'
|
|
2
|
+
import { Grammar, GrammarElement, Terminal, type Nonterminal } from './Grammar.js'
|
|
3
|
+
import { type FailedMatch, ParseError } from './ParseError.js'
|
|
4
|
+
|
|
5
|
+
//////////////////////////////////////////////////////////////////////////////////////////////
|
|
6
|
+
// Main parser function
|
|
7
|
+
//////////////////////////////////////////////////////////////////////////////////////////////
|
|
8
|
+
export function parse(inputString: string, grammar: Grammar<any>, options?: TopDownParserOptions) {
|
|
9
|
+
options = { ...options }
|
|
3
10
|
|
|
4
|
-
export function parse(inputString: string, grammar: Grammar<any>) {
|
|
5
11
|
const inputLength = inputString.length
|
|
6
12
|
|
|
7
|
-
|
|
13
|
+
const nonterminalStack: Nonterminal[] = []
|
|
14
|
+
|
|
15
|
+
let bestFailedMatches: FailedMatch[] = []
|
|
8
16
|
let bestFailedMatchesOffset = -1
|
|
9
17
|
|
|
18
|
+
const cacheKeyOffsetMultiplier = grammar.maxCacheId + 1
|
|
19
|
+
const parseResultsCache = new Map<number, ParseResult | null>()
|
|
20
|
+
|
|
10
21
|
function updateBestFailedMatchesIfNeeded(terminal: Terminal, startOffset: number) {
|
|
11
22
|
if (startOffset >= bestFailedMatchesOffset) {
|
|
23
|
+
const failedMatch: FailedMatch = {
|
|
24
|
+
terminal,
|
|
25
|
+
productionStack: [...nonterminalStack]
|
|
26
|
+
}
|
|
27
|
+
|
|
12
28
|
if (startOffset > bestFailedMatchesOffset) {
|
|
13
29
|
bestFailedMatchesOffset = startOffset
|
|
14
|
-
bestFailedMatches = [
|
|
30
|
+
bestFailedMatches = [failedMatch]
|
|
15
31
|
} else {
|
|
16
|
-
bestFailedMatches.push(
|
|
32
|
+
bestFailedMatches.push(failedMatch)
|
|
17
33
|
}
|
|
18
34
|
}
|
|
19
35
|
}
|
|
@@ -26,13 +42,9 @@ export function parse(inputString: string, grammar: Grammar<any>) {
|
|
|
26
42
|
}
|
|
27
43
|
}
|
|
28
44
|
|
|
29
|
-
const offsetMultiplier = grammar.maxCacheId + 1
|
|
30
|
-
|
|
31
|
-
const parseResultsCache = new Map<number, ParseResult | null>()
|
|
32
|
-
|
|
33
45
|
function tryParseCached(grammarElement: GrammarElement, startOffset: number): ParseResult | null {
|
|
34
46
|
const cacheId = grammarElement.cacheId!
|
|
35
|
-
const cacheKey = (startOffset *
|
|
47
|
+
const cacheKey = (startOffset * cacheKeyOffsetMultiplier) + cacheId
|
|
36
48
|
|
|
37
49
|
if (parseResultsCache.has(cacheKey)) {
|
|
38
50
|
return parseResultsCache.get(cacheKey)!
|
|
@@ -94,10 +106,6 @@ export function parse(inputString: string, grammar: Grammar<any>) {
|
|
|
94
106
|
|
|
95
107
|
if (groupsIndices.groups) {
|
|
96
108
|
namedGroupIndicesIdentifiers = Object.keys(groupsIndices.groups)
|
|
97
|
-
|
|
98
|
-
if (namedGroupIndicesIdentifiers.length !== groupsIndices.length - 1) {
|
|
99
|
-
throw new Error(`The regular expression /${grammarElement.regExp.source}/ contains a combination of named and unnamed groups. Due to limitations of the JavaScript RegExp engine, it is impossible to reliably identify the ordering of this combination, please use either all unnamed or named groups, but not both.`)
|
|
100
|
-
}
|
|
101
109
|
}
|
|
102
110
|
|
|
103
111
|
const children: ParseTreeNode[] = []
|
|
@@ -133,26 +141,44 @@ export function parse(inputString: string, grammar: Grammar<any>) {
|
|
|
133
141
|
}
|
|
134
142
|
|
|
135
143
|
case 'Nonterminal': {
|
|
144
|
+
nonterminalStack.push(grammarElement)
|
|
145
|
+
|
|
136
146
|
const result = tryParse(grammarElement.content, startOffset)
|
|
137
147
|
|
|
148
|
+
nonterminalStack.pop()
|
|
149
|
+
|
|
138
150
|
if (result === null) {
|
|
139
151
|
return null
|
|
140
152
|
}
|
|
141
153
|
|
|
142
|
-
|
|
143
|
-
name: grammarElement.name,
|
|
144
|
-
startOffset,
|
|
145
|
-
endOffset: result.endOffset,
|
|
146
|
-
sourceText: inputString.substring(startOffset, result.endOffset),
|
|
147
|
-
children: result.nodes,
|
|
148
|
-
}
|
|
154
|
+
const grammarElementName = grammarElement.name
|
|
149
155
|
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
156
|
+
if (grammarElement.unwrapped) {
|
|
157
|
+
const newResult: ParseResult = {
|
|
158
|
+
endOffset: result.endOffset,
|
|
159
|
+
nodes: result.nodes
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
return newResult
|
|
163
|
+
} else {
|
|
164
|
+
const newNode: ParseTreeNode = {
|
|
165
|
+
name: grammarElementName,
|
|
166
|
+
|
|
167
|
+
startOffset,
|
|
168
|
+
endOffset: result.endOffset,
|
|
169
|
+
|
|
170
|
+
sourceText: inputString.substring(startOffset, result.endOffset),
|
|
154
171
|
|
|
155
|
-
|
|
172
|
+
children: result.nodes,
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
const newResult: ParseResult = {
|
|
176
|
+
endOffset: result.endOffset,
|
|
177
|
+
nodes: [newNode]
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
return newResult
|
|
181
|
+
}
|
|
156
182
|
}
|
|
157
183
|
|
|
158
184
|
case 'Sequence': {
|
|
@@ -252,36 +278,22 @@ export function parse(inputString: string, grammar: Grammar<any>) {
|
|
|
252
278
|
if (result && result.endOffset >= inputLength) {
|
|
253
279
|
return result.nodes ?? []
|
|
254
280
|
} else {
|
|
255
|
-
|
|
256
|
-
const possibleMatches = bestFailedMatches.map(match => {
|
|
257
|
-
if (match.type === 'StringTerminal') {
|
|
258
|
-
return `'${match.content}'`
|
|
259
|
-
} else if (match.type === 'PatternTerminal') {
|
|
260
|
-
return `/${match.regExp.source}/`
|
|
261
|
-
} else {
|
|
262
|
-
throw new Error(`Invalid match type: '${(match as any).type}'`)
|
|
263
|
-
}
|
|
264
|
-
})
|
|
265
|
-
|
|
266
|
-
const possibleMatchesWithoutDuplicates = [...(new Set(possibleMatches))]
|
|
267
|
-
|
|
268
|
-
let possibleMatchesString: string
|
|
281
|
+
const failureOffset = bestFailedMatches.length > 0 ? bestFailedMatchesOffset : (result?.endOffset ?? 0)
|
|
269
282
|
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
} else {
|
|
273
|
-
possibleMatchesString = possibleMatchesWithoutDuplicates[0]
|
|
274
|
-
}
|
|
275
|
-
|
|
276
|
-
throw new Error(`Failed parsing the input text. Expected ${possibleMatchesString} at position ${bestFailedMatchesOffset}.`)
|
|
283
|
+
if (bestFailedMatches.length > 0) {
|
|
284
|
+
throw ParseError.createFailedParseError(inputString, failureOffset, bestFailedMatches)
|
|
277
285
|
} else {
|
|
278
286
|
const lastNode = result?.nodes?.[result.nodes.length - 1]
|
|
287
|
+
const parsedLength = lastNode?.endOffset ?? result?.endOffset ?? 0
|
|
279
288
|
|
|
280
|
-
throw
|
|
289
|
+
throw ParseError.createIncompleteParseError(inputString, failureOffset, parsedLength)
|
|
281
290
|
}
|
|
282
291
|
}
|
|
283
292
|
}
|
|
284
293
|
|
|
294
|
+
//////////////////////////////////////////////////////////////////////////////////////////////
|
|
295
|
+
// Types
|
|
296
|
+
//////////////////////////////////////////////////////////////////////////////////////////////
|
|
285
297
|
export interface ParseResult {
|
|
286
298
|
endOffset: number
|
|
287
299
|
nodes: ParseTreeNode[] | undefined
|
|
@@ -289,8 +301,14 @@ export interface ParseResult {
|
|
|
289
301
|
|
|
290
302
|
export interface ParseTreeNode {
|
|
291
303
|
name: string
|
|
304
|
+
|
|
292
305
|
startOffset: number
|
|
293
306
|
endOffset: number
|
|
307
|
+
|
|
294
308
|
sourceText: string
|
|
295
|
-
|
|
309
|
+
|
|
310
|
+
children?: ParseTreeNode[]
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
export interface TopDownParserOptions {
|
|
296
314
|
}
|
package/src/tests/Test.ts
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import { Timer } from '../utilities/Timer.js'
|
|
2
2
|
import { jsonSample1, jsonSample2 } from './test-data/TestData.js'
|
|
3
3
|
import { anyOf, buildGrammar } from '../exports/Exports.js'
|
|
4
|
-
import { JsonGrammar } from './test-grammars/JsonGrammar.js'
|
|
5
|
-
import { XmlGrammar } from './test-grammars/XmlGrammar.js'
|
|
6
|
-
import { RegExpGrammar } from './test-grammars/RegExpGrammar.js'
|
|
4
|
+
import { JsonGrammar, jsonGrammarUnwrappedNonterminalNames } from './test-grammars/JsonGrammar.js'
|
|
5
|
+
import { XmlGrammar, xmlGrammarUnwrappedNonterminalNames } from './test-grammars/XmlGrammar.js'
|
|
6
|
+
import { RegExpGrammar, regExpGrammarUnwrappedNonterminalNames } from './test-grammars/RegExpGrammar.js'
|
|
7
7
|
import { writeFile } from 'fs/promises'
|
|
8
8
|
import { SimpleTestGrammar1 } from './test-grammars/SimpleTestGrammar1.js'
|
|
9
9
|
|
|
@@ -30,7 +30,9 @@ function testBasic() {
|
|
|
30
30
|
function testJsonParser() {
|
|
31
31
|
const jsonString = jsonSample1
|
|
32
32
|
|
|
33
|
-
const grammar = buildGrammar(JsonGrammar, 'expression'
|
|
33
|
+
const grammar = buildGrammar(JsonGrammar, 'expression', {
|
|
34
|
+
unwrappedNonterminalNames: jsonGrammarUnwrappedNonterminalNames
|
|
35
|
+
})
|
|
34
36
|
|
|
35
37
|
const iterations = 1000
|
|
36
38
|
|
|
@@ -52,7 +54,7 @@ function testJsonParser() {
|
|
|
52
54
|
log(JSON.stringify(result1, undefined, 4))
|
|
53
55
|
}
|
|
54
56
|
|
|
55
|
-
function testXmlParser() {
|
|
57
|
+
async function testXmlParser() {
|
|
56
58
|
const xmlString = `
|
|
57
59
|
<!DOCTYPE web-app>
|
|
58
60
|
|
|
@@ -60,36 +62,33 @@ function testXmlParser() {
|
|
|
60
62
|
<header>Adobe SVG Viewer</header>
|
|
61
63
|
<item action="Open" id="Open">Open</item>
|
|
62
64
|
<item action="OpenNew" id="OpenNew">Open New</item>
|
|
63
|
-
<separator/>
|
|
64
|
-
<item action="ZoomIn" id="ZoomIn">Zoom In</item>
|
|
65
|
-
<item action="ZoomOut" id="ZoomOut">Zoom Out</item>
|
|
66
|
-
<separator/>
|
|
67
|
-
<item action="Quality" id="Quality">Quality</item>
|
|
68
|
-
<item action="Pause" id="Pause">Pause</item>
|
|
69
|
-
<item action="Mute" id="Mute">Mute</item>
|
|
70
|
-
<separator/>
|
|
71
|
-
<item action="Find" id="Find">Find...</item>
|
|
72
|
-
<item action="FindAgain" id="FindAgain">Find Again</item>
|
|
73
|
-
<item action="Copy" id="Copy">Copy</item>
|
|
74
65
|
</menu>
|
|
75
|
-
|
|
76
66
|
`
|
|
77
67
|
// Build the grammar. 'document' is the starting production
|
|
78
|
-
const grammar = buildGrammar(XmlGrammar, 'document'
|
|
68
|
+
const grammar = buildGrammar(XmlGrammar, 'document', {
|
|
69
|
+
unwrappedNonterminalNames: xmlGrammarUnwrappedNonterminalNames
|
|
70
|
+
})
|
|
79
71
|
|
|
80
72
|
// Parse the XML string
|
|
81
73
|
const parseTree = grammar.parse(xmlString)
|
|
82
74
|
|
|
83
|
-
|
|
75
|
+
const parseTreeJson = JSON.stringify(parseTree, undefined, 4)
|
|
76
|
+
|
|
77
|
+
log(parseTreeJson)
|
|
78
|
+
|
|
79
|
+
await writeFile('out/out.json', parseTreeJson)
|
|
84
80
|
}
|
|
85
81
|
|
|
86
82
|
async function testRegExpParser() {
|
|
87
|
-
const regExpString = /^([+]?[1]?(1 )?[-.+]?\(?\d{1}[- .+]*\d{1}[- .+]*\d{1}\)?[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1})$/.source
|
|
83
|
+
//const regExpString = /^([+]?[1]?(1 )?[-.+]?\(?\d{1}[- .+]*\d{1}[- .+]*\d{1}\)?[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1})$/.source
|
|
88
84
|
//const regExpString = /^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$/.source
|
|
89
85
|
//const regExpString = /(?=.*[!@#$%^&*])^mongodb:\/\/(?<user>[a-zA-Z0-9]+):(?<pass>[a-zA-Z0-9!@#$%^&*]{8,})@(?<host>[a-z0-9.-]+):(?<port>\d{2,5})$/.source
|
|
90
|
-
|
|
86
|
+
const regExpString = /^(abcd)*ef+g/.source
|
|
87
|
+
//const regExpString = '(abcd)(aa))'
|
|
91
88
|
|
|
92
|
-
const grammar = buildGrammar(RegExpGrammar, 'root'
|
|
89
|
+
const grammar = buildGrammar(RegExpGrammar, 'root', {
|
|
90
|
+
unwrappedNonterminalNames: regExpGrammarUnwrappedNonterminalNames
|
|
91
|
+
})
|
|
93
92
|
|
|
94
93
|
const parseTree = grammar.parse(regExpString)
|
|
95
94
|
|
|
@@ -131,8 +130,11 @@ function test1() {
|
|
|
131
130
|
}
|
|
132
131
|
|
|
133
132
|
|
|
134
|
-
//
|
|
133
|
+
//testParserError1()
|
|
134
|
+
//testParserError2()
|
|
135
135
|
|
|
136
|
-
|
|
136
|
+
//testJsonParser()
|
|
137
|
+
testXmlParser()
|
|
138
|
+
//testRegExpParser()
|
|
137
139
|
|
|
138
140
|
//test1()
|
|
@@ -11,7 +11,7 @@ export class JsonGrammar {
|
|
|
11
11
|
this.arrayExpression
|
|
12
12
|
)
|
|
13
13
|
|
|
14
|
-
stringLiteral = G.pattern([
|
|
14
|
+
stringLiteral = () => G.pattern([
|
|
15
15
|
zeroOrMoreWhitespace,
|
|
16
16
|
|
|
17
17
|
'"',
|
|
@@ -27,7 +27,7 @@ export class JsonGrammar {
|
|
|
27
27
|
zeroOrMoreWhitespace,
|
|
28
28
|
])
|
|
29
29
|
|
|
30
|
-
numberLiteral = G.pattern([
|
|
30
|
+
numberLiteral = () => G.pattern([
|
|
31
31
|
zeroOrMoreWhitespace,
|
|
32
32
|
|
|
33
33
|
R.captureAs('value', [
|
|
@@ -46,7 +46,7 @@ export class JsonGrammar {
|
|
|
46
46
|
zeroOrMoreWhitespace,
|
|
47
47
|
])
|
|
48
48
|
|
|
49
|
-
booleanLiteral = G.pattern([
|
|
49
|
+
booleanLiteral = () => G.pattern([
|
|
50
50
|
zeroOrMoreWhitespace,
|
|
51
51
|
|
|
52
52
|
R.captureAs('value',
|
|
@@ -56,7 +56,7 @@ export class JsonGrammar {
|
|
|
56
56
|
zeroOrMoreWhitespace,
|
|
57
57
|
])
|
|
58
58
|
|
|
59
|
-
nullLiteral = G.pattern([
|
|
59
|
+
nullLiteral = () => G.pattern([
|
|
60
60
|
zeroOrMoreWhitespace,
|
|
61
61
|
|
|
62
62
|
'null',
|
|
@@ -92,14 +92,14 @@ export class JsonGrammar {
|
|
|
92
92
|
this.closingSquareBracket
|
|
93
93
|
]
|
|
94
94
|
|
|
95
|
-
openingCurlyBrace = createPatternWithClearedWhitespace('{')
|
|
96
|
-
closingCurlyBrace = createPatternWithClearedWhitespace('}')
|
|
95
|
+
openingCurlyBrace = () => createPatternWithClearedWhitespace('{')
|
|
96
|
+
closingCurlyBrace = () => createPatternWithClearedWhitespace('}')
|
|
97
97
|
|
|
98
|
-
openingSquareBracket = createPatternWithClearedWhitespace('[')
|
|
99
|
-
closingSquareBracket = createPatternWithClearedWhitespace(']')
|
|
98
|
+
openingSquareBracket = () => createPatternWithClearedWhitespace('[')
|
|
99
|
+
closingSquareBracket = () => createPatternWithClearedWhitespace(']')
|
|
100
100
|
|
|
101
|
-
comma = createPatternWithClearedWhitespace(',')
|
|
102
|
-
colons = createPatternWithClearedWhitespace(':')
|
|
101
|
+
comma = () => createPatternWithClearedWhitespace(',')
|
|
102
|
+
colons = () => createPatternWithClearedWhitespace(':')
|
|
103
103
|
}
|
|
104
104
|
|
|
105
105
|
function createPatternWithClearedWhitespace(subpattern: R.Pattern) {
|
|
@@ -113,3 +113,21 @@ function createPatternWithClearedWhitespace(subpattern: R.Pattern) {
|
|
|
113
113
|
}
|
|
114
114
|
|
|
115
115
|
const zeroOrMoreWhitespace = R.zeroOrMore(R.whitespace)
|
|
116
|
+
|
|
117
|
+
//////////////////////////////////////////////////////////////////////////////////////////////
|
|
118
|
+
// Wrapped nonterminal names
|
|
119
|
+
//////////////////////////////////////////////////////////////////////////////////////////////
|
|
120
|
+
export const jsonGrammarUnwrappedNonterminalNames: G.GrammarNonterminalNames<JsonGrammar> = [
|
|
121
|
+
//'stringLiteral',
|
|
122
|
+
//'numberLiteral',
|
|
123
|
+
//'booleanLiteral',
|
|
124
|
+
//'nullLiteral',
|
|
125
|
+
|
|
126
|
+
'openingCurlyBrace',
|
|
127
|
+
'closingCurlyBrace',
|
|
128
|
+
'openingSquareBracket',
|
|
129
|
+
'closingSquareBracket',
|
|
130
|
+
|
|
131
|
+
'comma',
|
|
132
|
+
'colons',
|
|
133
|
+
]
|