grammar-composer 0.3.1 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +144 -89
- package/dist/exports/Exports.d.ts +2 -0
- package/dist/exports/Exports.d.ts.map +1 -1
- package/dist/exports/Exports.js +2 -0
- package/dist/exports/Exports.js.map +1 -1
- package/dist/parser-generator/Grammar.d.ts +12 -5
- package/dist/parser-generator/Grammar.d.ts.map +1 -1
- package/dist/parser-generator/Grammar.js +81 -263
- package/dist/parser-generator/Grammar.js.map +1 -1
- package/dist/parser-generator/ParseError.d.ts +37 -0
- package/dist/parser-generator/ParseError.d.ts.map +1 -0
- package/dist/parser-generator/ParseError.js +226 -0
- package/dist/parser-generator/ParseError.js.map +1 -0
- package/dist/parser-generator/StaticAnalysis.d.ts +6 -0
- package/dist/parser-generator/StaticAnalysis.d.ts.map +1 -0
- package/dist/parser-generator/StaticAnalysis.js +283 -0
- package/dist/parser-generator/StaticAnalysis.js.map +1 -0
- package/dist/parser-generator/TerminalToText.d.ts +15 -0
- package/dist/parser-generator/TerminalToText.d.ts.map +1 -0
- package/dist/parser-generator/TerminalToText.js +158 -0
- package/dist/parser-generator/TerminalToText.js.map +1 -0
- package/dist/parser-generator/TopDownParser.d.ts +4 -2
- package/dist/parser-generator/TopDownParser.d.ts.map +1 -1
- package/dist/parser-generator/TopDownParser.js +52 -52
- package/dist/parser-generator/TopDownParser.js.map +1 -1
- package/dist/tests/Test.js +36 -26
- package/dist/tests/Test.js.map +1 -1
- package/dist/tests/test-grammars/JsonGrammar.d.ts +12 -11
- package/dist/tests/test-grammars/JsonGrammar.d.ts.map +1 -1
- package/dist/tests/test-grammars/JsonGrammar.js +25 -10
- package/dist/tests/test-grammars/JsonGrammar.js.map +1 -1
- package/dist/tests/test-grammars/RegExpGrammar.d.ts +34 -28
- package/dist/tests/test-grammars/RegExpGrammar.d.ts.map +1 -1
- package/dist/tests/test-grammars/RegExpGrammar.js +104 -60
- package/dist/tests/test-grammars/RegExpGrammar.js.map +1 -1
- package/dist/tests/test-grammars/SimpleTestGrammar1.d.ts +5 -0
- package/dist/tests/test-grammars/SimpleTestGrammar1.d.ts.map +1 -0
- package/dist/tests/test-grammars/SimpleTestGrammar1.js +7 -0
- package/dist/tests/test-grammars/SimpleTestGrammar1.js.map +1 -0
- package/dist/tests/test-grammars/XmlGrammar.d.ts +11 -10
- package/dist/tests/test-grammars/XmlGrammar.d.ts.map +1 -1
- package/dist/tests/test-grammars/XmlGrammar.js +19 -8
- package/dist/tests/test-grammars/XmlGrammar.js.map +1 -1
- package/dist/utilities/LineAndColumn.d.ts +7 -0
- package/dist/utilities/LineAndColumn.d.ts.map +1 -0
- package/dist/utilities/LineAndColumn.js +52 -0
- package/dist/utilities/LineAndColumn.js.map +1 -0
- package/dist/utilities/Utilities.d.ts +7 -7
- package/dist/utilities/Utilities.d.ts.map +1 -1
- package/dist/utilities/Utilities.js.map +1 -1
- package/package.json +2 -2
- package/src/exports/Exports.ts +2 -0
- package/src/parser-generator/Grammar.ts +120 -308
- package/src/parser-generator/ParseError.ts +337 -0
- package/src/parser-generator/StaticAnalysis.ts +340 -0
- package/src/parser-generator/TerminalToText.ts +196 -0
- package/src/parser-generator/TopDownParser.ts +76 -64
- package/src/tests/Test.ts +44 -28
- package/src/tests/test-grammars/JsonGrammar.ts +28 -10
- package/src/tests/test-grammars/RegExpGrammar.ts +193 -96
- package/src/tests/test-grammars/SimpleTestGrammar1.ts +10 -0
- package/src/tests/test-grammars/XmlGrammar.ts +20 -8
- package/src/utilities/LineAndColumn.ts +61 -0
- package/src/utilities/Utilities.ts +7 -7
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
import type { Pattern, SpecialToken } from 'regexp-composer'
|
|
2
|
+
import type { PatternTerminal, StringTerminal, Terminal } from './Grammar.js'
|
|
3
|
+
|
|
4
|
+
/////////////////////////////////////////////////////////////////////////////////////////////////
|
|
5
|
+
// Terminal to text conversion
|
|
6
|
+
/////////////////////////////////////////////////////////////////////////////////////////////////
|
|
7
|
+
export class TerminalToText {
|
|
8
|
+
constructor(private stringifiedLengthLimit: number) {
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
stringifyTerminal(terminal: Terminal): string {
|
|
12
|
+
if (terminal.type === 'StringTerminal') {
|
|
13
|
+
return this.stringifyStringTerminal(terminal)
|
|
14
|
+
} else {
|
|
15
|
+
return this.stringifyPatternTerminal(terminal)
|
|
16
|
+
}
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
stringifyStringTerminal(terminal: StringTerminal): string {
|
|
20
|
+
return this.stringifyPatternStringLiteral(terminal.content)
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
stringifyPatternTerminal(terminal: PatternTerminal) {
|
|
24
|
+
let pattern = terminal.pattern
|
|
25
|
+
|
|
26
|
+
// Pattern terminals are automatically anchored at the start of the current parse
|
|
27
|
+
// position. This anchor is an implementation detail, so remove it.
|
|
28
|
+
if (Array.isArray(pattern) && pattern.length > 1) {
|
|
29
|
+
const firstElement = pattern[0]
|
|
30
|
+
|
|
31
|
+
if (!Array.isArray(firstElement) && typeof firstElement !== 'string' &&
|
|
32
|
+
firstElement.type === 'specialToken' && firstElement.name === 'inputStart') {
|
|
33
|
+
|
|
34
|
+
pattern = pattern.slice(1)
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
const stringifiedPattern = this.stringifyPattern(pattern)
|
|
39
|
+
const truncatedStringifiedPattern = this.truncateString(stringifiedPattern, this.stringifiedLengthLimit)
|
|
40
|
+
|
|
41
|
+
return truncatedStringifiedPattern
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
stringifyPattern(pattern: Pattern): string {
|
|
45
|
+
if (typeof pattern === 'string') {
|
|
46
|
+
return this.stringifyPatternStringLiteral(pattern)
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
if (Array.isArray(pattern)) {
|
|
50
|
+
return `[${pattern.map((pattern) => this.stringifyPattern(pattern)).join(', ')}]`
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
switch (pattern.type) {
|
|
54
|
+
case 'specialToken': {
|
|
55
|
+
return this.stringifySpecialToken(pattern)
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
case 'possibly': {
|
|
59
|
+
return `possibly(${this.stringifyPattern(pattern.content)})`
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
case 'zeroOrMore': {
|
|
63
|
+
return `zeroOrMore(${this.stringifyPattern(pattern.content)})`
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
case 'oneOrMore': {
|
|
67
|
+
return `oneOrMore(${this.stringifyPattern(pattern.content)})`
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
case 'repeated': {
|
|
71
|
+
const { minCount, maxCount } = pattern
|
|
72
|
+
const stringifiedContent = this.stringifyPattern(pattern.content)
|
|
73
|
+
|
|
74
|
+
if (minCount === maxCount) {
|
|
75
|
+
return `repeated(${minCount}, ${stringifiedContent})`
|
|
76
|
+
} else if (maxCount === Number.POSITIVE_INFINITY) {
|
|
77
|
+
return `repeated([${minCount}, ...], ${stringifiedContent})`
|
|
78
|
+
} else {
|
|
79
|
+
return `repeated([${minCount}, ${maxCount}], ${stringifiedContent})`
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
case 'capture': {
|
|
84
|
+
if (pattern.name !== undefined) {
|
|
85
|
+
return pattern.name
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
return `(${this.stringifyPattern(pattern.content)})`
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
case 'anyOf': {
|
|
92
|
+
const stringifiedMembers = pattern.members.map((pattern) => this.stringifyPattern(pattern))
|
|
93
|
+
|
|
94
|
+
let stringified = `anyOf(${stringifiedMembers.join(', ')})`
|
|
95
|
+
|
|
96
|
+
if (stringified.length > this.stringifiedLengthLimit && stringifiedMembers.length > 2) {
|
|
97
|
+
stringified = `anyOf(${stringifiedMembers[0]}, ${stringifiedMembers[1]}, …, ${stringifiedMembers[stringifiedMembers.length - 1]})`
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
return stringified
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
case 'notAnyOfChars': {
|
|
104
|
+
const stringifiedMembers = pattern.members.map((pattern) => this.stringifyPattern(pattern))
|
|
105
|
+
|
|
106
|
+
return `notAnyOfChars(${stringifiedMembers.join(', ')})`
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
case 'followedBy': {
|
|
110
|
+
return `followedBy(${this.stringifyPattern(pattern.content)})`
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
case 'notFollowedBy': {
|
|
114
|
+
return `notFollowedBy(${this.stringifyPattern(pattern.content)})`
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
case 'precededBy': {
|
|
118
|
+
return `precededBy(${this.stringifyPattern(pattern.content)})`
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
case 'notPrecededBy': {
|
|
122
|
+
return `notPrecededBy(${this.stringifyPattern(pattern.content)})`
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
case 'sameAs': {
|
|
126
|
+
const captureGroupNameOrIndex = pattern.captureGroupNameOrIndex
|
|
127
|
+
|
|
128
|
+
if (typeof captureGroupNameOrIndex === 'string') {
|
|
129
|
+
return `sameAs('${captureGroupNameOrIndex}')`
|
|
130
|
+
} else {
|
|
131
|
+
return `sameAs(${captureGroupNameOrIndex})`
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
default: {
|
|
136
|
+
throw new Error(`Unrecognized pattern type: ${(pattern as any).type}.`)
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
stringifyPatternStringLiteral(text: string): string {
|
|
142
|
+
let result = ''
|
|
143
|
+
|
|
144
|
+
for (const char of text) {
|
|
145
|
+
const escapedChar = escapeChar(char)
|
|
146
|
+
|
|
147
|
+
result += escapedChar
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
return `'${result}'`
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
stringifySpecialToken(token: SpecialToken): string {
|
|
154
|
+
if (token.name === 'charRange') {
|
|
155
|
+
const dashIndex = token.rawRegExp.indexOf('-')
|
|
156
|
+
|
|
157
|
+
if (dashIndex > 0 && dashIndex < token.rawRegExp.length - 1) {
|
|
158
|
+
const startChar = token.rawRegExp.substring(0, dashIndex).replace(/^\\/, '')
|
|
159
|
+
|
|
160
|
+
if (startChar.length > 0) {
|
|
161
|
+
return `charRange('${startChar}', '${token.rawRegExp.substring(dashIndex + 1)}')`
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
} else if (token.name !== undefined) {
|
|
165
|
+
return token.name
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
return token.rawRegExp
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
truncateString(str: string, lengthLimit: number): string {
|
|
172
|
+
if (str.length <= lengthLimit) {
|
|
173
|
+
return str
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
return `${str.substring(0, lengthLimit - 1)}…`
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
export function escapeChar(char: string): string {
|
|
181
|
+
const charMapping: Record<string, string> = {
|
|
182
|
+
//'\\': '\\\\',
|
|
183
|
+
"'": "\\'",
|
|
184
|
+
'\n': '\\n',
|
|
185
|
+
'\r': '\\r',
|
|
186
|
+
'\t': '\\t',
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
const mappedChar = charMapping[char]
|
|
190
|
+
|
|
191
|
+
if (mappedChar !== undefined) {
|
|
192
|
+
return mappedChar
|
|
193
|
+
} else {
|
|
194
|
+
return char
|
|
195
|
+
}
|
|
196
|
+
}
|
|
@@ -1,54 +1,60 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { isNumber } from '../utilities/Utilities.js'
|
|
2
|
+
import { Grammar, GrammarElement, Terminal, type Nonterminal } from './Grammar.js'
|
|
3
|
+
import { type FailedMatch, ParseError } from './ParseError.js'
|
|
4
|
+
|
|
5
|
+
//////////////////////////////////////////////////////////////////////////////////////////////
|
|
6
|
+
// Main parser function
|
|
7
|
+
//////////////////////////////////////////////////////////////////////////////////////////////
|
|
8
|
+
export function parse(inputString: string, grammar: Grammar<any>, options?: TopDownParserOptions) {
|
|
9
|
+
options = { ...options }
|
|
2
10
|
|
|
3
|
-
export function parse(inputString: string, grammar: Grammar<any>) {
|
|
4
11
|
const inputLength = inputString.length
|
|
5
12
|
|
|
6
|
-
|
|
13
|
+
const nonterminalStack: Nonterminal[] = []
|
|
14
|
+
|
|
15
|
+
let bestFailedMatches: FailedMatch[] = []
|
|
7
16
|
let bestFailedMatchesOffset = -1
|
|
8
17
|
|
|
18
|
+
const cacheKeyOffsetMultiplier = grammar.maxCacheId + 1
|
|
19
|
+
const parseResultsCache = new Map<number, ParseResult | null>()
|
|
20
|
+
|
|
9
21
|
function updateBestFailedMatchesIfNeeded(terminal: Terminal, startOffset: number) {
|
|
10
22
|
if (startOffset >= bestFailedMatchesOffset) {
|
|
23
|
+
const failedMatch: FailedMatch = {
|
|
24
|
+
terminal,
|
|
25
|
+
productionStack: [...nonterminalStack]
|
|
26
|
+
}
|
|
27
|
+
|
|
11
28
|
if (startOffset > bestFailedMatchesOffset) {
|
|
12
29
|
bestFailedMatchesOffset = startOffset
|
|
13
|
-
bestFailedMatches = [
|
|
30
|
+
bestFailedMatches = [failedMatch]
|
|
14
31
|
} else {
|
|
15
|
-
bestFailedMatches.push(
|
|
32
|
+
bestFailedMatches.push(failedMatch)
|
|
16
33
|
}
|
|
17
34
|
}
|
|
18
35
|
}
|
|
19
36
|
|
|
20
37
|
function tryParse(grammarElement: GrammarElement, startOffset: number): ParseResult | null {
|
|
21
|
-
if (grammarElement.
|
|
38
|
+
if (isNumber(grammarElement.cacheId)) {
|
|
22
39
|
return tryParseCached(grammarElement, startOffset)
|
|
23
40
|
} else {
|
|
24
41
|
return tryParseUncached(grammarElement, startOffset)
|
|
25
42
|
}
|
|
26
43
|
}
|
|
27
44
|
|
|
28
|
-
type Slot = Map<GrammarElement, ParseResult | null> | undefined
|
|
29
|
-
|
|
30
|
-
const cachedParseResults: Slot[] = new Array(inputLength)
|
|
31
|
-
|
|
32
45
|
function tryParseCached(grammarElement: GrammarElement, startOffset: number): ParseResult | null {
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
if (slot === undefined) {
|
|
36
|
-
slot = new Map<GrammarElement, ParseResult | null>()
|
|
46
|
+
const cacheId = grammarElement.cacheId!
|
|
47
|
+
const cacheKey = (startOffset * cacheKeyOffsetMultiplier) + cacheId
|
|
37
48
|
|
|
38
|
-
|
|
49
|
+
if (parseResultsCache.has(cacheKey)) {
|
|
50
|
+
return parseResultsCache.get(cacheKey)!
|
|
39
51
|
} else {
|
|
40
|
-
const
|
|
41
|
-
|
|
42
|
-
if (cachedResult !== undefined) {
|
|
43
|
-
return cachedResult
|
|
44
|
-
}
|
|
45
|
-
}
|
|
52
|
+
const parseResult = tryParseUncached(grammarElement, startOffset)
|
|
46
53
|
|
|
47
|
-
|
|
54
|
+
parseResultsCache.set(cacheKey, parseResult)
|
|
48
55
|
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
return parseResult
|
|
56
|
+
return parseResult
|
|
57
|
+
}
|
|
52
58
|
}
|
|
53
59
|
|
|
54
60
|
function tryParseUncached(grammarElement: GrammarElement, startOffset: number): ParseResult | null {
|
|
@@ -100,10 +106,6 @@ export function parse(inputString: string, grammar: Grammar<any>) {
|
|
|
100
106
|
|
|
101
107
|
if (groupsIndices.groups) {
|
|
102
108
|
namedGroupIndicesIdentifiers = Object.keys(groupsIndices.groups)
|
|
103
|
-
|
|
104
|
-
if (namedGroupIndicesIdentifiers.length !== groupsIndices.length - 1) {
|
|
105
|
-
throw new Error(`The regular expression /${grammarElement.regExp.source}/ contains a combination of named and unnamed groups. Due to limitations of the JavaScript RegExp engine, it is impossible to reliably identify the ordering of this combination, please use either all unnamed or named groups, but not both.`)
|
|
106
|
-
}
|
|
107
109
|
}
|
|
108
110
|
|
|
109
111
|
const children: ParseTreeNode[] = []
|
|
@@ -139,26 +141,44 @@ export function parse(inputString: string, grammar: Grammar<any>) {
|
|
|
139
141
|
}
|
|
140
142
|
|
|
141
143
|
case 'Nonterminal': {
|
|
144
|
+
nonterminalStack.push(grammarElement)
|
|
145
|
+
|
|
142
146
|
const result = tryParse(grammarElement.content, startOffset)
|
|
143
147
|
|
|
148
|
+
nonterminalStack.pop()
|
|
149
|
+
|
|
144
150
|
if (result === null) {
|
|
145
151
|
return null
|
|
146
152
|
}
|
|
147
153
|
|
|
148
|
-
|
|
149
|
-
name: grammarElement.name,
|
|
150
|
-
startOffset,
|
|
151
|
-
endOffset: result.endOffset,
|
|
152
|
-
sourceText: inputString.substring(startOffset, result.endOffset),
|
|
153
|
-
children: result.nodes,
|
|
154
|
-
}
|
|
154
|
+
const grammarElementName = grammarElement.name
|
|
155
155
|
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
156
|
+
if (grammarElement.unwrapped) {
|
|
157
|
+
const newResult: ParseResult = {
|
|
158
|
+
endOffset: result.endOffset,
|
|
159
|
+
nodes: result.nodes
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
return newResult
|
|
163
|
+
} else {
|
|
164
|
+
const newNode: ParseTreeNode = {
|
|
165
|
+
name: grammarElementName,
|
|
160
166
|
|
|
161
|
-
|
|
167
|
+
startOffset,
|
|
168
|
+
endOffset: result.endOffset,
|
|
169
|
+
|
|
170
|
+
sourceText: inputString.substring(startOffset, result.endOffset),
|
|
171
|
+
|
|
172
|
+
children: result.nodes,
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
const newResult: ParseResult = {
|
|
176
|
+
endOffset: result.endOffset,
|
|
177
|
+
nodes: [newNode]
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
return newResult
|
|
181
|
+
}
|
|
162
182
|
}
|
|
163
183
|
|
|
164
184
|
case 'Sequence': {
|
|
@@ -258,36 +278,22 @@ export function parse(inputString: string, grammar: Grammar<any>) {
|
|
|
258
278
|
if (result && result.endOffset >= inputLength) {
|
|
259
279
|
return result.nodes ?? []
|
|
260
280
|
} else {
|
|
261
|
-
|
|
262
|
-
const possibleMatches = bestFailedMatches.map(match => {
|
|
263
|
-
if (match.type === 'StringTerminal') {
|
|
264
|
-
return `'${match.content}'`
|
|
265
|
-
} else if (match.type === 'PatternTerminal') {
|
|
266
|
-
return `/${match.regExp.source}/`
|
|
267
|
-
} else {
|
|
268
|
-
throw new Error(`Invalid match type: '${(match as any).type}'`)
|
|
269
|
-
}
|
|
270
|
-
})
|
|
271
|
-
|
|
272
|
-
const possibleMatchesWithoutDuplicates = [...(new Set(possibleMatches))]
|
|
281
|
+
const failureOffset = bestFailedMatches.length > 0 ? bestFailedMatchesOffset : (result?.endOffset ?? 0)
|
|
273
282
|
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
if (possibleMatchesWithoutDuplicates.length > 1) {
|
|
277
|
-
possibleMatchesString = `any of ${possibleMatchesWithoutDuplicates.join(', ')}`
|
|
278
|
-
} else {
|
|
279
|
-
possibleMatchesString = possibleMatchesWithoutDuplicates[0]
|
|
280
|
-
}
|
|
281
|
-
|
|
282
|
-
throw new Error(`Failed parsing the input text. Expected ${possibleMatchesString} at position ${bestFailedMatchesOffset}.`)
|
|
283
|
+
if (bestFailedMatches.length > 0) {
|
|
284
|
+
throw ParseError.createFailedParseError(inputString, failureOffset, bestFailedMatches)
|
|
283
285
|
} else {
|
|
284
286
|
const lastNode = result?.nodes?.[result.nodes.length - 1]
|
|
287
|
+
const parsedLength = lastNode?.endOffset ?? result?.endOffset ?? 0
|
|
285
288
|
|
|
286
|
-
throw
|
|
289
|
+
throw ParseError.createIncompleteParseError(inputString, failureOffset, parsedLength)
|
|
287
290
|
}
|
|
288
291
|
}
|
|
289
292
|
}
|
|
290
293
|
|
|
294
|
+
//////////////////////////////////////////////////////////////////////////////////////////////
|
|
295
|
+
// Types
|
|
296
|
+
//////////////////////////////////////////////////////////////////////////////////////////////
|
|
291
297
|
export interface ParseResult {
|
|
292
298
|
endOffset: number
|
|
293
299
|
nodes: ParseTreeNode[] | undefined
|
|
@@ -295,8 +301,14 @@ export interface ParseResult {
|
|
|
295
301
|
|
|
296
302
|
export interface ParseTreeNode {
|
|
297
303
|
name: string
|
|
304
|
+
|
|
298
305
|
startOffset: number
|
|
299
306
|
endOffset: number
|
|
307
|
+
|
|
300
308
|
sourceText: string
|
|
301
|
-
|
|
309
|
+
|
|
310
|
+
children?: ParseTreeNode[]
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
export interface TopDownParserOptions {
|
|
302
314
|
}
|
package/src/tests/Test.ts
CHANGED
|
@@ -1,9 +1,11 @@
|
|
|
1
1
|
import { Timer } from '../utilities/Timer.js'
|
|
2
2
|
import { jsonSample1, jsonSample2 } from './test-data/TestData.js'
|
|
3
3
|
import { anyOf, buildGrammar } from '../exports/Exports.js'
|
|
4
|
-
import { JsonGrammar } from './test-grammars/JsonGrammar.js'
|
|
5
|
-
import { XmlGrammar } from './test-grammars/XmlGrammar.js'
|
|
6
|
-
import { RegExpGrammar } from './test-grammars/RegExpGrammar.js'
|
|
4
|
+
import { JsonGrammar, jsonGrammarUnwrappedNonterminalNames } from './test-grammars/JsonGrammar.js'
|
|
5
|
+
import { XmlGrammar, xmlGrammarUnwrappedNonterminalNames } from './test-grammars/XmlGrammar.js'
|
|
6
|
+
import { RegExpGrammar, regExpGrammarUnwrappedNonterminalNames } from './test-grammars/RegExpGrammar.js'
|
|
7
|
+
import { writeFile } from 'fs/promises'
|
|
8
|
+
import { SimpleTestGrammar1 } from './test-grammars/SimpleTestGrammar1.js'
|
|
7
9
|
|
|
8
10
|
const log = console.log
|
|
9
11
|
|
|
@@ -28,7 +30,9 @@ function testBasic() {
|
|
|
28
30
|
function testJsonParser() {
|
|
29
31
|
const jsonString = jsonSample1
|
|
30
32
|
|
|
31
|
-
const grammar = buildGrammar(JsonGrammar, 'expression'
|
|
33
|
+
const grammar = buildGrammar(JsonGrammar, 'expression', {
|
|
34
|
+
unwrappedNonterminalNames: jsonGrammarUnwrappedNonterminalNames
|
|
35
|
+
})
|
|
32
36
|
|
|
33
37
|
const iterations = 1000
|
|
34
38
|
|
|
@@ -50,7 +54,7 @@ function testJsonParser() {
|
|
|
50
54
|
log(JSON.stringify(result1, undefined, 4))
|
|
51
55
|
}
|
|
52
56
|
|
|
53
|
-
function testXmlParser() {
|
|
57
|
+
async function testXmlParser() {
|
|
54
58
|
const xmlString = `
|
|
55
59
|
<!DOCTYPE web-app>
|
|
56
60
|
|
|
@@ -58,34 +62,33 @@ function testXmlParser() {
|
|
|
58
62
|
<header>Adobe SVG Viewer</header>
|
|
59
63
|
<item action="Open" id="Open">Open</item>
|
|
60
64
|
<item action="OpenNew" id="OpenNew">Open New</item>
|
|
61
|
-
<separator/>
|
|
62
|
-
<item action="ZoomIn" id="ZoomIn">Zoom In</item>
|
|
63
|
-
<item action="ZoomOut" id="ZoomOut">Zoom Out</item>
|
|
64
|
-
<separator/>
|
|
65
|
-
<item action="Quality" id="Quality">Quality</item>
|
|
66
|
-
<item action="Pause" id="Pause">Pause</item>
|
|
67
|
-
<item action="Mute" id="Mute">Mute</item>
|
|
68
|
-
<separator/>
|
|
69
|
-
<item action="Find" id="Find">Find...</item>
|
|
70
|
-
<item action="FindAgain" id="FindAgain">Find Again</item>
|
|
71
|
-
<item action="Copy" id="Copy">Copy</item>
|
|
72
65
|
</menu>
|
|
73
|
-
|
|
74
66
|
`
|
|
75
67
|
// Build the grammar. 'document' is the starting production
|
|
76
|
-
const grammar = buildGrammar(XmlGrammar, 'document'
|
|
68
|
+
const grammar = buildGrammar(XmlGrammar, 'document', {
|
|
69
|
+
unwrappedNonterminalNames: xmlGrammarUnwrappedNonterminalNames
|
|
70
|
+
})
|
|
77
71
|
|
|
78
72
|
// Parse the XML string
|
|
79
73
|
const parseTree = grammar.parse(xmlString)
|
|
80
74
|
|
|
81
|
-
|
|
75
|
+
const parseTreeJson = JSON.stringify(parseTree, undefined, 4)
|
|
76
|
+
|
|
77
|
+
log(parseTreeJson)
|
|
78
|
+
|
|
79
|
+
await writeFile('out/out.json', parseTreeJson)
|
|
82
80
|
}
|
|
83
81
|
|
|
84
82
|
async function testRegExpParser() {
|
|
85
|
-
const regExpString = /^([+]?[1]?(1 )?[-.+]?\(?\d{1}[- .+]*\d{1}[- .+]*\d{1}\)?[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1})$/.source
|
|
86
|
-
//const regExpString = /^
|
|
83
|
+
//const regExpString = /^([+]?[1]?(1 )?[-.+]?\(?\d{1}[- .+]*\d{1}[- .+]*\d{1}\)?[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1}[- .+]*\d{1})$/.source
|
|
84
|
+
//const regExpString = /^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$/.source
|
|
85
|
+
//const regExpString = /(?=.*[!@#$%^&*])^mongodb:\/\/(?<user>[a-zA-Z0-9]+):(?<pass>[a-zA-Z0-9!@#$%^&*]{8,})@(?<host>[a-z0-9.-]+):(?<port>\d{2,5})$/.source
|
|
86
|
+
const regExpString = /^(abcd)*ef+g/.source
|
|
87
|
+
//const regExpString = '(abcd)(aa))'
|
|
87
88
|
|
|
88
|
-
const grammar = buildGrammar(RegExpGrammar, '
|
|
89
|
+
const grammar = buildGrammar(RegExpGrammar, 'root', {
|
|
90
|
+
unwrappedNonterminalNames: regExpGrammarUnwrappedNonterminalNames
|
|
91
|
+
})
|
|
89
92
|
|
|
90
93
|
const parseTree = grammar.parse(regExpString)
|
|
91
94
|
|
|
@@ -93,13 +96,10 @@ async function testRegExpParser() {
|
|
|
93
96
|
|
|
94
97
|
log(parseTreeJson)
|
|
95
98
|
|
|
96
|
-
const { writeFile } = await import('fs/promises')
|
|
97
|
-
|
|
98
99
|
await writeFile('out/out.json', parseTreeJson)
|
|
99
100
|
}
|
|
100
101
|
|
|
101
|
-
|
|
102
|
-
async function testParserError1() {
|
|
102
|
+
function testParserError1() {
|
|
103
103
|
const xmlData = `<hello> wo rld <!!! `
|
|
104
104
|
|
|
105
105
|
const grammar = buildGrammar(XmlGrammar, 'document')
|
|
@@ -109,7 +109,7 @@ async function testParserError1() {
|
|
|
109
109
|
console.log(JSON.stringify(result, undefined, 4))
|
|
110
110
|
}
|
|
111
111
|
|
|
112
|
-
|
|
112
|
+
function testParserError2() {
|
|
113
113
|
const jsonData = `{ "asdf": 12.5 `
|
|
114
114
|
|
|
115
115
|
const grammar = buildGrammar(JsonGrammar, 'expression')
|
|
@@ -119,6 +119,22 @@ async function testParserError2() {
|
|
|
119
119
|
console.log(JSON.stringify(result, undefined, 4))
|
|
120
120
|
}
|
|
121
121
|
|
|
122
|
+
function test1() {
|
|
123
|
+
const input = `abcdefg`
|
|
124
|
+
|
|
125
|
+
const grammar = buildGrammar(SimpleTestGrammar1, 'root')
|
|
126
|
+
|
|
127
|
+
const result = grammar.parse(input)
|
|
128
|
+
|
|
129
|
+
console.log(JSON.stringify(result, undefined, 4))
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
//testParserError1()
|
|
134
|
+
//testParserError2()
|
|
135
|
+
|
|
122
136
|
//testJsonParser()
|
|
137
|
+
testXmlParser()
|
|
138
|
+
//testRegExpParser()
|
|
123
139
|
|
|
124
|
-
|
|
140
|
+
//test1()
|
|
@@ -11,7 +11,7 @@ export class JsonGrammar {
|
|
|
11
11
|
this.arrayExpression
|
|
12
12
|
)
|
|
13
13
|
|
|
14
|
-
stringLiteral = G.pattern([
|
|
14
|
+
stringLiteral = () => G.pattern([
|
|
15
15
|
zeroOrMoreWhitespace,
|
|
16
16
|
|
|
17
17
|
'"',
|
|
@@ -27,7 +27,7 @@ export class JsonGrammar {
|
|
|
27
27
|
zeroOrMoreWhitespace,
|
|
28
28
|
])
|
|
29
29
|
|
|
30
|
-
numberLiteral = G.pattern([
|
|
30
|
+
numberLiteral = () => G.pattern([
|
|
31
31
|
zeroOrMoreWhitespace,
|
|
32
32
|
|
|
33
33
|
R.captureAs('value', [
|
|
@@ -46,7 +46,7 @@ export class JsonGrammar {
|
|
|
46
46
|
zeroOrMoreWhitespace,
|
|
47
47
|
])
|
|
48
48
|
|
|
49
|
-
booleanLiteral = G.pattern([
|
|
49
|
+
booleanLiteral = () => G.pattern([
|
|
50
50
|
zeroOrMoreWhitespace,
|
|
51
51
|
|
|
52
52
|
R.captureAs('value',
|
|
@@ -56,7 +56,7 @@ export class JsonGrammar {
|
|
|
56
56
|
zeroOrMoreWhitespace,
|
|
57
57
|
])
|
|
58
58
|
|
|
59
|
-
nullLiteral = G.pattern([
|
|
59
|
+
nullLiteral = () => G.pattern([
|
|
60
60
|
zeroOrMoreWhitespace,
|
|
61
61
|
|
|
62
62
|
'null',
|
|
@@ -92,14 +92,14 @@ export class JsonGrammar {
|
|
|
92
92
|
this.closingSquareBracket
|
|
93
93
|
]
|
|
94
94
|
|
|
95
|
-
openingCurlyBrace = createPatternWithClearedWhitespace('{')
|
|
96
|
-
closingCurlyBrace = createPatternWithClearedWhitespace('}')
|
|
95
|
+
openingCurlyBrace = () => createPatternWithClearedWhitespace('{')
|
|
96
|
+
closingCurlyBrace = () => createPatternWithClearedWhitespace('}')
|
|
97
97
|
|
|
98
|
-
openingSquareBracket = createPatternWithClearedWhitespace('[')
|
|
99
|
-
closingSquareBracket = createPatternWithClearedWhitespace(']')
|
|
98
|
+
openingSquareBracket = () => createPatternWithClearedWhitespace('[')
|
|
99
|
+
closingSquareBracket = () => createPatternWithClearedWhitespace(']')
|
|
100
100
|
|
|
101
|
-
comma = createPatternWithClearedWhitespace(',')
|
|
102
|
-
colons = createPatternWithClearedWhitespace(':')
|
|
101
|
+
comma = () => createPatternWithClearedWhitespace(',')
|
|
102
|
+
colons = () => createPatternWithClearedWhitespace(':')
|
|
103
103
|
}
|
|
104
104
|
|
|
105
105
|
function createPatternWithClearedWhitespace(subpattern: R.Pattern) {
|
|
@@ -113,3 +113,21 @@ function createPatternWithClearedWhitespace(subpattern: R.Pattern) {
|
|
|
113
113
|
}
|
|
114
114
|
|
|
115
115
|
const zeroOrMoreWhitespace = R.zeroOrMore(R.whitespace)
|
|
116
|
+
|
|
117
|
+
//////////////////////////////////////////////////////////////////////////////////////////////
|
|
118
|
+
// Wrapped nonterminal names
|
|
119
|
+
//////////////////////////////////////////////////////////////////////////////////////////////
|
|
120
|
+
export const jsonGrammarUnwrappedNonterminalNames: G.GrammarNonterminalNames<JsonGrammar> = [
|
|
121
|
+
//'stringLiteral',
|
|
122
|
+
//'numberLiteral',
|
|
123
|
+
//'booleanLiteral',
|
|
124
|
+
//'nullLiteral',
|
|
125
|
+
|
|
126
|
+
'openingCurlyBrace',
|
|
127
|
+
'closingCurlyBrace',
|
|
128
|
+
'openingSquareBracket',
|
|
129
|
+
'closingSquareBracket',
|
|
130
|
+
|
|
131
|
+
'comma',
|
|
132
|
+
'colons',
|
|
133
|
+
]
|