grammar-composer 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/README.md +144 -89
  2. package/dist/exports/Exports.d.ts +2 -0
  3. package/dist/exports/Exports.d.ts.map +1 -1
  4. package/dist/exports/Exports.js +2 -0
  5. package/dist/exports/Exports.js.map +1 -1
  6. package/dist/parser-generator/Grammar.d.ts +8 -1
  7. package/dist/parser-generator/Grammar.d.ts.map +1 -1
  8. package/dist/parser-generator/Grammar.js +30 -225
  9. package/dist/parser-generator/Grammar.js.map +1 -1
  10. package/dist/parser-generator/ParseError.d.ts +37 -0
  11. package/dist/parser-generator/ParseError.d.ts.map +1 -0
  12. package/dist/parser-generator/ParseError.js +226 -0
  13. package/dist/parser-generator/ParseError.js.map +1 -0
  14. package/dist/parser-generator/StaticAnalysis.d.ts +6 -0
  15. package/dist/parser-generator/StaticAnalysis.d.ts.map +1 -0
  16. package/dist/parser-generator/StaticAnalysis.js +283 -0
  17. package/dist/parser-generator/StaticAnalysis.js.map +1 -0
  18. package/dist/parser-generator/TerminalToText.d.ts +15 -0
  19. package/dist/parser-generator/TerminalToText.d.ts.map +1 -0
  20. package/dist/parser-generator/TerminalToText.js +158 -0
  21. package/dist/parser-generator/TerminalToText.js.map +1 -0
  22. package/dist/parser-generator/TopDownParser.d.ts +4 -2
  23. package/dist/parser-generator/TopDownParser.d.ts.map +1 -1
  24. package/dist/parser-generator/TopDownParser.js +44 -42
  25. package/dist/parser-generator/TopDownParser.js.map +1 -1
  26. package/dist/tests/Test.js +23 -23
  27. package/dist/tests/Test.js.map +1 -1
  28. package/dist/tests/test-grammars/JsonGrammar.d.ts +12 -11
  29. package/dist/tests/test-grammars/JsonGrammar.d.ts.map +1 -1
  30. package/dist/tests/test-grammars/JsonGrammar.js +25 -10
  31. package/dist/tests/test-grammars/JsonGrammar.js.map +1 -1
  32. package/dist/tests/test-grammars/RegExpGrammar.d.ts +25 -25
  33. package/dist/tests/test-grammars/RegExpGrammar.d.ts.map +1 -1
  34. package/dist/tests/test-grammars/RegExpGrammar.js +48 -31
  35. package/dist/tests/test-grammars/RegExpGrammar.js.map +1 -1
  36. package/dist/tests/test-grammars/XmlGrammar.d.ts +11 -10
  37. package/dist/tests/test-grammars/XmlGrammar.d.ts.map +1 -1
  38. package/dist/tests/test-grammars/XmlGrammar.js +19 -8
  39. package/dist/tests/test-grammars/XmlGrammar.js.map +1 -1
  40. package/dist/utilities/LineAndColumn.d.ts +7 -0
  41. package/dist/utilities/LineAndColumn.d.ts.map +1 -0
  42. package/dist/utilities/LineAndColumn.js +52 -0
  43. package/dist/utilities/LineAndColumn.js.map +1 -0
  44. package/dist/utilities/Timer.d.ts +7 -3
  45. package/dist/utilities/Timer.d.ts.map +1 -1
  46. package/dist/utilities/Timer.js +41 -39
  47. package/dist/utilities/Timer.js.map +1 -1
  48. package/package.json +3 -3
  49. package/src/exports/Exports.ts +2 -0
  50. package/src/parser-generator/Grammar.ts +59 -265
  51. package/src/parser-generator/ParseError.ts +337 -0
  52. package/src/parser-generator/StaticAnalysis.ts +340 -0
  53. package/src/parser-generator/TerminalToText.ts +196 -0
  54. package/src/parser-generator/TopDownParser.ts +67 -49
  55. package/src/tests/Test.ts +26 -24
  56. package/src/tests/test-grammars/JsonGrammar.ts +28 -10
  57. package/src/tests/test-grammars/RegExpGrammar.ts +65 -46
  58. package/src/tests/test-grammars/XmlGrammar.ts +20 -8
  59. package/src/utilities/LineAndColumn.ts +61 -0
  60. package/src/utilities/Timer.ts +50 -48
@@ -0,0 +1,337 @@
1
+ import type { Nonterminal, Terminal } from './Grammar.js'
2
+ import { buildCaretSpacing, expandTabs, getLineAndColumn } from '../utilities/LineAndColumn.js'
3
+ import { escapeChar, TerminalToText } from './TerminalToText.js'
4
+
5
+ /////////////////////////////////////////////////////////////////////////////////////////////////
6
+ // Parse error class
7
+ /////////////////////////////////////////////////////////////////////////////////////////////////
8
+ export class ParseError extends Error {
9
+ readonly input: string
10
+
11
+ readonly offset: number
12
+ readonly line: number
13
+ readonly column: number
14
+
15
+ readonly found?: string
16
+ readonly expected?: readonly ExpectedTerminal[]
17
+
18
+ readonly contextProduction?: string
19
+
20
+ constructor(details: ParseErrorDetails, maxDisplayedLineLength = 100) {
21
+ const { line, column } = getLineAndColumn(details.input, details.offset)
22
+
23
+ const errorMessage = buildErrorMessage(details, line, column, maxDisplayedLineLength)
24
+
25
+ super(errorMessage)
26
+
27
+ this.name = 'ParseError'
28
+
29
+ this.input = details.input
30
+
31
+ this.offset = details.offset
32
+ this.line = line
33
+ this.column = column
34
+
35
+
36
+ if (details.offset < details.input.length) {
37
+ this.found = details.input[details.offset]
38
+ }
39
+ this.expected = details.expected
40
+
41
+ this.contextProduction = details.contextProductionName
42
+ }
43
+
44
+ // Dedicated factory for the case where parsing failed because no expected
45
+ // terminal matched at some position.
46
+ static createFailedParseError(input: string, offset: number, bestFailedMatches: FailedMatch[]): ParseError {
47
+ const { expected, contextProduction } = analyzeFailedMatches(bestFailedMatches)
48
+
49
+ return new ParseError({
50
+ input,
51
+ offset,
52
+ expected,
53
+ contextProductionName: contextProduction.name
54
+ })
55
+ }
56
+
57
+ // Dedicated factory for the case where parsing consumed only part of the input
58
+ // without any specific terminal failing.
59
+ static createIncompleteParseError(input: string, offset: number, parsedLength: number): ParseError {
60
+ return new ParseError({ input, offset, parsedLength })
61
+ }
62
+ }
63
+
64
+ /////////////////////////////////////////////////////////////////////////////////////////////////
65
+ // Failed match analysis
66
+ /////////////////////////////////////////////////////////////////////////////////////////////////
67
+
68
+ // Computes the data needed to describe a parse failure: the deduplicated list of expected
69
+ // terminals (each retaining the production it belongs to) and the shared context production.
70
+ // The actual message formatting is owned by ParseError.
71
+ export function analyzeFailedMatches(failedMatches: FailedMatch[]): FailedMatchAnalysisResult {
72
+ const productionStacks = failedMatches.map(failedMatch => failedMatch.productionStack)
73
+
74
+ const commonProductionPrefix = findCommonProductionPrefix(productionStacks)
75
+ const contextProduction = commonProductionPrefix[commonProductionPrefix.length - 1]
76
+
77
+ // Collect every production each distinct terminal could be part of, together with
78
+ // the production stack that led to it. The production is the one directly below
79
+ // the shared context production.
80
+ const candidatesByTerminal = new Map<string, TerminalCandidate[]>()
81
+
82
+ for (const failedMatch of failedMatches) {
83
+ const subProduction = failedMatch.productionStack[commonProductionPrefix.length] ?? contextProduction
84
+
85
+ if (subProduction === undefined) {
86
+ continue
87
+ }
88
+
89
+ const stringifiedFailedTerminal = stringifyFailedTerminal(failedMatch)
90
+
91
+ const candidates = candidatesByTerminal.get(stringifiedFailedTerminal) ?? []
92
+
93
+ candidates.push({ production: subProduction, productionStack: failedMatch.productionStack })
94
+ candidatesByTerminal.set(stringifiedFailedTerminal, candidates)
95
+ }
96
+
97
+ // When the same terminal can be reached through several productions and one of
98
+ // them is an ancestor of another (e.g. the same pattern is referenced both
99
+ // directly and through a parent production), keep only the innermost production.
100
+ const expected: ExpectedTerminal[] = []
101
+
102
+ for (const [stringifiedFailedTerminal, candidates] of candidatesByTerminal) {
103
+ for (const candidate of candidates) {
104
+ if (isRedundantCandidate(candidate, candidates)) {
105
+ continue
106
+ }
107
+
108
+ if (!expected.some(entry =>
109
+ entry.productionName === candidate.production.name && entry.terminalString === stringifiedFailedTerminal)) {
110
+
111
+ expected.push({ productionName: candidate.production.name, terminalString: stringifiedFailedTerminal })
112
+ }
113
+ }
114
+ }
115
+
116
+ return {
117
+ expected,
118
+ contextProduction,
119
+ }
120
+ }
121
+
122
+ // A candidate is redundant when the same terminal is also expected by a deeper
123
+ // production within the same context, i.e. when another candidate's production
124
+ // appears below this candidate's production in its own stack.
125
+ function isRedundantCandidate(candidate: TerminalCandidate, candidates: TerminalCandidate[]): boolean {
126
+ const candidateIndex = stackIndexOf(candidate.productionStack, candidate.production)
127
+
128
+ for (const other of candidates) {
129
+ if (other === candidate) {
130
+ continue
131
+ }
132
+
133
+ const otherIndex = stackIndexOf(candidate.productionStack, other.production)
134
+
135
+ if (otherIndex > candidateIndex) {
136
+ return true
137
+ }
138
+ }
139
+
140
+ return false
141
+ }
142
+
143
+ // Compares nonterminals by their canonical grammar reference, because the same
144
+ // production can be represented by multiple cloned instances during a parse.
145
+ function stackIndexOf(productionStack: Nonterminal[], production: Nonterminal): number {
146
+ const canonical = production.grammarNonterminal ?? production
147
+
148
+ return productionStack.findIndex(nonterminal => (nonterminal.grammarNonterminal ?? nonterminal) === canonical)
149
+ }
150
+
151
+ function stringifyFailedTerminal(failedMatch: FailedMatch): string {
152
+ const { terminal, productionStack } = failedMatch
153
+
154
+ if (terminal.name !== undefined) {
155
+ return terminal.name
156
+ }
157
+
158
+ const innermostProduction = productionStack[productionStack.length - 1]
159
+
160
+ if (innermostProduction !== undefined && innermostProduction.content === terminal) {
161
+ return innermostProduction.name
162
+ }
163
+
164
+ const stringifiedLengthLimit = 100
165
+ const terminalToText = new TerminalToText(stringifiedLengthLimit)
166
+
167
+ return terminalToText.stringifyTerminal(terminal)
168
+ }
169
+
170
+ /////////////////////////////////////////////////////////////////////////////////////////////////
171
+ // Production grouping
172
+ /////////////////////////////////////////////////////////////////////////////////////////////////
173
+
174
+ // The stacks are compared by canonical nonterminal identity rather than object identity,
175
+ // because the same production can be represented by multiple distinct object instances
176
+ // during a single parse: the grammar builder creates spread copies of the canonical
177
+ // nonterminal for optional references and for references wrapped in cached(). Each clone
178
+ // keeps a reference to the original in 'grammarNonterminal'.
179
+ export function findCommonProductionPrefix(productionStacks: Nonterminal[][]): Nonterminal[] {
180
+ if (productionStacks.length === 0) {
181
+ return []
182
+ }
183
+
184
+ let commonPrefix = productionStacks[0]
185
+
186
+ for (let i = 1; i < productionStacks.length; i++) {
187
+ const stack = productionStacks[i]
188
+ let commonLength = 0
189
+
190
+ while (commonLength < commonPrefix.length &&
191
+ commonLength < stack.length &&
192
+ areSameNonterminal(commonPrefix[commonLength], stack[commonLength])) {
193
+
194
+ commonLength += 1
195
+ }
196
+
197
+ commonPrefix = commonPrefix.slice(0, commonLength)
198
+
199
+ if (commonPrefix.length === 0) {
200
+ break
201
+ }
202
+ }
203
+
204
+ return commonPrefix
205
+ }
206
+
207
+ function areSameNonterminal(a: Nonterminal, b: Nonterminal): boolean {
208
+ return (a.grammarNonterminal ?? a) === (b.grammarNonterminal ?? b)
209
+ }
210
+
211
+ /////////////////////////////////////////////////////////////////////////////////////////////////
212
+ // Parse error helpers
213
+ /////////////////////////////////////////////////////////////////////////////////////////////////
214
+ function buildErrorMessage(details: ParseErrorDetails, line: number, column: number, maxDisplayedLineLength: number): string {
215
+ const context = formatErrorContext(details.input, details.offset, maxDisplayedLineLength)
216
+
217
+ if (details.parsedLength !== undefined) {
218
+ return `Failed parsing the input text ${context}\n\n` +
219
+ `Stopped parsing at line ${line}, column ${column} (${details.parsedLength} of ${details.input.length} characters consumed).`
220
+ }
221
+
222
+ const expected = details.expected ?? []
223
+ const expectedMessage = expected.length > 0
224
+ ? buildExpectedGroupsMessage(expected)
225
+ : ''
226
+
227
+ const contextMessage = details.contextProductionName !== undefined
228
+ ? `While parsing '${details.contextProductionName}', expected one of the following:\n\n${expectedMessage}`
229
+ : `Expected one of the following:\n\n${expectedMessage}`
230
+
231
+ return `Failed parsing the input text ${context}\n\n${contextMessage}`
232
+ }
233
+
234
+ // Formats the expected terminals grouped by production, with a blank line between
235
+ // the groups so the message stays readable.
236
+ function buildExpectedGroupsMessage(expected: readonly ExpectedTerminal[]): string {
237
+ const expectedByProduction = new Map<string, string[]>()
238
+
239
+ for (const { productionName, terminalString } of expected) {
240
+ const stringifiedTerminals = expectedByProduction.get(productionName) ?? []
241
+
242
+ if (!stringifiedTerminals.includes(terminalString)) {
243
+ stringifiedTerminals.push(terminalString)
244
+ }
245
+
246
+ expectedByProduction.set(productionName, stringifiedTerminals)
247
+ }
248
+
249
+ return [...expectedByProduction.entries()]
250
+ .map(([productionName, stringifiedTerminals]) => `In '${productionName}': ${stringifiedTerminals.join(', ')}`)
251
+ .join('\n\n')
252
+ }
253
+
254
+ export function formatErrorContext(input: string, offset: number, maxDisplayedLineLength: number): string {
255
+ const { line, column } = getLineAndColumn(input, offset)
256
+
257
+ let lineStartOffset = offset
258
+ while (lineStartOffset > 0 && input[lineStartOffset - 1] !== '\n') {
259
+ lineStartOffset -= 1
260
+ }
261
+
262
+ let lineEndOffset = offset
263
+ while (lineEndOffset < input.length && input[lineEndOffset] !== '\n') {
264
+ lineEndOffset += 1
265
+ }
266
+
267
+ const lineText = input.substring(lineStartOffset, lineEndOffset).replace(/\r$/, '')
268
+
269
+ let displayedText = lineText
270
+ let displayedCaretColumn = column
271
+
272
+ // If the failing line is very long (e.g. minified input), show a window around the
273
+ // error position instead of the entire line.
274
+ if (lineText.length > maxDisplayedLineLength) {
275
+ const rawCaretOffset = Math.min(column - 1, lineText.length)
276
+ const radius = Math.floor(maxDisplayedLineLength / 2)
277
+
278
+ const contextStart = Math.max(0, rawCaretOffset - radius)
279
+ const contextEnd = Math.min(lineText.length, rawCaretOffset + radius)
280
+
281
+ const hasTruncatedStart = contextStart > 0
282
+ const hasTruncatedEnd = contextEnd < lineText.length
283
+
284
+ const truncatedLine = lineText.substring(contextStart, contextEnd)
285
+
286
+ displayedText = `${hasTruncatedStart ? '…' : ''}${truncatedLine}${hasTruncatedEnd ? '…' : ''}`
287
+ displayedCaretColumn = (rawCaretOffset - contextStart) + (hasTruncatedStart ? 1 : 0) + 1
288
+ }
289
+
290
+ let unexpectedTargetText: string
291
+
292
+ if (offset < input.length) {
293
+ const targetChar = input[offset]
294
+ const escapedTargetChar = escapeChar(targetChar)
295
+
296
+ unexpectedTargetText = ` (unexpected character '${escapedTargetChar}')`
297
+ } else {
298
+ unexpectedTargetText = ' (unexpected end of input)'
299
+ }
300
+
301
+ const text = `at line ${line}, column ${column}${unexpectedTargetText}:\n\n\t${expandTabs(displayedText)}\n\t${buildCaretSpacing(displayedText, displayedCaretColumn)}^`
302
+
303
+ return text
304
+ }
305
+
306
+ /////////////////////////////////////////////////////////////////////////////////////////////////
307
+ // Types
308
+ /////////////////////////////////////////////////////////////////////////////////////////////////
309
+ interface TerminalCandidate {
310
+ production: Nonterminal
311
+ productionStack: Nonterminal[]
312
+ }
313
+
314
+ interface FailedMatchAnalysisResult {
315
+ expected: ExpectedTerminal[]
316
+ contextProduction: Nonterminal
317
+ }
318
+
319
+ export interface FailedMatch {
320
+ terminal: Terminal
321
+ productionStack: Nonterminal[]
322
+ }
323
+
324
+ // A single expected terminal, retaining the production it belongs to.
325
+ export interface ExpectedTerminal {
326
+ productionName: string
327
+ terminalString: string
328
+ }
329
+
330
+ export interface ParseErrorDetails {
331
+ input: string
332
+ offset: number
333
+
334
+ expected?: readonly ExpectedTerminal[]
335
+ contextProductionName?: string
336
+ parsedLength?: number
337
+ }
@@ -0,0 +1,340 @@
1
+ import { Pattern } from 'regexp-composer'
2
+ import { isArray, isBoolean, isString } from '../utilities/Utilities.js'
3
+ import { GrammarElement } from './Grammar.js'
4
+
5
+ /////////////////////////////////////////////////////////////////////////////////////////////////
6
+ // Internal static analysis methods
7
+ /////////////////////////////////////////////////////////////////////////////////////////////////
8
+ export function detectAndAnnotateOptionalNodes(rootNode: GrammarElement) {
9
+ const visitedNodes = new Set<GrammarElement>()
10
+
11
+ const resolvedNodes = new Map<GrammarElement, boolean>()
12
+ const unresolvedNodes = new Map<GrammarElement, { dependencies: Set<GrammarElement>, isChoice: boolean }>()
13
+
14
+ function processDepthFirst(node: GrammarElement): boolean | undefined {
15
+ if (visitedNodes.has(node)) {
16
+ return resolvedNodes.get(node)
17
+ }
18
+
19
+ visitedNodes.add(node)
20
+
21
+ switch (node.type) {
22
+ case 'StringTerminal':
23
+ case 'PatternTerminal': {
24
+ resolvedNodes.set(node, node.optional)
25
+
26
+ return node.optional
27
+ }
28
+
29
+ case 'Nonterminal':
30
+ case 'Repetition': {
31
+ const result = processDepthFirst(node.content)
32
+
33
+ if (node.optional) {
34
+ resolvedNodes.set(node, true)
35
+
36
+ return true
37
+ } else if (isBoolean(result)) {
38
+ resolvedNodes.set(node, result)
39
+
40
+ return result
41
+ } else {
42
+ unresolvedNodes.set(node, { dependencies: new Set([node.content]), isChoice: false })
43
+
44
+ return undefined
45
+ }
46
+ }
47
+
48
+ case 'Sequence': {
49
+ const dependencies = new Set<GrammarElement>()
50
+
51
+ let hasNonOptionalResolvedMember = false
52
+
53
+ for (const element of node.members) {
54
+ const result = processDepthFirst(element)
55
+
56
+ if (isBoolean(result)) {
57
+ if (result === false) {
58
+ hasNonOptionalResolvedMember = true
59
+ }
60
+ } else {
61
+ dependencies.add(element)
62
+ }
63
+ }
64
+
65
+ if (node.optional === true) {
66
+ resolvedNodes.set(node, true)
67
+
68
+ return true
69
+ } else if (hasNonOptionalResolvedMember) {
70
+ resolvedNodes.set(node, false)
71
+
72
+ return false
73
+ } else if (dependencies.size === 0) {
74
+ resolvedNodes.set(node, true)
75
+
76
+ return true
77
+ } else {
78
+ unresolvedNodes.set(node, { dependencies, isChoice: false })
79
+
80
+ return undefined
81
+ }
82
+ }
83
+
84
+ case 'Choice': {
85
+ const dependencies = new Set<GrammarElement>()
86
+ let hasOptionalResolvedMember = false
87
+
88
+ for (const element of node.members) {
89
+ const result = processDepthFirst(element)
90
+
91
+ if (isBoolean(result)) {
92
+ if (result === true) {
93
+ hasOptionalResolvedMember = true
94
+ }
95
+ } else {
96
+ dependencies.add(element)
97
+ }
98
+ }
99
+
100
+ if (node.optional === true) {
101
+ resolvedNodes.set(node, true)
102
+
103
+ return true
104
+ } else if (hasOptionalResolvedMember) {
105
+ resolvedNodes.set(node, true)
106
+
107
+ return true
108
+ } else if (dependencies.size === 0) {
109
+ resolvedNodes.set(node, false)
110
+
111
+ return false
112
+ } else {
113
+ unresolvedNodes.set(node, { dependencies, isChoice: true })
114
+
115
+ return undefined
116
+ }
117
+ }
118
+ }
119
+
120
+ return undefined
121
+ }
122
+
123
+ // Process depth first to resolve the easy cases, for productions that contain
124
+ // no cyclic references:
125
+ processDepthFirst(rootNode)
126
+
127
+ // Now the remainder consists of nodes containing cyclic references that have not yet been resolved.
128
+ // Use a form of iterative elimination and substitution to resolve them:
129
+ while (unresolvedNodes.size > 0) {
130
+ // This variable tracks whether at least one dependency was resolved, in any node.
131
+ // If it stays false, it means that no improvement was made during the iteration,
132
+ // and we should exit the loop.
133
+ let atLastOneDependencyResolvedInAnyNode = false
134
+
135
+ const nodesToDelete: GrammarElement[] = []
136
+
137
+ // Scan the unresolved nodes to locate any new resolved dependencies
138
+ for (const [node, { dependencies, isChoice }] of unresolvedNodes) {
139
+ let nodeResolved = false
140
+
141
+ // Iterate over all unresolved dependencies for the node
142
+ for (const dependency of Array.from(dependencies)) {
143
+ // Check if the dependency has been resolved
144
+ const value = resolvedNodes.get(dependency)
145
+
146
+ if (value !== undefined) {
147
+ // If it did, record that some dependencies were resolved
148
+ atLastOneDependencyResolvedInAnyNode = true
149
+
150
+ if (isChoice) {
151
+ if (value === true) {
152
+ resolvedNodes.set(node, true)
153
+ nodeResolved = true
154
+ break
155
+ }
156
+ } else {
157
+ if (value === false) {
158
+ resolvedNodes.set(node, false)
159
+ nodeResolved = true
160
+ break
161
+ }
162
+ }
163
+
164
+ dependencies.delete(dependency)
165
+ }
166
+ }
167
+
168
+ if (nodeResolved) {
169
+ nodesToDelete.push(node)
170
+ } else if (dependencies.size === 0) {
171
+ if (isChoice) {
172
+ resolvedNodes.set(node, false)
173
+ } else {
174
+ resolvedNodes.set(node, true)
175
+ }
176
+
177
+ nodesToDelete.push(node)
178
+ }
179
+ }
180
+
181
+ for (const node of nodesToDelete) {
182
+ unresolvedNodes.delete(node)
183
+ }
184
+
185
+ // If not even one dependency was eliminated for any node,
186
+ // it means that only mutually cyclic nodes are left unresolved, so exit the loop.
187
+ if (!atLastOneDependencyResolvedInAnyNode) {
188
+ break
189
+ }
190
+ }
191
+
192
+ // All remaining unresolved nodes must now be optional,
193
+ // since they are all mutually cyclic and all their non-cyclic grammar elements are known to be optional.
194
+ for (const node of unresolvedNodes.keys()) {
195
+ resolvedNodes.set(node, true)
196
+ unresolvedNodes.delete(node)
197
+ }
198
+
199
+ // Finally set the 'optional' property of all nodes based on the detected values.
200
+ for (const [node, isOptional] of resolvedNodes) {
201
+ node.optional = isOptional
202
+ }
203
+ }
204
+
205
+ export function detectAndErrorOnLeftRecursion(rootNode: GrammarElement) {
206
+ const currentlyIteratedNodes = new Set<GrammarElement>()
207
+
208
+ function detect(node: GrammarElement) {
209
+ if (currentlyIteratedNodes.has(node)) {
210
+ if (node.type === 'Nonterminal') {
211
+ throw new Error(`Detected left recursion for nonterminal '${node.name}'.`)
212
+ } else {
213
+ throw new Error(`Detected left recursion for node: ${JSON.stringify(node, undefined, 4)}`)
214
+ }
215
+ }
216
+
217
+ currentlyIteratedNodes.add(node)
218
+
219
+ switch (node.type) {
220
+ case 'Nonterminal':
221
+ case 'Repetition': {
222
+ detect(node.content)
223
+
224
+ break
225
+ }
226
+
227
+ case 'Sequence': {
228
+ for (const member of node.members) {
229
+ detect(member)
230
+
231
+ if (!member.optional) {
232
+ break
233
+ }
234
+ }
235
+
236
+ break
237
+ }
238
+
239
+ case 'Choice': {
240
+ for (const member of node.members) {
241
+ detect(member)
242
+ }
243
+
244
+ break
245
+ }
246
+ }
247
+
248
+ currentlyIteratedNodes.delete(node)
249
+ }
250
+
251
+ detect(rootNode)
252
+ }
253
+
254
+ // Walks the regexp-composer Pattern AST to validate its capture groups.
255
+ //
256
+ // The JavaScript RegExp engine collapses multiple named groups that share a name
257
+ // into a single entry, and cannot reliably report the ordering of a mix of named
258
+ // and unnamed groups. Both cases are problematic for the parser, so we detect them
259
+ // here, once, at grammar build time (before the RegExp is even compiled), instead of
260
+ // during parsing.
261
+ //
262
+ // Mirrors the AST traversal performed by regexp-composer's own 'isPatternOptional'.
263
+ export function validatePatternCaptureGroups(pattern: Pattern): void {
264
+ const seenNamedCaptureGroupNames = new Set<string>()
265
+ let hasNamedCaptureGroup = false
266
+ let hasUnnamedCaptureGroup = false
267
+
268
+ function walk(node: Pattern): void {
269
+ if (isString(node)) {
270
+ return
271
+ }
272
+
273
+ if (isArray(node)) {
274
+ for (const element of node) {
275
+ walk(element)
276
+ }
277
+
278
+ return
279
+ }
280
+
281
+ switch (node.type) {
282
+ case 'capture': {
283
+ if (node.name !== undefined) {
284
+ if (seenNamedCaptureGroupNames.has(node.name)) {
285
+ throw new Error(`The regular expression pattern contains multiple named capture groups with the same name '${node.name}'. Named capture groups must have unique names.`)
286
+ }
287
+
288
+ seenNamedCaptureGroupNames.add(node.name)
289
+ hasNamedCaptureGroup = true
290
+ } else {
291
+ hasUnnamedCaptureGroup = true
292
+ }
293
+
294
+ walk(node.content)
295
+
296
+ return
297
+ }
298
+
299
+ // Composite nodes that wrap content without creating a capturing group:
300
+ case 'possibly':
301
+ case 'zeroOrMore':
302
+ case 'oneOrMore':
303
+ case 'repeated':
304
+ case 'followedBy':
305
+ case 'notFollowedBy':
306
+ case 'precededBy':
307
+ case 'notPrecededBy': {
308
+ walk(node.content)
309
+
310
+ return
311
+ }
312
+
313
+ case 'anyOf': {
314
+ for (const member of node.members) {
315
+ walk(member)
316
+ }
317
+
318
+ return
319
+ }
320
+
321
+ // These nodes never introduce capturing groups and contain only
322
+ // character-level tokens or backreferences:
323
+ case 'notAnyOfChars':
324
+ case 'specialToken':
325
+ case 'sameAs': {
326
+ return
327
+ }
328
+
329
+ default: {
330
+ throw new Error(`Unrecognized pattern type: ${(node as any).type}`)
331
+ }
332
+ }
333
+ }
334
+
335
+ walk(pattern)
336
+
337
+ if (hasNamedCaptureGroup && hasUnnamedCaptureGroup) {
338
+ throw new Error(`The regular expression pattern contains a combination of named and unnamed capture groups. Due to limitations of the JavaScript RegExp engine, it is impossible to reliably identify the ordering of this combination, please use either all unnamed or all named capture groups, but not both.`)
339
+ }
340
+ }