@scinorandex/sparse 0.1.1 → 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/workflows/ci.yml +41 -0
- package/AGENTS.md +31 -11
- package/README.md +289 -29
- package/dist/cli.js +113 -24
- package/dist/cli.js.map +1 -1
- package/dist/generator.d.ts +3 -1
- package/dist/generator.js +29 -5
- package/dist/generator.js.map +1 -1
- package/dist/index.d.ts +16 -4
- package/dist/index.js +34 -1
- package/dist/index.js.map +1 -1
- package/dist/meta/common.d.ts +17 -0
- package/dist/meta/common.js +66 -1
- package/dist/meta/common.js.map +1 -1
- package/dist/meta/selfhosted.d.ts +4 -5
- package/dist/meta/selfhosted.js +49 -36
- package/dist/meta/selfhosted.js.map +1 -1
- package/dist/parser.d.ts +55 -35
- package/dist/parser.js +141 -18
- package/dist/parser.js.map +1 -1
- package/dist/table/selfhosted.d.ts +10 -1
- package/dist/table/selfhosted.js +32 -12
- package/dist/table/selfhosted.js.map +1 -1
- package/dist/table/validate.d.ts +17 -0
- package/dist/table/validate.js +84 -0
- package/dist/table/validate.js.map +1 -0
- package/dist/utils/Stack.d.ts +4 -0
- package/dist/utils/Stack.js +16 -1
- package/dist/utils/Stack.js.map +1 -1
- package/dist/utils/errorWindowBuilder.js +11 -11
- package/dist/utils/errorWindowBuilder.js.map +1 -1
- package/dist/utils/loadFiles.d.ts +8 -0
- package/dist/utils/loadFiles.js +47 -0
- package/dist/utils/loadFiles.js.map +1 -0
- package/dist/utils/reducers.d.ts +9 -0
- package/dist/utils/reducers.js +56 -0
- package/dist/utils/reducers.js.map +1 -0
- package/example/LoLang/example.ts +16 -15
- package/example/kleene-test/example.ts +35 -17
- package/example/math/example.ts +12 -7
- package/example/selfhosted/example.ts +30 -24
- package/opencode.json +1 -1
- package/package.json +2 -2
- package/src/cli.ts +132 -23
- package/src/generator.ts +49 -8
- package/src/index.ts +55 -3
- package/src/meta/common.ts +116 -0
- package/src/meta/selfhosted.ts +42 -11
- package/src/parser.ts +243 -51
- package/src/table/selfhosted.ts +48 -12
- package/src/table/validate.ts +143 -0
- package/src/utils/Stack.ts +22 -2
- package/src/utils/errorWindowBuilder.ts +15 -13
- package/src/utils/loadFiles.ts +54 -0
- package/src/utils/reducers.ts +93 -0
- package/test/cli.test.ts +144 -0
- package/test/codegen.test.ts +75 -0
- package/test/grammar.test.ts +243 -0
- package/test/helpers.ts +59 -0
- package/test/lalr.test.ts +110 -0
- package/test/parser.test.ts +377 -0
- package/test/table.test.ts +144 -0
- package/tsconfig.json +1 -1
- package/example/LoLang/table2.txt +0 -497
- package/example/complicated/grammar.txt +0 -37
package/src/meta/common.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { ColumnAndRow, Token } from "@scinorandex/slex";
|
|
2
|
+
import { Result } from "../utils/Result";
|
|
2
3
|
|
|
3
4
|
export type Production = {
|
|
4
5
|
lhs: GrammarToken;
|
|
@@ -8,6 +9,8 @@ export type Production = {
|
|
|
8
9
|
rhs: { type: "terminal" | "variable"; token: GrammarToken; identifier: string; name: string | null }[];
|
|
9
10
|
};
|
|
10
11
|
|
|
12
|
+
export type ProductionRhsItem = Production["rhs"][number];
|
|
13
|
+
|
|
11
14
|
const dehydateGrammarToken = ({ column, lexeme, line, type }: GrammarToken) => {
|
|
12
15
|
return { column, lexeme, line, type };
|
|
13
16
|
};
|
|
@@ -50,6 +53,119 @@ export const hydrateProduction = (production: ReturnType<typeof dehydateProducti
|
|
|
50
53
|
};
|
|
51
54
|
|
|
52
55
|
export type GrammarTokenMetadata = {};
|
|
56
|
+
|
|
57
|
+
/** A readable one-liner for an unknown thrown value, used to turn it into a `Result` failure. */
|
|
58
|
+
export const describeThrowable = (err: unknown) =>
|
|
59
|
+
err instanceof Error ? err.message : `Unexpected error: ${String(err)}`;
|
|
60
|
+
|
|
61
|
+
/** A stand-in token used to report errors that do not belong to a specific place in a grammar. */
|
|
62
|
+
export const syntheticGrammarToken = () =>
|
|
63
|
+
new Token(GrammarTokenType.EOF, "", new ColumnAndRow(0, 0), {}) as GrammarToken;
|
|
64
|
+
|
|
65
|
+
export type ProductionWarningKind = "duplicate-rhs-name" | "unreachable-production";
|
|
66
|
+
|
|
67
|
+
export type ProductionWarning = {
|
|
68
|
+
kind: ProductionWarningKind;
|
|
69
|
+
reason: string;
|
|
70
|
+
token: GrammarToken;
|
|
71
|
+
};
|
|
72
|
+
|
|
73
|
+
export type ValidateProductionsOptions = {
|
|
74
|
+
/** Called once per warning. Warnings never stop the parser generator, they only flag suspicious grammars. */
|
|
75
|
+
onWarning?: (warning: ProductionWarning) => void;
|
|
76
|
+
};
|
|
77
|
+
|
|
78
|
+
const at = (token: GrammarToken) => `${token.line}:${token.column}`;
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Checks a set of unrolled productions for the mistakes that would otherwise surface as an
|
|
82
|
+
* inscrutable `TypeError` deep inside state generation or parsing:
|
|
83
|
+
*
|
|
84
|
+
* - a variable used on the right hand side that no production defines
|
|
85
|
+
* - a production with an empty right hand side (Sparse has no support for empty productions)
|
|
86
|
+
* - the start symbol being used on the right hand side (it must only appear on the left hand side)
|
|
87
|
+
*
|
|
88
|
+
* Suspicious-but-legal grammars (duplicate symbol names, unreachable productions) are reported
|
|
89
|
+
* through `onWarning` instead of failing.
|
|
90
|
+
*/
|
|
91
|
+
export const validateProductions = (
|
|
92
|
+
productions: Production[],
|
|
93
|
+
options: ValidateProductionsOptions = {},
|
|
94
|
+
): Result<Production[]> => {
|
|
95
|
+
const warn = (kind: ProductionWarningKind, reason: string, token: GrammarToken) => {
|
|
96
|
+
options.onWarning?.({ kind, reason, token });
|
|
97
|
+
};
|
|
98
|
+
|
|
99
|
+
if (productions.length === 0)
|
|
100
|
+
return {
|
|
101
|
+
success: false,
|
|
102
|
+
reason: "The grammar does not contain any productions. Is the grammar file empty?",
|
|
103
|
+
token: syntheticGrammarToken(),
|
|
104
|
+
};
|
|
105
|
+
|
|
106
|
+
const lhsIdentifiers = new Set(productions.map((production) => production.identifier));
|
|
107
|
+
const referencedIdentifiers = new Set<string>();
|
|
108
|
+
|
|
109
|
+
for (const production of productions) {
|
|
110
|
+
if (production.rhs.length === 0)
|
|
111
|
+
return {
|
|
112
|
+
success: false,
|
|
113
|
+
reason: `Production "${production.identifier}" (line ${at(production.lhs)}) has an empty right hand side. Sparse does not support empty productions, so a lone "(...)? " group such as "<X: x>: ([A])?;" is not a valid production.`,
|
|
114
|
+
token: production.lhs,
|
|
115
|
+
};
|
|
116
|
+
|
|
117
|
+
const seenNames = new Set<string>();
|
|
118
|
+
for (const rhsItem of production.rhs) {
|
|
119
|
+
if (rhsItem.type === "variable") {
|
|
120
|
+
referencedIdentifiers.add(rhsItem.identifier);
|
|
121
|
+
if (lhsIdentifiers.has(rhsItem.identifier) === false)
|
|
122
|
+
return {
|
|
123
|
+
success: false,
|
|
124
|
+
reason: `Variable "${rhsItem.identifier}" is used on the right hand side of production "${production.identifier}" (line ${at(production.lhs)}) but no production defines it as its left hand side.`,
|
|
125
|
+
token: rhsItem.token,
|
|
126
|
+
};
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
if (rhsItem.name != null) {
|
|
130
|
+
if (seenNames.has(rhsItem.name))
|
|
131
|
+
warn(
|
|
132
|
+
"duplicate-rhs-name",
|
|
133
|
+
`Production "${production.identifier}" (line ${at(production.lhs)}) names more than one symbol "${rhsItem.name}". Only the last one ends up on the reducer's "bag", use the positional input to reach the others.`,
|
|
134
|
+
rhsItem.token,
|
|
135
|
+
);
|
|
136
|
+
seenNames.add(rhsItem.name);
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
// The first production is the accept production: its left hand side is the start symbol, which
|
|
142
|
+
// by definition cannot be referenced from the right hand side of anything.
|
|
143
|
+
const startProduction = productions[0];
|
|
144
|
+
if (referencedIdentifiers.has(startProduction.identifier))
|
|
145
|
+
return {
|
|
146
|
+
success: false,
|
|
147
|
+
reason: `"${startProduction.identifier}" is the start symbol of the grammar, so it cannot be used on the right hand side of a production. It is used at line ${at(startProduction.lhs)}.`,
|
|
148
|
+
token: startProduction.lhs,
|
|
149
|
+
};
|
|
150
|
+
|
|
151
|
+
for (let i = 1; i < productions.length; i++) {
|
|
152
|
+
const production = productions[i];
|
|
153
|
+
if (referencedIdentifiers.has(production.identifier)) continue;
|
|
154
|
+
warn(
|
|
155
|
+
"unreachable-production",
|
|
156
|
+
`Production "${production.identifier}" (line ${at(production.lhs)}) is never used on the right hand side of another production, so the parser can never reach it.`,
|
|
157
|
+
production.lhs,
|
|
158
|
+
);
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
return { success: true, value: productions };
|
|
162
|
+
};
|
|
163
|
+
|
|
164
|
+
/** True when the identifier is a variable (`<FOO>`) rather than a terminal (`[FOO]`). */
|
|
165
|
+
export const isVariableIdentifier = (identifier: string) => identifier.startsWith("<") && identifier.endsWith(">");
|
|
166
|
+
|
|
167
|
+
export const terminalIdentifier = (name: string) => `[${name}]`;
|
|
168
|
+
export const variableIdentifier = (name: string) => `<${name}>`;
|
|
53
169
|
export enum GrammarTokenType {
|
|
54
170
|
IDENTIFIER,
|
|
55
171
|
L_ANGLE,
|
package/src/meta/selfhosted.ts
CHANGED
|
@@ -1,8 +1,17 @@
|
|
|
1
1
|
import { RegexEngine, Slex } from "@scinorandex/slex";
|
|
2
|
+
import { describeThrowable, syntheticGrammarToken } from "../meta/common";
|
|
2
3
|
import { Result, Sparse } from "../index";
|
|
3
4
|
import { selfhosted } from "./states";
|
|
4
|
-
import { TableState } from "../parser";
|
|
5
|
-
import {
|
|
5
|
+
import { LR1ParserGraveError, TableState } from "../parser";
|
|
6
|
+
import {
|
|
7
|
+
GrammarToken,
|
|
8
|
+
GrammarTokenMetadata,
|
|
9
|
+
GrammarTokenType,
|
|
10
|
+
hydrateProduction,
|
|
11
|
+
Production,
|
|
12
|
+
ValidateProductionsOptions,
|
|
13
|
+
validateProductions,
|
|
14
|
+
} from "./common";
|
|
6
15
|
|
|
7
16
|
export const grammarLexerGenerator = new Slex<GrammarTokenType, GrammarTokenMetadata>({
|
|
8
17
|
EOF_TYPE: GrammarTokenType.EOF,
|
|
@@ -21,9 +30,12 @@ grammarLexerGenerator.addRule(
|
|
|
21
30
|
grammarLexerGenerator.addRule("letter", "${lowercase} | ${uppercase}");
|
|
22
31
|
grammarLexerGenerator.addRule("digit", "0 | 1 | 2 | 3 | 4 | 5 | 6 | 7 | 8 | 9");
|
|
23
32
|
grammarLexerGenerator.addRule("alphanumeric", "${letter} | ${digit}");
|
|
33
|
+
// Dashes are allowed inside identifiers (but not first) so that the autogenerated productions of a
|
|
34
|
+
// "*" or "+" group, which are named "autogen-0", "autogen-1", ... survive a round trip through a
|
|
35
|
+
// table file.
|
|
24
36
|
grammarLexerGenerator.addRule(
|
|
25
37
|
"identifier",
|
|
26
|
-
"(${letter} | $_) (${letter} | ${digit} | $_)*",
|
|
38
|
+
"(${letter} | $_) (${letter} | ${digit} | $_ | $-)*",
|
|
27
39
|
GrammarTokenType.IDENTIFIER,
|
|
28
40
|
);
|
|
29
41
|
grammarLexerGenerator.addRule("colon", "$:", GrammarTokenType.COLON);
|
|
@@ -312,8 +324,15 @@ const reducers: { [key: string]: Reducer } = {
|
|
|
312
324
|
},
|
|
313
325
|
};
|
|
314
326
|
|
|
315
|
-
export const tryBuildProductions = (
|
|
316
|
-
|
|
327
|
+
export const tryBuildProductions = (
|
|
328
|
+
lexer: LexerInterface | string,
|
|
329
|
+
options: ValidateProductionsOptions = {},
|
|
330
|
+
): Result<Production[]> => {
|
|
331
|
+
try {
|
|
332
|
+
if (typeof lexer === "string") lexer = grammarLexerGenerator.generate(lexer, () => ({}));
|
|
333
|
+
} catch (err) {
|
|
334
|
+
return { success: false, reason: describeThrowable(err), token: syntheticGrammarToken() };
|
|
335
|
+
}
|
|
317
336
|
|
|
318
337
|
const parserGenerator = getSelfHostedParserGenerator();
|
|
319
338
|
const parser = parserGenerator.generate(lexer as RegexEngine<GrammarTokenType, GrammarTokenMetadata>, {
|
|
@@ -322,19 +341,31 @@ export const tryBuildProductions = (lexer: LexerInterface | string): Result<Prod
|
|
|
322
341
|
},
|
|
323
342
|
});
|
|
324
343
|
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
344
|
+
let programNode: ProgramNode;
|
|
345
|
+
try {
|
|
346
|
+
programNode = parser.parse().result as ProgramNode;
|
|
347
|
+
} catch (err) {
|
|
348
|
+
// The grammar parser has no error recovery, so a malformed grammar surfaces as a
|
|
349
|
+
// LR1ParserGraveError. Translate it into the Result failure every caller already handles.
|
|
350
|
+
if (err instanceof LR1ParserGraveError && err.currentToken != null)
|
|
351
|
+
return { success: false, reason: err.reason, token: err.currentToken as GrammarToken };
|
|
352
|
+
return { success: false, reason: describeThrowable(err), token: syntheticGrammarToken() };
|
|
353
|
+
}
|
|
354
|
+
|
|
355
|
+
return validateProductions(programNode.getProductions(), options);
|
|
328
356
|
};
|
|
329
357
|
|
|
330
|
-
interface LexerInterface {
|
|
358
|
+
export interface LexerInterface {
|
|
331
359
|
peekNextToken(): GrammarToken;
|
|
332
360
|
getNextToken(): GrammarToken;
|
|
333
361
|
hasNextToken(): boolean;
|
|
334
362
|
}
|
|
335
363
|
|
|
336
|
-
export const buildProductions = (
|
|
337
|
-
|
|
364
|
+
export const buildProductions = (
|
|
365
|
+
lexer: LexerInterface | string,
|
|
366
|
+
options: ValidateProductionsOptions = {},
|
|
367
|
+
): Production[] => {
|
|
368
|
+
const result = tryBuildProductions(lexer, options);
|
|
338
369
|
if (result.success === false)
|
|
339
370
|
throw new Error(
|
|
340
371
|
`Encountered error "${result.reason}" while parsing grammar at ${result.token.line}:${result.token.column}`,
|
package/src/parser.ts
CHANGED
|
@@ -2,15 +2,24 @@ import { Slex, Token } from "@scinorandex/slex";
|
|
|
2
2
|
import { generateStates } from "./generator";
|
|
3
3
|
import { Stack } from "./utils/Stack";
|
|
4
4
|
import { Result } from "./utils/Result";
|
|
5
|
-
import { Production } from "./meta/common";
|
|
5
|
+
import { GrammarToken, Production, ProductionWarning } from "./meta/common";
|
|
6
|
+
import { loadGrammar, loadTable } from "./utils/loadFiles";
|
|
7
|
+
import { validateTable } from "./table/validate";
|
|
6
8
|
|
|
7
9
|
export type TableAction = { type: "shift" | "reduce" | "goto"; value: number };
|
|
8
10
|
|
|
9
11
|
export class TableState {
|
|
10
12
|
public readonly actions: Map<string, TableAction>;
|
|
11
13
|
|
|
12
|
-
|
|
14
|
+
/**
|
|
15
|
+
* The actions exactly as they were written in a table file (`"s12"`, `"r3"`, `"7"`), kept so that
|
|
16
|
+
* malformed entries can be reported with the text the user actually typed.
|
|
17
|
+
*/
|
|
18
|
+
public readonly rawActions: Map<string, string>;
|
|
19
|
+
|
|
20
|
+
constructor(actions = new Map<string, TableAction>(), rawActions = new Map<string, string>()) {
|
|
13
21
|
this.actions = actions;
|
|
22
|
+
this.rawActions = rawActions;
|
|
14
23
|
}
|
|
15
24
|
|
|
16
25
|
public getTerminalAction(terminal: string): TableAction | undefined {
|
|
@@ -21,16 +30,41 @@ export class TableState {
|
|
|
21
30
|
return this.actions.get(variable);
|
|
22
31
|
}
|
|
23
32
|
|
|
33
|
+
/** Every terminal this state can consume, without the surrounding brackets, sorted. */
|
|
34
|
+
public getTerminalKeys(): string[] {
|
|
35
|
+
return this.getKeys("[", "]");
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/** Every variable this state can reduce past, without the surrounding angle brackets, sorted. */
|
|
39
|
+
public getVariableKeys(): string[] {
|
|
40
|
+
return this.getKeys("<", ">");
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
private getKeys(prefix: string, suffix: string): string[] {
|
|
44
|
+
const keys: string[] = [];
|
|
45
|
+
for (const key of this.actions.keys())
|
|
46
|
+
if (key.startsWith(prefix) && key.endsWith(suffix))
|
|
47
|
+
keys.push(key.substring(prefix.length, key.length - suffix.length));
|
|
48
|
+
return keys.sort();
|
|
49
|
+
}
|
|
50
|
+
|
|
24
51
|
toJSObject() {
|
|
25
52
|
return Object.fromEntries(this.actions);
|
|
26
53
|
}
|
|
27
54
|
|
|
28
55
|
static fromJSObject(json: ReturnType<TableState["toJSObject"]>) {
|
|
29
|
-
|
|
56
|
+
const actions = new Map(Object.entries(json));
|
|
57
|
+
const rawActions = new Map(
|
|
58
|
+
[...actions].map(([key, action]) => [
|
|
59
|
+
key,
|
|
60
|
+
action.type === "goto" ? `${action.value}` : `${action.type === "shift" ? "s" : "r"}${action.value}`,
|
|
61
|
+
]),
|
|
62
|
+
);
|
|
63
|
+
return new TableState(actions, rawActions);
|
|
30
64
|
}
|
|
31
65
|
}
|
|
32
66
|
|
|
33
|
-
type ReducerType<TokenType, Metadata, Node> = (
|
|
67
|
+
export type ReducerType<TokenType, Metadata, Node> = (
|
|
34
68
|
newInput: { bag: any; name: string | null },
|
|
35
69
|
oldInput: { input: LR1StackSymbol<TokenType, Metadata, Node>[]; index: number },
|
|
36
70
|
) => Node;
|
|
@@ -39,21 +73,46 @@ export type LR1StackSymbol<TokenType, Metadata, Node> =
|
|
|
39
73
|
| { type: "token"; token: Token<TokenType, Metadata> }
|
|
40
74
|
| { type: "node"; node: Node };
|
|
41
75
|
|
|
76
|
+
export type SparseOptions<TokenType> = {
|
|
77
|
+
productions: Production[];
|
|
78
|
+
states: TableState[];
|
|
79
|
+
toStringifiedTokenType: (tokenType: TokenType) => string;
|
|
80
|
+
/**
|
|
81
|
+
* Cross-check the productions against the parsing table right away, so a stale table is reported
|
|
82
|
+
* here instead of as an opaque crash in the middle of a parse. Off by default.
|
|
83
|
+
*/
|
|
84
|
+
validate?: boolean;
|
|
85
|
+
/** Name of the grammar file, used to make validation errors actionable. */
|
|
86
|
+
source?: string;
|
|
87
|
+
};
|
|
88
|
+
|
|
89
|
+
export type FromProductionsOptions<TokenType> = {
|
|
90
|
+
productions: Production[];
|
|
91
|
+
toStringifiedTokenType: (tokenType: TokenType) => string;
|
|
92
|
+
mode?: "lr1" | "lalr1";
|
|
93
|
+
/** Generating states can take a while for big grammars; set to true to silence the timing log. */
|
|
94
|
+
onWarning?: (warning: ProductionWarning) => void;
|
|
95
|
+
onProgress?: (statesGenerated: number) => void;
|
|
96
|
+
};
|
|
97
|
+
|
|
42
98
|
export class Sparse<TokenType, Metadata, Node> {
|
|
43
|
-
public constructor(
|
|
44
|
-
|
|
45
|
-
productions:
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
) {}
|
|
99
|
+
public constructor(public readonly options: SparseOptions<TokenType>) {
|
|
100
|
+
if (options.validate === true) {
|
|
101
|
+
const result = validateTable(options.productions, options.states, { source: options.source });
|
|
102
|
+
if (result.success === false) throw new Error(`The parsing table does not match the grammar: ${result.reason}`);
|
|
103
|
+
}
|
|
104
|
+
}
|
|
50
105
|
|
|
51
|
-
public static tryFromProductions<TokenType, Metadata, Node>(
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
const statesResult = generateStates(options.productions, {
|
|
106
|
+
public static tryFromProductions<TokenType, Metadata, Node>(
|
|
107
|
+
options: FromProductionsOptions<TokenType>,
|
|
108
|
+
): Result<Sparse<TokenType, Metadata, Node>> {
|
|
109
|
+
const startedAt = Date.now();
|
|
110
|
+
|
|
111
|
+
const statesResult = generateStates(options.productions, {
|
|
112
|
+
mode: options.mode,
|
|
113
|
+
onWarning: options.onWarning,
|
|
114
|
+
onProgress: options.onProgress,
|
|
115
|
+
});
|
|
57
116
|
if (statesResult.success === false) return statesResult;
|
|
58
117
|
const states = statesResult.value;
|
|
59
118
|
|
|
@@ -66,14 +125,54 @@ export class Sparse<TokenType, Metadata, Node> {
|
|
|
66
125
|
return { success: true, value: sparse };
|
|
67
126
|
}
|
|
68
127
|
|
|
69
|
-
public static fromProductions<TokenType, Metadata, Node>(options: {
|
|
70
|
-
productions: Production[];
|
|
71
|
-
toStringifiedTokenType: (tokenType: TokenType) => string;
|
|
72
|
-
mode?: "lr1" | "lalr1";
|
|
73
|
-
}) {
|
|
128
|
+
public static fromProductions<TokenType, Metadata, Node>(options: FromProductionsOptions<TokenType>) {
|
|
74
129
|
const result = Sparse.tryFromProductions<TokenType, Metadata, Node>(options);
|
|
75
|
-
if (result.success === false)
|
|
76
|
-
|
|
130
|
+
if (result.success === false) throw new SparseGrammarError(result.reason, result.token);
|
|
131
|
+
return result.value;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/** Builds a parser generator from a grammar file and a prebuilt parsing table file. */
|
|
135
|
+
public static async tryFromGrammarFile<TokenType, Metadata, Node>(options: {
|
|
136
|
+
grammarPath: string;
|
|
137
|
+
tablePath: string;
|
|
138
|
+
toStringifiedTokenType: (tokenType: TokenType) => string;
|
|
139
|
+
onWarning?: (warning: ProductionWarning) => void;
|
|
140
|
+
/** Set to false to skip cross-checking the table against the grammar. */
|
|
141
|
+
validate?: boolean;
|
|
142
|
+
}): Promise<Result<Sparse<TokenType, Metadata, Node>>> {
|
|
143
|
+
const productionsResult = await loadGrammar(options.grammarPath, { onWarning: options.onWarning });
|
|
144
|
+
if (productionsResult.success === false) return productionsResult;
|
|
145
|
+
|
|
146
|
+
const statesResult = await loadTable(options.tablePath, { productions: productionsResult.value });
|
|
147
|
+
if (statesResult.success === false) return statesResult;
|
|
148
|
+
|
|
149
|
+
if (options.validate !== false) {
|
|
150
|
+
const tableResult = validateTable(productionsResult.value, statesResult.value, {
|
|
151
|
+
source: options.tablePath,
|
|
152
|
+
});
|
|
153
|
+
if (tableResult.success === false) return tableResult;
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
return {
|
|
157
|
+
success: true,
|
|
158
|
+
value: new Sparse<TokenType, Metadata, Node>({
|
|
159
|
+
productions: productionsResult.value,
|
|
160
|
+
states: statesResult.value,
|
|
161
|
+
toStringifiedTokenType: options.toStringifiedTokenType,
|
|
162
|
+
source: options.grammarPath,
|
|
163
|
+
}),
|
|
164
|
+
};
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
public static async fromGrammarFile<TokenType, Metadata, Node>(options: {
|
|
168
|
+
grammarPath: string;
|
|
169
|
+
tablePath: string;
|
|
170
|
+
toStringifiedTokenType: (tokenType: TokenType) => string;
|
|
171
|
+
onWarning?: (warning: ProductionWarning) => void;
|
|
172
|
+
validate?: boolean;
|
|
173
|
+
}): Promise<Sparse<TokenType, Metadata, Node>> {
|
|
174
|
+
const result = await Sparse.tryFromGrammarFile<TokenType, Metadata, Node>(options);
|
|
175
|
+
if (result.success === false) throw new SparseGrammarError(result.reason, result.token);
|
|
77
176
|
return result.value;
|
|
78
177
|
}
|
|
79
178
|
|
|
@@ -99,6 +198,12 @@ export type ParserRecoveryFunction<TokenType, Metadata, Node> = (options: {
|
|
|
99
198
|
success: true;
|
|
100
199
|
token?: Token<TokenType, Metadata>;
|
|
101
200
|
};
|
|
201
|
+
/**
|
|
202
|
+
* Pushes a token that the input did not contain (a synthesized semicolon, a closing brace, ...)
|
|
203
|
+
* onto both stacks, exactly as if the lexer had produced it. Unlike `finish({ newToken })`, which
|
|
204
|
+
* only hands the token back to the parser as a lookahead, this one actually shifts it.
|
|
205
|
+
*/
|
|
206
|
+
insertToken: (token: Token<TokenType, Metadata>) => { success: false; reason: string } | undefined;
|
|
102
207
|
states: TableState[];
|
|
103
208
|
}) => { success: true; token?: Token<TokenType, Metadata> } | { success: false; reason: string };
|
|
104
209
|
|
|
@@ -108,38 +213,52 @@ export type ParserResult<TokenType, Metadata, Node> = {
|
|
|
108
213
|
errors: ParserError<TokenType, Metadata>[];
|
|
109
214
|
};
|
|
110
215
|
|
|
111
|
-
class LR1Parser<TokenType, Metadata, Node> {
|
|
216
|
+
export class LR1Parser<TokenType, Metadata, Node> {
|
|
112
217
|
statesStack: Stack<number> = new Stack([0]);
|
|
113
218
|
symbolsStack: Stack<LR1StackSymbol<TokenType, Metadata, Node>> = new Stack();
|
|
114
219
|
exceptions: ParserError<TokenType, Metadata>[] = [];
|
|
115
220
|
|
|
116
221
|
public constructor(
|
|
117
|
-
public readonly options:
|
|
118
|
-
productions: Production[];
|
|
119
|
-
states: TableState[];
|
|
120
|
-
toStringifiedTokenType: (tokenType: TokenType) => string;
|
|
121
|
-
},
|
|
222
|
+
public readonly options: SparseOptions<TokenType>,
|
|
122
223
|
public readonly reducer: ReducerType<TokenType, Metadata, Node>,
|
|
123
224
|
public readonly recover: ParserRecoveryFunction<TokenType, Metadata, Node> | null,
|
|
124
225
|
public readonly lexer: ReturnType<Slex<TokenType, Metadata>["generate"]>,
|
|
125
226
|
) {}
|
|
126
227
|
|
|
228
|
+
/**
|
|
229
|
+
* Clears the stacks and the collected errors so the parser can be handed to a fresh lexer.
|
|
230
|
+
* A parser instance consumes its lexer, so the usual way to parse a second input is to call
|
|
231
|
+
* `generator.generate(newLexer, options)` again.
|
|
232
|
+
*/
|
|
233
|
+
public reset() {
|
|
234
|
+
this.statesStack = new Stack([0]);
|
|
235
|
+
this.symbolsStack = new Stack();
|
|
236
|
+
this.exceptions = [];
|
|
237
|
+
}
|
|
238
|
+
|
|
127
239
|
public parse(): ParserResult<TokenType, Metadata, Node> {
|
|
128
240
|
const { productions, states, toStringifiedTokenType } = this.options;
|
|
129
241
|
|
|
242
|
+
if (states.length === 0)
|
|
243
|
+
throw new LR1ParserGraveError("The parsing table is empty, so no token can be parsed", null as any);
|
|
244
|
+
|
|
245
|
+
const currentStates = () => {
|
|
246
|
+
const state = states[this.statesStack.peek()];
|
|
247
|
+
if (state == null)
|
|
248
|
+
throw new LR1ParserGraveError(
|
|
249
|
+
`The parsing table is out of sync with the grammar: state ${this.statesStack.peek()} does not exist (the table has ${states.length} states)`,
|
|
250
|
+
null as any,
|
|
251
|
+
);
|
|
252
|
+
return state;
|
|
253
|
+
};
|
|
254
|
+
|
|
130
255
|
while (true) {
|
|
131
|
-
let currentState =
|
|
256
|
+
let currentState = currentStates();
|
|
132
257
|
let token = this.lexer.peekNextToken();
|
|
133
|
-
let action =
|
|
258
|
+
let action = currentState.getTerminalAction(toStringifiedTokenType(token.type));
|
|
134
259
|
|
|
135
260
|
if (action == null) {
|
|
136
|
-
if (this.recover === null)
|
|
137
|
-
throw new LR1ParserGraveError(
|
|
138
|
-
`Invalid syntax. Expected ${[...states[currentState].actions.keys()].join(
|
|
139
|
-
",",
|
|
140
|
-
)} but got ${toStringifiedTokenType(token.type)}`,
|
|
141
|
-
token,
|
|
142
|
-
);
|
|
261
|
+
if (this.recover === null) throw new LR1ParserGraveError(this.syntaxErrorMessage(currentState, token), token);
|
|
143
262
|
|
|
144
263
|
const recoveryResult = this.recover({
|
|
145
264
|
lexer: this.lexer,
|
|
@@ -149,22 +268,24 @@ class LR1Parser<TokenType, Metadata, Node> {
|
|
|
149
268
|
addError: (reason: string) => this.exceptions.push({ message: reason, token: this.lexer.peekNextToken() }),
|
|
150
269
|
crash: (reason: string) => ({ success: false, reason }),
|
|
151
270
|
finish: (options?: { newToken: Token<TokenType, Metadata> }) => ({ success: true, token: options?.newToken }),
|
|
152
|
-
|
|
153
|
-
|
|
271
|
+
insertToken: (newToken: Token<TokenType, Metadata>) => this.insertToken(newToken),
|
|
272
|
+
isSafe: () => currentStates().actions.has(`[${toStringifiedTokenType(this.lexer.peekNextToken().type)}]`),
|
|
154
273
|
});
|
|
155
274
|
|
|
156
275
|
if (recoveryResult.success === false) throw new LR1ParserGraveError(recoveryResult.reason, token);
|
|
157
276
|
|
|
158
277
|
token = recoveryResult.token ?? this.lexer.peekNextToken();
|
|
159
|
-
currentState =
|
|
160
|
-
action =
|
|
278
|
+
currentState = currentStates();
|
|
279
|
+
action = currentState.getTerminalAction(toStringifiedTokenType(token.type));
|
|
280
|
+
|
|
281
|
+
if (action == null) throw new LR1ParserGraveError(this.syntaxErrorMessage(currentState, token), token);
|
|
161
282
|
}
|
|
162
283
|
|
|
163
284
|
const fixedAction = action as TableAction;
|
|
164
285
|
if (fixedAction.type === "shift") {
|
|
165
286
|
// add current token to the stack and push the next state
|
|
166
287
|
token = this.lexer.getNextToken();
|
|
167
|
-
this.
|
|
288
|
+
this.pushState(fixedAction.value);
|
|
168
289
|
this.symbolsStack.push({ type: "token", token: token });
|
|
169
290
|
}
|
|
170
291
|
|
|
@@ -173,6 +294,12 @@ class LR1Parser<TokenType, Metadata, Node> {
|
|
|
173
294
|
if (fixedAction.value === 0) break;
|
|
174
295
|
const production = productions[fixedAction.value];
|
|
175
296
|
|
|
297
|
+
if (production == null)
|
|
298
|
+
throw new LR1ParserGraveError(
|
|
299
|
+
`The parsing table is out of sync with the grammar: it reduces by production ${fixedAction.value}, but the grammar only has ${productions.length} productions`,
|
|
300
|
+
token,
|
|
301
|
+
);
|
|
302
|
+
|
|
176
303
|
// pop the stack and reduce by this production
|
|
177
304
|
const popped: LR1StackSymbol<TokenType, Metadata, Node>[] = [];
|
|
178
305
|
for (let i = 0; i < production.rhs.length; i++) {
|
|
@@ -200,17 +327,18 @@ class LR1Parser<TokenType, Metadata, Node> {
|
|
|
200
327
|
const node = this.reducer(newInput, oldInput);
|
|
201
328
|
this.symbolsStack.push({ type: "node", node });
|
|
202
329
|
} catch (err) {
|
|
203
|
-
const
|
|
204
|
-
|
|
330
|
+
const cause = err instanceof Error ? err.message : String(err);
|
|
331
|
+
const input = truncate(JSON.stringify(popped), 500);
|
|
205
332
|
throw new LR1ParserGraveError(
|
|
206
|
-
`Error while performing reduction for production
|
|
333
|
+
`Error while performing the reduction for production ${productionIndex}${describeProduction(production)}: ${cause}\nReduction input: ${input}`,
|
|
207
334
|
token,
|
|
335
|
+
err instanceof Error ? err : undefined,
|
|
208
336
|
);
|
|
209
337
|
}
|
|
210
338
|
|
|
211
339
|
// get the top node and figure out what state to add to state stack
|
|
212
340
|
const topNode = this.statesStack.peek();
|
|
213
|
-
const gotoAction =
|
|
341
|
+
const gotoAction = currentStates().getVariableAction(production.identifier);
|
|
214
342
|
|
|
215
343
|
if (gotoAction == undefined)
|
|
216
344
|
throw new LR1ParserGraveError(
|
|
@@ -223,21 +351,85 @@ class LR1Parser<TokenType, Metadata, Node> {
|
|
|
223
351
|
token,
|
|
224
352
|
);
|
|
225
353
|
|
|
226
|
-
this.
|
|
354
|
+
this.pushState(gotoAction.value);
|
|
227
355
|
}
|
|
228
356
|
}
|
|
229
357
|
|
|
358
|
+
// A table can accept the empty input by reducing to production 0 straight away, in which case
|
|
359
|
+
// nothing was ever shifted and there is no tree to hand back.
|
|
360
|
+
if (this.symbolsStack.isEmpty) return { result: null, errors: this.exceptions };
|
|
361
|
+
|
|
230
362
|
const top = this.symbolsStack.peek();
|
|
231
363
|
return { result: top.type === "node" ? top.node : null, errors: this.exceptions };
|
|
232
364
|
}
|
|
365
|
+
|
|
366
|
+
private pushState(state: number) {
|
|
367
|
+
if (state === undefined || this.options.states[state] == null)
|
|
368
|
+
throw new LR1ParserGraveError(
|
|
369
|
+
`The parsing table is out of sync with the grammar: it points at state ${state}, but the table only has ${this.options.states.length} states`,
|
|
370
|
+
null as any,
|
|
371
|
+
);
|
|
372
|
+
this.statesStack.push(state);
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
/** Shifts a token that was not in the input, used by error recovery. */
|
|
376
|
+
private insertToken(token: Token<TokenType, Metadata>): { success: false; reason: string } | undefined {
|
|
377
|
+
const state = this.options.states[this.statesStack.peek()];
|
|
378
|
+
const action = state?.getTerminalAction(this.options.toStringifiedTokenType(token.type));
|
|
379
|
+
|
|
380
|
+
if (action == null)
|
|
381
|
+
return {
|
|
382
|
+
success: false,
|
|
383
|
+
reason: `Cannot insert ${this.options.toStringifiedTokenType(token.type)} here: state ${this.statesStack.peek()} does not accept it. Expected ${state?.getTerminalKeys().join(", ") ?? "nothing"}`,
|
|
384
|
+
};
|
|
385
|
+
|
|
386
|
+
if (action.type !== "shift")
|
|
387
|
+
return {
|
|
388
|
+
success: false,
|
|
389
|
+
reason: `Cannot insert ${this.options.toStringifiedTokenType(
|
|
390
|
+
token.type,
|
|
391
|
+
)} here: state ${this.statesStack.peek()} reduces on it instead of shifting it`,
|
|
392
|
+
};
|
|
393
|
+
|
|
394
|
+
this.pushState(action.value);
|
|
395
|
+
this.symbolsStack.push({ type: "token", token });
|
|
396
|
+
return undefined;
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
private syntaxErrorMessage(state: TableState, token: Token<TokenType, Metadata>) {
|
|
400
|
+
const expected = state.getTerminalKeys();
|
|
401
|
+
const got = this.options.toStringifiedTokenType(token.type);
|
|
402
|
+
const lexeme = token.lexeme === "" ? "" : ` ("${token.lexeme}")`;
|
|
403
|
+
return `Invalid syntax at ${token.line}:${token.column}: got ${got}${lexeme}, but expected ${
|
|
404
|
+
expected.length === 0 ? "the end of the input" : `one of [${expected.join("], [")}]`
|
|
405
|
+
}`;
|
|
406
|
+
}
|
|
233
407
|
}
|
|
234
408
|
|
|
235
409
|
export class LR1ParserGraveError<TokenType, Metadata> extends Error {
|
|
236
410
|
constructor(
|
|
237
411
|
public readonly reason: string,
|
|
238
|
-
public readonly currentToken: Token<TokenType, Metadata
|
|
412
|
+
public readonly currentToken: Token<TokenType, Metadata> | null,
|
|
239
413
|
err?: Error,
|
|
240
414
|
) {
|
|
241
|
-
super(reason);
|
|
415
|
+
super(reason, err === undefined ? undefined : { cause: err });
|
|
416
|
+
this.name = "LR1ParserGraveError";
|
|
242
417
|
}
|
|
243
418
|
}
|
|
419
|
+
|
|
420
|
+
/** A grammar (or table) problem found before parsing started, always tied to a place in the file. */
|
|
421
|
+
export class SparseGrammarError extends Error {
|
|
422
|
+
constructor(
|
|
423
|
+
public readonly reason: string,
|
|
424
|
+
public readonly token: GrammarToken,
|
|
425
|
+
) {
|
|
426
|
+
super(`Encountered error "${reason}" at ${token.line}:${token.column}`);
|
|
427
|
+
this.name = "SparseGrammarError";
|
|
428
|
+
}
|
|
429
|
+
}
|
|
430
|
+
|
|
431
|
+
const truncate = (value: string, max: number) =>
|
|
432
|
+
value.length <= max ? value : `${value.slice(0, max)}... (${value.length} characters)`;
|
|
433
|
+
|
|
434
|
+
const describeProduction = (production: Production) =>
|
|
435
|
+
production.name == null ? ` (${production.identifier})` : ` ("${production.name}")`;
|