@scinorandex/sparse 0.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +132 -0
- package/dist/cli.d.ts +2 -0
- package/dist/cli.js +47 -0
- package/dist/cli.js.map +1 -0
- package/dist/generator.d.ts +28 -0
- package/dist/generator.js +290 -0
- package/dist/generator.js.map +1 -0
- package/dist/grammarParser.d.ts +27 -0
- package/dist/grammarParser.js +73 -0
- package/dist/grammarParser.js.map +1 -0
- package/dist/index.d.ts +5 -0
- package/dist/index.js +15 -0
- package/dist/index.js.map +1 -0
- package/dist/parser.d.ts +102 -0
- package/dist/parser.js +125 -0
- package/dist/parser.js.map +1 -0
- package/dist/tableParser.d.ts +16 -0
- package/dist/tableParser.js +59 -0
- package/dist/tableParser.js.map +1 -0
- package/dist/utils/Result.d.ts +9 -0
- package/dist/utils/Result.js +3 -0
- package/dist/utils/Result.js.map +1 -0
- package/dist/utils/Stack.d.ts +7 -0
- package/dist/utils/Stack.js +26 -0
- package/dist/utils/Stack.js.map +1 -0
- package/dist/utils/errorWindowBuilder.d.ts +2 -0
- package/dist/utils/errorWindowBuilder.js +20 -0
- package/dist/utils/errorWindowBuilder.js.map +1 -0
- package/example/LoLang/Features_Array_Methods.lol +31 -0
- package/example/LoLang/example.ts +214 -0
- package/example/LoLang/grammar.txt +149 -0
- package/example/LoLang/table.txt +497 -0
- package/example/math/example.ts +49 -0
- package/example/math/grammar.txt +11 -0
- package/example/math/table.txt +22 -0
- package/package.json +36 -0
- package/src/cli.ts +49 -0
- package/src/generator.ts +394 -0
- package/src/grammarParser.ts +97 -0
- package/src/index.ts +12 -0
- package/src/parser.ts +207 -0
- package/src/tableParser.ts +69 -0
- package/src/utils/Result.ts +3 -0
- package/src/utils/Stack.ts +23 -0
- package/src/utils/errorWindowBuilder.ts +23 -0
- package/tsconfig.json +30 -0
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
import { Slex } from "@scinorandex/slex";
|
|
2
|
+
import { buildProductions, buildStates, LR1StackSymbol, Sparse } from "../../src/index";
|
|
3
|
+
import fs from "fs/promises";
|
|
4
|
+
|
|
5
|
+
// prettier-ignore
|
|
6
|
+
enum TokenType {
|
|
7
|
+
PLUS, MINUS, STAR, SLASH, NUMBER, EOF
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
type Metadata = {};
|
|
11
|
+
|
|
12
|
+
const lexerGenerator = new Slex<TokenType, Metadata>({
|
|
13
|
+
EOF_TYPE: TokenType.EOF,
|
|
14
|
+
isHigherPrecedence: ({ current, next }) => false,
|
|
15
|
+
});
|
|
16
|
+
|
|
17
|
+
lexerGenerator.addRule("plus", "$+", TokenType.PLUS);
|
|
18
|
+
lexerGenerator.addRule("minus", "$-", TokenType.MINUS);
|
|
19
|
+
lexerGenerator.addRule("star", "$*", TokenType.STAR);
|
|
20
|
+
lexerGenerator.addRule("forward_slash", "$/", TokenType.SLASH);
|
|
21
|
+
lexerGenerator.addRule("digit", "0|1|2|3|4|5|6|7|8|9");
|
|
22
|
+
lexerGenerator.addRule("float_number", "(${digit})+$.(${digit})+");
|
|
23
|
+
lexerGenerator.addRule("decimal_number", "(${digit})+");
|
|
24
|
+
lexerGenerator.addRule("number_literal", "${float_number}|${decimal_number}", TokenType.NUMBER);
|
|
25
|
+
|
|
26
|
+
const lexer = lexerGenerator.generate(`2.4 + 3.5 * 1 / 456.789`, () => ({}));
|
|
27
|
+
|
|
28
|
+
type StringifiedNode = (StringifiedNode | string)[];
|
|
29
|
+
class Node {
|
|
30
|
+
constructor(public readonly nodes: LR1StackSymbol<TokenType, Metadata, Node>[]) {}
|
|
31
|
+
toObject(): StringifiedNode {
|
|
32
|
+
return this.nodes.map((node) => (node.type === "token" ? node.token.lexeme : node.node.toObject()));
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
async function main() {
|
|
37
|
+
const toStringifiedTokenType = (type: TokenType) => TokenType[type];
|
|
38
|
+
const productions = buildProductions(await fs.readFile("./example/math/grammar.txt", "utf8"));
|
|
39
|
+
const states = buildStates(await fs.readFile("./example/math/table.txt", "utf8"));
|
|
40
|
+
const parserGenerator = new Sparse<TokenType, Metadata, Node>({ productions, states, toStringifiedTokenType });
|
|
41
|
+
|
|
42
|
+
const parser = parserGenerator.generate(lexer, {
|
|
43
|
+
reducer: (input, productionIndex) => new Node(input),
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
console.log(parser.parse().result!.toObject());
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
main();
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
<S>: <PROGRAM>;
|
|
2
|
+
<PROGRAM>: <EXPRESSION> [EOF];
|
|
3
|
+
<PROGRAM>: [EOF];
|
|
4
|
+
<EXPRESSION>: <TERM_EXPRESSION>;
|
|
5
|
+
<TERM_EXPRESSION>: <FACTOR_EXPRESSION> [PLUS] <TERM_EXPRESSION>;
|
|
6
|
+
<TERM_EXPRESSION>: <FACTOR_EXPRESSION> [MINUS] <TERM_EXPRESSION>;
|
|
7
|
+
<TERM_EXPRESSION>: <FACTOR_EXPRESSION>;
|
|
8
|
+
<FACTOR_EXPRESSION>: <ENDPOINT> [STAR] <FACTOR_EXPRESSION>;
|
|
9
|
+
<FACTOR_EXPRESSION>: <ENDPOINT> [SLASH] <FACTOR_EXPRESSION>;
|
|
10
|
+
<FACTOR_EXPRESSION>: <ENDPOINT>;
|
|
11
|
+
<ENDPOINT>: [NUMBER];
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
[EOF]=s6, [NUMBER]=s7, <PROGRAM>=1, <EXPRESSION>=2, <TERM_EXPRESSION>=3, <FACTOR_EXPRESSION>=4, <ENDPOINT>=5
|
|
2
|
+
[EOF]=r0
|
|
3
|
+
[EOF]=s8
|
|
4
|
+
[EOF]=r3
|
|
5
|
+
[EOF]=r6, [PLUS]=s9, [MINUS]=s10
|
|
6
|
+
[PLUS]=r9, [MINUS]=r9, [EOF]=r9, [STAR]=s11, [SLASH]=s12
|
|
7
|
+
[EOF]=r2
|
|
8
|
+
[STAR]=r10, [SLASH]=r10, [EOF]=r10, [PLUS]=r10, [MINUS]=r10
|
|
9
|
+
[EOF]=r1
|
|
10
|
+
[NUMBER]=s7, <TERM_EXPRESSION>=13, <FACTOR_EXPRESSION>=4, <ENDPOINT>=5
|
|
11
|
+
[NUMBER]=s7, <TERM_EXPRESSION>=14, <FACTOR_EXPRESSION>=4, <ENDPOINT>=5
|
|
12
|
+
[NUMBER]=s7, <FACTOR_EXPRESSION>=15, <ENDPOINT>=16
|
|
13
|
+
[NUMBER]=s7, <FACTOR_EXPRESSION>=17, <ENDPOINT>=16
|
|
14
|
+
[EOF]=r4
|
|
15
|
+
[EOF]=r5
|
|
16
|
+
[PLUS]=r7, [MINUS]=r7, [EOF]=r7
|
|
17
|
+
[EOF]=r9, [PLUS]=r9, [MINUS]=r9, [STAR]=s18, [SLASH]=s19
|
|
18
|
+
[PLUS]=r8, [MINUS]=r8, [EOF]=r8
|
|
19
|
+
[NUMBER]=s7, <FACTOR_EXPRESSION>=20, <ENDPOINT>=16
|
|
20
|
+
[NUMBER]=s7, <FACTOR_EXPRESSION>=21, <ENDPOINT>=16
|
|
21
|
+
[EOF]=r7, [PLUS]=r7, [MINUS]=r7
|
|
22
|
+
[EOF]=r8, [PLUS]=r8, [MINUS]=r8
|
package/package.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@scinorandex/sparse",
|
|
3
|
+
"version": "0.0.1",
|
|
4
|
+
"description": "Yet another parser generator",
|
|
5
|
+
"main": "dist/index.js",
|
|
6
|
+
"scripts": {
|
|
7
|
+
"build": "tsc",
|
|
8
|
+
"watch": "tsc -w"
|
|
9
|
+
},
|
|
10
|
+
"bin": {
|
|
11
|
+
"sparse": "dist/cli.js"
|
|
12
|
+
},
|
|
13
|
+
"repository": {
|
|
14
|
+
"type": "git",
|
|
15
|
+
"url": "github.com/scinscinscin/sparse"
|
|
16
|
+
},
|
|
17
|
+
"keywords": [
|
|
18
|
+
"parser",
|
|
19
|
+
"parser-generator",
|
|
20
|
+
"yacc",
|
|
21
|
+
"bison",
|
|
22
|
+
"antlr",
|
|
23
|
+
"programming-languages"
|
|
24
|
+
],
|
|
25
|
+
"author": "scinorandex",
|
|
26
|
+
"license": "MIT",
|
|
27
|
+
"devDependencies": {
|
|
28
|
+
"@types/minimist": "^1.2.5",
|
|
29
|
+
"@types/node": "^22.15.19"
|
|
30
|
+
},
|
|
31
|
+
"resolutions": {},
|
|
32
|
+
"dependencies": {
|
|
33
|
+
"@scinorandex/slex": "^0.0.2",
|
|
34
|
+
"minimist": "^1.2.8"
|
|
35
|
+
}
|
|
36
|
+
}
|
package/src/cli.ts
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
#! /usr/bin/env node
|
|
2
|
+
import fs from "fs/promises";
|
|
3
|
+
import { GrammarToken, tryBuildProductions } from "./grammarParser";
|
|
4
|
+
import { generateStates } from "./generator";
|
|
5
|
+
import minimist from "minimist";
|
|
6
|
+
import path from "path";
|
|
7
|
+
import { buildErrorWindow } from "./utils/errorWindowBuilder";
|
|
8
|
+
|
|
9
|
+
const args = minimist(process.argv.slice(2));
|
|
10
|
+
|
|
11
|
+
function checkFileExists(filepath: string) {
|
|
12
|
+
return new Promise<boolean>((resolve, reject) => {
|
|
13
|
+
fs.access(filepath, fs.constants.F_OK)
|
|
14
|
+
.then(() => resolve(true))
|
|
15
|
+
.catch(() => resolve(false));
|
|
16
|
+
});
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
async function main() {
|
|
20
|
+
if (typeof args.input !== "string" || typeof args.output !== "string") {
|
|
21
|
+
console.log("Usage: npx @scinorandex/sparse --input=<input> --output=<output>");
|
|
22
|
+
return;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
const inputFile = path.resolve(args.input);
|
|
26
|
+
const outputFile = path.resolve(args.output);
|
|
27
|
+
console.log(`Reading grammar from ${inputFile} and outputting table to ${outputFile}\n`);
|
|
28
|
+
|
|
29
|
+
if (!(await checkFileExists(inputFile))) return console.log(`File ${inputFile} does not exist`);
|
|
30
|
+
|
|
31
|
+
const grammar = await fs.readFile(args.input, "utf8");
|
|
32
|
+
const productionsResult = tryBuildProductions(grammar);
|
|
33
|
+
if (productionsResult.success === false) {
|
|
34
|
+
console.log(productionsResult.reason);
|
|
35
|
+
console.log(buildErrorWindow(grammar, productionsResult.token));
|
|
36
|
+
return;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
const generatorResult = generateStates(productionsResult.value);
|
|
40
|
+
if (generatorResult.success === false) {
|
|
41
|
+
console.log(generatorResult.reason);
|
|
42
|
+
console.log(buildErrorWindow(grammar, generatorResult.token));
|
|
43
|
+
return;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
await fs.writeFile(outputFile, generatorResult.value.toTable(), { encoding: "utf-8" });
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
main();
|
package/src/generator.ts
ADDED
|
@@ -0,0 +1,394 @@
|
|
|
1
|
+
import { GrammarToken, Production } from "./grammarParser";
|
|
2
|
+
import { TableState } from "./parser";
|
|
3
|
+
import { Result } from "./utils/Result";
|
|
4
|
+
|
|
5
|
+
const xContainsAllOfY = <T>(xs: Set<T>, ys: Set<T>) => [...ys].every((x) => xs.has(x));
|
|
6
|
+
const EOF_STRING = "[EOF]";
|
|
7
|
+
|
|
8
|
+
// Creates a list of first sets for each production
|
|
9
|
+
function computeFirstSets(allProductions: Production[]): Result<Map<string, Set<string>>> {
|
|
10
|
+
// Initialize the hashmap
|
|
11
|
+
const ret = new Map<string, Set<string>>();
|
|
12
|
+
for (const production of allProductions) ret.set(production.lhs.lexeme, new Set());
|
|
13
|
+
|
|
14
|
+
// Keep iterating through all productions until no edits are made
|
|
15
|
+
let wasEdited = true;
|
|
16
|
+
while (wasEdited) {
|
|
17
|
+
wasEdited = false;
|
|
18
|
+
|
|
19
|
+
for (const production of allProductions) {
|
|
20
|
+
const toBeModified = ret.get(production.lhs.lexeme)!;
|
|
21
|
+
|
|
22
|
+
const firstRhsToken = production.rhs[0];
|
|
23
|
+
if (firstRhsToken.type === "variable") {
|
|
24
|
+
const add = ret.get(firstRhsToken.token.lexeme);
|
|
25
|
+
|
|
26
|
+
if (add == null) {
|
|
27
|
+
return {
|
|
28
|
+
success: false,
|
|
29
|
+
reason: `Variable "${firstRhsToken.token.lexeme}" doesn't have a corresponding left hand side`,
|
|
30
|
+
token: firstRhsToken.token,
|
|
31
|
+
};
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
// check if the hashmap of the current production already contains all the items to be added
|
|
35
|
+
// if not then add them and set wasEdited to true so we iterate one more time
|
|
36
|
+
if (!xContainsAllOfY(toBeModified, add)) {
|
|
37
|
+
wasEdited = true;
|
|
38
|
+
ret.set(production.lhs.lexeme, new Set([...toBeModified, ...add]));
|
|
39
|
+
}
|
|
40
|
+
} else if (firstRhsToken.type === "terminal") {
|
|
41
|
+
// add terminal to toBeModified and set wasEdited to true
|
|
42
|
+
if (toBeModified.has(firstRhsToken.token.lexeme) == false) {
|
|
43
|
+
wasEdited = true;
|
|
44
|
+
ret.set(production.lhs.lexeme, new Set([...toBeModified, firstRhsToken.token.lexeme]));
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
return { success: true, value: ret };
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
function computeFollowSets(
|
|
54
|
+
allProductions: Production[]
|
|
55
|
+
): Result<{ firstSets: Map<string, Set<string>>; followSets: Map<string, Set<string>> }> {
|
|
56
|
+
const firstSetsResult = computeFirstSets(allProductions);
|
|
57
|
+
if (firstSetsResult.success === false) return firstSetsResult;
|
|
58
|
+
|
|
59
|
+
const firstSets = firstSetsResult.value;
|
|
60
|
+
|
|
61
|
+
// create a list of follow sets and add EOF to the initial production
|
|
62
|
+
const ret = new Map<string, Set<string>>();
|
|
63
|
+
for (const production of allProductions) ret.set(production.lhs.lexeme, new Set());
|
|
64
|
+
ret.get(allProductions[0].lhs.lexeme)!.add(EOF_STRING);
|
|
65
|
+
|
|
66
|
+
let wasEdited = true;
|
|
67
|
+
while (wasEdited) {
|
|
68
|
+
wasEdited = false;
|
|
69
|
+
|
|
70
|
+
for (const variable of ret.keys()) {
|
|
71
|
+
for (const currentProduction of allProductions) {
|
|
72
|
+
// check the rhs of current production for variable
|
|
73
|
+
const rhs = currentProduction.rhs;
|
|
74
|
+
|
|
75
|
+
for (let i = 0; i < rhs.length; i++) {
|
|
76
|
+
const currentRhsToken = rhs[i];
|
|
77
|
+
|
|
78
|
+
if (currentRhsToken.type === "variable" && currentRhsToken.token.lexeme === variable) {
|
|
79
|
+
if (i === rhs.length - 1) {
|
|
80
|
+
// we are at the end of rhs, so whatever is in currentProduction
|
|
81
|
+
// we need to also need to add to variable
|
|
82
|
+
const toBeAdded = ret.get(currentProduction.lhs.lexeme)!;
|
|
83
|
+
const receiver = ret.get(variable)!;
|
|
84
|
+
if (!xContainsAllOfY(receiver, toBeAdded)) {
|
|
85
|
+
wasEdited = true;
|
|
86
|
+
ret.set(variable, new Set([...toBeAdded, ...receiver]));
|
|
87
|
+
}
|
|
88
|
+
} else {
|
|
89
|
+
// not at the end of rhs, so we need to add the FIRST set of the next rhs token
|
|
90
|
+
const nextRhsToken = rhs[i + 1];
|
|
91
|
+
if (nextRhsToken.type === "terminal") {
|
|
92
|
+
// next token is a terminal, check if its in existng follow set and add if it doesn't exist
|
|
93
|
+
const existing = ret.get(variable)!;
|
|
94
|
+
if (!existing.has(nextRhsToken.token.lexeme)) {
|
|
95
|
+
wasEdited = true;
|
|
96
|
+
ret.set(variable, new Set([...existing, nextRhsToken.token.lexeme]));
|
|
97
|
+
}
|
|
98
|
+
} else if (nextRhsToken.type === "variable") {
|
|
99
|
+
// next token is a variable, get the first set of the variable
|
|
100
|
+
// check if it's not in the followset and add if not
|
|
101
|
+
const firstSet = firstSets.get(nextRhsToken.token.lexeme);
|
|
102
|
+
if (firstSet == null)
|
|
103
|
+
return {
|
|
104
|
+
success: false,
|
|
105
|
+
reason: `Variable "${nextRhsToken.token.lexeme}" doesn't have a corresponding left hand side`,
|
|
106
|
+
token: nextRhsToken.token,
|
|
107
|
+
};
|
|
108
|
+
|
|
109
|
+
if (!xContainsAllOfY(ret.get(variable)!, firstSet)) {
|
|
110
|
+
wasEdited = true;
|
|
111
|
+
ret.set(variable, new Set([...ret.get(variable)!, ...firstSet]));
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
return { success: true, value: { firstSets, followSets: ret } };
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
type Item = Production & { dot: number; lookahead: string[] };
|
|
125
|
+
type State = { itemSet: Item[]; kernel: Item; count: number };
|
|
126
|
+
|
|
127
|
+
export const generateStates = (productions: Production[]): Result<GeneratorResult> => {
|
|
128
|
+
const followSetsResult = computeFollowSets(productions);
|
|
129
|
+
if (followSetsResult.success == false) return followSetsResult;
|
|
130
|
+
|
|
131
|
+
const { firstSets, followSets } = followSetsResult.value;
|
|
132
|
+
|
|
133
|
+
// This funciton takes a token and determines what the next lookahead should be
|
|
134
|
+
// The implementation of this function ensures that the table is an LR(1) parsing table
|
|
135
|
+
const determineNextLookAhead = (
|
|
136
|
+
lhs: GrammarToken,
|
|
137
|
+
array: { type: "terminal" | "variable"; token: GrammarToken }[]
|
|
138
|
+
): Result<string[]> => {
|
|
139
|
+
if (array.length === 0) {
|
|
140
|
+
const followSet = followSets.get(lhs.lexeme);
|
|
141
|
+
if (followSet != undefined) return { success: true, value: [...followSet] };
|
|
142
|
+
|
|
143
|
+
return {
|
|
144
|
+
success: false,
|
|
145
|
+
reason: `Variable ${lhs.lexeme} not present in computed follow sets`,
|
|
146
|
+
token: lhs,
|
|
147
|
+
};
|
|
148
|
+
} else {
|
|
149
|
+
const next = array[0];
|
|
150
|
+
|
|
151
|
+
if (next.type === "terminal") return { success: true, value: [next.token.lexeme] };
|
|
152
|
+
const testing = firstSets.get(next.token.lexeme);
|
|
153
|
+
if (testing !== undefined) return { success: true, value: [...testing] };
|
|
154
|
+
|
|
155
|
+
return {
|
|
156
|
+
success: false,
|
|
157
|
+
reason: `Variable ${next.token.lexeme} not present in computed first sets`,
|
|
158
|
+
token: next.token,
|
|
159
|
+
};
|
|
160
|
+
}
|
|
161
|
+
};
|
|
162
|
+
|
|
163
|
+
// This fnuction creates the initial item set based on the
|
|
164
|
+
// production provided. It creating the initial item and expands it
|
|
165
|
+
function generateInitialItemSet(production: Production) {
|
|
166
|
+
const initialItem = { lhs: production.lhs, rhs: production.rhs, dot: 0, lookahead: [EOF_STRING] } as Item;
|
|
167
|
+
return expandItemSet([initialItem]);
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
// This function expands the item sets provided to it
|
|
171
|
+
function expandItemSet(items: Item[]): Result<{ itemSet: Item[]; kernel: Item }> {
|
|
172
|
+
const itemSet: Item[] = [...items];
|
|
173
|
+
|
|
174
|
+
// Create a queue of unprocessed items and keep shifting until there are no more items left
|
|
175
|
+
const unprocesssedItems: Item[] = [...items];
|
|
176
|
+
while (unprocesssedItems.length > 0) {
|
|
177
|
+
const currentItem = unprocesssedItems.shift()!;
|
|
178
|
+
|
|
179
|
+
// Check if the symbol after the dot is a non terminal
|
|
180
|
+
const after = currentItem.rhs[currentItem.dot];
|
|
181
|
+
if (after == null) continue; // TODO: check if this should be here
|
|
182
|
+
|
|
183
|
+
if (after.type === "variable") {
|
|
184
|
+
// Find prodctions whose left hand side is the symbol after the dot
|
|
185
|
+
const newProductions = productions.filter((p) => p.lhs.lexeme === after.token.lexeme);
|
|
186
|
+
|
|
187
|
+
// Compute the lookahead for the new productions to be added to the item set
|
|
188
|
+
const rest = currentItem.rhs.slice(currentItem.dot + 1);
|
|
189
|
+
|
|
190
|
+
const lookaheadResult = determineNextLookAhead(currentItem.lhs, rest);
|
|
191
|
+
if (lookaheadResult.success === false) return lookaheadResult;
|
|
192
|
+
const lookahead = lookaheadResult.value;
|
|
193
|
+
|
|
194
|
+
for (const newProduction of newProductions) {
|
|
195
|
+
// Create the new item and check if it already exists in the item set
|
|
196
|
+
// If it doesn't exist, add it to the item set and the queue of unprocessed items
|
|
197
|
+
const newItem = { lhs: newProduction.lhs, rhs: newProduction.rhs, dot: 0, lookahead } as Item;
|
|
198
|
+
const encoding = JSON.stringify(newItem);
|
|
199
|
+
|
|
200
|
+
if (itemSet.some((i) => JSON.stringify(i) === encoding) == false) {
|
|
201
|
+
unprocesssedItems.push(newItem);
|
|
202
|
+
itemSet.push(newItem);
|
|
203
|
+
}
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
// Return the item set and the kernel
|
|
209
|
+
return { success: true, value: { itemSet, kernel: items[0] } };
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
const initialItemSetResult = generateInitialItemSet(productions[0]);
|
|
213
|
+
if (initialItemSetResult.success === false) return initialItemSetResult;
|
|
214
|
+
const initialItemSet = initialItemSetResult.value;
|
|
215
|
+
|
|
216
|
+
function generateStates(_initialState: { itemSet: Item[]; kernel: Item }): Result<{
|
|
217
|
+
states: State[];
|
|
218
|
+
GotoTable: Map<string, number>[];
|
|
219
|
+
ActionTable: Map<string, { action: "shift" | "reduce"; value: number }>[];
|
|
220
|
+
}> {
|
|
221
|
+
const initialState = { ..._initialState, count: 0 };
|
|
222
|
+
// Create a list of states to be returned and another list of states that have not been vistited
|
|
223
|
+
const states = [initialState] as State[];
|
|
224
|
+
const unprocessedStates = [initialState] as State[];
|
|
225
|
+
|
|
226
|
+
// array of maps whose keys is a variable and the value are the state to goto next
|
|
227
|
+
const GotoTable = [] as Map<string, number>[];
|
|
228
|
+
// array of maps whose keys is a terminal is a value of either to shift or reduce
|
|
229
|
+
const ActionTable = [] as Map<string, { action: "shift" | "reduce"; value: number }>[];
|
|
230
|
+
|
|
231
|
+
// While there is an unprocessed state, visit it
|
|
232
|
+
while (unprocessedStates.length > 0) {
|
|
233
|
+
const currentState = unprocessedStates.shift()!;
|
|
234
|
+
const nextStates_Goto = new Map<string, Item[]>();
|
|
235
|
+
const nextStates_Shift = new Map<string, Item[]>();
|
|
236
|
+
|
|
237
|
+
// determine transitions out of the current state by iterating
|
|
238
|
+
// through all the items in the item set of the state
|
|
239
|
+
calculateNextItem: for (const item of currentState.itemSet) {
|
|
240
|
+
// check next symbol after the dot
|
|
241
|
+
const nextSymbol = item.rhs[item.dot];
|
|
242
|
+
|
|
243
|
+
// no next symbol is available, therefore we need to create a reduction
|
|
244
|
+
if (nextSymbol == null) {
|
|
245
|
+
// Find the prodction that this item reduces to using the symbols on its right hand side
|
|
246
|
+
// TODO: check if the find can fail, honestly this might be invariant
|
|
247
|
+
const productionToReduceTo = productions
|
|
248
|
+
.map((p, i) => [p, i] as const)
|
|
249
|
+
.find(
|
|
250
|
+
([p, _]) => JSON.stringify(p.rhs) === JSON.stringify(item.rhs) && p.lhs.lexeme === item.lhs.lexeme
|
|
251
|
+
)![1];
|
|
252
|
+
|
|
253
|
+
// for each lookahead in the item, create a new entry in the action table to reduce to the production found
|
|
254
|
+
for (const lookahead of item.lookahead) {
|
|
255
|
+
if (ActionTable[currentState.count] === undefined) ActionTable[currentState.count] = new Map();
|
|
256
|
+
ActionTable[currentState.count]!.set(lookahead, { action: "reduce", value: productionToReduceTo });
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
continue calculateNextItem;
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
// Create the next item by shifting the dot by 1 place
|
|
263
|
+
const newItem: Item = { ...item, dot: item.dot + 1 };
|
|
264
|
+
|
|
265
|
+
// Determine if we're going to create a GOTO or SHIFT based on the type of the symbol after the dot
|
|
266
|
+
if (nextSymbol.type === "variable") {
|
|
267
|
+
// Next symbol is a variable so we need to add it to the GOTO table
|
|
268
|
+
// Check if the GOTO table already has an entry for the next symbola and either push or create the array
|
|
269
|
+
if (nextStates_Goto.has(nextSymbol.token.lexeme)) nextStates_Goto.get(nextSymbol.token.lexeme)!.push(newItem);
|
|
270
|
+
else nextStates_Goto.set(nextSymbol.token.lexeme, [newItem]);
|
|
271
|
+
} else if (nextSymbol.type === "terminal") {
|
|
272
|
+
// Next symbol is a terminal so we need to add it to the SHIFT table
|
|
273
|
+
// Check if the SHIFT table already has an entry for the next symbola and either push or create the array
|
|
274
|
+
if (nextStates_Shift.has(nextSymbol.token.lexeme))
|
|
275
|
+
nextStates_Shift.get(nextSymbol.token.lexeme)!.push(newItem);
|
|
276
|
+
else nextStates_Shift.set(nextSymbol.token.lexeme, [newItem]);
|
|
277
|
+
}
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
// For each GOTO transition from the current state, expand the item set that it points to
|
|
281
|
+
// and check if a state exist with the same itemset already exists
|
|
282
|
+
// If it doesn't exist, create a new state and add it to the list of states
|
|
283
|
+
// Add the existing / created state to the GOTO table
|
|
284
|
+
for (const [gotoTransition, gotoItems] of nextStates_Goto.entries()) {
|
|
285
|
+
const expandedItemsResult = expandItemSet(gotoItems);
|
|
286
|
+
if (expandedItemsResult.success === false) return expandedItemsResult;
|
|
287
|
+
const expandedItems = expandedItemsResult.value;
|
|
288
|
+
|
|
289
|
+
// check if a state exist with the same item sets
|
|
290
|
+
let existingState = states.find((s) => {
|
|
291
|
+
return (
|
|
292
|
+
s.kernel.lhs === expandedItems.kernel.lhs &&
|
|
293
|
+
s.kernel.rhs.length === expandedItems.kernel.rhs.length &&
|
|
294
|
+
JSON.stringify(s.kernel.rhs) === JSON.stringify(expandedItems.kernel.rhs) &&
|
|
295
|
+
s.kernel.lookahead.length === expandedItems.kernel.lookahead.length &&
|
|
296
|
+
JSON.stringify(s.kernel.lookahead) === JSON.stringify(expandedItems.kernel.lookahead) &&
|
|
297
|
+
JSON.stringify(s.itemSet) === JSON.stringify(expandedItems.itemSet)
|
|
298
|
+
);
|
|
299
|
+
});
|
|
300
|
+
|
|
301
|
+
if (GotoTable[currentState.count] === undefined) GotoTable[currentState.count] = new Map();
|
|
302
|
+
if (existingState == undefined) {
|
|
303
|
+
const newState: State = {
|
|
304
|
+
itemSet: expandedItems.itemSet,
|
|
305
|
+
kernel: expandedItems.kernel,
|
|
306
|
+
count: states.length,
|
|
307
|
+
};
|
|
308
|
+
states.push(newState);
|
|
309
|
+
unprocessedStates.push(newState);
|
|
310
|
+
existingState = newState;
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
GotoTable[currentState.count]!.set(gotoTransition, existingState.count);
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
// For each SHIFT transition from the current state, expand the item set that it points to
|
|
317
|
+
// and check if a state exist with the same itemset already exists
|
|
318
|
+
// If it doesn't exist, create a new state and add it to the list of states
|
|
319
|
+
// Add the existing / created state to the SHIFT table
|
|
320
|
+
for (const [shiftTransition, shiftItems] of nextStates_Shift.entries()) {
|
|
321
|
+
const expandedItemsResult = expandItemSet(shiftItems);
|
|
322
|
+
if (expandedItemsResult.success === false) return expandedItemsResult;
|
|
323
|
+
const expandedItems = expandedItemsResult.value;
|
|
324
|
+
|
|
325
|
+
// check if a state exist with the same item sets
|
|
326
|
+
let existingState = states.find((s) => {
|
|
327
|
+
return (
|
|
328
|
+
s.kernel.lhs === expandedItems.kernel.lhs &&
|
|
329
|
+
s.kernel.rhs.length === expandedItems.kernel.rhs.length &&
|
|
330
|
+
JSON.stringify(s.kernel.rhs) === JSON.stringify(expandedItems.kernel.rhs) &&
|
|
331
|
+
s.kernel.lookahead.length === expandedItems.kernel.lookahead.length &&
|
|
332
|
+
JSON.stringify(s.kernel.lookahead) === JSON.stringify(expandedItems.kernel.lookahead) &&
|
|
333
|
+
JSON.stringify(s.itemSet) === JSON.stringify(expandedItems.itemSet)
|
|
334
|
+
);
|
|
335
|
+
});
|
|
336
|
+
|
|
337
|
+
// If there is no entry in the ActionTable for the current state, create one
|
|
338
|
+
if (ActionTable[currentState.count] === undefined) ActionTable[currentState.count] = new Map();
|
|
339
|
+
|
|
340
|
+
// If no such state exists, create a new state and add it to the list of states
|
|
341
|
+
if (existingState == undefined) {
|
|
342
|
+
const newState: State = {
|
|
343
|
+
itemSet: expandedItems.itemSet,
|
|
344
|
+
kernel: expandedItems.kernel,
|
|
345
|
+
count: states.length,
|
|
346
|
+
};
|
|
347
|
+
states.push(newState);
|
|
348
|
+
unprocessedStates.push(newState);
|
|
349
|
+
existingState = newState;
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
ActionTable[currentState.count]!.set(shiftTransition, { action: "shift", value: existingState.count });
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
// return the set of states, the GOTO table and the ACTION table
|
|
357
|
+
return { success: true, value: { states, GotoTable, ActionTable } };
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
const StatesResult = generateStates(initialItemSet);
|
|
361
|
+
if (StatesResult.success === false) return StatesResult;
|
|
362
|
+
const { states, ActionTable, GotoTable } = StatesResult.value;
|
|
363
|
+
return { success: true, value: new GeneratorResult(states, ActionTable, GotoTable) };
|
|
364
|
+
};
|
|
365
|
+
|
|
366
|
+
export class GeneratorResult {
|
|
367
|
+
constructor(
|
|
368
|
+
public readonly states: State[],
|
|
369
|
+
public readonly ActionTable: Map<string, { action: "shift" | "reduce"; value: number }>[],
|
|
370
|
+
public readonly GotoTable: Map<string, number>[]
|
|
371
|
+
) {}
|
|
372
|
+
|
|
373
|
+
toTable(): string {
|
|
374
|
+
return this.ActionTable.map((actions, idx) => {
|
|
375
|
+
return [
|
|
376
|
+
...[...actions.entries()].map(([k, { action, value }]) => `${k}=${action === "reduce" ? "r" : "s"}${value}`),
|
|
377
|
+
...(this.GotoTable[idx] === undefined ? [] : [...this.GotoTable[idx].entries()].map(([k, v]) => `${k}=${v}`)),
|
|
378
|
+
].join(", ");
|
|
379
|
+
}).join("\n");
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
toStates(): TableState[] {
|
|
383
|
+
return this.ActionTable.map((actions, idx) => {
|
|
384
|
+
const tableState = new TableState();
|
|
385
|
+
for (const [k, { action, value }] of actions.entries())
|
|
386
|
+
tableState.actions.set(k, { type: action === "reduce" ? "reduce" : "shift", value });
|
|
387
|
+
|
|
388
|
+
if (this.GotoTable[idx] !== undefined)
|
|
389
|
+
for (const [k, value] of this.GotoTable[idx].entries()) tableState.actions.set(k, { type: "goto", value });
|
|
390
|
+
|
|
391
|
+
return tableState;
|
|
392
|
+
});
|
|
393
|
+
}
|
|
394
|
+
}
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
import { Slex, Token } from "@scinorandex/slex";
|
|
2
|
+
import { Result } from "./utils/Result";
|
|
3
|
+
|
|
4
|
+
export type Production = { lhs: GrammarToken; rhs: { type: "terminal" | "variable"; token: GrammarToken }[] };
|
|
5
|
+
|
|
6
|
+
export type GrammarTokenMetadata = {};
|
|
7
|
+
export enum GrammarTokenType {
|
|
8
|
+
PRODUCTION_NAME,
|
|
9
|
+
TOKEN_NAME,
|
|
10
|
+
COLON,
|
|
11
|
+
SEMICOLON,
|
|
12
|
+
EOF,
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
export type GrammarToken = Token<GrammarTokenType, GrammarTokenMetadata>;
|
|
16
|
+
|
|
17
|
+
export const grammarLexerGenerator = new Slex<GrammarTokenType, GrammarTokenMetadata>({
|
|
18
|
+
EOF_TYPE: GrammarTokenType.EOF,
|
|
19
|
+
isHigherPrecedence: () => false,
|
|
20
|
+
});
|
|
21
|
+
|
|
22
|
+
grammarLexerGenerator.addRule(
|
|
23
|
+
"lowercase",
|
|
24
|
+
"a | b | c | d | e | f | g | h | i | j | k | l | m | n | o | p | q | r | s | t | u | v | w | x | y | z"
|
|
25
|
+
);
|
|
26
|
+
grammarLexerGenerator.addRule(
|
|
27
|
+
"uppercase",
|
|
28
|
+
"A | B | C | D | E | F | G | H | I | J | K | L | M | N | O | P | Q | R | S | T | U | V | W | X | Y | Z"
|
|
29
|
+
);
|
|
30
|
+
grammarLexerGenerator.addRule("letter", "${lowercase} | ${uppercase}");
|
|
31
|
+
grammarLexerGenerator.addRule("digit", "0 | 1 | 2 | 3 | 4 | 5 | 6 | 7 | 8 | 9");
|
|
32
|
+
grammarLexerGenerator.addRule("alphanumeric", "${letter} | ${digit}");
|
|
33
|
+
grammarLexerGenerator.addRule("identifier", "(${letter} | ${digit} | $_)*");
|
|
34
|
+
grammarLexerGenerator.addRule("colon", "$:", GrammarTokenType.COLON);
|
|
35
|
+
grammarLexerGenerator.addRule("semicolon", "$;", GrammarTokenType.SEMICOLON);
|
|
36
|
+
grammarLexerGenerator.addRule("production_name", "$< ${identifier} $>", GrammarTokenType.PRODUCTION_NAME);
|
|
37
|
+
grammarLexerGenerator.addRule("token_name", "$[ ${identifier} $]", GrammarTokenType.TOKEN_NAME);
|
|
38
|
+
|
|
39
|
+
interface LexerInterface {
|
|
40
|
+
peekNextToken(): GrammarToken;
|
|
41
|
+
getNextToken(): GrammarToken;
|
|
42
|
+
hasNextToken(): boolean;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export const tryBuildProductions = (lexer: LexerInterface | string): Result<Production[]> => {
|
|
46
|
+
if (typeof lexer === "string") lexer = grammarLexerGenerator.generate(lexer, () => ({}));
|
|
47
|
+
|
|
48
|
+
const productions: Production[] = [];
|
|
49
|
+
|
|
50
|
+
const expect = (type: GrammarTokenType) => {
|
|
51
|
+
const token = lexer.peekNextToken();
|
|
52
|
+
if (token.type !== type) throw new Error(`Expected ${type}, got ${token.type}`);
|
|
53
|
+
lexer.getNextToken();
|
|
54
|
+
};
|
|
55
|
+
|
|
56
|
+
const buildProduction = (): Result<Production> => {
|
|
57
|
+
const productionLHS = lexer.getNextToken();
|
|
58
|
+
expect(GrammarTokenType.COLON);
|
|
59
|
+
|
|
60
|
+
const production: Production = { lhs: productionLHS, rhs: [] };
|
|
61
|
+
|
|
62
|
+
while (lexer.peekNextToken().type != GrammarTokenType.SEMICOLON) {
|
|
63
|
+
const productionRHS = lexer.getNextToken();
|
|
64
|
+
|
|
65
|
+
if (productionRHS.type == GrammarTokenType.PRODUCTION_NAME)
|
|
66
|
+
production.rhs.push({ type: "variable", token: productionRHS });
|
|
67
|
+
else if (productionRHS.type == GrammarTokenType.TOKEN_NAME)
|
|
68
|
+
production.rhs.push({ type: "terminal", token: productionRHS });
|
|
69
|
+
else
|
|
70
|
+
return {
|
|
71
|
+
success: false,
|
|
72
|
+
reason: `Unexpected token type: ${GrammarTokenType[productionRHS.type]}`,
|
|
73
|
+
token: productionRHS,
|
|
74
|
+
};
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
expect(GrammarTokenType.SEMICOLON);
|
|
78
|
+
return { success: true, value: production };
|
|
79
|
+
};
|
|
80
|
+
|
|
81
|
+
while (lexer.hasNextToken()) {
|
|
82
|
+
const result = buildProduction();
|
|
83
|
+
if (result.success === false) return result;
|
|
84
|
+
productions.push(result.value);
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
return { success: true, value: productions };
|
|
88
|
+
};
|
|
89
|
+
|
|
90
|
+
export const buildProductions = (lexer: LexerInterface | string): Production[] => {
|
|
91
|
+
const result = tryBuildProductions(lexer);
|
|
92
|
+
if (result.success === false)
|
|
93
|
+
throw new Error(
|
|
94
|
+
`Encountered error "${result.reason}" while parsing grammar at ${result.token.line}:${result.token.column}`
|
|
95
|
+
);
|
|
96
|
+
return result.value;
|
|
97
|
+
};
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
export { GeneratorResult, generateStates } from "./generator";
|
|
2
|
+
export { Production, buildProductions, tryBuildProductions } from "./grammarParser";
|
|
3
|
+
export { Result } from "./utils/Result";
|
|
4
|
+
export {
|
|
5
|
+
ParserRecoveryFunction,
|
|
6
|
+
Sparse,
|
|
7
|
+
LR1ParserGraveError,
|
|
8
|
+
LR1StackSymbol,
|
|
9
|
+
ParserResult,
|
|
10
|
+
ParserError,
|
|
11
|
+
} from "./parser";
|
|
12
|
+
export { buildStates } from "./tableParser";
|