@scinorandex/sparse 0.1.1 → 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/workflows/ci.yml +41 -0
- package/AGENTS.md +31 -11
- package/README.md +289 -29
- package/dist/cli.js +113 -24
- package/dist/cli.js.map +1 -1
- package/dist/generator.d.ts +3 -1
- package/dist/generator.js +29 -5
- package/dist/generator.js.map +1 -1
- package/dist/index.d.ts +16 -4
- package/dist/index.js +34 -1
- package/dist/index.js.map +1 -1
- package/dist/meta/common.d.ts +17 -0
- package/dist/meta/common.js +66 -1
- package/dist/meta/common.js.map +1 -1
- package/dist/meta/selfhosted.d.ts +4 -5
- package/dist/meta/selfhosted.js +49 -36
- package/dist/meta/selfhosted.js.map +1 -1
- package/dist/parser.d.ts +55 -35
- package/dist/parser.js +141 -18
- package/dist/parser.js.map +1 -1
- package/dist/table/selfhosted.d.ts +10 -1
- package/dist/table/selfhosted.js +32 -12
- package/dist/table/selfhosted.js.map +1 -1
- package/dist/table/validate.d.ts +17 -0
- package/dist/table/validate.js +84 -0
- package/dist/table/validate.js.map +1 -0
- package/dist/utils/Stack.d.ts +4 -0
- package/dist/utils/Stack.js +16 -1
- package/dist/utils/Stack.js.map +1 -1
- package/dist/utils/errorWindowBuilder.js +11 -11
- package/dist/utils/errorWindowBuilder.js.map +1 -1
- package/dist/utils/loadFiles.d.ts +8 -0
- package/dist/utils/loadFiles.js +47 -0
- package/dist/utils/loadFiles.js.map +1 -0
- package/dist/utils/reducers.d.ts +9 -0
- package/dist/utils/reducers.js +56 -0
- package/dist/utils/reducers.js.map +1 -0
- package/example/LoLang/example.ts +16 -15
- package/example/kleene-test/example.ts +35 -17
- package/example/math/example.ts +12 -7
- package/example/selfhosted/example.ts +30 -24
- package/opencode.json +1 -1
- package/package.json +2 -2
- package/src/cli.ts +132 -23
- package/src/generator.ts +49 -8
- package/src/index.ts +55 -3
- package/src/meta/common.ts +116 -0
- package/src/meta/selfhosted.ts +42 -11
- package/src/parser.ts +243 -51
- package/src/table/selfhosted.ts +48 -12
- package/src/table/validate.ts +143 -0
- package/src/utils/Stack.ts +22 -2
- package/src/utils/errorWindowBuilder.ts +15 -13
- package/src/utils/loadFiles.ts +54 -0
- package/src/utils/reducers.ts +93 -0
- package/test/cli.test.ts +144 -0
- package/test/codegen.test.ts +75 -0
- package/test/grammar.test.ts +243 -0
- package/test/helpers.ts +59 -0
- package/test/lalr.test.ts +110 -0
- package/test/parser.test.ts +377 -0
- package/test/table.test.ts +144 -0
- package/tsconfig.json +1 -1
- package/example/LoLang/table2.txt +0 -497
- package/example/complicated/grammar.txt +0 -37
package/src/table/selfhosted.ts
CHANGED
|
@@ -1,7 +1,9 @@
|
|
|
1
|
-
import { GrammarToken } from "../meta/common";
|
|
1
|
+
import { GrammarToken, Production, describeThrowable, syntheticGrammarToken } from "../meta/common";
|
|
2
2
|
import { BaseNode, getSelfHostedParserGenerator, grammarLexerGenerator, ListNode } from "../meta/selfhosted";
|
|
3
|
-
import { TableAction, TableState } from "../parser";
|
|
3
|
+
import { LR1ParserGraveError, TableAction, TableState } from "../parser";
|
|
4
4
|
import { selfhosted } from "./states";
|
|
5
|
+
import { Result } from "../utils/Result";
|
|
6
|
+
import { TableFormatWarning, validateTableStates } from "./validate";
|
|
5
7
|
|
|
6
8
|
class StateNode extends BaseNode {
|
|
7
9
|
constructor(public readonly items: ListNode<ItemNode>) {
|
|
@@ -10,9 +12,11 @@ class StateNode extends BaseNode {
|
|
|
10
12
|
|
|
11
13
|
toTableState(): TableState {
|
|
12
14
|
const state = new TableState();
|
|
13
|
-
|
|
15
|
+
// The item list is built right-recursively (rest.add(item)), so it comes out reversed.
|
|
16
|
+
for (const item of this.items.getItemsReversed()) {
|
|
14
17
|
const key = item.type === "terminal" ? `[${item.name.lexeme}]` : `<${item.name.lexeme}>`;
|
|
15
18
|
state.actions.set(key, item.action);
|
|
19
|
+
state.rawActions.set(key, item.rawAction);
|
|
16
20
|
}
|
|
17
21
|
return state;
|
|
18
22
|
}
|
|
@@ -23,6 +27,7 @@ class ItemNode extends BaseNode {
|
|
|
23
27
|
public readonly type: "terminal" | "variable",
|
|
24
28
|
public readonly name: GrammarToken,
|
|
25
29
|
public readonly action: TableAction,
|
|
30
|
+
public readonly rawAction: string,
|
|
26
31
|
) {
|
|
27
32
|
super();
|
|
28
33
|
}
|
|
@@ -46,22 +51,53 @@ const reducers: { [key: string]: Reducer } = {
|
|
|
46
51
|
type: bag.action.lexeme.startsWith("s") ? "shift" : "reduce",
|
|
47
52
|
value: parseInt(bag.action.lexeme.substring(1)),
|
|
48
53
|
};
|
|
49
|
-
return new ItemNode("terminal", bag.name, action);
|
|
54
|
+
return new ItemNode("terminal", bag.name, action, bag.action.lexeme);
|
|
50
55
|
},
|
|
51
56
|
|
|
52
57
|
variable: (bag: { name: GrammarToken; state: GrammarToken }) => {
|
|
53
58
|
const action: TableAction = { type: "goto", value: parseInt(bag.state.lexeme) };
|
|
54
|
-
return new ItemNode("variable", bag.name, action);
|
|
59
|
+
return new ItemNode("variable", bag.name, action, bag.state.lexeme);
|
|
55
60
|
},
|
|
56
61
|
};
|
|
57
62
|
|
|
58
|
-
export
|
|
59
|
-
|
|
63
|
+
export type BuildStatesOptions = {
|
|
64
|
+
/** Pass the grammar's productions to also check every "reduce" action against them. */
|
|
65
|
+
productions?: Production[];
|
|
66
|
+
/** Name of the file the table was read from, used to make error messages actionable. */
|
|
67
|
+
source?: string;
|
|
68
|
+
onWarning?: (warning: TableFormatWarning) => void;
|
|
69
|
+
};
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Reads a parsing table written by the CLI (`sparse --input=grammar.txt --output=table.txt`).
|
|
73
|
+
* Returns a `Result` instead of throwing so that table problems can be reported alongside the
|
|
74
|
+
* offending grammar line.
|
|
75
|
+
*/
|
|
76
|
+
export const tryBuildStates = (table: string, options: BuildStatesOptions = {}): Result<TableState[]> => {
|
|
77
|
+
const lexer = grammarLexerGenerator.generate(table, () => ({}));
|
|
60
78
|
const parserGenerator = getSelfHostedParserGenerator(selfhosted as any);
|
|
61
|
-
const parser = parserGenerator.generate(lexer, {
|
|
62
|
-
reducer: ({ bag, name }) => reducers[name ?? ""](bag),
|
|
63
|
-
});
|
|
64
79
|
|
|
65
|
-
|
|
66
|
-
|
|
80
|
+
let states: ListNode<StateNode>;
|
|
81
|
+
try {
|
|
82
|
+
const parser = parserGenerator.generate(lexer, {
|
|
83
|
+
reducer: ({ bag, name }) => reducers[name ?? ""](bag),
|
|
84
|
+
});
|
|
85
|
+
states = parser.parse().result as ListNode<StateNode>;
|
|
86
|
+
} catch (err) {
|
|
87
|
+
if (err instanceof LR1ParserGraveError && err.currentToken != null)
|
|
88
|
+
return { success: false, reason: err.reason, token: err.currentToken as GrammarToken };
|
|
89
|
+
return { success: false, reason: describeThrowable(err), token: syntheticGrammarToken() };
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
const tableStates = states.getItemsReversed().map((state) => state.toTableState());
|
|
93
|
+
return validateTableStates(tableStates, options);
|
|
94
|
+
};
|
|
95
|
+
|
|
96
|
+
export const buildStates = (table: string, options: BuildStatesOptions = {}): TableState[] => {
|
|
97
|
+
const result = tryBuildStates(table, options);
|
|
98
|
+
if (result.success === false)
|
|
99
|
+
throw new Error(
|
|
100
|
+
`Encountered error "${result.reason}" while parsing the parsing table at ${result.token.line}:${result.token.column}`,
|
|
101
|
+
);
|
|
102
|
+
return result.value;
|
|
67
103
|
};
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
import { GrammarToken, syntheticGrammarToken } from "../meta/common";
|
|
2
|
+
import type { Production } from "../meta/common";
|
|
3
|
+
import type { TableAction, TableState } from "../parser";
|
|
4
|
+
import { Result } from "../utils/Result";
|
|
5
|
+
|
|
6
|
+
const SHIFT_REDUCE_RE = /^([sr])(\d+)$/;
|
|
7
|
+
const GOTO_RE = /^(\d+)$/;
|
|
8
|
+
|
|
9
|
+
const defaultRawAction = (action: TableAction): string =>
|
|
10
|
+
`${action.type === "goto" ? "" : action.type === "shift" ? "s" : "r"}${action.value}`;;
|
|
11
|
+
|
|
12
|
+
const STATE_NAME_PATTERN = /^(\[[^\]]+\]|<[^>]+>)$/;
|
|
13
|
+
|
|
14
|
+
export type TableFormatWarning = { reason: string; stateIndex: number };
|
|
15
|
+
|
|
16
|
+
export type ValidateTableStatesOptions = {
|
|
17
|
+
/** When provided, every "reduce" action is checked against the number of productions. */
|
|
18
|
+
productions?: Production[];
|
|
19
|
+
/** Name of the file the table was read from, used to make error messages actionable. */
|
|
20
|
+
source?: string;
|
|
21
|
+
onWarning?: (warning: TableFormatWarning) => void;
|
|
22
|
+
};
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Validates the shape of a parsing table: action syntax, and that every state a transition points
|
|
26
|
+
* at actually exists. Without this, a typo in a table file silently turns into nonsense behavior
|
|
27
|
+
* (`[A]=q3` becomes "reduce by production 3", `[A]=sX` becomes "shift to state NaN").
|
|
28
|
+
*/
|
|
29
|
+
export const validateTableStates = (
|
|
30
|
+
states: TableState[],
|
|
31
|
+
options: ValidateTableStatesOptions = {},
|
|
32
|
+
): Result<TableState[]> => {
|
|
33
|
+
const { productions, source } = options;
|
|
34
|
+
const suffix = source == null ? "" : ` in ${source}`;
|
|
35
|
+
const stateCount = states.length;
|
|
36
|
+
|
|
37
|
+
const fail = (reason: string, stateIndex: number): Result<TableState[]> => ({
|
|
38
|
+
success: false,
|
|
39
|
+
reason: `${reason}${suffix}`,
|
|
40
|
+
token: tokenFor(stateIndex, productions),
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
if (stateCount === 0)
|
|
44
|
+
return {
|
|
45
|
+
success: false,
|
|
46
|
+
reason: `The parsing table${suffix} does not contain any states`,
|
|
47
|
+
token: syntheticGrammarToken(),
|
|
48
|
+
};
|
|
49
|
+
|
|
50
|
+
for (let stateIndex = 0; stateIndex < stateCount; stateIndex++) {
|
|
51
|
+
const state = states[stateIndex];
|
|
52
|
+
if (state == null) return fail(`State ${stateIndex} is missing`, stateIndex);
|
|
53
|
+
|
|
54
|
+
if (stateIndex === 0 && state.actions.size === 0)
|
|
55
|
+
return fail("State 0 has no actions, so the parser cannot read the first token", stateIndex);
|
|
56
|
+
|
|
57
|
+
for (const [key, action] of state.actions.entries()) {
|
|
58
|
+
const nameMatch = STATE_NAME_PATTERN.exec(key);
|
|
59
|
+
if (nameMatch == null)
|
|
60
|
+
return fail(
|
|
61
|
+
`State ${stateIndex} has the entry "${key}", which is neither a terminal "[NAME]" nor a variable "<NAME>"`,
|
|
62
|
+
stateIndex,
|
|
63
|
+
);
|
|
64
|
+
|
|
65
|
+
const isVariable = nameMatch[1].startsWith("<");
|
|
66
|
+
if (isVariable !== (action.type === "goto"))
|
|
67
|
+
return fail(
|
|
68
|
+
`State ${stateIndex} maps the variable "${key}" to a "${action.type}" action, but variables can only be reached by a "goto" action`,
|
|
69
|
+
stateIndex,
|
|
70
|
+
);
|
|
71
|
+
|
|
72
|
+
// Prefer the text the user actually wrote: "[A]=q3" parses into a "reduce 3" action that
|
|
73
|
+
// looks valid unless you remember that "q3" is not a valid action.
|
|
74
|
+
const raw = state.rawActions.get(key) ?? defaultRawAction(action);
|
|
75
|
+
const expected = isVariable ? GOTO_RE : SHIFT_REDUCE_RE;
|
|
76
|
+
if (expected.test(raw) === false || Number.isInteger(action.value) === false)
|
|
77
|
+
return fail(
|
|
78
|
+
`State ${stateIndex} has the action "${key}=${raw}" for ${isVariable ? "a variable" : "a terminal"}, but it should be ${
|
|
79
|
+
isVariable ? "a state number" : '"sN" (shift) or "rN" (reduce)'
|
|
80
|
+
}`,
|
|
81
|
+
stateIndex,
|
|
82
|
+
);
|
|
83
|
+
|
|
84
|
+
if (isVariable || action.type === "shift") {
|
|
85
|
+
if (action.value >= stateCount)
|
|
86
|
+
return fail(
|
|
87
|
+
`State ${stateIndex} sends "${key}" to state ${action.value}, but the table only has ${stateCount} states (0 to ${stateCount - 1})`,
|
|
88
|
+
stateIndex,
|
|
89
|
+
);
|
|
90
|
+
} else if (productions != null && action.value >= productions.length)
|
|
91
|
+
return fail(
|
|
92
|
+
`State ${stateIndex} reduces "${key}" by production ${action.value}, but the grammar only has ${productions.length} productions (0 to ${productions.length - 1}). Is the table stale?`,
|
|
93
|
+
stateIndex,
|
|
94
|
+
);
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
const shiftTargets = new Set<number>();
|
|
98
|
+
for (const action of state.actions.values()) {
|
|
99
|
+
if (action.type !== "shift") continue;
|
|
100
|
+
if (shiftTargets.has(action.value))
|
|
101
|
+
options.onWarning?.({
|
|
102
|
+
reason: `State ${stateIndex} has two terminals that shift to state ${action.value}`,
|
|
103
|
+
stateIndex,
|
|
104
|
+
});
|
|
105
|
+
shiftTargets.add(action.value);
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
return { success: true, value: states };
|
|
110
|
+
};
|
|
111
|
+
|
|
112
|
+
export type ValidateTableOptions = { source?: string };
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* Cross-checks a prebuilt parsing table against the productions it is supposed to belong to.
|
|
116
|
+
* Catches the class of bug where a table is regenerated but the grammar is not (or vice versa),
|
|
117
|
+
* which otherwise shows up as an opaque crash in the middle of a parse.
|
|
118
|
+
*/
|
|
119
|
+
export const validateTable = (
|
|
120
|
+
productions: Production[],
|
|
121
|
+
states: TableState[],
|
|
122
|
+
options: ValidateTableOptions = {},
|
|
123
|
+
): Result<TableState[]> => {
|
|
124
|
+
const statesResult = validateTableStates(states, { ...options, productions });
|
|
125
|
+
if (statesResult.success === false) return statesResult;
|
|
126
|
+
|
|
127
|
+
const suffix = options.source == null ? "" : ` in ${options.source}`;
|
|
128
|
+
if (productions.length === 0)
|
|
129
|
+
return {
|
|
130
|
+
success: false,
|
|
131
|
+
reason: `The grammar${suffix} does not contain any productions, so the table cannot belong to it`,
|
|
132
|
+
token: syntheticGrammarToken(),
|
|
133
|
+
};
|
|
134
|
+
|
|
135
|
+
return { success: true, value: states };
|
|
136
|
+
};
|
|
137
|
+
|
|
138
|
+
/** Errors about a table are reported against the closest grammar token, or 1:1 when unavailable. */
|
|
139
|
+
const tokenFor = (stateIndex: number, productions: Production[] | undefined): GrammarToken => {
|
|
140
|
+
if (productions != null && productions.length > 0)
|
|
141
|
+
return productions[Math.min(stateIndex, productions.length - 1)].lhs;
|
|
142
|
+
return syntheticGrammarToken();
|
|
143
|
+
};
|
package/src/utils/Stack.ts
CHANGED
|
@@ -17,7 +17,27 @@ export class Stack<T> {
|
|
|
17
17
|
return ret;
|
|
18
18
|
}
|
|
19
19
|
|
|
20
|
+
/** Reads `depth` items below the top of the stack. Depth 0 is the top. */
|
|
21
|
+
public peekAt(depth: number): T {
|
|
22
|
+
const ret = this.items[this.items.length - 1 - depth];
|
|
23
|
+
if (ret === undefined) throw new Error(`Stack does not have an item at depth ${depth}`);
|
|
24
|
+
return ret;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
public get size() {
|
|
28
|
+
return this.items.length;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
public get isEmpty() {
|
|
32
|
+
return this.items.length === 0;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/** A copy of the stack, bottom first. Useful for error reporting inside a recovery function. */
|
|
36
|
+
public toArray(): T[] {
|
|
37
|
+
return [...this.items];
|
|
38
|
+
}
|
|
39
|
+
|
|
20
40
|
constructor(items: T[] = []) {
|
|
21
|
-
this.items = items;
|
|
41
|
+
this.items = [...items];
|
|
22
42
|
}
|
|
23
|
-
}
|
|
43
|
+
}
|
|
@@ -1,23 +1,25 @@
|
|
|
1
1
|
import { GrammarToken } from "../meta/common";
|
|
2
2
|
|
|
3
|
+
/**
|
|
4
|
+
* Renders the lines around a grammar token and points at the token itself, so that a syntax or
|
|
5
|
+
* table error tells the user exactly which part of their file to look at.
|
|
6
|
+
*/
|
|
3
7
|
export const buildErrorWindow = (file: string, token: GrammarToken) => {
|
|
4
8
|
const lines = file.split("\n");
|
|
9
|
+
const lineNumberWidth = `${token.line}`.length;
|
|
10
|
+
const gutter = (line: number) => `${`${line}`.padStart(lineNumberWidth)} | `;
|
|
5
11
|
|
|
6
|
-
const
|
|
7
|
-
|
|
8
|
-
third = "";
|
|
12
|
+
const sourceLine = lines[token.line - 1];
|
|
13
|
+
const nextLine = lines[token.line];
|
|
9
14
|
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
}
|
|
14
|
-
second += "|";
|
|
15
|
-
third += "┘";
|
|
15
|
+
const pointerColumn = Math.max(0, Math.min(token.column, sourceLine === undefined ? 0 : sourceLine.length));
|
|
16
|
+
const pointerLength = Math.max(1, Math.min(token.lexeme === "" ? 1 : token.lexeme.length, sourceLine === undefined ? 1 : sourceLine.length - pointerColumn));
|
|
17
|
+
const pointer = `${" ".repeat(pointerColumn)}${"~".repeat(pointerLength)}`;
|
|
16
18
|
|
|
17
|
-
let window = `${token.line
|
|
18
|
-
|
|
19
|
-
|
|
19
|
+
let window = `${gutter(token.line)}${sourceLine ?? "<end of file>"}\n`;
|
|
20
|
+
window += `${" ".repeat(lineNumberWidth)} | ${pointer}\n`;
|
|
21
|
+
|
|
22
|
+
if (nextLine !== undefined) window += `${gutter(token.line + 1)}${nextLine}\n`;
|
|
20
23
|
|
|
21
|
-
if (token.line < lines.length) window += `${(token.line + 1).toString().padStart(length)} | ${lines[token.line]}\n`;
|
|
22
24
|
return window;
|
|
23
25
|
};
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
import fs from "fs/promises";
|
|
2
|
+
import { Production, ValidateProductionsOptions, syntheticGrammarToken } from "../meta/common";
|
|
3
|
+
import { Result } from "./Result";
|
|
4
|
+
import { BuildStatesOptions, tryBuildStates } from "../table/selfhosted";
|
|
5
|
+
import { tryBuildProductions } from "../meta/selfhosted";
|
|
6
|
+
import type { TableState } from "../parser";
|
|
7
|
+
|
|
8
|
+
export type LoadGrammarOptions = ValidateProductionsOptions;
|
|
9
|
+
|
|
10
|
+
/** Reads a `.txt` grammar file and turns it into productions, reporting problems as a `Result`. */
|
|
11
|
+
export const loadGrammar = async (
|
|
12
|
+
grammarPath: string,
|
|
13
|
+
options: LoadGrammarOptions = {},
|
|
14
|
+
): Promise<Result<Production[]>> => {
|
|
15
|
+
let source: string;
|
|
16
|
+
try {
|
|
17
|
+
source = await fs.readFile(grammarPath, "utf8");
|
|
18
|
+
} catch (err) {
|
|
19
|
+
return {
|
|
20
|
+
success: false,
|
|
21
|
+
reason: `Could not read the grammar file: ${err instanceof Error ? err.message : String(err)}`,
|
|
22
|
+
token: syntheticGrammarToken(),
|
|
23
|
+
};
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
const result = tryBuildProductions(source, options);
|
|
27
|
+
if (result.success === false)
|
|
28
|
+
return { ...result, reason: `${result.reason} (in ${grammarPath})` };
|
|
29
|
+
return result;
|
|
30
|
+
};
|
|
31
|
+
|
|
32
|
+
export type LoadTableOptions = Omit<BuildStatesOptions, "source">;
|
|
33
|
+
|
|
34
|
+
/** Reads a parsing table file written by the CLI, reporting problems as a `Result`. */
|
|
35
|
+
export const loadTable = async (
|
|
36
|
+
tablePath: string,
|
|
37
|
+
options: LoadTableOptions = {},
|
|
38
|
+
): Promise<Result<TableState[]>> => {
|
|
39
|
+
let source: string;
|
|
40
|
+
try {
|
|
41
|
+
source = await fs.readFile(tablePath, "utf8");
|
|
42
|
+
} catch (err) {
|
|
43
|
+
return {
|
|
44
|
+
success: false,
|
|
45
|
+
reason: `Could not read the parsing table file: ${err instanceof Error ? err.message : String(err)}`,
|
|
46
|
+
token: syntheticGrammarToken(),
|
|
47
|
+
};
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
const result = tryBuildStates(source, { ...options, source: tablePath });
|
|
51
|
+
if (result.success === false)
|
|
52
|
+
return { ...result, reason: `${result.reason} (in ${tablePath})` };
|
|
53
|
+
return result;
|
|
54
|
+
};
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
import { Production } from "../meta/common";
|
|
2
|
+
|
|
3
|
+
export const KLEENE_REDUCER_NAME = "autogenerated-kleene";
|
|
4
|
+
|
|
5
|
+
/** The names of every production that has a `: name` in the grammar file. */
|
|
6
|
+
export const namedProductions = (productions: Production[]): string[] => {
|
|
7
|
+
const names = new Set<string>();
|
|
8
|
+
for (const production of productions)
|
|
9
|
+
if (production.name != null) names.add(production.name);
|
|
10
|
+
return [...names].sort();
|
|
11
|
+
};
|
|
12
|
+
|
|
13
|
+
/** Every named production in the grammar that `reducers` does not have a reducer for. */
|
|
14
|
+
export const missingReducerNames = (
|
|
15
|
+
productions: Production[],
|
|
16
|
+
reducers: Record<string, unknown>,
|
|
17
|
+
): string[] => namedProductions(productions).filter((name) => reducers[name] == null);
|
|
18
|
+
|
|
19
|
+
const describeKleene = (productions: Production[]) => {
|
|
20
|
+
const needed = productions.filter(
|
|
21
|
+
(production) =>
|
|
22
|
+
production.name === KLEENE_REDUCER_NAME ||
|
|
23
|
+
production.identifier.startsWith("<autogen-"),
|
|
24
|
+
);
|
|
25
|
+
if (needed.length === 0) return "";
|
|
26
|
+
|
|
27
|
+
return `\nNote: this grammar uses "*" or "+", so it has ${needed.length} autogenerated list production(s) named "${KLEENE_REDUCER_NAME}". That reducer has to flatten the list it is handed (see README, "Repetition with * and +").`;
|
|
28
|
+
};
|
|
29
|
+
|
|
30
|
+
export const missingReducerMessage = (
|
|
31
|
+
productions: Production[],
|
|
32
|
+
reducers: Record<string, unknown>,
|
|
33
|
+
): string => {
|
|
34
|
+
const missing = missingReducerNames(productions, reducers);
|
|
35
|
+
if (missing.length === 0) return "";
|
|
36
|
+
|
|
37
|
+
const has = namedProductions(productions).filter((name) => reducers[name] != null);
|
|
38
|
+
const kleene = missing.includes(KLEENE_REDUCER_NAME) ? describeKleene(productions) : "";
|
|
39
|
+
|
|
40
|
+
return (
|
|
41
|
+
`No reducer for ${missing.length === 1 ? "production" : "productions"} ${missing
|
|
42
|
+
.map((name) => `"${name}"`)
|
|
43
|
+
.join(", ")}.` +
|
|
44
|
+
`\nReducers are keyed by the name given in the grammar, e.g. "<PROGRAM: program>" is reduced by reducers.program.` +
|
|
45
|
+
(has.length === 0 ? "\nYour reducer map is empty." : `\nYou have reducers for: ${has.join(", ")}.`) +
|
|
46
|
+
kleene
|
|
47
|
+
);
|
|
48
|
+
};
|
|
49
|
+
|
|
50
|
+
export type ReducerMap<Node> = Record<string, (newInput: any, oldInput: any) => Node>;
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Wraps a map of reducers into the single reducer that `Sparse.generate` expects, keyed by the
|
|
54
|
+
* `: name` of each production in the grammar.
|
|
55
|
+
*
|
|
56
|
+
* A missing entry is reported by production name instead of surfacing as
|
|
57
|
+
* `reducers[name] is not a function` from deep inside the parse.
|
|
58
|
+
*/
|
|
59
|
+
export const defineReducers = <Reducers extends Record<string, (...args: never[]) => unknown>>(reducers: Reducers) =>
|
|
60
|
+
function sparseReducer(newInput: any, oldInput: any): any {
|
|
61
|
+
const name: string | null | undefined = newInput?.name;
|
|
62
|
+
const lookup = reducers as unknown as Record<string, (a: any, b: any) => any>;
|
|
63
|
+
const reducer = name == null ? undefined : lookup[name];
|
|
64
|
+
if (reducer == null)
|
|
65
|
+
throw new Error(
|
|
66
|
+
`No reducer for production "${name ?? ""}" (production ${oldInput?.index}). Reducers are keyed by the name given in the grammar.`,
|
|
67
|
+
);
|
|
68
|
+
return reducer(newInput, oldInput);
|
|
69
|
+
};
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Throws a descriptive error if `reducers` is missing any production the grammar names.
|
|
73
|
+
* Useful as a one-time check right after loading the grammar.
|
|
74
|
+
*/
|
|
75
|
+
export const assertReducersCoverGrammar = (productions: Production[], reducers: ReducerMap<unknown>) => {
|
|
76
|
+
const message = missingReducerMessage(productions, reducers);
|
|
77
|
+
if (message !== "") throw new Error(message);
|
|
78
|
+
};
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* The stringifier every example used to write by hand: `(type) => TokenType[type]`.
|
|
82
|
+
* Numeric TypeScript enums store their names in reverse, so a plain lookup is all it takes.
|
|
83
|
+
*
|
|
84
|
+
* ```ts
|
|
85
|
+
* const toStringifiedTokenType = enumToString<TokenType>(TokenType);
|
|
86
|
+
* ```
|
|
87
|
+
*/
|
|
88
|
+
export const enumToString =
|
|
89
|
+
<TokenType = unknown>(tokenTypes: object) =>
|
|
90
|
+
(tokenType: TokenType): string => {
|
|
91
|
+
const name = (tokenTypes as Record<string, unknown>)[tokenType as unknown as string];
|
|
92
|
+
return `${name ?? tokenType}`;
|
|
93
|
+
};
|
package/test/cli.test.ts
ADDED
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
import { describe, expect, it, beforeAll } from "vitest";
|
|
2
|
+
import { execFile } from "child_process";
|
|
3
|
+
import { promisify } from "util";
|
|
4
|
+
import fs from "fs/promises";
|
|
5
|
+
import os from "os";
|
|
6
|
+
import path from "path";
|
|
7
|
+
import { examplePath, root } from "./helpers";
|
|
8
|
+
|
|
9
|
+
const exec = promisify(execFile);
|
|
10
|
+
|
|
11
|
+
let cli = path.join(root, "dist/cli.js");
|
|
12
|
+
let tmp: string;
|
|
13
|
+
|
|
14
|
+
type Run = { code: number; stdout: string; stderr: string };
|
|
15
|
+
|
|
16
|
+
const run = async (...args: string[]): Promise<Run> => {
|
|
17
|
+
try {
|
|
18
|
+
const { stdout, stderr } = await exec("node", [cli, ...args], { cwd: root });
|
|
19
|
+
return { code: 0, stdout, stderr };
|
|
20
|
+
} catch (err) {
|
|
21
|
+
const failure = err as { code?: number; stdout: string; stderr: string };
|
|
22
|
+
return { code: failure.code ?? 1, stdout: failure.stdout ?? "", stderr: failure.stderr ?? "" };
|
|
23
|
+
}
|
|
24
|
+
};
|
|
25
|
+
|
|
26
|
+
beforeAll(async () => {
|
|
27
|
+
await exec("node", ["node_modules/typescript/bin/tsc", "-p", "tsconfig.build.json"], { cwd: root });
|
|
28
|
+
tmp = await fs.mkdtemp(path.join(os.tmpdir(), "sparse-cli-"));
|
|
29
|
+
}, 180_000);
|
|
30
|
+
|
|
31
|
+
describe("the sparse CLI", () => {
|
|
32
|
+
it("prints usage for --help and exits successfully", async () => {
|
|
33
|
+
const result = await run("--help");
|
|
34
|
+
expect(result.code).toBe(0);
|
|
35
|
+
expect(result.stdout).toContain("Usage: npx @scinorandex/sparse --input=<grammar> --output=<table>");
|
|
36
|
+
expect(result.stdout).toContain("--lalr");
|
|
37
|
+
expect(result.stdout).toContain("--check");
|
|
38
|
+
});
|
|
39
|
+
|
|
40
|
+
it("fails with usage when --input is missing", async () => {
|
|
41
|
+
const result = await run("--output=" + path.join(tmp, "table.txt"));
|
|
42
|
+
expect(result.code).toBe(1);
|
|
43
|
+
expect(result.stderr).toContain("Missing --input");
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
it("fails when the grammar does not exist", async () => {
|
|
47
|
+
const result = await run(`--input=${path.join(tmp, "nope.txt")}`, `--output=${path.join(tmp, "table.txt")}`);
|
|
48
|
+
expect(result.code).toBe(1);
|
|
49
|
+
expect(result.stderr).toContain("does not exist");
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
it("creates the output directory", async () => {
|
|
53
|
+
const output = path.join(tmp, "nested", "deeper", "table.txt");
|
|
54
|
+
const result = await run(`--input=${examplePath("math/grammar.txt")}`, `--output=${output}`);
|
|
55
|
+
|
|
56
|
+
expect(result.code).toBe(0);
|
|
57
|
+
expect(result.stdout).toContain("22 LR(1) states");
|
|
58
|
+
expect(await fs.readFile(output, "utf8")).toBe(await fs.readFile(examplePath("math/table.txt"), "utf8"));
|
|
59
|
+
});
|
|
60
|
+
|
|
61
|
+
it("reproduces the committed math table", async () => {
|
|
62
|
+
const output = path.join(tmp, "math.txt");
|
|
63
|
+
await run(`--input=${examplePath("math/grammar.txt")}`, `--output=${output}`);
|
|
64
|
+
|
|
65
|
+
expect((await fs.readFile(output, "utf8")).trim()).toBe(
|
|
66
|
+
(await fs.readFile(examplePath("math/table.txt"), "utf8")).trim(),
|
|
67
|
+
);
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
it("prints the table to stdout", async () => {
|
|
71
|
+
const result = await run(`--input=${examplePath("math/grammar.txt")}`, "--stdout");
|
|
72
|
+
expect(result.code).toBe(0);
|
|
73
|
+
expect(result.stdout.trim()).toBe((await fs.readFile(examplePath("math/table.txt"), "utf8")).trim());
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
it("shows a window into the grammar when the grammar is broken", async () => {
|
|
77
|
+
const grammarPath = path.join(tmp, "broken.txt");
|
|
78
|
+
await fs.writeFile(grammarPath, "<S>: <PROGRAM>;\n<PROGRAM: program>: [NUMBER] (;\n");
|
|
79
|
+
|
|
80
|
+
const result = await run(`--input=${grammarPath}`, `--output=${path.join(tmp, "broken-table.txt")}`);
|
|
81
|
+
|
|
82
|
+
expect(result.code).toBe(1);
|
|
83
|
+
expect(result.stderr).toContain("Invalid syntax at 2:30");
|
|
84
|
+
expect(result.stderr).toContain("<PROGRAM: program>: [NUMBER] (");
|
|
85
|
+
expect(result.stderr).toContain("~");
|
|
86
|
+
expect(result.stderr).not.toContain("at Object.");
|
|
87
|
+
});
|
|
88
|
+
|
|
89
|
+
it("reports a grammar that references an undefined variable", async () => {
|
|
90
|
+
const grammarPath = path.join(tmp, "dangling.txt");
|
|
91
|
+
await fs.writeFile(grammarPath, "<S>: <PROGRAM>;\n<PROGRAM: program>: [NUMBER] <NOPE> [EOF];\n");
|
|
92
|
+
|
|
93
|
+
const result = await run(`--input=${grammarPath}`, `--output=${path.join(tmp, "dangling-table.txt")}`);
|
|
94
|
+
|
|
95
|
+
expect(result.code).toBe(1);
|
|
96
|
+
expect(result.stderr).toContain(`Variable "<NOPE>"`);
|
|
97
|
+
});
|
|
98
|
+
|
|
99
|
+
it("warns about a suspicious grammar rule but still writes the table", async () => {
|
|
100
|
+
const grammarPath = path.join(tmp, "duplicate.txt");
|
|
101
|
+
const output = path.join(tmp, "duplicate-table.txt");
|
|
102
|
+
await fs.writeFile(grammarPath, "<S>: <PROGRAM>;\n<PROGRAM: program>: [NUMBER: n] [EOF: n];\n");
|
|
103
|
+
|
|
104
|
+
const result = await run(`--input=${grammarPath}`, `--output=${output}`);
|
|
105
|
+
|
|
106
|
+
expect(result.code).toBe(0);
|
|
107
|
+
expect(result.stderr).toContain("names more than one symbol");
|
|
108
|
+
|
|
109
|
+
const quiet = await run(`--input=${grammarPath}`, `--output=${output}`, "--quiet");
|
|
110
|
+
expect(quiet.stderr).toBe("");
|
|
111
|
+
});
|
|
112
|
+
|
|
113
|
+
it("accepts a table that is up to date with --check", async () => {
|
|
114
|
+
const output = path.join(tmp, "check.txt");
|
|
115
|
+
await run(`--input=${examplePath("math/grammar.txt")}`, `--output=${output}`);
|
|
116
|
+
|
|
117
|
+
const result = await run(`--input=${examplePath("math/grammar.txt")}`, `--output=${output}`, "--check");
|
|
118
|
+
expect(result.code).toBe(0);
|
|
119
|
+
expect(result.stdout).toContain("is up to date");
|
|
120
|
+
});
|
|
121
|
+
|
|
122
|
+
it("reports a stale table with --check", async () => {
|
|
123
|
+
const grammar = "<S>: <PROGRAM>;\n<PROGRAM: program>: [NUMBER] [EOF];\n";
|
|
124
|
+
const grammarPath = path.join(tmp, "check-grammar.txt");
|
|
125
|
+
const output = path.join(tmp, "check-stale.txt");
|
|
126
|
+
await fs.writeFile(grammarPath, grammar);
|
|
127
|
+
await fs.writeFile(output, "[EOF]=r0\n");
|
|
128
|
+
|
|
129
|
+
const result = await run(`--input=${grammarPath}`, `--output=${output}`, "--check");
|
|
130
|
+
expect(result.code).toBe(1);
|
|
131
|
+
expect(result.stderr).toContain("is out of date");
|
|
132
|
+
});
|
|
133
|
+
|
|
134
|
+
it("reports a table file that is not a table", async () => {
|
|
135
|
+
const grammarPath = path.join(tmp, "check-grammar2.txt");
|
|
136
|
+
const output = path.join(tmp, "check-garbage.txt");
|
|
137
|
+
await fs.writeFile(grammarPath, "<S>: <PROGRAM>;\n<PROGRAM: program>: [EOF];\n");
|
|
138
|
+
await fs.writeFile(output, "hello world\n");
|
|
139
|
+
|
|
140
|
+
const result = await run(`--input=${grammarPath}`, `--output=${output}`, "--check");
|
|
141
|
+
expect(result.code).toBe(1);
|
|
142
|
+
expect(result.stderr).toContain("not a valid parsing table");
|
|
143
|
+
});
|
|
144
|
+
});
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
import { describe, expect, it } from "vitest";
|
|
2
|
+
import fs from "fs/promises";
|
|
3
|
+
import { buildStates, dehydateProduction, generateStates, tryBuildProductions } from "../src/index";
|
|
4
|
+
import { grammarLexerGenerator } from "../src/meta/selfhosted";
|
|
5
|
+
import { selfhosted as metaSelfHosted } from "../src/meta/states";
|
|
6
|
+
import { selfhosted as tableSelfHosted } from "../src/table/states";
|
|
7
|
+
import { examplePath, readFile } from "./helpers";
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Mirrors example/selfhosted/example.ts: the LR(1) states that Sparse uses to read its own grammar
|
|
11
|
+
* and table file formats are generated from the grammar files in example/selfhosted, and the copies
|
|
12
|
+
* checked into src/ have to be identical. If this test fails, the bootstrap is broken: either the
|
|
13
|
+
* grammar files changed without regenerating the states, or the states changed without updating
|
|
14
|
+
* the grammar files.
|
|
15
|
+
*/
|
|
16
|
+
const regen = async (grammarName: string) => {
|
|
17
|
+
const grammar = await readFile(grammarName);
|
|
18
|
+
const lexer = grammarLexerGenerator.generate(grammar, () => ({}));
|
|
19
|
+
|
|
20
|
+
const productionsResult = tryBuildProductions(lexer);
|
|
21
|
+
if (productionsResult.success === false) throw new Error(productionsResult.reason);
|
|
22
|
+
const productions = productionsResult.value;
|
|
23
|
+
|
|
24
|
+
const statesResult = generateStates(productions);
|
|
25
|
+
if (statesResult.success === false) throw new Error(statesResult.reason);
|
|
26
|
+
|
|
27
|
+
return {
|
|
28
|
+
states: statesResult.value.toJSObject(),
|
|
29
|
+
productions: productions.map(dehydateProduction),
|
|
30
|
+
};
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
describe("self hosted code generation", () => {
|
|
34
|
+
it("regenerates the states checked into src/meta/states.ts byte for byte", async () => {
|
|
35
|
+
const regenerated = await regen("selfhosted/01-meta-grammar.txt");
|
|
36
|
+
|
|
37
|
+
expect(regenerated).toEqual({ states: metaSelfHosted.states, productions: metaSelfHosted.productions });
|
|
38
|
+
});
|
|
39
|
+
|
|
40
|
+
it("regenerates the states checked into src/table/states.ts byte for byte", async () => {
|
|
41
|
+
const regenerated = await regen("selfhosted/02-table-grammar.txt");
|
|
42
|
+
|
|
43
|
+
expect(regenerated).toEqual({ states: tableSelfHosted.states, productions: tableSelfHosted.productions });
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
it("keeps the committed codegen files in example/selfhosted in sync", async () => {
|
|
47
|
+
for (const [grammarName, codegenName] of [
|
|
48
|
+
["selfhosted/01-meta-grammar.txt", "selfhosted/codegen-meta.ts"],
|
|
49
|
+
["selfhosted/02-table-grammar.txt", "selfhosted/codegen-table.ts"],
|
|
50
|
+
]) {
|
|
51
|
+
const regenerated = await regen(grammarName as string);
|
|
52
|
+
const source = `export const states = ${JSON.stringify(regenerated.states, null, 2)};
|
|
53
|
+
export const productions = ${JSON.stringify(regenerated.productions, null, 2)};
|
|
54
|
+
export const selfhosted = { states, productions };`;
|
|
55
|
+
|
|
56
|
+
expect(await fs.readFile(examplePath(codegenName as string), "utf8")).toBe(source);
|
|
57
|
+
}
|
|
58
|
+
});
|
|
59
|
+
|
|
60
|
+
it("produces tables the library can read back", async () => {
|
|
61
|
+
const regenerated = await regen("selfhosted/01-meta-grammar.txt");
|
|
62
|
+
|
|
63
|
+
const fromJson = buildStates(
|
|
64
|
+
regenerated.states
|
|
65
|
+
.map((state) =>
|
|
66
|
+
Object.entries(state)
|
|
67
|
+
.map(([key, action]) => `${key}=${action.type === "goto" ? "" : action.type === "shift" ? "s" : "r"}${action.value}`)
|
|
68
|
+
.join(", "),
|
|
69
|
+
)
|
|
70
|
+
.join("\n"),
|
|
71
|
+
);
|
|
72
|
+
|
|
73
|
+
expect(fromJson.length).toBe(regenerated.states.length);
|
|
74
|
+
});
|
|
75
|
+
});
|