@scinorandex/sparse 0.1.1 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. package/.github/workflows/ci.yml +41 -0
  2. package/AGENTS.md +31 -11
  3. package/README.md +283 -29
  4. package/dist/cli.js +113 -24
  5. package/dist/cli.js.map +1 -1
  6. package/dist/generator.d.ts +3 -1
  7. package/dist/generator.js +29 -5
  8. package/dist/generator.js.map +1 -1
  9. package/dist/index.d.ts +16 -4
  10. package/dist/index.js +34 -1
  11. package/dist/index.js.map +1 -1
  12. package/dist/meta/common.d.ts +17 -0
  13. package/dist/meta/common.js +66 -1
  14. package/dist/meta/common.js.map +1 -1
  15. package/dist/meta/selfhosted.d.ts +4 -5
  16. package/dist/meta/selfhosted.js +49 -36
  17. package/dist/meta/selfhosted.js.map +1 -1
  18. package/dist/parser.d.ts +56 -35
  19. package/dist/parser.js +146 -18
  20. package/dist/parser.js.map +1 -1
  21. package/dist/table/selfhosted.d.ts +10 -1
  22. package/dist/table/selfhosted.js +32 -12
  23. package/dist/table/selfhosted.js.map +1 -1
  24. package/dist/table/validate.d.ts +17 -0
  25. package/dist/table/validate.js +84 -0
  26. package/dist/table/validate.js.map +1 -0
  27. package/dist/utils/Stack.d.ts +4 -0
  28. package/dist/utils/Stack.js +16 -1
  29. package/dist/utils/Stack.js.map +1 -1
  30. package/dist/utils/errorWindowBuilder.js +11 -11
  31. package/dist/utils/errorWindowBuilder.js.map +1 -1
  32. package/dist/utils/loadFiles.d.ts +8 -0
  33. package/dist/utils/loadFiles.js +47 -0
  34. package/dist/utils/loadFiles.js.map +1 -0
  35. package/dist/utils/reducers.d.ts +9 -0
  36. package/dist/utils/reducers.js +56 -0
  37. package/dist/utils/reducers.js.map +1 -0
  38. package/example/LoLang/example.ts +16 -15
  39. package/example/kleene-test/example.ts +35 -17
  40. package/example/math/example.ts +12 -7
  41. package/example/selfhosted/example.ts +30 -24
  42. package/opencode.json +1 -1
  43. package/package.json +2 -2
  44. package/src/cli.ts +132 -23
  45. package/src/generator.ts +49 -8
  46. package/src/index.ts +55 -3
  47. package/src/meta/common.ts +116 -0
  48. package/src/meta/selfhosted.ts +42 -11
  49. package/src/parser.ts +259 -49
  50. package/src/table/selfhosted.ts +48 -12
  51. package/src/table/validate.ts +143 -0
  52. package/src/utils/Stack.ts +22 -2
  53. package/src/utils/errorWindowBuilder.ts +15 -13
  54. package/src/utils/loadFiles.ts +54 -0
  55. package/src/utils/reducers.ts +93 -0
  56. package/test/cli.test.ts +144 -0
  57. package/test/codegen.test.ts +75 -0
  58. package/test/grammar.test.ts +243 -0
  59. package/test/helpers.ts +59 -0
  60. package/test/lalr.test.ts +114 -0
  61. package/test/parser.test.ts +392 -0
  62. package/test/table.test.ts +144 -0
  63. package/tsconfig.json +1 -1
  64. package/example/LoLang/table2.txt +0 -497
  65. package/example/complicated/grammar.txt +0 -37
@@ -1,4 +1,5 @@
1
1
  import { ColumnAndRow, Token } from "@scinorandex/slex";
2
+ import { Result } from "../utils/Result";
2
3
 
3
4
  export type Production = {
4
5
  lhs: GrammarToken;
@@ -8,6 +9,8 @@ export type Production = {
8
9
  rhs: { type: "terminal" | "variable"; token: GrammarToken; identifier: string; name: string | null }[];
9
10
  };
10
11
 
12
+ export type ProductionRhsItem = Production["rhs"][number];
13
+
11
14
  const dehydateGrammarToken = ({ column, lexeme, line, type }: GrammarToken) => {
12
15
  return { column, lexeme, line, type };
13
16
  };
@@ -50,6 +53,119 @@ export const hydrateProduction = (production: ReturnType<typeof dehydateProducti
50
53
  };
51
54
 
52
55
  export type GrammarTokenMetadata = {};
56
+
57
+ /** A readable one-liner for an unknown thrown value, used to turn it into a `Result` failure. */
58
+ export const describeThrowable = (err: unknown) =>
59
+ err instanceof Error ? err.message : `Unexpected error: ${String(err)}`;
60
+
61
+ /** A stand-in token used to report errors that do not belong to a specific place in a grammar. */
62
+ export const syntheticGrammarToken = () =>
63
+ new Token(GrammarTokenType.EOF, "", new ColumnAndRow(0, 0), {}) as GrammarToken;
64
+
65
+ export type ProductionWarningKind = "duplicate-rhs-name" | "unreachable-production";
66
+
67
+ export type ProductionWarning = {
68
+ kind: ProductionWarningKind;
69
+ reason: string;
70
+ token: GrammarToken;
71
+ };
72
+
73
+ export type ValidateProductionsOptions = {
74
+ /** Called once per warning. Warnings never stop the parser generator, they only flag suspicious grammars. */
75
+ onWarning?: (warning: ProductionWarning) => void;
76
+ };
77
+
78
+ const at = (token: GrammarToken) => `${token.line}:${token.column}`;
79
+
80
+ /**
81
+ * Checks a set of unrolled productions for the mistakes that would otherwise surface as an
82
+ * inscrutable `TypeError` deep inside state generation or parsing:
83
+ *
84
+ * - a variable used on the right hand side that no production defines
85
+ * - a production with an empty right hand side (Sparse has no support for empty productions)
86
+ * - the start symbol being used on the right hand side (it must only appear on the left hand side)
87
+ *
88
+ * Suspicious-but-legal grammars (duplicate symbol names, unreachable productions) are reported
89
+ * through `onWarning` instead of failing.
90
+ */
91
+ export const validateProductions = (
92
+ productions: Production[],
93
+ options: ValidateProductionsOptions = {},
94
+ ): Result<Production[]> => {
95
+ const warn = (kind: ProductionWarningKind, reason: string, token: GrammarToken) => {
96
+ options.onWarning?.({ kind, reason, token });
97
+ };
98
+
99
+ if (productions.length === 0)
100
+ return {
101
+ success: false,
102
+ reason: "The grammar does not contain any productions. Is the grammar file empty?",
103
+ token: syntheticGrammarToken(),
104
+ };
105
+
106
+ const lhsIdentifiers = new Set(productions.map((production) => production.identifier));
107
+ const referencedIdentifiers = new Set<string>();
108
+
109
+ for (const production of productions) {
110
+ if (production.rhs.length === 0)
111
+ return {
112
+ success: false,
113
+ reason: `Production "${production.identifier}" (line ${at(production.lhs)}) has an empty right hand side. Sparse does not support empty productions, so a lone "(...)? " group such as "<X: x>: ([A])?;" is not a valid production.`,
114
+ token: production.lhs,
115
+ };
116
+
117
+ const seenNames = new Set<string>();
118
+ for (const rhsItem of production.rhs) {
119
+ if (rhsItem.type === "variable") {
120
+ referencedIdentifiers.add(rhsItem.identifier);
121
+ if (lhsIdentifiers.has(rhsItem.identifier) === false)
122
+ return {
123
+ success: false,
124
+ reason: `Variable "${rhsItem.identifier}" is used on the right hand side of production "${production.identifier}" (line ${at(production.lhs)}) but no production defines it as its left hand side.`,
125
+ token: rhsItem.token,
126
+ };
127
+ }
128
+
129
+ if (rhsItem.name != null) {
130
+ if (seenNames.has(rhsItem.name))
131
+ warn(
132
+ "duplicate-rhs-name",
133
+ `Production "${production.identifier}" (line ${at(production.lhs)}) names more than one symbol "${rhsItem.name}". Only the last one ends up on the reducer's "bag", use the positional input to reach the others.`,
134
+ rhsItem.token,
135
+ );
136
+ seenNames.add(rhsItem.name);
137
+ }
138
+ }
139
+ }
140
+
141
+ // The first production is the accept production: its left hand side is the start symbol, which
142
+ // by definition cannot be referenced from the right hand side of anything.
143
+ const startProduction = productions[0];
144
+ if (referencedIdentifiers.has(startProduction.identifier))
145
+ return {
146
+ success: false,
147
+ reason: `"${startProduction.identifier}" is the start symbol of the grammar, so it cannot be used on the right hand side of a production. It is used at line ${at(startProduction.lhs)}.`,
148
+ token: startProduction.lhs,
149
+ };
150
+
151
+ for (let i = 1; i < productions.length; i++) {
152
+ const production = productions[i];
153
+ if (referencedIdentifiers.has(production.identifier)) continue;
154
+ warn(
155
+ "unreachable-production",
156
+ `Production "${production.identifier}" (line ${at(production.lhs)}) is never used on the right hand side of another production, so the parser can never reach it.`,
157
+ production.lhs,
158
+ );
159
+ }
160
+
161
+ return { success: true, value: productions };
162
+ };
163
+
164
+ /** True when the identifier is a variable (`<FOO>`) rather than a terminal (`[FOO]`). */
165
+ export const isVariableIdentifier = (identifier: string) => identifier.startsWith("<") && identifier.endsWith(">");
166
+
167
+ export const terminalIdentifier = (name: string) => `[${name}]`;
168
+ export const variableIdentifier = (name: string) => `<${name}>`;
53
169
  export enum GrammarTokenType {
54
170
  IDENTIFIER,
55
171
  L_ANGLE,
@@ -1,8 +1,17 @@
1
1
  import { RegexEngine, Slex } from "@scinorandex/slex";
2
+ import { describeThrowable, syntheticGrammarToken } from "../meta/common";
2
3
  import { Result, Sparse } from "../index";
3
4
  import { selfhosted } from "./states";
4
- import { TableState } from "../parser";
5
- import { GrammarToken, GrammarTokenMetadata, GrammarTokenType, hydrateProduction, Production } from "./common";
5
+ import { LR1ParserGraveError, TableState } from "../parser";
6
+ import {
7
+ GrammarToken,
8
+ GrammarTokenMetadata,
9
+ GrammarTokenType,
10
+ hydrateProduction,
11
+ Production,
12
+ ValidateProductionsOptions,
13
+ validateProductions,
14
+ } from "./common";
6
15
 
7
16
  export const grammarLexerGenerator = new Slex<GrammarTokenType, GrammarTokenMetadata>({
8
17
  EOF_TYPE: GrammarTokenType.EOF,
@@ -21,9 +30,12 @@ grammarLexerGenerator.addRule(
21
30
  grammarLexerGenerator.addRule("letter", "${lowercase} | ${uppercase}");
22
31
  grammarLexerGenerator.addRule("digit", "0 | 1 | 2 | 3 | 4 | 5 | 6 | 7 | 8 | 9");
23
32
  grammarLexerGenerator.addRule("alphanumeric", "${letter} | ${digit}");
33
+ // Dashes are allowed inside identifiers (but not first) so that the autogenerated productions of a
34
+ // "*" or "+" group, which are named "autogen-0", "autogen-1", ... survive a round trip through a
35
+ // table file.
24
36
  grammarLexerGenerator.addRule(
25
37
  "identifier",
26
- "(${letter} | $_) (${letter} | ${digit} | $_)*",
38
+ "(${letter} | $_) (${letter} | ${digit} | $_ | $-)*",
27
39
  GrammarTokenType.IDENTIFIER,
28
40
  );
29
41
  grammarLexerGenerator.addRule("colon", "$:", GrammarTokenType.COLON);
@@ -312,8 +324,15 @@ const reducers: { [key: string]: Reducer } = {
312
324
  },
313
325
  };
314
326
 
315
- export const tryBuildProductions = (lexer: LexerInterface | string): Result<Production[]> => {
316
- if (typeof lexer === "string") lexer = grammarLexerGenerator.generate(lexer, () => ({}));
327
+ export const tryBuildProductions = (
328
+ lexer: LexerInterface | string,
329
+ options: ValidateProductionsOptions = {},
330
+ ): Result<Production[]> => {
331
+ try {
332
+ if (typeof lexer === "string") lexer = grammarLexerGenerator.generate(lexer, () => ({}));
333
+ } catch (err) {
334
+ return { success: false, reason: describeThrowable(err), token: syntheticGrammarToken() };
335
+ }
317
336
 
318
337
  const parserGenerator = getSelfHostedParserGenerator();
319
338
  const parser = parserGenerator.generate(lexer as RegexEngine<GrammarTokenType, GrammarTokenMetadata>, {
@@ -322,19 +341,31 @@ export const tryBuildProductions = (lexer: LexerInterface | string): Result<Prod
322
341
  },
323
342
  });
324
343
 
325
- const parsingResult = parser.parse().result as ProgramNode;
326
- const productions = parsingResult.getProductions();
327
- return { success: true, value: productions };
344
+ let programNode: ProgramNode;
345
+ try {
346
+ programNode = parser.parse().result as ProgramNode;
347
+ } catch (err) {
348
+ // The grammar parser has no error recovery, so a malformed grammar surfaces as a
349
+ // LR1ParserGraveError. Translate it into the Result failure every caller already handles.
350
+ if (err instanceof LR1ParserGraveError && err.currentToken != null)
351
+ return { success: false, reason: err.reason, token: err.currentToken as GrammarToken };
352
+ return { success: false, reason: describeThrowable(err), token: syntheticGrammarToken() };
353
+ }
354
+
355
+ return validateProductions(programNode.getProductions(), options);
328
356
  };
329
357
 
330
- interface LexerInterface {
358
+ export interface LexerInterface {
331
359
  peekNextToken(): GrammarToken;
332
360
  getNextToken(): GrammarToken;
333
361
  hasNextToken(): boolean;
334
362
  }
335
363
 
336
- export const buildProductions = (lexer: LexerInterface | string): Production[] => {
337
- const result = tryBuildProductions(lexer);
364
+ export const buildProductions = (
365
+ lexer: LexerInterface | string,
366
+ options: ValidateProductionsOptions = {},
367
+ ): Production[] => {
368
+ const result = tryBuildProductions(lexer, options);
338
369
  if (result.success === false)
339
370
  throw new Error(
340
371
  `Encountered error "${result.reason}" while parsing grammar at ${result.token.line}:${result.token.column}`,
package/src/parser.ts CHANGED
@@ -2,15 +2,27 @@ import { Slex, Token } from "@scinorandex/slex";
2
2
  import { generateStates } from "./generator";
3
3
  import { Stack } from "./utils/Stack";
4
4
  import { Result } from "./utils/Result";
5
- import { Production } from "./meta/common";
5
+ import { GrammarToken, Production, ProductionWarning } from "./meta/common";
6
+ import { loadGrammar, loadTable } from "./utils/loadFiles";
7
+ import { validateTable } from "./table/validate";
6
8
 
7
9
  export type TableAction = { type: "shift" | "reduce" | "goto"; value: number };
8
10
 
9
11
  export class TableState {
10
12
  public readonly actions: Map<string, TableAction>;
11
13
 
12
- constructor(actions = new Map<string, TableAction>()) {
14
+ /**
15
+ * The actions exactly as they were written in a table file (`"s12"`, `"r3"`, `"7"`), kept so that
16
+ * malformed entries can be reported with the text the user actually typed.
17
+ */
18
+ public readonly rawActions: Map<string, string>;
19
+
20
+ constructor(
21
+ actions = new Map<string, TableAction>(),
22
+ rawActions = new Map<string, string>(),
23
+ ) {
13
24
  this.actions = actions;
25
+ this.rawActions = rawActions;
14
26
  }
15
27
 
16
28
  public getTerminalAction(terminal: string): TableAction | undefined {
@@ -21,16 +33,41 @@ export class TableState {
21
33
  return this.actions.get(variable);
22
34
  }
23
35
 
36
+ /** Every terminal this state can consume, without the surrounding brackets, sorted. */
37
+ public getTerminalKeys(): string[] {
38
+ return this.getKeys("[", "]");
39
+ }
40
+
41
+ /** Every variable this state can reduce past, without the surrounding angle brackets, sorted. */
42
+ public getVariableKeys(): string[] {
43
+ return this.getKeys("<", ">");
44
+ }
45
+
46
+ private getKeys(prefix: string, suffix: string): string[] {
47
+ const keys: string[] = [];
48
+ for (const key of this.actions.keys())
49
+ if (key.startsWith(prefix) && key.endsWith(suffix))
50
+ keys.push(key.substring(prefix.length, key.length - suffix.length));
51
+ return keys.sort();
52
+ }
53
+
24
54
  toJSObject() {
25
55
  return Object.fromEntries(this.actions);
26
56
  }
27
57
 
28
58
  static fromJSObject(json: ReturnType<TableState["toJSObject"]>) {
29
- return new TableState(new Map(Object.entries(json)));
59
+ const actions = new Map(Object.entries(json));
60
+ const rawActions = new Map(
61
+ [...actions].map(([key, action]) => [
62
+ key,
63
+ action.type === "goto" ? `${action.value}` : `${action.type === "shift" ? "s" : "r"}${action.value}`,
64
+ ]),
65
+ );
66
+ return new TableState(actions, rawActions);
30
67
  }
31
68
  }
32
69
 
33
- type ReducerType<TokenType, Metadata, Node> = (
70
+ export type ReducerType<TokenType, Metadata, Node> = (
34
71
  newInput: { bag: any; name: string | null },
35
72
  oldInput: { input: LR1StackSymbol<TokenType, Metadata, Node>[]; index: number },
36
73
  ) => Node;
@@ -39,21 +76,50 @@ export type LR1StackSymbol<TokenType, Metadata, Node> =
39
76
  | { type: "token"; token: Token<TokenType, Metadata> }
40
77
  | { type: "node"; node: Node };
41
78
 
79
+ export type SparseOptions<TokenType> = {
80
+ productions: Production[];
81
+ states: TableState[];
82
+ toStringifiedTokenType: (tokenType: TokenType) => string;
83
+ /**
84
+ * Cross-check the productions against the parsing table right away, so a stale table is reported
85
+ * here instead of as an opaque crash in the middle of a parse. Off by default.
86
+ */
87
+ validate?: boolean;
88
+ /** Name of the grammar file, used to make validation errors actionable. */
89
+ source?: string;
90
+ };
91
+
92
+ export type FromProductionsOptions<TokenType> = {
93
+ productions: Production[];
94
+ toStringifiedTokenType: (tokenType: TokenType) => string;
95
+ mode?: "lr1" | "lalr1";
96
+ /** Generating states can take a while for big grammars; set to true to silence the timing log. */
97
+ quiet?: boolean;
98
+ onWarning?: (warning: ProductionWarning) => void;
99
+ onProgress?: (statesGenerated: number) => void;
100
+ };
101
+
42
102
  export class Sparse<TokenType, Metadata, Node> {
43
103
  public constructor(
44
- public readonly options: {
45
- productions: Production[];
46
- states: TableState[];
47
- toStringifiedTokenType: (tokenType: TokenType) => string;
48
- },
49
- ) {}
104
+ public readonly options: SparseOptions<TokenType>,
105
+ ) {
106
+ if (options.validate === true) {
107
+ const result = validateTable(options.productions, options.states, { source: options.source });
108
+ if (result.success === false)
109
+ throw new Error(`The parsing table does not match the grammar: ${result.reason}`);
110
+ }
111
+ }
50
112
 
51
- public static tryFromProductions<TokenType, Metadata, Node>(options: {
52
- productions: Production[];
53
- toStringifiedTokenType: (tokenType: TokenType) => string;
54
- mode?: "lr1" | "lalr1";
55
- }): Result<Sparse<TokenType, Metadata, Node>> {
56
- const statesResult = generateStates(options.productions, { mode: options.mode });
113
+ public static tryFromProductions<TokenType, Metadata, Node>(
114
+ options: FromProductionsOptions<TokenType>,
115
+ ): Result<Sparse<TokenType, Metadata, Node>> {
116
+ const startedAt = Date.now();
117
+
118
+ const statesResult = generateStates(options.productions, {
119
+ mode: options.mode,
120
+ onWarning: options.onWarning,
121
+ onProgress: options.onProgress,
122
+ });
57
123
  if (statesResult.success === false) return statesResult;
58
124
  const states = statesResult.value;
59
125
 
@@ -63,17 +129,63 @@ export class Sparse<TokenType, Metadata, Node> {
63
129
  toStringifiedTokenType: options.toStringifiedTokenType,
64
130
  });
65
131
 
132
+ if (options.quiet !== true) {
133
+ const elapsed = Date.now() - startedAt;
134
+ const label = options.mode === "lalr1" ? "LALR(1)" : "LR(1)";
135
+ console.error(`Sparse generated ${sparse.options.states.length} ${label} states in ${elapsed}ms`);
136
+ }
137
+
66
138
  return { success: true, value: sparse };
67
139
  }
68
140
 
69
- public static fromProductions<TokenType, Metadata, Node>(options: {
70
- productions: Production[];
71
- toStringifiedTokenType: (tokenType: TokenType) => string;
72
- mode?: "lr1" | "lalr1";
73
- }) {
141
+ public static fromProductions<TokenType, Metadata, Node>(options: FromProductionsOptions<TokenType>) {
74
142
  const result = Sparse.tryFromProductions<TokenType, Metadata, Node>(options);
75
- if (result.success === false)
76
- throw new Error(`Encountered error "${result.reason}" at ${result.token.line}:${result.token.column}`);
143
+ if (result.success === false) throw new SparseGrammarError(result.reason, result.token);
144
+ return result.value;
145
+ }
146
+
147
+ /** Builds a parser generator from a grammar file and a prebuilt parsing table file. */
148
+ public static async tryFromGrammarFile<TokenType, Metadata, Node>(options: {
149
+ grammarPath: string;
150
+ tablePath: string;
151
+ toStringifiedTokenType: (tokenType: TokenType) => string;
152
+ onWarning?: (warning: ProductionWarning) => void;
153
+ /** Set to false to skip cross-checking the table against the grammar. */
154
+ validate?: boolean;
155
+ }): Promise<Result<Sparse<TokenType, Metadata, Node>>> {
156
+ const productionsResult = await loadGrammar(options.grammarPath, { onWarning: options.onWarning });
157
+ if (productionsResult.success === false) return productionsResult;
158
+
159
+ const statesResult = await loadTable(options.tablePath, { productions: productionsResult.value });
160
+ if (statesResult.success === false) return statesResult;
161
+
162
+ if (options.validate !== false) {
163
+ const tableResult = validateTable(productionsResult.value, statesResult.value, {
164
+ source: options.tablePath,
165
+ });
166
+ if (tableResult.success === false) return tableResult;
167
+ }
168
+
169
+ return {
170
+ success: true,
171
+ value: new Sparse<TokenType, Metadata, Node>({
172
+ productions: productionsResult.value,
173
+ states: statesResult.value,
174
+ toStringifiedTokenType: options.toStringifiedTokenType,
175
+ source: options.grammarPath,
176
+ }),
177
+ };
178
+ }
179
+
180
+ public static async fromGrammarFile<TokenType, Metadata, Node>(options: {
181
+ grammarPath: string;
182
+ tablePath: string;
183
+ toStringifiedTokenType: (tokenType: TokenType) => string;
184
+ onWarning?: (warning: ProductionWarning) => void;
185
+ validate?: boolean;
186
+ }): Promise<Sparse<TokenType, Metadata, Node>> {
187
+ const result = await Sparse.tryFromGrammarFile<TokenType, Metadata, Node>(options);
188
+ if (result.success === false) throw new SparseGrammarError(result.reason, result.token);
77
189
  return result.value;
78
190
  }
79
191
 
@@ -99,6 +211,12 @@ export type ParserRecoveryFunction<TokenType, Metadata, Node> = (options: {
99
211
  success: true;
100
212
  token?: Token<TokenType, Metadata>;
101
213
  };
214
+ /**
215
+ * Pushes a token that the input did not contain (a synthesized semicolon, a closing brace, ...)
216
+ * onto both stacks, exactly as if the lexer had produced it. Unlike `finish({ newToken })`, which
217
+ * only hands the token back to the parser as a lookahead, this one actually shifts it.
218
+ */
219
+ insertToken: (token: Token<TokenType, Metadata>) => { success: false; reason: string } | undefined;
102
220
  states: TableState[];
103
221
  }) => { success: true; token?: Token<TokenType, Metadata> } | { success: false; reason: string };
104
222
 
@@ -108,38 +226,52 @@ export type ParserResult<TokenType, Metadata, Node> = {
108
226
  errors: ParserError<TokenType, Metadata>[];
109
227
  };
110
228
 
111
- class LR1Parser<TokenType, Metadata, Node> {
229
+ export class LR1Parser<TokenType, Metadata, Node> {
112
230
  statesStack: Stack<number> = new Stack([0]);
113
231
  symbolsStack: Stack<LR1StackSymbol<TokenType, Metadata, Node>> = new Stack();
114
232
  exceptions: ParserError<TokenType, Metadata>[] = [];
115
233
 
116
234
  public constructor(
117
- public readonly options: {
118
- productions: Production[];
119
- states: TableState[];
120
- toStringifiedTokenType: (tokenType: TokenType) => string;
121
- },
235
+ public readonly options: SparseOptions<TokenType>,
122
236
  public readonly reducer: ReducerType<TokenType, Metadata, Node>,
123
237
  public readonly recover: ParserRecoveryFunction<TokenType, Metadata, Node> | null,
124
238
  public readonly lexer: ReturnType<Slex<TokenType, Metadata>["generate"]>,
125
239
  ) {}
126
240
 
241
+ /**
242
+ * Clears the stacks and the collected errors so the parser can be handed to a fresh lexer.
243
+ * A parser instance consumes its lexer, so the usual way to parse a second input is to call
244
+ * `generator.generate(newLexer, options)` again.
245
+ */
246
+ public reset() {
247
+ this.statesStack = new Stack([0]);
248
+ this.symbolsStack = new Stack();
249
+ this.exceptions = [];
250
+ }
251
+
127
252
  public parse(): ParserResult<TokenType, Metadata, Node> {
128
253
  const { productions, states, toStringifiedTokenType } = this.options;
129
254
 
255
+ if (states.length === 0)
256
+ throw new LR1ParserGraveError("The parsing table is empty, so no token can be parsed", null as any);
257
+
258
+ const currentStates = () => {
259
+ const state = states[this.statesStack.peek()];
260
+ if (state == null)
261
+ throw new LR1ParserGraveError(
262
+ `The parsing table is out of sync with the grammar: state ${this.statesStack.peek()} does not exist (the table has ${states.length} states)`,
263
+ null as any,
264
+ );
265
+ return state;
266
+ };
267
+
130
268
  while (true) {
131
- let currentState = this.statesStack.peek();
269
+ let currentState = currentStates();
132
270
  let token = this.lexer.peekNextToken();
133
- let action = states[currentState].getTerminalAction(toStringifiedTokenType(token.type));
271
+ let action = currentState.getTerminalAction(toStringifiedTokenType(token.type));
134
272
 
135
273
  if (action == null) {
136
- if (this.recover === null)
137
- throw new LR1ParserGraveError(
138
- `Invalid syntax. Expected ${[...states[currentState].actions.keys()].join(
139
- ",",
140
- )} but got ${toStringifiedTokenType(token.type)}`,
141
- token,
142
- );
274
+ if (this.recover === null) throw new LR1ParserGraveError(this.syntaxErrorMessage(currentState, token), token);
143
275
 
144
276
  const recoveryResult = this.recover({
145
277
  lexer: this.lexer,
@@ -149,22 +281,25 @@ class LR1Parser<TokenType, Metadata, Node> {
149
281
  addError: (reason: string) => this.exceptions.push({ message: reason, token: this.lexer.peekNextToken() }),
150
282
  crash: (reason: string) => ({ success: false, reason }),
151
283
  finish: (options?: { newToken: Token<TokenType, Metadata> }) => ({ success: true, token: options?.newToken }),
284
+ insertToken: (newToken: Token<TokenType, Metadata>) => this.insertToken(newToken),
152
285
  isSafe: () =>
153
- states[this.statesStack.peek()].actions.has(`[${toStringifiedTokenType(this.lexer.peekNextToken().type)}]`),
286
+ currentStates().actions.has(`[${toStringifiedTokenType(this.lexer.peekNextToken().type)}]`),
154
287
  });
155
288
 
156
289
  if (recoveryResult.success === false) throw new LR1ParserGraveError(recoveryResult.reason, token);
157
290
 
158
291
  token = recoveryResult.token ?? this.lexer.peekNextToken();
159
- currentState = this.statesStack.peek();
160
- action = states[currentState].getTerminalAction(toStringifiedTokenType(token.type));
292
+ currentState = currentStates();
293
+ action = currentState.getTerminalAction(toStringifiedTokenType(token.type));
294
+
295
+ if (action == null) throw new LR1ParserGraveError(this.syntaxErrorMessage(currentState, token), token);
161
296
  }
162
297
 
163
298
  const fixedAction = action as TableAction;
164
299
  if (fixedAction.type === "shift") {
165
300
  // add current token to the stack and push the next state
166
301
  token = this.lexer.getNextToken();
167
- this.statesStack.push(fixedAction.value);
302
+ this.pushState(fixedAction.value);
168
303
  this.symbolsStack.push({ type: "token", token: token });
169
304
  }
170
305
 
@@ -173,6 +308,12 @@ class LR1Parser<TokenType, Metadata, Node> {
173
308
  if (fixedAction.value === 0) break;
174
309
  const production = productions[fixedAction.value];
175
310
 
311
+ if (production == null)
312
+ throw new LR1ParserGraveError(
313
+ `The parsing table is out of sync with the grammar: it reduces by production ${fixedAction.value}, but the grammar only has ${productions.length} productions`,
314
+ token,
315
+ );
316
+
176
317
  // pop the stack and reduce by this production
177
318
  const popped: LR1StackSymbol<TokenType, Metadata, Node>[] = [];
178
319
  for (let i = 0; i < production.rhs.length; i++) {
@@ -200,17 +341,18 @@ class LR1Parser<TokenType, Metadata, Node> {
200
341
  const node = this.reducer(newInput, oldInput);
201
342
  this.symbolsStack.push({ type: "node", node });
202
343
  } catch (err) {
203
- const input = JSON.stringify(popped, null, 2);
204
- console.log(err);
344
+ const cause = err instanceof Error ? err.message : String(err);
345
+ const input = truncate(JSON.stringify(popped), 500);
205
346
  throw new LR1ParserGraveError(
206
- `Error while performing reduction for production: ${productionIndex}. Reduction input: ${input}`,
347
+ `Error while performing the reduction for production ${productionIndex}${describeProduction(production)}: ${cause}\nReduction input: ${input}`,
207
348
  token,
349
+ err instanceof Error ? err : undefined,
208
350
  );
209
351
  }
210
352
 
211
353
  // get the top node and figure out what state to add to state stack
212
354
  const topNode = this.statesStack.peek();
213
- const gotoAction = states[topNode].getVariableAction(production.identifier);
355
+ const gotoAction = currentStates().getVariableAction(production.identifier);
214
356
 
215
357
  if (gotoAction == undefined)
216
358
  throw new LR1ParserGraveError(
@@ -223,21 +365,89 @@ class LR1Parser<TokenType, Metadata, Node> {
223
365
  token,
224
366
  );
225
367
 
226
- this.statesStack.push(gotoAction.value);
368
+ this.pushState(gotoAction.value);
227
369
  }
228
370
  }
229
371
 
372
+ // A table can accept the empty input by reducing to production 0 straight away, in which case
373
+ // nothing was ever shifted and there is no tree to hand back.
374
+ if (this.symbolsStack.isEmpty) return { result: null, errors: this.exceptions };
375
+
230
376
  const top = this.symbolsStack.peek();
231
377
  return { result: top.type === "node" ? top.node : null, errors: this.exceptions };
232
378
  }
379
+
380
+ private pushState(state: number) {
381
+ if (state === undefined || this.options.states[state] == null)
382
+ throw new LR1ParserGraveError(
383
+ `The parsing table is out of sync with the grammar: it points at state ${state}, but the table only has ${this.options.states.length} states`,
384
+ null as any,
385
+ );
386
+ this.statesStack.push(state);
387
+ }
388
+
389
+ /** Shifts a token that was not in the input, used by error recovery. */
390
+ private insertToken(token: Token<TokenType, Metadata>): { success: false; reason: string } | undefined {
391
+ const state = this.options.states[this.statesStack.peek()];
392
+ const action = state?.getTerminalAction(this.options.toStringifiedTokenType(token.type));
393
+
394
+ if (action == null)
395
+ return {
396
+ success: false,
397
+ reason: `Cannot insert ${this.options.toStringifiedTokenType(token.type)} here: state ${
398
+ this.statesStack.peek()
399
+ } does not accept it. Expected ${state?.getTerminalKeys().join(", ") ?? "nothing"}`,
400
+ };
401
+
402
+ if (action.type !== "shift")
403
+ return {
404
+ success: false,
405
+ reason: `Cannot insert ${this.options.toStringifiedTokenType(
406
+ token.type,
407
+ )} here: state ${this.statesStack.peek()} reduces on it instead of shifting it`,
408
+ };
409
+
410
+ this.pushState(action.value);
411
+ this.symbolsStack.push({ type: "token", token });
412
+ return undefined;
413
+ }
414
+
415
+ private syntaxErrorMessage(state: TableState, token: Token<TokenType, Metadata>) {
416
+ const expected = state.getTerminalKeys();
417
+ const got = this.options.toStringifiedTokenType(token.type);
418
+ const lexeme = token.lexeme === "" ? "" : ` ("${token.lexeme}")`;
419
+ return (
420
+ `Invalid syntax at ${token.line}:${token.column}: got ${got}${lexeme}, but expected ${
421
+ expected.length === 0 ? "the end of the input" : `one of [${expected.join("], [")}]`
422
+ }`
423
+ );
424
+ }
233
425
  }
234
426
 
235
427
  export class LR1ParserGraveError<TokenType, Metadata> extends Error {
236
428
  constructor(
237
429
  public readonly reason: string,
238
- public readonly currentToken: Token<TokenType, Metadata>,
430
+ public readonly currentToken: Token<TokenType, Metadata> | null,
239
431
  err?: Error,
240
432
  ) {
241
- super(reason);
433
+ super(reason, err === undefined ? undefined : { cause: err });
434
+ this.name = "LR1ParserGraveError";
242
435
  }
243
436
  }
437
+
438
+ /** A grammar (or table) problem found before parsing started, always tied to a place in the file. */
439
+ export class SparseGrammarError extends Error {
440
+ constructor(
441
+ public readonly reason: string,
442
+ public readonly token: GrammarToken,
443
+ ) {
444
+ super(`Encountered error "${reason}" at ${token.line}:${token.column}`);
445
+ this.name = "SparseGrammarError";
446
+ }
447
+ }
448
+
449
+ const truncate = (value: string, max: number) =>
450
+ value.length <= max ? value : `${value.slice(0, max)}... (${value.length} characters)`;
451
+
452
+ const describeProduction = (production: Production) =>
453
+ production.name == null ? ` (${production.identifier})` : ` ("${production.name}")`;