@scinorandex/sparse 0.0.8 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. package/.github/workflows/ci.yml +41 -0
  2. package/AGENTS.md +60 -0
  3. package/README.md +286 -30
  4. package/dist/cli.js +113 -24
  5. package/dist/cli.js.map +1 -1
  6. package/dist/generator.d.ts +9 -2
  7. package/dist/generator.js +179 -50
  8. package/dist/generator.js.map +1 -1
  9. package/dist/index.d.ts +17 -4
  10. package/dist/index.js +34 -1
  11. package/dist/index.js.map +1 -1
  12. package/dist/meta/common.d.ts +17 -0
  13. package/dist/meta/common.js +66 -1
  14. package/dist/meta/common.js.map +1 -1
  15. package/dist/meta/selfhosted.d.ts +4 -5
  16. package/dist/meta/selfhosted.js +51 -38
  17. package/dist/meta/selfhosted.js.map +1 -1
  18. package/dist/parser.d.ts +56 -33
  19. package/dist/parser.js +146 -18
  20. package/dist/parser.js.map +1 -1
  21. package/dist/table/selfhosted.d.ts +10 -1
  22. package/dist/table/selfhosted.js +32 -12
  23. package/dist/table/selfhosted.js.map +1 -1
  24. package/dist/table/validate.d.ts +17 -0
  25. package/dist/table/validate.js +84 -0
  26. package/dist/table/validate.js.map +1 -0
  27. package/dist/utils/Stack.d.ts +4 -0
  28. package/dist/utils/Stack.js +16 -1
  29. package/dist/utils/Stack.js.map +1 -1
  30. package/dist/utils/errorWindowBuilder.js +11 -11
  31. package/dist/utils/errorWindowBuilder.js.map +1 -1
  32. package/dist/utils/loadFiles.d.ts +8 -0
  33. package/dist/utils/loadFiles.js +47 -0
  34. package/dist/utils/loadFiles.js.map +1 -0
  35. package/dist/utils/reducers.d.ts +9 -0
  36. package/dist/utils/reducers.js +56 -0
  37. package/dist/utils/reducers.js.map +1 -0
  38. package/example/LoLang/example.ts +42 -18
  39. package/example/LoLang/tablelalr.txt +381 -0
  40. package/example/kleene-test/example.ts +35 -17
  41. package/example/math/example.ts +12 -7
  42. package/example/selfhosted/example.ts +30 -24
  43. package/opencode.json +18 -0
  44. package/package.json +9 -3
  45. package/src/cli.ts +132 -23
  46. package/src/generator.ts +237 -67
  47. package/src/index.ts +56 -3
  48. package/src/meta/common.ts +116 -0
  49. package/src/meta/selfhosted.ts +42 -11
  50. package/src/parser.ts +259 -47
  51. package/src/table/selfhosted.ts +48 -12
  52. package/src/table/validate.ts +143 -0
  53. package/src/utils/Stack.ts +22 -2
  54. package/src/utils/errorWindowBuilder.ts +15 -13
  55. package/src/utils/loadFiles.ts +54 -0
  56. package/src/utils/reducers.ts +93 -0
  57. package/test/cli.test.ts +144 -0
  58. package/test/codegen.test.ts +75 -0
  59. package/test/grammar.test.ts +243 -0
  60. package/test/helpers.ts +59 -0
  61. package/test/lalr.test.ts +114 -0
  62. package/test/parser.test.ts +392 -0
  63. package/test/table.test.ts +144 -0
  64. package/tsconfig.json +1 -1
  65. package/example/complicated/grammar.txt +0 -37
@@ -0,0 +1,41 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+
8
+ jobs:
9
+ verify:
10
+ runs-on: ubuntu-latest
11
+
12
+ steps:
13
+ - uses: actions/checkout@v4
14
+
15
+ - uses: actions/setup-node@v4
16
+ with:
17
+ node-version: 20
18
+ cache: yarn
19
+
20
+ - run: yarn install --frozen-lockfile
21
+
22
+ - name: Typecheck
23
+ run: yarn tc
24
+
25
+ - name: Test
26
+ run: yarn test
27
+
28
+ - name: Build
29
+ run: yarn build
30
+
31
+ - name: Run the examples
32
+ run: |
33
+ npm install -g tsx
34
+ npx tsx example/math/example.ts
35
+ npx tsx example/kleene-test/example.ts
36
+ npx tsx example/LoLang/example.ts
37
+
38
+ - name: The committed tables are up to date
39
+ run: |
40
+ node dist/cli.js --input=example/math/grammar.txt --output=example/math/table.txt --check
41
+ node dist/cli.js --input=example/LoLang/grammar.txt --output=example/LoLang/table.txt --check
package/AGENTS.md ADDED
@@ -0,0 +1,60 @@
1
+ # AGENTS.md
2
+
3
+ `@scinorandex/sparse` — LR(1) parser generator library (TypeScript, CJS via tsc, yarn 1).
4
+ Single package: `src/` = library + CLI, `example/` = runnable demos, `test/` = vitest suite.
5
+ README.md is the user-facing doc (grammar syntax, reducer contract, recovery guide, API reference).
6
+
7
+ ## Commands
8
+ - Typecheck: `yarn tc` (covers src/, example/ and test/)
9
+ - Build: `yarn build` → `dist/` (CLI is `dist/cli.js`)
10
+ - Test: `yarn test` (vitest; builds `dist/` first because `test/cli.test.ts` runs the CLI)
11
+ - Run examples: `tsx example/<name>/example.ts` — tsx is global, not a repo dep; `node file.ts` fails (enums)
12
+ - Table CLI: `node dist/cli.js --input=<grammar> --output=<table>` (build first); output matches committed tables exactly
13
+ - `yarn bench` exists but there are no bench files. Verify with `yarn tc && yarn test` plus the examples.
14
+ - `dist/` and `yarn.lock` are gitignored.
15
+
16
+ ## Self-hosting / codegen (critical, easy to break)
17
+ - The lib parses its own grammar-file and table-file formats using **pre-generated** LR(1) states:
18
+ `src/meta/states.ts` (grammar format) and `src/table/states.ts` (table format) are generated files —
19
+ byte-identical copies of `example/selfhosted/codegen-meta.ts` and `codegen-table.ts`.
20
+ - To change the grammar/table file syntax: edit `example/selfhosted/01-meta-grammar.txt` and/or
21
+ `02-table-grammar.txt`, run `tsx example/selfhosted/example.ts`, then copy the regenerated
22
+ `codegen-*.ts` over the two `src/.../states.ts` files. The old (checked-in) parser must still
23
+ parse the modified grammar (bootstrap constraint).
24
+ - `test/codegen.test.ts` enforces this: it regenerates both state files in memory and compares them to
25
+ the checked-in ones. It compares objects (not file bytes) so the `cp` step stays manual.
26
+ - Changing the *lexer* rules in `grammarLexerGenerator` (e.g. allowing `-` inside identifiers) does NOT
27
+ change the generated states — the token types and positions stay the same.
28
+
29
+ ## Non-obvious behavior
30
+ - `?`, `*`, `+` groups and `|` alternatives are unrolled into extra productions; `*`/`+`
31
+ create `autogen-N` productions whose reducer is named `autogenerated-kleene` — user reducers must
32
+ handle it (see `example/kleene-test/`). `?` on a group that can match nothing produces an empty
33
+ production and is rejected by `validateProductions`.
34
+ - Modifiers only apply to parenthesized groups: `([A] [B])?`, not `[A]?`.
35
+ - `|` alternatives live inside a single token: `[A | B]`, `<A | B>`.
36
+ - Two construction paths: `Sparse.fromProductions(...)` builds states at runtime (slow, ~30s for
37
+ 150 productions) vs `Sparse.fromGrammarFile(...)` / `new Sparse({ productions, states })` (fast).
38
+ - Reducers are keyed by production name (`<LHS: name>` in the grammar file); nameless productions
39
+ have `name: null`. Production index 0 is the accept production —
40
+ the parse loop breaks on it without calling the reducer. `oldInput.index` is the pre-unrolling index.
41
+ - Table file format: one line per state, comma-separated `[TERMINAL]=sN|rN` and `<VARIABLE>=M`.
42
+ The reader preserves file order via `getItemsReversed()` (the item lists are built right-recursively).
43
+ - Errors flow as `Result<T>` (`{ success: false, reason, token }`); `buildProductions` /
44
+ `Sparse.fromProductions` / `buildStates` are the throwing wrappers.
45
+ - `GeneratorResult.ActionTable` is sparse: a state whose items only move to other states has no entry.
46
+ `toStates()`/`toTable()` use `Array.from`, not `map`, so state numbering stays aligned; `toTable()`
47
+ throws for such grammars because the file format cannot express an action-less state.
48
+ - LALR(1) mode merges same-core LR(1) states. Same-core states always agree on their shifts, gotos and
49
+ reduce targets, so the conflict branch in `mergeStatesForLALR` is currently unreachable.
50
+
51
+ ## Key files
52
+ - `src/generator.ts` — LR(1) state construction (first/follow sets, item-set expansion), `toTable()` serialization
53
+ - `src/parser.ts` — runtime parser loop + `recover` error-recovery callback API (`insertToken`, `finish`, `crash`)
54
+ - `src/meta/selfhosted.ts` — grammar-file parser + production unrolling
55
+ - `src/meta/common.ts` — `Production` type + `validateProductions`
56
+ - `src/table/selfhosted.ts` — table-file parser (`tryBuildStates`), `src/table/validate.ts` — table validation
57
+ - `src/cli.ts` — `--input/--output/--lalr/--check/--stdout/--quiet/--help`
58
+ - `example/` — `math` (prebuilt table), `kleene-test` (`*`/`+`), `LoLang` (large grammar + recovery, exports
59
+ its lexer for the tests), `selfhosted` (codegen)
60
+ - `test/` — `grammar`, `table`, `parser`, `lalr`, `codegen`, `cli`
package/README.md CHANGED
@@ -1,20 +1,33 @@
1
1
  ## Sparse - Scin's Parsing Library
2
2
 
3
- Sparse allows developers to easily create LR1 parsers and LR1 parsing tables.
3
+ Sparse allows developers to easily create LR(1) parsers and LR(1) parsing tables.
4
4
 
5
5
  **Features:**
6
6
  - Complete Error Handling API - when parsing fails, you have the final say on where parsing stops
7
- - Deferred Reductions - you create the nodes and Sparse builds the tree
7
+ - Deferred Reductions - you create the nodes and Sparse builds the tree
8
+ - Grammar and table validation - mistakes are reported with the line, column and a window into your file
8
9
  - Full [Slex](https://github.com/scinscinscin/slex) integration - define your entire language as a set of DFA and CFG rules
9
10
  - Optional Table Output - use Sparse to build your tables for use in other languages
10
11
 
11
- ## Getting Started
12
+ ## Table of contents
13
+
14
+ - [Getting started](#getting-started)
15
+ - [Grammar syntax reference](#grammar-syntax-reference)
16
+ - [Building a parser](#building-a-parser)
17
+ - [Reducers](#reducers)
18
+ - [Repetition with `*` and `+`](#repetition-with--and-)
19
+ - [Error recovery](#error-recovery)
20
+ - [Generating and shipping the parsing table](#generating-and-shipping-the-parsing-table)
21
+ - [API reference](#api-reference)
22
+ - [Known limitations](#known-limitations)
23
+
24
+ ## Getting started
12
25
 
13
26
  1. **Define the CFG of your language.**
14
27
 
15
- The CFG is defined by creating a file containing a list of productions. Variables are identifiers encased in angle brackets like `<STATEMENT>`, while terminals are encased in square-brackets like `[L_COLON]`.
28
+ The CFG is defined by creating a file containing a list of productions. Variables are identifiers encased in angle brackets like `<STATEMENT>`, while terminals are encased in square-brackets like `[L_COLON]`.
16
29
 
17
- An example production includes: `<IF_STATEMENT> : [IF] [L_PAREN] <EXPRESSION> [R_PAREN] <STATEMENT> <ELSE_IF_STATEMENTS> [ELSE] <STATEMENT>;`.
30
+ An example production includes: `<IF_STATEMENT> : [IF] [L_PAREN] <EXPRESSION> [R_PAREN] <STATEMENT> [ELSE] <STATEMENT>;`.
18
31
 
19
32
  An example CFG for a basic MDAS calculator is the following:
20
33
 
@@ -22,16 +35,20 @@ An example CFG for a basic MDAS calculator is the following:
22
35
  <S>: <PROGRAM>;
23
36
  <PROGRAM>: <EXPRESSION> [EOF];
24
37
  <PROGRAM>: [EOF];
38
+
39
+ // Begin parsing arithmetic expressions
25
40
  <EXPRESSION>: <TERM_EXPRESSION>;
26
41
  <TERM_EXPRESSION>: <FACTOR_EXPRESSION> [PLUS] <TERM_EXPRESSION>;
27
42
  <TERM_EXPRESSION>: <FACTOR_EXPRESSION> [MINUS] <TERM_EXPRESSION>;
28
43
  <TERM_EXPRESSION>: <FACTOR_EXPRESSION>;
29
44
  <FACTOR_EXPRESSION>: <ENDPOINT> [STAR] <FACTOR_EXPRESSION>;
30
- <FACTOR_EXPRESSION>: <ENDPOINT> [SLASH] <FACTOR_EXPRESSION>;
45
+ <FACTOR_EXPRESSION>: <ENDPOINT> [FORWARD_SLASH] <FACTOR_EXPRESSION>;
31
46
  <FACTOR_EXPRESSION>: <ENDPOINT>;
32
47
  <ENDPOINT>: [NUMBER];
33
48
  ```
34
49
 
50
+ The first production has to be the "accept" production: a variable that appears nowhere else, whose body is a single variable. It is the only production whose reducer Sparse never calls.
51
+
35
52
  2. **Create your [Slex](https://github.com/scinscinscin/slex) lexer.**
36
53
 
37
54
  ```ts
@@ -61,7 +78,7 @@ lexerGenerator.addRule("number_literal", "${float_number}|${decimal_number}", To
61
78
  const lexer = lexerGenerator.generate(`2.4 + 3.5 * 1 / 456.789`, () => ({}));
62
79
  ```
63
80
 
64
- 3. **Define the node representation.**
81
+ 3. **Define the node representation.**
65
82
 
66
83
  Sparse allows you to build the AST however you want, deferring to your functions when its time to make a reduction, giving you control over the representation.
67
84
 
@@ -71,7 +88,7 @@ type StringifiedNode = (StringifiedNode | string)[];
71
88
  // It doesn't have to be a class. It just has to be a structure that all nodes
72
89
  // in the AST adhere to. Classes allow this to be done easily through subclassing.
73
90
  class Node {
74
- constructor(public readonly nodes: LR1StackSymbol<TokenType, {}, Node>[]) {}
91
+ constructor(public readonly nodes: LR1StackSymbol<TokenType, Metadata, Node>[]) {}
75
92
  toObject(): StringifiedNode {
76
93
  return this.nodes.map((node) => (node.type === "token"
77
94
  ? node.token.lexeme
@@ -81,51 +98,290 @@ class Node {
81
98
  }
82
99
  ```
83
100
 
84
- 4. **Building the productions and the parser.**
101
+ 4. **Building the parser.**
85
102
 
86
- Productions are built using the `buildProductions()` function, which can be passed into Sparse alongside `toStringifiedTokenType`, which converts a numerical TypeScript enum to the name of the terminal.
103
+ With a prebuilt parsing table, `Sparse.fromGrammarFile` reads the grammar, reads the table, and checks that the two belong together:
87
104
 
88
105
  ```ts
106
+ import { Sparse, enumToString } from "@scinorandex/sparse";
107
+
108
+ async function main() {
109
+ const parserGenerator = await Sparse.fromGrammarFile<TokenType, Metadata, Node>({
110
+ grammarPath: "./example/math/grammar.txt",
111
+ tablePath: "./example/math/table.txt",
112
+ toStringifiedTokenType: enumToString<TokenType>(TokenType),
113
+ onWarning: ({ reason, token }) => console.warn(`${token.line}:${token.column} ${reason}`),
114
+ });
115
+
116
+ const parser = parserGenerator.generate(lexer, {
117
+ reducer: (_, { input }) => new Node(input),
118
+ });
119
+
120
+ console.log(parser.parse().result!.toObject());
121
+ }
122
+ ```
123
+
124
+ Generating the table at startup instead (fine for small grammars, slow for big ones):
125
+
126
+ ```ts
127
+ import { buildProductions, Sparse, enumToString } from "@scinorandex/sparse";
128
+
89
129
  async function main() {
90
- const toStringifiedTokenType = (type: TokenType) => TokenType[type];
91
130
  const productions = buildProductions(await fs.readFile("./example/math/grammar.txt", "utf8"));
92
- const parserGenerator = Sparse.fromProductions<TokenType, Metadata, Node>({ productions, toStringifiedTokenType });
131
+ const parserGenerator = Sparse.fromProductions<TokenType, Metadata, Node>({
132
+ productions,
133
+ toStringifiedTokenType: enumToString<TokenType>(TokenType),
134
+ });
93
135
  }
94
136
  ```
95
137
 
96
- 5. **Define your reducer and begin parsing.**
138
+ ## Grammar syntax reference
139
+
140
+ A grammar file is a list of productions. Whitespace is insignificant, `//` starts a line comment, and `/* ... */` spans lines.
141
+
142
+ | Syntax | Meaning |
143
+ | --- | --- |
144
+ | `<A>: <B> [C];` | `A` is made of one `B` followed by one `C` |
145
+ | `<A: name>: ...;` | The production is *named*: its reducer is looked up under `name` |
146
+ | `[TOK: name]` | Names the terminal on the right hand side so it shows up on the reducer's `bag` |
147
+ | `<VAR: name>` | Same, for a variable |
148
+ | `[A | B]` | Alternatives: unrolled into one production per alternative |
149
+ | `<A | B>` | Same, for variables |
150
+ | `([A] [B])?` | Optional group, unrolled into one production with and one without it |
151
+ | `// comment`, `/* comment */` | Comments |
152
+
153
+ Identifiers may contain letters, digits, `_` and `-`, and must start with a letter or `_`.
154
+
155
+ ### What Sparse checks for you
156
+
157
+ Grammar and table problems are reported as a `Result` failure with the offending token, so the CLI and
158
+ `loadGrammar` can print a window into your file:
97
159
 
98
- Finally, you can define the parser's reduction function and optionally define the recovery handling function.
160
+ - a variable used on the right hand side that no production defines
161
+ - an empty right hand side (a production like `<A: a>: ([X])?;` expands to nothing, and Sparse has no support for empty productions)
162
+ - the start symbol appearing on the right hand side
163
+ - two symbols in one production sharing a name (warns: only the last one reaches the `bag`)
164
+ - a production that is never reachable from the start symbol (warns)
165
+
166
+ ```
167
+ $ npx sparse --input=broken.txt --output=table.txt
168
+ Invalid syntax at 2:30: got SEMICOLON (";"), but expected one of [L_ANGLE], [L_BRACKET], [L_PAREN]
169
+
170
+ 2 | <PROGRAM: program>: [NUMBER] (;
171
+ | ~
172
+ 3 |
173
+ ```
174
+
175
+ ## Reducers
176
+
177
+ Reducers are keyed by the name in the grammar, and receive two arguments:
178
+
179
+ ```ts
180
+ type Reducer<TokenType, Metadata, Node> = (
181
+ newInput: { bag: Record<string, Node | Token<TokenType, Metadata>>; name: string | null },
182
+ oldInput: { input: LR1StackSymbol<TokenType, Metadata, Node>[]; index: number },
183
+ ) => Node;
184
+ ```
185
+
186
+ - `newInput.name` is the production's name, or `null` for an unnamed production.
187
+ - `newInput.bag` holds the right hand side symbols that were named in the grammar, keyed by those names. If two symbols share a name, the last one wins, so reach for `oldInput.input` when that happens.
188
+ - `oldInput.input` is the reduced symbols in order, as `{ type: "token", token }` or `{ type: "node", node }`.
189
+ - `oldInput.index` is the production's index *in the grammar file*, before `*`/`?`/`+`/`|` unrolling. Index 0 is the accept production, whose reducer is never called.
190
+
191
+ `defineReducers` turns a map of reducers into the reducer that `generate` wants, and tells you exactly which production has no reducer:
99
192
 
100
193
  ```ts
194
+ import { assertReducersCoverGrammar, defineReducers } from "@scinorandex/sparse";
195
+
196
+ const reducers = {
197
+ program: ({ bag }) => new Node(bag),
198
+ expression: ({ bag }, { input }) => new Node(input),
199
+ };
200
+
201
+ assertReducersCoverGrammar(productions, reducers); // throws if something is missing
202
+
101
203
  const parser = parserGenerator.generate(lexer, {
102
- reducer: (_, { input, index }) => new Node(input),
204
+ reducer: defineReducers(reducers),
103
205
  });
104
-
105
- console.log(parser.parse().result!.toObject());
106
206
  ```
107
207
 
108
- ## Exporting and Loading the Parsing Table
208
+ Without the helper the failure looks like this, halfway through a parse:
109
209
 
110
- It takes a while for Sparse to build states (around 30 seconds for a file containing 150 productions). This can be alleviated by generating the parsing table and loading the states directly instead. This method also allows you to edit the parsing table to resolve parsing conflicts.
210
+ ```
211
+ Error while performing the reduction for production 2 ("expression"): I cannot reduce a null bag
212
+ ```
111
213
 
112
- To create the the states, you can run `npx @scinorandex/sparse --input=<input> --output=<output>`.
214
+ ## Repetition with `*` and `+`
113
215
 
114
- Afterwards, you can create a parser with pre-built states like the following:
216
+ `[A]?`, `A B*`, and `A B+` are unrolled into extra productions. For `*` and `+` that means a new
217
+ production named **`autogenerated-kleene`** (with variables called `autogen-0`, `autogen-1`, ...), which
218
+ you **have** to implement: it is handed the items matched so far, and it has to flatten them into a list.
219
+ The first reduction has no `rest`, later ones do:
115
220
 
116
221
  ```ts
117
- async function main() {
118
- const toStringifiedTokenType = (type: TokenType) => TokenType[type];
119
- const productions = buildProductions(await fs.readFile("./example/math/grammar.txt", "utf8"));
222
+ class KleeneNode<T> extends Node {
223
+ contents: T[] = [];
224
+
225
+ constructor(nodes: LR1StackSymbol<TokenType, Metadata, Node>[], bag: T) {
226
+ super(nodes);
227
+ this.contents.unshift(bag);
228
+ }
229
+
230
+ add(nodes: LR1StackSymbol<TokenType, Metadata, Node>[], bag: T) {
231
+ this.nodes.unshift(...nodes.slice(0, nodes.length - 1));
232
+ this.contents.unshift(bag);
233
+ return this;
234
+ }
235
+ }
120
236
 
121
- // Notice how the "new Sparse()" constructor is used here instead of "Sparse.fromProductions()"
122
- const states = buildStates(await fs.readFile("./example/math/table.txt", "utf8"));
123
- const parserGenerator = new Sparse<TokenType, Metadata, Node>({ productions, states, toStringifiedTokenType });
237
+ const reducers = {
238
+ "autogenerated-kleene": ({ bag: { rest, ...item } }, { input }) =>
239
+ rest == null ? new KleeneNode(input, item) : rest.add(input, item),
240
+ program: (_, { input }) => new Node(input),
241
+ };
242
+ ```
243
+
244
+ See `example/kleene-test` for the whole thing.
245
+
246
+ ## Error recovery
247
+
248
+ By default a syntax error throws `LR1ParserGraveError`, which carries the token that could not be parsed:
249
+
250
+ ```ts
251
+ try {
252
+ parser.parse();
253
+ } catch (err) {
254
+ if (err instanceof LR1ParserGraveError) console.error(err.reason, err.currentToken);
124
255
  }
125
256
  ```
126
257
 
127
- ---
258
+ Pass a `recover` function to decide what happens instead. It receives the lexer, both stacks, the table,
259
+ and these helpers:
260
+
261
+ | Helper | What it does |
262
+ | --- | --- |
263
+ | `addError(reason)` | Records an error, keeps parsing, and returns it in `parse().errors` |
264
+ | `crash(reason)` | Stops parsing and throws a `LR1ParserGraveError` |
265
+ | `finish(options?)` | Stops recovery and carries on parsing from the current state |
266
+ | `insertToken(token)` | Shifts a token the input did not have, e.g. a synthesized `;` |
267
+ | `isSafe()` | True when the current state can consume the next real token |
268
+
269
+ ```ts
270
+ const parser = parserGenerator.generate(lexer, {
271
+ reducer: (_, { input }) => new Node(input),
272
+ recover({ lexer, states, statesStack, insertToken, addError, isSafe, finish, crash }) {
273
+ const token = lexer.peekNextToken();
274
+
275
+ // skip the extra semicolons the author left behind
276
+ while (token.type === TokenType.SEMICOLON) {
277
+ lexer.getNextToken();
278
+ if (isSafe()) return finish();
279
+ }
280
+
281
+ // or insert the semicolon they forgot
282
+ if (states[statesStack.peek()].getTerminalAction("SEMICOLON") != null) {
283
+ const semicolon = new Token(TokenType.SEMICOLON, ";", new ColumnAndRow(token.line, token.column), {});
284
+ const inserted = insertToken(semicolon);
285
+ if (inserted != null) return crash(inserted.reason);
286
+ addError(`Expected SEMICOLON but received ${TokenType[token.type]}`);
287
+ if (isSafe()) return finish();
288
+ }
289
+
290
+ return crash("Invalid syntax");
291
+ },
292
+ });
293
+
294
+ const { result, errors } = parser.parse();
295
+ ```
296
+
297
+ A parser consumes its lexer, so build a new one per input with `generate(...)`, or call `parser.reset()`
298
+ if you want to reuse the parser with a lexer that still has tokens left.
299
+
300
+ ## Generating and shipping the parsing table
301
+
302
+ Generating states for a large grammar takes a while (a few seconds for a few hundred productions), so
303
+ generate the table once at build time and commit it:
304
+
305
+ ```
306
+ npx sparse --input=grammar.txt --output=table.txt
307
+ ```
308
+
309
+ | Flag | Meaning |
310
+ | --- | --- |
311
+ | `--input=<file>` | Grammar to read (required) |
312
+ | `--output=<file>` | Table to write; missing directories are created |
313
+ | `--lalr` | Generate LALR(1) states instead of LR(1) |
314
+ | `--check` | Write nothing; verify that `<output>` already matches the grammar |
315
+ | `--stdout` | Print the table instead of writing it |
316
+ | `--quiet` | Hide warnings about suspicious rules |
317
+ | `--help` | Usage |
128
318
 
129
- ## Roadmap
319
+ By default, LR(1) states are generated. You can generate LALR(1) states instead by passing the `--lalr`
320
+ flag (or setting `mode: "lalr1"` in `Sparse.fromProductions`): the resulting table is never larger than
321
+ the LR(1) one. Using LALR(1) yields a ~35% performance boost over LR(1) for the same grammar (tested on
322
+ the LoLang example).
323
+
324
+ `--check` is useful in CI to prove that a committed table still matches its grammar:
325
+
326
+ ```
327
+ npx sparse --input=example/math/grammar.txt --output=example/math/table.txt --check
328
+ ```
130
329
 
131
- 1. Potentially add LALR(1) support. Currently, the parser only supports outputting LR(1) tables. Tables can be significantly smaller if it made LALR(1) tables instead.
330
+ Then load the table with `Sparse.fromGrammarFile`, as shown in [Getting started](#getting-started).
331
+
332
+ ## API reference
333
+
334
+ ### Reading and writing grammars
335
+
336
+ | Function | Returns |
337
+ | --- | --- |
338
+ | `buildProductions(source)` / `tryBuildProductions(source)` | The unrolled productions. Throws / returns a `Result` failure with the offending token |
339
+ | `loadGrammar(path)` | Same, but reads the file, turning missing files into failures too |
340
+ | `buildStates(table)` / `tryBuildStates(table)` | `TableState[]` from a table file, validating action syntax and state numbers |
341
+ | `loadTable(path)` | Same, but reads the file |
342
+ | `generateStates(productions, options?)` | `GeneratorResult`; `options` is `{ mode, onWarning, onProgress }` |
343
+ | `validateProductions(productions, options?)` | Fails on a grammar that cannot generate a correct table |
344
+ | `validateTable(productions, states)` | Fails when a table does not belong to the productions |
345
+ | `validateTableStates(states, options?)` | Fails on a malformed table |
346
+
347
+ ### Building parsers
348
+
349
+ | Function | Returns |
350
+ | --- | --- |
351
+ | `Sparse.fromGrammarFile({ grammarPath, tablePath, toStringifiedTokenType, onWarning?, validate? })` | Loads a grammar and its prebuilt table, cross-checking them |
352
+ | `Sparse.fromProductions({ productions, toStringifiedTokenType, mode?, quiet? })` | Generates states at startup |
353
+ | `Sparse.tryFromProductions(...)` | Same, as a `Result` |
354
+ | `new Sparse({ productions, states, toStringifiedTokenType, validate?, source? })` | Use when you already hold the states |
355
+ | `generator.generate(lexer, { reducer, recover? })` | A parser |
356
+ | `parser.parse()` | `{ result, errors }` |
357
+ | `parser.reset()` | Clears the stacks and errors |
358
+
359
+ ### Helpers
360
+
361
+ | Export | Purpose |
362
+ | --- | --- |
363
+ | `enumToString(TokenType)` | The `toStringifiedTokenType` every example used to write by hand |
364
+ | `defineReducers(reducers)` | Turns a map of named reducers into a reducer |
365
+ | `assertReducersCoverGrammar(productions, reducers)` | Fails up front when a named production has no reducer |
366
+ | `missingReducerNames(productions, reducers)` / `missingReducerMessage(...)` | Same, as data |
367
+ | `namedProductions(productions)` | Every production the grammar names |
368
+ | `TableState.fromJSObject(json)` / `GeneratorResult.toJSObject()` | Round trip a table through JSON |
369
+ | `hydrateProduction(json)` / `dehydateProduction(production)` | Round trip productions through JSON |
370
+ | `buildErrorWindow(source, token)` | Renders the window around a token |
371
+ | `Stack` | `push`, `pop`, `peek`, `peekAt`, `size`, `isEmpty`, `toArray` |
372
+
373
+ ## Known limitations
374
+
375
+ - `?`, `*` and `+` only apply to a parenthesized group, and a group that can match nothing
376
+ (`<A: a>: ([X])?;`) is rejected because Sparse has no support for empty productions. Write two
377
+ productions instead.
378
+ - Sparse does not detect grammar conflicts (shift/reduce, reduce/reduce). An ambiguous grammar produces a
379
+ table with one arbitrary resolution, so if your parser behaves strangely, check your grammar for ambiguity.
380
+ - A state with no actions cannot be written in the table file format; `toTable()` throws for such grammars.
381
+ Use `Sparse.fromProductions` to keep the states in memory instead.
382
+
383
+ ## AI Disclaimer
384
+
385
+ This project was originally written without the use of AI tools, the core LR(1) table generator was written by hand as per the algorithms described in the Dragon Book. Every release prior to v0.1 contained no AI generated code.
386
+
387
+ OpenCode and Qwen 3.8 27B were to implement performance improvements on the original LR(1) table generator and to implement LALR(1) support.
package/dist/cli.js CHANGED
@@ -5,43 +5,132 @@ var __importDefault = (this && this.__importDefault) || function (mod) {
5
5
  };
6
6
  Object.defineProperty(exports, "__esModule", { value: true });
7
7
  const promises_1 = __importDefault(require("fs/promises"));
8
- const index_1 = require("./index");
9
- const generator_1 = require("./generator");
10
- const minimist_1 = __importDefault(require("minimist"));
11
8
  const path_1 = __importDefault(require("path"));
12
- const errorWindowBuilder_1 = require("./utils/errorWindowBuilder");
9
+ const minimist_1 = __importDefault(require("minimist"));
10
+ const index_1 = require("./index");
13
11
  const args = (0, minimist_1.default)(process.argv.slice(2));
14
- function checkFileExists(filepath) {
15
- return new Promise((resolve, reject) => {
16
- promises_1.default.access(filepath, promises_1.default.constants.F_OK)
17
- .then(() => resolve(true))
18
- .catch(() => resolve(false));
19
- });
12
+ const BINARY_NAME = "@scinorandex/sparse";
13
+ const USAGE = `Usage: npx @scinorandex/sparse --input=<grammar> --output=<table> [options]
14
+
15
+ Generates an LR(1) parsing table from a grammar file.
16
+
17
+ Required:
18
+ --input=<file> Grammar file to read
19
+ --output=<file> Where to write the parsing table (created if the directory is missing)
20
+
21
+ Options:
22
+ --lalr Generate LALR(1) states instead of LR(1) (smaller table, ~35% faster parsing)
23
+ --check Do not write anything: verify that <output> already matches the grammar
24
+ --stdout Print the table instead of writing it to a file
25
+ --quiet Do not print warnings about suspicious grammar rules
26
+ --help Show this message
27
+
28
+ Run with npx ${BINARY_NAME} (the package is @scinorandex/sparse).`;
29
+ function fail(message) {
30
+ console.error(message);
31
+ process.exitCode = 1;
32
+ }
33
+ function toTableOrFail(result) {
34
+ try {
35
+ return result.toTable();
36
+ }
37
+ catch (err) {
38
+ fail(err instanceof Error ? err.message : String(err));
39
+ return null;
40
+ }
20
41
  }
21
42
  async function main() {
22
- if (typeof args.input !== "string" || typeof args.output !== "string") {
23
- console.log("Usage: npx @scinorandex/sparse --input=<input> --output=<output>");
43
+ if (args.help === true || args.h === true) {
44
+ console.log(USAGE);
45
+ return;
46
+ }
47
+ if (typeof args.input !== "string") {
48
+ fail(`Missing --input.\n\n${USAGE}`);
49
+ return;
50
+ }
51
+ const useStdout = args.stdout === true;
52
+ const check = args.check === true;
53
+ if (typeof args.output !== "string" && !useStdout) {
54
+ fail(`Missing --output.\n\n${USAGE}`);
24
55
  return;
25
56
  }
26
57
  const inputFile = path_1.default.resolve(args.input);
27
- const outputFile = path_1.default.resolve(args.output);
28
- console.log(`Reading grammar from ${inputFile} and outputting table to ${outputFile}\n`);
29
- if (!(await checkFileExists(inputFile)))
30
- return console.log(`File ${inputFile} does not exist`);
31
- const grammar = await promises_1.default.readFile(args.input, "utf8");
32
- const productionsResult = (0, index_1.tryBuildProductions)(grammar);
58
+ const outputFile = typeof args.output === "string" ? path_1.default.resolve(args.output) : undefined;
59
+ let grammar;
60
+ try {
61
+ grammar = await promises_1.default.readFile(inputFile, "utf8");
62
+ }
63
+ catch (_a) {
64
+ fail(`Cannot read grammar file ${inputFile}: it does not exist.`);
65
+ return;
66
+ }
67
+ const onWarning = (warning) => {
68
+ if (args.quiet === true)
69
+ return;
70
+ console.error(`Warning in ${inputFile} at ${warning.token.line}:${warning.token.column}: ${warning.reason}`);
71
+ };
72
+ const productionsResult = (0, index_1.tryBuildProductions)(grammar, { onWarning });
33
73
  if (productionsResult.success === false) {
34
- console.log(productionsResult.reason);
35
- console.log((0, errorWindowBuilder_1.buildErrorWindow)(grammar, productionsResult.token));
74
+ fail(`${productionsResult.reason}\n\n${(0, index_1.buildErrorWindow)(grammar, productionsResult.token)}`);
75
+ return;
76
+ }
77
+ const productions = productionsResult.value;
78
+ if (check) {
79
+ if (outputFile == null) {
80
+ fail(`--check needs --output to know which table to check.\n\n${USAGE}`);
81
+ return;
82
+ }
83
+ let existingTable;
84
+ try {
85
+ existingTable = await promises_1.default.readFile(outputFile, "utf8");
86
+ }
87
+ catch (_b) {
88
+ fail(`Cannot check ${outputFile}: it does not exist.`);
89
+ return;
90
+ }
91
+ const statesResult = (0, index_1.tryBuildStates)(existingTable, { productions });
92
+ if (statesResult.success === false) {
93
+ fail(`The existing table is not a valid parsing table:\n${statesResult.reason}`);
94
+ return;
95
+ }
96
+ const tableCheckResult = (0, index_1.validateTable)(productions, statesResult.value, { source: outputFile });
97
+ if (tableCheckResult.success === false)
98
+ fail(`The existing table does not match the grammar:\n${tableCheckResult.reason}`);
99
+ const expected = (0, index_1.generateStates)(productions, { mode: args.lalr === true ? "lalr1" : "lr1" });
100
+ if (expected.success === false) {
101
+ fail(`${expected.reason}\n\n${(0, index_1.buildErrorWindow)(grammar, expected.token)}`);
102
+ return;
103
+ }
104
+ const expectedTable = toTableOrFail(expected.value);
105
+ if (expectedTable == null)
106
+ return;
107
+ const actualTable = existingTable.trim();
108
+ if (expectedTable !== actualTable) {
109
+ fail(`${outputFile} is out of date with respect to ${inputFile}.\n` +
110
+ `Re-run without --check to regenerate it${args.lalr === true ? " (with --lalr)" : ""}.`);
111
+ return;
112
+ }
113
+ console.log(`${outputFile} is up to date with respect to ${inputFile} (${expectedTable.split("\n").length} states).`);
36
114
  return;
37
115
  }
38
- const generatorResult = (0, generator_1.generateStates)(productionsResult.value);
116
+ const generatorResult = (0, index_1.generateStates)(productions, { mode: args.lalr === true ? "lalr1" : "lr1" });
39
117
  if (generatorResult.success === false) {
40
- console.log(generatorResult.reason);
41
- console.log((0, errorWindowBuilder_1.buildErrorWindow)(grammar, generatorResult.token));
118
+ fail(`${generatorResult.reason}\n\n${(0, index_1.buildErrorWindow)(grammar, generatorResult.token)}`);
42
119
  return;
43
120
  }
44
- await promises_1.default.writeFile(outputFile, generatorResult.value.toTable(), { encoding: "utf-8" });
121
+ const table = toTableOrFail(generatorResult.value);
122
+ if (table == null)
123
+ return;
124
+ if (useStdout) {
125
+ console.log(table);
126
+ return;
127
+ }
128
+ if (outputFile == null)
129
+ return;
130
+ await promises_1.default.mkdir(path_1.default.dirname(outputFile), { recursive: true });
131
+ await promises_1.default.writeFile(outputFile, table, { encoding: "utf-8" });
132
+ const stateCount = table === "" ? 0 : table.split("\n").length;
133
+ console.log(`Wrote ${stateCount} ${args.lalr === true ? "LALR(1)" : "LR(1)"} states from ${inputFile} to ${outputFile}.`);
45
134
  }
46
135
  main();
47
136
  //# sourceMappingURL=cli.js.map