@scinorandex/sparse 0.1.1 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/workflows/ci.yml +41 -0
- package/AGENTS.md +31 -11
- package/README.md +283 -29
- package/dist/cli.js +113 -24
- package/dist/cli.js.map +1 -1
- package/dist/generator.d.ts +3 -1
- package/dist/generator.js +29 -5
- package/dist/generator.js.map +1 -1
- package/dist/index.d.ts +16 -4
- package/dist/index.js +34 -1
- package/dist/index.js.map +1 -1
- package/dist/meta/common.d.ts +17 -0
- package/dist/meta/common.js +66 -1
- package/dist/meta/common.js.map +1 -1
- package/dist/meta/selfhosted.d.ts +4 -5
- package/dist/meta/selfhosted.js +49 -36
- package/dist/meta/selfhosted.js.map +1 -1
- package/dist/parser.d.ts +56 -35
- package/dist/parser.js +146 -18
- package/dist/parser.js.map +1 -1
- package/dist/table/selfhosted.d.ts +10 -1
- package/dist/table/selfhosted.js +32 -12
- package/dist/table/selfhosted.js.map +1 -1
- package/dist/table/validate.d.ts +17 -0
- package/dist/table/validate.js +84 -0
- package/dist/table/validate.js.map +1 -0
- package/dist/utils/Stack.d.ts +4 -0
- package/dist/utils/Stack.js +16 -1
- package/dist/utils/Stack.js.map +1 -1
- package/dist/utils/errorWindowBuilder.js +11 -11
- package/dist/utils/errorWindowBuilder.js.map +1 -1
- package/dist/utils/loadFiles.d.ts +8 -0
- package/dist/utils/loadFiles.js +47 -0
- package/dist/utils/loadFiles.js.map +1 -0
- package/dist/utils/reducers.d.ts +9 -0
- package/dist/utils/reducers.js +56 -0
- package/dist/utils/reducers.js.map +1 -0
- package/example/LoLang/example.ts +16 -15
- package/example/kleene-test/example.ts +35 -17
- package/example/math/example.ts +12 -7
- package/example/selfhosted/example.ts +30 -24
- package/opencode.json +1 -1
- package/package.json +2 -2
- package/src/cli.ts +132 -23
- package/src/generator.ts +49 -8
- package/src/index.ts +55 -3
- package/src/meta/common.ts +116 -0
- package/src/meta/selfhosted.ts +42 -11
- package/src/parser.ts +259 -49
- package/src/table/selfhosted.ts +48 -12
- package/src/table/validate.ts +143 -0
- package/src/utils/Stack.ts +22 -2
- package/src/utils/errorWindowBuilder.ts +15 -13
- package/src/utils/loadFiles.ts +54 -0
- package/src/utils/reducers.ts +93 -0
- package/test/cli.test.ts +144 -0
- package/test/codegen.test.ts +75 -0
- package/test/grammar.test.ts +243 -0
- package/test/helpers.ts +59 -0
- package/test/lalr.test.ts +114 -0
- package/test/parser.test.ts +392 -0
- package/test/table.test.ts +144 -0
- package/tsconfig.json +1 -1
- package/example/LoLang/table2.txt +0 -497
- package/example/complicated/grammar.txt +0 -37
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
verify:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
|
|
12
|
+
steps:
|
|
13
|
+
- uses: actions/checkout@v4
|
|
14
|
+
|
|
15
|
+
- uses: actions/setup-node@v4
|
|
16
|
+
with:
|
|
17
|
+
node-version: 20
|
|
18
|
+
cache: yarn
|
|
19
|
+
|
|
20
|
+
- run: yarn install --frozen-lockfile
|
|
21
|
+
|
|
22
|
+
- name: Typecheck
|
|
23
|
+
run: yarn tc
|
|
24
|
+
|
|
25
|
+
- name: Test
|
|
26
|
+
run: yarn test
|
|
27
|
+
|
|
28
|
+
- name: Build
|
|
29
|
+
run: yarn build
|
|
30
|
+
|
|
31
|
+
- name: Run the examples
|
|
32
|
+
run: |
|
|
33
|
+
npm install -g tsx
|
|
34
|
+
npx tsx example/math/example.ts
|
|
35
|
+
npx tsx example/kleene-test/example.ts
|
|
36
|
+
npx tsx example/LoLang/example.ts
|
|
37
|
+
|
|
38
|
+
- name: The committed tables are up to date
|
|
39
|
+
run: |
|
|
40
|
+
node dist/cli.js --input=example/math/grammar.txt --output=example/math/table.txt --check
|
|
41
|
+
node dist/cli.js --input=example/LoLang/grammar.txt --output=example/LoLang/table.txt --check
|
package/AGENTS.md
CHANGED
|
@@ -1,14 +1,16 @@
|
|
|
1
1
|
# AGENTS.md
|
|
2
2
|
|
|
3
3
|
`@scinorandex/sparse` — LR(1) parser generator library (TypeScript, CJS via tsc, yarn 1).
|
|
4
|
-
Single package: `src/` = library + CLI, `example/` = runnable demos
|
|
4
|
+
Single package: `src/` = library + CLI, `example/` = runnable demos, `test/` = vitest suite.
|
|
5
|
+
README.md is the user-facing doc (grammar syntax, reducer contract, recovery guide, API reference).
|
|
5
6
|
|
|
6
7
|
## Commands
|
|
7
|
-
- Typecheck: `yarn tc` (covers src/ and
|
|
8
|
+
- Typecheck: `yarn tc` (covers src/, example/ and test/)
|
|
8
9
|
- Build: `yarn build` → `dist/` (CLI is `dist/cli.js`)
|
|
10
|
+
- Test: `yarn test` (vitest; builds `dist/` first because `test/cli.test.ts` runs the CLI)
|
|
9
11
|
- Run examples: `tsx example/<name>/example.ts` — tsx is global, not a repo dep; `node file.ts` fails (enums)
|
|
10
12
|
- Table CLI: `node dist/cli.js --input=<grammar> --output=<table>` (build first); output matches committed tables exactly
|
|
11
|
-
- `yarn
|
|
13
|
+
- `yarn bench` exists but there are no bench files. Verify with `yarn tc && yarn test` plus the examples.
|
|
12
14
|
- `dist/` and `yarn.lock` are gitignored.
|
|
13
15
|
|
|
14
16
|
## Self-hosting / codegen (critical, easy to break)
|
|
@@ -19,22 +21,40 @@ Single package: `src/` = library + CLI, `example/` = runnable demos. README.md i
|
|
|
19
21
|
`02-table-grammar.txt`, run `tsx example/selfhosted/example.ts`, then copy the regenerated
|
|
20
22
|
`codegen-*.ts` over the two `src/.../states.ts` files. The old (checked-in) parser must still
|
|
21
23
|
parse the modified grammar (bootstrap constraint).
|
|
24
|
+
- `test/codegen.test.ts` enforces this: it regenerates both state files in memory and compares them to
|
|
25
|
+
the checked-in ones. It compares objects (not file bytes) so the `cp` step stays manual.
|
|
26
|
+
- Changing the *lexer* rules in `grammarLexerGenerator` (e.g. allowing `-` inside identifiers) does NOT
|
|
27
|
+
change the generated states — the token types and positions stay the same.
|
|
22
28
|
|
|
23
29
|
## Non-obvious behavior
|
|
24
|
-
- `?`, `*`, `+` groups and
|
|
30
|
+
- `?`, `*`, `+` groups and `|` alternatives are unrolled into extra productions; `*`/`+`
|
|
25
31
|
create `autogen-N` productions whose reducer is named `autogenerated-kleene` — user reducers must
|
|
26
|
-
handle it (see `example/kleene-test/`).
|
|
32
|
+
handle it (see `example/kleene-test/`). `?` on a group that can match nothing produces an empty
|
|
33
|
+
production and is rejected by `validateProductions`.
|
|
34
|
+
- Modifiers only apply to parenthesized groups: `([A] [B])?`, not `[A]?`.
|
|
35
|
+
- `|` alternatives live inside a single token: `[A | B]`, `<A | B>`.
|
|
27
36
|
- Two construction paths: `Sparse.fromProductions(...)` builds states at runtime (slow, ~30s for
|
|
28
|
-
150 productions) vs `new Sparse({ productions, states
|
|
37
|
+
150 productions) vs `Sparse.fromGrammarFile(...)` / `new Sparse({ productions, states })` (fast).
|
|
29
38
|
- Reducers are keyed by production name (`<LHS: name>` in the grammar file); nameless productions
|
|
30
|
-
have `name: null
|
|
31
|
-
the parse loop breaks on it without calling the reducer.
|
|
39
|
+
have `name: null`. Production index 0 is the accept production —
|
|
40
|
+
the parse loop breaks on it without calling the reducer. `oldInput.index` is the pre-unrolling index.
|
|
32
41
|
- Table file format: one line per state, comma-separated `[TERMINAL]=sN|rN` and `<VARIABLE>=M`.
|
|
42
|
+
The reader preserves file order via `getItemsReversed()` (the item lists are built right-recursively).
|
|
33
43
|
- Errors flow as `Result<T>` (`{ success: false, reason, token }`); `buildProductions` /
|
|
34
|
-
`Sparse.fromProductions` are the throwing wrappers.
|
|
44
|
+
`Sparse.fromProductions` / `buildStates` are the throwing wrappers.
|
|
45
|
+
- `GeneratorResult.ActionTable` is sparse: a state whose items only move to other states has no entry.
|
|
46
|
+
`toStates()`/`toTable()` use `Array.from`, not `map`, so state numbering stays aligned; `toTable()`
|
|
47
|
+
throws for such grammars because the file format cannot express an action-less state.
|
|
48
|
+
- LALR(1) mode merges same-core LR(1) states. Same-core states always agree on their shifts, gotos and
|
|
49
|
+
reduce targets, so the conflict branch in `mergeStatesForLALR` is currently unreachable.
|
|
35
50
|
|
|
36
51
|
## Key files
|
|
37
52
|
- `src/generator.ts` — LR(1) state construction (first/follow sets, item-set expansion), `toTable()` serialization
|
|
38
|
-
- `src/parser.ts` — runtime parser loop + `recover` error-recovery callback API
|
|
53
|
+
- `src/parser.ts` — runtime parser loop + `recover` error-recovery callback API (`insertToken`, `finish`, `crash`)
|
|
39
54
|
- `src/meta/selfhosted.ts` — grammar-file parser + production unrolling
|
|
40
|
-
- `
|
|
55
|
+
- `src/meta/common.ts` — `Production` type + `validateProductions`
|
|
56
|
+
- `src/table/selfhosted.ts` — table-file parser (`tryBuildStates`), `src/table/validate.ts` — table validation
|
|
57
|
+
- `src/cli.ts` — `--input/--output/--lalr/--check/--stdout/--quiet/--help`
|
|
58
|
+
- `example/` — `math` (prebuilt table), `kleene-test` (`*`/`+`), `LoLang` (large grammar + recovery, exports
|
|
59
|
+
its lexer for the tests), `selfhosted` (codegen)
|
|
60
|
+
- `test/` — `grammar`, `table`, `parser`, `lalr`, `codegen`, `cli`
|
package/README.md
CHANGED
|
@@ -1,20 +1,33 @@
|
|
|
1
1
|
## Sparse - Scin's Parsing Library
|
|
2
2
|
|
|
3
|
-
Sparse allows developers to easily create
|
|
3
|
+
Sparse allows developers to easily create LR(1) parsers and LR(1) parsing tables.
|
|
4
4
|
|
|
5
5
|
**Features:**
|
|
6
6
|
- Complete Error Handling API - when parsing fails, you have the final say on where parsing stops
|
|
7
|
-
- Deferred Reductions - you create the nodes and Sparse builds the tree
|
|
7
|
+
- Deferred Reductions - you create the nodes and Sparse builds the tree
|
|
8
|
+
- Grammar and table validation - mistakes are reported with the line, column and a window into your file
|
|
8
9
|
- Full [Slex](https://github.com/scinscinscin/slex) integration - define your entire language as a set of DFA and CFG rules
|
|
9
10
|
- Optional Table Output - use Sparse to build your tables for use in other languages
|
|
10
11
|
|
|
11
|
-
##
|
|
12
|
+
## Table of contents
|
|
13
|
+
|
|
14
|
+
- [Getting started](#getting-started)
|
|
15
|
+
- [Grammar syntax reference](#grammar-syntax-reference)
|
|
16
|
+
- [Building a parser](#building-a-parser)
|
|
17
|
+
- [Reducers](#reducers)
|
|
18
|
+
- [Repetition with `*` and `+`](#repetition-with--and-)
|
|
19
|
+
- [Error recovery](#error-recovery)
|
|
20
|
+
- [Generating and shipping the parsing table](#generating-and-shipping-the-parsing-table)
|
|
21
|
+
- [API reference](#api-reference)
|
|
22
|
+
- [Known limitations](#known-limitations)
|
|
23
|
+
|
|
24
|
+
## Getting started
|
|
12
25
|
|
|
13
26
|
1. **Define the CFG of your language.**
|
|
14
27
|
|
|
15
|
-
The CFG is defined by creating a file containing a list of productions. Variables are identifiers encased in angle brackets like `<STATEMENT>`, while terminals are encased in square-brackets like `[L_COLON]`.
|
|
28
|
+
The CFG is defined by creating a file containing a list of productions. Variables are identifiers encased in angle brackets like `<STATEMENT>`, while terminals are encased in square-brackets like `[L_COLON]`.
|
|
16
29
|
|
|
17
|
-
An example production includes: `<IF_STATEMENT> : [IF] [L_PAREN] <EXPRESSION> [R_PAREN] <STATEMENT>
|
|
30
|
+
An example production includes: `<IF_STATEMENT> : [IF] [L_PAREN] <EXPRESSION> [R_PAREN] <STATEMENT> [ELSE] <STATEMENT>;`.
|
|
18
31
|
|
|
19
32
|
An example CFG for a basic MDAS calculator is the following:
|
|
20
33
|
|
|
@@ -22,16 +35,20 @@ An example CFG for a basic MDAS calculator is the following:
|
|
|
22
35
|
<S>: <PROGRAM>;
|
|
23
36
|
<PROGRAM>: <EXPRESSION> [EOF];
|
|
24
37
|
<PROGRAM>: [EOF];
|
|
38
|
+
|
|
39
|
+
// Begin parsing arithmetic expressions
|
|
25
40
|
<EXPRESSION>: <TERM_EXPRESSION>;
|
|
26
41
|
<TERM_EXPRESSION>: <FACTOR_EXPRESSION> [PLUS] <TERM_EXPRESSION>;
|
|
27
42
|
<TERM_EXPRESSION>: <FACTOR_EXPRESSION> [MINUS] <TERM_EXPRESSION>;
|
|
28
43
|
<TERM_EXPRESSION>: <FACTOR_EXPRESSION>;
|
|
29
44
|
<FACTOR_EXPRESSION>: <ENDPOINT> [STAR] <FACTOR_EXPRESSION>;
|
|
30
|
-
<FACTOR_EXPRESSION>: <ENDPOINT> [
|
|
45
|
+
<FACTOR_EXPRESSION>: <ENDPOINT> [FORWARD_SLASH] <FACTOR_EXPRESSION>;
|
|
31
46
|
<FACTOR_EXPRESSION>: <ENDPOINT>;
|
|
32
47
|
<ENDPOINT>: [NUMBER];
|
|
33
48
|
```
|
|
34
49
|
|
|
50
|
+
The first production has to be the "accept" production: a variable that appears nowhere else, whose body is a single variable. It is the only production whose reducer Sparse never calls.
|
|
51
|
+
|
|
35
52
|
2. **Create your [Slex](https://github.com/scinscinscin/slex) lexer.**
|
|
36
53
|
|
|
37
54
|
```ts
|
|
@@ -61,7 +78,7 @@ lexerGenerator.addRule("number_literal", "${float_number}|${decimal_number}", To
|
|
|
61
78
|
const lexer = lexerGenerator.generate(`2.4 + 3.5 * 1 / 456.789`, () => ({}));
|
|
62
79
|
```
|
|
63
80
|
|
|
64
|
-
3. **Define the node representation.**
|
|
81
|
+
3. **Define the node representation.**
|
|
65
82
|
|
|
66
83
|
Sparse allows you to build the AST however you want, deferring to your functions when its time to make a reduction, giving you control over the representation.
|
|
67
84
|
|
|
@@ -71,7 +88,7 @@ type StringifiedNode = (StringifiedNode | string)[];
|
|
|
71
88
|
// It doesn't have to be a class. It just has to be a structure that all nodes
|
|
72
89
|
// in the AST adhere to. Classes allow this to be done easily through subclassing.
|
|
73
90
|
class Node {
|
|
74
|
-
constructor(public readonly nodes: LR1StackSymbol<TokenType,
|
|
91
|
+
constructor(public readonly nodes: LR1StackSymbol<TokenType, Metadata, Node>[]) {}
|
|
75
92
|
toObject(): StringifiedNode {
|
|
76
93
|
return this.nodes.map((node) => (node.type === "token"
|
|
77
94
|
? node.token.lexeme
|
|
@@ -81,53 +98,290 @@ class Node {
|
|
|
81
98
|
}
|
|
82
99
|
```
|
|
83
100
|
|
|
84
|
-
4. **Building the
|
|
101
|
+
4. **Building the parser.**
|
|
85
102
|
|
|
86
|
-
|
|
103
|
+
With a prebuilt parsing table, `Sparse.fromGrammarFile` reads the grammar, reads the table, and checks that the two belong together:
|
|
87
104
|
|
|
88
105
|
```ts
|
|
106
|
+
import { Sparse, enumToString } from "@scinorandex/sparse";
|
|
107
|
+
|
|
108
|
+
async function main() {
|
|
109
|
+
const parserGenerator = await Sparse.fromGrammarFile<TokenType, Metadata, Node>({
|
|
110
|
+
grammarPath: "./example/math/grammar.txt",
|
|
111
|
+
tablePath: "./example/math/table.txt",
|
|
112
|
+
toStringifiedTokenType: enumToString<TokenType>(TokenType),
|
|
113
|
+
onWarning: ({ reason, token }) => console.warn(`${token.line}:${token.column} ${reason}`),
|
|
114
|
+
});
|
|
115
|
+
|
|
116
|
+
const parser = parserGenerator.generate(lexer, {
|
|
117
|
+
reducer: (_, { input }) => new Node(input),
|
|
118
|
+
});
|
|
119
|
+
|
|
120
|
+
console.log(parser.parse().result!.toObject());
|
|
121
|
+
}
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
Generating the table at startup instead (fine for small grammars, slow for big ones):
|
|
125
|
+
|
|
126
|
+
```ts
|
|
127
|
+
import { buildProductions, Sparse, enumToString } from "@scinorandex/sparse";
|
|
128
|
+
|
|
89
129
|
async function main() {
|
|
90
|
-
const toStringifiedTokenType = (type: TokenType) => TokenType[type];
|
|
91
130
|
const productions = buildProductions(await fs.readFile("./example/math/grammar.txt", "utf8"));
|
|
92
|
-
const parserGenerator = Sparse.fromProductions<TokenType, Metadata, Node>({
|
|
131
|
+
const parserGenerator = Sparse.fromProductions<TokenType, Metadata, Node>({
|
|
132
|
+
productions,
|
|
133
|
+
toStringifiedTokenType: enumToString<TokenType>(TokenType),
|
|
134
|
+
});
|
|
93
135
|
}
|
|
94
136
|
```
|
|
95
137
|
|
|
96
|
-
|
|
138
|
+
## Grammar syntax reference
|
|
139
|
+
|
|
140
|
+
A grammar file is a list of productions. Whitespace is insignificant, `//` starts a line comment, and `/* ... */` spans lines.
|
|
141
|
+
|
|
142
|
+
| Syntax | Meaning |
|
|
143
|
+
| --- | --- |
|
|
144
|
+
| `<A>: <B> [C];` | `A` is made of one `B` followed by one `C` |
|
|
145
|
+
| `<A: name>: ...;` | The production is *named*: its reducer is looked up under `name` |
|
|
146
|
+
| `[TOK: name]` | Names the terminal on the right hand side so it shows up on the reducer's `bag` |
|
|
147
|
+
| `<VAR: name>` | Same, for a variable |
|
|
148
|
+
| `[A | B]` | Alternatives: unrolled into one production per alternative |
|
|
149
|
+
| `<A | B>` | Same, for variables |
|
|
150
|
+
| `([A] [B])?` | Optional group, unrolled into one production with and one without it |
|
|
151
|
+
| `// comment`, `/* comment */` | Comments |
|
|
152
|
+
|
|
153
|
+
Identifiers may contain letters, digits, `_` and `-`, and must start with a letter or `_`.
|
|
154
|
+
|
|
155
|
+
### What Sparse checks for you
|
|
156
|
+
|
|
157
|
+
Grammar and table problems are reported as a `Result` failure with the offending token, so the CLI and
|
|
158
|
+
`loadGrammar` can print a window into your file:
|
|
159
|
+
|
|
160
|
+
- a variable used on the right hand side that no production defines
|
|
161
|
+
- an empty right hand side (a production like `<A: a>: ([X])?;` expands to nothing, and Sparse has no support for empty productions)
|
|
162
|
+
- the start symbol appearing on the right hand side
|
|
163
|
+
- two symbols in one production sharing a name (warns: only the last one reaches the `bag`)
|
|
164
|
+
- a production that is never reachable from the start symbol (warns)
|
|
165
|
+
|
|
166
|
+
```
|
|
167
|
+
$ npx sparse --input=broken.txt --output=table.txt
|
|
168
|
+
Invalid syntax at 2:30: got SEMICOLON (";"), but expected one of [L_ANGLE], [L_BRACKET], [L_PAREN]
|
|
169
|
+
|
|
170
|
+
2 | <PROGRAM: program>: [NUMBER] (;
|
|
171
|
+
| ~
|
|
172
|
+
3 |
|
|
173
|
+
```
|
|
97
174
|
|
|
98
|
-
|
|
175
|
+
## Reducers
|
|
176
|
+
|
|
177
|
+
Reducers are keyed by the name in the grammar, and receive two arguments:
|
|
99
178
|
|
|
100
179
|
```ts
|
|
180
|
+
type Reducer<TokenType, Metadata, Node> = (
|
|
181
|
+
newInput: { bag: Record<string, Node | Token<TokenType, Metadata>>; name: string | null },
|
|
182
|
+
oldInput: { input: LR1StackSymbol<TokenType, Metadata, Node>[]; index: number },
|
|
183
|
+
) => Node;
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
- `newInput.name` is the production's name, or `null` for an unnamed production.
|
|
187
|
+
- `newInput.bag` holds the right hand side symbols that were named in the grammar, keyed by those names. If two symbols share a name, the last one wins, so reach for `oldInput.input` when that happens.
|
|
188
|
+
- `oldInput.input` is the reduced symbols in order, as `{ type: "token", token }` or `{ type: "node", node }`.
|
|
189
|
+
- `oldInput.index` is the production's index *in the grammar file*, before `*`/`?`/`+`/`|` unrolling. Index 0 is the accept production, whose reducer is never called.
|
|
190
|
+
|
|
191
|
+
`defineReducers` turns a map of reducers into the reducer that `generate` wants, and tells you exactly which production has no reducer:
|
|
192
|
+
|
|
193
|
+
```ts
|
|
194
|
+
import { assertReducersCoverGrammar, defineReducers } from "@scinorandex/sparse";
|
|
195
|
+
|
|
196
|
+
const reducers = {
|
|
197
|
+
program: ({ bag }) => new Node(bag),
|
|
198
|
+
expression: ({ bag }, { input }) => new Node(input),
|
|
199
|
+
};
|
|
200
|
+
|
|
201
|
+
assertReducersCoverGrammar(productions, reducers); // throws if something is missing
|
|
202
|
+
|
|
101
203
|
const parser = parserGenerator.generate(lexer, {
|
|
102
|
-
reducer: (
|
|
204
|
+
reducer: defineReducers(reducers),
|
|
103
205
|
});
|
|
206
|
+
```
|
|
104
207
|
|
|
105
|
-
|
|
208
|
+
Without the helper the failure looks like this, halfway through a parse:
|
|
209
|
+
|
|
210
|
+
```
|
|
211
|
+
Error while performing the reduction for production 2 ("expression"): I cannot reduce a null bag
|
|
106
212
|
```
|
|
107
213
|
|
|
108
|
-
##
|
|
214
|
+
## Repetition with `*` and `+`
|
|
109
215
|
|
|
110
|
-
|
|
216
|
+
`[A]?`, `A B*`, and `A B+` are unrolled into extra productions. For `*` and `+` that means a new
|
|
217
|
+
production named **`autogenerated-kleene`** (with variables called `autogen-0`, `autogen-1`, ...), which
|
|
218
|
+
you **have** to implement: it is handed the items matched so far, and it has to flatten them into a list.
|
|
219
|
+
The first reduction has no `rest`, later ones do:
|
|
111
220
|
|
|
112
|
-
|
|
221
|
+
```ts
|
|
222
|
+
class KleeneNode<T> extends Node {
|
|
223
|
+
contents: T[] = [];
|
|
113
224
|
|
|
114
|
-
|
|
225
|
+
constructor(nodes: LR1StackSymbol<TokenType, Metadata, Node>[], bag: T) {
|
|
226
|
+
super(nodes);
|
|
227
|
+
this.contents.unshift(bag);
|
|
228
|
+
}
|
|
115
229
|
|
|
116
|
-
|
|
230
|
+
add(nodes: LR1StackSymbol<TokenType, Metadata, Node>[], bag: T) {
|
|
231
|
+
this.nodes.unshift(...nodes.slice(0, nodes.length - 1));
|
|
232
|
+
this.contents.unshift(bag);
|
|
233
|
+
return this;
|
|
234
|
+
}
|
|
235
|
+
}
|
|
117
236
|
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
237
|
+
const reducers = {
|
|
238
|
+
"autogenerated-kleene": ({ bag: { rest, ...item } }, { input }) =>
|
|
239
|
+
rest == null ? new KleeneNode(input, item) : rest.add(input, item),
|
|
240
|
+
program: (_, { input }) => new Node(input),
|
|
241
|
+
};
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
See `example/kleene-test` for the whole thing.
|
|
122
245
|
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
246
|
+
## Error recovery
|
|
247
|
+
|
|
248
|
+
By default a syntax error throws `LR1ParserGraveError`, which carries the token that could not be parsed:
|
|
249
|
+
|
|
250
|
+
```ts
|
|
251
|
+
try {
|
|
252
|
+
parser.parse();
|
|
253
|
+
} catch (err) {
|
|
254
|
+
if (err instanceof LR1ParserGraveError) console.error(err.reason, err.currentToken);
|
|
126
255
|
}
|
|
127
256
|
```
|
|
128
257
|
|
|
258
|
+
Pass a `recover` function to decide what happens instead. It receives the lexer, both stacks, the table,
|
|
259
|
+
and these helpers:
|
|
260
|
+
|
|
261
|
+
| Helper | What it does |
|
|
262
|
+
| --- | --- |
|
|
263
|
+
| `addError(reason)` | Records an error, keeps parsing, and returns it in `parse().errors` |
|
|
264
|
+
| `crash(reason)` | Stops parsing and throws a `LR1ParserGraveError` |
|
|
265
|
+
| `finish(options?)` | Stops recovery and carries on parsing from the current state |
|
|
266
|
+
| `insertToken(token)` | Shifts a token the input did not have, e.g. a synthesized `;` |
|
|
267
|
+
| `isSafe()` | True when the current state can consume the next real token |
|
|
268
|
+
|
|
269
|
+
```ts
|
|
270
|
+
const parser = parserGenerator.generate(lexer, {
|
|
271
|
+
reducer: (_, { input }) => new Node(input),
|
|
272
|
+
recover({ lexer, states, statesStack, insertToken, addError, isSafe, finish, crash }) {
|
|
273
|
+
const token = lexer.peekNextToken();
|
|
274
|
+
|
|
275
|
+
// skip the extra semicolons the author left behind
|
|
276
|
+
while (token.type === TokenType.SEMICOLON) {
|
|
277
|
+
lexer.getNextToken();
|
|
278
|
+
if (isSafe()) return finish();
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
// or insert the semicolon they forgot
|
|
282
|
+
if (states[statesStack.peek()].getTerminalAction("SEMICOLON") != null) {
|
|
283
|
+
const semicolon = new Token(TokenType.SEMICOLON, ";", new ColumnAndRow(token.line, token.column), {});
|
|
284
|
+
const inserted = insertToken(semicolon);
|
|
285
|
+
if (inserted != null) return crash(inserted.reason);
|
|
286
|
+
addError(`Expected SEMICOLON but received ${TokenType[token.type]}`);
|
|
287
|
+
if (isSafe()) return finish();
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
return crash("Invalid syntax");
|
|
291
|
+
},
|
|
292
|
+
});
|
|
293
|
+
|
|
294
|
+
const { result, errors } = parser.parse();
|
|
295
|
+
```
|
|
296
|
+
|
|
297
|
+
A parser consumes its lexer, so build a new one per input with `generate(...)`, or call `parser.reset()`
|
|
298
|
+
if you want to reuse the parser with a lexer that still has tokens left.
|
|
299
|
+
|
|
300
|
+
## Generating and shipping the parsing table
|
|
301
|
+
|
|
302
|
+
Generating states for a large grammar takes a while (a few seconds for a few hundred productions), so
|
|
303
|
+
generate the table once at build time and commit it:
|
|
304
|
+
|
|
305
|
+
```
|
|
306
|
+
npx sparse --input=grammar.txt --output=table.txt
|
|
307
|
+
```
|
|
308
|
+
|
|
309
|
+
| Flag | Meaning |
|
|
310
|
+
| --- | --- |
|
|
311
|
+
| `--input=<file>` | Grammar to read (required) |
|
|
312
|
+
| `--output=<file>` | Table to write; missing directories are created |
|
|
313
|
+
| `--lalr` | Generate LALR(1) states instead of LR(1) |
|
|
314
|
+
| `--check` | Write nothing; verify that `<output>` already matches the grammar |
|
|
315
|
+
| `--stdout` | Print the table instead of writing it |
|
|
316
|
+
| `--quiet` | Hide warnings about suspicious rules |
|
|
317
|
+
| `--help` | Usage |
|
|
318
|
+
|
|
319
|
+
By default, LR(1) states are generated. You can generate LALR(1) states instead by passing the `--lalr`
|
|
320
|
+
flag (or setting `mode: "lalr1"` in `Sparse.fromProductions`): the resulting table is never larger than
|
|
321
|
+
the LR(1) one. Using LALR(1) yields a ~35% performance boost over LR(1) for the same grammar (tested on
|
|
322
|
+
the LoLang example).
|
|
323
|
+
|
|
324
|
+
`--check` is useful in CI to prove that a committed table still matches its grammar:
|
|
325
|
+
|
|
326
|
+
```
|
|
327
|
+
npx sparse --input=example/math/grammar.txt --output=example/math/table.txt --check
|
|
328
|
+
```
|
|
329
|
+
|
|
330
|
+
Then load the table with `Sparse.fromGrammarFile`, as shown in [Getting started](#getting-started).
|
|
331
|
+
|
|
332
|
+
## API reference
|
|
333
|
+
|
|
334
|
+
### Reading and writing grammars
|
|
335
|
+
|
|
336
|
+
| Function | Returns |
|
|
337
|
+
| --- | --- |
|
|
338
|
+
| `buildProductions(source)` / `tryBuildProductions(source)` | The unrolled productions. Throws / returns a `Result` failure with the offending token |
|
|
339
|
+
| `loadGrammar(path)` | Same, but reads the file, turning missing files into failures too |
|
|
340
|
+
| `buildStates(table)` / `tryBuildStates(table)` | `TableState[]` from a table file, validating action syntax and state numbers |
|
|
341
|
+
| `loadTable(path)` | Same, but reads the file |
|
|
342
|
+
| `generateStates(productions, options?)` | `GeneratorResult`; `options` is `{ mode, onWarning, onProgress }` |
|
|
343
|
+
| `validateProductions(productions, options?)` | Fails on a grammar that cannot generate a correct table |
|
|
344
|
+
| `validateTable(productions, states)` | Fails when a table does not belong to the productions |
|
|
345
|
+
| `validateTableStates(states, options?)` | Fails on a malformed table |
|
|
346
|
+
|
|
347
|
+
### Building parsers
|
|
348
|
+
|
|
349
|
+
| Function | Returns |
|
|
350
|
+
| --- | --- |
|
|
351
|
+
| `Sparse.fromGrammarFile({ grammarPath, tablePath, toStringifiedTokenType, onWarning?, validate? })` | Loads a grammar and its prebuilt table, cross-checking them |
|
|
352
|
+
| `Sparse.fromProductions({ productions, toStringifiedTokenType, mode?, quiet? })` | Generates states at startup |
|
|
353
|
+
| `Sparse.tryFromProductions(...)` | Same, as a `Result` |
|
|
354
|
+
| `new Sparse({ productions, states, toStringifiedTokenType, validate?, source? })` | Use when you already hold the states |
|
|
355
|
+
| `generator.generate(lexer, { reducer, recover? })` | A parser |
|
|
356
|
+
| `parser.parse()` | `{ result, errors }` |
|
|
357
|
+
| `parser.reset()` | Clears the stacks and errors |
|
|
358
|
+
|
|
359
|
+
### Helpers
|
|
360
|
+
|
|
361
|
+
| Export | Purpose |
|
|
362
|
+
| --- | --- |
|
|
363
|
+
| `enumToString(TokenType)` | The `toStringifiedTokenType` every example used to write by hand |
|
|
364
|
+
| `defineReducers(reducers)` | Turns a map of named reducers into a reducer |
|
|
365
|
+
| `assertReducersCoverGrammar(productions, reducers)` | Fails up front when a named production has no reducer |
|
|
366
|
+
| `missingReducerNames(productions, reducers)` / `missingReducerMessage(...)` | Same, as data |
|
|
367
|
+
| `namedProductions(productions)` | Every production the grammar names |
|
|
368
|
+
| `TableState.fromJSObject(json)` / `GeneratorResult.toJSObject()` | Round trip a table through JSON |
|
|
369
|
+
| `hydrateProduction(json)` / `dehydateProduction(production)` | Round trip productions through JSON |
|
|
370
|
+
| `buildErrorWindow(source, token)` | Renders the window around a token |
|
|
371
|
+
| `Stack` | `push`, `pop`, `peek`, `peekAt`, `size`, `isEmpty`, `toArray` |
|
|
372
|
+
|
|
373
|
+
## Known limitations
|
|
374
|
+
|
|
375
|
+
- `?`, `*` and `+` only apply to a parenthesized group, and a group that can match nothing
|
|
376
|
+
(`<A: a>: ([X])?;`) is rejected because Sparse has no support for empty productions. Write two
|
|
377
|
+
productions instead.
|
|
378
|
+
- Sparse does not detect grammar conflicts (shift/reduce, reduce/reduce). An ambiguous grammar produces a
|
|
379
|
+
table with one arbitrary resolution, so if your parser behaves strangely, check your grammar for ambiguity.
|
|
380
|
+
- A state with no actions cannot be written in the table file format; `toTable()` throws for such grammars.
|
|
381
|
+
Use `Sparse.fromProductions` to keep the states in memory instead.
|
|
382
|
+
|
|
129
383
|
## AI Disclaimer
|
|
130
384
|
|
|
131
385
|
This project was originally written without the use of AI tools, the core LR(1) table generator was written by hand as per the algorithms described in the Dragon Book. Every release prior to v0.1 contained no AI generated code.
|
|
132
386
|
|
|
133
|
-
OpenCode and Qwen 3.8 27B were to implement performance improvements on the original LR(1) table generator and to implement LALR(1) support.
|
|
387
|
+
OpenCode and Qwen 3.8 27B were to implement performance improvements on the original LR(1) table generator and to implement LALR(1) support.
|
package/dist/cli.js
CHANGED
|
@@ -5,43 +5,132 @@ var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
|
5
5
|
};
|
|
6
6
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
7
7
|
const promises_1 = __importDefault(require("fs/promises"));
|
|
8
|
-
const index_1 = require("./index");
|
|
9
|
-
const generator_1 = require("./generator");
|
|
10
|
-
const minimist_1 = __importDefault(require("minimist"));
|
|
11
8
|
const path_1 = __importDefault(require("path"));
|
|
12
|
-
const
|
|
9
|
+
const minimist_1 = __importDefault(require("minimist"));
|
|
10
|
+
const index_1 = require("./index");
|
|
13
11
|
const args = (0, minimist_1.default)(process.argv.slice(2));
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
12
|
+
const BINARY_NAME = "@scinorandex/sparse";
|
|
13
|
+
const USAGE = `Usage: npx @scinorandex/sparse --input=<grammar> --output=<table> [options]
|
|
14
|
+
|
|
15
|
+
Generates an LR(1) parsing table from a grammar file.
|
|
16
|
+
|
|
17
|
+
Required:
|
|
18
|
+
--input=<file> Grammar file to read
|
|
19
|
+
--output=<file> Where to write the parsing table (created if the directory is missing)
|
|
20
|
+
|
|
21
|
+
Options:
|
|
22
|
+
--lalr Generate LALR(1) states instead of LR(1) (smaller table, ~35% faster parsing)
|
|
23
|
+
--check Do not write anything: verify that <output> already matches the grammar
|
|
24
|
+
--stdout Print the table instead of writing it to a file
|
|
25
|
+
--quiet Do not print warnings about suspicious grammar rules
|
|
26
|
+
--help Show this message
|
|
27
|
+
|
|
28
|
+
Run with npx ${BINARY_NAME} (the package is @scinorandex/sparse).`;
|
|
29
|
+
function fail(message) {
|
|
30
|
+
console.error(message);
|
|
31
|
+
process.exitCode = 1;
|
|
32
|
+
}
|
|
33
|
+
function toTableOrFail(result) {
|
|
34
|
+
try {
|
|
35
|
+
return result.toTable();
|
|
36
|
+
}
|
|
37
|
+
catch (err) {
|
|
38
|
+
fail(err instanceof Error ? err.message : String(err));
|
|
39
|
+
return null;
|
|
40
|
+
}
|
|
20
41
|
}
|
|
21
42
|
async function main() {
|
|
22
|
-
if (
|
|
23
|
-
console.log(
|
|
43
|
+
if (args.help === true || args.h === true) {
|
|
44
|
+
console.log(USAGE);
|
|
45
|
+
return;
|
|
46
|
+
}
|
|
47
|
+
if (typeof args.input !== "string") {
|
|
48
|
+
fail(`Missing --input.\n\n${USAGE}`);
|
|
49
|
+
return;
|
|
50
|
+
}
|
|
51
|
+
const useStdout = args.stdout === true;
|
|
52
|
+
const check = args.check === true;
|
|
53
|
+
if (typeof args.output !== "string" && !useStdout) {
|
|
54
|
+
fail(`Missing --output.\n\n${USAGE}`);
|
|
24
55
|
return;
|
|
25
56
|
}
|
|
26
57
|
const inputFile = path_1.default.resolve(args.input);
|
|
27
|
-
const outputFile = path_1.default.resolve(args.output);
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
58
|
+
const outputFile = typeof args.output === "string" ? path_1.default.resolve(args.output) : undefined;
|
|
59
|
+
let grammar;
|
|
60
|
+
try {
|
|
61
|
+
grammar = await promises_1.default.readFile(inputFile, "utf8");
|
|
62
|
+
}
|
|
63
|
+
catch (_a) {
|
|
64
|
+
fail(`Cannot read grammar file ${inputFile}: it does not exist.`);
|
|
65
|
+
return;
|
|
66
|
+
}
|
|
67
|
+
const onWarning = (warning) => {
|
|
68
|
+
if (args.quiet === true)
|
|
69
|
+
return;
|
|
70
|
+
console.error(`Warning in ${inputFile} at ${warning.token.line}:${warning.token.column}: ${warning.reason}`);
|
|
71
|
+
};
|
|
72
|
+
const productionsResult = (0, index_1.tryBuildProductions)(grammar, { onWarning });
|
|
33
73
|
if (productionsResult.success === false) {
|
|
34
|
-
|
|
35
|
-
|
|
74
|
+
fail(`${productionsResult.reason}\n\n${(0, index_1.buildErrorWindow)(grammar, productionsResult.token)}`);
|
|
75
|
+
return;
|
|
76
|
+
}
|
|
77
|
+
const productions = productionsResult.value;
|
|
78
|
+
if (check) {
|
|
79
|
+
if (outputFile == null) {
|
|
80
|
+
fail(`--check needs --output to know which table to check.\n\n${USAGE}`);
|
|
81
|
+
return;
|
|
82
|
+
}
|
|
83
|
+
let existingTable;
|
|
84
|
+
try {
|
|
85
|
+
existingTable = await promises_1.default.readFile(outputFile, "utf8");
|
|
86
|
+
}
|
|
87
|
+
catch (_b) {
|
|
88
|
+
fail(`Cannot check ${outputFile}: it does not exist.`);
|
|
89
|
+
return;
|
|
90
|
+
}
|
|
91
|
+
const statesResult = (0, index_1.tryBuildStates)(existingTable, { productions });
|
|
92
|
+
if (statesResult.success === false) {
|
|
93
|
+
fail(`The existing table is not a valid parsing table:\n${statesResult.reason}`);
|
|
94
|
+
return;
|
|
95
|
+
}
|
|
96
|
+
const tableCheckResult = (0, index_1.validateTable)(productions, statesResult.value, { source: outputFile });
|
|
97
|
+
if (tableCheckResult.success === false)
|
|
98
|
+
fail(`The existing table does not match the grammar:\n${tableCheckResult.reason}`);
|
|
99
|
+
const expected = (0, index_1.generateStates)(productions, { mode: args.lalr === true ? "lalr1" : "lr1" });
|
|
100
|
+
if (expected.success === false) {
|
|
101
|
+
fail(`${expected.reason}\n\n${(0, index_1.buildErrorWindow)(grammar, expected.token)}`);
|
|
102
|
+
return;
|
|
103
|
+
}
|
|
104
|
+
const expectedTable = toTableOrFail(expected.value);
|
|
105
|
+
if (expectedTable == null)
|
|
106
|
+
return;
|
|
107
|
+
const actualTable = existingTable.trim();
|
|
108
|
+
if (expectedTable !== actualTable) {
|
|
109
|
+
fail(`${outputFile} is out of date with respect to ${inputFile}.\n` +
|
|
110
|
+
`Re-run without --check to regenerate it${args.lalr === true ? " (with --lalr)" : ""}.`);
|
|
111
|
+
return;
|
|
112
|
+
}
|
|
113
|
+
console.log(`${outputFile} is up to date with respect to ${inputFile} (${expectedTable.split("\n").length} states).`);
|
|
36
114
|
return;
|
|
37
115
|
}
|
|
38
|
-
const generatorResult = (0,
|
|
116
|
+
const generatorResult = (0, index_1.generateStates)(productions, { mode: args.lalr === true ? "lalr1" : "lr1" });
|
|
39
117
|
if (generatorResult.success === false) {
|
|
40
|
-
|
|
41
|
-
console.log((0, errorWindowBuilder_1.buildErrorWindow)(grammar, generatorResult.token));
|
|
118
|
+
fail(`${generatorResult.reason}\n\n${(0, index_1.buildErrorWindow)(grammar, generatorResult.token)}`);
|
|
42
119
|
return;
|
|
43
120
|
}
|
|
44
|
-
|
|
121
|
+
const table = toTableOrFail(generatorResult.value);
|
|
122
|
+
if (table == null)
|
|
123
|
+
return;
|
|
124
|
+
if (useStdout) {
|
|
125
|
+
console.log(table);
|
|
126
|
+
return;
|
|
127
|
+
}
|
|
128
|
+
if (outputFile == null)
|
|
129
|
+
return;
|
|
130
|
+
await promises_1.default.mkdir(path_1.default.dirname(outputFile), { recursive: true });
|
|
131
|
+
await promises_1.default.writeFile(outputFile, table, { encoding: "utf-8" });
|
|
132
|
+
const stateCount = table === "" ? 0 : table.split("\n").length;
|
|
133
|
+
console.log(`Wrote ${stateCount} ${args.lalr === true ? "LALR(1)" : "LR(1)"} states from ${inputFile} to ${outputFile}.`);
|
|
45
134
|
}
|
|
46
135
|
main();
|
|
47
136
|
//# sourceMappingURL=cli.js.map
|