@scinorandex/sparse 0.0.2 → 0.0.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/README.md +2 -2
  2. package/dist/cli.js +2 -2
  3. package/dist/cli.js.map +1 -1
  4. package/dist/generator.d.ts +4 -1
  5. package/dist/generator.js +47 -32
  6. package/dist/generator.js.map +1 -1
  7. package/dist/index.d.ts +3 -2
  8. package/dist/index.js +5 -5
  9. package/dist/index.js.map +1 -1
  10. package/dist/meta/common.d.ts +57 -0
  11. package/dist/meta/common.js +62 -0
  12. package/dist/meta/common.js.map +1 -0
  13. package/dist/meta/handwritten.d.ts +12 -0
  14. package/dist/{grammarParser.js → meta/handwritten.js} +24 -24
  15. package/dist/meta/handwritten.js.map +1 -0
  16. package/dist/meta/selfhosted.d.ts +23 -0
  17. package/dist/meta/selfhosted.js +206 -0
  18. package/dist/meta/selfhosted.js.map +1 -0
  19. package/dist/meta/states.d.ts +1580 -0
  20. package/dist/meta/states.js +1493 -0
  21. package/dist/meta/states.js.map +1 -0
  22. package/dist/parser.d.ts +16 -5
  23. package/dist/parser.js +25 -7
  24. package/dist/parser.js.map +1 -1
  25. package/dist/{tableParser.d.ts → table/handwritten.d.ts} +1 -1
  26. package/dist/{tableParser.js → table/handwritten.js} +2 -2
  27. package/dist/table/handwritten.js.map +1 -0
  28. package/dist/table/selfhosted.d.ts +2 -0
  29. package/dist/table/selfhosted.js +65 -0
  30. package/dist/table/selfhosted.js.map +1 -0
  31. package/dist/table/states.d.ts +592 -0
  32. package/dist/table/states.js +593 -0
  33. package/dist/table/states.js.map +1 -0
  34. package/dist/utils/Result.d.ts +1 -1
  35. package/dist/utils/errorWindowBuilder.d.ts +1 -1
  36. package/example/LoLang/example.ts +14 -7
  37. package/example/complicated/grammar.txt +37 -0
  38. package/example/math/example.ts +1 -1
  39. package/example/selfhosted/01-meta-grammar.txt +10 -0
  40. package/example/selfhosted/02-table-grammar.txt +6 -0
  41. package/example/selfhosted/codegen-meta.ts +1489 -0
  42. package/example/selfhosted/codegen-table.ts +589 -0
  43. package/example/selfhosted/example.ts +39 -0
  44. package/package.json +3 -3
  45. package/src/cli.ts +1 -1
  46. package/src/generator.ts +52 -34
  47. package/src/index.ts +3 -2
  48. package/src/meta/common.ts +74 -0
  49. package/src/{grammarParser.ts → meta/handwritten.ts} +17 -21
  50. package/src/meta/selfhosted.ts +272 -0
  51. package/src/meta/states.ts +1489 -0
  52. package/src/parser.ts +53 -19
  53. package/src/{tableParser.ts → table/handwritten.ts} +5 -5
  54. package/src/table/selfhosted.ts +67 -0
  55. package/src/table/states.ts +589 -0
  56. package/src/utils/Result.ts +1 -1
  57. package/src/utils/errorWindowBuilder.ts +1 -1
  58. package/tsconfig.build.json +30 -0
  59. package/tsconfig.json +6 -6
  60. package/dist/grammarParser.d.ts +0 -27
  61. package/dist/grammarParser.js.map +0 -1
  62. package/dist/tableParser.js.map +0 -1
package/src/generator.ts CHANGED
@@ -1,5 +1,5 @@
1
- import { GrammarToken, Production } from "./grammarParser";
2
1
  import { TableState } from "./parser";
2
+ import { GrammarToken, Production } from "./meta/common";
3
3
  import { Result } from "./utils/Result";
4
4
 
5
5
  const xContainsAllOfY = <T>(xs: Set<T>, ys: Set<T>) => [...ys].every((x) => xs.has(x));
@@ -9,7 +9,7 @@ const EOF_STRING = "[EOF]";
9
9
  function computeFirstSets(allProductions: Production[]): Result<Map<string, Set<string>>> {
10
10
  // Initialize the hashmap
11
11
  const ret = new Map<string, Set<string>>();
12
- for (const production of allProductions) ret.set(production.lhs.lexeme, new Set());
12
+ for (const production of allProductions) ret.set(production.identifier, new Set());
13
13
 
14
14
  // Keep iterating through all productions until no edits are made
15
15
  let wasEdited = true;
@@ -17,16 +17,16 @@ function computeFirstSets(allProductions: Production[]): Result<Map<string, Set<
17
17
  wasEdited = false;
18
18
 
19
19
  for (const production of allProductions) {
20
- const toBeModified = ret.get(production.lhs.lexeme)!;
20
+ const toBeModified = ret.get(production.identifier)!;
21
21
 
22
22
  const firstRhsToken = production.rhs[0];
23
23
  if (firstRhsToken.type === "variable") {
24
- const add = ret.get(firstRhsToken.token.lexeme);
24
+ const add = ret.get(firstRhsToken.identifier);
25
25
 
26
26
  if (add == null) {
27
27
  return {
28
28
  success: false,
29
- reason: `Variable "${firstRhsToken.token.lexeme}" doesn't have a corresponding left hand side`,
29
+ reason: `Variable "${firstRhsToken.identifier}" doesn't have a corresponding left hand side`,
30
30
  token: firstRhsToken.token,
31
31
  };
32
32
  }
@@ -35,13 +35,13 @@ function computeFirstSets(allProductions: Production[]): Result<Map<string, Set<
35
35
  // if not then add them and set wasEdited to true so we iterate one more time
36
36
  if (!xContainsAllOfY(toBeModified, add)) {
37
37
  wasEdited = true;
38
- ret.set(production.lhs.lexeme, new Set([...toBeModified, ...add]));
38
+ ret.set(production.identifier, new Set([...toBeModified, ...add]));
39
39
  }
40
40
  } else if (firstRhsToken.type === "terminal") {
41
41
  // add terminal to toBeModified and set wasEdited to true
42
- if (toBeModified.has(firstRhsToken.token.lexeme) == false) {
42
+ if (toBeModified.has(firstRhsToken.identifier) == false) {
43
43
  wasEdited = true;
44
- ret.set(production.lhs.lexeme, new Set([...toBeModified, firstRhsToken.token.lexeme]));
44
+ ret.set(production.identifier, new Set([...toBeModified, firstRhsToken.identifier]));
45
45
  }
46
46
  }
47
47
  }
@@ -51,7 +51,7 @@ function computeFirstSets(allProductions: Production[]): Result<Map<string, Set<
51
51
  }
52
52
 
53
53
  function computeFollowSets(
54
- allProductions: Production[]
54
+ allProductions: Production[],
55
55
  ): Result<{ firstSets: Map<string, Set<string>>; followSets: Map<string, Set<string>> }> {
56
56
  const firstSetsResult = computeFirstSets(allProductions);
57
57
  if (firstSetsResult.success === false) return firstSetsResult;
@@ -60,8 +60,8 @@ function computeFollowSets(
60
60
 
61
61
  // create a list of follow sets and add EOF to the initial production
62
62
  const ret = new Map<string, Set<string>>();
63
- for (const production of allProductions) ret.set(production.lhs.lexeme, new Set());
64
- ret.get(allProductions[0].lhs.lexeme)!.add(EOF_STRING);
63
+ for (const production of allProductions) ret.set(production.identifier, new Set());
64
+ ret.get(allProductions[0].identifier)!.add(EOF_STRING);
65
65
 
66
66
  let wasEdited = true;
67
67
  while (wasEdited) {
@@ -75,11 +75,11 @@ function computeFollowSets(
75
75
  for (let i = 0; i < rhs.length; i++) {
76
76
  const currentRhsToken = rhs[i];
77
77
 
78
- if (currentRhsToken.type === "variable" && currentRhsToken.token.lexeme === variable) {
78
+ if (currentRhsToken.type === "variable" && currentRhsToken.identifier === variable) {
79
79
  if (i === rhs.length - 1) {
80
80
  // we are at the end of rhs, so whatever is in currentProduction
81
81
  // we need to also need to add to variable
82
- const toBeAdded = ret.get(currentProduction.lhs.lexeme)!;
82
+ const toBeAdded = ret.get(currentProduction.identifier)!;
83
83
  const receiver = ret.get(variable)!;
84
84
  if (!xContainsAllOfY(receiver, toBeAdded)) {
85
85
  wasEdited = true;
@@ -91,18 +91,18 @@ function computeFollowSets(
91
91
  if (nextRhsToken.type === "terminal") {
92
92
  // next token is a terminal, check if its in existng follow set and add if it doesn't exist
93
93
  const existing = ret.get(variable)!;
94
- if (!existing.has(nextRhsToken.token.lexeme)) {
94
+ if (!existing.has(nextRhsToken.identifier)) {
95
95
  wasEdited = true;
96
- ret.set(variable, new Set([...existing, nextRhsToken.token.lexeme]));
96
+ ret.set(variable, new Set([...existing, nextRhsToken.identifier]));
97
97
  }
98
98
  } else if (nextRhsToken.type === "variable") {
99
99
  // next token is a variable, get the first set of the variable
100
100
  // check if it's not in the followset and add if not
101
- const firstSet = firstSets.get(nextRhsToken.token.lexeme);
101
+ const firstSet = firstSets.get(nextRhsToken.identifier);
102
102
  if (firstSet == null)
103
103
  return {
104
104
  success: false,
105
- reason: `Variable "${nextRhsToken.token.lexeme}" doesn't have a corresponding left hand side`,
105
+ reason: `Variable "${nextRhsToken.identifier}" doesn't have a corresponding left hand side`,
106
106
  token: nextRhsToken.token,
107
107
  };
108
108
 
@@ -134,27 +134,28 @@ export const generateStates = (productions: Production[]): Result<GeneratorResul
134
134
  // The implementation of this function ensures that the table is an LR(1) parsing table
135
135
  const determineNextLookAhead = (
136
136
  lhs: GrammarToken,
137
- array: { type: "terminal" | "variable"; token: GrammarToken }[]
137
+ identifier: string,
138
+ array: { type: "terminal" | "variable"; token: GrammarToken; identifier: string }[],
138
139
  ): Result<string[]> => {
139
140
  if (array.length === 0) {
140
- const followSet = followSets.get(lhs.lexeme);
141
+ const followSet = followSets.get(identifier);
141
142
  if (followSet != undefined) return { success: true, value: [...followSet] };
142
143
 
143
144
  return {
144
145
  success: false,
145
- reason: `Variable ${lhs.lexeme} not present in computed follow sets`,
146
+ reason: `Variable ${identifier} not present in computed follow sets`,
146
147
  token: lhs,
147
148
  };
148
149
  } else {
149
150
  const next = array[0];
150
151
 
151
- if (next.type === "terminal") return { success: true, value: [next.token.lexeme] };
152
- const testing = firstSets.get(next.token.lexeme);
152
+ if (next.type === "terminal") return { success: true, value: [next.identifier] };
153
+ const testing = firstSets.get(next.identifier);
153
154
  if (testing !== undefined) return { success: true, value: [...testing] };
154
155
 
155
156
  return {
156
157
  success: false,
157
- reason: `Variable ${next.token.lexeme} not present in computed first sets`,
158
+ reason: `Variable ${next.identifier} not present in computed first sets`,
158
159
  token: next.token,
159
160
  };
160
161
  }
@@ -163,7 +164,14 @@ export const generateStates = (productions: Production[]): Result<GeneratorResul
163
164
  // This fnuction creates the initial item set based on the
164
165
  // production provided. It creating the initial item and expands it
165
166
  function generateInitialItemSet(production: Production) {
166
- const initialItem = { lhs: production.lhs, rhs: production.rhs, dot: 0, lookahead: [EOF_STRING] } as Item;
167
+ const initialItem = {
168
+ lhs: production.lhs,
169
+ identifier: production.identifier,
170
+ rhs: production.rhs,
171
+ dot: 0,
172
+ lookahead: [EOF_STRING],
173
+ } as Item;
174
+
167
175
  return expandItemSet([initialItem]);
168
176
  }
169
177
 
@@ -182,19 +190,25 @@ export const generateStates = (productions: Production[]): Result<GeneratorResul
182
190
 
183
191
  if (after.type === "variable") {
184
192
  // Find prodctions whose left hand side is the symbol after the dot
185
- const newProductions = productions.filter((p) => p.lhs.lexeme === after.token.lexeme);
193
+ const newProductions = productions.filter((p) => p.identifier === after.identifier);
186
194
 
187
195
  // Compute the lookahead for the new productions to be added to the item set
188
196
  const rest = currentItem.rhs.slice(currentItem.dot + 1);
189
197
 
190
- const lookaheadResult = determineNextLookAhead(currentItem.lhs, rest);
198
+ const lookaheadResult = determineNextLookAhead(currentItem.lhs, currentItem.identifier, rest);
191
199
  if (lookaheadResult.success === false) return lookaheadResult;
192
200
  const lookahead = lookaheadResult.value;
193
201
 
194
202
  for (const newProduction of newProductions) {
195
203
  // Create the new item and check if it already exists in the item set
196
204
  // If it doesn't exist, add it to the item set and the queue of unprocessed items
197
- const newItem = { lhs: newProduction.lhs, rhs: newProduction.rhs, dot: 0, lookahead } as Item;
205
+ const newItem = {
206
+ lhs: newProduction.lhs,
207
+ identifier: newProduction.identifier,
208
+ rhs: newProduction.rhs,
209
+ dot: 0,
210
+ lookahead,
211
+ } as Item;
198
212
  const encoding = JSON.stringify(newItem);
199
213
 
200
214
  if (itemSet.some((i) => JSON.stringify(i) === encoding) == false) {
@@ -247,7 +261,7 @@ export const generateStates = (productions: Production[]): Result<GeneratorResul
247
261
  const productionToReduceTo = productions
248
262
  .map((p, i) => [p, i] as const)
249
263
  .find(
250
- ([p, _]) => JSON.stringify(p.rhs) === JSON.stringify(item.rhs) && p.lhs.lexeme === item.lhs.lexeme
264
+ ([p, _]) => JSON.stringify(p.rhs) === JSON.stringify(item.rhs) && p.identifier === item.identifier,
251
265
  )![1];
252
266
 
253
267
  // for each lookahead in the item, create a new entry in the action table to reduce to the production found
@@ -266,14 +280,13 @@ export const generateStates = (productions: Production[]): Result<GeneratorResul
266
280
  if (nextSymbol.type === "variable") {
267
281
  // Next symbol is a variable so we need to add it to the GOTO table
268
282
  // Check if the GOTO table already has an entry for the next symbola and either push or create the array
269
- if (nextStates_Goto.has(nextSymbol.token.lexeme)) nextStates_Goto.get(nextSymbol.token.lexeme)!.push(newItem);
270
- else nextStates_Goto.set(nextSymbol.token.lexeme, [newItem]);
283
+ if (nextStates_Goto.has(nextSymbol.identifier)) nextStates_Goto.get(nextSymbol.identifier)!.push(newItem);
284
+ else nextStates_Goto.set(nextSymbol.identifier, [newItem]);
271
285
  } else if (nextSymbol.type === "terminal") {
272
286
  // Next symbol is a terminal so we need to add it to the SHIFT table
273
287
  // Check if the SHIFT table already has an entry for the next symbola and either push or create the array
274
- if (nextStates_Shift.has(nextSymbol.token.lexeme))
275
- nextStates_Shift.get(nextSymbol.token.lexeme)!.push(newItem);
276
- else nextStates_Shift.set(nextSymbol.token.lexeme, [newItem]);
288
+ if (nextStates_Shift.has(nextSymbol.identifier)) nextStates_Shift.get(nextSymbol.identifier)!.push(newItem);
289
+ else nextStates_Shift.set(nextSymbol.identifier, [newItem]);
277
290
  }
278
291
  }
279
292
 
@@ -367,7 +380,7 @@ export class GeneratorResult {
367
380
  constructor(
368
381
  public readonly states: State[],
369
382
  public readonly ActionTable: Map<string, { action: "shift" | "reduce"; value: number }>[],
370
- public readonly GotoTable: Map<string, number>[]
383
+ public readonly GotoTable: Map<string, number>[],
371
384
  ) {}
372
385
 
373
386
  toTable(): string {
@@ -382,6 +395,7 @@ export class GeneratorResult {
382
395
  toStates(): TableState[] {
383
396
  return this.ActionTable.map((actions, idx) => {
384
397
  const tableState = new TableState();
398
+
385
399
  for (const [k, { action, value }] of actions.entries())
386
400
  tableState.actions.set(k, { type: action === "reduce" ? "reduce" : "shift", value });
387
401
 
@@ -391,4 +405,8 @@ export class GeneratorResult {
391
405
  return tableState;
392
406
  });
393
407
  }
408
+
409
+ toJSObject() {
410
+ return this.toStates().map((tableState) => tableState.toJSObject());
411
+ }
394
412
  }
package/src/index.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  export { GeneratorResult, generateStates } from "./generator";
2
- export { Production, buildProductions, tryBuildProductions } from "./grammarParser";
2
+ export { buildProductions, tryBuildProductions } from "./meta/selfhosted";
3
3
  export { Result } from "./utils/Result";
4
4
  export {
5
5
  ParserRecoveryFunction,
@@ -9,4 +9,5 @@ export {
9
9
  ParserResult,
10
10
  ParserError,
11
11
  } from "./parser";
12
- export { buildStates } from "./tableParser";
12
+ export { buildStates } from "./table/selfhosted";
13
+ export { Production } from "./meta/common";
@@ -0,0 +1,74 @@
1
+ import { ColumnAndRow, Token } from "@scinorandex/slex";
2
+
3
+ export type Production = {
4
+ lhs: GrammarToken;
5
+ identifier: string;
6
+ originalProductionIndex: number;
7
+ name: string | null;
8
+ rhs: { type: "terminal" | "variable"; token: GrammarToken; identifier: string; name: string | null }[];
9
+ };
10
+
11
+ const dehydateGrammarToken = ({ column, lexeme, line, type }: GrammarToken) => {
12
+ return { column, lexeme, line, type };
13
+ };
14
+
15
+ const hydrateGrammarToken = (
16
+ token: ReturnType<typeof dehydateGrammarToken>,
17
+ ): Token<GrammarTokenType, GrammarTokenMetadata> => {
18
+ return new Token(token.type, token.lexeme, new ColumnAndRow(token.line, token.column), {});
19
+ };
20
+
21
+ export const dehydateProduction = (opts: Production) => {
22
+ const { lhs, identifier, rhs, originalProductionIndex, name } = opts;
23
+ return {
24
+ lhs: dehydateGrammarToken(lhs),
25
+ identifier,
26
+ name,
27
+ originalProductionIndex,
28
+ rhs: rhs.map(({ type, token, identifier, name }) => ({
29
+ type,
30
+ token: dehydateGrammarToken(token),
31
+ identifier,
32
+ name,
33
+ })),
34
+ };
35
+ };
36
+
37
+ export const hydrateProduction = (production: ReturnType<typeof dehydateProduction>): Production => {
38
+ return {
39
+ lhs: hydrateGrammarToken(production.lhs),
40
+ identifier: production.identifier,
41
+ originalProductionIndex: production.originalProductionIndex,
42
+ name: production.name,
43
+ rhs: production.rhs.map(({ type, token, name }) => ({
44
+ type,
45
+ token: hydrateGrammarToken(token),
46
+ identifier: token.lexeme,
47
+ name,
48
+ })),
49
+ };
50
+ };
51
+
52
+ export type GrammarTokenMetadata = {};
53
+ export enum GrammarTokenType {
54
+ IDENTIFIER,
55
+ L_ANGLE,
56
+ R_ANGLE,
57
+ L_BRACKET,
58
+ R_BRACKET,
59
+ L_PAREN,
60
+ R_PAREN,
61
+ PIPE,
62
+ QUESTION_MARK,
63
+ COLON,
64
+ SEMICOLON,
65
+ NUMBER,
66
+ EQUALS,
67
+ COMMA,
68
+ EOF,
69
+
70
+ PRODUCTION_NAME,
71
+ TOKEN_NAME,
72
+ }
73
+
74
+ export type GrammarToken = Token<GrammarTokenType, GrammarTokenMetadata>;
@@ -1,18 +1,6 @@
1
- import { Slex, Token } from "@scinorandex/slex";
2
- import { Result } from "./utils/Result";
3
-
4
- export type Production = { lhs: GrammarToken; rhs: { type: "terminal" | "variable"; token: GrammarToken }[] };
5
-
6
- export type GrammarTokenMetadata = {};
7
- export enum GrammarTokenType {
8
- PRODUCTION_NAME,
9
- TOKEN_NAME,
10
- COLON,
11
- SEMICOLON,
12
- EOF,
13
- }
14
-
15
- export type GrammarToken = Token<GrammarTokenType, GrammarTokenMetadata>;
1
+ import { Slex } from "@scinorandex/slex";
2
+ import { Result } from "../utils/Result";
3
+ import { GrammarToken, GrammarTokenMetadata, GrammarTokenType, Production } from "./common";
16
4
 
17
5
  export const grammarLexerGenerator = new Slex<GrammarTokenType, GrammarTokenMetadata>({
18
6
  EOF_TYPE: GrammarTokenType.EOF,
@@ -21,11 +9,11 @@ export const grammarLexerGenerator = new Slex<GrammarTokenType, GrammarTokenMeta
21
9
 
22
10
  grammarLexerGenerator.addRule(
23
11
  "lowercase",
24
- "a | b | c | d | e | f | g | h | i | j | k | l | m | n | o | p | q | r | s | t | u | v | w | x | y | z"
12
+ "a | b | c | d | e | f | g | h | i | j | k | l | m | n | o | p | q | r | s | t | u | v | w | x | y | z",
25
13
  );
26
14
  grammarLexerGenerator.addRule(
27
15
  "uppercase",
28
- "A | B | C | D | E | F | G | H | I | J | K | L | M | N | O | P | Q | R | S | T | U | V | W | X | Y | Z"
16
+ "A | B | C | D | E | F | G | H | I | J | K | L | M | N | O | P | Q | R | S | T | U | V | W | X | Y | Z",
29
17
  );
30
18
  grammarLexerGenerator.addRule("letter", "${lowercase} | ${uppercase}");
31
19
  grammarLexerGenerator.addRule("digit", "0 | 1 | 2 | 3 | 4 | 5 | 6 | 7 | 8 | 9");
@@ -53,19 +41,27 @@ export const tryBuildProductions = (lexer: LexerInterface | string): Result<Prod
53
41
  lexer.getNextToken();
54
42
  };
55
43
 
44
+ let productionIndex = 0;
45
+
56
46
  const buildProduction = (): Result<Production> => {
57
47
  const productionLHS = lexer.getNextToken();
58
48
  expect(GrammarTokenType.COLON);
59
49
 
60
- const production: Production = { lhs: productionLHS, rhs: [] };
50
+ const production: Production = {
51
+ lhs: productionLHS,
52
+ identifier: productionLHS.lexeme,
53
+ rhs: [],
54
+ name: null,
55
+ originalProductionIndex: productionIndex++,
56
+ };
61
57
 
62
58
  while (lexer.peekNextToken().type != GrammarTokenType.SEMICOLON) {
63
59
  const productionRHS = lexer.getNextToken();
64
60
 
65
61
  if (productionRHS.type == GrammarTokenType.PRODUCTION_NAME)
66
- production.rhs.push({ type: "variable", token: productionRHS });
62
+ production.rhs.push({ type: "variable", token: productionRHS, identifier: productionRHS.lexeme, name: null });
67
63
  else if (productionRHS.type == GrammarTokenType.TOKEN_NAME)
68
- production.rhs.push({ type: "terminal", token: productionRHS });
64
+ production.rhs.push({ type: "terminal", token: productionRHS, identifier: productionRHS.lexeme, name: null });
69
65
  else
70
66
  return {
71
67
  success: false,
@@ -91,7 +87,7 @@ export const buildProductions = (lexer: LexerInterface | string): Production[] =
91
87
  const result = tryBuildProductions(lexer);
92
88
  if (result.success === false)
93
89
  throw new Error(
94
- `Encountered error "${result.reason}" while parsing grammar at ${result.token.line}:${result.token.column}`
90
+ `Encountered error "${result.reason}" while parsing grammar at ${result.token.line}:${result.token.column}`,
95
91
  );
96
92
  return result.value;
97
93
  };
@@ -0,0 +1,272 @@
1
+ import { RegexEngine, Slex } from "@scinorandex/slex";
2
+ import { Result, Sparse } from "../index";
3
+ import { selfhosted } from "./states";
4
+ import { TableState } from "../parser";
5
+ import { GrammarToken, GrammarTokenMetadata, GrammarTokenType, hydrateProduction, Production } from "./common";
6
+
7
+ export const grammarLexerGenerator = new Slex<GrammarTokenType, GrammarTokenMetadata>({
8
+ EOF_TYPE: GrammarTokenType.EOF,
9
+ isHigherPrecedence: () => false,
10
+ });
11
+
12
+ grammarLexerGenerator.addRule(
13
+ "lowercase",
14
+ "a | b | c | d | e | f | g | h | i | j | k | l | m | n | o | p | q | r | s | t | u | v | w | x | y | z",
15
+ );
16
+ grammarLexerGenerator.addRule(
17
+ "uppercase",
18
+ "A | B | C | D | E | F | G | H | I | J | K | L | M | N | O | P | Q | R | S | T | U | V | W | X | Y | Z",
19
+ );
20
+ grammarLexerGenerator.addRule("letter", "${lowercase} | ${uppercase}");
21
+ grammarLexerGenerator.addRule("digit", "0 | 1 | 2 | 3 | 4 | 5 | 6 | 7 | 8 | 9");
22
+ grammarLexerGenerator.addRule("alphanumeric", "${letter} | ${digit}");
23
+ grammarLexerGenerator.addRule(
24
+ "identifier",
25
+ "(${letter} | $_) (${letter} | ${digit} | $_)*",
26
+ GrammarTokenType.IDENTIFIER,
27
+ );
28
+ grammarLexerGenerator.addRule("colon", "$:", GrammarTokenType.COLON);
29
+ grammarLexerGenerator.addRule("semicolon", "$;", GrammarTokenType.SEMICOLON);
30
+ grammarLexerGenerator.addRule("langle", "$<", GrammarTokenType.L_ANGLE);
31
+ grammarLexerGenerator.addRule("rangle", "$>", GrammarTokenType.R_ANGLE);
32
+ grammarLexerGenerator.addRule("lbracket", "$[", GrammarTokenType.L_BRACKET);
33
+ grammarLexerGenerator.addRule("rbracket", "$]", GrammarTokenType.R_BRACKET);
34
+ grammarLexerGenerator.addRule("lparen", "$(", GrammarTokenType.L_PAREN);
35
+ grammarLexerGenerator.addRule("rparen", "$)", GrammarTokenType.R_PAREN);
36
+ grammarLexerGenerator.addRule("pipe", "$|", GrammarTokenType.PIPE);
37
+ grammarLexerGenerator.addRule("equals", "$=", GrammarTokenType.EQUALS);
38
+ grammarLexerGenerator.addRule("comma", "$,", GrammarTokenType.COMMA);
39
+ grammarLexerGenerator.addRule("question_mark", "$?", GrammarTokenType.QUESTION_MARK);
40
+ grammarLexerGenerator.addRule("number", "(${digit})+", GrammarTokenType.NUMBER);
41
+
42
+ export abstract class BaseNode {}
43
+
44
+ export class ListNode<T> extends BaseNode {
45
+ private items: T[];
46
+
47
+ constructor(items: T | T[]) {
48
+ super();
49
+
50
+ if (Array.isArray(items)) this.items = items;
51
+ else this.items = [items];
52
+ }
53
+
54
+ add(item: T) {
55
+ this.items.push(item);
56
+ return this;
57
+ }
58
+
59
+ getItems() {
60
+ return this.items;
61
+ }
62
+
63
+ getItemsReversed() {
64
+ return this.items.toReversed();
65
+ }
66
+ }
67
+
68
+ class ProductionNode extends BaseNode {
69
+ constructor(
70
+ public readonly lhs: GrammarToken,
71
+ public readonly rhs: ListNode<TokenNode | GroupedTokenNode>,
72
+ public originalProductionIndex: number,
73
+ public readonly name?: string | undefined,
74
+ ) {
75
+ super();
76
+ }
77
+
78
+ toStruct(): Production {
79
+ if (this.originalProductionIndex === -1) throw new Error("ProductionNode has no originalProductionIndex");
80
+
81
+ const hasGrouped = this.rhs.getItemsReversed().some((node) => node instanceof GroupedTokenNode);
82
+ if (hasGrouped) throw new Error("Need to unroll before serializing");
83
+
84
+ return {
85
+ lhs: this.lhs,
86
+ identifier: `<${this.lhs.lexeme}>`,
87
+ name: this.name ?? null,
88
+ originalProductionIndex: this.originalProductionIndex,
89
+ rhs: this.rhs.getItemsReversed().map((node) => (node as TokenNode).toStruct()),
90
+ };
91
+ }
92
+
93
+ setOriginalProductionIndex(index: number) {
94
+ this.originalProductionIndex = index;
95
+ }
96
+ }
97
+
98
+ function expandProduction(production: ProductionNode): ProductionNode[] {
99
+ const rhs = production.rhs.getItemsReversed();
100
+
101
+ for (let i = 0; i < rhs.length; i++) {
102
+ const item = rhs[i];
103
+
104
+ const beforeItems = rhs.slice(0, i);
105
+ const afterItems = rhs.slice(i + 1);
106
+
107
+ if (item instanceof GroupedTokenNode) {
108
+ // need to create two new productions, one optional and the other not
109
+ // then apply unrolling on both
110
+
111
+ const withoutCurrentItem = new ProductionNode(
112
+ production.lhs,
113
+ new ListNode<TokenNode | GroupedTokenNode>([...beforeItems, ...afterItems].toReversed()),
114
+ production.originalProductionIndex,
115
+ production.name,
116
+ );
117
+
118
+ const withCurrentItem = new ProductionNode(
119
+ production.lhs,
120
+ new ListNode<TokenNode | GroupedTokenNode>(
121
+ [...beforeItems, ...item.inside.getItemsReversed(), ...afterItems].toReversed(),
122
+ ),
123
+ production.originalProductionIndex,
124
+ production.name,
125
+ );
126
+
127
+ const newProductions: ProductionNode[] = [withoutCurrentItem, withCurrentItem];
128
+ return newProductions.flatMap((prod) => expandProduction(prod));
129
+ } else if (item instanceof TokenNode) {
130
+ if (item.variables.getItems().length > 1) {
131
+ // this token node has many variants and we should unroll it
132
+ const variables = item.variables.getItemsReversed();
133
+
134
+ const newProductions = variables.map((variable) => {
135
+ return new ProductionNode(
136
+ production.lhs,
137
+ new ListNode<TokenNode | GroupedTokenNode>([
138
+ ...beforeItems,
139
+ new TokenNode("variable", new ListNode<GrammarToken>([variable]), item.name),
140
+ ]),
141
+ production.originalProductionIndex,
142
+ production.name,
143
+ );
144
+ });
145
+
146
+ return newProductions.flatMap((prod) => expandProduction(prod));
147
+ }
148
+ } else {
149
+ throw new Error("shoudn't reach here, ProductionNode RHS is neither GroupedTokenNode nor TokenNode");
150
+ }
151
+ }
152
+
153
+ // if it reached here then its normal
154
+ return [production];
155
+ }
156
+
157
+ class TokenNode extends BaseNode {
158
+ constructor(
159
+ public readonly type: "terminal" | "variable",
160
+ public readonly variables: ListNode<GrammarToken>,
161
+ public readonly name: string | undefined,
162
+ ) {
163
+ super();
164
+ }
165
+
166
+ toStruct() {
167
+ const items = this.variables.getItems();
168
+ if (items.length > 1) throw new Error("Not yet unrolled");
169
+
170
+ const lexeme = items[0].lexeme;
171
+ const identifier = this.type === "variable" ? `<${lexeme}>` : `[${lexeme}]`;
172
+ const name = this.name ?? null;
173
+
174
+ return { type: this.type, token: items[0], identifier, name };
175
+ }
176
+ }
177
+
178
+ class GroupedTokenNode extends BaseNode {
179
+ constructor(public readonly inside: ListNode<TokenNode>) {
180
+ super();
181
+ }
182
+ }
183
+
184
+ class ProgramNode extends BaseNode {
185
+ constructor(public readonly productions: ListNode<ProductionNode>) {
186
+ super();
187
+ }
188
+
189
+ unrollProductions() {
190
+ const productions = this.productions.getItemsReversed();
191
+ for (let i = 0; i < productions.length; i++) productions[i].setOriginalProductionIndex(i);
192
+ return productions.flatMap((node) => expandProduction(node));
193
+ }
194
+
195
+ getProductions() {
196
+ return this.unrollProductions().map((node) => node.toStruct());
197
+ }
198
+ }
199
+
200
+ export function getSelfHostedParserGenerator(e: typeof selfhosted = selfhosted) {
201
+ return new Sparse<GrammarTokenType, GrammarTokenMetadata, BaseNode>({
202
+ productions: e.productions.map((production) => hydrateProduction(production as any)),
203
+ states: e.states.map((state) => TableState.fromJSObject(state as any)),
204
+ toStringifiedTokenType: (type: GrammarTokenType) => GrammarTokenType[type],
205
+ });
206
+ }
207
+
208
+ type Reducer = (bag: any) => BaseNode;
209
+ const reducers: { [key: string]: Reducer } = {
210
+ program: (bag: { productions?: ListNode<ProductionNode> }) =>
211
+ new ProgramNode(bag.productions ?? new ListNode<ProductionNode>([])),
212
+
213
+ productions: (bag: { production: ProductionNode; rest?: ListNode<ProductionNode> }) => {
214
+ if (bag.rest == null) return new ListNode<ProductionNode>([bag.production]);
215
+ else return bag.rest.add(bag.production);
216
+ },
217
+
218
+ production: (bag: {
219
+ production_name: GrammarToken;
220
+ tokens: ListNode<TokenNode | GroupedTokenNode>;
221
+ uuid?: GrammarToken;
222
+ }) => new ProductionNode(bag.production_name, bag.tokens, -1, bag.uuid?.lexeme),
223
+
224
+ tokens: (bag: { token: TokenNode | GroupedTokenNode; rest?: ListNode<TokenNode | GroupedTokenNode> }) => {
225
+ if (bag.rest == null) return new ListNode<TokenNode | GroupedTokenNode>([bag.token]);
226
+ else return bag.rest.add(bag.token);
227
+ },
228
+
229
+ token: (bag: { token: TokenNode }) => bag.token,
230
+ grouped_token: (bag: { tokens: ListNode<TokenNode> }) => new GroupedTokenNode(bag.tokens),
231
+
232
+ variable: (bag: { inside: ListNode<GrammarToken>; token_name?: GrammarToken }) =>
233
+ new TokenNode("variable", bag.inside, bag.token_name?.lexeme),
234
+
235
+ terminal: (bag: { inside: ListNode<GrammarToken>; token_name?: GrammarToken }) =>
236
+ new TokenNode("terminal", bag.inside, bag.token_name?.lexeme),
237
+
238
+ inside: (bag: { identifier: GrammarToken; rest?: ListNode<GrammarToken> }) => {
239
+ if (bag.rest == null) return new ListNode<GrammarToken>([bag.identifier]);
240
+ else return bag.rest.add(bag.identifier);
241
+ },
242
+ };
243
+
244
+ export const tryBuildProductions = (lexer: LexerInterface | string): Result<Production[]> => {
245
+ if (typeof lexer === "string") lexer = grammarLexerGenerator.generate(lexer, () => ({}));
246
+
247
+ const parserGenerator = getSelfHostedParserGenerator();
248
+ const parser = parserGenerator.generate(lexer as RegexEngine<GrammarTokenType, GrammarTokenMetadata>, {
249
+ reducer: ({ bag, name }) => {
250
+ return reducers[name ?? ""](bag);
251
+ },
252
+ });
253
+
254
+ const parsingResult = parser.parse().result as ProgramNode;
255
+ const productions = parsingResult.getProductions();
256
+ return { success: true, value: productions };
257
+ };
258
+
259
+ interface LexerInterface {
260
+ peekNextToken(): GrammarToken;
261
+ getNextToken(): GrammarToken;
262
+ hasNextToken(): boolean;
263
+ }
264
+
265
+ export const buildProductions = (lexer: LexerInterface | string): Production[] => {
266
+ const result = tryBuildProductions(lexer);
267
+ if (result.success === false)
268
+ throw new Error(
269
+ `Encountered error "${result.reason}" while parsing grammar at ${result.token.line}:${result.token.column}`,
270
+ );
271
+ return result.value;
272
+ };