@scinorandex/sparse 0.0.7 → 0.0.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -380,11 +380,11 @@ export const states = [
380
380
  {
381
381
  "[R_ANGLE]": {
382
382
  "type": "reduce",
383
- "value": 16
383
+ "value": 18
384
384
  },
385
385
  "[COLON]": {
386
386
  "type": "reduce",
387
- "value": 16
387
+ "value": 18
388
388
  },
389
389
  "[PIPE]": {
390
390
  "type": "shift",
@@ -404,11 +404,11 @@ export const states = [
404
404
  {
405
405
  "[R_BRACKET]": {
406
406
  "type": "reduce",
407
- "value": 16
407
+ "value": 18
408
408
  },
409
409
  "[COLON]": {
410
410
  "type": "reduce",
411
- "value": 16
411
+ "value": 18
412
412
  },
413
413
  "[PIPE]": {
414
414
  "type": "shift",
@@ -459,6 +459,14 @@ export const states = [
459
459
  "[QUESTION_MARK]": {
460
460
  "type": "shift",
461
461
  "value": 41
462
+ },
463
+ "[PLUS]": {
464
+ "type": "shift",
465
+ "value": 42
466
+ },
467
+ "[STAR]": {
468
+ "type": "shift",
469
+ "value": 43
462
470
  }
463
471
  },
464
472
  {
@@ -470,83 +478,83 @@ export const states = [
470
478
  {
471
479
  "[SEMICOLON]": {
472
480
  "type": "reduce",
473
- "value": 12
481
+ "value": 14
474
482
  },
475
483
  "[R_PAREN]": {
476
484
  "type": "reduce",
477
- "value": 12
485
+ "value": 14
478
486
  },
479
487
  "[L_PAREN]": {
480
488
  "type": "reduce",
481
- "value": 12
489
+ "value": 14
482
490
  },
483
491
  "[L_ANGLE]": {
484
492
  "type": "reduce",
485
- "value": 12
493
+ "value": 14
486
494
  },
487
495
  "[L_BRACKET]": {
488
496
  "type": "reduce",
489
- "value": 12
497
+ "value": 14
490
498
  }
491
499
  },
492
500
  {
493
501
  "[IDENTIFIER]": {
494
502
  "type": "shift",
495
- "value": 42
503
+ "value": 44
496
504
  }
497
505
  },
498
506
  {
499
507
  "[IDENTIFIER]": {
500
508
  "type": "shift",
501
- "value": 44
509
+ "value": 46
502
510
  },
503
511
  "<INSIDE>": {
504
512
  "type": "goto",
505
- "value": 43
513
+ "value": 45
506
514
  }
507
515
  },
508
516
  {
509
517
  "[SEMICOLON]": {
510
518
  "type": "reduce",
511
- "value": 14
519
+ "value": 16
512
520
  },
513
521
  "[R_PAREN]": {
514
522
  "type": "reduce",
515
- "value": 14
523
+ "value": 16
516
524
  },
517
525
  "[L_PAREN]": {
518
526
  "type": "reduce",
519
- "value": 14
527
+ "value": 16
520
528
  },
521
529
  "[L_ANGLE]": {
522
530
  "type": "reduce",
523
- "value": 14
531
+ "value": 16
524
532
  },
525
533
  "[L_BRACKET]": {
526
534
  "type": "reduce",
527
- "value": 14
535
+ "value": 16
528
536
  }
529
537
  },
530
538
  {
531
539
  "[IDENTIFIER]": {
532
540
  "type": "shift",
533
- "value": 45
541
+ "value": 47
534
542
  }
535
543
  },
536
544
  {
537
545
  "[IDENTIFIER]": {
538
546
  "type": "shift",
539
- "value": 44
547
+ "value": 46
540
548
  },
541
549
  "<INSIDE>": {
542
550
  "type": "goto",
543
- "value": 46
551
+ "value": 48
544
552
  }
545
553
  },
546
554
  {
547
555
  "[SEMICOLON]": {
548
556
  "type": "shift",
549
- "value": 47
557
+ "value": 49
550
558
  }
551
559
  },
552
560
  {
@@ -571,54 +579,98 @@ export const states = [
571
579
  "value": 11
572
580
  }
573
581
  },
582
+ {
583
+ "[SEMICOLON]": {
584
+ "type": "reduce",
585
+ "value": 12
586
+ },
587
+ "[R_PAREN]": {
588
+ "type": "reduce",
589
+ "value": 12
590
+ },
591
+ "[L_PAREN]": {
592
+ "type": "reduce",
593
+ "value": 12
594
+ },
595
+ "[L_ANGLE]": {
596
+ "type": "reduce",
597
+ "value": 12
598
+ },
599
+ "[L_BRACKET]": {
600
+ "type": "reduce",
601
+ "value": 12
602
+ }
603
+ },
604
+ {
605
+ "[SEMICOLON]": {
606
+ "type": "reduce",
607
+ "value": 13
608
+ },
609
+ "[R_PAREN]": {
610
+ "type": "reduce",
611
+ "value": 13
612
+ },
613
+ "[L_PAREN]": {
614
+ "type": "reduce",
615
+ "value": 13
616
+ },
617
+ "[L_ANGLE]": {
618
+ "type": "reduce",
619
+ "value": 13
620
+ },
621
+ "[L_BRACKET]": {
622
+ "type": "reduce",
623
+ "value": 13
624
+ }
625
+ },
574
626
  {
575
627
  "[R_ANGLE]": {
576
628
  "type": "shift",
577
- "value": 48
629
+ "value": 50
578
630
  }
579
631
  },
580
632
  {
581
633
  "[R_ANGLE]": {
582
634
  "type": "reduce",
583
- "value": 17
635
+ "value": 19
584
636
  },
585
637
  "[COLON]": {
586
638
  "type": "reduce",
587
- "value": 17
639
+ "value": 19
588
640
  }
589
641
  },
590
642
  {
591
643
  "[R_ANGLE]": {
592
644
  "type": "reduce",
593
- "value": 16
645
+ "value": 18
594
646
  },
595
647
  "[COLON]": {
596
648
  "type": "reduce",
597
- "value": 16
649
+ "value": 18
598
650
  },
599
651
  "[R_BRACKET]": {
600
652
  "type": "reduce",
601
- "value": 16
653
+ "value": 18
602
654
  },
603
655
  "[PIPE]": {
604
656
  "type": "shift",
605
- "value": 49
657
+ "value": 51
606
658
  }
607
659
  },
608
660
  {
609
661
  "[R_BRACKET]": {
610
662
  "type": "shift",
611
- "value": 50
663
+ "value": 52
612
664
  }
613
665
  },
614
666
  {
615
667
  "[R_BRACKET]": {
616
668
  "type": "reduce",
617
- "value": 17
669
+ "value": 19
618
670
  },
619
671
  "[COLON]": {
620
672
  "type": "reduce",
621
- "value": 17
673
+ "value": 19
622
674
  }
623
675
  },
624
676
  {
@@ -634,69 +686,69 @@ export const states = [
634
686
  {
635
687
  "[SEMICOLON]": {
636
688
  "type": "reduce",
637
- "value": 13
689
+ "value": 15
638
690
  },
639
691
  "[R_PAREN]": {
640
692
  "type": "reduce",
641
- "value": 13
693
+ "value": 15
642
694
  },
643
695
  "[L_PAREN]": {
644
696
  "type": "reduce",
645
- "value": 13
697
+ "value": 15
646
698
  },
647
699
  "[L_ANGLE]": {
648
700
  "type": "reduce",
649
- "value": 13
701
+ "value": 15
650
702
  },
651
703
  "[L_BRACKET]": {
652
704
  "type": "reduce",
653
- "value": 13
705
+ "value": 15
654
706
  }
655
707
  },
656
708
  {
657
709
  "[IDENTIFIER]": {
658
710
  "type": "shift",
659
- "value": 44
711
+ "value": 46
660
712
  },
661
713
  "<INSIDE>": {
662
714
  "type": "goto",
663
- "value": 51
715
+ "value": 53
664
716
  }
665
717
  },
666
718
  {
667
719
  "[SEMICOLON]": {
668
720
  "type": "reduce",
669
- "value": 15
721
+ "value": 17
670
722
  },
671
723
  "[R_PAREN]": {
672
724
  "type": "reduce",
673
- "value": 15
725
+ "value": 17
674
726
  },
675
727
  "[L_PAREN]": {
676
728
  "type": "reduce",
677
- "value": 15
729
+ "value": 17
678
730
  },
679
731
  "[L_ANGLE]": {
680
732
  "type": "reduce",
681
- "value": 15
733
+ "value": 17
682
734
  },
683
735
  "[L_BRACKET]": {
684
736
  "type": "reduce",
685
- "value": 15
737
+ "value": 17
686
738
  }
687
739
  },
688
740
  {
689
741
  "[R_ANGLE]": {
690
742
  "type": "reduce",
691
- "value": 17
743
+ "value": 19
692
744
  },
693
745
  "[COLON]": {
694
746
  "type": "reduce",
695
- "value": 17
747
+ "value": 19
696
748
  },
697
749
  "[R_BRACKET]": {
698
750
  "type": "reduce",
699
- "value": 17
751
+ "value": 19
700
752
  }
701
753
  }
702
754
  ];
@@ -1183,7 +1235,121 @@ export const states = [
1183
1235
  "type": 0
1184
1236
  },
1185
1237
  "identifier": "[QUESTION_MARK]",
1186
- "name": "question_mark"
1238
+ "name": "modifier"
1239
+ }
1240
+ ]
1241
+ },
1242
+ {
1243
+ "lhs": {
1244
+ "column": 1,
1245
+ "lexeme": "TOKEN",
1246
+ "line": 7,
1247
+ "type": 0
1248
+ },
1249
+ "identifier": "<TOKEN>",
1250
+ "name": "grouped_token",
1251
+ "originalProductionIndex": 6,
1252
+ "rhs": [
1253
+ {
1254
+ "type": "terminal",
1255
+ "token": {
1256
+ "column": 25,
1257
+ "lexeme": "L_PAREN",
1258
+ "line": 7,
1259
+ "type": 0
1260
+ },
1261
+ "identifier": "[L_PAREN]",
1262
+ "name": null
1263
+ },
1264
+ {
1265
+ "type": "variable",
1266
+ "token": {
1267
+ "column": 35,
1268
+ "lexeme": "TOKENS",
1269
+ "line": 7,
1270
+ "type": 0
1271
+ },
1272
+ "identifier": "<TOKENS>",
1273
+ "name": "tokens"
1274
+ },
1275
+ {
1276
+ "type": "terminal",
1277
+ "token": {
1278
+ "column": 52,
1279
+ "lexeme": "R_PAREN",
1280
+ "line": 7,
1281
+ "type": 0
1282
+ },
1283
+ "identifier": "[R_PAREN]",
1284
+ "name": null
1285
+ },
1286
+ {
1287
+ "type": "terminal",
1288
+ "token": {
1289
+ "column": 78,
1290
+ "lexeme": "PLUS",
1291
+ "line": 7,
1292
+ "type": 0
1293
+ },
1294
+ "identifier": "[PLUS]",
1295
+ "name": "modifier"
1296
+ }
1297
+ ]
1298
+ },
1299
+ {
1300
+ "lhs": {
1301
+ "column": 1,
1302
+ "lexeme": "TOKEN",
1303
+ "line": 7,
1304
+ "type": 0
1305
+ },
1306
+ "identifier": "<TOKEN>",
1307
+ "name": "grouped_token",
1308
+ "originalProductionIndex": 6,
1309
+ "rhs": [
1310
+ {
1311
+ "type": "terminal",
1312
+ "token": {
1313
+ "column": 25,
1314
+ "lexeme": "L_PAREN",
1315
+ "line": 7,
1316
+ "type": 0
1317
+ },
1318
+ "identifier": "[L_PAREN]",
1319
+ "name": null
1320
+ },
1321
+ {
1322
+ "type": "variable",
1323
+ "token": {
1324
+ "column": 35,
1325
+ "lexeme": "TOKENS",
1326
+ "line": 7,
1327
+ "type": 0
1328
+ },
1329
+ "identifier": "<TOKENS>",
1330
+ "name": "tokens"
1331
+ },
1332
+ {
1333
+ "type": "terminal",
1334
+ "token": {
1335
+ "column": 52,
1336
+ "lexeme": "R_PAREN",
1337
+ "line": 7,
1338
+ "type": 0
1339
+ },
1340
+ "identifier": "[R_PAREN]",
1341
+ "name": null
1342
+ },
1343
+ {
1344
+ "type": "terminal",
1345
+ "token": {
1346
+ "column": 85,
1347
+ "lexeme": "STAR",
1348
+ "line": 7,
1349
+ "type": 0
1350
+ },
1351
+ "identifier": "[STAR]",
1352
+ "name": "modifier"
1187
1353
  }
1188
1354
  ]
1189
1355
  },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@scinorandex/sparse",
3
- "version": "0.0.7",
3
+ "version": "0.0.8",
4
4
  "description": "Yet another parser generator",
5
5
  "main": "dist/index.js",
6
6
  "scripts": {
@@ -60,6 +60,8 @@ export enum GrammarTokenType {
60
60
  R_PAREN,
61
61
  PIPE,
62
62
  QUESTION_MARK,
63
+ STAR,
64
+ PLUS,
63
65
  COLON,
64
66
  SEMICOLON,
65
67
  NUMBER,
@@ -38,6 +38,8 @@ grammarLexerGenerator.addRule("pipe", "$|", GrammarTokenType.PIPE);
38
38
  grammarLexerGenerator.addRule("equals", "$=", GrammarTokenType.EQUALS);
39
39
  grammarLexerGenerator.addRule("comma", "$,", GrammarTokenType.COMMA);
40
40
  grammarLexerGenerator.addRule("question_mark", "$?", GrammarTokenType.QUESTION_MARK);
41
+ grammarLexerGenerator.addRule("star", "$*", GrammarTokenType.STAR);
42
+ grammarLexerGenerator.addRule("plus", "$+", GrammarTokenType.PLUS);
41
43
  grammarLexerGenerator.addRule("number", "(${digit})+", GrammarTokenType.NUMBER);
42
44
 
43
45
  grammarLexerGenerator.addRule("single_line_comment", "$/$/(($\n)!)*", GrammarTokenType.SINGLE_LINE_COMMENT);
@@ -74,11 +76,16 @@ class ProductionNode extends BaseNode {
74
76
  public readonly lhs: GrammarToken,
75
77
  public readonly rhs: ListNode<TokenNode | GroupedTokenNode>,
76
78
  public originalProductionIndex: number,
79
+ public readonly identifier: string,
77
80
  public readonly name?: string | undefined,
78
81
  ) {
79
82
  super();
80
83
  }
81
84
 
85
+ fork(rhs: ListNode<TokenNode | GroupedTokenNode>) {
86
+ return new ProductionNode(this.lhs, rhs, this.originalProductionIndex, this.identifier, this.name);
87
+ }
88
+
82
89
  toStruct(): Production {
83
90
  if (this.originalProductionIndex === -1) throw new Error("ProductionNode has no originalProductionIndex");
84
91
 
@@ -87,7 +94,7 @@ class ProductionNode extends BaseNode {
87
94
 
88
95
  return {
89
96
  lhs: this.lhs,
90
- identifier: `<${this.lhs.lexeme}>`,
97
+ identifier: `<${this.identifier}>`,
91
98
  name: this.name ?? null,
92
99
  originalProductionIndex: this.originalProductionIndex,
93
100
  rhs: this.rhs.getItemsReversed().map((node) => (node as TokenNode).toStruct()),
@@ -99,7 +106,21 @@ class ProductionNode extends BaseNode {
99
106
  }
100
107
  }
101
108
 
102
- function expandProduction(production: ProductionNode): ProductionNode[] {
109
+ class ExpanderContext {
110
+ current = 0;
111
+
112
+ constructor(public currentProductionIndex: number) {}
113
+
114
+ getNextProductionIndex() {
115
+ return this.currentProductionIndex++;
116
+ }
117
+
118
+ getAutogeneratedId() {
119
+ return `autogen-${this.current++}`;
120
+ }
121
+ }
122
+
123
+ function expandProduction(production: ProductionNode, ctx: ExpanderContext): ProductionNode[] {
103
124
  const rhs = production.rhs.getItemsReversed();
104
125
 
105
126
  for (let i = 0; i < rhs.length; i++) {
@@ -109,35 +130,53 @@ function expandProduction(production: ProductionNode): ProductionNode[] {
109
130
  const afterItems = rhs.slice(i + 1);
110
131
 
111
132
  if (item instanceof GroupedTokenNode) {
112
- // need to create two new productions, one optional and the other not
113
- // then apply unrolling on both
114
-
115
- const withoutCurrentItem = new ProductionNode(
116
- production.lhs,
117
- new ListNode<TokenNode | GroupedTokenNode>([...beforeItems, ...afterItems].toReversed()),
118
- production.originalProductionIndex,
119
- production.name,
120
- );
121
-
122
- const withCurrentItem = new ProductionNode(
123
- production.lhs,
124
- new ListNode<TokenNode | GroupedTokenNode>(
125
- [...beforeItems, ...item.inside.getItemsReversed(), ...afterItems].toReversed(),
126
- ),
127
- production.originalProductionIndex,
128
- production.name,
129
- );
130
-
131
- const newProductions: ProductionNode[] = [withoutCurrentItem, withCurrentItem];
132
- return newProductions.flatMap((prod) => expandProduction(prod));
133
+ // The unrolling rules here depend on the modifier in the GroupedTokenNode
134
+
135
+ if (item.modifier === GrammarTokenType.QUESTION_MARK) {
136
+ const withoutCurrentItem = production.fork(
137
+ new ListNode<TokenNode | GroupedTokenNode>([...beforeItems, ...afterItems].toReversed()),
138
+ );
139
+
140
+ const withCurrentItem = production.fork(
141
+ new ListNode<TokenNode | GroupedTokenNode>(
142
+ [...beforeItems, ...item.inside.getItemsReversed(), ...afterItems].toReversed(),
143
+ ),
144
+ );
145
+
146
+ const newProductions: ProductionNode[] = [withoutCurrentItem, withCurrentItem];
147
+ return newProductions.flatMap((prod) => expandProduction(prod, ctx));
148
+ } else if (item.modifier === GrammarTokenType.STAR || item.modifier === GrammarTokenType.PLUS) {
149
+ // need to handle kleene-star and kleene-plus, which create entirely unique new productions
150
+ const inside = item.inside.getItemsReversed();
151
+ const newProductionIdentifier = ctx.getAutogeneratedId();
152
+
153
+ // create a dummy production that contains the contents of inside
154
+ const raw = new StubTokenNode("variable", new ListNode([production.lhs]), "rest", newProductionIdentifier);
155
+ const pump = new GroupedTokenNode(new ListNode([raw]), GrammarTokenType.QUESTION_MARK);
156
+
157
+ const stub = new ProductionNode(
158
+ production.lhs,
159
+ new ListNode<TokenNode | GroupedTokenNode>([...inside, pump].toReversed()),
160
+ ctx.getNextProductionIndex(),
161
+ newProductionIdentifier,
162
+ "autogenerated-kleene", // the user needs to implement a reducer named "autogenerated-kleene"
163
+ );
164
+
165
+ const currentReplacement = production.fork(
166
+ new ListNode(
167
+ [...beforeItems, item.modifier === GrammarTokenType.STAR ? pump : raw, ...afterItems].toReversed(),
168
+ ),
169
+ );
170
+
171
+ return [currentReplacement, stub].flatMap((prod) => expandProduction(prod, ctx));
172
+ } else throw new Error(`Unknown modifier ${GrammarTokenType[item.modifier]}`);
133
173
  } else if (item instanceof TokenNode) {
134
174
  if (item.variables.getItems().length > 1) {
135
175
  // this token node has many variants and we should unroll it
136
176
  const variables = item.variables.getItemsReversed();
137
177
 
138
178
  const newProductions = variables.map((variable) => {
139
- return new ProductionNode(
140
- production.lhs,
179
+ return production.fork(
141
180
  new ListNode<TokenNode | GroupedTokenNode>(
142
181
  [
143
182
  ...beforeItems,
@@ -145,12 +184,10 @@ function expandProduction(production: ProductionNode): ProductionNode[] {
145
184
  ...afterItems,
146
185
  ].toReversed(),
147
186
  ),
148
- production.originalProductionIndex,
149
- production.name,
150
187
  );
151
188
  });
152
189
 
153
- return newProductions.flatMap((prod) => expandProduction(prod));
190
+ return newProductions.flatMap((prod) => expandProduction(prod, ctx));
154
191
  }
155
192
  } else {
156
193
  throw new Error("shoudn't reach here, ProductionNode RHS is neither GroupedTokenNode nor TokenNode");
@@ -170,11 +207,17 @@ class TokenNode extends BaseNode {
170
207
  super();
171
208
  }
172
209
 
210
+ public getIdentifier() {
211
+ const items = this.variables.getItems();
212
+ if (items.length > 1) throw new Error("Not yet unrolled");
213
+ return items[0].lexeme;
214
+ }
215
+
173
216
  toStruct() {
174
217
  const items = this.variables.getItems();
175
218
  if (items.length > 1) throw new Error("Not yet unrolled");
176
219
 
177
- const lexeme = items[0].lexeme;
220
+ const lexeme = this.getIdentifier();
178
221
  const identifier = this.type === "variable" ? `<${lexeme}>` : `[${lexeme}]`;
179
222
  const name = this.name ?? null;
180
223
 
@@ -182,8 +225,26 @@ class TokenNode extends BaseNode {
182
225
  }
183
226
  }
184
227
 
228
+ class StubTokenNode extends TokenNode {
229
+ constructor(
230
+ public readonly type: "terminal" | "variable",
231
+ public readonly variables: ListNode<GrammarToken>,
232
+ public readonly name: string | undefined,
233
+ public readonly identifier: string,
234
+ ) {
235
+ super(type, variables, name);
236
+ }
237
+
238
+ public getIdentifier() {
239
+ return this.identifier;
240
+ }
241
+ }
242
+
185
243
  class GroupedTokenNode extends BaseNode {
186
- constructor(public readonly inside: ListNode<TokenNode>) {
244
+ constructor(
245
+ public readonly inside: ListNode<TokenNode>,
246
+ public readonly modifier: GrammarTokenType,
247
+ ) {
187
248
  super();
188
249
  }
189
250
  }
@@ -195,8 +256,10 @@ class ProgramNode extends BaseNode {
195
256
 
196
257
  unrollProductions() {
197
258
  const productions = this.productions.getItemsReversed();
259
+ const expanderContext = new ExpanderContext(this.productions.getItems().length);
260
+
198
261
  for (let i = 0; i < productions.length; i++) productions[i].setOriginalProductionIndex(i);
199
- return productions.flatMap((node) => expandProduction(node));
262
+ return productions.flatMap((node) => expandProduction(node, expanderContext));
200
263
  }
201
264
 
202
265
  getProductions() {
@@ -226,7 +289,7 @@ const reducers: { [key: string]: Reducer } = {
226
289
  production_name: GrammarToken;
227
290
  tokens: ListNode<TokenNode | GroupedTokenNode>;
228
291
  uuid?: GrammarToken;
229
- }) => new ProductionNode(bag.production_name, bag.tokens, -1, bag.uuid?.lexeme),
292
+ }) => new ProductionNode(bag.production_name, bag.tokens, -1, bag.production_name.lexeme, bag.uuid?.lexeme),
230
293
 
231
294
  tokens: (bag: { token: TokenNode | GroupedTokenNode; rest?: ListNode<TokenNode | GroupedTokenNode> }) => {
232
295
  if (bag.rest == null) return new ListNode<TokenNode | GroupedTokenNode>([bag.token]);
@@ -234,7 +297,8 @@ const reducers: { [key: string]: Reducer } = {
234
297
  },
235
298
 
236
299
  token: (bag: { token: TokenNode }) => bag.token,
237
- grouped_token: (bag: { tokens: ListNode<TokenNode> }) => new GroupedTokenNode(bag.tokens),
300
+ grouped_token: (bag: { tokens: ListNode<TokenNode>; modifier: GrammarToken }) =>
301
+ new GroupedTokenNode(bag.tokens, bag.modifier.type),
238
302
 
239
303
  variable: (bag: { inside: ListNode<GrammarToken>; token_name?: GrammarToken }) =>
240
304
  new TokenNode("variable", bag.inside, bag.token_name?.lexeme),