soml-lang 0.0.2 → 0.0.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,7 @@
1
- import { parse } from "./parse.js";
2
- import { isBareKeyCharacter, findNumberEnd, findLineEnd, findBlockStringEnd, } from "./shared.js";
1
+ import { parseWithTimes, } from "./parse.js";
2
+ import { isBareKeyCharacter, isSpace, findNumberEnd, findLineEnd, findBlockStringEnd, LF, DOUBLE_QUOTE, HASH, SINGLE_QUOTE, ASTERISK, COMMA, DOT, SLASH, BACKSLASH, OPEN_BRACKET, CLOSE_BRACKET, OPEN_BRACE, CLOSE_BRACE, } from "./shared.js";
3
3
  /**
4
- The properties of each node type that hold its child nodes, in source order. For tools that walk the tree, such as an ESLint language plugin.
4
+ The properties of each node type that hold its child nodes, in source order, for tools that walk the tree, such as an ESLint language plugin.
5
5
 
6
6
  @example
7
7
  ```
@@ -31,31 +31,16 @@ export const visitorKeys = Object.freeze({
31
31
  Duration: Object.freeze([]),
32
32
  });
33
33
  /* eslint-enable @typescript-eslint/naming-convention */
34
- const TAB = 0x09;
35
- const LF = 0x0A;
36
- const SPACE = 0x20;
37
- const DOUBLE_QUOTE = 0x22;
38
- const HASH = 0x23;
39
- const SINGLE_QUOTE = 0x27;
40
- const ASTERISK = 0x2A;
41
- const COMMA = 0x2C;
42
- const DOT = 0x2E;
43
- const SLASH = 0x2F;
44
- const BACKSLASH = 0x5C;
45
- const OPEN_BRACKET = 0x5B;
46
- const CLOSE_BRACKET = 0x5D;
47
- const OPEN_BRACE = 0x7B;
48
- const CLOSE_BRACE = 0x7D;
49
- const WHITESPACE = new Set([SPACE, TAB, LF]);
50
34
  /*
51
35
  `infinity` and `-infinity` are read as numbers, because they are floats.
52
36
  */
53
37
  const KEYWORDS = ['true', 'false', 'null'];
54
- const RADIX = { x: 16, o: 8, b: 2 };
38
+ // A map rather than an object, so that a property added to `Object.prototype` is never read as an entry.
39
+ const RADIX = new Map([['x', 16], ['o', 8], ['b', 2]]);
55
40
  /**
56
- Parse a document into a syntax tree, for tools such as linters and formatters.
41
+ Parse a document into a syntax tree, for tools such as linters and formatters. Returns a `Document` node, which also holds every token and comment.
57
42
 
58
- Every node, token, and comment has a `range` and a `loc`.
43
+ Every node, token, and comment has a `range`, which is `[start, end]` as UTF-16 offsets into `text`, and a `loc`, which is `{start: {line, column}, end: {line, column}}`, with a 1-based line and a 0-based column in UTF-16 code units, as in ESTree. A `ParseError` counts its column differently, for people to read, so use its `offset` to find the position in the tree. A scalar node or a key segment shares its `range` and `loc` with its token, and other nodes, except `Document`, share the positions in `loc` with their first and last token, so treat them as read-only.
59
44
 
60
45
  @param text - The document.
61
46
  @returns The root node, which also holds every token and comment.
@@ -80,21 +65,26 @@ export function parseTree(text) {
80
65
  throw new TypeError(`Expected a string, got ${text === null ? 'null' : typeof text}`);
81
66
  }
82
67
  // Validates the whole document, so the builder below can assume valid input.
83
- parse(text);
68
+ parseWithTimes(text);
84
69
  return new TreeBuilder(text).build();
85
70
  }
86
71
  /*
87
- The value of one scalar, read by the parser itself so that the tree and `parse()` cannot disagree. A scalar's validity does not depend on its context, so it parses on its own inside an array.
72
+ The value of one scalar, read by the parser itself so that the tree and `parse()` cannot disagree. A scalar's validity does not depend on its context, so it parses on its own inside an array. An instant or a duration is a `Time`.
88
73
  */
89
74
  function readScalar(raw) {
90
- return parse(`[${raw}]`)[0];
75
+ return parseWithTimes(`[${raw}]`)[0];
91
76
  }
77
+ /*
78
+ Every node is an object literal with the same property order, `type`, its own fields, `range`, and `loc`, rather than one generic function that spreads the fields, which is several times slower. A node that is one token, such as a scalar or a key segment, shares the token's `range` and `loc`.
79
+ */
92
80
  class TreeBuilder {
93
81
  #source;
94
82
  #index = 0;
95
83
  #tokens = [];
96
84
  #comments = [];
97
85
  #lineStarts = [0];
86
+ // The line of the last offset looked up. Offsets mostly arrive in order, so it is the best first guess for the next one.
87
+ #line = 0;
98
88
  constructor(source) {
99
89
  this.#source = source;
100
90
  for (let index = source.indexOf('\n'); index !== -1; index = source.indexOf('\n', index + 1)) {
@@ -109,29 +99,40 @@ class TreeBuilder {
109
99
  */
110
100
  #position(offset) {
111
101
  const lineStarts = this.#lineStarts;
112
- let low = 0;
113
- let high = lineStarts.length - 1;
114
- while (low < high) {
115
- const middle = Math.ceil((low + high) / 2);
116
- if (lineStarts[middle] <= offset) {
117
- low = middle;
118
- }
119
- else {
120
- high = middle - 1;
102
+ // The binary search runs only when the offset is not on the line of the last one or the line after it, which makes building the tree about twice as fast.
103
+ if (this.#isOnLine(offset, this.#line + 1)) {
104
+ this.#line++;
105
+ }
106
+ else if (!this.#isOnLine(offset, this.#line)) {
107
+ let low = 0;
108
+ let high = lineStarts.length - 1;
109
+ while (low < high) {
110
+ const middle = Math.ceil((low + high) / 2);
111
+ if (lineStarts[middle] <= offset) {
112
+ low = middle;
113
+ }
114
+ else {
115
+ high = middle - 1;
116
+ }
121
117
  }
118
+ this.#line = low;
122
119
  }
123
- return { line: low + 1, column: offset - lineStarts[low] };
120
+ return { line: this.#line + 1, column: offset - lineStarts[this.#line] };
124
121
  }
125
- #node(type, start, end, fields) {
126
- return {
122
+ #isOnLine(offset, line) {
123
+ const lineStarts = this.#lineStarts;
124
+ return line < lineStarts.length && lineStarts[line] <= offset && (line + 1 === lineStarts.length || lineStarts[line + 1] > offset);
125
+ }
126
+ #location(start, end) {
127
+ return { start: this.#position(start), end: this.#position(end) };
128
+ }
129
+ #token(type, start, end) {
130
+ const token = {
127
131
  type,
128
- ...fields,
132
+ value: this.#source.slice(start, end),
129
133
  range: [start, end],
130
- loc: { start: this.#position(start), end: this.#position(end) },
134
+ loc: this.#location(start, end),
131
135
  };
132
- }
133
- #token(type, start, end) {
134
- const token = this.#node(type, start, end, { value: this.#source.slice(start, end) });
135
136
  this.#tokens.push(token);
136
137
  return token;
137
138
  }
@@ -144,48 +145,73 @@ class TreeBuilder {
144
145
  for (;;) {
145
146
  const code = this.#code();
146
147
  const start = this.#index;
147
- if (WHITESPACE.has(code)) {
148
+ if (code === LF || isSpace(code)) {
148
149
  this.#index++;
149
150
  }
150
151
  else if (code === HASH) {
151
152
  this.#index = findLineEnd(source, start);
152
- this.#comments.push(this.#node('Line', start, this.#index, { value: source.slice(start + 1, this.#index) }));
153
+ this.#comment('Line', start, this.#index, source.slice(start + 1, this.#index));
153
154
  }
154
155
  else if (code === SLASH && this.#code(start + 1) === ASTERISK) {
155
156
  this.#index = source.indexOf('*/', start + 2) + 2;
156
- this.#comments.push(this.#node('Block', start, this.#index, { value: source.slice(start + 2, this.#index - 2) }));
157
+ this.#comment('Block', start, this.#index, source.slice(start + 2, this.#index - 2));
157
158
  }
158
159
  else {
159
160
  return;
160
161
  }
161
162
  }
162
163
  }
164
+ #comment(type, start, end, value) {
165
+ this.#comments.push({
166
+ type,
167
+ value,
168
+ range: [start, end],
169
+ loc: this.#location(start, end),
170
+ });
171
+ }
163
172
  #bareObject() {
164
173
  const members = [];
165
174
  do {
166
175
  members.push(this.#member());
167
176
  this.#skipTrivia();
168
177
  } while (this.#index < this.#source.length);
169
- return this.#node('Object', members[0].range[0], members.at(-1).range[1], { members, braced: false });
178
+ const first = members[0];
179
+ const last = members.at(-1);
180
+ return {
181
+ type: 'Object',
182
+ members,
183
+ braced: false,
184
+ range: [first.range[0], last.range[1]],
185
+ loc: spanLocation(first, last),
186
+ };
170
187
  }
171
188
  #object() {
172
- const start = this.#index;
173
- const members = this.#items(CLOSE_BRACE, () => this.#member());
174
- return this.#node('Object', start, this.#index, { members, braced: true });
189
+ const { items: members, opening, closing } = this.#items(CLOSE_BRACE, () => this.#member());
190
+ return {
191
+ type: 'Object',
192
+ members,
193
+ braced: true,
194
+ range: [opening.range[0], closing.range[1]],
195
+ loc: spanLocation(opening, closing),
196
+ };
175
197
  }
176
198
  #array() {
177
- const start = this.#index;
178
- const elements = this.#items(CLOSE_BRACKET, () => this.#value());
179
- return this.#node('Array', start, this.#index, { elements });
199
+ const { items: elements, opening, closing } = this.#items(CLOSE_BRACKET, () => this.#value());
200
+ return {
201
+ type: 'Array',
202
+ elements,
203
+ range: [opening.range[0], closing.range[1]],
204
+ loc: spanLocation(opening, closing),
205
+ };
180
206
  }
181
207
  /*
182
- The items between an opening bracket and its `closing` bracket, which is consumed too. The parser has already checked the commas.
208
+ The items between an opening bracket and its `closingCode` bracket, which is consumed too. Items are separated by a comma, a line break, or both, so a comma token may be absent. The parser has already checked the separators, including that a comma is on the line of the item before it.
183
209
  */
184
- #items(closing, parseItem) {
210
+ #items(closingCode, parseItem) {
185
211
  const items = [];
186
- this.#punctuator();
212
+ const opening = this.#punctuator();
187
213
  this.#skipTrivia();
188
- while (this.#code() !== closing) {
214
+ while (this.#code() !== closingCode) {
189
215
  items.push(parseItem());
190
216
  this.#skipTrivia();
191
217
  if (this.#code() !== COMMA) {
@@ -194,15 +220,20 @@ class TreeBuilder {
194
220
  this.#punctuator();
195
221
  this.#skipTrivia();
196
222
  }
197
- this.#punctuator();
198
- return items;
223
+ return { items, opening, closing: this.#punctuator() };
199
224
  }
200
225
  #member() {
201
226
  const key = this.#key();
202
227
  this.#punctuator(); // `:`
203
228
  this.#skipTrivia();
204
229
  const value = this.#value();
205
- return this.#node('Member', key.range[0], value.range[1], { key, value });
230
+ return {
231
+ type: 'Member',
232
+ key,
233
+ value,
234
+ range: [key.range[0], value.range[1]],
235
+ loc: spanLocation(key, value),
236
+ };
206
237
  }
207
238
  #key() {
208
239
  const segments = [this.#keySegment()];
@@ -210,20 +241,39 @@ class TreeBuilder {
210
241
  this.#punctuator();
211
242
  segments.push(this.#keySegment());
212
243
  }
213
- return this.#node('Key', segments[0].range[0], segments.at(-1).range[1], { segments });
244
+ const first = segments[0];
245
+ const last = segments.at(-1);
246
+ return {
247
+ type: 'Key',
248
+ segments,
249
+ range: [first.range[0], last.range[1]],
250
+ loc: spanLocation(first, last),
251
+ };
214
252
  }
215
253
  #keySegment() {
216
254
  const start = this.#index;
217
255
  const code = this.#code();
218
256
  if (code === SINGLE_QUOTE || code === DOUBLE_QUOTE) {
219
- const { value, style } = this.#singleLineString();
220
- return this.#node('KeySegment', start, this.#index, { value, style });
257
+ const { value, style, token: { range, loc } } = this.#singleLineString();
258
+ return {
259
+ type: 'KeySegment',
260
+ value,
261
+ style,
262
+ range,
263
+ loc,
264
+ };
221
265
  }
222
266
  while (isBareKeyCharacter(this.#code())) {
223
267
  this.#index++;
224
268
  }
225
- const token = this.#token('BareKey', start, this.#index);
226
- return this.#node('KeySegment', start, this.#index, { value: token.value, style: 'bare' });
269
+ const { value, range, loc } = this.#token('BareKey', start, this.#index);
270
+ return {
271
+ type: 'KeySegment',
272
+ value,
273
+ style: 'bare',
274
+ range,
275
+ loc,
276
+ };
227
277
  }
228
278
  #value() {
229
279
  const code = this.#code();
@@ -238,16 +288,30 @@ class TreeBuilder {
238
288
  if (this.#code(start + 1) === code && this.#code(start + 2) === code) {
239
289
  return this.#blockString(code);
240
290
  }
241
- const { value, style } = this.#singleLineString();
242
- return this.#node('String', start, this.#index, { value, style, block: false });
291
+ const { value, style, token: { range, loc } } = this.#singleLineString();
292
+ return {
293
+ type: 'String',
294
+ value,
295
+ style,
296
+ block: false,
297
+ range,
298
+ loc,
299
+ };
243
300
  }
244
- const keyword = KEYWORDS.find(word => this.#source.startsWith(word, start) && !isBareKeyCharacter(this.#code(start + word.length)));
301
+ // The document is valid, so a value that starts with a keyword is that keyword, as a number cannot start with a letter other than the `i` of `infinity`.
302
+ const keyword = KEYWORDS.find(word => this.#source.startsWith(word, start));
245
303
  if (keyword !== undefined) {
246
304
  this.#index += keyword.length;
247
- this.#token('Keyword', start, this.#index);
248
- return keyword === 'null'
249
- ? this.#node('Null', start, this.#index, {})
250
- : this.#node('Boolean', start, this.#index, { value: keyword === 'true' });
305
+ const { range, loc } = this.#token('Keyword', start, this.#index);
306
+ if (keyword === 'null') {
307
+ return { type: 'Null', range, loc };
308
+ }
309
+ return {
310
+ type: 'Boolean',
311
+ value: keyword === 'true',
312
+ range,
313
+ loc,
314
+ };
251
315
  }
252
316
  return this.#numberOrTime();
253
317
  }
@@ -270,8 +334,8 @@ class TreeBuilder {
270
334
  this.#index = end + 1;
271
335
  const token = this.#token('String', start, this.#index);
272
336
  return quote === SINGLE_QUOTE
273
- ? { value: source.slice(start + 1, end), style: 'literal' }
274
- : { value: readScalar(token.value), style: 'escaped' };
337
+ ? { value: source.slice(start + 1, end), style: 'literal', token }
338
+ : { value: readScalar(token.value), style: 'escaped', token };
275
339
  }
276
340
  /*
277
341
  The block ends at the first line whose first non-whitespace content is a run of exactly as many quotes as the opening delimiter.
@@ -285,12 +349,15 @@ class TreeBuilder {
285
349
  }
286
350
  const closing = findBlockStringEnd(source, source.indexOf('\n', start) + 1, quote, delimiterLength);
287
351
  this.#index = closing.delimiterStart + delimiterLength;
288
- const token = this.#token('String', start, this.#index);
289
- return this.#node('String', start, this.#index, {
290
- value: readScalar(token.value),
352
+ const { value, range, loc } = this.#token('String', start, this.#index);
353
+ return {
354
+ type: 'String',
355
+ value: readScalar(value),
291
356
  style: quote === SINGLE_QUOTE ? 'literal' : 'escaped',
292
357
  block: true,
293
- });
358
+ range,
359
+ loc,
360
+ };
294
361
  }
295
362
  #numberOrTime() {
296
363
  const start = this.#index;
@@ -299,16 +366,53 @@ class TreeBuilder {
299
366
  const raw = this.#source.slice(start, end);
300
367
  const value = readScalar(raw);
301
368
  if (typeof value === 'bigint') {
302
- this.#token('Integer', start, end);
303
- return this.#node('Integer', start, end, { value, radix: RADIX[raw.charAt(1)] ?? 10 });
369
+ const { range, loc } = this.#token('Integer', start, end);
370
+ return {
371
+ type: 'Integer',
372
+ value,
373
+ radix: RADIX.get(raw.charAt(1)) ?? 10,
374
+ range,
375
+ loc,
376
+ };
304
377
  }
305
378
  if (typeof value === 'number') {
306
- this.#token('Float', start, end);
307
- return this.#node('Float', start, end, { value });
379
+ const { range, loc } = this.#token('Float', start, end);
380
+ return {
381
+ type: 'Float',
382
+ value,
383
+ range,
384
+ loc,
385
+ };
386
+ }
387
+ const time = value;
388
+ const { range, loc } = this.#token(time.type, start, end);
389
+ let temporal;
390
+ if (time.type === 'Duration') {
391
+ // A getter, so that the `Temporal` object is made when `value` is first read.
392
+ const node = {
393
+ type: 'Duration',
394
+ get value() {
395
+ temporal ??= time.toTemporal();
396
+ return temporal;
397
+ },
398
+ negative: raw.startsWith('-'),
399
+ parts: readDurationParts(raw),
400
+ range,
401
+ loc,
402
+ };
403
+ return node;
308
404
  }
309
- const type = value instanceof Temporal.Instant ? 'Instant' : 'Duration';
310
- this.#token(type, start, end);
311
- return this.#node(type, start, end, { value });
405
+ // A getter, so that the `Temporal` object is made when `value` is first read.
406
+ const node = {
407
+ type: 'Instant',
408
+ get value() {
409
+ temporal ??= time.toTemporal();
410
+ return temporal;
411
+ },
412
+ range,
413
+ loc,
414
+ };
415
+ return node;
312
416
  }
313
417
  build() {
314
418
  this.#skipTrivia();
@@ -324,10 +428,41 @@ class TreeBuilder {
324
428
  body = this.#bareObject();
325
429
  }
326
430
  this.#skipTrivia();
327
- return this.#node('Document', 0, this.#source.length, {
431
+ return {
432
+ type: 'Document',
328
433
  body,
329
434
  tokens: this.#tokens,
330
435
  comments: this.#comments,
331
- });
436
+ range: [0, this.#source.length],
437
+ loc: this.#location(0, this.#source.length),
438
+ };
439
+ }
440
+ }
441
+ /*
442
+ The location from the start of `first` to the end of `last`, which shares their positions rather than looking them up again. A node that ends far from where it starts, such as a large object, would otherwise move the line cache in `#position` back and forth.
443
+ */
444
+ function spanLocation(first, last) {
445
+ return { start: first.loc.start, end: last.loc.end };
446
+ }
447
+ /*
448
+ The parts of a duration that is already valid. A part is a number, which has no letters, and then a unit, which has only lowercase letters.
449
+ */
450
+ function readDurationParts(raw) {
451
+ // From `a` to `z`.
452
+ const isUnitCharacter = (index) => raw.charCodeAt(index) >= 0x61 && raw.charCodeAt(index) <= 0x7A;
453
+ const parts = [];
454
+ let index = raw.startsWith('-') ? 1 : 0;
455
+ while (index < raw.length) {
456
+ let unitStart = index;
457
+ while (!isUnitCharacter(unitStart)) {
458
+ unitStart++;
459
+ }
460
+ let unitEnd = unitStart;
461
+ while (unitEnd < raw.length && isUnitCharacter(unitEnd)) {
462
+ unitEnd++;
463
+ }
464
+ parts.push({ number: raw.slice(index, unitStart), unit: raw.slice(unitStart, unitEnd) });
465
+ index = unitEnd;
332
466
  }
467
+ return parts;
333
468
  }
package/package.json CHANGED
@@ -1,8 +1,9 @@
1
1
  {
2
2
  "name": "soml-lang",
3
- "version": "0.0.2",
4
- "description": "Reference parser, canonical serializer, and formatter for SOML, a config format for humans",
3
+ "version": "0.0.3",
4
+ "description": "Reference parser, serializer, and formatter for SOML, a config format for humans",
5
5
  "license": "MIT",
6
+ "repository": "som-lang/soml-javascript",
6
7
  "author": "Sindre Sorhus <sindresorhus@gmail.com> (https://sindresorhus.com)",
7
8
  "type": "module",
8
9
  "exports": {
@@ -16,7 +17,7 @@
16
17
  "scripts": {
17
18
  "build": "tsc --project tsconfig.build.json",
18
19
  "prepack": "npm run build",
19
- "test": "xo && tsc && node --test",
20
+ "test": "xo && tsc && node --test 'test/*.ts'",
20
21
  "bench": "node bench/index.ts"
21
22
  },
22
23
  "files": [