soml-lang 0.0.1 → 0.0.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/distribution/edit.d.ts +56 -0
- package/distribution/edit.js +635 -0
- package/distribution/error.d.ts +49 -0
- package/distribution/error.js +156 -0
- package/distribution/format.d.ts +62 -0
- package/distribution/format.js +403 -0
- package/distribution/index.d.ts +8 -0
- package/distribution/index.js +10 -0
- package/distribution/parse.d.ts +131 -0
- package/distribution/parse.js +1587 -0
- package/distribution/shared.d.ts +102 -0
- package/distribution/shared.js +285 -0
- package/distribution/stringify.d.ts +115 -0
- package/distribution/stringify.js +473 -0
- package/distribution/tree.d.ts +264 -0
- package/distribution/tree.js +468 -0
- package/package.json +34 -34
- package/readme.md +326 -39
- package/index.d.ts +0 -151
- package/index.js +0 -3
- package/source/error.js +0 -130
- package/source/parse.js +0 -1574
- package/source/shared.js +0 -166
- package/source/stringify.js +0 -439
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
/**
|
|
2
|
+
Thrown when the input is not a valid document, by `parse()`, `parseTree()`, `format()`, and `edit()`. Extends `SyntaxError`.
|
|
3
|
+
|
|
4
|
+
The `message` includes the position and a code frame. Use `reason` for the message alone.
|
|
5
|
+
|
|
6
|
+
@example
|
|
7
|
+
```
|
|
8
|
+
import {parse, ParseError} from 'soml-lang';
|
|
9
|
+
|
|
10
|
+
try {
|
|
11
|
+
parse('name: api-gateway');
|
|
12
|
+
} catch (error) {
|
|
13
|
+
if (error instanceof ParseError) {
|
|
14
|
+
console.log(error.message);
|
|
15
|
+
}
|
|
16
|
+
}
|
|
17
|
+
// Unexpected "api-gateway". A string value must be quoted, as in 'api-gateway' at line 1, column 7
|
|
18
|
+
//
|
|
19
|
+
// > 1 | name: api-gateway
|
|
20
|
+
// | ^
|
|
21
|
+
```
|
|
22
|
+
*/
|
|
23
|
+
export declare class ParseError extends SyntaxError {
|
|
24
|
+
readonly name = "ParseError";
|
|
25
|
+
/**
|
|
26
|
+
What is wrong, without the position.
|
|
27
|
+
*/
|
|
28
|
+
readonly reason: string;
|
|
29
|
+
/**
|
|
30
|
+
The 1-based line of the error.
|
|
31
|
+
*/
|
|
32
|
+
readonly line: number;
|
|
33
|
+
/**
|
|
34
|
+
The 1-based column of the error, counted in Unicode code points.
|
|
35
|
+
*/
|
|
36
|
+
readonly column: number;
|
|
37
|
+
/**
|
|
38
|
+
The 0-based UTF-16 index of the error in the decoded text.
|
|
39
|
+
*/
|
|
40
|
+
readonly offset: number;
|
|
41
|
+
/**
|
|
42
|
+
Up to three lines ending at the error, with a caret under the position. A long line is clipped around the position.
|
|
43
|
+
*/
|
|
44
|
+
readonly codeFrame: string;
|
|
45
|
+
/**
|
|
46
|
+
Only the parser creates a `ParseError`.
|
|
47
|
+
*/
|
|
48
|
+
private constructor();
|
|
49
|
+
}
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
import { findLineEnd } from "./shared.js";
|
|
2
|
+
const CONTEXT_LINES = 2;
|
|
3
|
+
const MAX_LINE_WIDTH = 100;
|
|
4
|
+
// With the `v` flag, a surrogate in the class matches only when it is unpaired.
|
|
5
|
+
const UNPRINTABLE = /[\p{Bidi_Control}\u{D800}-\u{DFFF}[\p{Control}--\t]]/gv;
|
|
6
|
+
/**
|
|
7
|
+
Thrown when the input is not a valid document, by `parse()`, `parseTree()`, `format()`, and `edit()`. Extends `SyntaxError`.
|
|
8
|
+
|
|
9
|
+
The `message` includes the position and a code frame. Use `reason` for the message alone.
|
|
10
|
+
|
|
11
|
+
@example
|
|
12
|
+
```
|
|
13
|
+
import {parse, ParseError} from 'soml-lang';
|
|
14
|
+
|
|
15
|
+
try {
|
|
16
|
+
parse('name: api-gateway');
|
|
17
|
+
} catch (error) {
|
|
18
|
+
if (error instanceof ParseError) {
|
|
19
|
+
console.log(error.message);
|
|
20
|
+
}
|
|
21
|
+
}
|
|
22
|
+
// Unexpected "api-gateway". A string value must be quoted, as in 'api-gateway' at line 1, column 7
|
|
23
|
+
//
|
|
24
|
+
// > 1 | name: api-gateway
|
|
25
|
+
// | ^
|
|
26
|
+
```
|
|
27
|
+
*/
|
|
28
|
+
export class ParseError extends SyntaxError {
|
|
29
|
+
/**
|
|
30
|
+
For the parser, which cannot call the private constructor.
|
|
31
|
+
|
|
32
|
+
@internal
|
|
33
|
+
*/
|
|
34
|
+
static create(reason, source, offset) {
|
|
35
|
+
return new ParseError(reason, source, offset);
|
|
36
|
+
}
|
|
37
|
+
name = 'ParseError';
|
|
38
|
+
/**
|
|
39
|
+
What is wrong, without the position.
|
|
40
|
+
*/
|
|
41
|
+
reason;
|
|
42
|
+
/**
|
|
43
|
+
The 1-based line of the error.
|
|
44
|
+
*/
|
|
45
|
+
line;
|
|
46
|
+
/**
|
|
47
|
+
The 1-based column of the error, counted in Unicode code points.
|
|
48
|
+
*/
|
|
49
|
+
column;
|
|
50
|
+
/**
|
|
51
|
+
The 0-based UTF-16 index of the error in the decoded text.
|
|
52
|
+
*/
|
|
53
|
+
offset;
|
|
54
|
+
/**
|
|
55
|
+
Up to three lines ending at the error, with a caret under the position. A long line is clipped around the position.
|
|
56
|
+
*/
|
|
57
|
+
codeFrame;
|
|
58
|
+
/**
|
|
59
|
+
Only the parser creates a `ParseError`.
|
|
60
|
+
*/
|
|
61
|
+
constructor(reason, source, offset) {
|
|
62
|
+
const sanitizedReason = sanitize(reason);
|
|
63
|
+
const { line, column } = locate(source, offset);
|
|
64
|
+
const codeFrame = createCodeFrame(source, offset, line);
|
|
65
|
+
super(`${sanitizedReason} at line ${line}, column ${column}\n\n${codeFrame}`);
|
|
66
|
+
this.reason = sanitizedReason;
|
|
67
|
+
this.line = line;
|
|
68
|
+
this.column = column;
|
|
69
|
+
this.offset = offset;
|
|
70
|
+
this.codeFrame = codeFrame;
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
/*
|
|
74
|
+
Line and column are 1-based, and the column counts code points, so an emoji is one column.
|
|
75
|
+
*/
|
|
76
|
+
function locate(source, offset) {
|
|
77
|
+
let line = 1;
|
|
78
|
+
let lineStart = 0;
|
|
79
|
+
for (let index = source.indexOf('\n'); index !== -1 && index < offset; index = source.indexOf('\n', index + 1)) {
|
|
80
|
+
line++;
|
|
81
|
+
lineStart = index + 1;
|
|
82
|
+
}
|
|
83
|
+
return { line, column: countCodePoints(source.slice(lineStart, offset)) + 1 };
|
|
84
|
+
}
|
|
85
|
+
function countCodePoints(string) {
|
|
86
|
+
let count = 0;
|
|
87
|
+
for (let index = 0; index < string.length; index++) {
|
|
88
|
+
const code = string.charCodeAt(index);
|
|
89
|
+
// A high surrogate followed by a low one is one code point.
|
|
90
|
+
if (code >= 0xD8_00 && code <= 0xDB_FF) {
|
|
91
|
+
const next = string.charCodeAt(index + 1);
|
|
92
|
+
if (next >= 0xDC_00 && next <= 0xDF_FF) {
|
|
93
|
+
index++;
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
count++;
|
|
97
|
+
}
|
|
98
|
+
return count;
|
|
99
|
+
}
|
|
100
|
+
function createCodeFrame(source, offset, line) {
|
|
101
|
+
// The error line and up to two lines before it, found by scanning back from the error, so that an error late in a long document does not split the whole document.
|
|
102
|
+
const lineStarts = [offset === 0 ? 0 : source.lastIndexOf('\n', offset - 1) + 1];
|
|
103
|
+
while (lineStarts.length <= CONTEXT_LINES && lineStarts[0] > 0) {
|
|
104
|
+
const searchFrom = lineStarts[0] - 2;
|
|
105
|
+
lineStarts.unshift(searchFrom < 0 ? 0 : source.lastIndexOf('\n', searchFrom) + 1);
|
|
106
|
+
}
|
|
107
|
+
const firstLine = line - lineStarts.length + 1;
|
|
108
|
+
const gutterWidth = String(line).length;
|
|
109
|
+
const output = [];
|
|
110
|
+
for (const [index, lineStart] of lineStarts.entries()) {
|
|
111
|
+
const number = firstLine + index;
|
|
112
|
+
const isErrorLine = number === line;
|
|
113
|
+
const lineText = sanitize(source.slice(lineStart, findLineEnd(source, lineStart)));
|
|
114
|
+
const { text, pointerOffset } = clip(lineText, isErrorLine ? offset - lineStart : 0);
|
|
115
|
+
const gutter = `${isErrorLine ? '>' : ' '} ${String(number).padStart(gutterWidth)} |`;
|
|
116
|
+
output.push(text === '' ? gutter : `${gutter} ${text}`);
|
|
117
|
+
if (!isErrorLine) {
|
|
118
|
+
continue;
|
|
119
|
+
}
|
|
120
|
+
// Keep the tabs from the line, so the caret lines up whatever the tab width is.
|
|
121
|
+
const padding = text.slice(0, pointerOffset).replaceAll(/[^\t]/gv, ' ');
|
|
122
|
+
output.push(` ${' '.repeat(gutterWidth)} | ${padding}^`);
|
|
123
|
+
}
|
|
124
|
+
return output.join('\n');
|
|
125
|
+
}
|
|
126
|
+
/*
|
|
127
|
+
The rejected text can hold characters that a terminal acts on rather than shows: control characters such as an escape or a carriage return, bidirectional controls, and lone surrogates. Each becomes U+FFFD, which is also one code unit, so the caret still lines up. Tabs are kept.
|
|
128
|
+
*/
|
|
129
|
+
function sanitize(text) {
|
|
130
|
+
return text.replaceAll(UNPRINTABLE, '\u{FFFD}');
|
|
131
|
+
}
|
|
132
|
+
/*
|
|
133
|
+
A long line, such as a minified document, is cut to a window around the pointer. The cut never splits a surrogate pair.
|
|
134
|
+
*/
|
|
135
|
+
function clip(text, pointerOffset) {
|
|
136
|
+
if (text.length <= MAX_LINE_WIDTH) {
|
|
137
|
+
return { text, pointerOffset };
|
|
138
|
+
}
|
|
139
|
+
let start = Math.max(0, Math.min(pointerOffset - (MAX_LINE_WIDTH / 2), text.length - MAX_LINE_WIDTH));
|
|
140
|
+
let end = start + MAX_LINE_WIDTH;
|
|
141
|
+
if (isLowSurrogate(text.charCodeAt(start))) {
|
|
142
|
+
start++;
|
|
143
|
+
}
|
|
144
|
+
if (isLowSurrogate(text.charCodeAt(end))) {
|
|
145
|
+
end--;
|
|
146
|
+
}
|
|
147
|
+
const prefix = start > 0 ? '…' : '';
|
|
148
|
+
const suffix = end < text.length ? '…' : '';
|
|
149
|
+
return {
|
|
150
|
+
text: `${prefix}${text.slice(start, end)}${suffix}`,
|
|
151
|
+
pointerOffset: pointerOffset - start + prefix.length,
|
|
152
|
+
};
|
|
153
|
+
}
|
|
154
|
+
function isLowSurrogate(code) {
|
|
155
|
+
return code >= 0xDC_00 && code <= 0xDF_FF;
|
|
156
|
+
}
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
/**
|
|
2
|
+
Format a document. Returns the document with its layout normalized, ending with one line feed.
|
|
3
|
+
|
|
4
|
+
The layout follows the formatter in the specification: one tab per level, every member and item on its own line with no commas, and no trailing whitespace or runs of blank lines. An object or an array whose brackets are on one line stays on one line, as in `ports: [80, 443]`, with a comma and a space between its members or items. To give a one-line container one member or item per line, put a line break anywhere inside it. Comments, member order, dotted keys, block strings, and the spelling of every value stay as they are, so the value never changes. The one change inside a block comment is the same layout rule: trailing whitespace is removed, and runs of blank lines collapse to one.
|
|
5
|
+
|
|
6
|
+
In detail, as the specification states:
|
|
7
|
+
|
|
8
|
+
- A block comment that a value follows on the same line, as in `1, /* note *\/ 2`, stays in front of that value when each item goes on its own line.
|
|
9
|
+
- One space follows each `:`, as in canonical form.
|
|
10
|
+
- A value on the line after its `key:` moves up to that line, unless a comment comes between them. Then it stays on its own line, one level deeper than the key, and the blank lines between them are removed.
|
|
11
|
+
- A block string begins on the line after its key, and its delimiters and content have the indentation of that line. Its lines are not otherwise changed.
|
|
12
|
+
- There are no blank lines at the start of the file or directly inside brackets.
|
|
13
|
+
|
|
14
|
+
@param text - The document.
|
|
15
|
+
@returns The formatted document, ending with one line feed.
|
|
16
|
+
@throws {ParseError} When the document is not valid, the same as `parse()`.
|
|
17
|
+
@throws {TypeError} When `text` is not a string.
|
|
18
|
+
|
|
19
|
+
@example
|
|
20
|
+
```
|
|
21
|
+
import {format} from 'soml-lang';
|
|
22
|
+
|
|
23
|
+
format('pool: {min: 2, max: 16,} # Connections');
|
|
24
|
+
//=> 'pool: {min: 2, max: 16} # Connections\n'
|
|
25
|
+
|
|
26
|
+
format('pool: {\nmin: 2, max: 16}');
|
|
27
|
+
//=> 'pool: {\n\tmin: 2\n\tmax: 16\n}\n'
|
|
28
|
+
```
|
|
29
|
+
*/
|
|
30
|
+
export declare function format(text: string): string;
|
|
31
|
+
/**
|
|
32
|
+
A change to a document: a range of its text, and the text that replaces it.
|
|
33
|
+
*/
|
|
34
|
+
export type FormatEdit = {
|
|
35
|
+
/**
|
|
36
|
+
The start and end of the replaced text, as UTF-16 offsets into the document. They are equal when the edit only inserts.
|
|
37
|
+
*/
|
|
38
|
+
readonly range: readonly [start: number, end: number];
|
|
39
|
+
/**
|
|
40
|
+
The text that replaces the range. It is empty when the edit only removes.
|
|
41
|
+
*/
|
|
42
|
+
readonly text: string;
|
|
43
|
+
};
|
|
44
|
+
/**
|
|
45
|
+
The changes that `format()` makes to a document, as edits of its text. For an editor or a linter, which shows or applies each change where it is, rather than replacing the whole document.
|
|
46
|
+
|
|
47
|
+
The edits are sorted, do not overlap, and change only spaces, tabs, line feeds, and commas, as `format()` does. Each one is as small as possible, so it leaves out the characters at its ends that stay the same. Applying all of them gives the same text as `format()`, and a document that is already formatted gives no edits.
|
|
48
|
+
|
|
49
|
+
@param text - The document.
|
|
50
|
+
@returns The edits, in the order of their ranges.
|
|
51
|
+
@throws {ParseError} When the document is not valid, the same as `parse()`.
|
|
52
|
+
@throws {TypeError} When `text` is not a string.
|
|
53
|
+
|
|
54
|
+
@example
|
|
55
|
+
```
|
|
56
|
+
import {formatEdits} from 'soml-lang';
|
|
57
|
+
|
|
58
|
+
formatEdits('a: [1,\n2]\n');
|
|
59
|
+
//=> [{range: [4, 4], text: '\n\t'}, {range: [5, 7], text: '\n\t'}, {range: [8, 8], text: '\n'}]
|
|
60
|
+
```
|
|
61
|
+
*/
|
|
62
|
+
export declare function formatEdits(text: string): FormatEdit[];
|
|
@@ -0,0 +1,403 @@
|
|
|
1
|
+
import { parseTree, } from "./tree.js";
|
|
2
|
+
import { ASTERISK, isBlankLine, COMMA, HASH, SLASH, findLineEnd, isSpace, skipSpaces, skipSpacesBack, LF, } from "./shared.js";
|
|
3
|
+
const INDENT = '\t';
|
|
4
|
+
/**
|
|
5
|
+
Format a document. Returns the document with its layout normalized, ending with one line feed.
|
|
6
|
+
|
|
7
|
+
The layout follows the formatter in the specification: one tab per level, every member and item on its own line with no commas, and no trailing whitespace or runs of blank lines. An object or an array whose brackets are on one line stays on one line, as in `ports: [80, 443]`, with a comma and a space between its members or items. To give a one-line container one member or item per line, put a line break anywhere inside it. Comments, member order, dotted keys, block strings, and the spelling of every value stay as they are, so the value never changes. The one change inside a block comment is the same layout rule: trailing whitespace is removed, and runs of blank lines collapse to one.
|
|
8
|
+
|
|
9
|
+
In detail, as the specification states:
|
|
10
|
+
|
|
11
|
+
- A block comment that a value follows on the same line, as in `1, /* note *\/ 2`, stays in front of that value when each item goes on its own line.
|
|
12
|
+
- One space follows each `:`, as in canonical form.
|
|
13
|
+
- A value on the line after its `key:` moves up to that line, unless a comment comes between them. Then it stays on its own line, one level deeper than the key, and the blank lines between them are removed.
|
|
14
|
+
- A block string begins on the line after its key, and its delimiters and content have the indentation of that line. Its lines are not otherwise changed.
|
|
15
|
+
- There are no blank lines at the start of the file or directly inside brackets.
|
|
16
|
+
|
|
17
|
+
@param text - The document.
|
|
18
|
+
@returns The formatted document, ending with one line feed.
|
|
19
|
+
@throws {ParseError} When the document is not valid, the same as `parse()`.
|
|
20
|
+
@throws {TypeError} When `text` is not a string.
|
|
21
|
+
|
|
22
|
+
@example
|
|
23
|
+
```
|
|
24
|
+
import {format} from 'soml-lang';
|
|
25
|
+
|
|
26
|
+
format('pool: {min: 2, max: 16,} # Connections');
|
|
27
|
+
//=> 'pool: {min: 2, max: 16} # Connections\n'
|
|
28
|
+
|
|
29
|
+
format('pool: {\nmin: 2, max: 16}');
|
|
30
|
+
//=> 'pool: {\n\tmin: 2\n\tmax: 16\n}\n'
|
|
31
|
+
```
|
|
32
|
+
*/
|
|
33
|
+
export function format(text) {
|
|
34
|
+
return new Printer(text, parseTree(text)).print();
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
The changes that `format()` makes to a document, as edits of its text. For an editor or a linter, which shows or applies each change where it is, rather than replacing the whole document.
|
|
38
|
+
|
|
39
|
+
The edits are sorted, do not overlap, and change only spaces, tabs, line feeds, and commas, as `format()` does. Each one is as small as possible, so it leaves out the characters at its ends that stay the same. Applying all of them gives the same text as `format()`, and a document that is already formatted gives no edits.
|
|
40
|
+
|
|
41
|
+
@param text - The document.
|
|
42
|
+
@returns The edits, in the order of their ranges.
|
|
43
|
+
@throws {ParseError} When the document is not valid, the same as `parse()`.
|
|
44
|
+
@throws {TypeError} When `text` is not a string.
|
|
45
|
+
|
|
46
|
+
@example
|
|
47
|
+
```
|
|
48
|
+
import {formatEdits} from 'soml-lang';
|
|
49
|
+
|
|
50
|
+
formatEdits('a: [1,\n2]\n');
|
|
51
|
+
//=> [{range: [4, 4], text: '\n\t'}, {range: [5, 7], text: '\n\t'}, {range: [8, 8], text: '\n'}]
|
|
52
|
+
```
|
|
53
|
+
*/
|
|
54
|
+
export function formatEdits(text) {
|
|
55
|
+
const formatted = format(text);
|
|
56
|
+
const edits = [];
|
|
57
|
+
let textIndex = 0;
|
|
58
|
+
let formattedIndex = 0;
|
|
59
|
+
// Without the characters that the formatter changes, the two texts are the same. So walking both at once, every run of those characters that differs is one edit, which takes linear time, where a general diff takes quadratic time on a document with many changes.
|
|
60
|
+
while (textIndex < text.length || formattedIndex < formatted.length) {
|
|
61
|
+
const textStart = textIndex;
|
|
62
|
+
const formattedStart = formattedIndex;
|
|
63
|
+
const textGapEnd = skipLayout(text, textIndex);
|
|
64
|
+
const formattedGapEnd = skipLayout(formatted, formattedIndex);
|
|
65
|
+
if (text.slice(textIndex, textGapEnd) !== formatted.slice(formattedIndex, formattedGapEnd)) {
|
|
66
|
+
edits.push(createEdit(text, textIndex, textGapEnd, formatted.slice(formattedIndex, formattedGapEnd)));
|
|
67
|
+
}
|
|
68
|
+
textIndex = textGapEnd;
|
|
69
|
+
formattedIndex = formattedGapEnd;
|
|
70
|
+
while (textIndex < text.length && !isLayout(text.charCodeAt(textIndex)) && text.charCodeAt(textIndex) === formatted.charCodeAt(formattedIndex)) {
|
|
71
|
+
textIndex++;
|
|
72
|
+
formattedIndex++;
|
|
73
|
+
}
|
|
74
|
+
// Both stop at a different character, or one at its end, only when the formatter changed something else, and then the walk would never end.
|
|
75
|
+
if (textIndex === textStart && formattedIndex === formattedStart) {
|
|
76
|
+
throw new Error(`The formatter changed more than the layout at offset ${textIndex}. This is a bug in soml-lang.`);
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
return edits;
|
|
80
|
+
}
|
|
81
|
+
/*
|
|
82
|
+
The characters that the formatter changes.
|
|
83
|
+
*/
|
|
84
|
+
function isLayout(code) {
|
|
85
|
+
return isSpace(code) || code === LF || code === COMMA;
|
|
86
|
+
}
|
|
87
|
+
/*
|
|
88
|
+
The end of the run of layout characters at `index`.
|
|
89
|
+
*/
|
|
90
|
+
function skipLayout(text, index) {
|
|
91
|
+
while (index < text.length && isLayout(text.charCodeAt(index))) {
|
|
92
|
+
index++;
|
|
93
|
+
}
|
|
94
|
+
return index;
|
|
95
|
+
}
|
|
96
|
+
/*
|
|
97
|
+
The edit that replaces `text.slice(start, end)` with `replacement`, without the characters at its ends that stay the same, as in `, ` to `\n\t`.
|
|
98
|
+
*/
|
|
99
|
+
function createEdit(text, start, end, replacement) {
|
|
100
|
+
let prefixLength = 0;
|
|
101
|
+
while (start + prefixLength < end && prefixLength < replacement.length && text[start + prefixLength] === replacement[prefixLength]) {
|
|
102
|
+
prefixLength++;
|
|
103
|
+
}
|
|
104
|
+
let suffixLength = 0;
|
|
105
|
+
while (end - suffixLength > start + prefixLength && replacement.length - suffixLength > prefixLength && text[end - suffixLength - 1] === replacement[replacement.length - suffixLength - 1]) {
|
|
106
|
+
suffixLength++;
|
|
107
|
+
}
|
|
108
|
+
return {
|
|
109
|
+
range: [start + prefixLength, end - suffixLength],
|
|
110
|
+
text: replacement.slice(prefixLength, replacement.length - suffixLength),
|
|
111
|
+
};
|
|
112
|
+
}
|
|
113
|
+
class Printer {
|
|
114
|
+
#text;
|
|
115
|
+
#tree;
|
|
116
|
+
// Joined once at the end. Appending with `+=` builds a deep rope, and reading it one character at a time, as `formatEdits()` does, can take quadratic time once V8 deoptimizes the reader.
|
|
117
|
+
#output = [];
|
|
118
|
+
#commentIndex = 0;
|
|
119
|
+
constructor(text, tree) {
|
|
120
|
+
this.#text = text;
|
|
121
|
+
this.#tree = tree;
|
|
122
|
+
}
|
|
123
|
+
#write(text) {
|
|
124
|
+
this.#output.push(text);
|
|
125
|
+
}
|
|
126
|
+
/*
|
|
127
|
+
Starts a new line at `level`, with one blank line before it when the source had one.
|
|
128
|
+
*/
|
|
129
|
+
#newLine(level, isBlank) {
|
|
130
|
+
if (this.#output.length === 0) {
|
|
131
|
+
return;
|
|
132
|
+
}
|
|
133
|
+
this.#write(`${isBlank ? '\n\n' : '\n'}${INDENT.repeat(level)}`);
|
|
134
|
+
}
|
|
135
|
+
#isSameLine(start, end) {
|
|
136
|
+
return !this.#text.slice(start, end).includes('\n');
|
|
137
|
+
}
|
|
138
|
+
/*
|
|
139
|
+
Whether a line between `start` and `end` holds only spaces and tabs, if anything.
|
|
140
|
+
*/
|
|
141
|
+
#hasBlankLine(start, end) {
|
|
142
|
+
// A slice, so that the search for a line break stops at `end`.
|
|
143
|
+
const gap = this.#text.slice(start, end);
|
|
144
|
+
for (let index = gap.indexOf('\n'); index !== -1; index = gap.indexOf('\n', index + 1)) {
|
|
145
|
+
index = skipSpaces(gap, index + 1);
|
|
146
|
+
if (gap.charCodeAt(index) === LF) {
|
|
147
|
+
return true;
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
return false;
|
|
151
|
+
}
|
|
152
|
+
/*
|
|
153
|
+
The next comment that ends at or before `end`, without consuming it.
|
|
154
|
+
*/
|
|
155
|
+
#peekComment(end) {
|
|
156
|
+
const comment = this.#tree.comments[this.#commentIndex];
|
|
157
|
+
return comment !== undefined && comment.range[1] <= end ? comment : undefined;
|
|
158
|
+
}
|
|
159
|
+
#writeComment(comment) {
|
|
160
|
+
this.#commentIndex++;
|
|
161
|
+
if (comment.type === 'Line') {
|
|
162
|
+
this.#write(trimTrailingWhitespace(`#${comment.value}`));
|
|
163
|
+
return;
|
|
164
|
+
}
|
|
165
|
+
// The lines of a block comment keep their own indentation. Trailing whitespace is removed, and runs of blank lines collapse to one, as everywhere outside a block string.
|
|
166
|
+
const lines = `/*${comment.value}*/`.split('\n').map(line => trimTrailingWhitespace(line));
|
|
167
|
+
this.#write(lines.filter((line, index) => line !== '' || lines[index - 1] !== '').join('\n'));
|
|
168
|
+
}
|
|
169
|
+
/*
|
|
170
|
+
The offset of the `,` after an item in the source, or `undefined` when there is none.
|
|
171
|
+
*/
|
|
172
|
+
#findComma(start, end) {
|
|
173
|
+
const text = this.#text;
|
|
174
|
+
// Between two items there is only whitespace, comments, and at most one comma.
|
|
175
|
+
for (let index = start; index < end; index++) {
|
|
176
|
+
const code = text.charCodeAt(index);
|
|
177
|
+
if (code === COMMA) {
|
|
178
|
+
return index;
|
|
179
|
+
}
|
|
180
|
+
if (code === HASH) {
|
|
181
|
+
index = findLineEnd(text, index);
|
|
182
|
+
}
|
|
183
|
+
else if (code === SLASH && text.charCodeAt(index + 1) === ASTERISK) {
|
|
184
|
+
index = text.indexOf('*/', index + 2) + 1;
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
return undefined;
|
|
188
|
+
}
|
|
189
|
+
/*
|
|
190
|
+
Writes `items` one per line at `level`, with the comments between them, from `start` to `end` in the source.
|
|
191
|
+
|
|
192
|
+
A comment on the line of the item before it, or of the opening bracket, stays at the end of that line. The exception is a block comment after the comma that the next item follows on the same line, as in `[1, /* note *\/ 2]`: it stays in front of that item. Every other comment starts a new line, and a comment or an item that follows a block comment on its line stays on that line.
|
|
193
|
+
*/
|
|
194
|
+
#list(items, { start, end, level, isBracketed }) {
|
|
195
|
+
let previousEnd = start;
|
|
196
|
+
let itemEnd;
|
|
197
|
+
// Whether a comment may stay at the end of the line before it, which needs an item or an opening bracket there.
|
|
198
|
+
let canTrail = isBracketed;
|
|
199
|
+
let hasWritten = false;
|
|
200
|
+
let isOnSameLine = false;
|
|
201
|
+
for (const item of [...items, undefined]) {
|
|
202
|
+
const itemStart = item?.range[0] ?? end;
|
|
203
|
+
const comma = isBracketed ? this.#findComma(previousEnd, itemStart) : undefined;
|
|
204
|
+
// Found once per item rather than once per comment, which would be quadratic in the number of comments on one line.
|
|
205
|
+
const itemLineStart = item === undefined ? undefined : previousEnd + this.#text.slice(previousEnd, itemStart).lastIndexOf('\n') + 1;
|
|
206
|
+
for (let comment = this.#peekComment(itemStart); comment !== undefined; comment = this.#peekComment(itemStart)) {
|
|
207
|
+
const [commentStart, commentEnd] = comment.range;
|
|
208
|
+
const isTrailing = canTrail && !isOnSameLine && this.#isTrailing(comment, {
|
|
209
|
+
previousEnd,
|
|
210
|
+
itemEnd,
|
|
211
|
+
itemLineStart,
|
|
212
|
+
comma,
|
|
213
|
+
});
|
|
214
|
+
if (isTrailing || isOnSameLine) {
|
|
215
|
+
this.#write(' ');
|
|
216
|
+
}
|
|
217
|
+
else {
|
|
218
|
+
this.#newLine(level, hasWritten && this.#hasBlankLine(previousEnd, commentStart));
|
|
219
|
+
}
|
|
220
|
+
this.#writeComment(comment);
|
|
221
|
+
hasWritten ||= !isTrailing;
|
|
222
|
+
// A block comment that the next comment or item follows on the same line keeps it on that line.
|
|
223
|
+
const nextStart = this.#peekComment(itemStart)?.range[0] ?? itemStart;
|
|
224
|
+
isOnSameLine = comment.type === 'Block' && !isTrailing && this.#isSameLine(commentEnd, nextStart);
|
|
225
|
+
previousEnd = commentEnd;
|
|
226
|
+
}
|
|
227
|
+
if (item === undefined) {
|
|
228
|
+
return;
|
|
229
|
+
}
|
|
230
|
+
if (isOnSameLine) {
|
|
231
|
+
this.#write(' ');
|
|
232
|
+
}
|
|
233
|
+
else {
|
|
234
|
+
this.#newLine(level, hasWritten && this.#hasBlankLine(previousEnd, itemStart));
|
|
235
|
+
}
|
|
236
|
+
this.#item(item, level);
|
|
237
|
+
previousEnd = item.range[1];
|
|
238
|
+
itemEnd = previousEnd;
|
|
239
|
+
canTrail = true;
|
|
240
|
+
hasWritten = true;
|
|
241
|
+
isOnSameLine = false;
|
|
242
|
+
}
|
|
243
|
+
}
|
|
244
|
+
/*
|
|
245
|
+
Whether a comment stays at the end of the line before it. That line ends at the item before it, or at the comma after that item.
|
|
246
|
+
*/
|
|
247
|
+
#isTrailing(comment, { previousEnd, itemEnd, itemLineStart, comma }) {
|
|
248
|
+
const [commentStart, commentEnd] = comment.range;
|
|
249
|
+
const isAfterComma = comma !== undefined && commentStart > comma;
|
|
250
|
+
// A block comment after the comma that the next item follows on the same line stays in front of that item. The comment ends on the item's line when it ends after the start of that line.
|
|
251
|
+
if (itemLineStart !== undefined && comment.type === 'Block' && (comma === undefined || isAfterComma) && commentEnd >= itemLineStart) {
|
|
252
|
+
return false;
|
|
253
|
+
}
|
|
254
|
+
// A comment after the comma ends the comma's line, which is later than the item's when a block comment that spans lines comes between them. The comma is removed, so the comment moves to the end of the item's line, which only holds when nothing came between them.
|
|
255
|
+
const lineEnd = isAfterComma && previousEnd === itemEnd ? comma + 1 : previousEnd;
|
|
256
|
+
return this.#isSameLine(lineEnd, commentStart);
|
|
257
|
+
}
|
|
258
|
+
/*
|
|
259
|
+
Writes a container that is on one line in the source on one line, with a comma and a space between its members or items and no trailing comma. Only a block comment can be inside it, and each one stays where it is among the items and commas.
|
|
260
|
+
*/
|
|
261
|
+
#oneLine(items, { start, end, level }, opening, closing) {
|
|
262
|
+
this.#write(opening);
|
|
263
|
+
let previousEnd = start + 1;
|
|
264
|
+
let hasWrittenToken = false;
|
|
265
|
+
for (const item of [...items, undefined]) {
|
|
266
|
+
const itemStart = item?.range[0] ?? (end - 1);
|
|
267
|
+
// The comma after the last item is left out.
|
|
268
|
+
const comma = item === undefined ? undefined : this.#findComma(previousEnd, itemStart);
|
|
269
|
+
let hasWrittenComma = false;
|
|
270
|
+
for (let comment = this.#peekComment(itemStart); comment !== undefined; comment = this.#peekComment(itemStart)) {
|
|
271
|
+
if (!hasWrittenComma && comma !== undefined && comma < comment.range[0]) {
|
|
272
|
+
this.#write(',');
|
|
273
|
+
hasWrittenComma = true;
|
|
274
|
+
}
|
|
275
|
+
this.#write(hasWrittenToken ? ' ' : '');
|
|
276
|
+
this.#writeComment(comment);
|
|
277
|
+
hasWrittenToken = true;
|
|
278
|
+
}
|
|
279
|
+
if (item === undefined) {
|
|
280
|
+
break;
|
|
281
|
+
}
|
|
282
|
+
if (comma !== undefined && !hasWrittenComma) {
|
|
283
|
+
this.#write(', ');
|
|
284
|
+
}
|
|
285
|
+
else if (hasWrittenToken) {
|
|
286
|
+
this.#write(' ');
|
|
287
|
+
}
|
|
288
|
+
this.#item(item, level);
|
|
289
|
+
previousEnd = item.range[1];
|
|
290
|
+
hasWrittenToken = true;
|
|
291
|
+
}
|
|
292
|
+
this.#write(closing);
|
|
293
|
+
}
|
|
294
|
+
#item(item, level) {
|
|
295
|
+
if (item.type === 'Member') {
|
|
296
|
+
this.#member(item, level);
|
|
297
|
+
}
|
|
298
|
+
else {
|
|
299
|
+
this.#value(item, level);
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
#member(member, level) {
|
|
303
|
+
const { key, value } = member;
|
|
304
|
+
this.#write(`${this.#text.slice(...key.range)}:`);
|
|
305
|
+
const isBlockString = value.type === 'String' && value.block;
|
|
306
|
+
if (this.#peekComment(value.range[0]) === undefined) {
|
|
307
|
+
// A block string begins on the next line, one level deeper, so its delimiters and content line up.
|
|
308
|
+
if (isBlockString) {
|
|
309
|
+
this.#newLine(level + 1, false);
|
|
310
|
+
this.#value(value, level + 1);
|
|
311
|
+
return;
|
|
312
|
+
}
|
|
313
|
+
this.#write(' ');
|
|
314
|
+
this.#value(value, level);
|
|
315
|
+
return;
|
|
316
|
+
}
|
|
317
|
+
// With comments between the `:` and the value, the line breaks between them are kept, but not the blank lines, and a value on a new line is one level deeper.
|
|
318
|
+
let previousEnd = key.range[1] + 1;
|
|
319
|
+
let valueLevel = level;
|
|
320
|
+
for (let comment = this.#peekComment(value.range[0]); comment !== undefined; comment = this.#peekComment(value.range[0])) {
|
|
321
|
+
if (this.#writeSeparator(previousEnd, comment.range[0], level + 1)) {
|
|
322
|
+
valueLevel = level + 1;
|
|
323
|
+
}
|
|
324
|
+
this.#writeComment(comment);
|
|
325
|
+
previousEnd = comment.range[1];
|
|
326
|
+
}
|
|
327
|
+
if (isBlockString) {
|
|
328
|
+
this.#newLine(level + 1, false);
|
|
329
|
+
valueLevel = level + 1;
|
|
330
|
+
}
|
|
331
|
+
else if (this.#writeSeparator(previousEnd, value.range[0], level + 1)) {
|
|
332
|
+
valueLevel = level + 1;
|
|
333
|
+
}
|
|
334
|
+
this.#value(value, valueLevel);
|
|
335
|
+
}
|
|
336
|
+
/*
|
|
337
|
+
A space when `end` is on the line of `start` in the source, and otherwise a new line at `level`. Returns whether it started a new line.
|
|
338
|
+
*/
|
|
339
|
+
#writeSeparator(start, end, level) {
|
|
340
|
+
if (this.#isSameLine(start, end)) {
|
|
341
|
+
this.#write(' ');
|
|
342
|
+
return false;
|
|
343
|
+
}
|
|
344
|
+
this.#newLine(level, false);
|
|
345
|
+
return true;
|
|
346
|
+
}
|
|
347
|
+
#value(node, level) {
|
|
348
|
+
const [start, end] = node.range;
|
|
349
|
+
if (node.type === 'Object' || node.type === 'Array') {
|
|
350
|
+
const [opening, closing] = node.type === 'Object' ? ['{', '}'] : ['[', ']'];
|
|
351
|
+
const items = node.type === 'Object' ? node.members : node.elements;
|
|
352
|
+
if (items.length === 0 && this.#peekComment(end) === undefined) {
|
|
353
|
+
this.#write(opening + closing);
|
|
354
|
+
return;
|
|
355
|
+
}
|
|
356
|
+
if (this.#isSameLine(start, end)) {
|
|
357
|
+
this.#oneLine(items, { start, end, level }, opening, closing);
|
|
358
|
+
return;
|
|
359
|
+
}
|
|
360
|
+
this.#write(opening);
|
|
361
|
+
this.#list(items, {
|
|
362
|
+
start: start + 1,
|
|
363
|
+
end: end - 1,
|
|
364
|
+
level: level + 1,
|
|
365
|
+
isBracketed: true,
|
|
366
|
+
});
|
|
367
|
+
this.#newLine(level, false);
|
|
368
|
+
this.#write(closing);
|
|
369
|
+
return;
|
|
370
|
+
}
|
|
371
|
+
const text = this.#text.slice(start, end);
|
|
372
|
+
this.#write(node.type === 'String' && node.block ? reindentBlockString(text, INDENT.repeat(level)) : text);
|
|
373
|
+
}
|
|
374
|
+
print() {
|
|
375
|
+
const { body } = this.#tree;
|
|
376
|
+
const items = body.type === 'Object' && !body.braced ? body.members : [body];
|
|
377
|
+
// The document is a list without brackets: the members of a brace-less object, or one braced collection.
|
|
378
|
+
this.#list(items, {
|
|
379
|
+
start: 0,
|
|
380
|
+
end: this.#text.length,
|
|
381
|
+
level: 0,
|
|
382
|
+
isBracketed: false,
|
|
383
|
+
});
|
|
384
|
+
this.#output.push('\n');
|
|
385
|
+
return this.#output.join('');
|
|
386
|
+
}
|
|
387
|
+
}
|
|
388
|
+
/*
|
|
389
|
+
Only spaces and tabs are whitespace in SOML. Any other character, such as a no-break space, is content.
|
|
390
|
+
*/
|
|
391
|
+
function trimTrailingWhitespace(text) {
|
|
392
|
+
// A loop rather than a regular expression, which would take quadratic time on a long run of spaces inside a line.
|
|
393
|
+
return text.slice(0, skipSpacesBack(text, text.length));
|
|
394
|
+
}
|
|
395
|
+
/*
|
|
396
|
+
Moves a block string to a new indentation. Its content is relative to the closing delimiter's indentation, so replacing that indentation on every line keeps the value. A blank line is left as it is, because it becomes an empty line either way.
|
|
397
|
+
*/
|
|
398
|
+
function reindentBlockString(text, indentation) {
|
|
399
|
+
const lines = text.split('\n');
|
|
400
|
+
const closing = lines.at(-1);
|
|
401
|
+
const oldIndentation = closing.slice(0, skipSpaces(closing, 0));
|
|
402
|
+
return lines.map((line, index) => index === 0 || isBlankLine(line) ? line : indentation + line.slice(oldIndentation.length)).join('\n');
|
|
403
|
+
}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/// <reference lib="esnext.temporal" preserve="true" />
|
|
2
|
+
export { parse, type Value, type ObjectValue, type Document, type ParseOptions, } from './parse.ts';
|
|
3
|
+
export { parseTree, visitorKeys, type Position, type SourceLocation, type Token, type Comment, type DocumentNode, type ObjectNode, type MemberNode, type ArrayNode, type KeyNode, type KeySegmentNode, type StringNode, type IntegerNode, type FloatNode, type BooleanNode, type NullNode, type InstantNode, type DurationNode, type DurationPart, type ValueNode, type Node, } from './tree.ts';
|
|
4
|
+
export { format, formatEdits, type FormatEdit } from './format.ts';
|
|
5
|
+
export { edit, type EditOptions, type PathSegment } from './edit.ts';
|
|
6
|
+
export { stringify, stringifyValue, compareKeys, type StringifyOptions, } from './stringify.ts';
|
|
7
|
+
export { isBareKey } from './shared.ts';
|
|
8
|
+
export { ParseError } from './error.ts';
|