soml-lang 0.0.1 → 0.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/distribution/error.d.ts +32 -0
- package/distribution/error.js +139 -0
- package/distribution/format.d.ts +18 -0
- package/distribution/format.js +243 -0
- package/distribution/index.d.ts +5 -0
- package/distribution/index.js +5 -0
- package/distribution/parse.d.ts +68 -0
- package/distribution/parse.js +1171 -0
- package/distribution/shared.d.ts +48 -0
- package/distribution/shared.js +203 -0
- package/distribution/stringify.d.ts +35 -0
- package/distribution/stringify.js +359 -0
- package/distribution/tree.d.ts +153 -0
- package/distribution/tree.js +333 -0
- package/package.json +33 -34
- package/readme.md +90 -10
- package/index.d.ts +0 -151
- package/index.js +0 -3
- package/source/error.js +0 -130
- package/source/parse.js +0 -1574
- package/source/shared.js +0 -166
- package/source/stringify.js +0 -439
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
export declare const MAX_DEPTH = 100;
|
|
2
|
+
export declare const INT64_MIN: bigint;
|
|
3
|
+
export declare const INT64_MAX: bigint;
|
|
4
|
+
export declare const MIN_INSTANT = -62135596800000000000n;
|
|
5
|
+
export declare const MAX_INSTANT = 253402300799999999999n;
|
|
6
|
+
export declare function requireTemporal(): typeof globalThis.Temporal;
|
|
7
|
+
export declare const DURATION_UNITS: Map<string, bigint>;
|
|
8
|
+
/**
|
|
9
|
+
Remove the trailing zeros of a run of digits. A loop rather than a regular expression, which would be quadratic on a long run of zeros followed by another digit.
|
|
10
|
+
*/
|
|
11
|
+
export declare function trimTrailingZeros(digits: string): string;
|
|
12
|
+
/**
|
|
13
|
+
A duration as a `Temporal.Duration` in hours and smaller units, which hold any int64 count of nanoseconds exactly.
|
|
14
|
+
*/
|
|
15
|
+
export declare function createDuration(nanoseconds: bigint): Temporal.Duration;
|
|
16
|
+
export declare function validateIntegersOption(integers: unknown): asserts integers is 'bigint' | 'number';
|
|
17
|
+
/**
|
|
18
|
+
Whether a UTF-16 code unit can be part of a bare key. `NaN`, which `charCodeAt()` returns past the end of a string, cannot.
|
|
19
|
+
*/
|
|
20
|
+
export declare function isBareKeyCharacter(code: number): boolean;
|
|
21
|
+
/**
|
|
22
|
+
Whether a key can be written without quotes: one or more ASCII letters, digits, `_`, or `-`.
|
|
23
|
+
*/
|
|
24
|
+
export declare function isBareKey(key: string): boolean;
|
|
25
|
+
/**
|
|
26
|
+
The end of the number, instant, or duration token that starts at `start`, which is a digit or `-`.
|
|
27
|
+
*/
|
|
28
|
+
export declare function findNumberEnd(source: string, start: number): number;
|
|
29
|
+
/**
|
|
30
|
+
The offset of the line feed that ends the line containing `offset`, or the length of `source` on the last line.
|
|
31
|
+
*/
|
|
32
|
+
export declare function findLineEnd(source: string, offset: number): number;
|
|
33
|
+
/**
|
|
34
|
+
Find the closing delimiter of a block string: the first line from `offset` on that holds, after zero or more spaces and tabs, a run of exactly `length` of the `quote` character. A longer run is content. Returns where that line and its delimiter start, or `undefined` when no line closes the block.
|
|
35
|
+
*/
|
|
36
|
+
export declare function findBlockStringEnd(source: string, offset: number, quote: number, length: number): {
|
|
37
|
+
lineStart: number;
|
|
38
|
+
delimiterStart: number;
|
|
39
|
+
} | undefined;
|
|
40
|
+
/**
|
|
41
|
+
Shorten text quoted in an error message, so a huge token does not make a huge message. The cut never splits a surrogate pair.
|
|
42
|
+
*/
|
|
43
|
+
export declare function abbreviate(text: string, maximumLength?: number): string;
|
|
44
|
+
export declare function formatCodePoint(codePoint: number): string;
|
|
45
|
+
/**
|
|
46
|
+
Describe a character for an error message, such as `"$"` or `U+00A0 (NO-BREAK SPACE)`.
|
|
47
|
+
*/
|
|
48
|
+
export declare function describeCharacter(codePoint: number): string;
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
/*
|
|
2
|
+
The deepest nesting either direction accepts. A recursive-descent parser would otherwise end in an engine stack overflow.
|
|
3
|
+
*/
|
|
4
|
+
export const MAX_DEPTH = 100;
|
|
5
|
+
export const INT64_MIN = -(2n ** 63n);
|
|
6
|
+
export const INT64_MAX = (2n ** 63n) - 1n;
|
|
7
|
+
/*
|
|
8
|
+
The instant range is checked in UTC, so that every instant has a canonical form that can be read back.
|
|
9
|
+
*/
|
|
10
|
+
export const MIN_INSTANT = -62135596800000000000n; // 0001-01-01T00:00:00Z
|
|
11
|
+
export const MAX_INSTANT = 253402300799999999999n; // 9999-12-31T23:59:59.999999999Z
|
|
12
|
+
/*
|
|
13
|
+
`Temporal` is built into Node.js 26 and later. Before that, only instants and durations need it, so that everything else works without a polyfill. It is read at import, so a polyfill must be loaded before this package.
|
|
14
|
+
*/
|
|
15
|
+
const temporal = typeof Temporal === 'undefined' ? undefined : Temporal;
|
|
16
|
+
export function requireTemporal() {
|
|
17
|
+
if (temporal === undefined) {
|
|
18
|
+
throw new Error('Instants and durations need `Temporal`. Use Node.js 26 or later, or load a polyfill such as `temporal-polyfill/global` before this package');
|
|
19
|
+
}
|
|
20
|
+
return temporal;
|
|
21
|
+
}
|
|
22
|
+
/*
|
|
23
|
+
The duration units in their required order, with their length in nanoseconds.
|
|
24
|
+
*/
|
|
25
|
+
export const DURATION_UNITS = new Map([
|
|
26
|
+
['h', 3600000000000n],
|
|
27
|
+
['m', 60000000000n],
|
|
28
|
+
['s', 1000000000n],
|
|
29
|
+
['ms', 1000000n],
|
|
30
|
+
['us', 1000n],
|
|
31
|
+
['ns', 1n],
|
|
32
|
+
]);
|
|
33
|
+
/**
|
|
34
|
+
Remove the trailing zeros of a run of digits. A loop rather than a regular expression, which would be quadratic on a long run of zeros followed by another digit.
|
|
35
|
+
*/
|
|
36
|
+
export function trimTrailingZeros(digits) {
|
|
37
|
+
let end = digits.length;
|
|
38
|
+
while (end > 0 && digits.charCodeAt(end - 1) === 0x30) {
|
|
39
|
+
end--;
|
|
40
|
+
}
|
|
41
|
+
return digits.slice(0, end);
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
A duration as a `Temporal.Duration` in hours and smaller units, which hold any int64 count of nanoseconds exactly.
|
|
45
|
+
*/
|
|
46
|
+
export function createDuration(nanoseconds) {
|
|
47
|
+
const sign = nanoseconds < 0n ? -1n : 1n;
|
|
48
|
+
let rest = nanoseconds * sign;
|
|
49
|
+
const parts = [];
|
|
50
|
+
for (const size of DURATION_UNITS.values()) {
|
|
51
|
+
parts.push(Number(sign * (rest / size)));
|
|
52
|
+
rest %= size;
|
|
53
|
+
}
|
|
54
|
+
const [hours, minutes, seconds, milliseconds, microseconds, nanosecondsPart] = parts;
|
|
55
|
+
return new (requireTemporal().Duration)(0, 0, 0, 0, hours, minutes, seconds, milliseconds, microseconds, nanosecondsPart);
|
|
56
|
+
}
|
|
57
|
+
export function validateIntegersOption(integers) {
|
|
58
|
+
if (integers !== 'bigint' && integers !== 'number') {
|
|
59
|
+
throw new TypeError(`The \`integers\` option must be 'bigint' or 'number', got ${typeof integers === 'string' ? `'${abbreviate(integers)}'` : typeof integers}`);
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
/*
|
|
63
|
+
Names for the characters that are invisible or easy to mistake, so an error can say what it found.
|
|
64
|
+
*/
|
|
65
|
+
const CHARACTER_NAMES = new Map([
|
|
66
|
+
[0x06_1C, 'ARABIC LETTER MARK'],
|
|
67
|
+
[0x85, 'NEXT LINE'],
|
|
68
|
+
[0xA0, 'NO-BREAK SPACE'],
|
|
69
|
+
[0xAD, 'SOFT HYPHEN'],
|
|
70
|
+
[0x16_80, 'OGHAM SPACE MARK'],
|
|
71
|
+
[0x20_00, 'EN QUAD'],
|
|
72
|
+
[0x20_01, 'EM QUAD'],
|
|
73
|
+
[0x20_02, 'EN SPACE'],
|
|
74
|
+
[0x20_03, 'EM SPACE'],
|
|
75
|
+
[0x20_04, 'THREE-PER-EM SPACE'],
|
|
76
|
+
[0x20_05, 'FOUR-PER-EM SPACE'],
|
|
77
|
+
[0x20_06, 'SIX-PER-EM SPACE'],
|
|
78
|
+
[0x20_07, 'FIGURE SPACE'],
|
|
79
|
+
[0x20_08, 'PUNCTUATION SPACE'],
|
|
80
|
+
[0x20_09, 'THIN SPACE'],
|
|
81
|
+
[0x20_0A, 'HAIR SPACE'],
|
|
82
|
+
[0x20_0B, 'ZERO WIDTH SPACE'],
|
|
83
|
+
[0x20_0C, 'ZERO WIDTH NON-JOINER'],
|
|
84
|
+
[0x20_0D, 'ZERO WIDTH JOINER'],
|
|
85
|
+
[0x20_0E, 'LEFT-TO-RIGHT MARK'],
|
|
86
|
+
[0x20_0F, 'RIGHT-TO-LEFT MARK'],
|
|
87
|
+
[0x20_28, 'LINE SEPARATOR'],
|
|
88
|
+
[0x20_29, 'PARAGRAPH SEPARATOR'],
|
|
89
|
+
[0x20_2A, 'LEFT-TO-RIGHT EMBEDDING'],
|
|
90
|
+
[0x20_2B, 'RIGHT-TO-LEFT EMBEDDING'],
|
|
91
|
+
[0x20_2C, 'POP DIRECTIONAL FORMATTING'],
|
|
92
|
+
[0x20_2D, 'LEFT-TO-RIGHT OVERRIDE'],
|
|
93
|
+
[0x20_2E, 'RIGHT-TO-LEFT OVERRIDE'],
|
|
94
|
+
[0x20_2F, 'NARROW NO-BREAK SPACE'],
|
|
95
|
+
[0x20_5F, 'MEDIUM MATHEMATICAL SPACE'],
|
|
96
|
+
[0x20_60, 'WORD JOINER'],
|
|
97
|
+
[0x20_66, 'LEFT-TO-RIGHT ISOLATE'],
|
|
98
|
+
[0x20_67, 'RIGHT-TO-LEFT ISOLATE'],
|
|
99
|
+
[0x20_68, 'FIRST STRONG ISOLATE'],
|
|
100
|
+
[0x20_69, 'POP DIRECTIONAL ISOLATE'],
|
|
101
|
+
[0x30_00, 'IDEOGRAPHIC SPACE'],
|
|
102
|
+
[0xFE_FF, 'ZERO WIDTH NO-BREAK SPACE'],
|
|
103
|
+
]);
|
|
104
|
+
/*
|
|
105
|
+
Letters, digits, `_`, and `-`, the characters of a bare key, as a lookup table by code unit.
|
|
106
|
+
*/
|
|
107
|
+
const bareKeyCharacters = new Uint8Array(128);
|
|
108
|
+
for (const character of 'ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789_-') {
|
|
109
|
+
bareKeyCharacters[character.charCodeAt(0)] = 1;
|
|
110
|
+
}
|
|
111
|
+
/*
|
|
112
|
+
A number, an instant, or a duration is lexed as one maximal run of these characters, and then validated as a whole. The validation is a hand-written scan rather than a regular expression, because a regular expression with a repeated group runs out of stack on a token of a few million characters.
|
|
113
|
+
*/
|
|
114
|
+
const numberCharacters = new Uint8Array(128);
|
|
115
|
+
for (const character of 'ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789_.:+-') {
|
|
116
|
+
numberCharacters[character.charCodeAt(0)] = 1;
|
|
117
|
+
}
|
|
118
|
+
/**
|
|
119
|
+
Whether a UTF-16 code unit can be part of a bare key. `NaN`, which `charCodeAt()` returns past the end of a string, cannot.
|
|
120
|
+
*/
|
|
121
|
+
export function isBareKeyCharacter(code) {
|
|
122
|
+
return code < 128 && bareKeyCharacters[code] === 1;
|
|
123
|
+
}
|
|
124
|
+
/**
|
|
125
|
+
Whether a key can be written without quotes: one or more ASCII letters, digits, `_`, or `-`.
|
|
126
|
+
*/
|
|
127
|
+
export function isBareKey(key) {
|
|
128
|
+
if (key === '') {
|
|
129
|
+
return false;
|
|
130
|
+
}
|
|
131
|
+
for (let index = 0; index < key.length; index++) {
|
|
132
|
+
if (!isBareKeyCharacter(key.charCodeAt(index))) {
|
|
133
|
+
return false;
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
return true;
|
|
137
|
+
}
|
|
138
|
+
/**
|
|
139
|
+
The end of the number, instant, or duration token that starts at `start`, which is a digit or `-`.
|
|
140
|
+
*/
|
|
141
|
+
export function findNumberEnd(source, start) {
|
|
142
|
+
let end = start + 1;
|
|
143
|
+
while (isNumberCharacter(source.charCodeAt(end))) {
|
|
144
|
+
end++;
|
|
145
|
+
}
|
|
146
|
+
return end;
|
|
147
|
+
}
|
|
148
|
+
function isNumberCharacter(code) {
|
|
149
|
+
return code < 128 && numberCharacters[code] === 1;
|
|
150
|
+
}
|
|
151
|
+
/**
|
|
152
|
+
The offset of the line feed that ends the line containing `offset`, or the length of `source` on the last line.
|
|
153
|
+
*/
|
|
154
|
+
export function findLineEnd(source, offset) {
|
|
155
|
+
const end = source.indexOf('\n', offset);
|
|
156
|
+
return end === -1 ? source.length : end;
|
|
157
|
+
}
|
|
158
|
+
/**
|
|
159
|
+
Find the closing delimiter of a block string: the first line from `offset` on that holds, after zero or more spaces and tabs, a run of exactly `length` of the `quote` character. A longer run is content. Returns where that line and its delimiter start, or `undefined` when no line closes the block.
|
|
160
|
+
*/
|
|
161
|
+
export function findBlockStringEnd(source, offset, quote, length) {
|
|
162
|
+
for (let lineStart = offset; lineStart < source.length; lineStart = findLineEnd(source, lineStart) + 1) {
|
|
163
|
+
let delimiterStart = lineStart;
|
|
164
|
+
while (source.charCodeAt(delimiterStart) === 0x20 || source.charCodeAt(delimiterStart) === 0x09) {
|
|
165
|
+
delimiterStart++;
|
|
166
|
+
}
|
|
167
|
+
let runLength = 0;
|
|
168
|
+
while (source.charCodeAt(delimiterStart + runLength) === quote) {
|
|
169
|
+
runLength++;
|
|
170
|
+
}
|
|
171
|
+
if (runLength === length) {
|
|
172
|
+
return { lineStart, delimiterStart };
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
return undefined;
|
|
176
|
+
}
|
|
177
|
+
/**
|
|
178
|
+
Shorten text quoted in an error message, so a huge token does not make a huge message. The cut never splits a surrogate pair.
|
|
179
|
+
*/
|
|
180
|
+
export function abbreviate(text, maximumLength = 40) {
|
|
181
|
+
if (text.length <= maximumLength) {
|
|
182
|
+
return text;
|
|
183
|
+
}
|
|
184
|
+
const lastCode = text.charCodeAt(maximumLength - 1);
|
|
185
|
+
const end = lastCode >= 0xD8_00 && lastCode <= 0xDB_FF ? maximumLength - 1 : maximumLength;
|
|
186
|
+
return `${text.slice(0, end)}…`;
|
|
187
|
+
}
|
|
188
|
+
export function formatCodePoint(codePoint) {
|
|
189
|
+
return `U+${codePoint.toString(16).toUpperCase().padStart(4, '0')}`;
|
|
190
|
+
}
|
|
191
|
+
/**
|
|
192
|
+
Describe a character for an error message, such as `"$"` or `U+00A0 (NO-BREAK SPACE)`.
|
|
193
|
+
*/
|
|
194
|
+
export function describeCharacter(codePoint) {
|
|
195
|
+
const name = CHARACTER_NAMES.get(codePoint);
|
|
196
|
+
if (name !== undefined) {
|
|
197
|
+
const isWhitespace = /^[\p{White_Space}\u{200B}\u{FEFF}]$/v.test(String.fromCodePoint(codePoint));
|
|
198
|
+
return `${formatCodePoint(codePoint)} (${name}${isWhitespace ? '; only space, tab, and line feed are whitespace' : ''})`;
|
|
199
|
+
}
|
|
200
|
+
const character = String.fromCodePoint(codePoint);
|
|
201
|
+
// Control, format, private-use, unassigned, surrogate, and separator characters are invisible or ambiguous, so they are shown by code point.
|
|
202
|
+
return codePoint !== 0x20 && /^[\p{Other}\p{Separator}]$/v.test(character) ? formatCodePoint(codePoint) : `"${character}"`;
|
|
203
|
+
}
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
export type StringifyOptions = {
|
|
2
|
+
/**
|
|
3
|
+
How an int is represented in the value.
|
|
4
|
+
|
|
5
|
+
- `'bigint'`: A `bigint` is written as an int, and a `number` is always written as a float, so `8080` becomes `8080.0`.
|
|
6
|
+
- `'number'`: A `number` that is a safe integer is written as an int, and any other `number` as a float. A `bigint` is still written as an int.
|
|
7
|
+
|
|
8
|
+
@default 'bigint'
|
|
9
|
+
*/
|
|
10
|
+
readonly integers?: 'bigint' | 'number';
|
|
11
|
+
};
|
|
12
|
+
/**
|
|
13
|
+
Serialize an object or an array to canonical form.
|
|
14
|
+
|
|
15
|
+
Canonical form is unique for a value: two equal values produce the same bytes, so the output can be hashed, signed, or compared. Members are sorted by key, and comments, dotted keys, and block strings are never written.
|
|
16
|
+
|
|
17
|
+
A `bigint` is written as an int and a `number` as a float, so `{port: 8080}` becomes `port: 8080.0`. Use `8080n`, or the `integers: 'number'` option.
|
|
18
|
+
|
|
19
|
+
A `Date` is accepted and written as an instant. A member whose value is `undefined` is left out.
|
|
20
|
+
|
|
21
|
+
@param value - A plain object or an array.
|
|
22
|
+
@param options - How to write ints.
|
|
23
|
+
@returns The canonical form, ending with one line feed.
|
|
24
|
+
@throws {TypeError} When the value contains something that cannot be represented, such as `NaN`, a function, a class instance, an invalid `Date`, a `Temporal.Duration` with years, months, weeks, or days, a circular reference, or a string or key with a lone surrogate or a carriage return. Also when `options.integers` is not `'bigint'` or `'number'`.
|
|
25
|
+
@throws {RangeError} When an int is outside the 64-bit range, an instant is outside the years 0001 to 9999, a duration is outside the 64-bit range of nanoseconds, or the value is nested more than 100 levels deep.
|
|
26
|
+
|
|
27
|
+
@example
|
|
28
|
+
```
|
|
29
|
+
import {stringify} from 'soml-lang';
|
|
30
|
+
|
|
31
|
+
stringify({name: 'api-gateway', replicas: 3n, timeout: 30});
|
|
32
|
+
//=> "name: 'api-gateway'\nreplicas: 3\ntimeout: 30.0\n"
|
|
33
|
+
```
|
|
34
|
+
*/
|
|
35
|
+
export declare function stringify(value: object, options?: StringifyOptions): string;
|
|
@@ -0,0 +1,359 @@
|
|
|
1
|
+
import { MAX_DEPTH, INT64_MIN, INT64_MAX, MIN_INSTANT, MAX_INSTANT, DURATION_UNITS, trimTrailingZeros, validateIntegersOption, requireTemporal, isBareKey, abbreviate, } from "./shared.js";
|
|
2
|
+
const dateTime = Date.prototype.getTime;
|
|
3
|
+
const durationField = (name) => Object.getOwnPropertyDescriptor(Temporal.Duration.prototype, name).get;
|
|
4
|
+
const durationSizes = DURATION_UNITS.values().toArray();
|
|
5
|
+
// Undefined without `Temporal`, and then `requireTemporal()` throws before it is used.
|
|
6
|
+
const temporal = typeof Temporal === 'undefined'
|
|
7
|
+
? undefined
|
|
8
|
+
: {
|
|
9
|
+
instantNanoseconds: Object.getOwnPropertyDescriptor(Temporal.Instant.prototype, 'epochNanoseconds').get,
|
|
10
|
+
durationCalendarFields: ['years', 'months', 'weeks', 'days'].map(name => durationField(name)),
|
|
11
|
+
// In the order of `DURATION_UNITS`.
|
|
12
|
+
durationTimeFields: ['hours', 'minutes', 'seconds', 'milliseconds', 'microseconds', 'nanoseconds'].map(name => durationField(name)),
|
|
13
|
+
durationToString: Temporal.Duration.prototype.toString,
|
|
14
|
+
};
|
|
15
|
+
const HIGH_CODE_UNIT = /[\u{D800}-\u{10FFFF}]/v;
|
|
16
|
+
// eslint-disable-next-line no-control-regex, regexp/no-control-character -- Control characters are exactly what must be found.
|
|
17
|
+
const NEEDS_ESCAPED_STRING = /[\u{0}-\u{1F}'\u{7F}]/v;
|
|
18
|
+
// eslint-disable-next-line no-control-regex, regexp/no-control-character -- As above, plus a lone surrogate, which the `v` flag matches only when it is unpaired.
|
|
19
|
+
const NEEDS_ATTENTION = /[\u{0}-\u{1F}'\u{7F}\u{D800}-\u{DFFF}]/v;
|
|
20
|
+
// eslint-disable-next-line no-control-regex, regexp/no-control-character -- Control characters are exactly what must be escaped.
|
|
21
|
+
const ESCAPED_CHARACTER = /[\u{0}-\u{1F}"\\\u{7F}]/gv;
|
|
22
|
+
const ESCAPES = {
|
|
23
|
+
'\\': '\\\\',
|
|
24
|
+
'"': String.raw `\"`,
|
|
25
|
+
'\n': String.raw `\n`,
|
|
26
|
+
'\t': String.raw `\t`,
|
|
27
|
+
};
|
|
28
|
+
/**
|
|
29
|
+
Serialize an object or an array to canonical form.
|
|
30
|
+
|
|
31
|
+
Canonical form is unique for a value: two equal values produce the same bytes, so the output can be hashed, signed, or compared. Members are sorted by key, and comments, dotted keys, and block strings are never written.
|
|
32
|
+
|
|
33
|
+
A `bigint` is written as an int and a `number` as a float, so `{port: 8080}` becomes `port: 8080.0`. Use `8080n`, or the `integers: 'number'` option.
|
|
34
|
+
|
|
35
|
+
A `Date` is accepted and written as an instant. A member whose value is `undefined` is left out.
|
|
36
|
+
|
|
37
|
+
@param value - A plain object or an array.
|
|
38
|
+
@param options - How to write ints.
|
|
39
|
+
@returns The canonical form, ending with one line feed.
|
|
40
|
+
@throws {TypeError} When the value contains something that cannot be represented, such as `NaN`, a function, a class instance, an invalid `Date`, a `Temporal.Duration` with years, months, weeks, or days, a circular reference, or a string or key with a lone surrogate or a carriage return. Also when `options.integers` is not `'bigint'` or `'number'`.
|
|
41
|
+
@throws {RangeError} When an int is outside the 64-bit range, an instant is outside the years 0001 to 9999, a duration is outside the 64-bit range of nanoseconds, or the value is nested more than 100 levels deep.
|
|
42
|
+
|
|
43
|
+
@example
|
|
44
|
+
```
|
|
45
|
+
import {stringify} from 'soml-lang';
|
|
46
|
+
|
|
47
|
+
stringify({name: 'api-gateway', replicas: 3n, timeout: 30});
|
|
48
|
+
//=> "name: 'api-gateway'\nreplicas: 3\ntimeout: 30.0\n"
|
|
49
|
+
```
|
|
50
|
+
*/
|
|
51
|
+
// `object` rather than `Record<string, unknown>`, so that a value typed with an interface is accepted too. The value is checked at runtime.
|
|
52
|
+
export function stringify(value, options = {}) {
|
|
53
|
+
const { integers = 'bigint' } = options;
|
|
54
|
+
validateIntegersOption(integers);
|
|
55
|
+
if (!Array.isArray(value) && !isPlainObject(value)) {
|
|
56
|
+
throw new TypeError(`The top-level value must be an object or an array, because a document is a collection. Got ${describeType(value)}`);
|
|
57
|
+
}
|
|
58
|
+
return new Writer(integers).writeDocument(value);
|
|
59
|
+
}
|
|
60
|
+
/*
|
|
61
|
+
Appends to one array of parts that is joined once at the end, so the time is linear in the output whatever the nesting. A scalar and its line are one part.
|
|
62
|
+
*/
|
|
63
|
+
class Writer {
|
|
64
|
+
#integers;
|
|
65
|
+
#ancestors = new Set();
|
|
66
|
+
#parts = [];
|
|
67
|
+
constructor(integers) {
|
|
68
|
+
this.#integers = integers;
|
|
69
|
+
}
|
|
70
|
+
/*
|
|
71
|
+
Writes `prefix`, the value, and `suffix`. A container's own depth is `depth`, and its lines are indented by `indent`.
|
|
72
|
+
*/
|
|
73
|
+
#writeItem(prefix, value, depth, indent, suffix, key) {
|
|
74
|
+
const scalar = this.#formatScalar(value, key);
|
|
75
|
+
if (scalar === undefined) {
|
|
76
|
+
this.#parts.push(prefix);
|
|
77
|
+
this.#writeContainer(value, depth, indent);
|
|
78
|
+
this.#parts.push(suffix);
|
|
79
|
+
}
|
|
80
|
+
else {
|
|
81
|
+
this.#parts.push(`${prefix}${scalar}${suffix}`);
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
/*
|
|
85
|
+
Formats a scalar, or returns `undefined` for an array or a plain object.
|
|
86
|
+
*/
|
|
87
|
+
#formatScalar(value, key) {
|
|
88
|
+
switch (typeof value) {
|
|
89
|
+
case 'string': {
|
|
90
|
+
return formatString(value, 'string');
|
|
91
|
+
}
|
|
92
|
+
case 'bigint': {
|
|
93
|
+
return formatInteger(value);
|
|
94
|
+
}
|
|
95
|
+
case 'number': {
|
|
96
|
+
return this.#integers === 'number' && Number.isSafeInteger(value) ? formatInteger(BigInt(value)) : formatFloat(value);
|
|
97
|
+
}
|
|
98
|
+
case 'boolean': {
|
|
99
|
+
return value ? 'true' : 'false';
|
|
100
|
+
}
|
|
101
|
+
case 'object': {
|
|
102
|
+
break;
|
|
103
|
+
}
|
|
104
|
+
default: {
|
|
105
|
+
throw new TypeError(`Cannot serialize ${describeType(value)}${key === undefined ? '' : ` at key "${abbreviate(key)}"`}`);
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
if (value === null) {
|
|
109
|
+
return 'null';
|
|
110
|
+
}
|
|
111
|
+
if (Array.isArray(value) || isPlainObject(value)) {
|
|
112
|
+
return undefined;
|
|
113
|
+
}
|
|
114
|
+
// A brand check rather than `instanceof`, so that an instant or a date from another realm, such as a `vm` context, is accepted too. The tag is only a cheap filter, since `Symbol.toStringTag` can fake it. The getters read internal slots, so they throw for anything else.
|
|
115
|
+
const brand = Object.prototype.toString.call(value);
|
|
116
|
+
if (brand === '[object Temporal.Instant]') {
|
|
117
|
+
requireTemporal();
|
|
118
|
+
return formatInstant(readBranded(temporal.instantNanoseconds, value, 'Temporal.Instant'));
|
|
119
|
+
}
|
|
120
|
+
if (brand === '[object Temporal.Duration]') {
|
|
121
|
+
return formatDuration(value);
|
|
122
|
+
}
|
|
123
|
+
if (brand === '[object Date]') {
|
|
124
|
+
const milliseconds = readBranded(dateTime, value, 'Date');
|
|
125
|
+
if (Number.isNaN(milliseconds)) {
|
|
126
|
+
throw new TypeError('Cannot serialize an invalid Date');
|
|
127
|
+
}
|
|
128
|
+
return formatInstant(BigInt(milliseconds) * 1000000n);
|
|
129
|
+
}
|
|
130
|
+
throw new TypeError(`Cannot serialize ${describeType(value)}. Only plain objects, arrays, and the scalar types can be serialized`);
|
|
131
|
+
}
|
|
132
|
+
#enter(value, depth) {
|
|
133
|
+
if (depth > MAX_DEPTH) {
|
|
134
|
+
throw new RangeError(`Cannot serialize a value nested more than ${MAX_DEPTH} levels deep`);
|
|
135
|
+
}
|
|
136
|
+
if (this.#ancestors.has(value)) {
|
|
137
|
+
throw new TypeError('Cannot serialize a circular structure');
|
|
138
|
+
}
|
|
139
|
+
this.#ancestors.add(value);
|
|
140
|
+
}
|
|
141
|
+
#writeContainer(value, depth, indent) {
|
|
142
|
+
this.#enter(value, depth);
|
|
143
|
+
const parts = this.#parts;
|
|
144
|
+
const innerIndent = `${indent}\t`;
|
|
145
|
+
if (Array.isArray(value)) {
|
|
146
|
+
if (value.length === 0) {
|
|
147
|
+
parts.push('[]');
|
|
148
|
+
}
|
|
149
|
+
else {
|
|
150
|
+
parts.push('[\n');
|
|
151
|
+
for (const [index, item] of value.entries()) {
|
|
152
|
+
// A hole reads as `undefined`, and both would otherwise have to be guessed into something.
|
|
153
|
+
if (item === undefined) {
|
|
154
|
+
throw new TypeError(`Cannot serialize undefined in an array, at index ${index}`);
|
|
155
|
+
}
|
|
156
|
+
this.#writeItem(innerIndent, item, depth + 1, innerIndent, ',\n');
|
|
157
|
+
}
|
|
158
|
+
parts.push(`${indent}]`);
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
else {
|
|
162
|
+
const members = sortedMembers(value);
|
|
163
|
+
if (members.length === 0) {
|
|
164
|
+
parts.push('{}');
|
|
165
|
+
}
|
|
166
|
+
else {
|
|
167
|
+
parts.push('{\n');
|
|
168
|
+
for (const [key, member] of members) {
|
|
169
|
+
this.#writeItem(`${innerIndent}${formatKey(key)}: `, member, depth + 1, innerIndent, ',\n', key);
|
|
170
|
+
}
|
|
171
|
+
parts.push(`${indent}}`);
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
this.#ancestors.delete(value);
|
|
175
|
+
}
|
|
176
|
+
writeDocument(value) {
|
|
177
|
+
if (Array.isArray(value)) {
|
|
178
|
+
this.#writeContainer(value, 1, '');
|
|
179
|
+
this.#parts.push('\n');
|
|
180
|
+
return this.#parts.join('');
|
|
181
|
+
}
|
|
182
|
+
this.#enter(value, 1);
|
|
183
|
+
const members = sortedMembers(value);
|
|
184
|
+
for (const [key, member] of members) {
|
|
185
|
+
this.#writeItem(`${formatKey(key)}: `, member, 2, '', '\n', key);
|
|
186
|
+
}
|
|
187
|
+
this.#ancestors.delete(value);
|
|
188
|
+
// A brace-less object needs at least one entry, so the empty object keeps its braces.
|
|
189
|
+
return members.length === 0 ? '{}\n' : this.#parts.join('');
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
/*
|
|
193
|
+
The members to write, sorted by key, with each value read once. A member whose value is `undefined` is left out, as `JSON.stringify` does.
|
|
194
|
+
*/
|
|
195
|
+
function sortedMembers(object) {
|
|
196
|
+
const keys = Object.keys(object);
|
|
197
|
+
// The default sort compares UTF-16 code units, which matches code point order unless a key has a code unit from U+D800 up.
|
|
198
|
+
if (keys.some(key => HIGH_CODE_UNIT.test(key))) {
|
|
199
|
+
keys.sort(compareCodePoints);
|
|
200
|
+
}
|
|
201
|
+
else {
|
|
202
|
+
keys.sort(); // The native string order is the point, and it is much faster than a comparator.
|
|
203
|
+
}
|
|
204
|
+
const members = [];
|
|
205
|
+
for (const key of keys) {
|
|
206
|
+
const value = object[key];
|
|
207
|
+
if (value !== undefined) {
|
|
208
|
+
members.push([key, value]);
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
return members;
|
|
212
|
+
}
|
|
213
|
+
function isPlainObject(value) {
|
|
214
|
+
if (typeof value !== 'object' || value === null) {
|
|
215
|
+
return false;
|
|
216
|
+
}
|
|
217
|
+
const prototype = Object.getPrototypeOf(value);
|
|
218
|
+
return prototype === null || prototype === Object.prototype || Object.getPrototypeOf(prototype) === null;
|
|
219
|
+
}
|
|
220
|
+
function describeType(value) {
|
|
221
|
+
if (value === null) {
|
|
222
|
+
return 'null';
|
|
223
|
+
}
|
|
224
|
+
if (typeof value === 'number' && Number.isNaN(value)) {
|
|
225
|
+
return 'NaN, which is not representable';
|
|
226
|
+
}
|
|
227
|
+
if (typeof value === 'function') {
|
|
228
|
+
return 'a function';
|
|
229
|
+
}
|
|
230
|
+
if (typeof value === 'object') {
|
|
231
|
+
const name = value.constructor?.name;
|
|
232
|
+
return Array.isArray(value) ? 'an array' : (name === undefined || name === '' ? 'an object' : `an instance of ${name}`);
|
|
233
|
+
}
|
|
234
|
+
return value === undefined ? 'undefined' : `a ${typeof value}`;
|
|
235
|
+
}
|
|
236
|
+
/*
|
|
237
|
+
Members are sorted by the Unicode scalar values of their keys. `Array#sort` compares UTF-16 code units, which orders U+E000 to U+FFFF after every astral character, so the code units are compared with surrogates moved above that range.
|
|
238
|
+
*/
|
|
239
|
+
function compareCodePoints(left, right) {
|
|
240
|
+
const length = Math.min(left.length, right.length);
|
|
241
|
+
for (let index = 0; index < length; index++) {
|
|
242
|
+
let leftCode = left.charCodeAt(index);
|
|
243
|
+
let rightCode = right.charCodeAt(index);
|
|
244
|
+
if (leftCode !== rightCode) {
|
|
245
|
+
if (leftCode >= 0xD8_00 && rightCode >= 0xD8_00) {
|
|
246
|
+
leftCode = toCodePointOrder(leftCode);
|
|
247
|
+
rightCode = toCodePointOrder(rightCode);
|
|
248
|
+
}
|
|
249
|
+
return leftCode - rightCode;
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
return left.length - right.length;
|
|
253
|
+
}
|
|
254
|
+
/*
|
|
255
|
+
Moves U+E000 to U+FFFF down by 0x800 and the surrogates up by 0x2000, so the surrogates sort last.
|
|
256
|
+
*/
|
|
257
|
+
function toCodePointOrder(code) {
|
|
258
|
+
if (code >= 0xE0_00) {
|
|
259
|
+
return code - 0x8_00;
|
|
260
|
+
}
|
|
261
|
+
return code >= 0xD8_00 ? code + 0x20_00 : code;
|
|
262
|
+
}
|
|
263
|
+
function formatKey(key) {
|
|
264
|
+
return isBareKey(key) ? key : formatString(key, 'key');
|
|
265
|
+
}
|
|
266
|
+
function formatString(string, kind) {
|
|
267
|
+
// One scan settles the common case: no quote, no control character, and no lone surrogate.
|
|
268
|
+
if (!NEEDS_ATTENTION.test(string)) {
|
|
269
|
+
return `'${string}'`;
|
|
270
|
+
}
|
|
271
|
+
if (!string.isWellFormed()) {
|
|
272
|
+
throw new TypeError(`Cannot serialize a ${kind} containing a lone surrogate, which is not a Unicode scalar value`);
|
|
273
|
+
}
|
|
274
|
+
if (string.includes('\r')) {
|
|
275
|
+
throw new TypeError(`Cannot serialize a ${kind} containing a carriage return (U+000D), which is not representable`);
|
|
276
|
+
}
|
|
277
|
+
return NEEDS_ESCAPED_STRING.test(string) ? `"${string.replaceAll(ESCAPED_CHARACTER, character => ESCAPES[character] ?? String.raw `\u{${character.codePointAt(0).toString(16)}}`)}"` : `'${string}'`;
|
|
278
|
+
}
|
|
279
|
+
function formatInteger(value) {
|
|
280
|
+
if (value < INT64_MIN || value > INT64_MAX) {
|
|
281
|
+
throw new RangeError(`Cannot serialize the integer ${value}, because it is outside the 64-bit range`);
|
|
282
|
+
}
|
|
283
|
+
return String(value);
|
|
284
|
+
}
|
|
285
|
+
/*
|
|
286
|
+
The shortest decimal that reads back as the same binary64 value, laid out as ECMAScript `Number::toString` does, which is also RFC 8785's choice. A fractional part is added when there is neither one nor an exponent, so that a float never reads back as an int.
|
|
287
|
+
*/
|
|
288
|
+
function formatFloat(value) {
|
|
289
|
+
if (Number.isNaN(value)) {
|
|
290
|
+
throw new TypeError('Cannot serialize NaN, which is not representable. Use null for a missing value');
|
|
291
|
+
}
|
|
292
|
+
if (value === Infinity) {
|
|
293
|
+
return 'infinity';
|
|
294
|
+
}
|
|
295
|
+
if (value === -Infinity) {
|
|
296
|
+
return '-infinity';
|
|
297
|
+
}
|
|
298
|
+
// Covers -0 too, since zero has one value whatever its sign.
|
|
299
|
+
if (value === 0) {
|
|
300
|
+
return '0.0';
|
|
301
|
+
}
|
|
302
|
+
const text = String(value);
|
|
303
|
+
if (text.includes('e')) {
|
|
304
|
+
return text.replace('e+', 'e');
|
|
305
|
+
}
|
|
306
|
+
return text.includes('.') ? text : `${text}.0`;
|
|
307
|
+
}
|
|
308
|
+
function readBranded(getter, value, name) {
|
|
309
|
+
try {
|
|
310
|
+
return getter.call(value);
|
|
311
|
+
}
|
|
312
|
+
catch {
|
|
313
|
+
throw new TypeError(`Cannot serialize an object that claims to be a ${name} but is not one`);
|
|
314
|
+
}
|
|
315
|
+
}
|
|
316
|
+
/*
|
|
317
|
+
The total is computed from the fields as BigInts, because `Temporal.Duration#total()` returns a float, which cannot hold every int64 count of nanoseconds.
|
|
318
|
+
*/
|
|
319
|
+
function formatDuration(duration) {
|
|
320
|
+
requireTemporal();
|
|
321
|
+
const { durationCalendarFields, durationTimeFields, durationToString } = temporal;
|
|
322
|
+
if (durationCalendarFields.some(getter => readBranded(getter, duration, 'Temporal.Duration') !== 0)) {
|
|
323
|
+
throw new TypeError(`Cannot serialize the duration ${durationToString.call(duration)}, because years, months, weeks, and days are not a fixed length. Use hours or smaller units`);
|
|
324
|
+
}
|
|
325
|
+
let nanoseconds = 0n;
|
|
326
|
+
for (const [index, getter] of durationTimeFields.entries()) {
|
|
327
|
+
nanoseconds += BigInt(getter.call(duration)) * durationSizes[index];
|
|
328
|
+
}
|
|
329
|
+
if (nanoseconds < INT64_MIN || nanoseconds > INT64_MAX) {
|
|
330
|
+
throw new RangeError(`Cannot serialize the duration ${durationToString.call(duration)}, because it is outside the 64-bit range of nanoseconds`);
|
|
331
|
+
}
|
|
332
|
+
if (nanoseconds === 0n) {
|
|
333
|
+
return '0s';
|
|
334
|
+
}
|
|
335
|
+
const magnitude = nanoseconds < 0n ? -nanoseconds : nanoseconds;
|
|
336
|
+
const hours = magnitude / 3600000000000n;
|
|
337
|
+
const minutes = (magnitude / 60000000000n) % 60n;
|
|
338
|
+
const seconds = (magnitude / 1000000000n) % 60n;
|
|
339
|
+
const fraction = magnitude % 1000000000n;
|
|
340
|
+
let text = nanoseconds < 0n ? '-' : '';
|
|
341
|
+
if (hours > 0n) {
|
|
342
|
+
text += `${hours}h`;
|
|
343
|
+
}
|
|
344
|
+
if (minutes > 0n) {
|
|
345
|
+
text += `${minutes}m`;
|
|
346
|
+
}
|
|
347
|
+
if (seconds > 0n || fraction > 0n) {
|
|
348
|
+
// The fraction is written like an instant's: nine digits with the trailing zeros removed.
|
|
349
|
+
text += fraction === 0n ? `${seconds}s` : `${seconds}.${trimTrailingZeros(String(fraction).padStart(9, '0'))}s`;
|
|
350
|
+
}
|
|
351
|
+
return text;
|
|
352
|
+
}
|
|
353
|
+
function formatInstant(nanoseconds) {
|
|
354
|
+
const instant = new (requireTemporal().Instant)(nanoseconds);
|
|
355
|
+
if (nanoseconds < MIN_INSTANT || nanoseconds > MAX_INSTANT) {
|
|
356
|
+
throw new RangeError(`Cannot serialize the instant ${instant.toString()}, because it is outside the years 0001 to 9999`);
|
|
357
|
+
}
|
|
358
|
+
return instant.toString();
|
|
359
|
+
}
|