@amritk/nish-aarch64-darwin 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/std/json.ts ADDED
@@ -0,0 +1,402 @@
1
+ /**
2
+ * `std/json` — the value of one field of one flat JSON object.
3
+ *
4
+ * It exists because the compiler's own machine-readable surface is JSON: under
5
+ * `--json` every diagnostic is one flat object on a line of stdout, whose `code`
6
+ * is a stable rule identifier (`AGENTS.md`, and `docs/wp12-release.md` for the
7
+ * exit-code bands). A Nish program that reads that surface — a wrapper, an
8
+ * editor plug-in, or `tests/nish/cli.ts`, which pins the contract — needs the
9
+ * value of a named field and nothing else, and the language has no `JSON.parse`
10
+ * to give it.
11
+ *
12
+ * So this is a **reader, not a parser**, and the difference is worth stating:
13
+ *
14
+ * - It answers **text**. A string field comes back with its escapes decoded
15
+ * and without its quotes; a number, a `true` or a `null` comes back as the
16
+ * bytes it was written with, for `parseInt` or `===` at the call site. A
17
+ * field whose value is `null` and one whose value is the *string* `"null"`
18
+ * therefore answer the same thing, which is the one ambiguity a program that
19
+ * needs to tell them apart cannot live with — and the sign that it wants a
20
+ * real parser rather than this.
21
+ * - It answers `null` for a field that is not there, so "absent" and "empty"
22
+ * stay apart the way they do in `getenv`.
23
+ * - It **does not validate**. A nested object or array is stepped over so that
24
+ * the fields after it are still reachable, and a truncated object still
25
+ * answers the fields that precede the truncation. A program that has to know
26
+ * whether the whole line was well-formed is asking a question this does not
27
+ * answer.
28
+ *
29
+ * There is one exported function on purpose. A `std/` module is compiled into
30
+ * the program that imports it and the symbol namespace is flat
31
+ * (`std/README.md`), so every name here is a name its importer cannot use: the
32
+ * helpers are `json`-prefixed and the reader is the only export. Splitting a
33
+ * stream into lines and picking the ones that start with `{` is two calls the
34
+ * caller already has — `splitLines` from `std/text` and `startsWith` — and
35
+ * duplicating them here would cost more names than it saves.
36
+ *
37
+ * import { jsonField } from "nish/json";
38
+ *
39
+ * for (const line of splitLines(stdout)) {
40
+ * if (!line.startsWith("{")) {
41
+ * continue;
42
+ * }
43
+ * const code = jsonField(line, "code");
44
+ * if (code !== null && code === "NL0003") {
45
+ * panic("the compiler crashed");
46
+ * }
47
+ * }
48
+ *
49
+ * Every offset here is a **byte** offset and every width is spelled, with each
50
+ * length and byte a builtin answers read through `toI32` — `.length` and
51
+ * `charCodeAt` answer `number`, which is `f64` under `--number-mode f64`, so the
52
+ * conversion is what makes this one module in both modes rather than two
53
+ * (`docs/wp26-stdlib.md` §4).
54
+ */
55
+
56
+ const JSON_TAB: i32 = 9;
57
+ const JSON_NEWLINE: i32 = 10;
58
+ const JSON_CARRIAGE_RETURN: i32 = 13;
59
+ const JSON_SPACE: i32 = 32;
60
+ const JSON_QUOTE: i32 = 34;
61
+ const JSON_COMMA: i32 = 44;
62
+ const JSON_COLON: i32 = 58;
63
+ const JSON_BACKSLASH: i32 = 92;
64
+ const JSON_OPEN_BRACKET: i32 = 91;
65
+ const JSON_CLOSE_BRACKET: i32 = 93;
66
+ const JSON_OPEN_BRACE: i32 = 123;
67
+ const JSON_CLOSE_BRACE: i32 = 125;
68
+ /**
69
+ * The UTF-8 tags a `\u` escape is rebuilt out of: a two-byte lead is
70
+ * `110xxxxx`, a three-byte lead `1110xxxx`, a continuation byte `10xxxxxx`, and
71
+ * `LOW_SIX` is the six bits each continuation carries.
72
+ *
73
+ * They are named constants rather than the literals `192`, `224`, `128` and `63`
74
+ * for a reason that is the module's rule and not a style preference: a bare
75
+ * numeric literal is a `number`, which is an `f64` under `--number-mode f64`, so
76
+ * `192 + (code >> 6)` adds an `f64` to an `i32` there and does not compile at all
77
+ * (`docs/wp26-stdlib.md` §4, and `tests/link/std_text_f64` is what says so). A
78
+ * constant declared `i32` means the same thing in both modes.
79
+ */
80
+ const JSON_UTF8_TWO_BYTE_LEAD: i32 = 192;
81
+ const JSON_UTF8_THREE_BYTE_LEAD: i32 = 224;
82
+ const JSON_UTF8_CONTINUATION: i32 = 128;
83
+ const JSON_UTF8_LOW_SIX: i32 = 63;
84
+ /** The first code point that needs two UTF-8 bytes, and the first that needs three. */
85
+ const JSON_UTF8_TWO_BYTE_FLOOR: i32 = 128;
86
+ const JSON_UTF8_THREE_BYTE_FLOOR: i32 = 2048;
87
+ /** The base `jsonHex4` accumulates in. */
88
+ const JSON_HEX_BASE: i32 = 16;
89
+
90
+ /** The four bytes JSON allows between tokens. */
91
+ const jsonIsBlankByte = (code: i32): boolean =>
92
+ code === JSON_SPACE || code === JSON_TAB || code === JSON_NEWLINE || code === JSON_CARRIAGE_RETURN;
93
+
94
+ /**
95
+ * The first index at or after `from` that is not blank, or the length.
96
+ *
97
+ * Every caller hands in an index it has already found, so `from` is never
98
+ * negative; the guard says so in a form the bounds proof reads, and is what
99
+ * lets the loop below read `text` without a check.
100
+ */
101
+ const jsonSkipBlank = (text: string, from: i32): i32 => {
102
+ if (from < 0) {
103
+ return from;
104
+ }
105
+ const length: i32 = toI32(text.length);
106
+ let i: i32 = from;
107
+ while (i < length && jsonIsBlankByte(toI32(text.charCodeAt(i)))) {
108
+ i += 1;
109
+ }
110
+ return i;
111
+ };
112
+
113
+ /**
114
+ * The index just past the string literal whose opening quote is at `at`, or `-1`
115
+ * when the quote is never closed. A backslash takes the byte after it with it,
116
+ * which is all a scanner needs to know about escapes: `\"` cannot end the
117
+ * literal and `\\` cannot make the next quote an escape.
118
+ */
119
+ const jsonEndOfString = (text: string, at: i32): i32 => {
120
+ const length: i32 = toI32(text.length);
121
+ if (at < 0) {
122
+ return -1;
123
+ }
124
+ let i: i32 = at + 1;
125
+ while (i < length) {
126
+ const code: i32 = toI32(text.charCodeAt(i));
127
+ if (code === JSON_BACKSLASH) {
128
+ i += 2;
129
+ continue;
130
+ }
131
+ if (code === JSON_QUOTE) {
132
+ return i + 1;
133
+ }
134
+ i += 1;
135
+ }
136
+ return -1;
137
+ };
138
+
139
+ /**
140
+ * The index just past the value that starts at `at`, or `-1` when it does not
141
+ * end.
142
+ *
143
+ * A string ends at its quote and an object or array at the closer that brings
144
+ * the depth back to zero, with strings inside it skipped whole so that a `}` in
145
+ * a message cannot close it. Anything else — a number, `true`, `false`, `null` —
146
+ * ends at the first byte that cannot be part of it, which is the comma, the
147
+ * closer or the blank that follows.
148
+ *
149
+ * The string skip inside an object is a flag in the one loop rather than a call
150
+ * to `jsonEndOfString` that moves the cursor to wherever it answers: a cursor
151
+ * that only ever steps forward keeps the lower bound the bounds proof needs,
152
+ * and one assigned a callee's answer does not.
153
+ */
154
+ const jsonEndOfValue = (text: string, at: i32): i32 => {
155
+ const length: i32 = toI32(text.length);
156
+ if (at < 0 || at >= length) {
157
+ return -1;
158
+ }
159
+ const first: i32 = toI32(text.charCodeAt(at));
160
+ if (first === JSON_QUOTE) {
161
+ return jsonEndOfString(text, at);
162
+ }
163
+ if (first === JSON_OPEN_BRACE || first === JSON_OPEN_BRACKET) {
164
+ let depth: i32 = 0;
165
+ let inString: boolean = false;
166
+ let i: i32 = at;
167
+ while (i < length) {
168
+ const code: i32 = toI32(text.charCodeAt(i));
169
+ if (inString) {
170
+ // `jsonEndOfString`'s rule: a backslash takes the byte after it along.
171
+ if (code === JSON_BACKSLASH) {
172
+ i += 2;
173
+ continue;
174
+ }
175
+ if (code === JSON_QUOTE) {
176
+ inString = false;
177
+ }
178
+ i += 1;
179
+ continue;
180
+ }
181
+ if (code === JSON_QUOTE) {
182
+ inString = true;
183
+ } else if (code === JSON_OPEN_BRACE || code === JSON_OPEN_BRACKET) {
184
+ depth += 1;
185
+ } else if (code === JSON_CLOSE_BRACE || code === JSON_CLOSE_BRACKET) {
186
+ depth -= 1;
187
+ if (depth === 0) {
188
+ return i + 1;
189
+ }
190
+ }
191
+ i += 1;
192
+ }
193
+ return -1;
194
+ }
195
+ let i: i32 = at;
196
+ while (i < length) {
197
+ const code: i32 = toI32(text.charCodeAt(i));
198
+ if (code === JSON_COMMA || code === JSON_CLOSE_BRACE || code === JSON_CLOSE_BRACKET) {
199
+ return i;
200
+ }
201
+ if (jsonIsBlankByte(code)) {
202
+ return i;
203
+ }
204
+ i += 1;
205
+ }
206
+ return length;
207
+ };
208
+
209
+ /** The value of one lower-case or upper-case hex digit, or `-1`. */
210
+ const jsonHexDigit = (code: i32): i32 => {
211
+ if (code >= 48 && code <= 57) {
212
+ return code - 48;
213
+ }
214
+ if (code >= 97 && code <= 102) {
215
+ return code - 87;
216
+ }
217
+ if (code >= 65 && code <= 70) {
218
+ return code - 55;
219
+ }
220
+ return -1;
221
+ };
222
+
223
+ /** The four hex digits at `at` as one code point, or `-1` when they are not four hex digits. */
224
+ const jsonHex4 = (text: string, at: i32, end: i32): i32 => {
225
+ if (at + 4 > end) {
226
+ return -1;
227
+ }
228
+ let value: i32 = 0;
229
+ let i: i32 = 0;
230
+ while (i < 4) {
231
+ const digit: i32 = jsonHexDigit(toI32(text.charCodeAt(at + i)));
232
+ if (digit < 0) {
233
+ return -1;
234
+ }
235
+ value = value * JSON_HEX_BASE + digit;
236
+ i += 1;
237
+ }
238
+ return value;
239
+ };
240
+
241
+ /**
242
+ * `code` as UTF-8, one to three bytes.
243
+ *
244
+ * A surrogate pair is **not** recombined: each half becomes its own three-byte
245
+ * sequence, which is what `😀` reads as here. Nothing this module
246
+ * exists to read produces one — `JSON.stringify`, and `jsonQuote` in
247
+ * `self/strings.ts` which matches it byte for byte, escape only the seven short
248
+ * forms and `\u00xx` below `0x20`, and pass every other byte through as itself —
249
+ * so the alternative would be code with no caller to keep it honest.
250
+ */
251
+ const jsonUtf8 = (code: i32): string => {
252
+ if (code < JSON_UTF8_TWO_BYTE_FLOOR) {
253
+ return String.fromCharCode(code);
254
+ }
255
+ if (code < JSON_UTF8_THREE_BYTE_FLOOR) {
256
+ const lead: string = String.fromCharCode(JSON_UTF8_TWO_BYTE_LEAD + (code >> 6));
257
+ return `${lead}${String.fromCharCode(JSON_UTF8_CONTINUATION + (code & JSON_UTF8_LOW_SIX))}`;
258
+ }
259
+ const lead: string = String.fromCharCode(JSON_UTF8_THREE_BYTE_LEAD + (code >> 12));
260
+ const middle: string = String.fromCharCode(JSON_UTF8_CONTINUATION + ((code >> 6) & JSON_UTF8_LOW_SIX));
261
+ return `${lead}${middle}${String.fromCharCode(JSON_UTF8_CONTINUATION + (code & JSON_UTF8_LOW_SIX))}`;
262
+ };
263
+
264
+ /**
265
+ * The bytes of `text[at..end)` with the JSON escapes decoded — the body of a
266
+ * string literal, without its quotes.
267
+ *
268
+ * A literal with no backslash in it is answered as a substring, which is the
269
+ * common case and the whole reason for the first loop: a diagnostic's `file` and
270
+ * `code` never carry an escape, and building them out of parts would allocate an
271
+ * array to copy a string that was already there. An unknown escape stands for
272
+ * the byte it escapes, which is what `\"`, `\\` and `\/` want and is the lenient
273
+ * reading of anything else.
274
+ */
275
+ const jsonUnescape = (text: string, at: i32, end: i32): string => {
276
+ // Both callers pass the inside of a literal `jsonEndOfString` found, so
277
+ // `0 <= at` and `end <= text.length` always hold. Saying so here is what
278
+ // proves every read below in range and drops the `substring` clamps.
279
+ if (at < 0 || end > toI32(text.length)) {
280
+ return "";
281
+ }
282
+ let i: i32 = at;
283
+ while (i < end && toI32(text.charCodeAt(i)) !== JSON_BACKSLASH) {
284
+ i += 1;
285
+ }
286
+ if (i >= end) {
287
+ return text.substring(at, end);
288
+ }
289
+ const parts: string[] = [];
290
+ let start: i32 = at;
291
+ while (i < end) {
292
+ if (toI32(text.charCodeAt(i)) !== JSON_BACKSLASH) {
293
+ i += 1;
294
+ continue;
295
+ }
296
+ // `start === i` is an escape straight after another, and the empty run
297
+ // between them adds nothing to the join. The `start >= 0` half is always
298
+ // true — `start` is `at` or a cursor that has only moved forward — and is
299
+ // there because `start` is reassigned in the loop, which is where the
300
+ // bounds proof stops following it.
301
+ if (start >= 0 && start < i) {
302
+ parts.push(text.substring(start, i));
303
+ }
304
+ // A trailing backslash has nothing after it: `0` is no escape letter, so it
305
+ // falls to the default below and stands for itself.
306
+ const letter: i32 = i + 1 < end ? toI32(text.charCodeAt(i + 1)) : 0;
307
+ if (letter === 110) {
308
+ parts.push("\n");
309
+ i += 2;
310
+ } else if (letter === 116) {
311
+ parts.push("\t");
312
+ i += 2;
313
+ } else if (letter === 114) {
314
+ parts.push("\r");
315
+ i += 2;
316
+ } else if (letter === 98) {
317
+ parts.push(String.fromCharCode(8));
318
+ i += 2;
319
+ } else if (letter === 102) {
320
+ parts.push(String.fromCharCode(12));
321
+ i += 2;
322
+ } else if (letter === 117) {
323
+ const code: i32 = jsonHex4(text, i + 2, end);
324
+ if (code < 0) {
325
+ // Not four hex digits after the `u`: the escape is broken, and copying it
326
+ // through unchanged is the reading that loses nothing.
327
+ parts.push("\\u");
328
+ i += 2;
329
+ } else {
330
+ parts.push(jsonUtf8(code));
331
+ i += 6;
332
+ }
333
+ } else if (letter === 0) {
334
+ parts.push("\\");
335
+ i += 1;
336
+ } else {
337
+ parts.push(String.fromCharCode(letter));
338
+ i += 2;
339
+ }
340
+ start = i;
341
+ }
342
+ parts.push(text.substring(start, end));
343
+ return parts.join("");
344
+ };
345
+
346
+ /**
347
+ * The value of `name` in `object`, or `null` when the object does not have that
348
+ * field.
349
+ *
350
+ * A string value comes back unquoted and unescaped; any other value comes back
351
+ * as the bytes it was written with, so `parseInt` reads a number and `=== "true"`
352
+ * reads a boolean. The scan is left to right and answers the **first** field of
353
+ * that name, which is what a duplicated key means to every JSON reader that does
354
+ * not build a map.
355
+ *
356
+ * A nested value is stepped over rather than searched, so `name` is a field of
357
+ * this object and not a path into it: that is the "flat" in the module header,
358
+ * and it is the shape the compiler's `--json` line has.
359
+ */
360
+ export const jsonField = (object: string, name: string): string | null => {
361
+ const length: i32 = toI32(object.length);
362
+ let i: i32 = jsonSkipBlank(object, 0);
363
+ if (i >= length || toI32(object.charCodeAt(i)) !== JSON_OPEN_BRACE) {
364
+ return null;
365
+ }
366
+ i = jsonSkipBlank(object, i + 1);
367
+ // `jsonSkipBlank` never answers a negative index for a non-negative one, but
368
+ // the bounds proof does not look inside a callee, so every cursor it answers
369
+ // is tested for `i < 0` beside `i >= length`: the pair is what lets each read
370
+ // below go without a check.
371
+ while (i >= 0 && i < length && toI32(object.charCodeAt(i)) === JSON_QUOTE) {
372
+ const keyEnd: i32 = jsonEndOfString(object, i);
373
+ if (keyEnd < 0) {
374
+ return null;
375
+ }
376
+ const key: string = jsonUnescape(object, i + 1, keyEnd - 1);
377
+ i = jsonSkipBlank(object, keyEnd);
378
+ if (i < 0 || i >= length || toI32(object.charCodeAt(i)) !== JSON_COLON) {
379
+ return null;
380
+ }
381
+ const valueAt: i32 = jsonSkipBlank(object, i + 1);
382
+ const valueEnd: i32 = jsonEndOfValue(object, valueAt);
383
+ // `jsonEndOfValue` answers `-1` for a `valueAt` outside the object and never
384
+ // answers past its end, so the last three tests change no answer: they are
385
+ // the range the value's first byte and its `substring` are proved in.
386
+ if (valueEnd < 0 || valueAt < 0 || valueAt >= length || valueEnd > toI32(object.length)) {
387
+ return null;
388
+ }
389
+ if (key === name) {
390
+ if (toI32(object.charCodeAt(valueAt)) === JSON_QUOTE) {
391
+ return jsonUnescape(object, valueAt + 1, valueEnd - 1);
392
+ }
393
+ return object.substring(valueAt, valueEnd);
394
+ }
395
+ i = jsonSkipBlank(object, valueEnd);
396
+ if (i < 0 || i >= length || toI32(object.charCodeAt(i)) !== JSON_COMMA) {
397
+ return null;
398
+ }
399
+ i = jsonSkipBlank(object, i + 1);
400
+ }
401
+ return null;
402
+ };
package/std/pair.ts ADDED
@@ -0,0 +1,28 @@
1
+ /**
2
+ * `std/pair` — two values answered by one call.
3
+ *
4
+ * It exists for the function that has two things to say, like the lexer's
5
+ * `scanEscape`: where an escape ended, and whether it was malformed. Without a
6
+ * pair, the second answer lives in a field on some object, written by the
7
+ * callee and read back by the caller three lines later. With one, both come
8
+ * back in the return:
9
+ *
10
+ * import { Pair } from "nish/pair";
11
+ *
12
+ * const scanEscape = (at: i32): Pair<i32, boolean> => ({ first: at + 2, second: true });
13
+ *
14
+ * It is **for returning two values, not for storing them**. A struct held in an
15
+ * array or a class field should have named fields, and parallel arrays are the
16
+ * house style for two values side by side (`docs/wp15-performance.md` §1a), so
17
+ * a `Pair[]` is usually the slower shape and the less readable one.
18
+ *
19
+ * It is an `interface` rather than a class, so there is no constructor to call
20
+ * and nothing to import but the type: an object literal takes its type from the
21
+ * annotation, and the checker holds it to setting `first` and `second` exactly
22
+ * once each. There is no tuple syntax behind it and no special case in the
23
+ * compiler; `docs/wp23-language-surface.md` §5 is the argument for that.
24
+ */
25
+ export interface Pair<A, B> {
26
+ first: A;
27
+ second: B;
28
+ }