@amritk/nish-x86_64-linux 0.10.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/INSTALL.md +22 -5
- package/bin/nish +0 -0
- package/package.json +1 -1
- package/runtime/LICENSE-ryu +23 -0
- package/runtime/nish.d.ts +11 -0
- package/runtime/nish.h +9 -1
- package/runtime/nish.mjs +13 -0
- package/runtime/runtime.c +10 -1
- package/std/README.md +17 -10
- package/std/collections.ts +585 -0
- package/std/map.ts +37 -0
- package/std/threads.ts +150 -0
package/INSTALL.md
CHANGED
|
@@ -234,12 +234,13 @@ routes are
|
|
|
234
234
|
If you need one of these to be a first-class install, that is the conversation
|
|
235
235
|
to have on the issue tracker rather than a workaround to discover here.
|
|
236
236
|
|
|
237
|
-
**
|
|
238
|
-
|
|
239
|
-
[docs/wp12-release.md](wp12-release.md#release-procedure) step 4
|
|
240
|
-
|
|
237
|
+
**On the registry from 0.10.0.** Every release publishes the main package and
|
|
238
|
+
its platform packages to npm from `release.yml`, by trusted publishing
|
|
239
|
+
([docs/wp12-release.md](wp12-release.md#release-procedure) step 4), so
|
|
240
|
+
`npm install -g @amritk/nish` is the whole install. The rest of this section is
|
|
241
|
+
for a machine that cannot reach the registry, or a release before 0.10.0.
|
|
241
242
|
|
|
242
|
-
From a release tarball on GitHub — the same package `npm publish`
|
|
243
|
+
From a release tarball on GitHub — the same package `npm publish` uploads,
|
|
243
244
|
carried by the release instead of the registry. The `Release` workflow attaches
|
|
244
245
|
the npm tarball to the release it builds for a `v*` tag. `npm pack` names it
|
|
245
246
|
after `package.json#name`, so it is `amritk-nish-<version>.tgz` from 0.4.0 on
|
|
@@ -445,6 +446,21 @@ Other build profiles: `--profile size` (smallest binary), `--profile debug`
|
|
|
445
446
|
(no optimisation, symbols kept). Run `nish --help` for every flag, and
|
|
446
447
|
see the [README](../README.md) for the language subset.
|
|
447
448
|
|
|
449
|
+
To use a program as a script rather than keep the binary, run it:
|
|
450
|
+
|
|
451
|
+
```bash
|
|
452
|
+
nish run hello.ts
|
|
453
|
+
# hello from Nish
|
|
454
|
+
```
|
|
455
|
+
|
|
456
|
+
`nish run [options] <file.ts> [args ...]` compiles the file, links it into
|
|
457
|
+
`$XDG_CACHE_HOME/nish/run` (or `~/.cache/nish/run`) when that program has not
|
|
458
|
+
been linked before, and starts it with the arguments after the file. The first
|
|
459
|
+
run of a new edit pays for one link, and later runs start the cached binary.
|
|
460
|
+
It needs clang only for that link. It needs `HOME` or `XDG_CACHE_HOME` set,
|
|
461
|
+
and refuses the run without them rather than keep a binary in a shared
|
|
462
|
+
directory.
|
|
463
|
+
|
|
448
464
|
## 4. Exit codes
|
|
449
465
|
|
|
450
466
|
| Code | Meaning |
|
|
@@ -453,6 +469,7 @@ see the [README](../README.md) for the language subset.
|
|
|
453
469
|
| 1 | the program was rejected: compile error (`file:line:col: error: ...`), missing input file, or an `-o` layout that does not fit the module count |
|
|
454
470
|
| 2 | usage error: unknown flag, missing argument, no input files |
|
|
455
471
|
| 3 | toolchain error: `--link` found no `clang` (`CC` overrides), or `scripts/build.sh` failed (its output is shown; the `.ll` files are still written) — or the `nish` command found no prebuilt compiler for this platform, or one that would not start. Under `--json` all of these are one `NL0002` object |
|
|
472
|
+
| | `nish run` answers these codes until the program starts, and the program's own exit status (`128 + n` for a signal) after that. With neither `HOME` nor `XDG_CACHE_HOME` set it answers 3 |
|
|
456
473
|
| 70 | internal compiler error: an unexpected exception. Please report it at <https://github.com/amritk/nish/issues> with the input and command line; `NISH_DEBUG=1` prints the stack trace |
|
|
457
474
|
|
|
458
475
|
## Troubleshooting
|
package/bin/nish
CHANGED
|
Binary file
|
package/package.json
CHANGED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
Boost Software License - Version 1.0 - August 17th, 2003
|
|
2
|
+
|
|
3
|
+
Permission is hereby granted, free of charge, to any person or organization
|
|
4
|
+
obtaining a copy of the software and accompanying documentation covered by
|
|
5
|
+
this license (the "Software") to use, reproduce, display, distribute,
|
|
6
|
+
execute, and transmit the Software, and to prepare derivative works of the
|
|
7
|
+
Software, and to permit third-parties to whom the Software is furnished to
|
|
8
|
+
do so, all subject to the following:
|
|
9
|
+
|
|
10
|
+
The copyright notices in the Software and this entire statement, including
|
|
11
|
+
the above license grant, this restriction and the following disclaimer,
|
|
12
|
+
must be included in all copies of the Software, in whole or in part, and
|
|
13
|
+
all derivative works of the Software, unless such copies or derivative
|
|
14
|
+
works are solely in the form of machine-executable object code generated by
|
|
15
|
+
a source language processor.
|
|
16
|
+
|
|
17
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
18
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
19
|
+
FITNESS FOR A PARTICULAR PURPOSE, TITLE AND NON-INFRINGEMENT. IN NO EVENT
|
|
20
|
+
SHALL THE COPYRIGHT HOLDERS OR ANYONE DISTRIBUTING THE SOFTWARE BE LIABLE
|
|
21
|
+
FOR ANY DAMAGES OR OTHER LIABILITY, WHETHER IN CONTRACT, TORT OR OTHERWISE,
|
|
22
|
+
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
|
23
|
+
DEALINGS IN THE SOFTWARE.
|
package/runtime/nish.d.ts
CHANGED
|
@@ -25,6 +25,17 @@
|
|
|
25
25
|
* So a program that `tsc` accepts may still be rejected by `nish`; the
|
|
26
26
|
* reverse should never happen, and a case where it does is a bug in this file.
|
|
27
27
|
* `docs/LANGUAGE.md` is the normative description.
|
|
28
|
+
*
|
|
29
|
+
* **`Map` and `Set` are not declared here**, although `nish` has them as
|
|
30
|
+
* globals: the `"lib": ["ES2022"]` a program is checked against already
|
|
31
|
+
* declares JavaScript's, and a second declaration would clash with it.
|
|
32
|
+
* `nish`'s are that surface less what it defers — no `entries`, no `forEach`,
|
|
33
|
+
* no `for...of` over a `Map` itself — with `keys()` and `values()` admitted
|
|
34
|
+
* only as the iterable of a `for...of`, and `get`'s `V | undefined` only as a
|
|
35
|
+
* `const`'s initialiser, the left of `??` or an operand of `=== undefined`,
|
|
36
|
+
* none of which `tsc` can know (docs/LANGUAGE.md -> `Map` and `Set`).
|
|
37
|
+
* `nish/map`'s `reserve` and `getOrInsert` are not declared here either: they
|
|
38
|
+
* are ordinary source, `std/map.ts`, which `tsconfig.json` maps `nish/*` to.
|
|
28
39
|
*/
|
|
29
40
|
|
|
30
41
|
// ---- Numeric widths (docs/LANGUAGE.md -> Types) ------------------------------
|
package/runtime/nish.h
CHANGED
|
@@ -179,7 +179,15 @@ void nish_append_file(const nish_str *path, const nish_str *data);
|
|
|
179
179
|
* `push` that grows it copies the elements into the arena and leaves your
|
|
180
180
|
* buffer behind; the callee must not retain the pointer beyond the call.
|
|
181
181
|
* A returned array lives in the arena (valid until the next reset/release):
|
|
182
|
-
* copy `len` elements out of `data` before recycling.
|
|
182
|
+
* copy `len` elements out of `data` before recycling.
|
|
183
|
+
*
|
|
184
|
+
* An array field stored inside its object (`self/inline_arrays.ts`) is this
|
|
185
|
+
* header followed by its `K` slots, `struct { nish_array h; T slots[K]; }`
|
|
186
|
+
* with `h.data == (char *)slots` and `h.cap == K`; `self/runtime.ts`'s
|
|
187
|
+
* `ARRAY_TYPE` and `self/structs.ts`'s `INLINE_HEADER_BYTES` are the same 24
|
|
188
|
+
* bytes, and `tests/layout/inline_array.c` holds the three to it. It only
|
|
189
|
+
* happens where no header, `.d.ts` or N-API shim describes the class, so a
|
|
190
|
+
* host that includes a generated header never meets one. */
|
|
183
191
|
typedef struct nish_array { uint64_t len; uint64_t cap; char *data; } nish_array;
|
|
184
192
|
/* A fresh arena array of `len` uninitialised elements (`len == cap`), for a
|
|
185
193
|
* host that wants the runtime to own the storage (the wasm loader does).
|
package/runtime/nish.mjs
CHANGED
|
@@ -54,6 +54,7 @@
|
|
|
54
54
|
* harness already uses, so the two cannot drift apart: a semantic fixed there
|
|
55
55
|
* is fixed here in the same commit.
|
|
56
56
|
*/
|
|
57
|
+
import { registerHooks } from "node:module";
|
|
57
58
|
import * as shim from "./shim.mjs";
|
|
58
59
|
|
|
59
60
|
/** Install `value` as a global unless the program declared its own. */
|
|
@@ -141,3 +142,15 @@ provide("Arena", {
|
|
|
141
142
|
reset: shim.arenaReset,
|
|
142
143
|
used: shim.arenaUsed,
|
|
143
144
|
});
|
|
145
|
+
|
|
146
|
+
// The standard library. A program imports it as `nish/<module>`, which the
|
|
147
|
+
// compiler resolves to `std/<module>.ts` beside itself; this does the same for
|
|
148
|
+
// Node, relative to this file rather than to the program, so the same
|
|
149
|
+
// specifier works from any directory. Only the `nish/` package is answered
|
|
150
|
+
// here, and every other specifier goes to Node as it was written.
|
|
151
|
+
registerHooks({
|
|
152
|
+
resolve: (specifier, context, next) =>
|
|
153
|
+
specifier.startsWith("nish/")
|
|
154
|
+
? next(new URL(`../std/${specifier.slice("nish/".length)}.ts`, import.meta.url).href, context)
|
|
155
|
+
: next(specifier, context),
|
|
156
|
+
});
|
package/runtime/runtime.c
CHANGED
|
@@ -277,6 +277,13 @@ nish_str *nish_str_from_u64(uint64_t v) { return str_from_digits(v, 0); }
|
|
|
277
277
|
|
|
278
278
|
/* ---- Shortest round-trip digits (Ryu)
|
|
279
279
|
|
|
280
|
+
Adapted from Ryu (https://github.com/ulfjack/ryu), ryu/d2s.c and
|
|
281
|
+
ryu/common.h. Copyright 2018 Ulf Adams. Used under the Boost Software
|
|
282
|
+
License, Version 1.0; see runtime/LICENSE-ryu. The code from here to the end
|
|
283
|
+
of `nish_shortest_digits` follows upstream's `d2d()` and its helpers, which
|
|
284
|
+
is why it keeps their names, constants and comments; the two tables are
|
|
285
|
+
regenerated by scripts/gen-pow5-tables.py rather than copied.
|
|
286
|
+
|
|
280
287
|
`String(x)` in JavaScript prints the *fewest* digits that read back as the
|
|
281
288
|
same double, and the language promises that spelling. Reaching it by asking
|
|
282
289
|
snprintf for k digits and strtod whether they round-trip -- which is what
|
|
@@ -1079,7 +1086,9 @@ nish_str *nish_str_from_f64(double v) {
|
|
|
1079
1086
|
return nish_str_new(out, o - out);
|
|
1080
1087
|
}
|
|
1081
1088
|
|
|
1082
|
-
/* ---- Math.random: xorshift64*, seeded lazily from time and pid.
|
|
1089
|
+
/* ---- Math.random: xorshift64*, seeded lazily from time and pid. The shifts
|
|
1090
|
+
and the multiplier are Marsaglia's and Vigna's published constants, which
|
|
1091
|
+
are public domain.
|
|
1083
1092
|
Thread-local under -DNISH_THREADS (WP20 T0 §3.2): a shared seed word is a
|
|
1084
1093
|
race, and a per-thread one also makes each worker's stream its own rather
|
|
1085
1094
|
than an interleaving of everyone's. Two threads that start in the same
|
package/std/README.md
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
# `std/` — the standard library
|
|
2
2
|
|
|
3
3
|
Nish modules written in Nish, for Nish programs to import. There is no magic
|
|
4
|
-
here and
|
|
5
|
-
ordinary Nish source file, compiled as part of
|
|
6
|
-
and subject to the same rules as `examples/` or
|
|
7
|
-
([`docs/LANGUAGE.md`](../docs/LANGUAGE.md) is the style guide).
|
|
4
|
+
here and — with three exceptions, `threads.ts`, `collections.ts` and `map.ts` —
|
|
5
|
+
nothing the compiler knows about: a module in this directory is an ordinary Nish source file, compiled as part of
|
|
6
|
+
whatever program imports it, and subject to the same rules as `examples/` or
|
|
7
|
+
`self/` ([`docs/LANGUAGE.md`](../docs/LANGUAGE.md) is the style guide).
|
|
8
8
|
|
|
9
9
|
| Module | What it is |
|
|
10
10
|
| --- | --- |
|
|
@@ -12,6 +12,9 @@ and subject to the same rules as `examples/` or `self/`
|
|
|
12
12
|
| [`text.ts`](./text.ts) | the string operations a program would otherwise write inline: `splitLines`, `splitWhitespace`, `trim` and its halves, `contains`, `replaceAll`, and `firstDifference` over two arrays of lines |
|
|
13
13
|
| [`json.ts`](./json.ts) | `jsonField(object, name)`: the value of one field of one flat JSON object, which is the shape the compiler's own `--json` diagnostics have. A reader and not a parser — it answers text, answers `null` for a field that is not there, and does not validate |
|
|
14
14
|
| [`pair.ts`](./pair.ts) | `Pair<A, B>`: an interface with `first` and `second`, for a function that answers two values from one call. A type and nothing else — the caller writes an object literal at the return — and for returning two values rather than storing them side by side |
|
|
15
|
+
| [`collections.ts`](./collections.ts) | the global `Map<K, V>` and `Set<T>`: insertion-ordered tables whose buckets carry a hash fingerprint beside the entry index and whose entries keep their full hash, so every `get`, `set`, `add`, `has` and `delete` is one probe. `get` is not a method here: its `V | undefined` never crosses a call, so the compiler lowers it to `probe` and, where the key was found, `valueAt`. Nor are `keys()` and `values()`: an iterator is not a value, so a `for...of` over one is lowered to `walkOpen`, `walkNext`, `keyAt` or `valueAt`, and `walkClose`, and a count of live walks defers compaction until no loop is walking the table. A program never imports it: naming `Map` or `Set` loads it, and the compiler emits what a module uses of it into that module ([`docs/wp32-map.md`](../docs/wp32-map.md), [`docs/LANGUAGE.md`](../docs/LANGUAGE.md#map-and-set)). Its `hashKey`, `sameKey` and `storedKey` are lowered by the compiler per key type |
|
|
16
|
+
| [`map.ts`](./map.ts) | `reserve(m, n)` and `getOrInsert(m, k, v)` for the global `Map`. Their bodies are the meaning, and what runs under Node: `reserve` does nothing, and `getOrInsert` is a `get`, and a `set` of `v` when the key was missing. Natively the compiler lowers every call in place — `reserve` to the table's `reserveSlots`, which grows the buckets once so that `n` entries fit without a rebuild, and `getOrInsert` to one `probe` and a `valueAt` or an `insertAt` through its answer — so, like `collections.ts`, it writes no `.ll` of its own ([`docs/wp32-map.md`](../docs/wp32-map.md) §9.2, [`docs/LANGUAGE.md`](../docs/LANGUAGE.md#map-and-set)) |
|
|
17
|
+
| [`threads.ts`](./threads.ts) | `parallelMapInto(src, dst, f)` and `parallelReduce(src, f, identity)`: a function over every element of an array, on as many threads as the length is worth. Its bodies are the sequential meaning, which is what runs under Node; the compiler recognises the two templates by module and name, lowers the one loop in each onto `nish_parallel_range`, holds the function to the rules that make that safe, and compiles an importing program with `--threads` ([`docs/LANGUAGE.md`](../docs/LANGUAGE.md#data-parallelism-nishthreads)). `tests/link/par_*` are its programs |
|
|
15
18
|
|
|
16
19
|
## How a program imports it
|
|
17
20
|
|
|
@@ -62,12 +65,16 @@ reasons that are only true of it:
|
|
|
62
65
|
package and is versioned with it, so "the `std/` beside this binary" is not a
|
|
63
66
|
guess a resolver makes — it is the only `std/` that can be correct for the
|
|
64
67
|
compiler reading it. No version can be skewed against it.
|
|
65
|
-
- It **is** what Node
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
`
|
|
69
|
-
|
|
70
|
-
|
|
68
|
+
- It **is** what Node and `tsc` resolve. The package is published as
|
|
69
|
+
`@amritk/nish`, so a bare `nish/text` is not a package self-reference on its
|
|
70
|
+
own; `runtime/nish.mjs`, the prelude a program runs under Node with
|
|
71
|
+
(`node --experimental-strip-types --import ./runtime/nish.mjs`), resolves
|
|
72
|
+
`nish/<module>` to `std/<module>.ts` beside itself, and the repository's
|
|
73
|
+
`tsconfig.json` maps `nish/*` to `./std/*`, which is what gives `tsc` and an
|
|
74
|
+
editor go-to-definition into the real source. `package.json` still declares
|
|
75
|
+
`"./*": "./std/*.ts"` in `exports`, so `@amritk/nish/text` reaches the same
|
|
76
|
+
file. The compiler short-circuits to that answer rather than walking
|
|
77
|
+
`node_modules` to reach it.
|
|
71
78
|
|
|
72
79
|
So this is WP21's first slice rather than a detour around it: the spelling is
|
|
73
80
|
the one WP21 specifies, and what is still missing is resolution for specifiers
|
|
@@ -0,0 +1,585 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `std/collections` — the global `Map` and `Set` (docs/wp32-map.md).
|
|
3
|
+
*
|
|
4
|
+
* A program does not import this module. Naming `Map` or `Set` is what loads
|
|
5
|
+
* it, unless the module declares or imports a `Map` or `Set` of its own, and
|
|
6
|
+
* the compiler emits every instance used, and every function here that it
|
|
7
|
+
* reaches, into the module that uses it as `internal` functions. So this file
|
|
8
|
+
* writes no `.ll` of its own, and a one-file program that names `Map` is still
|
|
9
|
+
* one module (§4.1). Under Node, and under `tsc`, `Map` and `Set` are the
|
|
10
|
+
* platform's, so nothing here runs there.
|
|
11
|
+
*
|
|
12
|
+
* **The layout is §2's.** Entries sit in insertion order in parallel arrays,
|
|
13
|
+
* `entryKeys`, `entryValues` and `entryHashes`, and `entryHashes` holds the
|
|
14
|
+
* full 32-bit hash of each. The bucket table `slots` is open addressing with
|
|
15
|
+
* linear probing, and a bucket is one `u32`:
|
|
16
|
+
*
|
|
17
|
+
* bits 31..24 the top eight bits of the key's hash (the fingerprint)
|
|
18
|
+
* bits 23..0 the entry's index plus one
|
|
19
|
+
*
|
|
20
|
+
* An empty bucket is 0. A deleted entry's bucket is `0x01000000` (16777216),
|
|
21
|
+
* fingerprint 1 and index field 0, which no live entry has; its stored hash is
|
|
22
|
+
* set to 0, which is why a computed hash of 0 is moved to 1. A probe reads the
|
|
23
|
+
* entry arrays only when a bucket's fingerprint matches, and compares the
|
|
24
|
+
* stored hash before the key, so a miss almost never touches a key, and a hit
|
|
25
|
+
* compares one. Growth and compaction re-file the buckets from the stored
|
|
26
|
+
* hashes and never hash a key again.
|
|
27
|
+
*
|
|
28
|
+
* **Every operation is one probe.** `probe` answers a packed `i64`: the entry
|
|
29
|
+
* it found and the bucket that points at it, or the empty bucket it stopped at
|
|
30
|
+
* and the key's hash. `set` and `add` insert through that result, and nothing
|
|
31
|
+
* here asks `has` and then `set`. `probe`, `valueAt`, `setValueAt` and
|
|
32
|
+
* `insertAt` are the pieces a fused lookup writes through, and they write
|
|
33
|
+
* nothing but what their names say: `probe` writes no memory at all.
|
|
34
|
+
*
|
|
35
|
+
* A program sees only the JavaScript members (§7), `size`, `get`, `set`/`add`,
|
|
36
|
+
* `has`, `delete` and `clear`, and `keys()` and `values()` as the iterable of
|
|
37
|
+
* a `for...of`. Neither `get` nor the iterators has a method here: `get`'s
|
|
38
|
+
* `V | undefined` never crosses a call, so the compiler lowers it to `probe`
|
|
39
|
+
* and, where found, `valueAt` (§3.2), and an iterator is not a value, so a
|
|
40
|
+
* `for...of` over one is lowered to `walkOpen`, `walkNext`, `keyAt` or
|
|
41
|
+
* `valueAt`, and `walkClose` (§6.2). The rest of this file is refused by name
|
|
42
|
+
* outside it.
|
|
43
|
+
*/
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* The most entries a table holds, dead ones included until a rebuild: 2^24 - 1,
|
|
47
|
+
* what the 24-bit index field holds as an index plus one. Node's own `Map` and
|
|
48
|
+
* `Set` hold one more, 2^24, and throw on the next insert.
|
|
49
|
+
*/
|
|
50
|
+
const INDEX_CAP: i32 = 16777215;
|
|
51
|
+
|
|
52
|
+
/** The first bucket count. A power of two, as every bucket count is. */
|
|
53
|
+
const INITIAL_SLOTS: i32 = 8;
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* The key's hash, never 0: FNV-1a over a string's bytes, murmur3's `fmix32`
|
|
57
|
+
* for an integer of 32 bits or fewer, `fmix64` folded to 32 bits for a 64-bit
|
|
58
|
+
* integer, a float (normalised first, so that -0 and +0, and every NaN, hash
|
|
59
|
+
* alike) and a class instance's address (§5.2). FNV, MurmurHash3 and their
|
|
60
|
+
* constants are public domain.
|
|
61
|
+
*
|
|
62
|
+
* The body is never emitted. Every call is lowered in place, per key type, by
|
|
63
|
+
* `self/emit_map.ts`; the constant is what the checker and the whole-program
|
|
64
|
+
* facts see, and the facts of the function that calls it are what decide its
|
|
65
|
+
* attributes, because every call is inside a probe that reads the table.
|
|
66
|
+
*/
|
|
67
|
+
const hashKey = <K>(key: K): u32 => 1;
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* JavaScript's key equality, SameValueZero: `===`, except that NaN equals NaN.
|
|
71
|
+
* Lowered in place per key type as well: `nish_str_eq` for a string, one
|
|
72
|
+
* `icmp` for an integer, a boolean, an enum and a class instance, and for a
|
|
73
|
+
* float `a == b` or both unordered.
|
|
74
|
+
*/
|
|
75
|
+
const sameKey = <K>(a: K, b: K): boolean => a === b;
|
|
76
|
+
|
|
77
|
+
/** The first bucket for `h`: its low bits, folded with the high half. */
|
|
78
|
+
const homeBucket = (h: u32, mask: i32): i32 => toI32(h ^ (h >>> 16)) & mask;
|
|
79
|
+
|
|
80
|
+
/** The bucket word for entry `index` of hash `h`. */
|
|
81
|
+
const slotWord = (h: u32, index: i32): u32 => ((h >>> 24) << 24) | toU32(index + 1);
|
|
82
|
+
|
|
83
|
+
/** A probe that found entry `index`, pointed at by `bucket`. Never negative. */
|
|
84
|
+
const foundAt = (bucket: i32, index: i32): i64 => (toI64(bucket) << 32) | toI64(index);
|
|
85
|
+
|
|
86
|
+
/** A probe that stopped at the empty `bucket` for hash `h`. Always negative. */
|
|
87
|
+
const absentAt = (bucket: i32, h: u32): i64 => toI64(-1) - ((toI64(bucket) << 32) | toI64(h));
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* The one probe, for both classes: linear probing from `h`'s home bucket to
|
|
91
|
+
* the bucket pointing at `key`, or to the first empty one. A tombstone is
|
|
92
|
+
* walked past — its index field is 0, so its entry index is -1 and the range
|
|
93
|
+
* test turns it away — and the load bound keeps a quarter of the buckets
|
|
94
|
+
* empty, so the walk ends.
|
|
95
|
+
*/
|
|
96
|
+
const probeTable = <K>(slots: u32[], mask: i32, hashes: u32[], keys: K[], key: K): i64 => {
|
|
97
|
+
const h = hashKey(key);
|
|
98
|
+
const fingerprint = h >>> 24;
|
|
99
|
+
let bucket = homeBucket(h, mask);
|
|
100
|
+
// The length is read in the condition rather than once: the key compare is
|
|
101
|
+
// a call, and a call ends every length fact the bounds proof holds.
|
|
102
|
+
while (bucket >= 0 && bucket < toI32(slots.length)) {
|
|
103
|
+
const word = slots[bucket];
|
|
104
|
+
if (word === 0) {
|
|
105
|
+
return absentAt(bucket, h);
|
|
106
|
+
}
|
|
107
|
+
if (word >>> 24 === fingerprint) {
|
|
108
|
+
const at = toI32(word & 16777215) - 1;
|
|
109
|
+
if (at >= 0 && at < toI32(hashes.length) && hashes[at] === h && at < toI32(keys.length) && sameKey(keys[at], key)) {
|
|
110
|
+
return foundAt(bucket, at);
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
bucket = (bucket + 1) & mask;
|
|
114
|
+
}
|
|
115
|
+
panic("collections: a probe ran out of buckets");
|
|
116
|
+
};
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* Point the first empty bucket from `h`'s home at entry `index`. A rebuild's
|
|
120
|
+
* re-filing: every key is already known to be distinct, so no key is
|
|
121
|
+
* compared, and the stored hash is all it needs.
|
|
122
|
+
*/
|
|
123
|
+
const fileEntry = (slots: u32[], mask: i32, h: u32, index: i32): void => {
|
|
124
|
+
const word = slotWord(h, index);
|
|
125
|
+
let bucket = homeBucket(h, mask);
|
|
126
|
+
while (bucket >= 0 && bucket < toI32(slots.length)) {
|
|
127
|
+
if (slots[bucket] === 0) {
|
|
128
|
+
slots[bucket] = word;
|
|
129
|
+
return;
|
|
130
|
+
}
|
|
131
|
+
bucket = (bucket + 1) & mask;
|
|
132
|
+
}
|
|
133
|
+
};
|
|
134
|
+
|
|
135
|
+
/** Slide the live entries of `items` down over the dead ones, in order, and drop the tail. */
|
|
136
|
+
const compactEntries = <T>(items: T[], hashes: u32[]): void => {
|
|
137
|
+
const used = toI32(items.length);
|
|
138
|
+
let to: i32 = 0;
|
|
139
|
+
for (let from: i32 = 0; from < used && from < toI32(hashes.length); from++) {
|
|
140
|
+
if (hashes[from] !== 0 && to >= 0 && to < used && from < toI32(items.length)) {
|
|
141
|
+
items[to] = items[from];
|
|
142
|
+
to++;
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
while (toI32(items.length) > to) {
|
|
146
|
+
items.pop();
|
|
147
|
+
}
|
|
148
|
+
};
|
|
149
|
+
|
|
150
|
+
/** The stored hashes compacted the same way, last, since the two above read them. */
|
|
151
|
+
const compactHashes = (hashes: u32[]): void => {
|
|
152
|
+
const used = toI32(hashes.length);
|
|
153
|
+
let to: i32 = 0;
|
|
154
|
+
for (let from: i32 = 0; from < used; from++) {
|
|
155
|
+
const h = hashes[from];
|
|
156
|
+
if (h !== 0 && to >= 0 && to < used) {
|
|
157
|
+
hashes[to] = h;
|
|
158
|
+
to++;
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
while (toI32(hashes.length) > to) {
|
|
162
|
+
hashes.pop();
|
|
163
|
+
}
|
|
164
|
+
};
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* The bucket table after a rebuild of `live` entries out of `used`. More than
|
|
168
|
+
* half dead compacts at the same size and clears the table in place — a new
|
|
169
|
+
* array would leave the old one in the arena, once per compaction, which a
|
|
170
|
+
* table that churns forever cannot afford (§6.1). Otherwise it doubles, and
|
|
171
|
+
* the table never shrinks.
|
|
172
|
+
*/
|
|
173
|
+
const rebuiltSlots = (slots: u32[], live: i32, used: i32): u32[] => {
|
|
174
|
+
const n = toI32(slots.length);
|
|
175
|
+
if (live * 2 < used) {
|
|
176
|
+
clearSlots(slots);
|
|
177
|
+
return slots;
|
|
178
|
+
}
|
|
179
|
+
return new Array<u32>(n * 2);
|
|
180
|
+
};
|
|
181
|
+
|
|
182
|
+
/**
|
|
183
|
+
* Re-file every live entry from its stored hash. After a compaction none is
|
|
184
|
+
* dead; after a rebuild during a walk, which does not compact, a dead entry
|
|
185
|
+
* keeps its place and takes no bucket, since no probe can find it.
|
|
186
|
+
*/
|
|
187
|
+
const refile = (slots: u32[], hashes: u32[]): void => {
|
|
188
|
+
const mask = toI32(slots.length) - 1;
|
|
189
|
+
for (let i: i32 = 0; i < toI32(hashes.length); i++) {
|
|
190
|
+
const h = hashes[i];
|
|
191
|
+
if (h !== 0) {
|
|
192
|
+
fileEntry(slots, mask, h, i);
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
};
|
|
196
|
+
|
|
197
|
+
/**
|
|
198
|
+
* `delete`'s write through a found probe result: the bucket becomes a
|
|
199
|
+
* tombstone and the entry's stored hash 0. The key and value stay in the entry
|
|
200
|
+
* until a rebuild compacts them away (§6.1).
|
|
201
|
+
*/
|
|
202
|
+
const killEntry = (slots: u32[], hashes: u32[], found: i64): void => {
|
|
203
|
+
const at = toI32(found);
|
|
204
|
+
const bucket = toI32(found >> 32);
|
|
205
|
+
if (bucket >= 0 && bucket < toI32(slots.length)) {
|
|
206
|
+
slots[bucket] = 16777216;
|
|
207
|
+
}
|
|
208
|
+
if (at >= 0 && at < toI32(hashes.length)) {
|
|
209
|
+
hashes[at] = 0;
|
|
210
|
+
}
|
|
211
|
+
};
|
|
212
|
+
|
|
213
|
+
/**
|
|
214
|
+
* Zero every element in place: the buckets, which emptying a table and
|
|
215
|
+
* compacting one both start with, and under a walk the stored hashes, which
|
|
216
|
+
* is how `clear` marks every entry dead without moving one (§6.1).
|
|
217
|
+
*/
|
|
218
|
+
const clearSlots = (slots: u32[]): void => {
|
|
219
|
+
for (let i: i32 = 0; i < toI32(slots.length); i++) {
|
|
220
|
+
slots[i] = 0;
|
|
221
|
+
}
|
|
222
|
+
};
|
|
223
|
+
|
|
224
|
+
/**
|
|
225
|
+
* The index of the first live entry at `from` or after it, or -1 when there is
|
|
226
|
+
* none: a `for...of` walk's step. It reads the entry count on every call, so
|
|
227
|
+
* an entry appended during the walk is reached, and it skips a dead entry,
|
|
228
|
+
* whose stored hash is 0, so a key deleted before the walk reaches it is not
|
|
229
|
+
* visited (docs/wp32-map.md §6.2).
|
|
230
|
+
*/
|
|
231
|
+
const nextLive = (hashes: u32[], from: i32): i32 => {
|
|
232
|
+
for (let i: i32 = from; i >= 0 && i < toI32(hashes.length); i++) {
|
|
233
|
+
if (hashes[i] !== 0) {
|
|
234
|
+
return i;
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
return -1;
|
|
238
|
+
};
|
|
239
|
+
|
|
240
|
+
/** Drop every element of `items`, keeping its capacity. */
|
|
241
|
+
const truncate = <T>(items: T[]): void => {
|
|
242
|
+
while (toI32(items.length) > 0) {
|
|
243
|
+
items.pop();
|
|
244
|
+
}
|
|
245
|
+
};
|
|
246
|
+
|
|
247
|
+
/**
|
|
248
|
+
* Point bucket `bucket`, the empty one an absent probe stopped at, at the entry
|
|
249
|
+
* just appended as index `used - 1`; or, when a rebuild made room first and
|
|
250
|
+
* moved the buckets (`bucket` is -1), file it from its hash.
|
|
251
|
+
*/
|
|
252
|
+
const fileAppended = (slots: u32[], mask: i32, bucket: i32, h: u32, used: i32): void => {
|
|
253
|
+
if (bucket >= 0 && bucket < toI32(slots.length)) {
|
|
254
|
+
slots[bucket] = slotWord(h, used - 1);
|
|
255
|
+
} else {
|
|
256
|
+
fileEntry(slots, mask, h, used - 1);
|
|
257
|
+
}
|
|
258
|
+
};
|
|
259
|
+
|
|
260
|
+
/**
|
|
261
|
+
* A key-value table in insertion order, with JavaScript's semantics: keys are
|
|
262
|
+
* compared by SameValueZero, a key set again keeps its place, and a deleted key
|
|
263
|
+
* set again goes to the end.
|
|
264
|
+
*/
|
|
265
|
+
// biome-ignore lint/suspicious/noShadowRestrictedNames: this is the global `Map`, which the compiler loads for a program that names it
|
|
266
|
+
export class Map<K, V> {
|
|
267
|
+
/** How many entries are live. The one field a program may read, and it may not write it. */
|
|
268
|
+
size: number = 0;
|
|
269
|
+
/** Bucket -> fingerprint and entry index plus one; see the header. */
|
|
270
|
+
slots: u32[];
|
|
271
|
+
/** `slots.length - 1`: the table is a power of two. */
|
|
272
|
+
mask: i32 = 7;
|
|
273
|
+
/** Live entries, as an `i32` for the load and compaction arithmetic. */
|
|
274
|
+
live: i32 = 0;
|
|
275
|
+
entryKeys: K[];
|
|
276
|
+
entryValues: V[];
|
|
277
|
+
/** The full hash of each entry; 0 once the entry is deleted. */
|
|
278
|
+
entryHashes: u32[];
|
|
279
|
+
/**
|
|
280
|
+
* How many `for...of` loops are walking the table now. While it is above 0
|
|
281
|
+
* a rebuild doubles rather than compacts, and `clear` marks entries dead
|
|
282
|
+
* rather than truncating, so no entry moves under a walk's cursor (§6.2).
|
|
283
|
+
*/
|
|
284
|
+
walks: i32 = 0;
|
|
285
|
+
|
|
286
|
+
constructor() {
|
|
287
|
+
this.slots = new Array<u32>(INITIAL_SLOTS);
|
|
288
|
+
this.entryKeys = [];
|
|
289
|
+
this.entryValues = [];
|
|
290
|
+
this.entryHashes = [];
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
/** The packed probe result for `key`: see `probeTable`, `foundAt` and `absentAt`. */
|
|
294
|
+
probe(key: K): i64 {
|
|
295
|
+
return probeTable(this.slots, this.mask, this.entryHashes, this.entryKeys, key);
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
has(key: K): boolean {
|
|
299
|
+
return this.probe(key) >= 0;
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
/** Set `key` to `value`, in place when it is there and at the end when it is not: one probe. */
|
|
303
|
+
set(key: K, value: V): Map<K, V> {
|
|
304
|
+
const found = this.probe(key);
|
|
305
|
+
if (found >= 0) {
|
|
306
|
+
this.setValueAt(toI32(found), value);
|
|
307
|
+
} else {
|
|
308
|
+
this.insertAt(found, key, value);
|
|
309
|
+
}
|
|
310
|
+
return this;
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
delete(key: K): boolean {
|
|
314
|
+
const found = this.probe(key);
|
|
315
|
+
if (found < 0) {
|
|
316
|
+
return false;
|
|
317
|
+
}
|
|
318
|
+
killEntry(this.slots, this.entryHashes, found);
|
|
319
|
+
this.live = this.live - 1;
|
|
320
|
+
this.size = this.size - 1;
|
|
321
|
+
return true;
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
/**
|
|
325
|
+
* Empty the table in place: the buckets are zeroed and the entries
|
|
326
|
+
* truncated, or, while a loop walks the table, marked dead and kept (§6.1).
|
|
327
|
+
*/
|
|
328
|
+
clear(): void {
|
|
329
|
+
clearSlots(this.slots);
|
|
330
|
+
if (this.walks > 0) {
|
|
331
|
+
clearSlots(this.entryHashes); // every entry dead, and the count kept
|
|
332
|
+
} else {
|
|
333
|
+
truncate(this.entryKeys);
|
|
334
|
+
truncate(this.entryValues);
|
|
335
|
+
truncate(this.entryHashes);
|
|
336
|
+
}
|
|
337
|
+
this.live = 0;
|
|
338
|
+
this.size = 0;
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
/**
|
|
342
|
+
* The four pieces a `for...of` over `keys()` or `values()` is lowered to
|
|
343
|
+
* (`emitForOf`, docs/wp32-map.md §6.2): `walkOpen` where the loop is
|
|
344
|
+
* entered, `walkNext` for the first live entry and after each pass,
|
|
345
|
+
* `keyAt` or `valueAt` for the loop variable, and `walkClose` on every edge
|
|
346
|
+
* that leaves the loop. Two stores a loop, and none per entry.
|
|
347
|
+
*/
|
|
348
|
+
walkOpen(): void {
|
|
349
|
+
this.walks = this.walks + 1;
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
walkNext(from: i32): i32 {
|
|
353
|
+
return nextLive(this.entryHashes, from);
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
walkClose(): void {
|
|
357
|
+
this.walks = this.walks - 1;
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
/** The key of entry `index`, which `walkNext` answered. */
|
|
361
|
+
keyAt(index: i32): K {
|
|
362
|
+
if (index < 0 || index >= toI32(this.entryKeys.length)) {
|
|
363
|
+
panic("Map: no entry at this index");
|
|
364
|
+
}
|
|
365
|
+
return this.entryKeys[index];
|
|
366
|
+
}
|
|
367
|
+
|
|
368
|
+
/** The value of the entry a probe found, or a walk reached. */
|
|
369
|
+
valueAt(index: i32): V {
|
|
370
|
+
if (index < 0 || index >= toI32(this.entryValues.length)) {
|
|
371
|
+
panic("Map: no entry at this index");
|
|
372
|
+
}
|
|
373
|
+
return this.entryValues[index];
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
/** Overwrite the value of the entry a probe found; it keeps its place. */
|
|
377
|
+
setValueAt(index: i32, value: V): void {
|
|
378
|
+
if (index >= 0 && index < toI32(this.entryValues.length)) {
|
|
379
|
+
this.entryValues[index] = value;
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
/** Append an entry for `key` at the empty bucket an absent probe result names, reusing its hash. */
|
|
384
|
+
insertAt(absent: i64, key: K, value: V): void {
|
|
385
|
+
const packed = -1 - absent;
|
|
386
|
+
let bucket = toI32(packed >> 32);
|
|
387
|
+
const h = toU32(packed);
|
|
388
|
+
if (toI32(this.entryKeys.length) >= INDEX_CAP) {
|
|
389
|
+
// Dead entries hold the cap: compact them away, then find the bucket
|
|
390
|
+
// again. A walk defers compaction, so under one the cap is full.
|
|
391
|
+
if (this.live >= INDEX_CAP || this.walks > 0) {
|
|
392
|
+
panic("Map maximum size exceeded");
|
|
393
|
+
}
|
|
394
|
+
this.rebuild();
|
|
395
|
+
bucket = -1;
|
|
396
|
+
}
|
|
397
|
+
this.entryKeys.push(storedKey(key));
|
|
398
|
+
this.entryValues.push(value);
|
|
399
|
+
this.entryHashes.push(h);
|
|
400
|
+
this.live = this.live + 1;
|
|
401
|
+
this.size = this.size + 1;
|
|
402
|
+
// Every entry takes a bucket, live or dead, until a rebuild, which files
|
|
403
|
+
// the new entry with the rest; otherwise it takes the bucket the probe found.
|
|
404
|
+
const used = toI32(this.entryKeys.length);
|
|
405
|
+
if (used * 4 > toI32(this.slots.length) * 3) {
|
|
406
|
+
this.rebuild();
|
|
407
|
+
} else {
|
|
408
|
+
fileAppended(this.slots, this.mask, bucket, h, used);
|
|
409
|
+
}
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
/**
|
|
413
|
+
* Compact or double, then re-file the buckets from the stored hashes (§6.1).
|
|
414
|
+
* While a loop walks the table it always doubles and moves no entry, and the
|
|
415
|
+
* first rebuild after the walk compacts (§6.2).
|
|
416
|
+
*/
|
|
417
|
+
rebuild(): void {
|
|
418
|
+
const used = toI32(this.entryKeys.length);
|
|
419
|
+
const walking = this.walks > 0;
|
|
420
|
+
const slots = rebuiltSlots(this.slots, walking ? used : this.live, used);
|
|
421
|
+
if (!walking && this.live < used) {
|
|
422
|
+
compactEntries(this.entryKeys, this.entryHashes);
|
|
423
|
+
compactEntries(this.entryValues, this.entryHashes);
|
|
424
|
+
compactHashes(this.entryHashes);
|
|
425
|
+
}
|
|
426
|
+
this.slots = slots;
|
|
427
|
+
this.mask = toI32(slots.length) - 1;
|
|
428
|
+
refile(slots, this.entryHashes);
|
|
429
|
+
}
|
|
430
|
+
|
|
431
|
+
/**
|
|
432
|
+
* `reserve(m, n)` from `nish/map` (docs/wp32-map.md §9.2): grow the bucket
|
|
433
|
+
* table until `n` entries, dead ones included, fit under the load bound, so
|
|
434
|
+
* that the inserts up to `n` rebuild nothing. It only ever grows the buckets
|
|
435
|
+
* and re-files them from the stored hashes, so no entry moves and it is as
|
|
436
|
+
* safe under a walk as the doubling a walk already allows (§6.2). A count
|
|
437
|
+
* that is not positive, or not a number, does nothing, as `reserve` does
|
|
438
|
+
* under Node; one past the entry cap is the cap.
|
|
439
|
+
*/
|
|
440
|
+
reserveSlots(n: number): void {
|
|
441
|
+
if (!(n > 0)) {
|
|
442
|
+
return;
|
|
443
|
+
}
|
|
444
|
+
const want: i32 = n >= 16777215 ? INDEX_CAP : toI32(n);
|
|
445
|
+
const have = toI32(this.slots.length);
|
|
446
|
+
let size = have;
|
|
447
|
+
while (want * 4 > size * 3) {
|
|
448
|
+
size = size * 2;
|
|
449
|
+
}
|
|
450
|
+
if (size > have) {
|
|
451
|
+
const slots = new Array<u32>(size);
|
|
452
|
+
this.slots = slots;
|
|
453
|
+
this.mask = size - 1;
|
|
454
|
+
refile(slots, this.entryHashes);
|
|
455
|
+
}
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
/** A set of keys in insertion order: `Map`'s table with no values, and the same probe. */
|
|
460
|
+
// biome-ignore lint/suspicious/noShadowRestrictedNames: this is the global `Set`, which the compiler loads for a program that names it
|
|
461
|
+
export class Set<T> {
|
|
462
|
+
/** How many elements are live. The one field a program may read, and it may not write it. */
|
|
463
|
+
size: number = 0;
|
|
464
|
+
slots: u32[];
|
|
465
|
+
mask: i32 = 7;
|
|
466
|
+
live: i32 = 0;
|
|
467
|
+
entryKeys: T[];
|
|
468
|
+
entryHashes: u32[];
|
|
469
|
+
/** `Map`'s walk count: how many `for...of` loops are walking the table now. */
|
|
470
|
+
walks: i32 = 0;
|
|
471
|
+
|
|
472
|
+
constructor() {
|
|
473
|
+
this.slots = new Array<u32>(INITIAL_SLOTS);
|
|
474
|
+
this.entryKeys = [];
|
|
475
|
+
this.entryHashes = [];
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
probe(key: T): i64 {
|
|
479
|
+
return probeTable(this.slots, this.mask, this.entryHashes, this.entryKeys, key);
|
|
480
|
+
}
|
|
481
|
+
|
|
482
|
+
has(key: T): boolean {
|
|
483
|
+
return this.probe(key) >= 0;
|
|
484
|
+
}
|
|
485
|
+
|
|
486
|
+
/** Add `key` at the end when it is not there already: one probe. */
|
|
487
|
+
add(key: T): Set<T> {
|
|
488
|
+
const found = this.probe(key);
|
|
489
|
+
if (found < 0) {
|
|
490
|
+
this.insertAt(found, key);
|
|
491
|
+
}
|
|
492
|
+
return this;
|
|
493
|
+
}
|
|
494
|
+
|
|
495
|
+
delete(key: T): boolean {
|
|
496
|
+
const found = this.probe(key);
|
|
497
|
+
if (found < 0) {
|
|
498
|
+
return false;
|
|
499
|
+
}
|
|
500
|
+
killEntry(this.slots, this.entryHashes, found);
|
|
501
|
+
this.live = this.live - 1;
|
|
502
|
+
this.size = this.size - 1;
|
|
503
|
+
return true;
|
|
504
|
+
}
|
|
505
|
+
|
|
506
|
+
clear(): void {
|
|
507
|
+
clearSlots(this.slots);
|
|
508
|
+
if (this.walks > 0) {
|
|
509
|
+
clearSlots(this.entryHashes); // every entry dead, and the count kept
|
|
510
|
+
} else {
|
|
511
|
+
truncate(this.entryKeys);
|
|
512
|
+
truncate(this.entryHashes);
|
|
513
|
+
}
|
|
514
|
+
this.live = 0;
|
|
515
|
+
this.size = 0;
|
|
516
|
+
}
|
|
517
|
+
|
|
518
|
+
/** `Map`'s walk: `for (const x of s)`, `s.keys()` and `s.values()` are all this one. */
|
|
519
|
+
walkOpen(): void {
|
|
520
|
+
this.walks = this.walks + 1;
|
|
521
|
+
}
|
|
522
|
+
|
|
523
|
+
walkNext(from: i32): i32 {
|
|
524
|
+
return nextLive(this.entryHashes, from);
|
|
525
|
+
}
|
|
526
|
+
|
|
527
|
+
walkClose(): void {
|
|
528
|
+
this.walks = this.walks - 1;
|
|
529
|
+
}
|
|
530
|
+
|
|
531
|
+
keyAt(index: i32): T {
|
|
532
|
+
if (index < 0 || index >= toI32(this.entryKeys.length)) {
|
|
533
|
+
panic("Set: no entry at this index");
|
|
534
|
+
}
|
|
535
|
+
return this.entryKeys[index];
|
|
536
|
+
}
|
|
537
|
+
|
|
538
|
+
insertAt(absent: i64, key: T): void {
|
|
539
|
+
const packed = -1 - absent;
|
|
540
|
+
let bucket = toI32(packed >> 32);
|
|
541
|
+
const h = toU32(packed);
|
|
542
|
+
if (toI32(this.entryKeys.length) >= INDEX_CAP) {
|
|
543
|
+
if (this.live >= INDEX_CAP || this.walks > 0) {
|
|
544
|
+
panic("Set maximum size exceeded");
|
|
545
|
+
}
|
|
546
|
+
this.rebuild();
|
|
547
|
+
bucket = -1;
|
|
548
|
+
}
|
|
549
|
+
this.entryKeys.push(storedKey(key));
|
|
550
|
+
this.entryHashes.push(h);
|
|
551
|
+
this.live = this.live + 1;
|
|
552
|
+
this.size = this.size + 1;
|
|
553
|
+
// Every entry takes a bucket, live or dead, until a rebuild, which files
|
|
554
|
+
// the new entry with the rest; otherwise it takes the bucket the probe found.
|
|
555
|
+
const used = toI32(this.entryKeys.length);
|
|
556
|
+
if (used * 4 > toI32(this.slots.length) * 3) {
|
|
557
|
+
this.rebuild();
|
|
558
|
+
} else {
|
|
559
|
+
fileAppended(this.slots, this.mask, bucket, h, used);
|
|
560
|
+
}
|
|
561
|
+
}
|
|
562
|
+
|
|
563
|
+
rebuild(): void {
|
|
564
|
+
const used = toI32(this.entryKeys.length);
|
|
565
|
+
const walking = this.walks > 0;
|
|
566
|
+
const slots = rebuiltSlots(this.slots, walking ? used : this.live, used);
|
|
567
|
+
if (!walking && this.live < used) {
|
|
568
|
+
compactEntries(this.entryKeys, this.entryHashes);
|
|
569
|
+
compactHashes(this.entryHashes);
|
|
570
|
+
}
|
|
571
|
+
this.slots = slots;
|
|
572
|
+
this.mask = toI32(slots.length) - 1;
|
|
573
|
+
refile(slots, this.entryHashes);
|
|
574
|
+
}
|
|
575
|
+
}
|
|
576
|
+
|
|
577
|
+
/**
|
|
578
|
+
* The key as an entry stores it: a float's -0 becomes +0, as ECMA-262's
|
|
579
|
+
* `Map.prototype.set` and `Set.prototype.add` store it, so a walk over the
|
|
580
|
+
* keys yields +0 whichever zero was inserted. Lowered in place, like `hashKey`
|
|
581
|
+
* and `sameKey`: `fadd` of +0 for a float, and nothing at all for every other
|
|
582
|
+
* key. It sits last in the file so that adding it moved no line a `-g` build
|
|
583
|
+
* of an earlier function records.
|
|
584
|
+
*/
|
|
585
|
+
const storedKey = <K>(key: K): K => key;
|
package/std/map.ts
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `nish/map` — two operations on the global `Map` that JavaScript's does not
|
|
3
|
+
* have, written so that a program using them still runs unmodified under Node
|
|
4
|
+
* (docs/wp32-map.md §9.2).
|
|
5
|
+
*
|
|
6
|
+
* The bodies below are the meaning, and they are what runs under Node:
|
|
7
|
+
* `reserve` does nothing, and `getOrInsert` is a `get`, and a `set` when the
|
|
8
|
+
* key was missing. Natively neither body is ever called. The compiler lowers
|
|
9
|
+
* every call in place, as it lowers `hashKey` and `sameKey`: `reserve` is the
|
|
10
|
+
* table's `reserveSlots`, which grows the buckets so that `n` entries fit
|
|
11
|
+
* without a rebuild, and `getOrInsert` is one `probe` of the key, then the
|
|
12
|
+
* value it found or an insert through the empty bucket the probe stopped at,
|
|
13
|
+
* with the hash it already computed. So this module writes no `.ll` of its
|
|
14
|
+
* own, and a one-file program that imports it is still one module.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Make room in `m` for `n` entries in all, so that inserting up to `n` never
|
|
19
|
+
* grows the table. Only the speed of a program depends on it: natively it
|
|
20
|
+
* presizes the bucket table, and under Node, which has no such call, it does
|
|
21
|
+
* nothing. A count that is not positive does nothing either way.
|
|
22
|
+
*/
|
|
23
|
+
export const reserve = <K, V>(m: Map<K, V>, n: number): void => {};
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* The value of `key` in `m`; or, when `key` is missing, `value`, after
|
|
27
|
+
* setting `key` to it. One hash and one probe natively, whichever it was.
|
|
28
|
+
* `value` is evaluated either way, as every argument is.
|
|
29
|
+
*/
|
|
30
|
+
export const getOrInsert = <K, V>(m: Map<K, V>, key: K, value: V): V => {
|
|
31
|
+
const found = m.get(key);
|
|
32
|
+
if (found !== undefined) {
|
|
33
|
+
return found;
|
|
34
|
+
}
|
|
35
|
+
m.set(key, value);
|
|
36
|
+
return value;
|
|
37
|
+
};
|
package/std/threads.ts
ADDED
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `std/threads` — data parallelism, and nothing else (docs/wp29-thread-surface.md §4.1).
|
|
3
|
+
*
|
|
4
|
+
* import { parallelMapInto, parallelReduce } from "nish/threads";
|
|
5
|
+
*
|
|
6
|
+
* parallelMapInto(src, dst, (x) => x * 3);
|
|
7
|
+
* const total = parallelReduce(src, (a, b) => a + b, 0);
|
|
8
|
+
*
|
|
9
|
+
* **What is written here is the meaning, not the implementation.** Each body
|
|
10
|
+
* below is the sequential program, and it is what runs under Node, what
|
|
11
|
+
* `npm run check` type-checks, and what the compiler checks the call against.
|
|
12
|
+
* The compiler then recognises the two exported templates by module and name
|
|
13
|
+
* and lowers an instance of either onto `nish_parallel_range`
|
|
14
|
+
* (runtime/runtime_parallel.c): it replaces the one call that walks the whole
|
|
15
|
+
* range — `mapRange` for a map, `reduceBlocks` for a reduce — with a region that
|
|
16
|
+
* hands each thread a contiguous piece of it, and emits the rest of the body as
|
|
17
|
+
* written. So the length check, its message and the order a reduce combines in
|
|
18
|
+
* are this file's, whichever way the program is compiled.
|
|
19
|
+
*
|
|
20
|
+
* What makes the region safe is checked at the call, not trusted
|
|
21
|
+
* (docs/LANGUAGE.md, "Data parallelism"): the function passed as `f` may write
|
|
22
|
+
* nothing its caller could observe, may allocate only temporaries it drops
|
|
23
|
+
* before it returns — they are given back after every element — `dst` may not
|
|
24
|
+
* be reachable from an element of `src`, and the result type is a number or a
|
|
25
|
+
* `boolean`. Importing this module compiles the program with `--threads`,
|
|
26
|
+
* because every worker needs an arena of its own.
|
|
27
|
+
*
|
|
28
|
+
* A reduce is deterministic. `src` is split into `min(64, ceil(n / BLOCK))`
|
|
29
|
+
* blocks, each block is folded from `identity`, and the block results are
|
|
30
|
+
* combined left to right — here, on one thread, and by the compiled program on
|
|
31
|
+
* any number of them — so an `f64` sum is the same bits on one core or sixty
|
|
32
|
+
* four. That needs `f` to be associative and `identity` to be its identity,
|
|
33
|
+
* which the checker enforces for an arrow whose body is one operator on its two
|
|
34
|
+
* parameters and cannot see through a named function.
|
|
35
|
+
*/
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Elements per block of a reduce: 2^20, about a millisecond of simple work
|
|
39
|
+
* (docs/wp20-threads.md §8e). It decides the blocking, and so the answer's
|
|
40
|
+
* bits: an array this short is one block, folded as the loop it would have
|
|
41
|
+
* been. It is independent of the map's grain (`mapGrain` in
|
|
42
|
+
* `self/parallel.ts`), which decides only how a map is divided and may
|
|
43
|
+
* change without changing any result; this one may not.
|
|
44
|
+
*/
|
|
45
|
+
const BLOCK: i32 = 1048576;
|
|
46
|
+
|
|
47
|
+
/** The most blocks a reduce is split into: the partitioner's own ceiling on threads. */
|
|
48
|
+
const MAX_BLOCKS: i32 = 64;
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Writes `f(src[i])` into `dst[i]` for every `i` in `[lo, hi)`: one thread's
|
|
52
|
+
* share of a map. Each length is read in the loop condition and `dst`'s again
|
|
53
|
+
* after the call, because a call ends every length fact the bounds proof holds
|
|
54
|
+
* (docs/LANGUAGE.md, "Arrays") and this is what keeps both accesses unchecked.
|
|
55
|
+
* The caller has already checked that the range is inside both arrays, so
|
|
56
|
+
* neither test is ever the one that stops the loop.
|
|
57
|
+
*/
|
|
58
|
+
const mapRange = <T, U>(src: T[], dst: U[], f: (x: T) => U, lo: i32, hi: i32): void => {
|
|
59
|
+
for (let i: i32 = lo; i >= 0 && i < hi && i < toI32(src.length); i++) {
|
|
60
|
+
const y = f(src[i]);
|
|
61
|
+
if (i < toI32(dst.length)) {
|
|
62
|
+
dst[i] = y;
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
};
|
|
66
|
+
|
|
67
|
+
/** `f` folded over `src[lo..hi)` from `identity`: one block of a reduce. */
|
|
68
|
+
const reduceRange = <T>(src: T[], f: (acc: T, x: T) => T, identity: T, lo: i32, hi: i32): T => {
|
|
69
|
+
let acc = identity;
|
|
70
|
+
for (let i: i32 = lo; i >= 0 && i < hi && i < toI32(src.length); i++) {
|
|
71
|
+
acc = f(acc, src[i]);
|
|
72
|
+
}
|
|
73
|
+
return acc;
|
|
74
|
+
};
|
|
75
|
+
|
|
76
|
+
/** How many blocks a reduce over `n` elements is split into: `min(MAX_BLOCKS, ceil(n / BLOCK))`. */
|
|
77
|
+
const reduceBlockCount = (n: i32): i32 => {
|
|
78
|
+
const wanted: i32 = toI32((toI64(n) + toI64(BLOCK) - 1) / toI64(BLOCK));
|
|
79
|
+
return wanted < MAX_BLOCKS ? wanted : MAX_BLOCKS;
|
|
80
|
+
};
|
|
81
|
+
|
|
82
|
+
/** Where block `k` of `blocks` over `n` elements starts; block `blocks` starts at `n`. */
|
|
83
|
+
const reduceBlockStart = (n: i32, blocks: i32, k: i32): i32 => toI32((toI64(n) * toI64(k)) / toI64(blocks));
|
|
84
|
+
|
|
85
|
+
/** Folds blocks `[lo, hi)` of `src` into `partials`, one result per block: one thread's share of a reduce. */
|
|
86
|
+
const reduceBlocks = <T>(
|
|
87
|
+
src: T[],
|
|
88
|
+
f: (acc: T, x: T) => T,
|
|
89
|
+
identity: T,
|
|
90
|
+
partials: T[],
|
|
91
|
+
lo: i32,
|
|
92
|
+
hi: i32
|
|
93
|
+
): void => {
|
|
94
|
+
const n: i32 = toI32(src.length);
|
|
95
|
+
const blocks: i32 = toI32(partials.length);
|
|
96
|
+
for (let k: i32 = lo; k >= 0 && k < hi && k < blocks; k++) {
|
|
97
|
+
const partial = reduceRange(
|
|
98
|
+
src,
|
|
99
|
+
f,
|
|
100
|
+
identity,
|
|
101
|
+
reduceBlockStart(n, blocks, k),
|
|
102
|
+
reduceBlockStart(n, blocks, k + 1)
|
|
103
|
+
);
|
|
104
|
+
if (k < toI32(partials.length)) {
|
|
105
|
+
partials[k] = partial;
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
};
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* The panic of a map whose `dst` is shorter than its `src`. A function of its
|
|
112
|
+
* own so that the message it builds is its allocation and not the map's: an
|
|
113
|
+
* allocation anywhere in `parallelMapInto` would give every call an arena mark
|
|
114
|
+
* and release, which is most of what a map over a few elements costs.
|
|
115
|
+
*/
|
|
116
|
+
const dstTooShort = (have: i32, want: i32): void => {
|
|
117
|
+
panic(`parallelMapInto: dst has ${have} elements and src has ${want}`);
|
|
118
|
+
};
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* `dst[i] = f(src[i])` for every index of `src`, on as many threads as the
|
|
122
|
+
* machine has and the length is worth. `dst` must be at least as long as
|
|
123
|
+
* `src`, which is checked once, before any element is written.
|
|
124
|
+
*/
|
|
125
|
+
export const parallelMapInto = <T, U>(src: T[], dst: U[], f: (x: T) => U): void => {
|
|
126
|
+
const n: i32 = toI32(src.length);
|
|
127
|
+
if (toI32(dst.length) < n) {
|
|
128
|
+
dstTooShort(toI32(dst.length), n);
|
|
129
|
+
}
|
|
130
|
+
mapRange(src, dst, f, 0, n);
|
|
131
|
+
};
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* `f` folded over `src`, blockwise from `identity` and then left to right over
|
|
135
|
+
* the blocks, so the answer does not depend on how many threads computed it.
|
|
136
|
+
* An empty `src` answers `identity`.
|
|
137
|
+
*/
|
|
138
|
+
export const parallelReduce = <T>(src: T[], f: (acc: T, x: T) => T, identity: T): T => {
|
|
139
|
+
const blocks: i32 = reduceBlockCount(toI32(src.length));
|
|
140
|
+
if (blocks === 0) {
|
|
141
|
+
return identity;
|
|
142
|
+
}
|
|
143
|
+
const partials = new Array<T>(blocks);
|
|
144
|
+
reduceBlocks(src, f, identity, partials, 0, blocks);
|
|
145
|
+
let acc = partials[0];
|
|
146
|
+
for (let k: i32 = 1; k < toI32(partials.length); k++) {
|
|
147
|
+
acc = f(acc, partials[k]);
|
|
148
|
+
}
|
|
149
|
+
return acc;
|
|
150
|
+
};
|