@amritk/nish 0.10.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +29 -13
- package/docs/AI.md +259 -16
- package/docs/INSTALL.md +22 -5
- package/package.json +7 -6
- package/runtime/LICENSE-ryu +23 -0
- package/runtime/nish.d.ts +11 -0
- package/runtime/nish.h +9 -1
- package/runtime/nish.mjs +13 -0
- package/runtime/runtime.c +10 -1
- package/scripts/ci-profile.mjs +3 -2
- package/scripts/gen-pow5-tables.py +5 -0
- package/std/README.md +17 -10
- package/std/collections.ts +585 -0
- package/std/map.ts +37 -0
- package/std/threads.ts +150 -0
package/std/map.ts
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `nish/map` — two operations on the global `Map` that JavaScript's does not
|
|
3
|
+
* have, written so that a program using them still runs unmodified under Node
|
|
4
|
+
* (docs/wp32-map.md §9.2).
|
|
5
|
+
*
|
|
6
|
+
* The bodies below are the meaning, and they are what runs under Node:
|
|
7
|
+
* `reserve` does nothing, and `getOrInsert` is a `get`, and a `set` when the
|
|
8
|
+
* key was missing. Natively neither body is ever called. The compiler lowers
|
|
9
|
+
* every call in place, as it lowers `hashKey` and `sameKey`: `reserve` is the
|
|
10
|
+
* table's `reserveSlots`, which grows the buckets so that `n` entries fit
|
|
11
|
+
* without a rebuild, and `getOrInsert` is one `probe` of the key, then the
|
|
12
|
+
* value it found or an insert through the empty bucket the probe stopped at,
|
|
13
|
+
* with the hash it already computed. So this module writes no `.ll` of its
|
|
14
|
+
* own, and a one-file program that imports it is still one module.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Make room in `m` for `n` entries in all, so that inserting up to `n` never
|
|
19
|
+
* grows the table. Only the speed of a program depends on it: natively it
|
|
20
|
+
* presizes the bucket table, and under Node, which has no such call, it does
|
|
21
|
+
* nothing. A count that is not positive does nothing either way.
|
|
22
|
+
*/
|
|
23
|
+
export const reserve = <K, V>(m: Map<K, V>, n: number): void => {};
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* The value of `key` in `m`; or, when `key` is missing, `value`, after
|
|
27
|
+
* setting `key` to it. One hash and one probe natively, whichever it was.
|
|
28
|
+
* `value` is evaluated either way, as every argument is.
|
|
29
|
+
*/
|
|
30
|
+
export const getOrInsert = <K, V>(m: Map<K, V>, key: K, value: V): V => {
|
|
31
|
+
const found = m.get(key);
|
|
32
|
+
if (found !== undefined) {
|
|
33
|
+
return found;
|
|
34
|
+
}
|
|
35
|
+
m.set(key, value);
|
|
36
|
+
return value;
|
|
37
|
+
};
|
package/std/threads.ts
ADDED
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `std/threads` — data parallelism, and nothing else (docs/wp29-thread-surface.md §4.1).
|
|
3
|
+
*
|
|
4
|
+
* import { parallelMapInto, parallelReduce } from "nish/threads";
|
|
5
|
+
*
|
|
6
|
+
* parallelMapInto(src, dst, (x) => x * 3);
|
|
7
|
+
* const total = parallelReduce(src, (a, b) => a + b, 0);
|
|
8
|
+
*
|
|
9
|
+
* **What is written here is the meaning, not the implementation.** Each body
|
|
10
|
+
* below is the sequential program, and it is what runs under Node, what
|
|
11
|
+
* `npm run check` type-checks, and what the compiler checks the call against.
|
|
12
|
+
* The compiler then recognises the two exported templates by module and name
|
|
13
|
+
* and lowers an instance of either onto `nish_parallel_range`
|
|
14
|
+
* (runtime/runtime_parallel.c): it replaces the one call that walks the whole
|
|
15
|
+
* range — `mapRange` for a map, `reduceBlocks` for a reduce — with a region that
|
|
16
|
+
* hands each thread a contiguous piece of it, and emits the rest of the body as
|
|
17
|
+
* written. So the length check, its message and the order a reduce combines in
|
|
18
|
+
* are this file's, whichever way the program is compiled.
|
|
19
|
+
*
|
|
20
|
+
* What makes the region safe is checked at the call, not trusted
|
|
21
|
+
* (docs/LANGUAGE.md, "Data parallelism"): the function passed as `f` may write
|
|
22
|
+
* nothing its caller could observe, may allocate only temporaries it drops
|
|
23
|
+
* before it returns — they are given back after every element — `dst` may not
|
|
24
|
+
* be reachable from an element of `src`, and the result type is a number or a
|
|
25
|
+
* `boolean`. Importing this module compiles the program with `--threads`,
|
|
26
|
+
* because every worker needs an arena of its own.
|
|
27
|
+
*
|
|
28
|
+
* A reduce is deterministic. `src` is split into `min(64, ceil(n / BLOCK))`
|
|
29
|
+
* blocks, each block is folded from `identity`, and the block results are
|
|
30
|
+
* combined left to right — here, on one thread, and by the compiled program on
|
|
31
|
+
* any number of them — so an `f64` sum is the same bits on one core or sixty
|
|
32
|
+
* four. That needs `f` to be associative and `identity` to be its identity,
|
|
33
|
+
* which the checker enforces for an arrow whose body is one operator on its two
|
|
34
|
+
* parameters and cannot see through a named function.
|
|
35
|
+
*/
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Elements per block of a reduce: 2^20, about a millisecond of simple work
|
|
39
|
+
* (docs/wp20-threads.md §8e). It decides the blocking, and so the answer's
|
|
40
|
+
* bits: an array this short is one block, folded as the loop it would have
|
|
41
|
+
* been. It is independent of the map's grain (`mapGrain` in
|
|
42
|
+
* `self/parallel.ts`), which decides only how a map is divided and may
|
|
43
|
+
* change without changing any result; this one may not.
|
|
44
|
+
*/
|
|
45
|
+
const BLOCK: i32 = 1048576;
|
|
46
|
+
|
|
47
|
+
/** The most blocks a reduce is split into: the partitioner's own ceiling on threads. */
|
|
48
|
+
const MAX_BLOCKS: i32 = 64;
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Writes `f(src[i])` into `dst[i]` for every `i` in `[lo, hi)`: one thread's
|
|
52
|
+
* share of a map. Each length is read in the loop condition and `dst`'s again
|
|
53
|
+
* after the call, because a call ends every length fact the bounds proof holds
|
|
54
|
+
* (docs/LANGUAGE.md, "Arrays") and this is what keeps both accesses unchecked.
|
|
55
|
+
* The caller has already checked that the range is inside both arrays, so
|
|
56
|
+
* neither test is ever the one that stops the loop.
|
|
57
|
+
*/
|
|
58
|
+
const mapRange = <T, U>(src: T[], dst: U[], f: (x: T) => U, lo: i32, hi: i32): void => {
|
|
59
|
+
for (let i: i32 = lo; i >= 0 && i < hi && i < toI32(src.length); i++) {
|
|
60
|
+
const y = f(src[i]);
|
|
61
|
+
if (i < toI32(dst.length)) {
|
|
62
|
+
dst[i] = y;
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
};
|
|
66
|
+
|
|
67
|
+
/** `f` folded over `src[lo..hi)` from `identity`: one block of a reduce. */
|
|
68
|
+
const reduceRange = <T>(src: T[], f: (acc: T, x: T) => T, identity: T, lo: i32, hi: i32): T => {
|
|
69
|
+
let acc = identity;
|
|
70
|
+
for (let i: i32 = lo; i >= 0 && i < hi && i < toI32(src.length); i++) {
|
|
71
|
+
acc = f(acc, src[i]);
|
|
72
|
+
}
|
|
73
|
+
return acc;
|
|
74
|
+
};
|
|
75
|
+
|
|
76
|
+
/** How many blocks a reduce over `n` elements is split into: `min(MAX_BLOCKS, ceil(n / BLOCK))`. */
|
|
77
|
+
const reduceBlockCount = (n: i32): i32 => {
|
|
78
|
+
const wanted: i32 = toI32((toI64(n) + toI64(BLOCK) - 1) / toI64(BLOCK));
|
|
79
|
+
return wanted < MAX_BLOCKS ? wanted : MAX_BLOCKS;
|
|
80
|
+
};
|
|
81
|
+
|
|
82
|
+
/** Where block `k` of `blocks` over `n` elements starts; block `blocks` starts at `n`. */
|
|
83
|
+
const reduceBlockStart = (n: i32, blocks: i32, k: i32): i32 => toI32((toI64(n) * toI64(k)) / toI64(blocks));
|
|
84
|
+
|
|
85
|
+
/** Folds blocks `[lo, hi)` of `src` into `partials`, one result per block: one thread's share of a reduce. */
|
|
86
|
+
const reduceBlocks = <T>(
|
|
87
|
+
src: T[],
|
|
88
|
+
f: (acc: T, x: T) => T,
|
|
89
|
+
identity: T,
|
|
90
|
+
partials: T[],
|
|
91
|
+
lo: i32,
|
|
92
|
+
hi: i32
|
|
93
|
+
): void => {
|
|
94
|
+
const n: i32 = toI32(src.length);
|
|
95
|
+
const blocks: i32 = toI32(partials.length);
|
|
96
|
+
for (let k: i32 = lo; k >= 0 && k < hi && k < blocks; k++) {
|
|
97
|
+
const partial = reduceRange(
|
|
98
|
+
src,
|
|
99
|
+
f,
|
|
100
|
+
identity,
|
|
101
|
+
reduceBlockStart(n, blocks, k),
|
|
102
|
+
reduceBlockStart(n, blocks, k + 1)
|
|
103
|
+
);
|
|
104
|
+
if (k < toI32(partials.length)) {
|
|
105
|
+
partials[k] = partial;
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
};
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* The panic of a map whose `dst` is shorter than its `src`. A function of its
|
|
112
|
+
* own so that the message it builds is its allocation and not the map's: an
|
|
113
|
+
* allocation anywhere in `parallelMapInto` would give every call an arena mark
|
|
114
|
+
* and release, which is most of what a map over a few elements costs.
|
|
115
|
+
*/
|
|
116
|
+
const dstTooShort = (have: i32, want: i32): void => {
|
|
117
|
+
panic(`parallelMapInto: dst has ${have} elements and src has ${want}`);
|
|
118
|
+
};
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* `dst[i] = f(src[i])` for every index of `src`, on as many threads as the
|
|
122
|
+
* machine has and the length is worth. `dst` must be at least as long as
|
|
123
|
+
* `src`, which is checked once, before any element is written.
|
|
124
|
+
*/
|
|
125
|
+
export const parallelMapInto = <T, U>(src: T[], dst: U[], f: (x: T) => U): void => {
|
|
126
|
+
const n: i32 = toI32(src.length);
|
|
127
|
+
if (toI32(dst.length) < n) {
|
|
128
|
+
dstTooShort(toI32(dst.length), n);
|
|
129
|
+
}
|
|
130
|
+
mapRange(src, dst, f, 0, n);
|
|
131
|
+
};
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* `f` folded over `src`, blockwise from `identity` and then left to right over
|
|
135
|
+
* the blocks, so the answer does not depend on how many threads computed it.
|
|
136
|
+
* An empty `src` answers `identity`.
|
|
137
|
+
*/
|
|
138
|
+
export const parallelReduce = <T>(src: T[], f: (acc: T, x: T) => T, identity: T): T => {
|
|
139
|
+
const blocks: i32 = reduceBlockCount(toI32(src.length));
|
|
140
|
+
if (blocks === 0) {
|
|
141
|
+
return identity;
|
|
142
|
+
}
|
|
143
|
+
const partials = new Array<T>(blocks);
|
|
144
|
+
reduceBlocks(src, f, identity, partials, 0, blocks);
|
|
145
|
+
let acc = partials[0];
|
|
146
|
+
for (let k: i32 = 1; k < toI32(partials.length); k++) {
|
|
147
|
+
acc = f(acc, partials[k]);
|
|
148
|
+
}
|
|
149
|
+
return acc;
|
|
150
|
+
};
|