@amritk/nish 0.10.0 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/std/map.ts ADDED
@@ -0,0 +1,37 @@
1
+ /**
2
+ * `nish/map` — two operations on the global `Map` that JavaScript's does not
3
+ * have, written so that a program using them still runs unmodified under Node
4
+ * (docs/wp32-map.md §9.2).
5
+ *
6
+ * The bodies below are the meaning, and they are what runs under Node:
7
+ * `reserve` does nothing, and `getOrInsert` is a `get`, and a `set` when the
8
+ * key was missing. Natively neither body is ever called. The compiler lowers
9
+ * every call in place, as it lowers `hashKey` and `sameKey`: `reserve` is the
10
+ * table's `reserveSlots`, which grows the buckets so that `n` entries fit
11
+ * without a rebuild, and `getOrInsert` is one `probe` of the key, then the
12
+ * value it found or an insert through the empty bucket the probe stopped at,
13
+ * with the hash it already computed. So this module writes no `.ll` of its
14
+ * own, and a one-file program that imports it is still one module.
15
+ */
16
+
17
+ /**
18
+ * Make room in `m` for `n` entries in all, so that inserting up to `n` never
19
+ * grows the table. Only the speed of a program depends on it: natively it
20
+ * presizes the bucket table, and under Node, which has no such call, it does
21
+ * nothing. A count that is not positive does nothing either way.
22
+ */
23
+ export const reserve = <K, V>(m: Map<K, V>, n: number): void => {};
24
+
25
+ /**
26
+ * The value of `key` in `m`; or, when `key` is missing, `value`, after
27
+ * setting `key` to it. One hash and one probe natively, whichever it was.
28
+ * `value` is evaluated either way, as every argument is.
29
+ */
30
+ export const getOrInsert = <K, V>(m: Map<K, V>, key: K, value: V): V => {
31
+ const found = m.get(key);
32
+ if (found !== undefined) {
33
+ return found;
34
+ }
35
+ m.set(key, value);
36
+ return value;
37
+ };
package/std/threads.ts ADDED
@@ -0,0 +1,150 @@
1
+ /**
2
+ * `std/threads` — data parallelism, and nothing else (docs/wp29-thread-surface.md §4.1).
3
+ *
4
+ * import { parallelMapInto, parallelReduce } from "nish/threads";
5
+ *
6
+ * parallelMapInto(src, dst, (x) => x * 3);
7
+ * const total = parallelReduce(src, (a, b) => a + b, 0);
8
+ *
9
+ * **What is written here is the meaning, not the implementation.** Each body
10
+ * below is the sequential program, and it is what runs under Node, what
11
+ * `npm run check` type-checks, and what the compiler checks the call against.
12
+ * The compiler then recognises the two exported templates by module and name
13
+ * and lowers an instance of either onto `nish_parallel_range`
14
+ * (runtime/runtime_parallel.c): it replaces the one call that walks the whole
15
+ * range — `mapRange` for a map, `reduceBlocks` for a reduce — with a region that
16
+ * hands each thread a contiguous piece of it, and emits the rest of the body as
17
+ * written. So the length check, its message and the order a reduce combines in
18
+ * are this file's, whichever way the program is compiled.
19
+ *
20
+ * What makes the region safe is checked at the call, not trusted
21
+ * (docs/LANGUAGE.md, "Data parallelism"): the function passed as `f` may write
22
+ * nothing its caller could observe, may allocate only temporaries it drops
23
+ * before it returns — they are given back after every element — `dst` may not
24
+ * be reachable from an element of `src`, and the result type is a number or a
25
+ * `boolean`. Importing this module compiles the program with `--threads`,
26
+ * because every worker needs an arena of its own.
27
+ *
28
+ * A reduce is deterministic. `src` is split into `min(64, ceil(n / BLOCK))`
29
+ * blocks, each block is folded from `identity`, and the block results are
30
+ * combined left to right — here, on one thread, and by the compiled program on
31
+ * any number of them — so an `f64` sum is the same bits on one core or sixty
32
+ * four. That needs `f` to be associative and `identity` to be its identity,
33
+ * which the checker enforces for an arrow whose body is one operator on its two
34
+ * parameters and cannot see through a named function.
35
+ */
36
+
37
+ /**
38
+ * Elements per block of a reduce: 2^20, about a millisecond of simple work
39
+ * (docs/wp20-threads.md §8e). It decides the blocking, and so the answer's
40
+ * bits: an array this short is one block, folded as the loop it would have
41
+ * been. It is independent of the map's grain (`mapGrain` in
42
+ * `self/parallel.ts`), which decides only how a map is divided and may
43
+ * change without changing any result; this one may not.
44
+ */
45
+ const BLOCK: i32 = 1048576;
46
+
47
+ /** The most blocks a reduce is split into: the partitioner's own ceiling on threads. */
48
+ const MAX_BLOCKS: i32 = 64;
49
+
50
+ /**
51
+ * Writes `f(src[i])` into `dst[i]` for every `i` in `[lo, hi)`: one thread's
52
+ * share of a map. Each length is read in the loop condition and `dst`'s again
53
+ * after the call, because a call ends every length fact the bounds proof holds
54
+ * (docs/LANGUAGE.md, "Arrays") and this is what keeps both accesses unchecked.
55
+ * The caller has already checked that the range is inside both arrays, so
56
+ * neither test is ever the one that stops the loop.
57
+ */
58
+ const mapRange = <T, U>(src: T[], dst: U[], f: (x: T) => U, lo: i32, hi: i32): void => {
59
+ for (let i: i32 = lo; i >= 0 && i < hi && i < toI32(src.length); i++) {
60
+ const y = f(src[i]);
61
+ if (i < toI32(dst.length)) {
62
+ dst[i] = y;
63
+ }
64
+ }
65
+ };
66
+
67
+ /** `f` folded over `src[lo..hi)` from `identity`: one block of a reduce. */
68
+ const reduceRange = <T>(src: T[], f: (acc: T, x: T) => T, identity: T, lo: i32, hi: i32): T => {
69
+ let acc = identity;
70
+ for (let i: i32 = lo; i >= 0 && i < hi && i < toI32(src.length); i++) {
71
+ acc = f(acc, src[i]);
72
+ }
73
+ return acc;
74
+ };
75
+
76
+ /** How many blocks a reduce over `n` elements is split into: `min(MAX_BLOCKS, ceil(n / BLOCK))`. */
77
+ const reduceBlockCount = (n: i32): i32 => {
78
+ const wanted: i32 = toI32((toI64(n) + toI64(BLOCK) - 1) / toI64(BLOCK));
79
+ return wanted < MAX_BLOCKS ? wanted : MAX_BLOCKS;
80
+ };
81
+
82
+ /** Where block `k` of `blocks` over `n` elements starts; block `blocks` starts at `n`. */
83
+ const reduceBlockStart = (n: i32, blocks: i32, k: i32): i32 => toI32((toI64(n) * toI64(k)) / toI64(blocks));
84
+
85
+ /** Folds blocks `[lo, hi)` of `src` into `partials`, one result per block: one thread's share of a reduce. */
86
+ const reduceBlocks = <T>(
87
+ src: T[],
88
+ f: (acc: T, x: T) => T,
89
+ identity: T,
90
+ partials: T[],
91
+ lo: i32,
92
+ hi: i32
93
+ ): void => {
94
+ const n: i32 = toI32(src.length);
95
+ const blocks: i32 = toI32(partials.length);
96
+ for (let k: i32 = lo; k >= 0 && k < hi && k < blocks; k++) {
97
+ const partial = reduceRange(
98
+ src,
99
+ f,
100
+ identity,
101
+ reduceBlockStart(n, blocks, k),
102
+ reduceBlockStart(n, blocks, k + 1)
103
+ );
104
+ if (k < toI32(partials.length)) {
105
+ partials[k] = partial;
106
+ }
107
+ }
108
+ };
109
+
110
+ /**
111
+ * The panic of a map whose `dst` is shorter than its `src`. A function of its
112
+ * own so that the message it builds is its allocation and not the map's: an
113
+ * allocation anywhere in `parallelMapInto` would give every call an arena mark
114
+ * and release, which is most of what a map over a few elements costs.
115
+ */
116
+ const dstTooShort = (have: i32, want: i32): void => {
117
+ panic(`parallelMapInto: dst has ${have} elements and src has ${want}`);
118
+ };
119
+
120
+ /**
121
+ * `dst[i] = f(src[i])` for every index of `src`, on as many threads as the
122
+ * machine has and the length is worth. `dst` must be at least as long as
123
+ * `src`, which is checked once, before any element is written.
124
+ */
125
+ export const parallelMapInto = <T, U>(src: T[], dst: U[], f: (x: T) => U): void => {
126
+ const n: i32 = toI32(src.length);
127
+ if (toI32(dst.length) < n) {
128
+ dstTooShort(toI32(dst.length), n);
129
+ }
130
+ mapRange(src, dst, f, 0, n);
131
+ };
132
+
133
+ /**
134
+ * `f` folded over `src`, blockwise from `identity` and then left to right over
135
+ * the blocks, so the answer does not depend on how many threads computed it.
136
+ * An empty `src` answers `identity`.
137
+ */
138
+ export const parallelReduce = <T>(src: T[], f: (acc: T, x: T) => T, identity: T): T => {
139
+ const blocks: i32 = reduceBlockCount(toI32(src.length));
140
+ if (blocks === 0) {
141
+ return identity;
142
+ }
143
+ const partials = new Array<T>(blocks);
144
+ reduceBlocks(src, f, identity, partials, 0, blocks);
145
+ let acc = partials[0];
146
+ for (let k: i32 = 1; k < toI32(partials.length); k++) {
147
+ acc = f(acc, partials[k]);
148
+ }
149
+ return acc;
150
+ };