@amritk/nish-aarch64-linux 0.12.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/std/threads.ts CHANGED
@@ -1,17 +1,23 @@
1
1
  /**
2
- * `std/threads` — data parallelism, and nothing else (docs/wp29-thread-surface.md §4.1).
2
+ * `std/threads` — data parallelism, and a scope for heterogeneous tasks
3
+ * (docs/wp29-thread-surface.md §4.1 and §4.2).
3
4
  *
4
- * import { parallelMapInto, parallelReduce } from "nish/threads";
5
+ * import { parallelMapInto, parallelReduce, scope } from "nish/threads";
5
6
  *
6
7
  * parallelMapInto(src, dst, (x) => x * 3);
7
8
  * const total = parallelReduce(src, (a, b) => a + b, 0);
9
+ * {
10
+ * using s = scope();
11
+ * s.spawn(sumOf, xs, out, 0);
12
+ * s.spawn(maxOf, ys, out, 1);
13
+ * } // both tasks have run, and `out` holds both answers
8
14
  *
9
15
  * **What is written here is the meaning, not the implementation.** Each body
10
16
  * below is the sequential program, and it is what runs under Node, what
11
17
  * `npm run check` type-checks, and what the compiler checks the call against.
12
18
  * The compiler then recognises the two exported templates by module and name
13
19
  * and lowers an instance of either onto `nish_parallel_range`
14
- * (runtime/runtime_parallel.c): it replaces the one call that walks the whole
20
+ * (runtime/runtime-parallel.c): it replaces the one call that walks the whole
15
21
  * range — `mapRange` for a map, `reduceBlocks` for a reduce — with a region that
16
22
  * hands each thread a contiguous piece of it, and emits the rest of the body as
17
23
  * written. So the length check, its message and the order a reduce combines in
@@ -39,13 +45,13 @@
39
45
  * (docs/wp20-threads.md §8e). It decides the blocking, and so the answer's
40
46
  * bits: an array this short is one block, folded as the loop it would have
41
47
  * been. It is independent of the map's grain (`mapGrain` in
42
- * `self/parallel.ts`), which decides only how a map is divided and may
48
+ * `src/parallel.ts`), which decides only how a map is divided and may
43
49
  * change without changing any result; this one may not.
44
50
  */
45
- const BLOCK: i32 = 1048576;
51
+ const BLOCK: i32 = 1048576
46
52
 
47
53
  /** The most blocks a reduce is split into: the partitioner's own ceiling on threads. */
48
- const MAX_BLOCKS: i32 = 64;
54
+ const MAX_BLOCKS: i32 = 64
49
55
 
50
56
  /**
51
57
  * Writes `f(src[i])` into `dst[i]` for every `i` in `[lo, hi)`: one thread's
@@ -57,30 +63,47 @@ const MAX_BLOCKS: i32 = 64;
57
63
  */
58
64
  const mapRange = <T, U>(src: T[], dst: U[], f: (x: T) => U, lo: i32, hi: i32): void => {
59
65
  for (let i: i32 = lo; i >= 0 && i < hi && i < toI32(src.length); i++) {
60
- const y = f(src[i]);
66
+ const y = f(src[i])
61
67
  if (i < toI32(dst.length)) {
62
- dst[i] = y;
68
+ dst[i] = y
63
69
  }
64
70
  }
65
- };
71
+ }
66
72
 
67
73
  /** `f` folded over `src[lo..hi)` from `identity`: one block of a reduce. */
68
74
  const reduceRange = <T>(src: T[], f: (acc: T, x: T) => T, identity: T, lo: i32, hi: i32): T => {
69
- let acc = identity;
75
+ let acc = identity
70
76
  for (let i: i32 = lo; i >= 0 && i < hi && i < toI32(src.length); i++) {
71
- acc = f(acc, src[i]);
77
+ acc = f(acc, src[i])
72
78
  }
73
- return acc;
74
- };
79
+ return acc
80
+ }
75
81
 
76
- /** How many blocks a reduce over `n` elements is split into: `min(MAX_BLOCKS, ceil(n / BLOCK))`. */
82
+ /**
83
+ * How many blocks a reduce over `n` elements is split into: `min(MAX_BLOCKS, ceil(n / BLOCK))`.
84
+ *
85
+ * This and `reduceBlockStart` compute in `f64` rather than `i64`, because this
86
+ * file is also what runs under Node (docs/RUN_UNDER_NODE.md), where `toI64`
87
+ * answers a BigInt that throws the moment it meets the `1` of a `+ 1`. A double
88
+ * holds every value either function reaches exactly: `n` is below 2^31 and
89
+ * dividing by `BLOCK`, a power of two, loses nothing.
90
+ */
77
91
  const reduceBlockCount = (n: i32): i32 => {
78
- const wanted: i32 = toI32((toI64(n) + toI64(BLOCK) - 1) / toI64(BLOCK));
79
- return wanted < MAX_BLOCKS ? wanted : MAX_BLOCKS;
80
- };
92
+ const wanted: i32 = toI32(Math.ceil(toF64(n) / toF64(BLOCK)))
93
+ return wanted < MAX_BLOCKS ? wanted : MAX_BLOCKS
94
+ }
81
95
 
82
- /** Where block `k` of `blocks` over `n` elements starts; block `blocks` starts at `n`. */
83
- const reduceBlockStart = (n: i32, blocks: i32, k: i32): i32 => toI32((toI64(n) * toI64(k)) / toI64(blocks));
96
+ /**
97
+ * Where block `k` of `blocks` over `n` elements starts, `floor(n * k / blocks)`;
98
+ * block `blocks` starts at `n`. `n * k` is below 2^37, past `i32` but exact in
99
+ * a double, and the rounded quotient floors to the true one: when `n * k` is
100
+ * not a multiple of `blocks` the true quotient is at least `1 / blocks` below
101
+ * the next integer, and the rounding error is under 2^37 * 2^-53, far less than
102
+ * `1 / MAX_BLOCKS`. So these are the same blocks the `i64` form computed, which
103
+ * the determinism of a reduce depends on (`tests/link/par_reduce_blocks`).
104
+ */
105
+ const reduceBlockStart = (n: i32, blocks: i32, k: i32): i32 =>
106
+ toI32(Math.floor((toF64(n) * toF64(k)) / toF64(blocks)))
84
107
 
85
108
  /** Folds blocks `[lo, hi)` of `src` into `partials`, one result per block: one thread's share of a reduce. */
86
109
  const reduceBlocks = <T>(
@@ -91,8 +114,8 @@ const reduceBlocks = <T>(
91
114
  lo: i32,
92
115
  hi: i32
93
116
  ): void => {
94
- const n: i32 = toI32(src.length);
95
- const blocks: i32 = toI32(partials.length);
117
+ const n: i32 = toI32(src.length)
118
+ const blocks: i32 = toI32(partials.length)
96
119
  for (let k: i32 = lo; k >= 0 && k < hi && k < blocks; k++) {
97
120
  const partial = reduceRange(
98
121
  src,
@@ -100,12 +123,12 @@ const reduceBlocks = <T>(
100
123
  identity,
101
124
  reduceBlockStart(n, blocks, k),
102
125
  reduceBlockStart(n, blocks, k + 1)
103
- );
126
+ )
104
127
  if (k < toI32(partials.length)) {
105
- partials[k] = partial;
128
+ partials[k] = partial
106
129
  }
107
130
  }
108
- };
131
+ }
109
132
 
110
133
  /**
111
134
  * The panic of a map whose `dst` is shorter than its `src`. A function of its
@@ -114,8 +137,8 @@ const reduceBlocks = <T>(
114
137
  * and release, which is most of what a map over a few elements costs.
115
138
  */
116
139
  const dstTooShort = (have: i32, want: i32): void => {
117
- panic(`parallelMapInto: dst has ${have} elements and src has ${want}`);
118
- };
140
+ panic(`parallelMapInto: dst has ${have} elements and src has ${want}`)
141
+ }
119
142
 
120
143
  /**
121
144
  * `dst[i] = f(src[i])` for every index of `src`, on as many threads as the
@@ -123,12 +146,12 @@ const dstTooShort = (have: i32, want: i32): void => {
123
146
  * `src`, which is checked once, before any element is written.
124
147
  */
125
148
  export const parallelMapInto = <T, U>(src: T[], dst: U[], f: (x: T) => U): void => {
126
- const n: i32 = toI32(src.length);
149
+ const n: i32 = toI32(src.length)
127
150
  if (toI32(dst.length) < n) {
128
- dstTooShort(toI32(dst.length), n);
151
+ dstTooShort(toI32(dst.length), n)
129
152
  }
130
- mapRange(src, dst, f, 0, n);
131
- };
153
+ mapRange(src, dst, f, 0, n)
154
+ }
132
155
 
133
156
  /**
134
157
  * `f` folded over `src`, blockwise from `identity` and then left to right over
@@ -136,15 +159,80 @@ export const parallelMapInto = <T, U>(src: T[], dst: U[], f: (x: T) => U): void
136
159
  * An empty `src` answers `identity`.
137
160
  */
138
161
  export const parallelReduce = <T>(src: T[], f: (acc: T, x: T) => T, identity: T): T => {
139
- const blocks: i32 = reduceBlockCount(toI32(src.length));
162
+ const blocks: i32 = reduceBlockCount(toI32(src.length))
140
163
  if (blocks === 0) {
141
- return identity;
164
+ return identity
142
165
  }
143
- const partials = new Array<T>(blocks);
144
- reduceBlocks(src, f, identity, partials, 0, blocks);
145
- let acc = partials[0];
166
+ const partials = new Array<T>(blocks)
167
+ reduceBlocks(src, f, identity, partials, 0, blocks)
168
+ let acc = partials[0]
146
169
  for (let k: i32 = 1; k < toI32(partials.length); k++) {
147
- acc = f(acc, partials[k]);
170
+ acc = f(acc, partials[k])
171
+ }
172
+ return acc
173
+ }
174
+
175
+ /**
176
+ * The panic of a task whose destination slot is not in its array. A function of
177
+ * its own for the reason `dstTooShort` is one: the message is its allocation.
178
+ */
179
+ const slotOutOfRange = (at: i32, length: i32): void => {
180
+ panic(`spawn: destination index ${at} is out of range for an array of ${length} elements`)
181
+ }
182
+
183
+ /** `dst[at] = r`, checked: what the scope does with a task's answer when it joins. */
184
+ const storeResult = <R>(dst: R[], at: i32, r: R): void => {
185
+ if (at >= 0 && at < toI32(dst.length)) {
186
+ dst[at] = r
187
+ } else {
188
+ slotOutOfRange(at, toI32(dst.length))
189
+ }
190
+ }
191
+
192
+ /**
193
+ * One task: `entry(arg)`, stored into `dst[at]`. Natively the compiler splits
194
+ * this call in two (`src/emit-parallel.ts`): `entry(arg)` runs on a thread of its
195
+ * own when the scope closes, and the store runs on the thread that opened the
196
+ * scope, once every task of it has finished. Here, under Node, it is the one
197
+ * call it reads as, made at the spawn.
198
+ */
199
+ const runTask = <A, R>(entry: (arg: A) => R, arg: A, dst: R[], at: i32): void =>
200
+ storeResult(dst, at, entry(arg))
201
+
202
+ /**
203
+ * A scope of tasks, joined when the block that declares it ends
204
+ * (docs/wp29-thread-surface.md §4.2). It is only ever made by `scope()` in a
205
+ * `using` declaration, and only ever used as the receiver of `spawn`
206
+ * (docs/LANGUAGE.md, "Scoped tasks"), so no task can outlive it.
207
+ */
208
+ export class ThreadScope {
209
+ /**
210
+ * Nothing reads or writes it. It is here so that each scope is an object of
211
+ * its own, whose address is what the runtime files the scope's tasks under.
212
+ */
213
+ tag: i32 = 0
214
+
215
+ /**
216
+ * Run `entry(arg)` on a thread of its own and store its answer in `dst[at]`
217
+ * when the scope closes. `at` is checked here, before the task is taken, and
218
+ * again when its answer is stored.
219
+ */
220
+ spawn<A, R>(entry: (arg: A) => R, arg: A, dst: R[], at: i32): void {
221
+ if (at < 0 || at >= toI32(dst.length)) {
222
+ slotOutOfRange(at, toI32(dst.length))
223
+ }
224
+ runTask(entry, arg, dst, at)
148
225
  }
149
- return acc;
150
- };
226
+
227
+ /**
228
+ * What `using` calls when the block ends. Natively the compiler emits the
229
+ * join itself at every exit of the block, and under Node every task has
230
+ * already run by the time this is reached, so there is nothing left to do.
231
+ */
232
+ [Symbol.dispose](): void {
233
+ // Nothing left to join: see above.
234
+ }
235
+ }
236
+
237
+ /** A new scope, for a `using` declaration: `using s = scope()`. */
238
+ export const scope = (): ThreadScope => new ThreadScope()