@graphty/webgpu-graph-algorithms 0.6.26 → 0.6.28
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +56 -6
- package/dist/acquire.d.ts +2 -0
- package/dist/browser.js +18 -1
- package/dist/browser.js.map +1 -1
- package/dist/chunks/accelerator-B-FjQwaA.js +19173 -0
- package/dist/chunks/accelerator-B-FjQwaA.js.map +1 -0
- package/dist/chunks/managed-D_GdQtnu.js +98 -0
- package/dist/chunks/managed-D_GdQtnu.js.map +1 -0
- package/dist/node.js +18 -1
- package/dist/node.js.map +1 -1
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +5 -3
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/all-pairs.d.ts.map +1 -1
- package/dist/src/algorithms/all-pairs.js +72 -47
- package/dist/src/algorithms/all-pairs.js.map +1 -1
- package/dist/src/algorithms/betweenness.d.ts +1 -1
- package/dist/src/algorithms/betweenness.js +2 -2
- package/dist/src/algorithms/closeness.d.ts +45 -42
- package/dist/src/algorithms/closeness.d.ts.map +1 -1
- package/dist/src/algorithms/closeness.js +295 -226
- package/dist/src/algorithms/closeness.js.map +1 -1
- package/dist/src/browser/index.d.ts +10 -0
- package/dist/src/browser/index.d.ts.map +1 -1
- package/dist/src/browser/index.js +22 -0
- package/dist/src/browser/index.js.map +1 -1
- package/dist/src/constants.d.ts +13 -3
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +13 -3
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernels.d.ts +14 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +62 -28
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/force-simulation.d.ts.map +1 -1
- package/dist/src/layouts/force-simulation.js +0 -1
- package/dist/src/layouts/force-simulation.js.map +1 -1
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
- package/dist/src/layouts/forceatlas2.js +0 -1
- package/dist/src/layouts/forceatlas2.js.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.js +0 -1
- package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
- package/dist/src/layouts/repulsion-grid.d.ts +1 -3
- package/dist/src/layouts/repulsion-grid.d.ts.map +1 -1
- package/dist/src/layouts/repulsion-grid.js +1 -6
- package/dist/src/layouts/repulsion-grid.js.map +1 -1
- package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
- package/dist/src/layouts/spring-electrical.js +0 -1
- package/dist/src/layouts/spring-electrical.js.map +1 -1
- package/dist/src/managed.d.ts +11 -0
- package/dist/src/managed.d.ts.map +1 -0
- package/dist/src/managed.js +129 -0
- package/dist/src/managed.js.map +1 -0
- package/dist/src/node/index.d.ts +11 -0
- package/dist/src/node/index.d.ts.map +1 -1
- package/dist/src/node/index.js +21 -0
- package/dist/src/node/index.js.map +1 -1
- package/dist/src/primitives/grid-pyramid.d.ts +16 -15
- package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
- package/dist/src/primitives/grid-pyramid.js +20 -28
- package/dist/src/primitives/grid-pyramid.js.map +1 -1
- package/dist/src/types/accelerator.d.ts +2 -0
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/managed.d.ts +81 -0
- package/dist/src/types/managed.d.ts.map +1 -0
- package/dist/src/types/managed.js +7 -0
- package/dist/src/types/managed.js.map +1 -0
- package/dist/src/wgsl/bc-forward.wgsl.d.ts +1 -1
- package/dist/src/wgsl/bc-forward.wgsl.js +1 -1
- package/dist/src/wgsl/closeness-level.wgsl.d.ts +37 -0
- package/dist/src/wgsl/closeness-level.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-level.wgsl.js +204 -0
- package/dist/src/wgsl/closeness-level.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts +11 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.js +42 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.js.map +1 -0
- package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +2 -2
- package/dist/src/wgsl/grid-centroid-hub.wgsl.js +2 -2
- package/dist/webgpu-graph-algorithms.js +142 -15586
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +10 -4
- package/src/accelerator.ts +5 -3
- package/src/algorithms/all-pairs.ts +86 -56
- package/src/algorithms/betweenness.ts +2 -2
- package/src/algorithms/closeness.ts +353 -256
- package/src/browser/index.ts +37 -0
- package/src/constants.ts +13 -3
- package/src/kernels.ts +65 -36
- package/src/layouts/force-simulation.ts +0 -1
- package/src/layouts/forceatlas2.ts +0 -1
- package/src/layouts/fruchterman-reingold.ts +0 -1
- package/src/layouts/repulsion-grid.ts +2 -7
- package/src/layouts/spring-electrical.ts +0 -1
- package/src/managed.ts +172 -0
- package/src/node/index.ts +36 -0
- package/src/primitives/grid-pyramid.ts +29 -41
- package/src/types/accelerator.ts +2 -0
- package/src/types/managed.ts +86 -0
- package/src/wgsl/bc-forward.wgsl.ts +1 -1
- package/src/wgsl/closeness-level.wgsl.ts +203 -0
- package/src/wgsl/closeness-rowsum.wgsl.ts +41 -0
- package/src/wgsl/grid-centroid-hub.wgsl.ts +2 -2
- package/dist/chunks/context-BZY6SMsM.js +0 -3615
- package/dist/chunks/context-BZY6SMsM.js.map +0 -1
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +0 -20
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +0 -1
- package/dist/src/wgsl/closeness-reduce.wgsl.js +0 -69
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +0 -1
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +0 -22
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +0 -1
- package/dist/src/wgsl/closeness-sweep.wgsl.js +0 -106
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +0 -1
- package/src/wgsl/closeness-reduce.wgsl.ts +0 -68
- package/src/wgsl/closeness-sweep.wgsl.ts +0 -105
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `closeness-level` kernel body: one level of the bit-parallel multi-source breadth-first search of closeness, in
|
|
3
|
+
* ONE dispatch, plus the two seed roles of a batch. A batch runs `32 x P.words` sources at once: bit `b` of word `j`
|
|
4
|
+
* of node `v` is source lane `32 j + b`. The `bits` buffer holds five regions of `P.base` words, `P.words` words per
|
|
5
|
+
* node: `visited` at 0, three frontier regions at `P.base`, `2 P.base` and `3 P.base` that rotate by the level
|
|
6
|
+
* (level `L` reads region `1 + L % 3`, writes region `1 + (L + 1) % 3` and zeroes region `1 + (L + 2) % 3`, the
|
|
7
|
+
* frontier of level `L - 1` that nothing reads any more, so no fill runs between levels; a stale bit could never
|
|
8
|
+
* claim anything, since its node's neighbours were claimed the level after, so the clear only keeps a later frontier
|
|
9
|
+
* from re-walking old nodes), then a sampled run's
|
|
10
|
+
* per-node distance sums at `4 P.base`. The `table` buffer holds the per-level claim counts of the submit (row `P.row`
|
|
11
|
+
* at `P.row x 32 P.words`, one word per source lane), then a ring of three control slots of four words at `P.ctrl`
|
|
12
|
+
* (`any` @0: the level claimed something; `arcs` @1: the out-degree of every (node, word) that joined the next
|
|
13
|
+
* frontier; `pull` @2: `arcs` passed `P.pullAt`), then a sampled run's source list at `P.sourcesAt`.
|
|
14
|
+
*
|
|
15
|
+
* Role 0 (one invocation per node and per arc, `max(n, arcs)` in all): a level whose predecessor claimed nothing
|
|
16
|
+
* returns at once (one uniform load per workgroup), so the host can record more levels than the batch needs.
|
|
17
|
+
* Otherwise the level chooses its step from the slot its predecessor filled, the same way for every invocation:
|
|
18
|
+
* - PUSH (the frontier is cheap to expand): one invocation per out-arc `(u, x)` -- `u` found by a binary search of
|
|
19
|
+
* `rowPtr` -- claims the sources at `u` that `x` has not seen with `atomicOr` on `x`'s visited word; the bits it
|
|
20
|
+
* won (`fresh`) join the next frontier;
|
|
21
|
+
* - PULL (the frontier's arcs pass `P.pullAt`, and `P.pullOk`): one invocation per node not yet reached by every
|
|
22
|
+
* source walks its IN-arcs, ORs the neighbours' frontier words, and keeps the bits it had not seen; it stops at
|
|
23
|
+
* the first in-arc after which every source has reached it (the early exit). Only the owner writes a node's
|
|
24
|
+
* words. The host allows the pull only when no node has more than a few thousand in-arcs, so no
|
|
25
|
+
* invocation of either step loops more than a few tens of thousands of times (llvmpipe silently ends every loop
|
|
26
|
+
* of an invocation past 65,535 iterations).
|
|
27
|
+
* Every claimed bit is tallied per source lane in workgroup memory and flushed with one global `atomicAdd` per lane
|
|
28
|
+
* per workgroup into the level's row; the host turns the counts into exact sums (`count x (L + 1)`) and harmonic
|
|
29
|
+
* sums (`count / (L + 1)`). Role 1 (one invocation per word): zeroes the regions, sets every dead lane of a partial
|
|
30
|
+
* batch as already visited (so the early exit and the "reached by every source" test see a full word), and zeroes the
|
|
31
|
+
* control ring. Role 2 (one invocation per lane): seeds lane `b`'s source -- `P.source + b`, or word `P.source + b` of
|
|
32
|
+
* the source list -- into `visited` and level 0's frontier with `atomicOr` (a node listed twice carries both bits),
|
|
33
|
+
* and fills the control slot level 0 reads. Body only; the text is normative: the sabotage rows of
|
|
34
|
+
* test/helpers/sabotage.ts are textual edits of it.
|
|
35
|
+
*/
|
|
36
|
+
export declare const closenessLevelWgsl = "\nconst max_words: u32 = 8u;\nvar<workgroup> tally: array<atomic<u32>, 32u * max_words>; // 32 x max_words: this workgroup's claims per source lane\nvar<workgroup> wlive: u32;\nvar<workgroup> wpull: u32;\nvar<workgroup> wany: atomic<u32>;\nvar<workgroup> warcs: atomic<u32>;\nvar<workgroup> wover: atomic<u32>;\n\nfn dead_lanes(j: u32) -> u32 { // the lanes of word j at or past the batch's source count\n let first = 32u * j;\n if (P.count >= first + 32u) { return 0u; }\n if (P.count <= first) { return U32_MAX; }\n return ~((1u << (P.count - first)) - 1u);\n}\n\nfn add_arcs(slot: u32, value: u32) { // the arcs total of a control slot; pull once it passes P.pullAt\n if (value == 0u) { return; }\n let before = atomicAdd(&table[slot + 1u], value);\n if (value > P.pullAt || before + value > P.pullAt || before + value < before) {\n atomicStore(&table[slot + 2u], 1u);\n }\n}\n\nfn record(j: u32, node: u32, fresh: u32, dist: u32) { // the claims of one word: tallied per lane, a sampled run's sums\n if (P.perNode == 1u) { atomicAdd(&bits[4u * P.base + node], countOneBits(fresh) * dist); }\n var b = fresh;\n loop {\n if (b == 0u) { break; }\n atomicAdd(&tally[32u * j + firstTrailingBit(b)], 1u);\n b = b & (b - 1u);\n }\n}\n\n@compute @workgroup_size(WG)\nfn closeness_level(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n let W = P.words;\n if (P.role == 1u) { // seed, part 1: one invocation per word\n if (i == 0u) {\n for (var k = 0u; k < 12u; k = k + 1u) { atomicStore(&table[P.ctrl + k], 0u); }\n }\n if (i < P.total) { atomicStore(&bits[i], select(0u, dead_lanes(i % W), i < P.base)); }\n return; // uniform: P.role is\n }\n if (P.role == 2u) { // seed, part 2: one invocation per source lane\n if (i < P.count) {\n var v = P.source + i;\n if (P.sourcesAt != 0u) { v = atomicLoad(&table[P.sourcesAt + P.source + i]); }\n let w = v * W + i / 32u;\n let bit = 1u << (i % 32u);\n atomicOr(&bits[w], bit); // visited\n let before = atomicOr(&bits[P.base + w], bit); // level 0's frontier (region 1)\n let slot = P.ctrl + 8u; // the slot of \"level -1\", which level 0 reads\n atomicStore(&table[slot], 1u);\n if (before == 0u) { add_arcs(slot, rowPtr[v + 1u] - rowPtr[v]); }\n }\n return;\n }\n\n // role 0: one level\n let L = P.level;\n let prev = P.ctrl + 4u * ((L + 2u) % 3u);\n let cur = P.ctrl + 4u * (L % 3u);\n if (lid.x == 0u) {\n wlive = atomicLoad(&table[prev]);\n wpull = select(0u, atomicLoad(&table[prev + 2u]), P.pullOk == 1u);\n atomicStore(&wany, 0u);\n atomicStore(&warcs, 0u);\n atomicStore(&wover, 0u);\n }\n for (var k = lid.x; k < 32u * W; k = k + WG) { atomicStore(&tally[k], 0u); }\n let live = workgroupUniformLoad(&wlive); // uniform; includes a barrier\n if (live == 0u) { return; } // the previous level claimed nothing\n let pull = workgroupUniformLoad(&wpull) == 1u;\n if (i == 0u) { // the slot of level L + 1 held level L - 2's\n let nxt = P.ctrl + 4u * ((L + 1u) % 3u);\n atomicStore(&table[nxt], 0u);\n atomicStore(&table[nxt + 1u], 0u);\n atomicStore(&table[nxt + 2u], 0u);\n }\n let frontierBase = P.base * (1u + L % 3u);\n let nextBase = P.base * (1u + (L + 1u) % 3u);\n let staleBase = P.base * (1u + (L + 2u) % 3u);\n let dist = L + 1u;\n var arcs = 0u;\n if (i < P.n) {\n for (var j = 0u; j < W; j = j + 1u) { atomicStore(&bits[staleBase + i * W + j], 0u); }\n }\n if (pull) {\n if (i < P.n) { // one invocation per node: its in-arcs\n var vis: array<u32, max_words>;\n var unseen = 0u;\n for (var j = 0u; j < W; j = j + 1u) {\n vis[j] = atomicLoad(&bits[i * W + j]);\n unseen = unseen | ~vis[j];\n }\n if (unseen != 0u) { // some source has not reached this node yet\n var acc: array<u32, max_words>;\n let end = inRowPtr[i + 1u];\n for (var a = inRowPtr[i]; a < end; a = a + 1u) {\n let u = inColIdx[a];\n var missing = 0u;\n for (var j = 0u; j < W; j = j + 1u) {\n acc[j] = acc[j] | atomicLoad(&bits[frontierBase + u * W + j]);\n missing = missing | ~(acc[j] | vis[j]);\n }\n if (missing == 0u) { break; } // every source has reached it: the early exit\n }\n let degree = rowPtr[i + 1u] - rowPtr[i];\n for (var j = 0u; j < W; j = j + 1u) {\n let fresh = acc[j] & ~vis[j];\n if (fresh != 0u) {\n atomicStore(&bits[i * W + j], vis[j] | fresh);\n atomicStore(&bits[nextBase + i * W + j], fresh);\n arcs = arcs + degree;\n record(j, i, fresh, dist);\n }\n }\n }\n }\n } else if (i < P.arcCount) { // one invocation per arc: no row is walked whole\n var lo = 0u; // the arc's source: the last row starting at or before it\n var hi = P.n - 1u;\n loop {\n if (lo >= hi) { break; }\n let mid = (lo + hi + 1u) / 2u;\n if (rowPtr[mid] <= i) { lo = mid; } else { hi = mid - 1u; }\n }\n let u = lo;\n let x = colIdx[i];\n for (var j = 0u; j < W; j = j + 1u) {\n let mask = atomicLoad(&bits[frontierBase + u * W + j]) & ~atomicLoad(&bits[x * W + j]); // at u, not yet at x\n if (mask == 0u) { continue; }\n let fresh = mask & ~atomicOr(&bits[x * W + j], mask); // the claims this invocation won\n if (fresh == 0u) { continue; }\n if (atomicOr(&bits[nextBase + x * W + j], fresh) == 0u) {\n arcs = arcs + (rowPtr[x + 1u] - rowPtr[x]);\n }\n record(j, x, fresh, dist);\n }\n }\n if (arcs > P.pullAt) {\n atomicStore(&wover, 1u);\n } else if (arcs != 0u) {\n let before = atomicAdd(&warcs, arcs);\n if (before + arcs > P.pullAt || before + arcs < before) { atomicStore(&wover, 1u); }\n }\n workgroupBarrier();\n let row = P.row * 32u * W;\n for (var k = lid.x; k < 32u * W; k = k + WG) { // ONE global atomic per source lane per workgroup\n let c = atomicLoad(&tally[k]);\n if (c != 0u) {\n atomicAdd(&table[row + k], c);\n atomicStore(&wany, 1u);\n }\n }\n workgroupBarrier();\n if (lid.x == 0u) {\n if (atomicLoad(&wany) != 0u) { atomicStore(&table[cur], 1u); }\n if (atomicLoad(&wover) != 0u) {\n atomicStore(&table[cur + 2u], 1u);\n } else {\n add_arcs(cur, atomicLoad(&warcs));\n }\n }\n}\n";
|
|
37
|
+
//# sourceMappingURL=closeness-level.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"closeness-level.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-level.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkCG;AACH,eAAO,MAAM,kBAAkB,ohPAuK9B,CAAC"}
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `closeness-level` kernel body: one level of the bit-parallel multi-source breadth-first search of closeness, in
|
|
3
|
+
* ONE dispatch, plus the two seed roles of a batch. A batch runs `32 x P.words` sources at once: bit `b` of word `j`
|
|
4
|
+
* of node `v` is source lane `32 j + b`. The `bits` buffer holds five regions of `P.base` words, `P.words` words per
|
|
5
|
+
* node: `visited` at 0, three frontier regions at `P.base`, `2 P.base` and `3 P.base` that rotate by the level
|
|
6
|
+
* (level `L` reads region `1 + L % 3`, writes region `1 + (L + 1) % 3` and zeroes region `1 + (L + 2) % 3`, the
|
|
7
|
+
* frontier of level `L - 1` that nothing reads any more, so no fill runs between levels; a stale bit could never
|
|
8
|
+
* claim anything, since its node's neighbours were claimed the level after, so the clear only keeps a later frontier
|
|
9
|
+
* from re-walking old nodes), then a sampled run's
|
|
10
|
+
* per-node distance sums at `4 P.base`. The `table` buffer holds the per-level claim counts of the submit (row `P.row`
|
|
11
|
+
* at `P.row x 32 P.words`, one word per source lane), then a ring of three control slots of four words at `P.ctrl`
|
|
12
|
+
* (`any` @0: the level claimed something; `arcs` @1: the out-degree of every (node, word) that joined the next
|
|
13
|
+
* frontier; `pull` @2: `arcs` passed `P.pullAt`), then a sampled run's source list at `P.sourcesAt`.
|
|
14
|
+
*
|
|
15
|
+
* Role 0 (one invocation per node and per arc, `max(n, arcs)` in all): a level whose predecessor claimed nothing
|
|
16
|
+
* returns at once (one uniform load per workgroup), so the host can record more levels than the batch needs.
|
|
17
|
+
* Otherwise the level chooses its step from the slot its predecessor filled, the same way for every invocation:
|
|
18
|
+
* - PUSH (the frontier is cheap to expand): one invocation per out-arc `(u, x)` -- `u` found by a binary search of
|
|
19
|
+
* `rowPtr` -- claims the sources at `u` that `x` has not seen with `atomicOr` on `x`'s visited word; the bits it
|
|
20
|
+
* won (`fresh`) join the next frontier;
|
|
21
|
+
* - PULL (the frontier's arcs pass `P.pullAt`, and `P.pullOk`): one invocation per node not yet reached by every
|
|
22
|
+
* source walks its IN-arcs, ORs the neighbours' frontier words, and keeps the bits it had not seen; it stops at
|
|
23
|
+
* the first in-arc after which every source has reached it (the early exit). Only the owner writes a node's
|
|
24
|
+
* words. The host allows the pull only when no node has more than a few thousand in-arcs, so no
|
|
25
|
+
* invocation of either step loops more than a few tens of thousands of times (llvmpipe silently ends every loop
|
|
26
|
+
* of an invocation past 65,535 iterations).
|
|
27
|
+
* Every claimed bit is tallied per source lane in workgroup memory and flushed with one global `atomicAdd` per lane
|
|
28
|
+
* per workgroup into the level's row; the host turns the counts into exact sums (`count x (L + 1)`) and harmonic
|
|
29
|
+
* sums (`count / (L + 1)`). Role 1 (one invocation per word): zeroes the regions, sets every dead lane of a partial
|
|
30
|
+
* batch as already visited (so the early exit and the "reached by every source" test see a full word), and zeroes the
|
|
31
|
+
* control ring. Role 2 (one invocation per lane): seeds lane `b`'s source -- `P.source + b`, or word `P.source + b` of
|
|
32
|
+
* the source list -- into `visited` and level 0's frontier with `atomicOr` (a node listed twice carries both bits),
|
|
33
|
+
* and fills the control slot level 0 reads. Body only; the text is normative: the sabotage rows of
|
|
34
|
+
* test/helpers/sabotage.ts are textual edits of it.
|
|
35
|
+
*/
|
|
36
|
+
export const closenessLevelWgsl = /* wgsl */ `
|
|
37
|
+
const max_words: u32 = 8u;
|
|
38
|
+
var<workgroup> tally: array<atomic<u32>, 32u * max_words>; // 32 x max_words: this workgroup's claims per source lane
|
|
39
|
+
var<workgroup> wlive: u32;
|
|
40
|
+
var<workgroup> wpull: u32;
|
|
41
|
+
var<workgroup> wany: atomic<u32>;
|
|
42
|
+
var<workgroup> warcs: atomic<u32>;
|
|
43
|
+
var<workgroup> wover: atomic<u32>;
|
|
44
|
+
|
|
45
|
+
fn dead_lanes(j: u32) -> u32 { // the lanes of word j at or past the batch's source count
|
|
46
|
+
let first = 32u * j;
|
|
47
|
+
if (P.count >= first + 32u) { return 0u; }
|
|
48
|
+
if (P.count <= first) { return U32_MAX; }
|
|
49
|
+
return ~((1u << (P.count - first)) - 1u);
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
fn add_arcs(slot: u32, value: u32) { // the arcs total of a control slot; pull once it passes P.pullAt
|
|
53
|
+
if (value == 0u) { return; }
|
|
54
|
+
let before = atomicAdd(&table[slot + 1u], value);
|
|
55
|
+
if (value > P.pullAt || before + value > P.pullAt || before + value < before) {
|
|
56
|
+
atomicStore(&table[slot + 2u], 1u);
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
fn record(j: u32, node: u32, fresh: u32, dist: u32) { // the claims of one word: tallied per lane, a sampled run's sums
|
|
61
|
+
if (P.perNode == 1u) { atomicAdd(&bits[4u * P.base + node], countOneBits(fresh) * dist); }
|
|
62
|
+
var b = fresh;
|
|
63
|
+
loop {
|
|
64
|
+
if (b == 0u) { break; }
|
|
65
|
+
atomicAdd(&tally[32u * j + firstTrailingBit(b)], 1u);
|
|
66
|
+
b = b & (b - 1u);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
@compute @workgroup_size(WG)
|
|
71
|
+
fn closeness_level(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
72
|
+
let i = linear_id(wid, lid.x);
|
|
73
|
+
let W = P.words;
|
|
74
|
+
if (P.role == 1u) { // seed, part 1: one invocation per word
|
|
75
|
+
if (i == 0u) {
|
|
76
|
+
for (var k = 0u; k < 12u; k = k + 1u) { atomicStore(&table[P.ctrl + k], 0u); }
|
|
77
|
+
}
|
|
78
|
+
if (i < P.total) { atomicStore(&bits[i], select(0u, dead_lanes(i % W), i < P.base)); }
|
|
79
|
+
return; // uniform: P.role is
|
|
80
|
+
}
|
|
81
|
+
if (P.role == 2u) { // seed, part 2: one invocation per source lane
|
|
82
|
+
if (i < P.count) {
|
|
83
|
+
var v = P.source + i;
|
|
84
|
+
if (P.sourcesAt != 0u) { v = atomicLoad(&table[P.sourcesAt + P.source + i]); }
|
|
85
|
+
let w = v * W + i / 32u;
|
|
86
|
+
let bit = 1u << (i % 32u);
|
|
87
|
+
atomicOr(&bits[w], bit); // visited
|
|
88
|
+
let before = atomicOr(&bits[P.base + w], bit); // level 0's frontier (region 1)
|
|
89
|
+
let slot = P.ctrl + 8u; // the slot of "level -1", which level 0 reads
|
|
90
|
+
atomicStore(&table[slot], 1u);
|
|
91
|
+
if (before == 0u) { add_arcs(slot, rowPtr[v + 1u] - rowPtr[v]); }
|
|
92
|
+
}
|
|
93
|
+
return;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
// role 0: one level
|
|
97
|
+
let L = P.level;
|
|
98
|
+
let prev = P.ctrl + 4u * ((L + 2u) % 3u);
|
|
99
|
+
let cur = P.ctrl + 4u * (L % 3u);
|
|
100
|
+
if (lid.x == 0u) {
|
|
101
|
+
wlive = atomicLoad(&table[prev]);
|
|
102
|
+
wpull = select(0u, atomicLoad(&table[prev + 2u]), P.pullOk == 1u);
|
|
103
|
+
atomicStore(&wany, 0u);
|
|
104
|
+
atomicStore(&warcs, 0u);
|
|
105
|
+
atomicStore(&wover, 0u);
|
|
106
|
+
}
|
|
107
|
+
for (var k = lid.x; k < 32u * W; k = k + WG) { atomicStore(&tally[k], 0u); }
|
|
108
|
+
let live = workgroupUniformLoad(&wlive); // uniform; includes a barrier
|
|
109
|
+
if (live == 0u) { return; } // the previous level claimed nothing
|
|
110
|
+
let pull = workgroupUniformLoad(&wpull) == 1u;
|
|
111
|
+
if (i == 0u) { // the slot of level L + 1 held level L - 2's
|
|
112
|
+
let nxt = P.ctrl + 4u * ((L + 1u) % 3u);
|
|
113
|
+
atomicStore(&table[nxt], 0u);
|
|
114
|
+
atomicStore(&table[nxt + 1u], 0u);
|
|
115
|
+
atomicStore(&table[nxt + 2u], 0u);
|
|
116
|
+
}
|
|
117
|
+
let frontierBase = P.base * (1u + L % 3u);
|
|
118
|
+
let nextBase = P.base * (1u + (L + 1u) % 3u);
|
|
119
|
+
let staleBase = P.base * (1u + (L + 2u) % 3u);
|
|
120
|
+
let dist = L + 1u;
|
|
121
|
+
var arcs = 0u;
|
|
122
|
+
if (i < P.n) {
|
|
123
|
+
for (var j = 0u; j < W; j = j + 1u) { atomicStore(&bits[staleBase + i * W + j], 0u); }
|
|
124
|
+
}
|
|
125
|
+
if (pull) {
|
|
126
|
+
if (i < P.n) { // one invocation per node: its in-arcs
|
|
127
|
+
var vis: array<u32, max_words>;
|
|
128
|
+
var unseen = 0u;
|
|
129
|
+
for (var j = 0u; j < W; j = j + 1u) {
|
|
130
|
+
vis[j] = atomicLoad(&bits[i * W + j]);
|
|
131
|
+
unseen = unseen | ~vis[j];
|
|
132
|
+
}
|
|
133
|
+
if (unseen != 0u) { // some source has not reached this node yet
|
|
134
|
+
var acc: array<u32, max_words>;
|
|
135
|
+
let end = inRowPtr[i + 1u];
|
|
136
|
+
for (var a = inRowPtr[i]; a < end; a = a + 1u) {
|
|
137
|
+
let u = inColIdx[a];
|
|
138
|
+
var missing = 0u;
|
|
139
|
+
for (var j = 0u; j < W; j = j + 1u) {
|
|
140
|
+
acc[j] = acc[j] | atomicLoad(&bits[frontierBase + u * W + j]);
|
|
141
|
+
missing = missing | ~(acc[j] | vis[j]);
|
|
142
|
+
}
|
|
143
|
+
if (missing == 0u) { break; } // every source has reached it: the early exit
|
|
144
|
+
}
|
|
145
|
+
let degree = rowPtr[i + 1u] - rowPtr[i];
|
|
146
|
+
for (var j = 0u; j < W; j = j + 1u) {
|
|
147
|
+
let fresh = acc[j] & ~vis[j];
|
|
148
|
+
if (fresh != 0u) {
|
|
149
|
+
atomicStore(&bits[i * W + j], vis[j] | fresh);
|
|
150
|
+
atomicStore(&bits[nextBase + i * W + j], fresh);
|
|
151
|
+
arcs = arcs + degree;
|
|
152
|
+
record(j, i, fresh, dist);
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
} else if (i < P.arcCount) { // one invocation per arc: no row is walked whole
|
|
158
|
+
var lo = 0u; // the arc's source: the last row starting at or before it
|
|
159
|
+
var hi = P.n - 1u;
|
|
160
|
+
loop {
|
|
161
|
+
if (lo >= hi) { break; }
|
|
162
|
+
let mid = (lo + hi + 1u) / 2u;
|
|
163
|
+
if (rowPtr[mid] <= i) { lo = mid; } else { hi = mid - 1u; }
|
|
164
|
+
}
|
|
165
|
+
let u = lo;
|
|
166
|
+
let x = colIdx[i];
|
|
167
|
+
for (var j = 0u; j < W; j = j + 1u) {
|
|
168
|
+
let mask = atomicLoad(&bits[frontierBase + u * W + j]) & ~atomicLoad(&bits[x * W + j]); // at u, not yet at x
|
|
169
|
+
if (mask == 0u) { continue; }
|
|
170
|
+
let fresh = mask & ~atomicOr(&bits[x * W + j], mask); // the claims this invocation won
|
|
171
|
+
if (fresh == 0u) { continue; }
|
|
172
|
+
if (atomicOr(&bits[nextBase + x * W + j], fresh) == 0u) {
|
|
173
|
+
arcs = arcs + (rowPtr[x + 1u] - rowPtr[x]);
|
|
174
|
+
}
|
|
175
|
+
record(j, x, fresh, dist);
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
if (arcs > P.pullAt) {
|
|
179
|
+
atomicStore(&wover, 1u);
|
|
180
|
+
} else if (arcs != 0u) {
|
|
181
|
+
let before = atomicAdd(&warcs, arcs);
|
|
182
|
+
if (before + arcs > P.pullAt || before + arcs < before) { atomicStore(&wover, 1u); }
|
|
183
|
+
}
|
|
184
|
+
workgroupBarrier();
|
|
185
|
+
let row = P.row * 32u * W;
|
|
186
|
+
for (var k = lid.x; k < 32u * W; k = k + WG) { // ONE global atomic per source lane per workgroup
|
|
187
|
+
let c = atomicLoad(&tally[k]);
|
|
188
|
+
if (c != 0u) {
|
|
189
|
+
atomicAdd(&table[row + k], c);
|
|
190
|
+
atomicStore(&wany, 1u);
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
workgroupBarrier();
|
|
194
|
+
if (lid.x == 0u) {
|
|
195
|
+
if (atomicLoad(&wany) != 0u) { atomicStore(&table[cur], 1u); }
|
|
196
|
+
if (atomicLoad(&wover) != 0u) {
|
|
197
|
+
atomicStore(&table[cur + 2u], 1u);
|
|
198
|
+
} else {
|
|
199
|
+
add_arcs(cur, atomicLoad(&warcs));
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
`;
|
|
204
|
+
//# sourceMappingURL=closeness-level.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"closeness-level.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/closeness-level.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkCG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAuK5C,CAAC"}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `closeness-rowsum` kernel body: closeness from a finished all-pairs distance matrix, one workgroup per row.
|
|
3
|
+
* Every lane walks its strided columns of row `r` (the distances FROM `r`), skips the diagonal and every unreachable
|
|
4
|
+
* entry (`+Infinity`, compared by its bit pattern because WGSL lets a compiler assume no infinities), and adds, by
|
|
5
|
+
* `P.role`: 0 the hop count as an integer (exact: a row of at most 23,170 hops below 23,170 sums below 2^32), 1 the
|
|
6
|
+
* f32 distance, 2 its reciprocal (harmonic closeness; a zero distance adds nothing, as in the CPU port). A tree reduction in workgroup memory folds the lanes and lane
|
|
7
|
+
* 0 writes `out[r]` -- the integer, or the f32 bit pattern. The host turns the row into the score. Body only; the text
|
|
8
|
+
* is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
9
|
+
*/
|
|
10
|
+
export declare const closenessRowsumWgsl = "\nvar<workgroup> partial: array<u32, WG>;\n\n@compute @workgroup_size(WG)\nfn closeness_rowsum(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let row = group_id(wid); // uniform: one workgroup per row\n if (row >= P.n) { return; }\n var whole = 0u;\n var real = 0.0;\n for (var c = lid.x; c < P.n; c = c + WG) {\n let d = dist[row * P.n + c];\n if (c == row || bitcast<u32>(d) == F32_INF_BITS) { continue; }\n if (P.role == 0u) {\n whole = whole + u32(d);\n } else if (P.role == 1u) {\n real = real + d;\n } else if (d > 0.0) {\n real = real + 1.0 / d; // a zero distance adds nothing, as on the CPU\n }\n }\n partial[lid.x] = select(whole, bitcast<u32>(real), P.role != 0u);\n for (var s = WG / 2u; s > 0u; s = s / 2u) {\n workgroupBarrier();\n if (lid.x < s) {\n let a = partial[lid.x];\n let b = partial[lid.x + s];\n partial[lid.x] = select(a + b, bitcast<u32>(bitcast<f32>(a) + bitcast<f32>(b)), P.role != 0u);\n }\n }\n if (lid.x == 0u) { out[row] = partial[0]; }\n}\n";
|
|
11
|
+
//# sourceMappingURL=closeness-rowsum.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"closeness-rowsum.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-rowsum.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AACH,eAAO,MAAM,mBAAmB,uuCA+B/B,CAAC"}
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `closeness-rowsum` kernel body: closeness from a finished all-pairs distance matrix, one workgroup per row.
|
|
3
|
+
* Every lane walks its strided columns of row `r` (the distances FROM `r`), skips the diagonal and every unreachable
|
|
4
|
+
* entry (`+Infinity`, compared by its bit pattern because WGSL lets a compiler assume no infinities), and adds, by
|
|
5
|
+
* `P.role`: 0 the hop count as an integer (exact: a row of at most 23,170 hops below 23,170 sums below 2^32), 1 the
|
|
6
|
+
* f32 distance, 2 its reciprocal (harmonic closeness; a zero distance adds nothing, as in the CPU port). A tree reduction in workgroup memory folds the lanes and lane
|
|
7
|
+
* 0 writes `out[r]` -- the integer, or the f32 bit pattern. The host turns the row into the score. Body only; the text
|
|
8
|
+
* is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
9
|
+
*/
|
|
10
|
+
export const closenessRowsumWgsl = /* wgsl */ `
|
|
11
|
+
var<workgroup> partial: array<u32, WG>;
|
|
12
|
+
|
|
13
|
+
@compute @workgroup_size(WG)
|
|
14
|
+
fn closeness_rowsum(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
15
|
+
let row = group_id(wid); // uniform: one workgroup per row
|
|
16
|
+
if (row >= P.n) { return; }
|
|
17
|
+
var whole = 0u;
|
|
18
|
+
var real = 0.0;
|
|
19
|
+
for (var c = lid.x; c < P.n; c = c + WG) {
|
|
20
|
+
let d = dist[row * P.n + c];
|
|
21
|
+
if (c == row || bitcast<u32>(d) == F32_INF_BITS) { continue; }
|
|
22
|
+
if (P.role == 0u) {
|
|
23
|
+
whole = whole + u32(d);
|
|
24
|
+
} else if (P.role == 1u) {
|
|
25
|
+
real = real + d;
|
|
26
|
+
} else if (d > 0.0) {
|
|
27
|
+
real = real + 1.0 / d; // a zero distance adds nothing, as on the CPU
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
partial[lid.x] = select(whole, bitcast<u32>(real), P.role != 0u);
|
|
31
|
+
for (var s = WG / 2u; s > 0u; s = s / 2u) {
|
|
32
|
+
workgroupBarrier();
|
|
33
|
+
if (lid.x < s) {
|
|
34
|
+
let a = partial[lid.x];
|
|
35
|
+
let b = partial[lid.x + s];
|
|
36
|
+
partial[lid.x] = select(a + b, bitcast<u32>(bitcast<f32>(a) + bitcast<f32>(b)), P.role != 0u);
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
if (lid.x == 0u) { out[row] = partial[0]; }
|
|
40
|
+
}
|
|
41
|
+
`;
|
|
42
|
+
//# sourceMappingURL=closeness-rowsum.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"closeness-rowsum.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/closeness-rowsum.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AACH,MAAM,CAAC,MAAM,mBAAmB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA+B7C,CAAC"}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per
|
|
3
|
-
*
|
|
2
|
+
* G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per word of hubList, dispatched
|
|
3
|
+
* directly (issue #732), so the workgroups past hubCounters[0] idle; a WG-strided mass-weighted sum reduced by the
|
|
4
4
|
* prelude's tree. The work is guarded by `valid`, never an early return, so the reduction is uniform (PD-13). Body
|
|
5
5
|
* only; normative text.
|
|
6
6
|
*/
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per
|
|
3
|
-
*
|
|
2
|
+
* G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per word of hubList, dispatched
|
|
3
|
+
* directly (issue #732), so the workgroups past hubCounters[0] idle; a WG-strided mass-weighted sum reduced by the
|
|
4
4
|
* prelude's tree. The work is guarded by `valid`, never an early return, so the reduction is uniform (PD-13). Body
|
|
5
5
|
* only; normative text.
|
|
6
6
|
*/
|