@graphty/webgpu-graph-algorithms 0.6.26 → 0.6.28
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +56 -6
- package/dist/acquire.d.ts +2 -0
- package/dist/browser.js +18 -1
- package/dist/browser.js.map +1 -1
- package/dist/chunks/accelerator-B-FjQwaA.js +19173 -0
- package/dist/chunks/accelerator-B-FjQwaA.js.map +1 -0
- package/dist/chunks/managed-D_GdQtnu.js +98 -0
- package/dist/chunks/managed-D_GdQtnu.js.map +1 -0
- package/dist/node.js +18 -1
- package/dist/node.js.map +1 -1
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +5 -3
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/all-pairs.d.ts.map +1 -1
- package/dist/src/algorithms/all-pairs.js +72 -47
- package/dist/src/algorithms/all-pairs.js.map +1 -1
- package/dist/src/algorithms/betweenness.d.ts +1 -1
- package/dist/src/algorithms/betweenness.js +2 -2
- package/dist/src/algorithms/closeness.d.ts +45 -42
- package/dist/src/algorithms/closeness.d.ts.map +1 -1
- package/dist/src/algorithms/closeness.js +295 -226
- package/dist/src/algorithms/closeness.js.map +1 -1
- package/dist/src/browser/index.d.ts +10 -0
- package/dist/src/browser/index.d.ts.map +1 -1
- package/dist/src/browser/index.js +22 -0
- package/dist/src/browser/index.js.map +1 -1
- package/dist/src/constants.d.ts +13 -3
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +13 -3
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernels.d.ts +14 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +62 -28
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/force-simulation.d.ts.map +1 -1
- package/dist/src/layouts/force-simulation.js +0 -1
- package/dist/src/layouts/force-simulation.js.map +1 -1
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
- package/dist/src/layouts/forceatlas2.js +0 -1
- package/dist/src/layouts/forceatlas2.js.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.js +0 -1
- package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
- package/dist/src/layouts/repulsion-grid.d.ts +1 -3
- package/dist/src/layouts/repulsion-grid.d.ts.map +1 -1
- package/dist/src/layouts/repulsion-grid.js +1 -6
- package/dist/src/layouts/repulsion-grid.js.map +1 -1
- package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
- package/dist/src/layouts/spring-electrical.js +0 -1
- package/dist/src/layouts/spring-electrical.js.map +1 -1
- package/dist/src/managed.d.ts +11 -0
- package/dist/src/managed.d.ts.map +1 -0
- package/dist/src/managed.js +129 -0
- package/dist/src/managed.js.map +1 -0
- package/dist/src/node/index.d.ts +11 -0
- package/dist/src/node/index.d.ts.map +1 -1
- package/dist/src/node/index.js +21 -0
- package/dist/src/node/index.js.map +1 -1
- package/dist/src/primitives/grid-pyramid.d.ts +16 -15
- package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
- package/dist/src/primitives/grid-pyramid.js +20 -28
- package/dist/src/primitives/grid-pyramid.js.map +1 -1
- package/dist/src/types/accelerator.d.ts +2 -0
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/managed.d.ts +81 -0
- package/dist/src/types/managed.d.ts.map +1 -0
- package/dist/src/types/managed.js +7 -0
- package/dist/src/types/managed.js.map +1 -0
- package/dist/src/wgsl/bc-forward.wgsl.d.ts +1 -1
- package/dist/src/wgsl/bc-forward.wgsl.js +1 -1
- package/dist/src/wgsl/closeness-level.wgsl.d.ts +37 -0
- package/dist/src/wgsl/closeness-level.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-level.wgsl.js +204 -0
- package/dist/src/wgsl/closeness-level.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts +11 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.js +42 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.js.map +1 -0
- package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +2 -2
- package/dist/src/wgsl/grid-centroid-hub.wgsl.js +2 -2
- package/dist/webgpu-graph-algorithms.js +142 -15586
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +10 -4
- package/src/accelerator.ts +5 -3
- package/src/algorithms/all-pairs.ts +86 -56
- package/src/algorithms/betweenness.ts +2 -2
- package/src/algorithms/closeness.ts +353 -256
- package/src/browser/index.ts +37 -0
- package/src/constants.ts +13 -3
- package/src/kernels.ts +65 -36
- package/src/layouts/force-simulation.ts +0 -1
- package/src/layouts/forceatlas2.ts +0 -1
- package/src/layouts/fruchterman-reingold.ts +0 -1
- package/src/layouts/repulsion-grid.ts +2 -7
- package/src/layouts/spring-electrical.ts +0 -1
- package/src/managed.ts +172 -0
- package/src/node/index.ts +36 -0
- package/src/primitives/grid-pyramid.ts +29 -41
- package/src/types/accelerator.ts +2 -0
- package/src/types/managed.ts +86 -0
- package/src/wgsl/bc-forward.wgsl.ts +1 -1
- package/src/wgsl/closeness-level.wgsl.ts +203 -0
- package/src/wgsl/closeness-rowsum.wgsl.ts +41 -0
- package/src/wgsl/grid-centroid-hub.wgsl.ts +2 -2
- package/dist/chunks/context-BZY6SMsM.js +0 -3615
- package/dist/chunks/context-BZY6SMsM.js.map +0 -1
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +0 -20
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +0 -1
- package/dist/src/wgsl/closeness-reduce.wgsl.js +0 -69
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +0 -1
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +0 -22
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +0 -1
- package/dist/src/wgsl/closeness-sweep.wgsl.js +0 -106
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +0 -1
- package/src/wgsl/closeness-reduce.wgsl.ts +0 -68
- package/src/wgsl/closeness-sweep.wgsl.ts +0 -105
|
@@ -1,20 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* The `closeness-reduce` kernel body (design 8.4, 9.7 "integer distances before division"; P8-T11, the P8 plan's
|
|
3
|
-
* PD-13): the one-lane bookkeeping of the bit-parallel sweep, two roles by `P.role`. Role 0 is the level boundary,
|
|
4
|
-
* recorded BEFORE the level's `compact` and sweep: `done` is whether the previous level's compacted count is 0 (so the
|
|
5
|
-
* traversal ends one level after the last claim, one empty sweep and no wrong sum), then for every source `s` the
|
|
6
|
-
* claims the level just run made (`newCount[s]`) are folded into `reached[s]` and into the 64-bit `sum[s]` at the
|
|
7
|
-
* distance `level + 1` -- the product as a 16-bit split into a low and a high word, then the add with its carry --
|
|
8
|
-
* `newCount[s]` is zeroed and `level` advances. Role 1 is the seed of a batch: for the batch's `k = min(32, n -
|
|
9
|
-
* P.source)` sources, bit `s` into `visited[source_s]` and into the frontier region level 0 reads (region 1, since
|
|
10
|
-
* level 0's parity is 0), `flags[source_s] = 1` (level 0's `compact` turns the flags into the list; a seeded list
|
|
11
|
-
* would be overwritten by a compaction of all-zero flags), `counters[0] = k` (not done) and `level = U32_MAX` (so
|
|
12
|
-
* level 0's boundary accumulates nothing and brings the word to 0, and level 1's counts the distance-1 claims at 1).
|
|
13
|
-
* Role 2 is role 1 for a sampled run: source `s` of the batch is word `P.source + s` of the source list the host
|
|
14
|
-
* wrote after the per-node sums (`perSource[128 + P.bitsBase + ...]`), `P.n` is the list's length, and a node listed
|
|
15
|
-
* twice in one batch carries both bits (the seed ORs, so a duplicate runs twice).
|
|
16
|
-
* No barrier follows the early return of the other lanes (3.5 rule 1). Body only (spec 3.5, D9); the text is
|
|
17
|
-
* normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
18
|
-
*/
|
|
19
|
-
export declare const closenessReduceWgsl = "\n@compute @workgroup_size(WG)\nfn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role != 0u) { // the seed of a batch: P.source is its first source\n let k = min(32u, P.n - P.source);\n for (var s = 0u; s < k; s = s + 1u) {\n var v = P.source + s;\n if (P.role == 2u) { v = atomicLoad(&perSource[128u + P.bitsBase + P.source + s]); } // a sampled run's list\n let bit = 1u << s;\n bits[v] = bits[v] | bit; // visited\n bits[P.bitsBase + v] = bits[P.bitsBase + v] | bit; // the frontier level 0 reads (region 1: level 0's parity is 0)\n bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list\n }\n atomicStore(&counters[0], k); // not done\n atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0\n atomicStore(&counters[15], 0u); // done\n return;\n }\n // role 0: the level boundary -- done from the previous level's compacted count, then the accumulation\n let count = atomicLoad(&counters[0]);\n atomicStore(&counters[15], select(0u, 1u, count == 0u));\n let level = atomicLoad(&counters[11]);\n let d = level + 1u; // the distance of the claims the level just run made\n for (var s = 0u; s < 32u; s = s + 1u) {\n let c = atomicLoad(&perSource[s]); // newCount[s]\n atomicStore(&perSource[32u + s], atomicLoad(&perSource[32u + s]) + c); // reached[s]\n // sum[s] += c x d in 64 bits: the 16-bit split product (pLo, pHi), then the add with its carry\n let cLo = c & 0xFFFFu;\n let cHi = c >> 16u;\n let dLo = d & 0xFFFFu;\n let dHi = d >> 16u;\n let ll = cLo * dLo;\n let lh = cLo * dHi;\n let hl = cHi * dLo;\n let mid = (ll >> 16u) + (lh & 0xFFFFu) + (hl & 0xFFFFu);\n let pLo = (ll & 0xFFFFu) | ((mid & 0xFFFFu) << 16u);\n let pHi = (cHi * dHi) + (lh >> 16u) + (hl >> 16u) + (mid >> 16u);\n var lo = atomicLoad(&perSource[64u + s]); // sumLo[s]\n var hi = atomicLoad(&perSource[96u + s]); // sumHi[s]\n let before = lo;\n lo = lo + pLo;\n hi = hi + pHi + select(0u, 1u, lo < before);\n atomicStore(&perSource[64u + s], lo);\n atomicStore(&perSource[96u + s], hi);\n atomicStore(&perSource[s], 0u);\n }\n atomicStore(&counters[11], level + 1u);\n}\n";
|
|
20
|
-
//# sourceMappingURL=closeness-reduce.wgsl.d.ts.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"closeness-reduce.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-reduce.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,eAAO,MAAM,mBAAmB,qyFAiD/B,CAAC"}
|
|
@@ -1,69 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* The `closeness-reduce` kernel body (design 8.4, 9.7 "integer distances before division"; P8-T11, the P8 plan's
|
|
3
|
-
* PD-13): the one-lane bookkeeping of the bit-parallel sweep, two roles by `P.role`. Role 0 is the level boundary,
|
|
4
|
-
* recorded BEFORE the level's `compact` and sweep: `done` is whether the previous level's compacted count is 0 (so the
|
|
5
|
-
* traversal ends one level after the last claim, one empty sweep and no wrong sum), then for every source `s` the
|
|
6
|
-
* claims the level just run made (`newCount[s]`) are folded into `reached[s]` and into the 64-bit `sum[s]` at the
|
|
7
|
-
* distance `level + 1` -- the product as a 16-bit split into a low and a high word, then the add with its carry --
|
|
8
|
-
* `newCount[s]` is zeroed and `level` advances. Role 1 is the seed of a batch: for the batch's `k = min(32, n -
|
|
9
|
-
* P.source)` sources, bit `s` into `visited[source_s]` and into the frontier region level 0 reads (region 1, since
|
|
10
|
-
* level 0's parity is 0), `flags[source_s] = 1` (level 0's `compact` turns the flags into the list; a seeded list
|
|
11
|
-
* would be overwritten by a compaction of all-zero flags), `counters[0] = k` (not done) and `level = U32_MAX` (so
|
|
12
|
-
* level 0's boundary accumulates nothing and brings the word to 0, and level 1's counts the distance-1 claims at 1).
|
|
13
|
-
* Role 2 is role 1 for a sampled run: source `s` of the batch is word `P.source + s` of the source list the host
|
|
14
|
-
* wrote after the per-node sums (`perSource[128 + P.bitsBase + ...]`), `P.n` is the list's length, and a node listed
|
|
15
|
-
* twice in one batch carries both bits (the seed ORs, so a duplicate runs twice).
|
|
16
|
-
* No barrier follows the early return of the other lanes (3.5 rule 1). Body only (spec 3.5, D9); the text is
|
|
17
|
-
* normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
18
|
-
*/
|
|
19
|
-
export const closenessReduceWgsl = /* wgsl */ `
|
|
20
|
-
@compute @workgroup_size(WG)
|
|
21
|
-
fn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
22
|
-
if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
|
|
23
|
-
if (P.role != 0u) { // the seed of a batch: P.source is its first source
|
|
24
|
-
let k = min(32u, P.n - P.source);
|
|
25
|
-
for (var s = 0u; s < k; s = s + 1u) {
|
|
26
|
-
var v = P.source + s;
|
|
27
|
-
if (P.role == 2u) { v = atomicLoad(&perSource[128u + P.bitsBase + P.source + s]); } // a sampled run's list
|
|
28
|
-
let bit = 1u << s;
|
|
29
|
-
bits[v] = bits[v] | bit; // visited
|
|
30
|
-
bits[P.bitsBase + v] = bits[P.bitsBase + v] | bit; // the frontier level 0 reads (region 1: level 0's parity is 0)
|
|
31
|
-
bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list
|
|
32
|
-
}
|
|
33
|
-
atomicStore(&counters[0], k); // not done
|
|
34
|
-
atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0
|
|
35
|
-
atomicStore(&counters[15], 0u); // done
|
|
36
|
-
return;
|
|
37
|
-
}
|
|
38
|
-
// role 0: the level boundary -- done from the previous level's compacted count, then the accumulation
|
|
39
|
-
let count = atomicLoad(&counters[0]);
|
|
40
|
-
atomicStore(&counters[15], select(0u, 1u, count == 0u));
|
|
41
|
-
let level = atomicLoad(&counters[11]);
|
|
42
|
-
let d = level + 1u; // the distance of the claims the level just run made
|
|
43
|
-
for (var s = 0u; s < 32u; s = s + 1u) {
|
|
44
|
-
let c = atomicLoad(&perSource[s]); // newCount[s]
|
|
45
|
-
atomicStore(&perSource[32u + s], atomicLoad(&perSource[32u + s]) + c); // reached[s]
|
|
46
|
-
// sum[s] += c x d in 64 bits: the 16-bit split product (pLo, pHi), then the add with its carry
|
|
47
|
-
let cLo = c & 0xFFFFu;
|
|
48
|
-
let cHi = c >> 16u;
|
|
49
|
-
let dLo = d & 0xFFFFu;
|
|
50
|
-
let dHi = d >> 16u;
|
|
51
|
-
let ll = cLo * dLo;
|
|
52
|
-
let lh = cLo * dHi;
|
|
53
|
-
let hl = cHi * dLo;
|
|
54
|
-
let mid = (ll >> 16u) + (lh & 0xFFFFu) + (hl & 0xFFFFu);
|
|
55
|
-
let pLo = (ll & 0xFFFFu) | ((mid & 0xFFFFu) << 16u);
|
|
56
|
-
let pHi = (cHi * dHi) + (lh >> 16u) + (hl >> 16u) + (mid >> 16u);
|
|
57
|
-
var lo = atomicLoad(&perSource[64u + s]); // sumLo[s]
|
|
58
|
-
var hi = atomicLoad(&perSource[96u + s]); // sumHi[s]
|
|
59
|
-
let before = lo;
|
|
60
|
-
lo = lo + pLo;
|
|
61
|
-
hi = hi + pHi + select(0u, 1u, lo < before);
|
|
62
|
-
atomicStore(&perSource[64u + s], lo);
|
|
63
|
-
atomicStore(&perSource[96u + s], hi);
|
|
64
|
-
atomicStore(&perSource[s], 0u);
|
|
65
|
-
}
|
|
66
|
-
atomicStore(&counters[11], level + 1u);
|
|
67
|
-
}
|
|
68
|
-
`;
|
|
69
|
-
//# sourceMappingURL=closeness-reduce.wgsl.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"closeness-reduce.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/closeness-reduce.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,CAAC,MAAM,mBAAmB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAiD7C,CAAC"}
|
|
@@ -1,22 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* The `closeness-sweep` kernel body (design 8.4 "32 sources per u32 word"; P8-T11, the P8 plan's PD-13 / DEP-P8-E):
|
|
3
|
-
* one level of the bit-parallel multi-source breadth-first search. The batch's state is one `bits` buffer of four
|
|
4
|
-
* regions of `P.bitsBase` words each -- `visited` at 0, the two frontier regions at `P.bitsBase` and `2 x P.bitsBase`
|
|
5
|
-
* (which one is the frontier and which the next swaps by the level's parity, `P.mode`, so nothing is copied between
|
|
6
|
-
* levels), `flags` at `3 x P.bitsBase` -- with bit `s` of word `v` meaning "source `s` has reached / is at / is next
|
|
7
|
-
* at `v`". The expansion is `advance-expand`'s block-mapped strip (P8-T5): each workgroup loads up to `WG` entries of
|
|
8
|
-
* the frontier LIST (the vertices any source is at, compacted from the flags by the host's `compact`), scans their
|
|
9
|
-
* degrees with the inlined Hillis-Steele scan of `bfs-contract` (this is not a twin kernel: `needs: []`), and every
|
|
10
|
-
* invocation strips the aggregate by binary search. The claim is inline: for the arc `(u, x)` the mask is the sources
|
|
11
|
-
* at `u` that have not reached `x`; `atomicOr` on `visited[x]` returns the bits this lane won (`fresh`), which go into
|
|
12
|
-
* the next region, set `flags[x]` (the flags region is part of the one atomic binding, so a plain store is a compile
|
|
13
|
-
* error), and are tallied per source in WORKGROUP memory -- one global `atomicAdd` per source per workgroup after the
|
|
14
|
-
* strip loop, never one per arc. Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan and the
|
|
15
|
-
* `workgroupUniformLoad` sit unconditionally after the guard, the strip loop is bounded by the uniform aggregate, and
|
|
16
|
-
* the flush's barrier follows it in uniform control flow. A sampled run (`P.perNode == 1`) also adds each claim's
|
|
17
|
-
* distance (`level + 1`, one per won bit) into the per-node sum of `x`, `perSource[128 + x]`: the distance from each
|
|
18
|
-
* of the batch's sources TO `x`, which is what a node's sampled closeness sums on an undirected graph. Body only (spec 3.5, D9); the text is normative: the
|
|
19
|
-
* sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
20
|
-
*/
|
|
21
|
-
export declare const closenessSweepWgsl = "\nvar<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan\nvar<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row\nvar<workgroup> rowOf: array<u32, WG>; // the frontier vertex of each entry (the source end of its arcs)\nvar<workgroup> local: array<atomic<u32>, 32>; // this workgroup's fresh claims per source\nvar<workgroup> wcount: u32; // the frontier list's length\nvar<workgroup> wdist: u32; // the distance of this level's claims\n\n@compute @workgroup_size(WG)\nfn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x == 0u) {\n wcount = atomicLoad(&counters[0]); // the frontier list's length (compact's total)\n wdist = atomicLoad(&counters[11]) + 1u; // the level word: this level claims at level + 1\n }\n let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers\n let dist = workgroupUniformLoad(&wdist);\n let nextBase = select(2u * P.bitsBase, P.bitsBase, P.mode == 1u); // the region that is next this level\n let frontierBase = 3u * P.bitsBase - nextBase; // the other one: the region that is the frontier\n for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries\n let i = b0 + lid.x; // this lane's frontier entry\n var deg = 0u;\n var start = 0u;\n var u = 0u;\n if (i < count) { // guarded loads into locals (3.5 rule 1)\n u = frontierList[i];\n let lo = max(rowPtr[u], P.arcBase);\n let hi = min(rowPtr[u + 1u], P.arcEnd);\n start = lo;\n deg = select(0u, hi - lo, hi > lo);\n }\n if (lid.x < 32u) { atomicStore(&local[lid.x], 0u); } // zeroed before the strip loop (WebGPU zero-initialises workgroup memory; said anyway)\n sh[lid.x] = deg;\n rowStart[lid.x] = start;\n rowOf[lid.x] = u;\n workgroupBarrier();\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees (bfs-contract's, inlined)\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier\n for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...\n var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p\n var hi = WG;\n loop {\n if (lo >= hi) { break; }\n let mid = (lo + hi) / 2u;\n if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }\n }\n let k = lo;\n var exclusive = 0u;\n if (k > 0u) { exclusive = sh[k - 1u]; }\n let arc = rowStart[k] + (p - exclusive);\n let x = colIdx[arc - P.arcBase];\n let mask = atomicLoad(&bits[frontierBase + rowOf[k]]) & ~atomicLoad(&bits[x]); // the sources at u that have not reached x\n if (mask != 0u) {\n let old = atomicOr(&bits[x], mask); // visited: the claim, one read-modify-write\n let fresh = mask & ~old; // the sources whose claim this lane won\n if (fresh != 0u) {\n atomicOr(&bits[nextBase + x], fresh);\n atomicStore(&bits[3u * P.bitsBase + x], 1u); // flags: x is in the next frontier list (compact reads it)\n if (P.perNode == 1u) { // a sampled run: x's distance to each source won\n atomicAdd(&perSource[128u + x], countOneBits(fresh) * dist);\n }\n var b = fresh;\n loop { // one tally per set bit of fresh\n if (b == 0u) { break; }\n let s = firstTrailingBit(b);\n atomicAdd(&local[s], 1u);\n b = b & (b - 1u);\n }\n }\n }\n }\n workgroupBarrier(); // uniform: the loop's bound is the uniform aggregate\n if (lid.x < 32u) { // ONE global atomic per source per workgroup\n let c = atomicLoad(&local[lid.x]);\n if (c != 0u) { atomicAdd(&perSource[lid.x], c); } // newCount[s]\n }\n workgroupBarrier(); // sh, rowStart, rowOf and local are reused by the next block\n }\n}\n";
|
|
22
|
-
//# sourceMappingURL=closeness-sweep.wgsl.d.ts.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"closeness-sweep.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-sweep.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AACH,eAAO,MAAM,kBAAkB,inKAoF9B,CAAC"}
|
|
@@ -1,106 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* The `closeness-sweep` kernel body (design 8.4 "32 sources per u32 word"; P8-T11, the P8 plan's PD-13 / DEP-P8-E):
|
|
3
|
-
* one level of the bit-parallel multi-source breadth-first search. The batch's state is one `bits` buffer of four
|
|
4
|
-
* regions of `P.bitsBase` words each -- `visited` at 0, the two frontier regions at `P.bitsBase` and `2 x P.bitsBase`
|
|
5
|
-
* (which one is the frontier and which the next swaps by the level's parity, `P.mode`, so nothing is copied between
|
|
6
|
-
* levels), `flags` at `3 x P.bitsBase` -- with bit `s` of word `v` meaning "source `s` has reached / is at / is next
|
|
7
|
-
* at `v`". The expansion is `advance-expand`'s block-mapped strip (P8-T5): each workgroup loads up to `WG` entries of
|
|
8
|
-
* the frontier LIST (the vertices any source is at, compacted from the flags by the host's `compact`), scans their
|
|
9
|
-
* degrees with the inlined Hillis-Steele scan of `bfs-contract` (this is not a twin kernel: `needs: []`), and every
|
|
10
|
-
* invocation strips the aggregate by binary search. The claim is inline: for the arc `(u, x)` the mask is the sources
|
|
11
|
-
* at `u` that have not reached `x`; `atomicOr` on `visited[x]` returns the bits this lane won (`fresh`), which go into
|
|
12
|
-
* the next region, set `flags[x]` (the flags region is part of the one atomic binding, so a plain store is a compile
|
|
13
|
-
* error), and are tallied per source in WORKGROUP memory -- one global `atomicAdd` per source per workgroup after the
|
|
14
|
-
* strip loop, never one per arc. Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan and the
|
|
15
|
-
* `workgroupUniformLoad` sit unconditionally after the guard, the strip loop is bounded by the uniform aggregate, and
|
|
16
|
-
* the flush's barrier follows it in uniform control flow. A sampled run (`P.perNode == 1`) also adds each claim's
|
|
17
|
-
* distance (`level + 1`, one per won bit) into the per-node sum of `x`, `perSource[128 + x]`: the distance from each
|
|
18
|
-
* of the batch's sources TO `x`, which is what a node's sampled closeness sums on an undirected graph. Body only (spec 3.5, D9); the text is normative: the
|
|
19
|
-
* sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
20
|
-
*/
|
|
21
|
-
export const closenessSweepWgsl = /* wgsl */ `
|
|
22
|
-
var<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan
|
|
23
|
-
var<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row
|
|
24
|
-
var<workgroup> rowOf: array<u32, WG>; // the frontier vertex of each entry (the source end of its arcs)
|
|
25
|
-
var<workgroup> local: array<atomic<u32>, 32>; // this workgroup's fresh claims per source
|
|
26
|
-
var<workgroup> wcount: u32; // the frontier list's length
|
|
27
|
-
var<workgroup> wdist: u32; // the distance of this level's claims
|
|
28
|
-
|
|
29
|
-
@compute @workgroup_size(WG)
|
|
30
|
-
fn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
31
|
-
if (lid.x == 0u) {
|
|
32
|
-
wcount = atomicLoad(&counters[0]); // the frontier list's length (compact's total)
|
|
33
|
-
wdist = atomicLoad(&counters[11]) + 1u; // the level word: this level claims at level + 1
|
|
34
|
-
}
|
|
35
|
-
let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
|
|
36
|
-
let dist = workgroupUniformLoad(&wdist);
|
|
37
|
-
let nextBase = select(2u * P.bitsBase, P.bitsBase, P.mode == 1u); // the region that is next this level
|
|
38
|
-
let frontierBase = 3u * P.bitsBase - nextBase; // the other one: the region that is the frontier
|
|
39
|
-
for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
|
|
40
|
-
let i = b0 + lid.x; // this lane's frontier entry
|
|
41
|
-
var deg = 0u;
|
|
42
|
-
var start = 0u;
|
|
43
|
-
var u = 0u;
|
|
44
|
-
if (i < count) { // guarded loads into locals (3.5 rule 1)
|
|
45
|
-
u = frontierList[i];
|
|
46
|
-
let lo = max(rowPtr[u], P.arcBase);
|
|
47
|
-
let hi = min(rowPtr[u + 1u], P.arcEnd);
|
|
48
|
-
start = lo;
|
|
49
|
-
deg = select(0u, hi - lo, hi > lo);
|
|
50
|
-
}
|
|
51
|
-
if (lid.x < 32u) { atomicStore(&local[lid.x], 0u); } // zeroed before the strip loop (WebGPU zero-initialises workgroup memory; said anyway)
|
|
52
|
-
sh[lid.x] = deg;
|
|
53
|
-
rowStart[lid.x] = start;
|
|
54
|
-
rowOf[lid.x] = u;
|
|
55
|
-
workgroupBarrier();
|
|
56
|
-
for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees (bfs-contract's, inlined)
|
|
57
|
-
var t = 0u;
|
|
58
|
-
if (lid.x >= s) { t = sh[lid.x - s]; }
|
|
59
|
-
workgroupBarrier();
|
|
60
|
-
sh[lid.x] = sh[lid.x] + t;
|
|
61
|
-
workgroupBarrier();
|
|
62
|
-
}
|
|
63
|
-
let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier
|
|
64
|
-
for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
|
|
65
|
-
var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p
|
|
66
|
-
var hi = WG;
|
|
67
|
-
loop {
|
|
68
|
-
if (lo >= hi) { break; }
|
|
69
|
-
let mid = (lo + hi) / 2u;
|
|
70
|
-
if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }
|
|
71
|
-
}
|
|
72
|
-
let k = lo;
|
|
73
|
-
var exclusive = 0u;
|
|
74
|
-
if (k > 0u) { exclusive = sh[k - 1u]; }
|
|
75
|
-
let arc = rowStart[k] + (p - exclusive);
|
|
76
|
-
let x = colIdx[arc - P.arcBase];
|
|
77
|
-
let mask = atomicLoad(&bits[frontierBase + rowOf[k]]) & ~atomicLoad(&bits[x]); // the sources at u that have not reached x
|
|
78
|
-
if (mask != 0u) {
|
|
79
|
-
let old = atomicOr(&bits[x], mask); // visited: the claim, one read-modify-write
|
|
80
|
-
let fresh = mask & ~old; // the sources whose claim this lane won
|
|
81
|
-
if (fresh != 0u) {
|
|
82
|
-
atomicOr(&bits[nextBase + x], fresh);
|
|
83
|
-
atomicStore(&bits[3u * P.bitsBase + x], 1u); // flags: x is in the next frontier list (compact reads it)
|
|
84
|
-
if (P.perNode == 1u) { // a sampled run: x's distance to each source won
|
|
85
|
-
atomicAdd(&perSource[128u + x], countOneBits(fresh) * dist);
|
|
86
|
-
}
|
|
87
|
-
var b = fresh;
|
|
88
|
-
loop { // one tally per set bit of fresh
|
|
89
|
-
if (b == 0u) { break; }
|
|
90
|
-
let s = firstTrailingBit(b);
|
|
91
|
-
atomicAdd(&local[s], 1u);
|
|
92
|
-
b = b & (b - 1u);
|
|
93
|
-
}
|
|
94
|
-
}
|
|
95
|
-
}
|
|
96
|
-
}
|
|
97
|
-
workgroupBarrier(); // uniform: the loop's bound is the uniform aggregate
|
|
98
|
-
if (lid.x < 32u) { // ONE global atomic per source per workgroup
|
|
99
|
-
let c = atomicLoad(&local[lid.x]);
|
|
100
|
-
if (c != 0u) { atomicAdd(&perSource[lid.x], c); } // newCount[s]
|
|
101
|
-
}
|
|
102
|
-
workgroupBarrier(); // sh, rowStart, rowOf and local are reused by the next block
|
|
103
|
-
}
|
|
104
|
-
}
|
|
105
|
-
`;
|
|
106
|
-
//# sourceMappingURL=closeness-sweep.wgsl.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"closeness-sweep.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/closeness-sweep.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAoF5C,CAAC"}
|
|
@@ -1,68 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* The `closeness-reduce` kernel body (design 8.4, 9.7 "integer distances before division"; P8-T11, the P8 plan's
|
|
3
|
-
* PD-13): the one-lane bookkeeping of the bit-parallel sweep, two roles by `P.role`. Role 0 is the level boundary,
|
|
4
|
-
* recorded BEFORE the level's `compact` and sweep: `done` is whether the previous level's compacted count is 0 (so the
|
|
5
|
-
* traversal ends one level after the last claim, one empty sweep and no wrong sum), then for every source `s` the
|
|
6
|
-
* claims the level just run made (`newCount[s]`) are folded into `reached[s]` and into the 64-bit `sum[s]` at the
|
|
7
|
-
* distance `level + 1` -- the product as a 16-bit split into a low and a high word, then the add with its carry --
|
|
8
|
-
* `newCount[s]` is zeroed and `level` advances. Role 1 is the seed of a batch: for the batch's `k = min(32, n -
|
|
9
|
-
* P.source)` sources, bit `s` into `visited[source_s]` and into the frontier region level 0 reads (region 1, since
|
|
10
|
-
* level 0's parity is 0), `flags[source_s] = 1` (level 0's `compact` turns the flags into the list; a seeded list
|
|
11
|
-
* would be overwritten by a compaction of all-zero flags), `counters[0] = k` (not done) and `level = U32_MAX` (so
|
|
12
|
-
* level 0's boundary accumulates nothing and brings the word to 0, and level 1's counts the distance-1 claims at 1).
|
|
13
|
-
* Role 2 is role 1 for a sampled run: source `s` of the batch is word `P.source + s` of the source list the host
|
|
14
|
-
* wrote after the per-node sums (`perSource[128 + P.bitsBase + ...]`), `P.n` is the list's length, and a node listed
|
|
15
|
-
* twice in one batch carries both bits (the seed ORs, so a duplicate runs twice).
|
|
16
|
-
* No barrier follows the early return of the other lanes (3.5 rule 1). Body only (spec 3.5, D9); the text is
|
|
17
|
-
* normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
18
|
-
*/
|
|
19
|
-
export const closenessReduceWgsl = /* wgsl */ `
|
|
20
|
-
@compute @workgroup_size(WG)
|
|
21
|
-
fn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
22
|
-
if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
|
|
23
|
-
if (P.role != 0u) { // the seed of a batch: P.source is its first source
|
|
24
|
-
let k = min(32u, P.n - P.source);
|
|
25
|
-
for (var s = 0u; s < k; s = s + 1u) {
|
|
26
|
-
var v = P.source + s;
|
|
27
|
-
if (P.role == 2u) { v = atomicLoad(&perSource[128u + P.bitsBase + P.source + s]); } // a sampled run's list
|
|
28
|
-
let bit = 1u << s;
|
|
29
|
-
bits[v] = bits[v] | bit; // visited
|
|
30
|
-
bits[P.bitsBase + v] = bits[P.bitsBase + v] | bit; // the frontier level 0 reads (region 1: level 0's parity is 0)
|
|
31
|
-
bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list
|
|
32
|
-
}
|
|
33
|
-
atomicStore(&counters[0], k); // not done
|
|
34
|
-
atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0
|
|
35
|
-
atomicStore(&counters[15], 0u); // done
|
|
36
|
-
return;
|
|
37
|
-
}
|
|
38
|
-
// role 0: the level boundary -- done from the previous level's compacted count, then the accumulation
|
|
39
|
-
let count = atomicLoad(&counters[0]);
|
|
40
|
-
atomicStore(&counters[15], select(0u, 1u, count == 0u));
|
|
41
|
-
let level = atomicLoad(&counters[11]);
|
|
42
|
-
let d = level + 1u; // the distance of the claims the level just run made
|
|
43
|
-
for (var s = 0u; s < 32u; s = s + 1u) {
|
|
44
|
-
let c = atomicLoad(&perSource[s]); // newCount[s]
|
|
45
|
-
atomicStore(&perSource[32u + s], atomicLoad(&perSource[32u + s]) + c); // reached[s]
|
|
46
|
-
// sum[s] += c x d in 64 bits: the 16-bit split product (pLo, pHi), then the add with its carry
|
|
47
|
-
let cLo = c & 0xFFFFu;
|
|
48
|
-
let cHi = c >> 16u;
|
|
49
|
-
let dLo = d & 0xFFFFu;
|
|
50
|
-
let dHi = d >> 16u;
|
|
51
|
-
let ll = cLo * dLo;
|
|
52
|
-
let lh = cLo * dHi;
|
|
53
|
-
let hl = cHi * dLo;
|
|
54
|
-
let mid = (ll >> 16u) + (lh & 0xFFFFu) + (hl & 0xFFFFu);
|
|
55
|
-
let pLo = (ll & 0xFFFFu) | ((mid & 0xFFFFu) << 16u);
|
|
56
|
-
let pHi = (cHi * dHi) + (lh >> 16u) + (hl >> 16u) + (mid >> 16u);
|
|
57
|
-
var lo = atomicLoad(&perSource[64u + s]); // sumLo[s]
|
|
58
|
-
var hi = atomicLoad(&perSource[96u + s]); // sumHi[s]
|
|
59
|
-
let before = lo;
|
|
60
|
-
lo = lo + pLo;
|
|
61
|
-
hi = hi + pHi + select(0u, 1u, lo < before);
|
|
62
|
-
atomicStore(&perSource[64u + s], lo);
|
|
63
|
-
atomicStore(&perSource[96u + s], hi);
|
|
64
|
-
atomicStore(&perSource[s], 0u);
|
|
65
|
-
}
|
|
66
|
-
atomicStore(&counters[11], level + 1u);
|
|
67
|
-
}
|
|
68
|
-
`;
|
|
@@ -1,105 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* The `closeness-sweep` kernel body (design 8.4 "32 sources per u32 word"; P8-T11, the P8 plan's PD-13 / DEP-P8-E):
|
|
3
|
-
* one level of the bit-parallel multi-source breadth-first search. The batch's state is one `bits` buffer of four
|
|
4
|
-
* regions of `P.bitsBase` words each -- `visited` at 0, the two frontier regions at `P.bitsBase` and `2 x P.bitsBase`
|
|
5
|
-
* (which one is the frontier and which the next swaps by the level's parity, `P.mode`, so nothing is copied between
|
|
6
|
-
* levels), `flags` at `3 x P.bitsBase` -- with bit `s` of word `v` meaning "source `s` has reached / is at / is next
|
|
7
|
-
* at `v`". The expansion is `advance-expand`'s block-mapped strip (P8-T5): each workgroup loads up to `WG` entries of
|
|
8
|
-
* the frontier LIST (the vertices any source is at, compacted from the flags by the host's `compact`), scans their
|
|
9
|
-
* degrees with the inlined Hillis-Steele scan of `bfs-contract` (this is not a twin kernel: `needs: []`), and every
|
|
10
|
-
* invocation strips the aggregate by binary search. The claim is inline: for the arc `(u, x)` the mask is the sources
|
|
11
|
-
* at `u` that have not reached `x`; `atomicOr` on `visited[x]` returns the bits this lane won (`fresh`), which go into
|
|
12
|
-
* the next region, set `flags[x]` (the flags region is part of the one atomic binding, so a plain store is a compile
|
|
13
|
-
* error), and are tallied per source in WORKGROUP memory -- one global `atomicAdd` per source per workgroup after the
|
|
14
|
-
* strip loop, never one per arc. Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan and the
|
|
15
|
-
* `workgroupUniformLoad` sit unconditionally after the guard, the strip loop is bounded by the uniform aggregate, and
|
|
16
|
-
* the flush's barrier follows it in uniform control flow. A sampled run (`P.perNode == 1`) also adds each claim's
|
|
17
|
-
* distance (`level + 1`, one per won bit) into the per-node sum of `x`, `perSource[128 + x]`: the distance from each
|
|
18
|
-
* of the batch's sources TO `x`, which is what a node's sampled closeness sums on an undirected graph. Body only (spec 3.5, D9); the text is normative: the
|
|
19
|
-
* sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
20
|
-
*/
|
|
21
|
-
export const closenessSweepWgsl = /* wgsl */ `
|
|
22
|
-
var<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan
|
|
23
|
-
var<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row
|
|
24
|
-
var<workgroup> rowOf: array<u32, WG>; // the frontier vertex of each entry (the source end of its arcs)
|
|
25
|
-
var<workgroup> local: array<atomic<u32>, 32>; // this workgroup's fresh claims per source
|
|
26
|
-
var<workgroup> wcount: u32; // the frontier list's length
|
|
27
|
-
var<workgroup> wdist: u32; // the distance of this level's claims
|
|
28
|
-
|
|
29
|
-
@compute @workgroup_size(WG)
|
|
30
|
-
fn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
31
|
-
if (lid.x == 0u) {
|
|
32
|
-
wcount = atomicLoad(&counters[0]); // the frontier list's length (compact's total)
|
|
33
|
-
wdist = atomicLoad(&counters[11]) + 1u; // the level word: this level claims at level + 1
|
|
34
|
-
}
|
|
35
|
-
let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
|
|
36
|
-
let dist = workgroupUniformLoad(&wdist);
|
|
37
|
-
let nextBase = select(2u * P.bitsBase, P.bitsBase, P.mode == 1u); // the region that is next this level
|
|
38
|
-
let frontierBase = 3u * P.bitsBase - nextBase; // the other one: the region that is the frontier
|
|
39
|
-
for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
|
|
40
|
-
let i = b0 + lid.x; // this lane's frontier entry
|
|
41
|
-
var deg = 0u;
|
|
42
|
-
var start = 0u;
|
|
43
|
-
var u = 0u;
|
|
44
|
-
if (i < count) { // guarded loads into locals (3.5 rule 1)
|
|
45
|
-
u = frontierList[i];
|
|
46
|
-
let lo = max(rowPtr[u], P.arcBase);
|
|
47
|
-
let hi = min(rowPtr[u + 1u], P.arcEnd);
|
|
48
|
-
start = lo;
|
|
49
|
-
deg = select(0u, hi - lo, hi > lo);
|
|
50
|
-
}
|
|
51
|
-
if (lid.x < 32u) { atomicStore(&local[lid.x], 0u); } // zeroed before the strip loop (WebGPU zero-initialises workgroup memory; said anyway)
|
|
52
|
-
sh[lid.x] = deg;
|
|
53
|
-
rowStart[lid.x] = start;
|
|
54
|
-
rowOf[lid.x] = u;
|
|
55
|
-
workgroupBarrier();
|
|
56
|
-
for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees (bfs-contract's, inlined)
|
|
57
|
-
var t = 0u;
|
|
58
|
-
if (lid.x >= s) { t = sh[lid.x - s]; }
|
|
59
|
-
workgroupBarrier();
|
|
60
|
-
sh[lid.x] = sh[lid.x] + t;
|
|
61
|
-
workgroupBarrier();
|
|
62
|
-
}
|
|
63
|
-
let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier
|
|
64
|
-
for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
|
|
65
|
-
var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p
|
|
66
|
-
var hi = WG;
|
|
67
|
-
loop {
|
|
68
|
-
if (lo >= hi) { break; }
|
|
69
|
-
let mid = (lo + hi) / 2u;
|
|
70
|
-
if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }
|
|
71
|
-
}
|
|
72
|
-
let k = lo;
|
|
73
|
-
var exclusive = 0u;
|
|
74
|
-
if (k > 0u) { exclusive = sh[k - 1u]; }
|
|
75
|
-
let arc = rowStart[k] + (p - exclusive);
|
|
76
|
-
let x = colIdx[arc - P.arcBase];
|
|
77
|
-
let mask = atomicLoad(&bits[frontierBase + rowOf[k]]) & ~atomicLoad(&bits[x]); // the sources at u that have not reached x
|
|
78
|
-
if (mask != 0u) {
|
|
79
|
-
let old = atomicOr(&bits[x], mask); // visited: the claim, one read-modify-write
|
|
80
|
-
let fresh = mask & ~old; // the sources whose claim this lane won
|
|
81
|
-
if (fresh != 0u) {
|
|
82
|
-
atomicOr(&bits[nextBase + x], fresh);
|
|
83
|
-
atomicStore(&bits[3u * P.bitsBase + x], 1u); // flags: x is in the next frontier list (compact reads it)
|
|
84
|
-
if (P.perNode == 1u) { // a sampled run: x's distance to each source won
|
|
85
|
-
atomicAdd(&perSource[128u + x], countOneBits(fresh) * dist);
|
|
86
|
-
}
|
|
87
|
-
var b = fresh;
|
|
88
|
-
loop { // one tally per set bit of fresh
|
|
89
|
-
if (b == 0u) { break; }
|
|
90
|
-
let s = firstTrailingBit(b);
|
|
91
|
-
atomicAdd(&local[s], 1u);
|
|
92
|
-
b = b & (b - 1u);
|
|
93
|
-
}
|
|
94
|
-
}
|
|
95
|
-
}
|
|
96
|
-
}
|
|
97
|
-
workgroupBarrier(); // uniform: the loop's bound is the uniform aggregate
|
|
98
|
-
if (lid.x < 32u) { // ONE global atomic per source per workgroup
|
|
99
|
-
let c = atomicLoad(&local[lid.x]);
|
|
100
|
-
if (c != 0u) { atomicAdd(&perSource[lid.x], c); } // newCount[s]
|
|
101
|
-
}
|
|
102
|
-
workgroupBarrier(); // sh, rowStart, rowOf and local are reused by the next block
|
|
103
|
-
}
|
|
104
|
-
}
|
|
105
|
-
`;
|