@graphty/webgpu-graph-algorithms 0.6.2 → 0.6.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +62 -32
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-BXqgCifx.js → context-Dvq-Cc6v.js} +71 -25
- package/dist/chunks/context-Dvq-Cc6v.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +8 -6
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +57 -6
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/bellman-ford.d.ts +60 -0
- package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
- package/dist/src/algorithms/bellman-ford.js +301 -0
- package/dist/src/algorithms/bellman-ford.js.map +1 -0
- package/dist/src/algorithms/bfs.d.ts +67 -0
- package/dist/src/algorithms/bfs.d.ts.map +1 -0
- package/dist/src/algorithms/bfs.js +534 -0
- package/dist/src/algorithms/bfs.js.map +1 -0
- package/dist/src/algorithms/closeness.d.ts +53 -0
- package/dist/src/algorithms/closeness.d.ts.map +1 -0
- package/dist/src/algorithms/closeness.js +323 -0
- package/dist/src/algorithms/closeness.js.map +1 -0
- package/dist/src/algorithms/scope.d.ts +5 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +3 -0
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +71 -0
- package/dist/src/algorithms/sssp.d.ts.map +1 -0
- package/dist/src/algorithms/sssp.js +585 -0
- package/dist/src/algorithms/sssp.js.map +1 -0
- package/dist/src/constants.d.ts +12 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +12 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/index.d.ts +8 -2
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +7 -1
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts +4 -4
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +39 -5
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernel/uniform-ring.d.ts +8 -0
- package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
- package/dist/src/kernel/uniform-ring.js +13 -0
- package/dist/src/kernel/uniform-ring.js.map +1 -1
- package/dist/src/kernels.d.ts +44 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +371 -3
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +62 -0
- package/dist/src/primitives/advance.d.ts.map +1 -0
- package/dist/src/primitives/advance.js +95 -0
- package/dist/src/primitives/advance.js.map +1 -0
- package/dist/src/primitives/compact.d.ts +89 -0
- package/dist/src/primitives/compact.d.ts.map +1 -0
- package/dist/src/primitives/compact.js +233 -0
- package/dist/src/primitives/compact.js.map +1 -0
- package/dist/src/primitives/core-shape.d.ts +22 -1
- package/dist/src/primitives/core-shape.d.ts.map +1 -1
- package/dist/src/primitives/core-shape.js +33 -3
- package/dist/src/primitives/core-shape.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +156 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -0
- package/dist/src/primitives/frontier.js +259 -0
- package/dist/src/primitives/frontier.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +16 -7
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/traversal.d.ts +53 -0
- package/dist/src/types/traversal.d.ts.map +1 -0
- package/dist/src/types/traversal.js +10 -0
- package/dist/src/types/traversal.js.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
- package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +59 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js +210 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
- package/dist/webgpu-graph-algorithms.js +3207 -377
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +5 -4
- package/src/accelerator.ts +65 -7
- package/src/algorithms/bellman-ford.ts +387 -0
- package/src/algorithms/bfs.ts +626 -0
- package/src/algorithms/closeness.ts +395 -0
- package/src/algorithms/scope.ts +13 -3
- package/src/algorithms/sssp.ts +767 -0
- package/src/constants.ts +12 -0
- package/src/index.ts +14 -1
- package/src/kernel/prelude.ts +39 -4
- package/src/kernel/uniform-ring.ts +14 -0
- package/src/kernels.ts +450 -6
- package/src/primitives/advance.ts +130 -0
- package/src/primitives/compact.ts +323 -0
- package/src/primitives/core-shape.ts +41 -3
- package/src/primitives/frontier.ts +388 -0
- package/src/types/accelerator.ts +18 -5
- package/src/types/traversal.ts +56 -0
- package/src/wgsl/advance-expand.wgsl.ts +68 -0
- package/src/wgsl/bf-relax.wgsl.ts +57 -0
- package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
- package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
- package/src/wgsl/bfs-contract.wgsl.ts +54 -0
- package/src/wgsl/bfs-fused.wgsl.ts +77 -0
- package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
- package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
- package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
- package/src/wgsl/compact-scatter.wgsl.ts +16 -0
- package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
- package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
- package/src/wgsl/frontier-finalize.wgsl.ts +209 -0
- package/src/wgsl/sssp-pred.wgsl.ts +79 -0
- package/src/wgsl/sssp-relax.wgsl.ts +71 -0
- package/dist/chunks/context-BXqgCifx.js.map +0 -1
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `closeness-reduce` kernel body (design 8.4, 9.7 "integer distances before division"; P8-T11, the P8 plan's
|
|
3
|
+
* PD-13): the one-lane bookkeeping of the bit-parallel sweep, two roles by `P.role`. Role 0 is the level boundary,
|
|
4
|
+
* recorded BEFORE the level's `compact` and sweep: `done` is whether the previous level's compacted count is 0 (so the
|
|
5
|
+
* traversal ends one level after the last claim, one empty sweep and no wrong sum), then for every source `s` the
|
|
6
|
+
* claims the level just run made (`newCount[s]`) are folded into `reached[s]` and into the 64-bit `sum[s]` at the
|
|
7
|
+
* distance `level + 1` -- the product as a 16-bit split into a low and a high word, then the add with its carry --
|
|
8
|
+
* `newCount[s]` is zeroed and `level` advances. Role 1 is the seed of a batch: for the batch's `k = min(32, n -
|
|
9
|
+
* P.source)` sources, bit `s` into `visited[source_s]` and into the frontier region level 0 reads (region 1, since
|
|
10
|
+
* level 0's parity is 0), `flags[source_s] = 1` (level 0's `compact` turns the flags into the list; a seeded list
|
|
11
|
+
* would be overwritten by a compaction of all-zero flags), `counters[0] = k` (not done) and `level = U32_MAX` (so
|
|
12
|
+
* level 0's boundary accumulates nothing and brings the word to 0, and level 1's counts the distance-1 claims at 1).
|
|
13
|
+
* No barrier follows the early return of the other lanes (3.5 rule 1). Body only (spec 3.5, D9); the text is
|
|
14
|
+
* normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
15
|
+
*/
|
|
16
|
+
export declare const closenessReduceWgsl = "\n@compute @workgroup_size(WG)\nfn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role == 1u) { // the seed of a batch: P.source is its first source\n let k = min(32u, P.n - P.source);\n for (var s = 0u; s < k; s = s + 1u) {\n let v = P.source + s;\n let bit = 1u << s;\n bits[v] = bit; // visited\n bits[P.bitsBase + v] = bit; // the frontier level 0 reads (region 1: level 0's parity is 0)\n bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list\n }\n atomicStore(&counters[0], k); // not done\n atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0\n atomicStore(&counters[15], 0u); // done\n return;\n }\n // role 0: the level boundary -- done from the previous level's compacted count, then the accumulation\n let count = atomicLoad(&counters[0]);\n atomicStore(&counters[15], select(0u, 1u, count == 0u));\n let level = atomicLoad(&counters[11]);\n let d = level + 1u; // the distance of the claims the level just run made\n for (var s = 0u; s < 32u; s = s + 1u) {\n let c = atomicLoad(&perSource[s]); // newCount[s]\n atomicStore(&perSource[32u + s], atomicLoad(&perSource[32u + s]) + c); // reached[s]\n // sum[s] += c x d in 64 bits: the 16-bit split product (pLo, pHi), then the add with its carry\n let cLo = c & 0xFFFFu;\n let cHi = c >> 16u;\n let dLo = d & 0xFFFFu;\n let dHi = d >> 16u;\n let ll = cLo * dLo;\n let lh = cLo * dHi;\n let hl = cHi * dLo;\n let mid = (ll >> 16u) + (lh & 0xFFFFu) + (hl & 0xFFFFu);\n let pLo = (ll & 0xFFFFu) | ((mid & 0xFFFFu) << 16u);\n let pHi = (cHi * dHi) + (lh >> 16u) + (hl >> 16u) + (mid >> 16u);\n var lo = atomicLoad(&perSource[64u + s]); // sumLo[s]\n var hi = atomicLoad(&perSource[96u + s]); // sumHi[s]\n let before = lo;\n lo = lo + pLo;\n hi = hi + pHi + select(0u, 1u, lo < before);\n atomicStore(&perSource[64u + s], lo);\n atomicStore(&perSource[96u + s], hi);\n atomicStore(&perSource[s], 0u);\n }\n atomicStore(&counters[11], level + 1u);\n}\n";
|
|
17
|
+
//# sourceMappingURL=closeness-reduce.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"closeness-reduce.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-reduce.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AACH,eAAO,MAAM,mBAAmB,0qFAgD/B,CAAC"}
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `closeness-reduce` kernel body (design 8.4, 9.7 "integer distances before division"; P8-T11, the P8 plan's
|
|
3
|
+
* PD-13): the one-lane bookkeeping of the bit-parallel sweep, two roles by `P.role`. Role 0 is the level boundary,
|
|
4
|
+
* recorded BEFORE the level's `compact` and sweep: `done` is whether the previous level's compacted count is 0 (so the
|
|
5
|
+
* traversal ends one level after the last claim, one empty sweep and no wrong sum), then for every source `s` the
|
|
6
|
+
* claims the level just run made (`newCount[s]`) are folded into `reached[s]` and into the 64-bit `sum[s]` at the
|
|
7
|
+
* distance `level + 1` -- the product as a 16-bit split into a low and a high word, then the add with its carry --
|
|
8
|
+
* `newCount[s]` is zeroed and `level` advances. Role 1 is the seed of a batch: for the batch's `k = min(32, n -
|
|
9
|
+
* P.source)` sources, bit `s` into `visited[source_s]` and into the frontier region level 0 reads (region 1, since
|
|
10
|
+
* level 0's parity is 0), `flags[source_s] = 1` (level 0's `compact` turns the flags into the list; a seeded list
|
|
11
|
+
* would be overwritten by a compaction of all-zero flags), `counters[0] = k` (not done) and `level = U32_MAX` (so
|
|
12
|
+
* level 0's boundary accumulates nothing and brings the word to 0, and level 1's counts the distance-1 claims at 1).
|
|
13
|
+
* No barrier follows the early return of the other lanes (3.5 rule 1). Body only (spec 3.5, D9); the text is
|
|
14
|
+
* normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
15
|
+
*/
|
|
16
|
+
export const closenessReduceWgsl = /* wgsl */ `
|
|
17
|
+
@compute @workgroup_size(WG)
|
|
18
|
+
fn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
19
|
+
if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
|
|
20
|
+
if (P.role == 1u) { // the seed of a batch: P.source is its first source
|
|
21
|
+
let k = min(32u, P.n - P.source);
|
|
22
|
+
for (var s = 0u; s < k; s = s + 1u) {
|
|
23
|
+
let v = P.source + s;
|
|
24
|
+
let bit = 1u << s;
|
|
25
|
+
bits[v] = bit; // visited
|
|
26
|
+
bits[P.bitsBase + v] = bit; // the frontier level 0 reads (region 1: level 0's parity is 0)
|
|
27
|
+
bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list
|
|
28
|
+
}
|
|
29
|
+
atomicStore(&counters[0], k); // not done
|
|
30
|
+
atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0
|
|
31
|
+
atomicStore(&counters[15], 0u); // done
|
|
32
|
+
return;
|
|
33
|
+
}
|
|
34
|
+
// role 0: the level boundary -- done from the previous level's compacted count, then the accumulation
|
|
35
|
+
let count = atomicLoad(&counters[0]);
|
|
36
|
+
atomicStore(&counters[15], select(0u, 1u, count == 0u));
|
|
37
|
+
let level = atomicLoad(&counters[11]);
|
|
38
|
+
let d = level + 1u; // the distance of the claims the level just run made
|
|
39
|
+
for (var s = 0u; s < 32u; s = s + 1u) {
|
|
40
|
+
let c = atomicLoad(&perSource[s]); // newCount[s]
|
|
41
|
+
atomicStore(&perSource[32u + s], atomicLoad(&perSource[32u + s]) + c); // reached[s]
|
|
42
|
+
// sum[s] += c x d in 64 bits: the 16-bit split product (pLo, pHi), then the add with its carry
|
|
43
|
+
let cLo = c & 0xFFFFu;
|
|
44
|
+
let cHi = c >> 16u;
|
|
45
|
+
let dLo = d & 0xFFFFu;
|
|
46
|
+
let dHi = d >> 16u;
|
|
47
|
+
let ll = cLo * dLo;
|
|
48
|
+
let lh = cLo * dHi;
|
|
49
|
+
let hl = cHi * dLo;
|
|
50
|
+
let mid = (ll >> 16u) + (lh & 0xFFFFu) + (hl & 0xFFFFu);
|
|
51
|
+
let pLo = (ll & 0xFFFFu) | ((mid & 0xFFFFu) << 16u);
|
|
52
|
+
let pHi = (cHi * dHi) + (lh >> 16u) + (hl >> 16u) + (mid >> 16u);
|
|
53
|
+
var lo = atomicLoad(&perSource[64u + s]); // sumLo[s]
|
|
54
|
+
var hi = atomicLoad(&perSource[96u + s]); // sumHi[s]
|
|
55
|
+
let before = lo;
|
|
56
|
+
lo = lo + pLo;
|
|
57
|
+
hi = hi + pHi + select(0u, 1u, lo < before);
|
|
58
|
+
atomicStore(&perSource[64u + s], lo);
|
|
59
|
+
atomicStore(&perSource[96u + s], hi);
|
|
60
|
+
atomicStore(&perSource[s], 0u);
|
|
61
|
+
}
|
|
62
|
+
atomicStore(&counters[11], level + 1u);
|
|
63
|
+
}
|
|
64
|
+
`;
|
|
65
|
+
//# sourceMappingURL=closeness-reduce.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"closeness-reduce.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/closeness-reduce.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AACH,MAAM,CAAC,MAAM,mBAAmB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAgD7C,CAAC"}
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `closeness-sweep` kernel body (design 8.4 "32 sources per u32 word"; P8-T11, the P8 plan's PD-13 / DEP-P8-E):
|
|
3
|
+
* one level of the bit-parallel multi-source breadth-first search. The batch's state is one `bits` buffer of four
|
|
4
|
+
* regions of `P.bitsBase` words each -- `visited` at 0, the two frontier regions at `P.bitsBase` and `2 x P.bitsBase`
|
|
5
|
+
* (which one is the frontier and which the next swaps by the level's parity, `P.mode`, so nothing is copied between
|
|
6
|
+
* levels), `flags` at `3 x P.bitsBase` -- with bit `s` of word `v` meaning "source `s` has reached / is at / is next
|
|
7
|
+
* at `v`". The expansion is `advance-expand`'s block-mapped strip (P8-T5): each workgroup loads up to `WG` entries of
|
|
8
|
+
* the frontier LIST (the vertices any source is at, compacted from the flags by the host's `compact`), scans their
|
|
9
|
+
* degrees with the inlined Hillis-Steele scan of `bfs-contract` (this is not a twin kernel: `needs: []`), and every
|
|
10
|
+
* invocation strips the aggregate by binary search. The claim is inline: for the arc `(u, x)` the mask is the sources
|
|
11
|
+
* at `u` that have not reached `x`; `atomicOr` on `visited[x]` returns the bits this lane won (`fresh`), which go into
|
|
12
|
+
* the next region, set `flags[x]` (the flags region is part of the one atomic binding, so a plain store is a compile
|
|
13
|
+
* error), and are tallied per source in WORKGROUP memory -- one global `atomicAdd` per source per workgroup after the
|
|
14
|
+
* strip loop, never one per arc. Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan and the
|
|
15
|
+
* `workgroupUniformLoad` sit unconditionally after the guard, the strip loop is bounded by the uniform aggregate, and
|
|
16
|
+
* the flush's barrier follows it in uniform control flow. Body only (spec 3.5, D9); the text is normative: the
|
|
17
|
+
* sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
18
|
+
*/
|
|
19
|
+
export declare const closenessSweepWgsl = "\nvar<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan\nvar<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row\nvar<workgroup> rowOf: array<u32, WG>; // the frontier vertex of each entry (the source end of its arcs)\nvar<workgroup> local: array<atomic<u32>, 32>; // this workgroup's fresh claims per source\nvar<workgroup> wcount: u32; // the frontier list's length\n\n@compute @workgroup_size(WG)\nfn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x == 0u) { wcount = atomicLoad(&counters[0]); } // the frontier list's length (compact's total)\n let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers\n let nextBase = select(2u * P.bitsBase, P.bitsBase, P.mode == 1u); // the region that is next this level\n let frontierBase = 3u * P.bitsBase - nextBase; // the other one: the region that is the frontier\n for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries\n let i = b0 + lid.x; // this lane's frontier entry\n var deg = 0u;\n var start = 0u;\n var u = 0u;\n if (i < count) { // guarded loads into locals (3.5 rule 1)\n u = frontierList[i];\n let lo = max(rowPtr[u], P.arcBase);\n let hi = min(rowPtr[u + 1u], P.arcEnd);\n start = lo;\n deg = select(0u, hi - lo, hi > lo);\n }\n if (lid.x < 32u) { atomicStore(&local[lid.x], 0u); } // zeroed before the strip loop (WebGPU zero-initialises workgroup memory; said anyway)\n sh[lid.x] = deg;\n rowStart[lid.x] = start;\n rowOf[lid.x] = u;\n workgroupBarrier();\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees (bfs-contract's, inlined)\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier\n for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...\n var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p\n var hi = WG;\n loop {\n if (lo >= hi) { break; }\n let mid = (lo + hi) / 2u;\n if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }\n }\n let k = lo;\n var exclusive = 0u;\n if (k > 0u) { exclusive = sh[k - 1u]; }\n let arc = rowStart[k] + (p - exclusive);\n let x = colIdx[arc - P.arcBase];\n let mask = atomicLoad(&bits[frontierBase + rowOf[k]]) & ~atomicLoad(&bits[x]); // the sources at u that have not reached x\n if (mask != 0u) {\n let old = atomicOr(&bits[x], mask); // visited: the claim, one read-modify-write\n let fresh = mask & ~old; // the sources whose claim this lane won\n if (fresh != 0u) {\n atomicOr(&bits[nextBase + x], fresh);\n atomicStore(&bits[3u * P.bitsBase + x], 1u); // flags: x is in the next frontier list (compact reads it)\n var b = fresh;\n loop { // one tally per set bit of fresh\n if (b == 0u) { break; }\n let s = firstTrailingBit(b);\n atomicAdd(&local[s], 1u);\n b = b & (b - 1u);\n }\n }\n }\n }\n workgroupBarrier(); // uniform: the loop's bound is the uniform aggregate\n if (lid.x < 32u) { // ONE global atomic per source per workgroup\n let c = atomicLoad(&local[lid.x]);\n if (c != 0u) { atomicAdd(&perSource[lid.x], c); } // newCount[s]\n }\n workgroupBarrier(); // sh, rowStart, rowOf and local are reused by the next block\n }\n}\n";
|
|
20
|
+
//# sourceMappingURL=closeness-sweep.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"closeness-sweep.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-sweep.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,eAAO,MAAM,kBAAkB,ymJA4E9B,CAAC"}
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `closeness-sweep` kernel body (design 8.4 "32 sources per u32 word"; P8-T11, the P8 plan's PD-13 / DEP-P8-E):
|
|
3
|
+
* one level of the bit-parallel multi-source breadth-first search. The batch's state is one `bits` buffer of four
|
|
4
|
+
* regions of `P.bitsBase` words each -- `visited` at 0, the two frontier regions at `P.bitsBase` and `2 x P.bitsBase`
|
|
5
|
+
* (which one is the frontier and which the next swaps by the level's parity, `P.mode`, so nothing is copied between
|
|
6
|
+
* levels), `flags` at `3 x P.bitsBase` -- with bit `s` of word `v` meaning "source `s` has reached / is at / is next
|
|
7
|
+
* at `v`". The expansion is `advance-expand`'s block-mapped strip (P8-T5): each workgroup loads up to `WG` entries of
|
|
8
|
+
* the frontier LIST (the vertices any source is at, compacted from the flags by the host's `compact`), scans their
|
|
9
|
+
* degrees with the inlined Hillis-Steele scan of `bfs-contract` (this is not a twin kernel: `needs: []`), and every
|
|
10
|
+
* invocation strips the aggregate by binary search. The claim is inline: for the arc `(u, x)` the mask is the sources
|
|
11
|
+
* at `u` that have not reached `x`; `atomicOr` on `visited[x]` returns the bits this lane won (`fresh`), which go into
|
|
12
|
+
* the next region, set `flags[x]` (the flags region is part of the one atomic binding, so a plain store is a compile
|
|
13
|
+
* error), and are tallied per source in WORKGROUP memory -- one global `atomicAdd` per source per workgroup after the
|
|
14
|
+
* strip loop, never one per arc. Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan and the
|
|
15
|
+
* `workgroupUniformLoad` sit unconditionally after the guard, the strip loop is bounded by the uniform aggregate, and
|
|
16
|
+
* the flush's barrier follows it in uniform control flow. Body only (spec 3.5, D9); the text is normative: the
|
|
17
|
+
* sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
18
|
+
*/
|
|
19
|
+
export const closenessSweepWgsl = /* wgsl */ `
|
|
20
|
+
var<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan
|
|
21
|
+
var<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row
|
|
22
|
+
var<workgroup> rowOf: array<u32, WG>; // the frontier vertex of each entry (the source end of its arcs)
|
|
23
|
+
var<workgroup> local: array<atomic<u32>, 32>; // this workgroup's fresh claims per source
|
|
24
|
+
var<workgroup> wcount: u32; // the frontier list's length
|
|
25
|
+
|
|
26
|
+
@compute @workgroup_size(WG)
|
|
27
|
+
fn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
28
|
+
if (lid.x == 0u) { wcount = atomicLoad(&counters[0]); } // the frontier list's length (compact's total)
|
|
29
|
+
let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
|
|
30
|
+
let nextBase = select(2u * P.bitsBase, P.bitsBase, P.mode == 1u); // the region that is next this level
|
|
31
|
+
let frontierBase = 3u * P.bitsBase - nextBase; // the other one: the region that is the frontier
|
|
32
|
+
for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
|
|
33
|
+
let i = b0 + lid.x; // this lane's frontier entry
|
|
34
|
+
var deg = 0u;
|
|
35
|
+
var start = 0u;
|
|
36
|
+
var u = 0u;
|
|
37
|
+
if (i < count) { // guarded loads into locals (3.5 rule 1)
|
|
38
|
+
u = frontierList[i];
|
|
39
|
+
let lo = max(rowPtr[u], P.arcBase);
|
|
40
|
+
let hi = min(rowPtr[u + 1u], P.arcEnd);
|
|
41
|
+
start = lo;
|
|
42
|
+
deg = select(0u, hi - lo, hi > lo);
|
|
43
|
+
}
|
|
44
|
+
if (lid.x < 32u) { atomicStore(&local[lid.x], 0u); } // zeroed before the strip loop (WebGPU zero-initialises workgroup memory; said anyway)
|
|
45
|
+
sh[lid.x] = deg;
|
|
46
|
+
rowStart[lid.x] = start;
|
|
47
|
+
rowOf[lid.x] = u;
|
|
48
|
+
workgroupBarrier();
|
|
49
|
+
for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees (bfs-contract's, inlined)
|
|
50
|
+
var t = 0u;
|
|
51
|
+
if (lid.x >= s) { t = sh[lid.x - s]; }
|
|
52
|
+
workgroupBarrier();
|
|
53
|
+
sh[lid.x] = sh[lid.x] + t;
|
|
54
|
+
workgroupBarrier();
|
|
55
|
+
}
|
|
56
|
+
let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier
|
|
57
|
+
for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
|
|
58
|
+
var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p
|
|
59
|
+
var hi = WG;
|
|
60
|
+
loop {
|
|
61
|
+
if (lo >= hi) { break; }
|
|
62
|
+
let mid = (lo + hi) / 2u;
|
|
63
|
+
if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }
|
|
64
|
+
}
|
|
65
|
+
let k = lo;
|
|
66
|
+
var exclusive = 0u;
|
|
67
|
+
if (k > 0u) { exclusive = sh[k - 1u]; }
|
|
68
|
+
let arc = rowStart[k] + (p - exclusive);
|
|
69
|
+
let x = colIdx[arc - P.arcBase];
|
|
70
|
+
let mask = atomicLoad(&bits[frontierBase + rowOf[k]]) & ~atomicLoad(&bits[x]); // the sources at u that have not reached x
|
|
71
|
+
if (mask != 0u) {
|
|
72
|
+
let old = atomicOr(&bits[x], mask); // visited: the claim, one read-modify-write
|
|
73
|
+
let fresh = mask & ~old; // the sources whose claim this lane won
|
|
74
|
+
if (fresh != 0u) {
|
|
75
|
+
atomicOr(&bits[nextBase + x], fresh);
|
|
76
|
+
atomicStore(&bits[3u * P.bitsBase + x], 1u); // flags: x is in the next frontier list (compact reads it)
|
|
77
|
+
var b = fresh;
|
|
78
|
+
loop { // one tally per set bit of fresh
|
|
79
|
+
if (b == 0u) { break; }
|
|
80
|
+
let s = firstTrailingBit(b);
|
|
81
|
+
atomicAdd(&local[s], 1u);
|
|
82
|
+
b = b & (b - 1u);
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
workgroupBarrier(); // uniform: the loop's bound is the uniform aggregate
|
|
88
|
+
if (lid.x < 32u) { // ONE global atomic per source per workgroup
|
|
89
|
+
let c = atomicLoad(&local[lid.x]);
|
|
90
|
+
if (c != 0u) { atomicAdd(&perSource[lid.x], c); } // newCount[s]
|
|
91
|
+
}
|
|
92
|
+
workgroupBarrier(); // sh, rowStart, rowOf and local are reused by the next block
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
`;
|
|
96
|
+
//# sourceMappingURL=closeness-sweep.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"closeness-sweep.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/closeness-sweep.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA4E5C,CAAC"}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `compact-scatter` kernel body (spec 6 row 4; P8-T3): the scatter step of `compact`, after the caller's flags
|
|
3
|
+
* have been exclusive-scanned into `offsets`. Every flagged entry lands at its offset, in queue order, so the output
|
|
4
|
+
* is bitwise reproducible; lane 0 writes the total -- the last offset plus the last flag -- into `outCount[P.outIndex]`
|
|
5
|
+
* (the block is bound whole and indexed because a four-byte word is never 256-aligned). The planner never dispatches
|
|
6
|
+
* this body for count 0, so `P.count - 1u` never wraps. Body only (spec 3.5, D9); normative text.
|
|
7
|
+
*/
|
|
8
|
+
export declare const compactScatterWgsl = "\n@compute @workgroup_size(WG)\nfn compact_scatter(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n if (i == 0u) { outCount[P.outIndex] = offsets[P.count - 1u] + flags[P.count - 1u]; } // the exclusive scan's total; the planner never dispatches for count 0\n if (i >= P.count) { return; } // no barrier follows\n if (flags[i] != 0u) { out[offsets[i]] = queue[i]; }\n}\n";
|
|
9
|
+
//# sourceMappingURL=compact-scatter.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"compact-scatter.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/compact-scatter.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,eAAO,MAAM,kBAAkB,sgBAQ9B,CAAC"}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `compact-scatter` kernel body (spec 6 row 4; P8-T3): the scatter step of `compact`, after the caller's flags
|
|
3
|
+
* have been exclusive-scanned into `offsets`. Every flagged entry lands at its offset, in queue order, so the output
|
|
4
|
+
* is bitwise reproducible; lane 0 writes the total -- the last offset plus the last flag -- into `outCount[P.outIndex]`
|
|
5
|
+
* (the block is bound whole and indexed because a four-byte word is never 256-aligned). The planner never dispatches
|
|
6
|
+
* this body for count 0, so `P.count - 1u` never wraps. Body only (spec 3.5, D9); normative text.
|
|
7
|
+
*/
|
|
8
|
+
export const compactScatterWgsl = /* wgsl */ `
|
|
9
|
+
@compute @workgroup_size(WG)
|
|
10
|
+
fn compact_scatter(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
11
|
+
let i = linear_id(wid, lid.x);
|
|
12
|
+
if (i == 0u) { outCount[P.outIndex] = offsets[P.count - 1u] + flags[P.count - 1u]; } // the exclusive scan's total; the planner never dispatches for count 0
|
|
13
|
+
if (i >= P.count) { return; } // no barrier follows
|
|
14
|
+
if (flags[i] != 0u) { out[offsets[i]] = queue[i]; }
|
|
15
|
+
}
|
|
16
|
+
`;
|
|
17
|
+
//# sourceMappingURL=compact-scatter.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"compact-scatter.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/compact-scatter.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,UAAU,CAAC;;;;;;;;CAQ5C,CAAC"}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `dedupe-claim` kernel body (spec 6 row 4; P8-T3): the first of `dedupe`'s two dispatches. Every entry stores
|
|
3
|
+
* its own index into `owner[queue[i]]` (relaxed atomics: between two dispatches last-writer-wins is well defined, so
|
|
4
|
+
* exactly one index per distinct vertex is the owner). The entry count is `P.count`, or the device word
|
|
5
|
+
* `counters[P.countIndex]` clamped to `P.count` when `P.countIndex` is not `U32_MAX` (the SSSP piles only know their
|
|
6
|
+
* count on the device). `owner` needs no reset between calls: a stale or garbage word is only ever read by an entry
|
|
7
|
+
* whose vertex a current entry has just overwritten. Body only (spec 3.5, D9); normative text.
|
|
8
|
+
*/
|
|
9
|
+
export declare const dedupeClaimWgsl = "\n@compute @workgroup_size(WG)\nfn dedupe_claim(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n var count = P.count;\n if (P.countIndex != U32_MAX) { count = min(atomicLoad(&counters[P.countIndex]), P.count); } // a device-side count, clamped to the capacity\n for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // grid-stride; no barrier anywhere\n atomicStore(&owner[queue[i]], i);\n }\n}\n";
|
|
10
|
+
//# sourceMappingURL=dedupe-claim.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"dedupe-claim.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/dedupe-claim.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AACH,eAAO,MAAM,eAAe,6dAS3B,CAAC"}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `dedupe-claim` kernel body (spec 6 row 4; P8-T3): the first of `dedupe`'s two dispatches. Every entry stores
|
|
3
|
+
* its own index into `owner[queue[i]]` (relaxed atomics: between two dispatches last-writer-wins is well defined, so
|
|
4
|
+
* exactly one index per distinct vertex is the owner). The entry count is `P.count`, or the device word
|
|
5
|
+
* `counters[P.countIndex]` clamped to `P.count` when `P.countIndex` is not `U32_MAX` (the SSSP piles only know their
|
|
6
|
+
* count on the device). `owner` needs no reset between calls: a stale or garbage word is only ever read by an entry
|
|
7
|
+
* whose vertex a current entry has just overwritten. Body only (spec 3.5, D9); normative text.
|
|
8
|
+
*/
|
|
9
|
+
export const dedupeClaimWgsl = /* wgsl */ `
|
|
10
|
+
@compute @workgroup_size(WG)
|
|
11
|
+
fn dedupe_claim(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
12
|
+
var count = P.count;
|
|
13
|
+
if (P.countIndex != U32_MAX) { count = min(atomicLoad(&counters[P.countIndex]), P.count); } // a device-side count, clamped to the capacity
|
|
14
|
+
for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // grid-stride; no barrier anywhere
|
|
15
|
+
atomicStore(&owner[queue[i]], i);
|
|
16
|
+
}
|
|
17
|
+
}
|
|
18
|
+
`;
|
|
19
|
+
//# sourceMappingURL=dedupe-claim.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"dedupe-claim.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/dedupe-claim.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AACH,MAAM,CAAC,MAAM,eAAe,GAAG,UAAU,CAAC;;;;;;;;;CASzC,CAAC"}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `dedupe-filter` kernel body (spec 6 row 4; P8-T3): the second of `dedupe`'s two dispatches, a SEPARATE
|
|
3
|
+
* dispatch because a plain store read back in the same one is a data race (WGSL 6.5.7). An entry survives iff
|
|
4
|
+
* `owner[queue[i]]` still holds its own index; the survivors are packed by a Hillis-Steele inclusive scan of the keep
|
|
5
|
+
* bits in workgroup memory (scan-block's rounds and barrier placement, verbatim) and ONE `atomicAdd` per workgroup on
|
|
6
|
+
* `outCount[P.outIndex]` for the block's aggregate, so the output is set-deterministic: the surviving SET is fixed,
|
|
7
|
+
* the order inside `out` follows the schedule. The entry count is `P.count` or the device word
|
|
8
|
+
* `outCount[P.countIndex]` clamped to it -- the same block and the same word `dedupe-claim` read. Every lane reaches
|
|
9
|
+
* every barrier: the guarded work goes into locals (spec 3.5 rule 1). Body only (spec 3.5, D9); normative text.
|
|
10
|
+
*/
|
|
11
|
+
export declare const dedupeFilterWgsl = "\nvar<workgroup> sh: array<u32, WG>;\nvar<workgroup> base: u32;\nvar<workgroup> wcount: u32;\n\n@compute @workgroup_size(WG)\nfn dedupe_filter(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x == 0u) {\n var count = P.count;\n if (P.countIndex != U32_MAX) { count = min(atomicLoad(&outCount[P.countIndex]), P.count); } // the same count source as the claim\n wcount = count;\n }\n let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers\n for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries\n let i = b0 + lid.x;\n var keep = 0u;\n var v = 0u;\n if (i < count) { v = queue[i]; keep = select(0u, 1u, atomicLoad(&owner[v]) == i); } // guarded work into locals\n sh[lid.x] = keep;\n workgroupBarrier(); // every lane, unconditionally (3.5 rule 1)\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of keep\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let inclusive = sh[lid.x];\n if (lid.x == WG - 1u) { base = atomicAdd(&outCount[P.outIndex], inclusive); } // ONE atomic per workgroup: the block's aggregate\n workgroupBarrier();\n if (keep == 1u) { out[base + inclusive - 1u] = v; }\n workgroupBarrier(); // sh and base are reused by the next block\n }\n}\n";
|
|
12
|
+
//# sourceMappingURL=dedupe-filter.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"dedupe-filter.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/dedupe-filter.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AACH,eAAO,MAAM,gBAAgB,szDAkC5B,CAAC"}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `dedupe-filter` kernel body (spec 6 row 4; P8-T3): the second of `dedupe`'s two dispatches, a SEPARATE
|
|
3
|
+
* dispatch because a plain store read back in the same one is a data race (WGSL 6.5.7). An entry survives iff
|
|
4
|
+
* `owner[queue[i]]` still holds its own index; the survivors are packed by a Hillis-Steele inclusive scan of the keep
|
|
5
|
+
* bits in workgroup memory (scan-block's rounds and barrier placement, verbatim) and ONE `atomicAdd` per workgroup on
|
|
6
|
+
* `outCount[P.outIndex]` for the block's aggregate, so the output is set-deterministic: the surviving SET is fixed,
|
|
7
|
+
* the order inside `out` follows the schedule. The entry count is `P.count` or the device word
|
|
8
|
+
* `outCount[P.countIndex]` clamped to it -- the same block and the same word `dedupe-claim` read. Every lane reaches
|
|
9
|
+
* every barrier: the guarded work goes into locals (spec 3.5 rule 1). Body only (spec 3.5, D9); normative text.
|
|
10
|
+
*/
|
|
11
|
+
export const dedupeFilterWgsl = /* wgsl */ `
|
|
12
|
+
var<workgroup> sh: array<u32, WG>;
|
|
13
|
+
var<workgroup> base: u32;
|
|
14
|
+
var<workgroup> wcount: u32;
|
|
15
|
+
|
|
16
|
+
@compute @workgroup_size(WG)
|
|
17
|
+
fn dedupe_filter(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
18
|
+
if (lid.x == 0u) {
|
|
19
|
+
var count = P.count;
|
|
20
|
+
if (P.countIndex != U32_MAX) { count = min(atomicLoad(&outCount[P.countIndex]), P.count); } // the same count source as the claim
|
|
21
|
+
wcount = count;
|
|
22
|
+
}
|
|
23
|
+
let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
|
|
24
|
+
for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
|
|
25
|
+
let i = b0 + lid.x;
|
|
26
|
+
var keep = 0u;
|
|
27
|
+
var v = 0u;
|
|
28
|
+
if (i < count) { v = queue[i]; keep = select(0u, 1u, atomicLoad(&owner[v]) == i); } // guarded work into locals
|
|
29
|
+
sh[lid.x] = keep;
|
|
30
|
+
workgroupBarrier(); // every lane, unconditionally (3.5 rule 1)
|
|
31
|
+
for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of keep
|
|
32
|
+
var t = 0u;
|
|
33
|
+
if (lid.x >= s) { t = sh[lid.x - s]; }
|
|
34
|
+
workgroupBarrier();
|
|
35
|
+
sh[lid.x] = sh[lid.x] + t;
|
|
36
|
+
workgroupBarrier();
|
|
37
|
+
}
|
|
38
|
+
let inclusive = sh[lid.x];
|
|
39
|
+
if (lid.x == WG - 1u) { base = atomicAdd(&outCount[P.outIndex], inclusive); } // ONE atomic per workgroup: the block's aggregate
|
|
40
|
+
workgroupBarrier();
|
|
41
|
+
if (keep == 1u) { out[base + inclusive - 1u] = v; }
|
|
42
|
+
workgroupBarrier(); // sh and base are reused by the next block
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
`;
|
|
46
|
+
//# sourceMappingURL=dedupe-filter.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"dedupe-filter.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/dedupe-filter.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAkC1C,CAAC"}
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `frontier-finalize` kernel body (design 5.4, 6 row 7; P8-T4, the P8 plan's PD-3 / PD-23 / DEP-P8-C): the
|
|
3
|
+
* device-side selector of the frontier family. One workgroup, one lane, no barrier after the early return (spec 3.5
|
|
4
|
+
* rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's dispatch sizes become
|
|
5
|
+
* known at two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`,
|
|
6
|
+
* advances `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`), chooses the path and writes this
|
|
7
|
+
* level's seven 16-byte slots at `P.slotBase`; role 1 runs once the edge queue is filled -- clamps `edgeCount` to
|
|
8
|
+
* `P.edgeCapacity`, sizes the contract slot, or, when `edgeCountUnclamped` exceeds the capacity, zeroes it and sizes
|
|
9
|
+
* the fused-retry slot from `frontierCount` instead (PD-23). Two rules: a boundary that finds `done` set zeroes its
|
|
10
|
+
* slots and moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when
|
|
11
|
+
* role 0 chose one, which it reads from slot 0's `x` (a storage write of one dispatch is visible to the next of the
|
|
12
|
+
* same pass). The `(x, y)` arithmetic is `indirect-finalize`'s verbatim (P4), so a count above 2^32 - wg cannot wrap
|
|
13
|
+
* and a group count above MAX_WORKGROUPS_PER_DIM splits in 2D; slots 2 and 6 are sized one WORKGROUP per entry for
|
|
14
|
+
* `bfs-fused`; slots 3, 4 and 5 (the bits fill over `ceil(n / 32)` words, the bitset build over the frontier, the sweep
|
|
15
|
+
* over the unvisited list) are the bottom-up level's.
|
|
16
|
+
*
|
|
17
|
+
* Roles 2 and 3 are the SSSP round boundary of the near-far loop (P8-T9, PD-20), over the same block read in its
|
|
18
|
+
* SSSP sense (word 1 the raw near half's appends, 21 the raw far half's, 0 and 20 the deduped pile counts, 22 the
|
|
19
|
+
* threshold, 23 the delta, 4 the previous threshold, 14 the round's mode) and the same seven slots: role 2 finds a
|
|
20
|
+
* non-empty raw near half and sizes the near dedupe (slots 0 and 1, count word 1, output word 0) in mode 0; finds it
|
|
21
|
+
* empty and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
|
|
22
|
+
* unchanged sets `done 3`, the host's E_UNSUPPORTED), remembers the previous threshold in word 4 and sizes the far
|
|
23
|
+
* dedupe (slots 3 and 4, count word 21, output word 20) in mode 1; finds both empty and sets `done 1`; and finds a
|
|
24
|
+
* raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in `level` when it
|
|
25
|
+
* picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds dispatched; a
|
|
26
|
+
* boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed: mode 0 sizes slot 2 (the
|
|
27
|
+
* relax over nearIn) from word 0 and restarts the raw near half (word 1 to 0); mode 1 sizes slot 5 (the pass-through
|
|
28
|
+
* over farIn) from word 20 and restarts the raw far half (word 21 to 0).
|
|
29
|
+
*
|
|
30
|
+
* Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a switch can be counted at
|
|
31
|
+
* the done boundary too, which the host model of the tests mirrors): top-down switches to bottom-up when
|
|
32
|
+
* `frontierDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's `max(1, floor(arcCount / n))`
|
|
33
|
+
* unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up switches back when
|
|
34
|
+
* `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no admitted device
|
|
35
|
+
* reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the direction at 0. Every
|
|
36
|
+
* change is counted in `switches`, the previous direction is word 14. The two unvisited words the test reads are
|
|
37
|
+
* rebuilt exactly once per submit by `bfs-unvisited-flags` (PD-18) and maintained here by subtraction: the count is
|
|
38
|
+
* subtracted from the SECOND boundary of a submit on and the degree sum from the THIRD on, because a boundary may
|
|
39
|
+
* only subtract what the submit's rebuild counted, and the frontier whose degree sum the second boundary holds was
|
|
40
|
+
* claimed before the rebuild ran (the rebuild counts the vertices unclaimed when it runs; the frontier rotated in at
|
|
41
|
+
* boundary 0 was claimed by the previous submit's last contract, so it was never in the sum; boundary b subtracts
|
|
42
|
+
* `next = |F_b|`, inside the sum iff b >= 1, and `degSum = deg(F_{b-1})`, inside it iff b >= 2). The degree sum is
|
|
43
|
+
* the "unvisited degree estimate" of the design rather than an exact count for two reasons: it is one level stale
|
|
44
|
+
* (a frontier's degree sum is only known once it has been expanded), and a bottom-up level expands nothing, so the
|
|
45
|
+
* word stops falling while bottom-up runs and overstates the set afterwards. The bias is one-directional -- an
|
|
46
|
+
* overstated m_u makes the switch INTO bottom-up harder, never easier -- and the next submit's rebuild makes it
|
|
47
|
+
* exact again. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
|
|
48
|
+
* textual edits of it.
|
|
49
|
+
*
|
|
50
|
+
* Since 2026-09-25 nothing dispatches FROM the slots (G8-F5: Dawn's validation of an indirect dispatch cost about
|
|
51
|
+
* 0.4 ms of device time each, whether or not it dispatched anything, and the seven slots of thirty-two recorded
|
|
52
|
+
* levels were 97 % of a traversal's wall time). Every level kernel is a direct grid-stride dispatch that reads the
|
|
53
|
+
* `path` word (24) this kernel writes -- 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3
|
|
54
|
+
* bottom-up, 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2) -- and the SSSP dedupes read
|
|
55
|
+
* their counts from words 8 and 9, which role 2 writes. The slots stay as the selector's recorded decision, read
|
|
56
|
+
* back by the frontier tests; deleting them with those tests is the follow-up.
|
|
57
|
+
*/
|
|
58
|
+
export declare const frontierFinalizeWgsl = "\nfn write_slot_groups(slot: u32, groups: u32, count: u32) { // groups workgroups, split in 2D above the per-dim limit\n var x = groups;\n var y = 1u;\n if (groups > MAX_WORKGROUPS_PER_DIM) {\n x = MAX_WORKGROUPS_PER_DIM;\n y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;\n }\n let base = 4u * (P.slotBase + slot); // 16-byte slots: (x, y, 1, count)\n args[base] = x; args[base + 1u] = y; args[base + 2u] = 1u; args[base + 3u] = count;\n}\nfn write_slot(slot: u32, count: u32) { // one INVOCATION per entry: ceil(count / wg) workgroups\n let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap (indirect-finalize's rule)\n write_slot_groups(slot, groups, count);\n}\nfn zero_slot(slot: u32) {\n let base = 4u * (P.slotBase + slot);\n args[base] = 0u; args[base + 1u] = 0u; args[base + 2u] = 1u; args[base + 3u] = 0u;\n}\n\n@compute @workgroup_size(WG)\nfn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role == 0u) { // the level boundary\n if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end\n for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); } // this slotBase holds the previous submit's args\n return; // no counter word moves (P8-T6's levels formula reads them)\n }\n let finished = atomicLoad(&counters[0]);\n let next = atomicLoad(&counters[1]);\n let degSum = atomicLoad(&counters[2]);\n atomicStore(&counters[3], finished); // prevFrontierCount\n atomicStore(&counters[4], degSum); // prevDegreeSum\n atomicStore(&counters[0], next); // the rotation\n atomicStore(&counters[1], 0u);\n atomicStore(&counters[2], 0u);\n atomicStore(&counters[8], 0u); // edgeCount\n atomicStore(&counters[9], 0u); // edgeCountUnclamped\n atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount\n if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped to 2 (P8-T8, PD-18)\n atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount (exact): F_b was inside the submit's rebuilt sum iff b >= 1\n }\n if (P.firstOfSubmit >= 2u) {\n atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2\n }\n let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0\n atomicStore(&counters[11], level);\n let done = (next == 0u) || (level >= P.maxDepth);\n atomicStore(&counters[15], select(0u, 1u, done));\n var direction = atomicLoad(&counters[14]);\n if (P.mode == 1u) {\n direction = 0u; // top-down only (the test seam)\n } else if (direction == 0u) {\n if (degSum > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing\n } else {\n if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking\n }\n if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches\n var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)\n if (done) {\n for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }\n } else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep\n zero_slot(0u); zero_slot(1u); zero_slot(2u); zero_slot(6u);\n write_slot(3u, (P.n + 31u) / 32u); write_slot(4u, next); write_slot(5u, atomicLoad(&counters[7]));\n path = 3u;\n atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);\n } else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)\n zero_slot(0u); zero_slot(1u); zero_slot(3u); zero_slot(4u); zero_slot(5u); zero_slot(6u);\n write_slot_groups(2u, next, next); // bfs-fused is one WORKGROUP per frontier entry\n path = 2u;\n atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);\n } else {\n zero_slot(2u); zero_slot(3u); zero_slot(4u); zero_slot(5u);\n write_slot(0u, next); // slots 1 and 6 are role 1's\n path = 1u;\n }\n atomicStore(&counters[14], direction);\n atomicStore(&counters[24], path);\n } else if (P.role == 1u) { // the edge queue is filled\n if (args[4u * P.slotBase] == 0u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to size, nothing to count\n zero_slot(1u); zero_slot(6u);\n return;\n }\n let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);\n atomicStore(&counters[8], clamped);\n if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry\n let entries = atomicLoad(&counters[0]);\n zero_slot(1u); write_slot_groups(6u, entries, entries); // one workgroup per frontier entry, as slot 2\n atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry, bfs-contract nothing\n atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);\n atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);\n } else {\n write_slot(1u, clamped); zero_slot(6u);\n atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)\n }\n } else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes\n atomicStore(&counters[8], 0u); // the dedupe counts (words 8 and 9, the SSSP sense) and the path word: nothing unless a pile is chosen below\n atomicStore(&counters[9], 0u);\n atomicStore(&counters[24], 0u);\n if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)\n for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }\n return;\n }\n let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped\n let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped\n if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE\n atomicStore(&counters[15], 2u);\n for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }\n return;\n }\n zero_slot(2u); zero_slot(5u); zero_slot(6u); // role 3 sizes the relax slots once the piles are deduped\n if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn\n atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter\n atomicStore(&counters[14], 0u); // mode 0\n write_slot(0u, nearRaw); write_slot(1u, nearRaw); // dedupe-claim, dedupe-filter over the near half\n atomicStore(&counters[8], nearRaw); // the near dedupe's count word\n atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing\n zero_slot(3u); zero_slot(4u);\n atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)\n } else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile\n let threshold = bitcast<f32>(atomicLoad(&counters[22]));\n let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)\n if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED\n atomicStore(&counters[15], 3u);\n for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }\n return;\n }\n atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below\n atomicStore(&counters[22], bitcast<u32>(raised));\n atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter\n atomicStore(&counters[14], 1u); // mode 1\n zero_slot(0u); zero_slot(1u);\n write_slot(3u, farRaw); write_slot(4u, farRaw); // dedupe-claim, dedupe-filter over the far half\n atomicStore(&counters[9], farRaw); // the far dedupe's count word\n atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing\n atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);\n } else { // both piles empty: finished\n atomicStore(&counters[15], 1u);\n for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }\n }\n } else if (P.role == 3u) { // the piles are deduped: size the relax\n if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 zeroed every slot of the round\n if (atomicLoad(&counters[14]) == 0u) {\n write_slot(2u, atomicLoad(&counters[0])); // the near round over nearIn\n atomicStore(&counters[1], 0u); // the raw near half restarts\n } else {\n write_slot(5u, atomicLoad(&counters[20])); // the pass-through over farIn\n atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far)\n }\n }\n}\n";
|
|
59
|
+
//# sourceMappingURL=frontier-finalize.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"frontier-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAwDG;AACH,eAAO,MAAM,oBAAoB,0mWAuJhC,CAAC"}
|