@graphty/webgpu-graph-algorithms 0.6.3 → 0.6.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +62 -32
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-BXqgCifx.js → context-Dvq-Cc6v.js} +71 -25
- package/dist/chunks/context-Dvq-Cc6v.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +8 -6
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +57 -6
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/bellman-ford.d.ts +60 -0
- package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
- package/dist/src/algorithms/bellman-ford.js +301 -0
- package/dist/src/algorithms/bellman-ford.js.map +1 -0
- package/dist/src/algorithms/bfs.d.ts +67 -0
- package/dist/src/algorithms/bfs.d.ts.map +1 -0
- package/dist/src/algorithms/bfs.js +534 -0
- package/dist/src/algorithms/bfs.js.map +1 -0
- package/dist/src/algorithms/closeness.d.ts +53 -0
- package/dist/src/algorithms/closeness.d.ts.map +1 -0
- package/dist/src/algorithms/closeness.js +323 -0
- package/dist/src/algorithms/closeness.js.map +1 -0
- package/dist/src/algorithms/scope.d.ts +5 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +3 -0
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +71 -0
- package/dist/src/algorithms/sssp.d.ts.map +1 -0
- package/dist/src/algorithms/sssp.js +585 -0
- package/dist/src/algorithms/sssp.js.map +1 -0
- package/dist/src/constants.d.ts +12 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +12 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/index.d.ts +8 -2
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +7 -1
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts +4 -4
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +39 -5
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernel/uniform-ring.d.ts +8 -0
- package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
- package/dist/src/kernel/uniform-ring.js +13 -0
- package/dist/src/kernel/uniform-ring.js.map +1 -1
- package/dist/src/kernels.d.ts +44 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +371 -3
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +62 -0
- package/dist/src/primitives/advance.d.ts.map +1 -0
- package/dist/src/primitives/advance.js +95 -0
- package/dist/src/primitives/advance.js.map +1 -0
- package/dist/src/primitives/compact.d.ts +89 -0
- package/dist/src/primitives/compact.d.ts.map +1 -0
- package/dist/src/primitives/compact.js +233 -0
- package/dist/src/primitives/compact.js.map +1 -0
- package/dist/src/primitives/core-shape.d.ts +22 -1
- package/dist/src/primitives/core-shape.d.ts.map +1 -1
- package/dist/src/primitives/core-shape.js +33 -3
- package/dist/src/primitives/core-shape.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +156 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -0
- package/dist/src/primitives/frontier.js +259 -0
- package/dist/src/primitives/frontier.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +16 -7
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/traversal.d.ts +53 -0
- package/dist/src/types/traversal.d.ts.map +1 -0
- package/dist/src/types/traversal.js +10 -0
- package/dist/src/types/traversal.js.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
- package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +59 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js +210 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
- package/dist/webgpu-graph-algorithms.js +3207 -377
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +5 -4
- package/src/accelerator.ts +65 -7
- package/src/algorithms/bellman-ford.ts +387 -0
- package/src/algorithms/bfs.ts +626 -0
- package/src/algorithms/closeness.ts +395 -0
- package/src/algorithms/scope.ts +13 -3
- package/src/algorithms/sssp.ts +767 -0
- package/src/constants.ts +12 -0
- package/src/index.ts +14 -1
- package/src/kernel/prelude.ts +39 -4
- package/src/kernel/uniform-ring.ts +14 -0
- package/src/kernels.ts +450 -6
- package/src/primitives/advance.ts +130 -0
- package/src/primitives/compact.ts +323 -0
- package/src/primitives/core-shape.ts +41 -3
- package/src/primitives/frontier.ts +388 -0
- package/src/types/accelerator.ts +18 -5
- package/src/types/traversal.ts +56 -0
- package/src/wgsl/advance-expand.wgsl.ts +68 -0
- package/src/wgsl/bf-relax.wgsl.ts +57 -0
- package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
- package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
- package/src/wgsl/bfs-contract.wgsl.ts +54 -0
- package/src/wgsl/bfs-fused.wgsl.ts +77 -0
- package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
- package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
- package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
- package/src/wgsl/compact-scatter.wgsl.ts +16 -0
- package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
- package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
- package/src/wgsl/frontier-finalize.wgsl.ts +209 -0
- package/src/wgsl/sssp-pred.wgsl.ts +79 -0
- package/src/wgsl/sssp-relax.wgsl.ts +71 -0
- package/dist/chunks/context-BXqgCifx.js.map +0 -1
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bfs-bottom-up` kernel body (design 8.4 "the bottom-up sweep"; P8-T8, the P8 plan's PD-18 / PD-21): one
|
|
3
|
+
* invocation per entry of the unvisited list (`unvisitedListLen`, word 7; the list is the first region of the
|
|
4
|
+
* read-only `sweepIn` buffer, the frontier bitset its second at `P.bitsBase`), over the REVERSE core bound in group 0
|
|
5
|
+
* (`coreOfView(residency.view(s, "reverse"))`: the forward arrays on an undirected snapshot, which P7's residency
|
|
6
|
+
* aliases at zero upload cost). A lane whose vertex is still at `INVALID_INDEX` -- the list is up to 32 levels
|
|
7
|
+
* stale, and an entry claimed since the rebuild is skipped on that test -- walks its in-neighbours until the FIRST
|
|
8
|
+
* one whose bit is set and breaks: that early exit is the whole point of the direction (a vertex with one frontier
|
|
9
|
+
* neighbour among thousands costs one read), and `arcsScanned` (word 16, one `atomicAdd` per workgroup of the lanes'
|
|
10
|
+
* reads through `wg_reduce_u32`) is the counter the sabotage row that removes the exit is pinned to, never a timing.
|
|
11
|
+
* The winners are packed into the output vertex queue by `bfs-contract`'s workgroup scan and one `atomicAdd` per
|
|
12
|
+
* workgroup on `nextFrontierCount`, and claim with a plain `atomicStore`: the list holds every vertex once and the
|
|
13
|
+
* sweep is vertex-parallel, so no two lanes claim one vertex. It adds nothing to `frontierDegreeSum` (a bottom-up
|
|
14
|
+
* level expands nothing), which is why `unvisitedDegreeSum` stops falling while bottom-up runs (the selector's
|
|
15
|
+
* JSDoc). Uniformity (spec 3.5 rule 1): the guarded walk writes locals, the scan and the reduction run
|
|
16
|
+
* unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
|
|
17
|
+
* test/helpers/sabotage.ts are textual edits of it.
|
|
18
|
+
*/
|
|
19
|
+
export const bfsBottomUpWgsl = /* wgsl */ `
|
|
20
|
+
var<workgroup> sh: array<u32, WG>;
|
|
21
|
+
var<workgroup> base: u32;
|
|
22
|
+
var<workgroup> wcount: u32; // the unvisited list's length on a bottom-up level, 0 on any other
|
|
23
|
+
|
|
24
|
+
@compute @workgroup_size(WG)
|
|
25
|
+
fn bfs_bottom_up(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
26
|
+
let claim = atomicLoad(&counters[11]) + 1u;
|
|
27
|
+
if (lid.x == 0u) { wcount = select(0u, atomicLoad(&counters[7]), atomicLoad(&counters[24]) == 3u); } // unvisitedListLen, on the bottom-up path only (the path word)
|
|
28
|
+
let len = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
|
|
29
|
+
for (var b0 = group_id(wid) * WG; b0 < len; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
|
|
30
|
+
let i = b0 + lid.x;
|
|
31
|
+
var won = 0u;
|
|
32
|
+
var v = 0u;
|
|
33
|
+
var reads = 0u;
|
|
34
|
+
if (i < len) { // guarded work into locals
|
|
35
|
+
v = sweepIn[i];
|
|
36
|
+
if (atomicLoad(&depth[v]) == INVALID_INDEX) { // a stale entry, claimed since the rebuild, is skipped
|
|
37
|
+
let end = min(rowPtr[v + 1u], P.arcEnd);
|
|
38
|
+
for (var a = max(rowPtr[v], P.arcBase); a < end; a = a + 1u) { // in-neighbours through the reverse core
|
|
39
|
+
reads = reads + 1u;
|
|
40
|
+
let u = colIdx[a - P.arcBase];
|
|
41
|
+
if (mask_bit(sweepIn[P.bitsBase + (u >> 5u)], u)) { won = 1u; break; } // the early exit: a real break, never a flag
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
sh[lid.x] = won;
|
|
46
|
+
workgroupBarrier();
|
|
47
|
+
for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of won (bfs-contract's)
|
|
48
|
+
var t = 0u;
|
|
49
|
+
if (lid.x >= s) { t = sh[lid.x - s]; }
|
|
50
|
+
workgroupBarrier();
|
|
51
|
+
sh[lid.x] = sh[lid.x] + t;
|
|
52
|
+
workgroupBarrier();
|
|
53
|
+
}
|
|
54
|
+
let inclusive = sh[lid.x];
|
|
55
|
+
let readsTotal = wg_reduce_u32(reads, lid.x, 0u); // arcsScanned, one atomic per workgroup (the sabotage witness, Step 6)
|
|
56
|
+
if (lid.x == WG - 1u) { base = atomicAdd(&counters[1], inclusive); }
|
|
57
|
+
if (lid.x == 0u) { atomicAdd(&counters[16], readsTotal); }
|
|
58
|
+
workgroupBarrier();
|
|
59
|
+
if (won == 1u) {
|
|
60
|
+
atomicStore(&depth[v], claim); // no claim race: the list holds v once and the sweep is vertex-parallel
|
|
61
|
+
frontierOut[base + inclusive - 1u] = v;
|
|
62
|
+
}
|
|
63
|
+
workgroupBarrier(); // sh and base are reused by the next block
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
`;
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bfs-contract` kernel body (design 8.4, 8.10 "BFS contract"; P8-T6, the P8 plan's PD-5 / PD-6 / PD-24 /
|
|
3
|
+
* DEP-P8-B): the contraction phase of the two-phase top-down level. One invocation per edge-queue entry (the target
|
|
4
|
+
* vertex `advance-expand` wrote), a direct grid-stride dispatch looping to the `edgeCount` `frontier-finalize` role 1
|
|
5
|
+
* clamped, and only when the block's `path` word says the level is two-phase. The claim is `atomicMin(&depth[v], level + 1)` and the invocation that
|
|
6
|
+
* observes `INVALID_INDEX` is the unique winner (PD-6): `atomicMin` is one read-modify-write, so only the first
|
|
7
|
+
* claimant of this level can see the sentinel; a second claimant of the same level observes `claim`, and a vertex
|
|
8
|
+
* already at a smaller depth returns that depth and keeps it. Design 8.4 says why this is not
|
|
9
|
+
* `atomicCompareExchangeWeak`: WGSL 17.8.5 lets it fail spuriously, so a compare-exchange claim needs a retry loop
|
|
10
|
+
* and a bounded loop could leave a vertex unclaimed for its level; `atomicMin` has no such failure mode. The winners
|
|
11
|
+
* are packed by a Hillis-Steele inclusive scan of the workgroup's `won` flags and ONE `atomicAdd` per workgroup on
|
|
12
|
+
* `nextFrontierCount` reserves the block's span in the output vertex queue. The edge queue holds duplicates (a vertex
|
|
13
|
+
* with three frontier neighbours appears three times); exactly one claimant wins, exactly one appends, so the next
|
|
14
|
+
* frontier is duplicate-free by construction and no ownership dedupe follows (DEP-P8-B). Nothing here writes a
|
|
15
|
+
* parent (PD-24: the post-pass does). Uniformity (spec 3.5 rule 1): the guarded work writes locals, every barrier is
|
|
16
|
+
* reached unconditionally. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
|
|
17
|
+
* test/helpers/sabotage.ts are textual edits of it.
|
|
18
|
+
*/
|
|
19
|
+
export const bfsContractWgsl = /* wgsl */ `
|
|
20
|
+
var<workgroup> sh: array<u32, WG>;
|
|
21
|
+
var<workgroup> base: u32;
|
|
22
|
+
var<workgroup> wcount: u32; // the clamped edge count on a two-phase level, 0 on any other
|
|
23
|
+
|
|
24
|
+
@compute @workgroup_size(WG)
|
|
25
|
+
fn bfs_contract(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
26
|
+
if (lid.x == 0u) { wcount = select(0u, atomicLoad(&counters[8]), atomicLoad(&counters[24]) == 1u); } // edgeCount, clamped by role 1; the two-phase path only (the path word)
|
|
27
|
+
let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
|
|
28
|
+
let claim = atomicLoad(&counters[11]) + 1u; // the depth this level assigns
|
|
29
|
+
for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
|
|
30
|
+
let i = b0 + lid.x;
|
|
31
|
+
var won = 0u;
|
|
32
|
+
var v = 0u;
|
|
33
|
+
if (i < count) { // guarded work into locals
|
|
34
|
+
v = edgeQueue[i];
|
|
35
|
+
let old = atomicMin(&depth[v], claim);
|
|
36
|
+
won = select(0u, 1u, old == INVALID_INDEX); // PD-6: the unique winner
|
|
37
|
+
}
|
|
38
|
+
sh[lid.x] = won;
|
|
39
|
+
workgroupBarrier();
|
|
40
|
+
for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of won
|
|
41
|
+
var t = 0u;
|
|
42
|
+
if (lid.x >= s) { t = sh[lid.x - s]; }
|
|
43
|
+
workgroupBarrier();
|
|
44
|
+
sh[lid.x] = sh[lid.x] + t;
|
|
45
|
+
workgroupBarrier();
|
|
46
|
+
}
|
|
47
|
+
let inclusive = sh[lid.x];
|
|
48
|
+
if (lid.x == WG - 1u) { base = atomicAdd(&counters[1], inclusive); } // nextFrontierCount: one reservation per workgroup
|
|
49
|
+
workgroupBarrier();
|
|
50
|
+
if (won == 1u) { frontierOut[base + inclusive - 1u] = v; }
|
|
51
|
+
workgroupBarrier(); // sh and base are reused by the next block
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
`;
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bfs-fused` kernel body (design 8.4 "the fused variant", 6 row 8 "the workgroup-per-row tier", 8.10 "BFS
|
|
3
|
+
* fused expand-contract"; P8-T7, the P8 plan's PD-23 / PD-24): the expansion and the contraction of one level in
|
|
4
|
+
* ONE dispatch, chosen by `frontier-finalize` for a frontier below `P.fusedMax` entries (`path` 2) and for the retry
|
|
5
|
+
* of a level whose edge queue overflowed (`path` 4, PD-23). One WORKGROUP per frontier entry, the workgroups striding
|
|
6
|
+
* the entries by the dispatch's group count (`P.stride`): lane 0 reads the entry's row clipped
|
|
7
|
+
* to the bound arc window, adds its degree to `frontierDegreeSum` (Beamer's m_f, so P8-T8's test sees fused levels
|
|
8
|
+
* too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim inline --
|
|
9
|
+
* `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
|
|
10
|
+
* packing the strip's winners into the output vertex queue by the same Hillis-Steele scan and one `atomicAdd` per
|
|
11
|
+
* strip on `nextFrontierCount`. No edge queue is written or read, which is the whole win for a tiny frontier
|
|
12
|
+
* (Merrill's fleeting iterations) and what makes the overflow retry exact: the partial edge queue is never consulted.
|
|
13
|
+
* On a retry level `advance-expand` has already added the frontier's degree to `frontierDegreeSum`, so that word
|
|
14
|
+
* holds 2 x m_f for the level and the next boundary's Beamer test and degree-sum subtraction see the doubled value;
|
|
15
|
+
* reachable only with a faked capacity or an absurd graph, accepted and said here rather than guarded. Nothing here
|
|
16
|
+
* writes a parent (PD-24: the post-pass does), which is what keeps the kernel at the eight-storage-buffer budget
|
|
17
|
+
* with the four graph slots. Uniformity (spec 3.5 rule 1): the strip loop's bound and the row start are
|
|
18
|
+
* `workgroupUniformLoad`s, so every barrier of the per-strip append is in uniform control flow; the guarded claim
|
|
19
|
+
* writes locals; a trailing barrier protects `sh` and `base` before the next strip reuses them. The claim line and
|
|
20
|
+
* the scan are `bfs-contract`'s verbatim so a reader can diff the two bodies; the scan's comment differs on purpose
|
|
21
|
+
* (the sabotage rows need distinct find strings). Body only (spec 3.5, D9); the text is normative: the sabotage rows
|
|
22
|
+
* of test/helpers/sabotage.ts are textual edits of it.
|
|
23
|
+
*/
|
|
24
|
+
export const bfsFusedWgsl = /* wgsl */ `
|
|
25
|
+
var<workgroup> sh: array<u32, WG>;
|
|
26
|
+
var<workgroup> wdeg: u32;
|
|
27
|
+
var<workgroup> wstart: u32;
|
|
28
|
+
var<workgroup> base: u32;
|
|
29
|
+
var<workgroup> wcount: u32; // the frontier's length on a fused or retry level, 0 on any other
|
|
30
|
+
|
|
31
|
+
@compute @workgroup_size(WG)
|
|
32
|
+
fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
33
|
+
let claim = atomicLoad(&counters[11]) + 1u;
|
|
34
|
+
if (lid.x == 0u) {
|
|
35
|
+
let path = atomicLoad(&counters[24]); // 2 the fused level, 4 the overflow retry (PD-23)
|
|
36
|
+
wcount = select(0u, atomicLoad(&counters[0]), path == 2u || path == 4u);
|
|
37
|
+
}
|
|
38
|
+
let count = workgroupUniformLoad(&wcount); // uniform: the entry loop below holds barriers
|
|
39
|
+
for (var g = group_id(wid); g < count; g = g + P.stride) { // one workgroup per frontier entry; P.stride is the dispatch's GROUP count
|
|
40
|
+
if (lid.x == 0u) {
|
|
41
|
+
let u = frontierIn[g];
|
|
42
|
+
let a0 = max(rowPtr[u], P.arcBase);
|
|
43
|
+
let a1 = min(rowPtr[u + 1u], P.arcEnd);
|
|
44
|
+
let d = select(0u, a1 - a0, a1 > a0);
|
|
45
|
+
wdeg = d;
|
|
46
|
+
wstart = a0;
|
|
47
|
+
atomicAdd(&counters[2], d); // frontierDegreeSum, so Beamer's test (P8-T8) sees fused levels too
|
|
48
|
+
}
|
|
49
|
+
let deg = workgroupUniformLoad(&wdeg); // uniform: the loop below may hold barriers
|
|
50
|
+
let start = workgroupUniformLoad(&wstart);
|
|
51
|
+
for (var p0 = 0u; p0 < deg; p0 = p0 + WG) { // strip the row WG arcs at a time
|
|
52
|
+
let p = p0 + lid.x;
|
|
53
|
+
var won = 0u;
|
|
54
|
+
var v = 0u;
|
|
55
|
+
if (p < deg) { // guarded claim into locals
|
|
56
|
+
v = colIdx[start + p - P.arcBase];
|
|
57
|
+
let old = atomicMin(&depth[v], claim);
|
|
58
|
+
won = select(0u, 1u, old == INVALID_INDEX);
|
|
59
|
+
}
|
|
60
|
+
sh[lid.x] = won;
|
|
61
|
+
workgroupBarrier();
|
|
62
|
+
for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of won (bfs-contract's, verbatim)
|
|
63
|
+
var t = 0u;
|
|
64
|
+
if (lid.x >= s) { t = sh[lid.x - s]; }
|
|
65
|
+
workgroupBarrier();
|
|
66
|
+
sh[lid.x] = sh[lid.x] + t;
|
|
67
|
+
workgroupBarrier();
|
|
68
|
+
}
|
|
69
|
+
let inclusive = sh[lid.x];
|
|
70
|
+
if (lid.x == WG - 1u) { base = atomicAdd(&counters[1], inclusive); }
|
|
71
|
+
workgroupBarrier();
|
|
72
|
+
if (won == 1u) { frontierOut[base + inclusive - 1u] = v; }
|
|
73
|
+
workgroupBarrier(); // sh and base are reused by the next strip
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
`;
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bfs-unvisited-flags` kernel body (design 8.4; P8-T8, the P8 plan's PD-18): the producer of the unvisited set
|
|
3
|
+
* Beamer's test is against, run ONCE per submit before the levels. It grid-strides over the vertices
|
|
4
|
+
* (`planGridStride(n)`, `P.stride` the plan's stride) and, per lane, counts the vertices still at `INVALID_INDEX`
|
|
5
|
+
* (`unvisitedCount`, word 5), sums their OUT-degrees (`unvisitedDegreeSum`, word 6: Beamer's m_u counts the edges
|
|
6
|
+
* top-down would examine), and flags the unvisited vertices with a non-zero IN-degree (`unvisitedListLen`, word 7:
|
|
7
|
+
* what the bottom-up sweep iterates, since a vertex nobody points at can never be claimed by it); the three lane
|
|
8
|
+
* sums are reduced by the prelude's `wg_reduce_u32` (its sum code) and ONE `atomicAdd` per word per workgroup lands
|
|
9
|
+
* them in the counters block. `compact` over an iota queue then turns `flags` into the unvisited list. The list is
|
|
10
|
+
* up to `MAX_LEVELS_PER_SUBMIT` levels stale by the time the sweep reads it: it holds vertices claimed since the
|
|
11
|
+
* rebuild, which the sweep skips on the `depth == INVALID_INDEX` test it makes anyway, so staleness costs a few
|
|
12
|
+
* wasted reads and never a wrong depth. Between rebuilds `frontier-finalize` maintains words 5 and 6 by subtraction
|
|
13
|
+
* (its JSDoc states the boundary rule). Uniformity (spec 3.5 rule 1): the loop holds no barrier, and the three
|
|
14
|
+
* reductions run unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
|
|
15
|
+
* test/helpers/sabotage.ts are textual edits of it.
|
|
16
|
+
*/
|
|
17
|
+
export const bfsUnvisitedFlagsWgsl = /* wgsl */ `
|
|
18
|
+
@compute @workgroup_size(WG)
|
|
19
|
+
fn bfs_unvisited_flags(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
20
|
+
let first = linear_id(wid, lid.x);
|
|
21
|
+
var cnt = 0u;
|
|
22
|
+
var degSum = 0u;
|
|
23
|
+
var len = 0u;
|
|
24
|
+
for (var v = first; v < P.n; v = v + P.stride) { // no barrier inside: the trip count is per lane
|
|
25
|
+
let unv = depth[v] == INVALID_INDEX;
|
|
26
|
+
let listed = unv && (inDegree[v] != 0u);
|
|
27
|
+
flags[v] = select(0u, 1u, listed);
|
|
28
|
+
cnt = cnt + select(0u, 1u, unv);
|
|
29
|
+
degSum = degSum + select(0u, outDegree[v], unv); // the OUT-degree: Beamer's m_u counts the edges top-down would examine
|
|
30
|
+
len = len + select(0u, 1u, listed);
|
|
31
|
+
}
|
|
32
|
+
let c = wg_reduce_u32(cnt, lid.x, 0u); // the prelude's workgroup sum (combine_u's sum code); uniform: after the loop
|
|
33
|
+
let d = wg_reduce_u32(degSum, lid.x, 0u);
|
|
34
|
+
let l = wg_reduce_u32(len, lid.x, 0u);
|
|
35
|
+
if (lid.x == 0u) { // ONE atomic per word per workgroup
|
|
36
|
+
atomicAdd(&counters[5], c);
|
|
37
|
+
atomicAdd(&counters[6], d);
|
|
38
|
+
atomicAdd(&counters[7], l);
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
`;
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `closeness-reduce` kernel body (design 8.4, 9.7 "integer distances before division"; P8-T11, the P8 plan's
|
|
3
|
+
* PD-13): the one-lane bookkeeping of the bit-parallel sweep, two roles by `P.role`. Role 0 is the level boundary,
|
|
4
|
+
* recorded BEFORE the level's `compact` and sweep: `done` is whether the previous level's compacted count is 0 (so the
|
|
5
|
+
* traversal ends one level after the last claim, one empty sweep and no wrong sum), then for every source `s` the
|
|
6
|
+
* claims the level just run made (`newCount[s]`) are folded into `reached[s]` and into the 64-bit `sum[s]` at the
|
|
7
|
+
* distance `level + 1` -- the product as a 16-bit split into a low and a high word, then the add with its carry --
|
|
8
|
+
* `newCount[s]` is zeroed and `level` advances. Role 1 is the seed of a batch: for the batch's `k = min(32, n -
|
|
9
|
+
* P.source)` sources, bit `s` into `visited[source_s]` and into the frontier region level 0 reads (region 1, since
|
|
10
|
+
* level 0's parity is 0), `flags[source_s] = 1` (level 0's `compact` turns the flags into the list; a seeded list
|
|
11
|
+
* would be overwritten by a compaction of all-zero flags), `counters[0] = k` (not done) and `level = U32_MAX` (so
|
|
12
|
+
* level 0's boundary accumulates nothing and brings the word to 0, and level 1's counts the distance-1 claims at 1).
|
|
13
|
+
* No barrier follows the early return of the other lanes (3.5 rule 1). Body only (spec 3.5, D9); the text is
|
|
14
|
+
* normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
15
|
+
*/
|
|
16
|
+
export const closenessReduceWgsl = /* wgsl */ `
|
|
17
|
+
@compute @workgroup_size(WG)
|
|
18
|
+
fn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
19
|
+
if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
|
|
20
|
+
if (P.role == 1u) { // the seed of a batch: P.source is its first source
|
|
21
|
+
let k = min(32u, P.n - P.source);
|
|
22
|
+
for (var s = 0u; s < k; s = s + 1u) {
|
|
23
|
+
let v = P.source + s;
|
|
24
|
+
let bit = 1u << s;
|
|
25
|
+
bits[v] = bit; // visited
|
|
26
|
+
bits[P.bitsBase + v] = bit; // the frontier level 0 reads (region 1: level 0's parity is 0)
|
|
27
|
+
bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list
|
|
28
|
+
}
|
|
29
|
+
atomicStore(&counters[0], k); // not done
|
|
30
|
+
atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0
|
|
31
|
+
atomicStore(&counters[15], 0u); // done
|
|
32
|
+
return;
|
|
33
|
+
}
|
|
34
|
+
// role 0: the level boundary -- done from the previous level's compacted count, then the accumulation
|
|
35
|
+
let count = atomicLoad(&counters[0]);
|
|
36
|
+
atomicStore(&counters[15], select(0u, 1u, count == 0u));
|
|
37
|
+
let level = atomicLoad(&counters[11]);
|
|
38
|
+
let d = level + 1u; // the distance of the claims the level just run made
|
|
39
|
+
for (var s = 0u; s < 32u; s = s + 1u) {
|
|
40
|
+
let c = atomicLoad(&perSource[s]); // newCount[s]
|
|
41
|
+
atomicStore(&perSource[32u + s], atomicLoad(&perSource[32u + s]) + c); // reached[s]
|
|
42
|
+
// sum[s] += c x d in 64 bits: the 16-bit split product (pLo, pHi), then the add with its carry
|
|
43
|
+
let cLo = c & 0xFFFFu;
|
|
44
|
+
let cHi = c >> 16u;
|
|
45
|
+
let dLo = d & 0xFFFFu;
|
|
46
|
+
let dHi = d >> 16u;
|
|
47
|
+
let ll = cLo * dLo;
|
|
48
|
+
let lh = cLo * dHi;
|
|
49
|
+
let hl = cHi * dLo;
|
|
50
|
+
let mid = (ll >> 16u) + (lh & 0xFFFFu) + (hl & 0xFFFFu);
|
|
51
|
+
let pLo = (ll & 0xFFFFu) | ((mid & 0xFFFFu) << 16u);
|
|
52
|
+
let pHi = (cHi * dHi) + (lh >> 16u) + (hl >> 16u) + (mid >> 16u);
|
|
53
|
+
var lo = atomicLoad(&perSource[64u + s]); // sumLo[s]
|
|
54
|
+
var hi = atomicLoad(&perSource[96u + s]); // sumHi[s]
|
|
55
|
+
let before = lo;
|
|
56
|
+
lo = lo + pLo;
|
|
57
|
+
hi = hi + pHi + select(0u, 1u, lo < before);
|
|
58
|
+
atomicStore(&perSource[64u + s], lo);
|
|
59
|
+
atomicStore(&perSource[96u + s], hi);
|
|
60
|
+
atomicStore(&perSource[s], 0u);
|
|
61
|
+
}
|
|
62
|
+
atomicStore(&counters[11], level + 1u);
|
|
63
|
+
}
|
|
64
|
+
`;
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `closeness-sweep` kernel body (design 8.4 "32 sources per u32 word"; P8-T11, the P8 plan's PD-13 / DEP-P8-E):
|
|
3
|
+
* one level of the bit-parallel multi-source breadth-first search. The batch's state is one `bits` buffer of four
|
|
4
|
+
* regions of `P.bitsBase` words each -- `visited` at 0, the two frontier regions at `P.bitsBase` and `2 x P.bitsBase`
|
|
5
|
+
* (which one is the frontier and which the next swaps by the level's parity, `P.mode`, so nothing is copied between
|
|
6
|
+
* levels), `flags` at `3 x P.bitsBase` -- with bit `s` of word `v` meaning "source `s` has reached / is at / is next
|
|
7
|
+
* at `v`". The expansion is `advance-expand`'s block-mapped strip (P8-T5): each workgroup loads up to `WG` entries of
|
|
8
|
+
* the frontier LIST (the vertices any source is at, compacted from the flags by the host's `compact`), scans their
|
|
9
|
+
* degrees with the inlined Hillis-Steele scan of `bfs-contract` (this is not a twin kernel: `needs: []`), and every
|
|
10
|
+
* invocation strips the aggregate by binary search. The claim is inline: for the arc `(u, x)` the mask is the sources
|
|
11
|
+
* at `u` that have not reached `x`; `atomicOr` on `visited[x]` returns the bits this lane won (`fresh`), which go into
|
|
12
|
+
* the next region, set `flags[x]` (the flags region is part of the one atomic binding, so a plain store is a compile
|
|
13
|
+
* error), and are tallied per source in WORKGROUP memory -- one global `atomicAdd` per source per workgroup after the
|
|
14
|
+
* strip loop, never one per arc. Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan and the
|
|
15
|
+
* `workgroupUniformLoad` sit unconditionally after the guard, the strip loop is bounded by the uniform aggregate, and
|
|
16
|
+
* the flush's barrier follows it in uniform control flow. Body only (spec 3.5, D9); the text is normative: the
|
|
17
|
+
* sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
18
|
+
*/
|
|
19
|
+
export const closenessSweepWgsl = /* wgsl */ `
|
|
20
|
+
var<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan
|
|
21
|
+
var<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row
|
|
22
|
+
var<workgroup> rowOf: array<u32, WG>; // the frontier vertex of each entry (the source end of its arcs)
|
|
23
|
+
var<workgroup> local: array<atomic<u32>, 32>; // this workgroup's fresh claims per source
|
|
24
|
+
var<workgroup> wcount: u32; // the frontier list's length
|
|
25
|
+
|
|
26
|
+
@compute @workgroup_size(WG)
|
|
27
|
+
fn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
28
|
+
if (lid.x == 0u) { wcount = atomicLoad(&counters[0]); } // the frontier list's length (compact's total)
|
|
29
|
+
let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
|
|
30
|
+
let nextBase = select(2u * P.bitsBase, P.bitsBase, P.mode == 1u); // the region that is next this level
|
|
31
|
+
let frontierBase = 3u * P.bitsBase - nextBase; // the other one: the region that is the frontier
|
|
32
|
+
for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
|
|
33
|
+
let i = b0 + lid.x; // this lane's frontier entry
|
|
34
|
+
var deg = 0u;
|
|
35
|
+
var start = 0u;
|
|
36
|
+
var u = 0u;
|
|
37
|
+
if (i < count) { // guarded loads into locals (3.5 rule 1)
|
|
38
|
+
u = frontierList[i];
|
|
39
|
+
let lo = max(rowPtr[u], P.arcBase);
|
|
40
|
+
let hi = min(rowPtr[u + 1u], P.arcEnd);
|
|
41
|
+
start = lo;
|
|
42
|
+
deg = select(0u, hi - lo, hi > lo);
|
|
43
|
+
}
|
|
44
|
+
if (lid.x < 32u) { atomicStore(&local[lid.x], 0u); } // zeroed before the strip loop (WebGPU zero-initialises workgroup memory; said anyway)
|
|
45
|
+
sh[lid.x] = deg;
|
|
46
|
+
rowStart[lid.x] = start;
|
|
47
|
+
rowOf[lid.x] = u;
|
|
48
|
+
workgroupBarrier();
|
|
49
|
+
for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees (bfs-contract's, inlined)
|
|
50
|
+
var t = 0u;
|
|
51
|
+
if (lid.x >= s) { t = sh[lid.x - s]; }
|
|
52
|
+
workgroupBarrier();
|
|
53
|
+
sh[lid.x] = sh[lid.x] + t;
|
|
54
|
+
workgroupBarrier();
|
|
55
|
+
}
|
|
56
|
+
let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier
|
|
57
|
+
for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
|
|
58
|
+
var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p
|
|
59
|
+
var hi = WG;
|
|
60
|
+
loop {
|
|
61
|
+
if (lo >= hi) { break; }
|
|
62
|
+
let mid = (lo + hi) / 2u;
|
|
63
|
+
if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }
|
|
64
|
+
}
|
|
65
|
+
let k = lo;
|
|
66
|
+
var exclusive = 0u;
|
|
67
|
+
if (k > 0u) { exclusive = sh[k - 1u]; }
|
|
68
|
+
let arc = rowStart[k] + (p - exclusive);
|
|
69
|
+
let x = colIdx[arc - P.arcBase];
|
|
70
|
+
let mask = atomicLoad(&bits[frontierBase + rowOf[k]]) & ~atomicLoad(&bits[x]); // the sources at u that have not reached x
|
|
71
|
+
if (mask != 0u) {
|
|
72
|
+
let old = atomicOr(&bits[x], mask); // visited: the claim, one read-modify-write
|
|
73
|
+
let fresh = mask & ~old; // the sources whose claim this lane won
|
|
74
|
+
if (fresh != 0u) {
|
|
75
|
+
atomicOr(&bits[nextBase + x], fresh);
|
|
76
|
+
atomicStore(&bits[3u * P.bitsBase + x], 1u); // flags: x is in the next frontier list (compact reads it)
|
|
77
|
+
var b = fresh;
|
|
78
|
+
loop { // one tally per set bit of fresh
|
|
79
|
+
if (b == 0u) { break; }
|
|
80
|
+
let s = firstTrailingBit(b);
|
|
81
|
+
atomicAdd(&local[s], 1u);
|
|
82
|
+
b = b & (b - 1u);
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
workgroupBarrier(); // uniform: the loop's bound is the uniform aggregate
|
|
88
|
+
if (lid.x < 32u) { // ONE global atomic per source per workgroup
|
|
89
|
+
let c = atomicLoad(&local[lid.x]);
|
|
90
|
+
if (c != 0u) { atomicAdd(&perSource[lid.x], c); } // newCount[s]
|
|
91
|
+
}
|
|
92
|
+
workgroupBarrier(); // sh, rowStart, rowOf and local are reused by the next block
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
`;
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `compact-scatter` kernel body (spec 6 row 4; P8-T3): the scatter step of `compact`, after the caller's flags
|
|
3
|
+
* have been exclusive-scanned into `offsets`. Every flagged entry lands at its offset, in queue order, so the output
|
|
4
|
+
* is bitwise reproducible; lane 0 writes the total -- the last offset plus the last flag -- into `outCount[P.outIndex]`
|
|
5
|
+
* (the block is bound whole and indexed because a four-byte word is never 256-aligned). The planner never dispatches
|
|
6
|
+
* this body for count 0, so `P.count - 1u` never wraps. Body only (spec 3.5, D9); normative text.
|
|
7
|
+
*/
|
|
8
|
+
export const compactScatterWgsl = /* wgsl */ `
|
|
9
|
+
@compute @workgroup_size(WG)
|
|
10
|
+
fn compact_scatter(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
11
|
+
let i = linear_id(wid, lid.x);
|
|
12
|
+
if (i == 0u) { outCount[P.outIndex] = offsets[P.count - 1u] + flags[P.count - 1u]; } // the exclusive scan's total; the planner never dispatches for count 0
|
|
13
|
+
if (i >= P.count) { return; } // no barrier follows
|
|
14
|
+
if (flags[i] != 0u) { out[offsets[i]] = queue[i]; }
|
|
15
|
+
}
|
|
16
|
+
`;
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `dedupe-claim` kernel body (spec 6 row 4; P8-T3): the first of `dedupe`'s two dispatches. Every entry stores
|
|
3
|
+
* its own index into `owner[queue[i]]` (relaxed atomics: between two dispatches last-writer-wins is well defined, so
|
|
4
|
+
* exactly one index per distinct vertex is the owner). The entry count is `P.count`, or the device word
|
|
5
|
+
* `counters[P.countIndex]` clamped to `P.count` when `P.countIndex` is not `U32_MAX` (the SSSP piles only know their
|
|
6
|
+
* count on the device). `owner` needs no reset between calls: a stale or garbage word is only ever read by an entry
|
|
7
|
+
* whose vertex a current entry has just overwritten. Body only (spec 3.5, D9); normative text.
|
|
8
|
+
*/
|
|
9
|
+
export const dedupeClaimWgsl = /* wgsl */ `
|
|
10
|
+
@compute @workgroup_size(WG)
|
|
11
|
+
fn dedupe_claim(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
12
|
+
var count = P.count;
|
|
13
|
+
if (P.countIndex != U32_MAX) { count = min(atomicLoad(&counters[P.countIndex]), P.count); } // a device-side count, clamped to the capacity
|
|
14
|
+
for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // grid-stride; no barrier anywhere
|
|
15
|
+
atomicStore(&owner[queue[i]], i);
|
|
16
|
+
}
|
|
17
|
+
}
|
|
18
|
+
`;
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `dedupe-filter` kernel body (spec 6 row 4; P8-T3): the second of `dedupe`'s two dispatches, a SEPARATE
|
|
3
|
+
* dispatch because a plain store read back in the same one is a data race (WGSL 6.5.7). An entry survives iff
|
|
4
|
+
* `owner[queue[i]]` still holds its own index; the survivors are packed by a Hillis-Steele inclusive scan of the keep
|
|
5
|
+
* bits in workgroup memory (scan-block's rounds and barrier placement, verbatim) and ONE `atomicAdd` per workgroup on
|
|
6
|
+
* `outCount[P.outIndex]` for the block's aggregate, so the output is set-deterministic: the surviving SET is fixed,
|
|
7
|
+
* the order inside `out` follows the schedule. The entry count is `P.count` or the device word
|
|
8
|
+
* `outCount[P.countIndex]` clamped to it -- the same block and the same word `dedupe-claim` read. Every lane reaches
|
|
9
|
+
* every barrier: the guarded work goes into locals (spec 3.5 rule 1). Body only (spec 3.5, D9); normative text.
|
|
10
|
+
*/
|
|
11
|
+
export const dedupeFilterWgsl = /* wgsl */ `
|
|
12
|
+
var<workgroup> sh: array<u32, WG>;
|
|
13
|
+
var<workgroup> base: u32;
|
|
14
|
+
var<workgroup> wcount: u32;
|
|
15
|
+
|
|
16
|
+
@compute @workgroup_size(WG)
|
|
17
|
+
fn dedupe_filter(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
18
|
+
if (lid.x == 0u) {
|
|
19
|
+
var count = P.count;
|
|
20
|
+
if (P.countIndex != U32_MAX) { count = min(atomicLoad(&outCount[P.countIndex]), P.count); } // the same count source as the claim
|
|
21
|
+
wcount = count;
|
|
22
|
+
}
|
|
23
|
+
let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
|
|
24
|
+
for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
|
|
25
|
+
let i = b0 + lid.x;
|
|
26
|
+
var keep = 0u;
|
|
27
|
+
var v = 0u;
|
|
28
|
+
if (i < count) { v = queue[i]; keep = select(0u, 1u, atomicLoad(&owner[v]) == i); } // guarded work into locals
|
|
29
|
+
sh[lid.x] = keep;
|
|
30
|
+
workgroupBarrier(); // every lane, unconditionally (3.5 rule 1)
|
|
31
|
+
for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of keep
|
|
32
|
+
var t = 0u;
|
|
33
|
+
if (lid.x >= s) { t = sh[lid.x - s]; }
|
|
34
|
+
workgroupBarrier();
|
|
35
|
+
sh[lid.x] = sh[lid.x] + t;
|
|
36
|
+
workgroupBarrier();
|
|
37
|
+
}
|
|
38
|
+
let inclusive = sh[lid.x];
|
|
39
|
+
if (lid.x == WG - 1u) { base = atomicAdd(&outCount[P.outIndex], inclusive); } // ONE atomic per workgroup: the block's aggregate
|
|
40
|
+
workgroupBarrier();
|
|
41
|
+
if (keep == 1u) { out[base + inclusive - 1u] = v; }
|
|
42
|
+
workgroupBarrier(); // sh and base are reused by the next block
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
`;
|