@graphty/webgpu-graph-algorithms 0.6.14 → 0.6.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +52 -52
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-Bi6AhScG.js → context-VIvatQOo.js} +69 -34
- package/dist/chunks/context-VIvatQOo.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +5 -3
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +101 -5
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/all-pairs.d.ts +41 -0
- package/dist/src/algorithms/all-pairs.d.ts.map +1 -0
- package/dist/src/algorithms/all-pairs.js +181 -0
- package/dist/src/algorithms/all-pairs.js.map +1 -0
- package/dist/src/algorithms/betweenness.d.ts +70 -0
- package/dist/src/algorithms/betweenness.d.ts.map +1 -0
- package/dist/src/algorithms/betweenness.js +538 -0
- package/dist/src/algorithms/betweenness.js.map +1 -0
- package/dist/src/algorithms/closeness.d.ts +15 -5
- package/dist/src/algorithms/closeness.d.ts.map +1 -1
- package/dist/src/algorithms/closeness.js +112 -26
- package/dist/src/algorithms/closeness.js.map +1 -1
- package/dist/src/algorithms/components.d.ts +9 -1
- package/dist/src/algorithms/components.d.ts.map +1 -1
- package/dist/src/algorithms/components.js +2 -2
- package/dist/src/algorithms/components.js.map +1 -1
- package/dist/src/algorithms/label-propagation.d.ts +31 -0
- package/dist/src/algorithms/label-propagation.d.ts.map +1 -0
- package/dist/src/algorithms/label-propagation.js +254 -0
- package/dist/src/algorithms/label-propagation.js.map +1 -0
- package/dist/src/algorithms/simple-symmetric.d.ts +88 -0
- package/dist/src/algorithms/simple-symmetric.d.ts.map +1 -0
- package/dist/src/algorithms/simple-symmetric.js +347 -0
- package/dist/src/algorithms/simple-symmetric.js.map +1 -0
- package/dist/src/algorithms/triangles.d.ts +34 -0
- package/dist/src/algorithms/triangles.d.ts.map +1 -0
- package/dist/src/algorithms/triangles.js +203 -0
- package/dist/src/algorithms/triangles.js.map +1 -0
- package/dist/src/constants.d.ts +53 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +53 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/index.d.ts +12 -3
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +8 -1
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +4 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernels.d.ts +24 -6
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +373 -7
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/memory/residency.js +15 -4
- package/dist/src/memory/residency.js.map +1 -1
- package/dist/src/primitives/coo-to-csr.d.ts +73 -0
- package/dist/src/primitives/coo-to-csr.d.ts.map +1 -0
- package/dist/src/primitives/coo-to-csr.js +183 -0
- package/dist/src/primitives/coo-to-csr.js.map +1 -0
- package/dist/src/primitives/frontier.d.ts +2 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -1
- package/dist/src/primitives/frontier.js +2 -0
- package/dist/src/primitives/frontier.js.map +1 -1
- package/dist/src/primitives/group-by-key.d.ts +82 -0
- package/dist/src/primitives/group-by-key.d.ts.map +1 -0
- package/dist/src/primitives/group-by-key.js +147 -0
- package/dist/src/primitives/group-by-key.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +19 -7
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/algorithms.d.ts +4 -0
- package/dist/src/types/algorithms.d.ts.map +1 -1
- package/dist/src/types/all-pairs.d.ts +35 -0
- package/dist/src/types/all-pairs.d.ts.map +1 -0
- package/dist/src/types/all-pairs.js +8 -0
- package/dist/src/types/all-pairs.js.map +1 -0
- package/dist/src/types/betweenness.d.ts +35 -0
- package/dist/src/types/betweenness.d.ts.map +1 -0
- package/dist/src/types/betweenness.js +7 -0
- package/dist/src/types/betweenness.js.map +1 -0
- package/dist/src/types/community.d.ts +18 -0
- package/dist/src/types/community.d.ts.map +1 -0
- package/dist/src/types/community.js +5 -0
- package/dist/src/types/community.js.map +1 -0
- package/dist/src/types/structure.d.ts +27 -0
- package/dist/src/types/structure.d.ts.map +1 -0
- package/dist/src/types/structure.js +8 -0
- package/dist/src/types/structure.js.map +1 -0
- package/dist/src/wgsl/apsp-fw.wgsl.d.ts +25 -0
- package/dist/src/wgsl/apsp-fw.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/apsp-fw.wgsl.js +113 -0
- package/dist/src/wgsl/apsp-fw.wgsl.js.map +1 -0
- package/dist/src/wgsl/apsp-init.wgsl.d.ts +12 -0
- package/dist/src/wgsl/apsp-init.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/apsp-init.wgsl.js +26 -0
- package/dist/src/wgsl/apsp-init.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-backward.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bc-backward.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-backward.wgsl.js +34 -0
- package/dist/src/wgsl/bc-backward.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts +12 -0
- package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-edge-gather.wgsl.js +36 -0
- package/dist/src/wgsl/bc-edge-gather.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-finalize.wgsl.d.ts +21 -0
- package/dist/src/wgsl/bc-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-finalize.wgsl.js +47 -0
- package/dist/src/wgsl/bc-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-forward-edge.wgsl.js +76 -0
- package/dist/src/wgsl/bc-forward-edge.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-forward.wgsl.d.ts +23 -0
- package/dist/src/wgsl/bc-forward.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-forward.wgsl.js +106 -0
- package/dist/src/wgsl/bc-forward.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-gather.wgsl.d.ts +9 -0
- package/dist/src/wgsl/bc-gather.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-gather.wgsl.js +20 -0
- package/dist/src/wgsl/bc-gather.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +4 -1
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/closeness-reduce.wgsl.js +8 -4
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -1
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +4 -2
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/closeness-sweep.wgsl.js +12 -2
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -1
- package/dist/src/wgsl/coo-emit.wgsl.d.ts +10 -0
- package/dist/src/wgsl/coo-emit.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/coo-emit.wgsl.js +33 -0
- package/dist/src/wgsl/coo-emit.wgsl.js.map +1 -0
- package/dist/src/wgsl/coo-scatter.wgsl.d.ts +15 -0
- package/dist/src/wgsl/coo-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/coo-scatter.wgsl.js +32 -0
- package/dist/src/wgsl/coo-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/group-by-key-row.wgsl.d.ts +26 -0
- package/dist/src/wgsl/group-by-key-row.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/group-by-key-row.wgsl.js +146 -0
- package/dist/src/wgsl/group-by-key-row.wgsl.js.map +1 -0
- package/dist/src/wgsl/lpa-step.wgsl.d.ts +10 -0
- package/dist/src/wgsl/lpa-step.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/lpa-step.wgsl.js +35 -0
- package/dist/src/wgsl/lpa-step.wgsl.js.map +1 -0
- package/dist/src/wgsl/orient-flags.wgsl.d.ts +9 -0
- package/dist/src/wgsl/orient-flags.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/orient-flags.wgsl.js +21 -0
- package/dist/src/wgsl/orient-flags.wgsl.js.map +1 -0
- package/dist/src/wgsl/run-flags.wgsl.d.ts +8 -0
- package/dist/src/wgsl/run-flags.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/run-flags.wgsl.js +18 -0
- package/dist/src/wgsl/run-flags.wgsl.js.map +1 -0
- package/dist/src/wgsl/tri-intersect.wgsl.d.ts +11 -0
- package/dist/src/wgsl/tri-intersect.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/tri-intersect.wgsl.js +64 -0
- package/dist/src/wgsl/tri-intersect.wgsl.js.map +1 -0
- package/dist/webgpu-graph-algorithms.js +2828 -321
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +5 -5
- package/src/accelerator.ts +130 -7
- package/src/algorithms/all-pairs.ts +228 -0
- package/src/algorithms/betweenness.ts +739 -0
- package/src/algorithms/closeness.ts +124 -32
- package/src/algorithms/components.ts +2 -2
- package/src/algorithms/label-propagation.ts +280 -0
- package/src/algorithms/simple-symmetric.ts +409 -0
- package/src/algorithms/triangles.ts +240 -0
- package/src/constants.ts +53 -0
- package/src/index.ts +20 -1
- package/src/kernel/prelude.ts +6 -0
- package/src/kernels.ts +411 -10
- package/src/memory/residency.ts +15 -4
- package/src/primitives/coo-to-csr.ts +251 -0
- package/src/primitives/frontier.ts +4 -0
- package/src/primitives/group-by-key.ts +209 -0
- package/src/types/accelerator.ts +26 -6
- package/src/types/algorithms.ts +5 -0
- package/src/types/all-pairs.ts +37 -0
- package/src/types/betweenness.ts +38 -0
- package/src/types/community.ts +18 -0
- package/src/types/structure.ts +28 -0
- package/src/wgsl/apsp-fw.wgsl.ts +112 -0
- package/src/wgsl/apsp-init.wgsl.ts +25 -0
- package/src/wgsl/bc-backward.wgsl.ts +33 -0
- package/src/wgsl/bc-edge-gather.wgsl.ts +35 -0
- package/src/wgsl/bc-finalize.wgsl.ts +46 -0
- package/src/wgsl/bc-forward-edge.wgsl.ts +75 -0
- package/src/wgsl/bc-forward.wgsl.ts +105 -0
- package/src/wgsl/bc-gather.wgsl.ts +19 -0
- package/src/wgsl/closeness-reduce.wgsl.ts +8 -4
- package/src/wgsl/closeness-sweep.wgsl.ts +12 -2
- package/src/wgsl/coo-emit.wgsl.ts +32 -0
- package/src/wgsl/coo-scatter.wgsl.ts +31 -0
- package/src/wgsl/group-by-key-row.wgsl.ts +145 -0
- package/src/wgsl/lpa-step.wgsl.ts +34 -0
- package/src/wgsl/orient-flags.wgsl.ts +20 -0
- package/src/wgsl/run-flags.wgsl.ts +17 -0
- package/src/wgsl/tri-intersect.wgsl.ts +63 -0
- package/dist/chunks/context-Bi6AhScG.js.map +0 -1
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bc-forward-edge` kernel body (design 8.4 "the McLaughlin-Bader online switch to the edge-parallel form", 8.8
|
|
3
|
+
* row 7): one level of a betweenness batch's forward pass, edge-parallel -- every logical edge of the `edgeList` view
|
|
4
|
+
* for every source of the batch, grid-striding over the edges inside a loop over the k sources. For the edge `(u, x)`
|
|
5
|
+
* and source `s`, when `depth[s][u]` is the level the edge is relaxed toward `x` with EXACTLY the claim and the count
|
|
6
|
+
* of `bc-forward` (the pre-check, `atomicMin`, the winner appends `s * n + x` to the same claim log, every arc on a
|
|
7
|
+
* shortest path adds `sigma[s][u]`, a u32 wrap raises word 27); `UNDIRECTED` relaxes the other direction too, the
|
|
8
|
+
* edge list holding each undirected edge once. So the two forward bodies are interchangeable level by level and the
|
|
9
|
+
* backward pass cannot tell which ran. A workgroup does nothing when the level is empty (the done boundary). The
|
|
10
|
+
* winners of a strip are packed into the log with one global `atomicAdd` per strip; every barrier is in uniform
|
|
11
|
+
* control flow (the loop bounds are uniforms and the workgroup id). Body only; the sabotage rows of
|
|
12
|
+
* test/helpers/sabotage.ts are textual edits of it.
|
|
13
|
+
*/
|
|
14
|
+
export declare const bcForwardEdgeWgsl = "\nvar<workgroup> wlive: u32; // 1 when the level has entries\nvar<workgroup> wwon: atomic<u32>; // the strip's winners\nvar<workgroup> wbase: u32; // where the strip's winners go in the log\n\nfn claim(x: u32, next: u32) -> bool {\n if (atomicLoad(&depthK[x]) != INVALID_INDEX) { return false; } // the pre-check of design 16.1\n return atomicMin(&depthK[x], next) == INVALID_INDEX;\n}\n\nfn count_paths(origin: u32, x: u32, next: u32) {\n if (atomicLoad(&depthK[x]) == next) { // every arc on a shortest path adds\n let add = atomicLoad(&sigmaK[origin]);\n let old = atomicAdd(&sigmaK[x], add);\n if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported\n }\n}\n\n@compute @workgroup_size(WG)\nfn bc_forward_edge(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let level = atomicLoad(&counters[11]);\n if (lid.x == 0u) { wlive = select(0u, 1u, ends[level + 1u] > ends[level]); }\n if (workgroupUniformLoad(&wlive) == 0u) { return; } // uniform: nothing below runs on an empty level\n let next = level + 1u;\n for (var s = 0u; s < P.k; s = s + 1u) {\n let base = s * P.n;\n for (var e0 = group_id(wid) * WG; e0 < P.count; e0 = e0 + P.stride) { // grid-stride over the edges\n let e = e0 + lid.x;\n var a = INVALID_INDEX; // the claims this lane won\n var b = INVALID_INDEX;\n if (e < P.count) {\n let u = base + edgeSrc[e];\n let x = base + edgeDst[e];\n if (atomicLoad(&depthK[u]) == level) {\n if (claim(x, next)) { a = x; }\n count_paths(u, x, next);\n }\n if (UNDIRECTED) { // the other direction of an undirected edge\n if (atomicLoad(&depthK[x]) == level) {\n if (claim(u, next)) { b = u; }\n count_paths(x, u, next);\n }\n }\n }\n let mine = select(0u, 1u, a != INVALID_INDEX) + select(0u, 1u, b != INVALID_INDEX);\n var slot = 0u;\n if (mine != 0u) { slot = atomicAdd(&wwon, mine); }\n workgroupBarrier();\n if (lid.x == 0u) {\n wbase = atomicAdd(&counters[26], atomicLoad(&wwon)); // stackTop: one global atomic per strip\n atomicStore(&wwon, 0u);\n }\n workgroupBarrier();\n if (a != INVALID_INDEX) {\n S[wbase + slot] = a;\n slot = slot + 1u;\n }\n if (b != INVALID_INDEX) { S[wbase + slot] = b; }\n }\n }\n}\n";
|
|
15
|
+
//# sourceMappingURL=bc-forward-edge.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"bc-forward-edge.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-forward-edge.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AACH,eAAO,MAAM,iBAAiB,m0FA6D7B,CAAC"}
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bc-forward-edge` kernel body (design 8.4 "the McLaughlin-Bader online switch to the edge-parallel form", 8.8
|
|
3
|
+
* row 7): one level of a betweenness batch's forward pass, edge-parallel -- every logical edge of the `edgeList` view
|
|
4
|
+
* for every source of the batch, grid-striding over the edges inside a loop over the k sources. For the edge `(u, x)`
|
|
5
|
+
* and source `s`, when `depth[s][u]` is the level the edge is relaxed toward `x` with EXACTLY the claim and the count
|
|
6
|
+
* of `bc-forward` (the pre-check, `atomicMin`, the winner appends `s * n + x` to the same claim log, every arc on a
|
|
7
|
+
* shortest path adds `sigma[s][u]`, a u32 wrap raises word 27); `UNDIRECTED` relaxes the other direction too, the
|
|
8
|
+
* edge list holding each undirected edge once. So the two forward bodies are interchangeable level by level and the
|
|
9
|
+
* backward pass cannot tell which ran. A workgroup does nothing when the level is empty (the done boundary). The
|
|
10
|
+
* winners of a strip are packed into the log with one global `atomicAdd` per strip; every barrier is in uniform
|
|
11
|
+
* control flow (the loop bounds are uniforms and the workgroup id). Body only; the sabotage rows of
|
|
12
|
+
* test/helpers/sabotage.ts are textual edits of it.
|
|
13
|
+
*/
|
|
14
|
+
export const bcForwardEdgeWgsl = /* wgsl */ `
|
|
15
|
+
var<workgroup> wlive: u32; // 1 when the level has entries
|
|
16
|
+
var<workgroup> wwon: atomic<u32>; // the strip's winners
|
|
17
|
+
var<workgroup> wbase: u32; // where the strip's winners go in the log
|
|
18
|
+
|
|
19
|
+
fn claim(x: u32, next: u32) -> bool {
|
|
20
|
+
if (atomicLoad(&depthK[x]) != INVALID_INDEX) { return false; } // the pre-check of design 16.1
|
|
21
|
+
return atomicMin(&depthK[x], next) == INVALID_INDEX;
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
fn count_paths(origin: u32, x: u32, next: u32) {
|
|
25
|
+
if (atomicLoad(&depthK[x]) == next) { // every arc on a shortest path adds
|
|
26
|
+
let add = atomicLoad(&sigmaK[origin]);
|
|
27
|
+
let old = atomicAdd(&sigmaK[x], add);
|
|
28
|
+
if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
@compute @workgroup_size(WG)
|
|
33
|
+
fn bc_forward_edge(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
34
|
+
let level = atomicLoad(&counters[11]);
|
|
35
|
+
if (lid.x == 0u) { wlive = select(0u, 1u, ends[level + 1u] > ends[level]); }
|
|
36
|
+
if (workgroupUniformLoad(&wlive) == 0u) { return; } // uniform: nothing below runs on an empty level
|
|
37
|
+
let next = level + 1u;
|
|
38
|
+
for (var s = 0u; s < P.k; s = s + 1u) {
|
|
39
|
+
let base = s * P.n;
|
|
40
|
+
for (var e0 = group_id(wid) * WG; e0 < P.count; e0 = e0 + P.stride) { // grid-stride over the edges
|
|
41
|
+
let e = e0 + lid.x;
|
|
42
|
+
var a = INVALID_INDEX; // the claims this lane won
|
|
43
|
+
var b = INVALID_INDEX;
|
|
44
|
+
if (e < P.count) {
|
|
45
|
+
let u = base + edgeSrc[e];
|
|
46
|
+
let x = base + edgeDst[e];
|
|
47
|
+
if (atomicLoad(&depthK[u]) == level) {
|
|
48
|
+
if (claim(x, next)) { a = x; }
|
|
49
|
+
count_paths(u, x, next);
|
|
50
|
+
}
|
|
51
|
+
if (UNDIRECTED) { // the other direction of an undirected edge
|
|
52
|
+
if (atomicLoad(&depthK[x]) == level) {
|
|
53
|
+
if (claim(u, next)) { b = u; }
|
|
54
|
+
count_paths(x, u, next);
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
let mine = select(0u, 1u, a != INVALID_INDEX) + select(0u, 1u, b != INVALID_INDEX);
|
|
59
|
+
var slot = 0u;
|
|
60
|
+
if (mine != 0u) { slot = atomicAdd(&wwon, mine); }
|
|
61
|
+
workgroupBarrier();
|
|
62
|
+
if (lid.x == 0u) {
|
|
63
|
+
wbase = atomicAdd(&counters[26], atomicLoad(&wwon)); // stackTop: one global atomic per strip
|
|
64
|
+
atomicStore(&wwon, 0u);
|
|
65
|
+
}
|
|
66
|
+
workgroupBarrier();
|
|
67
|
+
if (a != INVALID_INDEX) {
|
|
68
|
+
S[wbase + slot] = a;
|
|
69
|
+
slot = slot + 1u;
|
|
70
|
+
}
|
|
71
|
+
if (b != INVALID_INDEX) { S[wbase + slot] = b; }
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
`;
|
|
76
|
+
//# sourceMappingURL=bc-forward-edge.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"bc-forward-edge.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-forward-edge.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AACH,MAAM,CAAC,MAAM,iBAAiB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA6D3C,CAAC"}
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bc-forward` kernel body (design 8.4 "forward pass = BFS with sigma as array<atomic<u32>>", 8.10 "BC forward
|
|
3
|
+
* (tagged)", 16.1): one level of the tagged multi-source breadth-first search of a betweenness batch. The level's
|
|
4
|
+
* frontier is the range `S[ends[level] .. ends[level + 1])` of the claim log, every entry a packed `s * n + u`; the
|
|
5
|
+
* expansion is `closeness-sweep`'s block-mapped strip (each workgroup loads up to `WG` entries, scans their degrees
|
|
6
|
+
* in workgroup memory, and every lane strips the aggregate by an upper-bound binary search), fused with the claim, so
|
|
7
|
+
* no edge queue exists.
|
|
8
|
+
*
|
|
9
|
+
* For the arc `(u, x)` of the entry `(u, s)`, with `t = s * n + x`: the CLAIM -- a relaxed pre-check that skips the
|
|
10
|
+
* atomic when `t` is already claimed (design 16.1: a stale "unclaimed" costs one redundant atomic, a stale "claimed"
|
|
11
|
+
* cannot happen because a claim is never revoked), then `atomicMin(&depthK[t], level + 1)`, the invocation that
|
|
12
|
+
* observes INVALID_INDEX the unique winner, which appends `t` to the log; and, as a SEPARATE condition, the COUNT:
|
|
13
|
+
* every arc that reaches `t` at `level + 1` adds `sigma[s][u]` into `sigma[s][x]`, winner or not, which is what makes
|
|
14
|
+
* sigma the number of shortest paths rather than of claims. The add detects a u32 wrap from `atomicAdd`'s return
|
|
15
|
+
* value (`old + add < old`) and raises `sigmaOverflow` (counters word 27); the count is never clamped. The winners of
|
|
16
|
+
* a strip are packed into the log with one workgroup-memory counter and ONE global `atomicAdd` on `stackTop` (word
|
|
17
|
+
* 26) per strip. Uniformity (spec 3.5 rule 1): the range and the aggregate are `workgroupUniformLoad`s and the strip
|
|
18
|
+
* loop steps a uniform `p0`, so every barrier is in uniform control flow. The order of the log inside a level is
|
|
19
|
+
* not deterministic; nothing downstream depends on it (the counts are integers and every dependency is written by
|
|
20
|
+
* index). Body only; the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
21
|
+
*/
|
|
22
|
+
export declare const bcForwardWgsl = "\nvar<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan\nvar<workgroup> rowStart: array<u32, WG>; // the first arc of each entry's row\nvar<workgroup> entryOf: array<u32, WG>; // each entry, s * n + u\nvar<workgroup> wstart: u32; // the level's first log index\nvar<workgroup> wcount: u32; // the level's entry count\nvar<workgroup> wwon: atomic<u32>; // the strip's winners\nvar<workgroup> wbase: u32; // where the strip's winners go in the log\n\n@compute @workgroup_size(WG)\nfn bc_forward(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let level = atomicLoad(&counters[11]);\n if (lid.x == 0u) {\n let lo = ends[level];\n wstart = lo;\n wcount = ends[level + 1u] - lo;\n }\n let start = workgroupUniformLoad(&wstart);\n let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers\n let next = level + 1u;\n for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries\n let i = b0 + lid.x;\n var deg = 0u;\n var first = 0u;\n var entry = 0u;\n if (i < count) { // guarded loads into locals (3.5 rule 1)\n entry = S[start + i];\n let u = entry % P.n;\n first = rowPtr[u];\n deg = rowPtr[u + 1u] - first;\n }\n sh[lid.x] = deg;\n rowStart[lid.x] = first;\n entryOf[lid.x] = entry;\n workgroupBarrier();\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier\n for (var p0 = 0u; p0 < aggregate; p0 = p0 + WG) { // strip [0, aggregate) WG arcs at a time\n let p = p0 + lid.x;\n var won = false;\n var claimed = 0u;\n if (p < aggregate) {\n var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p\n var hi = WG;\n loop {\n if (lo >= hi) { break; }\n let mid = (lo + hi) / 2u;\n if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }\n }\n let k = lo;\n var exclusive = 0u;\n if (k > 0u) { exclusive = sh[k - 1u]; }\n let origin = entryOf[k]; // s * n + u\n let x = (origin - (origin % P.n)) + colIdx[rowStart[k] + (p - exclusive)]; // s * n + x\n if (atomicLoad(&depthK[x]) == INVALID_INDEX) { // the pre-check of design 16.1\n won = atomicMin(&depthK[x], next) == INVALID_INDEX; // the claim: the one winner appends\n }\n if (atomicLoad(&depthK[x]) == next) { // the count: EVERY arc on a shortest path adds\n let add = atomicLoad(&sigmaK[origin]);\n let old = atomicAdd(&sigmaK[x], add);\n if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported\n }\n claimed = x;\n }\n var slot = 0u;\n if (won) { slot = atomicAdd(&wwon, 1u); }\n workgroupBarrier();\n if (lid.x == 0u) {\n wbase = atomicAdd(&counters[26], atomicLoad(&wwon)); // stackTop: one global atomic per strip\n atomicStore(&wwon, 0u);\n }\n workgroupBarrier();\n if (won) { S[wbase + slot] = claimed; }\n }\n workgroupBarrier(); // sh, rowStart and entryOf are reused by the next block\n }\n}\n";
|
|
23
|
+
//# sourceMappingURL=bc-forward.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"bc-forward.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-forward.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,eAAO,MAAM,aAAa,+nIAmFzB,CAAC"}
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bc-forward` kernel body (design 8.4 "forward pass = BFS with sigma as array<atomic<u32>>", 8.10 "BC forward
|
|
3
|
+
* (tagged)", 16.1): one level of the tagged multi-source breadth-first search of a betweenness batch. The level's
|
|
4
|
+
* frontier is the range `S[ends[level] .. ends[level + 1])` of the claim log, every entry a packed `s * n + u`; the
|
|
5
|
+
* expansion is `closeness-sweep`'s block-mapped strip (each workgroup loads up to `WG` entries, scans their degrees
|
|
6
|
+
* in workgroup memory, and every lane strips the aggregate by an upper-bound binary search), fused with the claim, so
|
|
7
|
+
* no edge queue exists.
|
|
8
|
+
*
|
|
9
|
+
* For the arc `(u, x)` of the entry `(u, s)`, with `t = s * n + x`: the CLAIM -- a relaxed pre-check that skips the
|
|
10
|
+
* atomic when `t` is already claimed (design 16.1: a stale "unclaimed" costs one redundant atomic, a stale "claimed"
|
|
11
|
+
* cannot happen because a claim is never revoked), then `atomicMin(&depthK[t], level + 1)`, the invocation that
|
|
12
|
+
* observes INVALID_INDEX the unique winner, which appends `t` to the log; and, as a SEPARATE condition, the COUNT:
|
|
13
|
+
* every arc that reaches `t` at `level + 1` adds `sigma[s][u]` into `sigma[s][x]`, winner or not, which is what makes
|
|
14
|
+
* sigma the number of shortest paths rather than of claims. The add detects a u32 wrap from `atomicAdd`'s return
|
|
15
|
+
* value (`old + add < old`) and raises `sigmaOverflow` (counters word 27); the count is never clamped. The winners of
|
|
16
|
+
* a strip are packed into the log with one workgroup-memory counter and ONE global `atomicAdd` on `stackTop` (word
|
|
17
|
+
* 26) per strip. Uniformity (spec 3.5 rule 1): the range and the aggregate are `workgroupUniformLoad`s and the strip
|
|
18
|
+
* loop steps a uniform `p0`, so every barrier is in uniform control flow. The order of the log inside a level is
|
|
19
|
+
* not deterministic; nothing downstream depends on it (the counts are integers and every dependency is written by
|
|
20
|
+
* index). Body only; the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
21
|
+
*/
|
|
22
|
+
export const bcForwardWgsl = /* wgsl */ `
|
|
23
|
+
var<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan
|
|
24
|
+
var<workgroup> rowStart: array<u32, WG>; // the first arc of each entry's row
|
|
25
|
+
var<workgroup> entryOf: array<u32, WG>; // each entry, s * n + u
|
|
26
|
+
var<workgroup> wstart: u32; // the level's first log index
|
|
27
|
+
var<workgroup> wcount: u32; // the level's entry count
|
|
28
|
+
var<workgroup> wwon: atomic<u32>; // the strip's winners
|
|
29
|
+
var<workgroup> wbase: u32; // where the strip's winners go in the log
|
|
30
|
+
|
|
31
|
+
@compute @workgroup_size(WG)
|
|
32
|
+
fn bc_forward(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
33
|
+
let level = atomicLoad(&counters[11]);
|
|
34
|
+
if (lid.x == 0u) {
|
|
35
|
+
let lo = ends[level];
|
|
36
|
+
wstart = lo;
|
|
37
|
+
wcount = ends[level + 1u] - lo;
|
|
38
|
+
}
|
|
39
|
+
let start = workgroupUniformLoad(&wstart);
|
|
40
|
+
let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
|
|
41
|
+
let next = level + 1u;
|
|
42
|
+
for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
|
|
43
|
+
let i = b0 + lid.x;
|
|
44
|
+
var deg = 0u;
|
|
45
|
+
var first = 0u;
|
|
46
|
+
var entry = 0u;
|
|
47
|
+
if (i < count) { // guarded loads into locals (3.5 rule 1)
|
|
48
|
+
entry = S[start + i];
|
|
49
|
+
let u = entry % P.n;
|
|
50
|
+
first = rowPtr[u];
|
|
51
|
+
deg = rowPtr[u + 1u] - first;
|
|
52
|
+
}
|
|
53
|
+
sh[lid.x] = deg;
|
|
54
|
+
rowStart[lid.x] = first;
|
|
55
|
+
entryOf[lid.x] = entry;
|
|
56
|
+
workgroupBarrier();
|
|
57
|
+
for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees
|
|
58
|
+
var t = 0u;
|
|
59
|
+
if (lid.x >= s) { t = sh[lid.x - s]; }
|
|
60
|
+
workgroupBarrier();
|
|
61
|
+
sh[lid.x] = sh[lid.x] + t;
|
|
62
|
+
workgroupBarrier();
|
|
63
|
+
}
|
|
64
|
+
let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier
|
|
65
|
+
for (var p0 = 0u; p0 < aggregate; p0 = p0 + WG) { // strip [0, aggregate) WG arcs at a time
|
|
66
|
+
let p = p0 + lid.x;
|
|
67
|
+
var won = false;
|
|
68
|
+
var claimed = 0u;
|
|
69
|
+
if (p < aggregate) {
|
|
70
|
+
var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p
|
|
71
|
+
var hi = WG;
|
|
72
|
+
loop {
|
|
73
|
+
if (lo >= hi) { break; }
|
|
74
|
+
let mid = (lo + hi) / 2u;
|
|
75
|
+
if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }
|
|
76
|
+
}
|
|
77
|
+
let k = lo;
|
|
78
|
+
var exclusive = 0u;
|
|
79
|
+
if (k > 0u) { exclusive = sh[k - 1u]; }
|
|
80
|
+
let origin = entryOf[k]; // s * n + u
|
|
81
|
+
let x = (origin - (origin % P.n)) + colIdx[rowStart[k] + (p - exclusive)]; // s * n + x
|
|
82
|
+
if (atomicLoad(&depthK[x]) == INVALID_INDEX) { // the pre-check of design 16.1
|
|
83
|
+
won = atomicMin(&depthK[x], next) == INVALID_INDEX; // the claim: the one winner appends
|
|
84
|
+
}
|
|
85
|
+
if (atomicLoad(&depthK[x]) == next) { // the count: EVERY arc on a shortest path adds
|
|
86
|
+
let add = atomicLoad(&sigmaK[origin]);
|
|
87
|
+
let old = atomicAdd(&sigmaK[x], add);
|
|
88
|
+
if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported
|
|
89
|
+
}
|
|
90
|
+
claimed = x;
|
|
91
|
+
}
|
|
92
|
+
var slot = 0u;
|
|
93
|
+
if (won) { slot = atomicAdd(&wwon, 1u); }
|
|
94
|
+
workgroupBarrier();
|
|
95
|
+
if (lid.x == 0u) {
|
|
96
|
+
wbase = atomicAdd(&counters[26], atomicLoad(&wwon)); // stackTop: one global atomic per strip
|
|
97
|
+
atomicStore(&wwon, 0u);
|
|
98
|
+
}
|
|
99
|
+
workgroupBarrier();
|
|
100
|
+
if (won) { S[wbase + slot] = claimed; }
|
|
101
|
+
}
|
|
102
|
+
workgroupBarrier(); // sh, rowStart and entryOf are reused by the next block
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
`;
|
|
106
|
+
//# sourceMappingURL=bc-forward.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"bc-forward.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-forward.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,MAAM,CAAC,MAAM,aAAa,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAmFvC,CAAC"}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bc-gather` kernel body (design 8.4 "one gather kernel after the batch's last backward level", 8.10 "BC
|
|
3
|
+
* gather"): one invocation per vertex `w`, `bc[w] += delta[0][w] + ... + delta[k - 1][w]` in source order -- k reads,
|
|
4
|
+
* one write, no atomic, a fixed summation order, so the scores are bitwise reproducible across runs and adapters.
|
|
5
|
+
* A source's own dependency is 0 (the backward pass never visits depth 0), so nothing is skipped here. Grid-stride
|
|
6
|
+
* loop (`P.stride`). Body only; the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
7
|
+
*/
|
|
8
|
+
export declare const bcGatherWgsl = "\n@compute @workgroup_size(WG)\nfn bc_gather(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n for (var w = linear_id(wid, lid.x); w < P.n; w = w + P.stride) {\n var acc = bc[w];\n for (var s = 0u; s < P.k; s = s + 1u) {\n acc = acc + deltaK[s * P.n + w];\n }\n bc[w] = acc;\n }\n}\n";
|
|
9
|
+
//# sourceMappingURL=bc-gather.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"bc-gather.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-gather.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,eAAO,MAAM,YAAY,oXAWxB,CAAC"}
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bc-gather` kernel body (design 8.4 "one gather kernel after the batch's last backward level", 8.10 "BC
|
|
3
|
+
* gather"): one invocation per vertex `w`, `bc[w] += delta[0][w] + ... + delta[k - 1][w]` in source order -- k reads,
|
|
4
|
+
* one write, no atomic, a fixed summation order, so the scores are bitwise reproducible across runs and adapters.
|
|
5
|
+
* A source's own dependency is 0 (the backward pass never visits depth 0), so nothing is skipped here. Grid-stride
|
|
6
|
+
* loop (`P.stride`). Body only; the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
7
|
+
*/
|
|
8
|
+
export const bcGatherWgsl = /* wgsl */ `
|
|
9
|
+
@compute @workgroup_size(WG)
|
|
10
|
+
fn bc_gather(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
11
|
+
for (var w = linear_id(wid, lid.x); w < P.n; w = w + P.stride) {
|
|
12
|
+
var acc = bc[w];
|
|
13
|
+
for (var s = 0u; s < P.k; s = s + 1u) {
|
|
14
|
+
acc = acc + deltaK[s * P.n + w];
|
|
15
|
+
}
|
|
16
|
+
bc[w] = acc;
|
|
17
|
+
}
|
|
18
|
+
}
|
|
19
|
+
`;
|
|
20
|
+
//# sourceMappingURL=bc-gather.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"bc-gather.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-gather.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,MAAM,CAAC,MAAM,YAAY,GAAG,UAAU,CAAC;;;;;;;;;;;CAWtC,CAAC"}
|
|
@@ -10,8 +10,11 @@
|
|
|
10
10
|
* level 0's parity is 0), `flags[source_s] = 1` (level 0's `compact` turns the flags into the list; a seeded list
|
|
11
11
|
* would be overwritten by a compaction of all-zero flags), `counters[0] = k` (not done) and `level = U32_MAX` (so
|
|
12
12
|
* level 0's boundary accumulates nothing and brings the word to 0, and level 1's counts the distance-1 claims at 1).
|
|
13
|
+
* Role 2 is role 1 for a sampled run: source `s` of the batch is word `P.source + s` of the source list the host
|
|
14
|
+
* wrote after the per-node sums (`perSource[128 + P.bitsBase + ...]`), `P.n` is the list's length, and a node listed
|
|
15
|
+
* twice in one batch carries both bits (the seed ORs, so a duplicate runs twice).
|
|
13
16
|
* No barrier follows the early return of the other lanes (3.5 rule 1). Body only (spec 3.5, D9); the text is
|
|
14
17
|
* normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
15
18
|
*/
|
|
16
|
-
export declare const closenessReduceWgsl = "\n@compute @workgroup_size(WG)\nfn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role
|
|
19
|
+
export declare const closenessReduceWgsl = "\n@compute @workgroup_size(WG)\nfn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role != 0u) { // the seed of a batch: P.source is its first source\n let k = min(32u, P.n - P.source);\n for (var s = 0u; s < k; s = s + 1u) {\n var v = P.source + s;\n if (P.role == 2u) { v = atomicLoad(&perSource[128u + P.bitsBase + P.source + s]); } // a sampled run's list\n let bit = 1u << s;\n bits[v] = bits[v] | bit; // visited\n bits[P.bitsBase + v] = bits[P.bitsBase + v] | bit; // the frontier level 0 reads (region 1: level 0's parity is 0)\n bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list\n }\n atomicStore(&counters[0], k); // not done\n atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0\n atomicStore(&counters[15], 0u); // done\n return;\n }\n // role 0: the level boundary -- done from the previous level's compacted count, then the accumulation\n let count = atomicLoad(&counters[0]);\n atomicStore(&counters[15], select(0u, 1u, count == 0u));\n let level = atomicLoad(&counters[11]);\n let d = level + 1u; // the distance of the claims the level just run made\n for (var s = 0u; s < 32u; s = s + 1u) {\n let c = atomicLoad(&perSource[s]); // newCount[s]\n atomicStore(&perSource[32u + s], atomicLoad(&perSource[32u + s]) + c); // reached[s]\n // sum[s] += c x d in 64 bits: the 16-bit split product (pLo, pHi), then the add with its carry\n let cLo = c & 0xFFFFu;\n let cHi = c >> 16u;\n let dLo = d & 0xFFFFu;\n let dHi = d >> 16u;\n let ll = cLo * dLo;\n let lh = cLo * dHi;\n let hl = cHi * dLo;\n let mid = (ll >> 16u) + (lh & 0xFFFFu) + (hl & 0xFFFFu);\n let pLo = (ll & 0xFFFFu) | ((mid & 0xFFFFu) << 16u);\n let pHi = (cHi * dHi) + (lh >> 16u) + (hl >> 16u) + (mid >> 16u);\n var lo = atomicLoad(&perSource[64u + s]); // sumLo[s]\n var hi = atomicLoad(&perSource[96u + s]); // sumHi[s]\n let before = lo;\n lo = lo + pLo;\n hi = hi + pHi + select(0u, 1u, lo < before);\n atomicStore(&perSource[64u + s], lo);\n atomicStore(&perSource[96u + s], hi);\n atomicStore(&perSource[s], 0u);\n }\n atomicStore(&counters[11], level + 1u);\n}\n";
|
|
17
20
|
//# sourceMappingURL=closeness-reduce.wgsl.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"closeness-reduce.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-reduce.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"closeness-reduce.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-reduce.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,eAAO,MAAM,mBAAmB,qyFAiD/B,CAAC"}
|
|
@@ -10,6 +10,9 @@
|
|
|
10
10
|
* level 0's parity is 0), `flags[source_s] = 1` (level 0's `compact` turns the flags into the list; a seeded list
|
|
11
11
|
* would be overwritten by a compaction of all-zero flags), `counters[0] = k` (not done) and `level = U32_MAX` (so
|
|
12
12
|
* level 0's boundary accumulates nothing and brings the word to 0, and level 1's counts the distance-1 claims at 1).
|
|
13
|
+
* Role 2 is role 1 for a sampled run: source `s` of the batch is word `P.source + s` of the source list the host
|
|
14
|
+
* wrote after the per-node sums (`perSource[128 + P.bitsBase + ...]`), `P.n` is the list's length, and a node listed
|
|
15
|
+
* twice in one batch carries both bits (the seed ORs, so a duplicate runs twice).
|
|
13
16
|
* No barrier follows the early return of the other lanes (3.5 rule 1). Body only (spec 3.5, D9); the text is
|
|
14
17
|
* normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
15
18
|
*/
|
|
@@ -17,13 +20,14 @@ export const closenessReduceWgsl = /* wgsl */ `
|
|
|
17
20
|
@compute @workgroup_size(WG)
|
|
18
21
|
fn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
19
22
|
if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
|
|
20
|
-
if (P.role
|
|
23
|
+
if (P.role != 0u) { // the seed of a batch: P.source is its first source
|
|
21
24
|
let k = min(32u, P.n - P.source);
|
|
22
25
|
for (var s = 0u; s < k; s = s + 1u) {
|
|
23
|
-
|
|
26
|
+
var v = P.source + s;
|
|
27
|
+
if (P.role == 2u) { v = atomicLoad(&perSource[128u + P.bitsBase + P.source + s]); } // a sampled run's list
|
|
24
28
|
let bit = 1u << s;
|
|
25
|
-
bits[v] = bit;
|
|
26
|
-
bits[P.bitsBase + v] = bit;
|
|
29
|
+
bits[v] = bits[v] | bit; // visited
|
|
30
|
+
bits[P.bitsBase + v] = bits[P.bitsBase + v] | bit; // the frontier level 0 reads (region 1: level 0's parity is 0)
|
|
27
31
|
bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list
|
|
28
32
|
}
|
|
29
33
|
atomicStore(&counters[0], k); // not done
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"closeness-reduce.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/closeness-reduce.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"closeness-reduce.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/closeness-reduce.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,CAAC,MAAM,mBAAmB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAiD7C,CAAC"}
|
|
@@ -13,8 +13,10 @@
|
|
|
13
13
|
* error), and are tallied per source in WORKGROUP memory -- one global `atomicAdd` per source per workgroup after the
|
|
14
14
|
* strip loop, never one per arc. Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan and the
|
|
15
15
|
* `workgroupUniformLoad` sit unconditionally after the guard, the strip loop is bounded by the uniform aggregate, and
|
|
16
|
-
* the flush's barrier follows it in uniform control flow.
|
|
16
|
+
* the flush's barrier follows it in uniform control flow. A sampled run (`P.perNode == 1`) also adds each claim's
|
|
17
|
+
* distance (`level + 1`, one per won bit) into the per-node sum of `x`, `perSource[128 + x]`: the distance from each
|
|
18
|
+
* of the batch's sources TO `x`, which is what a node's sampled closeness sums on an undirected graph. Body only (spec 3.5, D9); the text is normative: the
|
|
17
19
|
* sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
18
20
|
*/
|
|
19
|
-
export declare const closenessSweepWgsl = "\nvar<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan\nvar<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row\nvar<workgroup> rowOf: array<u32, WG>; // the frontier vertex of each entry (the source end of its arcs)\nvar<workgroup> local: array<atomic<u32>, 32>; // this workgroup's fresh claims per source\nvar<workgroup> wcount: u32; // the frontier list's length\n\n@compute @workgroup_size(WG)\nfn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x == 0u) {
|
|
21
|
+
export declare const closenessSweepWgsl = "\nvar<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan\nvar<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row\nvar<workgroup> rowOf: array<u32, WG>; // the frontier vertex of each entry (the source end of its arcs)\nvar<workgroup> local: array<atomic<u32>, 32>; // this workgroup's fresh claims per source\nvar<workgroup> wcount: u32; // the frontier list's length\nvar<workgroup> wdist: u32; // the distance of this level's claims\n\n@compute @workgroup_size(WG)\nfn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x == 0u) {\n wcount = atomicLoad(&counters[0]); // the frontier list's length (compact's total)\n wdist = atomicLoad(&counters[11]) + 1u; // the level word: this level claims at level + 1\n }\n let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers\n let dist = workgroupUniformLoad(&wdist);\n let nextBase = select(2u * P.bitsBase, P.bitsBase, P.mode == 1u); // the region that is next this level\n let frontierBase = 3u * P.bitsBase - nextBase; // the other one: the region that is the frontier\n for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries\n let i = b0 + lid.x; // this lane's frontier entry\n var deg = 0u;\n var start = 0u;\n var u = 0u;\n if (i < count) { // guarded loads into locals (3.5 rule 1)\n u = frontierList[i];\n let lo = max(rowPtr[u], P.arcBase);\n let hi = min(rowPtr[u + 1u], P.arcEnd);\n start = lo;\n deg = select(0u, hi - lo, hi > lo);\n }\n if (lid.x < 32u) { atomicStore(&local[lid.x], 0u); } // zeroed before the strip loop (WebGPU zero-initialises workgroup memory; said anyway)\n sh[lid.x] = deg;\n rowStart[lid.x] = start;\n rowOf[lid.x] = u;\n workgroupBarrier();\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees (bfs-contract's, inlined)\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier\n for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...\n var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p\n var hi = WG;\n loop {\n if (lo >= hi) { break; }\n let mid = (lo + hi) / 2u;\n if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }\n }\n let k = lo;\n var exclusive = 0u;\n if (k > 0u) { exclusive = sh[k - 1u]; }\n let arc = rowStart[k] + (p - exclusive);\n let x = colIdx[arc - P.arcBase];\n let mask = atomicLoad(&bits[frontierBase + rowOf[k]]) & ~atomicLoad(&bits[x]); // the sources at u that have not reached x\n if (mask != 0u) {\n let old = atomicOr(&bits[x], mask); // visited: the claim, one read-modify-write\n let fresh = mask & ~old; // the sources whose claim this lane won\n if (fresh != 0u) {\n atomicOr(&bits[nextBase + x], fresh);\n atomicStore(&bits[3u * P.bitsBase + x], 1u); // flags: x is in the next frontier list (compact reads it)\n if (P.perNode == 1u) { // a sampled run: x's distance to each source won\n atomicAdd(&perSource[128u + x], countOneBits(fresh) * dist);\n }\n var b = fresh;\n loop { // one tally per set bit of fresh\n if (b == 0u) { break; }\n let s = firstTrailingBit(b);\n atomicAdd(&local[s], 1u);\n b = b & (b - 1u);\n }\n }\n }\n }\n workgroupBarrier(); // uniform: the loop's bound is the uniform aggregate\n if (lid.x < 32u) { // ONE global atomic per source per workgroup\n let c = atomicLoad(&local[lid.x]);\n if (c != 0u) { atomicAdd(&perSource[lid.x], c); } // newCount[s]\n }\n workgroupBarrier(); // sh, rowStart, rowOf and local are reused by the next block\n }\n}\n";
|
|
20
22
|
//# sourceMappingURL=closeness-sweep.wgsl.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"closeness-sweep.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-sweep.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"closeness-sweep.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-sweep.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AACH,eAAO,MAAM,kBAAkB,inKAoF9B,CAAC"}
|
|
@@ -13,7 +13,9 @@
|
|
|
13
13
|
* error), and are tallied per source in WORKGROUP memory -- one global `atomicAdd` per source per workgroup after the
|
|
14
14
|
* strip loop, never one per arc. Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan and the
|
|
15
15
|
* `workgroupUniformLoad` sit unconditionally after the guard, the strip loop is bounded by the uniform aggregate, and
|
|
16
|
-
* the flush's barrier follows it in uniform control flow.
|
|
16
|
+
* the flush's barrier follows it in uniform control flow. A sampled run (`P.perNode == 1`) also adds each claim's
|
|
17
|
+
* distance (`level + 1`, one per won bit) into the per-node sum of `x`, `perSource[128 + x]`: the distance from each
|
|
18
|
+
* of the batch's sources TO `x`, which is what a node's sampled closeness sums on an undirected graph. Body only (spec 3.5, D9); the text is normative: the
|
|
17
19
|
* sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
18
20
|
*/
|
|
19
21
|
export const closenessSweepWgsl = /* wgsl */ `
|
|
@@ -22,11 +24,16 @@ var<workgroup> rowStart: array<u32, WG>; // the first bound arc of each
|
|
|
22
24
|
var<workgroup> rowOf: array<u32, WG>; // the frontier vertex of each entry (the source end of its arcs)
|
|
23
25
|
var<workgroup> local: array<atomic<u32>, 32>; // this workgroup's fresh claims per source
|
|
24
26
|
var<workgroup> wcount: u32; // the frontier list's length
|
|
27
|
+
var<workgroup> wdist: u32; // the distance of this level's claims
|
|
25
28
|
|
|
26
29
|
@compute @workgroup_size(WG)
|
|
27
30
|
fn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
28
|
-
if (lid.x == 0u) {
|
|
31
|
+
if (lid.x == 0u) {
|
|
32
|
+
wcount = atomicLoad(&counters[0]); // the frontier list's length (compact's total)
|
|
33
|
+
wdist = atomicLoad(&counters[11]) + 1u; // the level word: this level claims at level + 1
|
|
34
|
+
}
|
|
29
35
|
let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
|
|
36
|
+
let dist = workgroupUniformLoad(&wdist);
|
|
30
37
|
let nextBase = select(2u * P.bitsBase, P.bitsBase, P.mode == 1u); // the region that is next this level
|
|
31
38
|
let frontierBase = 3u * P.bitsBase - nextBase; // the other one: the region that is the frontier
|
|
32
39
|
for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
|
|
@@ -74,6 +81,9 @@ fn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocat
|
|
|
74
81
|
if (fresh != 0u) {
|
|
75
82
|
atomicOr(&bits[nextBase + x], fresh);
|
|
76
83
|
atomicStore(&bits[3u * P.bitsBase + x], 1u); // flags: x is in the next frontier list (compact reads it)
|
|
84
|
+
if (P.perNode == 1u) { // a sampled run: x's distance to each source won
|
|
85
|
+
atomicAdd(&perSource[128u + x], countOneBits(fresh) * dist);
|
|
86
|
+
}
|
|
77
87
|
var b = fresh;
|
|
78
88
|
loop { // one tally per set bit of fresh
|
|
79
89
|
if (b == 0u) { break; }
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"closeness-sweep.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/closeness-sweep.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"closeness-sweep.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/closeness-sweep.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAoF5C,CAAC"}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `coo-emit` kernel body (design 6 row 10, 8.6; the first step of the simple symmetric graph build): position `i`
|
|
3
|
+
* takes arc `a` -- `i` itself, or `order[i]` under INDEXED, which is how a sorted permutation is materialised -- and
|
|
4
|
+
* writes that arc's source, target and weight. Arc `2e` is logical edge `e` as declared and arc `2e + 1` its reverse,
|
|
5
|
+
* so every edge lands in both directions and the result is symmetric whatever the snapshot's directedness. A
|
|
6
|
+
* self-loop is dropped: both of its arcs are written as `INVALID_INDEX`, which sorts after every node index and which
|
|
7
|
+
* `run-flags` never marks. Weights are 1 without WEIGHTED. Body only (spec 3.5, D9); normative text.
|
|
8
|
+
*/
|
|
9
|
+
export declare const cooEmitWgsl = "\n@compute @workgroup_size(WG)\nfn coo_emit(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n if (i >= P.count) { return; } // no barrier follows\n var a = i;\n if (INDEXED) { a = order[i]; } // the arc this position takes\n let e = a / 2u; // arc 2e is edge e as declared, arc 2e + 1 its reverse\n let forward = (a % 2u) == 0u;\n let u = edgeSrc[e];\n let v = edgeDst[e];\n var w = 1.0;\n if (WEIGHTED) { w = edgeWeight[e]; }\n if (u == v) { // a self-loop is dropped: both arcs sort last and never open a run\n outSrc[i] = INVALID_INDEX;\n outDst[i] = INVALID_INDEX;\n outWeight[i] = 0.0;\n return;\n }\n outSrc[i] = select(v, u, forward);\n outDst[i] = select(u, v, forward);\n outWeight[i] = w;\n}\n";
|
|
10
|
+
//# sourceMappingURL=coo-emit.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"coo-emit.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/coo-emit.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AACH,eAAO,MAAM,WAAW,s/BAuBvB,CAAC"}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `coo-emit` kernel body (design 6 row 10, 8.6; the first step of the simple symmetric graph build): position `i`
|
|
3
|
+
* takes arc `a` -- `i` itself, or `order[i]` under INDEXED, which is how a sorted permutation is materialised -- and
|
|
4
|
+
* writes that arc's source, target and weight. Arc `2e` is logical edge `e` as declared and arc `2e + 1` its reverse,
|
|
5
|
+
* so every edge lands in both directions and the result is symmetric whatever the snapshot's directedness. A
|
|
6
|
+
* self-loop is dropped: both of its arcs are written as `INVALID_INDEX`, which sorts after every node index and which
|
|
7
|
+
* `run-flags` never marks. Weights are 1 without WEIGHTED. Body only (spec 3.5, D9); normative text.
|
|
8
|
+
*/
|
|
9
|
+
export const cooEmitWgsl = /* wgsl */ `
|
|
10
|
+
@compute @workgroup_size(WG)
|
|
11
|
+
fn coo_emit(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
12
|
+
let i = linear_id(wid, lid.x);
|
|
13
|
+
if (i >= P.count) { return; } // no barrier follows
|
|
14
|
+
var a = i;
|
|
15
|
+
if (INDEXED) { a = order[i]; } // the arc this position takes
|
|
16
|
+
let e = a / 2u; // arc 2e is edge e as declared, arc 2e + 1 its reverse
|
|
17
|
+
let forward = (a % 2u) == 0u;
|
|
18
|
+
let u = edgeSrc[e];
|
|
19
|
+
let v = edgeDst[e];
|
|
20
|
+
var w = 1.0;
|
|
21
|
+
if (WEIGHTED) { w = edgeWeight[e]; }
|
|
22
|
+
if (u == v) { // a self-loop is dropped: both arcs sort last and never open a run
|
|
23
|
+
outSrc[i] = INVALID_INDEX;
|
|
24
|
+
outDst[i] = INVALID_INDEX;
|
|
25
|
+
outWeight[i] = 0.0;
|
|
26
|
+
return;
|
|
27
|
+
}
|
|
28
|
+
outSrc[i] = select(v, u, forward);
|
|
29
|
+
outDst[i] = select(u, v, forward);
|
|
30
|
+
outWeight[i] = w;
|
|
31
|
+
}
|
|
32
|
+
`;
|
|
33
|
+
//# sourceMappingURL=coo-emit.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"coo-emit.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/coo-emit.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AACH,MAAM,CAAC,MAAM,WAAW,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;CAuBrC,CAAC"}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `coo-scatter` kernel body (design 6 row 10): the last step of `cooToCsr`, after the histogram of the sources
|
|
3
|
+
* and its exclusive scan into `rowPtr`. Two modes under SORTED_INPUT.
|
|
4
|
+
*
|
|
5
|
+
* SORTED_INPUT false is the design's cursor scatter: an arc reserves its slot with `atomicAdd` on its row's cursor
|
|
6
|
+
* (`cursors`, zeroed by the caller), so slots are handed out in race order and a row is NOT sorted by target even when
|
|
7
|
+
* the input was.
|
|
8
|
+
*
|
|
9
|
+
* SORTED_INPUT true takes arcs already ordered by source: arc `i` then sits at its own index minus its row's start
|
|
10
|
+
* inside the row, no cursor and no atomic, so the write is a pure function of the input and the input order survives
|
|
11
|
+
* into every row. The precondition is checked, not trusted: an arc whose source is below its predecessor's raises
|
|
12
|
+
* `cursors[0]`, the word the driver reads back and refuses. Body only (spec 3.5, D9); normative text.
|
|
13
|
+
*/
|
|
14
|
+
export declare const cooScatterWgsl = "\n@compute @workgroup_size(WG)\nfn coo_scatter(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n if (i >= P.count) { return; } // no barrier follows\n let s = src[i];\n var slot = 0u;\n if (SORTED_INPUT) {\n if (i > 0u && s < src[i - 1u]) { atomicStore(&cursors[0], 1u); } // unsorted input: the flag the driver refuses\n let within = i - rowPtr[s]; // the arc's place in its row\n slot = rowPtr[s] + within;\n } else {\n slot = rowPtr[s] + atomicAdd(&cursors[s], 1u); // race order: the row is not sorted in this mode\n }\n colIdx[slot] = dst[i];\n if (WEIGHTED) { outWeight[slot] = weight[i]; }\n}\n";
|
|
15
|
+
//# sourceMappingURL=coo-scatter.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"coo-scatter.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/coo-scatter.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AACH,eAAO,MAAM,cAAc,2yBAiB1B,CAAC"}
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `coo-scatter` kernel body (design 6 row 10): the last step of `cooToCsr`, after the histogram of the sources
|
|
3
|
+
* and its exclusive scan into `rowPtr`. Two modes under SORTED_INPUT.
|
|
4
|
+
*
|
|
5
|
+
* SORTED_INPUT false is the design's cursor scatter: an arc reserves its slot with `atomicAdd` on its row's cursor
|
|
6
|
+
* (`cursors`, zeroed by the caller), so slots are handed out in race order and a row is NOT sorted by target even when
|
|
7
|
+
* the input was.
|
|
8
|
+
*
|
|
9
|
+
* SORTED_INPUT true takes arcs already ordered by source: arc `i` then sits at its own index minus its row's start
|
|
10
|
+
* inside the row, no cursor and no atomic, so the write is a pure function of the input and the input order survives
|
|
11
|
+
* into every row. The precondition is checked, not trusted: an arc whose source is below its predecessor's raises
|
|
12
|
+
* `cursors[0]`, the word the driver reads back and refuses. Body only (spec 3.5, D9); normative text.
|
|
13
|
+
*/
|
|
14
|
+
export const cooScatterWgsl = /* wgsl */ `
|
|
15
|
+
@compute @workgroup_size(WG)
|
|
16
|
+
fn coo_scatter(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
17
|
+
let i = linear_id(wid, lid.x);
|
|
18
|
+
if (i >= P.count) { return; } // no barrier follows
|
|
19
|
+
let s = src[i];
|
|
20
|
+
var slot = 0u;
|
|
21
|
+
if (SORTED_INPUT) {
|
|
22
|
+
if (i > 0u && s < src[i - 1u]) { atomicStore(&cursors[0], 1u); } // unsorted input: the flag the driver refuses
|
|
23
|
+
let within = i - rowPtr[s]; // the arc's place in its row
|
|
24
|
+
slot = rowPtr[s] + within;
|
|
25
|
+
} else {
|
|
26
|
+
slot = rowPtr[s] + atomicAdd(&cursors[s], 1u); // race order: the row is not sorted in this mode
|
|
27
|
+
}
|
|
28
|
+
colIdx[slot] = dst[i];
|
|
29
|
+
if (WEIGHTED) { outWeight[slot] = weight[i]; }
|
|
30
|
+
}
|
|
31
|
+
`;
|
|
32
|
+
//# sourceMappingURL=coo-scatter.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"coo-scatter.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/coo-scatter.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AACH,MAAM,CAAC,MAAM,cAAc,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;CAiBxC,CAAC"}
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `group-by-key-row` kernel body (design 8.6; the per-row group-by-key): for every listed row `v`, the arcs of
|
|
3
|
+
* the row are grouped by the key of their target (`keyIn[colIdx[a]]`), the weights are summed per key, and the key
|
|
4
|
+
* with the largest sum wins, ties going to the LOWEST key -- which makes the answer independent of the order the arcs
|
|
5
|
+
* are visited in, and so bitwise reproducible. `bestKey[v]` gets the key (`INVALID_INDEX` for an empty row) and
|
|
6
|
+
* `bestScore[v]` the summed weight.
|
|
7
|
+
*
|
|
8
|
+
* The sums are u32 fixed point, because WGSL has no float atomic: every weight is scaled by `2^s`, a power of two
|
|
9
|
+
* chosen from the exponents of the row's largest weight and of its degree alone so that `maxWeight x degree x 2^s`
|
|
10
|
+
* lies in [2^28, 2^30), and rounded to the nearest integer, halves up. No step rounds a float: the scale needs no
|
|
11
|
+
* product, scaling by a power of two is exact, and so are the integer part and the fraction of the scaled weight.
|
|
12
|
+
* That matters because WGSL lets `x + y` and `x * y` round to EITHER neighbour of an inexact result, so a rounded
|
|
13
|
+
* step could differ between devices; as written, both tiers, every device and any CPU reference compute identical
|
|
14
|
+
* integers. A weight smaller than `2^-s / 2` contributes nothing, and a negative weight counts as zero. Without
|
|
15
|
+
* WEIGHTED every weight is 1.
|
|
16
|
+
*
|
|
17
|
+
* TIER 0: one thread per row (`rows[P.rowsBase + i]`), a pairwise scan in registers -- for short rows, and at most
|
|
18
|
+
* GROUP_ROW_THREAD_LIMIT arcs, since llvmpipe stops an invocation's loops after 65,535 steps. Any other TIER: one
|
|
19
|
+
* workgroup per row (`rows[P.rowsBase + g]`) over its own open-addressing region of `GROUP_HASH_LOAD_FACTOR x degree`
|
|
20
|
+
* slot pairs (key, sum) starting at word `rows[P.basesBase + g]` of `hashRegion`, cleared by the workgroup itself;
|
|
21
|
+
* a key is claimed by a bounded compare-exchange loop with linear probing, and a lane that exhausts its bound raises
|
|
22
|
+
* `hashRegion[0]`, which the driver reads and refuses. Every barrier is reached in uniform control flow: the only
|
|
23
|
+
* early return keys on the workgroup id and a uniform. Body only (spec 3.5, D9); normative text.
|
|
24
|
+
*/
|
|
25
|
+
export declare const groupByKeyRowWgsl = "\nvar<workgroup> shMax: array<f32, WG>;\nvar<workgroup> shKey: array<u32, WG>;\nvar<workgroup> shSum: array<u32, WG>;\n\nfn weight_of(a: u32) -> f32 {\n if (WEIGHTED) { return weights[a]; }\n return 1.0;\n}\nfn scale_of(maxW: f32, d: u32) -> f32 { // 2^s, maxW x d x 2^s in [2^28, 2^30), from exponents alone\n let em = i32((bitcast<u32>(maxW) >> 23u) & 255u) - 127;\n let ed = i32(firstLeadingBit(max(d, 1u)));\n let s = clamp(28 - em - ed, -126, 126);\n return bitcast<f32>(u32(s + 127) << 23u);\n}\nfn inverse_of(scale: f32) -> f32 { // 2^-s, exact\n let s = i32((bitcast<u32>(scale) >> 23u) & 255u) - 127;\n return bitcast<f32>(u32(127 - s) << 23u);\n}\nfn quantize(w: f32, scale: f32) -> u32 { // nearest integer, halves up; every step exact\n let x = max(w * scale, 0.0);\n let i = u32(x);\n return select(i, i + 1u, x - f32(i) >= 0.5);\n}\nfn better(sum: u32, key: u32, bestSum: u32, bestKey0: u32) -> bool { return sum > bestSum || (sum == bestSum && key < bestKey0); }\n\nfn row_thread(v: u32) {\n let lo = rowPtr[v];\n let hi = rowPtr[v + 1u];\n var maxW = 0.0;\n for (var a = lo; a < hi; a = a + 1u) { maxW = max(maxW, weight_of(a)); }\n let scale = scale_of(maxW, hi - lo);\n var bk = INVALID_INDEX;\n var bs = 0u;\n for (var a = lo; a < hi; a = a + 1u) {\n let k = keyIn[colIdx[a]];\n var seen = false;\n for (var b = lo; b < a; b = b + 1u) { if (keyIn[colIdx[b]] == k) { seen = true; break; } }\n if (seen) { continue; } // this key was summed at its first arc\n var sum = 0u;\n for (var b = a; b < hi; b = b + 1u) { if (keyIn[colIdx[b]] == k) { sum = sum + quantize(weight_of(b), scale); } }\n if (better(sum, k, bs, bk)) { bk = k; bs = sum; }\n }\n bestKey[v] = bk;\n bestScore[v] = f32(bs) * inverse_of(scale);\n}\n\nfn row_hash(g: u32, lid: u32) {\n let v = rows[P.rowsBase + g];\n let base = rows[P.basesBase + g];\n let lo = rowPtr[v];\n let hi = rowPtr[v + 1u];\n let cap = GROUP_HASH_LOAD_FACTOR * (hi - lo);\n var m = 0.0;\n for (var a = lo + lid; a < hi; a = a + WG) { m = max(m, weight_of(a)); }\n shMax[lid] = m;\n workgroupBarrier();\n for (var s = WG / 2u; s > 0u; s = s / 2u) {\n if (lid < s) { shMax[lid] = max(shMax[lid], shMax[lid + s]); }\n workgroupBarrier();\n }\n let scale = scale_of(shMax[0], hi - lo);\n for (var j = lid; j < cap; j = j + WG) { // the region is this workgroup's alone: clear it\n atomicStore(&hashRegion[base + 2u * j], INVALID_INDEX);\n atomicStore(&hashRegion[base + 2u * j + 1u], 0u);\n }\n storageBarrier();\n var exhausted = false;\n for (var a = lo + lid; a < hi; a = a + WG) {\n let k = keyIn[colIdx[a]];\n let q = quantize(weight_of(a), scale);\n var slot = lowbias32(k) % cap;\n var steps = 0u;\n loop {\n if (steps >= cap + 64u) { exhausted = true; break; } // bounded: a spurious compare-exchange failure is legal (WGSL 17.8.5)\n steps = steps + 1u;\n let r = atomicCompareExchangeWeak(&hashRegion[base + 2u * slot], INVALID_INDEX, k);\n if (r.exchanged || r.old_value == k) { atomicAdd(&hashRegion[base + 2u * slot + 1u], q); break; }\n if (r.old_value == INVALID_INDEX) { continue; } // a spurious failure: the same slot again\n slot = (slot + 1u) % cap; // linear probing\n }\n }\n if (exhausted) { atomicStore(&hashRegion[0], 1u); }\n storageBarrier();\n var bk = INVALID_INDEX;\n var bs = 0u;\n for (var j = lid; j < cap; j = j + WG) {\n let k = atomicLoad(&hashRegion[base + 2u * j]);\n if (k != INVALID_INDEX) {\n let sum = atomicLoad(&hashRegion[base + 2u * j + 1u]);\n if (better(sum, k, bs, bk)) { bk = k; bs = sum; }\n }\n }\n shKey[lid] = bk;\n shSum[lid] = bs;\n workgroupBarrier();\n for (var s = WG / 2u; s > 0u; s = s / 2u) {\n if (lid < s && better(shSum[lid + s], shKey[lid + s], shSum[lid], shKey[lid])) {\n shKey[lid] = shKey[lid + s];\n shSum[lid] = shSum[lid + s];\n }\n workgroupBarrier();\n }\n if (lid == 0u) {\n bestKey[v] = shKey[0];\n bestScore[v] = f32(shSum[0]) * inverse_of(scale);\n }\n}\n\n@compute @workgroup_size(WG)\nfn group_by_key_row(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n if (TIER == 0u) {\n let i = linear_id(wid, lid.x);\n if (i < P.count) { row_thread(rows[P.rowsBase + i]); }\n return; // TIER is an override: uniform\n }\n let g = group_id(wid);\n if (g >= P.count) { return; } // uniform: the workgroup id and a uniform\n row_hash(g, lid.x);\n}\n";
|
|
26
|
+
//# sourceMappingURL=group-by-key-row.wgsl.d.ts.map
|