@graphty/webgpu-graph-algorithms 0.6.23 → 0.6.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +1 -1
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-B40Z6lV_.js → context-BZY6SMsM.js} +41 -35
- package/dist/chunks/context-BZY6SMsM.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/algorithms/betweenness.d.ts +24 -9
- package/dist/src/algorithms/betweenness.d.ts.map +1 -1
- package/dist/src/algorithms/betweenness.js +105 -39
- package/dist/src/algorithms/betweenness.js.map +1 -1
- package/dist/src/constants.d.ts +11 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +11 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +4 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernels.d.ts +4 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +43 -13
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/types/betweenness.d.ts +5 -2
- package/dist/src/types/betweenness.d.ts.map +1 -1
- package/dist/src/wgsl/bc-backward.wgsl.d.ts +5 -2
- package/dist/src/wgsl/bc-backward.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bc-backward.wgsl.js +12 -3
- package/dist/src/wgsl/bc-backward.wgsl.js.map +1 -1
- package/dist/src/wgsl/bc-count.wgsl.d.ts +22 -0
- package/dist/src/wgsl/bc-count.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-count.wgsl.js +46 -0
- package/dist/src/wgsl/bc-count.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts +3 -2
- package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bc-edge-gather.wgsl.js +10 -2
- package/dist/src/wgsl/bc-edge-gather.wgsl.js.map +1 -1
- package/dist/src/wgsl/bc-finalize.wgsl.d.ts +4 -3
- package/dist/src/wgsl/bc-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bc-finalize.wgsl.js +5 -3
- package/dist/src/wgsl/bc-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts +2 -2
- package/dist/src/wgsl/bc-forward-edge.wgsl.js +2 -2
- package/dist/src/wgsl/bc-forward.wgsl.d.ts +4 -2
- package/dist/src/wgsl/bc-forward.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bc-forward.wgsl.js +4 -2
- package/dist/src/wgsl/bc-forward.wgsl.js.map +1 -1
- package/dist/webgpu-graph-algorithms.js +171 -47
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +3 -3
- package/src/algorithms/betweenness.ts +155 -48
- package/src/constants.ts +11 -0
- package/src/kernel/prelude.ts +4 -0
- package/src/kernels.ts +45 -13
- package/src/types/betweenness.ts +5 -2
- package/src/wgsl/bc-backward.wgsl.ts +12 -3
- package/src/wgsl/bc-count.wgsl.ts +45 -0
- package/src/wgsl/bc-edge-gather.wgsl.ts +10 -2
- package/src/wgsl/bc-finalize.wgsl.ts +5 -3
- package/src/wgsl/bc-forward-edge.wgsl.ts +2 -2
- package/src/wgsl/bc-forward.wgsl.ts +4 -2
- package/dist/chunks/context-B40Z6lV_.js.map +0 -1
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"bc-count.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-count.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AACH,eAAO,MAAM,WAAW,61CAwBvB,CAAC"}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bc-count` kernel body (design 8.4, issue #719): the shortest-path counts of the depth a betweenness forward
|
|
3
|
+
* level has just claimed, PULLED once the claims are complete. It runs only in a batch rerun because its u32 counts
|
|
4
|
+
* wrapped (the forward bodies then run with `SCALED` and count nothing); `sigmaK` holds f32 bits throughout. One invocation per entry `t = s * n + x` of the range
|
|
5
|
+
* `S[ends[level + 1] .. stackTop)` (the entries the forward dispatch appended), each summing `sigma[s][v]` over the
|
|
6
|
+
* in-arcs `(v, x)` with `depth[s][v] == level` in CSR order -- the reverse adjacency, which on an undirected snapshot
|
|
7
|
+
* is the forward one. No atomic touches a count, so the sums are bitwise reproducible, and both forward bodies leave
|
|
8
|
+
* the same counts.
|
|
9
|
+
*
|
|
10
|
+
* The counts are f32 and rescaled per depth, because a lattice outgrows any fixed width (a 40 x 40 grid has C(78, 39),
|
|
11
|
+
* about 2.6e22, corner-to-corner paths): each sum is multiplied by `2^-sigma_shift(levelMax[level])`, the power of
|
|
12
|
+
* two that brings the previous depth's largest count down to 2^BC_SIGMA_EXPONENT_CAP, and the result's bits go into
|
|
13
|
+
* `levelMax[level + 1]` by `atomicMax` (positive f32 bits order like the values). Every count of one (depth, batch)
|
|
14
|
+
* shares that scale, and Brandes' backward pass reads only ratios of a count to its successors', so `bc-backward`
|
|
15
|
+
* undoes the one step between two depths and nothing else. A count that leaves f32's normal range anyway -- zero,
|
|
16
|
+
* subnormal, infinite -- raises `sigmaOverflow` (counters word 27): the counts at one depth spread wider than f32 can
|
|
17
|
+
* hold, and the scores are wrong. Grid-stride loop (`P.stride`); the range is read from the device, so the host
|
|
18
|
+
* dispatches for the largest possible level. Body only; the sabotage rows of test/helpers/sabotage.ts are textual
|
|
19
|
+
* edits of it.
|
|
20
|
+
*/
|
|
21
|
+
export const bcCountWgsl = /* wgsl */ `
|
|
22
|
+
@compute @workgroup_size(WG)
|
|
23
|
+
fn bc_count(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
24
|
+
let level = atomicLoad(&counters[11]);
|
|
25
|
+
let start = ends[level + 1u]; // the boundary closed the claim-from level here
|
|
26
|
+
let count = atomicLoad(&counters[26]) - start; // stackTop: what the forward dispatch appended
|
|
27
|
+
let shift = sigma_shift(atomicLoad(&levelMax[level]));
|
|
28
|
+
for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) {
|
|
29
|
+
let t = S[start + i]; // s * n + x, at depth level + 1
|
|
30
|
+
let x = t % P.n;
|
|
31
|
+
let base = t - x; // s * n
|
|
32
|
+
var acc = 0.0;
|
|
33
|
+
for (var a = rowPtr[x]; a < rowPtr[x + 1u]; a = a + 1u) { // the in-arcs (v, x)
|
|
34
|
+
let v = base + colIdx[a];
|
|
35
|
+
if (depthK[v] == level) { acc = acc + bitcast<f32>(sigmaK[v]); } // every predecessor's paths, CSR order
|
|
36
|
+
}
|
|
37
|
+
let sigma = ldexp(acc, -shift);
|
|
38
|
+
let bits = bitcast<u32>(sigma);
|
|
39
|
+
sigmaK[t] = bits;
|
|
40
|
+
let exponent = (bits >> 23u) & 0xffu;
|
|
41
|
+
if (exponent == 0u || exponent == 0xffu) { atomicOr(&counters[27], 1u); } // out of f32's normal range
|
|
42
|
+
atomicMax(&levelMax[level + 1u], bits);
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
`;
|
|
46
|
+
//# sourceMappingURL=bc-count.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"bc-count.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-count.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AACH,MAAM,CAAC,MAAM,WAAW,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;CAwBrC,CAAC"}
|
|
@@ -3,10 +3,11 @@
|
|
|
3
3
|
* twin of `bc-gather`, run once per batch after the backward sweep. One invocation per ARC (`P.count` arcs,
|
|
4
4
|
* grid-stride): it finds the row `w` that owns the arc by an upper-bound search over `rowPtr`, then adds over the
|
|
5
5
|
* batch's sources in order the term the backward pass summed -- `sigma[s][w] / sigma[s][v] * (1 + delta[s][v])`
|
|
6
|
-
* whenever `w` was reached and `depth[s][v] == depth[s][w] + 1
|
|
6
|
+
* whenever `w` was reached and `depth[s][v] == depth[s][w] + 1`, with `bc-backward`'s depth scale step under `SCALED` -- into
|
|
7
|
+
* `arcScores[arc]`. Each arc is written by one
|
|
7
8
|
* invocation: no atomic, a fixed order. Arc-parallel rather than row-parallel so a hub row is not one lane's loop
|
|
8
9
|
* (llvmpipe caps a shader loop at 65,535 iterations, and a 1,000-arc hub times 64 sources passed it). Body only;
|
|
9
10
|
* the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
10
11
|
*/
|
|
11
|
-
export declare const bcEdgeGatherWgsl = "\n@compute @workgroup_size(WG)\nfn bc_edge_gather(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n for (var a = linear_id(wid, lid.x); a < P.count; a = a + P.stride) {\n var lo = 0u; // the row w with rowPtr[w] <= a < rowPtr[w + 1]\n var hi = P.n;\n loop {\n if (lo >= hi) { break; }\n let mid = (lo + hi) / 2u;\n if (rowPtr[mid + 1u] <= a) { lo = mid + 1u; } else { hi = mid; }\n }\n let w = lo;\n let nbr = colIdx[a];\n var acc = arcScores[a];\n for (var s = 0u; s < P.k; s = s + 1u) {\n let base = s * P.n;\n let dw = depthK[base + w];\n if (dw != INVALID_INDEX && depthK[base + nbr] == dw + 1u) { // (w, nbr) is on a shortest path from s\n
|
|
12
|
+
export declare const bcEdgeGatherWgsl = "\nfn sigma_of(word: u32) -> f32 { // a stored count: u32, or f32 bits when SCALED\n if (SCALED) { return bitcast<f32>(word); }\n return f32(word);\n}\n\n@compute @workgroup_size(WG)\nfn bc_edge_gather(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n for (var a = linear_id(wid, lid.x); a < P.count; a = a + P.stride) {\n var lo = 0u; // the row w with rowPtr[w] <= a < rowPtr[w + 1]\n var hi = P.n;\n loop {\n if (lo >= hi) { break; }\n let mid = (lo + hi) / 2u;\n if (rowPtr[mid + 1u] <= a) { lo = mid + 1u; } else { hi = mid; }\n }\n let w = lo;\n let nbr = colIdx[a];\n var acc = arcScores[a];\n for (var s = 0u; s < P.k; s = s + 1u) {\n let base = s * P.n;\n let dw = depthK[base + w];\n if (dw != INVALID_INDEX && depthK[base + nbr] == dw + 1u) { // (w, nbr) is on a shortest path from s\n let shift = select(0i, sigma_shift(levelMax[dw]), SCALED);\n let ratio = ldexp(sigma_of(sigmaK[base + w]) / sigma_of(sigmaK[base + nbr]), -shift);\n acc = acc + ratio * (1.0 + deltaK[base + nbr]);\n }\n }\n arcScores[a] = acc;\n }\n}\n";
|
|
12
13
|
//# sourceMappingURL=bc-edge-gather.wgsl.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"bc-edge-gather.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-edge-gather.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"bc-edge-gather.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-edge-gather.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AACH,eAAO,MAAM,gBAAgB,q1CA+B5B,CAAC"}
|
|
@@ -3,12 +3,18 @@
|
|
|
3
3
|
* twin of `bc-gather`, run once per batch after the backward sweep. One invocation per ARC (`P.count` arcs,
|
|
4
4
|
* grid-stride): it finds the row `w` that owns the arc by an upper-bound search over `rowPtr`, then adds over the
|
|
5
5
|
* batch's sources in order the term the backward pass summed -- `sigma[s][w] / sigma[s][v] * (1 + delta[s][v])`
|
|
6
|
-
* whenever `w` was reached and `depth[s][v] == depth[s][w] + 1
|
|
6
|
+
* whenever `w` was reached and `depth[s][v] == depth[s][w] + 1`, with `bc-backward`'s depth scale step under `SCALED` -- into
|
|
7
|
+
* `arcScores[arc]`. Each arc is written by one
|
|
7
8
|
* invocation: no atomic, a fixed order. Arc-parallel rather than row-parallel so a hub row is not one lane's loop
|
|
8
9
|
* (llvmpipe caps a shader loop at 65,535 iterations, and a 1,000-arc hub times 64 sources passed it). Body only;
|
|
9
10
|
* the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
10
11
|
*/
|
|
11
12
|
export const bcEdgeGatherWgsl = /* wgsl */ `
|
|
13
|
+
fn sigma_of(word: u32) -> f32 { // a stored count: u32, or f32 bits when SCALED
|
|
14
|
+
if (SCALED) { return bitcast<f32>(word); }
|
|
15
|
+
return f32(word);
|
|
16
|
+
}
|
|
17
|
+
|
|
12
18
|
@compute @workgroup_size(WG)
|
|
13
19
|
fn bc_edge_gather(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
14
20
|
for (var a = linear_id(wid, lid.x); a < P.count; a = a + P.stride) {
|
|
@@ -26,7 +32,9 @@ fn bc_edge_gather(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
|
|
|
26
32
|
let base = s * P.n;
|
|
27
33
|
let dw = depthK[base + w];
|
|
28
34
|
if (dw != INVALID_INDEX && depthK[base + nbr] == dw + 1u) { // (w, nbr) is on a shortest path from s
|
|
29
|
-
|
|
35
|
+
let shift = select(0i, sigma_shift(levelMax[dw]), SCALED);
|
|
36
|
+
let ratio = ldexp(sigma_of(sigmaK[base + w]) / sigma_of(sigmaK[base + nbr]), -shift);
|
|
37
|
+
acc = acc + ratio * (1.0 + deltaK[base + nbr]);
|
|
30
38
|
}
|
|
31
39
|
}
|
|
32
40
|
arcScores[a] = acc;
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"bc-edge-gather.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-edge-gather.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"bc-edge-gather.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-edge-gather.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA+B1C,CAAC"}
|
|
@@ -5,8 +5,9 @@
|
|
|
5
5
|
* entries at depth `L` are `S[ends[L] .. ends[L + 1])`. That log is Brandes' stack, so the backward pass walks the
|
|
6
6
|
* same ranges from the deepest level up and nothing is ever copied between levels.
|
|
7
7
|
*
|
|
8
|
-
* Role 1 seeds the batch: the k seed entries the host wrote into `S[0 .. k)` get depth 0 and one shortest path
|
|
9
|
-
*
|
|
8
|
+
* Role 1 seeds the batch: the k seed entries the host wrote into `S[0 .. k)` get depth 0 and one shortest path (the
|
|
9
|
+
* u32 1, or under `SCALED` the bits of the f32 1.0), `levelMax[0]` the bits of 1.0 (the largest count at depth 0,
|
|
10
|
+
* which `bc-count` scales depth 1 from), `ends[0] = 0`, the append cursor `stackTop` (counters word 26) starts at k, the overflow flag (word 27) is cleared,
|
|
10
11
|
* `level` (word 11) is U32_MAX so the first boundary lands on 0, and `done` (word 15) is cleared.
|
|
11
12
|
*
|
|
12
13
|
* Role 0 is the level boundary, recorded before every forward level: it advances `level`, closes the level just
|
|
@@ -17,5 +18,5 @@
|
|
|
17
18
|
* early return of the others (spec 3.5 rule 1). Body only; the sabotage rows of test/helpers/sabotage.ts are
|
|
18
19
|
* textual edits of it.
|
|
19
20
|
*/
|
|
20
|
-
export declare const bcFinalizeWgsl = "\n@compute @workgroup_size(WG)\nfn bc_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role == 1u) { // the seed of a batch\n for (var i = 0u; i < P.k; i = i + 1u) {\n let t = S[i];\n depthK[t] = 0u; // the source is at depth 0\n sigmaK[t] = 1u;
|
|
21
|
+
export declare const bcFinalizeWgsl = "\n@compute @workgroup_size(WG)\nfn bc_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role == 1u) { // the seed of a batch\n for (var i = 0u; i < P.k; i = i + 1u) {\n let t = S[i];\n depthK[t] = 0u; // the source is at depth 0\n sigmaK[t] = select(1u, bitcast<u32>(1.0), SCALED); // with one shortest path, itself (f32 bits when SCALED)\n }\n levelMax[0] = bitcast<u32>(1.0); // the largest count at depth 0\n ends[0] = 0u;\n atomicStore(&counters[26], P.k); // stackTop: the seeds are the log's first k entries\n atomicStore(&counters[27], 0u); // sigmaOverflow\n atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0\n atomicStore(&counters[15], 0u); // done\n return;\n }\n if (atomicLoad(&counters[15]) != 0u) { return; } // done: a no-op level the host recorded past the end\n let level = atomicLoad(&counters[11]) + 1u;\n let top = atomicLoad(&counters[26]);\n ends[level + 1u] = top; // the level's entries end where the log ends now\n let count = top - ends[level];\n atomicStore(&counters[0], count); // frontierCount (the inspect seam reads it)\n atomicStore(&counters[11], level);\n atomicStore(&counters[15], select(0u, 1u, count == 0u)); // an empty level ends the batch\n}\n";
|
|
21
22
|
//# sourceMappingURL=bc-finalize.wgsl.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"bc-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-finalize.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"bc-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AACH,eAAO,MAAM,cAAc,4wDA2B1B,CAAC"}
|
|
@@ -5,8 +5,9 @@
|
|
|
5
5
|
* entries at depth `L` are `S[ends[L] .. ends[L + 1])`. That log is Brandes' stack, so the backward pass walks the
|
|
6
6
|
* same ranges from the deepest level up and nothing is ever copied between levels.
|
|
7
7
|
*
|
|
8
|
-
* Role 1 seeds the batch: the k seed entries the host wrote into `S[0 .. k)` get depth 0 and one shortest path
|
|
9
|
-
*
|
|
8
|
+
* Role 1 seeds the batch: the k seed entries the host wrote into `S[0 .. k)` get depth 0 and one shortest path (the
|
|
9
|
+
* u32 1, or under `SCALED` the bits of the f32 1.0), `levelMax[0]` the bits of 1.0 (the largest count at depth 0,
|
|
10
|
+
* which `bc-count` scales depth 1 from), `ends[0] = 0`, the append cursor `stackTop` (counters word 26) starts at k, the overflow flag (word 27) is cleared,
|
|
10
11
|
* `level` (word 11) is U32_MAX so the first boundary lands on 0, and `done` (word 15) is cleared.
|
|
11
12
|
*
|
|
12
13
|
* Role 0 is the level boundary, recorded before every forward level: it advances `level`, closes the level just
|
|
@@ -25,8 +26,9 @@ fn bc_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
25
26
|
for (var i = 0u; i < P.k; i = i + 1u) {
|
|
26
27
|
let t = S[i];
|
|
27
28
|
depthK[t] = 0u; // the source is at depth 0
|
|
28
|
-
sigmaK[t] = 1u;
|
|
29
|
+
sigmaK[t] = select(1u, bitcast<u32>(1.0), SCALED); // with one shortest path, itself (f32 bits when SCALED)
|
|
29
30
|
}
|
|
31
|
+
levelMax[0] = bitcast<u32>(1.0); // the largest count at depth 0
|
|
30
32
|
ends[0] = 0u;
|
|
31
33
|
atomicStore(&counters[26], P.k); // stackTop: the seeds are the log's first k entries
|
|
32
34
|
atomicStore(&counters[27], 0u); // sigmaOverflow
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"bc-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-finalize.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"bc-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AACH,MAAM,CAAC,MAAM,cAAc,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;CA2BxC,CAAC"}
|
|
@@ -5,11 +5,11 @@
|
|
|
5
5
|
* and source `s`, when `depth[s][u]` is the level the edge is relaxed toward `x` with EXACTLY the claim and the count
|
|
6
6
|
* of `bc-forward` (the pre-check, `atomicMin`, the winner appends `s * n + x` to the same claim log, every arc on a
|
|
7
7
|
* shortest path adds `sigma[s][u]`, a u32 wrap raises word 27); `UNDIRECTED` relaxes the other direction too, the
|
|
8
|
-
* edge list holding each undirected edge once. So the two forward bodies are interchangeable level by level and the
|
|
8
|
+
* edge list holding each undirected edge once. Under `SCALED` the count is skipped, as in `bc-forward`. So the two forward bodies are interchangeable level by level and the
|
|
9
9
|
* backward pass cannot tell which ran. A workgroup does nothing when the level is empty (the done boundary). The
|
|
10
10
|
* winners of a strip are packed into the log with one global `atomicAdd` per strip; every barrier is in uniform
|
|
11
11
|
* control flow (the loop bounds are uniforms and the workgroup id). Body only; the sabotage rows of
|
|
12
12
|
* test/helpers/sabotage.ts are textual edits of it.
|
|
13
13
|
*/
|
|
14
|
-
export declare const bcForwardEdgeWgsl = "\nvar<workgroup> wlive: u32; // 1 when the level has entries\nvar<workgroup> wwon: atomic<u32>; // the strip's winners\nvar<workgroup> wbase: u32; // where the strip's winners go in the log\n\nfn claim(x: u32, next: u32) -> bool {\n if (atomicLoad(&depthK[x]) != INVALID_INDEX) { return false; } // the pre-check of design 16.1\n return atomicMin(&depthK[x], next) == INVALID_INDEX;\n}\n\nfn count_paths(origin: u32, x: u32, next: u32) {\n if (atomicLoad(&depthK[x]) == next) {
|
|
14
|
+
export declare const bcForwardEdgeWgsl = "\nvar<workgroup> wlive: u32; // 1 when the level has entries\nvar<workgroup> wwon: atomic<u32>; // the strip's winners\nvar<workgroup> wbase: u32; // where the strip's winners go in the log\n\nfn claim(x: u32, next: u32) -> bool {\n if (atomicLoad(&depthK[x]) != INVALID_INDEX) { return false; } // the pre-check of design 16.1\n return atomicMin(&depthK[x], next) == INVALID_INDEX;\n}\n\nfn count_paths(origin: u32, x: u32, next: u32) {\n if (!SCALED && atomicLoad(&depthK[x]) == next) { // every arc on a shortest path adds\n let add = atomicLoad(&sigmaK[origin]);\n let old = atomicAdd(&sigmaK[x], add);\n if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported\n }\n}\n\n@compute @workgroup_size(WG)\nfn bc_forward_edge(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let level = atomicLoad(&counters[11]);\n if (lid.x == 0u) { wlive = select(0u, 1u, ends[level + 1u] > ends[level]); }\n if (workgroupUniformLoad(&wlive) == 0u) { return; } // uniform: nothing below runs on an empty level\n let next = level + 1u;\n for (var s = 0u; s < P.k; s = s + 1u) {\n let base = s * P.n;\n for (var e0 = group_id(wid) * WG; e0 < P.count; e0 = e0 + P.stride) { // grid-stride over the edges\n let e = e0 + lid.x;\n var a = INVALID_INDEX; // the claims this lane won\n var b = INVALID_INDEX;\n if (e < P.count) {\n let u = base + edgeSrc[e];\n let x = base + edgeDst[e];\n if (atomicLoad(&depthK[u]) == level) {\n if (claim(x, next)) { a = x; }\n count_paths(u, x, next);\n }\n if (UNDIRECTED) { // the other direction of an undirected edge\n if (atomicLoad(&depthK[x]) == level) {\n if (claim(u, next)) { b = u; }\n count_paths(x, u, next);\n }\n }\n }\n let mine = select(0u, 1u, a != INVALID_INDEX) + select(0u, 1u, b != INVALID_INDEX);\n var slot = 0u;\n if (mine != 0u) { slot = atomicAdd(&wwon, mine); }\n workgroupBarrier();\n if (lid.x == 0u) {\n wbase = atomicAdd(&counters[26], atomicLoad(&wwon)); // stackTop: one global atomic per strip\n atomicStore(&wwon, 0u);\n }\n workgroupBarrier();\n if (a != INVALID_INDEX) {\n S[wbase + slot] = a;\n slot = slot + 1u;\n }\n if (b != INVALID_INDEX) { S[wbase + slot] = b; }\n }\n }\n}\n";
|
|
15
15
|
//# sourceMappingURL=bc-forward-edge.wgsl.d.ts.map
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
* and source `s`, when `depth[s][u]` is the level the edge is relaxed toward `x` with EXACTLY the claim and the count
|
|
6
6
|
* of `bc-forward` (the pre-check, `atomicMin`, the winner appends `s * n + x` to the same claim log, every arc on a
|
|
7
7
|
* shortest path adds `sigma[s][u]`, a u32 wrap raises word 27); `UNDIRECTED` relaxes the other direction too, the
|
|
8
|
-
* edge list holding each undirected edge once. So the two forward bodies are interchangeable level by level and the
|
|
8
|
+
* edge list holding each undirected edge once. Under `SCALED` the count is skipped, as in `bc-forward`. So the two forward bodies are interchangeable level by level and the
|
|
9
9
|
* backward pass cannot tell which ran. A workgroup does nothing when the level is empty (the done boundary). The
|
|
10
10
|
* winners of a strip are packed into the log with one global `atomicAdd` per strip; every barrier is in uniform
|
|
11
11
|
* control flow (the loop bounds are uniforms and the workgroup id). Body only; the sabotage rows of
|
|
@@ -22,7 +22,7 @@ fn claim(x: u32, next: u32) -> bool {
|
|
|
22
22
|
}
|
|
23
23
|
|
|
24
24
|
fn count_paths(origin: u32, x: u32, next: u32) {
|
|
25
|
-
if (atomicLoad(&depthK[x]) == next) {
|
|
25
|
+
if (!SCALED && atomicLoad(&depthK[x]) == next) { // every arc on a shortest path adds
|
|
26
26
|
let add = atomicLoad(&sigmaK[origin]);
|
|
27
27
|
let old = atomicAdd(&sigmaK[x], add);
|
|
28
28
|
if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported
|
|
@@ -12,12 +12,14 @@
|
|
|
12
12
|
* observes INVALID_INDEX the unique winner, which appends `t` to the log; and, as a SEPARATE condition, the COUNT:
|
|
13
13
|
* every arc that reaches `t` at `level + 1` adds `sigma[s][u]` into `sigma[s][x]`, winner or not, which is what makes
|
|
14
14
|
* sigma the number of shortest paths rather than of claims. The add detects a u32 wrap from `atomicAdd`'s return
|
|
15
|
-
* value (`old + add < old`) and raises `sigmaOverflow` (counters word 27); the count is never clamped.
|
|
15
|
+
* value (`old + add < old`) and raises `sigmaOverflow` (counters word 27); the count is never clamped. Under
|
|
16
|
+
* `SCALED` (a batch rerun because the u32 counts wrapped) the count is skipped and `bc-count` pulls rescaled f32
|
|
17
|
+
* counts after the level instead. The winners of
|
|
16
18
|
* a strip are packed into the log with one workgroup-memory counter and ONE global `atomicAdd` on `stackTop` (word
|
|
17
19
|
* 26) per strip. Uniformity (spec 3.5 rule 1): the range and the aggregate are `workgroupUniformLoad`s and the strip
|
|
18
20
|
* loop steps a uniform `p0`, so every barrier is in uniform control flow. The order of the log inside a level is
|
|
19
21
|
* not deterministic; nothing downstream depends on it (the counts are integers and every dependency is written by
|
|
20
22
|
* index). Body only; the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
21
23
|
*/
|
|
22
|
-
export declare const bcForwardWgsl = "\nvar<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan\nvar<workgroup> rowStart: array<u32, WG>; // the first arc of each entry's row\nvar<workgroup> entryOf: array<u32, WG>; // each entry, s * n + u\nvar<workgroup> wstart: u32; // the level's first log index\nvar<workgroup> wcount: u32; // the level's entry count\nvar<workgroup> wwon: atomic<u32>; // the strip's winners\nvar<workgroup> wbase: u32; // where the strip's winners go in the log\n\n@compute @workgroup_size(WG)\nfn bc_forward(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let level = atomicLoad(&counters[11]);\n if (lid.x == 0u) {\n let lo = ends[level];\n wstart = lo;\n wcount = ends[level + 1u] - lo;\n }\n let start = workgroupUniformLoad(&wstart);\n let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers\n let next = level + 1u;\n for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries\n let i = b0 + lid.x;\n var deg = 0u;\n var first = 0u;\n var entry = 0u;\n if (i < count) { // guarded loads into locals (3.5 rule 1)\n entry = S[start + i];\n let u = entry % P.n;\n first = rowPtr[u];\n deg = rowPtr[u + 1u] - first;\n }\n sh[lid.x] = deg;\n rowStart[lid.x] = first;\n entryOf[lid.x] = entry;\n workgroupBarrier();\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier\n for (var p0 = 0u; p0 < aggregate; p0 = p0 + WG) { // strip [0, aggregate) WG arcs at a time\n let p = p0 + lid.x;\n var won = false;\n var claimed = 0u;\n if (p < aggregate) {\n var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p\n var hi = WG;\n loop {\n if (lo >= hi) { break; }\n let mid = (lo + hi) / 2u;\n if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }\n }\n let k = lo;\n var exclusive = 0u;\n if (k > 0u) { exclusive = sh[k - 1u]; }\n let origin = entryOf[k]; // s * n + u\n let x = (origin - (origin % P.n)) + colIdx[rowStart[k] + (p - exclusive)]; // s * n + x\n if (atomicLoad(&depthK[x]) == INVALID_INDEX) { // the pre-check of design 16.1\n won = atomicMin(&depthK[x], next) == INVALID_INDEX; // the claim: the one winner appends\n }\n if (atomicLoad(&depthK[x]) == next) {
|
|
24
|
+
export declare const bcForwardWgsl = "\nvar<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan\nvar<workgroup> rowStart: array<u32, WG>; // the first arc of each entry's row\nvar<workgroup> entryOf: array<u32, WG>; // each entry, s * n + u\nvar<workgroup> wstart: u32; // the level's first log index\nvar<workgroup> wcount: u32; // the level's entry count\nvar<workgroup> wwon: atomic<u32>; // the strip's winners\nvar<workgroup> wbase: u32; // where the strip's winners go in the log\n\n@compute @workgroup_size(WG)\nfn bc_forward(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let level = atomicLoad(&counters[11]);\n if (lid.x == 0u) {\n let lo = ends[level];\n wstart = lo;\n wcount = ends[level + 1u] - lo;\n }\n let start = workgroupUniformLoad(&wstart);\n let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers\n let next = level + 1u;\n for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries\n let i = b0 + lid.x;\n var deg = 0u;\n var first = 0u;\n var entry = 0u;\n if (i < count) { // guarded loads into locals (3.5 rule 1)\n entry = S[start + i];\n let u = entry % P.n;\n first = rowPtr[u];\n deg = rowPtr[u + 1u] - first;\n }\n sh[lid.x] = deg;\n rowStart[lid.x] = first;\n entryOf[lid.x] = entry;\n workgroupBarrier();\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier\n for (var p0 = 0u; p0 < aggregate; p0 = p0 + WG) { // strip [0, aggregate) WG arcs at a time\n let p = p0 + lid.x;\n var won = false;\n var claimed = 0u;\n if (p < aggregate) {\n var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p\n var hi = WG;\n loop {\n if (lo >= hi) { break; }\n let mid = (lo + hi) / 2u;\n if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }\n }\n let k = lo;\n var exclusive = 0u;\n if (k > 0u) { exclusive = sh[k - 1u]; }\n let origin = entryOf[k]; // s * n + u\n let x = (origin - (origin % P.n)) + colIdx[rowStart[k] + (p - exclusive)]; // s * n + x\n if (atomicLoad(&depthK[x]) == INVALID_INDEX) { // the pre-check of design 16.1\n won = atomicMin(&depthK[x], next) == INVALID_INDEX; // the claim: the one winner appends\n }\n if (!SCALED && atomicLoad(&depthK[x]) == next) { // the count: EVERY arc on a shortest path adds\n let add = atomicLoad(&sigmaK[origin]);\n let old = atomicAdd(&sigmaK[x], add);\n if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported\n }\n claimed = x;\n }\n var slot = 0u;\n if (won) { slot = atomicAdd(&wwon, 1u); }\n workgroupBarrier();\n if (lid.x == 0u) {\n wbase = atomicAdd(&counters[26], atomicLoad(&wwon)); // stackTop: one global atomic per strip\n atomicStore(&wwon, 0u);\n }\n workgroupBarrier();\n if (won) { S[wbase + slot] = claimed; }\n }\n workgroupBarrier(); // sh, rowStart and entryOf are reused by the next block\n }\n}\n";
|
|
23
25
|
//# sourceMappingURL=bc-forward.wgsl.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"bc-forward.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-forward.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"bc-forward.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-forward.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;GAsBG;AACH,eAAO,MAAM,aAAa,+nIAmFzB,CAAC"}
|
|
@@ -12,7 +12,9 @@
|
|
|
12
12
|
* observes INVALID_INDEX the unique winner, which appends `t` to the log; and, as a SEPARATE condition, the COUNT:
|
|
13
13
|
* every arc that reaches `t` at `level + 1` adds `sigma[s][u]` into `sigma[s][x]`, winner or not, which is what makes
|
|
14
14
|
* sigma the number of shortest paths rather than of claims. The add detects a u32 wrap from `atomicAdd`'s return
|
|
15
|
-
* value (`old + add < old`) and raises `sigmaOverflow` (counters word 27); the count is never clamped.
|
|
15
|
+
* value (`old + add < old`) and raises `sigmaOverflow` (counters word 27); the count is never clamped. Under
|
|
16
|
+
* `SCALED` (a batch rerun because the u32 counts wrapped) the count is skipped and `bc-count` pulls rescaled f32
|
|
17
|
+
* counts after the level instead. The winners of
|
|
16
18
|
* a strip are packed into the log with one workgroup-memory counter and ONE global `atomicAdd` on `stackTop` (word
|
|
17
19
|
* 26) per strip. Uniformity (spec 3.5 rule 1): the range and the aggregate are `workgroupUniformLoad`s and the strip
|
|
18
20
|
* loop steps a uniform `p0`, so every barrier is in uniform control flow. The order of the log inside a level is
|
|
@@ -82,7 +84,7 @@ fn bc_forward(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_i
|
|
|
82
84
|
if (atomicLoad(&depthK[x]) == INVALID_INDEX) { // the pre-check of design 16.1
|
|
83
85
|
won = atomicMin(&depthK[x], next) == INVALID_INDEX; // the claim: the one winner appends
|
|
84
86
|
}
|
|
85
|
-
if (atomicLoad(&depthK[x]) == next) {
|
|
87
|
+
if (!SCALED && atomicLoad(&depthK[x]) == next) { // the count: EVERY arc on a shortest path adds
|
|
86
88
|
let add = atomicLoad(&sigmaK[origin]);
|
|
87
89
|
let old = atomicAdd(&sigmaK[x], add);
|
|
88
90
|
if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"bc-forward.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-forward.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"bc-forward.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-forward.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;GAsBG;AACH,MAAM,CAAC,MAAM,aAAa,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAmFvC,CAAC"}
|