@graphty/webgpu-graph-algorithms 0.6.24 → 0.6.25

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/LICENSE +1 -1
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-B40Z6lV_.js → context-BZY6SMsM.js} +41 -35
  4. package/dist/chunks/context-BZY6SMsM.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/algorithms/betweenness.d.ts +24 -9
  7. package/dist/src/algorithms/betweenness.d.ts.map +1 -1
  8. package/dist/src/algorithms/betweenness.js +105 -39
  9. package/dist/src/algorithms/betweenness.js.map +1 -1
  10. package/dist/src/constants.d.ts +11 -0
  11. package/dist/src/constants.d.ts.map +1 -1
  12. package/dist/src/constants.js +11 -0
  13. package/dist/src/constants.js.map +1 -1
  14. package/dist/src/kernel/prelude.d.ts.map +1 -1
  15. package/dist/src/kernel/prelude.js +4 -1
  16. package/dist/src/kernel/prelude.js.map +1 -1
  17. package/dist/src/kernels.d.ts +4 -4
  18. package/dist/src/kernels.d.ts.map +1 -1
  19. package/dist/src/kernels.js +43 -13
  20. package/dist/src/kernels.js.map +1 -1
  21. package/dist/src/types/betweenness.d.ts +5 -2
  22. package/dist/src/types/betweenness.d.ts.map +1 -1
  23. package/dist/src/wgsl/bc-backward.wgsl.d.ts +5 -2
  24. package/dist/src/wgsl/bc-backward.wgsl.d.ts.map +1 -1
  25. package/dist/src/wgsl/bc-backward.wgsl.js +12 -3
  26. package/dist/src/wgsl/bc-backward.wgsl.js.map +1 -1
  27. package/dist/src/wgsl/bc-count.wgsl.d.ts +22 -0
  28. package/dist/src/wgsl/bc-count.wgsl.d.ts.map +1 -0
  29. package/dist/src/wgsl/bc-count.wgsl.js +46 -0
  30. package/dist/src/wgsl/bc-count.wgsl.js.map +1 -0
  31. package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts +3 -2
  32. package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts.map +1 -1
  33. package/dist/src/wgsl/bc-edge-gather.wgsl.js +10 -2
  34. package/dist/src/wgsl/bc-edge-gather.wgsl.js.map +1 -1
  35. package/dist/src/wgsl/bc-finalize.wgsl.d.ts +4 -3
  36. package/dist/src/wgsl/bc-finalize.wgsl.d.ts.map +1 -1
  37. package/dist/src/wgsl/bc-finalize.wgsl.js +5 -3
  38. package/dist/src/wgsl/bc-finalize.wgsl.js.map +1 -1
  39. package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts +2 -2
  40. package/dist/src/wgsl/bc-forward-edge.wgsl.js +2 -2
  41. package/dist/src/wgsl/bc-forward.wgsl.d.ts +4 -2
  42. package/dist/src/wgsl/bc-forward.wgsl.d.ts.map +1 -1
  43. package/dist/src/wgsl/bc-forward.wgsl.js +4 -2
  44. package/dist/src/wgsl/bc-forward.wgsl.js.map +1 -1
  45. package/dist/webgpu-graph-algorithms.js +171 -47
  46. package/dist/webgpu-graph-algorithms.js.map +1 -1
  47. package/package.json +3 -3
  48. package/src/algorithms/betweenness.ts +155 -48
  49. package/src/constants.ts +11 -0
  50. package/src/kernel/prelude.ts +4 -0
  51. package/src/kernels.ts +45 -13
  52. package/src/types/betweenness.ts +5 -2
  53. package/src/wgsl/bc-backward.wgsl.ts +12 -3
  54. package/src/wgsl/bc-count.wgsl.ts +45 -0
  55. package/src/wgsl/bc-edge-gather.wgsl.ts +10 -2
  56. package/src/wgsl/bc-finalize.wgsl.ts +5 -3
  57. package/src/wgsl/bc-forward-edge.wgsl.ts +2 -2
  58. package/src/wgsl/bc-forward.wgsl.ts +4 -2
  59. package/dist/chunks/context-B40Z6lV_.js.map +0 -1
@@ -4,7 +4,10 @@
4
4
  * level's range `S[P.start .. P.start + P.count)`, which the host planned from the `ends` it read back. Each entry
5
5
  * walks the out-arcs of `w` -- the rows the forward pass expanded -- and sums `sigma[s][w] / sigma[s][v] * (1 +
6
6
  * delta[s][v])` over the successors `v` (`depth[s][v] == depth[s][w] + 1`), whose dependencies the previous (deeper)
7
- * dispatch wrote, then writes `delta[s][w]` ONCE. No float is ever accumulated through an atomic, so the result is
7
+ * dispatch wrote, then writes `delta[s][w]` ONCE. The stored counts are scaled per depth (`bc-count`), so the ratio
8
+ * is taken back to true scale with `ldexp(..., -sigma_shift(levelMax[depth]))`, the step `bc-count` applied between
9
+ * `w`'s depth and its successors'. Only under `SCALED` (the batch reran with `bc-count`'s f32 counts); otherwise the
10
+ * counts are exact u32 and the step is 0, which leaves every ratio bit for bit what it was. No float is ever accumulated through an atomic, so the result is
8
11
  * bitwise reproducible. The sources (depth 0) are never dispatched: their dependency stays 0, which is what keeps a
9
12
  * source's own dependency out of its score. Grid-stride loop (`P.stride`). ponytail: one lane walks one row, so a
10
13
  * vertex of degree above 65,535 exceeds llvmpipe's per-invocation loop cap (CLAUDE.md, Verified Platform Facts) and
@@ -12,6 +15,11 @@
12
15
  * test/helpers/sabotage.ts are textual edits of it.
13
16
  */
14
17
  export const bcBackwardWgsl = /* wgsl */ `
18
+ fn sigma_of(word: u32) -> f32 { // a stored count: u32, or f32 bits when SCALED
19
+ if (SCALED) { return bitcast<f32>(word); }
20
+ return f32(word);
21
+ }
22
+
15
23
  @compute @workgroup_size(WG)
16
24
  fn bc_backward(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
17
25
  for (var i = linear_id(wid, lid.x); i < P.count; i = i + P.stride) {
@@ -19,12 +27,13 @@ fn bc_backward(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_
19
27
  let w = t % P.n;
20
28
  let base = t - w; // s * n
21
29
  let succ = depthK[t] + 1u;
22
- let sw = f32(sigmaK[t]);
30
+ let sw = sigma_of(sigmaK[t]);
31
+ let shift = select(0i, sigma_shift(levelMax[depthK[t]]), SCALED); // the scale step to the successors' depth
23
32
  var acc = 0.0;
24
33
  for (var a = rowPtr[w]; a < rowPtr[w + 1u]; a = a + 1u) {
25
34
  let v = base + colIdx[a];
26
35
  if (depthK[v] == succ) { // v is a successor of w for source s
27
- acc = acc + (sw / f32(sigmaK[v])) * (1.0 + deltaK[v]);
36
+ acc = acc + ldexp(sw / sigma_of(sigmaK[v]), -shift) * (1.0 + deltaK[v]);
28
37
  }
29
38
  }
30
39
  deltaK[t] = acc; // written once per (w, s)
@@ -0,0 +1,45 @@
1
+ /**
2
+ * The `bc-count` kernel body (design 8.4, issue #719): the shortest-path counts of the depth a betweenness forward
3
+ * level has just claimed, PULLED once the claims are complete. It runs only in a batch rerun because its u32 counts
4
+ * wrapped (the forward bodies then run with `SCALED` and count nothing); `sigmaK` holds f32 bits throughout. One invocation per entry `t = s * n + x` of the range
5
+ * `S[ends[level + 1] .. stackTop)` (the entries the forward dispatch appended), each summing `sigma[s][v]` over the
6
+ * in-arcs `(v, x)` with `depth[s][v] == level` in CSR order -- the reverse adjacency, which on an undirected snapshot
7
+ * is the forward one. No atomic touches a count, so the sums are bitwise reproducible, and both forward bodies leave
8
+ * the same counts.
9
+ *
10
+ * The counts are f32 and rescaled per depth, because a lattice outgrows any fixed width (a 40 x 40 grid has C(78, 39),
11
+ * about 2.6e22, corner-to-corner paths): each sum is multiplied by `2^-sigma_shift(levelMax[level])`, the power of
12
+ * two that brings the previous depth's largest count down to 2^BC_SIGMA_EXPONENT_CAP, and the result's bits go into
13
+ * `levelMax[level + 1]` by `atomicMax` (positive f32 bits order like the values). Every count of one (depth, batch)
14
+ * shares that scale, and Brandes' backward pass reads only ratios of a count to its successors', so `bc-backward`
15
+ * undoes the one step between two depths and nothing else. A count that leaves f32's normal range anyway -- zero,
16
+ * subnormal, infinite -- raises `sigmaOverflow` (counters word 27): the counts at one depth spread wider than f32 can
17
+ * hold, and the scores are wrong. Grid-stride loop (`P.stride`); the range is read from the device, so the host
18
+ * dispatches for the largest possible level. Body only; the sabotage rows of test/helpers/sabotage.ts are textual
19
+ * edits of it.
20
+ */
21
+ export const bcCountWgsl = /* wgsl */ `
22
+ @compute @workgroup_size(WG)
23
+ fn bc_count(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
24
+ let level = atomicLoad(&counters[11]);
25
+ let start = ends[level + 1u]; // the boundary closed the claim-from level here
26
+ let count = atomicLoad(&counters[26]) - start; // stackTop: what the forward dispatch appended
27
+ let shift = sigma_shift(atomicLoad(&levelMax[level]));
28
+ for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) {
29
+ let t = S[start + i]; // s * n + x, at depth level + 1
30
+ let x = t % P.n;
31
+ let base = t - x; // s * n
32
+ var acc = 0.0;
33
+ for (var a = rowPtr[x]; a < rowPtr[x + 1u]; a = a + 1u) { // the in-arcs (v, x)
34
+ let v = base + colIdx[a];
35
+ if (depthK[v] == level) { acc = acc + bitcast<f32>(sigmaK[v]); } // every predecessor's paths, CSR order
36
+ }
37
+ let sigma = ldexp(acc, -shift);
38
+ let bits = bitcast<u32>(sigma);
39
+ sigmaK[t] = bits;
40
+ let exponent = (bits >> 23u) & 0xffu;
41
+ if (exponent == 0u || exponent == 0xffu) { atomicOr(&counters[27], 1u); } // out of f32's normal range
42
+ atomicMax(&levelMax[level + 1u], bits);
43
+ }
44
+ }
45
+ `;
@@ -3,12 +3,18 @@
3
3
  * twin of `bc-gather`, run once per batch after the backward sweep. One invocation per ARC (`P.count` arcs,
4
4
  * grid-stride): it finds the row `w` that owns the arc by an upper-bound search over `rowPtr`, then adds over the
5
5
  * batch's sources in order the term the backward pass summed -- `sigma[s][w] / sigma[s][v] * (1 + delta[s][v])`
6
- * whenever `w` was reached and `depth[s][v] == depth[s][w] + 1` -- into `arcScores[arc]`. Each arc is written by one
6
+ * whenever `w` was reached and `depth[s][v] == depth[s][w] + 1`, with `bc-backward`'s depth scale step under `SCALED` -- into
7
+ * `arcScores[arc]`. Each arc is written by one
7
8
  * invocation: no atomic, a fixed order. Arc-parallel rather than row-parallel so a hub row is not one lane's loop
8
9
  * (llvmpipe caps a shader loop at 65,535 iterations, and a 1,000-arc hub times 64 sources passed it). Body only;
9
10
  * the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
10
11
  */
11
12
  export const bcEdgeGatherWgsl = /* wgsl */ `
13
+ fn sigma_of(word: u32) -> f32 { // a stored count: u32, or f32 bits when SCALED
14
+ if (SCALED) { return bitcast<f32>(word); }
15
+ return f32(word);
16
+ }
17
+
12
18
  @compute @workgroup_size(WG)
13
19
  fn bc_edge_gather(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
14
20
  for (var a = linear_id(wid, lid.x); a < P.count; a = a + P.stride) {
@@ -26,7 +32,9 @@ fn bc_edge_gather(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
26
32
  let base = s * P.n;
27
33
  let dw = depthK[base + w];
28
34
  if (dw != INVALID_INDEX && depthK[base + nbr] == dw + 1u) { // (w, nbr) is on a shortest path from s
29
- acc = acc + (f32(sigmaK[base + w]) / f32(sigmaK[base + nbr])) * (1.0 + deltaK[base + nbr]);
35
+ let shift = select(0i, sigma_shift(levelMax[dw]), SCALED);
36
+ let ratio = ldexp(sigma_of(sigmaK[base + w]) / sigma_of(sigmaK[base + nbr]), -shift);
37
+ acc = acc + ratio * (1.0 + deltaK[base + nbr]);
30
38
  }
31
39
  }
32
40
  arcScores[a] = acc;
@@ -5,8 +5,9 @@
5
5
  * entries at depth `L` are `S[ends[L] .. ends[L + 1])`. That log is Brandes' stack, so the backward pass walks the
6
6
  * same ranges from the deepest level up and nothing is ever copied between levels.
7
7
  *
8
- * Role 1 seeds the batch: the k seed entries the host wrote into `S[0 .. k)` get depth 0 and one shortest path,
9
- * `ends[0] = 0`, the append cursor `stackTop` (counters word 26) starts at k, the overflow flag (word 27) is cleared,
8
+ * Role 1 seeds the batch: the k seed entries the host wrote into `S[0 .. k)` get depth 0 and one shortest path (the
9
+ * u32 1, or under `SCALED` the bits of the f32 1.0), `levelMax[0]` the bits of 1.0 (the largest count at depth 0,
10
+ * which `bc-count` scales depth 1 from), `ends[0] = 0`, the append cursor `stackTop` (counters word 26) starts at k, the overflow flag (word 27) is cleared,
10
11
  * `level` (word 11) is U32_MAX so the first boundary lands on 0, and `done` (word 15) is cleared.
11
12
  *
12
13
  * Role 0 is the level boundary, recorded before every forward level: it advances `level`, closes the level just
@@ -25,8 +26,9 @@ fn bc_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
25
26
  for (var i = 0u; i < P.k; i = i + 1u) {
26
27
  let t = S[i];
27
28
  depthK[t] = 0u; // the source is at depth 0
28
- sigmaK[t] = 1u; // with one shortest path, itself
29
+ sigmaK[t] = select(1u, bitcast<u32>(1.0), SCALED); // with one shortest path, itself (f32 bits when SCALED)
29
30
  }
31
+ levelMax[0] = bitcast<u32>(1.0); // the largest count at depth 0
30
32
  ends[0] = 0u;
31
33
  atomicStore(&counters[26], P.k); // stackTop: the seeds are the log's first k entries
32
34
  atomicStore(&counters[27], 0u); // sigmaOverflow
@@ -5,7 +5,7 @@
5
5
  * and source `s`, when `depth[s][u]` is the level the edge is relaxed toward `x` with EXACTLY the claim and the count
6
6
  * of `bc-forward` (the pre-check, `atomicMin`, the winner appends `s * n + x` to the same claim log, every arc on a
7
7
  * shortest path adds `sigma[s][u]`, a u32 wrap raises word 27); `UNDIRECTED` relaxes the other direction too, the
8
- * edge list holding each undirected edge once. So the two forward bodies are interchangeable level by level and the
8
+ * edge list holding each undirected edge once. Under `SCALED` the count is skipped, as in `bc-forward`. So the two forward bodies are interchangeable level by level and the
9
9
  * backward pass cannot tell which ran. A workgroup does nothing when the level is empty (the done boundary). The
10
10
  * winners of a strip are packed into the log with one global `atomicAdd` per strip; every barrier is in uniform
11
11
  * control flow (the loop bounds are uniforms and the workgroup id). Body only; the sabotage rows of
@@ -22,7 +22,7 @@ fn claim(x: u32, next: u32) -> bool {
22
22
  }
23
23
 
24
24
  fn count_paths(origin: u32, x: u32, next: u32) {
25
- if (atomicLoad(&depthK[x]) == next) { // every arc on a shortest path adds
25
+ if (!SCALED && atomicLoad(&depthK[x]) == next) { // every arc on a shortest path adds
26
26
  let add = atomicLoad(&sigmaK[origin]);
27
27
  let old = atomicAdd(&sigmaK[x], add);
28
28
  if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported
@@ -12,7 +12,9 @@
12
12
  * observes INVALID_INDEX the unique winner, which appends `t` to the log; and, as a SEPARATE condition, the COUNT:
13
13
  * every arc that reaches `t` at `level + 1` adds `sigma[s][u]` into `sigma[s][x]`, winner or not, which is what makes
14
14
  * sigma the number of shortest paths rather than of claims. The add detects a u32 wrap from `atomicAdd`'s return
15
- * value (`old + add < old`) and raises `sigmaOverflow` (counters word 27); the count is never clamped. The winners of
15
+ * value (`old + add < old`) and raises `sigmaOverflow` (counters word 27); the count is never clamped. Under
16
+ * `SCALED` (a batch rerun because the u32 counts wrapped) the count is skipped and `bc-count` pulls rescaled f32
17
+ * counts after the level instead. The winners of
16
18
  * a strip are packed into the log with one workgroup-memory counter and ONE global `atomicAdd` on `stackTop` (word
17
19
  * 26) per strip. Uniformity (spec 3.5 rule 1): the range and the aggregate are `workgroupUniformLoad`s and the strip
18
20
  * loop steps a uniform `p0`, so every barrier is in uniform control flow. The order of the log inside a level is
@@ -82,7 +84,7 @@ fn bc_forward(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_i
82
84
  if (atomicLoad(&depthK[x]) == INVALID_INDEX) { // the pre-check of design 16.1
83
85
  won = atomicMin(&depthK[x], next) == INVALID_INDEX; // the claim: the one winner appends
84
86
  }
85
- if (atomicLoad(&depthK[x]) == next) { // the count: EVERY arc on a shortest path adds
87
+ if (!SCALED && atomicLoad(&depthK[x]) == next) { // the count: EVERY arc on a shortest path adds
86
88
  let add = atomicLoad(&sigmaK[origin]);
87
89
  let old = atomicAdd(&sigmaK[x], add);
88
90
  if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported