@graphty/webgpu-graph-algorithms 0.6.13 → 0.6.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. package/README.md +52 -52
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-Bi6AhScG.js → context-oXphO3yj.js} +36 -28
  4. package/dist/chunks/context-oXphO3yj.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/accelerator.d.ts +3 -2
  7. package/dist/src/accelerator.d.ts.map +1 -1
  8. package/dist/src/accelerator.js +48 -4
  9. package/dist/src/accelerator.js.map +1 -1
  10. package/dist/src/algorithms/betweenness.d.ts +70 -0
  11. package/dist/src/algorithms/betweenness.d.ts.map +1 -0
  12. package/dist/src/algorithms/betweenness.js +538 -0
  13. package/dist/src/algorithms/betweenness.js.map +1 -0
  14. package/dist/src/algorithms/closeness.d.ts +15 -5
  15. package/dist/src/algorithms/closeness.d.ts.map +1 -1
  16. package/dist/src/algorithms/closeness.js +112 -26
  17. package/dist/src/algorithms/closeness.js.map +1 -1
  18. package/dist/src/constants.d.ts +8 -0
  19. package/dist/src/constants.d.ts.map +1 -1
  20. package/dist/src/constants.js +8 -0
  21. package/dist/src/constants.js.map +1 -1
  22. package/dist/src/index.d.ts +4 -2
  23. package/dist/src/index.d.ts.map +1 -1
  24. package/dist/src/index.js +1 -0
  25. package/dist/src/index.js.map +1 -1
  26. package/dist/src/kernels.d.ts +12 -6
  27. package/dist/src/kernels.d.ts.map +1 -1
  28. package/dist/src/kernels.js +153 -7
  29. package/dist/src/kernels.js.map +1 -1
  30. package/dist/src/primitives/frontier.d.ts +2 -0
  31. package/dist/src/primitives/frontier.d.ts.map +1 -1
  32. package/dist/src/primitives/frontier.js +2 -0
  33. package/dist/src/primitives/frontier.js.map +1 -1
  34. package/dist/src/types/accelerator.d.ts +11 -7
  35. package/dist/src/types/accelerator.d.ts.map +1 -1
  36. package/dist/src/types/algorithms.d.ts +4 -0
  37. package/dist/src/types/algorithms.d.ts.map +1 -1
  38. package/dist/src/types/betweenness.d.ts +35 -0
  39. package/dist/src/types/betweenness.d.ts.map +1 -0
  40. package/dist/src/types/betweenness.js +7 -0
  41. package/dist/src/types/betweenness.js.map +1 -0
  42. package/dist/src/wgsl/bc-backward.wgsl.d.ts +15 -0
  43. package/dist/src/wgsl/bc-backward.wgsl.d.ts.map +1 -0
  44. package/dist/src/wgsl/bc-backward.wgsl.js +34 -0
  45. package/dist/src/wgsl/bc-backward.wgsl.js.map +1 -0
  46. package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts +12 -0
  47. package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts.map +1 -0
  48. package/dist/src/wgsl/bc-edge-gather.wgsl.js +36 -0
  49. package/dist/src/wgsl/bc-edge-gather.wgsl.js.map +1 -0
  50. package/dist/src/wgsl/bc-finalize.wgsl.d.ts +21 -0
  51. package/dist/src/wgsl/bc-finalize.wgsl.d.ts.map +1 -0
  52. package/dist/src/wgsl/bc-finalize.wgsl.js +47 -0
  53. package/dist/src/wgsl/bc-finalize.wgsl.js.map +1 -0
  54. package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts +15 -0
  55. package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts.map +1 -0
  56. package/dist/src/wgsl/bc-forward-edge.wgsl.js +76 -0
  57. package/dist/src/wgsl/bc-forward-edge.wgsl.js.map +1 -0
  58. package/dist/src/wgsl/bc-forward.wgsl.d.ts +23 -0
  59. package/dist/src/wgsl/bc-forward.wgsl.d.ts.map +1 -0
  60. package/dist/src/wgsl/bc-forward.wgsl.js +106 -0
  61. package/dist/src/wgsl/bc-forward.wgsl.js.map +1 -0
  62. package/dist/src/wgsl/bc-gather.wgsl.d.ts +9 -0
  63. package/dist/src/wgsl/bc-gather.wgsl.d.ts.map +1 -0
  64. package/dist/src/wgsl/bc-gather.wgsl.js +20 -0
  65. package/dist/src/wgsl/bc-gather.wgsl.js.map +1 -0
  66. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +4 -1
  67. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -1
  68. package/dist/src/wgsl/closeness-reduce.wgsl.js +8 -4
  69. package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -1
  70. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +4 -2
  71. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -1
  72. package/dist/src/wgsl/closeness-sweep.wgsl.js +12 -2
  73. package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -1
  74. package/dist/webgpu-graph-algorithms.js +1037 -168
  75. package/dist/webgpu-graph-algorithms.js.map +1 -1
  76. package/package.json +5 -5
  77. package/src/accelerator.ts +75 -7
  78. package/src/algorithms/betweenness.ts +739 -0
  79. package/src/algorithms/closeness.ts +124 -32
  80. package/src/constants.ts +8 -0
  81. package/src/index.ts +8 -0
  82. package/src/kernels.ts +169 -10
  83. package/src/primitives/frontier.ts +4 -0
  84. package/src/types/accelerator.ts +18 -6
  85. package/src/types/algorithms.ts +5 -0
  86. package/src/types/betweenness.ts +38 -0
  87. package/src/wgsl/bc-backward.wgsl.ts +33 -0
  88. package/src/wgsl/bc-edge-gather.wgsl.ts +35 -0
  89. package/src/wgsl/bc-finalize.wgsl.ts +46 -0
  90. package/src/wgsl/bc-forward-edge.wgsl.ts +75 -0
  91. package/src/wgsl/bc-forward.wgsl.ts +105 -0
  92. package/src/wgsl/bc-gather.wgsl.ts +19 -0
  93. package/src/wgsl/closeness-reduce.wgsl.ts +8 -4
  94. package/src/wgsl/closeness-sweep.wgsl.ts +12 -2
  95. package/dist/chunks/context-Bi6AhScG.js.map +0 -1
@@ -0,0 +1,35 @@
1
+ /**
2
+ * The betweenness result records (spec 3.3 lines 833-834). Types only: this file imports nothing at runtime. The
3
+ * options are the CPU seam's own `BetweennessAcceleratorOptions` (`normalized`, `endpoints`, `sources`, `k`), as the
4
+ * traversals take the seam's option types.
5
+ */
6
+ import type { F32 } from "@graphty/graph-format";
7
+ import type { GpuScoresResult } from "./algorithms.js";
8
+ /**
9
+ * Vertex betweenness (spec 3.3 line 833). A SAMPLED run (`sources` or `k`) returns the UNSCALED sum over the sources
10
+ * actually run -- never extrapolated by `n / k` -- and `sourcesUsed` says how many that was; a caller who wants the
11
+ * estimator of the full sum multiplies by `n / sourcesUsed`.
12
+ * @public
13
+ */
14
+ export interface GpuBetweennessResult extends GpuScoresResult {
15
+ /** How many sources the scores sum over: `n` for an exact run, the sample size for a sampled one. */
16
+ readonly sourcesUsed: number;
17
+ /**
18
+ * True when some pair of vertices is joined by more than 2^32 shortest paths: the u32 path counts wrapped and the
19
+ * scores are WRONG, not approximate. Never clamped, never silent.
20
+ */
21
+ readonly sigmaOverflow: boolean;
22
+ }
23
+ /**
24
+ * Edge betweenness (spec 3.3 line 834): one score per logical edge (`edgeCount`), the per-arc scores folded with
25
+ * `foldArcs(s, perArc, "sum")` and halved on an undirected snapshot (the two arcs carry the pairs crossing the edge
26
+ * in each direction). `sourcesUsed` and `sigmaOverflow` mean what they mean on `GpuBetweennessResult`.
27
+ * @public
28
+ */
29
+ export interface GpuEdgeScoresResult {
30
+ readonly scores: F32;
31
+ readonly precision: "f32";
32
+ readonly sourcesUsed: number;
33
+ readonly sigmaOverflow: boolean;
34
+ }
35
+ //# sourceMappingURL=betweenness.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"betweenness.d.ts","sourceRoot":"","sources":["../../../src/types/betweenness.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAEH,OAAO,KAAK,EAAE,GAAG,EAAE,MAAM,uBAAuB,CAAC;AAEjD,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,iBAAiB,CAAC;AAEvD;;;;;GAKG;AACH,MAAM,WAAW,oBAAqB,SAAQ,eAAe;IACzD,qGAAqG;IACrG,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAC7B;;;OAGG;IACH,QAAQ,CAAC,aAAa,EAAE,OAAO,CAAC;CACnC;AAED;;;;;GAKG;AACH,MAAM,WAAW,mBAAmB;IAChC,QAAQ,CAAC,MAAM,EAAE,GAAG,CAAC;IACrB,QAAQ,CAAC,SAAS,EAAE,KAAK,CAAC;IAC1B,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAC7B,QAAQ,CAAC,aAAa,EAAE,OAAO,CAAC;CACnC"}
@@ -0,0 +1,7 @@
1
+ /**
2
+ * The betweenness result records (spec 3.3 lines 833-834). Types only: this file imports nothing at runtime. The
3
+ * options are the CPU seam's own `BetweennessAcceleratorOptions` (`normalized`, `endpoints`, `sources`, `k`), as the
4
+ * traversals take the seam's option types.
5
+ */
6
+ export {};
7
+ //# sourceMappingURL=betweenness.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"betweenness.js","sourceRoot":"","sources":["../../../src/types/betweenness.ts"],"names":[],"mappings":"AAAA;;;;GAIG"}
@@ -0,0 +1,15 @@
1
+ /**
2
+ * The `bc-backward` kernel body (design 8.4 "each (w, s) PULLS over its successors", 8.10 "BC backward (successor
3
+ * pull)"): one level of a betweenness batch's dependency accumulation, one invocation per log entry `(w, s)` of the
4
+ * level's range `S[P.start .. P.start + P.count)`, which the host planned from the `ends` it read back. Each entry
5
+ * walks the out-arcs of `w` -- the rows the forward pass expanded -- and sums `sigma[s][w] / sigma[s][v] * (1 +
6
+ * delta[s][v])` over the successors `v` (`depth[s][v] == depth[s][w] + 1`), whose dependencies the previous (deeper)
7
+ * dispatch wrote, then writes `delta[s][w]` ONCE. No float is ever accumulated through an atomic, so the result is
8
+ * bitwise reproducible. The sources (depth 0) are never dispatched: their dependency stays 0, which is what keeps a
9
+ * source's own dependency out of its score. Grid-stride loop (`P.stride`). ponytail: one lane walks one row, so a
10
+ * vertex of degree above 65,535 exceeds llvmpipe's per-invocation loop cap (CLAUDE.md, Verified Platform Facts) and
11
+ * would need a workgroup-per-row form there; hardware adapters have no such cap. Body only; the sabotage rows of
12
+ * test/helpers/sabotage.ts are textual edits of it.
13
+ */
14
+ export declare const bcBackwardWgsl = "\n@compute @workgroup_size(WG)\nfn bc_backward(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n for (var i = linear_id(wid, lid.x); i < P.count; i = i + P.stride) {\n let t = S[P.start + i]; // s * n + w\n let w = t % P.n;\n let base = t - w; // s * n\n let succ = depthK[t] + 1u;\n let sw = f32(sigmaK[t]);\n var acc = 0.0;\n for (var a = rowPtr[w]; a < rowPtr[w + 1u]; a = a + 1u) {\n let v = base + colIdx[a];\n if (depthK[v] == succ) { // v is a successor of w for source s\n acc = acc + (sw / f32(sigmaK[v])) * (1.0 + deltaK[v]);\n }\n }\n deltaK[t] = acc; // written once per (w, s)\n }\n}\n";
15
+ //# sourceMappingURL=bc-backward.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bc-backward.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-backward.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AACH,eAAO,MAAM,cAAc,m5BAmB1B,CAAC"}
@@ -0,0 +1,34 @@
1
+ /**
2
+ * The `bc-backward` kernel body (design 8.4 "each (w, s) PULLS over its successors", 8.10 "BC backward (successor
3
+ * pull)"): one level of a betweenness batch's dependency accumulation, one invocation per log entry `(w, s)` of the
4
+ * level's range `S[P.start .. P.start + P.count)`, which the host planned from the `ends` it read back. Each entry
5
+ * walks the out-arcs of `w` -- the rows the forward pass expanded -- and sums `sigma[s][w] / sigma[s][v] * (1 +
6
+ * delta[s][v])` over the successors `v` (`depth[s][v] == depth[s][w] + 1`), whose dependencies the previous (deeper)
7
+ * dispatch wrote, then writes `delta[s][w]` ONCE. No float is ever accumulated through an atomic, so the result is
8
+ * bitwise reproducible. The sources (depth 0) are never dispatched: their dependency stays 0, which is what keeps a
9
+ * source's own dependency out of its score. Grid-stride loop (`P.stride`). ponytail: one lane walks one row, so a
10
+ * vertex of degree above 65,535 exceeds llvmpipe's per-invocation loop cap (CLAUDE.md, Verified Platform Facts) and
11
+ * would need a workgroup-per-row form there; hardware adapters have no such cap. Body only; the sabotage rows of
12
+ * test/helpers/sabotage.ts are textual edits of it.
13
+ */
14
+ export const bcBackwardWgsl = /* wgsl */ `
15
+ @compute @workgroup_size(WG)
16
+ fn bc_backward(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
17
+ for (var i = linear_id(wid, lid.x); i < P.count; i = i + P.stride) {
18
+ let t = S[P.start + i]; // s * n + w
19
+ let w = t % P.n;
20
+ let base = t - w; // s * n
21
+ let succ = depthK[t] + 1u;
22
+ let sw = f32(sigmaK[t]);
23
+ var acc = 0.0;
24
+ for (var a = rowPtr[w]; a < rowPtr[w + 1u]; a = a + 1u) {
25
+ let v = base + colIdx[a];
26
+ if (depthK[v] == succ) { // v is a successor of w for source s
27
+ acc = acc + (sw / f32(sigmaK[v])) * (1.0 + deltaK[v]);
28
+ }
29
+ }
30
+ deltaK[t] = acc; // written once per (w, s)
31
+ }
32
+ }
33
+ `;
34
+ //# sourceMappingURL=bc-backward.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bc-backward.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-backward.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AACH,MAAM,CAAC,MAAM,cAAc,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;CAmBxC,CAAC"}
@@ -0,0 +1,12 @@
1
+ /**
2
+ * The `bc-edge-gather` kernel body (design 8.4 "edge BC accumulates per arc from the same n x k deltas"): the per-arc
3
+ * twin of `bc-gather`, run once per batch after the backward sweep. One invocation per ARC (`P.count` arcs,
4
+ * grid-stride): it finds the row `w` that owns the arc by an upper-bound search over `rowPtr`, then adds over the
5
+ * batch's sources in order the term the backward pass summed -- `sigma[s][w] / sigma[s][v] * (1 + delta[s][v])`
6
+ * whenever `w` was reached and `depth[s][v] == depth[s][w] + 1` -- into `arcScores[arc]`. Each arc is written by one
7
+ * invocation: no atomic, a fixed order. Arc-parallel rather than row-parallel so a hub row is not one lane's loop
8
+ * (llvmpipe caps a shader loop at 65,535 iterations, and a 1,000-arc hub times 64 sources passed it). Body only;
9
+ * the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
10
+ */
11
+ export declare const bcEdgeGatherWgsl = "\n@compute @workgroup_size(WG)\nfn bc_edge_gather(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n for (var a = linear_id(wid, lid.x); a < P.count; a = a + P.stride) {\n var lo = 0u; // the row w with rowPtr[w] <= a < rowPtr[w + 1]\n var hi = P.n;\n loop {\n if (lo >= hi) { break; }\n let mid = (lo + hi) / 2u;\n if (rowPtr[mid + 1u] <= a) { lo = mid + 1u; } else { hi = mid; }\n }\n let w = lo;\n let nbr = colIdx[a];\n var acc = arcScores[a];\n for (var s = 0u; s < P.k; s = s + 1u) {\n let base = s * P.n;\n let dw = depthK[base + w];\n if (dw != INVALID_INDEX && depthK[base + nbr] == dw + 1u) { // (w, nbr) is on a shortest path from s\n acc = acc + (f32(sigmaK[base + w]) / f32(sigmaK[base + nbr])) * (1.0 + deltaK[base + nbr]);\n }\n }\n arcScores[a] = acc;\n }\n}\n";
12
+ //# sourceMappingURL=bc-edge-gather.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bc-edge-gather.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-edge-gather.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AACH,eAAO,MAAM,gBAAgB,6gCAwB5B,CAAC"}
@@ -0,0 +1,36 @@
1
+ /**
2
+ * The `bc-edge-gather` kernel body (design 8.4 "edge BC accumulates per arc from the same n x k deltas"): the per-arc
3
+ * twin of `bc-gather`, run once per batch after the backward sweep. One invocation per ARC (`P.count` arcs,
4
+ * grid-stride): it finds the row `w` that owns the arc by an upper-bound search over `rowPtr`, then adds over the
5
+ * batch's sources in order the term the backward pass summed -- `sigma[s][w] / sigma[s][v] * (1 + delta[s][v])`
6
+ * whenever `w` was reached and `depth[s][v] == depth[s][w] + 1` -- into `arcScores[arc]`. Each arc is written by one
7
+ * invocation: no atomic, a fixed order. Arc-parallel rather than row-parallel so a hub row is not one lane's loop
8
+ * (llvmpipe caps a shader loop at 65,535 iterations, and a 1,000-arc hub times 64 sources passed it). Body only;
9
+ * the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
10
+ */
11
+ export const bcEdgeGatherWgsl = /* wgsl */ `
12
+ @compute @workgroup_size(WG)
13
+ fn bc_edge_gather(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
14
+ for (var a = linear_id(wid, lid.x); a < P.count; a = a + P.stride) {
15
+ var lo = 0u; // the row w with rowPtr[w] <= a < rowPtr[w + 1]
16
+ var hi = P.n;
17
+ loop {
18
+ if (lo >= hi) { break; }
19
+ let mid = (lo + hi) / 2u;
20
+ if (rowPtr[mid + 1u] <= a) { lo = mid + 1u; } else { hi = mid; }
21
+ }
22
+ let w = lo;
23
+ let nbr = colIdx[a];
24
+ var acc = arcScores[a];
25
+ for (var s = 0u; s < P.k; s = s + 1u) {
26
+ let base = s * P.n;
27
+ let dw = depthK[base + w];
28
+ if (dw != INVALID_INDEX && depthK[base + nbr] == dw + 1u) { // (w, nbr) is on a shortest path from s
29
+ acc = acc + (f32(sigmaK[base + w]) / f32(sigmaK[base + nbr])) * (1.0 + deltaK[base + nbr]);
30
+ }
31
+ }
32
+ arcScores[a] = acc;
33
+ }
34
+ }
35
+ `;
36
+ //# sourceMappingURL=bc-edge-gather.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bc-edge-gather.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-edge-gather.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;CAwB1C,CAAC"}
@@ -0,0 +1,21 @@
1
+ /**
2
+ * The `bc-finalize` kernel body (design 8.4, 5.4): the one-lane bookkeeping of a betweenness source batch, two roles
3
+ * by `P.role`. The batch keeps ONE append-only claim log `S` -- every `(vertex, source)` pair the forward pass
4
+ * claims, packed as the index `s * n + v` into the `n x k` arrays -- and `ends`, the level boundaries into it: the
5
+ * entries at depth `L` are `S[ends[L] .. ends[L + 1])`. That log is Brandes' stack, so the backward pass walks the
6
+ * same ranges from the deepest level up and nothing is ever copied between levels.
7
+ *
8
+ * Role 1 seeds the batch: the k seed entries the host wrote into `S[0 .. k)` get depth 0 and one shortest path,
9
+ * `ends[0] = 0`, the append cursor `stackTop` (counters word 26) starts at k, the overflow flag (word 27) is cleared,
10
+ * `level` (word 11) is U32_MAX so the first boundary lands on 0, and `done` (word 15) is cleared.
11
+ *
12
+ * Role 0 is the level boundary, recorded before every forward level: it advances `level`, closes the level just
13
+ * claimed by writing `ends[level + 1] = stackTop`, publishes the new level's size in `frontierCount` (word 0) and sets
14
+ * `done` when that level is empty. A boundary that finds `done` set moves nothing, so the levels the host records
15
+ * past the end are no-ops. The forward kernels dispatch directly and read their range from `ends` (no indirect
16
+ * dispatch: design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md). One lane; no barrier follows the
17
+ * early return of the others (spec 3.5 rule 1). Body only; the sabotage rows of test/helpers/sabotage.ts are
18
+ * textual edits of it.
19
+ */
20
+ export declare const bcFinalizeWgsl = "\n@compute @workgroup_size(WG)\nfn bc_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role == 1u) { // the seed of a batch\n for (var i = 0u; i < P.k; i = i + 1u) {\n let t = S[i];\n depthK[t] = 0u; // the source is at depth 0\n sigmaK[t] = 1u; // with one shortest path, itself\n }\n ends[0] = 0u;\n atomicStore(&counters[26], P.k); // stackTop: the seeds are the log's first k entries\n atomicStore(&counters[27], 0u); // sigmaOverflow\n atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0\n atomicStore(&counters[15], 0u); // done\n return;\n }\n if (atomicLoad(&counters[15]) != 0u) { return; } // done: a no-op level the host recorded past the end\n let level = atomicLoad(&counters[11]) + 1u;\n let top = atomicLoad(&counters[26]);\n ends[level + 1u] = top; // the level's entries end where the log ends now\n let count = top - ends[level];\n atomicStore(&counters[0], count); // frontierCount (the inspect seam reads it)\n atomicStore(&counters[11], level);\n atomicStore(&counters[15], select(0u, 1u, count == 0u)); // an empty level ends the batch\n}\n";
21
+ //# sourceMappingURL=bc-finalize.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bc-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;GAkBG;AACH,eAAO,MAAM,cAAc,+oDA0B1B,CAAC"}
@@ -0,0 +1,47 @@
1
+ /**
2
+ * The `bc-finalize` kernel body (design 8.4, 5.4): the one-lane bookkeeping of a betweenness source batch, two roles
3
+ * by `P.role`. The batch keeps ONE append-only claim log `S` -- every `(vertex, source)` pair the forward pass
4
+ * claims, packed as the index `s * n + v` into the `n x k` arrays -- and `ends`, the level boundaries into it: the
5
+ * entries at depth `L` are `S[ends[L] .. ends[L + 1])`. That log is Brandes' stack, so the backward pass walks the
6
+ * same ranges from the deepest level up and nothing is ever copied between levels.
7
+ *
8
+ * Role 1 seeds the batch: the k seed entries the host wrote into `S[0 .. k)` get depth 0 and one shortest path,
9
+ * `ends[0] = 0`, the append cursor `stackTop` (counters word 26) starts at k, the overflow flag (word 27) is cleared,
10
+ * `level` (word 11) is U32_MAX so the first boundary lands on 0, and `done` (word 15) is cleared.
11
+ *
12
+ * Role 0 is the level boundary, recorded before every forward level: it advances `level`, closes the level just
13
+ * claimed by writing `ends[level + 1] = stackTop`, publishes the new level's size in `frontierCount` (word 0) and sets
14
+ * `done` when that level is empty. A boundary that finds `done` set moves nothing, so the levels the host records
15
+ * past the end are no-ops. The forward kernels dispatch directly and read their range from `ends` (no indirect
16
+ * dispatch: design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md). One lane; no barrier follows the
17
+ * early return of the others (spec 3.5 rule 1). Body only; the sabotage rows of test/helpers/sabotage.ts are
18
+ * textual edits of it.
19
+ */
20
+ export const bcFinalizeWgsl = /* wgsl */ `
21
+ @compute @workgroup_size(WG)
22
+ fn bc_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
23
+ if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
24
+ if (P.role == 1u) { // the seed of a batch
25
+ for (var i = 0u; i < P.k; i = i + 1u) {
26
+ let t = S[i];
27
+ depthK[t] = 0u; // the source is at depth 0
28
+ sigmaK[t] = 1u; // with one shortest path, itself
29
+ }
30
+ ends[0] = 0u;
31
+ atomicStore(&counters[26], P.k); // stackTop: the seeds are the log's first k entries
32
+ atomicStore(&counters[27], 0u); // sigmaOverflow
33
+ atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0
34
+ atomicStore(&counters[15], 0u); // done
35
+ return;
36
+ }
37
+ if (atomicLoad(&counters[15]) != 0u) { return; } // done: a no-op level the host recorded past the end
38
+ let level = atomicLoad(&counters[11]) + 1u;
39
+ let top = atomicLoad(&counters[26]);
40
+ ends[level + 1u] = top; // the level's entries end where the log ends now
41
+ let count = top - ends[level];
42
+ atomicStore(&counters[0], count); // frontierCount (the inspect seam reads it)
43
+ atomicStore(&counters[11], level);
44
+ atomicStore(&counters[15], select(0u, 1u, count == 0u)); // an empty level ends the batch
45
+ }
46
+ `;
47
+ //# sourceMappingURL=bc-finalize.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bc-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;GAkBG;AACH,MAAM,CAAC,MAAM,cAAc,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;CA0BxC,CAAC"}
@@ -0,0 +1,15 @@
1
+ /**
2
+ * The `bc-forward-edge` kernel body (design 8.4 "the McLaughlin-Bader online switch to the edge-parallel form", 8.8
3
+ * row 7): one level of a betweenness batch's forward pass, edge-parallel -- every logical edge of the `edgeList` view
4
+ * for every source of the batch, grid-striding over the edges inside a loop over the k sources. For the edge `(u, x)`
5
+ * and source `s`, when `depth[s][u]` is the level the edge is relaxed toward `x` with EXACTLY the claim and the count
6
+ * of `bc-forward` (the pre-check, `atomicMin`, the winner appends `s * n + x` to the same claim log, every arc on a
7
+ * shortest path adds `sigma[s][u]`, a u32 wrap raises word 27); `UNDIRECTED` relaxes the other direction too, the
8
+ * edge list holding each undirected edge once. So the two forward bodies are interchangeable level by level and the
9
+ * backward pass cannot tell which ran. A workgroup does nothing when the level is empty (the done boundary). The
10
+ * winners of a strip are packed into the log with one global `atomicAdd` per strip; every barrier is in uniform
11
+ * control flow (the loop bounds are uniforms and the workgroup id). Body only; the sabotage rows of
12
+ * test/helpers/sabotage.ts are textual edits of it.
13
+ */
14
+ export declare const bcForwardEdgeWgsl = "\nvar<workgroup> wlive: u32; // 1 when the level has entries\nvar<workgroup> wwon: atomic<u32>; // the strip's winners\nvar<workgroup> wbase: u32; // where the strip's winners go in the log\n\nfn claim(x: u32, next: u32) -> bool {\n if (atomicLoad(&depthK[x]) != INVALID_INDEX) { return false; } // the pre-check of design 16.1\n return atomicMin(&depthK[x], next) == INVALID_INDEX;\n}\n\nfn count_paths(origin: u32, x: u32, next: u32) {\n if (atomicLoad(&depthK[x]) == next) { // every arc on a shortest path adds\n let add = atomicLoad(&sigmaK[origin]);\n let old = atomicAdd(&sigmaK[x], add);\n if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported\n }\n}\n\n@compute @workgroup_size(WG)\nfn bc_forward_edge(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let level = atomicLoad(&counters[11]);\n if (lid.x == 0u) { wlive = select(0u, 1u, ends[level + 1u] > ends[level]); }\n if (workgroupUniformLoad(&wlive) == 0u) { return; } // uniform: nothing below runs on an empty level\n let next = level + 1u;\n for (var s = 0u; s < P.k; s = s + 1u) {\n let base = s * P.n;\n for (var e0 = group_id(wid) * WG; e0 < P.count; e0 = e0 + P.stride) { // grid-stride over the edges\n let e = e0 + lid.x;\n var a = INVALID_INDEX; // the claims this lane won\n var b = INVALID_INDEX;\n if (e < P.count) {\n let u = base + edgeSrc[e];\n let x = base + edgeDst[e];\n if (atomicLoad(&depthK[u]) == level) {\n if (claim(x, next)) { a = x; }\n count_paths(u, x, next);\n }\n if (UNDIRECTED) { // the other direction of an undirected edge\n if (atomicLoad(&depthK[x]) == level) {\n if (claim(u, next)) { b = u; }\n count_paths(x, u, next);\n }\n }\n }\n let mine = select(0u, 1u, a != INVALID_INDEX) + select(0u, 1u, b != INVALID_INDEX);\n var slot = 0u;\n if (mine != 0u) { slot = atomicAdd(&wwon, mine); }\n workgroupBarrier();\n if (lid.x == 0u) {\n wbase = atomicAdd(&counters[26], atomicLoad(&wwon)); // stackTop: one global atomic per strip\n atomicStore(&wwon, 0u);\n }\n workgroupBarrier();\n if (a != INVALID_INDEX) {\n S[wbase + slot] = a;\n slot = slot + 1u;\n }\n if (b != INVALID_INDEX) { S[wbase + slot] = b; }\n }\n }\n}\n";
15
+ //# sourceMappingURL=bc-forward-edge.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bc-forward-edge.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-forward-edge.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AACH,eAAO,MAAM,iBAAiB,m0FA6D7B,CAAC"}
@@ -0,0 +1,76 @@
1
+ /**
2
+ * The `bc-forward-edge` kernel body (design 8.4 "the McLaughlin-Bader online switch to the edge-parallel form", 8.8
3
+ * row 7): one level of a betweenness batch's forward pass, edge-parallel -- every logical edge of the `edgeList` view
4
+ * for every source of the batch, grid-striding over the edges inside a loop over the k sources. For the edge `(u, x)`
5
+ * and source `s`, when `depth[s][u]` is the level the edge is relaxed toward `x` with EXACTLY the claim and the count
6
+ * of `bc-forward` (the pre-check, `atomicMin`, the winner appends `s * n + x` to the same claim log, every arc on a
7
+ * shortest path adds `sigma[s][u]`, a u32 wrap raises word 27); `UNDIRECTED` relaxes the other direction too, the
8
+ * edge list holding each undirected edge once. So the two forward bodies are interchangeable level by level and the
9
+ * backward pass cannot tell which ran. A workgroup does nothing when the level is empty (the done boundary). The
10
+ * winners of a strip are packed into the log with one global `atomicAdd` per strip; every barrier is in uniform
11
+ * control flow (the loop bounds are uniforms and the workgroup id). Body only; the sabotage rows of
12
+ * test/helpers/sabotage.ts are textual edits of it.
13
+ */
14
+ export const bcForwardEdgeWgsl = /* wgsl */ `
15
+ var<workgroup> wlive: u32; // 1 when the level has entries
16
+ var<workgroup> wwon: atomic<u32>; // the strip's winners
17
+ var<workgroup> wbase: u32; // where the strip's winners go in the log
18
+
19
+ fn claim(x: u32, next: u32) -> bool {
20
+ if (atomicLoad(&depthK[x]) != INVALID_INDEX) { return false; } // the pre-check of design 16.1
21
+ return atomicMin(&depthK[x], next) == INVALID_INDEX;
22
+ }
23
+
24
+ fn count_paths(origin: u32, x: u32, next: u32) {
25
+ if (atomicLoad(&depthK[x]) == next) { // every arc on a shortest path adds
26
+ let add = atomicLoad(&sigmaK[origin]);
27
+ let old = atomicAdd(&sigmaK[x], add);
28
+ if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported
29
+ }
30
+ }
31
+
32
+ @compute @workgroup_size(WG)
33
+ fn bc_forward_edge(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
34
+ let level = atomicLoad(&counters[11]);
35
+ if (lid.x == 0u) { wlive = select(0u, 1u, ends[level + 1u] > ends[level]); }
36
+ if (workgroupUniformLoad(&wlive) == 0u) { return; } // uniform: nothing below runs on an empty level
37
+ let next = level + 1u;
38
+ for (var s = 0u; s < P.k; s = s + 1u) {
39
+ let base = s * P.n;
40
+ for (var e0 = group_id(wid) * WG; e0 < P.count; e0 = e0 + P.stride) { // grid-stride over the edges
41
+ let e = e0 + lid.x;
42
+ var a = INVALID_INDEX; // the claims this lane won
43
+ var b = INVALID_INDEX;
44
+ if (e < P.count) {
45
+ let u = base + edgeSrc[e];
46
+ let x = base + edgeDst[e];
47
+ if (atomicLoad(&depthK[u]) == level) {
48
+ if (claim(x, next)) { a = x; }
49
+ count_paths(u, x, next);
50
+ }
51
+ if (UNDIRECTED) { // the other direction of an undirected edge
52
+ if (atomicLoad(&depthK[x]) == level) {
53
+ if (claim(u, next)) { b = u; }
54
+ count_paths(x, u, next);
55
+ }
56
+ }
57
+ }
58
+ let mine = select(0u, 1u, a != INVALID_INDEX) + select(0u, 1u, b != INVALID_INDEX);
59
+ var slot = 0u;
60
+ if (mine != 0u) { slot = atomicAdd(&wwon, mine); }
61
+ workgroupBarrier();
62
+ if (lid.x == 0u) {
63
+ wbase = atomicAdd(&counters[26], atomicLoad(&wwon)); // stackTop: one global atomic per strip
64
+ atomicStore(&wwon, 0u);
65
+ }
66
+ workgroupBarrier();
67
+ if (a != INVALID_INDEX) {
68
+ S[wbase + slot] = a;
69
+ slot = slot + 1u;
70
+ }
71
+ if (b != INVALID_INDEX) { S[wbase + slot] = b; }
72
+ }
73
+ }
74
+ }
75
+ `;
76
+ //# sourceMappingURL=bc-forward-edge.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bc-forward-edge.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-forward-edge.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AACH,MAAM,CAAC,MAAM,iBAAiB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA6D3C,CAAC"}
@@ -0,0 +1,23 @@
1
+ /**
2
+ * The `bc-forward` kernel body (design 8.4 "forward pass = BFS with sigma as array<atomic<u32>>", 8.10 "BC forward
3
+ * (tagged)", 16.1): one level of the tagged multi-source breadth-first search of a betweenness batch. The level's
4
+ * frontier is the range `S[ends[level] .. ends[level + 1])` of the claim log, every entry a packed `s * n + u`; the
5
+ * expansion is `closeness-sweep`'s block-mapped strip (each workgroup loads up to `WG` entries, scans their degrees
6
+ * in workgroup memory, and every lane strips the aggregate by an upper-bound binary search), fused with the claim, so
7
+ * no edge queue exists.
8
+ *
9
+ * For the arc `(u, x)` of the entry `(u, s)`, with `t = s * n + x`: the CLAIM -- a relaxed pre-check that skips the
10
+ * atomic when `t` is already claimed (design 16.1: a stale "unclaimed" costs one redundant atomic, a stale "claimed"
11
+ * cannot happen because a claim is never revoked), then `atomicMin(&depthK[t], level + 1)`, the invocation that
12
+ * observes INVALID_INDEX the unique winner, which appends `t` to the log; and, as a SEPARATE condition, the COUNT:
13
+ * every arc that reaches `t` at `level + 1` adds `sigma[s][u]` into `sigma[s][x]`, winner or not, which is what makes
14
+ * sigma the number of shortest paths rather than of claims. The add detects a u32 wrap from `atomicAdd`'s return
15
+ * value (`old + add < old`) and raises `sigmaOverflow` (counters word 27); the count is never clamped. The winners of
16
+ * a strip are packed into the log with one workgroup-memory counter and ONE global `atomicAdd` on `stackTop` (word
17
+ * 26) per strip. Uniformity (spec 3.5 rule 1): the range and the aggregate are `workgroupUniformLoad`s and the strip
18
+ * loop steps a uniform `p0`, so every barrier is in uniform control flow. The order of the log inside a level is
19
+ * not deterministic; nothing downstream depends on it (the counts are integers and every dependency is written by
20
+ * index). Body only; the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
21
+ */
22
+ export declare const bcForwardWgsl = "\nvar<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan\nvar<workgroup> rowStart: array<u32, WG>; // the first arc of each entry's row\nvar<workgroup> entryOf: array<u32, WG>; // each entry, s * n + u\nvar<workgroup> wstart: u32; // the level's first log index\nvar<workgroup> wcount: u32; // the level's entry count\nvar<workgroup> wwon: atomic<u32>; // the strip's winners\nvar<workgroup> wbase: u32; // where the strip's winners go in the log\n\n@compute @workgroup_size(WG)\nfn bc_forward(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let level = atomicLoad(&counters[11]);\n if (lid.x == 0u) {\n let lo = ends[level];\n wstart = lo;\n wcount = ends[level + 1u] - lo;\n }\n let start = workgroupUniformLoad(&wstart);\n let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers\n let next = level + 1u;\n for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries\n let i = b0 + lid.x;\n var deg = 0u;\n var first = 0u;\n var entry = 0u;\n if (i < count) { // guarded loads into locals (3.5 rule 1)\n entry = S[start + i];\n let u = entry % P.n;\n first = rowPtr[u];\n deg = rowPtr[u + 1u] - first;\n }\n sh[lid.x] = deg;\n rowStart[lid.x] = first;\n entryOf[lid.x] = entry;\n workgroupBarrier();\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier\n for (var p0 = 0u; p0 < aggregate; p0 = p0 + WG) { // strip [0, aggregate) WG arcs at a time\n let p = p0 + lid.x;\n var won = false;\n var claimed = 0u;\n if (p < aggregate) {\n var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p\n var hi = WG;\n loop {\n if (lo >= hi) { break; }\n let mid = (lo + hi) / 2u;\n if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }\n }\n let k = lo;\n var exclusive = 0u;\n if (k > 0u) { exclusive = sh[k - 1u]; }\n let origin = entryOf[k]; // s * n + u\n let x = (origin - (origin % P.n)) + colIdx[rowStart[k] + (p - exclusive)]; // s * n + x\n if (atomicLoad(&depthK[x]) == INVALID_INDEX) { // the pre-check of design 16.1\n won = atomicMin(&depthK[x], next) == INVALID_INDEX; // the claim: the one winner appends\n }\n if (atomicLoad(&depthK[x]) == next) { // the count: EVERY arc on a shortest path adds\n let add = atomicLoad(&sigmaK[origin]);\n let old = atomicAdd(&sigmaK[x], add);\n if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported\n }\n claimed = x;\n }\n var slot = 0u;\n if (won) { slot = atomicAdd(&wwon, 1u); }\n workgroupBarrier();\n if (lid.x == 0u) {\n wbase = atomicAdd(&counters[26], atomicLoad(&wwon)); // stackTop: one global atomic per strip\n atomicStore(&wwon, 0u);\n }\n workgroupBarrier();\n if (won) { S[wbase + slot] = claimed; }\n }\n workgroupBarrier(); // sh, rowStart and entryOf are reused by the next block\n }\n}\n";
23
+ //# sourceMappingURL=bc-forward.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bc-forward.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-forward.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,eAAO,MAAM,aAAa,+nIAmFzB,CAAC"}
@@ -0,0 +1,106 @@
1
+ /**
2
+ * The `bc-forward` kernel body (design 8.4 "forward pass = BFS with sigma as array<atomic<u32>>", 8.10 "BC forward
3
+ * (tagged)", 16.1): one level of the tagged multi-source breadth-first search of a betweenness batch. The level's
4
+ * frontier is the range `S[ends[level] .. ends[level + 1])` of the claim log, every entry a packed `s * n + u`; the
5
+ * expansion is `closeness-sweep`'s block-mapped strip (each workgroup loads up to `WG` entries, scans their degrees
6
+ * in workgroup memory, and every lane strips the aggregate by an upper-bound binary search), fused with the claim, so
7
+ * no edge queue exists.
8
+ *
9
+ * For the arc `(u, x)` of the entry `(u, s)`, with `t = s * n + x`: the CLAIM -- a relaxed pre-check that skips the
10
+ * atomic when `t` is already claimed (design 16.1: a stale "unclaimed" costs one redundant atomic, a stale "claimed"
11
+ * cannot happen because a claim is never revoked), then `atomicMin(&depthK[t], level + 1)`, the invocation that
12
+ * observes INVALID_INDEX the unique winner, which appends `t` to the log; and, as a SEPARATE condition, the COUNT:
13
+ * every arc that reaches `t` at `level + 1` adds `sigma[s][u]` into `sigma[s][x]`, winner or not, which is what makes
14
+ * sigma the number of shortest paths rather than of claims. The add detects a u32 wrap from `atomicAdd`'s return
15
+ * value (`old + add < old`) and raises `sigmaOverflow` (counters word 27); the count is never clamped. The winners of
16
+ * a strip are packed into the log with one workgroup-memory counter and ONE global `atomicAdd` on `stackTop` (word
17
+ * 26) per strip. Uniformity (spec 3.5 rule 1): the range and the aggregate are `workgroupUniformLoad`s and the strip
18
+ * loop steps a uniform `p0`, so every barrier is in uniform control flow. The order of the log inside a level is
19
+ * not deterministic; nothing downstream depends on it (the counts are integers and every dependency is written by
20
+ * index). Body only; the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
21
+ */
22
+ export const bcForwardWgsl = /* wgsl */ `
23
+ var<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan
24
+ var<workgroup> rowStart: array<u32, WG>; // the first arc of each entry's row
25
+ var<workgroup> entryOf: array<u32, WG>; // each entry, s * n + u
26
+ var<workgroup> wstart: u32; // the level's first log index
27
+ var<workgroup> wcount: u32; // the level's entry count
28
+ var<workgroup> wwon: atomic<u32>; // the strip's winners
29
+ var<workgroup> wbase: u32; // where the strip's winners go in the log
30
+
31
+ @compute @workgroup_size(WG)
32
+ fn bc_forward(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
33
+ let level = atomicLoad(&counters[11]);
34
+ if (lid.x == 0u) {
35
+ let lo = ends[level];
36
+ wstart = lo;
37
+ wcount = ends[level + 1u] - lo;
38
+ }
39
+ let start = workgroupUniformLoad(&wstart);
40
+ let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
41
+ let next = level + 1u;
42
+ for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
43
+ let i = b0 + lid.x;
44
+ var deg = 0u;
45
+ var first = 0u;
46
+ var entry = 0u;
47
+ if (i < count) { // guarded loads into locals (3.5 rule 1)
48
+ entry = S[start + i];
49
+ let u = entry % P.n;
50
+ first = rowPtr[u];
51
+ deg = rowPtr[u + 1u] - first;
52
+ }
53
+ sh[lid.x] = deg;
54
+ rowStart[lid.x] = first;
55
+ entryOf[lid.x] = entry;
56
+ workgroupBarrier();
57
+ for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees
58
+ var t = 0u;
59
+ if (lid.x >= s) { t = sh[lid.x - s]; }
60
+ workgroupBarrier();
61
+ sh[lid.x] = sh[lid.x] + t;
62
+ workgroupBarrier();
63
+ }
64
+ let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier
65
+ for (var p0 = 0u; p0 < aggregate; p0 = p0 + WG) { // strip [0, aggregate) WG arcs at a time
66
+ let p = p0 + lid.x;
67
+ var won = false;
68
+ var claimed = 0u;
69
+ if (p < aggregate) {
70
+ var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p
71
+ var hi = WG;
72
+ loop {
73
+ if (lo >= hi) { break; }
74
+ let mid = (lo + hi) / 2u;
75
+ if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }
76
+ }
77
+ let k = lo;
78
+ var exclusive = 0u;
79
+ if (k > 0u) { exclusive = sh[k - 1u]; }
80
+ let origin = entryOf[k]; // s * n + u
81
+ let x = (origin - (origin % P.n)) + colIdx[rowStart[k] + (p - exclusive)]; // s * n + x
82
+ if (atomicLoad(&depthK[x]) == INVALID_INDEX) { // the pre-check of design 16.1
83
+ won = atomicMin(&depthK[x], next) == INVALID_INDEX; // the claim: the one winner appends
84
+ }
85
+ if (atomicLoad(&depthK[x]) == next) { // the count: EVERY arc on a shortest path adds
86
+ let add = atomicLoad(&sigmaK[origin]);
87
+ let old = atomicAdd(&sigmaK[x], add);
88
+ if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported
89
+ }
90
+ claimed = x;
91
+ }
92
+ var slot = 0u;
93
+ if (won) { slot = atomicAdd(&wwon, 1u); }
94
+ workgroupBarrier();
95
+ if (lid.x == 0u) {
96
+ wbase = atomicAdd(&counters[26], atomicLoad(&wwon)); // stackTop: one global atomic per strip
97
+ atomicStore(&wwon, 0u);
98
+ }
99
+ workgroupBarrier();
100
+ if (won) { S[wbase + slot] = claimed; }
101
+ }
102
+ workgroupBarrier(); // sh, rowStart and entryOf are reused by the next block
103
+ }
104
+ }
105
+ `;
106
+ //# sourceMappingURL=bc-forward.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bc-forward.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-forward.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,MAAM,CAAC,MAAM,aAAa,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAmFvC,CAAC"}
@@ -0,0 +1,9 @@
1
+ /**
2
+ * The `bc-gather` kernel body (design 8.4 "one gather kernel after the batch's last backward level", 8.10 "BC
3
+ * gather"): one invocation per vertex `w`, `bc[w] += delta[0][w] + ... + delta[k - 1][w]` in source order -- k reads,
4
+ * one write, no atomic, a fixed summation order, so the scores are bitwise reproducible across runs and adapters.
5
+ * A source's own dependency is 0 (the backward pass never visits depth 0), so nothing is skipped here. Grid-stride
6
+ * loop (`P.stride`). Body only; the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
7
+ */
8
+ export declare const bcGatherWgsl = "\n@compute @workgroup_size(WG)\nfn bc_gather(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n for (var w = linear_id(wid, lid.x); w < P.n; w = w + P.stride) {\n var acc = bc[w];\n for (var s = 0u; s < P.k; s = s + 1u) {\n acc = acc + deltaK[s * P.n + w];\n }\n bc[w] = acc;\n }\n}\n";
9
+ //# sourceMappingURL=bc-gather.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bc-gather.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bc-gather.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,eAAO,MAAM,YAAY,oXAWxB,CAAC"}
@@ -0,0 +1,20 @@
1
+ /**
2
+ * The `bc-gather` kernel body (design 8.4 "one gather kernel after the batch's last backward level", 8.10 "BC
3
+ * gather"): one invocation per vertex `w`, `bc[w] += delta[0][w] + ... + delta[k - 1][w]` in source order -- k reads,
4
+ * one write, no atomic, a fixed summation order, so the scores are bitwise reproducible across runs and adapters.
5
+ * A source's own dependency is 0 (the backward pass never visits depth 0), so nothing is skipped here. Grid-stride
6
+ * loop (`P.stride`). Body only; the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
7
+ */
8
+ export const bcGatherWgsl = /* wgsl */ `
9
+ @compute @workgroup_size(WG)
10
+ fn bc_gather(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
11
+ for (var w = linear_id(wid, lid.x); w < P.n; w = w + P.stride) {
12
+ var acc = bc[w];
13
+ for (var s = 0u; s < P.k; s = s + 1u) {
14
+ acc = acc + deltaK[s * P.n + w];
15
+ }
16
+ bc[w] = acc;
17
+ }
18
+ }
19
+ `;
20
+ //# sourceMappingURL=bc-gather.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bc-gather.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bc-gather.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,MAAM,CAAC,MAAM,YAAY,GAAG,UAAU,CAAC;;;;;;;;;;;CAWtC,CAAC"}
@@ -10,8 +10,11 @@
10
10
  * level 0's parity is 0), `flags[source_s] = 1` (level 0's `compact` turns the flags into the list; a seeded list
11
11
  * would be overwritten by a compaction of all-zero flags), `counters[0] = k` (not done) and `level = U32_MAX` (so
12
12
  * level 0's boundary accumulates nothing and brings the word to 0, and level 1's counts the distance-1 claims at 1).
13
+ * Role 2 is role 1 for a sampled run: source `s` of the batch is word `P.source + s` of the source list the host
14
+ * wrote after the per-node sums (`perSource[128 + P.bitsBase + ...]`), `P.n` is the list's length, and a node listed
15
+ * twice in one batch carries both bits (the seed ORs, so a duplicate runs twice).
13
16
  * No barrier follows the early return of the other lanes (3.5 rule 1). Body only (spec 3.5, D9); the text is
14
17
  * normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
15
18
  */
16
- export declare const closenessReduceWgsl = "\n@compute @workgroup_size(WG)\nfn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role == 1u) { // the seed of a batch: P.source is its first source\n let k = min(32u, P.n - P.source);\n for (var s = 0u; s < k; s = s + 1u) {\n let v = P.source + s;\n let bit = 1u << s;\n bits[v] = bit; // visited\n bits[P.bitsBase + v] = bit; // the frontier level 0 reads (region 1: level 0's parity is 0)\n bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list\n }\n atomicStore(&counters[0], k); // not done\n atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0\n atomicStore(&counters[15], 0u); // done\n return;\n }\n // role 0: the level boundary -- done from the previous level's compacted count, then the accumulation\n let count = atomicLoad(&counters[0]);\n atomicStore(&counters[15], select(0u, 1u, count == 0u));\n let level = atomicLoad(&counters[11]);\n let d = level + 1u; // the distance of the claims the level just run made\n for (var s = 0u; s < 32u; s = s + 1u) {\n let c = atomicLoad(&perSource[s]); // newCount[s]\n atomicStore(&perSource[32u + s], atomicLoad(&perSource[32u + s]) + c); // reached[s]\n // sum[s] += c x d in 64 bits: the 16-bit split product (pLo, pHi), then the add with its carry\n let cLo = c & 0xFFFFu;\n let cHi = c >> 16u;\n let dLo = d & 0xFFFFu;\n let dHi = d >> 16u;\n let ll = cLo * dLo;\n let lh = cLo * dHi;\n let hl = cHi * dLo;\n let mid = (ll >> 16u) + (lh & 0xFFFFu) + (hl & 0xFFFFu);\n let pLo = (ll & 0xFFFFu) | ((mid & 0xFFFFu) << 16u);\n let pHi = (cHi * dHi) + (lh >> 16u) + (hl >> 16u) + (mid >> 16u);\n var lo = atomicLoad(&perSource[64u + s]); // sumLo[s]\n var hi = atomicLoad(&perSource[96u + s]); // sumHi[s]\n let before = lo;\n lo = lo + pLo;\n hi = hi + pHi + select(0u, 1u, lo < before);\n atomicStore(&perSource[64u + s], lo);\n atomicStore(&perSource[96u + s], hi);\n atomicStore(&perSource[s], 0u);\n }\n atomicStore(&counters[11], level + 1u);\n}\n";
19
+ export declare const closenessReduceWgsl = "\n@compute @workgroup_size(WG)\nfn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role != 0u) { // the seed of a batch: P.source is its first source\n let k = min(32u, P.n - P.source);\n for (var s = 0u; s < k; s = s + 1u) {\n var v = P.source + s;\n if (P.role == 2u) { v = atomicLoad(&perSource[128u + P.bitsBase + P.source + s]); } // a sampled run's list\n let bit = 1u << s;\n bits[v] = bits[v] | bit; // visited\n bits[P.bitsBase + v] = bits[P.bitsBase + v] | bit; // the frontier level 0 reads (region 1: level 0's parity is 0)\n bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list\n }\n atomicStore(&counters[0], k); // not done\n atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0\n atomicStore(&counters[15], 0u); // done\n return;\n }\n // role 0: the level boundary -- done from the previous level's compacted count, then the accumulation\n let count = atomicLoad(&counters[0]);\n atomicStore(&counters[15], select(0u, 1u, count == 0u));\n let level = atomicLoad(&counters[11]);\n let d = level + 1u; // the distance of the claims the level just run made\n for (var s = 0u; s < 32u; s = s + 1u) {\n let c = atomicLoad(&perSource[s]); // newCount[s]\n atomicStore(&perSource[32u + s], atomicLoad(&perSource[32u + s]) + c); // reached[s]\n // sum[s] += c x d in 64 bits: the 16-bit split product (pLo, pHi), then the add with its carry\n let cLo = c & 0xFFFFu;\n let cHi = c >> 16u;\n let dLo = d & 0xFFFFu;\n let dHi = d >> 16u;\n let ll = cLo * dLo;\n let lh = cLo * dHi;\n let hl = cHi * dLo;\n let mid = (ll >> 16u) + (lh & 0xFFFFu) + (hl & 0xFFFFu);\n let pLo = (ll & 0xFFFFu) | ((mid & 0xFFFFu) << 16u);\n let pHi = (cHi * dHi) + (lh >> 16u) + (hl >> 16u) + (mid >> 16u);\n var lo = atomicLoad(&perSource[64u + s]); // sumLo[s]\n var hi = atomicLoad(&perSource[96u + s]); // sumHi[s]\n let before = lo;\n lo = lo + pLo;\n hi = hi + pHi + select(0u, 1u, lo < before);\n atomicStore(&perSource[64u + s], lo);\n atomicStore(&perSource[96u + s], hi);\n atomicStore(&perSource[s], 0u);\n }\n atomicStore(&counters[11], level + 1u);\n}\n";
17
20
  //# sourceMappingURL=closeness-reduce.wgsl.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"closeness-reduce.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-reduce.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AACH,eAAO,MAAM,mBAAmB,0qFAgD/B,CAAC"}
1
+ {"version":3,"file":"closeness-reduce.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-reduce.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,eAAO,MAAM,mBAAmB,qyFAiD/B,CAAC"}
@@ -10,6 +10,9 @@
10
10
  * level 0's parity is 0), `flags[source_s] = 1` (level 0's `compact` turns the flags into the list; a seeded list
11
11
  * would be overwritten by a compaction of all-zero flags), `counters[0] = k` (not done) and `level = U32_MAX` (so
12
12
  * level 0's boundary accumulates nothing and brings the word to 0, and level 1's counts the distance-1 claims at 1).
13
+ * Role 2 is role 1 for a sampled run: source `s` of the batch is word `P.source + s` of the source list the host
14
+ * wrote after the per-node sums (`perSource[128 + P.bitsBase + ...]`), `P.n` is the list's length, and a node listed
15
+ * twice in one batch carries both bits (the seed ORs, so a duplicate runs twice).
13
16
  * No barrier follows the early return of the other lanes (3.5 rule 1). Body only (spec 3.5, D9); the text is
14
17
  * normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
15
18
  */
@@ -17,13 +20,14 @@ export const closenessReduceWgsl = /* wgsl */ `
17
20
  @compute @workgroup_size(WG)
18
21
  fn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {
19
22
  if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
20
- if (P.role == 1u) { // the seed of a batch: P.source is its first source
23
+ if (P.role != 0u) { // the seed of a batch: P.source is its first source
21
24
  let k = min(32u, P.n - P.source);
22
25
  for (var s = 0u; s < k; s = s + 1u) {
23
- let v = P.source + s;
26
+ var v = P.source + s;
27
+ if (P.role == 2u) { v = atomicLoad(&perSource[128u + P.bitsBase + P.source + s]); } // a sampled run's list
24
28
  let bit = 1u << s;
25
- bits[v] = bit; // visited
26
- bits[P.bitsBase + v] = bit; // the frontier level 0 reads (region 1: level 0's parity is 0)
29
+ bits[v] = bits[v] | bit; // visited
30
+ bits[P.bitsBase + v] = bits[P.bitsBase + v] | bit; // the frontier level 0 reads (region 1: level 0's parity is 0)
27
31
  bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list
28
32
  }
29
33
  atomicStore(&counters[0], k); // not done