@graphty/webgpu-graph-algorithms 0.6.26 → 0.6.28

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/README.md +56 -6
  2. package/dist/acquire.d.ts +2 -0
  3. package/dist/browser.js +18 -1
  4. package/dist/browser.js.map +1 -1
  5. package/dist/chunks/accelerator-B-FjQwaA.js +19173 -0
  6. package/dist/chunks/accelerator-B-FjQwaA.js.map +1 -0
  7. package/dist/chunks/managed-D_GdQtnu.js +98 -0
  8. package/dist/chunks/managed-D_GdQtnu.js.map +1 -0
  9. package/dist/node.js +18 -1
  10. package/dist/node.js.map +1 -1
  11. package/dist/src/accelerator.d.ts.map +1 -1
  12. package/dist/src/accelerator.js +5 -3
  13. package/dist/src/accelerator.js.map +1 -1
  14. package/dist/src/algorithms/all-pairs.d.ts.map +1 -1
  15. package/dist/src/algorithms/all-pairs.js +72 -47
  16. package/dist/src/algorithms/all-pairs.js.map +1 -1
  17. package/dist/src/algorithms/betweenness.d.ts +1 -1
  18. package/dist/src/algorithms/betweenness.js +2 -2
  19. package/dist/src/algorithms/closeness.d.ts +45 -42
  20. package/dist/src/algorithms/closeness.d.ts.map +1 -1
  21. package/dist/src/algorithms/closeness.js +295 -226
  22. package/dist/src/algorithms/closeness.js.map +1 -1
  23. package/dist/src/browser/index.d.ts +10 -0
  24. package/dist/src/browser/index.d.ts.map +1 -1
  25. package/dist/src/browser/index.js +22 -0
  26. package/dist/src/browser/index.js.map +1 -1
  27. package/dist/src/constants.d.ts +13 -3
  28. package/dist/src/constants.d.ts.map +1 -1
  29. package/dist/src/constants.js +13 -3
  30. package/dist/src/constants.js.map +1 -1
  31. package/dist/src/kernels.d.ts +14 -4
  32. package/dist/src/kernels.d.ts.map +1 -1
  33. package/dist/src/kernels.js +62 -28
  34. package/dist/src/kernels.js.map +1 -1
  35. package/dist/src/layouts/force-simulation.d.ts.map +1 -1
  36. package/dist/src/layouts/force-simulation.js +0 -1
  37. package/dist/src/layouts/force-simulation.js.map +1 -1
  38. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  39. package/dist/src/layouts/forceatlas2.js +0 -1
  40. package/dist/src/layouts/forceatlas2.js.map +1 -1
  41. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  42. package/dist/src/layouts/fruchterman-reingold.js +0 -1
  43. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  44. package/dist/src/layouts/repulsion-grid.d.ts +1 -3
  45. package/dist/src/layouts/repulsion-grid.d.ts.map +1 -1
  46. package/dist/src/layouts/repulsion-grid.js +1 -6
  47. package/dist/src/layouts/repulsion-grid.js.map +1 -1
  48. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  49. package/dist/src/layouts/spring-electrical.js +0 -1
  50. package/dist/src/layouts/spring-electrical.js.map +1 -1
  51. package/dist/src/managed.d.ts +11 -0
  52. package/dist/src/managed.d.ts.map +1 -0
  53. package/dist/src/managed.js +129 -0
  54. package/dist/src/managed.js.map +1 -0
  55. package/dist/src/node/index.d.ts +11 -0
  56. package/dist/src/node/index.d.ts.map +1 -1
  57. package/dist/src/node/index.js +21 -0
  58. package/dist/src/node/index.js.map +1 -1
  59. package/dist/src/primitives/grid-pyramid.d.ts +16 -15
  60. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  61. package/dist/src/primitives/grid-pyramid.js +20 -28
  62. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  63. package/dist/src/types/accelerator.d.ts +2 -0
  64. package/dist/src/types/accelerator.d.ts.map +1 -1
  65. package/dist/src/types/managed.d.ts +81 -0
  66. package/dist/src/types/managed.d.ts.map +1 -0
  67. package/dist/src/types/managed.js +7 -0
  68. package/dist/src/types/managed.js.map +1 -0
  69. package/dist/src/wgsl/bc-forward.wgsl.d.ts +1 -1
  70. package/dist/src/wgsl/bc-forward.wgsl.js +1 -1
  71. package/dist/src/wgsl/closeness-level.wgsl.d.ts +37 -0
  72. package/dist/src/wgsl/closeness-level.wgsl.d.ts.map +1 -0
  73. package/dist/src/wgsl/closeness-level.wgsl.js +204 -0
  74. package/dist/src/wgsl/closeness-level.wgsl.js.map +1 -0
  75. package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts +11 -0
  76. package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts.map +1 -0
  77. package/dist/src/wgsl/closeness-rowsum.wgsl.js +42 -0
  78. package/dist/src/wgsl/closeness-rowsum.wgsl.js.map +1 -0
  79. package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +2 -2
  80. package/dist/src/wgsl/grid-centroid-hub.wgsl.js +2 -2
  81. package/dist/webgpu-graph-algorithms.js +142 -15586
  82. package/dist/webgpu-graph-algorithms.js.map +1 -1
  83. package/package.json +10 -4
  84. package/src/accelerator.ts +5 -3
  85. package/src/algorithms/all-pairs.ts +86 -56
  86. package/src/algorithms/betweenness.ts +2 -2
  87. package/src/algorithms/closeness.ts +353 -256
  88. package/src/browser/index.ts +37 -0
  89. package/src/constants.ts +13 -3
  90. package/src/kernels.ts +65 -36
  91. package/src/layouts/force-simulation.ts +0 -1
  92. package/src/layouts/forceatlas2.ts +0 -1
  93. package/src/layouts/fruchterman-reingold.ts +0 -1
  94. package/src/layouts/repulsion-grid.ts +2 -7
  95. package/src/layouts/spring-electrical.ts +0 -1
  96. package/src/managed.ts +172 -0
  97. package/src/node/index.ts +36 -0
  98. package/src/primitives/grid-pyramid.ts +29 -41
  99. package/src/types/accelerator.ts +2 -0
  100. package/src/types/managed.ts +86 -0
  101. package/src/wgsl/bc-forward.wgsl.ts +1 -1
  102. package/src/wgsl/closeness-level.wgsl.ts +203 -0
  103. package/src/wgsl/closeness-rowsum.wgsl.ts +41 -0
  104. package/src/wgsl/grid-centroid-hub.wgsl.ts +2 -2
  105. package/dist/chunks/context-BZY6SMsM.js +0 -3615
  106. package/dist/chunks/context-BZY6SMsM.js.map +0 -1
  107. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +0 -20
  108. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +0 -1
  109. package/dist/src/wgsl/closeness-reduce.wgsl.js +0 -69
  110. package/dist/src/wgsl/closeness-reduce.wgsl.js.map +0 -1
  111. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +0 -22
  112. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +0 -1
  113. package/dist/src/wgsl/closeness-sweep.wgsl.js +0 -106
  114. package/dist/src/wgsl/closeness-sweep.wgsl.js.map +0 -1
  115. package/src/wgsl/closeness-reduce.wgsl.ts +0 -68
  116. package/src/wgsl/closeness-sweep.wgsl.ts +0 -105
@@ -1,20 +0,0 @@
1
- /**
2
- * The `closeness-reduce` kernel body (design 8.4, 9.7 "integer distances before division"; P8-T11, the P8 plan's
3
- * PD-13): the one-lane bookkeeping of the bit-parallel sweep, two roles by `P.role`. Role 0 is the level boundary,
4
- * recorded BEFORE the level's `compact` and sweep: `done` is whether the previous level's compacted count is 0 (so the
5
- * traversal ends one level after the last claim, one empty sweep and no wrong sum), then for every source `s` the
6
- * claims the level just run made (`newCount[s]`) are folded into `reached[s]` and into the 64-bit `sum[s]` at the
7
- * distance `level + 1` -- the product as a 16-bit split into a low and a high word, then the add with its carry --
8
- * `newCount[s]` is zeroed and `level` advances. Role 1 is the seed of a batch: for the batch's `k = min(32, n -
9
- * P.source)` sources, bit `s` into `visited[source_s]` and into the frontier region level 0 reads (region 1, since
10
- * level 0's parity is 0), `flags[source_s] = 1` (level 0's `compact` turns the flags into the list; a seeded list
11
- * would be overwritten by a compaction of all-zero flags), `counters[0] = k` (not done) and `level = U32_MAX` (so
12
- * level 0's boundary accumulates nothing and brings the word to 0, and level 1's counts the distance-1 claims at 1).
13
- * Role 2 is role 1 for a sampled run: source `s` of the batch is word `P.source + s` of the source list the host
14
- * wrote after the per-node sums (`perSource[128 + P.bitsBase + ...]`), `P.n` is the list's length, and a node listed
15
- * twice in one batch carries both bits (the seed ORs, so a duplicate runs twice).
16
- * No barrier follows the early return of the other lanes (3.5 rule 1). Body only (spec 3.5, D9); the text is
17
- * normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
18
- */
19
- export declare const closenessReduceWgsl = "\n@compute @workgroup_size(WG)\nfn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role != 0u) { // the seed of a batch: P.source is its first source\n let k = min(32u, P.n - P.source);\n for (var s = 0u; s < k; s = s + 1u) {\n var v = P.source + s;\n if (P.role == 2u) { v = atomicLoad(&perSource[128u + P.bitsBase + P.source + s]); } // a sampled run's list\n let bit = 1u << s;\n bits[v] = bits[v] | bit; // visited\n bits[P.bitsBase + v] = bits[P.bitsBase + v] | bit; // the frontier level 0 reads (region 1: level 0's parity is 0)\n bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list\n }\n atomicStore(&counters[0], k); // not done\n atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0\n atomicStore(&counters[15], 0u); // done\n return;\n }\n // role 0: the level boundary -- done from the previous level's compacted count, then the accumulation\n let count = atomicLoad(&counters[0]);\n atomicStore(&counters[15], select(0u, 1u, count == 0u));\n let level = atomicLoad(&counters[11]);\n let d = level + 1u; // the distance of the claims the level just run made\n for (var s = 0u; s < 32u; s = s + 1u) {\n let c = atomicLoad(&perSource[s]); // newCount[s]\n atomicStore(&perSource[32u + s], atomicLoad(&perSource[32u + s]) + c); // reached[s]\n // sum[s] += c x d in 64 bits: the 16-bit split product (pLo, pHi), then the add with its carry\n let cLo = c & 0xFFFFu;\n let cHi = c >> 16u;\n let dLo = d & 0xFFFFu;\n let dHi = d >> 16u;\n let ll = cLo * dLo;\n let lh = cLo * dHi;\n let hl = cHi * dLo;\n let mid = (ll >> 16u) + (lh & 0xFFFFu) + (hl & 0xFFFFu);\n let pLo = (ll & 0xFFFFu) | ((mid & 0xFFFFu) << 16u);\n let pHi = (cHi * dHi) + (lh >> 16u) + (hl >> 16u) + (mid >> 16u);\n var lo = atomicLoad(&perSource[64u + s]); // sumLo[s]\n var hi = atomicLoad(&perSource[96u + s]); // sumHi[s]\n let before = lo;\n lo = lo + pLo;\n hi = hi + pHi + select(0u, 1u, lo < before);\n atomicStore(&perSource[64u + s], lo);\n atomicStore(&perSource[96u + s], hi);\n atomicStore(&perSource[s], 0u);\n }\n atomicStore(&counters[11], level + 1u);\n}\n";
20
- //# sourceMappingURL=closeness-reduce.wgsl.d.ts.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"closeness-reduce.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-reduce.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,eAAO,MAAM,mBAAmB,qyFAiD/B,CAAC"}
@@ -1,69 +0,0 @@
1
- /**
2
- * The `closeness-reduce` kernel body (design 8.4, 9.7 "integer distances before division"; P8-T11, the P8 plan's
3
- * PD-13): the one-lane bookkeeping of the bit-parallel sweep, two roles by `P.role`. Role 0 is the level boundary,
4
- * recorded BEFORE the level's `compact` and sweep: `done` is whether the previous level's compacted count is 0 (so the
5
- * traversal ends one level after the last claim, one empty sweep and no wrong sum), then for every source `s` the
6
- * claims the level just run made (`newCount[s]`) are folded into `reached[s]` and into the 64-bit `sum[s]` at the
7
- * distance `level + 1` -- the product as a 16-bit split into a low and a high word, then the add with its carry --
8
- * `newCount[s]` is zeroed and `level` advances. Role 1 is the seed of a batch: for the batch's `k = min(32, n -
9
- * P.source)` sources, bit `s` into `visited[source_s]` and into the frontier region level 0 reads (region 1, since
10
- * level 0's parity is 0), `flags[source_s] = 1` (level 0's `compact` turns the flags into the list; a seeded list
11
- * would be overwritten by a compaction of all-zero flags), `counters[0] = k` (not done) and `level = U32_MAX` (so
12
- * level 0's boundary accumulates nothing and brings the word to 0, and level 1's counts the distance-1 claims at 1).
13
- * Role 2 is role 1 for a sampled run: source `s` of the batch is word `P.source + s` of the source list the host
14
- * wrote after the per-node sums (`perSource[128 + P.bitsBase + ...]`), `P.n` is the list's length, and a node listed
15
- * twice in one batch carries both bits (the seed ORs, so a duplicate runs twice).
16
- * No barrier follows the early return of the other lanes (3.5 rule 1). Body only (spec 3.5, D9); the text is
17
- * normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
18
- */
19
- export const closenessReduceWgsl = /* wgsl */ `
20
- @compute @workgroup_size(WG)
21
- fn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {
22
- if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
23
- if (P.role != 0u) { // the seed of a batch: P.source is its first source
24
- let k = min(32u, P.n - P.source);
25
- for (var s = 0u; s < k; s = s + 1u) {
26
- var v = P.source + s;
27
- if (P.role == 2u) { v = atomicLoad(&perSource[128u + P.bitsBase + P.source + s]); } // a sampled run's list
28
- let bit = 1u << s;
29
- bits[v] = bits[v] | bit; // visited
30
- bits[P.bitsBase + v] = bits[P.bitsBase + v] | bit; // the frontier level 0 reads (region 1: level 0's parity is 0)
31
- bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list
32
- }
33
- atomicStore(&counters[0], k); // not done
34
- atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0
35
- atomicStore(&counters[15], 0u); // done
36
- return;
37
- }
38
- // role 0: the level boundary -- done from the previous level's compacted count, then the accumulation
39
- let count = atomicLoad(&counters[0]);
40
- atomicStore(&counters[15], select(0u, 1u, count == 0u));
41
- let level = atomicLoad(&counters[11]);
42
- let d = level + 1u; // the distance of the claims the level just run made
43
- for (var s = 0u; s < 32u; s = s + 1u) {
44
- let c = atomicLoad(&perSource[s]); // newCount[s]
45
- atomicStore(&perSource[32u + s], atomicLoad(&perSource[32u + s]) + c); // reached[s]
46
- // sum[s] += c x d in 64 bits: the 16-bit split product (pLo, pHi), then the add with its carry
47
- let cLo = c & 0xFFFFu;
48
- let cHi = c >> 16u;
49
- let dLo = d & 0xFFFFu;
50
- let dHi = d >> 16u;
51
- let ll = cLo * dLo;
52
- let lh = cLo * dHi;
53
- let hl = cHi * dLo;
54
- let mid = (ll >> 16u) + (lh & 0xFFFFu) + (hl & 0xFFFFu);
55
- let pLo = (ll & 0xFFFFu) | ((mid & 0xFFFFu) << 16u);
56
- let pHi = (cHi * dHi) + (lh >> 16u) + (hl >> 16u) + (mid >> 16u);
57
- var lo = atomicLoad(&perSource[64u + s]); // sumLo[s]
58
- var hi = atomicLoad(&perSource[96u + s]); // sumHi[s]
59
- let before = lo;
60
- lo = lo + pLo;
61
- hi = hi + pHi + select(0u, 1u, lo < before);
62
- atomicStore(&perSource[64u + s], lo);
63
- atomicStore(&perSource[96u + s], hi);
64
- atomicStore(&perSource[s], 0u);
65
- }
66
- atomicStore(&counters[11], level + 1u);
67
- }
68
- `;
69
- //# sourceMappingURL=closeness-reduce.wgsl.js.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"closeness-reduce.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/closeness-reduce.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,CAAC,MAAM,mBAAmB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAiD7C,CAAC"}
@@ -1,22 +0,0 @@
1
- /**
2
- * The `closeness-sweep` kernel body (design 8.4 "32 sources per u32 word"; P8-T11, the P8 plan's PD-13 / DEP-P8-E):
3
- * one level of the bit-parallel multi-source breadth-first search. The batch's state is one `bits` buffer of four
4
- * regions of `P.bitsBase` words each -- `visited` at 0, the two frontier regions at `P.bitsBase` and `2 x P.bitsBase`
5
- * (which one is the frontier and which the next swaps by the level's parity, `P.mode`, so nothing is copied between
6
- * levels), `flags` at `3 x P.bitsBase` -- with bit `s` of word `v` meaning "source `s` has reached / is at / is next
7
- * at `v`". The expansion is `advance-expand`'s block-mapped strip (P8-T5): each workgroup loads up to `WG` entries of
8
- * the frontier LIST (the vertices any source is at, compacted from the flags by the host's `compact`), scans their
9
- * degrees with the inlined Hillis-Steele scan of `bfs-contract` (this is not a twin kernel: `needs: []`), and every
10
- * invocation strips the aggregate by binary search. The claim is inline: for the arc `(u, x)` the mask is the sources
11
- * at `u` that have not reached `x`; `atomicOr` on `visited[x]` returns the bits this lane won (`fresh`), which go into
12
- * the next region, set `flags[x]` (the flags region is part of the one atomic binding, so a plain store is a compile
13
- * error), and are tallied per source in WORKGROUP memory -- one global `atomicAdd` per source per workgroup after the
14
- * strip loop, never one per arc. Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan and the
15
- * `workgroupUniformLoad` sit unconditionally after the guard, the strip loop is bounded by the uniform aggregate, and
16
- * the flush's barrier follows it in uniform control flow. A sampled run (`P.perNode == 1`) also adds each claim's
17
- * distance (`level + 1`, one per won bit) into the per-node sum of `x`, `perSource[128 + x]`: the distance from each
18
- * of the batch's sources TO `x`, which is what a node's sampled closeness sums on an undirected graph. Body only (spec 3.5, D9); the text is normative: the
19
- * sabotage rows of test/helpers/sabotage.ts are textual edits of it.
20
- */
21
- export declare const closenessSweepWgsl = "\nvar<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan\nvar<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row\nvar<workgroup> rowOf: array<u32, WG>; // the frontier vertex of each entry (the source end of its arcs)\nvar<workgroup> local: array<atomic<u32>, 32>; // this workgroup's fresh claims per source\nvar<workgroup> wcount: u32; // the frontier list's length\nvar<workgroup> wdist: u32; // the distance of this level's claims\n\n@compute @workgroup_size(WG)\nfn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x == 0u) {\n wcount = atomicLoad(&counters[0]); // the frontier list's length (compact's total)\n wdist = atomicLoad(&counters[11]) + 1u; // the level word: this level claims at level + 1\n }\n let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers\n let dist = workgroupUniformLoad(&wdist);\n let nextBase = select(2u * P.bitsBase, P.bitsBase, P.mode == 1u); // the region that is next this level\n let frontierBase = 3u * P.bitsBase - nextBase; // the other one: the region that is the frontier\n for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries\n let i = b0 + lid.x; // this lane's frontier entry\n var deg = 0u;\n var start = 0u;\n var u = 0u;\n if (i < count) { // guarded loads into locals (3.5 rule 1)\n u = frontierList[i];\n let lo = max(rowPtr[u], P.arcBase);\n let hi = min(rowPtr[u + 1u], P.arcEnd);\n start = lo;\n deg = select(0u, hi - lo, hi > lo);\n }\n if (lid.x < 32u) { atomicStore(&local[lid.x], 0u); } // zeroed before the strip loop (WebGPU zero-initialises workgroup memory; said anyway)\n sh[lid.x] = deg;\n rowStart[lid.x] = start;\n rowOf[lid.x] = u;\n workgroupBarrier();\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees (bfs-contract's, inlined)\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier\n for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...\n var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p\n var hi = WG;\n loop {\n if (lo >= hi) { break; }\n let mid = (lo + hi) / 2u;\n if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }\n }\n let k = lo;\n var exclusive = 0u;\n if (k > 0u) { exclusive = sh[k - 1u]; }\n let arc = rowStart[k] + (p - exclusive);\n let x = colIdx[arc - P.arcBase];\n let mask = atomicLoad(&bits[frontierBase + rowOf[k]]) & ~atomicLoad(&bits[x]); // the sources at u that have not reached x\n if (mask != 0u) {\n let old = atomicOr(&bits[x], mask); // visited: the claim, one read-modify-write\n let fresh = mask & ~old; // the sources whose claim this lane won\n if (fresh != 0u) {\n atomicOr(&bits[nextBase + x], fresh);\n atomicStore(&bits[3u * P.bitsBase + x], 1u); // flags: x is in the next frontier list (compact reads it)\n if (P.perNode == 1u) { // a sampled run: x's distance to each source won\n atomicAdd(&perSource[128u + x], countOneBits(fresh) * dist);\n }\n var b = fresh;\n loop { // one tally per set bit of fresh\n if (b == 0u) { break; }\n let s = firstTrailingBit(b);\n atomicAdd(&local[s], 1u);\n b = b & (b - 1u);\n }\n }\n }\n }\n workgroupBarrier(); // uniform: the loop's bound is the uniform aggregate\n if (lid.x < 32u) { // ONE global atomic per source per workgroup\n let c = atomicLoad(&local[lid.x]);\n if (c != 0u) { atomicAdd(&perSource[lid.x], c); } // newCount[s]\n }\n workgroupBarrier(); // sh, rowStart, rowOf and local are reused by the next block\n }\n}\n";
22
- //# sourceMappingURL=closeness-sweep.wgsl.d.ts.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"closeness-sweep.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-sweep.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AACH,eAAO,MAAM,kBAAkB,inKAoF9B,CAAC"}
@@ -1,106 +0,0 @@
1
- /**
2
- * The `closeness-sweep` kernel body (design 8.4 "32 sources per u32 word"; P8-T11, the P8 plan's PD-13 / DEP-P8-E):
3
- * one level of the bit-parallel multi-source breadth-first search. The batch's state is one `bits` buffer of four
4
- * regions of `P.bitsBase` words each -- `visited` at 0, the two frontier regions at `P.bitsBase` and `2 x P.bitsBase`
5
- * (which one is the frontier and which the next swaps by the level's parity, `P.mode`, so nothing is copied between
6
- * levels), `flags` at `3 x P.bitsBase` -- with bit `s` of word `v` meaning "source `s` has reached / is at / is next
7
- * at `v`". The expansion is `advance-expand`'s block-mapped strip (P8-T5): each workgroup loads up to `WG` entries of
8
- * the frontier LIST (the vertices any source is at, compacted from the flags by the host's `compact`), scans their
9
- * degrees with the inlined Hillis-Steele scan of `bfs-contract` (this is not a twin kernel: `needs: []`), and every
10
- * invocation strips the aggregate by binary search. The claim is inline: for the arc `(u, x)` the mask is the sources
11
- * at `u` that have not reached `x`; `atomicOr` on `visited[x]` returns the bits this lane won (`fresh`), which go into
12
- * the next region, set `flags[x]` (the flags region is part of the one atomic binding, so a plain store is a compile
13
- * error), and are tallied per source in WORKGROUP memory -- one global `atomicAdd` per source per workgroup after the
14
- * strip loop, never one per arc. Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan and the
15
- * `workgroupUniformLoad` sit unconditionally after the guard, the strip loop is bounded by the uniform aggregate, and
16
- * the flush's barrier follows it in uniform control flow. A sampled run (`P.perNode == 1`) also adds each claim's
17
- * distance (`level + 1`, one per won bit) into the per-node sum of `x`, `perSource[128 + x]`: the distance from each
18
- * of the batch's sources TO `x`, which is what a node's sampled closeness sums on an undirected graph. Body only (spec 3.5, D9); the text is normative: the
19
- * sabotage rows of test/helpers/sabotage.ts are textual edits of it.
20
- */
21
- export const closenessSweepWgsl = /* wgsl */ `
22
- var<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan
23
- var<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row
24
- var<workgroup> rowOf: array<u32, WG>; // the frontier vertex of each entry (the source end of its arcs)
25
- var<workgroup> local: array<atomic<u32>, 32>; // this workgroup's fresh claims per source
26
- var<workgroup> wcount: u32; // the frontier list's length
27
- var<workgroup> wdist: u32; // the distance of this level's claims
28
-
29
- @compute @workgroup_size(WG)
30
- fn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
31
- if (lid.x == 0u) {
32
- wcount = atomicLoad(&counters[0]); // the frontier list's length (compact's total)
33
- wdist = atomicLoad(&counters[11]) + 1u; // the level word: this level claims at level + 1
34
- }
35
- let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
36
- let dist = workgroupUniformLoad(&wdist);
37
- let nextBase = select(2u * P.bitsBase, P.bitsBase, P.mode == 1u); // the region that is next this level
38
- let frontierBase = 3u * P.bitsBase - nextBase; // the other one: the region that is the frontier
39
- for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
40
- let i = b0 + lid.x; // this lane's frontier entry
41
- var deg = 0u;
42
- var start = 0u;
43
- var u = 0u;
44
- if (i < count) { // guarded loads into locals (3.5 rule 1)
45
- u = frontierList[i];
46
- let lo = max(rowPtr[u], P.arcBase);
47
- let hi = min(rowPtr[u + 1u], P.arcEnd);
48
- start = lo;
49
- deg = select(0u, hi - lo, hi > lo);
50
- }
51
- if (lid.x < 32u) { atomicStore(&local[lid.x], 0u); } // zeroed before the strip loop (WebGPU zero-initialises workgroup memory; said anyway)
52
- sh[lid.x] = deg;
53
- rowStart[lid.x] = start;
54
- rowOf[lid.x] = u;
55
- workgroupBarrier();
56
- for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees (bfs-contract's, inlined)
57
- var t = 0u;
58
- if (lid.x >= s) { t = sh[lid.x - s]; }
59
- workgroupBarrier();
60
- sh[lid.x] = sh[lid.x] + t;
61
- workgroupBarrier();
62
- }
63
- let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier
64
- for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
65
- var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p
66
- var hi = WG;
67
- loop {
68
- if (lo >= hi) { break; }
69
- let mid = (lo + hi) / 2u;
70
- if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }
71
- }
72
- let k = lo;
73
- var exclusive = 0u;
74
- if (k > 0u) { exclusive = sh[k - 1u]; }
75
- let arc = rowStart[k] + (p - exclusive);
76
- let x = colIdx[arc - P.arcBase];
77
- let mask = atomicLoad(&bits[frontierBase + rowOf[k]]) & ~atomicLoad(&bits[x]); // the sources at u that have not reached x
78
- if (mask != 0u) {
79
- let old = atomicOr(&bits[x], mask); // visited: the claim, one read-modify-write
80
- let fresh = mask & ~old; // the sources whose claim this lane won
81
- if (fresh != 0u) {
82
- atomicOr(&bits[nextBase + x], fresh);
83
- atomicStore(&bits[3u * P.bitsBase + x], 1u); // flags: x is in the next frontier list (compact reads it)
84
- if (P.perNode == 1u) { // a sampled run: x's distance to each source won
85
- atomicAdd(&perSource[128u + x], countOneBits(fresh) * dist);
86
- }
87
- var b = fresh;
88
- loop { // one tally per set bit of fresh
89
- if (b == 0u) { break; }
90
- let s = firstTrailingBit(b);
91
- atomicAdd(&local[s], 1u);
92
- b = b & (b - 1u);
93
- }
94
- }
95
- }
96
- }
97
- workgroupBarrier(); // uniform: the loop's bound is the uniform aggregate
98
- if (lid.x < 32u) { // ONE global atomic per source per workgroup
99
- let c = atomicLoad(&local[lid.x]);
100
- if (c != 0u) { atomicAdd(&perSource[lid.x], c); } // newCount[s]
101
- }
102
- workgroupBarrier(); // sh, rowStart, rowOf and local are reused by the next block
103
- }
104
- }
105
- `;
106
- //# sourceMappingURL=closeness-sweep.wgsl.js.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"closeness-sweep.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/closeness-sweep.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAoF5C,CAAC"}
@@ -1,68 +0,0 @@
1
- /**
2
- * The `closeness-reduce` kernel body (design 8.4, 9.7 "integer distances before division"; P8-T11, the P8 plan's
3
- * PD-13): the one-lane bookkeeping of the bit-parallel sweep, two roles by `P.role`. Role 0 is the level boundary,
4
- * recorded BEFORE the level's `compact` and sweep: `done` is whether the previous level's compacted count is 0 (so the
5
- * traversal ends one level after the last claim, one empty sweep and no wrong sum), then for every source `s` the
6
- * claims the level just run made (`newCount[s]`) are folded into `reached[s]` and into the 64-bit `sum[s]` at the
7
- * distance `level + 1` -- the product as a 16-bit split into a low and a high word, then the add with its carry --
8
- * `newCount[s]` is zeroed and `level` advances. Role 1 is the seed of a batch: for the batch's `k = min(32, n -
9
- * P.source)` sources, bit `s` into `visited[source_s]` and into the frontier region level 0 reads (region 1, since
10
- * level 0's parity is 0), `flags[source_s] = 1` (level 0's `compact` turns the flags into the list; a seeded list
11
- * would be overwritten by a compaction of all-zero flags), `counters[0] = k` (not done) and `level = U32_MAX` (so
12
- * level 0's boundary accumulates nothing and brings the word to 0, and level 1's counts the distance-1 claims at 1).
13
- * Role 2 is role 1 for a sampled run: source `s` of the batch is word `P.source + s` of the source list the host
14
- * wrote after the per-node sums (`perSource[128 + P.bitsBase + ...]`), `P.n` is the list's length, and a node listed
15
- * twice in one batch carries both bits (the seed ORs, so a duplicate runs twice).
16
- * No barrier follows the early return of the other lanes (3.5 rule 1). Body only (spec 3.5, D9); the text is
17
- * normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
18
- */
19
- export const closenessReduceWgsl = /* wgsl */ `
20
- @compute @workgroup_size(WG)
21
- fn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {
22
- if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
23
- if (P.role != 0u) { // the seed of a batch: P.source is its first source
24
- let k = min(32u, P.n - P.source);
25
- for (var s = 0u; s < k; s = s + 1u) {
26
- var v = P.source + s;
27
- if (P.role == 2u) { v = atomicLoad(&perSource[128u + P.bitsBase + P.source + s]); } // a sampled run's list
28
- let bit = 1u << s;
29
- bits[v] = bits[v] | bit; // visited
30
- bits[P.bitsBase + v] = bits[P.bitsBase + v] | bit; // the frontier level 0 reads (region 1: level 0's parity is 0)
31
- bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list
32
- }
33
- atomicStore(&counters[0], k); // not done
34
- atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0
35
- atomicStore(&counters[15], 0u); // done
36
- return;
37
- }
38
- // role 0: the level boundary -- done from the previous level's compacted count, then the accumulation
39
- let count = atomicLoad(&counters[0]);
40
- atomicStore(&counters[15], select(0u, 1u, count == 0u));
41
- let level = atomicLoad(&counters[11]);
42
- let d = level + 1u; // the distance of the claims the level just run made
43
- for (var s = 0u; s < 32u; s = s + 1u) {
44
- let c = atomicLoad(&perSource[s]); // newCount[s]
45
- atomicStore(&perSource[32u + s], atomicLoad(&perSource[32u + s]) + c); // reached[s]
46
- // sum[s] += c x d in 64 bits: the 16-bit split product (pLo, pHi), then the add with its carry
47
- let cLo = c & 0xFFFFu;
48
- let cHi = c >> 16u;
49
- let dLo = d & 0xFFFFu;
50
- let dHi = d >> 16u;
51
- let ll = cLo * dLo;
52
- let lh = cLo * dHi;
53
- let hl = cHi * dLo;
54
- let mid = (ll >> 16u) + (lh & 0xFFFFu) + (hl & 0xFFFFu);
55
- let pLo = (ll & 0xFFFFu) | ((mid & 0xFFFFu) << 16u);
56
- let pHi = (cHi * dHi) + (lh >> 16u) + (hl >> 16u) + (mid >> 16u);
57
- var lo = atomicLoad(&perSource[64u + s]); // sumLo[s]
58
- var hi = atomicLoad(&perSource[96u + s]); // sumHi[s]
59
- let before = lo;
60
- lo = lo + pLo;
61
- hi = hi + pHi + select(0u, 1u, lo < before);
62
- atomicStore(&perSource[64u + s], lo);
63
- atomicStore(&perSource[96u + s], hi);
64
- atomicStore(&perSource[s], 0u);
65
- }
66
- atomicStore(&counters[11], level + 1u);
67
- }
68
- `;
@@ -1,105 +0,0 @@
1
- /**
2
- * The `closeness-sweep` kernel body (design 8.4 "32 sources per u32 word"; P8-T11, the P8 plan's PD-13 / DEP-P8-E):
3
- * one level of the bit-parallel multi-source breadth-first search. The batch's state is one `bits` buffer of four
4
- * regions of `P.bitsBase` words each -- `visited` at 0, the two frontier regions at `P.bitsBase` and `2 x P.bitsBase`
5
- * (which one is the frontier and which the next swaps by the level's parity, `P.mode`, so nothing is copied between
6
- * levels), `flags` at `3 x P.bitsBase` -- with bit `s` of word `v` meaning "source `s` has reached / is at / is next
7
- * at `v`". The expansion is `advance-expand`'s block-mapped strip (P8-T5): each workgroup loads up to `WG` entries of
8
- * the frontier LIST (the vertices any source is at, compacted from the flags by the host's `compact`), scans their
9
- * degrees with the inlined Hillis-Steele scan of `bfs-contract` (this is not a twin kernel: `needs: []`), and every
10
- * invocation strips the aggregate by binary search. The claim is inline: for the arc `(u, x)` the mask is the sources
11
- * at `u` that have not reached `x`; `atomicOr` on `visited[x]` returns the bits this lane won (`fresh`), which go into
12
- * the next region, set `flags[x]` (the flags region is part of the one atomic binding, so a plain store is a compile
13
- * error), and are tallied per source in WORKGROUP memory -- one global `atomicAdd` per source per workgroup after the
14
- * strip loop, never one per arc. Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan and the
15
- * `workgroupUniformLoad` sit unconditionally after the guard, the strip loop is bounded by the uniform aggregate, and
16
- * the flush's barrier follows it in uniform control flow. A sampled run (`P.perNode == 1`) also adds each claim's
17
- * distance (`level + 1`, one per won bit) into the per-node sum of `x`, `perSource[128 + x]`: the distance from each
18
- * of the batch's sources TO `x`, which is what a node's sampled closeness sums on an undirected graph. Body only (spec 3.5, D9); the text is normative: the
19
- * sabotage rows of test/helpers/sabotage.ts are textual edits of it.
20
- */
21
- export const closenessSweepWgsl = /* wgsl */ `
22
- var<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan
23
- var<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row
24
- var<workgroup> rowOf: array<u32, WG>; // the frontier vertex of each entry (the source end of its arcs)
25
- var<workgroup> local: array<atomic<u32>, 32>; // this workgroup's fresh claims per source
26
- var<workgroup> wcount: u32; // the frontier list's length
27
- var<workgroup> wdist: u32; // the distance of this level's claims
28
-
29
- @compute @workgroup_size(WG)
30
- fn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
31
- if (lid.x == 0u) {
32
- wcount = atomicLoad(&counters[0]); // the frontier list's length (compact's total)
33
- wdist = atomicLoad(&counters[11]) + 1u; // the level word: this level claims at level + 1
34
- }
35
- let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
36
- let dist = workgroupUniformLoad(&wdist);
37
- let nextBase = select(2u * P.bitsBase, P.bitsBase, P.mode == 1u); // the region that is next this level
38
- let frontierBase = 3u * P.bitsBase - nextBase; // the other one: the region that is the frontier
39
- for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
40
- let i = b0 + lid.x; // this lane's frontier entry
41
- var deg = 0u;
42
- var start = 0u;
43
- var u = 0u;
44
- if (i < count) { // guarded loads into locals (3.5 rule 1)
45
- u = frontierList[i];
46
- let lo = max(rowPtr[u], P.arcBase);
47
- let hi = min(rowPtr[u + 1u], P.arcEnd);
48
- start = lo;
49
- deg = select(0u, hi - lo, hi > lo);
50
- }
51
- if (lid.x < 32u) { atomicStore(&local[lid.x], 0u); } // zeroed before the strip loop (WebGPU zero-initialises workgroup memory; said anyway)
52
- sh[lid.x] = deg;
53
- rowStart[lid.x] = start;
54
- rowOf[lid.x] = u;
55
- workgroupBarrier();
56
- for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees (bfs-contract's, inlined)
57
- var t = 0u;
58
- if (lid.x >= s) { t = sh[lid.x - s]; }
59
- workgroupBarrier();
60
- sh[lid.x] = sh[lid.x] + t;
61
- workgroupBarrier();
62
- }
63
- let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier
64
- for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
65
- var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p
66
- var hi = WG;
67
- loop {
68
- if (lo >= hi) { break; }
69
- let mid = (lo + hi) / 2u;
70
- if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }
71
- }
72
- let k = lo;
73
- var exclusive = 0u;
74
- if (k > 0u) { exclusive = sh[k - 1u]; }
75
- let arc = rowStart[k] + (p - exclusive);
76
- let x = colIdx[arc - P.arcBase];
77
- let mask = atomicLoad(&bits[frontierBase + rowOf[k]]) & ~atomicLoad(&bits[x]); // the sources at u that have not reached x
78
- if (mask != 0u) {
79
- let old = atomicOr(&bits[x], mask); // visited: the claim, one read-modify-write
80
- let fresh = mask & ~old; // the sources whose claim this lane won
81
- if (fresh != 0u) {
82
- atomicOr(&bits[nextBase + x], fresh);
83
- atomicStore(&bits[3u * P.bitsBase + x], 1u); // flags: x is in the next frontier list (compact reads it)
84
- if (P.perNode == 1u) { // a sampled run: x's distance to each source won
85
- atomicAdd(&perSource[128u + x], countOneBits(fresh) * dist);
86
- }
87
- var b = fresh;
88
- loop { // one tally per set bit of fresh
89
- if (b == 0u) { break; }
90
- let s = firstTrailingBit(b);
91
- atomicAdd(&local[s], 1u);
92
- b = b & (b - 1u);
93
- }
94
- }
95
- }
96
- }
97
- workgroupBarrier(); // uniform: the loop's bound is the uniform aggregate
98
- if (lid.x < 32u) { // ONE global atomic per source per workgroup
99
- let c = atomicLoad(&local[lid.x]);
100
- if (c != 0u) { atomicAdd(&perSource[lid.x], c); } // newCount[s]
101
- }
102
- workgroupBarrier(); // sh, rowStart, rowOf and local are reused by the next block
103
- }
104
- }
105
- `;