@graphty/webgpu-graph-algorithms 0.6.26 → 0.6.28

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/README.md +56 -6
  2. package/dist/acquire.d.ts +2 -0
  3. package/dist/browser.js +18 -1
  4. package/dist/browser.js.map +1 -1
  5. package/dist/chunks/accelerator-B-FjQwaA.js +19173 -0
  6. package/dist/chunks/accelerator-B-FjQwaA.js.map +1 -0
  7. package/dist/chunks/managed-D_GdQtnu.js +98 -0
  8. package/dist/chunks/managed-D_GdQtnu.js.map +1 -0
  9. package/dist/node.js +18 -1
  10. package/dist/node.js.map +1 -1
  11. package/dist/src/accelerator.d.ts.map +1 -1
  12. package/dist/src/accelerator.js +5 -3
  13. package/dist/src/accelerator.js.map +1 -1
  14. package/dist/src/algorithms/all-pairs.d.ts.map +1 -1
  15. package/dist/src/algorithms/all-pairs.js +72 -47
  16. package/dist/src/algorithms/all-pairs.js.map +1 -1
  17. package/dist/src/algorithms/betweenness.d.ts +1 -1
  18. package/dist/src/algorithms/betweenness.js +2 -2
  19. package/dist/src/algorithms/closeness.d.ts +45 -42
  20. package/dist/src/algorithms/closeness.d.ts.map +1 -1
  21. package/dist/src/algorithms/closeness.js +295 -226
  22. package/dist/src/algorithms/closeness.js.map +1 -1
  23. package/dist/src/browser/index.d.ts +10 -0
  24. package/dist/src/browser/index.d.ts.map +1 -1
  25. package/dist/src/browser/index.js +22 -0
  26. package/dist/src/browser/index.js.map +1 -1
  27. package/dist/src/constants.d.ts +13 -3
  28. package/dist/src/constants.d.ts.map +1 -1
  29. package/dist/src/constants.js +13 -3
  30. package/dist/src/constants.js.map +1 -1
  31. package/dist/src/kernels.d.ts +14 -4
  32. package/dist/src/kernels.d.ts.map +1 -1
  33. package/dist/src/kernels.js +62 -28
  34. package/dist/src/kernels.js.map +1 -1
  35. package/dist/src/layouts/force-simulation.d.ts.map +1 -1
  36. package/dist/src/layouts/force-simulation.js +0 -1
  37. package/dist/src/layouts/force-simulation.js.map +1 -1
  38. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  39. package/dist/src/layouts/forceatlas2.js +0 -1
  40. package/dist/src/layouts/forceatlas2.js.map +1 -1
  41. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  42. package/dist/src/layouts/fruchterman-reingold.js +0 -1
  43. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  44. package/dist/src/layouts/repulsion-grid.d.ts +1 -3
  45. package/dist/src/layouts/repulsion-grid.d.ts.map +1 -1
  46. package/dist/src/layouts/repulsion-grid.js +1 -6
  47. package/dist/src/layouts/repulsion-grid.js.map +1 -1
  48. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  49. package/dist/src/layouts/spring-electrical.js +0 -1
  50. package/dist/src/layouts/spring-electrical.js.map +1 -1
  51. package/dist/src/managed.d.ts +11 -0
  52. package/dist/src/managed.d.ts.map +1 -0
  53. package/dist/src/managed.js +129 -0
  54. package/dist/src/managed.js.map +1 -0
  55. package/dist/src/node/index.d.ts +11 -0
  56. package/dist/src/node/index.d.ts.map +1 -1
  57. package/dist/src/node/index.js +21 -0
  58. package/dist/src/node/index.js.map +1 -1
  59. package/dist/src/primitives/grid-pyramid.d.ts +16 -15
  60. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  61. package/dist/src/primitives/grid-pyramid.js +20 -28
  62. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  63. package/dist/src/types/accelerator.d.ts +2 -0
  64. package/dist/src/types/accelerator.d.ts.map +1 -1
  65. package/dist/src/types/managed.d.ts +81 -0
  66. package/dist/src/types/managed.d.ts.map +1 -0
  67. package/dist/src/types/managed.js +7 -0
  68. package/dist/src/types/managed.js.map +1 -0
  69. package/dist/src/wgsl/bc-forward.wgsl.d.ts +1 -1
  70. package/dist/src/wgsl/bc-forward.wgsl.js +1 -1
  71. package/dist/src/wgsl/closeness-level.wgsl.d.ts +37 -0
  72. package/dist/src/wgsl/closeness-level.wgsl.d.ts.map +1 -0
  73. package/dist/src/wgsl/closeness-level.wgsl.js +204 -0
  74. package/dist/src/wgsl/closeness-level.wgsl.js.map +1 -0
  75. package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts +11 -0
  76. package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts.map +1 -0
  77. package/dist/src/wgsl/closeness-rowsum.wgsl.js +42 -0
  78. package/dist/src/wgsl/closeness-rowsum.wgsl.js.map +1 -0
  79. package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +2 -2
  80. package/dist/src/wgsl/grid-centroid-hub.wgsl.js +2 -2
  81. package/dist/webgpu-graph-algorithms.js +142 -15586
  82. package/dist/webgpu-graph-algorithms.js.map +1 -1
  83. package/package.json +10 -4
  84. package/src/accelerator.ts +5 -3
  85. package/src/algorithms/all-pairs.ts +86 -56
  86. package/src/algorithms/betweenness.ts +2 -2
  87. package/src/algorithms/closeness.ts +353 -256
  88. package/src/browser/index.ts +37 -0
  89. package/src/constants.ts +13 -3
  90. package/src/kernels.ts +65 -36
  91. package/src/layouts/force-simulation.ts +0 -1
  92. package/src/layouts/forceatlas2.ts +0 -1
  93. package/src/layouts/fruchterman-reingold.ts +0 -1
  94. package/src/layouts/repulsion-grid.ts +2 -7
  95. package/src/layouts/spring-electrical.ts +0 -1
  96. package/src/managed.ts +172 -0
  97. package/src/node/index.ts +36 -0
  98. package/src/primitives/grid-pyramid.ts +29 -41
  99. package/src/types/accelerator.ts +2 -0
  100. package/src/types/managed.ts +86 -0
  101. package/src/wgsl/bc-forward.wgsl.ts +1 -1
  102. package/src/wgsl/closeness-level.wgsl.ts +203 -0
  103. package/src/wgsl/closeness-rowsum.wgsl.ts +41 -0
  104. package/src/wgsl/grid-centroid-hub.wgsl.ts +2 -2
  105. package/dist/chunks/context-BZY6SMsM.js +0 -3615
  106. package/dist/chunks/context-BZY6SMsM.js.map +0 -1
  107. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +0 -20
  108. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +0 -1
  109. package/dist/src/wgsl/closeness-reduce.wgsl.js +0 -69
  110. package/dist/src/wgsl/closeness-reduce.wgsl.js.map +0 -1
  111. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +0 -22
  112. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +0 -1
  113. package/dist/src/wgsl/closeness-sweep.wgsl.js +0 -106
  114. package/dist/src/wgsl/closeness-sweep.wgsl.js.map +0 -1
  115. package/src/wgsl/closeness-reduce.wgsl.ts +0 -68
  116. package/src/wgsl/closeness-sweep.wgsl.ts +0 -105
@@ -0,0 +1,37 @@
1
+ /**
2
+ * The `closeness-level` kernel body: one level of the bit-parallel multi-source breadth-first search of closeness, in
3
+ * ONE dispatch, plus the two seed roles of a batch. A batch runs `32 x P.words` sources at once: bit `b` of word `j`
4
+ * of node `v` is source lane `32 j + b`. The `bits` buffer holds five regions of `P.base` words, `P.words` words per
5
+ * node: `visited` at 0, three frontier regions at `P.base`, `2 P.base` and `3 P.base` that rotate by the level
6
+ * (level `L` reads region `1 + L % 3`, writes region `1 + (L + 1) % 3` and zeroes region `1 + (L + 2) % 3`, the
7
+ * frontier of level `L - 1` that nothing reads any more, so no fill runs between levels; a stale bit could never
8
+ * claim anything, since its node's neighbours were claimed the level after, so the clear only keeps a later frontier
9
+ * from re-walking old nodes), then a sampled run's
10
+ * per-node distance sums at `4 P.base`. The `table` buffer holds the per-level claim counts of the submit (row `P.row`
11
+ * at `P.row x 32 P.words`, one word per source lane), then a ring of three control slots of four words at `P.ctrl`
12
+ * (`any` @0: the level claimed something; `arcs` @1: the out-degree of every (node, word) that joined the next
13
+ * frontier; `pull` @2: `arcs` passed `P.pullAt`), then a sampled run's source list at `P.sourcesAt`.
14
+ *
15
+ * Role 0 (one invocation per node and per arc, `max(n, arcs)` in all): a level whose predecessor claimed nothing
16
+ * returns at once (one uniform load per workgroup), so the host can record more levels than the batch needs.
17
+ * Otherwise the level chooses its step from the slot its predecessor filled, the same way for every invocation:
18
+ * - PUSH (the frontier is cheap to expand): one invocation per out-arc `(u, x)` -- `u` found by a binary search of
19
+ * `rowPtr` -- claims the sources at `u` that `x` has not seen with `atomicOr` on `x`'s visited word; the bits it
20
+ * won (`fresh`) join the next frontier;
21
+ * - PULL (the frontier's arcs pass `P.pullAt`, and `P.pullOk`): one invocation per node not yet reached by every
22
+ * source walks its IN-arcs, ORs the neighbours' frontier words, and keeps the bits it had not seen; it stops at
23
+ * the first in-arc after which every source has reached it (the early exit). Only the owner writes a node's
24
+ * words. The host allows the pull only when no node has more than a few thousand in-arcs, so no
25
+ * invocation of either step loops more than a few tens of thousands of times (llvmpipe silently ends every loop
26
+ * of an invocation past 65,535 iterations).
27
+ * Every claimed bit is tallied per source lane in workgroup memory and flushed with one global `atomicAdd` per lane
28
+ * per workgroup into the level's row; the host turns the counts into exact sums (`count x (L + 1)`) and harmonic
29
+ * sums (`count / (L + 1)`). Role 1 (one invocation per word): zeroes the regions, sets every dead lane of a partial
30
+ * batch as already visited (so the early exit and the "reached by every source" test see a full word), and zeroes the
31
+ * control ring. Role 2 (one invocation per lane): seeds lane `b`'s source -- `P.source + b`, or word `P.source + b` of
32
+ * the source list -- into `visited` and level 0's frontier with `atomicOr` (a node listed twice carries both bits),
33
+ * and fills the control slot level 0 reads. Body only; the text is normative: the sabotage rows of
34
+ * test/helpers/sabotage.ts are textual edits of it.
35
+ */
36
+ export declare const closenessLevelWgsl = "\nconst max_words: u32 = 8u;\nvar<workgroup> tally: array<atomic<u32>, 32u * max_words>; // 32 x max_words: this workgroup's claims per source lane\nvar<workgroup> wlive: u32;\nvar<workgroup> wpull: u32;\nvar<workgroup> wany: atomic<u32>;\nvar<workgroup> warcs: atomic<u32>;\nvar<workgroup> wover: atomic<u32>;\n\nfn dead_lanes(j: u32) -> u32 { // the lanes of word j at or past the batch's source count\n let first = 32u * j;\n if (P.count >= first + 32u) { return 0u; }\n if (P.count <= first) { return U32_MAX; }\n return ~((1u << (P.count - first)) - 1u);\n}\n\nfn add_arcs(slot: u32, value: u32) { // the arcs total of a control slot; pull once it passes P.pullAt\n if (value == 0u) { return; }\n let before = atomicAdd(&table[slot + 1u], value);\n if (value > P.pullAt || before + value > P.pullAt || before + value < before) {\n atomicStore(&table[slot + 2u], 1u);\n }\n}\n\nfn record(j: u32, node: u32, fresh: u32, dist: u32) { // the claims of one word: tallied per lane, a sampled run's sums\n if (P.perNode == 1u) { atomicAdd(&bits[4u * P.base + node], countOneBits(fresh) * dist); }\n var b = fresh;\n loop {\n if (b == 0u) { break; }\n atomicAdd(&tally[32u * j + firstTrailingBit(b)], 1u);\n b = b & (b - 1u);\n }\n}\n\n@compute @workgroup_size(WG)\nfn closeness_level(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n let W = P.words;\n if (P.role == 1u) { // seed, part 1: one invocation per word\n if (i == 0u) {\n for (var k = 0u; k < 12u; k = k + 1u) { atomicStore(&table[P.ctrl + k], 0u); }\n }\n if (i < P.total) { atomicStore(&bits[i], select(0u, dead_lanes(i % W), i < P.base)); }\n return; // uniform: P.role is\n }\n if (P.role == 2u) { // seed, part 2: one invocation per source lane\n if (i < P.count) {\n var v = P.source + i;\n if (P.sourcesAt != 0u) { v = atomicLoad(&table[P.sourcesAt + P.source + i]); }\n let w = v * W + i / 32u;\n let bit = 1u << (i % 32u);\n atomicOr(&bits[w], bit); // visited\n let before = atomicOr(&bits[P.base + w], bit); // level 0's frontier (region 1)\n let slot = P.ctrl + 8u; // the slot of \"level -1\", which level 0 reads\n atomicStore(&table[slot], 1u);\n if (before == 0u) { add_arcs(slot, rowPtr[v + 1u] - rowPtr[v]); }\n }\n return;\n }\n\n // role 0: one level\n let L = P.level;\n let prev = P.ctrl + 4u * ((L + 2u) % 3u);\n let cur = P.ctrl + 4u * (L % 3u);\n if (lid.x == 0u) {\n wlive = atomicLoad(&table[prev]);\n wpull = select(0u, atomicLoad(&table[prev + 2u]), P.pullOk == 1u);\n atomicStore(&wany, 0u);\n atomicStore(&warcs, 0u);\n atomicStore(&wover, 0u);\n }\n for (var k = lid.x; k < 32u * W; k = k + WG) { atomicStore(&tally[k], 0u); }\n let live = workgroupUniformLoad(&wlive); // uniform; includes a barrier\n if (live == 0u) { return; } // the previous level claimed nothing\n let pull = workgroupUniformLoad(&wpull) == 1u;\n if (i == 0u) { // the slot of level L + 1 held level L - 2's\n let nxt = P.ctrl + 4u * ((L + 1u) % 3u);\n atomicStore(&table[nxt], 0u);\n atomicStore(&table[nxt + 1u], 0u);\n atomicStore(&table[nxt + 2u], 0u);\n }\n let frontierBase = P.base * (1u + L % 3u);\n let nextBase = P.base * (1u + (L + 1u) % 3u);\n let staleBase = P.base * (1u + (L + 2u) % 3u);\n let dist = L + 1u;\n var arcs = 0u;\n if (i < P.n) {\n for (var j = 0u; j < W; j = j + 1u) { atomicStore(&bits[staleBase + i * W + j], 0u); }\n }\n if (pull) {\n if (i < P.n) { // one invocation per node: its in-arcs\n var vis: array<u32, max_words>;\n var unseen = 0u;\n for (var j = 0u; j < W; j = j + 1u) {\n vis[j] = atomicLoad(&bits[i * W + j]);\n unseen = unseen | ~vis[j];\n }\n if (unseen != 0u) { // some source has not reached this node yet\n var acc: array<u32, max_words>;\n let end = inRowPtr[i + 1u];\n for (var a = inRowPtr[i]; a < end; a = a + 1u) {\n let u = inColIdx[a];\n var missing = 0u;\n for (var j = 0u; j < W; j = j + 1u) {\n acc[j] = acc[j] | atomicLoad(&bits[frontierBase + u * W + j]);\n missing = missing | ~(acc[j] | vis[j]);\n }\n if (missing == 0u) { break; } // every source has reached it: the early exit\n }\n let degree = rowPtr[i + 1u] - rowPtr[i];\n for (var j = 0u; j < W; j = j + 1u) {\n let fresh = acc[j] & ~vis[j];\n if (fresh != 0u) {\n atomicStore(&bits[i * W + j], vis[j] | fresh);\n atomicStore(&bits[nextBase + i * W + j], fresh);\n arcs = arcs + degree;\n record(j, i, fresh, dist);\n }\n }\n }\n }\n } else if (i < P.arcCount) { // one invocation per arc: no row is walked whole\n var lo = 0u; // the arc's source: the last row starting at or before it\n var hi = P.n - 1u;\n loop {\n if (lo >= hi) { break; }\n let mid = (lo + hi + 1u) / 2u;\n if (rowPtr[mid] <= i) { lo = mid; } else { hi = mid - 1u; }\n }\n let u = lo;\n let x = colIdx[i];\n for (var j = 0u; j < W; j = j + 1u) {\n let mask = atomicLoad(&bits[frontierBase + u * W + j]) & ~atomicLoad(&bits[x * W + j]); // at u, not yet at x\n if (mask == 0u) { continue; }\n let fresh = mask & ~atomicOr(&bits[x * W + j], mask); // the claims this invocation won\n if (fresh == 0u) { continue; }\n if (atomicOr(&bits[nextBase + x * W + j], fresh) == 0u) {\n arcs = arcs + (rowPtr[x + 1u] - rowPtr[x]);\n }\n record(j, x, fresh, dist);\n }\n }\n if (arcs > P.pullAt) {\n atomicStore(&wover, 1u);\n } else if (arcs != 0u) {\n let before = atomicAdd(&warcs, arcs);\n if (before + arcs > P.pullAt || before + arcs < before) { atomicStore(&wover, 1u); }\n }\n workgroupBarrier();\n let row = P.row * 32u * W;\n for (var k = lid.x; k < 32u * W; k = k + WG) { // ONE global atomic per source lane per workgroup\n let c = atomicLoad(&tally[k]);\n if (c != 0u) {\n atomicAdd(&table[row + k], c);\n atomicStore(&wany, 1u);\n }\n }\n workgroupBarrier();\n if (lid.x == 0u) {\n if (atomicLoad(&wany) != 0u) { atomicStore(&table[cur], 1u); }\n if (atomicLoad(&wover) != 0u) {\n atomicStore(&table[cur + 2u], 1u);\n } else {\n add_arcs(cur, atomicLoad(&warcs));\n }\n }\n}\n";
37
+ //# sourceMappingURL=closeness-level.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"closeness-level.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-level.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkCG;AACH,eAAO,MAAM,kBAAkB,ohPAuK9B,CAAC"}
@@ -0,0 +1,204 @@
1
+ /**
2
+ * The `closeness-level` kernel body: one level of the bit-parallel multi-source breadth-first search of closeness, in
3
+ * ONE dispatch, plus the two seed roles of a batch. A batch runs `32 x P.words` sources at once: bit `b` of word `j`
4
+ * of node `v` is source lane `32 j + b`. The `bits` buffer holds five regions of `P.base` words, `P.words` words per
5
+ * node: `visited` at 0, three frontier regions at `P.base`, `2 P.base` and `3 P.base` that rotate by the level
6
+ * (level `L` reads region `1 + L % 3`, writes region `1 + (L + 1) % 3` and zeroes region `1 + (L + 2) % 3`, the
7
+ * frontier of level `L - 1` that nothing reads any more, so no fill runs between levels; a stale bit could never
8
+ * claim anything, since its node's neighbours were claimed the level after, so the clear only keeps a later frontier
9
+ * from re-walking old nodes), then a sampled run's
10
+ * per-node distance sums at `4 P.base`. The `table` buffer holds the per-level claim counts of the submit (row `P.row`
11
+ * at `P.row x 32 P.words`, one word per source lane), then a ring of three control slots of four words at `P.ctrl`
12
+ * (`any` @0: the level claimed something; `arcs` @1: the out-degree of every (node, word) that joined the next
13
+ * frontier; `pull` @2: `arcs` passed `P.pullAt`), then a sampled run's source list at `P.sourcesAt`.
14
+ *
15
+ * Role 0 (one invocation per node and per arc, `max(n, arcs)` in all): a level whose predecessor claimed nothing
16
+ * returns at once (one uniform load per workgroup), so the host can record more levels than the batch needs.
17
+ * Otherwise the level chooses its step from the slot its predecessor filled, the same way for every invocation:
18
+ * - PUSH (the frontier is cheap to expand): one invocation per out-arc `(u, x)` -- `u` found by a binary search of
19
+ * `rowPtr` -- claims the sources at `u` that `x` has not seen with `atomicOr` on `x`'s visited word; the bits it
20
+ * won (`fresh`) join the next frontier;
21
+ * - PULL (the frontier's arcs pass `P.pullAt`, and `P.pullOk`): one invocation per node not yet reached by every
22
+ * source walks its IN-arcs, ORs the neighbours' frontier words, and keeps the bits it had not seen; it stops at
23
+ * the first in-arc after which every source has reached it (the early exit). Only the owner writes a node's
24
+ * words. The host allows the pull only when no node has more than a few thousand in-arcs, so no
25
+ * invocation of either step loops more than a few tens of thousands of times (llvmpipe silently ends every loop
26
+ * of an invocation past 65,535 iterations).
27
+ * Every claimed bit is tallied per source lane in workgroup memory and flushed with one global `atomicAdd` per lane
28
+ * per workgroup into the level's row; the host turns the counts into exact sums (`count x (L + 1)`) and harmonic
29
+ * sums (`count / (L + 1)`). Role 1 (one invocation per word): zeroes the regions, sets every dead lane of a partial
30
+ * batch as already visited (so the early exit and the "reached by every source" test see a full word), and zeroes the
31
+ * control ring. Role 2 (one invocation per lane): seeds lane `b`'s source -- `P.source + b`, or word `P.source + b` of
32
+ * the source list -- into `visited` and level 0's frontier with `atomicOr` (a node listed twice carries both bits),
33
+ * and fills the control slot level 0 reads. Body only; the text is normative: the sabotage rows of
34
+ * test/helpers/sabotage.ts are textual edits of it.
35
+ */
36
+ export const closenessLevelWgsl = /* wgsl */ `
37
+ const max_words: u32 = 8u;
38
+ var<workgroup> tally: array<atomic<u32>, 32u * max_words>; // 32 x max_words: this workgroup's claims per source lane
39
+ var<workgroup> wlive: u32;
40
+ var<workgroup> wpull: u32;
41
+ var<workgroup> wany: atomic<u32>;
42
+ var<workgroup> warcs: atomic<u32>;
43
+ var<workgroup> wover: atomic<u32>;
44
+
45
+ fn dead_lanes(j: u32) -> u32 { // the lanes of word j at or past the batch's source count
46
+ let first = 32u * j;
47
+ if (P.count >= first + 32u) { return 0u; }
48
+ if (P.count <= first) { return U32_MAX; }
49
+ return ~((1u << (P.count - first)) - 1u);
50
+ }
51
+
52
+ fn add_arcs(slot: u32, value: u32) { // the arcs total of a control slot; pull once it passes P.pullAt
53
+ if (value == 0u) { return; }
54
+ let before = atomicAdd(&table[slot + 1u], value);
55
+ if (value > P.pullAt || before + value > P.pullAt || before + value < before) {
56
+ atomicStore(&table[slot + 2u], 1u);
57
+ }
58
+ }
59
+
60
+ fn record(j: u32, node: u32, fresh: u32, dist: u32) { // the claims of one word: tallied per lane, a sampled run's sums
61
+ if (P.perNode == 1u) { atomicAdd(&bits[4u * P.base + node], countOneBits(fresh) * dist); }
62
+ var b = fresh;
63
+ loop {
64
+ if (b == 0u) { break; }
65
+ atomicAdd(&tally[32u * j + firstTrailingBit(b)], 1u);
66
+ b = b & (b - 1u);
67
+ }
68
+ }
69
+
70
+ @compute @workgroup_size(WG)
71
+ fn closeness_level(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
72
+ let i = linear_id(wid, lid.x);
73
+ let W = P.words;
74
+ if (P.role == 1u) { // seed, part 1: one invocation per word
75
+ if (i == 0u) {
76
+ for (var k = 0u; k < 12u; k = k + 1u) { atomicStore(&table[P.ctrl + k], 0u); }
77
+ }
78
+ if (i < P.total) { atomicStore(&bits[i], select(0u, dead_lanes(i % W), i < P.base)); }
79
+ return; // uniform: P.role is
80
+ }
81
+ if (P.role == 2u) { // seed, part 2: one invocation per source lane
82
+ if (i < P.count) {
83
+ var v = P.source + i;
84
+ if (P.sourcesAt != 0u) { v = atomicLoad(&table[P.sourcesAt + P.source + i]); }
85
+ let w = v * W + i / 32u;
86
+ let bit = 1u << (i % 32u);
87
+ atomicOr(&bits[w], bit); // visited
88
+ let before = atomicOr(&bits[P.base + w], bit); // level 0's frontier (region 1)
89
+ let slot = P.ctrl + 8u; // the slot of "level -1", which level 0 reads
90
+ atomicStore(&table[slot], 1u);
91
+ if (before == 0u) { add_arcs(slot, rowPtr[v + 1u] - rowPtr[v]); }
92
+ }
93
+ return;
94
+ }
95
+
96
+ // role 0: one level
97
+ let L = P.level;
98
+ let prev = P.ctrl + 4u * ((L + 2u) % 3u);
99
+ let cur = P.ctrl + 4u * (L % 3u);
100
+ if (lid.x == 0u) {
101
+ wlive = atomicLoad(&table[prev]);
102
+ wpull = select(0u, atomicLoad(&table[prev + 2u]), P.pullOk == 1u);
103
+ atomicStore(&wany, 0u);
104
+ atomicStore(&warcs, 0u);
105
+ atomicStore(&wover, 0u);
106
+ }
107
+ for (var k = lid.x; k < 32u * W; k = k + WG) { atomicStore(&tally[k], 0u); }
108
+ let live = workgroupUniformLoad(&wlive); // uniform; includes a barrier
109
+ if (live == 0u) { return; } // the previous level claimed nothing
110
+ let pull = workgroupUniformLoad(&wpull) == 1u;
111
+ if (i == 0u) { // the slot of level L + 1 held level L - 2's
112
+ let nxt = P.ctrl + 4u * ((L + 1u) % 3u);
113
+ atomicStore(&table[nxt], 0u);
114
+ atomicStore(&table[nxt + 1u], 0u);
115
+ atomicStore(&table[nxt + 2u], 0u);
116
+ }
117
+ let frontierBase = P.base * (1u + L % 3u);
118
+ let nextBase = P.base * (1u + (L + 1u) % 3u);
119
+ let staleBase = P.base * (1u + (L + 2u) % 3u);
120
+ let dist = L + 1u;
121
+ var arcs = 0u;
122
+ if (i < P.n) {
123
+ for (var j = 0u; j < W; j = j + 1u) { atomicStore(&bits[staleBase + i * W + j], 0u); }
124
+ }
125
+ if (pull) {
126
+ if (i < P.n) { // one invocation per node: its in-arcs
127
+ var vis: array<u32, max_words>;
128
+ var unseen = 0u;
129
+ for (var j = 0u; j < W; j = j + 1u) {
130
+ vis[j] = atomicLoad(&bits[i * W + j]);
131
+ unseen = unseen | ~vis[j];
132
+ }
133
+ if (unseen != 0u) { // some source has not reached this node yet
134
+ var acc: array<u32, max_words>;
135
+ let end = inRowPtr[i + 1u];
136
+ for (var a = inRowPtr[i]; a < end; a = a + 1u) {
137
+ let u = inColIdx[a];
138
+ var missing = 0u;
139
+ for (var j = 0u; j < W; j = j + 1u) {
140
+ acc[j] = acc[j] | atomicLoad(&bits[frontierBase + u * W + j]);
141
+ missing = missing | ~(acc[j] | vis[j]);
142
+ }
143
+ if (missing == 0u) { break; } // every source has reached it: the early exit
144
+ }
145
+ let degree = rowPtr[i + 1u] - rowPtr[i];
146
+ for (var j = 0u; j < W; j = j + 1u) {
147
+ let fresh = acc[j] & ~vis[j];
148
+ if (fresh != 0u) {
149
+ atomicStore(&bits[i * W + j], vis[j] | fresh);
150
+ atomicStore(&bits[nextBase + i * W + j], fresh);
151
+ arcs = arcs + degree;
152
+ record(j, i, fresh, dist);
153
+ }
154
+ }
155
+ }
156
+ }
157
+ } else if (i < P.arcCount) { // one invocation per arc: no row is walked whole
158
+ var lo = 0u; // the arc's source: the last row starting at or before it
159
+ var hi = P.n - 1u;
160
+ loop {
161
+ if (lo >= hi) { break; }
162
+ let mid = (lo + hi + 1u) / 2u;
163
+ if (rowPtr[mid] <= i) { lo = mid; } else { hi = mid - 1u; }
164
+ }
165
+ let u = lo;
166
+ let x = colIdx[i];
167
+ for (var j = 0u; j < W; j = j + 1u) {
168
+ let mask = atomicLoad(&bits[frontierBase + u * W + j]) & ~atomicLoad(&bits[x * W + j]); // at u, not yet at x
169
+ if (mask == 0u) { continue; }
170
+ let fresh = mask & ~atomicOr(&bits[x * W + j], mask); // the claims this invocation won
171
+ if (fresh == 0u) { continue; }
172
+ if (atomicOr(&bits[nextBase + x * W + j], fresh) == 0u) {
173
+ arcs = arcs + (rowPtr[x + 1u] - rowPtr[x]);
174
+ }
175
+ record(j, x, fresh, dist);
176
+ }
177
+ }
178
+ if (arcs > P.pullAt) {
179
+ atomicStore(&wover, 1u);
180
+ } else if (arcs != 0u) {
181
+ let before = atomicAdd(&warcs, arcs);
182
+ if (before + arcs > P.pullAt || before + arcs < before) { atomicStore(&wover, 1u); }
183
+ }
184
+ workgroupBarrier();
185
+ let row = P.row * 32u * W;
186
+ for (var k = lid.x; k < 32u * W; k = k + WG) { // ONE global atomic per source lane per workgroup
187
+ let c = atomicLoad(&tally[k]);
188
+ if (c != 0u) {
189
+ atomicAdd(&table[row + k], c);
190
+ atomicStore(&wany, 1u);
191
+ }
192
+ }
193
+ workgroupBarrier();
194
+ if (lid.x == 0u) {
195
+ if (atomicLoad(&wany) != 0u) { atomicStore(&table[cur], 1u); }
196
+ if (atomicLoad(&wover) != 0u) {
197
+ atomicStore(&table[cur + 2u], 1u);
198
+ } else {
199
+ add_arcs(cur, atomicLoad(&warcs));
200
+ }
201
+ }
202
+ }
203
+ `;
204
+ //# sourceMappingURL=closeness-level.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"closeness-level.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/closeness-level.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkCG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAuK5C,CAAC"}
@@ -0,0 +1,11 @@
1
+ /**
2
+ * The `closeness-rowsum` kernel body: closeness from a finished all-pairs distance matrix, one workgroup per row.
3
+ * Every lane walks its strided columns of row `r` (the distances FROM `r`), skips the diagonal and every unreachable
4
+ * entry (`+Infinity`, compared by its bit pattern because WGSL lets a compiler assume no infinities), and adds, by
5
+ * `P.role`: 0 the hop count as an integer (exact: a row of at most 23,170 hops below 23,170 sums below 2^32), 1 the
6
+ * f32 distance, 2 its reciprocal (harmonic closeness; a zero distance adds nothing, as in the CPU port). A tree reduction in workgroup memory folds the lanes and lane
7
+ * 0 writes `out[r]` -- the integer, or the f32 bit pattern. The host turns the row into the score. Body only; the text
8
+ * is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
9
+ */
10
+ export declare const closenessRowsumWgsl = "\nvar<workgroup> partial: array<u32, WG>;\n\n@compute @workgroup_size(WG)\nfn closeness_rowsum(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let row = group_id(wid); // uniform: one workgroup per row\n if (row >= P.n) { return; }\n var whole = 0u;\n var real = 0.0;\n for (var c = lid.x; c < P.n; c = c + WG) {\n let d = dist[row * P.n + c];\n if (c == row || bitcast<u32>(d) == F32_INF_BITS) { continue; }\n if (P.role == 0u) {\n whole = whole + u32(d);\n } else if (P.role == 1u) {\n real = real + d;\n } else if (d > 0.0) {\n real = real + 1.0 / d; // a zero distance adds nothing, as on the CPU\n }\n }\n partial[lid.x] = select(whole, bitcast<u32>(real), P.role != 0u);\n for (var s = WG / 2u; s > 0u; s = s / 2u) {\n workgroupBarrier();\n if (lid.x < s) {\n let a = partial[lid.x];\n let b = partial[lid.x + s];\n partial[lid.x] = select(a + b, bitcast<u32>(bitcast<f32>(a) + bitcast<f32>(b)), P.role != 0u);\n }\n }\n if (lid.x == 0u) { out[row] = partial[0]; }\n}\n";
11
+ //# sourceMappingURL=closeness-rowsum.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"closeness-rowsum.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/closeness-rowsum.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AACH,eAAO,MAAM,mBAAmB,uuCA+B/B,CAAC"}
@@ -0,0 +1,42 @@
1
+ /**
2
+ * The `closeness-rowsum` kernel body: closeness from a finished all-pairs distance matrix, one workgroup per row.
3
+ * Every lane walks its strided columns of row `r` (the distances FROM `r`), skips the diagonal and every unreachable
4
+ * entry (`+Infinity`, compared by its bit pattern because WGSL lets a compiler assume no infinities), and adds, by
5
+ * `P.role`: 0 the hop count as an integer (exact: a row of at most 23,170 hops below 23,170 sums below 2^32), 1 the
6
+ * f32 distance, 2 its reciprocal (harmonic closeness; a zero distance adds nothing, as in the CPU port). A tree reduction in workgroup memory folds the lanes and lane
7
+ * 0 writes `out[r]` -- the integer, or the f32 bit pattern. The host turns the row into the score. Body only; the text
8
+ * is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
9
+ */
10
+ export const closenessRowsumWgsl = /* wgsl */ `
11
+ var<workgroup> partial: array<u32, WG>;
12
+
13
+ @compute @workgroup_size(WG)
14
+ fn closeness_rowsum(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
15
+ let row = group_id(wid); // uniform: one workgroup per row
16
+ if (row >= P.n) { return; }
17
+ var whole = 0u;
18
+ var real = 0.0;
19
+ for (var c = lid.x; c < P.n; c = c + WG) {
20
+ let d = dist[row * P.n + c];
21
+ if (c == row || bitcast<u32>(d) == F32_INF_BITS) { continue; }
22
+ if (P.role == 0u) {
23
+ whole = whole + u32(d);
24
+ } else if (P.role == 1u) {
25
+ real = real + d;
26
+ } else if (d > 0.0) {
27
+ real = real + 1.0 / d; // a zero distance adds nothing, as on the CPU
28
+ }
29
+ }
30
+ partial[lid.x] = select(whole, bitcast<u32>(real), P.role != 0u);
31
+ for (var s = WG / 2u; s > 0u; s = s / 2u) {
32
+ workgroupBarrier();
33
+ if (lid.x < s) {
34
+ let a = partial[lid.x];
35
+ let b = partial[lid.x + s];
36
+ partial[lid.x] = select(a + b, bitcast<u32>(bitcast<f32>(a) + bitcast<f32>(b)), P.role != 0u);
37
+ }
38
+ }
39
+ if (lid.x == 0u) { out[row] = partial[0]; }
40
+ }
41
+ `;
42
+ //# sourceMappingURL=closeness-rowsum.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"closeness-rowsum.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/closeness-rowsum.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AACH,MAAM,CAAC,MAAM,mBAAmB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA+B7C,CAAC"}
@@ -1,6 +1,6 @@
1
1
  /**
2
- * G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per hub cell of hubList, dispatched
3
- * indirectly from hubArgs (the T1 finalize over hubCounters[0]); a WG-strided mass-weighted sum reduced by the
2
+ * G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per word of hubList, dispatched
3
+ * directly (issue #732), so the workgroups past hubCounters[0] idle; a WG-strided mass-weighted sum reduced by the
4
4
  * prelude's tree. The work is guarded by `valid`, never an early return, so the reduction is uniform (PD-13). Body
5
5
  * only; normative text.
6
6
  */
@@ -1,6 +1,6 @@
1
1
  /**
2
- * G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per hub cell of hubList, dispatched
3
- * indirectly from hubArgs (the T1 finalize over hubCounters[0]); a WG-strided mass-weighted sum reduced by the
2
+ * G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per word of hubList, dispatched
3
+ * directly (issue #732), so the workgroups past hubCounters[0] idle; a WG-strided mass-weighted sum reduced by the
4
4
  * prelude's tree. The work is guarded by `valid`, never an early return, so the reduction is uniform (PD-13). Body
5
5
  * only; normative text.
6
6
  */