@graphty/webgpu-graph-algorithms 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (264) hide show
  1. package/README.md +104 -52
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-CRbw2Wyo.js → context-BXqgCifx.js} +225 -33
  4. package/dist/chunks/context-BXqgCifx.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/accelerator.d.ts +12 -10
  7. package/dist/src/accelerator.d.ts.map +1 -1
  8. package/dist/src/accelerator.js +32 -10
  9. package/dist/src/accelerator.js.map +1 -1
  10. package/dist/src/algorithms/components.d.ts.map +1 -1
  11. package/dist/src/algorithms/components.js +12 -13
  12. package/dist/src/algorithms/components.js.map +1 -1
  13. package/dist/src/algorithms/degree.d.ts +6 -8
  14. package/dist/src/algorithms/degree.d.ts.map +1 -1
  15. package/dist/src/algorithms/degree.js +58 -35
  16. package/dist/src/algorithms/degree.js.map +1 -1
  17. package/dist/src/algorithms/pagerank.d.ts.map +1 -1
  18. package/dist/src/algorithms/pagerank.js +16 -14
  19. package/dist/src/algorithms/pagerank.js.map +1 -1
  20. package/dist/src/algorithms/power-iteration.d.ts +2 -2
  21. package/dist/src/algorithms/power-iteration.d.ts.map +1 -1
  22. package/dist/src/algorithms/power-iteration.js +17 -14
  23. package/dist/src/algorithms/power-iteration.js.map +1 -1
  24. package/dist/src/constants.d.ts +85 -8
  25. package/dist/src/constants.d.ts.map +1 -1
  26. package/dist/src/constants.js +85 -8
  27. package/dist/src/constants.js.map +1 -1
  28. package/dist/src/errors.d.ts +3 -2
  29. package/dist/src/errors.d.ts.map +1 -1
  30. package/dist/src/errors.js +2 -1
  31. package/dist/src/errors.js.map +1 -1
  32. package/dist/src/index.d.ts +10 -5
  33. package/dist/src/index.d.ts.map +1 -1
  34. package/dist/src/index.js +14 -5
  35. package/dist/src/index.js.map +1 -1
  36. package/dist/src/kernel/dispatch.d.ts +8 -3
  37. package/dist/src/kernel/dispatch.d.ts.map +1 -1
  38. package/dist/src/kernel/dispatch.js +18 -7
  39. package/dist/src/kernel/dispatch.js.map +1 -1
  40. package/dist/src/kernel/kernel.d.ts +30 -1
  41. package/dist/src/kernel/kernel.d.ts.map +1 -1
  42. package/dist/src/kernel/kernel.js +49 -5
  43. package/dist/src/kernel/kernel.js.map +1 -1
  44. package/dist/src/kernel/prelude.d.ts.map +1 -1
  45. package/dist/src/kernel/prelude.js +9 -1
  46. package/dist/src/kernel/prelude.js.map +1 -1
  47. package/dist/src/kernel/profiler.d.ts +15 -3
  48. package/dist/src/kernel/profiler.d.ts.map +1 -1
  49. package/dist/src/kernel/profiler.js +27 -4
  50. package/dist/src/kernel/profiler.js.map +1 -1
  51. package/dist/src/kernels.d.ts +18 -8
  52. package/dist/src/kernels.d.ts.map +1 -1
  53. package/dist/src/kernels.js +345 -22
  54. package/dist/src/kernels.js.map +1 -1
  55. package/dist/src/layouts/calibrate.d.ts +51 -0
  56. package/dist/src/layouts/calibrate.d.ts.map +1 -0
  57. package/dist/src/layouts/calibrate.js +172 -0
  58. package/dist/src/layouts/calibrate.js.map +1 -0
  59. package/dist/src/layouts/force-simulation.d.ts +42 -5
  60. package/dist/src/layouts/force-simulation.d.ts.map +1 -1
  61. package/dist/src/layouts/force-simulation.js +84 -22
  62. package/dist/src/layouts/force-simulation.js.map +1 -1
  63. package/dist/src/layouts/forceatlas2.d.ts +107 -38
  64. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  65. package/dist/src/layouts/forceatlas2.js +297 -290
  66. package/dist/src/layouts/forceatlas2.js.map +1 -1
  67. package/dist/src/layouts/fruchterman-reingold.d.ts +241 -0
  68. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -0
  69. package/dist/src/layouts/fruchterman-reingold.js +739 -0
  70. package/dist/src/layouts/fruchterman-reingold.js.map +1 -0
  71. package/dist/src/layouts/model-common.d.ts +140 -0
  72. package/dist/src/layouts/model-common.d.ts.map +1 -0
  73. package/dist/src/layouts/model-common.js +269 -0
  74. package/dist/src/layouts/model-common.js.map +1 -0
  75. package/dist/src/layouts/repulsion-grid.d.ts +152 -0
  76. package/dist/src/layouts/repulsion-grid.d.ts.map +1 -0
  77. package/dist/src/layouts/repulsion-grid.js +318 -0
  78. package/dist/src/layouts/repulsion-grid.js.map +1 -0
  79. package/dist/src/layouts/spring-electrical.d.ts +224 -0
  80. package/dist/src/layouts/spring-electrical.d.ts.map +1 -0
  81. package/dist/src/layouts/spring-electrical.js +665 -0
  82. package/dist/src/layouts/spring-electrical.js.map +1 -0
  83. package/dist/src/memory/residency.d.ts +6 -2
  84. package/dist/src/memory/residency.d.ts.map +1 -1
  85. package/dist/src/memory/residency.js +84 -14
  86. package/dist/src/memory/residency.js.map +1 -1
  87. package/dist/src/primitives/core-shape.d.ts +38 -2
  88. package/dist/src/primitives/core-shape.d.ts.map +1 -1
  89. package/dist/src/primitives/core-shape.js +71 -3
  90. package/dist/src/primitives/core-shape.js.map +1 -1
  91. package/dist/src/primitives/grid-pyramid.d.ts +71 -0
  92. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -0
  93. package/dist/src/primitives/grid-pyramid.js +143 -0
  94. package/dist/src/primitives/grid-pyramid.js.map +1 -0
  95. package/dist/src/primitives/grid.d.ts +118 -0
  96. package/dist/src/primitives/grid.d.ts.map +1 -0
  97. package/dist/src/primitives/grid.js +225 -0
  98. package/dist/src/primitives/grid.js.map +1 -0
  99. package/dist/src/primitives/histogram.d.ts +67 -0
  100. package/dist/src/primitives/histogram.d.ts.map +1 -0
  101. package/dist/src/primitives/histogram.js +190 -0
  102. package/dist/src/primitives/histogram.js.map +1 -0
  103. package/dist/src/primitives/radix-sort.d.ts +75 -0
  104. package/dist/src/primitives/radix-sort.d.ts.map +1 -0
  105. package/dist/src/primitives/radix-sort.js +168 -0
  106. package/dist/src/primitives/radix-sort.js.map +1 -0
  107. package/dist/src/primitives/scan.d.ts +44 -0
  108. package/dist/src/primitives/scan.d.ts.map +1 -0
  109. package/dist/src/primitives/scan.js +151 -0
  110. package/dist/src/primitives/scan.js.map +1 -0
  111. package/dist/src/primitives/segmented-reduce.d.ts +25 -17
  112. package/dist/src/primitives/segmented-reduce.d.ts.map +1 -1
  113. package/dist/src/primitives/segmented-reduce.js +166 -47
  114. package/dist/src/primitives/segmented-reduce.js.map +1 -1
  115. package/dist/src/primitives/spmv.d.ts +18 -14
  116. package/dist/src/primitives/spmv.d.ts.map +1 -1
  117. package/dist/src/primitives/spmv.js +94 -58
  118. package/dist/src/primitives/spmv.js.map +1 -1
  119. package/dist/src/primitives/verify.d.ts +49 -0
  120. package/dist/src/primitives/verify.d.ts.map +1 -0
  121. package/dist/src/primitives/verify.js +229 -0
  122. package/dist/src/primitives/verify.js.map +1 -0
  123. package/dist/src/types/accelerator.d.ts +7 -3
  124. package/dist/src/types/accelerator.d.ts.map +1 -1
  125. package/dist/src/types/context.d.ts +53 -0
  126. package/dist/src/types/context.d.ts.map +1 -1
  127. package/dist/src/types/layout.d.ts +52 -0
  128. package/dist/src/types/layout.d.ts.map +1 -1
  129. package/dist/src/types/options.d.ts +43 -1
  130. package/dist/src/types/options.d.ts.map +1 -1
  131. package/dist/src/wgsl/counting-scatter.wgsl.d.ts +8 -0
  132. package/dist/src/wgsl/counting-scatter.wgsl.d.ts.map +1 -0
  133. package/dist/src/wgsl/counting-scatter.wgsl.js +17 -0
  134. package/dist/src/wgsl/counting-scatter.wgsl.js.map +1 -0
  135. package/dist/src/wgsl/fa2-attraction.wgsl.d.ts +23 -8
  136. package/dist/src/wgsl/fa2-attraction.wgsl.d.ts.map +1 -1
  137. package/dist/src/wgsl/fa2-attraction.wgsl.js +100 -17
  138. package/dist/src/wgsl/fa2-attraction.wgsl.js.map +1 -1
  139. package/dist/src/wgsl/fa2-integrate.wgsl.d.ts +7 -2
  140. package/dist/src/wgsl/fa2-integrate.wgsl.d.ts.map +1 -1
  141. package/dist/src/wgsl/fa2-integrate.wgsl.js +28 -2
  142. package/dist/src/wgsl/fa2-integrate.wgsl.js.map +1 -1
  143. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -2
  144. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
  145. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +14 -5
  146. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
  147. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +12 -1
  148. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
  149. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +54 -0
  150. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
  151. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +8 -0
  152. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -0
  153. package/dist/src/wgsl/grid-cell-key.wgsl.js +30 -0
  154. package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -0
  155. package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +8 -0
  156. package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts.map +1 -0
  157. package/dist/src/wgsl/grid-centroid-hub.wgsl.js +29 -0
  158. package/dist/src/wgsl/grid-centroid-hub.wgsl.js.map +1 -0
  159. package/dist/src/wgsl/grid-centroid.wgsl.d.ts +8 -0
  160. package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -0
  161. package/dist/src/wgsl/grid-centroid.wgsl.js +29 -0
  162. package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -0
  163. package/dist/src/wgsl/grid-downsample.wgsl.d.ts +7 -0
  164. package/dist/src/wgsl/grid-downsample.wgsl.d.ts.map +1 -0
  165. package/dist/src/wgsl/grid-downsample.wgsl.js +28 -0
  166. package/dist/src/wgsl/grid-downsample.wgsl.js.map +1 -0
  167. package/dist/src/wgsl/grid-far-field.wgsl.d.ts +13 -0
  168. package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -0
  169. package/dist/src/wgsl/grid-far-field.wgsl.js +98 -0
  170. package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -0
  171. package/dist/src/wgsl/grid-near-field.wgsl.d.ts +19 -0
  172. package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -0
  173. package/dist/src/wgsl/grid-near-field.wgsl.js +129 -0
  174. package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -0
  175. package/dist/src/wgsl/histogram.wgsl.d.ts +7 -0
  176. package/dist/src/wgsl/histogram.wgsl.d.ts.map +1 -0
  177. package/dist/src/wgsl/histogram.wgsl.js +15 -0
  178. package/dist/src/wgsl/histogram.wgsl.js.map +1 -0
  179. package/dist/src/wgsl/indirect-finalize.wgsl.d.ts +8 -0
  180. package/dist/src/wgsl/indirect-finalize.wgsl.d.ts.map +1 -0
  181. package/dist/src/wgsl/indirect-finalize.wgsl.js +26 -0
  182. package/dist/src/wgsl/indirect-finalize.wgsl.js.map +1 -0
  183. package/dist/src/wgsl/radix-hist.wgsl.d.ts +9 -0
  184. package/dist/src/wgsl/radix-hist.wgsl.d.ts.map +1 -0
  185. package/dist/src/wgsl/radix-hist.wgsl.js +31 -0
  186. package/dist/src/wgsl/radix-hist.wgsl.js.map +1 -0
  187. package/dist/src/wgsl/radix-scatter.wgsl.d.ts +9 -0
  188. package/dist/src/wgsl/radix-scatter.wgsl.d.ts.map +1 -0
  189. package/dist/src/wgsl/radix-scatter.wgsl.js +40 -0
  190. package/dist/src/wgsl/radix-scatter.wgsl.js.map +1 -0
  191. package/dist/src/wgsl/scan-add.wgsl.d.ts +6 -0
  192. package/dist/src/wgsl/scan-add.wgsl.d.ts.map +1 -0
  193. package/dist/src/wgsl/scan-add.wgsl.js +14 -0
  194. package/dist/src/wgsl/scan-add.wgsl.js.map +1 -0
  195. package/dist/src/wgsl/scan-block.wgsl.d.ts +8 -0
  196. package/dist/src/wgsl/scan-block.wgsl.d.ts.map +1 -0
  197. package/dist/src/wgsl/scan-block.wgsl.js +30 -0
  198. package/dist/src/wgsl/scan-block.wgsl.js.map +1 -0
  199. package/dist/src/wgsl/segmented-reduce.wgsl.d.ts +22 -8
  200. package/dist/src/wgsl/segmented-reduce.wgsl.d.ts.map +1 -1
  201. package/dist/src/wgsl/segmented-reduce.wgsl.js +84 -15
  202. package/dist/src/wgsl/segmented-reduce.wgsl.js.map +1 -1
  203. package/dist/src/wgsl/spmv-pull.wgsl.d.ts +22 -11
  204. package/dist/src/wgsl/spmv-pull.wgsl.d.ts.map +1 -1
  205. package/dist/src/wgsl/spmv-pull.wgsl.js +110 -36
  206. package/dist/src/wgsl/spmv-pull.wgsl.js.map +1 -1
  207. package/dist/tsconfig.build.tsbuildinfo +1 -1
  208. package/dist/webgpu-graph-algorithms.js +5016 -1130
  209. package/dist/webgpu-graph-algorithms.js.map +1 -1
  210. package/package.json +10 -7
  211. package/src/accelerator.ts +46 -12
  212. package/src/algorithms/components.ts +12 -16
  213. package/src/algorithms/degree.ts +58 -43
  214. package/src/algorithms/pagerank.ts +20 -18
  215. package/src/algorithms/power-iteration.ts +19 -18
  216. package/src/constants.ts +108 -8
  217. package/src/errors.ts +3 -1
  218. package/src/index.ts +25 -5
  219. package/src/kernel/dispatch.ts +18 -7
  220. package/src/kernel/kernel.ts +59 -5
  221. package/src/kernel/prelude.ts +15 -0
  222. package/src/kernel/profiler.ts +28 -4
  223. package/src/kernels.ts +378 -24
  224. package/src/layouts/calibrate.ts +187 -0
  225. package/src/layouts/force-simulation.ts +111 -26
  226. package/src/layouts/forceatlas2.ts +346 -324
  227. package/src/layouts/fruchterman-reingold.ts +918 -0
  228. package/src/layouts/model-common.ts +323 -0
  229. package/src/layouts/repulsion-grid.ts +451 -0
  230. package/src/layouts/spring-electrical.ts +845 -0
  231. package/src/memory/residency.ts +126 -20
  232. package/src/primitives/core-shape.ts +91 -4
  233. package/src/primitives/grid-pyramid.ts +221 -0
  234. package/src/primitives/grid.ts +349 -0
  235. package/src/primitives/histogram.ts +273 -0
  236. package/src/primitives/radix-sort.ts +246 -0
  237. package/src/primitives/scan.ts +197 -0
  238. package/src/primitives/segmented-reduce.ts +214 -56
  239. package/src/primitives/spmv.ts +125 -65
  240. package/src/primitives/verify.ts +249 -0
  241. package/src/types/accelerator.ts +15 -3
  242. package/src/types/context.ts +56 -0
  243. package/src/types/layout.ts +58 -0
  244. package/src/types/options.ts +45 -1
  245. package/src/wgsl/counting-scatter.wgsl.ts +16 -0
  246. package/src/wgsl/fa2-attraction.wgsl.ts +100 -17
  247. package/src/wgsl/fa2-integrate.wgsl.ts +28 -2
  248. package/src/wgsl/fa2-repulsion-exact.wgsl.ts +14 -5
  249. package/src/wgsl/fa2-stats-finalize.wgsl.ts +54 -0
  250. package/src/wgsl/grid-cell-key.wgsl.ts +29 -0
  251. package/src/wgsl/grid-centroid-hub.wgsl.ts +28 -0
  252. package/src/wgsl/grid-centroid.wgsl.ts +28 -0
  253. package/src/wgsl/grid-downsample.wgsl.ts +27 -0
  254. package/src/wgsl/grid-far-field.wgsl.ts +97 -0
  255. package/src/wgsl/grid-near-field.wgsl.ts +128 -0
  256. package/src/wgsl/histogram.wgsl.ts +14 -0
  257. package/src/wgsl/indirect-finalize.wgsl.ts +25 -0
  258. package/src/wgsl/radix-hist.wgsl.ts +30 -0
  259. package/src/wgsl/radix-scatter.wgsl.ts +39 -0
  260. package/src/wgsl/scan-add.wgsl.ts +13 -0
  261. package/src/wgsl/scan-block.wgsl.ts +29 -0
  262. package/src/wgsl/segmented-reduce.wgsl.ts +84 -15
  263. package/src/wgsl/spmv-pull.wgsl.ts +110 -36
  264. package/dist/chunks/context-CRbw2Wyo.js.map +0 -1
@@ -4,7 +4,12 @@
4
4
  * networkx mode `m |F|`, contract 4.6), the position update with no displacement clamp (D25), z never integrated
5
5
  * in 2D (spec 7.13), `oldForce` stored in paper mode for fixed nodes too (spec 7.11), and the partials A
6
6
  * (sum p, sum |p - c|^2, min, max with max |p - c|^2 in `max.w`) and C (sum |dp| over free rows, free count) in
7
- * uniform control flow.
7
+ * uniform control flow. `APPLY` (P5, spec 7.20) picks the integrator: 0 = the FA2 text, 1 = the Fruchterman-Reingold
8
+ * temperature cap `min(|F|, t)` along F, 2 = ngraph's semi-implicit Euler step with drag and the unit speed clamp
9
+ * over the velocity that lives in the `oldForce` slot (PD-2), which also folds the free kinetic energy into
10
+ * `partials.swingTraction.x` (PD-4); mode 1 folds the free force energy `sum |F|^2` into the same slot for K1's
11
+ * adaptive cooling and takes its temperature from the state block when `FA2_FLAG_ADAPTIVE` is set. The energy
12
+ * reduction runs under every mode; only its write is conditional.
8
13
  *
9
14
  * Body only (spec 3.5, D9); normative text (contract 4.5); the K5 sabotage mutations (P3-T5) are textual edits of it.
10
15
  */
@@ -25,6 +30,7 @@ fn integrate(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
25
30
  var p = vec4f(0.0);
26
31
  var free = false;
27
32
  var valid = false;
33
+ var ke = 0.0;
28
34
  if (i < P.n) {
29
35
  valid = true;
30
36
  let f = load_force(i);
@@ -33,7 +39,25 @@ fn integrate(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
33
39
  if (SWING_MODE == 0u) { swing_i = p.w * length(f - load_old(i)); } // paper: m |F(t) - F(t-1)|, recomputed inline (7.2)
34
40
  let factor = S.speed / (1.0 + sqrt(S.speed * swing_i));
35
41
  let fixed = mask_bit(fixedMask[i >> 5u], i);
36
- dp = select(f * factor, vec3f(0.0), fixed); // no clamp on dp (D25)
42
+ dp = select(f * factor, vec3f(0.0), fixed); // APPLY 0 (FA2): no clamp on dp (D25)
43
+ if (APPLY == 1u) { // APPLY 1 (FR, 7.20): move along F by min(|F|, t); a fixed node stays
44
+ let mag = length(f);
45
+ let t = select(P.temperature, S.temperature, (P.flags & FA2_FLAG_ADAPTIVE) != 0u); // adaptive cooling: K1's state temperature
46
+ dp = vec3f(0.0);
47
+ if (mag > 0.0 && !fixed) { dp = f * (min(mag, t) / mag); }
48
+ ke = select(0.0, dot(f, f), !fixed); // the force energy of Hu's step control, folded like the spring preset's kinetic energy
49
+ }
50
+ if (APPLY == 2u) { // APPLY 2 (spring-electrical): ngraph's Euler step over the velocity in the oldForce slot (PD-2)
51
+ var v = load_old(i);
52
+ let fd = f - P.dragCoefficient * v; // drag (generateCreateDragForce.js:18)
53
+ v = v + (P.timeStep / p.w) * fd; // v += (dt / m) F (generateIntegrator.js:27-29)
54
+ let sp = length(v);
55
+ if (sp > 1.0) { v = v / sp; } // the unit speed clamp (generateIntegrator.js:33-37)
56
+ if (P.dim == 2u) { v.z = 0.0; }
57
+ dp = select(P.timeStep * v, vec3f(0.0), fixed); // dp = dt v; a pinned body is skipped (generateIntegrator.js:21, 39-41)
58
+ if (!fixed) { store_old(i, v); }
59
+ ke = select(0.0, 0.5 * p.w * dot(v, v), !fixed); // partials B under APPLY 2 (PD-4)
60
+ }
37
61
  if (P.dim == 2u) { dp.z = 0.0; } // 2D never integrates z (7.13)
38
62
  p = vec4f(p.xyz + dp, p.w);
39
63
  pos[i] = p;
@@ -59,11 +83,13 @@ fn integrate(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
59
83
  let tHi = wg_reduce_vec4(hi, lid.x, 2u);
60
84
  let tDl = wg_reduce_f32(dl, lid.x, 0u);
61
85
  let tFr = wg_reduce_u32(fr, lid.x, 0u);
86
+ let tKe = wg_reduce_f32(ke, lid.x, 0u);
62
87
  if (lid.x == 0u) {
63
88
  let g = group_id(wid);
64
89
  partials[g].sum = tSum;
65
90
  partials[g].min = tLo;
66
91
  partials[g].max = tHi;
67
92
  partials[g].dispFree = vec2f(tDl, f32(tFr));
93
+ if (APPLY != 0u) { partials[g].swingTraction = vec2f(tKe, 0.0); } // overwrites K3's epilogue: K4 never runs under FR or the preset (PD-4)
68
94
  }
69
95
  }`;
@@ -1,6 +1,8 @@
1
1
  /**
2
2
  * The `fa2-repulsion-exact` kernel body (K3; contract 4.5; spec 7.6, 7.9, 7.10): the tiled all-pairs repulsion
3
- * `|F| = k m_i m_j / d` with the 0.01 distance floor and the antisymmetric coincident kick, the gravity epilogue
3
+ * `|F| = k m_i m_j / d` with the 0.01 distance floor and the antisymmetric coincident kick (`LAW` 0; P5's `LAW` 1
4
+ * is Fruchterman-Reingold `k^2 / d` and `LAW` 2 ngraph's Coulomb `-g m_i m_j / d^2`, both unfloored, spec 7.20,
5
+ * with the kick's magnitude taken from the law at d = 0.01, PD-10), the gravity epilogue
4
6
  * (GRAVITY_CENTER 0 = centroid, 1 = origin; STRONG_GRAVITY) added under the `valid` guard, `force += f`, and the
5
7
  * swing / traction workgroup reduction in uniform control flow (SWING_MODE 0 = paper: free nodes only against
6
8
  * `oldForce`; 1 = NetworkX: positions and forces mixed, every node) whose lane 0 writes
@@ -18,6 +20,11 @@ fn store_force(i: u32, f: vec3f) {
18
20
  force[3u * i + 2u] = f.z;
19
21
  }
20
22
  fn load_old(i: u32) -> vec3f { return vec3f(oldForce[3u * i], oldForce[3u * i + 1u], oldForce[3u * i + 2u]); }
23
+ fn kick_magnitude(mi: f32, mj: f32) -> f32 { // the law's magnitude at d = FA2_DIST_FLOOR (PD-10)
24
+ if (LAW == 1u) { return P.frK * P.frK / FA2_DIST_FLOOR; }
25
+ if (LAW == 2u) { return -P.coulomb * mi * mj / FA2_DIST_FLOOR_SQ; }
26
+ return P.scalingRatio * mi * mj / FA2_DIST_FLOOR;
27
+ }
21
28
  fn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: centroid (GRAVITY_CENTER 0) or origin (1); regular or strong
22
29
  var q = pi.xyz;
23
30
  if (GRAVITY_CENTER == 0u) { q = pi.xyz - S.centroid.xyz; }
@@ -45,13 +52,15 @@ fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
45
52
  if (o.w > 0.0 && jj != i) { // mass > 0 for every real node, 0 for the pad
46
53
  let d = pi.xyz - o.xyz;
47
54
  var d2 = dot(d, d);
48
- if (d2 < FA2_COINCIDENT_SQ) { // coincident: antisymmetric unit kick of magnitude k m_i m_j / 0.01 (7.2)
49
- f = f + kick_dir(i, jj, P.dim) * (P.scalingRatio * pi.w * o.w / FA2_DIST_FLOOR);
55
+ if (d2 < FA2_COINCIDENT_SQ) { // coincident: antisymmetric unit kick of the law's magnitude at d = 0.01 (7.2; PD-10)
56
+ f = f + kick_dir(i, jj, P.dim) * kick_magnitude(pi.w, o.w);
50
57
  continue;
51
58
  }
52
- d2 = max(d2, FA2_DIST_FLOOR_SQ); // d >= 0.01
59
+ if (LAW == 0u) { d2 = max(d2, FA2_DIST_FLOOR_SQ); } // FA2 alone floors d >= 0.01 (7.2); FR and coulomb are unfloored (7.20)
53
60
  let k = P.scalingRatio * pi.w * o.w;
54
- f = f + d * (k / d2); // |F| = k m_i m_j / d along d / d
61
+ if (LAW == 0u) { f = f + d * (k / d2); } // LAW 0 (FA2): |F| = k m_i m_j / d along d / d
62
+ if (LAW == 1u) { f = f + d * (P.frK * P.frK / d2); } // LAW 1 (FR, 7.20): |F| = k^2 / d, mass ignored
63
+ if (LAW == 2u) { f = f + d * (-P.coulomb * pi.w * o.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb, ngraph generateQuadTree.js:131-132): |F| = -g m_i m_j / d^2
55
64
  }
56
65
  }
57
66
  workgroupBarrier();
@@ -4,6 +4,17 @@
4
4
  * exact spec 3.3 value from `partials.max.w`), bounding box, mean displacement over free nodes (0 when every node
5
5
  * is fixed), the settle counter -- increments the iteration counter and writes the K1 half of the trace record.
6
6
  * On the first iteration after load() (`FA2_FLAG_FIRST`) it folds nothing and keeps the host-written state.
7
+ * `STATS_MODE` (P5, spec 7.20) adds the model statistic: 0 = the FA2 text, 1 = the Fruchterman-Reingold temperature
8
+ * of this iteration into `S.temperature` and the trace's `modelScalar` -- the uniform's under the linear schedule,
9
+ * or, with `FA2_FLAG_ADAPTIVE` set, the adaptive one: the previous iteration's force energy (K5 folds sum |F|^2 over
10
+ * free nodes into `partials.swingTraction.x`) against `S.frEnergy` grows the temperature by 1 / FR_COOLING_STEP after
11
+ * FR_COOLING_PATIENCE consecutive falls and shrinks it by FR_COOLING_STEP on a rise (Yifan Hu 2005, section 3.2);
12
+ * 2 = the spring-electrical kinetic energy K5 folded into `partials.swingTraction.x` (PD-4) into `S.kineticEnergy`
13
+ * and the trace. On the grid tier (`P.gridMax > 0`, P4-T10, PD-14) it also derives the grid frame of the next build
14
+ * from the fold (`extent = max(min(bboxExtent * GRID_BBOX_MARGIN, extentFactor * rmsRadius), GRID_EXTENT_FLOOR)`,
15
+ * `cellSize = extent / G`, `gridMin = centroid - extent / 2` with `cellSize` in `.w`, `invCellSize`, `eps = 0.25
16
+ * cellSize`), copies the previous iteration's pseudo-cell count and occupancy max into the state, and resets the
17
+ * hub counters; the exact tier writes `gridMax: 0` and binds two dummies, so the block is dead there.
7
18
  *
8
19
  * This file holds the kernel BODY only (spec 3.5, D9): no bind-group lines and no `override` lines -- the composer
9
20
  * emits them from the registry entry in src/kernels.ts (contract 3.10.1). The text is normative (contract 4.5) and
@@ -21,6 +32,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
21
32
  var hi = vec4f(-F32_MAX);
22
33
  var disp = 0.0;
23
34
  var free = 0u;
35
+ var ke = 0.0;
24
36
  if (fold) {
25
37
  for (var g = lid.x; g < groups; g = g + WG) { // sequential per lane in index order: deterministic
26
38
  let q = partials[g];
@@ -29,6 +41,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
29
41
  hi = max(hi, q.max);
30
42
  disp = disp + q.dispFree.x;
31
43
  free = free + u32(q.dispFree.y);
44
+ ke = ke + q.swingTraction.x;
32
45
  }
33
46
  }
34
47
  let tSum = wg_reduce_vec4(sum, lid.x, 0u);
@@ -36,6 +49,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
36
49
  let tHi = wg_reduce_vec4(hi, lid.x, 2u);
37
50
  let tDisp = wg_reduce_f32(disp, lid.x, 0u);
38
51
  let tFree = wg_reduce_u32(free, lid.x, 0u);
52
+ let tKe = wg_reduce_f32(ke, lid.x, 0u);
39
53
  if (lid.x == 0u) {
40
54
  if (fold) {
41
55
  let n = f32(P.n);
@@ -53,5 +67,45 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
53
67
  T[P.iterationIndex].meanDisplacement = S.meanDisplacement;
54
68
  T[P.iterationIndex].settledCount = S.settledCount;
55
69
  T[P.iterationIndex].iteration = S.iteration;
70
+ if (P.gridMax > 0u) { // the grid tier (7.7): the robust extent, the cell size, eps, last iteration's counts, the hub counter reset (PD-14)
71
+ let cells = P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u);
72
+ if (fold) {
73
+ let box = (S.max.xyz - S.min.xyz) * GRID_BBOX_MARGIN;
74
+ var bboxExtent = max(box.x, box.y);
75
+ if (P.dim == 3u) { bboxExtent = max(bboxExtent, box.z); }
76
+ let extent = max(min(bboxExtent, P.extentFactor * S.rmsRadius), GRID_EXTENT_FLOOR); // min(bbox, extentFactor x rms), floored (7.7)
77
+ let cellSize = extent / f32(P.gridMax);
78
+ S.gridMin = vec4f(S.centroid.xyz - vec3f(0.5 * extent), cellSize); // gridMin.w carries cellSize
79
+ S.invCellSize = 1.0 / cellSize;
80
+ S.eps = 0.25 * cellSize;
81
+ }
82
+ S.outsideGrid = cellHist[cells]; // the previous iteration's pseudo-cell count (0 after load)
83
+ S.maxCellOccupancy = atomicLoad(&hubCounters[1]);
84
+ atomicStore(&hubCounters[0], 0u);
85
+ atomicStore(&hubCounters[1], 0u);
86
+ }
87
+ if (STATS_MODE == 1u) { // FR: this iteration's temperature (7.20) into the state and the trace
88
+ if ((P.flags & FA2_FLAG_ADAPTIVE) != 0u) { // adaptive cooling (Yifan Hu 2005 3.2): tKe is the previous iteration's sum |F|^2 over free nodes
89
+ if (fold) {
90
+ var t = S.temperature;
91
+ if (tKe < S.frEnergy) {
92
+ S.frProgress = S.frProgress + 1u;
93
+ if (S.frProgress >= FR_COOLING_PATIENCE) { S.frProgress = 0u; t = t / FR_COOLING_STEP; }
94
+ } else {
95
+ S.frProgress = 0u;
96
+ t = t * FR_COOLING_STEP;
97
+ }
98
+ S.frEnergy = tKe;
99
+ S.temperature = t;
100
+ }
101
+ } else {
102
+ S.temperature = P.temperature;
103
+ }
104
+ T[P.iterationIndex].modelScalar = S.temperature;
105
+ }
106
+ if (STATS_MODE == 2u) { // spring-electrical: the kinetic energy K5 folded into partials B (PD-4); 0 on the first iteration after load()
107
+ S.kineticEnergy = tKe;
108
+ T[P.iterationIndex].modelScalar = tKe;
109
+ }
56
110
  }
57
111
  }`;
@@ -0,0 +1,29 @@
1
+ /**
2
+ * G1, the `grid-cell-key` kernel body (spec 7.7; P4-T8): the finest cell of every node from the state's robust extent,
3
+ * `floor((p - gridMin) * invCellSize)` (a multiply, correctly rounded everywhere: PD-10), linearised when every axis
4
+ * is in [0, G) and the outside pseudo-cell `cells` otherwise; `cellVal[i] = i`. The clamp before the floor keeps a
5
+ * far-away or NaN coordinate out of an out-of-range float-to-int conversion. Body only; normative text.
6
+ */
7
+ export const gridCellKeyWgsl = /* wgsl */ `
8
+ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
9
+
10
+ @compute @workgroup_size(WG)
11
+ fn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
12
+ let i = linear_id(wid, lid.x);
13
+ if (i >= P.n) { return; } // no barrier follows
14
+ let cells = grid_cells();
15
+ let gf = f32(P.gridMax);
16
+ let q = (pos[i].xyz - S.gridMin.xyz) * S.invCellSize; // PD-10: never a division
17
+ let c = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));
18
+ let g = i32(P.gridMax);
19
+ var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;
20
+ if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }
21
+ var key = cells; // the outside pseudo-cell (7.7)
22
+ if (inside) {
23
+ key = u32(c.x) + P.gridMax * u32(c.y);
24
+ if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }
25
+ }
26
+ cellKey[i] = key;
27
+ cellVal[i] = i;
28
+ }
29
+ `;
@@ -0,0 +1,28 @@
1
+ /**
2
+ * G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per hub cell of hubList, dispatched
3
+ * indirectly from hubArgs (the T1 finalize over hubCounters[0]); a WG-strided mass-weighted sum reduced by the
4
+ * prelude's tree. The work is guarded by `valid`, never an early return, so the reduction is uniform (PD-13). Body
5
+ * only; normative text.
6
+ */
7
+ export const gridCentroidHubWgsl = /* wgsl */ `
8
+ @compute @workgroup_size(WG)
9
+ fn grid_centroid_hub(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
10
+ let h = group_id(wid);
11
+ let valid = h < hubCount[0]; // a workgroup past the count sums nothing
12
+ var c = 0u;
13
+ var start = 0u;
14
+ var count = 0u;
15
+ if (valid) {
16
+ c = hubList[h];
17
+ start = cellStart[c];
18
+ count = cellStart[c + 1u] - start;
19
+ }
20
+ var acc = vec4f(0.0);
21
+ for (var k = start + lid.x; k < start + count; k = k + WG) { // strided over the cell's sorted range
22
+ let p = pos[sortedIdx[k]];
23
+ acc = acc + vec4f(p.xyz * p.w, p.w);
24
+ }
25
+ let t = wg_reduce_vec4(acc, lid.x, 0u); // uniform control flow: 256 -> 1
26
+ if (valid && lid.x == 0u) { pyramid[c] = t; }
27
+ }
28
+ `;
@@ -0,0 +1,28 @@
1
+ /**
2
+ * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the pseudo-cell included; the
3
+ * mass-weighted position sum of a cell's sorted range in index order (no atomics: deterministic), the largest
4
+ * occupancy into hubCounters[1], and cells above GRID_HUB_CELL entries appended to hubList for G4b (PD-13). Body
5
+ * only; normative text.
6
+ */
7
+ export const gridCentroidWgsl = /* wgsl */ `
8
+ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
9
+
10
+ @compute @workgroup_size(WG)
11
+ fn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
12
+ let c = linear_id(wid, lid.x);
13
+ if (c > grid_cells()) { return; } // cells [0, cells]: the pseudo-cell is index cells; no barrier follows
14
+ let start = cellStart[c];
15
+ let count = cellStart[c + 1u] - start;
16
+ atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration
17
+ if (count > GRID_HUB_CELL) { // a hub cell: G4b sums it (PD-13)
18
+ hubList[atomicAdd(&hubCounters[0], 1u)] = c;
19
+ return;
20
+ }
21
+ var acc = vec4f(0.0);
22
+ for (var k = start; k < start + count; k = k + 1u) { // sorted order: deterministic
23
+ let p = pos[sortedIdx[k]];
24
+ acc = acc + vec4f(p.xyz * p.w, p.w); // (sum m x, sum m y, sum m z, sum m)
25
+ }
26
+ pyramid[c] = acc;
27
+ }
28
+ `;
@@ -0,0 +1,27 @@
1
+ /**
2
+ * G5, the `grid-downsample` kernel body (spec 7.7; P4-T9): one dispatch per coarser level; every parent cell is the
3
+ * sum of its 4 (2D) or 8 (3D) children at the level below, read at P.childBase and written at P.parentBase (the
4
+ * pseudo-cell, index cells of level 0, is never a child). No atomics. Body only; normative text.
5
+ */
6
+ export const gridDownsampleWgsl = /* wgsl */ `
7
+ @compute @workgroup_size(WG)
8
+ fn grid_downsample(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
9
+ let pc = linear_id(wid, lid.x); // the parent cell inside its level
10
+ if (pc >= P.parentCells) { return; } // no barrier follows
11
+ let side = P.parentSide;
12
+ let cs = 2u * side; // the child level's side
13
+ let px = pc % side;
14
+ let py = (pc / side) % side;
15
+ let pz = pc / (side * side);
16
+ var acc = vec4f(0.0);
17
+ for (var dz = 0u; dz < P.depth; dz = dz + 1u) {
18
+ for (var dy = 0u; dy < 2u; dy = dy + 1u) {
19
+ for (var dx = 0u; dx < 2u; dx = dx + 1u) {
20
+ let child = (2u * px + dx) + cs * ((2u * py + dy) + cs * (2u * pz + dz));
21
+ acc = acc + pyramid[P.childBase + child];
22
+ }
23
+ }
24
+ }
25
+ pyramid[P.parentBase + pc] = acc;
26
+ }
27
+ `;
@@ -0,0 +1,97 @@
1
+ /**
2
+ * G6, the `grid-far-field` kernel body (spec 7.7; P4-T10; D24): per node `i = sortedIdx[t]`, its finest cell
3
+ * recomputed from `pos[i]` and the state (PD-10); for an inside node the coarsest level minus the 3x3 (3x3x3)
4
+ * around its coarsest cell, then at every finer level the 6x6 (6x6x6) block that is the parent's 3x3 minus this
5
+ * level's own 3x3 -- space tiled exactly once, no theta -- plus the outside pseudo-cell's centroid; for an outside
6
+ * node the coarsest level in full and no pseudo-cell. Every cell term is the per-cell law on the mass-weighted
7
+ * centroid (Gephi Region semantics), softened by `eps^2`: `LAW` 0 (FA2) `d * (k m_i M / d2)`, `LAW` 1 (FR, 7.20)
8
+ * `d * (k^2 M / d2)` (mass 1 per node, so `M` is the cell's count), `LAW` 2 (coulomb) `d * (-g m_i M / d2^1.5)`
9
+ * (P4-T13, PD-22). `force += f` (K2 wrote it). The loop bounds are `P.levels` and `P.gridMax` from the uniform,
10
+ * not a `LEVELS` override (PD-16, DEP-P4-G). Body only; normative text.
11
+ */
12
+ export const gridFarFieldWgsl = /* wgsl */ `
13
+ fn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }
14
+ fn store_force(i: u32, f: vec3f) {
15
+ force[3u * i] = f.x;
16
+ force[3u * i + 1u] = f.y;
17
+ force[3u * i + 2u] = f.z;
18
+ }
19
+ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
20
+ fn grid_side(level: u32) -> u32 { return P.gridMax >> level; }
21
+ fn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cell at index cells)
22
+ var base = 0u;
23
+ for (var l = 0u; l < level; l = l + 1u) {
24
+ let s = grid_side(l);
25
+ base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, 1u, l == 0u);
26
+ }
27
+ return base;
28
+ }
29
+ fn cell_at(level: u32, cx: i32, cy: i32, cz: i32) -> u32 {
30
+ let s = grid_side(level);
31
+ return level_base(level) + u32(cx) + s * (u32(cy) + select(0u, s * u32(cz), P.dim == 3u));
32
+ }
33
+ fn cell_force(pi: vec4f, q: vec4f) -> vec3f { // one far-field term, softened by state.eps (7.7)
34
+ if (q.w <= 0.0) { return vec3f(0.0); } // an empty cell
35
+ let d = pi.xyz - q.xyz / q.w; // to the mass-weighted centroid
36
+ let d2 = dot(d, d) + S.eps * S.eps;
37
+ if (LAW == 1u) { return d * (P.frK * P.frK * q.w / d2); } // LAW 1 (FR, 7.20): k^2 / d per node, q.w nodes at the centroid
38
+ if (LAW == 2u) { return d * (-P.coulomb * pi.w * q.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb): -g m_i M_cell / d^2
39
+ return d * (P.scalingRatio * pi.w * q.w / d2); // LAW 0 (FA2): |F| = k m_i M_cell / d
40
+ }
41
+
42
+ @compute @workgroup_size(WG)
43
+ fn grid_far_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
44
+ let t = linear_id(wid, lid.x);
45
+ if (t >= P.n) { return; } // no barrier follows
46
+ let i = sortedIdx[t]; // sorted order (D24)
47
+ let pi = pos[i];
48
+ let gf = f32(P.gridMax);
49
+ let q = (pi.xyz - S.gridMin.xyz) * S.invCellSize; // PD-10
50
+ var c0 = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));
51
+ if (P.dim == 2u) { c0.z = 0; } // 2D: one z plane; the loops below visit cz = 0 only, so the 3x3 test must see cz - 0
52
+ let g = i32(P.gridMax);
53
+ var inside = c0.x >= 0 && c0.x < g && c0.y >= 0 && c0.y < g;
54
+ if (P.dim == 3u) { inside = inside && c0.z >= 0 && c0.z < g; }
55
+ let top = P.levels - 1u;
56
+ let ts = i32(grid_side(top)); // the coarsest side (4)
57
+ let zTop = select(0, ts - 1, P.dim == 3u); // z ranges: one plane in 2D
58
+ var f = vec3f(0.0);
59
+ if (inside) {
60
+ let ct = c0 / i32(1u << top); // the node's coarsest cell
61
+ for (var cz = 0; cz <= zTop; cz = cz + 1) {
62
+ for (var cy = 0; cy < ts; cy = cy + 1) {
63
+ for (var cx = 0; cx < ts; cx = cx + 1) {
64
+ if (abs(cx - ct.x) <= 1 && abs(cy - ct.y) <= 1 && abs(cz - ct.z) <= 1) { continue; } // the 3x3(x3) is finer levels' work
65
+ f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);
66
+ }
67
+ }
68
+ }
69
+ for (var l = top; l > 0u; l = l - 1u) { // level l - 1: the parent's 3x3 at level l, refined, minus this level's own 3x3
70
+ let level = l - 1u;
71
+ let cl = c0 / i32(1u << level);
72
+ let cp = cl / 2;
73
+ let side = i32(grid_side(level));
74
+ let zLo = select(0, max(0, 2 * (cp.z - 1)), P.dim == 3u);
75
+ let zHi = select(0, min(side - 1, 2 * (cp.z + 1) + 1), P.dim == 3u);
76
+ for (var cz = zLo; cz <= zHi; cz = cz + 1) {
77
+ for (var cy = max(0, 2 * (cp.y - 1)); cy <= min(side - 1, 2 * (cp.y + 1) + 1); cy = cy + 1) {
78
+ for (var cx = max(0, 2 * (cp.x - 1)); cx <= min(side - 1, 2 * (cp.x + 1) + 1); cx = cx + 1) {
79
+ if (abs(cx - cl.x) <= 1 && abs(cy - cl.y) <= 1 && abs(cz - cl.z) <= 1) { continue; }
80
+ f = f + cell_force(pi, pyramid[cell_at(level, cx, cy, cz)]);
81
+ }
82
+ }
83
+ }
84
+ }
85
+ f = f + cell_force(pi, pyramid[grid_cells()]); // the outside pseudo-cell as one far-field term
86
+ } else {
87
+ for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (it would include itself)
88
+ for (var cy = 0; cy < ts; cy = cy + 1) {
89
+ for (var cx = 0; cx < ts; cx = cx + 1) {
90
+ f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);
91
+ }
92
+ }
93
+ }
94
+ }
95
+ store_force(i, load_force(i) + f);
96
+ }
97
+ `;
@@ -0,0 +1,128 @@
1
+ /**
2
+ * G7, the `grid-near-field` kernel body (spec 7.7, 7.6, 7.20; P4-T10, P4-T13; D24): per node `i = sortedIdx[t]`, the
3
+ * exact pair law of K3 (`LAW` 0: `|F| = k m_i m_j / d` with the 0.01 floor; `LAW` 1: FR's unfloored `k^2 / d`;
4
+ * `LAW` 2: the unfloored coulomb `-g m_i m_j / d^2`; the antisymmetric coincident kick at the law's magnitude at
5
+ * d = 0.01, PD-22) over the 9 (27) finest
6
+ * cells around its own, or over the outside pseudo-cell alone for an outside node; a cell above `nearMax` entries
7
+ * is sampled by `nearMax` INDEPENDENT draws with replacement, draw `k` reading the slot
8
+ * `lowbias32(((c ^ (iteration * 0x9E3779B9)) ^ seed) ^ (k * 0x85EBCA6B)) % count` (every slot's inclusion
9
+ * probability is `nearMax / count` whatever its position in the sorted order, so a duplicated draw is counted twice
10
+ * and the node itself, when drawn, is skipped and not replaced), and scaled by `others / sampled` where `sampled` is
11
+ * the realised number of draws that were not the node (the Horvitz-Thompson form of PD-15 / DEP-P4-K: given
12
+ * `sampled = s`, those `s` draws are i.i.d. uniform over the `others` slots, so the expectation of the scaled sum is
13
+ * the exact cell sum whenever `s >= 1`; the G4 record's G4-F2 row carries the measurement); then
14
+ * the fused epilogue of K3 (gravity, `force +=`, the swing / traction workgroup reduction in uniform control flow).
15
+ * The helpers `load_force`, `store_force`, `load_old`, `kick_magnitude`, `gravity_force` and the epilogue are K3's
16
+ * text. Body only; normative text.
17
+ */
18
+ export const gridNearFieldWgsl = /* wgsl */ `
19
+ fn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }
20
+ fn store_force(i: u32, f: vec3f) {
21
+ force[3u * i] = f.x;
22
+ force[3u * i + 1u] = f.y;
23
+ force[3u * i + 2u] = f.z;
24
+ }
25
+ fn load_old(i: u32) -> vec3f { return vec3f(oldForce[3u * i], oldForce[3u * i + 1u], oldForce[3u * i + 2u]); }
26
+ fn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: centroid (GRAVITY_CENTER 0) or origin (1); regular or strong
27
+ var q = pi.xyz;
28
+ if (GRAVITY_CENTER == 0u) { q = pi.xyz - S.centroid.xyz; }
29
+ if (STRONG_GRAVITY) { return -P.gravity * pi.w * q; }
30
+ let d = length(q);
31
+ if (d > FA2_DIST_FLOOR) { return -P.gravity * pi.w * q / d; }
32
+ return vec3f(0.0);
33
+ }
34
+ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
35
+ fn kick_magnitude(mi: f32, mj: f32) -> f32 { // the law's magnitude at d = FA2_DIST_FLOOR (the P5 plan's PD-10)
36
+ if (LAW == 1u) { return P.frK * P.frK / FA2_DIST_FLOOR; }
37
+ if (LAW == 2u) { return -P.coulomb * mi * mj / FA2_DIST_FLOOR_SQ; }
38
+ return P.scalingRatio * mi * mj / FA2_DIST_FLOOR;
39
+ }
40
+ fn pair_force(i: u32, pi: vec4f, jj: u32, o: vec4f) -> vec3f { // the exact pair law of K3 (7.6, 7.20): the floor (FA2 only), the coincident kick
41
+ let d = pi.xyz - o.xyz;
42
+ var d2 = dot(d, d);
43
+ if (d2 < FA2_COINCIDENT_SQ) { return kick_dir(i, jj, P.dim) * kick_magnitude(pi.w, o.w); }
44
+ if (LAW == 0u) { d2 = max(d2, FA2_DIST_FLOOR_SQ); }
45
+ let k = P.scalingRatio * pi.w * o.w;
46
+ if (LAW == 1u) { return d * (P.frK * P.frK / d2); }
47
+ if (LAW == 2u) { return d * (-P.coulomb * pi.w * o.w / (d2 * sqrt(d2))); }
48
+ return d * (k / d2);
49
+ }
50
+ fn cell_sum(i: u32, pi: vec4f, c: u32, own: bool) -> vec3f { // one finest cell: exact below nearMax entries, Horvitz-Thompson above (PD-15)
51
+ let start = cellStart[c];
52
+ let count = cellStart[c + 1u] - start;
53
+ var f = vec3f(0.0);
54
+ if (count <= P.nearMax) {
55
+ for (var k = start; k < start + count; k = k + 1u) {
56
+ let jj = sortedIdx[k];
57
+ if (jj != i) { f = f + pair_force(i, pi, jj, pos[jj]); }
58
+ }
59
+ return f;
60
+ }
61
+ let base = (c ^ (P.iterationIndex * 0x9E3779B9u)) ^ P.seed; // the per-iteration draw seed (7.16)
62
+ var sampled = 0u;
63
+ for (var k = 0u; k < P.nearMax; k = k + 1u) {
64
+ let jj = sortedIdx[start + (lowbias32(base ^ (k * 0x85EBCA6Bu)) % count)]; // draw k: independent inclusion, with replacement (PD-15)
65
+ if (jj == i) { continue; }
66
+ f = f + pair_force(i, pi, jj, pos[jj]);
67
+ sampled = sampled + 1u;
68
+ }
69
+ if (sampled == 0u) { return vec3f(0.0); }
70
+ let others = select(count, count - 1u, own);
71
+ return f * (f32(others) / f32(sampled)); // others / sampled over the realised sample (DEP-P4-K)
72
+ }
73
+
74
+ @compute @workgroup_size(WG)
75
+ fn grid_near_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
76
+ let t = linear_id(wid, lid.x);
77
+ let valid = t < P.n;
78
+ var i = 0u;
79
+ var pi = vec4f(0.0);
80
+ var f = vec3f(0.0);
81
+ if (valid) {
82
+ i = sortedIdx[t]; // sorted order (D24)
83
+ pi = pos[i];
84
+ let gf = f32(P.gridMax);
85
+ let q = (pi.xyz - S.gridMin.xyz) * S.invCellSize;
86
+ var c0 = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));
87
+ if (P.dim == 2u) { c0.z = 0; } // 2D: the one z plane
88
+ let g = i32(P.gridMax);
89
+ var inside = c0.x >= 0 && c0.x < g && c0.y >= 0 && c0.y < g;
90
+ if (P.dim == 3u) { inside = inside && c0.z >= 0 && c0.z < g; }
91
+ if (inside) {
92
+ let zr = select(0, 1, P.dim == 3u);
93
+ for (var dz = -zr; dz <= zr; dz = dz + 1) {
94
+ for (var dy = -1; dy <= 1; dy = dy + 1) {
95
+ for (var dx = -1; dx <= 1; dx = dx + 1) {
96
+ let cx = c0.x + dx;
97
+ let cy = c0.y + dy;
98
+ let cz = c0.z + dz;
99
+ if (cx < 0 || cx >= g || cy < 0 || cy >= g || cz < 0 || cz >= g) { continue; }
100
+ let c = u32(cx) + P.gridMax * (u32(cy) + select(0u, P.gridMax * u32(cz), P.dim == 3u));
101
+ f = f + cell_sum(i, pi, c, dx == 0 && dy == 0 && dz == 0);
102
+ }
103
+ }
104
+ }
105
+ } else {
106
+ f = cell_sum(i, pi, grid_cells(), true); // an outside node: the pseudo-cell alone
107
+ }
108
+ }
109
+ // epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it (K3's text)
110
+ var sw = 0.0;
111
+ var tr = 0.0;
112
+ if (valid) {
113
+ f = f + gravity_force(pi);
114
+ let fnew = load_force(i) + f;
115
+ store_force(i, fnew);
116
+ if (SWING_MODE == 1u) { // NetworkX: positions and forces mixed, every node (7.2)
117
+ sw = pi.w * length(pi.xyz - fnew);
118
+ tr = 0.5 * pi.w * length(pi.xyz + fnew);
119
+ } else if (!mask_bit(fixedMask[i >> 5u], i)) { // paper: free nodes only
120
+ let fold = load_old(i);
121
+ sw = pi.w * length(fnew - fold);
122
+ tr = 0.5 * pi.w * length(fnew + fold);
123
+ }
124
+ }
125
+ let tt = wg_reduce_vec4(vec4f(sw, tr, 0.0, 0.0), lid.x, 0u); // uniform control flow: 256 -> 1
126
+ if (lid.x == 0u) { partials[group_id(wid)].swingTraction = tt.xy; }
127
+ }
128
+ `;
@@ -0,0 +1,14 @@
1
+ /**
2
+ * The `histogram` kernel body (spec 6 row 5; P4-T3): one atomicAdd per key into the global `hist` (zeroed by a fill
3
+ * dispatch earlier in the pass, PD-4). Order-independent, hence deterministic. A key >= P.bins is not counted (the
4
+ * caller's contract; the grid's keys are always < cells + 1). Body only; normative text.
5
+ */
6
+ export const histogramWgsl = /* wgsl */ `
7
+ @compute @workgroup_size(WG)
8
+ fn histogram(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
9
+ let i = linear_id(wid, lid.x);
10
+ if (i >= P.count) { return; } // no barrier follows
11
+ let k = keys[i];
12
+ if (k < P.bins) { atomicAdd(&hist[k], 1u); } // order-independent: the count is the same whatever the schedule
13
+ }
14
+ `;
@@ -0,0 +1,25 @@
1
+ /**
2
+ * The `indirect-finalize` kernel body (spec 5.4; P4-T1): one lane turns the device-side count `counters[P.countIndex]`
3
+ * into the `(x, y, 1)` of an indirect dispatch by plan1d's rule (spec 5.2) and writes it with the count into the
4
+ * 16-byte slot `P.slot` of `args` (PD-2). No barrier follows the early return of the other lanes. Body only (spec 3.5,
5
+ * D9); the text is normative: the P4 sabotage rows are textual edits of it.
6
+ */
7
+ export const indirectFinalizeWgsl = /* wgsl */ `
8
+ @compute @workgroup_size(WG)
9
+ fn indirect_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
10
+ if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
11
+ let count = counters[P.countIndex];
12
+ let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap of (count + wg - 1) above 2^32 - wg: plan1d's rule (5.2) for ANY u32 count
13
+ var x = groups;
14
+ var y = 1u;
15
+ if (groups > MAX_WORKGROUPS_PER_DIM) { // the 2D split; y <= 1,025 for any u32 count
16
+ x = MAX_WORKGROUPS_PER_DIM;
17
+ y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;
18
+ }
19
+ let base = 4u * P.slot; // 16-byte slots: (x, y, 1, count) (PD-2)
20
+ args[base] = x;
21
+ args[base + 1u] = y;
22
+ args[base + 2u] = 1u;
23
+ args[base + 3u] = count;
24
+ }
25
+ `;
@@ -0,0 +1,30 @@
1
+ /**
2
+ * The `radix-hist` kernel body (spec 6 row 6; P4-T4): the RADIX_BINS-bin (2^8) digit histogram of one WG-wide block
3
+ * of keys, privatised in workgroup memory (atomics on workgroup memory are order-independent) and written
4
+ * DIGIT-MAJOR, `hist[digit * P.groups + group]`, so one exclusiveScan over the table yields, per digit, the offsets
5
+ * of the workgroups in workgroup order -- what a stable LSD scatter needs. The bitwise operators act on a KEY and a
6
+ * DIGIT, never on an arc index (house rule). Body only; normative text.
7
+ */
8
+ export const radixHistWgsl = /* wgsl */ `
9
+ var<workgroup> local: array<atomic<u32>, RADIX_BINS>;
10
+
11
+ @compute @workgroup_size(WG)
12
+ fn radix_hist(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
13
+ for (var b = lid.x; b < RADIX_BINS; b = b + WG) { atomicStore(&local[b], 0u); }
14
+ workgroupBarrier();
15
+ let g = group_id(wid);
16
+ let i = g * WG + lid.x;
17
+ if (i < P.count) {
18
+ let d = (keys[i] >> P.shift) & RADIX_DIGIT_MASK; // the pass's digit
19
+ atomicAdd(&local[d], 1u);
20
+ }
21
+ workgroupBarrier();
22
+ // Above MAX_1D_ITEMS plan1d pads the 2D grid to x * y >= P.groups workgroups; a padding workgroup (g >= P.groups)
23
+ // holds an all-zero table and its digit-major slot b * P.groups + g is digit b + 1's slot of a REAL group, so it
24
+ // must not store. Uniform per workgroup, no barrier inside. Every test size fits 1D (the ladder tops at 2^22); no
25
+ // case reaches this branch, so the guard is proved by reading, not by a run.
26
+ if (g < P.groups) {
27
+ for (var b = lid.x; b < RADIX_BINS; b = b + WG) { hist[b * P.groups + g] = atomicLoad(&local[b]); } // digit-major
28
+ }
29
+ }
30
+ `;
@@ -0,0 +1,39 @@
1
+ /**
2
+ * The `radix-scatter` kernel body (spec 6 row 6; P4-T4): the stable scatter of one LSD pass. Lane 0 ranks the block's
3
+ * keys serially in index order with a RADIX_BINS-entry counter table (PD-5: deterministic and stable, WG steps per
4
+ * workgroup; ponytail: the 8-way split ranking of GraphWaGu is the upgrade if T-6 shows the sort on the critical
5
+ * path), then every lane writes its key and value at `offsets[digit * P.groups + group] + rank`. Body only;
6
+ * normative text.
7
+ */
8
+ export const radixScatterWgsl = /* wgsl */ `
9
+ var<workgroup> rank: array<u32, WG>;
10
+ var<workgroup> cnt: array<u32, RADIX_BINS>;
11
+
12
+ @compute @workgroup_size(WG)
13
+ fn radix_scatter(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
14
+ for (var b = lid.x; b < RADIX_BINS; b = b + WG) { cnt[b] = 0u; }
15
+ let g = group_id(wid);
16
+ let i = g * WG + lid.x;
17
+ var key = 0u;
18
+ var d = 0u;
19
+ if (i < P.count) {
20
+ key = keys[i];
21
+ d = (key >> P.shift) & RADIX_DIGIT_MASK;
22
+ }
23
+ workgroupBarrier();
24
+ if (lid.x == 0u) { // serial stable ranking (PD-5)
25
+ let last = min(WG, P.count - g * WG);
26
+ for (var s = 0u; s < last; s = s + 1u) {
27
+ let ds = (keys[g * WG + s] >> P.shift) & RADIX_DIGIT_MASK;
28
+ rank[s] = cnt[ds];
29
+ cnt[ds] = cnt[ds] + 1u;
30
+ }
31
+ }
32
+ workgroupBarrier();
33
+ if (i < P.count) {
34
+ let dst = offsets[d * P.groups + g] + rank[lid.x];
35
+ keysOut[dst] = key;
36
+ valsOut[dst] = vals[i];
37
+ }
38
+ }
39
+ `;