@graphty/webgpu-graph-algorithms 0.5.1 → 0.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (243) hide show
  1. package/README.md +459 -58
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-BR7fx3vR.js → context-BXqgCifx.js} +190 -40
  4. package/dist/chunks/context-BXqgCifx.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/algorithms/components.d.ts.map +1 -1
  7. package/dist/src/algorithms/components.js +12 -13
  8. package/dist/src/algorithms/components.js.map +1 -1
  9. package/dist/src/algorithms/degree.d.ts +6 -8
  10. package/dist/src/algorithms/degree.d.ts.map +1 -1
  11. package/dist/src/algorithms/degree.js +58 -35
  12. package/dist/src/algorithms/degree.js.map +1 -1
  13. package/dist/src/algorithms/pagerank.d.ts.map +1 -1
  14. package/dist/src/algorithms/pagerank.js +16 -14
  15. package/dist/src/algorithms/pagerank.js.map +1 -1
  16. package/dist/src/algorithms/power-iteration.d.ts +2 -2
  17. package/dist/src/algorithms/power-iteration.d.ts.map +1 -1
  18. package/dist/src/algorithms/power-iteration.js +17 -14
  19. package/dist/src/algorithms/power-iteration.js.map +1 -1
  20. package/dist/src/constants.d.ts +38 -8
  21. package/dist/src/constants.d.ts.map +1 -1
  22. package/dist/src/constants.js +38 -8
  23. package/dist/src/constants.js.map +1 -1
  24. package/dist/src/errors.d.ts +3 -2
  25. package/dist/src/errors.d.ts.map +1 -1
  26. package/dist/src/errors.js +2 -1
  27. package/dist/src/errors.js.map +1 -1
  28. package/dist/src/index.d.ts +6 -4
  29. package/dist/src/index.d.ts.map +1 -1
  30. package/dist/src/index.js +8 -3
  31. package/dist/src/index.js.map +1 -1
  32. package/dist/src/kernel/dispatch.d.ts +8 -3
  33. package/dist/src/kernel/dispatch.d.ts.map +1 -1
  34. package/dist/src/kernel/dispatch.js +18 -7
  35. package/dist/src/kernel/dispatch.js.map +1 -1
  36. package/dist/src/kernel/kernel.d.ts +30 -1
  37. package/dist/src/kernel/kernel.d.ts.map +1 -1
  38. package/dist/src/kernel/kernel.js +49 -5
  39. package/dist/src/kernel/kernel.js.map +1 -1
  40. package/dist/src/kernel/prelude.d.ts.map +1 -1
  41. package/dist/src/kernel/prelude.js +6 -1
  42. package/dist/src/kernel/prelude.js.map +1 -1
  43. package/dist/src/kernel/profiler.d.ts +15 -3
  44. package/dist/src/kernel/profiler.d.ts.map +1 -1
  45. package/dist/src/kernel/profiler.js +27 -4
  46. package/dist/src/kernel/profiler.js.map +1 -1
  47. package/dist/src/kernels.d.ts +17 -7
  48. package/dist/src/kernels.d.ts.map +1 -1
  49. package/dist/src/kernels.js +323 -16
  50. package/dist/src/kernels.js.map +1 -1
  51. package/dist/src/layouts/calibrate.d.ts +51 -0
  52. package/dist/src/layouts/calibrate.d.ts.map +1 -0
  53. package/dist/src/layouts/calibrate.js +172 -0
  54. package/dist/src/layouts/calibrate.js.map +1 -0
  55. package/dist/src/layouts/force-simulation.d.ts +39 -4
  56. package/dist/src/layouts/force-simulation.d.ts.map +1 -1
  57. package/dist/src/layouts/force-simulation.js +71 -19
  58. package/dist/src/layouts/force-simulation.js.map +1 -1
  59. package/dist/src/layouts/forceatlas2.d.ts +107 -36
  60. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  61. package/dist/src/layouts/forceatlas2.js +296 -100
  62. package/dist/src/layouts/forceatlas2.js.map +1 -1
  63. package/dist/src/layouts/fruchterman-reingold.d.ts +73 -27
  64. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  65. package/dist/src/layouts/fruchterman-reingold.js +230 -70
  66. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  67. package/dist/src/layouts/model-common.d.ts +41 -3
  68. package/dist/src/layouts/model-common.d.ts.map +1 -1
  69. package/dist/src/layouts/model-common.js +74 -3
  70. package/dist/src/layouts/model-common.js.map +1 -1
  71. package/dist/src/layouts/repulsion-grid.d.ts +152 -0
  72. package/dist/src/layouts/repulsion-grid.d.ts.map +1 -0
  73. package/dist/src/layouts/repulsion-grid.js +318 -0
  74. package/dist/src/layouts/repulsion-grid.js.map +1 -0
  75. package/dist/src/layouts/spring-electrical.d.ts +75 -30
  76. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  77. package/dist/src/layouts/spring-electrical.js +231 -74
  78. package/dist/src/layouts/spring-electrical.js.map +1 -1
  79. package/dist/src/memory/residency.d.ts +6 -2
  80. package/dist/src/memory/residency.d.ts.map +1 -1
  81. package/dist/src/memory/residency.js +84 -14
  82. package/dist/src/memory/residency.js.map +1 -1
  83. package/dist/src/primitives/core-shape.d.ts +38 -2
  84. package/dist/src/primitives/core-shape.d.ts.map +1 -1
  85. package/dist/src/primitives/core-shape.js +71 -3
  86. package/dist/src/primitives/core-shape.js.map +1 -1
  87. package/dist/src/primitives/grid-pyramid.d.ts +71 -0
  88. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -0
  89. package/dist/src/primitives/grid-pyramid.js +143 -0
  90. package/dist/src/primitives/grid-pyramid.js.map +1 -0
  91. package/dist/src/primitives/grid.d.ts +118 -0
  92. package/dist/src/primitives/grid.d.ts.map +1 -0
  93. package/dist/src/primitives/grid.js +225 -0
  94. package/dist/src/primitives/grid.js.map +1 -0
  95. package/dist/src/primitives/histogram.d.ts +67 -0
  96. package/dist/src/primitives/histogram.d.ts.map +1 -0
  97. package/dist/src/primitives/histogram.js +190 -0
  98. package/dist/src/primitives/histogram.js.map +1 -0
  99. package/dist/src/primitives/radix-sort.d.ts +75 -0
  100. package/dist/src/primitives/radix-sort.d.ts.map +1 -0
  101. package/dist/src/primitives/radix-sort.js +168 -0
  102. package/dist/src/primitives/radix-sort.js.map +1 -0
  103. package/dist/src/primitives/scan.d.ts +44 -0
  104. package/dist/src/primitives/scan.d.ts.map +1 -0
  105. package/dist/src/primitives/scan.js +151 -0
  106. package/dist/src/primitives/scan.js.map +1 -0
  107. package/dist/src/primitives/segmented-reduce.d.ts +25 -17
  108. package/dist/src/primitives/segmented-reduce.d.ts.map +1 -1
  109. package/dist/src/primitives/segmented-reduce.js +166 -47
  110. package/dist/src/primitives/segmented-reduce.js.map +1 -1
  111. package/dist/src/primitives/spmv.d.ts +18 -14
  112. package/dist/src/primitives/spmv.d.ts.map +1 -1
  113. package/dist/src/primitives/spmv.js +94 -58
  114. package/dist/src/primitives/spmv.js.map +1 -1
  115. package/dist/src/primitives/verify.d.ts +49 -0
  116. package/dist/src/primitives/verify.d.ts.map +1 -0
  117. package/dist/src/primitives/verify.js +229 -0
  118. package/dist/src/primitives/verify.js.map +1 -0
  119. package/dist/src/types/context.d.ts +53 -0
  120. package/dist/src/types/context.d.ts.map +1 -1
  121. package/dist/src/types/layout.d.ts +20 -0
  122. package/dist/src/types/layout.d.ts.map +1 -1
  123. package/dist/src/wgsl/counting-scatter.wgsl.d.ts +8 -0
  124. package/dist/src/wgsl/counting-scatter.wgsl.d.ts.map +1 -0
  125. package/dist/src/wgsl/counting-scatter.wgsl.js +17 -0
  126. package/dist/src/wgsl/counting-scatter.wgsl.js.map +1 -0
  127. package/dist/src/wgsl/fa2-attraction.wgsl.d.ts +23 -11
  128. package/dist/src/wgsl/fa2-attraction.wgsl.d.ts.map +1 -1
  129. package/dist/src/wgsl/fa2-attraction.wgsl.js +98 -20
  130. package/dist/src/wgsl/fa2-attraction.wgsl.js.map +1 -1
  131. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +6 -2
  132. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
  133. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +22 -1
  134. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
  135. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +8 -0
  136. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -0
  137. package/dist/src/wgsl/grid-cell-key.wgsl.js +30 -0
  138. package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -0
  139. package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +8 -0
  140. package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts.map +1 -0
  141. package/dist/src/wgsl/grid-centroid-hub.wgsl.js +29 -0
  142. package/dist/src/wgsl/grid-centroid-hub.wgsl.js.map +1 -0
  143. package/dist/src/wgsl/grid-centroid.wgsl.d.ts +8 -0
  144. package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -0
  145. package/dist/src/wgsl/grid-centroid.wgsl.js +29 -0
  146. package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -0
  147. package/dist/src/wgsl/grid-downsample.wgsl.d.ts +7 -0
  148. package/dist/src/wgsl/grid-downsample.wgsl.d.ts.map +1 -0
  149. package/dist/src/wgsl/grid-downsample.wgsl.js +28 -0
  150. package/dist/src/wgsl/grid-downsample.wgsl.js.map +1 -0
  151. package/dist/src/wgsl/grid-far-field.wgsl.d.ts +13 -0
  152. package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -0
  153. package/dist/src/wgsl/grid-far-field.wgsl.js +98 -0
  154. package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -0
  155. package/dist/src/wgsl/grid-near-field.wgsl.d.ts +19 -0
  156. package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -0
  157. package/dist/src/wgsl/grid-near-field.wgsl.js +129 -0
  158. package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -0
  159. package/dist/src/wgsl/histogram.wgsl.d.ts +7 -0
  160. package/dist/src/wgsl/histogram.wgsl.d.ts.map +1 -0
  161. package/dist/src/wgsl/histogram.wgsl.js +15 -0
  162. package/dist/src/wgsl/histogram.wgsl.js.map +1 -0
  163. package/dist/src/wgsl/indirect-finalize.wgsl.d.ts +8 -0
  164. package/dist/src/wgsl/indirect-finalize.wgsl.d.ts.map +1 -0
  165. package/dist/src/wgsl/indirect-finalize.wgsl.js +26 -0
  166. package/dist/src/wgsl/indirect-finalize.wgsl.js.map +1 -0
  167. package/dist/src/wgsl/radix-hist.wgsl.d.ts +9 -0
  168. package/dist/src/wgsl/radix-hist.wgsl.d.ts.map +1 -0
  169. package/dist/src/wgsl/radix-hist.wgsl.js +31 -0
  170. package/dist/src/wgsl/radix-hist.wgsl.js.map +1 -0
  171. package/dist/src/wgsl/radix-scatter.wgsl.d.ts +9 -0
  172. package/dist/src/wgsl/radix-scatter.wgsl.d.ts.map +1 -0
  173. package/dist/src/wgsl/radix-scatter.wgsl.js +40 -0
  174. package/dist/src/wgsl/radix-scatter.wgsl.js.map +1 -0
  175. package/dist/src/wgsl/scan-add.wgsl.d.ts +6 -0
  176. package/dist/src/wgsl/scan-add.wgsl.d.ts.map +1 -0
  177. package/dist/src/wgsl/scan-add.wgsl.js +14 -0
  178. package/dist/src/wgsl/scan-add.wgsl.js.map +1 -0
  179. package/dist/src/wgsl/scan-block.wgsl.d.ts +8 -0
  180. package/dist/src/wgsl/scan-block.wgsl.d.ts.map +1 -0
  181. package/dist/src/wgsl/scan-block.wgsl.js +30 -0
  182. package/dist/src/wgsl/scan-block.wgsl.js.map +1 -0
  183. package/dist/src/wgsl/segmented-reduce.wgsl.d.ts +22 -8
  184. package/dist/src/wgsl/segmented-reduce.wgsl.d.ts.map +1 -1
  185. package/dist/src/wgsl/segmented-reduce.wgsl.js +84 -15
  186. package/dist/src/wgsl/segmented-reduce.wgsl.js.map +1 -1
  187. package/dist/src/wgsl/spmv-pull.wgsl.d.ts +22 -11
  188. package/dist/src/wgsl/spmv-pull.wgsl.d.ts.map +1 -1
  189. package/dist/src/wgsl/spmv-pull.wgsl.js +110 -36
  190. package/dist/src/wgsl/spmv-pull.wgsl.js.map +1 -1
  191. package/dist/tsconfig.build.tsbuildinfo +1 -1
  192. package/dist/webgpu-graph-algorithms.js +3815 -1003
  193. package/dist/webgpu-graph-algorithms.js.map +1 -1
  194. package/package.json +9 -8
  195. package/src/algorithms/components.ts +12 -16
  196. package/src/algorithms/degree.ts +58 -43
  197. package/src/algorithms/pagerank.ts +20 -18
  198. package/src/algorithms/power-iteration.ts +19 -18
  199. package/src/constants.ts +38 -8
  200. package/src/errors.ts +3 -1
  201. package/src/index.ts +14 -4
  202. package/src/kernel/dispatch.ts +18 -7
  203. package/src/kernel/kernel.ts +59 -5
  204. package/src/kernel/prelude.ts +9 -0
  205. package/src/kernel/profiler.ts +28 -4
  206. package/src/kernels.ts +356 -18
  207. package/src/layouts/calibrate.ts +187 -0
  208. package/src/layouts/force-simulation.ts +91 -23
  209. package/src/layouts/forceatlas2.ts +331 -106
  210. package/src/layouts/fruchterman-reingold.ts +255 -74
  211. package/src/layouts/model-common.ts +98 -3
  212. package/src/layouts/repulsion-grid.ts +451 -0
  213. package/src/layouts/spring-electrical.ts +257 -78
  214. package/src/memory/residency.ts +126 -20
  215. package/src/primitives/core-shape.ts +91 -4
  216. package/src/primitives/grid-pyramid.ts +221 -0
  217. package/src/primitives/grid.ts +349 -0
  218. package/src/primitives/histogram.ts +273 -0
  219. package/src/primitives/radix-sort.ts +246 -0
  220. package/src/primitives/scan.ts +197 -0
  221. package/src/primitives/segmented-reduce.ts +214 -56
  222. package/src/primitives/spmv.ts +125 -65
  223. package/src/primitives/verify.ts +249 -0
  224. package/src/types/context.ts +56 -0
  225. package/src/types/layout.ts +22 -0
  226. package/src/wgsl/counting-scatter.wgsl.ts +16 -0
  227. package/src/wgsl/fa2-attraction.wgsl.ts +98 -20
  228. package/src/wgsl/fa2-stats-finalize.wgsl.ts +22 -1
  229. package/src/wgsl/grid-cell-key.wgsl.ts +29 -0
  230. package/src/wgsl/grid-centroid-hub.wgsl.ts +28 -0
  231. package/src/wgsl/grid-centroid.wgsl.ts +28 -0
  232. package/src/wgsl/grid-downsample.wgsl.ts +27 -0
  233. package/src/wgsl/grid-far-field.wgsl.ts +97 -0
  234. package/src/wgsl/grid-near-field.wgsl.ts +128 -0
  235. package/src/wgsl/histogram.wgsl.ts +14 -0
  236. package/src/wgsl/indirect-finalize.wgsl.ts +25 -0
  237. package/src/wgsl/radix-hist.wgsl.ts +30 -0
  238. package/src/wgsl/radix-scatter.wgsl.ts +39 -0
  239. package/src/wgsl/scan-add.wgsl.ts +13 -0
  240. package/src/wgsl/scan-block.wgsl.ts +29 -0
  241. package/src/wgsl/segmented-reduce.wgsl.ts +84 -15
  242. package/src/wgsl/spmv-pull.wgsl.ts +110 -36
  243. package/dist/chunks/context-BR7fx3vR.js.map +0 -1
@@ -1 +1 @@
1
- {"version":3,"file":"layout.d.ts","sourceRoot":"","sources":["../../../src/types/layout.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH,OAAO,KAAK,EAAE,GAAG,EAAE,aAAa,EAAE,QAAQ,EAAE,MAAM,uBAAuB,CAAC;AAE1E,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,kBAAkB,CAAC;AAEzD,wGAAwG;AACxG,MAAM,WAAW,eAAe;IAC5B,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,QAAQ,CAAC,gBAAgB,EAAE,MAAM,CAAC;IAClC,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;IAC9B,QAAQ,CAAC,QAAQ,EAAE,SAAS,CAAC,MAAM,EAAE,MAAM,EAAE,MAAM,CAAC,CAAC;IACrD,QAAQ,CAAC,aAAa,EAAE,OAAO,GAAG,MAAM,CAAC;IACzC,QAAQ,CAAC,gBAAgB,EAAE,MAAM,GAAG,IAAI,CAAC;IACzC,QAAQ,CAAC,WAAW,EAAE,MAAM,GAAG,IAAI,CAAC;IACpC,QAAQ,CAAC,cAAc,EAAE,MAAM,GAAG,IAAI,CAAC;CAC1C;AAED;;;;GAIG;AACH,MAAM,WAAW,sBAAsB;IACnC,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,eAAe,EAAE,MAAM,CAAC;IACjC,QAAQ,CAAC,gBAAgB,EAAE,MAAM,CAAC;IAClC,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;CACjC;AAED,2CAA2C;AAC3C,MAAM,WAAW,gBAAiB,SAAQ,eAAe;IACrD,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,eAAe,EAAE,MAAM,CAAC;IACjC,QAAQ,CAAC,KAAK,EAAE,aAAa,CAAC,sBAAsB,CAAC,CAAC;CACzD;AAED;;;;GAIG;AACH,MAAM,WAAW,8BAA8B;IAC3C,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAC7B,QAAQ,CAAC,gBAAgB,EAAE,MAAM,CAAC;IAClC,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;CACjC;AAED,gHAAgH;AAChH,MAAM,WAAW,wBAAyB,SAAQ,eAAe;IAC7D,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAC7B,QAAQ,CAAC,KAAK,EAAE,aAAa,CAAC,8BAA8B,CAAC,CAAC;CACjE;AAED;;;;;;GAMG;AACH,MAAM,WAAW,2BAA2B;IACxC,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAC;IAC/B,QAAQ,CAAC,gBAAgB,EAAE,MAAM,CAAC;IAClC,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;CACjC;AAED,0IAA0I;AAC1I,MAAM,WAAW,qBAAsB,SAAQ,eAAe;IAC1D,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAC;IAC/B,QAAQ,CAAC,KAAK,EAAE,aAAa,CAAC,2BAA2B,CAAC,CAAC;CAC9D;AAED,qDAAqD;AACrD,MAAM,WAAW,UAAU;IACvB,QAAQ,CAAC,OAAO,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACtC,QAAQ,CAAC,KAAK,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACpC,QAAQ,CAAC,MAAM,CAAC,EAAE,WAAW,GAAG,SAAS,CAAC;CAC7C;AAED,4GAA4G;AAC5G,MAAM,WAAW,mBAAmB,CAAC,OAAO,EAAE,KAAK,SAAS,eAAe,CAAE,SAAQ,gBAAgB;IACjG,IAAI,CAAC,QAAQ,EAAE,aAAa,EAAE,SAAS,EAAE,GAAG,GAAG,IAAI,CAAC;IACpD;;;;;;OAMG;IACH,IAAI,CAAC,UAAU,CAAC,EAAE,MAAM,GAAG,OAAO,CAAC,IAAI,CAAC,CAAC;IACzC,QAAQ,CAAC,OAAO,EAAE,OAAO,CAAC;IAC1B,QAAQ,CAAC,IAAI,EAAE,QAAQ,GAAG,IAAI,CAAC;IAC/B,WAAW,CAAC,KAAK,EAAE,MAAM,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAClE,OAAO,IAAI,IAAI,CAAC;IAChB,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,QAAQ,CAAC,cAAc,EAAE,MAAM,CAAC;IAChC,QAAQ,CAAC,KAAK,EAAE,KAAK,CAAC;IACtB,KAAK,IAAI,OAAO,CAAC,IAAI,CAAC,CAAC;IACvB,MAAM,IAAI,IAAI,CAAC;IACf,SAAS,CAAC,KAAK,EAAE,OAAO,CAAC,OAAO,CAAC,GAAG,IAAI,CAAC;IACzC,GAAG,CAAC,OAAO,CAAC,EAAE,UAAU,GAAG,OAAO,CAAC,KAAK,CAAC,CAAC;IAC1C,OAAO,CAAC,CAAC,IAAI,EAAE,MAAM,GAAG,OAAO,CAAC,YAAY,GAAG,WAAW,CAAC,CAAC;CAC/D;AAED;;;GAGG;AACH,MAAM,WAAW,eAAe;IAC5B,QAAQ,CAAC,SAAS,CAAC,EAAE,OAAO,GAAG,MAAM,GAAG,MAAM,GAAG,SAAS,CAAC;IAC3D,QAAQ,CAAC,aAAa,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC5C,QAAQ,CAAC,OAAO,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACtC,QAAQ,CAAC,aAAa,CAAC,EAAE,OAAO,GAAG,SAAS,CAAC;IAC7C,QAAQ,CAAC,SAAS,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACxC,QAAQ,CAAC,SAAS,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACxC,QAAQ,CAAC,YAAY,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC3C,QAAQ,CAAC,MAAM,CAAC,EAAE,OAAO,GAAG,UAAU,GAAG,SAAS,CAAC;CACtD;AAED,8FAA8F;AAC9F,MAAM,WAAW,oBAAoB;IACjC,QAAQ,CAAC,SAAS,EAAE,OAAO,GAAG,MAAM,GAAG,MAAM,CAAC;IAC9C,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAC;IAC/B,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;IACzB,QAAQ,CAAC,aAAa,EAAE,OAAO,CAAC;IAChC,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;IAC9B,QAAQ,CAAC,MAAM,EAAE,OAAO,GAAG,UAAU,CAAC;CACzC"}
1
+ {"version":3,"file":"layout.d.ts","sourceRoot":"","sources":["../../../src/types/layout.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH,OAAO,KAAK,EAAE,GAAG,EAAE,aAAa,EAAE,QAAQ,EAAE,MAAM,uBAAuB,CAAC;AAE1E,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,kBAAkB,CAAC;AAEzD,wGAAwG;AACxG,MAAM,WAAW,eAAe;IAC5B,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,QAAQ,CAAC,gBAAgB,EAAE,MAAM,CAAC;IAClC,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;IAC9B,QAAQ,CAAC,QAAQ,EAAE,SAAS,CAAC,MAAM,EAAE,MAAM,EAAE,MAAM,CAAC,CAAC;IACrD,QAAQ,CAAC,aAAa,EAAE,OAAO,GAAG,MAAM,CAAC;IACzC,QAAQ,CAAC,gBAAgB,EAAE,MAAM,GAAG,IAAI,CAAC;IACzC,QAAQ,CAAC,WAAW,EAAE,MAAM,GAAG,IAAI,CAAC;IACpC,QAAQ,CAAC,cAAc,EAAE,MAAM,GAAG,IAAI,CAAC;CAC1C;AAED;;;;GAIG;AACH,MAAM,WAAW,sBAAsB;IACnC,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,eAAe,EAAE,MAAM,CAAC;IACjC,QAAQ,CAAC,gBAAgB,EAAE,MAAM,CAAC;IAClC,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;CACjC;AAED,2CAA2C;AAC3C,MAAM,WAAW,gBAAiB,SAAQ,eAAe;IACrD,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,eAAe,EAAE,MAAM,CAAC;IACjC,QAAQ,CAAC,KAAK,EAAE,aAAa,CAAC,sBAAsB,CAAC,CAAC;CACzD;AAED;;;;GAIG;AACH,MAAM,WAAW,8BAA8B;IAC3C,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAC7B,QAAQ,CAAC,gBAAgB,EAAE,MAAM,CAAC;IAClC,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;CACjC;AAED,gHAAgH;AAChH,MAAM,WAAW,wBAAyB,SAAQ,eAAe;IAC7D,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAC7B,QAAQ,CAAC,KAAK,EAAE,aAAa,CAAC,8BAA8B,CAAC,CAAC;CACjE;AAED;;;;;;GAMG;AACH,MAAM,WAAW,2BAA2B;IACxC,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAC;IAC/B,QAAQ,CAAC,gBAAgB,EAAE,MAAM,CAAC;IAClC,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;CACjC;AAED,0IAA0I;AAC1I,MAAM,WAAW,qBAAsB,SAAQ,eAAe;IAC1D,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAC;IAC/B,QAAQ,CAAC,KAAK,EAAE,aAAa,CAAC,2BAA2B,CAAC,CAAC;CAC9D;AAED,qDAAqD;AACrD,MAAM,WAAW,UAAU;IACvB,QAAQ,CAAC,OAAO,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACtC,QAAQ,CAAC,KAAK,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACpC,QAAQ,CAAC,MAAM,CAAC,EAAE,WAAW,GAAG,SAAS,CAAC;CAC7C;AAED,4GAA4G;AAC5G,MAAM,WAAW,mBAAmB,CAAC,OAAO,EAAE,KAAK,SAAS,eAAe,CAAE,SAAQ,gBAAgB;IACjG,IAAI,CAAC,QAAQ,EAAE,aAAa,EAAE,SAAS,EAAE,GAAG,GAAG,IAAI,CAAC;IACpD;;;;;;OAMG;IACH,IAAI,CAAC,UAAU,CAAC,EAAE,MAAM,GAAG,OAAO,CAAC,IAAI,CAAC,CAAC;IACzC,QAAQ,CAAC,OAAO,EAAE,OAAO,CAAC;IAC1B,QAAQ,CAAC,IAAI,EAAE,QAAQ,GAAG,IAAI,CAAC;IAC/B,WAAW,CAAC,KAAK,EAAE,MAAM,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAClE,OAAO,IAAI,IAAI,CAAC;IAChB,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,QAAQ,CAAC,cAAc,EAAE,MAAM,CAAC;IAChC,QAAQ,CAAC,KAAK,EAAE,KAAK,CAAC;IACtB,KAAK,IAAI,OAAO,CAAC,IAAI,CAAC,CAAC;IACvB,MAAM,IAAI,IAAI,CAAC;IACf,SAAS,CAAC,KAAK,EAAE,OAAO,CAAC,OAAO,CAAC,GAAG,IAAI,CAAC;IACzC,GAAG,CAAC,OAAO,CAAC,EAAE,UAAU,GAAG,OAAO,CAAC,KAAK,CAAC,CAAC;IAC1C,OAAO,CAAC,CAAC,IAAI,EAAE,MAAM,GAAG,OAAO,CAAC,YAAY,GAAG,WAAW,CAAC,CAAC;CAC/D;AAED;;;GAGG;AACH,MAAM,WAAW,eAAe;IAC5B,QAAQ,CAAC,SAAS,CAAC,EAAE,OAAO,GAAG,MAAM,GAAG,MAAM,GAAG,SAAS,CAAC;IAC3D,QAAQ,CAAC,aAAa,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC5C,QAAQ,CAAC,OAAO,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACtC,QAAQ,CAAC,aAAa,CAAC,EAAE,OAAO,GAAG,SAAS,CAAC;IAC7C,QAAQ,CAAC,SAAS,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACxC,QAAQ,CAAC,SAAS,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACxC,QAAQ,CAAC,YAAY,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC3C,QAAQ,CAAC,MAAM,CAAC,EAAE,OAAO,GAAG,UAAU,GAAG,SAAS,CAAC;CACtD;AAED,8FAA8F;AAC9F,MAAM,WAAW,oBAAoB;IACjC,QAAQ,CAAC,SAAS,EAAE,OAAO,GAAG,MAAM,GAAG,MAAM,CAAC;IAC9C,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAC;IAC/B,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;IACzB,QAAQ,CAAC,aAAa,EAAE,OAAO,CAAC;IAChC,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;IAC9B,QAAQ,CAAC,MAAM,EAAE,OAAO,GAAG,UAAU,CAAC;CACzC;AAED,8GAA8G;AAC9G,MAAM,WAAW,gBAAgB;IAC7B,QAAQ,CAAC,KAAK,CAAC,EAAE,SAAS,MAAM,EAAE,GAAG,SAAS,CAAC;CAClD;AAED;;;;;;;;GAQG;AACH,MAAM,WAAW,cAAc;IAC3B,QAAQ,CAAC,cAAc,EAAE,MAAM,CAAC;IAChC,QAAQ,CAAC,cAAc,EAAE,QAAQ,CAAC,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC,CAAC;IAC1D,QAAQ,CAAC,aAAa,EAAE,QAAQ,CAAC,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC,CAAC;IACzD,QAAQ,CAAC,sBAAsB,EAAE,MAAM,CAAC;IACxC,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;CAChC"}
@@ -0,0 +1,8 @@
1
+ /**
2
+ * The `counting-scatter` kernel body (spec 6 row 5; P4-T3): the scatter of a counting sort. `start[k]` is the
3
+ * exclusive scan of the histogram, `cursor[k]` a zeroed per-bin atomic; an element takes the slot `start[k] +
4
+ * atomicAdd(&cursor[k], 1u)`. The order inside a bin depends on the schedule (set-deterministic, design 6). Body
5
+ * only; normative text.
6
+ */
7
+ export declare const countingScatterWgsl = "\n@compute @workgroup_size(WG)\nfn counting_scatter(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n if (i >= P.count) { return; } // no barrier follows\n let k = keys[i];\n let slot = atomicAdd(&cursor[k], 1u); // the per-bin cursor (6 row 5)\n outIndex[start[k] + slot] = i;\n}\n";
8
+ //# sourceMappingURL=counting-scatter.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"counting-scatter.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/counting-scatter.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,eAAO,MAAM,mBAAmB,gbAS/B,CAAC"}
@@ -0,0 +1,17 @@
1
+ /**
2
+ * The `counting-scatter` kernel body (spec 6 row 5; P4-T3): the scatter of a counting sort. `start[k]` is the
3
+ * exclusive scan of the histogram, `cursor[k]` a zeroed per-bin atomic; an element takes the slot `start[k] +
4
+ * atomicAdd(&cursor[k], 1u)`. The order inside a bin depends on the schedule (set-deterministic, design 6). Body
5
+ * only; normative text.
6
+ */
7
+ export const countingScatterWgsl = /* wgsl */ `
8
+ @compute @workgroup_size(WG)
9
+ fn counting_scatter(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
10
+ let i = linear_id(wid, lid.x);
11
+ if (i >= P.count) { return; } // no barrier follows
12
+ let k = keys[i];
13
+ let slot = atomicAdd(&cursor[k], 1u); // the per-bin cursor (6 row 5)
14
+ outIndex[start[k] + slot] = i;
15
+ }
16
+ `;
17
+ //# sourceMappingURL=counting-scatter.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"counting-scatter.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/counting-scatter.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,CAAC,MAAM,mBAAmB,GAAG,UAAU,CAAC;;;;;;;;;CAS7C,CAAC"}
@@ -1,15 +1,27 @@
1
1
  /**
2
- * K2 of the ForceAtlas2 iteration, `fa2-attraction` (spec 7.5; contract 4.5): the thread-per-row gather of the
3
- * attraction force over the undirected CSR rows (both arcs present, so the sum is symmetric with no atomics), the
4
- * linear or linlog law, the optional weights, the distributed-action division by the mass in `pos.w`, written as
5
- * the FIRST writer of `force` each iteration. P3 ships the thread-per-row tier over `[tierStart, tierEnd)` = `[0, n)`
6
- * with `USE_PERM = false`; the subgroup / workgroup tiers arrive with P4 (`TIER`). `LAW` (P5, spec 7.20) picks the
7
- * pair law: 0 = the FA2 text, 1 = Fruchterman-Reingold `d^2 / k` (unfloored), 2 = ngraph's Hooke spring
8
- * `k_s (d - L)`; under 1 / 2 the models compile `LINLOG = false` and `HAS_WEIGHTS = false`, and the law overwrites
9
- * `w` so weights are ignored either way.
2
+ * K2 of the ForceAtlas2 iteration, `fa2-attraction` (spec 7.5; contract 4.5): the gather of the attraction force
3
+ * over the undirected CSR rows (both arcs present, so the sum is symmetric with no atomics), the linear or linlog
4
+ * law, the optional weights, the distributed-action division by the mass in `pos.w`, written as the FIRST writer of
5
+ * `force` each iteration (or combined into it under `P.accumulate`, the windowed pattern of 4.2). The arcs are read
6
+ * inside the bound window [P.arcBase, P.arcEnd). P4-T5 (PD-6, PD-7) gives it the three degree tiers: TIER 0 is one
7
+ * row per thread over `[P.tierStart, P.tierEnd)`; TIER 1 is 32 lanes per row over `[P.hiEnd, P.midEnd)` with a
8
+ * five-step tree in workgroup memory; TIER 2 is one workgroup per row over `[0, P.hiEnd)` through `wg_reduce_vec4`
9
+ * (hence `needs: ["subgroups"]`); the row is `perm[row]` under USE_PERM. `LAW` (P5, spec 7.20) picks the pair law:
10
+ * 0 = the FA2 text, 1 = Fruchterman-Reingold `d^2 / k` (unfloored), 2 = ngraph's Hooke spring `k_s (d - L)`; under
11
+ * 1 / 2 the models compile `LINLOG = false` and `HAS_WEIGHTS = false`, and the law overwrites `w` so weights are
12
+ * ignored either way.
10
13
  *
11
- * Body only (spec 3.5, D9); normative text (contract 4.5); the K2 sabotage mutations (P3-T5) are textual edits of it.
14
+ * Body only (spec 3.5, D9); normative text (contract 4.5); the K2 sabotage mutations (P3-T5, P5-T7, P4-T5 / T6) are
15
+ * textual edits of it.
16
+ *
17
+ * TIER 0 folds its row through `row_force_dense`, a stride-one copy of `row_force`, because the shader compiler
18
+ * emits `row_force(i, 0u, 1u)` as a call and leaves the stride in a parameter: the loop then walks the row with a
19
+ * runtime step, which costs the strength-reduced addressing into colIdx / weights and the unrolling that keeps
20
+ * several loads in flight per thread. TIER 0 runs on every load -- alone when no row reaches degree 32, and over
21
+ * the low-degree rows, which are most of them, when the tiers are bound. The two folds spell their locals apart
22
+ * (`arc` / `nbr` / `weight` / `total` against `a` / `j` / `w` / `f`) so that each sabotage row names exactly one
23
+ * of them.
12
24
  */
13
- /** The K2 body: entry point `attraction`; an early return is legal here because no barrier follows (spec 3.5 rule 1). */
14
- export declare const fa2AttractionWgsl = "fn store_force(i: u32, f: vec3f) {\n force[3u * i] = f.x;\n force[3u * i + 1u] = f.y;\n force[3u * i + 2u] = f.z;\n}\n\n@compute @workgroup_size(WG)\nfn attraction(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let row = linear_id(wid, lid.x) + P.tierStart;\n if (row >= P.tierEnd) { return; } // no barrier follows in this tier (3.5 rule 1)\n let i = select(row, perm[row], USE_PERM);\n let pi = pos[i]; // xyz + mass in one load (D23)\n var f = vec3f(0.0);\n for (var a = rowPtr[i]; a < rowPtr[i + 1u]; a = a + 1u) {\n let j = colIdx[a];\n if (j == i) { continue; } // a self-loop exerts no force\n var w = 1.0;\n if (HAS_WEIGHTS) { w = weights[a]; }\n let d = pos[j].xyz - pi.xyz; // toward j\n let len = max(length(d), FA2_DIST_FLOOR);\n if (LAW == 1u) { w = length(d) / P.frK; } // LAW 1 (FR, 7.20): |F| = d^2 / k along d / d, unfloored; the linear select below applies w as is\n if (LAW == 2u) { w = P.springCoefficient * (len - P.springLength) / len; } // LAW 2 (spring, ngraph generateCreateSpringForce.js:33-36): Hooke k_s (d - L) toward j\n let mag = select(w, w * log(1.0 + len) / len, LINLOG); // linear: |F| = w len; linlog: |F| = w log(1 + len)\n f = f + d * mag;\n }\n if (DISTRIBUTED) { f = f / pi.w; }\n store_force(i, f); // overwrites: attraction is the first writer of force each iteration\n}";
25
+ /** The K2 body: entry point `attraction`; the tier bodies are functions called under the uniform `TIER` override, so the barriers of `tiered` are reached in uniform control flow and `tier0`'s early return is legal (spec 3.5 rule 1). */
26
+ export declare const fa2AttractionWgsl = "fn store_force(i: u32, f: vec3f) {\n force[3u * i] = f.x;\n force[3u * i + 1u] = f.y;\n force[3u * i + 2u] = f.z;\n}\nfn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }\nfn row_node(row: u32) -> u32 { return select(row, perm[row], USE_PERM); }\nfn row_force(i: u32, lane: u32, step: u32) -> vec3f { // the arcs of row i this lane walks inside the bound window [P.arcBase, P.arcEnd) (4.2)\n let pi = pos[i]; // xyz + mass in one load (D23)\n let a0 = max(rowPtr[i], P.arcBase);\n let a1 = min(rowPtr[i + 1u], P.arcEnd);\n var f = vec3f(0.0);\n for (var a = a0 + lane; a < a1; a = a + step) {\n let j = colIdx[a - P.arcBase];\n if (j == i) { continue; } // a self-loop exerts no force\n var w = 1.0;\n if (HAS_WEIGHTS) { w = weights[a - P.arcBase]; }\n let d = pos[j].xyz - pi.xyz; // toward j\n let len = max(length(d), FA2_DIST_FLOOR);\n if (LAW == 1u) { w = length(d) / P.frK; } // LAW 1 (FR, 7.20): |F| = d^2 / k along d / d, unfloored; the linear select below applies w as is\n if (LAW == 2u) { w = P.springCoefficient * (len - P.springLength) / len; } // LAW 2 (spring, ngraph generateCreateSpringForce.js:33-36): Hooke k_s (d - L) toward j\n let mag = select(w, w * log(1.0 + len) / len, LINLOG); // linear: |F| = w len; linlog: |F| = w log(1 + len)\n f = f + d * mag;\n }\n return f;\n}\nfn row_force_dense(i: u32) -> vec3f { // TIER 0's stride-one twin of row_force (see the header)\n let pi = pos[i];\n let lo = max(rowPtr[i], P.arcBase);\n let hi = min(rowPtr[i + 1u], P.arcEnd);\n var total = vec3f(0.0);\n for (var arc = lo; arc < hi; arc = arc + 1u) {\n let k = arc - P.arcBase; // the window-local index; this walk is contiguous\n let nbr = colIdx[k];\n if (nbr == i) { continue; } // a self-loop exerts no force\n var weight = 1.0;\n if (HAS_WEIGHTS) { weight = weights[k]; }\n let d = pos[nbr].xyz - pi.xyz; // toward the neighbour\n let len = max(length(d), FA2_DIST_FLOOR);\n if (LAW == 1u) { weight = length(d) / P.frK; } // LAW 1 (FR, 7.20), as in row_force\n if (LAW == 2u) { weight = P.springCoefficient * (len - P.springLength) / len; } // LAW 2 (spring), as in row_force\n let mag = select(weight, weight * log(1.0 + len) / len, LINLOG);\n total = total + d * mag;\n }\n return total;\n}\nfn finish(i: u32, f0: vec3f) {\n var f = f0;\n if (DISTRIBUTED) { f = f / pos[i].w; }\n if (P.accumulate == 1u) { f = f + load_force(i); } // the windowed loop of 4.2 (arcBase != 0 dispatches after the first)\n store_force(i, f); // overwrites: attraction is the first writer of force each iteration\n}\nfn tier0(wid: vec3<u32>, lane: u32) { // TIER 0: one row per thread over [tierStart, tierEnd); no barrier, so the early return is legal (3.5 rule 1)\n let row = linear_id(wid, lane) + P.tierStart;\n if (row >= P.tierEnd) { return; }\n let i = row_node(row);\n finish(i, row_force_dense(i));\n}\n\nvar<workgroup> sh: array<vec3f, WG>;\n\nfn tiered(wid: vec3<u32>, lid: u32) { // TIER 1: 32 lanes per row over [hiEnd, midEnd); TIER 2: WG lanes per row over [0, hiEnd) (PD-6, PD-7)\n let g = group_id(wid);\n var row = g;\n var end = P.hiEnd;\n var lane = lid;\n var step = WG;\n if (TIER == 1u) { row = P.hiEnd + g * (WG / 32u) + lid / 32u; end = P.midEnd; lane = lid % 32u; step = 32u; }\n let valid = row < end;\n var i = 0u;\n var f = vec3f(0.0);\n if (valid) { i = row_node(row); f = row_force(i, lane, step); }\n if (TIER == 1u) {\n sh[lid] = f;\n workgroupBarrier();\n for (var s = 16u; s >= 1u; s = s / 2u) { // the five-step tree over each 32-lane group; every lane runs every step\n var t = vec3f(0.0);\n if (lane < s) { t = sh[lid + s]; }\n workgroupBarrier();\n sh[lid] = sh[lid] + t;\n workgroupBarrier();\n }\n if (valid && lane == 0u) { finish(i, sh[lid]); }\n }\n if (TIER == 2u) {\n let t = wg_reduce_vec4(vec4f(f, 0.0), lid, 0u);\n if (valid && lid == 0u) { finish(i, t.xyz); }\n }\n}\n\n@compute @workgroup_size(WG)\nfn attraction(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n if (TIER == 0u) { tier0(wid, lid.x); } else { tiered(wid, lid.x); }\n}";
15
27
  //# sourceMappingURL=fa2-attraction.wgsl.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"fa2-attraction.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-attraction.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;GAWG;AAEH,yHAAyH;AACzH,eAAO,MAAM,iBAAiB,0mDA2B5B,CAAC"}
1
+ {"version":3,"file":"fa2-attraction.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-attraction.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AAEH,4OAA4O;AAC5O,eAAO,MAAM,iBAAiB,msJA6F5B,CAAC"}
@@ -1,34 +1,45 @@
1
1
  /**
2
- * K2 of the ForceAtlas2 iteration, `fa2-attraction` (spec 7.5; contract 4.5): the thread-per-row gather of the
3
- * attraction force over the undirected CSR rows (both arcs present, so the sum is symmetric with no atomics), the
4
- * linear or linlog law, the optional weights, the distributed-action division by the mass in `pos.w`, written as
5
- * the FIRST writer of `force` each iteration. P3 ships the thread-per-row tier over `[tierStart, tierEnd)` = `[0, n)`
6
- * with `USE_PERM = false`; the subgroup / workgroup tiers arrive with P4 (`TIER`). `LAW` (P5, spec 7.20) picks the
7
- * pair law: 0 = the FA2 text, 1 = Fruchterman-Reingold `d^2 / k` (unfloored), 2 = ngraph's Hooke spring
8
- * `k_s (d - L)`; under 1 / 2 the models compile `LINLOG = false` and `HAS_WEIGHTS = false`, and the law overwrites
9
- * `w` so weights are ignored either way.
2
+ * K2 of the ForceAtlas2 iteration, `fa2-attraction` (spec 7.5; contract 4.5): the gather of the attraction force
3
+ * over the undirected CSR rows (both arcs present, so the sum is symmetric with no atomics), the linear or linlog
4
+ * law, the optional weights, the distributed-action division by the mass in `pos.w`, written as the FIRST writer of
5
+ * `force` each iteration (or combined into it under `P.accumulate`, the windowed pattern of 4.2). The arcs are read
6
+ * inside the bound window [P.arcBase, P.arcEnd). P4-T5 (PD-6, PD-7) gives it the three degree tiers: TIER 0 is one
7
+ * row per thread over `[P.tierStart, P.tierEnd)`; TIER 1 is 32 lanes per row over `[P.hiEnd, P.midEnd)` with a
8
+ * five-step tree in workgroup memory; TIER 2 is one workgroup per row over `[0, P.hiEnd)` through `wg_reduce_vec4`
9
+ * (hence `needs: ["subgroups"]`); the row is `perm[row]` under USE_PERM. `LAW` (P5, spec 7.20) picks the pair law:
10
+ * 0 = the FA2 text, 1 = Fruchterman-Reingold `d^2 / k` (unfloored), 2 = ngraph's Hooke spring `k_s (d - L)`; under
11
+ * 1 / 2 the models compile `LINLOG = false` and `HAS_WEIGHTS = false`, and the law overwrites `w` so weights are
12
+ * ignored either way.
10
13
  *
11
- * Body only (spec 3.5, D9); normative text (contract 4.5); the K2 sabotage mutations (P3-T5) are textual edits of it.
14
+ * Body only (spec 3.5, D9); normative text (contract 4.5); the K2 sabotage mutations (P3-T5, P5-T7, P4-T5 / T6) are
15
+ * textual edits of it.
16
+ *
17
+ * TIER 0 folds its row through `row_force_dense`, a stride-one copy of `row_force`, because the shader compiler
18
+ * emits `row_force(i, 0u, 1u)` as a call and leaves the stride in a parameter: the loop then walks the row with a
19
+ * runtime step, which costs the strength-reduced addressing into colIdx / weights and the unrolling that keeps
20
+ * several loads in flight per thread. TIER 0 runs on every load -- alone when no row reaches degree 32, and over
21
+ * the low-degree rows, which are most of them, when the tiers are bound. The two folds spell their locals apart
22
+ * (`arc` / `nbr` / `weight` / `total` against `a` / `j` / `w` / `f`) so that each sabotage row names exactly one
23
+ * of them.
12
24
  */
13
- /** The K2 body: entry point `attraction`; an early return is legal here because no barrier follows (spec 3.5 rule 1). */
25
+ /** The K2 body: entry point `attraction`; the tier bodies are functions called under the uniform `TIER` override, so the barriers of `tiered` are reached in uniform control flow and `tier0`'s early return is legal (spec 3.5 rule 1). */
14
26
  export const fa2AttractionWgsl = /* wgsl */ `fn store_force(i: u32, f: vec3f) {
15
27
  force[3u * i] = f.x;
16
28
  force[3u * i + 1u] = f.y;
17
29
  force[3u * i + 2u] = f.z;
18
30
  }
19
-
20
- @compute @workgroup_size(WG)
21
- fn attraction(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
22
- let row = linear_id(wid, lid.x) + P.tierStart;
23
- if (row >= P.tierEnd) { return; } // no barrier follows in this tier (3.5 rule 1)
24
- let i = select(row, perm[row], USE_PERM);
31
+ fn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }
32
+ fn row_node(row: u32) -> u32 { return select(row, perm[row], USE_PERM); }
33
+ fn row_force(i: u32, lane: u32, step: u32) -> vec3f { // the arcs of row i this lane walks inside the bound window [P.arcBase, P.arcEnd) (4.2)
25
34
  let pi = pos[i]; // xyz + mass in one load (D23)
35
+ let a0 = max(rowPtr[i], P.arcBase);
36
+ let a1 = min(rowPtr[i + 1u], P.arcEnd);
26
37
  var f = vec3f(0.0);
27
- for (var a = rowPtr[i]; a < rowPtr[i + 1u]; a = a + 1u) {
28
- let j = colIdx[a];
38
+ for (var a = a0 + lane; a < a1; a = a + step) {
39
+ let j = colIdx[a - P.arcBase];
29
40
  if (j == i) { continue; } // a self-loop exerts no force
30
41
  var w = 1.0;
31
- if (HAS_WEIGHTS) { w = weights[a]; }
42
+ if (HAS_WEIGHTS) { w = weights[a - P.arcBase]; }
32
43
  let d = pos[j].xyz - pi.xyz; // toward j
33
44
  let len = max(length(d), FA2_DIST_FLOOR);
34
45
  if (LAW == 1u) { w = length(d) / P.frK; } // LAW 1 (FR, 7.20): |F| = d^2 / k along d / d, unfloored; the linear select below applies w as is
@@ -36,7 +47,74 @@ fn attraction(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_i
36
47
  let mag = select(w, w * log(1.0 + len) / len, LINLOG); // linear: |F| = w len; linlog: |F| = w log(1 + len)
37
48
  f = f + d * mag;
38
49
  }
39
- if (DISTRIBUTED) { f = f / pi.w; }
50
+ return f;
51
+ }
52
+ fn row_force_dense(i: u32) -> vec3f { // TIER 0's stride-one twin of row_force (see the header)
53
+ let pi = pos[i];
54
+ let lo = max(rowPtr[i], P.arcBase);
55
+ let hi = min(rowPtr[i + 1u], P.arcEnd);
56
+ var total = vec3f(0.0);
57
+ for (var arc = lo; arc < hi; arc = arc + 1u) {
58
+ let k = arc - P.arcBase; // the window-local index; this walk is contiguous
59
+ let nbr = colIdx[k];
60
+ if (nbr == i) { continue; } // a self-loop exerts no force
61
+ var weight = 1.0;
62
+ if (HAS_WEIGHTS) { weight = weights[k]; }
63
+ let d = pos[nbr].xyz - pi.xyz; // toward the neighbour
64
+ let len = max(length(d), FA2_DIST_FLOOR);
65
+ if (LAW == 1u) { weight = length(d) / P.frK; } // LAW 1 (FR, 7.20), as in row_force
66
+ if (LAW == 2u) { weight = P.springCoefficient * (len - P.springLength) / len; } // LAW 2 (spring), as in row_force
67
+ let mag = select(weight, weight * log(1.0 + len) / len, LINLOG);
68
+ total = total + d * mag;
69
+ }
70
+ return total;
71
+ }
72
+ fn finish(i: u32, f0: vec3f) {
73
+ var f = f0;
74
+ if (DISTRIBUTED) { f = f / pos[i].w; }
75
+ if (P.accumulate == 1u) { f = f + load_force(i); } // the windowed loop of 4.2 (arcBase != 0 dispatches after the first)
40
76
  store_force(i, f); // overwrites: attraction is the first writer of force each iteration
77
+ }
78
+ fn tier0(wid: vec3<u32>, lane: u32) { // TIER 0: one row per thread over [tierStart, tierEnd); no barrier, so the early return is legal (3.5 rule 1)
79
+ let row = linear_id(wid, lane) + P.tierStart;
80
+ if (row >= P.tierEnd) { return; }
81
+ let i = row_node(row);
82
+ finish(i, row_force_dense(i));
83
+ }
84
+
85
+ var<workgroup> sh: array<vec3f, WG>;
86
+
87
+ fn tiered(wid: vec3<u32>, lid: u32) { // TIER 1: 32 lanes per row over [hiEnd, midEnd); TIER 2: WG lanes per row over [0, hiEnd) (PD-6, PD-7)
88
+ let g = group_id(wid);
89
+ var row = g;
90
+ var end = P.hiEnd;
91
+ var lane = lid;
92
+ var step = WG;
93
+ if (TIER == 1u) { row = P.hiEnd + g * (WG / 32u) + lid / 32u; end = P.midEnd; lane = lid % 32u; step = 32u; }
94
+ let valid = row < end;
95
+ var i = 0u;
96
+ var f = vec3f(0.0);
97
+ if (valid) { i = row_node(row); f = row_force(i, lane, step); }
98
+ if (TIER == 1u) {
99
+ sh[lid] = f;
100
+ workgroupBarrier();
101
+ for (var s = 16u; s >= 1u; s = s / 2u) { // the five-step tree over each 32-lane group; every lane runs every step
102
+ var t = vec3f(0.0);
103
+ if (lane < s) { t = sh[lid + s]; }
104
+ workgroupBarrier();
105
+ sh[lid] = sh[lid] + t;
106
+ workgroupBarrier();
107
+ }
108
+ if (valid && lane == 0u) { finish(i, sh[lid]); }
109
+ }
110
+ if (TIER == 2u) {
111
+ let t = wg_reduce_vec4(vec4f(f, 0.0), lid, 0u);
112
+ if (valid && lid == 0u) { finish(i, t.xyz); }
113
+ }
114
+ }
115
+
116
+ @compute @workgroup_size(WG)
117
+ fn attraction(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
118
+ if (TIER == 0u) { tier0(wid, lid.x); } else { tiered(wid, lid.x); }
41
119
  }`;
42
120
  //# sourceMappingURL=fa2-attraction.wgsl.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"fa2-attraction.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-attraction.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;GAWG;AAEH,yHAAyH;AACzH,MAAM,CAAC,MAAM,iBAAiB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;EA2B1C,CAAC"}
1
+ {"version":3,"file":"fa2-attraction.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-attraction.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AAEH,4OAA4O;AAC5O,MAAM,CAAC,MAAM,iBAAiB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EA6F1C,CAAC"}
@@ -10,12 +10,16 @@
10
10
  * free nodes into `partials.swingTraction.x`) against `S.frEnergy` grows the temperature by 1 / FR_COOLING_STEP after
11
11
  * FR_COOLING_PATIENCE consecutive falls and shrinks it by FR_COOLING_STEP on a rise (Yifan Hu 2005, section 3.2);
12
12
  * 2 = the spring-electrical kinetic energy K5 folded into `partials.swingTraction.x` (PD-4) into `S.kineticEnergy`
13
- * and the trace.
13
+ * and the trace. On the grid tier (`P.gridMax > 0`, P4-T10, PD-14) it also derives the grid frame of the next build
14
+ * from the fold (`extent = max(min(bboxExtent * GRID_BBOX_MARGIN, extentFactor * rmsRadius), GRID_EXTENT_FLOOR)`,
15
+ * `cellSize = extent / G`, `gridMin = centroid - extent / 2` with `cellSize` in `.w`, `invCellSize`, `eps = 0.25
16
+ * cellSize`), copies the previous iteration's pseudo-cell count and occupancy max into the state, and resets the
17
+ * hub counters; the exact tier writes `gridMax: 0` and binds two dummies, so the block is dead there.
14
18
  *
15
19
  * This file holds the kernel BODY only (spec 3.5, D9): no bind-group lines and no `override` lines -- the composer
16
20
  * emits them from the registry entry in src/kernels.ts (contract 3.10.1). The text is normative (contract 4.5) and
17
21
  * is the target of the K1 sabotage mutations (test/helpers/sabotage.ts, P3-T5); amend the contract before editing.
18
22
  */
19
23
  /** The K1 body: entry point `stats_finalize`; calls the reduction helpers (`needs: ["subgroups"]`, contract 4.3). */
20
- export declare const fa2StatsFinalizeWgsl = "// K1: folds the previous integrate's partials into the state block (spec 7.4); one workgroup\n@compute @workgroup_size(WG)\nfn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n let groups = (P.n + WG - 1u) / WG;\n let fold = (P.flags & FA2_FLAG_FIRST) == 0u; // the first iteration after load() keeps the host-written state\n var sum = vec4f(0.0);\n var lo = vec4f(F32_MAX);\n var hi = vec4f(-F32_MAX);\n var disp = 0.0;\n var free = 0u;\n var ke = 0.0;\n if (fold) {\n for (var g = lid.x; g < groups; g = g + WG) { // sequential per lane in index order: deterministic\n let q = partials[g];\n sum = sum + q.sum;\n lo = min(lo, q.min);\n hi = max(hi, q.max);\n disp = disp + q.dispFree.x;\n free = free + u32(q.dispFree.y);\n ke = ke + q.swingTraction.x;\n }\n }\n let tSum = wg_reduce_vec4(sum, lid.x, 0u);\n let tLo = wg_reduce_vec4(lo, lid.x, 1u);\n let tHi = wg_reduce_vec4(hi, lid.x, 2u);\n let tDisp = wg_reduce_f32(disp, lid.x, 0u);\n let tFree = wg_reduce_u32(free, lid.x, 0u);\n let tKe = wg_reduce_f32(ke, lid.x, 0u);\n if (lid.x == 0u) {\n if (fold) {\n let n = f32(P.n);\n let c = tSum.xyz / n;\n S.centroid = vec4f(c, 0.0);\n S.rmsRadius = sqrt(max(tSum.w, 0.0) / n); // RMS radius about the previous centroid (7.17)\n S.min = vec4f(tLo.xyz, 0.0);\n S.max = vec4f(tHi.xyz, 0.0);\n S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)\n let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)\n S.meanDisplacement = meanDisp;\n S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);\n }\n S.iteration = S.iteration + 1u;\n T[P.iterationIndex].meanDisplacement = S.meanDisplacement;\n T[P.iterationIndex].settledCount = S.settledCount;\n T[P.iterationIndex].iteration = S.iteration;\n if (STATS_MODE == 1u) { // FR: this iteration's temperature (7.20) into the state and the trace\n if ((P.flags & FA2_FLAG_ADAPTIVE) != 0u) { // adaptive cooling (Yifan Hu 2005 3.2): tKe is the previous iteration's sum |F|^2 over free nodes\n if (fold) {\n var t = S.temperature;\n if (tKe < S.frEnergy) {\n S.frProgress = S.frProgress + 1u;\n if (S.frProgress >= FR_COOLING_PATIENCE) { S.frProgress = 0u; t = t / FR_COOLING_STEP; }\n } else {\n S.frProgress = 0u;\n t = t * FR_COOLING_STEP;\n }\n S.frEnergy = tKe;\n S.temperature = t;\n }\n } else {\n S.temperature = P.temperature;\n }\n T[P.iterationIndex].modelScalar = S.temperature;\n }\n if (STATS_MODE == 2u) { // spring-electrical: the kinetic energy K5 folded into partials B (PD-4); 0 on the first iteration after load()\n S.kineticEnergy = tKe;\n T[P.iterationIndex].modelScalar = tKe;\n }\n }\n}";
24
+ export declare const fa2StatsFinalizeWgsl = "// K1: folds the previous integrate's partials into the state block (spec 7.4); one workgroup\n@compute @workgroup_size(WG)\nfn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n let groups = (P.n + WG - 1u) / WG;\n let fold = (P.flags & FA2_FLAG_FIRST) == 0u; // the first iteration after load() keeps the host-written state\n var sum = vec4f(0.0);\n var lo = vec4f(F32_MAX);\n var hi = vec4f(-F32_MAX);\n var disp = 0.0;\n var free = 0u;\n var ke = 0.0;\n if (fold) {\n for (var g = lid.x; g < groups; g = g + WG) { // sequential per lane in index order: deterministic\n let q = partials[g];\n sum = sum + q.sum;\n lo = min(lo, q.min);\n hi = max(hi, q.max);\n disp = disp + q.dispFree.x;\n free = free + u32(q.dispFree.y);\n ke = ke + q.swingTraction.x;\n }\n }\n let tSum = wg_reduce_vec4(sum, lid.x, 0u);\n let tLo = wg_reduce_vec4(lo, lid.x, 1u);\n let tHi = wg_reduce_vec4(hi, lid.x, 2u);\n let tDisp = wg_reduce_f32(disp, lid.x, 0u);\n let tFree = wg_reduce_u32(free, lid.x, 0u);\n let tKe = wg_reduce_f32(ke, lid.x, 0u);\n if (lid.x == 0u) {\n if (fold) {\n let n = f32(P.n);\n let c = tSum.xyz / n;\n S.centroid = vec4f(c, 0.0);\n S.rmsRadius = sqrt(max(tSum.w, 0.0) / n); // RMS radius about the previous centroid (7.17)\n S.min = vec4f(tLo.xyz, 0.0);\n S.max = vec4f(tHi.xyz, 0.0);\n S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)\n let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)\n S.meanDisplacement = meanDisp;\n S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);\n }\n S.iteration = S.iteration + 1u;\n T[P.iterationIndex].meanDisplacement = S.meanDisplacement;\n T[P.iterationIndex].settledCount = S.settledCount;\n T[P.iterationIndex].iteration = S.iteration;\n if (P.gridMax > 0u) { // the grid tier (7.7): the robust extent, the cell size, eps, last iteration's counts, the hub counter reset (PD-14)\n let cells = P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u);\n if (fold) {\n let box = (S.max.xyz - S.min.xyz) * GRID_BBOX_MARGIN;\n var bboxExtent = max(box.x, box.y);\n if (P.dim == 3u) { bboxExtent = max(bboxExtent, box.z); }\n let extent = max(min(bboxExtent, P.extentFactor * S.rmsRadius), GRID_EXTENT_FLOOR); // min(bbox, extentFactor x rms), floored (7.7)\n let cellSize = extent / f32(P.gridMax);\n S.gridMin = vec4f(S.centroid.xyz - vec3f(0.5 * extent), cellSize); // gridMin.w carries cellSize\n S.invCellSize = 1.0 / cellSize;\n S.eps = 0.25 * cellSize;\n }\n S.outsideGrid = cellHist[cells]; // the previous iteration's pseudo-cell count (0 after load)\n S.maxCellOccupancy = atomicLoad(&hubCounters[1]);\n atomicStore(&hubCounters[0], 0u);\n atomicStore(&hubCounters[1], 0u);\n }\n if (STATS_MODE == 1u) { // FR: this iteration's temperature (7.20) into the state and the trace\n if ((P.flags & FA2_FLAG_ADAPTIVE) != 0u) { // adaptive cooling (Yifan Hu 2005 3.2): tKe is the previous iteration's sum |F|^2 over free nodes\n if (fold) {\n var t = S.temperature;\n if (tKe < S.frEnergy) {\n S.frProgress = S.frProgress + 1u;\n if (S.frProgress >= FR_COOLING_PATIENCE) { S.frProgress = 0u; t = t / FR_COOLING_STEP; }\n } else {\n S.frProgress = 0u;\n t = t * FR_COOLING_STEP;\n }\n S.frEnergy = tKe;\n S.temperature = t;\n }\n } else {\n S.temperature = P.temperature;\n }\n T[P.iterationIndex].modelScalar = S.temperature;\n }\n if (STATS_MODE == 2u) { // spring-electrical: the kinetic energy K5 folded into partials B (PD-4); 0 on the first iteration after load()\n S.kineticEnergy = tKe;\n T[P.iterationIndex].modelScalar = tKe;\n }\n }\n}";
21
25
  //# sourceMappingURL=fa2-stats-finalize.wgsl.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"fa2-stats-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AAEH,qHAAqH;AACrH,eAAO,MAAM,oBAAoB,86GAqE/B,CAAC"}
1
+ {"version":3,"file":"fa2-stats-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH,qHAAqH;AACrH,eAAO,MAAM,oBAAoB,snJAsF/B,CAAC"}
@@ -10,7 +10,11 @@
10
10
  * free nodes into `partials.swingTraction.x`) against `S.frEnergy` grows the temperature by 1 / FR_COOLING_STEP after
11
11
  * FR_COOLING_PATIENCE consecutive falls and shrinks it by FR_COOLING_STEP on a rise (Yifan Hu 2005, section 3.2);
12
12
  * 2 = the spring-electrical kinetic energy K5 folded into `partials.swingTraction.x` (PD-4) into `S.kineticEnergy`
13
- * and the trace.
13
+ * and the trace. On the grid tier (`P.gridMax > 0`, P4-T10, PD-14) it also derives the grid frame of the next build
14
+ * from the fold (`extent = max(min(bboxExtent * GRID_BBOX_MARGIN, extentFactor * rmsRadius), GRID_EXTENT_FLOOR)`,
15
+ * `cellSize = extent / G`, `gridMin = centroid - extent / 2` with `cellSize` in `.w`, `invCellSize`, `eps = 0.25
16
+ * cellSize`), copies the previous iteration's pseudo-cell count and occupancy max into the state, and resets the
17
+ * hub counters; the exact tier writes `gridMax: 0` and binds two dummies, so the block is dead there.
14
18
  *
15
19
  * This file holds the kernel BODY only (spec 3.5, D9): no bind-group lines and no `override` lines -- the composer
16
20
  * emits them from the registry entry in src/kernels.ts (contract 3.10.1). The text is normative (contract 4.5) and
@@ -62,6 +66,23 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
62
66
  T[P.iterationIndex].meanDisplacement = S.meanDisplacement;
63
67
  T[P.iterationIndex].settledCount = S.settledCount;
64
68
  T[P.iterationIndex].iteration = S.iteration;
69
+ if (P.gridMax > 0u) { // the grid tier (7.7): the robust extent, the cell size, eps, last iteration's counts, the hub counter reset (PD-14)
70
+ let cells = P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u);
71
+ if (fold) {
72
+ let box = (S.max.xyz - S.min.xyz) * GRID_BBOX_MARGIN;
73
+ var bboxExtent = max(box.x, box.y);
74
+ if (P.dim == 3u) { bboxExtent = max(bboxExtent, box.z); }
75
+ let extent = max(min(bboxExtent, P.extentFactor * S.rmsRadius), GRID_EXTENT_FLOOR); // min(bbox, extentFactor x rms), floored (7.7)
76
+ let cellSize = extent / f32(P.gridMax);
77
+ S.gridMin = vec4f(S.centroid.xyz - vec3f(0.5 * extent), cellSize); // gridMin.w carries cellSize
78
+ S.invCellSize = 1.0 / cellSize;
79
+ S.eps = 0.25 * cellSize;
80
+ }
81
+ S.outsideGrid = cellHist[cells]; // the previous iteration's pseudo-cell count (0 after load)
82
+ S.maxCellOccupancy = atomicLoad(&hubCounters[1]);
83
+ atomicStore(&hubCounters[0], 0u);
84
+ atomicStore(&hubCounters[1], 0u);
85
+ }
65
86
  if (STATS_MODE == 1u) { // FR: this iteration's temperature (7.20) into the state and the trace
66
87
  if ((P.flags & FA2_FLAG_ADAPTIVE) != 0u) { // adaptive cooling (Yifan Hu 2005 3.2): tKe is the previous iteration's sum |F|^2 over free nodes
67
88
  if (fold) {
@@ -1 +1 @@
1
- {"version":3,"file":"fa2-stats-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AAEH,qHAAqH;AACrH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EAqE7C,CAAC"}
1
+ {"version":3,"file":"fa2-stats-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH,qHAAqH;AACrH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EAsF7C,CAAC"}
@@ -0,0 +1,8 @@
1
+ /**
2
+ * G1, the `grid-cell-key` kernel body (spec 7.7; P4-T8): the finest cell of every node from the state's robust extent,
3
+ * `floor((p - gridMin) * invCellSize)` (a multiply, correctly rounded everywhere: PD-10), linearised when every axis
4
+ * is in [0, G) and the outside pseudo-cell `cells` otherwise; `cellVal[i] = i`. The clamp before the floor keeps a
5
+ * far-away or NaN coordinate out of an out-of-range float-to-int conversion. Body only; normative text.
6
+ */
7
+ export declare const gridCellKeyWgsl = "\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\n\n@compute @workgroup_size(WG)\nfn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n if (i >= P.n) { return; } // no barrier follows\n let cells = grid_cells();\n let gf = f32(P.gridMax);\n let q = (pos[i].xyz - S.gridMin.xyz) * S.invCellSize; // PD-10: never a division\n let c = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));\n let g = i32(P.gridMax);\n var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;\n if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }\n var key = cells; // the outside pseudo-cell (7.7)\n if (inside) {\n key = u32(c.x) + P.gridMax * u32(c.y);\n if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }\n }\n cellKey[i] = key;\n cellVal[i] = i;\n}\n";
8
+ //# sourceMappingURL=grid-cell-key.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-cell-key.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-cell-key.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,eAAO,MAAM,eAAe,uhCAsB3B,CAAC"}
@@ -0,0 +1,30 @@
1
+ /**
2
+ * G1, the `grid-cell-key` kernel body (spec 7.7; P4-T8): the finest cell of every node from the state's robust extent,
3
+ * `floor((p - gridMin) * invCellSize)` (a multiply, correctly rounded everywhere: PD-10), linearised when every axis
4
+ * is in [0, G) and the outside pseudo-cell `cells` otherwise; `cellVal[i] = i`. The clamp before the floor keeps a
5
+ * far-away or NaN coordinate out of an out-of-range float-to-int conversion. Body only; normative text.
6
+ */
7
+ export const gridCellKeyWgsl = /* wgsl */ `
8
+ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
9
+
10
+ @compute @workgroup_size(WG)
11
+ fn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
12
+ let i = linear_id(wid, lid.x);
13
+ if (i >= P.n) { return; } // no barrier follows
14
+ let cells = grid_cells();
15
+ let gf = f32(P.gridMax);
16
+ let q = (pos[i].xyz - S.gridMin.xyz) * S.invCellSize; // PD-10: never a division
17
+ let c = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));
18
+ let g = i32(P.gridMax);
19
+ var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;
20
+ if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }
21
+ var key = cells; // the outside pseudo-cell (7.7)
22
+ if (inside) {
23
+ key = u32(c.x) + P.gridMax * u32(c.y);
24
+ if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }
25
+ }
26
+ cellKey[i] = key;
27
+ cellVal[i] = i;
28
+ }
29
+ `;
30
+ //# sourceMappingURL=grid-cell-key.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-cell-key.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-cell-key.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,CAAC,MAAM,eAAe,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;CAsBzC,CAAC"}
@@ -0,0 +1,8 @@
1
+ /**
2
+ * G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per hub cell of hubList, dispatched
3
+ * indirectly from hubArgs (the T1 finalize over hubCounters[0]); a WG-strided mass-weighted sum reduced by the
4
+ * prelude's tree. The work is guarded by `valid`, never an early return, so the reduction is uniform (PD-13). Body
5
+ * only; normative text.
6
+ */
7
+ export declare const gridCentroidHubWgsl = "\n@compute @workgroup_size(WG)\nfn grid_centroid_hub(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let h = group_id(wid);\n let valid = h < hubCount[0]; // a workgroup past the count sums nothing\n var c = 0u;\n var start = 0u;\n var count = 0u;\n if (valid) {\n c = hubList[h];\n start = cellStart[c];\n count = cellStart[c + 1u] - start;\n }\n var acc = vec4f(0.0);\n for (var k = start + lid.x; k < start + count; k = k + WG) { // strided over the cell's sorted range\n let p = pos[sortedIdx[k]];\n acc = acc + vec4f(p.xyz * p.w, p.w);\n }\n let t = wg_reduce_vec4(acc, lid.x, 0u); // uniform control flow: 256 -> 1\n if (valid && lid.x == 0u) { pyramid[c] = t; }\n}\n";
8
+ //# sourceMappingURL=grid-centroid-hub.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-centroid-hub.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid-hub.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,eAAO,MAAM,mBAAmB,i1BAqB/B,CAAC"}
@@ -0,0 +1,29 @@
1
+ /**
2
+ * G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per hub cell of hubList, dispatched
3
+ * indirectly from hubArgs (the T1 finalize over hubCounters[0]); a WG-strided mass-weighted sum reduced by the
4
+ * prelude's tree. The work is guarded by `valid`, never an early return, so the reduction is uniform (PD-13). Body
5
+ * only; normative text.
6
+ */
7
+ export const gridCentroidHubWgsl = /* wgsl */ `
8
+ @compute @workgroup_size(WG)
9
+ fn grid_centroid_hub(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
10
+ let h = group_id(wid);
11
+ let valid = h < hubCount[0]; // a workgroup past the count sums nothing
12
+ var c = 0u;
13
+ var start = 0u;
14
+ var count = 0u;
15
+ if (valid) {
16
+ c = hubList[h];
17
+ start = cellStart[c];
18
+ count = cellStart[c + 1u] - start;
19
+ }
20
+ var acc = vec4f(0.0);
21
+ for (var k = start + lid.x; k < start + count; k = k + WG) { // strided over the cell's sorted range
22
+ let p = pos[sortedIdx[k]];
23
+ acc = acc + vec4f(p.xyz * p.w, p.w);
24
+ }
25
+ let t = wg_reduce_vec4(acc, lid.x, 0u); // uniform control flow: 256 -> 1
26
+ if (valid && lid.x == 0u) { pyramid[c] = t; }
27
+ }
28
+ `;
29
+ //# sourceMappingURL=grid-centroid-hub.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-centroid-hub.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid-hub.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,CAAC,MAAM,mBAAmB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;CAqB7C,CAAC"}
@@ -0,0 +1,8 @@
1
+ /**
2
+ * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the pseudo-cell included; the
3
+ * mass-weighted position sum of a cell's sorted range in index order (no atomics: deterministic), the largest
4
+ * occupancy into hubCounters[1], and cells above GRID_HUB_CELL entries appended to hubList for G4b (PD-13). Body
5
+ * only; normative text.
6
+ */
7
+ export declare const gridCentroidWgsl = "\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\n\n@compute @workgroup_size(WG)\nfn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let c = linear_id(wid, lid.x);\n if (c > grid_cells()) { return; } // cells [0, cells]: the pseudo-cell is index cells; no barrier follows\n let start = cellStart[c];\n let count = cellStart[c + 1u] - start;\n atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration\n if (count > GRID_HUB_CELL) { // a hub cell: G4b sums it (PD-13)\n hubList[atomicAdd(&hubCounters[0], 1u)] = c;\n return;\n }\n var acc = vec4f(0.0);\n for (var k = start; k < start + count; k = k + 1u) { // sorted order: deterministic\n let p = pos[sortedIdx[k]];\n acc = acc + vec4f(p.xyz * p.w, p.w); // (sum m x, sum m y, sum m z, sum m)\n }\n pyramid[c] = acc;\n}\n";
8
+ //# sourceMappingURL=grid-centroid.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-centroid.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,eAAO,MAAM,gBAAgB,+jCAqB5B,CAAC"}
@@ -0,0 +1,29 @@
1
+ /**
2
+ * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the pseudo-cell included; the
3
+ * mass-weighted position sum of a cell's sorted range in index order (no atomics: deterministic), the largest
4
+ * occupancy into hubCounters[1], and cells above GRID_HUB_CELL entries appended to hubList for G4b (PD-13). Body
5
+ * only; normative text.
6
+ */
7
+ export const gridCentroidWgsl = /* wgsl */ `
8
+ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
9
+
10
+ @compute @workgroup_size(WG)
11
+ fn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
12
+ let c = linear_id(wid, lid.x);
13
+ if (c > grid_cells()) { return; } // cells [0, cells]: the pseudo-cell is index cells; no barrier follows
14
+ let start = cellStart[c];
15
+ let count = cellStart[c + 1u] - start;
16
+ atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration
17
+ if (count > GRID_HUB_CELL) { // a hub cell: G4b sums it (PD-13)
18
+ hubList[atomicAdd(&hubCounters[0], 1u)] = c;
19
+ return;
20
+ }
21
+ var acc = vec4f(0.0);
22
+ for (var k = start; k < start + count; k = k + 1u) { // sorted order: deterministic
23
+ let p = pos[sortedIdx[k]];
24
+ acc = acc + vec4f(p.xyz * p.w, p.w); // (sum m x, sum m y, sum m z, sum m)
25
+ }
26
+ pyramid[c] = acc;
27
+ }
28
+ `;
29
+ //# sourceMappingURL=grid-centroid.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-centroid.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;CAqB1C,CAAC"}
@@ -0,0 +1,7 @@
1
+ /**
2
+ * G5, the `grid-downsample` kernel body (spec 7.7; P4-T9): one dispatch per coarser level; every parent cell is the
3
+ * sum of its 4 (2D) or 8 (3D) children at the level below, read at P.childBase and written at P.parentBase (the
4
+ * pseudo-cell, index cells of level 0, is never a child). No atomics. Body only; normative text.
5
+ */
6
+ export declare const gridDownsampleWgsl = "\n@compute @workgroup_size(WG)\nfn grid_downsample(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let pc = linear_id(wid, lid.x); // the parent cell inside its level\n if (pc >= P.parentCells) { return; } // no barrier follows\n let side = P.parentSide;\n let cs = 2u * side; // the child level's side\n let px = pc % side;\n let py = (pc / side) % side;\n let pz = pc / (side * side);\n var acc = vec4f(0.0);\n for (var dz = 0u; dz < P.depth; dz = dz + 1u) {\n for (var dy = 0u; dy < 2u; dy = dy + 1u) {\n for (var dx = 0u; dx < 2u; dx = dx + 1u) {\n let child = (2u * px + dx) + cs * ((2u * py + dy) + cs * (2u * pz + dz));\n acc = acc + pyramid[P.childBase + child];\n }\n }\n }\n pyramid[P.parentBase + pc] = acc;\n}\n";
7
+ //# sourceMappingURL=grid-downsample.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-downsample.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-downsample.wgsl.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH,eAAO,MAAM,kBAAkB,w8BAqB9B,CAAC"}
@@ -0,0 +1,28 @@
1
+ /**
2
+ * G5, the `grid-downsample` kernel body (spec 7.7; P4-T9): one dispatch per coarser level; every parent cell is the
3
+ * sum of its 4 (2D) or 8 (3D) children at the level below, read at P.childBase and written at P.parentBase (the
4
+ * pseudo-cell, index cells of level 0, is never a child). No atomics. Body only; normative text.
5
+ */
6
+ export const gridDownsampleWgsl = /* wgsl */ `
7
+ @compute @workgroup_size(WG)
8
+ fn grid_downsample(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
9
+ let pc = linear_id(wid, lid.x); // the parent cell inside its level
10
+ if (pc >= P.parentCells) { return; } // no barrier follows
11
+ let side = P.parentSide;
12
+ let cs = 2u * side; // the child level's side
13
+ let px = pc % side;
14
+ let py = (pc / side) % side;
15
+ let pz = pc / (side * side);
16
+ var acc = vec4f(0.0);
17
+ for (var dz = 0u; dz < P.depth; dz = dz + 1u) {
18
+ for (var dy = 0u; dy < 2u; dy = dy + 1u) {
19
+ for (var dx = 0u; dx < 2u; dx = dx + 1u) {
20
+ let child = (2u * px + dx) + cs * ((2u * py + dy) + cs * (2u * pz + dz));
21
+ acc = acc + pyramid[P.childBase + child];
22
+ }
23
+ }
24
+ }
25
+ pyramid[P.parentBase + pc] = acc;
26
+ }
27
+ `;
28
+ //# sourceMappingURL=grid-downsample.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-downsample.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-downsample.wgsl.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;CAqB5C,CAAC"}
@@ -0,0 +1,13 @@
1
+ /**
2
+ * G6, the `grid-far-field` kernel body (spec 7.7; P4-T10; D24): per node `i = sortedIdx[t]`, its finest cell
3
+ * recomputed from `pos[i]` and the state (PD-10); for an inside node the coarsest level minus the 3x3 (3x3x3)
4
+ * around its coarsest cell, then at every finer level the 6x6 (6x6x6) block that is the parent's 3x3 minus this
5
+ * level's own 3x3 -- space tiled exactly once, no theta -- plus the outside pseudo-cell's centroid; for an outside
6
+ * node the coarsest level in full and no pseudo-cell. Every cell term is the per-cell law on the mass-weighted
7
+ * centroid (Gephi Region semantics), softened by `eps^2`: `LAW` 0 (FA2) `d * (k m_i M / d2)`, `LAW` 1 (FR, 7.20)
8
+ * `d * (k^2 M / d2)` (mass 1 per node, so `M` is the cell's count), `LAW` 2 (coulomb) `d * (-g m_i M / d2^1.5)`
9
+ * (P4-T13, PD-22). `force += f` (K2 wrote it). The loop bounds are `P.levels` and `P.gridMax` from the uniform,
10
+ * not a `LEVELS` override (PD-16, DEP-P4-G). Body only; normative text.
11
+ */
12
+ export declare const gridFarFieldWgsl = "\nfn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }\nfn store_force(i: u32, f: vec3f) {\n force[3u * i] = f.x;\n force[3u * i + 1u] = f.y;\n force[3u * i + 2u] = f.z;\n}\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\nfn grid_side(level: u32) -> u32 { return P.gridMax >> level; }\nfn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cell at index cells)\n var base = 0u;\n for (var l = 0u; l < level; l = l + 1u) {\n let s = grid_side(l);\n base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, 1u, l == 0u);\n }\n return base;\n}\nfn cell_at(level: u32, cx: i32, cy: i32, cz: i32) -> u32 {\n let s = grid_side(level);\n return level_base(level) + u32(cx) + s * (u32(cy) + select(0u, s * u32(cz), P.dim == 3u));\n}\nfn cell_force(pi: vec4f, q: vec4f) -> vec3f { // one far-field term, softened by state.eps (7.7)\n if (q.w <= 0.0) { return vec3f(0.0); } // an empty cell\n let d = pi.xyz - q.xyz / q.w; // to the mass-weighted centroid\n let d2 = dot(d, d) + S.eps * S.eps;\n if (LAW == 1u) { return d * (P.frK * P.frK * q.w / d2); } // LAW 1 (FR, 7.20): k^2 / d per node, q.w nodes at the centroid\n if (LAW == 2u) { return d * (-P.coulomb * pi.w * q.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb): -g m_i M_cell / d^2\n return d * (P.scalingRatio * pi.w * q.w / d2); // LAW 0 (FA2): |F| = k m_i M_cell / d\n}\n\n@compute @workgroup_size(WG)\nfn grid_far_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let t = linear_id(wid, lid.x);\n if (t >= P.n) { return; } // no barrier follows\n let i = sortedIdx[t]; // sorted order (D24)\n let pi = pos[i];\n let gf = f32(P.gridMax);\n let q = (pi.xyz - S.gridMin.xyz) * S.invCellSize; // PD-10\n var c0 = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));\n if (P.dim == 2u) { c0.z = 0; } // 2D: one z plane; the loops below visit cz = 0 only, so the 3x3 test must see cz - 0\n let g = i32(P.gridMax);\n var inside = c0.x >= 0 && c0.x < g && c0.y >= 0 && c0.y < g;\n if (P.dim == 3u) { inside = inside && c0.z >= 0 && c0.z < g; }\n let top = P.levels - 1u;\n let ts = i32(grid_side(top)); // the coarsest side (4)\n let zTop = select(0, ts - 1, P.dim == 3u); // z ranges: one plane in 2D\n var f = vec3f(0.0);\n if (inside) {\n let ct = c0 / i32(1u << top); // the node's coarsest cell\n for (var cz = 0; cz <= zTop; cz = cz + 1) {\n for (var cy = 0; cy < ts; cy = cy + 1) {\n for (var cx = 0; cx < ts; cx = cx + 1) {\n if (abs(cx - ct.x) <= 1 && abs(cy - ct.y) <= 1 && abs(cz - ct.z) <= 1) { continue; } // the 3x3(x3) is finer levels' work\n f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);\n }\n }\n }\n for (var l = top; l > 0u; l = l - 1u) { // level l - 1: the parent's 3x3 at level l, refined, minus this level's own 3x3\n let level = l - 1u;\n let cl = c0 / i32(1u << level);\n let cp = cl / 2;\n let side = i32(grid_side(level));\n let zLo = select(0, max(0, 2 * (cp.z - 1)), P.dim == 3u);\n let zHi = select(0, min(side - 1, 2 * (cp.z + 1) + 1), P.dim == 3u);\n for (var cz = zLo; cz <= zHi; cz = cz + 1) {\n for (var cy = max(0, 2 * (cp.y - 1)); cy <= min(side - 1, 2 * (cp.y + 1) + 1); cy = cy + 1) {\n for (var cx = max(0, 2 * (cp.x - 1)); cx <= min(side - 1, 2 * (cp.x + 1) + 1); cx = cx + 1) {\n if (abs(cx - cl.x) <= 1 && abs(cy - cl.y) <= 1 && abs(cz - cl.z) <= 1) { continue; }\n f = f + cell_force(pi, pyramid[cell_at(level, cx, cy, cz)]);\n }\n }\n }\n }\n f = f + cell_force(pi, pyramid[grid_cells()]); // the outside pseudo-cell as one far-field term\n } else {\n for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (it would include itself)\n for (var cy = 0; cy < ts; cy = cy + 1) {\n for (var cx = 0; cx < ts; cx = cx + 1) {\n f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);\n }\n }\n }\n }\n store_force(i, load_force(i) + f);\n}\n";
13
+ //# sourceMappingURL=grid-far-field.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-far-field.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-far-field.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AACH,eAAO,MAAM,gBAAgB,mzJAqF5B,CAAC"}