@graphty/webgpu-graph-algorithms 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (264) hide show
  1. package/README.md +104 -52
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-CRbw2Wyo.js → context-BXqgCifx.js} +225 -33
  4. package/dist/chunks/context-BXqgCifx.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/accelerator.d.ts +12 -10
  7. package/dist/src/accelerator.d.ts.map +1 -1
  8. package/dist/src/accelerator.js +32 -10
  9. package/dist/src/accelerator.js.map +1 -1
  10. package/dist/src/algorithms/components.d.ts.map +1 -1
  11. package/dist/src/algorithms/components.js +12 -13
  12. package/dist/src/algorithms/components.js.map +1 -1
  13. package/dist/src/algorithms/degree.d.ts +6 -8
  14. package/dist/src/algorithms/degree.d.ts.map +1 -1
  15. package/dist/src/algorithms/degree.js +58 -35
  16. package/dist/src/algorithms/degree.js.map +1 -1
  17. package/dist/src/algorithms/pagerank.d.ts.map +1 -1
  18. package/dist/src/algorithms/pagerank.js +16 -14
  19. package/dist/src/algorithms/pagerank.js.map +1 -1
  20. package/dist/src/algorithms/power-iteration.d.ts +2 -2
  21. package/dist/src/algorithms/power-iteration.d.ts.map +1 -1
  22. package/dist/src/algorithms/power-iteration.js +17 -14
  23. package/dist/src/algorithms/power-iteration.js.map +1 -1
  24. package/dist/src/constants.d.ts +85 -8
  25. package/dist/src/constants.d.ts.map +1 -1
  26. package/dist/src/constants.js +85 -8
  27. package/dist/src/constants.js.map +1 -1
  28. package/dist/src/errors.d.ts +3 -2
  29. package/dist/src/errors.d.ts.map +1 -1
  30. package/dist/src/errors.js +2 -1
  31. package/dist/src/errors.js.map +1 -1
  32. package/dist/src/index.d.ts +10 -5
  33. package/dist/src/index.d.ts.map +1 -1
  34. package/dist/src/index.js +14 -5
  35. package/dist/src/index.js.map +1 -1
  36. package/dist/src/kernel/dispatch.d.ts +8 -3
  37. package/dist/src/kernel/dispatch.d.ts.map +1 -1
  38. package/dist/src/kernel/dispatch.js +18 -7
  39. package/dist/src/kernel/dispatch.js.map +1 -1
  40. package/dist/src/kernel/kernel.d.ts +30 -1
  41. package/dist/src/kernel/kernel.d.ts.map +1 -1
  42. package/dist/src/kernel/kernel.js +49 -5
  43. package/dist/src/kernel/kernel.js.map +1 -1
  44. package/dist/src/kernel/prelude.d.ts.map +1 -1
  45. package/dist/src/kernel/prelude.js +9 -1
  46. package/dist/src/kernel/prelude.js.map +1 -1
  47. package/dist/src/kernel/profiler.d.ts +15 -3
  48. package/dist/src/kernel/profiler.d.ts.map +1 -1
  49. package/dist/src/kernel/profiler.js +27 -4
  50. package/dist/src/kernel/profiler.js.map +1 -1
  51. package/dist/src/kernels.d.ts +18 -8
  52. package/dist/src/kernels.d.ts.map +1 -1
  53. package/dist/src/kernels.js +345 -22
  54. package/dist/src/kernels.js.map +1 -1
  55. package/dist/src/layouts/calibrate.d.ts +51 -0
  56. package/dist/src/layouts/calibrate.d.ts.map +1 -0
  57. package/dist/src/layouts/calibrate.js +172 -0
  58. package/dist/src/layouts/calibrate.js.map +1 -0
  59. package/dist/src/layouts/force-simulation.d.ts +42 -5
  60. package/dist/src/layouts/force-simulation.d.ts.map +1 -1
  61. package/dist/src/layouts/force-simulation.js +84 -22
  62. package/dist/src/layouts/force-simulation.js.map +1 -1
  63. package/dist/src/layouts/forceatlas2.d.ts +107 -38
  64. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  65. package/dist/src/layouts/forceatlas2.js +297 -290
  66. package/dist/src/layouts/forceatlas2.js.map +1 -1
  67. package/dist/src/layouts/fruchterman-reingold.d.ts +241 -0
  68. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -0
  69. package/dist/src/layouts/fruchterman-reingold.js +739 -0
  70. package/dist/src/layouts/fruchterman-reingold.js.map +1 -0
  71. package/dist/src/layouts/model-common.d.ts +140 -0
  72. package/dist/src/layouts/model-common.d.ts.map +1 -0
  73. package/dist/src/layouts/model-common.js +269 -0
  74. package/dist/src/layouts/model-common.js.map +1 -0
  75. package/dist/src/layouts/repulsion-grid.d.ts +152 -0
  76. package/dist/src/layouts/repulsion-grid.d.ts.map +1 -0
  77. package/dist/src/layouts/repulsion-grid.js +318 -0
  78. package/dist/src/layouts/repulsion-grid.js.map +1 -0
  79. package/dist/src/layouts/spring-electrical.d.ts +224 -0
  80. package/dist/src/layouts/spring-electrical.d.ts.map +1 -0
  81. package/dist/src/layouts/spring-electrical.js +665 -0
  82. package/dist/src/layouts/spring-electrical.js.map +1 -0
  83. package/dist/src/memory/residency.d.ts +6 -2
  84. package/dist/src/memory/residency.d.ts.map +1 -1
  85. package/dist/src/memory/residency.js +84 -14
  86. package/dist/src/memory/residency.js.map +1 -1
  87. package/dist/src/primitives/core-shape.d.ts +38 -2
  88. package/dist/src/primitives/core-shape.d.ts.map +1 -1
  89. package/dist/src/primitives/core-shape.js +71 -3
  90. package/dist/src/primitives/core-shape.js.map +1 -1
  91. package/dist/src/primitives/grid-pyramid.d.ts +71 -0
  92. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -0
  93. package/dist/src/primitives/grid-pyramid.js +143 -0
  94. package/dist/src/primitives/grid-pyramid.js.map +1 -0
  95. package/dist/src/primitives/grid.d.ts +118 -0
  96. package/dist/src/primitives/grid.d.ts.map +1 -0
  97. package/dist/src/primitives/grid.js +225 -0
  98. package/dist/src/primitives/grid.js.map +1 -0
  99. package/dist/src/primitives/histogram.d.ts +67 -0
  100. package/dist/src/primitives/histogram.d.ts.map +1 -0
  101. package/dist/src/primitives/histogram.js +190 -0
  102. package/dist/src/primitives/histogram.js.map +1 -0
  103. package/dist/src/primitives/radix-sort.d.ts +75 -0
  104. package/dist/src/primitives/radix-sort.d.ts.map +1 -0
  105. package/dist/src/primitives/radix-sort.js +168 -0
  106. package/dist/src/primitives/radix-sort.js.map +1 -0
  107. package/dist/src/primitives/scan.d.ts +44 -0
  108. package/dist/src/primitives/scan.d.ts.map +1 -0
  109. package/dist/src/primitives/scan.js +151 -0
  110. package/dist/src/primitives/scan.js.map +1 -0
  111. package/dist/src/primitives/segmented-reduce.d.ts +25 -17
  112. package/dist/src/primitives/segmented-reduce.d.ts.map +1 -1
  113. package/dist/src/primitives/segmented-reduce.js +166 -47
  114. package/dist/src/primitives/segmented-reduce.js.map +1 -1
  115. package/dist/src/primitives/spmv.d.ts +18 -14
  116. package/dist/src/primitives/spmv.d.ts.map +1 -1
  117. package/dist/src/primitives/spmv.js +94 -58
  118. package/dist/src/primitives/spmv.js.map +1 -1
  119. package/dist/src/primitives/verify.d.ts +49 -0
  120. package/dist/src/primitives/verify.d.ts.map +1 -0
  121. package/dist/src/primitives/verify.js +229 -0
  122. package/dist/src/primitives/verify.js.map +1 -0
  123. package/dist/src/types/accelerator.d.ts +7 -3
  124. package/dist/src/types/accelerator.d.ts.map +1 -1
  125. package/dist/src/types/context.d.ts +53 -0
  126. package/dist/src/types/context.d.ts.map +1 -1
  127. package/dist/src/types/layout.d.ts +52 -0
  128. package/dist/src/types/layout.d.ts.map +1 -1
  129. package/dist/src/types/options.d.ts +43 -1
  130. package/dist/src/types/options.d.ts.map +1 -1
  131. package/dist/src/wgsl/counting-scatter.wgsl.d.ts +8 -0
  132. package/dist/src/wgsl/counting-scatter.wgsl.d.ts.map +1 -0
  133. package/dist/src/wgsl/counting-scatter.wgsl.js +17 -0
  134. package/dist/src/wgsl/counting-scatter.wgsl.js.map +1 -0
  135. package/dist/src/wgsl/fa2-attraction.wgsl.d.ts +23 -8
  136. package/dist/src/wgsl/fa2-attraction.wgsl.d.ts.map +1 -1
  137. package/dist/src/wgsl/fa2-attraction.wgsl.js +100 -17
  138. package/dist/src/wgsl/fa2-attraction.wgsl.js.map +1 -1
  139. package/dist/src/wgsl/fa2-integrate.wgsl.d.ts +7 -2
  140. package/dist/src/wgsl/fa2-integrate.wgsl.d.ts.map +1 -1
  141. package/dist/src/wgsl/fa2-integrate.wgsl.js +28 -2
  142. package/dist/src/wgsl/fa2-integrate.wgsl.js.map +1 -1
  143. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -2
  144. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
  145. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +14 -5
  146. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
  147. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +12 -1
  148. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
  149. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +54 -0
  150. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
  151. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +8 -0
  152. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -0
  153. package/dist/src/wgsl/grid-cell-key.wgsl.js +30 -0
  154. package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -0
  155. package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +8 -0
  156. package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts.map +1 -0
  157. package/dist/src/wgsl/grid-centroid-hub.wgsl.js +29 -0
  158. package/dist/src/wgsl/grid-centroid-hub.wgsl.js.map +1 -0
  159. package/dist/src/wgsl/grid-centroid.wgsl.d.ts +8 -0
  160. package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -0
  161. package/dist/src/wgsl/grid-centroid.wgsl.js +29 -0
  162. package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -0
  163. package/dist/src/wgsl/grid-downsample.wgsl.d.ts +7 -0
  164. package/dist/src/wgsl/grid-downsample.wgsl.d.ts.map +1 -0
  165. package/dist/src/wgsl/grid-downsample.wgsl.js +28 -0
  166. package/dist/src/wgsl/grid-downsample.wgsl.js.map +1 -0
  167. package/dist/src/wgsl/grid-far-field.wgsl.d.ts +13 -0
  168. package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -0
  169. package/dist/src/wgsl/grid-far-field.wgsl.js +98 -0
  170. package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -0
  171. package/dist/src/wgsl/grid-near-field.wgsl.d.ts +19 -0
  172. package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -0
  173. package/dist/src/wgsl/grid-near-field.wgsl.js +129 -0
  174. package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -0
  175. package/dist/src/wgsl/histogram.wgsl.d.ts +7 -0
  176. package/dist/src/wgsl/histogram.wgsl.d.ts.map +1 -0
  177. package/dist/src/wgsl/histogram.wgsl.js +15 -0
  178. package/dist/src/wgsl/histogram.wgsl.js.map +1 -0
  179. package/dist/src/wgsl/indirect-finalize.wgsl.d.ts +8 -0
  180. package/dist/src/wgsl/indirect-finalize.wgsl.d.ts.map +1 -0
  181. package/dist/src/wgsl/indirect-finalize.wgsl.js +26 -0
  182. package/dist/src/wgsl/indirect-finalize.wgsl.js.map +1 -0
  183. package/dist/src/wgsl/radix-hist.wgsl.d.ts +9 -0
  184. package/dist/src/wgsl/radix-hist.wgsl.d.ts.map +1 -0
  185. package/dist/src/wgsl/radix-hist.wgsl.js +31 -0
  186. package/dist/src/wgsl/radix-hist.wgsl.js.map +1 -0
  187. package/dist/src/wgsl/radix-scatter.wgsl.d.ts +9 -0
  188. package/dist/src/wgsl/radix-scatter.wgsl.d.ts.map +1 -0
  189. package/dist/src/wgsl/radix-scatter.wgsl.js +40 -0
  190. package/dist/src/wgsl/radix-scatter.wgsl.js.map +1 -0
  191. package/dist/src/wgsl/scan-add.wgsl.d.ts +6 -0
  192. package/dist/src/wgsl/scan-add.wgsl.d.ts.map +1 -0
  193. package/dist/src/wgsl/scan-add.wgsl.js +14 -0
  194. package/dist/src/wgsl/scan-add.wgsl.js.map +1 -0
  195. package/dist/src/wgsl/scan-block.wgsl.d.ts +8 -0
  196. package/dist/src/wgsl/scan-block.wgsl.d.ts.map +1 -0
  197. package/dist/src/wgsl/scan-block.wgsl.js +30 -0
  198. package/dist/src/wgsl/scan-block.wgsl.js.map +1 -0
  199. package/dist/src/wgsl/segmented-reduce.wgsl.d.ts +22 -8
  200. package/dist/src/wgsl/segmented-reduce.wgsl.d.ts.map +1 -1
  201. package/dist/src/wgsl/segmented-reduce.wgsl.js +84 -15
  202. package/dist/src/wgsl/segmented-reduce.wgsl.js.map +1 -1
  203. package/dist/src/wgsl/spmv-pull.wgsl.d.ts +22 -11
  204. package/dist/src/wgsl/spmv-pull.wgsl.d.ts.map +1 -1
  205. package/dist/src/wgsl/spmv-pull.wgsl.js +110 -36
  206. package/dist/src/wgsl/spmv-pull.wgsl.js.map +1 -1
  207. package/dist/tsconfig.build.tsbuildinfo +1 -1
  208. package/dist/webgpu-graph-algorithms.js +5016 -1130
  209. package/dist/webgpu-graph-algorithms.js.map +1 -1
  210. package/package.json +10 -7
  211. package/src/accelerator.ts +46 -12
  212. package/src/algorithms/components.ts +12 -16
  213. package/src/algorithms/degree.ts +58 -43
  214. package/src/algorithms/pagerank.ts +20 -18
  215. package/src/algorithms/power-iteration.ts +19 -18
  216. package/src/constants.ts +108 -8
  217. package/src/errors.ts +3 -1
  218. package/src/index.ts +25 -5
  219. package/src/kernel/dispatch.ts +18 -7
  220. package/src/kernel/kernel.ts +59 -5
  221. package/src/kernel/prelude.ts +15 -0
  222. package/src/kernel/profiler.ts +28 -4
  223. package/src/kernels.ts +378 -24
  224. package/src/layouts/calibrate.ts +187 -0
  225. package/src/layouts/force-simulation.ts +111 -26
  226. package/src/layouts/forceatlas2.ts +346 -324
  227. package/src/layouts/fruchterman-reingold.ts +918 -0
  228. package/src/layouts/model-common.ts +323 -0
  229. package/src/layouts/repulsion-grid.ts +451 -0
  230. package/src/layouts/spring-electrical.ts +845 -0
  231. package/src/memory/residency.ts +126 -20
  232. package/src/primitives/core-shape.ts +91 -4
  233. package/src/primitives/grid-pyramid.ts +221 -0
  234. package/src/primitives/grid.ts +349 -0
  235. package/src/primitives/histogram.ts +273 -0
  236. package/src/primitives/radix-sort.ts +246 -0
  237. package/src/primitives/scan.ts +197 -0
  238. package/src/primitives/segmented-reduce.ts +214 -56
  239. package/src/primitives/spmv.ts +125 -65
  240. package/src/primitives/verify.ts +249 -0
  241. package/src/types/accelerator.ts +15 -3
  242. package/src/types/context.ts +56 -0
  243. package/src/types/layout.ts +58 -0
  244. package/src/types/options.ts +45 -1
  245. package/src/wgsl/counting-scatter.wgsl.ts +16 -0
  246. package/src/wgsl/fa2-attraction.wgsl.ts +100 -17
  247. package/src/wgsl/fa2-integrate.wgsl.ts +28 -2
  248. package/src/wgsl/fa2-repulsion-exact.wgsl.ts +14 -5
  249. package/src/wgsl/fa2-stats-finalize.wgsl.ts +54 -0
  250. package/src/wgsl/grid-cell-key.wgsl.ts +29 -0
  251. package/src/wgsl/grid-centroid-hub.wgsl.ts +28 -0
  252. package/src/wgsl/grid-centroid.wgsl.ts +28 -0
  253. package/src/wgsl/grid-downsample.wgsl.ts +27 -0
  254. package/src/wgsl/grid-far-field.wgsl.ts +97 -0
  255. package/src/wgsl/grid-near-field.wgsl.ts +128 -0
  256. package/src/wgsl/histogram.wgsl.ts +14 -0
  257. package/src/wgsl/indirect-finalize.wgsl.ts +25 -0
  258. package/src/wgsl/radix-hist.wgsl.ts +30 -0
  259. package/src/wgsl/radix-scatter.wgsl.ts +39 -0
  260. package/src/wgsl/scan-add.wgsl.ts +13 -0
  261. package/src/wgsl/scan-block.wgsl.ts +29 -0
  262. package/src/wgsl/segmented-reduce.wgsl.ts +84 -15
  263. package/src/wgsl/spmv-pull.wgsl.ts +110 -36
  264. package/dist/chunks/context-CRbw2Wyo.js.map +0 -1
@@ -1 +1 @@
1
- {"version":3,"file":"fa2-repulsion-exact.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-repulsion-exact.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AACH,MAAM,CAAC,MAAM,qBAAqB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAmE/C,CAAC"}
1
+ {"version":3,"file":"fa2-repulsion-exact.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-repulsion-exact.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;GAWG;AACH,MAAM,CAAC,MAAM,qBAAqB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA0E/C,CAAC"}
@@ -4,11 +4,22 @@
4
4
  * exact spec 3.3 value from `partials.max.w`), bounding box, mean displacement over free nodes (0 when every node
5
5
  * is fixed), the settle counter -- increments the iteration counter and writes the K1 half of the trace record.
6
6
  * On the first iteration after load() (`FA2_FLAG_FIRST`) it folds nothing and keeps the host-written state.
7
+ * `STATS_MODE` (P5, spec 7.20) adds the model statistic: 0 = the FA2 text, 1 = the Fruchterman-Reingold temperature
8
+ * of this iteration into `S.temperature` and the trace's `modelScalar` -- the uniform's under the linear schedule,
9
+ * or, with `FA2_FLAG_ADAPTIVE` set, the adaptive one: the previous iteration's force energy (K5 folds sum |F|^2 over
10
+ * free nodes into `partials.swingTraction.x`) against `S.frEnergy` grows the temperature by 1 / FR_COOLING_STEP after
11
+ * FR_COOLING_PATIENCE consecutive falls and shrinks it by FR_COOLING_STEP on a rise (Yifan Hu 2005, section 3.2);
12
+ * 2 = the spring-electrical kinetic energy K5 folded into `partials.swingTraction.x` (PD-4) into `S.kineticEnergy`
13
+ * and the trace. On the grid tier (`P.gridMax > 0`, P4-T10, PD-14) it also derives the grid frame of the next build
14
+ * from the fold (`extent = max(min(bboxExtent * GRID_BBOX_MARGIN, extentFactor * rmsRadius), GRID_EXTENT_FLOOR)`,
15
+ * `cellSize = extent / G`, `gridMin = centroid - extent / 2` with `cellSize` in `.w`, `invCellSize`, `eps = 0.25
16
+ * cellSize`), copies the previous iteration's pseudo-cell count and occupancy max into the state, and resets the
17
+ * hub counters; the exact tier writes `gridMax: 0` and binds two dummies, so the block is dead there.
7
18
  *
8
19
  * This file holds the kernel BODY only (spec 3.5, D9): no bind-group lines and no `override` lines -- the composer
9
20
  * emits them from the registry entry in src/kernels.ts (contract 3.10.1). The text is normative (contract 4.5) and
10
21
  * is the target of the K1 sabotage mutations (test/helpers/sabotage.ts, P3-T5); amend the contract before editing.
11
22
  */
12
23
  /** The K1 body: entry point `stats_finalize`; calls the reduction helpers (`needs: ["subgroups"]`, contract 4.3). */
13
- export declare const fa2StatsFinalizeWgsl = "// K1: folds the previous integrate's partials into the state block (spec 7.4); one workgroup\n@compute @workgroup_size(WG)\nfn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n let groups = (P.n + WG - 1u) / WG;\n let fold = (P.flags & FA2_FLAG_FIRST) == 0u; // the first iteration after load() keeps the host-written state\n var sum = vec4f(0.0);\n var lo = vec4f(F32_MAX);\n var hi = vec4f(-F32_MAX);\n var disp = 0.0;\n var free = 0u;\n if (fold) {\n for (var g = lid.x; g < groups; g = g + WG) { // sequential per lane in index order: deterministic\n let q = partials[g];\n sum = sum + q.sum;\n lo = min(lo, q.min);\n hi = max(hi, q.max);\n disp = disp + q.dispFree.x;\n free = free + u32(q.dispFree.y);\n }\n }\n let tSum = wg_reduce_vec4(sum, lid.x, 0u);\n let tLo = wg_reduce_vec4(lo, lid.x, 1u);\n let tHi = wg_reduce_vec4(hi, lid.x, 2u);\n let tDisp = wg_reduce_f32(disp, lid.x, 0u);\n let tFree = wg_reduce_u32(free, lid.x, 0u);\n if (lid.x == 0u) {\n if (fold) {\n let n = f32(P.n);\n let c = tSum.xyz / n;\n S.centroid = vec4f(c, 0.0);\n S.rmsRadius = sqrt(max(tSum.w, 0.0) / n); // RMS radius about the previous centroid (7.17)\n S.min = vec4f(tLo.xyz, 0.0);\n S.max = vec4f(tHi.xyz, 0.0);\n S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)\n let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)\n S.meanDisplacement = meanDisp;\n S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);\n }\n S.iteration = S.iteration + 1u;\n T[P.iterationIndex].meanDisplacement = S.meanDisplacement;\n T[P.iterationIndex].settledCount = S.settledCount;\n T[P.iterationIndex].iteration = S.iteration;\n }\n}";
24
+ export declare const fa2StatsFinalizeWgsl = "// K1: folds the previous integrate's partials into the state block (spec 7.4); one workgroup\n@compute @workgroup_size(WG)\nfn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n let groups = (P.n + WG - 1u) / WG;\n let fold = (P.flags & FA2_FLAG_FIRST) == 0u; // the first iteration after load() keeps the host-written state\n var sum = vec4f(0.0);\n var lo = vec4f(F32_MAX);\n var hi = vec4f(-F32_MAX);\n var disp = 0.0;\n var free = 0u;\n var ke = 0.0;\n if (fold) {\n for (var g = lid.x; g < groups; g = g + WG) { // sequential per lane in index order: deterministic\n let q = partials[g];\n sum = sum + q.sum;\n lo = min(lo, q.min);\n hi = max(hi, q.max);\n disp = disp + q.dispFree.x;\n free = free + u32(q.dispFree.y);\n ke = ke + q.swingTraction.x;\n }\n }\n let tSum = wg_reduce_vec4(sum, lid.x, 0u);\n let tLo = wg_reduce_vec4(lo, lid.x, 1u);\n let tHi = wg_reduce_vec4(hi, lid.x, 2u);\n let tDisp = wg_reduce_f32(disp, lid.x, 0u);\n let tFree = wg_reduce_u32(free, lid.x, 0u);\n let tKe = wg_reduce_f32(ke, lid.x, 0u);\n if (lid.x == 0u) {\n if (fold) {\n let n = f32(P.n);\n let c = tSum.xyz / n;\n S.centroid = vec4f(c, 0.0);\n S.rmsRadius = sqrt(max(tSum.w, 0.0) / n); // RMS radius about the previous centroid (7.17)\n S.min = vec4f(tLo.xyz, 0.0);\n S.max = vec4f(tHi.xyz, 0.0);\n S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)\n let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)\n S.meanDisplacement = meanDisp;\n S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);\n }\n S.iteration = S.iteration + 1u;\n T[P.iterationIndex].meanDisplacement = S.meanDisplacement;\n T[P.iterationIndex].settledCount = S.settledCount;\n T[P.iterationIndex].iteration = S.iteration;\n if (P.gridMax > 0u) { // the grid tier (7.7): the robust extent, the cell size, eps, last iteration's counts, the hub counter reset (PD-14)\n let cells = P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u);\n if (fold) {\n let box = (S.max.xyz - S.min.xyz) * GRID_BBOX_MARGIN;\n var bboxExtent = max(box.x, box.y);\n if (P.dim == 3u) { bboxExtent = max(bboxExtent, box.z); }\n let extent = max(min(bboxExtent, P.extentFactor * S.rmsRadius), GRID_EXTENT_FLOOR); // min(bbox, extentFactor x rms), floored (7.7)\n let cellSize = extent / f32(P.gridMax);\n S.gridMin = vec4f(S.centroid.xyz - vec3f(0.5 * extent), cellSize); // gridMin.w carries cellSize\n S.invCellSize = 1.0 / cellSize;\n S.eps = 0.25 * cellSize;\n }\n S.outsideGrid = cellHist[cells]; // the previous iteration's pseudo-cell count (0 after load)\n S.maxCellOccupancy = atomicLoad(&hubCounters[1]);\n atomicStore(&hubCounters[0], 0u);\n atomicStore(&hubCounters[1], 0u);\n }\n if (STATS_MODE == 1u) { // FR: this iteration's temperature (7.20) into the state and the trace\n if ((P.flags & FA2_FLAG_ADAPTIVE) != 0u) { // adaptive cooling (Yifan Hu 2005 3.2): tKe is the previous iteration's sum |F|^2 over free nodes\n if (fold) {\n var t = S.temperature;\n if (tKe < S.frEnergy) {\n S.frProgress = S.frProgress + 1u;\n if (S.frProgress >= FR_COOLING_PATIENCE) { S.frProgress = 0u; t = t / FR_COOLING_STEP; }\n } else {\n S.frProgress = 0u;\n t = t * FR_COOLING_STEP;\n }\n S.frEnergy = tKe;\n S.temperature = t;\n }\n } else {\n S.temperature = P.temperature;\n }\n T[P.iterationIndex].modelScalar = S.temperature;\n }\n if (STATS_MODE == 2u) { // spring-electrical: the kinetic energy K5 folded into partials B (PD-4); 0 on the first iteration after load()\n S.kineticEnergy = tKe;\n T[P.iterationIndex].modelScalar = tKe;\n }\n }\n}";
14
25
  //# sourceMappingURL=fa2-stats-finalize.wgsl.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"fa2-stats-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AAEH,qHAAqH;AACrH,eAAO,MAAM,oBAAoB,gkEA2C/B,CAAC"}
1
+ {"version":3,"file":"fa2-stats-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH,qHAAqH;AACrH,eAAO,MAAM,oBAAoB,snJAsF/B,CAAC"}
@@ -4,6 +4,17 @@
4
4
  * exact spec 3.3 value from `partials.max.w`), bounding box, mean displacement over free nodes (0 when every node
5
5
  * is fixed), the settle counter -- increments the iteration counter and writes the K1 half of the trace record.
6
6
  * On the first iteration after load() (`FA2_FLAG_FIRST`) it folds nothing and keeps the host-written state.
7
+ * `STATS_MODE` (P5, spec 7.20) adds the model statistic: 0 = the FA2 text, 1 = the Fruchterman-Reingold temperature
8
+ * of this iteration into `S.temperature` and the trace's `modelScalar` -- the uniform's under the linear schedule,
9
+ * or, with `FA2_FLAG_ADAPTIVE` set, the adaptive one: the previous iteration's force energy (K5 folds sum |F|^2 over
10
+ * free nodes into `partials.swingTraction.x`) against `S.frEnergy` grows the temperature by 1 / FR_COOLING_STEP after
11
+ * FR_COOLING_PATIENCE consecutive falls and shrinks it by FR_COOLING_STEP on a rise (Yifan Hu 2005, section 3.2);
12
+ * 2 = the spring-electrical kinetic energy K5 folded into `partials.swingTraction.x` (PD-4) into `S.kineticEnergy`
13
+ * and the trace. On the grid tier (`P.gridMax > 0`, P4-T10, PD-14) it also derives the grid frame of the next build
14
+ * from the fold (`extent = max(min(bboxExtent * GRID_BBOX_MARGIN, extentFactor * rmsRadius), GRID_EXTENT_FLOOR)`,
15
+ * `cellSize = extent / G`, `gridMin = centroid - extent / 2` with `cellSize` in `.w`, `invCellSize`, `eps = 0.25
16
+ * cellSize`), copies the previous iteration's pseudo-cell count and occupancy max into the state, and resets the
17
+ * hub counters; the exact tier writes `gridMax: 0` and binds two dummies, so the block is dead there.
7
18
  *
8
19
  * This file holds the kernel BODY only (spec 3.5, D9): no bind-group lines and no `override` lines -- the composer
9
20
  * emits them from the registry entry in src/kernels.ts (contract 3.10.1). The text is normative (contract 4.5) and
@@ -20,6 +31,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
20
31
  var hi = vec4f(-F32_MAX);
21
32
  var disp = 0.0;
22
33
  var free = 0u;
34
+ var ke = 0.0;
23
35
  if (fold) {
24
36
  for (var g = lid.x; g < groups; g = g + WG) { // sequential per lane in index order: deterministic
25
37
  let q = partials[g];
@@ -28,6 +40,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
28
40
  hi = max(hi, q.max);
29
41
  disp = disp + q.dispFree.x;
30
42
  free = free + u32(q.dispFree.y);
43
+ ke = ke + q.swingTraction.x;
31
44
  }
32
45
  }
33
46
  let tSum = wg_reduce_vec4(sum, lid.x, 0u);
@@ -35,6 +48,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
35
48
  let tHi = wg_reduce_vec4(hi, lid.x, 2u);
36
49
  let tDisp = wg_reduce_f32(disp, lid.x, 0u);
37
50
  let tFree = wg_reduce_u32(free, lid.x, 0u);
51
+ let tKe = wg_reduce_f32(ke, lid.x, 0u);
38
52
  if (lid.x == 0u) {
39
53
  if (fold) {
40
54
  let n = f32(P.n);
@@ -52,6 +66,46 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
52
66
  T[P.iterationIndex].meanDisplacement = S.meanDisplacement;
53
67
  T[P.iterationIndex].settledCount = S.settledCount;
54
68
  T[P.iterationIndex].iteration = S.iteration;
69
+ if (P.gridMax > 0u) { // the grid tier (7.7): the robust extent, the cell size, eps, last iteration's counts, the hub counter reset (PD-14)
70
+ let cells = P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u);
71
+ if (fold) {
72
+ let box = (S.max.xyz - S.min.xyz) * GRID_BBOX_MARGIN;
73
+ var bboxExtent = max(box.x, box.y);
74
+ if (P.dim == 3u) { bboxExtent = max(bboxExtent, box.z); }
75
+ let extent = max(min(bboxExtent, P.extentFactor * S.rmsRadius), GRID_EXTENT_FLOOR); // min(bbox, extentFactor x rms), floored (7.7)
76
+ let cellSize = extent / f32(P.gridMax);
77
+ S.gridMin = vec4f(S.centroid.xyz - vec3f(0.5 * extent), cellSize); // gridMin.w carries cellSize
78
+ S.invCellSize = 1.0 / cellSize;
79
+ S.eps = 0.25 * cellSize;
80
+ }
81
+ S.outsideGrid = cellHist[cells]; // the previous iteration's pseudo-cell count (0 after load)
82
+ S.maxCellOccupancy = atomicLoad(&hubCounters[1]);
83
+ atomicStore(&hubCounters[0], 0u);
84
+ atomicStore(&hubCounters[1], 0u);
85
+ }
86
+ if (STATS_MODE == 1u) { // FR: this iteration's temperature (7.20) into the state and the trace
87
+ if ((P.flags & FA2_FLAG_ADAPTIVE) != 0u) { // adaptive cooling (Yifan Hu 2005 3.2): tKe is the previous iteration's sum |F|^2 over free nodes
88
+ if (fold) {
89
+ var t = S.temperature;
90
+ if (tKe < S.frEnergy) {
91
+ S.frProgress = S.frProgress + 1u;
92
+ if (S.frProgress >= FR_COOLING_PATIENCE) { S.frProgress = 0u; t = t / FR_COOLING_STEP; }
93
+ } else {
94
+ S.frProgress = 0u;
95
+ t = t * FR_COOLING_STEP;
96
+ }
97
+ S.frEnergy = tKe;
98
+ S.temperature = t;
99
+ }
100
+ } else {
101
+ S.temperature = P.temperature;
102
+ }
103
+ T[P.iterationIndex].modelScalar = S.temperature;
104
+ }
105
+ if (STATS_MODE == 2u) { // spring-electrical: the kinetic energy K5 folded into partials B (PD-4); 0 on the first iteration after load()
106
+ S.kineticEnergy = tKe;
107
+ T[P.iterationIndex].modelScalar = tKe;
108
+ }
55
109
  }
56
110
  }`;
57
111
  //# sourceMappingURL=fa2-stats-finalize.wgsl.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"fa2-stats-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AAEH,qHAAqH;AACrH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EA2C7C,CAAC"}
1
+ {"version":3,"file":"fa2-stats-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH,qHAAqH;AACrH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EAsF7C,CAAC"}
@@ -0,0 +1,8 @@
1
+ /**
2
+ * G1, the `grid-cell-key` kernel body (spec 7.7; P4-T8): the finest cell of every node from the state's robust extent,
3
+ * `floor((p - gridMin) * invCellSize)` (a multiply, correctly rounded everywhere: PD-10), linearised when every axis
4
+ * is in [0, G) and the outside pseudo-cell `cells` otherwise; `cellVal[i] = i`. The clamp before the floor keeps a
5
+ * far-away or NaN coordinate out of an out-of-range float-to-int conversion. Body only; normative text.
6
+ */
7
+ export declare const gridCellKeyWgsl = "\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\n\n@compute @workgroup_size(WG)\nfn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n if (i >= P.n) { return; } // no barrier follows\n let cells = grid_cells();\n let gf = f32(P.gridMax);\n let q = (pos[i].xyz - S.gridMin.xyz) * S.invCellSize; // PD-10: never a division\n let c = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));\n let g = i32(P.gridMax);\n var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;\n if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }\n var key = cells; // the outside pseudo-cell (7.7)\n if (inside) {\n key = u32(c.x) + P.gridMax * u32(c.y);\n if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }\n }\n cellKey[i] = key;\n cellVal[i] = i;\n}\n";
8
+ //# sourceMappingURL=grid-cell-key.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-cell-key.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-cell-key.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,eAAO,MAAM,eAAe,uhCAsB3B,CAAC"}
@@ -0,0 +1,30 @@
1
+ /**
2
+ * G1, the `grid-cell-key` kernel body (spec 7.7; P4-T8): the finest cell of every node from the state's robust extent,
3
+ * `floor((p - gridMin) * invCellSize)` (a multiply, correctly rounded everywhere: PD-10), linearised when every axis
4
+ * is in [0, G) and the outside pseudo-cell `cells` otherwise; `cellVal[i] = i`. The clamp before the floor keeps a
5
+ * far-away or NaN coordinate out of an out-of-range float-to-int conversion. Body only; normative text.
6
+ */
7
+ export const gridCellKeyWgsl = /* wgsl */ `
8
+ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
9
+
10
+ @compute @workgroup_size(WG)
11
+ fn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
12
+ let i = linear_id(wid, lid.x);
13
+ if (i >= P.n) { return; } // no barrier follows
14
+ let cells = grid_cells();
15
+ let gf = f32(P.gridMax);
16
+ let q = (pos[i].xyz - S.gridMin.xyz) * S.invCellSize; // PD-10: never a division
17
+ let c = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));
18
+ let g = i32(P.gridMax);
19
+ var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;
20
+ if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }
21
+ var key = cells; // the outside pseudo-cell (7.7)
22
+ if (inside) {
23
+ key = u32(c.x) + P.gridMax * u32(c.y);
24
+ if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }
25
+ }
26
+ cellKey[i] = key;
27
+ cellVal[i] = i;
28
+ }
29
+ `;
30
+ //# sourceMappingURL=grid-cell-key.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-cell-key.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-cell-key.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,CAAC,MAAM,eAAe,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;CAsBzC,CAAC"}
@@ -0,0 +1,8 @@
1
+ /**
2
+ * G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per hub cell of hubList, dispatched
3
+ * indirectly from hubArgs (the T1 finalize over hubCounters[0]); a WG-strided mass-weighted sum reduced by the
4
+ * prelude's tree. The work is guarded by `valid`, never an early return, so the reduction is uniform (PD-13). Body
5
+ * only; normative text.
6
+ */
7
+ export declare const gridCentroidHubWgsl = "\n@compute @workgroup_size(WG)\nfn grid_centroid_hub(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let h = group_id(wid);\n let valid = h < hubCount[0]; // a workgroup past the count sums nothing\n var c = 0u;\n var start = 0u;\n var count = 0u;\n if (valid) {\n c = hubList[h];\n start = cellStart[c];\n count = cellStart[c + 1u] - start;\n }\n var acc = vec4f(0.0);\n for (var k = start + lid.x; k < start + count; k = k + WG) { // strided over the cell's sorted range\n let p = pos[sortedIdx[k]];\n acc = acc + vec4f(p.xyz * p.w, p.w);\n }\n let t = wg_reduce_vec4(acc, lid.x, 0u); // uniform control flow: 256 -> 1\n if (valid && lid.x == 0u) { pyramid[c] = t; }\n}\n";
8
+ //# sourceMappingURL=grid-centroid-hub.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-centroid-hub.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid-hub.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,eAAO,MAAM,mBAAmB,i1BAqB/B,CAAC"}
@@ -0,0 +1,29 @@
1
+ /**
2
+ * G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per hub cell of hubList, dispatched
3
+ * indirectly from hubArgs (the T1 finalize over hubCounters[0]); a WG-strided mass-weighted sum reduced by the
4
+ * prelude's tree. The work is guarded by `valid`, never an early return, so the reduction is uniform (PD-13). Body
5
+ * only; normative text.
6
+ */
7
+ export const gridCentroidHubWgsl = /* wgsl */ `
8
+ @compute @workgroup_size(WG)
9
+ fn grid_centroid_hub(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
10
+ let h = group_id(wid);
11
+ let valid = h < hubCount[0]; // a workgroup past the count sums nothing
12
+ var c = 0u;
13
+ var start = 0u;
14
+ var count = 0u;
15
+ if (valid) {
16
+ c = hubList[h];
17
+ start = cellStart[c];
18
+ count = cellStart[c + 1u] - start;
19
+ }
20
+ var acc = vec4f(0.0);
21
+ for (var k = start + lid.x; k < start + count; k = k + WG) { // strided over the cell's sorted range
22
+ let p = pos[sortedIdx[k]];
23
+ acc = acc + vec4f(p.xyz * p.w, p.w);
24
+ }
25
+ let t = wg_reduce_vec4(acc, lid.x, 0u); // uniform control flow: 256 -> 1
26
+ if (valid && lid.x == 0u) { pyramid[c] = t; }
27
+ }
28
+ `;
29
+ //# sourceMappingURL=grid-centroid-hub.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-centroid-hub.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid-hub.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,CAAC,MAAM,mBAAmB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;CAqB7C,CAAC"}
@@ -0,0 +1,8 @@
1
+ /**
2
+ * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the pseudo-cell included; the
3
+ * mass-weighted position sum of a cell's sorted range in index order (no atomics: deterministic), the largest
4
+ * occupancy into hubCounters[1], and cells above GRID_HUB_CELL entries appended to hubList for G4b (PD-13). Body
5
+ * only; normative text.
6
+ */
7
+ export declare const gridCentroidWgsl = "\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\n\n@compute @workgroup_size(WG)\nfn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let c = linear_id(wid, lid.x);\n if (c > grid_cells()) { return; } // cells [0, cells]: the pseudo-cell is index cells; no barrier follows\n let start = cellStart[c];\n let count = cellStart[c + 1u] - start;\n atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration\n if (count > GRID_HUB_CELL) { // a hub cell: G4b sums it (PD-13)\n hubList[atomicAdd(&hubCounters[0], 1u)] = c;\n return;\n }\n var acc = vec4f(0.0);\n for (var k = start; k < start + count; k = k + 1u) { // sorted order: deterministic\n let p = pos[sortedIdx[k]];\n acc = acc + vec4f(p.xyz * p.w, p.w); // (sum m x, sum m y, sum m z, sum m)\n }\n pyramid[c] = acc;\n}\n";
8
+ //# sourceMappingURL=grid-centroid.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-centroid.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,eAAO,MAAM,gBAAgB,+jCAqB5B,CAAC"}
@@ -0,0 +1,29 @@
1
+ /**
2
+ * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the pseudo-cell included; the
3
+ * mass-weighted position sum of a cell's sorted range in index order (no atomics: deterministic), the largest
4
+ * occupancy into hubCounters[1], and cells above GRID_HUB_CELL entries appended to hubList for G4b (PD-13). Body
5
+ * only; normative text.
6
+ */
7
+ export const gridCentroidWgsl = /* wgsl */ `
8
+ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
9
+
10
+ @compute @workgroup_size(WG)
11
+ fn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
12
+ let c = linear_id(wid, lid.x);
13
+ if (c > grid_cells()) { return; } // cells [0, cells]: the pseudo-cell is index cells; no barrier follows
14
+ let start = cellStart[c];
15
+ let count = cellStart[c + 1u] - start;
16
+ atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration
17
+ if (count > GRID_HUB_CELL) { // a hub cell: G4b sums it (PD-13)
18
+ hubList[atomicAdd(&hubCounters[0], 1u)] = c;
19
+ return;
20
+ }
21
+ var acc = vec4f(0.0);
22
+ for (var k = start; k < start + count; k = k + 1u) { // sorted order: deterministic
23
+ let p = pos[sortedIdx[k]];
24
+ acc = acc + vec4f(p.xyz * p.w, p.w); // (sum m x, sum m y, sum m z, sum m)
25
+ }
26
+ pyramid[c] = acc;
27
+ }
28
+ `;
29
+ //# sourceMappingURL=grid-centroid.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-centroid.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;CAqB1C,CAAC"}
@@ -0,0 +1,7 @@
1
+ /**
2
+ * G5, the `grid-downsample` kernel body (spec 7.7; P4-T9): one dispatch per coarser level; every parent cell is the
3
+ * sum of its 4 (2D) or 8 (3D) children at the level below, read at P.childBase and written at P.parentBase (the
4
+ * pseudo-cell, index cells of level 0, is never a child). No atomics. Body only; normative text.
5
+ */
6
+ export declare const gridDownsampleWgsl = "\n@compute @workgroup_size(WG)\nfn grid_downsample(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let pc = linear_id(wid, lid.x); // the parent cell inside its level\n if (pc >= P.parentCells) { return; } // no barrier follows\n let side = P.parentSide;\n let cs = 2u * side; // the child level's side\n let px = pc % side;\n let py = (pc / side) % side;\n let pz = pc / (side * side);\n var acc = vec4f(0.0);\n for (var dz = 0u; dz < P.depth; dz = dz + 1u) {\n for (var dy = 0u; dy < 2u; dy = dy + 1u) {\n for (var dx = 0u; dx < 2u; dx = dx + 1u) {\n let child = (2u * px + dx) + cs * ((2u * py + dy) + cs * (2u * pz + dz));\n acc = acc + pyramid[P.childBase + child];\n }\n }\n }\n pyramid[P.parentBase + pc] = acc;\n}\n";
7
+ //# sourceMappingURL=grid-downsample.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-downsample.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-downsample.wgsl.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH,eAAO,MAAM,kBAAkB,w8BAqB9B,CAAC"}
@@ -0,0 +1,28 @@
1
+ /**
2
+ * G5, the `grid-downsample` kernel body (spec 7.7; P4-T9): one dispatch per coarser level; every parent cell is the
3
+ * sum of its 4 (2D) or 8 (3D) children at the level below, read at P.childBase and written at P.parentBase (the
4
+ * pseudo-cell, index cells of level 0, is never a child). No atomics. Body only; normative text.
5
+ */
6
+ export const gridDownsampleWgsl = /* wgsl */ `
7
+ @compute @workgroup_size(WG)
8
+ fn grid_downsample(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
9
+ let pc = linear_id(wid, lid.x); // the parent cell inside its level
10
+ if (pc >= P.parentCells) { return; } // no barrier follows
11
+ let side = P.parentSide;
12
+ let cs = 2u * side; // the child level's side
13
+ let px = pc % side;
14
+ let py = (pc / side) % side;
15
+ let pz = pc / (side * side);
16
+ var acc = vec4f(0.0);
17
+ for (var dz = 0u; dz < P.depth; dz = dz + 1u) {
18
+ for (var dy = 0u; dy < 2u; dy = dy + 1u) {
19
+ for (var dx = 0u; dx < 2u; dx = dx + 1u) {
20
+ let child = (2u * px + dx) + cs * ((2u * py + dy) + cs * (2u * pz + dz));
21
+ acc = acc + pyramid[P.childBase + child];
22
+ }
23
+ }
24
+ }
25
+ pyramid[P.parentBase + pc] = acc;
26
+ }
27
+ `;
28
+ //# sourceMappingURL=grid-downsample.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-downsample.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-downsample.wgsl.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;CAqB5C,CAAC"}
@@ -0,0 +1,13 @@
1
+ /**
2
+ * G6, the `grid-far-field` kernel body (spec 7.7; P4-T10; D24): per node `i = sortedIdx[t]`, its finest cell
3
+ * recomputed from `pos[i]` and the state (PD-10); for an inside node the coarsest level minus the 3x3 (3x3x3)
4
+ * around its coarsest cell, then at every finer level the 6x6 (6x6x6) block that is the parent's 3x3 minus this
5
+ * level's own 3x3 -- space tiled exactly once, no theta -- plus the outside pseudo-cell's centroid; for an outside
6
+ * node the coarsest level in full and no pseudo-cell. Every cell term is the per-cell law on the mass-weighted
7
+ * centroid (Gephi Region semantics), softened by `eps^2`: `LAW` 0 (FA2) `d * (k m_i M / d2)`, `LAW` 1 (FR, 7.20)
8
+ * `d * (k^2 M / d2)` (mass 1 per node, so `M` is the cell's count), `LAW` 2 (coulomb) `d * (-g m_i M / d2^1.5)`
9
+ * (P4-T13, PD-22). `force += f` (K2 wrote it). The loop bounds are `P.levels` and `P.gridMax` from the uniform,
10
+ * not a `LEVELS` override (PD-16, DEP-P4-G). Body only; normative text.
11
+ */
12
+ export declare const gridFarFieldWgsl = "\nfn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }\nfn store_force(i: u32, f: vec3f) {\n force[3u * i] = f.x;\n force[3u * i + 1u] = f.y;\n force[3u * i + 2u] = f.z;\n}\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\nfn grid_side(level: u32) -> u32 { return P.gridMax >> level; }\nfn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cell at index cells)\n var base = 0u;\n for (var l = 0u; l < level; l = l + 1u) {\n let s = grid_side(l);\n base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, 1u, l == 0u);\n }\n return base;\n}\nfn cell_at(level: u32, cx: i32, cy: i32, cz: i32) -> u32 {\n let s = grid_side(level);\n return level_base(level) + u32(cx) + s * (u32(cy) + select(0u, s * u32(cz), P.dim == 3u));\n}\nfn cell_force(pi: vec4f, q: vec4f) -> vec3f { // one far-field term, softened by state.eps (7.7)\n if (q.w <= 0.0) { return vec3f(0.0); } // an empty cell\n let d = pi.xyz - q.xyz / q.w; // to the mass-weighted centroid\n let d2 = dot(d, d) + S.eps * S.eps;\n if (LAW == 1u) { return d * (P.frK * P.frK * q.w / d2); } // LAW 1 (FR, 7.20): k^2 / d per node, q.w nodes at the centroid\n if (LAW == 2u) { return d * (-P.coulomb * pi.w * q.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb): -g m_i M_cell / d^2\n return d * (P.scalingRatio * pi.w * q.w / d2); // LAW 0 (FA2): |F| = k m_i M_cell / d\n}\n\n@compute @workgroup_size(WG)\nfn grid_far_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let t = linear_id(wid, lid.x);\n if (t >= P.n) { return; } // no barrier follows\n let i = sortedIdx[t]; // sorted order (D24)\n let pi = pos[i];\n let gf = f32(P.gridMax);\n let q = (pi.xyz - S.gridMin.xyz) * S.invCellSize; // PD-10\n var c0 = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));\n if (P.dim == 2u) { c0.z = 0; } // 2D: one z plane; the loops below visit cz = 0 only, so the 3x3 test must see cz - 0\n let g = i32(P.gridMax);\n var inside = c0.x >= 0 && c0.x < g && c0.y >= 0 && c0.y < g;\n if (P.dim == 3u) { inside = inside && c0.z >= 0 && c0.z < g; }\n let top = P.levels - 1u;\n let ts = i32(grid_side(top)); // the coarsest side (4)\n let zTop = select(0, ts - 1, P.dim == 3u); // z ranges: one plane in 2D\n var f = vec3f(0.0);\n if (inside) {\n let ct = c0 / i32(1u << top); // the node's coarsest cell\n for (var cz = 0; cz <= zTop; cz = cz + 1) {\n for (var cy = 0; cy < ts; cy = cy + 1) {\n for (var cx = 0; cx < ts; cx = cx + 1) {\n if (abs(cx - ct.x) <= 1 && abs(cy - ct.y) <= 1 && abs(cz - ct.z) <= 1) { continue; } // the 3x3(x3) is finer levels' work\n f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);\n }\n }\n }\n for (var l = top; l > 0u; l = l - 1u) { // level l - 1: the parent's 3x3 at level l, refined, minus this level's own 3x3\n let level = l - 1u;\n let cl = c0 / i32(1u << level);\n let cp = cl / 2;\n let side = i32(grid_side(level));\n let zLo = select(0, max(0, 2 * (cp.z - 1)), P.dim == 3u);\n let zHi = select(0, min(side - 1, 2 * (cp.z + 1) + 1), P.dim == 3u);\n for (var cz = zLo; cz <= zHi; cz = cz + 1) {\n for (var cy = max(0, 2 * (cp.y - 1)); cy <= min(side - 1, 2 * (cp.y + 1) + 1); cy = cy + 1) {\n for (var cx = max(0, 2 * (cp.x - 1)); cx <= min(side - 1, 2 * (cp.x + 1) + 1); cx = cx + 1) {\n if (abs(cx - cl.x) <= 1 && abs(cy - cl.y) <= 1 && abs(cz - cl.z) <= 1) { continue; }\n f = f + cell_force(pi, pyramid[cell_at(level, cx, cy, cz)]);\n }\n }\n }\n }\n f = f + cell_force(pi, pyramid[grid_cells()]); // the outside pseudo-cell as one far-field term\n } else {\n for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (it would include itself)\n for (var cy = 0; cy < ts; cy = cy + 1) {\n for (var cx = 0; cx < ts; cx = cx + 1) {\n f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);\n }\n }\n }\n }\n store_force(i, load_force(i) + f);\n}\n";
13
+ //# sourceMappingURL=grid-far-field.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-far-field.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-far-field.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AACH,eAAO,MAAM,gBAAgB,mzJAqF5B,CAAC"}
@@ -0,0 +1,98 @@
1
+ /**
2
+ * G6, the `grid-far-field` kernel body (spec 7.7; P4-T10; D24): per node `i = sortedIdx[t]`, its finest cell
3
+ * recomputed from `pos[i]` and the state (PD-10); for an inside node the coarsest level minus the 3x3 (3x3x3)
4
+ * around its coarsest cell, then at every finer level the 6x6 (6x6x6) block that is the parent's 3x3 minus this
5
+ * level's own 3x3 -- space tiled exactly once, no theta -- plus the outside pseudo-cell's centroid; for an outside
6
+ * node the coarsest level in full and no pseudo-cell. Every cell term is the per-cell law on the mass-weighted
7
+ * centroid (Gephi Region semantics), softened by `eps^2`: `LAW` 0 (FA2) `d * (k m_i M / d2)`, `LAW` 1 (FR, 7.20)
8
+ * `d * (k^2 M / d2)` (mass 1 per node, so `M` is the cell's count), `LAW` 2 (coulomb) `d * (-g m_i M / d2^1.5)`
9
+ * (P4-T13, PD-22). `force += f` (K2 wrote it). The loop bounds are `P.levels` and `P.gridMax` from the uniform,
10
+ * not a `LEVELS` override (PD-16, DEP-P4-G). Body only; normative text.
11
+ */
12
+ export const gridFarFieldWgsl = /* wgsl */ `
13
+ fn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }
14
+ fn store_force(i: u32, f: vec3f) {
15
+ force[3u * i] = f.x;
16
+ force[3u * i + 1u] = f.y;
17
+ force[3u * i + 2u] = f.z;
18
+ }
19
+ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
20
+ fn grid_side(level: u32) -> u32 { return P.gridMax >> level; }
21
+ fn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cell at index cells)
22
+ var base = 0u;
23
+ for (var l = 0u; l < level; l = l + 1u) {
24
+ let s = grid_side(l);
25
+ base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, 1u, l == 0u);
26
+ }
27
+ return base;
28
+ }
29
+ fn cell_at(level: u32, cx: i32, cy: i32, cz: i32) -> u32 {
30
+ let s = grid_side(level);
31
+ return level_base(level) + u32(cx) + s * (u32(cy) + select(0u, s * u32(cz), P.dim == 3u));
32
+ }
33
+ fn cell_force(pi: vec4f, q: vec4f) -> vec3f { // one far-field term, softened by state.eps (7.7)
34
+ if (q.w <= 0.0) { return vec3f(0.0); } // an empty cell
35
+ let d = pi.xyz - q.xyz / q.w; // to the mass-weighted centroid
36
+ let d2 = dot(d, d) + S.eps * S.eps;
37
+ if (LAW == 1u) { return d * (P.frK * P.frK * q.w / d2); } // LAW 1 (FR, 7.20): k^2 / d per node, q.w nodes at the centroid
38
+ if (LAW == 2u) { return d * (-P.coulomb * pi.w * q.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb): -g m_i M_cell / d^2
39
+ return d * (P.scalingRatio * pi.w * q.w / d2); // LAW 0 (FA2): |F| = k m_i M_cell / d
40
+ }
41
+
42
+ @compute @workgroup_size(WG)
43
+ fn grid_far_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
44
+ let t = linear_id(wid, lid.x);
45
+ if (t >= P.n) { return; } // no barrier follows
46
+ let i = sortedIdx[t]; // sorted order (D24)
47
+ let pi = pos[i];
48
+ let gf = f32(P.gridMax);
49
+ let q = (pi.xyz - S.gridMin.xyz) * S.invCellSize; // PD-10
50
+ var c0 = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));
51
+ if (P.dim == 2u) { c0.z = 0; } // 2D: one z plane; the loops below visit cz = 0 only, so the 3x3 test must see cz - 0
52
+ let g = i32(P.gridMax);
53
+ var inside = c0.x >= 0 && c0.x < g && c0.y >= 0 && c0.y < g;
54
+ if (P.dim == 3u) { inside = inside && c0.z >= 0 && c0.z < g; }
55
+ let top = P.levels - 1u;
56
+ let ts = i32(grid_side(top)); // the coarsest side (4)
57
+ let zTop = select(0, ts - 1, P.dim == 3u); // z ranges: one plane in 2D
58
+ var f = vec3f(0.0);
59
+ if (inside) {
60
+ let ct = c0 / i32(1u << top); // the node's coarsest cell
61
+ for (var cz = 0; cz <= zTop; cz = cz + 1) {
62
+ for (var cy = 0; cy < ts; cy = cy + 1) {
63
+ for (var cx = 0; cx < ts; cx = cx + 1) {
64
+ if (abs(cx - ct.x) <= 1 && abs(cy - ct.y) <= 1 && abs(cz - ct.z) <= 1) { continue; } // the 3x3(x3) is finer levels' work
65
+ f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);
66
+ }
67
+ }
68
+ }
69
+ for (var l = top; l > 0u; l = l - 1u) { // level l - 1: the parent's 3x3 at level l, refined, minus this level's own 3x3
70
+ let level = l - 1u;
71
+ let cl = c0 / i32(1u << level);
72
+ let cp = cl / 2;
73
+ let side = i32(grid_side(level));
74
+ let zLo = select(0, max(0, 2 * (cp.z - 1)), P.dim == 3u);
75
+ let zHi = select(0, min(side - 1, 2 * (cp.z + 1) + 1), P.dim == 3u);
76
+ for (var cz = zLo; cz <= zHi; cz = cz + 1) {
77
+ for (var cy = max(0, 2 * (cp.y - 1)); cy <= min(side - 1, 2 * (cp.y + 1) + 1); cy = cy + 1) {
78
+ for (var cx = max(0, 2 * (cp.x - 1)); cx <= min(side - 1, 2 * (cp.x + 1) + 1); cx = cx + 1) {
79
+ if (abs(cx - cl.x) <= 1 && abs(cy - cl.y) <= 1 && abs(cz - cl.z) <= 1) { continue; }
80
+ f = f + cell_force(pi, pyramid[cell_at(level, cx, cy, cz)]);
81
+ }
82
+ }
83
+ }
84
+ }
85
+ f = f + cell_force(pi, pyramid[grid_cells()]); // the outside pseudo-cell as one far-field term
86
+ } else {
87
+ for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (it would include itself)
88
+ for (var cy = 0; cy < ts; cy = cy + 1) {
89
+ for (var cx = 0; cx < ts; cx = cx + 1) {
90
+ f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);
91
+ }
92
+ }
93
+ }
94
+ }
95
+ store_force(i, load_force(i) + f);
96
+ }
97
+ `;
98
+ //# sourceMappingURL=grid-far-field.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-far-field.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-far-field.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAqF1C,CAAC"}
@@ -0,0 +1,19 @@
1
+ /**
2
+ * G7, the `grid-near-field` kernel body (spec 7.7, 7.6, 7.20; P4-T10, P4-T13; D24): per node `i = sortedIdx[t]`, the
3
+ * exact pair law of K3 (`LAW` 0: `|F| = k m_i m_j / d` with the 0.01 floor; `LAW` 1: FR's unfloored `k^2 / d`;
4
+ * `LAW` 2: the unfloored coulomb `-g m_i m_j / d^2`; the antisymmetric coincident kick at the law's magnitude at
5
+ * d = 0.01, PD-22) over the 9 (27) finest
6
+ * cells around its own, or over the outside pseudo-cell alone for an outside node; a cell above `nearMax` entries
7
+ * is sampled by `nearMax` INDEPENDENT draws with replacement, draw `k` reading the slot
8
+ * `lowbias32(((c ^ (iteration * 0x9E3779B9)) ^ seed) ^ (k * 0x85EBCA6B)) % count` (every slot's inclusion
9
+ * probability is `nearMax / count` whatever its position in the sorted order, so a duplicated draw is counted twice
10
+ * and the node itself, when drawn, is skipped and not replaced), and scaled by `others / sampled` where `sampled` is
11
+ * the realised number of draws that were not the node (the Horvitz-Thompson form of PD-15 / DEP-P4-K: given
12
+ * `sampled = s`, those `s` draws are i.i.d. uniform over the `others` slots, so the expectation of the scaled sum is
13
+ * the exact cell sum whenever `s >= 1`; the G4 record's G4-F2 row carries the measurement); then
14
+ * the fused epilogue of K3 (gravity, `force +=`, the swing / traction workgroup reduction in uniform control flow).
15
+ * The helpers `load_force`, `store_force`, `load_old`, `kick_magnitude`, `gravity_force` and the epilogue are K3's
16
+ * text. Body only; normative text.
17
+ */
18
+ export declare const gridNearFieldWgsl = "\nfn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }\nfn store_force(i: u32, f: vec3f) {\n force[3u * i] = f.x;\n force[3u * i + 1u] = f.y;\n force[3u * i + 2u] = f.z;\n}\nfn load_old(i: u32) -> vec3f { return vec3f(oldForce[3u * i], oldForce[3u * i + 1u], oldForce[3u * i + 2u]); }\nfn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: centroid (GRAVITY_CENTER 0) or origin (1); regular or strong\n var q = pi.xyz;\n if (GRAVITY_CENTER == 0u) { q = pi.xyz - S.centroid.xyz; }\n if (STRONG_GRAVITY) { return -P.gravity * pi.w * q; }\n let d = length(q);\n if (d > FA2_DIST_FLOOR) { return -P.gravity * pi.w * q / d; }\n return vec3f(0.0);\n}\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\nfn kick_magnitude(mi: f32, mj: f32) -> f32 { // the law's magnitude at d = FA2_DIST_FLOOR (the P5 plan's PD-10)\n if (LAW == 1u) { return P.frK * P.frK / FA2_DIST_FLOOR; }\n if (LAW == 2u) { return -P.coulomb * mi * mj / FA2_DIST_FLOOR_SQ; }\n return P.scalingRatio * mi * mj / FA2_DIST_FLOOR;\n}\nfn pair_force(i: u32, pi: vec4f, jj: u32, o: vec4f) -> vec3f { // the exact pair law of K3 (7.6, 7.20): the floor (FA2 only), the coincident kick\n let d = pi.xyz - o.xyz;\n var d2 = dot(d, d);\n if (d2 < FA2_COINCIDENT_SQ) { return kick_dir(i, jj, P.dim) * kick_magnitude(pi.w, o.w); }\n if (LAW == 0u) { d2 = max(d2, FA2_DIST_FLOOR_SQ); }\n let k = P.scalingRatio * pi.w * o.w;\n if (LAW == 1u) { return d * (P.frK * P.frK / d2); }\n if (LAW == 2u) { return d * (-P.coulomb * pi.w * o.w / (d2 * sqrt(d2))); }\n return d * (k / d2);\n}\nfn cell_sum(i: u32, pi: vec4f, c: u32, own: bool) -> vec3f { // one finest cell: exact below nearMax entries, Horvitz-Thompson above (PD-15)\n let start = cellStart[c];\n let count = cellStart[c + 1u] - start;\n var f = vec3f(0.0);\n if (count <= P.nearMax) {\n for (var k = start; k < start + count; k = k + 1u) {\n let jj = sortedIdx[k];\n if (jj != i) { f = f + pair_force(i, pi, jj, pos[jj]); }\n }\n return f;\n }\n let base = (c ^ (P.iterationIndex * 0x9E3779B9u)) ^ P.seed; // the per-iteration draw seed (7.16)\n var sampled = 0u;\n for (var k = 0u; k < P.nearMax; k = k + 1u) {\n let jj = sortedIdx[start + (lowbias32(base ^ (k * 0x85EBCA6Bu)) % count)]; // draw k: independent inclusion, with replacement (PD-15)\n if (jj == i) { continue; }\n f = f + pair_force(i, pi, jj, pos[jj]);\n sampled = sampled + 1u;\n }\n if (sampled == 0u) { return vec3f(0.0); }\n let others = select(count, count - 1u, own);\n return f * (f32(others) / f32(sampled)); // others / sampled over the realised sample (DEP-P4-K)\n}\n\n@compute @workgroup_size(WG)\nfn grid_near_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let t = linear_id(wid, lid.x);\n let valid = t < P.n;\n var i = 0u;\n var pi = vec4f(0.0);\n var f = vec3f(0.0);\n if (valid) {\n i = sortedIdx[t]; // sorted order (D24)\n pi = pos[i];\n let gf = f32(P.gridMax);\n let q = (pi.xyz - S.gridMin.xyz) * S.invCellSize;\n var c0 = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));\n if (P.dim == 2u) { c0.z = 0; } // 2D: the one z plane\n let g = i32(P.gridMax);\n var inside = c0.x >= 0 && c0.x < g && c0.y >= 0 && c0.y < g;\n if (P.dim == 3u) { inside = inside && c0.z >= 0 && c0.z < g; }\n if (inside) {\n let zr = select(0, 1, P.dim == 3u);\n for (var dz = -zr; dz <= zr; dz = dz + 1) {\n for (var dy = -1; dy <= 1; dy = dy + 1) {\n for (var dx = -1; dx <= 1; dx = dx + 1) {\n let cx = c0.x + dx;\n let cy = c0.y + dy;\n let cz = c0.z + dz;\n if (cx < 0 || cx >= g || cy < 0 || cy >= g || cz < 0 || cz >= g) { continue; }\n let c = u32(cx) + P.gridMax * (u32(cy) + select(0u, P.gridMax * u32(cz), P.dim == 3u));\n f = f + cell_sum(i, pi, c, dx == 0 && dy == 0 && dz == 0);\n }\n }\n }\n } else {\n f = cell_sum(i, pi, grid_cells(), true); // an outside node: the pseudo-cell alone\n }\n }\n // epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it (K3's text)\n var sw = 0.0;\n var tr = 0.0;\n if (valid) {\n f = f + gravity_force(pi);\n let fnew = load_force(i) + f;\n store_force(i, fnew);\n if (SWING_MODE == 1u) { // NetworkX: positions and forces mixed, every node (7.2)\n sw = pi.w * length(pi.xyz - fnew);\n tr = 0.5 * pi.w * length(pi.xyz + fnew);\n } else if (!mask_bit(fixedMask[i >> 5u], i)) { // paper: free nodes only\n let fold = load_old(i);\n sw = pi.w * length(fnew - fold);\n tr = 0.5 * pi.w * length(fnew + fold);\n }\n }\n let tt = wg_reduce_vec4(vec4f(sw, tr, 0.0, 0.0), lid.x, 0u); // uniform control flow: 256 -> 1\n if (lid.x == 0u) { partials[group_id(wid)].swingTraction = tt.xy; }\n}\n";
19
+ //# sourceMappingURL=grid-near-field.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"grid-near-field.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-near-field.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;GAgBG;AACH,eAAO,MAAM,iBAAiB,27KA8G7B,CAAC"}