@graphty/webgpu-graph-algorithms 0.0.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (269) hide show
  1. package/README.md +344 -23
  2. package/dist/browser.d.ts +1 -0
  3. package/dist/browser.js +32 -0
  4. package/dist/browser.js.map +1 -0
  5. package/dist/chunks/context-E6iKaeuJ.js +3136 -0
  6. package/dist/chunks/context-E6iKaeuJ.js.map +1 -0
  7. package/dist/node.d.ts +1 -0
  8. package/dist/node.js +131 -0
  9. package/dist/node.js.map +1 -0
  10. package/dist/src/accelerator.d.ts +26 -0
  11. package/dist/src/accelerator.d.ts.map +1 -0
  12. package/dist/src/accelerator.js +101 -0
  13. package/dist/src/accelerator.js.map +1 -0
  14. package/dist/src/algorithms/degree.d.ts +35 -0
  15. package/dist/src/algorithms/degree.d.ts.map +1 -0
  16. package/dist/src/algorithms/degree.js +119 -0
  17. package/dist/src/algorithms/degree.js.map +1 -0
  18. package/dist/src/browser/index.d.ts +23 -0
  19. package/dist/src/browser/index.d.ts.map +1 -0
  20. package/dist/src/browser/index.js +48 -0
  21. package/dist/src/browser/index.js.map +1 -0
  22. package/dist/src/constants.d.ts +92 -0
  23. package/dist/src/constants.d.ts.map +1 -0
  24. package/dist/src/constants.js +92 -0
  25. package/dist/src/constants.js.map +1 -0
  26. package/dist/src/context.d.ts +84 -0
  27. package/dist/src/context.d.ts.map +1 -0
  28. package/dist/src/context.js +304 -0
  29. package/dist/src/context.js.map +1 -0
  30. package/dist/src/device/acquire.d.ts +57 -0
  31. package/dist/src/device/acquire.d.ts.map +1 -0
  32. package/dist/src/device/acquire.js +232 -0
  33. package/dist/src/device/acquire.js.map +1 -0
  34. package/dist/src/device/caps.d.ts +43 -0
  35. package/dist/src/device/caps.d.ts.map +1 -0
  36. package/dist/src/device/caps.js +104 -0
  37. package/dist/src/device/caps.js.map +1 -0
  38. package/dist/src/device/error-scope.d.ts +75 -0
  39. package/dist/src/device/error-scope.d.ts.map +1 -0
  40. package/dist/src/device/error-scope.js +152 -0
  41. package/dist/src/device/error-scope.js.map +1 -0
  42. package/dist/src/device/lost.d.ts +51 -0
  43. package/dist/src/device/lost.d.ts.map +1 -0
  44. package/dist/src/device/lost.js +130 -0
  45. package/dist/src/device/lost.js.map +1 -0
  46. package/dist/src/device/webgpu-constants.d.ts +31 -0
  47. package/dist/src/device/webgpu-constants.d.ts.map +1 -0
  48. package/dist/src/device/webgpu-constants.js +31 -0
  49. package/dist/src/device/webgpu-constants.js.map +1 -0
  50. package/dist/src/errors.d.ts +56 -0
  51. package/dist/src/errors.d.ts.map +1 -0
  52. package/dist/src/errors.js +57 -0
  53. package/dist/src/errors.js.map +1 -0
  54. package/dist/src/index.d.ts +29 -0
  55. package/dist/src/index.d.ts.map +1 -0
  56. package/dist/src/index.js +27 -0
  57. package/dist/src/index.js.map +1 -0
  58. package/dist/src/kernel/batch.d.ts +116 -0
  59. package/dist/src/kernel/batch.d.ts.map +1 -0
  60. package/dist/src/kernel/batch.js +335 -0
  61. package/dist/src/kernel/batch.js.map +1 -0
  62. package/dist/src/kernel/dispatch.d.ts +59 -0
  63. package/dist/src/kernel/dispatch.d.ts.map +1 -0
  64. package/dist/src/kernel/dispatch.js +139 -0
  65. package/dist/src/kernel/dispatch.js.map +1 -0
  66. package/dist/src/kernel/kernel.d.ts +84 -0
  67. package/dist/src/kernel/kernel.d.ts.map +1 -0
  68. package/dist/src/kernel/kernel.js +239 -0
  69. package/dist/src/kernel/kernel.js.map +1 -0
  70. package/dist/src/kernel/pipeline-cache.d.ts +90 -0
  71. package/dist/src/kernel/pipeline-cache.d.ts.map +1 -0
  72. package/dist/src/kernel/pipeline-cache.js +251 -0
  73. package/dist/src/kernel/pipeline-cache.js.map +1 -0
  74. package/dist/src/kernel/prelude.d.ts +35 -0
  75. package/dist/src/kernel/prelude.d.ts.map +1 -0
  76. package/dist/src/kernel/prelude.js +211 -0
  77. package/dist/src/kernel/prelude.js.map +1 -0
  78. package/dist/src/kernel/profiler.d.ts +64 -0
  79. package/dist/src/kernel/profiler.d.ts.map +1 -0
  80. package/dist/src/kernel/profiler.js +120 -0
  81. package/dist/src/kernel/profiler.js.map +1 -0
  82. package/dist/src/kernel/struct-block.d.ts +122 -0
  83. package/dist/src/kernel/struct-block.d.ts.map +1 -0
  84. package/dist/src/kernel/struct-block.js +353 -0
  85. package/dist/src/kernel/struct-block.js.map +1 -0
  86. package/dist/src/kernel/uniform-ring.d.ts +70 -0
  87. package/dist/src/kernel/uniform-ring.d.ts.map +1 -0
  88. package/dist/src/kernel/uniform-ring.js +146 -0
  89. package/dist/src/kernel/uniform-ring.js.map +1 -0
  90. package/dist/src/kernel/wgsl.d.ts +88 -0
  91. package/dist/src/kernel/wgsl.d.ts.map +1 -0
  92. package/dist/src/kernel/wgsl.js +390 -0
  93. package/dist/src/kernel/wgsl.js.map +1 -0
  94. package/dist/src/kernels.d.ts +81 -0
  95. package/dist/src/kernels.d.ts.map +1 -0
  96. package/dist/src/kernels.js +417 -0
  97. package/dist/src/kernels.js.map +1 -0
  98. package/dist/src/layouts/force-simulation.d.ts +498 -0
  99. package/dist/src/layouts/force-simulation.d.ts.map +1 -0
  100. package/dist/src/layouts/force-simulation.js +1650 -0
  101. package/dist/src/layouts/force-simulation.js.map +1 -0
  102. package/dist/src/layouts/forceatlas2.d.ts +210 -0
  103. package/dist/src/layouts/forceatlas2.d.ts.map +1 -0
  104. package/dist/src/layouts/forceatlas2.js +759 -0
  105. package/dist/src/layouts/forceatlas2.js.map +1 -0
  106. package/dist/src/layouts/inputs.d.ts +40 -0
  107. package/dist/src/layouts/inputs.d.ts.map +1 -0
  108. package/dist/src/layouts/inputs.js +185 -0
  109. package/dist/src/layouts/inputs.js.map +1 -0
  110. package/dist/src/layouts/repulsion-exact.d.ts +85 -0
  111. package/dist/src/layouts/repulsion-exact.d.ts.map +1 -0
  112. package/dist/src/layouts/repulsion-exact.js +134 -0
  113. package/dist/src/layouts/repulsion-exact.js.map +1 -0
  114. package/dist/src/layouts/seed.d.ts +56 -0
  115. package/dist/src/layouts/seed.d.ts.map +1 -0
  116. package/dist/src/layouts/seed.js +173 -0
  117. package/dist/src/layouts/seed.js.map +1 -0
  118. package/dist/src/memory/buffer-pool.d.ts +73 -0
  119. package/dist/src/memory/buffer-pool.d.ts.map +1 -0
  120. package/dist/src/memory/buffer-pool.js +170 -0
  121. package/dist/src/memory/buffer-pool.js.map +1 -0
  122. package/dist/src/memory/lease.d.ts +53 -0
  123. package/dist/src/memory/lease.d.ts.map +1 -0
  124. package/dist/src/memory/lease.js +85 -0
  125. package/dist/src/memory/lease.js.map +1 -0
  126. package/dist/src/memory/readback.d.ts +143 -0
  127. package/dist/src/memory/readback.d.ts.map +1 -0
  128. package/dist/src/memory/readback.js +375 -0
  129. package/dist/src/memory/readback.js.map +1 -0
  130. package/dist/src/memory/residency.d.ts +83 -0
  131. package/dist/src/memory/residency.d.ts.map +1 -0
  132. package/dist/src/memory/residency.js +573 -0
  133. package/dist/src/memory/residency.js.map +1 -0
  134. package/dist/src/memory/upload-plan.d.ts +101 -0
  135. package/dist/src/memory/upload-plan.d.ts.map +1 -0
  136. package/dist/src/memory/upload-plan.js +265 -0
  137. package/dist/src/memory/upload-plan.js.map +1 -0
  138. package/dist/src/node/index.d.ts +64 -0
  139. package/dist/src/node/index.d.ts.map +1 -0
  140. package/dist/src/node/index.js +183 -0
  141. package/dist/src/node/index.js.map +1 -0
  142. package/dist/src/primitives/reduce.d.ts +57 -0
  143. package/dist/src/primitives/reduce.d.ts.map +1 -0
  144. package/dist/src/primitives/reduce.js +161 -0
  145. package/dist/src/primitives/reduce.js.map +1 -0
  146. package/dist/src/primitives/segmented-reduce.d.ts +38 -0
  147. package/dist/src/primitives/segmented-reduce.d.ts.map +1 -0
  148. package/dist/src/primitives/segmented-reduce.js +211 -0
  149. package/dist/src/primitives/segmented-reduce.js.map +1 -0
  150. package/dist/src/types/accelerator.d.ts +209 -0
  151. package/dist/src/types/accelerator.d.ts.map +1 -0
  152. package/dist/src/types/accelerator.js +8 -0
  153. package/dist/src/types/accelerator.js.map +1 -0
  154. package/dist/src/types/context.d.ts +114 -0
  155. package/dist/src/types/context.d.ts.map +1 -0
  156. package/dist/src/types/context.js +7 -0
  157. package/dist/src/types/context.js.map +1 -0
  158. package/dist/src/types/layout.d.ts +95 -0
  159. package/dist/src/types/layout.d.ts.map +1 -0
  160. package/dist/src/types/layout.js +6 -0
  161. package/dist/src/types/layout.js.map +1 -0
  162. package/dist/src/types/memory.d.ts +22 -0
  163. package/dist/src/types/memory.d.ts.map +1 -0
  164. package/dist/src/types/memory.js +7 -0
  165. package/dist/src/types/memory.js.map +1 -0
  166. package/dist/src/types/options.d.ts +78 -0
  167. package/dist/src/types/options.d.ts.map +1 -0
  168. package/dist/src/types/options.js +7 -0
  169. package/dist/src/types/options.js.map +1 -0
  170. package/dist/src/types/run.d.ts +13 -0
  171. package/dist/src/types/run.d.ts.map +1 -0
  172. package/dist/src/types/run.js +6 -0
  173. package/dist/src/types/run.js.map +1 -0
  174. package/dist/src/wgsl/degree.wgsl.d.ts +10 -0
  175. package/dist/src/wgsl/degree.wgsl.d.ts.map +1 -0
  176. package/dist/src/wgsl/degree.wgsl.js +25 -0
  177. package/dist/src/wgsl/degree.wgsl.js.map +1 -0
  178. package/dist/src/wgsl/fa2-attraction.wgsl.d.ts +12 -0
  179. package/dist/src/wgsl/fa2-attraction.wgsl.d.ts.map +1 -0
  180. package/dist/src/wgsl/fa2-attraction.wgsl.js +37 -0
  181. package/dist/src/wgsl/fa2-attraction.wgsl.js.map +1 -0
  182. package/dist/src/wgsl/fa2-integrate.wgsl.d.ts +13 -0
  183. package/dist/src/wgsl/fa2-integrate.wgsl.d.ts.map +1 -0
  184. package/dist/src/wgsl/fa2-integrate.wgsl.js +69 -0
  185. package/dist/src/wgsl/fa2-integrate.wgsl.js.map +1 -0
  186. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +12 -0
  187. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -0
  188. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +79 -0
  189. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -0
  190. package/dist/src/wgsl/fa2-speed-finalize.wgsl.d.ts +15 -0
  191. package/dist/src/wgsl/fa2-speed-finalize.wgsl.d.ts.map +1 -0
  192. package/dist/src/wgsl/fa2-speed-finalize.wgsl.js +54 -0
  193. package/dist/src/wgsl/fa2-speed-finalize.wgsl.js.map +1 -0
  194. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +14 -0
  195. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -0
  196. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +57 -0
  197. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -0
  198. package/dist/src/wgsl/fa2-to-scene.wgsl.d.ts +11 -0
  199. package/dist/src/wgsl/fa2-to-scene.wgsl.d.ts.map +1 -0
  200. package/dist/src/wgsl/fa2-to-scene.wgsl.js +19 -0
  201. package/dist/src/wgsl/fa2-to-scene.wgsl.js.map +1 -0
  202. package/dist/src/wgsl/fill.wgsl.d.ts +7 -0
  203. package/dist/src/wgsl/fill.wgsl.d.ts.map +1 -0
  204. package/dist/src/wgsl/fill.wgsl.js +14 -0
  205. package/dist/src/wgsl/fill.wgsl.js.map +1 -0
  206. package/dist/src/wgsl/reduce.wgsl.d.ts +10 -0
  207. package/dist/src/wgsl/reduce.wgsl.d.ts.map +1 -0
  208. package/dist/src/wgsl/reduce.wgsl.js +63 -0
  209. package/dist/src/wgsl/reduce.wgsl.js.map +1 -0
  210. package/dist/src/wgsl/segmented-reduce.wgsl.d.ts +13 -0
  211. package/dist/src/wgsl/segmented-reduce.wgsl.d.ts.map +1 -0
  212. package/dist/src/wgsl/segmented-reduce.wgsl.js +35 -0
  213. package/dist/src/wgsl/segmented-reduce.wgsl.js.map +1 -0
  214. package/dist/tsconfig.build.tsbuildinfo +1 -0
  215. package/dist/webgpu-graph-algorithms.d.ts +1 -0
  216. package/dist/webgpu-graph-algorithms.js +4454 -0
  217. package/dist/webgpu-graph-algorithms.js.map +1 -0
  218. package/package.json +108 -17
  219. package/src/accelerator.ts +117 -0
  220. package/src/algorithms/degree.ts +142 -0
  221. package/src/browser/index.ts +57 -0
  222. package/src/constants.ts +116 -0
  223. package/src/context.ts +399 -0
  224. package/src/device/acquire.ts +256 -0
  225. package/src/device/caps.ts +122 -0
  226. package/src/device/error-scope.ts +171 -0
  227. package/src/device/lost.ts +142 -0
  228. package/src/device/webgpu-constants.ts +44 -0
  229. package/src/errors.ts +94 -0
  230. package/src/index.ts +102 -0
  231. package/src/kernel/batch.ts +427 -0
  232. package/src/kernel/dispatch.ts +162 -0
  233. package/src/kernel/kernel.ts +311 -0
  234. package/src/kernel/pipeline-cache.ts +288 -0
  235. package/src/kernel/prelude.ts +229 -0
  236. package/src/kernel/profiler.ts +148 -0
  237. package/src/kernel/struct-block.ts +439 -0
  238. package/src/kernel/uniform-ring.ts +184 -0
  239. package/src/kernel/wgsl.ts +490 -0
  240. package/src/kernels.ts +511 -0
  241. package/src/layouts/force-simulation.ts +2111 -0
  242. package/src/layouts/forceatlas2.ts +942 -0
  243. package/src/layouts/inputs.ts +252 -0
  244. package/src/layouts/repulsion-exact.ts +183 -0
  245. package/src/layouts/seed.ts +198 -0
  246. package/src/memory/buffer-pool.ts +204 -0
  247. package/src/memory/lease.ts +93 -0
  248. package/src/memory/readback.ts +429 -0
  249. package/src/memory/residency.ts +753 -0
  250. package/src/memory/upload-plan.ts +350 -0
  251. package/src/node/index.ts +230 -0
  252. package/src/primitives/reduce.ts +233 -0
  253. package/src/primitives/segmented-reduce.ts +270 -0
  254. package/src/types/accelerator.ts +236 -0
  255. package/src/types/context.ts +135 -0
  256. package/src/types/layout.ts +103 -0
  257. package/src/types/memory.ts +23 -0
  258. package/src/types/options.ts +84 -0
  259. package/src/types/run.ts +13 -0
  260. package/src/wgsl/degree.wgsl.ts +24 -0
  261. package/src/wgsl/fa2-attraction.wgsl.ts +37 -0
  262. package/src/wgsl/fa2-integrate.wgsl.ts +69 -0
  263. package/src/wgsl/fa2-repulsion-exact.wgsl.ts +78 -0
  264. package/src/wgsl/fa2-speed-finalize.wgsl.ts +53 -0
  265. package/src/wgsl/fa2-stats-finalize.wgsl.ts +57 -0
  266. package/src/wgsl/fa2-to-scene.wgsl.ts +19 -0
  267. package/src/wgsl/fill.wgsl.ts +13 -0
  268. package/src/wgsl/reduce.wgsl.ts +62 -0
  269. package/src/wgsl/segmented-reduce.wgsl.ts +35 -0
@@ -0,0 +1,6 @@
1
+ /**
2
+ * The run options every algorithm function accepts beside its own (spec 3.3, design 10.7; contract 3.3).
3
+ * Types only.
4
+ */
5
+ export {};
6
+ //# sourceMappingURL=run.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"run.js","sourceRoot":"","sources":["../../../src/types/run.ts"],"names":[],"mappings":"AAAA;;;GAGG"}
@@ -0,0 +1,10 @@
1
+ /**
2
+ * The `degree` kernel body (contract 4.5; spec 6 row 3, 11.5): the out-degree of every row of `[P.start, P.end)`
3
+ * through the row-walking gather with the USE_PERM dummy pattern, counting the arcs of the bound window whose
4
+ * neighbour index is a valid node; `P.accumulate == 1` adds into `out[i]` instead of overwriting (the P4 windowed
5
+ * loop). Bindings, overrides and the `RangeParams` block come from the registry entry (src/kernels.ts); WG,
6
+ * USE_PERM and `linear_id` from the prelude. The text is normative: test/helpers/sabotage.ts mutates it textually,
7
+ * so it is copied from the contract and never re-derived or restyled.
8
+ */
9
+ export declare const degreeWgsl = "\n@compute @workgroup_size(WG)\nfn degree(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let row = linear_id(wid, lid.x) + P.start;\n if (row >= P.end) { return; }\n let i = select(row, perm[row], USE_PERM);\n let a0 = max(rowPtr[i], P.arcBase);\n let a1 = min(rowPtr[i + 1u], P.arcEnd);\n var d = 0u;\n for (var arc = a0; arc < a1; arc = arc + 1u) {\n let nbr = colIdx[arc - P.arcBase]; // `target` is a WGSL reserved word (spec 16.2); never use it as an identifier\n d = d + select(0u, 1u, nbr < P.n);\n }\n out[i] = select(d, out[i] + d, P.accumulate == 1u);\n}\n";
10
+ //# sourceMappingURL=degree.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"degree.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/degree.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AACH,eAAO,MAAM,UAAU,4pBAetB,CAAC"}
@@ -0,0 +1,25 @@
1
+ /**
2
+ * The `degree` kernel body (contract 4.5; spec 6 row 3, 11.5): the out-degree of every row of `[P.start, P.end)`
3
+ * through the row-walking gather with the USE_PERM dummy pattern, counting the arcs of the bound window whose
4
+ * neighbour index is a valid node; `P.accumulate == 1` adds into `out[i]` instead of overwriting (the P4 windowed
5
+ * loop). Bindings, overrides and the `RangeParams` block come from the registry entry (src/kernels.ts); WG,
6
+ * USE_PERM and `linear_id` from the prelude. The text is normative: test/helpers/sabotage.ts mutates it textually,
7
+ * so it is copied from the contract and never re-derived or restyled.
8
+ */
9
+ export const degreeWgsl = /* wgsl */ `
10
+ @compute @workgroup_size(WG)
11
+ fn degree(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
12
+ let row = linear_id(wid, lid.x) + P.start;
13
+ if (row >= P.end) { return; }
14
+ let i = select(row, perm[row], USE_PERM);
15
+ let a0 = max(rowPtr[i], P.arcBase);
16
+ let a1 = min(rowPtr[i + 1u], P.arcEnd);
17
+ var d = 0u;
18
+ for (var arc = a0; arc < a1; arc = arc + 1u) {
19
+ let nbr = colIdx[arc - P.arcBase]; // \`target\` is a WGSL reserved word (spec 16.2); never use it as an identifier
20
+ d = d + select(0u, 1u, nbr < P.n);
21
+ }
22
+ out[i] = select(d, out[i] + d, P.accumulate == 1u);
23
+ }
24
+ `;
25
+ //# sourceMappingURL=degree.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"degree.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/degree.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AACH,MAAM,CAAC,MAAM,UAAU,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;CAepC,CAAC"}
@@ -0,0 +1,12 @@
1
+ /**
2
+ * K2 of the ForceAtlas2 iteration, `fa2-attraction` (spec 7.5; contract 4.5): the thread-per-row gather of the
3
+ * attraction force over the undirected CSR rows (both arcs present, so the sum is symmetric with no atomics), the
4
+ * linear or linlog law, the optional weights, the distributed-action division by the mass in `pos.w`, written as
5
+ * the FIRST writer of `force` each iteration. P3 ships the thread-per-row tier over `[tierStart, tierEnd)` = `[0, n)`
6
+ * with `USE_PERM = false`; the subgroup / workgroup tiers arrive with P4 (`TIER`).
7
+ *
8
+ * Body only (spec 3.5, D9); normative text (contract 4.5); the K2 sabotage mutations (P3-T5) are textual edits of it.
9
+ */
10
+ /** The K2 body: entry point `attraction`; an early return is legal here because no barrier follows (spec 3.5 rule 1). */
11
+ export declare const fa2AttractionWgsl = "fn store_force(i: u32, f: vec3f) {\n force[3u * i] = f.x;\n force[3u * i + 1u] = f.y;\n force[3u * i + 2u] = f.z;\n}\n\n@compute @workgroup_size(WG)\nfn attraction(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let row = linear_id(wid, lid.x) + P.tierStart;\n if (row >= P.tierEnd) { return; } // no barrier follows in this tier (3.5 rule 1)\n let i = select(row, perm[row], USE_PERM);\n let pi = pos[i]; // xyz + mass in one load (D23)\n var f = vec3f(0.0);\n for (var a = rowPtr[i]; a < rowPtr[i + 1u]; a = a + 1u) {\n let j = colIdx[a];\n if (j == i) { continue; } // a self-loop exerts no force\n var w = 1.0;\n if (HAS_WEIGHTS) { w = weights[a]; }\n let d = pos[j].xyz - pi.xyz; // toward j\n let len = max(length(d), FA2_DIST_FLOOR);\n let mag = select(w, w * log(1.0 + len) / len, LINLOG); // linear: |F| = w len; linlog: |F| = w log(1 + len)\n f = f + d * mag;\n }\n if (DISTRIBUTED) { f = f / pi.w; }\n store_force(i, f); // overwrites: attraction is the first writer of force each iteration\n}";
12
+ //# sourceMappingURL=fa2-attraction.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"fa2-attraction.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-attraction.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,yHAAyH;AACzH,eAAO,MAAM,iBAAiB,wxCAyB5B,CAAC"}
@@ -0,0 +1,37 @@
1
+ /**
2
+ * K2 of the ForceAtlas2 iteration, `fa2-attraction` (spec 7.5; contract 4.5): the thread-per-row gather of the
3
+ * attraction force over the undirected CSR rows (both arcs present, so the sum is symmetric with no atomics), the
4
+ * linear or linlog law, the optional weights, the distributed-action division by the mass in `pos.w`, written as
5
+ * the FIRST writer of `force` each iteration. P3 ships the thread-per-row tier over `[tierStart, tierEnd)` = `[0, n)`
6
+ * with `USE_PERM = false`; the subgroup / workgroup tiers arrive with P4 (`TIER`).
7
+ *
8
+ * Body only (spec 3.5, D9); normative text (contract 4.5); the K2 sabotage mutations (P3-T5) are textual edits of it.
9
+ */
10
+ /** The K2 body: entry point `attraction`; an early return is legal here because no barrier follows (spec 3.5 rule 1). */
11
+ export const fa2AttractionWgsl = /* wgsl */ `fn store_force(i: u32, f: vec3f) {
12
+ force[3u * i] = f.x;
13
+ force[3u * i + 1u] = f.y;
14
+ force[3u * i + 2u] = f.z;
15
+ }
16
+
17
+ @compute @workgroup_size(WG)
18
+ fn attraction(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
19
+ let row = linear_id(wid, lid.x) + P.tierStart;
20
+ if (row >= P.tierEnd) { return; } // no barrier follows in this tier (3.5 rule 1)
21
+ let i = select(row, perm[row], USE_PERM);
22
+ let pi = pos[i]; // xyz + mass in one load (D23)
23
+ var f = vec3f(0.0);
24
+ for (var a = rowPtr[i]; a < rowPtr[i + 1u]; a = a + 1u) {
25
+ let j = colIdx[a];
26
+ if (j == i) { continue; } // a self-loop exerts no force
27
+ var w = 1.0;
28
+ if (HAS_WEIGHTS) { w = weights[a]; }
29
+ let d = pos[j].xyz - pi.xyz; // toward j
30
+ let len = max(length(d), FA2_DIST_FLOOR);
31
+ let mag = select(w, w * log(1.0 + len) / len, LINLOG); // linear: |F| = w len; linlog: |F| = w log(1 + len)
32
+ f = f + d * mag;
33
+ }
34
+ if (DISTRIBUTED) { f = f / pi.w; }
35
+ store_force(i, f); // overwrites: attraction is the first writer of force each iteration
36
+ }`;
37
+ //# sourceMappingURL=fa2-attraction.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"fa2-attraction.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-attraction.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,yHAAyH;AACzH,MAAM,CAAC,MAAM,iBAAiB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;EAyB1C,CAAC"}
@@ -0,0 +1,13 @@
1
+ /**
2
+ * K5 of the ForceAtlas2 iteration, `fa2-integrate` (spec 7.11; contract 4.5): the per-node speed factor
3
+ * `speed / (1 + sqrt(speed * swing_i))` with the local swing recomputed inline (paper mode `m |F(t) - F(t-1)|`,
4
+ * networkx mode `m |F|`, contract 4.6), the position update with no displacement clamp (D25), z never integrated
5
+ * in 2D (spec 7.13), `oldForce` stored in paper mode for fixed nodes too (spec 7.11), and the partials A
6
+ * (sum p, sum |p - c|^2, min, max with max |p - c|^2 in `max.w`) and C (sum |dp| over free rows, free count) in
7
+ * uniform control flow.
8
+ *
9
+ * Body only (spec 3.5, D9); normative text (contract 4.5); the K5 sabotage mutations (P3-T5) are textual edits of it.
10
+ */
11
+ /** The K5 body: entry point `integrate`; calls the reduction helpers (`needs: ["subgroups"]`, contract 4.3). */
12
+ export declare const fa2IntegrateWgsl = "fn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }\nfn load_old(i: u32) -> vec3f { return vec3f(oldForce[3u * i], oldForce[3u * i + 1u], oldForce[3u * i + 2u]); }\nfn store_old(i: u32, f: vec3f) {\n oldForce[3u * i] = f.x;\n oldForce[3u * i + 1u] = f.y;\n oldForce[3u * i + 2u] = f.z;\n}\n\n@compute @workgroup_size(WG)\nfn integrate(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n var dp = vec3f(0.0);\n var p = vec4f(0.0);\n var free = false;\n var valid = false;\n if (i < P.n) {\n valid = true;\n let f = load_force(i);\n p = pos[i];\n var swing_i = p.w * length(f); // SWING_MODE 1: NetworkX's local swinging m |F| (layout.py line 1497)\n if (SWING_MODE == 0u) { swing_i = p.w * length(f - load_old(i)); } // paper: m |F(t) - F(t-1)|, recomputed inline (7.2)\n let factor = S.speed / (1.0 + sqrt(S.speed * swing_i));\n let fixed = mask_bit(fixedMask[i >> 5u], i);\n dp = select(f * factor, vec3f(0.0), fixed); // no clamp on dp (D25)\n if (P.dim == 2u) { dp.z = 0.0; } // 2D never integrates z (7.13)\n p = vec4f(p.xyz + dp, p.w);\n pos[i] = p;\n if (SWING_MODE == 0u) { store_old(i, f); } // fixed nodes too, so a later unpin sees no stale swing (7.11)\n free = !fixed;\n }\n // uniform control flow from here (3.5 rule 1): partials A (sum p, sum |p - c|^2, min, max over valid rows) and C (sum |dp|, free count)\n let c = S.centroid.xyz;\n var sumv = vec4f(0.0);\n var lo = vec4f(F32_MAX);\n var hi = vec4f(-F32_MAX);\n var dl = 0.0;\n var fr = 0u;\n if (valid) {\n let q = p.xyz - c;\n sumv = vec4f(p.xyz, dot(q, q));\n lo = vec4f(p.xyz, 0.0);\n hi = vec4f(p.xyz, dot(q, q)); // max.w carries max |p - c|^2 so K1 can write the exact layoutRadius (spec 3.3)\n }\n if (free) { dl = length(dp); fr = 1u; }\n let tSum = wg_reduce_vec4(sumv, lid.x, 0u);\n let tLo = wg_reduce_vec4(lo, lid.x, 1u);\n let tHi = wg_reduce_vec4(hi, lid.x, 2u);\n let tDl = wg_reduce_f32(dl, lid.x, 0u);\n let tFr = wg_reduce_u32(fr, lid.x, 0u);\n if (lid.x == 0u) {\n let g = group_id(wid);\n partials[g].sum = tSum;\n partials[g].min = tLo;\n partials[g].max = tHi;\n partials[g].dispFree = vec2f(tDl, f32(tFr));\n }\n}";
13
+ //# sourceMappingURL=fa2-integrate.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"fa2-integrate.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-integrate.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AAEH,gHAAgH;AAChH,eAAO,MAAM,gBAAgB,4jFAwD3B,CAAC"}
@@ -0,0 +1,69 @@
1
+ /**
2
+ * K5 of the ForceAtlas2 iteration, `fa2-integrate` (spec 7.11; contract 4.5): the per-node speed factor
3
+ * `speed / (1 + sqrt(speed * swing_i))` with the local swing recomputed inline (paper mode `m |F(t) - F(t-1)|`,
4
+ * networkx mode `m |F|`, contract 4.6), the position update with no displacement clamp (D25), z never integrated
5
+ * in 2D (spec 7.13), `oldForce` stored in paper mode for fixed nodes too (spec 7.11), and the partials A
6
+ * (sum p, sum |p - c|^2, min, max with max |p - c|^2 in `max.w`) and C (sum |dp| over free rows, free count) in
7
+ * uniform control flow.
8
+ *
9
+ * Body only (spec 3.5, D9); normative text (contract 4.5); the K5 sabotage mutations (P3-T5) are textual edits of it.
10
+ */
11
+ /** The K5 body: entry point `integrate`; calls the reduction helpers (`needs: ["subgroups"]`, contract 4.3). */
12
+ export const fa2IntegrateWgsl = /* wgsl */ `fn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }
13
+ fn load_old(i: u32) -> vec3f { return vec3f(oldForce[3u * i], oldForce[3u * i + 1u], oldForce[3u * i + 2u]); }
14
+ fn store_old(i: u32, f: vec3f) {
15
+ oldForce[3u * i] = f.x;
16
+ oldForce[3u * i + 1u] = f.y;
17
+ oldForce[3u * i + 2u] = f.z;
18
+ }
19
+
20
+ @compute @workgroup_size(WG)
21
+ fn integrate(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
22
+ let i = linear_id(wid, lid.x);
23
+ var dp = vec3f(0.0);
24
+ var p = vec4f(0.0);
25
+ var free = false;
26
+ var valid = false;
27
+ if (i < P.n) {
28
+ valid = true;
29
+ let f = load_force(i);
30
+ p = pos[i];
31
+ var swing_i = p.w * length(f); // SWING_MODE 1: NetworkX's local swinging m |F| (layout.py line 1497)
32
+ if (SWING_MODE == 0u) { swing_i = p.w * length(f - load_old(i)); } // paper: m |F(t) - F(t-1)|, recomputed inline (7.2)
33
+ let factor = S.speed / (1.0 + sqrt(S.speed * swing_i));
34
+ let fixed = mask_bit(fixedMask[i >> 5u], i);
35
+ dp = select(f * factor, vec3f(0.0), fixed); // no clamp on dp (D25)
36
+ if (P.dim == 2u) { dp.z = 0.0; } // 2D never integrates z (7.13)
37
+ p = vec4f(p.xyz + dp, p.w);
38
+ pos[i] = p;
39
+ if (SWING_MODE == 0u) { store_old(i, f); } // fixed nodes too, so a later unpin sees no stale swing (7.11)
40
+ free = !fixed;
41
+ }
42
+ // uniform control flow from here (3.5 rule 1): partials A (sum p, sum |p - c|^2, min, max over valid rows) and C (sum |dp|, free count)
43
+ let c = S.centroid.xyz;
44
+ var sumv = vec4f(0.0);
45
+ var lo = vec4f(F32_MAX);
46
+ var hi = vec4f(-F32_MAX);
47
+ var dl = 0.0;
48
+ var fr = 0u;
49
+ if (valid) {
50
+ let q = p.xyz - c;
51
+ sumv = vec4f(p.xyz, dot(q, q));
52
+ lo = vec4f(p.xyz, 0.0);
53
+ hi = vec4f(p.xyz, dot(q, q)); // max.w carries max |p - c|^2 so K1 can write the exact layoutRadius (spec 3.3)
54
+ }
55
+ if (free) { dl = length(dp); fr = 1u; }
56
+ let tSum = wg_reduce_vec4(sumv, lid.x, 0u);
57
+ let tLo = wg_reduce_vec4(lo, lid.x, 1u);
58
+ let tHi = wg_reduce_vec4(hi, lid.x, 2u);
59
+ let tDl = wg_reduce_f32(dl, lid.x, 0u);
60
+ let tFr = wg_reduce_u32(fr, lid.x, 0u);
61
+ if (lid.x == 0u) {
62
+ let g = group_id(wid);
63
+ partials[g].sum = tSum;
64
+ partials[g].min = tLo;
65
+ partials[g].max = tHi;
66
+ partials[g].dispFree = vec2f(tDl, f32(tFr));
67
+ }
68
+ }`;
69
+ //# sourceMappingURL=fa2-integrate.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"fa2-integrate.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-integrate.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AAEH,gHAAgH;AAChH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EAwDzC,CAAC"}
@@ -0,0 +1,12 @@
1
+ /**
2
+ * The `fa2-repulsion-exact` kernel body (K3; contract 4.5; spec 7.6, 7.9, 7.10): the tiled all-pairs repulsion
3
+ * `|F| = k m_i m_j / d` with the 0.01 distance floor and the antisymmetric coincident kick, the gravity epilogue
4
+ * (GRAVITY_CENTER 0 = centroid, 1 = origin; STRONG_GRAVITY) added under the `valid` guard, `force += f`, and the
5
+ * swing / traction workgroup reduction in uniform control flow (SWING_MODE 0 = paper: free nodes only against
6
+ * `oldForce`; 1 = NetworkX: positions and forces mixed, every node) whose lane 0 writes
7
+ * `partials[group].swingTraction`. `pos` is `array<vec4f>` with the mass in `.w`; `force` / `oldForce` are stride-3
8
+ * `array<f32>` read through the per-body helpers (4.4 rule 6). Normative text, copied verbatim: the P1-T5 sabotage
9
+ * rows (gravity sign, `k / d2`, the `jj != i` guard, the `.w` mass lane) are textual edits of this string.
10
+ */
11
+ export declare const fa2RepulsionExactWgsl = "\nvar<workgroup> tile: array<vec4f, WG>; // xyz + mass, 4 KiB at WG = 256\n\nfn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }\nfn store_force(i: u32, f: vec3f) {\n force[3u * i] = f.x;\n force[3u * i + 1u] = f.y;\n force[3u * i + 2u] = f.z;\n}\nfn load_old(i: u32) -> vec3f { return vec3f(oldForce[3u * i], oldForce[3u * i + 1u], oldForce[3u * i + 2u]); }\nfn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: centroid (GRAVITY_CENTER 0) or origin (1); regular or strong\n var q = pi.xyz;\n if (GRAVITY_CENTER == 0u) { q = pi.xyz - S.centroid.xyz; }\n if (STRONG_GRAVITY) { return -P.gravity * pi.w * q; }\n let d = length(q);\n if (d > FA2_DIST_FLOOR) { return -P.gravity * pi.w * q / d; }\n return vec3f(0.0);\n}\n\n@compute @workgroup_size(WG)\nfn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n let valid = i < P.n;\n var pi = vec4f(0.0);\n if (valid) { pi = pos[i]; }\n var f = vec3f(0.0);\n let tiles = (P.n + WG - 1u) / WG;\n for (var t = 0u; t < tiles; t = t + 1u) {\n let j = t * WG + lid.x;\n if (j < P.n) { tile[lid.x] = pos[j]; } else { tile[lid.x] = vec4f(0.0); } // guarded fill; mass 0 marks the pad\n workgroupBarrier(); // uniform: every invocation reaches it\n for (var s = 0u; s < WG; s = s + 1u) {\n let o = tile[s];\n let jj = t * WG + s;\n if (o.w > 0.0 && jj != i) { // mass > 0 for every real node, 0 for the pad\n let d = pi.xyz - o.xyz;\n var d2 = dot(d, d);\n if (d2 < FA2_COINCIDENT_SQ) { // coincident: antisymmetric unit kick of magnitude k m_i m_j / 0.01 (7.2)\n f = f + kick_dir(i, jj, P.dim) * (P.scalingRatio * pi.w * o.w / FA2_DIST_FLOOR);\n continue;\n }\n d2 = max(d2, FA2_DIST_FLOOR_SQ); // d >= 0.01\n let k = P.scalingRatio * pi.w * o.w;\n f = f + d * (k / d2); // |F| = k m_i m_j / d along d / d\n }\n }\n workgroupBarrier();\n }\n // epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it\n var sw = 0.0;\n var tr = 0.0;\n if (valid) {\n f = f + gravity_force(pi);\n let fnew = load_force(i) + f;\n store_force(i, fnew);\n if (SWING_MODE == 1u) { // NetworkX: positions and forces mixed, every node (7.2)\n sw = pi.w * length(pi.xyz - fnew);\n tr = 0.5 * pi.w * length(pi.xyz + fnew);\n } else if (!mask_bit(fixedMask[i >> 5u], i)) { // paper: free nodes only (Gephi ForceAtlas2.java 283-293)\n let fold = load_old(i);\n sw = pi.w * length(fnew - fold);\n tr = 0.5 * pi.w * length(fnew + fold);\n }\n }\n let t = wg_reduce_vec4(vec4f(sw, tr, 0.0, 0.0), lid.x, 0u); // uniform control flow: 256 -> 1\n if (lid.x == 0u) { partials[group_id(wid)].swingTraction = t.xy; }\n}\n";
12
+ //# sourceMappingURL=fa2-repulsion-exact.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"fa2-repulsion-exact.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-repulsion-exact.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AACH,eAAO,MAAM,qBAAqB,g4GAmEjC,CAAC"}
@@ -0,0 +1,79 @@
1
+ /**
2
+ * The `fa2-repulsion-exact` kernel body (K3; contract 4.5; spec 7.6, 7.9, 7.10): the tiled all-pairs repulsion
3
+ * `|F| = k m_i m_j / d` with the 0.01 distance floor and the antisymmetric coincident kick, the gravity epilogue
4
+ * (GRAVITY_CENTER 0 = centroid, 1 = origin; STRONG_GRAVITY) added under the `valid` guard, `force += f`, and the
5
+ * swing / traction workgroup reduction in uniform control flow (SWING_MODE 0 = paper: free nodes only against
6
+ * `oldForce`; 1 = NetworkX: positions and forces mixed, every node) whose lane 0 writes
7
+ * `partials[group].swingTraction`. `pos` is `array<vec4f>` with the mass in `.w`; `force` / `oldForce` are stride-3
8
+ * `array<f32>` read through the per-body helpers (4.4 rule 6). Normative text, copied verbatim: the P1-T5 sabotage
9
+ * rows (gravity sign, `k / d2`, the `jj != i` guard, the `.w` mass lane) are textual edits of this string.
10
+ */
11
+ export const fa2RepulsionExactWgsl = /* wgsl */ `
12
+ var<workgroup> tile: array<vec4f, WG>; // xyz + mass, 4 KiB at WG = 256
13
+
14
+ fn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }
15
+ fn store_force(i: u32, f: vec3f) {
16
+ force[3u * i] = f.x;
17
+ force[3u * i + 1u] = f.y;
18
+ force[3u * i + 2u] = f.z;
19
+ }
20
+ fn load_old(i: u32) -> vec3f { return vec3f(oldForce[3u * i], oldForce[3u * i + 1u], oldForce[3u * i + 2u]); }
21
+ fn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: centroid (GRAVITY_CENTER 0) or origin (1); regular or strong
22
+ var q = pi.xyz;
23
+ if (GRAVITY_CENTER == 0u) { q = pi.xyz - S.centroid.xyz; }
24
+ if (STRONG_GRAVITY) { return -P.gravity * pi.w * q; }
25
+ let d = length(q);
26
+ if (d > FA2_DIST_FLOOR) { return -P.gravity * pi.w * q / d; }
27
+ return vec3f(0.0);
28
+ }
29
+
30
+ @compute @workgroup_size(WG)
31
+ fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
32
+ let i = linear_id(wid, lid.x);
33
+ let valid = i < P.n;
34
+ var pi = vec4f(0.0);
35
+ if (valid) { pi = pos[i]; }
36
+ var f = vec3f(0.0);
37
+ let tiles = (P.n + WG - 1u) / WG;
38
+ for (var t = 0u; t < tiles; t = t + 1u) {
39
+ let j = t * WG + lid.x;
40
+ if (j < P.n) { tile[lid.x] = pos[j]; } else { tile[lid.x] = vec4f(0.0); } // guarded fill; mass 0 marks the pad
41
+ workgroupBarrier(); // uniform: every invocation reaches it
42
+ for (var s = 0u; s < WG; s = s + 1u) {
43
+ let o = tile[s];
44
+ let jj = t * WG + s;
45
+ if (o.w > 0.0 && jj != i) { // mass > 0 for every real node, 0 for the pad
46
+ let d = pi.xyz - o.xyz;
47
+ var d2 = dot(d, d);
48
+ if (d2 < FA2_COINCIDENT_SQ) { // coincident: antisymmetric unit kick of magnitude k m_i m_j / 0.01 (7.2)
49
+ f = f + kick_dir(i, jj, P.dim) * (P.scalingRatio * pi.w * o.w / FA2_DIST_FLOOR);
50
+ continue;
51
+ }
52
+ d2 = max(d2, FA2_DIST_FLOOR_SQ); // d >= 0.01
53
+ let k = P.scalingRatio * pi.w * o.w;
54
+ f = f + d * (k / d2); // |F| = k m_i m_j / d along d / d
55
+ }
56
+ }
57
+ workgroupBarrier();
58
+ }
59
+ // epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it
60
+ var sw = 0.0;
61
+ var tr = 0.0;
62
+ if (valid) {
63
+ f = f + gravity_force(pi);
64
+ let fnew = load_force(i) + f;
65
+ store_force(i, fnew);
66
+ if (SWING_MODE == 1u) { // NetworkX: positions and forces mixed, every node (7.2)
67
+ sw = pi.w * length(pi.xyz - fnew);
68
+ tr = 0.5 * pi.w * length(pi.xyz + fnew);
69
+ } else if (!mask_bit(fixedMask[i >> 5u], i)) { // paper: free nodes only (Gephi ForceAtlas2.java 283-293)
70
+ let fold = load_old(i);
71
+ sw = pi.w * length(fnew - fold);
72
+ tr = 0.5 * pi.w * length(fnew + fold);
73
+ }
74
+ }
75
+ let t = wg_reduce_vec4(vec4f(sw, tr, 0.0, 0.0), lid.x, 0u); // uniform control flow: 256 -> 1
76
+ if (lid.x == 0u) { partials[group_id(wid)].swingTraction = t.xy; }
77
+ }
78
+ `;
79
+ //# sourceMappingURL=fa2-repulsion-exact.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"fa2-repulsion-exact.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-repulsion-exact.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AACH,MAAM,CAAC,MAAM,qBAAqB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAmE/C,CAAC"}
@@ -0,0 +1,15 @@
1
+ /**
2
+ * The `fa2-speed-finalize` kernel body (K4; contract 4.5; spec 7.10): ONE workgroup sums `partials[*].swingTraction`
3
+ * sequentially per lane (deterministic), reduces in uniform control flow, then lane 0 runs the line-for-line port
4
+ * of the CPU `estimateFactor` (SWING_MODE 1 accumulates NetworkX's sums across iterations from S.swing / S.traction)
5
+ * and writes `S.speed`, `S.speedEfficiency`, `S.swing`, `S.traction` and the four controller fields of
6
+ * `T[P.iterationIndex]`. `target` is a WGSL reserved word, hence `targetSpeed`. Normative text, copied verbatim:
7
+ * the P1-T5 sabotage rows (`* 0.5` -> `* 0.9`, the 1.3 rise, the swing / traction swap) are textual edits of it.
8
+ * The halving predicate is `swing > 2.0 * tr`, the exact form of the port's `swing / traction > 2` (contract 4.5
9
+ * CONTRACT DECISION K4-1, G3 finding G3-F6): every paper-mode first iteration after load() has oldForce = 0, so
10
+ * traction is EXACTLY half the swing and the predicate sits on its knife edge; a multiplication by 2 and the
11
+ * comparison are exact on every device, while WGSL grants f32 division 2.5 ULP and Dawn on NVIDIA returns 2 + 1 ulp
12
+ * for 2.2% of the inputs x / (x / 2) (tmp/p3-fix/div-probe.mjs), which took the branch the CPU reference never takes.
13
+ */
14
+ export declare const fa2SpeedFinalizeWgsl = "\n@compute @workgroup_size(WG)\nfn speed_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n let groups = (P.n + WG - 1u) / WG;\n var st = vec2f(0.0);\n for (var g = lid.x; g < groups; g = g + WG) { st = st + partials[g].swingTraction; } // sequential per lane: deterministic\n let t = wg_reduce_vec4(vec4f(st, 0.0, 0.0), lid.x, 0u);\n if (lid.x == 0u) {\n var swing = t.x;\n var traction = t.y;\n if (SWING_MODE == 1u) { swing = S.swing + t.x; traction = S.traction + t.y; } // NetworkX accumulates across iterations from 1\n let n = f32(P.n);\n let optJitter = 0.05 * sqrt(n);\n let minJitter = sqrt(optJitter);\n let maxJitter = 10.0;\n let tr = max(traction, 1.0e-30); // guards the division only (7.10)\n let other = min(maxJitter, optJitter * traction / (n * n));\n var jitter = P.jitterTolerance * max(minJitter, other);\n var eff = S.speedEfficiency;\n if (swing > 2.0 * tr) { // swing / traction > 2 in the exact form (contract 4.5 CONTRACT DECISION K4-1: 2 x is exact, a WGSL f32 division is not)\n if (eff > 0.05) { eff = eff * 0.5; } // the CPU's conditional multiply (7.2)\n jitter = max(jitter, P.jitterTolerance);\n }\n let targetSpeed = select(jitter * eff * traction / swing, 1.0e30, swing == 0.0); // +Inf in the port; 1e30 gives the same min() below (`target` is reserved)\n if (swing > jitter * traction) {\n if (eff > 0.05) { eff = eff * 0.7; }\n } else if (S.speed < 1000.0) {\n eff = eff * 1.3;\n }\n S.speed = S.speed + min(targetSpeed - S.speed, 0.5 * S.speed);\n S.speedEfficiency = eff;\n S.swing = swing;\n S.traction = traction;\n T[P.iterationIndex].swing = swing;\n T[P.iterationIndex].traction = traction;\n T[P.iterationIndex].speed = S.speed;\n T[P.iterationIndex].speedEfficiency = eff;\n }\n}\n";
15
+ //# sourceMappingURL=fa2-speed-finalize.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"fa2-speed-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-speed-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AACH,eAAO,MAAM,oBAAoB,qkEAuChC,CAAC"}
@@ -0,0 +1,54 @@
1
+ /**
2
+ * The `fa2-speed-finalize` kernel body (K4; contract 4.5; spec 7.10): ONE workgroup sums `partials[*].swingTraction`
3
+ * sequentially per lane (deterministic), reduces in uniform control flow, then lane 0 runs the line-for-line port
4
+ * of the CPU `estimateFactor` (SWING_MODE 1 accumulates NetworkX's sums across iterations from S.swing / S.traction)
5
+ * and writes `S.speed`, `S.speedEfficiency`, `S.swing`, `S.traction` and the four controller fields of
6
+ * `T[P.iterationIndex]`. `target` is a WGSL reserved word, hence `targetSpeed`. Normative text, copied verbatim:
7
+ * the P1-T5 sabotage rows (`* 0.5` -> `* 0.9`, the 1.3 rise, the swing / traction swap) are textual edits of it.
8
+ * The halving predicate is `swing > 2.0 * tr`, the exact form of the port's `swing / traction > 2` (contract 4.5
9
+ * CONTRACT DECISION K4-1, G3 finding G3-F6): every paper-mode first iteration after load() has oldForce = 0, so
10
+ * traction is EXACTLY half the swing and the predicate sits on its knife edge; a multiplication by 2 and the
11
+ * comparison are exact on every device, while WGSL grants f32 division 2.5 ULP and Dawn on NVIDIA returns 2 + 1 ulp
12
+ * for 2.2% of the inputs x / (x / 2) (tmp/p3-fix/div-probe.mjs), which took the branch the CPU reference never takes.
13
+ */
14
+ export const fa2SpeedFinalizeWgsl = /* wgsl */ `
15
+ @compute @workgroup_size(WG)
16
+ fn speed_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
17
+ let groups = (P.n + WG - 1u) / WG;
18
+ var st = vec2f(0.0);
19
+ for (var g = lid.x; g < groups; g = g + WG) { st = st + partials[g].swingTraction; } // sequential per lane: deterministic
20
+ let t = wg_reduce_vec4(vec4f(st, 0.0, 0.0), lid.x, 0u);
21
+ if (lid.x == 0u) {
22
+ var swing = t.x;
23
+ var traction = t.y;
24
+ if (SWING_MODE == 1u) { swing = S.swing + t.x; traction = S.traction + t.y; } // NetworkX accumulates across iterations from 1
25
+ let n = f32(P.n);
26
+ let optJitter = 0.05 * sqrt(n);
27
+ let minJitter = sqrt(optJitter);
28
+ let maxJitter = 10.0;
29
+ let tr = max(traction, 1.0e-30); // guards the division only (7.10)
30
+ let other = min(maxJitter, optJitter * traction / (n * n));
31
+ var jitter = P.jitterTolerance * max(minJitter, other);
32
+ var eff = S.speedEfficiency;
33
+ if (swing > 2.0 * tr) { // swing / traction > 2 in the exact form (contract 4.5 CONTRACT DECISION K4-1: 2 x is exact, a WGSL f32 division is not)
34
+ if (eff > 0.05) { eff = eff * 0.5; } // the CPU's conditional multiply (7.2)
35
+ jitter = max(jitter, P.jitterTolerance);
36
+ }
37
+ let targetSpeed = select(jitter * eff * traction / swing, 1.0e30, swing == 0.0); // +Inf in the port; 1e30 gives the same min() below (\`target\` is reserved)
38
+ if (swing > jitter * traction) {
39
+ if (eff > 0.05) { eff = eff * 0.7; }
40
+ } else if (S.speed < 1000.0) {
41
+ eff = eff * 1.3;
42
+ }
43
+ S.speed = S.speed + min(targetSpeed - S.speed, 0.5 * S.speed);
44
+ S.speedEfficiency = eff;
45
+ S.swing = swing;
46
+ S.traction = traction;
47
+ T[P.iterationIndex].swing = swing;
48
+ T[P.iterationIndex].traction = traction;
49
+ T[P.iterationIndex].speed = S.speed;
50
+ T[P.iterationIndex].speedEfficiency = eff;
51
+ }
52
+ }
53
+ `;
54
+ //# sourceMappingURL=fa2-speed-finalize.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"fa2-speed-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-speed-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAuC9C,CAAC"}
@@ -0,0 +1,14 @@
1
+ /**
2
+ * K1 of the ForceAtlas2 iteration, `fa2-stats-finalize` (spec 7.4; contract 4.5): one workgroup folds the partials
3
+ * the previous integrate (K5) wrote into the state block -- centroid, RMS radius, layout radius (max |p - c|, the
4
+ * exact spec 3.3 value from `partials.max.w`), bounding box, mean displacement over free nodes (0 when every node
5
+ * is fixed), the settle counter -- increments the iteration counter and writes the K1 half of the trace record.
6
+ * On the first iteration after load() (`FA2_FLAG_FIRST`) it folds nothing and keeps the host-written state.
7
+ *
8
+ * This file holds the kernel BODY only (spec 3.5, D9): no bind-group lines and no `override` lines -- the composer
9
+ * emits them from the registry entry in src/kernels.ts (contract 3.10.1). The text is normative (contract 4.5) and
10
+ * is the target of the K1 sabotage mutations (test/helpers/sabotage.ts, P3-T5); amend the contract before editing.
11
+ */
12
+ /** The K1 body: entry point `stats_finalize`; calls the reduction helpers (`needs: ["subgroups"]`, contract 4.3). */
13
+ export declare const fa2StatsFinalizeWgsl = "// K1: folds the previous integrate's partials into the state block (spec 7.4); one workgroup\n@compute @workgroup_size(WG)\nfn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n let groups = (P.n + WG - 1u) / WG;\n let fold = (P.flags & FA2_FLAG_FIRST) == 0u; // the first iteration after load() keeps the host-written state\n var sum = vec4f(0.0);\n var lo = vec4f(F32_MAX);\n var hi = vec4f(-F32_MAX);\n var disp = 0.0;\n var free = 0u;\n if (fold) {\n for (var g = lid.x; g < groups; g = g + WG) { // sequential per lane in index order: deterministic\n let q = partials[g];\n sum = sum + q.sum;\n lo = min(lo, q.min);\n hi = max(hi, q.max);\n disp = disp + q.dispFree.x;\n free = free + u32(q.dispFree.y);\n }\n }\n let tSum = wg_reduce_vec4(sum, lid.x, 0u);\n let tLo = wg_reduce_vec4(lo, lid.x, 1u);\n let tHi = wg_reduce_vec4(hi, lid.x, 2u);\n let tDisp = wg_reduce_f32(disp, lid.x, 0u);\n let tFree = wg_reduce_u32(free, lid.x, 0u);\n if (lid.x == 0u) {\n if (fold) {\n let n = f32(P.n);\n let c = tSum.xyz / n;\n S.centroid = vec4f(c, 0.0);\n S.rmsRadius = sqrt(max(tSum.w, 0.0) / n); // RMS radius about the previous centroid (7.17)\n S.min = vec4f(tLo.xyz, 0.0);\n S.max = vec4f(tHi.xyz, 0.0);\n S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)\n let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)\n S.meanDisplacement = meanDisp;\n S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);\n }\n S.iteration = S.iteration + 1u;\n T[P.iterationIndex].meanDisplacement = S.meanDisplacement;\n T[P.iterationIndex].settledCount = S.settledCount;\n T[P.iterationIndex].iteration = S.iteration;\n }\n}";
14
+ //# sourceMappingURL=fa2-stats-finalize.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"fa2-stats-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AAEH,qHAAqH;AACrH,eAAO,MAAM,oBAAoB,gkEA2C/B,CAAC"}
@@ -0,0 +1,57 @@
1
+ /**
2
+ * K1 of the ForceAtlas2 iteration, `fa2-stats-finalize` (spec 7.4; contract 4.5): one workgroup folds the partials
3
+ * the previous integrate (K5) wrote into the state block -- centroid, RMS radius, layout radius (max |p - c|, the
4
+ * exact spec 3.3 value from `partials.max.w`), bounding box, mean displacement over free nodes (0 when every node
5
+ * is fixed), the settle counter -- increments the iteration counter and writes the K1 half of the trace record.
6
+ * On the first iteration after load() (`FA2_FLAG_FIRST`) it folds nothing and keeps the host-written state.
7
+ *
8
+ * This file holds the kernel BODY only (spec 3.5, D9): no bind-group lines and no `override` lines -- the composer
9
+ * emits them from the registry entry in src/kernels.ts (contract 3.10.1). The text is normative (contract 4.5) and
10
+ * is the target of the K1 sabotage mutations (test/helpers/sabotage.ts, P3-T5); amend the contract before editing.
11
+ */
12
+ /** The K1 body: entry point `stats_finalize`; calls the reduction helpers (`needs: ["subgroups"]`, contract 4.3). */
13
+ export const fa2StatsFinalizeWgsl = /* wgsl */ `// K1: folds the previous integrate's partials into the state block (spec 7.4); one workgroup
14
+ @compute @workgroup_size(WG)
15
+ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
16
+ let groups = (P.n + WG - 1u) / WG;
17
+ let fold = (P.flags & FA2_FLAG_FIRST) == 0u; // the first iteration after load() keeps the host-written state
18
+ var sum = vec4f(0.0);
19
+ var lo = vec4f(F32_MAX);
20
+ var hi = vec4f(-F32_MAX);
21
+ var disp = 0.0;
22
+ var free = 0u;
23
+ if (fold) {
24
+ for (var g = lid.x; g < groups; g = g + WG) { // sequential per lane in index order: deterministic
25
+ let q = partials[g];
26
+ sum = sum + q.sum;
27
+ lo = min(lo, q.min);
28
+ hi = max(hi, q.max);
29
+ disp = disp + q.dispFree.x;
30
+ free = free + u32(q.dispFree.y);
31
+ }
32
+ }
33
+ let tSum = wg_reduce_vec4(sum, lid.x, 0u);
34
+ let tLo = wg_reduce_vec4(lo, lid.x, 1u);
35
+ let tHi = wg_reduce_vec4(hi, lid.x, 2u);
36
+ let tDisp = wg_reduce_f32(disp, lid.x, 0u);
37
+ let tFree = wg_reduce_u32(free, lid.x, 0u);
38
+ if (lid.x == 0u) {
39
+ if (fold) {
40
+ let n = f32(P.n);
41
+ let c = tSum.xyz / n;
42
+ S.centroid = vec4f(c, 0.0);
43
+ S.rmsRadius = sqrt(max(tSum.w, 0.0) / n); // RMS radius about the previous centroid (7.17)
44
+ S.min = vec4f(tLo.xyz, 0.0);
45
+ S.max = vec4f(tHi.xyz, 0.0);
46
+ S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)
47
+ let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)
48
+ S.meanDisplacement = meanDisp;
49
+ S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);
50
+ }
51
+ S.iteration = S.iteration + 1u;
52
+ T[P.iterationIndex].meanDisplacement = S.meanDisplacement;
53
+ T[P.iterationIndex].settledCount = S.settledCount;
54
+ T[P.iterationIndex].iteration = S.iteration;
55
+ }
56
+ }`;
57
+ //# sourceMappingURL=fa2-stats-finalize.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"fa2-stats-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AAEH,qHAAqH;AACrH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EA2C7C,CAAC"}
@@ -0,0 +1,11 @@
1
+ /**
2
+ * The per-batch `fa2-to-scene` kernel (spec 7.13, 7.18; contract 4.5): unpacks the vec4f layout-unit positions into
3
+ * the stride-3 scene array the staging copy reads back, applying `scale` and `center`; in 2D the third component is
4
+ * `center.z` on every readback whatever z the device holds.
5
+ *
6
+ * Body only (spec 3.5, D9); normative text (contract 4.5); exempt from the sabotage matrix (SABOTAGE_EXEMPT: a wrong
7
+ * toScene fails the exact-equality position tests directly).
8
+ */
9
+ /** The toScene body: entry point `to_scene`; an early return is legal here because no barrier follows (spec 3.5 rule 1). */
10
+ export declare const fa2ToSceneWgsl = "@compute @workgroup_size(WG)\nfn to_scene(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n if (i >= P.n) { return; }\n let s = pos[i].xyz * P.scale + P.center.xyz;\n scene[3u * i] = s.x;\n scene[3u * i + 1u] = s.y;\n scene[3u * i + 2u] = select(s.z, P.center.z, P.dim == 2u); // 2D writes z = center.z on every readback (7.13)\n}";
11
+ //# sourceMappingURL=fa2-to-scene.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"fa2-to-scene.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-to-scene.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAEH,4HAA4H;AAC5H,eAAO,MAAM,cAAc,+aAQzB,CAAC"}
@@ -0,0 +1,19 @@
1
+ /**
2
+ * The per-batch `fa2-to-scene` kernel (spec 7.13, 7.18; contract 4.5): unpacks the vec4f layout-unit positions into
3
+ * the stride-3 scene array the staging copy reads back, applying `scale` and `center`; in 2D the third component is
4
+ * `center.z` on every readback whatever z the device holds.
5
+ *
6
+ * Body only (spec 3.5, D9); normative text (contract 4.5); exempt from the sabotage matrix (SABOTAGE_EXEMPT: a wrong
7
+ * toScene fails the exact-equality position tests directly).
8
+ */
9
+ /** The toScene body: entry point `to_scene`; an early return is legal here because no barrier follows (spec 3.5 rule 1). */
10
+ export const fa2ToSceneWgsl = /* wgsl */ `@compute @workgroup_size(WG)
11
+ fn to_scene(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
12
+ let i = linear_id(wid, lid.x);
13
+ if (i >= P.n) { return; }
14
+ let s = pos[i].xyz * P.scale + P.center.xyz;
15
+ scene[3u * i] = s.x;
16
+ scene[3u * i + 1u] = s.y;
17
+ scene[3u * i + 2u] = select(s.z, P.center.z, P.dim == 2u); // 2D writes z = center.z on every readback (7.13)
18
+ }`;
19
+ //# sourceMappingURL=fa2-to-scene.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"fa2-to-scene.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-to-scene.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAEH,4HAA4H;AAC5H,MAAM,CAAC,MAAM,cAAc,GAAG,UAAU,CAAC;;;;;;;;EAQvC,CAAC"}
@@ -0,0 +1,7 @@
1
+ /**
2
+ * The `fill` kernel body (contract 4.5): `dst[i] = P.value` (mode 0) or `i + P.value` (mode 1, iota) for every
3
+ * `i < P.count`, `i` from the prelude's `linear_id` so a 2D dispatch covers counts above MAX_1D_ITEMS. The 17M-item
4
+ * linear_id test of spec 5.2 / 11.5 is mode 1 over 16,776,961 words. Normative text, copied verbatim.
5
+ */
6
+ export declare const fillWgsl = "\n@compute @workgroup_size(WG)\nfn fill(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n if (i >= P.count) { return; }\n dst[i] = select(P.value, i + P.value, P.mode == 1u);\n}\n";
7
+ //# sourceMappingURL=fill.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"fill.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fill.wgsl.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH,eAAO,MAAM,QAAQ,yQAOpB,CAAC"}
@@ -0,0 +1,14 @@
1
+ /**
2
+ * The `fill` kernel body (contract 4.5): `dst[i] = P.value` (mode 0) or `i + P.value` (mode 1, iota) for every
3
+ * `i < P.count`, `i` from the prelude's `linear_id` so a 2D dispatch covers counts above MAX_1D_ITEMS. The 17M-item
4
+ * linear_id test of spec 5.2 / 11.5 is mode 1 over 16,776,961 words. Normative text, copied verbatim.
5
+ */
6
+ export const fillWgsl = /* wgsl */ `
7
+ @compute @workgroup_size(WG)
8
+ fn fill(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
9
+ let i = linear_id(wid, lid.x);
10
+ if (i >= P.count) { return; }
11
+ dst[i] = select(P.value, i + P.value, P.mode == 1u);
12
+ }
13
+ `;
14
+ //# sourceMappingURL=fill.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"fill.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fill.wgsl.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH,MAAM,CAAC,MAAM,QAAQ,GAAG,UAAU,CAAC;;;;;;;CAOlC,CAAC"}
@@ -0,0 +1,10 @@
1
+ /**
2
+ * The `reduce` kernel body (contract 4.5; spec 6 row 1): one level of the 2-3 level reduction. Overrides: OP
3
+ * 0 sum / 1 min / 2 max, DTYPE 0 f32 / 1 u32 / 2 vec4f (four words per element, loaded through bitcast from the
4
+ * u32 storage view), FINAL true for the one-workgroup level that folds the partials sequentially per lane in
5
+ * index order (deterministic) and writes one element at `out[P.outOffset]`; the level-1 form writes
6
+ * `out[P.outOffset + group]`. Calls the prelude's `wg_reduce_*` helpers (needs: ["subgroups"]) in uniform
7
+ * control flow: DTYPE is a pipeline constant, so the branch on it is uniform. Normative text, copied verbatim.
8
+ */
9
+ export declare const reduceWgsl = "\n// DTYPE 0 = f32, 1 = u32, 2 = vec4f (4 words per element); OP 0 = sum, 1 = min, 2 = max; FINAL = the one-workgroup level\nfn identity_f() -> f32 { if (OP == 1u) { return F32_MAX; } if (OP == 2u) { return -F32_MAX; } return 0.0; }\nfn identity_u() -> u32 { if (OP == 1u) { return U32_MAX; } return 0u; } // U32_MAX from the prelude: the literal is forbidden in bodies (4.1)\nfn comb_f(a: f32, b: f32) -> f32 { if (OP == 1u) { return min(a, b); } if (OP == 2u) { return max(a, b); } return a + b; }\nfn comb_u(a: u32, b: u32) -> u32 { if (OP == 1u) { return min(a, b); } if (OP == 2u) { return max(a, b); } return a + b; }\nfn comb_v(a: vec4f, b: vec4f) -> vec4f { if (OP == 1u) { return min(a, b); } if (OP == 2u) { return max(a, b); } return a + b; }\nfn load_f(i: u32) -> f32 { return bitcast<f32>(src[i]); }\nfn load_v(i: u32) -> vec4f {\n return vec4f(bitcast<f32>(src[4u * i]), bitcast<f32>(src[4u * i + 1u]), bitcast<f32>(src[4u * i + 2u]), bitcast<f32>(src[4u * i + 3u]));\n}\nfn store_f(i: u32, v: f32) { out[i] = bitcast<u32>(v); }\nfn store_v(i: u32, v: vec4f) {\n out[4u * i] = bitcast<u32>(v.x);\n out[4u * i + 1u] = bitcast<u32>(v.y);\n out[4u * i + 2u] = bitcast<u32>(v.z);\n out[4u * i + 3u] = bitcast<u32>(v.w);\n}\n\n@compute @workgroup_size(WG)\nfn reduce(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n var accF = identity_f();\n var accU = identity_u();\n var accV = vec4f(identity_f());\n if (FINAL) {\n for (var i = lid.x; i < P.count; i = i + WG) { // sequential per lane in index order: deterministic\n if (DTYPE == 0u) { accF = comb_f(accF, load_f(i)); }\n else if (DTYPE == 1u) { accU = comb_u(accU, src[i]); }\n else { accV = comb_v(accV, load_v(i)); }\n }\n } else {\n let i = linear_id(wid, lid.x);\n if (i < P.count) {\n if (DTYPE == 0u) { accF = load_f(i); }\n else if (DTYPE == 1u) { accU = src[i]; }\n else { accV = load_v(i); }\n }\n }\n // uniform control flow: the workgroup reduction of the selected dtype (DTYPE is a pipeline constant, so the branch is uniform)\n var tF = 0.0;\n var tU = 0u;\n var tV = vec4f(0.0);\n if (DTYPE == 0u) { tF = wg_reduce_f32(accF, lid.x, OP); }\n else if (DTYPE == 1u) { tU = wg_reduce_u32(accU, lid.x, OP); }\n else { tV = wg_reduce_vec4(accV, lid.x, OP); }\n if (lid.x == 0u) {\n let g = select(group_id(wid), 0u, FINAL);\n let o = P.outOffset + g;\n if (DTYPE == 0u) { store_f(o, tF); }\n else if (DTYPE == 1u) { out[o] = tU; }\n else { store_v(o, tV); }\n }\n}\n";
10
+ //# sourceMappingURL=reduce.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"reduce.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/reduce.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AACH,eAAO,MAAM,UAAU,woFAqDtB,CAAC"}