@graphty/webgpu-graph-algorithms 0.6.5 → 0.6.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/README.md +38 -17
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-hzGggHeM.js → context-Cezi7qpi.js} +46 -22
  4. package/dist/chunks/context-Cezi7qpi.js.map +1 -0
  5. package/dist/node.js +19 -11
  6. package/dist/node.js.map +1 -1
  7. package/dist/src/algorithms/bfs.d.ts +10 -6
  8. package/dist/src/algorithms/bfs.d.ts.map +1 -1
  9. package/dist/src/algorithms/bfs.js +32 -9
  10. package/dist/src/algorithms/bfs.js.map +1 -1
  11. package/dist/src/algorithms/pagerank.d.ts.map +1 -1
  12. package/dist/src/algorithms/pagerank.js +19 -5
  13. package/dist/src/algorithms/pagerank.js.map +1 -1
  14. package/dist/src/algorithms/power-iteration.d.ts.map +1 -1
  15. package/dist/src/algorithms/power-iteration.js +8 -2
  16. package/dist/src/algorithms/power-iteration.js.map +1 -1
  17. package/dist/src/constants.d.ts +33 -0
  18. package/dist/src/constants.d.ts.map +1 -1
  19. package/dist/src/constants.js +33 -0
  20. package/dist/src/constants.js.map +1 -1
  21. package/dist/src/kernel/dispatch.d.ts +2 -2
  22. package/dist/src/kernel/dispatch.d.ts.map +1 -1
  23. package/dist/src/kernel/kernel.d.ts +1 -1
  24. package/dist/src/kernel/kernel.js +2 -2
  25. package/dist/src/kernel/kernel.js.map +1 -1
  26. package/dist/src/kernel/prelude.d.ts.map +1 -1
  27. package/dist/src/kernel/prelude.js +2 -1
  28. package/dist/src/kernel/prelude.js.map +1 -1
  29. package/dist/src/kernels.d.ts +10 -7
  30. package/dist/src/kernels.d.ts.map +1 -1
  31. package/dist/src/kernels.js +33 -9
  32. package/dist/src/kernels.js.map +1 -1
  33. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  34. package/dist/src/layouts/forceatlas2.js +2 -1
  35. package/dist/src/layouts/forceatlas2.js.map +1 -1
  36. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  37. package/dist/src/layouts/fruchterman-reingold.js +4 -2
  38. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  39. package/dist/src/layouts/repulsion-exact.d.ts +16 -0
  40. package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
  41. package/dist/src/layouts/repulsion-exact.js +21 -1
  42. package/dist/src/layouts/repulsion-exact.js.map +1 -1
  43. package/dist/src/layouts/repulsion-grid.d.ts +1 -1
  44. package/dist/src/layouts/repulsion-grid.js +1 -1
  45. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  46. package/dist/src/layouts/spring-electrical.js +6 -2
  47. package/dist/src/layouts/spring-electrical.js.map +1 -1
  48. package/dist/src/memory/residency.js +14 -4
  49. package/dist/src/memory/residency.js.map +1 -1
  50. package/dist/src/node/index.d.ts +13 -8
  51. package/dist/src/node/index.d.ts.map +1 -1
  52. package/dist/src/node/index.js +36 -17
  53. package/dist/src/node/index.js.map +1 -1
  54. package/dist/src/primitives/frontier.d.ts +1 -0
  55. package/dist/src/primitives/frontier.d.ts.map +1 -1
  56. package/dist/src/primitives/frontier.js +1 -0
  57. package/dist/src/primitives/frontier.js.map +1 -1
  58. package/dist/src/primitives/grid-pyramid.d.ts +4 -4
  59. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  60. package/dist/src/primitives/grid-pyramid.js +4 -3
  61. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  62. package/dist/src/primitives/grid.d.ts +13 -10
  63. package/dist/src/primitives/grid.d.ts.map +1 -1
  64. package/dist/src/primitives/grid.js +10 -7
  65. package/dist/src/primitives/grid.js.map +1 -1
  66. package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
  67. package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
  68. package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
  69. package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
  70. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
  71. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
  72. package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
  73. package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
  74. package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
  75. package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
  76. package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
  77. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
  78. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
  79. package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
  80. package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
  81. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
  82. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
  83. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
  84. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
  85. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
  86. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
  87. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
  88. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
  89. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +21 -20
  90. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
  91. package/dist/src/wgsl/frontier-finalize.wgsl.js +26 -25
  92. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
  93. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
  94. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
  95. package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
  96. package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
  97. package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
  98. package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
  99. package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
  100. package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
  101. package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
  102. package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
  103. package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
  104. package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
  105. package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
  106. package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
  107. package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
  108. package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
  109. package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
  110. package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
  111. package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
  112. package/dist/src/wgsl/histogram.wgsl.js +1 -1
  113. package/dist/webgpu-graph-algorithms.js +144 -45
  114. package/dist/webgpu-graph-algorithms.js.map +1 -1
  115. package/package.json +3 -3
  116. package/src/algorithms/bfs.ts +33 -9
  117. package/src/algorithms/pagerank.ts +19 -5
  118. package/src/algorithms/power-iteration.ts +8 -2
  119. package/src/constants.ts +35 -0
  120. package/src/kernel/dispatch.ts +2 -2
  121. package/src/kernel/kernel.ts +2 -2
  122. package/src/kernel/prelude.ts +2 -0
  123. package/src/kernels.ts +35 -9
  124. package/src/layouts/forceatlas2.ts +2 -0
  125. package/src/layouts/fruchterman-reingold.ts +4 -1
  126. package/src/layouts/repulsion-exact.ts +29 -1
  127. package/src/layouts/repulsion-grid.ts +1 -1
  128. package/src/layouts/spring-electrical.ts +8 -1
  129. package/src/memory/residency.ts +14 -4
  130. package/src/node/index.ts +42 -18
  131. package/src/primitives/frontier.ts +2 -0
  132. package/src/primitives/grid-pyramid.ts +6 -5
  133. package/src/primitives/grid.ts +17 -12
  134. package/src/wgsl/advance-expand.wgsl.ts +4 -3
  135. package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
  136. package/src/wgsl/bfs-fused.wgsl.ts +6 -6
  137. package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
  138. package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
  139. package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
  140. package/src/wgsl/frontier-finalize.wgsl.ts +26 -25
  141. package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
  142. package/src/wgsl/grid-centroid.wgsl.ts +3 -2
  143. package/src/wgsl/grid-downsample.wgsl.ts +1 -1
  144. package/src/wgsl/grid-far-field.wgsl.ts +15 -8
  145. package/src/wgsl/grid-near-field.wgsl.ts +6 -2
  146. package/src/wgsl/histogram.wgsl.ts +1 -1
  147. package/dist/chunks/context-hzGggHeM.js.map +0 -1
@@ -9,6 +9,9 @@
9
9
  * `partials[group].swingTraction`. `pos` is `array<vec4f>` with the mass in `.w`; `force` / `oldForce` are stride-3
10
10
  * `array<f32>` read through the per-body helpers (4.4 rule 6). Normative text, copied verbatim: the P1-T5 sabotage
11
11
  * rows (gravity sign, `k / d2`, the `jj != i` guard, the `.w` mass lane) are textual edits of this string.
12
+ * The j range is split into passes of at most EXACT_TILES_PER_PASS tiles (issue #87: llvmpipe's per-invocation loop
13
+ * budget); an earlier pass adds its partial sum into `force`, the last one runs gravity and the epilogue. A graph of
14
+ * at most 32,768 nodes is one pass, bitwise the single-pass kernel. Record it through recordExactRepulsion.
12
15
  */
13
16
  export const fa2RepulsionExactWgsl = /* wgsl */ `
14
17
  var<workgroup> tile: array<vec4f, WG>; // xyz + mass, 4 KiB at WG = 256
@@ -35,14 +38,23 @@ fn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: cent
35
38
  }
36
39
 
37
40
  @compute @workgroup_size(WG)
38
- fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
41
+ fn repulsion(
42
+ @builtin(workgroup_id) wid: vec3<u32>,
43
+ @builtin(local_invocation_id) lid: vec3<u32>,
44
+ @builtin(num_workgroups) nwg: vec3<u32>,
45
+ ) {
46
+ // issue #87: pass p of the tile range is dispatched with p + 1 z slices; only the last slice works, so the pass
47
+ // index needs no uniform. Uniform: keyed on workgroup_id and num_workgroups only.
48
+ if (wid.z + 1u < nwg.z) { return; }
39
49
  let i = linear_id(wid, lid.x);
40
50
  let valid = i < P.n;
41
51
  var pi = vec4f(0.0);
42
52
  if (valid) { pi = pos[i]; }
43
53
  var f = vec3f(0.0);
44
54
  let tiles = (P.n + WG - 1u) / WG;
45
- for (var t = 0u; t < tiles; t = t + 1u) {
55
+ let tileBegin = (nwg.z - 1u) * EXACT_TILES_PER_PASS;
56
+ let tileEnd = min(tiles, tileBegin + EXACT_TILES_PER_PASS);
57
+ for (var t = tileBegin; t < tileEnd; t = t + 1u) {
46
58
  let j = t * WG + lid.x;
47
59
  if (j < P.n) { tile[lid.x] = pos[j]; } else { tile[lid.x] = vec4f(0.0); } // guarded fill; mass 0 marks the pad
48
60
  workgroupBarrier(); // uniform: every invocation reaches it
@@ -65,6 +77,10 @@ fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
65
77
  }
66
78
  workgroupBarrier();
67
79
  }
80
+ if (tileEnd < tiles) { // an earlier pass: its partial sum only (uniform: P.n, nwg)
81
+ if (valid) { store_force(i, load_force(i) + f); }
82
+ return;
83
+ }
68
84
  // epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it
69
85
  var sw = 0.0;
70
86
  var tr = 0.0;
@@ -1 +1 @@
1
- {"version":3,"file":"fa2-repulsion-exact.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-repulsion-exact.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;GAWG;AACH,MAAM,CAAC,MAAM,qBAAqB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA0E/C,CAAC"}
1
+ {"version":3,"file":"fa2-repulsion-exact.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-repulsion-exact.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AACH,MAAM,CAAC,MAAM,qBAAqB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAuF/C,CAAC"}
@@ -21,5 +21,5 @@
21
21
  * is the target of the K1 sabotage mutations (test/helpers/sabotage.ts, P3-T5); amend the contract before editing.
22
22
  */
23
23
  /** The K1 body: entry point `stats_finalize`; calls the reduction helpers (`needs: ["subgroups"]`, contract 4.3). */
24
- export declare const fa2StatsFinalizeWgsl = "// K1: folds the previous integrate's partials into the state block (spec 7.4); one workgroup\n@compute @workgroup_size(WG)\nfn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n let groups = (P.n + WG - 1u) / WG;\n let fold = (P.flags & FA2_FLAG_FIRST) == 0u; // the first iteration after load() keeps the host-written state\n var sum = vec4f(0.0);\n var lo = vec4f(F32_MAX);\n var hi = vec4f(-F32_MAX);\n var disp = 0.0;\n var free = 0u;\n var ke = 0.0;\n if (fold) {\n for (var g = lid.x; g < groups; g = g + WG) { // sequential per lane in index order: deterministic\n let q = partials[g];\n sum = sum + q.sum;\n lo = min(lo, q.min);\n hi = max(hi, q.max);\n disp = disp + q.dispFree.x;\n free = free + u32(q.dispFree.y);\n ke = ke + q.swingTraction.x;\n }\n }\n let tSum = wg_reduce_vec4(sum, lid.x, 0u);\n let tLo = wg_reduce_vec4(lo, lid.x, 1u);\n let tHi = wg_reduce_vec4(hi, lid.x, 2u);\n let tDisp = wg_reduce_f32(disp, lid.x, 0u);\n let tFree = wg_reduce_u32(free, lid.x, 0u);\n let tKe = wg_reduce_f32(ke, lid.x, 0u);\n if (lid.x == 0u) {\n if (fold) {\n let n = f32(P.n);\n let c = tSum.xyz / n;\n S.centroid = vec4f(c, 0.0);\n S.rmsRadius = sqrt(max(tSum.w, 0.0) / n); // RMS radius about the previous centroid (7.17)\n S.min = vec4f(tLo.xyz, 0.0);\n S.max = vec4f(tHi.xyz, 0.0);\n S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)\n let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)\n S.meanDisplacement = meanDisp;\n S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);\n }\n S.iteration = S.iteration + 1u;\n T[P.iterationIndex].meanDisplacement = S.meanDisplacement;\n T[P.iterationIndex].settledCount = S.settledCount;\n T[P.iterationIndex].iteration = S.iteration;\n if (P.gridMax > 0u) { // the grid tier (7.7): the robust extent, the cell size, eps, last iteration's counts, the hub counter reset (PD-14)\n let cells = P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u);\n if (fold) {\n let box = (S.max.xyz - S.min.xyz) * GRID_BBOX_MARGIN;\n var bboxExtent = max(box.x, box.y);\n if (P.dim == 3u) { bboxExtent = max(bboxExtent, box.z); }\n let extent = max(min(bboxExtent, P.extentFactor * S.rmsRadius), GRID_EXTENT_FLOOR); // min(bbox, extentFactor x rms), floored (7.7)\n let cellSize = extent / f32(P.gridMax);\n S.gridMin = vec4f(S.centroid.xyz - vec3f(0.5 * extent), cellSize); // gridMin.w carries cellSize\n S.invCellSize = 1.0 / cellSize;\n S.eps = 0.25 * cellSize;\n }\n S.outsideGrid = cellHist[cells]; // the previous iteration's pseudo-cell count (0 after load)\n S.maxCellOccupancy = atomicLoad(&hubCounters[1]);\n atomicStore(&hubCounters[0], 0u);\n atomicStore(&hubCounters[1], 0u);\n }\n if (STATS_MODE == 1u) { // FR: this iteration's temperature (7.20) into the state and the trace\n if ((P.flags & FA2_FLAG_ADAPTIVE) != 0u) { // adaptive cooling (Yifan Hu 2005 3.2): tKe is the previous iteration's sum |F|^2 over free nodes\n if (fold) {\n var t = S.temperature;\n if (tKe < S.frEnergy) {\n S.frProgress = S.frProgress + 1u;\n if (S.frProgress >= FR_COOLING_PATIENCE) { S.frProgress = 0u; t = t / FR_COOLING_STEP; }\n } else {\n S.frProgress = 0u;\n t = t * FR_COOLING_STEP;\n }\n S.frEnergy = tKe;\n S.temperature = t;\n }\n } else {\n S.temperature = P.temperature;\n }\n T[P.iterationIndex].modelScalar = S.temperature;\n }\n if (STATS_MODE == 2u) { // spring-electrical: the kinetic energy K5 folded into partials B (PD-4); 0 on the first iteration after load()\n S.kineticEnergy = tKe;\n T[P.iterationIndex].modelScalar = tKe;\n }\n }\n}";
24
+ export declare const fa2StatsFinalizeWgsl = "// K1: folds the previous integrate's partials into the state block (spec 7.4); one workgroup\n@compute @workgroup_size(WG)\nfn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n let groups = (P.n + WG - 1u) / WG;\n let fold = (P.flags & FA2_FLAG_FIRST) == 0u; // the first iteration after load() keeps the host-written state\n var sum = vec4f(0.0);\n var lo = vec4f(F32_MAX);\n var hi = vec4f(-F32_MAX);\n var disp = 0.0;\n var free = 0u;\n var ke = 0.0;\n if (fold) {\n for (var g = lid.x; g < groups; g = g + WG) { // sequential per lane in index order: deterministic\n let q = partials[g];\n sum = sum + q.sum;\n lo = min(lo, q.min);\n hi = max(hi, q.max);\n disp = disp + q.dispFree.x;\n free = free + u32(q.dispFree.y);\n ke = ke + q.swingTraction.x;\n }\n }\n let tSum = wg_reduce_vec4(sum, lid.x, 0u);\n let tLo = wg_reduce_vec4(lo, lid.x, 1u);\n let tHi = wg_reduce_vec4(hi, lid.x, 2u);\n let tDisp = wg_reduce_f32(disp, lid.x, 0u);\n let tFree = wg_reduce_u32(free, lid.x, 0u);\n let tKe = wg_reduce_f32(ke, lid.x, 0u);\n if (lid.x == 0u) {\n if (fold) {\n let n = f32(P.n);\n let c = tSum.xyz / n;\n S.centroid = vec4f(c, 0.0);\n S.rmsRadius = sqrt(max(tSum.w, 0.0) / n); // RMS radius about the previous centroid (7.17)\n S.min = vec4f(tLo.xyz, 0.0);\n S.max = vec4f(tHi.xyz, 0.0);\n S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)\n let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)\n S.meanDisplacement = meanDisp;\n S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= min(P.settleThreshold * S.rmsRadius, P.settleFloor)); // relative AND absolute (issue #97)\n }\n S.iteration = S.iteration + 1u;\n T[P.iterationIndex].meanDisplacement = S.meanDisplacement;\n T[P.iterationIndex].settledCount = S.settledCount;\n T[P.iterationIndex].iteration = S.iteration;\n if (P.gridMax > 0u) { // the grid tier (7.7): the robust extent, the cell size, eps, last iteration's counts, the hub counter reset (PD-14)\n let cells = P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u);\n if (fold) {\n let box = (S.max.xyz - S.min.xyz) * GRID_BBOX_MARGIN;\n var bboxExtent = max(box.x, box.y);\n if (P.dim == 3u) { bboxExtent = max(bboxExtent, box.z); }\n let extent = max(min(bboxExtent, P.extentFactor * S.rmsRadius), GRID_EXTENT_FLOOR); // min(bbox, extentFactor x rms), floored (7.7)\n let cellSize = extent / f32(P.gridMax);\n S.gridMin = vec4f(S.centroid.xyz - vec3f(0.5 * extent), cellSize); // gridMin.w carries cellSize\n S.invCellSize = 1.0 / cellSize;\n S.eps = 0.25 * cellSize;\n }\n var outside = 0u; // the previous iteration's pseudo-cell counts, one per orthant (issue #90; 0 after load)\n for (var o = 0u; o < select(4u, 8u, P.dim == 3u); o = o + 1u) { outside = outside + cellHist[cells + o]; }\n S.outsideGrid = outside;\n S.maxCellOccupancy = atomicLoad(&hubCounters[1]);\n atomicStore(&hubCounters[0], 0u);\n atomicStore(&hubCounters[1], 0u);\n }\n if (STATS_MODE == 1u) { // FR: this iteration's temperature (7.20) into the state and the trace\n if ((P.flags & FA2_FLAG_ADAPTIVE) != 0u) { // adaptive cooling (Yifan Hu 2005 3.2): tKe is the previous iteration's sum |F|^2 over free nodes\n if (fold) {\n var t = S.temperature;\n if (tKe < S.frEnergy) {\n S.frProgress = S.frProgress + 1u;\n if (S.frProgress >= FR_COOLING_PATIENCE) { S.frProgress = 0u; t = t / FR_COOLING_STEP; }\n } else {\n S.frProgress = 0u;\n t = t * FR_COOLING_STEP;\n }\n S.frEnergy = tKe;\n S.temperature = t;\n }\n } else {\n S.temperature = P.temperature;\n }\n T[P.iterationIndex].modelScalar = S.temperature;\n }\n if (STATS_MODE == 2u) { // spring-electrical: the kinetic energy K5 folded into partials B (PD-4); 0 on the first iteration after load()\n S.kineticEnergy = tKe;\n T[P.iterationIndex].modelScalar = tKe;\n }\n }\n}";
25
25
  //# sourceMappingURL=fa2-stats-finalize.wgsl.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"fa2-stats-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH,qHAAqH;AACrH,eAAO,MAAM,oBAAoB,snJAsF/B,CAAC"}
1
+ {"version":3,"file":"fa2-stats-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH,qHAAqH;AACrH,eAAO,MAAM,oBAAoB,42JAwF/B,CAAC"}
@@ -60,7 +60,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
60
60
  S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)
61
61
  let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)
62
62
  S.meanDisplacement = meanDisp;
63
- S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);
63
+ S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= min(P.settleThreshold * S.rmsRadius, P.settleFloor)); // relative AND absolute (issue #97)
64
64
  }
65
65
  S.iteration = S.iteration + 1u;
66
66
  T[P.iterationIndex].meanDisplacement = S.meanDisplacement;
@@ -78,7 +78,9 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
78
78
  S.invCellSize = 1.0 / cellSize;
79
79
  S.eps = 0.25 * cellSize;
80
80
  }
81
- S.outsideGrid = cellHist[cells]; // the previous iteration's pseudo-cell count (0 after load)
81
+ var outside = 0u; // the previous iteration's pseudo-cell counts, one per orthant (issue #90; 0 after load)
82
+ for (var o = 0u; o < select(4u, 8u, P.dim == 3u); o = o + 1u) { outside = outside + cellHist[cells + o]; }
83
+ S.outsideGrid = outside;
82
84
  S.maxCellOccupancy = atomicLoad(&hubCounters[1]);
83
85
  atomicStore(&hubCounters[0], 0u);
84
86
  atomicStore(&hubCounters[1], 0u);
@@ -1 +1 @@
1
- {"version":3,"file":"fa2-stats-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH,qHAAqH;AACrH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EAsF7C,CAAC"}
1
+ {"version":3,"file":"fa2-stats-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH,qHAAqH;AACrH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EAwF7C,CAAC"}
@@ -29,25 +29,26 @@
29
29
  * raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
30
30
  * to 0); the relax kernels size themselves from words 0 and 20.
31
31
  *
32
- * Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a switch can be counted at
33
- * the done boundary too, which the host model of the tests mirrors): top-down switches to bottom-up when
34
- * `frontierDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's `max(1, floor(arcCount / n))`
35
- * unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up switches back when
36
- * `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no admitted device
37
- * reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the direction at 0. Every
38
- * change is counted in `switches`, the previous direction is word 14. The two unvisited words the test reads are
39
- * rebuilt exactly once per submit by `bfs-unvisited-flags` (PD-18) and maintained here by subtraction: the count is
40
- * subtracted from the SECOND boundary of a submit on and the degree sum from the THIRD on, because a boundary may
41
- * only subtract what the submit's rebuild counted, and the frontier whose degree sum the second boundary holds was
42
- * claimed before the rebuild ran (the rebuild counts the vertices unclaimed when it runs; the frontier rotated in at
43
- * boundary 0 was claimed by the previous submit's last contract, so it was never in the sum; boundary b subtracts
44
- * `next = |F_b|`, inside the sum iff b >= 1, and `degSum = deg(F_{b-1})`, inside it iff b >= 2). The degree sum is
45
- * the "unvisited degree estimate" of the design rather than an exact count for two reasons: it is one level stale
46
- * (a frontier's degree sum is only known once it has been expanded), and a bottom-up level expands nothing, so the
47
- * word stops falling while bottom-up runs and overstates the set afterwards. The bias is one-directional -- an
48
- * overstated m_u makes the switch INTO bottom-up harder, never easier -- and the next submit's rebuild makes it
49
- * exact again. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
50
- * textual edits of it.
32
+ * Beamer's test (P8-T8, PD-21; amended for issue #391), evaluated at every boundary BEFORE the `done` branch (so a
33
+ * switch can be counted at the done boundary too, which the host model of the tests mirrors): top-down switches to
34
+ * bottom-up when `nextDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's
35
+ * `max(1, floor(arcCount / n))` unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up
36
+ * switches back when `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no
37
+ * admitted device reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the
38
+ * direction at 0. Every change is counted in `switches`, the previous direction is word 14. `nextDegreeSum` (word
39
+ * 25) is Beamer's m_f measured EXACTLY: `bfs-next-degree` sums the out-degrees of the vertices a level claims at
40
+ * the end of that level, so the boundary that rotates them in as `next` compares the degree of the frontier it is
41
+ * about to expand -- not, as before the amendment, `frontierDegreeSum` (word 2), the degree of the frontier the
42
+ * previous level EXPANDED, one level stale and 0 after a bottom-up level, which on the 1M / 10M R-MAT missed the
43
+ * switch at the level holding 13.6M of the 21M arcs. Word 2 is still accumulated by the expansion and rotated into
44
+ * word 4 for the inspect seam. The two unvisited words are rebuilt exactly once per submit by `bfs-unvisited-flags`
45
+ * (PD-18) and maintained here by subtraction from the SECOND boundary of a submit on, because a boundary may only
46
+ * subtract what the submit's rebuild counted: the rebuild counts the vertices unclaimed when it runs, the frontier
47
+ * rotated in at boundary 0 was claimed by the previous submit's last level, so it was never in the sums, and
48
+ * boundary b subtracts `next = |F_b|` and `nextDegreeSum = deg(F_b)`, both inside the sums iff b >= 1. Both words
49
+ * are therefore exact at every boundary, bottom-up levels included (the sweep's claims are summed like any other).
50
+ * Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits
51
+ * of it.
51
52
  */
52
- export declare const frontierFinalizeWgsl = "\n@compute @workgroup_size(WG)\nfn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role == 0u) { // the level boundary\n if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end\n atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)\n return;\n }\n let finished = atomicLoad(&counters[0]);\n let next = atomicLoad(&counters[1]);\n let degSum = atomicLoad(&counters[2]);\n atomicStore(&counters[3], finished); // prevFrontierCount\n atomicStore(&counters[4], degSum); // prevDegreeSum\n atomicStore(&counters[0], next); // the rotation\n atomicStore(&counters[1], 0u);\n atomicStore(&counters[2], 0u);\n atomicStore(&counters[8], 0u); // edgeCount\n atomicStore(&counters[9], 0u); // edgeCountUnclamped\n atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount\n if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped to 2 (P8-T8, PD-18)\n atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount (exact): F_b was inside the submit's rebuilt sum iff b >= 1\n }\n if (P.firstOfSubmit >= 2u) {\n atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2\n }\n let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0\n atomicStore(&counters[11], level);\n let done = (next == 0u) || (level >= P.maxDepth);\n atomicStore(&counters[15], select(0u, 1u, done));\n var direction = atomicLoad(&counters[14]);\n if (P.mode == 1u) {\n direction = 0u; // top-down only (the test seam)\n } else if (direction == 0u) {\n if (degSum > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing\n } else {\n if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking\n }\n if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches\n var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)\n if (done) {\n path = 0u;\n } else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep\n path = 3u;\n atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);\n } else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)\n path = 2u; // bfs-fused: one WORKGROUP per frontier entry\n atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);\n } else {\n path = 1u; // advance-expand, then role 1 and bfs-contract\n }\n atomicStore(&counters[14], direction);\n atomicStore(&counters[24], path);\n } else if (P.role == 1u) { // the edge queue is filled\n if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count\n return;\n }\n let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);\n atomicStore(&counters[8], clamped);\n if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry\n atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing\n atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);\n atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);\n } else {\n atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)\n }\n } else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes\n atomicStore(&counters[8], 0u); // the dedupe counts (words 8 and 9, the SSSP sense) and the path word: nothing unless a pile is chosen below\n atomicStore(&counters[9], 0u);\n atomicStore(&counters[24], 0u);\n if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)\n return;\n }\n let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped\n let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped\n if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE\n atomicStore(&counters[15], 2u);\n return;\n }\n if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn\n atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter\n atomicStore(&counters[14], 0u); // mode 0\n atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)\n atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing\n atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)\n } else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile\n let threshold = bitcast<f32>(atomicLoad(&counters[22]));\n let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)\n if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED\n atomicStore(&counters[15], 3u);\n return;\n }\n atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below\n atomicStore(&counters[22], bitcast<u32>(raised));\n atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter\n atomicStore(&counters[14], 1u); // mode 1\n atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)\n atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing\n atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);\n } else { // both piles empty: finished\n atomicStore(&counters[15], 1u);\n }\n } else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed\n if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round\n if (atomicLoad(&counters[14]) == 0u) {\n atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)\n } else {\n atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)\n }\n }\n}\n";
53
+ export declare const frontierFinalizeWgsl = "\n@compute @workgroup_size(WG)\nfn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role == 0u) { // the level boundary\n if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end\n atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)\n return;\n }\n let finished = atomicLoad(&counters[0]);\n let next = atomicLoad(&counters[1]);\n let degSum = atomicLoad(&counters[2]);\n let nextDeg = atomicLoad(&counters[25]); // deg(F_b), summed by bfs-next-degree when F_b was claimed (issue #391)\n atomicStore(&counters[3], finished); // prevFrontierCount\n atomicStore(&counters[4], degSum); // prevDegreeSum\n atomicStore(&counters[0], next); // the rotation\n atomicStore(&counters[1], 0u);\n atomicStore(&counters[2], 0u);\n atomicStore(&counters[25], 0u); // the next level's claims sum from 0\n atomicStore(&counters[8], 0u); // edgeCount\n atomicStore(&counters[9], 0u); // edgeCountUnclamped\n atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount\n if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped (P8-T8, PD-18): F_b was inside the submit's rebuilt sums iff b >= 1\n atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount, exact\n atomicStore(&counters[6], atomicLoad(&counters[6]) - nextDeg); // unvisitedDegreeSum, exact (issue #391: no longer one level stale)\n }\n let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0\n atomicStore(&counters[11], level);\n let done = (next == 0u) || (level >= P.maxDepth);\n atomicStore(&counters[15], select(0u, 1u, done));\n var direction = atomicLoad(&counters[14]);\n if (P.mode == 1u) {\n direction = 0u; // top-down only (the test seam)\n } else if (direction == 0u) {\n if (nextDeg > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing, m_f the degree of the frontier about to be expanded\n } else {\n if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking\n }\n if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches\n var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)\n if (done) {\n path = 0u;\n } else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep\n path = 3u;\n atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);\n } else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)\n path = 2u; // bfs-fused: one WORKGROUP per frontier entry\n atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);\n } else {\n path = 1u; // advance-expand, then role 1 and bfs-contract\n }\n atomicStore(&counters[14], direction);\n atomicStore(&counters[24], path);\n } else if (P.role == 1u) { // the edge queue is filled\n if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count\n return;\n }\n let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);\n atomicStore(&counters[8], clamped);\n if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry\n atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing\n atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);\n atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);\n } else {\n atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)\n }\n } else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes\n atomicStore(&counters[8], 0u); // the dedupe counts (words 8 and 9, the SSSP sense) and the path word: nothing unless a pile is chosen below\n atomicStore(&counters[9], 0u);\n atomicStore(&counters[24], 0u);\n if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)\n return;\n }\n let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped\n let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped\n if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE\n atomicStore(&counters[15], 2u);\n return;\n }\n if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn\n atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter\n atomicStore(&counters[14], 0u); // mode 0\n atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)\n atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing\n atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)\n } else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile\n let threshold = bitcast<f32>(atomicLoad(&counters[22]));\n let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)\n if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED\n atomicStore(&counters[15], 3u);\n return;\n }\n atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below\n atomicStore(&counters[22], bitcast<u32>(raised));\n atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter\n atomicStore(&counters[14], 1u); // mode 1\n atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)\n atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing\n atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);\n } else { // both piles empty: finished\n atomicStore(&counters[15], 1u);\n }\n } else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed\n if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round\n if (atomicLoad(&counters[14]) == 0u) {\n atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)\n } else {\n atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)\n }\n }\n}\n";
53
54
  //# sourceMappingURL=frontier-finalize.wgsl.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"frontier-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkDG;AACH,eAAO,MAAM,oBAAoB,spRA+GhC,CAAC"}
1
+ {"version":3,"file":"frontier-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAmDG;AACH,eAAO,MAAM,oBAAoB,s5RA+GhC,CAAC"}
@@ -29,25 +29,26 @@
29
29
  * raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
30
30
  * to 0); the relax kernels size themselves from words 0 and 20.
31
31
  *
32
- * Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a switch can be counted at
33
- * the done boundary too, which the host model of the tests mirrors): top-down switches to bottom-up when
34
- * `frontierDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's `max(1, floor(arcCount / n))`
35
- * unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up switches back when
36
- * `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no admitted device
37
- * reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the direction at 0. Every
38
- * change is counted in `switches`, the previous direction is word 14. The two unvisited words the test reads are
39
- * rebuilt exactly once per submit by `bfs-unvisited-flags` (PD-18) and maintained here by subtraction: the count is
40
- * subtracted from the SECOND boundary of a submit on and the degree sum from the THIRD on, because a boundary may
41
- * only subtract what the submit's rebuild counted, and the frontier whose degree sum the second boundary holds was
42
- * claimed before the rebuild ran (the rebuild counts the vertices unclaimed when it runs; the frontier rotated in at
43
- * boundary 0 was claimed by the previous submit's last contract, so it was never in the sum; boundary b subtracts
44
- * `next = |F_b|`, inside the sum iff b >= 1, and `degSum = deg(F_{b-1})`, inside it iff b >= 2). The degree sum is
45
- * the "unvisited degree estimate" of the design rather than an exact count for two reasons: it is one level stale
46
- * (a frontier's degree sum is only known once it has been expanded), and a bottom-up level expands nothing, so the
47
- * word stops falling while bottom-up runs and overstates the set afterwards. The bias is one-directional -- an
48
- * overstated m_u makes the switch INTO bottom-up harder, never easier -- and the next submit's rebuild makes it
49
- * exact again. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
50
- * textual edits of it.
32
+ * Beamer's test (P8-T8, PD-21; amended for issue #391), evaluated at every boundary BEFORE the `done` branch (so a
33
+ * switch can be counted at the done boundary too, which the host model of the tests mirrors): top-down switches to
34
+ * bottom-up when `nextDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's
35
+ * `max(1, floor(arcCount / n))` unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up
36
+ * switches back when `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no
37
+ * admitted device reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the
38
+ * direction at 0. Every change is counted in `switches`, the previous direction is word 14. `nextDegreeSum` (word
39
+ * 25) is Beamer's m_f measured EXACTLY: `bfs-next-degree` sums the out-degrees of the vertices a level claims at
40
+ * the end of that level, so the boundary that rotates them in as `next` compares the degree of the frontier it is
41
+ * about to expand -- not, as before the amendment, `frontierDegreeSum` (word 2), the degree of the frontier the
42
+ * previous level EXPANDED, one level stale and 0 after a bottom-up level, which on the 1M / 10M R-MAT missed the
43
+ * switch at the level holding 13.6M of the 21M arcs. Word 2 is still accumulated by the expansion and rotated into
44
+ * word 4 for the inspect seam. The two unvisited words are rebuilt exactly once per submit by `bfs-unvisited-flags`
45
+ * (PD-18) and maintained here by subtraction from the SECOND boundary of a submit on, because a boundary may only
46
+ * subtract what the submit's rebuild counted: the rebuild counts the vertices unclaimed when it runs, the frontier
47
+ * rotated in at boundary 0 was claimed by the previous submit's last level, so it was never in the sums, and
48
+ * boundary b subtracts `next = |F_b|` and `nextDegreeSum = deg(F_b)`, both inside the sums iff b >= 1. Both words
49
+ * are therefore exact at every boundary, bottom-up levels included (the sweep's claims are summed like any other).
50
+ * Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits
51
+ * of it.
51
52
  */
52
53
  export const frontierFinalizeWgsl = /* wgsl */ `
53
54
  @compute @workgroup_size(WG)
@@ -61,19 +62,19 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
61
62
  let finished = atomicLoad(&counters[0]);
62
63
  let next = atomicLoad(&counters[1]);
63
64
  let degSum = atomicLoad(&counters[2]);
65
+ let nextDeg = atomicLoad(&counters[25]); // deg(F_b), summed by bfs-next-degree when F_b was claimed (issue #391)
64
66
  atomicStore(&counters[3], finished); // prevFrontierCount
65
67
  atomicStore(&counters[4], degSum); // prevDegreeSum
66
68
  atomicStore(&counters[0], next); // the rotation
67
69
  atomicStore(&counters[1], 0u);
68
70
  atomicStore(&counters[2], 0u);
71
+ atomicStore(&counters[25], 0u); // the next level's claims sum from 0
69
72
  atomicStore(&counters[8], 0u); // edgeCount
70
73
  atomicStore(&counters[9], 0u); // edgeCountUnclamped
71
74
  atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount
72
- if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped to 2 (P8-T8, PD-18)
73
- atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount (exact): F_b was inside the submit's rebuilt sum iff b >= 1
74
- }
75
- if (P.firstOfSubmit >= 2u) {
76
- atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2
75
+ if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped (P8-T8, PD-18): F_b was inside the submit's rebuilt sums iff b >= 1
76
+ atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount, exact
77
+ atomicStore(&counters[6], atomicLoad(&counters[6]) - nextDeg); // unvisitedDegreeSum, exact (issue #391: no longer one level stale)
77
78
  }
78
79
  let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0
79
80
  atomicStore(&counters[11], level);
@@ -83,7 +84,7 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
83
84
  if (P.mode == 1u) {
84
85
  direction = 0u; // top-down only (the test seam)
85
86
  } else if (direction == 0u) {
86
- if (degSum > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing
87
+ if (nextDeg > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing, m_f the degree of the frontier about to be expanded
87
88
  } else {
88
89
  if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking
89
90
  }
@@ -1 +1 @@
1
- {"version":3,"file":"frontier-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkDG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA+G9C,CAAC"}
1
+ {"version":3,"file":"frontier-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAmDG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA+G9C,CAAC"}
@@ -1,8 +1,9 @@
1
1
  /**
2
2
  * G1, the `grid-cell-key` kernel body (spec 7.7; P4-T8): the finest cell of every node from the state's robust extent,
3
3
  * `floor((p - gridMin) * invCellSize)` (a multiply, correctly rounded everywhere: PD-10), linearised when every axis
4
- * is in [0, G) and the outside pseudo-cell `cells` otherwise; `cellVal[i] = i`. The clamp before the floor keeps a
4
+ * is in [0, G), and otherwise one of the 2^dim outside pseudo-cells `cells + orthant`, the orthant of the cell about the
5
+ * grid centre (bit a set when `c[a] >= G / 2`; issue #90); `cellVal[i] = i`. The clamp before the floor keeps a
5
6
  * far-away or NaN coordinate out of an out-of-range float-to-int conversion. Body only; normative text.
6
7
  */
7
- export declare const gridCellKeyWgsl = "\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\n\n@compute @workgroup_size(WG)\nfn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n if (i >= P.n) { return; } // no barrier follows\n let cells = grid_cells();\n let gf = f32(P.gridMax);\n let q = (pos[i].xyz - S.gridMin.xyz) * S.invCellSize; // PD-10: never a division\n let c = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));\n let g = i32(P.gridMax);\n var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;\n if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }\n var key = cells; // the outside pseudo-cell (7.7)\n if (inside) {\n key = u32(c.x) + P.gridMax * u32(c.y);\n if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }\n }\n cellKey[i] = key;\n cellVal[i] = i;\n}\n";
8
+ export declare const gridCellKeyWgsl = "\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\n\n@compute @workgroup_size(WG)\nfn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n if (i >= P.n) { return; } // no barrier follows\n let cells = grid_cells();\n let gf = f32(P.gridMax);\n let q = (pos[i].xyz - S.gridMin.xyz) * S.invCellSize; // PD-10: never a division\n let c = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));\n let g = i32(P.gridMax);\n var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;\n if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }\n var key = cells + select(0u, 1u, c.x >= g / 2) + select(0u, 2u, c.y >= g / 2); // an outside pseudo-cell: its orthant (issue #90)\n if (P.dim == 3u) { key = key + select(0u, 4u, c.z >= g / 2); }\n if (inside) {\n key = u32(c.x) + P.gridMax * u32(c.y);\n if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }\n }\n cellKey[i] = key;\n cellVal[i] = i;\n}\n";
8
9
  //# sourceMappingURL=grid-cell-key.wgsl.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"grid-cell-key.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-cell-key.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,eAAO,MAAM,eAAe,uhCAsB3B,CAAC"}
1
+ {"version":3,"file":"grid-cell-key.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-cell-key.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,eAAO,MAAM,eAAe,+nCAuB3B,CAAC"}
@@ -1,7 +1,8 @@
1
1
  /**
2
2
  * G1, the `grid-cell-key` kernel body (spec 7.7; P4-T8): the finest cell of every node from the state's robust extent,
3
3
  * `floor((p - gridMin) * invCellSize)` (a multiply, correctly rounded everywhere: PD-10), linearised when every axis
4
- * is in [0, G) and the outside pseudo-cell `cells` otherwise; `cellVal[i] = i`. The clamp before the floor keeps a
4
+ * is in [0, G), and otherwise one of the 2^dim outside pseudo-cells `cells + orthant`, the orthant of the cell about the
5
+ * grid centre (bit a set when `c[a] >= G / 2`; issue #90); `cellVal[i] = i`. The clamp before the floor keeps a
5
6
  * far-away or NaN coordinate out of an out-of-range float-to-int conversion. Body only; normative text.
6
7
  */
7
8
  export const gridCellKeyWgsl = /* wgsl */ `
@@ -18,7 +19,8 @@ fn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocatio
18
19
  let g = i32(P.gridMax);
19
20
  var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;
20
21
  if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }
21
- var key = cells; // the outside pseudo-cell (7.7)
22
+ var key = cells + select(0u, 1u, c.x >= g / 2) + select(0u, 2u, c.y >= g / 2); // an outside pseudo-cell: its orthant (issue #90)
23
+ if (P.dim == 3u) { key = key + select(0u, 4u, c.z >= g / 2); }
22
24
  if (inside) {
23
25
  key = u32(c.x) + P.gridMax * u32(c.y);
24
26
  if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }
@@ -1 +1 @@
1
- {"version":3,"file":"grid-cell-key.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-cell-key.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,CAAC,MAAM,eAAe,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;CAsBzC,CAAC"}
1
+ {"version":3,"file":"grid-cell-key.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-cell-key.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,MAAM,CAAC,MAAM,eAAe,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;CAuBzC,CAAC"}
@@ -1,8 +1,9 @@
1
1
  /**
2
- * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the pseudo-cell included; the
2
+ * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the 2^dim outside pseudo-cells
3
+ * included (issue #90); the
3
4
  * mass-weighted position sum of a cell's sorted range in index order (no atomics: deterministic), the largest
4
5
  * occupancy into hubCounters[1], and cells above GRID_HUB_CELL entries appended to hubList for G4b (PD-13). Body
5
6
  * only; normative text.
6
7
  */
7
- export declare const gridCentroidWgsl = "\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\n\n@compute @workgroup_size(WG)\nfn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let c = linear_id(wid, lid.x);\n if (c > grid_cells()) { return; } // cells [0, cells]: the pseudo-cell is index cells; no barrier follows\n let start = cellStart[c];\n let count = cellStart[c + 1u] - start;\n atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration\n if (count > GRID_HUB_CELL) { // a hub cell: G4b sums it (PD-13)\n hubList[atomicAdd(&hubCounters[0], 1u)] = c;\n return;\n }\n var acc = vec4f(0.0);\n for (var k = start; k < start + count; k = k + 1u) { // sorted order: deterministic\n let p = pos[sortedIdx[k]];\n acc = acc + vec4f(p.xyz * p.w, p.w); // (sum m x, sum m y, sum m z, sum m)\n }\n pyramid[c] = acc;\n}\n";
8
+ export declare const gridCentroidWgsl = "\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\n\n@compute @workgroup_size(WG)\nfn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let c = linear_id(wid, lid.x);\n if (c >= grid_cells() + select(4u, 8u, P.dim == 3u)) { return; } // cells [0, cells + 2^dim): the pseudo-cells follow the real ones; no barrier follows\n let start = cellStart[c];\n let count = cellStart[c + 1u] - start;\n atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration\n if (count > GRID_HUB_CELL) { // a hub cell: G4b sums it (PD-13)\n hubList[atomicAdd(&hubCounters[0], 1u)] = c;\n return;\n }\n var acc = vec4f(0.0);\n for (var k = start; k < start + count; k = k + 1u) { // sorted order: deterministic\n let p = pos[sortedIdx[k]];\n acc = acc + vec4f(p.xyz * p.w, p.w); // (sum m x, sum m y, sum m z, sum m)\n }\n pyramid[c] = acc;\n}\n";
8
9
  //# sourceMappingURL=grid-centroid.wgsl.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"grid-centroid.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,eAAO,MAAM,gBAAgB,+jCAqB5B,CAAC"}
1
+ {"version":3,"file":"grid-centroid.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,eAAO,MAAM,gBAAgB,klCAqB5B,CAAC"}
@@ -1,5 +1,6 @@
1
1
  /**
2
- * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the pseudo-cell included; the
2
+ * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the 2^dim outside pseudo-cells
3
+ * included (issue #90); the
3
4
  * mass-weighted position sum of a cell's sorted range in index order (no atomics: deterministic), the largest
4
5
  * occupancy into hubCounters[1], and cells above GRID_HUB_CELL entries appended to hubList for G4b (PD-13). Body
5
6
  * only; normative text.
@@ -10,7 +11,7 @@ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.
10
11
  @compute @workgroup_size(WG)
11
12
  fn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
12
13
  let c = linear_id(wid, lid.x);
13
- if (c > grid_cells()) { return; } // cells [0, cells]: the pseudo-cell is index cells; no barrier follows
14
+ if (c >= grid_cells() + select(4u, 8u, P.dim == 3u)) { return; } // cells [0, cells + 2^dim): the pseudo-cells follow the real ones; no barrier follows
14
15
  let start = cellStart[c];
15
16
  let count = cellStart[c + 1u] - start;
16
17
  atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration
@@ -1 +1 @@
1
- {"version":3,"file":"grid-centroid.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;CAqB1C,CAAC"}
1
+ {"version":3,"file":"grid-centroid.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;CAqB1C,CAAC"}
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * G5, the `grid-downsample` kernel body (spec 7.7; P4-T9): one dispatch per coarser level; every parent cell is the
3
3
  * sum of its 4 (2D) or 8 (3D) children at the level below, read at P.childBase and written at P.parentBase (the
4
- * pseudo-cell, index cells of level 0, is never a child). No atomics. Body only; normative text.
4
+ * pseudo-cells, indices cells .. of level 0, are never children). No atomics. Body only; normative text.
5
5
  */
6
6
  export declare const gridDownsampleWgsl = "\n@compute @workgroup_size(WG)\nfn grid_downsample(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let pc = linear_id(wid, lid.x); // the parent cell inside its level\n if (pc >= P.parentCells) { return; } // no barrier follows\n let side = P.parentSide;\n let cs = 2u * side; // the child level's side\n let px = pc % side;\n let py = (pc / side) % side;\n let pz = pc / (side * side);\n var acc = vec4f(0.0);\n for (var dz = 0u; dz < P.depth; dz = dz + 1u) {\n for (var dy = 0u; dy < 2u; dy = dy + 1u) {\n for (var dx = 0u; dx < 2u; dx = dx + 1u) {\n let child = (2u * px + dx) + cs * ((2u * py + dy) + cs * (2u * pz + dz));\n acc = acc + pyramid[P.childBase + child];\n }\n }\n }\n pyramid[P.parentBase + pc] = acc;\n}\n";
7
7
  //# sourceMappingURL=grid-downsample.wgsl.d.ts.map
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * G5, the `grid-downsample` kernel body (spec 7.7; P4-T9): one dispatch per coarser level; every parent cell is the
3
3
  * sum of its 4 (2D) or 8 (3D) children at the level below, read at P.childBase and written at P.parentBase (the
4
- * pseudo-cell, index cells of level 0, is never a child). No atomics. Body only; normative text.
4
+ * pseudo-cells, indices cells .. of level 0, are never children). No atomics. Body only; normative text.
5
5
  */
6
6
  export const gridDownsampleWgsl = /* wgsl */ `
7
7
  @compute @workgroup_size(WG)
@@ -2,12 +2,14 @@
2
2
  * G6, the `grid-far-field` kernel body (spec 7.7; P4-T10; D24): per node `i = sortedIdx[t]`, its finest cell
3
3
  * recomputed from `pos[i]` and the state (PD-10); for an inside node the coarsest level minus the 3x3 (3x3x3)
4
4
  * around its coarsest cell, then at every finer level the 6x6 (6x6x6) block that is the parent's 3x3 minus this
5
- * level's own 3x3 -- space tiled exactly once, no theta -- plus the outside pseudo-cell's centroid; for an outside
6
- * node the coarsest level in full and no pseudo-cell. Every cell term is the per-cell law on the mass-weighted
7
- * centroid (Gephi Region semantics), softened by `eps^2`: `LAW` 0 (FA2) `d * (k m_i M / d2)`, `LAW` 1 (FR, 7.20)
5
+ * level's own 3x3 -- space tiled exactly once, no theta -- plus the centroid of each of the 2^dim outside
6
+ * pseudo-cells, one per orthant about the grid centre (issue #90); for an outside node the coarsest level in full and
7
+ * no pseudo-cell. Every cell term is the per-cell law on the mass-weighted
8
+ * centroid (Gephi Region semantics), softened by `eps^2`: `LAW` 0 (FA2) `d * (k m_i M / d2)` with `|d|^2` first
9
+ * floored at 0.01^2 like K3's pair law (issue #89), `LAW` 1 (FR, 7.20)
8
10
  * `d * (k^2 M / d2)` (mass 1 per node, so `M` is the cell's count), `LAW` 2 (coulomb) `d * (-g m_i M / d2^1.5)`
9
11
  * (P4-T13, PD-22). `force += f` (K2 wrote it). The loop bounds are `P.levels` and `P.gridMax` from the uniform,
10
12
  * not a `LEVELS` override (PD-16, DEP-P4-G). Body only; normative text.
11
13
  */
12
- export declare const gridFarFieldWgsl = "\nfn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }\nfn store_force(i: u32, f: vec3f) {\n force[3u * i] = f.x;\n force[3u * i + 1u] = f.y;\n force[3u * i + 2u] = f.z;\n}\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\nfn grid_side(level: u32) -> u32 { return P.gridMax >> level; }\nfn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cell at index cells)\n var base = 0u;\n for (var l = 0u; l < level; l = l + 1u) {\n let s = grid_side(l);\n base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, 1u, l == 0u);\n }\n return base;\n}\nfn cell_at(level: u32, cx: i32, cy: i32, cz: i32) -> u32 {\n let s = grid_side(level);\n return level_base(level) + u32(cx) + s * (u32(cy) + select(0u, s * u32(cz), P.dim == 3u));\n}\nfn cell_force(pi: vec4f, q: vec4f) -> vec3f { // one far-field term, softened by state.eps (7.7)\n if (q.w <= 0.0) { return vec3f(0.0); } // an empty cell\n let d = pi.xyz - q.xyz / q.w; // to the mass-weighted centroid\n let d2 = dot(d, d) + S.eps * S.eps;\n if (LAW == 1u) { return d * (P.frK * P.frK * q.w / d2); } // LAW 1 (FR, 7.20): k^2 / d per node, q.w nodes at the centroid\n if (LAW == 2u) { return d * (-P.coulomb * pi.w * q.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb): -g m_i M_cell / d^2\n return d * (P.scalingRatio * pi.w * q.w / d2); // LAW 0 (FA2): |F| = k m_i M_cell / d\n}\n\n@compute @workgroup_size(WG)\nfn grid_far_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let t = linear_id(wid, lid.x);\n if (t >= P.n) { return; } // no barrier follows\n let i = sortedIdx[t]; // sorted order (D24)\n let pi = pos[i];\n let gf = f32(P.gridMax);\n let q = (pi.xyz - S.gridMin.xyz) * S.invCellSize; // PD-10\n var c0 = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));\n if (P.dim == 2u) { c0.z = 0; } // 2D: one z plane; the loops below visit cz = 0 only, so the 3x3 test must see cz - 0\n let g = i32(P.gridMax);\n var inside = c0.x >= 0 && c0.x < g && c0.y >= 0 && c0.y < g;\n if (P.dim == 3u) { inside = inside && c0.z >= 0 && c0.z < g; }\n let top = P.levels - 1u;\n let ts = i32(grid_side(top)); // the coarsest side (4)\n let zTop = select(0, ts - 1, P.dim == 3u); // z ranges: one plane in 2D\n var f = vec3f(0.0);\n if (inside) {\n let ct = c0 / i32(1u << top); // the node's coarsest cell\n for (var cz = 0; cz <= zTop; cz = cz + 1) {\n for (var cy = 0; cy < ts; cy = cy + 1) {\n for (var cx = 0; cx < ts; cx = cx + 1) {\n if (abs(cx - ct.x) <= 1 && abs(cy - ct.y) <= 1 && abs(cz - ct.z) <= 1) { continue; } // the 3x3(x3) is finer levels' work\n f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);\n }\n }\n }\n for (var l = top; l > 0u; l = l - 1u) { // level l - 1: the parent's 3x3 at level l, refined, minus this level's own 3x3\n let level = l - 1u;\n let cl = c0 / i32(1u << level);\n let cp = cl / 2;\n let side = i32(grid_side(level));\n let zLo = select(0, max(0, 2 * (cp.z - 1)), P.dim == 3u);\n let zHi = select(0, min(side - 1, 2 * (cp.z + 1) + 1), P.dim == 3u);\n for (var cz = zLo; cz <= zHi; cz = cz + 1) {\n for (var cy = max(0, 2 * (cp.y - 1)); cy <= min(side - 1, 2 * (cp.y + 1) + 1); cy = cy + 1) {\n for (var cx = max(0, 2 * (cp.x - 1)); cx <= min(side - 1, 2 * (cp.x + 1) + 1); cx = cx + 1) {\n if (abs(cx - cl.x) <= 1 && abs(cy - cl.y) <= 1 && abs(cz - cl.z) <= 1) { continue; }\n f = f + cell_force(pi, pyramid[cell_at(level, cx, cy, cz)]);\n }\n }\n }\n }\n f = f + cell_force(pi, pyramid[grid_cells()]); // the outside pseudo-cell as one far-field term\n } else {\n for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (it would include itself)\n for (var cy = 0; cy < ts; cy = cy + 1) {\n for (var cx = 0; cx < ts; cx = cx + 1) {\n f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);\n }\n }\n }\n }\n store_force(i, load_force(i) + f);\n}\n";
14
+ export declare const gridFarFieldWgsl = "\nfn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }\nfn store_force(i: u32, f: vec3f) {\n force[3u * i] = f.x;\n force[3u * i + 1u] = f.y;\n force[3u * i + 2u] = f.z;\n}\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\nfn grid_side(level: u32) -> u32 { return P.gridMax >> level; }\nfn outside_cells() -> u32 { return select(4u, 8u, P.dim == 3u); } // one pseudo-cell per orthant (issue #90)\nfn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cells at cells ..)\n var base = 0u;\n for (var l = 0u; l < level; l = l + 1u) {\n let s = grid_side(l);\n base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, outside_cells(), l == 0u);\n }\n return base;\n}\nfn cell_at(level: u32, cx: i32, cy: i32, cz: i32) -> u32 {\n let s = grid_side(level);\n return level_base(level) + u32(cx) + s * (u32(cy) + select(0u, s * u32(cz), P.dim == 3u));\n}\nfn cell_force(pi: vec4f, q: vec4f) -> vec3f { // one far-field term, softened by state.eps (7.7)\n if (q.w <= 0.0) { return vec3f(0.0); } // an empty cell\n let d = pi.xyz - q.xyz / q.w; // to the mass-weighted centroid\n var d2 = dot(d, d);\n if (LAW == 0u) { d2 = max(d2, FA2_DIST_FLOOR_SQ); } // FA2 alone floors d >= 0.01, as K3 and G7 do (issue #89); FR and coulomb are unfloored (7.20)\n d2 = d2 + S.eps * S.eps;\n if (LAW == 1u) { return d * (P.frK * P.frK * q.w / d2); } // LAW 1 (FR, 7.20): k^2 / d per node, q.w nodes at the centroid\n if (LAW == 2u) { return d * (-P.coulomb * pi.w * q.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb): -g m_i M_cell / d^2\n return d * (P.scalingRatio * pi.w * q.w / d2); // LAW 0 (FA2): |F| = k m_i M_cell / d\n}\n\n@compute @workgroup_size(WG)\nfn grid_far_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let t = linear_id(wid, lid.x);\n if (t >= P.n) { return; } // no barrier follows\n let i = sortedIdx[t]; // sorted order (D24)\n let pi = pos[i];\n let gf = f32(P.gridMax);\n let q = (pi.xyz - S.gridMin.xyz) * S.invCellSize; // PD-10\n var c0 = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));\n if (P.dim == 2u) { c0.z = 0; } // 2D: one z plane; the loops below visit cz = 0 only, so the 3x3 test must see cz - 0\n let g = i32(P.gridMax);\n var inside = c0.x >= 0 && c0.x < g && c0.y >= 0 && c0.y < g;\n if (P.dim == 3u) { inside = inside && c0.z >= 0 && c0.z < g; }\n let top = P.levels - 1u;\n let ts = i32(grid_side(top)); // the coarsest side (4)\n let zTop = select(0, ts - 1, P.dim == 3u); // z ranges: one plane in 2D\n var f = vec3f(0.0);\n if (inside) {\n let ct = c0 / i32(1u << top); // the node's coarsest cell\n for (var cz = 0; cz <= zTop; cz = cz + 1) {\n for (var cy = 0; cy < ts; cy = cy + 1) {\n for (var cx = 0; cx < ts; cx = cx + 1) {\n if (abs(cx - ct.x) <= 1 && abs(cy - ct.y) <= 1 && abs(cz - ct.z) <= 1) { continue; } // the 3x3(x3) is finer levels' work\n f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);\n }\n }\n }\n for (var l = top; l > 0u; l = l - 1u) { // level l - 1: the parent's 3x3 at level l, refined, minus this level's own 3x3\n let level = l - 1u;\n let cl = c0 / i32(1u << level);\n let cp = cl / 2;\n let side = i32(grid_side(level));\n let zLo = select(0, max(0, 2 * (cp.z - 1)), P.dim == 3u);\n let zHi = select(0, min(side - 1, 2 * (cp.z + 1) + 1), P.dim == 3u);\n for (var cz = zLo; cz <= zHi; cz = cz + 1) {\n for (var cy = max(0, 2 * (cp.y - 1)); cy <= min(side - 1, 2 * (cp.y + 1) + 1); cy = cy + 1) {\n for (var cx = max(0, 2 * (cp.x - 1)); cx <= min(side - 1, 2 * (cp.x + 1) + 1); cx = cx + 1) {\n if (abs(cx - cl.x) <= 1 && abs(cy - cl.y) <= 1 && abs(cz - cl.z) <= 1) { continue; }\n f = f + cell_force(pi, pyramid[cell_at(level, cx, cy, cz)]);\n }\n }\n }\n }\n for (var o = 0u; o < outside_cells(); o = o + 1u) { // every outside pseudo-cell: one far-field term per orthant\n f = f + cell_force(pi, pyramid[grid_cells() + o]);\n }\n } else {\n for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (G7 sums them pair by pair)\n for (var cy = 0; cy < ts; cy = cy + 1) {\n for (var cx = 0; cx < ts; cx = cx + 1) {\n f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);\n }\n }\n }\n }\n store_force(i, load_force(i) + f);\n}\n";
13
15
  //# sourceMappingURL=grid-far-field.wgsl.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"grid-far-field.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-far-field.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AACH,eAAO,MAAM,gBAAgB,mzJAqF5B,CAAC"}
1
+ {"version":3,"file":"grid-far-field.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-far-field.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AACH,eAAO,MAAM,gBAAgB,qrKA0F5B,CAAC"}
@@ -2,9 +2,11 @@
2
2
  * G6, the `grid-far-field` kernel body (spec 7.7; P4-T10; D24): per node `i = sortedIdx[t]`, its finest cell
3
3
  * recomputed from `pos[i]` and the state (PD-10); for an inside node the coarsest level minus the 3x3 (3x3x3)
4
4
  * around its coarsest cell, then at every finer level the 6x6 (6x6x6) block that is the parent's 3x3 minus this
5
- * level's own 3x3 -- space tiled exactly once, no theta -- plus the outside pseudo-cell's centroid; for an outside
6
- * node the coarsest level in full and no pseudo-cell. Every cell term is the per-cell law on the mass-weighted
7
- * centroid (Gephi Region semantics), softened by `eps^2`: `LAW` 0 (FA2) `d * (k m_i M / d2)`, `LAW` 1 (FR, 7.20)
5
+ * level's own 3x3 -- space tiled exactly once, no theta -- plus the centroid of each of the 2^dim outside
6
+ * pseudo-cells, one per orthant about the grid centre (issue #90); for an outside node the coarsest level in full and
7
+ * no pseudo-cell. Every cell term is the per-cell law on the mass-weighted
8
+ * centroid (Gephi Region semantics), softened by `eps^2`: `LAW` 0 (FA2) `d * (k m_i M / d2)` with `|d|^2` first
9
+ * floored at 0.01^2 like K3's pair law (issue #89), `LAW` 1 (FR, 7.20)
8
10
  * `d * (k^2 M / d2)` (mass 1 per node, so `M` is the cell's count), `LAW` 2 (coulomb) `d * (-g m_i M / d2^1.5)`
9
11
  * (P4-T13, PD-22). `force += f` (K2 wrote it). The loop bounds are `P.levels` and `P.gridMax` from the uniform,
10
12
  * not a `LEVELS` override (PD-16, DEP-P4-G). Body only; normative text.
@@ -18,11 +20,12 @@ fn store_force(i: u32, f: vec3f) {
18
20
  }
19
21
  fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
20
22
  fn grid_side(level: u32) -> u32 { return P.gridMax >> level; }
21
- fn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cell at index cells)
23
+ fn outside_cells() -> u32 { return select(4u, 8u, P.dim == 3u); } // one pseudo-cell per orthant (issue #90)
24
+ fn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cells at cells ..)
22
25
  var base = 0u;
23
26
  for (var l = 0u; l < level; l = l + 1u) {
24
27
  let s = grid_side(l);
25
- base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, 1u, l == 0u);
28
+ base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, outside_cells(), l == 0u);
26
29
  }
27
30
  return base;
28
31
  }
@@ -33,7 +36,9 @@ fn cell_at(level: u32, cx: i32, cy: i32, cz: i32) -> u32 {
33
36
  fn cell_force(pi: vec4f, q: vec4f) -> vec3f { // one far-field term, softened by state.eps (7.7)
34
37
  if (q.w <= 0.0) { return vec3f(0.0); } // an empty cell
35
38
  let d = pi.xyz - q.xyz / q.w; // to the mass-weighted centroid
36
- let d2 = dot(d, d) + S.eps * S.eps;
39
+ var d2 = dot(d, d);
40
+ if (LAW == 0u) { d2 = max(d2, FA2_DIST_FLOOR_SQ); } // FA2 alone floors d >= 0.01, as K3 and G7 do (issue #89); FR and coulomb are unfloored (7.20)
41
+ d2 = d2 + S.eps * S.eps;
37
42
  if (LAW == 1u) { return d * (P.frK * P.frK * q.w / d2); } // LAW 1 (FR, 7.20): k^2 / d per node, q.w nodes at the centroid
38
43
  if (LAW == 2u) { return d * (-P.coulomb * pi.w * q.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb): -g m_i M_cell / d^2
39
44
  return d * (P.scalingRatio * pi.w * q.w / d2); // LAW 0 (FA2): |F| = k m_i M_cell / d
@@ -82,9 +87,11 @@ fn grid_far_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
82
87
  }
83
88
  }
84
89
  }
85
- f = f + cell_force(pi, pyramid[grid_cells()]); // the outside pseudo-cell as one far-field term
90
+ for (var o = 0u; o < outside_cells(); o = o + 1u) { // every outside pseudo-cell: one far-field term per orthant
91
+ f = f + cell_force(pi, pyramid[grid_cells() + o]);
92
+ }
86
93
  } else {
87
- for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (it would include itself)
94
+ for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (G7 sums them pair by pair)
88
95
  for (var cy = 0; cy < ts; cy = cy + 1) {
89
96
  for (var cx = 0; cx < ts; cx = cx + 1) {
90
97
  f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);
@@ -1 +1 @@
1
- {"version":3,"file":"grid-far-field.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-far-field.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAqF1C,CAAC"}
1
+ {"version":3,"file":"grid-far-field.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-far-field.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA0F1C,CAAC"}