@graphty/webgpu-graph-algorithms 0.6.5 → 0.6.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. package/README.md +38 -17
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-hzGggHeM.js → context-DiSr6eiz.js} +32 -18
  4. package/dist/chunks/context-DiSr6eiz.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/algorithms/bfs.d.ts +10 -6
  7. package/dist/src/algorithms/bfs.d.ts.map +1 -1
  8. package/dist/src/algorithms/bfs.js +32 -9
  9. package/dist/src/algorithms/bfs.js.map +1 -1
  10. package/dist/src/constants.d.ts +33 -0
  11. package/dist/src/constants.d.ts.map +1 -1
  12. package/dist/src/constants.js +33 -0
  13. package/dist/src/constants.js.map +1 -1
  14. package/dist/src/kernel/dispatch.d.ts +2 -2
  15. package/dist/src/kernel/dispatch.d.ts.map +1 -1
  16. package/dist/src/kernel/kernel.d.ts +1 -1
  17. package/dist/src/kernel/kernel.js +2 -2
  18. package/dist/src/kernel/kernel.js.map +1 -1
  19. package/dist/src/kernel/prelude.d.ts.map +1 -1
  20. package/dist/src/kernel/prelude.js +2 -1
  21. package/dist/src/kernel/prelude.js.map +1 -1
  22. package/dist/src/kernels.d.ts +10 -7
  23. package/dist/src/kernels.d.ts.map +1 -1
  24. package/dist/src/kernels.js +33 -9
  25. package/dist/src/kernels.js.map +1 -1
  26. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  27. package/dist/src/layouts/forceatlas2.js +2 -1
  28. package/dist/src/layouts/forceatlas2.js.map +1 -1
  29. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  30. package/dist/src/layouts/fruchterman-reingold.js +4 -2
  31. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  32. package/dist/src/layouts/repulsion-exact.d.ts +16 -0
  33. package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
  34. package/dist/src/layouts/repulsion-exact.js +21 -1
  35. package/dist/src/layouts/repulsion-exact.js.map +1 -1
  36. package/dist/src/layouts/repulsion-grid.d.ts +1 -1
  37. package/dist/src/layouts/repulsion-grid.js +1 -1
  38. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  39. package/dist/src/layouts/spring-electrical.js +6 -2
  40. package/dist/src/layouts/spring-electrical.js.map +1 -1
  41. package/dist/src/primitives/frontier.d.ts +1 -0
  42. package/dist/src/primitives/frontier.d.ts.map +1 -1
  43. package/dist/src/primitives/frontier.js +1 -0
  44. package/dist/src/primitives/frontier.js.map +1 -1
  45. package/dist/src/primitives/grid-pyramid.d.ts +4 -4
  46. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  47. package/dist/src/primitives/grid-pyramid.js +4 -3
  48. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  49. package/dist/src/primitives/grid.d.ts +13 -10
  50. package/dist/src/primitives/grid.d.ts.map +1 -1
  51. package/dist/src/primitives/grid.js +10 -7
  52. package/dist/src/primitives/grid.js.map +1 -1
  53. package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
  54. package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
  55. package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
  56. package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
  57. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
  58. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
  59. package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
  60. package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
  61. package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
  62. package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
  63. package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
  64. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
  65. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
  66. package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
  67. package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
  68. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
  69. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
  70. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
  71. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
  72. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
  73. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
  74. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
  75. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
  76. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +21 -20
  77. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
  78. package/dist/src/wgsl/frontier-finalize.wgsl.js +26 -25
  79. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
  80. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
  81. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
  82. package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
  83. package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
  84. package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
  85. package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
  86. package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
  87. package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
  88. package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
  89. package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
  90. package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
  91. package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
  92. package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
  93. package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
  94. package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
  95. package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
  96. package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
  97. package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
  98. package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
  99. package/dist/src/wgsl/histogram.wgsl.js +1 -1
  100. package/dist/webgpu-graph-algorithms.js +124 -40
  101. package/dist/webgpu-graph-algorithms.js.map +1 -1
  102. package/package.json +3 -3
  103. package/src/algorithms/bfs.ts +33 -9
  104. package/src/algorithms/pagerank.ts +19 -5
  105. package/src/algorithms/power-iteration.ts +8 -2
  106. package/src/constants.ts +35 -0
  107. package/src/kernel/dispatch.ts +2 -2
  108. package/src/kernel/kernel.ts +2 -2
  109. package/src/kernel/prelude.ts +2 -0
  110. package/src/kernels.ts +35 -9
  111. package/src/layouts/forceatlas2.ts +2 -0
  112. package/src/layouts/fruchterman-reingold.ts +4 -1
  113. package/src/layouts/repulsion-exact.ts +29 -1
  114. package/src/layouts/repulsion-grid.ts +1 -1
  115. package/src/layouts/spring-electrical.ts +8 -1
  116. package/src/memory/residency.ts +14 -4
  117. package/src/primitives/frontier.ts +2 -0
  118. package/src/primitives/grid-pyramid.ts +6 -5
  119. package/src/primitives/grid.ts +17 -12
  120. package/src/wgsl/advance-expand.wgsl.ts +4 -3
  121. package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
  122. package/src/wgsl/bfs-fused.wgsl.ts +6 -6
  123. package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
  124. package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
  125. package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
  126. package/src/wgsl/frontier-finalize.wgsl.ts +26 -25
  127. package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
  128. package/src/wgsl/grid-centroid.wgsl.ts +3 -2
  129. package/src/wgsl/grid-downsample.wgsl.ts +1 -1
  130. package/src/wgsl/grid-far-field.wgsl.ts +15 -8
  131. package/src/wgsl/grid-near-field.wgsl.ts +6 -2
  132. package/src/wgsl/histogram.wgsl.ts +1 -1
  133. package/dist/chunks/context-hzGggHeM.js.map +0 -1
@@ -579,7 +579,9 @@ export class GraphResidency {
579
579
  }
580
580
 
581
581
  /**
582
- * One resident per array (spec 4.3: views upload in perArray mode, never into the arena).
582
+ * One resident per array (spec 4.3: views upload in perArray mode, never into the arena). An empty array (the
583
+ * colIdx of an edgeless directed reverse view, the src / dst of an edgeless edgeList) is skipped: spec 5.6 never
584
+ * uploads a zero-length array, and Kernel.bind rejects a zero-size binding, so it is absent as in core().
583
585
  * @param record - the owning record
584
586
  * @param arrays - the named arrays
585
587
  * @param label - the buffer label prefix
@@ -592,6 +594,9 @@ export class GraphResidency {
592
594
  ): Readonly<Record<string, Binding>> {
593
595
  const bindings: Record<string, Binding> = {};
594
596
  for (const [name, array] of arrays) {
597
+ if (array.byteLength === 0) {
598
+ continue;
599
+ }
595
600
  const resident = this.upload(record, array, array, `${label}:${name}`);
596
601
  bindings[name] = { buffer: resident.buffer, offset: 0, size: resident.byteLength, window: null };
597
602
  }
@@ -606,19 +611,24 @@ export class GraphResidency {
606
611
  * lengths, so keying the packed buffer on `rev.rowPtr` would make the packed and the unpacked view of one
607
612
  * snapshot collide -- whichever was built second would get the other's buffer. The record still owns the
608
613
  * resident, so release(s) destroys it with the rest. Offsets are STORAGE_ALIGN-aligned because Kernel.bind
609
- * rejects any other offset synchronously (E_INVALID_ARGUMENT { argument: "offset" }).
614
+ * rejects any other offset synchronously (E_INVALID_ARGUMENT { argument: "offset" }). Empty arrays are left
615
+ * out as in separateArrays; when nothing is left, nothing is uploaded.
610
616
  * @param record - the owning record
611
- * @param arrays - the named arrays, in buffer order
617
+ * @param all - the named arrays, in buffer order
612
618
  * @param key - the marker object the resident is keyed on
613
619
  * @param label - the buffer label
614
620
  * @returns the bindings by name, all into the one buffer
615
621
  */
616
622
  private packArrays(
617
623
  record: ResidencyRecord,
618
- arrays: readonly (readonly [string, TypedArrayData])[],
624
+ all: readonly (readonly [string, TypedArrayData])[],
619
625
  key: object,
620
626
  label: string,
621
627
  ): Readonly<Record<string, Binding>> {
628
+ const arrays = all.filter(([, array]) => array.byteLength > 0);
629
+ if (arrays.length === 0) {
630
+ return Object.freeze({});
631
+ }
622
632
  const offsets: number[] = [];
623
633
  let total = 0;
624
634
  for (const [, array] of arrays) {
@@ -71,6 +71,7 @@ export const W: Readonly<{
71
71
  thresholdBits: 22;
72
72
  deltaBits: 23;
73
73
  path: 24;
74
+ nextDegreeSum: 25;
74
75
  }> = Object.freeze({
75
76
  frontierCount: 0,
76
77
  nextFrontierCount: 1,
@@ -97,6 +98,7 @@ export const W: Readonly<{
97
98
  thresholdBits: 22,
98
99
  deltaBits: 23,
99
100
  path: 24,
101
+ nextDegreeSum: 25,
100
102
  });
101
103
 
102
104
  /** The words a `reset` seeds (every other word is zeroed). */
@@ -1,11 +1,11 @@
1
1
  /**
2
2
  * The grid pyramid (spec 6 row 12, 7.7 G4-G5; P4-T9): the planner that records, into the caller's pass, the finest
3
- * centroids (G4, `grid-centroid`: thread per cell over `cells + 1`, the pseudo-cell included), the hub-cell
3
+ * centroids (G4, `grid-centroid`: thread per cell over `cells + outsideCells`, the pseudo-cells included), the hub-cell
4
4
  * completion (G4a: the T1 `indirect-finalize` over `hubCounters[0]` into `hubArgs` with `wg = 1`, so the finalize's
5
5
  * `ceil(count / wg)` is ONE workgroup per hub cell; G4b: `grid-centroid-hub`, one workgroup per hub cell, dispatched
6
6
  * indirectly; PD-13, DEP-P4-I) and one `grid-downsample` dispatch per coarser
7
7
  * level (G5). Level 0 holds `[sum m x, sum m y, sum m z, sum m]` per cell; every parent is the sum of its 2^dim
8
- * children; the pseudo-cell (index `cells` of level 0) is never a child. No atomics touch the sums (design 6 row 12:
8
+ * children; the pseudo-cells (indices `cells ..` of level 0) are never children. No atomics touch the sums (design 6 row 12:
9
9
  * bitwise reproducible); the only atomics are the hub append and the occupancy max.
10
10
  *
11
11
  * The named grid buffers (`pyramid`, `hubList`, `hubCounters`, `hubArgs`) are the caller's (the model's
@@ -34,9 +34,9 @@ export interface GridPyramidBindings {
34
34
  readonly params: Binding;
35
35
  /** `n` words: the sorted node indices (the T8 build). */
36
36
  readonly sortedIdx: Binding;
37
- /** `cells + 2` words: the exclusive scan of the cell histogram (the T8 build). */
37
+ /** `histWords` (`cells + 2^dim + 1`) words: the exclusive scan of the cell histogram (the T8 build). */
38
38
  readonly cellStart: Binding;
39
- /** `pyramidCells` vec4f: every level, level 0 first with the pseudo-cell at index `cells`. */
39
+ /** `pyramidCells` vec4f: every level, level 0 first with the 2^dim orthant pseudo-cells from index `cells`. */
40
40
  readonly pyramid: Binding;
41
41
  /** The hub cells' indices, appended by G4 (at least one word; at most `floor(n / (GRID_HUB_CELL + 1))` are ever written, so `ceil(n / GRID_HUB_CELL)` words always suffice). */
42
42
  readonly hubList: Binding;
@@ -204,7 +204,8 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
204
204
  }
205
205
  const { centroid, finalize, hub, downsample } = this.kernels;
206
206
  const one: DispatchPlan = { x: 1, y: 1, z: 1, items: 1, stride: null };
207
- centroid.dispatch(pass, bound.centroid, plan1d(spec.cells + 1, scope.workgroupSize, scope.caps), [paramsOffset]);
207
+ const level0 = spec.cells + spec.outsideCells;
208
+ centroid.dispatch(pass, bound.centroid, plan1d(level0, scope.workgroupSize, scope.caps), [paramsOffset]);
208
209
  finalize.dispatch(pass, bound.finalize, one, [bound.finalizeOffset]);
209
210
  hub.dispatchIndirect(pass, bound.hub, bound.hubArgs, 0, [paramsOffset]);
210
211
  this.dispatches = 3;
@@ -3,8 +3,9 @@
3
3
  * the caller's pass, the cell keys (G1, `grid-cell-key`), the stable sort by key (G2: `radixSort` at GRID_SORT_BITS,
4
4
  * or `countingSortByKey` when the caller asks for the set-deterministic path) and the per-cell histogram with its
5
5
  * exclusive scan (G3: the `histogram` kernel over `cellKey` and the `scan` of it; DEP-P4-I names no grid-specific
6
- * id). `cellHist` and `cellStart` hold `cells + 2` words: every real cell, the outside pseudo-cell at index `cells`
7
- * and one more so `cellStart[cells + 1] === n` closes the last range. Every zeroing is a `fill` dispatch inside the
6
+ * id). `cellHist` and `cellStart` hold `histWords = cells + 2^dim + 1` words: every real cell, the 2^dim outside
7
+ * pseudo-cells (one per orthant about the grid centre, issue #90) from index `cells`, and one more so
8
+ * `cellStart[histWords - 1] === n` closes the last range. Every zeroing is a `fill` dispatch inside the
8
9
  * pass (PD-12), never an encoder clear.
9
10
  *
10
11
  * The named grid buffers (`cellKey`, `cellVal`, `sortedKey`, `sortedIdx`, `cellHist`, `cellStart`) are the caller's
@@ -39,11 +40,13 @@ export interface GridSpec {
39
40
  readonly g: number;
40
41
  /** `log2(G / GRID_COARSEST_SIDE) + 1`. */
41
42
  readonly levels: number;
42
- /** `G^dim` finest cells; the outside pseudo-cell is index `cells`. */
43
+ /** `G^dim` finest cells; the outside pseudo-cells are indices `cells .. cells + outsideCells - 1`. */
43
44
  readonly cells: number;
44
- /** `cells + 2`: the length of `cellHist` / `cellStart`. */
45
+ /** `2^dim`: one outside pseudo-cell per orthant about the grid centre (issue #90). */
46
+ readonly outsideCells: number;
47
+ /** `cells + outsideCells + 1`: the length of `cellHist` / `cellStart`. */
45
48
  readonly histWords: number;
46
- /** The first cell of every level inside the pyramid: `levelOffsets[0] = 0`, level 0 holds `cells + 1` (the pseudo-cell), level L `(G / 2^L)^dim`. */
49
+ /** The first cell of every level inside the pyramid: `levelOffsets[0] = 0`, level 0 holds `cells + outsideCells` (the pseudo-cells last), level L `(G / 2^L)^dim`. */
47
50
  readonly levelOffsets: readonly number[];
48
51
  /** Every level's cells together: `levelOffsets[levels - 1] + GRID_COARSEST_SIDE^dim`. */
49
52
  readonly pyramidCells: number;
@@ -81,8 +84,8 @@ function floorPow2(x: number): number {
81
84
  * The grid of `n` nodes in `dim` dimensions under the tuning (spec 7.7 geometry table; PD-9): `G = clamp(nextPow2(2 *
82
85
  * ceil(n^(1 / dim))), GRID_MIN_SIDE, floorPow2(gridMax))` where `gridMax` is `gridMax2D` or `gridMax3D`, rounded DOWN
83
86
  * to a power of two so every level's side is an integer (512 and 128 stay; 100 becomes 64); `levels = log2(G /
84
- * GRID_COARSEST_SIDE) + 1`. At the caps: 349,521 pyramid cells in 2D, 2,396,737 in 3D (the design's counts plus the
85
- * pseudo-cell).
87
+ * GRID_COARSEST_SIDE) + 1`. At the caps: 349,524 pyramid cells in 2D, 2,396,744 in 3D (the design's counts plus the
88
+ * 2^dim pseudo-cells).
86
89
  * @param n - the node count (>= 0)
87
90
  * @param dim - 2 or 3
88
91
  * @param tuning - the resolved layout tuning (`gridMax2D`, `gridMax3D`, `deterministic`)
@@ -102,10 +105,11 @@ export function gridSpecFor(
102
105
  levels++;
103
106
  }
104
107
  const cells = g ** dim;
108
+ const outsideCells = 2 ** dim;
105
109
  const levelOffsets: number[] = [0];
106
110
  let s = g;
107
111
  for (let level = 0; level + 1 < levels; level++) {
108
- levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? 1 : 0));
112
+ levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? outsideCells : 0));
109
113
  s /= 2;
110
114
  }
111
115
  return {
@@ -113,7 +117,8 @@ export function gridSpecFor(
113
117
  g,
114
118
  levels,
115
119
  cells,
116
- histWords: cells + 2,
120
+ outsideCells,
121
+ histWords: cells + outsideCells + 1,
117
122
  levelOffsets: Object.freeze(levelOffsets),
118
123
  pyramidCells: levelOffsets[levels - 1] + GRID_COARSEST_SIDE ** dim,
119
124
  deterministic: tuning.deterministic,
@@ -121,7 +126,7 @@ export function gridSpecFor(
121
126
  }
122
127
 
123
128
  /**
124
- * The bytes of the pyramid (spec 7.7: 16 B per cell, every level, the pseudo-cell included): 38,347,792 at the 3D cap.
129
+ * The bytes of the pyramid (spec 7.7: 16 B per cell, every level, the pseudo-cells included): 38,347,904 at the 3D cap.
125
130
  * @param spec - the grid
126
131
  * @returns the byte length
127
132
  */
@@ -155,9 +160,9 @@ export interface GridBuildBindings {
155
160
  readonly sortedKey: Binding;
156
161
  /** `n` words: the sorted node indices. */
157
162
  readonly sortedIdx: Binding;
158
- /** `cells + 2` words: the per-cell counts. */
163
+ /** `histWords` (`cells + 2^dim + 1`) words: the per-cell counts. */
159
164
  readonly cellHist: Binding;
160
- /** `cells + 2` words: the exclusive scan of `cellHist`. */
165
+ /** `histWords` (`cells + 2^dim + 1`) words: the exclusive scan of `cellHist`. */
161
166
  readonly cellStart: Binding;
162
167
  }
163
168
 
@@ -7,8 +7,9 @@
7
7
  * the twin's two compilations -- and then every invocation strips the range `[0, aggregate)` with a binary search
8
8
  * (`upper_bound`) over the scanned degrees to find which entry its arc belongs to. One `atomicAdd` per WORKGROUP
9
9
  * reserves the block's span in the queue (`edgeCount`), the same aggregate lands in `edgeCountUnclamped` (the overflow
10
- * detector, never clamped) and in `frontierDegreeSum` (Beamer's m_f); a lane whose queue position is at or past
11
- * `P.edgeCapacity` writes nothing (the clamp). The queue holds the TARGET vertex of each arc only (PD-24: `parent`
10
+ * detector, never clamped) and in `frontierDegreeSum` (the inspect seam's per-level expansion count, rotated into
11
+ * `prevDegreeSum` by the boundary; Beamer's m_f is `nextDegreeSum`, measured by `bfs-next-degree` -- issue #391);
12
+ * a lane whose queue position is at or past `P.edgeCapacity` writes nothing (the clamp). The queue holds the TARGET vertex of each arc only (PD-24: `parent`
12
13
  * comes from the post-pass). Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan call and the
13
14
  * `workgroupUniformLoad` sit unconditionally after the guard, and the strip loop is bounded by a uniform value.
14
15
  * There is no `TIER` override: a hub row is balanced over all `WG` lanes inside its block, and the small-frontier
@@ -44,7 +45,7 @@ fn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
44
45
  if (lid.x == 0u) {
45
46
  base = atomicAdd(&counters[8], aggregate); // edgeCount: ONE reservation per workgroup, not one per arc
46
47
  atomicAdd(&counters[9], aggregate); // edgeCountUnclamped: the overflow detector (PD-23)
47
- atomicAdd(&counters[2], aggregate); // frontierDegreeSum: Beamer's m_f (P8-T8)
48
+ atomicAdd(&counters[2], aggregate); // frontierDegreeSum: what this level expanded (the inspect seam)
48
49
  }
49
50
  workgroupBarrier();
50
51
  for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
@@ -11,9 +11,10 @@
11
11
  * The winners are packed into the output vertex queue by `bfs-contract`'s workgroup scan and one `atomicAdd` per
12
12
  * workgroup on `nextFrontierCount`, and claim with a plain `atomicStore`: the list holds every vertex once and the
13
13
  * sweep is vertex-parallel, so no two lanes claim one vertex. It adds nothing to `frontierDegreeSum` (a bottom-up
14
- * level expands nothing), which is why `unvisitedDegreeSum` stops falling while bottom-up runs (the selector's
15
- * JSDoc). Uniformity (spec 3.5 rule 1): the guarded walk writes locals, the scan and the reduction run
16
- * unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
14
+ * level expands nothing), which is why that word is not Beamer's m_f: `bfs-next-degree` sums the degree of what
15
+ * this sweep CLAIMS into `nextDegreeSum`, so the boundary's test and its `unvisitedDegreeSum` subtraction are exact
16
+ * on a bottom-up level like any other (issue #391; the selector's JSDoc). Uniformity (spec 3.5 rule 1): the
17
+ * guarded walk writes locals, the scan and the reduction run unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
17
18
  * test/helpers/sabotage.ts are textual edits of it.
18
19
  */
19
20
  export const bfsBottomUpWgsl = /* wgsl */ `
@@ -4,15 +4,15 @@
4
4
  * ONE dispatch, chosen by `frontier-finalize` for a frontier below `P.fusedMax` entries (`path` 2) and for the retry
5
5
  * of a level whose edge queue overflowed (`path` 4, PD-23). One WORKGROUP per frontier entry, the workgroups striding
6
6
  * the entries by the dispatch's group count (`P.stride`): lane 0 reads the entry's row clipped
7
- * to the bound arc window, adds its degree to `frontierDegreeSum` (Beamer's m_f, so P8-T8's test sees fused levels
8
- * too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim inline --
9
- * `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
7
+ * to the bound arc window, adds its degree to `frontierDegreeSum` (so the inspect seam's per-level expansion count
8
+ * covers fused levels too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim
9
+ * inline -- `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
10
10
  * packing the strip's winners into the output vertex queue by the same Hillis-Steele scan and one `atomicAdd` per
11
11
  * strip on `nextFrontierCount`. No edge queue is written or read, which is the whole win for a tiny frontier
12
12
  * (Merrill's fleeting iterations) and what makes the overflow retry exact: the partial edge queue is never consulted.
13
13
  * On a retry level `advance-expand` has already added the frontier's degree to `frontierDegreeSum`, so that word
14
- * holds 2 x m_f for the level and the next boundary's Beamer test and degree-sum subtraction see the doubled value;
15
- * reachable only with a faked capacity or an absurd graph, accepted and said here rather than guarded. Nothing here
14
+ * holds twice the level's expanded degree; since issue #391 no decision reads it (Beamer's m_f is `nextDegreeSum`
15
+ * and the boundary subtracts that), so the doubling only reaches the inspect seam's `prevDegreeSum`. Nothing here
16
16
  * writes a parent (PD-24: the post-pass does), which is what keeps the kernel at the eight-storage-buffer budget
17
17
  * with the four graph slots. Uniformity (spec 3.5 rule 1): the strip loop's bound and the row start are
18
18
  * `workgroupUniformLoad`s, so every barrier of the per-strip append is in uniform control flow; the guarded claim
@@ -44,7 +44,7 @@ fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
44
44
  let d = select(0u, a1 - a0, a1 > a0);
45
45
  wdeg = d;
46
46
  wstart = a0;
47
- atomicAdd(&counters[2], d); // frontierDegreeSum, so Beamer's test (P8-T8) sees fused levels too
47
+ atomicAdd(&counters[2], d); // frontierDegreeSum, so the inspect seam sees fused levels too
48
48
  }
49
49
  let deg = workgroupUniformLoad(&wdeg); // uniform: the loop below may hold barriers
50
50
  let start = workgroupUniformLoad(&wstart);
@@ -0,0 +1,33 @@
1
+ /**
2
+ * The `bfs-next-degree` kernel body (design 8.4; issue #391, the amendment to P8-T8's PD-21): Beamer's m_f measured
3
+ * EXACTLY, at the end of every level, as the out-degree sum of the vertices the level just claimed -- the next
4
+ * frontier, which is what the next boundary decides the direction FOR. It grid-strides over the output vertex
5
+ * queue (`nextFrontierCount`, word 1, the claim kernels' append span; `P.stride` the plan's stride), reads each
6
+ * entry's out-degree from the `outDegree` view, reduces the lane sums with the prelude's `wg_reduce_u32` and lands
7
+ * ONE `atomicAdd` per workgroup in `nextDegreeSum` (word 25), which `frontier-finalize` role 0 reads for the
8
+ * switch-into-bottom-up test, subtracts from `unvisitedDegreeSum` and zeroes for the next level. It runs on every
9
+ * path that claims (the path word 24 non-zero: two-phase, fused, bottom-up, the retry) and does nothing on a level
10
+ * past the end.
11
+ *
12
+ * Why a kernel of its own: before it, the test used `frontierDegreeSum` (word 2), which the EXPANSION of the
13
+ * previous frontier accumulates, so the boundary compared the degree of the frontier it had just finished with the
14
+ * unvisited set, one level stale, and a bottom-up level (which expands nothing) left it at 0. On the 1M / 10M R-MAT
15
+ * that misses the switch at the level that matters: the frontier of 46,524 hubs at level 1 has 13.6M out-arcs, the
16
+ * unvisited set 7.3M, and the boundary saw the source's 86,405 instead -- top-down wrote 13.6M edge-queue entries
17
+ * where the bottom-up sweep reads 0.6M. Measuring the next frontier's degree at claim time is Beamer's own m_f, and
18
+ * the same word makes `unvisitedDegreeSum` exact after a bottom-up level too. Uniformity (spec 3.5 rule 1): the
19
+ * loop holds no barrier (its trip count is per lane), and the reduction runs unconditionally after it. Body only
20
+ * (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
21
+ */
22
+ export const bfsNextDegreeWgsl = /* wgsl */ `
23
+ @compute @workgroup_size(WG)
24
+ fn bfs_next_degree(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
25
+ let count = select(0u, atomicLoad(&counters[1]), atomicLoad(&counters[24]) != 0u); // nextFrontierCount, on a level that claimed (the path word)
26
+ var sum = 0u;
27
+ for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // no barrier inside: the trip count is per lane
28
+ sum = sum + outDegree[frontier[i]];
29
+ }
30
+ let total = wg_reduce_u32(sum, lid.x, 0u); // the prelude's workgroup sum; uniform: after the loop
31
+ if (lid.x == 0u) { atomicAdd(&counters[25], total); } // nextDegreeSum: ONE atomic per workgroup
32
+ }
33
+ `;
@@ -9,6 +9,9 @@
9
9
  * `partials[group].swingTraction`. `pos` is `array<vec4f>` with the mass in `.w`; `force` / `oldForce` are stride-3
10
10
  * `array<f32>` read through the per-body helpers (4.4 rule 6). Normative text, copied verbatim: the P1-T5 sabotage
11
11
  * rows (gravity sign, `k / d2`, the `jj != i` guard, the `.w` mass lane) are textual edits of this string.
12
+ * The j range is split into passes of at most EXACT_TILES_PER_PASS tiles (issue #87: llvmpipe's per-invocation loop
13
+ * budget); an earlier pass adds its partial sum into `force`, the last one runs gravity and the epilogue. A graph of
14
+ * at most 32,768 nodes is one pass, bitwise the single-pass kernel. Record it through recordExactRepulsion.
12
15
  */
13
16
  export const fa2RepulsionExactWgsl = /* wgsl */ `
14
17
  var<workgroup> tile: array<vec4f, WG>; // xyz + mass, 4 KiB at WG = 256
@@ -35,14 +38,23 @@ fn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: cent
35
38
  }
36
39
 
37
40
  @compute @workgroup_size(WG)
38
- fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
41
+ fn repulsion(
42
+ @builtin(workgroup_id) wid: vec3<u32>,
43
+ @builtin(local_invocation_id) lid: vec3<u32>,
44
+ @builtin(num_workgroups) nwg: vec3<u32>,
45
+ ) {
46
+ // issue #87: pass p of the tile range is dispatched with p + 1 z slices; only the last slice works, so the pass
47
+ // index needs no uniform. Uniform: keyed on workgroup_id and num_workgroups only.
48
+ if (wid.z + 1u < nwg.z) { return; }
39
49
  let i = linear_id(wid, lid.x);
40
50
  let valid = i < P.n;
41
51
  var pi = vec4f(0.0);
42
52
  if (valid) { pi = pos[i]; }
43
53
  var f = vec3f(0.0);
44
54
  let tiles = (P.n + WG - 1u) / WG;
45
- for (var t = 0u; t < tiles; t = t + 1u) {
55
+ let tileBegin = (nwg.z - 1u) * EXACT_TILES_PER_PASS;
56
+ let tileEnd = min(tiles, tileBegin + EXACT_TILES_PER_PASS);
57
+ for (var t = tileBegin; t < tileEnd; t = t + 1u) {
46
58
  let j = t * WG + lid.x;
47
59
  if (j < P.n) { tile[lid.x] = pos[j]; } else { tile[lid.x] = vec4f(0.0); } // guarded fill; mass 0 marks the pad
48
60
  workgroupBarrier(); // uniform: every invocation reaches it
@@ -65,6 +77,10 @@ fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
65
77
  }
66
78
  workgroupBarrier();
67
79
  }
80
+ if (tileEnd < tiles) { // an earlier pass: its partial sum only (uniform: P.n, nwg)
81
+ if (valid) { store_force(i, load_force(i) + f); }
82
+ return;
83
+ }
68
84
  // epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it
69
85
  var sw = 0.0;
70
86
  var tr = 0.0;
@@ -61,7 +61,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
61
61
  S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)
62
62
  let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)
63
63
  S.meanDisplacement = meanDisp;
64
- S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);
64
+ S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= min(P.settleThreshold * S.rmsRadius, P.settleFloor)); // relative AND absolute (issue #97)
65
65
  }
66
66
  S.iteration = S.iteration + 1u;
67
67
  T[P.iterationIndex].meanDisplacement = S.meanDisplacement;
@@ -79,7 +79,9 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
79
79
  S.invCellSize = 1.0 / cellSize;
80
80
  S.eps = 0.25 * cellSize;
81
81
  }
82
- S.outsideGrid = cellHist[cells]; // the previous iteration's pseudo-cell count (0 after load)
82
+ var outside = 0u; // the previous iteration's pseudo-cell counts, one per orthant (issue #90; 0 after load)
83
+ for (var o = 0u; o < select(4u, 8u, P.dim == 3u); o = o + 1u) { outside = outside + cellHist[cells + o]; }
84
+ S.outsideGrid = outside;
83
85
  S.maxCellOccupancy = atomicLoad(&hubCounters[1]);
84
86
  atomicStore(&hubCounters[0], 0u);
85
87
  atomicStore(&hubCounters[1], 0u);
@@ -29,25 +29,26 @@
29
29
  * raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
30
30
  * to 0); the relax kernels size themselves from words 0 and 20.
31
31
  *
32
- * Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a switch can be counted at
33
- * the done boundary too, which the host model of the tests mirrors): top-down switches to bottom-up when
34
- * `frontierDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's `max(1, floor(arcCount / n))`
35
- * unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up switches back when
36
- * `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no admitted device
37
- * reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the direction at 0. Every
38
- * change is counted in `switches`, the previous direction is word 14. The two unvisited words the test reads are
39
- * rebuilt exactly once per submit by `bfs-unvisited-flags` (PD-18) and maintained here by subtraction: the count is
40
- * subtracted from the SECOND boundary of a submit on and the degree sum from the THIRD on, because a boundary may
41
- * only subtract what the submit's rebuild counted, and the frontier whose degree sum the second boundary holds was
42
- * claimed before the rebuild ran (the rebuild counts the vertices unclaimed when it runs; the frontier rotated in at
43
- * boundary 0 was claimed by the previous submit's last contract, so it was never in the sum; boundary b subtracts
44
- * `next = |F_b|`, inside the sum iff b >= 1, and `degSum = deg(F_{b-1})`, inside it iff b >= 2). The degree sum is
45
- * the "unvisited degree estimate" of the design rather than an exact count for two reasons: it is one level stale
46
- * (a frontier's degree sum is only known once it has been expanded), and a bottom-up level expands nothing, so the
47
- * word stops falling while bottom-up runs and overstates the set afterwards. The bias is one-directional -- an
48
- * overstated m_u makes the switch INTO bottom-up harder, never easier -- and the next submit's rebuild makes it
49
- * exact again. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
50
- * textual edits of it.
32
+ * Beamer's test (P8-T8, PD-21; amended for issue #391), evaluated at every boundary BEFORE the `done` branch (so a
33
+ * switch can be counted at the done boundary too, which the host model of the tests mirrors): top-down switches to
34
+ * bottom-up when `nextDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's
35
+ * `max(1, floor(arcCount / n))` unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up
36
+ * switches back when `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no
37
+ * admitted device reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the
38
+ * direction at 0. Every change is counted in `switches`, the previous direction is word 14. `nextDegreeSum` (word
39
+ * 25) is Beamer's m_f measured EXACTLY: `bfs-next-degree` sums the out-degrees of the vertices a level claims at
40
+ * the end of that level, so the boundary that rotates them in as `next` compares the degree of the frontier it is
41
+ * about to expand -- not, as before the amendment, `frontierDegreeSum` (word 2), the degree of the frontier the
42
+ * previous level EXPANDED, one level stale and 0 after a bottom-up level, which on the 1M / 10M R-MAT missed the
43
+ * switch at the level holding 13.6M of the 21M arcs. Word 2 is still accumulated by the expansion and rotated into
44
+ * word 4 for the inspect seam. The two unvisited words are rebuilt exactly once per submit by `bfs-unvisited-flags`
45
+ * (PD-18) and maintained here by subtraction from the SECOND boundary of a submit on, because a boundary may only
46
+ * subtract what the submit's rebuild counted: the rebuild counts the vertices unclaimed when it runs, the frontier
47
+ * rotated in at boundary 0 was claimed by the previous submit's last level, so it was never in the sums, and
48
+ * boundary b subtracts `next = |F_b|` and `nextDegreeSum = deg(F_b)`, both inside the sums iff b >= 1. Both words
49
+ * are therefore exact at every boundary, bottom-up levels included (the sweep's claims are summed like any other).
50
+ * Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits
51
+ * of it.
51
52
  */
52
53
  export const frontierFinalizeWgsl = /* wgsl */ `
53
54
  @compute @workgroup_size(WG)
@@ -61,19 +62,19 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
61
62
  let finished = atomicLoad(&counters[0]);
62
63
  let next = atomicLoad(&counters[1]);
63
64
  let degSum = atomicLoad(&counters[2]);
65
+ let nextDeg = atomicLoad(&counters[25]); // deg(F_b), summed by bfs-next-degree when F_b was claimed (issue #391)
64
66
  atomicStore(&counters[3], finished); // prevFrontierCount
65
67
  atomicStore(&counters[4], degSum); // prevDegreeSum
66
68
  atomicStore(&counters[0], next); // the rotation
67
69
  atomicStore(&counters[1], 0u);
68
70
  atomicStore(&counters[2], 0u);
71
+ atomicStore(&counters[25], 0u); // the next level's claims sum from 0
69
72
  atomicStore(&counters[8], 0u); // edgeCount
70
73
  atomicStore(&counters[9], 0u); // edgeCountUnclamped
71
74
  atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount
72
- if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped to 2 (P8-T8, PD-18)
73
- atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount (exact): F_b was inside the submit's rebuilt sum iff b >= 1
74
- }
75
- if (P.firstOfSubmit >= 2u) {
76
- atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2
75
+ if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped (P8-T8, PD-18): F_b was inside the submit's rebuilt sums iff b >= 1
76
+ atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount, exact
77
+ atomicStore(&counters[6], atomicLoad(&counters[6]) - nextDeg); // unvisitedDegreeSum, exact (issue #391: no longer one level stale)
77
78
  }
78
79
  let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0
79
80
  atomicStore(&counters[11], level);
@@ -83,7 +84,7 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
83
84
  if (P.mode == 1u) {
84
85
  direction = 0u; // top-down only (the test seam)
85
86
  } else if (direction == 0u) {
86
- if (degSum > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing
87
+ if (nextDeg > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing, m_f the degree of the frontier about to be expanded
87
88
  } else {
88
89
  if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking
89
90
  }
@@ -1,7 +1,8 @@
1
1
  /**
2
2
  * G1, the `grid-cell-key` kernel body (spec 7.7; P4-T8): the finest cell of every node from the state's robust extent,
3
3
  * `floor((p - gridMin) * invCellSize)` (a multiply, correctly rounded everywhere: PD-10), linearised when every axis
4
- * is in [0, G) and the outside pseudo-cell `cells` otherwise; `cellVal[i] = i`. The clamp before the floor keeps a
4
+ * is in [0, G), and otherwise one of the 2^dim outside pseudo-cells `cells + orthant`, the orthant of the cell about the
5
+ * grid centre (bit a set when `c[a] >= G / 2`; issue #90); `cellVal[i] = i`. The clamp before the floor keeps a
5
6
  * far-away or NaN coordinate out of an out-of-range float-to-int conversion. Body only; normative text.
6
7
  */
7
8
  export const gridCellKeyWgsl = /* wgsl */ `
@@ -18,7 +19,8 @@ fn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocatio
18
19
  let g = i32(P.gridMax);
19
20
  var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;
20
21
  if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }
21
- var key = cells; // the outside pseudo-cell (7.7)
22
+ var key = cells + select(0u, 1u, c.x >= g / 2) + select(0u, 2u, c.y >= g / 2); // an outside pseudo-cell: its orthant (issue #90)
23
+ if (P.dim == 3u) { key = key + select(0u, 4u, c.z >= g / 2); }
22
24
  if (inside) {
23
25
  key = u32(c.x) + P.gridMax * u32(c.y);
24
26
  if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }
@@ -1,5 +1,6 @@
1
1
  /**
2
- * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the pseudo-cell included; the
2
+ * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the 2^dim outside pseudo-cells
3
+ * included (issue #90); the
3
4
  * mass-weighted position sum of a cell's sorted range in index order (no atomics: deterministic), the largest
4
5
  * occupancy into hubCounters[1], and cells above GRID_HUB_CELL entries appended to hubList for G4b (PD-13). Body
5
6
  * only; normative text.
@@ -10,7 +11,7 @@ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.
10
11
  @compute @workgroup_size(WG)
11
12
  fn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
12
13
  let c = linear_id(wid, lid.x);
13
- if (c > grid_cells()) { return; } // cells [0, cells]: the pseudo-cell is index cells; no barrier follows
14
+ if (c >= grid_cells() + select(4u, 8u, P.dim == 3u)) { return; } // cells [0, cells + 2^dim): the pseudo-cells follow the real ones; no barrier follows
14
15
  let start = cellStart[c];
15
16
  let count = cellStart[c + 1u] - start;
16
17
  atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * G5, the `grid-downsample` kernel body (spec 7.7; P4-T9): one dispatch per coarser level; every parent cell is the
3
3
  * sum of its 4 (2D) or 8 (3D) children at the level below, read at P.childBase and written at P.parentBase (the
4
- * pseudo-cell, index cells of level 0, is never a child). No atomics. Body only; normative text.
4
+ * pseudo-cells, indices cells .. of level 0, are never children). No atomics. Body only; normative text.
5
5
  */
6
6
  export const gridDownsampleWgsl = /* wgsl */ `
7
7
  @compute @workgroup_size(WG)
@@ -2,9 +2,11 @@
2
2
  * G6, the `grid-far-field` kernel body (spec 7.7; P4-T10; D24): per node `i = sortedIdx[t]`, its finest cell
3
3
  * recomputed from `pos[i]` and the state (PD-10); for an inside node the coarsest level minus the 3x3 (3x3x3)
4
4
  * around its coarsest cell, then at every finer level the 6x6 (6x6x6) block that is the parent's 3x3 minus this
5
- * level's own 3x3 -- space tiled exactly once, no theta -- plus the outside pseudo-cell's centroid; for an outside
6
- * node the coarsest level in full and no pseudo-cell. Every cell term is the per-cell law on the mass-weighted
7
- * centroid (Gephi Region semantics), softened by `eps^2`: `LAW` 0 (FA2) `d * (k m_i M / d2)`, `LAW` 1 (FR, 7.20)
5
+ * level's own 3x3 -- space tiled exactly once, no theta -- plus the centroid of each of the 2^dim outside
6
+ * pseudo-cells, one per orthant about the grid centre (issue #90); for an outside node the coarsest level in full and
7
+ * no pseudo-cell. Every cell term is the per-cell law on the mass-weighted
8
+ * centroid (Gephi Region semantics), softened by `eps^2`: `LAW` 0 (FA2) `d * (k m_i M / d2)` with `|d|^2` first
9
+ * floored at 0.01^2 like K3's pair law (issue #89), `LAW` 1 (FR, 7.20)
8
10
  * `d * (k^2 M / d2)` (mass 1 per node, so `M` is the cell's count), `LAW` 2 (coulomb) `d * (-g m_i M / d2^1.5)`
9
11
  * (P4-T13, PD-22). `force += f` (K2 wrote it). The loop bounds are `P.levels` and `P.gridMax` from the uniform,
10
12
  * not a `LEVELS` override (PD-16, DEP-P4-G). Body only; normative text.
@@ -18,11 +20,12 @@ fn store_force(i: u32, f: vec3f) {
18
20
  }
19
21
  fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
20
22
  fn grid_side(level: u32) -> u32 { return P.gridMax >> level; }
21
- fn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cell at index cells)
23
+ fn outside_cells() -> u32 { return select(4u, 8u, P.dim == 3u); } // one pseudo-cell per orthant (issue #90)
24
+ fn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cells at cells ..)
22
25
  var base = 0u;
23
26
  for (var l = 0u; l < level; l = l + 1u) {
24
27
  let s = grid_side(l);
25
- base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, 1u, l == 0u);
28
+ base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, outside_cells(), l == 0u);
26
29
  }
27
30
  return base;
28
31
  }
@@ -33,7 +36,9 @@ fn cell_at(level: u32, cx: i32, cy: i32, cz: i32) -> u32 {
33
36
  fn cell_force(pi: vec4f, q: vec4f) -> vec3f { // one far-field term, softened by state.eps (7.7)
34
37
  if (q.w <= 0.0) { return vec3f(0.0); } // an empty cell
35
38
  let d = pi.xyz - q.xyz / q.w; // to the mass-weighted centroid
36
- let d2 = dot(d, d) + S.eps * S.eps;
39
+ var d2 = dot(d, d);
40
+ if (LAW == 0u) { d2 = max(d2, FA2_DIST_FLOOR_SQ); } // FA2 alone floors d >= 0.01, as K3 and G7 do (issue #89); FR and coulomb are unfloored (7.20)
41
+ d2 = d2 + S.eps * S.eps;
37
42
  if (LAW == 1u) { return d * (P.frK * P.frK * q.w / d2); } // LAW 1 (FR, 7.20): k^2 / d per node, q.w nodes at the centroid
38
43
  if (LAW == 2u) { return d * (-P.coulomb * pi.w * q.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb): -g m_i M_cell / d^2
39
44
  return d * (P.scalingRatio * pi.w * q.w / d2); // LAW 0 (FA2): |F| = k m_i M_cell / d
@@ -82,9 +87,11 @@ fn grid_far_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
82
87
  }
83
88
  }
84
89
  }
85
- f = f + cell_force(pi, pyramid[grid_cells()]); // the outside pseudo-cell as one far-field term
90
+ for (var o = 0u; o < outside_cells(); o = o + 1u) { // every outside pseudo-cell: one far-field term per orthant
91
+ f = f + cell_force(pi, pyramid[grid_cells() + o]);
92
+ }
86
93
  } else {
87
- for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (it would include itself)
94
+ for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (G7 sums them pair by pair)
88
95
  for (var cy = 0; cy < ts; cy = cy + 1) {
89
96
  for (var cx = 0; cx < ts; cx = cx + 1) {
90
97
  f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);
@@ -3,7 +3,7 @@
3
3
  * exact pair law of K3 (`LAW` 0: `|F| = k m_i m_j / d` with the 0.01 floor; `LAW` 1: FR's unfloored `k^2 / d`;
4
4
  * `LAW` 2: the unfloored coulomb `-g m_i m_j / d^2`; the antisymmetric coincident kick at the law's magnitude at
5
5
  * d = 0.01, PD-22) over the 9 (27) finest
6
- * cells around its own, or over the outside pseudo-cell alone for an outside node; a cell above `nearMax` entries
6
+ * cells around its own, or over the 2^dim outside pseudo-cells (one per orthant, issue #90) for an outside node; a cell above `nearMax` entries
7
7
  * is sampled by `nearMax` INDEPENDENT draws with replacement, draw `k` reading the slot
8
8
  * `lowbias32(((c ^ (iteration * 0x9E3779B9)) ^ seed) ^ (k * 0x85EBCA6B)) % count` (every slot's inclusion
9
9
  * probability is `nearMax / count` whatever its position in the sorted order, so a duplicated draw is counted twice
@@ -103,7 +103,11 @@ fn grid_near_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocat
103
103
  }
104
104
  }
105
105
  } else {
106
- f = cell_sum(i, pi, grid_cells(), true); // an outside node: the pseudo-cell alone
106
+ var own = grid_cells() + select(0u, 1u, c0.x >= g / 2) + select(0u, 2u, c0.y >= g / 2); // G1's orthant key
107
+ if (P.dim == 3u) { own = own + select(0u, 4u, c0.z >= g / 2); }
108
+ for (var o = grid_cells(); o < grid_cells() + select(4u, 8u, P.dim == 3u); o = o + 1u) { // an outside node: every outside pseudo-cell
109
+ f = f + cell_sum(i, pi, o, o == own);
110
+ }
107
111
  }
108
112
  }
109
113
  // epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it (K3's text)
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * The `histogram` kernel body (spec 6 row 5; P4-T3): one atomicAdd per key into the global `hist` (zeroed by a fill
3
3
  * dispatch earlier in the pass, PD-4). Order-independent, hence deterministic. A key >= P.bins is not counted (the
4
- * caller's contract; the grid's keys are always < cells + 1). Body only; normative text.
4
+ * caller's contract; the grid's keys are always < cells + 2^dim). Body only; normative text.
5
5
  */
6
6
  export const histogramWgsl = /* wgsl */ `
7
7
  @compute @workgroup_size(WG)