@graphty/webgpu-graph-algorithms 0.6.5 → 0.6.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +38 -17
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-hzGggHeM.js → context-DiSr6eiz.js} +32 -18
- package/dist/chunks/context-DiSr6eiz.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/algorithms/bfs.d.ts +10 -6
- package/dist/src/algorithms/bfs.d.ts.map +1 -1
- package/dist/src/algorithms/bfs.js +32 -9
- package/dist/src/algorithms/bfs.js.map +1 -1
- package/dist/src/constants.d.ts +33 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +33 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernel/dispatch.d.ts +2 -2
- package/dist/src/kernel/dispatch.d.ts.map +1 -1
- package/dist/src/kernel/kernel.d.ts +1 -1
- package/dist/src/kernel/kernel.js +2 -2
- package/dist/src/kernel/kernel.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +2 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernels.d.ts +10 -7
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +33 -9
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
- package/dist/src/layouts/forceatlas2.js +2 -1
- package/dist/src/layouts/forceatlas2.js.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.js +4 -2
- package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
- package/dist/src/layouts/repulsion-exact.d.ts +16 -0
- package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
- package/dist/src/layouts/repulsion-exact.js +21 -1
- package/dist/src/layouts/repulsion-exact.js.map +1 -1
- package/dist/src/layouts/repulsion-grid.d.ts +1 -1
- package/dist/src/layouts/repulsion-grid.js +1 -1
- package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
- package/dist/src/layouts/spring-electrical.js +6 -2
- package/dist/src/layouts/spring-electrical.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +1 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -1
- package/dist/src/primitives/frontier.js +1 -0
- package/dist/src/primitives/frontier.js.map +1 -1
- package/dist/src/primitives/grid-pyramid.d.ts +4 -4
- package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
- package/dist/src/primitives/grid-pyramid.js +4 -3
- package/dist/src/primitives/grid-pyramid.js.map +1 -1
- package/dist/src/primitives/grid.d.ts +13 -10
- package/dist/src/primitives/grid.d.ts.map +1 -1
- package/dist/src/primitives/grid.js +10 -7
- package/dist/src/primitives/grid.js.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +21 -20
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.js +26 -25
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
- package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
- package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
- package/dist/src/wgsl/histogram.wgsl.js +1 -1
- package/dist/webgpu-graph-algorithms.js +124 -40
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +3 -3
- package/src/algorithms/bfs.ts +33 -9
- package/src/algorithms/pagerank.ts +19 -5
- package/src/algorithms/power-iteration.ts +8 -2
- package/src/constants.ts +35 -0
- package/src/kernel/dispatch.ts +2 -2
- package/src/kernel/kernel.ts +2 -2
- package/src/kernel/prelude.ts +2 -0
- package/src/kernels.ts +35 -9
- package/src/layouts/forceatlas2.ts +2 -0
- package/src/layouts/fruchterman-reingold.ts +4 -1
- package/src/layouts/repulsion-exact.ts +29 -1
- package/src/layouts/repulsion-grid.ts +1 -1
- package/src/layouts/spring-electrical.ts +8 -1
- package/src/memory/residency.ts +14 -4
- package/src/primitives/frontier.ts +2 -0
- package/src/primitives/grid-pyramid.ts +6 -5
- package/src/primitives/grid.ts +17 -12
- package/src/wgsl/advance-expand.wgsl.ts +4 -3
- package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
- package/src/wgsl/bfs-fused.wgsl.ts +6 -6
- package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
- package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
- package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
- package/src/wgsl/frontier-finalize.wgsl.ts +26 -25
- package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
- package/src/wgsl/grid-centroid.wgsl.ts +3 -2
- package/src/wgsl/grid-downsample.wgsl.ts +1 -1
- package/src/wgsl/grid-far-field.wgsl.ts +15 -8
- package/src/wgsl/grid-near-field.wgsl.ts +6 -2
- package/src/wgsl/histogram.wgsl.ts +1 -1
- package/dist/chunks/context-hzGggHeM.js.map +0 -1
package/src/memory/residency.ts
CHANGED
|
@@ -579,7 +579,9 @@ export class GraphResidency {
|
|
|
579
579
|
}
|
|
580
580
|
|
|
581
581
|
/**
|
|
582
|
-
* One resident per array (spec 4.3: views upload in perArray mode, never into the arena).
|
|
582
|
+
* One resident per array (spec 4.3: views upload in perArray mode, never into the arena). An empty array (the
|
|
583
|
+
* colIdx of an edgeless directed reverse view, the src / dst of an edgeless edgeList) is skipped: spec 5.6 never
|
|
584
|
+
* uploads a zero-length array, and Kernel.bind rejects a zero-size binding, so it is absent as in core().
|
|
583
585
|
* @param record - the owning record
|
|
584
586
|
* @param arrays - the named arrays
|
|
585
587
|
* @param label - the buffer label prefix
|
|
@@ -592,6 +594,9 @@ export class GraphResidency {
|
|
|
592
594
|
): Readonly<Record<string, Binding>> {
|
|
593
595
|
const bindings: Record<string, Binding> = {};
|
|
594
596
|
for (const [name, array] of arrays) {
|
|
597
|
+
if (array.byteLength === 0) {
|
|
598
|
+
continue;
|
|
599
|
+
}
|
|
595
600
|
const resident = this.upload(record, array, array, `${label}:${name}`);
|
|
596
601
|
bindings[name] = { buffer: resident.buffer, offset: 0, size: resident.byteLength, window: null };
|
|
597
602
|
}
|
|
@@ -606,19 +611,24 @@ export class GraphResidency {
|
|
|
606
611
|
* lengths, so keying the packed buffer on `rev.rowPtr` would make the packed and the unpacked view of one
|
|
607
612
|
* snapshot collide -- whichever was built second would get the other's buffer. The record still owns the
|
|
608
613
|
* resident, so release(s) destroys it with the rest. Offsets are STORAGE_ALIGN-aligned because Kernel.bind
|
|
609
|
-
* rejects any other offset synchronously (E_INVALID_ARGUMENT { argument: "offset" }).
|
|
614
|
+
* rejects any other offset synchronously (E_INVALID_ARGUMENT { argument: "offset" }). Empty arrays are left
|
|
615
|
+
* out as in separateArrays; when nothing is left, nothing is uploaded.
|
|
610
616
|
* @param record - the owning record
|
|
611
|
-
* @param
|
|
617
|
+
* @param all - the named arrays, in buffer order
|
|
612
618
|
* @param key - the marker object the resident is keyed on
|
|
613
619
|
* @param label - the buffer label
|
|
614
620
|
* @returns the bindings by name, all into the one buffer
|
|
615
621
|
*/
|
|
616
622
|
private packArrays(
|
|
617
623
|
record: ResidencyRecord,
|
|
618
|
-
|
|
624
|
+
all: readonly (readonly [string, TypedArrayData])[],
|
|
619
625
|
key: object,
|
|
620
626
|
label: string,
|
|
621
627
|
): Readonly<Record<string, Binding>> {
|
|
628
|
+
const arrays = all.filter(([, array]) => array.byteLength > 0);
|
|
629
|
+
if (arrays.length === 0) {
|
|
630
|
+
return Object.freeze({});
|
|
631
|
+
}
|
|
622
632
|
const offsets: number[] = [];
|
|
623
633
|
let total = 0;
|
|
624
634
|
for (const [, array] of arrays) {
|
|
@@ -71,6 +71,7 @@ export const W: Readonly<{
|
|
|
71
71
|
thresholdBits: 22;
|
|
72
72
|
deltaBits: 23;
|
|
73
73
|
path: 24;
|
|
74
|
+
nextDegreeSum: 25;
|
|
74
75
|
}> = Object.freeze({
|
|
75
76
|
frontierCount: 0,
|
|
76
77
|
nextFrontierCount: 1,
|
|
@@ -97,6 +98,7 @@ export const W: Readonly<{
|
|
|
97
98
|
thresholdBits: 22,
|
|
98
99
|
deltaBits: 23,
|
|
99
100
|
path: 24,
|
|
101
|
+
nextDegreeSum: 25,
|
|
100
102
|
});
|
|
101
103
|
|
|
102
104
|
/** The words a `reset` seeds (every other word is zeroed). */
|
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The grid pyramid (spec 6 row 12, 7.7 G4-G5; P4-T9): the planner that records, into the caller's pass, the finest
|
|
3
|
-
* centroids (G4, `grid-centroid`: thread per cell over `cells +
|
|
3
|
+
* centroids (G4, `grid-centroid`: thread per cell over `cells + outsideCells`, the pseudo-cells included), the hub-cell
|
|
4
4
|
* completion (G4a: the T1 `indirect-finalize` over `hubCounters[0]` into `hubArgs` with `wg = 1`, so the finalize's
|
|
5
5
|
* `ceil(count / wg)` is ONE workgroup per hub cell; G4b: `grid-centroid-hub`, one workgroup per hub cell, dispatched
|
|
6
6
|
* indirectly; PD-13, DEP-P4-I) and one `grid-downsample` dispatch per coarser
|
|
7
7
|
* level (G5). Level 0 holds `[sum m x, sum m y, sum m z, sum m]` per cell; every parent is the sum of its 2^dim
|
|
8
|
-
* children; the pseudo-
|
|
8
|
+
* children; the pseudo-cells (indices `cells ..` of level 0) are never children. No atomics touch the sums (design 6 row 12:
|
|
9
9
|
* bitwise reproducible); the only atomics are the hub append and the occupancy max.
|
|
10
10
|
*
|
|
11
11
|
* The named grid buffers (`pyramid`, `hubList`, `hubCounters`, `hubArgs`) are the caller's (the model's
|
|
@@ -34,9 +34,9 @@ export interface GridPyramidBindings {
|
|
|
34
34
|
readonly params: Binding;
|
|
35
35
|
/** `n` words: the sorted node indices (the T8 build). */
|
|
36
36
|
readonly sortedIdx: Binding;
|
|
37
|
-
/** `cells + 2` words: the exclusive scan of the cell histogram (the T8 build). */
|
|
37
|
+
/** `histWords` (`cells + 2^dim + 1`) words: the exclusive scan of the cell histogram (the T8 build). */
|
|
38
38
|
readonly cellStart: Binding;
|
|
39
|
-
/** `pyramidCells` vec4f: every level, level 0 first with the pseudo-
|
|
39
|
+
/** `pyramidCells` vec4f: every level, level 0 first with the 2^dim orthant pseudo-cells from index `cells`. */
|
|
40
40
|
readonly pyramid: Binding;
|
|
41
41
|
/** The hub cells' indices, appended by G4 (at least one word; at most `floor(n / (GRID_HUB_CELL + 1))` are ever written, so `ceil(n / GRID_HUB_CELL)` words always suffice). */
|
|
42
42
|
readonly hubList: Binding;
|
|
@@ -204,7 +204,8 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
|
|
|
204
204
|
}
|
|
205
205
|
const { centroid, finalize, hub, downsample } = this.kernels;
|
|
206
206
|
const one: DispatchPlan = { x: 1, y: 1, z: 1, items: 1, stride: null };
|
|
207
|
-
|
|
207
|
+
const level0 = spec.cells + spec.outsideCells;
|
|
208
|
+
centroid.dispatch(pass, bound.centroid, plan1d(level0, scope.workgroupSize, scope.caps), [paramsOffset]);
|
|
208
209
|
finalize.dispatch(pass, bound.finalize, one, [bound.finalizeOffset]);
|
|
209
210
|
hub.dispatchIndirect(pass, bound.hub, bound.hubArgs, 0, [paramsOffset]);
|
|
210
211
|
this.dispatches = 3;
|
package/src/primitives/grid.ts
CHANGED
|
@@ -3,8 +3,9 @@
|
|
|
3
3
|
* the caller's pass, the cell keys (G1, `grid-cell-key`), the stable sort by key (G2: `radixSort` at GRID_SORT_BITS,
|
|
4
4
|
* or `countingSortByKey` when the caller asks for the set-deterministic path) and the per-cell histogram with its
|
|
5
5
|
* exclusive scan (G3: the `histogram` kernel over `cellKey` and the `scan` of it; DEP-P4-I names no grid-specific
|
|
6
|
-
* id). `cellHist` and `cellStart` hold `cells + 2` words: every real cell, the outside
|
|
7
|
-
*
|
|
6
|
+
* id). `cellHist` and `cellStart` hold `histWords = cells + 2^dim + 1` words: every real cell, the 2^dim outside
|
|
7
|
+
* pseudo-cells (one per orthant about the grid centre, issue #90) from index `cells`, and one more so
|
|
8
|
+
* `cellStart[histWords - 1] === n` closes the last range. Every zeroing is a `fill` dispatch inside the
|
|
8
9
|
* pass (PD-12), never an encoder clear.
|
|
9
10
|
*
|
|
10
11
|
* The named grid buffers (`cellKey`, `cellVal`, `sortedKey`, `sortedIdx`, `cellHist`, `cellStart`) are the caller's
|
|
@@ -39,11 +40,13 @@ export interface GridSpec {
|
|
|
39
40
|
readonly g: number;
|
|
40
41
|
/** `log2(G / GRID_COARSEST_SIDE) + 1`. */
|
|
41
42
|
readonly levels: number;
|
|
42
|
-
/** `G^dim` finest cells; the outside pseudo-
|
|
43
|
+
/** `G^dim` finest cells; the outside pseudo-cells are indices `cells .. cells + outsideCells - 1`. */
|
|
43
44
|
readonly cells: number;
|
|
44
|
-
/** `
|
|
45
|
+
/** `2^dim`: one outside pseudo-cell per orthant about the grid centre (issue #90). */
|
|
46
|
+
readonly outsideCells: number;
|
|
47
|
+
/** `cells + outsideCells + 1`: the length of `cellHist` / `cellStart`. */
|
|
45
48
|
readonly histWords: number;
|
|
46
|
-
/** The first cell of every level inside the pyramid: `levelOffsets[0] = 0`, level 0 holds `cells +
|
|
49
|
+
/** The first cell of every level inside the pyramid: `levelOffsets[0] = 0`, level 0 holds `cells + outsideCells` (the pseudo-cells last), level L `(G / 2^L)^dim`. */
|
|
47
50
|
readonly levelOffsets: readonly number[];
|
|
48
51
|
/** Every level's cells together: `levelOffsets[levels - 1] + GRID_COARSEST_SIDE^dim`. */
|
|
49
52
|
readonly pyramidCells: number;
|
|
@@ -81,8 +84,8 @@ function floorPow2(x: number): number {
|
|
|
81
84
|
* The grid of `n` nodes in `dim` dimensions under the tuning (spec 7.7 geometry table; PD-9): `G = clamp(nextPow2(2 *
|
|
82
85
|
* ceil(n^(1 / dim))), GRID_MIN_SIDE, floorPow2(gridMax))` where `gridMax` is `gridMax2D` or `gridMax3D`, rounded DOWN
|
|
83
86
|
* to a power of two so every level's side is an integer (512 and 128 stay; 100 becomes 64); `levels = log2(G /
|
|
84
|
-
* GRID_COARSEST_SIDE) + 1`. At the caps: 349,
|
|
85
|
-
* pseudo-
|
|
87
|
+
* GRID_COARSEST_SIDE) + 1`. At the caps: 349,524 pyramid cells in 2D, 2,396,744 in 3D (the design's counts plus the
|
|
88
|
+
* 2^dim pseudo-cells).
|
|
86
89
|
* @param n - the node count (>= 0)
|
|
87
90
|
* @param dim - 2 or 3
|
|
88
91
|
* @param tuning - the resolved layout tuning (`gridMax2D`, `gridMax3D`, `deterministic`)
|
|
@@ -102,10 +105,11 @@ export function gridSpecFor(
|
|
|
102
105
|
levels++;
|
|
103
106
|
}
|
|
104
107
|
const cells = g ** dim;
|
|
108
|
+
const outsideCells = 2 ** dim;
|
|
105
109
|
const levelOffsets: number[] = [0];
|
|
106
110
|
let s = g;
|
|
107
111
|
for (let level = 0; level + 1 < levels; level++) {
|
|
108
|
-
levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ?
|
|
112
|
+
levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? outsideCells : 0));
|
|
109
113
|
s /= 2;
|
|
110
114
|
}
|
|
111
115
|
return {
|
|
@@ -113,7 +117,8 @@ export function gridSpecFor(
|
|
|
113
117
|
g,
|
|
114
118
|
levels,
|
|
115
119
|
cells,
|
|
116
|
-
|
|
120
|
+
outsideCells,
|
|
121
|
+
histWords: cells + outsideCells + 1,
|
|
117
122
|
levelOffsets: Object.freeze(levelOffsets),
|
|
118
123
|
pyramidCells: levelOffsets[levels - 1] + GRID_COARSEST_SIDE ** dim,
|
|
119
124
|
deterministic: tuning.deterministic,
|
|
@@ -121,7 +126,7 @@ export function gridSpecFor(
|
|
|
121
126
|
}
|
|
122
127
|
|
|
123
128
|
/**
|
|
124
|
-
* The bytes of the pyramid (spec 7.7: 16 B per cell, every level, the pseudo-
|
|
129
|
+
* The bytes of the pyramid (spec 7.7: 16 B per cell, every level, the pseudo-cells included): 38,347,904 at the 3D cap.
|
|
125
130
|
* @param spec - the grid
|
|
126
131
|
* @returns the byte length
|
|
127
132
|
*/
|
|
@@ -155,9 +160,9 @@ export interface GridBuildBindings {
|
|
|
155
160
|
readonly sortedKey: Binding;
|
|
156
161
|
/** `n` words: the sorted node indices. */
|
|
157
162
|
readonly sortedIdx: Binding;
|
|
158
|
-
/** `cells + 2` words: the per-cell counts. */
|
|
163
|
+
/** `histWords` (`cells + 2^dim + 1`) words: the per-cell counts. */
|
|
159
164
|
readonly cellHist: Binding;
|
|
160
|
-
/** `cells + 2` words: the exclusive scan of `cellHist`. */
|
|
165
|
+
/** `histWords` (`cells + 2^dim + 1`) words: the exclusive scan of `cellHist`. */
|
|
161
166
|
readonly cellStart: Binding;
|
|
162
167
|
}
|
|
163
168
|
|
|
@@ -7,8 +7,9 @@
|
|
|
7
7
|
* the twin's two compilations -- and then every invocation strips the range `[0, aggregate)` with a binary search
|
|
8
8
|
* (`upper_bound`) over the scanned degrees to find which entry its arc belongs to. One `atomicAdd` per WORKGROUP
|
|
9
9
|
* reserves the block's span in the queue (`edgeCount`), the same aggregate lands in `edgeCountUnclamped` (the overflow
|
|
10
|
-
* detector, never clamped) and in `frontierDegreeSum` (
|
|
11
|
-
* `
|
|
10
|
+
* detector, never clamped) and in `frontierDegreeSum` (the inspect seam's per-level expansion count, rotated into
|
|
11
|
+
* `prevDegreeSum` by the boundary; Beamer's m_f is `nextDegreeSum`, measured by `bfs-next-degree` -- issue #391);
|
|
12
|
+
* a lane whose queue position is at or past `P.edgeCapacity` writes nothing (the clamp). The queue holds the TARGET vertex of each arc only (PD-24: `parent`
|
|
12
13
|
* comes from the post-pass). Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan call and the
|
|
13
14
|
* `workgroupUniformLoad` sit unconditionally after the guard, and the strip loop is bounded by a uniform value.
|
|
14
15
|
* There is no `TIER` override: a hub row is balanced over all `WG` lanes inside its block, and the small-frontier
|
|
@@ -44,7 +45,7 @@ fn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
|
|
|
44
45
|
if (lid.x == 0u) {
|
|
45
46
|
base = atomicAdd(&counters[8], aggregate); // edgeCount: ONE reservation per workgroup, not one per arc
|
|
46
47
|
atomicAdd(&counters[9], aggregate); // edgeCountUnclamped: the overflow detector (PD-23)
|
|
47
|
-
atomicAdd(&counters[2], aggregate); // frontierDegreeSum:
|
|
48
|
+
atomicAdd(&counters[2], aggregate); // frontierDegreeSum: what this level expanded (the inspect seam)
|
|
48
49
|
}
|
|
49
50
|
workgroupBarrier();
|
|
50
51
|
for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
|
|
@@ -11,9 +11,10 @@
|
|
|
11
11
|
* The winners are packed into the output vertex queue by `bfs-contract`'s workgroup scan and one `atomicAdd` per
|
|
12
12
|
* workgroup on `nextFrontierCount`, and claim with a plain `atomicStore`: the list holds every vertex once and the
|
|
13
13
|
* sweep is vertex-parallel, so no two lanes claim one vertex. It adds nothing to `frontierDegreeSum` (a bottom-up
|
|
14
|
-
* level expands nothing), which is why
|
|
15
|
-
*
|
|
16
|
-
*
|
|
14
|
+
* level expands nothing), which is why that word is not Beamer's m_f: `bfs-next-degree` sums the degree of what
|
|
15
|
+
* this sweep CLAIMS into `nextDegreeSum`, so the boundary's test and its `unvisitedDegreeSum` subtraction are exact
|
|
16
|
+
* on a bottom-up level like any other (issue #391; the selector's JSDoc). Uniformity (spec 3.5 rule 1): the
|
|
17
|
+
* guarded walk writes locals, the scan and the reduction run unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
|
|
17
18
|
* test/helpers/sabotage.ts are textual edits of it.
|
|
18
19
|
*/
|
|
19
20
|
export const bfsBottomUpWgsl = /* wgsl */ `
|
|
@@ -4,15 +4,15 @@
|
|
|
4
4
|
* ONE dispatch, chosen by `frontier-finalize` for a frontier below `P.fusedMax` entries (`path` 2) and for the retry
|
|
5
5
|
* of a level whose edge queue overflowed (`path` 4, PD-23). One WORKGROUP per frontier entry, the workgroups striding
|
|
6
6
|
* the entries by the dispatch's group count (`P.stride`): lane 0 reads the entry's row clipped
|
|
7
|
-
* to the bound arc window, adds its degree to `frontierDegreeSum` (
|
|
8
|
-
* too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim
|
|
9
|
-
* `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
|
|
7
|
+
* to the bound arc window, adds its degree to `frontierDegreeSum` (so the inspect seam's per-level expansion count
|
|
8
|
+
* covers fused levels too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim
|
|
9
|
+
* inline -- `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
|
|
10
10
|
* packing the strip's winners into the output vertex queue by the same Hillis-Steele scan and one `atomicAdd` per
|
|
11
11
|
* strip on `nextFrontierCount`. No edge queue is written or read, which is the whole win for a tiny frontier
|
|
12
12
|
* (Merrill's fleeting iterations) and what makes the overflow retry exact: the partial edge queue is never consulted.
|
|
13
13
|
* On a retry level `advance-expand` has already added the frontier's degree to `frontierDegreeSum`, so that word
|
|
14
|
-
* holds
|
|
15
|
-
*
|
|
14
|
+
* holds twice the level's expanded degree; since issue #391 no decision reads it (Beamer's m_f is `nextDegreeSum`
|
|
15
|
+
* and the boundary subtracts that), so the doubling only reaches the inspect seam's `prevDegreeSum`. Nothing here
|
|
16
16
|
* writes a parent (PD-24: the post-pass does), which is what keeps the kernel at the eight-storage-buffer budget
|
|
17
17
|
* with the four graph slots. Uniformity (spec 3.5 rule 1): the strip loop's bound and the row start are
|
|
18
18
|
* `workgroupUniformLoad`s, so every barrier of the per-strip append is in uniform control flow; the guarded claim
|
|
@@ -44,7 +44,7 @@ fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
|
|
|
44
44
|
let d = select(0u, a1 - a0, a1 > a0);
|
|
45
45
|
wdeg = d;
|
|
46
46
|
wstart = a0;
|
|
47
|
-
atomicAdd(&counters[2], d); // frontierDegreeSum, so
|
|
47
|
+
atomicAdd(&counters[2], d); // frontierDegreeSum, so the inspect seam sees fused levels too
|
|
48
48
|
}
|
|
49
49
|
let deg = workgroupUniformLoad(&wdeg); // uniform: the loop below may hold barriers
|
|
50
50
|
let start = workgroupUniformLoad(&wstart);
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bfs-next-degree` kernel body (design 8.4; issue #391, the amendment to P8-T8's PD-21): Beamer's m_f measured
|
|
3
|
+
* EXACTLY, at the end of every level, as the out-degree sum of the vertices the level just claimed -- the next
|
|
4
|
+
* frontier, which is what the next boundary decides the direction FOR. It grid-strides over the output vertex
|
|
5
|
+
* queue (`nextFrontierCount`, word 1, the claim kernels' append span; `P.stride` the plan's stride), reads each
|
|
6
|
+
* entry's out-degree from the `outDegree` view, reduces the lane sums with the prelude's `wg_reduce_u32` and lands
|
|
7
|
+
* ONE `atomicAdd` per workgroup in `nextDegreeSum` (word 25), which `frontier-finalize` role 0 reads for the
|
|
8
|
+
* switch-into-bottom-up test, subtracts from `unvisitedDegreeSum` and zeroes for the next level. It runs on every
|
|
9
|
+
* path that claims (the path word 24 non-zero: two-phase, fused, bottom-up, the retry) and does nothing on a level
|
|
10
|
+
* past the end.
|
|
11
|
+
*
|
|
12
|
+
* Why a kernel of its own: before it, the test used `frontierDegreeSum` (word 2), which the EXPANSION of the
|
|
13
|
+
* previous frontier accumulates, so the boundary compared the degree of the frontier it had just finished with the
|
|
14
|
+
* unvisited set, one level stale, and a bottom-up level (which expands nothing) left it at 0. On the 1M / 10M R-MAT
|
|
15
|
+
* that misses the switch at the level that matters: the frontier of 46,524 hubs at level 1 has 13.6M out-arcs, the
|
|
16
|
+
* unvisited set 7.3M, and the boundary saw the source's 86,405 instead -- top-down wrote 13.6M edge-queue entries
|
|
17
|
+
* where the bottom-up sweep reads 0.6M. Measuring the next frontier's degree at claim time is Beamer's own m_f, and
|
|
18
|
+
* the same word makes `unvisitedDegreeSum` exact after a bottom-up level too. Uniformity (spec 3.5 rule 1): the
|
|
19
|
+
* loop holds no barrier (its trip count is per lane), and the reduction runs unconditionally after it. Body only
|
|
20
|
+
* (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
21
|
+
*/
|
|
22
|
+
export const bfsNextDegreeWgsl = /* wgsl */ `
|
|
23
|
+
@compute @workgroup_size(WG)
|
|
24
|
+
fn bfs_next_degree(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
25
|
+
let count = select(0u, atomicLoad(&counters[1]), atomicLoad(&counters[24]) != 0u); // nextFrontierCount, on a level that claimed (the path word)
|
|
26
|
+
var sum = 0u;
|
|
27
|
+
for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // no barrier inside: the trip count is per lane
|
|
28
|
+
sum = sum + outDegree[frontier[i]];
|
|
29
|
+
}
|
|
30
|
+
let total = wg_reduce_u32(sum, lid.x, 0u); // the prelude's workgroup sum; uniform: after the loop
|
|
31
|
+
if (lid.x == 0u) { atomicAdd(&counters[25], total); } // nextDegreeSum: ONE atomic per workgroup
|
|
32
|
+
}
|
|
33
|
+
`;
|
|
@@ -9,6 +9,9 @@
|
|
|
9
9
|
* `partials[group].swingTraction`. `pos` is `array<vec4f>` with the mass in `.w`; `force` / `oldForce` are stride-3
|
|
10
10
|
* `array<f32>` read through the per-body helpers (4.4 rule 6). Normative text, copied verbatim: the P1-T5 sabotage
|
|
11
11
|
* rows (gravity sign, `k / d2`, the `jj != i` guard, the `.w` mass lane) are textual edits of this string.
|
|
12
|
+
* The j range is split into passes of at most EXACT_TILES_PER_PASS tiles (issue #87: llvmpipe's per-invocation loop
|
|
13
|
+
* budget); an earlier pass adds its partial sum into `force`, the last one runs gravity and the epilogue. A graph of
|
|
14
|
+
* at most 32,768 nodes is one pass, bitwise the single-pass kernel. Record it through recordExactRepulsion.
|
|
12
15
|
*/
|
|
13
16
|
export const fa2RepulsionExactWgsl = /* wgsl */ `
|
|
14
17
|
var<workgroup> tile: array<vec4f, WG>; // xyz + mass, 4 KiB at WG = 256
|
|
@@ -35,14 +38,23 @@ fn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: cent
|
|
|
35
38
|
}
|
|
36
39
|
|
|
37
40
|
@compute @workgroup_size(WG)
|
|
38
|
-
fn repulsion(
|
|
41
|
+
fn repulsion(
|
|
42
|
+
@builtin(workgroup_id) wid: vec3<u32>,
|
|
43
|
+
@builtin(local_invocation_id) lid: vec3<u32>,
|
|
44
|
+
@builtin(num_workgroups) nwg: vec3<u32>,
|
|
45
|
+
) {
|
|
46
|
+
// issue #87: pass p of the tile range is dispatched with p + 1 z slices; only the last slice works, so the pass
|
|
47
|
+
// index needs no uniform. Uniform: keyed on workgroup_id and num_workgroups only.
|
|
48
|
+
if (wid.z + 1u < nwg.z) { return; }
|
|
39
49
|
let i = linear_id(wid, lid.x);
|
|
40
50
|
let valid = i < P.n;
|
|
41
51
|
var pi = vec4f(0.0);
|
|
42
52
|
if (valid) { pi = pos[i]; }
|
|
43
53
|
var f = vec3f(0.0);
|
|
44
54
|
let tiles = (P.n + WG - 1u) / WG;
|
|
45
|
-
|
|
55
|
+
let tileBegin = (nwg.z - 1u) * EXACT_TILES_PER_PASS;
|
|
56
|
+
let tileEnd = min(tiles, tileBegin + EXACT_TILES_PER_PASS);
|
|
57
|
+
for (var t = tileBegin; t < tileEnd; t = t + 1u) {
|
|
46
58
|
let j = t * WG + lid.x;
|
|
47
59
|
if (j < P.n) { tile[lid.x] = pos[j]; } else { tile[lid.x] = vec4f(0.0); } // guarded fill; mass 0 marks the pad
|
|
48
60
|
workgroupBarrier(); // uniform: every invocation reaches it
|
|
@@ -65,6 +77,10 @@ fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
|
|
|
65
77
|
}
|
|
66
78
|
workgroupBarrier();
|
|
67
79
|
}
|
|
80
|
+
if (tileEnd < tiles) { // an earlier pass: its partial sum only (uniform: P.n, nwg)
|
|
81
|
+
if (valid) { store_force(i, load_force(i) + f); }
|
|
82
|
+
return;
|
|
83
|
+
}
|
|
68
84
|
// epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it
|
|
69
85
|
var sw = 0.0;
|
|
70
86
|
var tr = 0.0;
|
|
@@ -61,7 +61,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
61
61
|
S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)
|
|
62
62
|
let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)
|
|
63
63
|
S.meanDisplacement = meanDisp;
|
|
64
|
-
S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);
|
|
64
|
+
S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= min(P.settleThreshold * S.rmsRadius, P.settleFloor)); // relative AND absolute (issue #97)
|
|
65
65
|
}
|
|
66
66
|
S.iteration = S.iteration + 1u;
|
|
67
67
|
T[P.iterationIndex].meanDisplacement = S.meanDisplacement;
|
|
@@ -79,7 +79,9 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
79
79
|
S.invCellSize = 1.0 / cellSize;
|
|
80
80
|
S.eps = 0.25 * cellSize;
|
|
81
81
|
}
|
|
82
|
-
|
|
82
|
+
var outside = 0u; // the previous iteration's pseudo-cell counts, one per orthant (issue #90; 0 after load)
|
|
83
|
+
for (var o = 0u; o < select(4u, 8u, P.dim == 3u); o = o + 1u) { outside = outside + cellHist[cells + o]; }
|
|
84
|
+
S.outsideGrid = outside;
|
|
83
85
|
S.maxCellOccupancy = atomicLoad(&hubCounters[1]);
|
|
84
86
|
atomicStore(&hubCounters[0], 0u);
|
|
85
87
|
atomicStore(&hubCounters[1], 0u);
|
|
@@ -29,25 +29,26 @@
|
|
|
29
29
|
* raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
|
|
30
30
|
* to 0); the relax kernels size themselves from words 0 and 20.
|
|
31
31
|
*
|
|
32
|
-
* Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a
|
|
33
|
-
* the done boundary too, which the host model of the tests mirrors): top-down switches to
|
|
34
|
-
* `
|
|
35
|
-
* unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up
|
|
36
|
-
* `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no
|
|
37
|
-
* reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the
|
|
38
|
-
* change is counted in `switches`, the previous direction is word 14.
|
|
39
|
-
*
|
|
40
|
-
*
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
*
|
|
47
|
-
*
|
|
48
|
-
*
|
|
49
|
-
*
|
|
50
|
-
*
|
|
32
|
+
* Beamer's test (P8-T8, PD-21; amended for issue #391), evaluated at every boundary BEFORE the `done` branch (so a
|
|
33
|
+
* switch can be counted at the done boundary too, which the host model of the tests mirrors): top-down switches to
|
|
34
|
+
* bottom-up when `nextDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's
|
|
35
|
+
* `max(1, floor(arcCount / n))` unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up
|
|
36
|
+
* switches back when `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no
|
|
37
|
+
* admitted device reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the
|
|
38
|
+
* direction at 0. Every change is counted in `switches`, the previous direction is word 14. `nextDegreeSum` (word
|
|
39
|
+
* 25) is Beamer's m_f measured EXACTLY: `bfs-next-degree` sums the out-degrees of the vertices a level claims at
|
|
40
|
+
* the end of that level, so the boundary that rotates them in as `next` compares the degree of the frontier it is
|
|
41
|
+
* about to expand -- not, as before the amendment, `frontierDegreeSum` (word 2), the degree of the frontier the
|
|
42
|
+
* previous level EXPANDED, one level stale and 0 after a bottom-up level, which on the 1M / 10M R-MAT missed the
|
|
43
|
+
* switch at the level holding 13.6M of the 21M arcs. Word 2 is still accumulated by the expansion and rotated into
|
|
44
|
+
* word 4 for the inspect seam. The two unvisited words are rebuilt exactly once per submit by `bfs-unvisited-flags`
|
|
45
|
+
* (PD-18) and maintained here by subtraction from the SECOND boundary of a submit on, because a boundary may only
|
|
46
|
+
* subtract what the submit's rebuild counted: the rebuild counts the vertices unclaimed when it runs, the frontier
|
|
47
|
+
* rotated in at boundary 0 was claimed by the previous submit's last level, so it was never in the sums, and
|
|
48
|
+
* boundary b subtracts `next = |F_b|` and `nextDegreeSum = deg(F_b)`, both inside the sums iff b >= 1. Both words
|
|
49
|
+
* are therefore exact at every boundary, bottom-up levels included (the sweep's claims are summed like any other).
|
|
50
|
+
* Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits
|
|
51
|
+
* of it.
|
|
51
52
|
*/
|
|
52
53
|
export const frontierFinalizeWgsl = /* wgsl */ `
|
|
53
54
|
@compute @workgroup_size(WG)
|
|
@@ -61,19 +62,19 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
61
62
|
let finished = atomicLoad(&counters[0]);
|
|
62
63
|
let next = atomicLoad(&counters[1]);
|
|
63
64
|
let degSum = atomicLoad(&counters[2]);
|
|
65
|
+
let nextDeg = atomicLoad(&counters[25]); // deg(F_b), summed by bfs-next-degree when F_b was claimed (issue #391)
|
|
64
66
|
atomicStore(&counters[3], finished); // prevFrontierCount
|
|
65
67
|
atomicStore(&counters[4], degSum); // prevDegreeSum
|
|
66
68
|
atomicStore(&counters[0], next); // the rotation
|
|
67
69
|
atomicStore(&counters[1], 0u);
|
|
68
70
|
atomicStore(&counters[2], 0u);
|
|
71
|
+
atomicStore(&counters[25], 0u); // the next level's claims sum from 0
|
|
69
72
|
atomicStore(&counters[8], 0u); // edgeCount
|
|
70
73
|
atomicStore(&counters[9], 0u); // edgeCountUnclamped
|
|
71
74
|
atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount
|
|
72
|
-
if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped
|
|
73
|
-
atomicStore(&counters[5], atomicLoad(&counters[5]) - next);
|
|
74
|
-
|
|
75
|
-
if (P.firstOfSubmit >= 2u) {
|
|
76
|
-
atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2
|
|
75
|
+
if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped (P8-T8, PD-18): F_b was inside the submit's rebuilt sums iff b >= 1
|
|
76
|
+
atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount, exact
|
|
77
|
+
atomicStore(&counters[6], atomicLoad(&counters[6]) - nextDeg); // unvisitedDegreeSum, exact (issue #391: no longer one level stale)
|
|
77
78
|
}
|
|
78
79
|
let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0
|
|
79
80
|
atomicStore(&counters[11], level);
|
|
@@ -83,7 +84,7 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
83
84
|
if (P.mode == 1u) {
|
|
84
85
|
direction = 0u; // top-down only (the test seam)
|
|
85
86
|
} else if (direction == 0u) {
|
|
86
|
-
if (
|
|
87
|
+
if (nextDeg > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing, m_f the degree of the frontier about to be expanded
|
|
87
88
|
} else {
|
|
88
89
|
if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking
|
|
89
90
|
}
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* G1, the `grid-cell-key` kernel body (spec 7.7; P4-T8): the finest cell of every node from the state's robust extent,
|
|
3
3
|
* `floor((p - gridMin) * invCellSize)` (a multiply, correctly rounded everywhere: PD-10), linearised when every axis
|
|
4
|
-
* is in [0, G) and the outside pseudo-
|
|
4
|
+
* is in [0, G), and otherwise one of the 2^dim outside pseudo-cells `cells + orthant`, the orthant of the cell about the
|
|
5
|
+
* grid centre (bit a set when `c[a] >= G / 2`; issue #90); `cellVal[i] = i`. The clamp before the floor keeps a
|
|
5
6
|
* far-away or NaN coordinate out of an out-of-range float-to-int conversion. Body only; normative text.
|
|
6
7
|
*/
|
|
7
8
|
export const gridCellKeyWgsl = /* wgsl */ `
|
|
@@ -18,7 +19,8 @@ fn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocatio
|
|
|
18
19
|
let g = i32(P.gridMax);
|
|
19
20
|
var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;
|
|
20
21
|
if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }
|
|
21
|
-
var key = cells;
|
|
22
|
+
var key = cells + select(0u, 1u, c.x >= g / 2) + select(0u, 2u, c.y >= g / 2); // an outside pseudo-cell: its orthant (issue #90)
|
|
23
|
+
if (P.dim == 3u) { key = key + select(0u, 4u, c.z >= g / 2); }
|
|
22
24
|
if (inside) {
|
|
23
25
|
key = u32(c.x) + P.gridMax * u32(c.y);
|
|
24
26
|
if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the pseudo-
|
|
2
|
+
* G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the 2^dim outside pseudo-cells
|
|
3
|
+
* included (issue #90); the
|
|
3
4
|
* mass-weighted position sum of a cell's sorted range in index order (no atomics: deterministic), the largest
|
|
4
5
|
* occupancy into hubCounters[1], and cells above GRID_HUB_CELL entries appended to hubList for G4b (PD-13). Body
|
|
5
6
|
* only; normative text.
|
|
@@ -10,7 +11,7 @@ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.
|
|
|
10
11
|
@compute @workgroup_size(WG)
|
|
11
12
|
fn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
12
13
|
let c = linear_id(wid, lid.x);
|
|
13
|
-
if (c
|
|
14
|
+
if (c >= grid_cells() + select(4u, 8u, P.dim == 3u)) { return; } // cells [0, cells + 2^dim): the pseudo-cells follow the real ones; no barrier follows
|
|
14
15
|
let start = cellStart[c];
|
|
15
16
|
let count = cellStart[c + 1u] - start;
|
|
16
17
|
atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* G5, the `grid-downsample` kernel body (spec 7.7; P4-T9): one dispatch per coarser level; every parent cell is the
|
|
3
3
|
* sum of its 4 (2D) or 8 (3D) children at the level below, read at P.childBase and written at P.parentBase (the
|
|
4
|
-
* pseudo-
|
|
4
|
+
* pseudo-cells, indices cells .. of level 0, are never children). No atomics. Body only; normative text.
|
|
5
5
|
*/
|
|
6
6
|
export const gridDownsampleWgsl = /* wgsl */ `
|
|
7
7
|
@compute @workgroup_size(WG)
|
|
@@ -2,9 +2,11 @@
|
|
|
2
2
|
* G6, the `grid-far-field` kernel body (spec 7.7; P4-T10; D24): per node `i = sortedIdx[t]`, its finest cell
|
|
3
3
|
* recomputed from `pos[i]` and the state (PD-10); for an inside node the coarsest level minus the 3x3 (3x3x3)
|
|
4
4
|
* around its coarsest cell, then at every finer level the 6x6 (6x6x6) block that is the parent's 3x3 minus this
|
|
5
|
-
* level's own 3x3 -- space tiled exactly once, no theta -- plus the
|
|
6
|
-
*
|
|
7
|
-
*
|
|
5
|
+
* level's own 3x3 -- space tiled exactly once, no theta -- plus the centroid of each of the 2^dim outside
|
|
6
|
+
* pseudo-cells, one per orthant about the grid centre (issue #90); for an outside node the coarsest level in full and
|
|
7
|
+
* no pseudo-cell. Every cell term is the per-cell law on the mass-weighted
|
|
8
|
+
* centroid (Gephi Region semantics), softened by `eps^2`: `LAW` 0 (FA2) `d * (k m_i M / d2)` with `|d|^2` first
|
|
9
|
+
* floored at 0.01^2 like K3's pair law (issue #89), `LAW` 1 (FR, 7.20)
|
|
8
10
|
* `d * (k^2 M / d2)` (mass 1 per node, so `M` is the cell's count), `LAW` 2 (coulomb) `d * (-g m_i M / d2^1.5)`
|
|
9
11
|
* (P4-T13, PD-22). `force += f` (K2 wrote it). The loop bounds are `P.levels` and `P.gridMax` from the uniform,
|
|
10
12
|
* not a `LEVELS` override (PD-16, DEP-P4-G). Body only; normative text.
|
|
@@ -18,11 +20,12 @@ fn store_force(i: u32, f: vec3f) {
|
|
|
18
20
|
}
|
|
19
21
|
fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
|
|
20
22
|
fn grid_side(level: u32) -> u32 { return P.gridMax >> level; }
|
|
21
|
-
fn
|
|
23
|
+
fn outside_cells() -> u32 { return select(4u, 8u, P.dim == 3u); } // one pseudo-cell per orthant (issue #90)
|
|
24
|
+
fn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cells at cells ..)
|
|
22
25
|
var base = 0u;
|
|
23
26
|
for (var l = 0u; l < level; l = l + 1u) {
|
|
24
27
|
let s = grid_side(l);
|
|
25
|
-
base = base + s * s * select(1u, s, P.dim == 3u) + select(0u,
|
|
28
|
+
base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, outside_cells(), l == 0u);
|
|
26
29
|
}
|
|
27
30
|
return base;
|
|
28
31
|
}
|
|
@@ -33,7 +36,9 @@ fn cell_at(level: u32, cx: i32, cy: i32, cz: i32) -> u32 {
|
|
|
33
36
|
fn cell_force(pi: vec4f, q: vec4f) -> vec3f { // one far-field term, softened by state.eps (7.7)
|
|
34
37
|
if (q.w <= 0.0) { return vec3f(0.0); } // an empty cell
|
|
35
38
|
let d = pi.xyz - q.xyz / q.w; // to the mass-weighted centroid
|
|
36
|
-
|
|
39
|
+
var d2 = dot(d, d);
|
|
40
|
+
if (LAW == 0u) { d2 = max(d2, FA2_DIST_FLOOR_SQ); } // FA2 alone floors d >= 0.01, as K3 and G7 do (issue #89); FR and coulomb are unfloored (7.20)
|
|
41
|
+
d2 = d2 + S.eps * S.eps;
|
|
37
42
|
if (LAW == 1u) { return d * (P.frK * P.frK * q.w / d2); } // LAW 1 (FR, 7.20): k^2 / d per node, q.w nodes at the centroid
|
|
38
43
|
if (LAW == 2u) { return d * (-P.coulomb * pi.w * q.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb): -g m_i M_cell / d^2
|
|
39
44
|
return d * (P.scalingRatio * pi.w * q.w / d2); // LAW 0 (FA2): |F| = k m_i M_cell / d
|
|
@@ -82,9 +87,11 @@ fn grid_far_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
|
|
|
82
87
|
}
|
|
83
88
|
}
|
|
84
89
|
}
|
|
85
|
-
|
|
90
|
+
for (var o = 0u; o < outside_cells(); o = o + 1u) { // every outside pseudo-cell: one far-field term per orthant
|
|
91
|
+
f = f + cell_force(pi, pyramid[grid_cells() + o]);
|
|
92
|
+
}
|
|
86
93
|
} else {
|
|
87
|
-
for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (
|
|
94
|
+
for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (G7 sums them pair by pair)
|
|
88
95
|
for (var cy = 0; cy < ts; cy = cy + 1) {
|
|
89
96
|
for (var cx = 0; cx < ts; cx = cx + 1) {
|
|
90
97
|
f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
* exact pair law of K3 (`LAW` 0: `|F| = k m_i m_j / d` with the 0.01 floor; `LAW` 1: FR's unfloored `k^2 / d`;
|
|
4
4
|
* `LAW` 2: the unfloored coulomb `-g m_i m_j / d^2`; the antisymmetric coincident kick at the law's magnitude at
|
|
5
5
|
* d = 0.01, PD-22) over the 9 (27) finest
|
|
6
|
-
* cells around its own, or over the outside pseudo-
|
|
6
|
+
* cells around its own, or over the 2^dim outside pseudo-cells (one per orthant, issue #90) for an outside node; a cell above `nearMax` entries
|
|
7
7
|
* is sampled by `nearMax` INDEPENDENT draws with replacement, draw `k` reading the slot
|
|
8
8
|
* `lowbias32(((c ^ (iteration * 0x9E3779B9)) ^ seed) ^ (k * 0x85EBCA6B)) % count` (every slot's inclusion
|
|
9
9
|
* probability is `nearMax / count` whatever its position in the sorted order, so a duplicated draw is counted twice
|
|
@@ -103,7 +103,11 @@ fn grid_near_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocat
|
|
|
103
103
|
}
|
|
104
104
|
}
|
|
105
105
|
} else {
|
|
106
|
-
|
|
106
|
+
var own = grid_cells() + select(0u, 1u, c0.x >= g / 2) + select(0u, 2u, c0.y >= g / 2); // G1's orthant key
|
|
107
|
+
if (P.dim == 3u) { own = own + select(0u, 4u, c0.z >= g / 2); }
|
|
108
|
+
for (var o = grid_cells(); o < grid_cells() + select(4u, 8u, P.dim == 3u); o = o + 1u) { // an outside node: every outside pseudo-cell
|
|
109
|
+
f = f + cell_sum(i, pi, o, o == own);
|
|
110
|
+
}
|
|
107
111
|
}
|
|
108
112
|
}
|
|
109
113
|
// epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it (K3's text)
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The `histogram` kernel body (spec 6 row 5; P4-T3): one atomicAdd per key into the global `hist` (zeroed by a fill
|
|
3
3
|
* dispatch earlier in the pass, PD-4). Order-independent, hence deterministic. A key >= P.bins is not counted (the
|
|
4
|
-
* caller's contract; the grid's keys are always < cells +
|
|
4
|
+
* caller's contract; the grid's keys are always < cells + 2^dim). Body only; normative text.
|
|
5
5
|
*/
|
|
6
6
|
export const histogramWgsl = /* wgsl */ `
|
|
7
7
|
@compute @workgroup_size(WG)
|