@graphty/webgpu-graph-algorithms 0.6.4 → 0.6.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +38 -17
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-Dvq-Cc6v.js → context-DiSr6eiz.js} +45 -33
- package/dist/chunks/context-DiSr6eiz.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/algorithms/bfs.d.ts +11 -7
- package/dist/src/algorithms/bfs.d.ts.map +1 -1
- package/dist/src/algorithms/bfs.js +33 -10
- package/dist/src/algorithms/bfs.js.map +1 -1
- package/dist/src/algorithms/scope.d.ts +3 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +0 -2
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +4 -3
- package/dist/src/algorithms/sssp.d.ts.map +1 -1
- package/dist/src/algorithms/sssp.js +4 -3
- package/dist/src/algorithms/sssp.js.map +1 -1
- package/dist/src/constants.d.ts +33 -2
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +33 -2
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernel/dispatch.d.ts +2 -2
- package/dist/src/kernel/dispatch.d.ts.map +1 -1
- package/dist/src/kernel/kernel.d.ts +1 -1
- package/dist/src/kernel/kernel.js +2 -2
- package/dist/src/kernel/kernel.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +2 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernels.d.ts +15 -11
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +42 -21
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
- package/dist/src/layouts/forceatlas2.js +2 -1
- package/dist/src/layouts/forceatlas2.js.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.js +4 -2
- package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
- package/dist/src/layouts/repulsion-exact.d.ts +16 -0
- package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
- package/dist/src/layouts/repulsion-exact.js +21 -1
- package/dist/src/layouts/repulsion-exact.js.map +1 -1
- package/dist/src/layouts/repulsion-grid.d.ts +1 -1
- package/dist/src/layouts/repulsion-grid.js +1 -1
- package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
- package/dist/src/layouts/spring-electrical.js +6 -2
- package/dist/src/layouts/spring-electrical.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +3 -2
- package/dist/src/primitives/advance.d.ts.map +1 -1
- package/dist/src/primitives/advance.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +34 -38
- package/dist/src/primitives/frontier.d.ts.map +1 -1
- package/dist/src/primitives/frontier.js +24 -32
- package/dist/src/primitives/frontier.js.map +1 -1
- package/dist/src/primitives/grid-pyramid.d.ts +4 -4
- package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
- package/dist/src/primitives/grid-pyramid.js +4 -3
- package/dist/src/primitives/grid-pyramid.js.map +1 -1
- package/dist/src/primitives/grid.d.ts +13 -10
- package/dist/src/primitives/grid.d.ts.map +1 -1
- package/dist/src/primitives/grid.js +10 -7
- package/dist/src/primitives/grid.js.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +44 -49
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.js +62 -107
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
- package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
- package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
- package/dist/src/wgsl/histogram.wgsl.js +1 -1
- package/dist/webgpu-graph-algorithms.js +144 -119
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +3 -3
- package/src/algorithms/bfs.ts +34 -10
- package/src/algorithms/pagerank.ts +19 -5
- package/src/algorithms/power-iteration.ts +8 -2
- package/src/algorithms/scope.ts +3 -10
- package/src/algorithms/sssp.ts +4 -3
- package/src/constants.ts +35 -2
- package/src/kernel/dispatch.ts +2 -2
- package/src/kernel/kernel.ts +2 -2
- package/src/kernel/prelude.ts +2 -0
- package/src/kernels.ts +44 -21
- package/src/layouts/forceatlas2.ts +2 -0
- package/src/layouts/fruchterman-reingold.ts +4 -1
- package/src/layouts/repulsion-exact.ts +29 -1
- package/src/layouts/repulsion-grid.ts +1 -1
- package/src/layouts/spring-electrical.ts +8 -1
- package/src/memory/residency.ts +14 -4
- package/src/primitives/advance.ts +5 -4
- package/src/primitives/frontier.ts +42 -56
- package/src/primitives/grid-pyramid.ts +6 -5
- package/src/primitives/grid.ts +17 -12
- package/src/wgsl/advance-expand.wgsl.ts +4 -3
- package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
- package/src/wgsl/bfs-fused.wgsl.ts +6 -6
- package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
- package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
- package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
- package/src/wgsl/frontier-finalize.wgsl.ts +62 -107
- package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
- package/src/wgsl/grid-centroid.wgsl.ts +3 -2
- package/src/wgsl/grid-downsample.wgsl.ts +1 -1
- package/src/wgsl/grid-far-field.wgsl.ts +15 -8
- package/src/wgsl/grid-near-field.wgsl.ts +6 -2
- package/src/wgsl/histogram.wgsl.ts +1 -1
- package/dist/chunks/context-Dvq-Cc6v.js.map +0 -1
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bfs-next-degree` kernel body (design 8.4; issue #391, the amendment to P8-T8's PD-21): Beamer's m_f measured
|
|
3
|
+
* EXACTLY, at the end of every level, as the out-degree sum of the vertices the level just claimed -- the next
|
|
4
|
+
* frontier, which is what the next boundary decides the direction FOR. It grid-strides over the output vertex
|
|
5
|
+
* queue (`nextFrontierCount`, word 1, the claim kernels' append span; `P.stride` the plan's stride), reads each
|
|
6
|
+
* entry's out-degree from the `outDegree` view, reduces the lane sums with the prelude's `wg_reduce_u32` and lands
|
|
7
|
+
* ONE `atomicAdd` per workgroup in `nextDegreeSum` (word 25), which `frontier-finalize` role 0 reads for the
|
|
8
|
+
* switch-into-bottom-up test, subtracts from `unvisitedDegreeSum` and zeroes for the next level. It runs on every
|
|
9
|
+
* path that claims (the path word 24 non-zero: two-phase, fused, bottom-up, the retry) and does nothing on a level
|
|
10
|
+
* past the end.
|
|
11
|
+
*
|
|
12
|
+
* Why a kernel of its own: before it, the test used `frontierDegreeSum` (word 2), which the EXPANSION of the
|
|
13
|
+
* previous frontier accumulates, so the boundary compared the degree of the frontier it had just finished with the
|
|
14
|
+
* unvisited set, one level stale, and a bottom-up level (which expands nothing) left it at 0. On the 1M / 10M R-MAT
|
|
15
|
+
* that misses the switch at the level that matters: the frontier of 46,524 hubs at level 1 has 13.6M out-arcs, the
|
|
16
|
+
* unvisited set 7.3M, and the boundary saw the source's 86,405 instead -- top-down wrote 13.6M edge-queue entries
|
|
17
|
+
* where the bottom-up sweep reads 0.6M. Measuring the next frontier's degree at claim time is Beamer's own m_f, and
|
|
18
|
+
* the same word makes `unvisitedDegreeSum` exact after a bottom-up level too. Uniformity (spec 3.5 rule 1): the
|
|
19
|
+
* loop holds no barrier (its trip count is per lane), and the reduction runs unconditionally after it. Body only
|
|
20
|
+
* (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
21
|
+
*/
|
|
22
|
+
export const bfsNextDegreeWgsl = /* wgsl */ `
|
|
23
|
+
@compute @workgroup_size(WG)
|
|
24
|
+
fn bfs_next_degree(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
25
|
+
let count = select(0u, atomicLoad(&counters[1]), atomicLoad(&counters[24]) != 0u); // nextFrontierCount, on a level that claimed (the path word)
|
|
26
|
+
var sum = 0u;
|
|
27
|
+
for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // no barrier inside: the trip count is per lane
|
|
28
|
+
sum = sum + outDegree[frontier[i]];
|
|
29
|
+
}
|
|
30
|
+
let total = wg_reduce_u32(sum, lid.x, 0u); // the prelude's workgroup sum; uniform: after the loop
|
|
31
|
+
if (lid.x == 0u) { atomicAdd(&counters[25], total); } // nextDegreeSum: ONE atomic per workgroup
|
|
32
|
+
}
|
|
33
|
+
`;
|
|
@@ -9,6 +9,9 @@
|
|
|
9
9
|
* `partials[group].swingTraction`. `pos` is `array<vec4f>` with the mass in `.w`; `force` / `oldForce` are stride-3
|
|
10
10
|
* `array<f32>` read through the per-body helpers (4.4 rule 6). Normative text, copied verbatim: the P1-T5 sabotage
|
|
11
11
|
* rows (gravity sign, `k / d2`, the `jj != i` guard, the `.w` mass lane) are textual edits of this string.
|
|
12
|
+
* The j range is split into passes of at most EXACT_TILES_PER_PASS tiles (issue #87: llvmpipe's per-invocation loop
|
|
13
|
+
* budget); an earlier pass adds its partial sum into `force`, the last one runs gravity and the epilogue. A graph of
|
|
14
|
+
* at most 32,768 nodes is one pass, bitwise the single-pass kernel. Record it through recordExactRepulsion.
|
|
12
15
|
*/
|
|
13
16
|
export const fa2RepulsionExactWgsl = /* wgsl */ `
|
|
14
17
|
var<workgroup> tile: array<vec4f, WG>; // xyz + mass, 4 KiB at WG = 256
|
|
@@ -35,14 +38,23 @@ fn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: cent
|
|
|
35
38
|
}
|
|
36
39
|
|
|
37
40
|
@compute @workgroup_size(WG)
|
|
38
|
-
fn repulsion(
|
|
41
|
+
fn repulsion(
|
|
42
|
+
@builtin(workgroup_id) wid: vec3<u32>,
|
|
43
|
+
@builtin(local_invocation_id) lid: vec3<u32>,
|
|
44
|
+
@builtin(num_workgroups) nwg: vec3<u32>,
|
|
45
|
+
) {
|
|
46
|
+
// issue #87: pass p of the tile range is dispatched with p + 1 z slices; only the last slice works, so the pass
|
|
47
|
+
// index needs no uniform. Uniform: keyed on workgroup_id and num_workgroups only.
|
|
48
|
+
if (wid.z + 1u < nwg.z) { return; }
|
|
39
49
|
let i = linear_id(wid, lid.x);
|
|
40
50
|
let valid = i < P.n;
|
|
41
51
|
var pi = vec4f(0.0);
|
|
42
52
|
if (valid) { pi = pos[i]; }
|
|
43
53
|
var f = vec3f(0.0);
|
|
44
54
|
let tiles = (P.n + WG - 1u) / WG;
|
|
45
|
-
|
|
55
|
+
let tileBegin = (nwg.z - 1u) * EXACT_TILES_PER_PASS;
|
|
56
|
+
let tileEnd = min(tiles, tileBegin + EXACT_TILES_PER_PASS);
|
|
57
|
+
for (var t = tileBegin; t < tileEnd; t = t + 1u) {
|
|
46
58
|
let j = t * WG + lid.x;
|
|
47
59
|
if (j < P.n) { tile[lid.x] = pos[j]; } else { tile[lid.x] = vec4f(0.0); } // guarded fill; mass 0 marks the pad
|
|
48
60
|
workgroupBarrier(); // uniform: every invocation reaches it
|
|
@@ -65,6 +77,10 @@ fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
|
|
|
65
77
|
}
|
|
66
78
|
workgroupBarrier();
|
|
67
79
|
}
|
|
80
|
+
if (tileEnd < tiles) { // an earlier pass: its partial sum only (uniform: P.n, nwg)
|
|
81
|
+
if (valid) { store_force(i, load_force(i) + f); }
|
|
82
|
+
return;
|
|
83
|
+
}
|
|
68
84
|
// epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it
|
|
69
85
|
var sw = 0.0;
|
|
70
86
|
var tr = 0.0;
|
|
@@ -61,7 +61,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
61
61
|
S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)
|
|
62
62
|
let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)
|
|
63
63
|
S.meanDisplacement = meanDisp;
|
|
64
|
-
S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);
|
|
64
|
+
S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= min(P.settleThreshold * S.rmsRadius, P.settleFloor)); // relative AND absolute (issue #97)
|
|
65
65
|
}
|
|
66
66
|
S.iteration = S.iteration + 1u;
|
|
67
67
|
T[P.iterationIndex].meanDisplacement = S.meanDisplacement;
|
|
@@ -79,7 +79,9 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
79
79
|
S.invCellSize = 1.0 / cellSize;
|
|
80
80
|
S.eps = 0.25 * cellSize;
|
|
81
81
|
}
|
|
82
|
-
|
|
82
|
+
var outside = 0u; // the previous iteration's pseudo-cell counts, one per orthant (issue #90; 0 after load)
|
|
83
|
+
for (var o = 0u; o < select(4u, 8u, P.dim == 3u); o = o + 1u) { outside = outside + cellHist[cells + o]; }
|
|
84
|
+
S.outsideGrid = outside;
|
|
83
85
|
S.maxCellOccupancy = atomicLoad(&hubCounters[1]);
|
|
84
86
|
atomicStore(&hubCounters[0], 0u);
|
|
85
87
|
atomicStore(&hubCounters[1], 0u);
|
|
@@ -1,104 +1,80 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The `frontier-finalize` kernel body (design 5.4, 6 row 7; P8-T4, the P8 plan's PD-3 / PD-23 / DEP-P8-C): the
|
|
3
3
|
* device-side selector of the frontier family. One workgroup, one lane, no barrier after the early return (spec 3.5
|
|
4
|
-
* rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
4
|
+
* rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's counts become known at
|
|
5
|
+
* two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`, advances
|
|
6
|
+
* `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`) and chooses the path; role 1 runs once the
|
|
7
|
+
* edge queue is filled -- clamps `edgeCount` to `P.edgeCapacity`, or, when `edgeCountUnclamped` exceeds the
|
|
8
|
+
* capacity, switches the path to the fused retry over `frontierCount` (PD-23). The decision is ONE word of the
|
|
9
|
+
* counters block, `path` (word 24): 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3 bottom-up,
|
|
10
|
+
* 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2). Every level kernel is a direct grid-stride
|
|
11
|
+
* dispatch that reads that word first and runs only when it names it (decision record
|
|
12
|
+
* design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md: Dawn's validation of an indirect dispatch cost
|
|
13
|
+
* about 0.4 ms of device time each, and the seven indirect slots this kernel once wrote per level were 97 % of a
|
|
14
|
+
* traversal's wall time; the slots and their args buffer are gone). Two rules: a boundary that finds `done` set
|
|
15
|
+
* moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when role 0
|
|
16
|
+
* chose one, which it reads from the path word (a storage write of one dispatch is visible to the next of the same
|
|
17
|
+
* pass).
|
|
16
18
|
*
|
|
17
19
|
* Roles 2 and 3 are the SSSP round boundary of the near-far loop (P8-T9, PD-20), over the same block read in its
|
|
18
20
|
* SSSP sense (word 1 the raw near half's appends, 21 the raw far half's, 0 and 20 the deduped pile counts, 22 the
|
|
19
|
-
* threshold, 23 the delta, 4 the previous threshold, 14 the round's mode)
|
|
20
|
-
*
|
|
21
|
-
*
|
|
21
|
+
* threshold, 23 the delta, 4 the previous threshold, 14 the round's mode): role 2 finds a non-empty raw near half
|
|
22
|
+
* and sizes the near dedupe (count word 1, output word 0; the dedupe's count in word 8) in mode 0; finds it empty
|
|
23
|
+
* and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
|
|
22
24
|
* unchanged sets `done 3`, the host's E_UNSUPPORTED), remembers the previous threshold in word 4 and sizes the far
|
|
23
|
-
* dedupe (
|
|
24
|
-
* raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in
|
|
25
|
-
* picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds
|
|
26
|
-
* boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed
|
|
27
|
-
*
|
|
28
|
-
*
|
|
25
|
+
* dedupe (count word 21, output word 20; the dedupe's count in word 9) in mode 1; finds both empty and sets
|
|
26
|
+
* `done 1`; and finds a raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in
|
|
27
|
+
* `level` when it picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds
|
|
28
|
+
* dispatched; a boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed and restarts the
|
|
29
|
+
* raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
|
|
30
|
+
* to 0); the relax kernels size themselves from words 0 and 20.
|
|
29
31
|
*
|
|
30
|
-
* Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a
|
|
31
|
-
* the done boundary too, which the host model of the tests mirrors): top-down switches to
|
|
32
|
-
* `
|
|
33
|
-
* unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up
|
|
34
|
-
* `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no
|
|
35
|
-
* reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the
|
|
36
|
-
* change is counted in `switches`, the previous direction is word 14.
|
|
37
|
-
*
|
|
38
|
-
*
|
|
39
|
-
*
|
|
40
|
-
*
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
*
|
|
47
|
-
*
|
|
48
|
-
*
|
|
49
|
-
*
|
|
50
|
-
* Since 2026-09-25 nothing dispatches FROM the slots (G8-F5: Dawn's validation of an indirect dispatch cost about
|
|
51
|
-
* 0.4 ms of device time each, whether or not it dispatched anything, and the seven slots of thirty-two recorded
|
|
52
|
-
* levels were 97 % of a traversal's wall time). Every level kernel is a direct grid-stride dispatch that reads the
|
|
53
|
-
* `path` word (24) this kernel writes -- 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3
|
|
54
|
-
* bottom-up, 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2) -- and the SSSP dedupes read
|
|
55
|
-
* their counts from words 8 and 9, which role 2 writes. The slots stay as the selector's recorded decision, read
|
|
56
|
-
* back by the frontier tests; deleting them with those tests is the follow-up.
|
|
32
|
+
* Beamer's test (P8-T8, PD-21; amended for issue #391), evaluated at every boundary BEFORE the `done` branch (so a
|
|
33
|
+
* switch can be counted at the done boundary too, which the host model of the tests mirrors): top-down switches to
|
|
34
|
+
* bottom-up when `nextDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's
|
|
35
|
+
* `max(1, floor(arcCount / n))` unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up
|
|
36
|
+
* switches back when `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no
|
|
37
|
+
* admitted device reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the
|
|
38
|
+
* direction at 0. Every change is counted in `switches`, the previous direction is word 14. `nextDegreeSum` (word
|
|
39
|
+
* 25) is Beamer's m_f measured EXACTLY: `bfs-next-degree` sums the out-degrees of the vertices a level claims at
|
|
40
|
+
* the end of that level, so the boundary that rotates them in as `next` compares the degree of the frontier it is
|
|
41
|
+
* about to expand -- not, as before the amendment, `frontierDegreeSum` (word 2), the degree of the frontier the
|
|
42
|
+
* previous level EXPANDED, one level stale and 0 after a bottom-up level, which on the 1M / 10M R-MAT missed the
|
|
43
|
+
* switch at the level holding 13.6M of the 21M arcs. Word 2 is still accumulated by the expansion and rotated into
|
|
44
|
+
* word 4 for the inspect seam. The two unvisited words are rebuilt exactly once per submit by `bfs-unvisited-flags`
|
|
45
|
+
* (PD-18) and maintained here by subtraction from the SECOND boundary of a submit on, because a boundary may only
|
|
46
|
+
* subtract what the submit's rebuild counted: the rebuild counts the vertices unclaimed when it runs, the frontier
|
|
47
|
+
* rotated in at boundary 0 was claimed by the previous submit's last level, so it was never in the sums, and
|
|
48
|
+
* boundary b subtracts `next = |F_b|` and `nextDegreeSum = deg(F_b)`, both inside the sums iff b >= 1. Both words
|
|
49
|
+
* are therefore exact at every boundary, bottom-up levels included (the sweep's claims are summed like any other).
|
|
50
|
+
* Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits
|
|
51
|
+
* of it.
|
|
57
52
|
*/
|
|
58
53
|
export const frontierFinalizeWgsl = /* wgsl */ `
|
|
59
|
-
fn write_slot_groups(slot: u32, groups: u32, count: u32) { // groups workgroups, split in 2D above the per-dim limit
|
|
60
|
-
var x = groups;
|
|
61
|
-
var y = 1u;
|
|
62
|
-
if (groups > MAX_WORKGROUPS_PER_DIM) {
|
|
63
|
-
x = MAX_WORKGROUPS_PER_DIM;
|
|
64
|
-
y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;
|
|
65
|
-
}
|
|
66
|
-
let base = 4u * (P.slotBase + slot); // 16-byte slots: (x, y, 1, count)
|
|
67
|
-
args[base] = x; args[base + 1u] = y; args[base + 2u] = 1u; args[base + 3u] = count;
|
|
68
|
-
}
|
|
69
|
-
fn write_slot(slot: u32, count: u32) { // one INVOCATION per entry: ceil(count / wg) workgroups
|
|
70
|
-
let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap (indirect-finalize's rule)
|
|
71
|
-
write_slot_groups(slot, groups, count);
|
|
72
|
-
}
|
|
73
|
-
fn zero_slot(slot: u32) {
|
|
74
|
-
let base = 4u * (P.slotBase + slot);
|
|
75
|
-
args[base] = 0u; args[base + 1u] = 0u; args[base + 2u] = 1u; args[base + 3u] = 0u;
|
|
76
|
-
}
|
|
77
|
-
|
|
78
54
|
@compute @workgroup_size(WG)
|
|
79
55
|
fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
80
56
|
if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
|
|
81
57
|
if (P.role == 0u) { // the level boundary
|
|
82
58
|
if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end
|
|
83
|
-
|
|
84
|
-
return;
|
|
59
|
+
atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)
|
|
60
|
+
return;
|
|
85
61
|
}
|
|
86
62
|
let finished = atomicLoad(&counters[0]);
|
|
87
63
|
let next = atomicLoad(&counters[1]);
|
|
88
64
|
let degSum = atomicLoad(&counters[2]);
|
|
65
|
+
let nextDeg = atomicLoad(&counters[25]); // deg(F_b), summed by bfs-next-degree when F_b was claimed (issue #391)
|
|
89
66
|
atomicStore(&counters[3], finished); // prevFrontierCount
|
|
90
67
|
atomicStore(&counters[4], degSum); // prevDegreeSum
|
|
91
68
|
atomicStore(&counters[0], next); // the rotation
|
|
92
69
|
atomicStore(&counters[1], 0u);
|
|
93
70
|
atomicStore(&counters[2], 0u);
|
|
71
|
+
atomicStore(&counters[25], 0u); // the next level's claims sum from 0
|
|
94
72
|
atomicStore(&counters[8], 0u); // edgeCount
|
|
95
73
|
atomicStore(&counters[9], 0u); // edgeCountUnclamped
|
|
96
74
|
atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount
|
|
97
|
-
if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped
|
|
98
|
-
atomicStore(&counters[5], atomicLoad(&counters[5]) - next);
|
|
99
|
-
|
|
100
|
-
if (P.firstOfSubmit >= 2u) {
|
|
101
|
-
atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2
|
|
75
|
+
if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped (P8-T8, PD-18): F_b was inside the submit's rebuilt sums iff b >= 1
|
|
76
|
+
atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount, exact
|
|
77
|
+
atomicStore(&counters[6], atomicLoad(&counters[6]) - nextDeg); // unvisitedDegreeSum, exact (issue #391: no longer one level stale)
|
|
102
78
|
}
|
|
103
79
|
let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0
|
|
104
80
|
atomicStore(&counters[11], level);
|
|
@@ -108,46 +84,36 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
108
84
|
if (P.mode == 1u) {
|
|
109
85
|
direction = 0u; // top-down only (the test seam)
|
|
110
86
|
} else if (direction == 0u) {
|
|
111
|
-
if (
|
|
87
|
+
if (nextDeg > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing, m_f the degree of the frontier about to be expanded
|
|
112
88
|
} else {
|
|
113
89
|
if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking
|
|
114
90
|
}
|
|
115
91
|
if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches
|
|
116
92
|
var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)
|
|
117
93
|
if (done) {
|
|
118
|
-
|
|
94
|
+
path = 0u;
|
|
119
95
|
} else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep
|
|
120
|
-
zero_slot(0u); zero_slot(1u); zero_slot(2u); zero_slot(6u);
|
|
121
|
-
write_slot(3u, (P.n + 31u) / 32u); write_slot(4u, next); write_slot(5u, atomicLoad(&counters[7]));
|
|
122
96
|
path = 3u;
|
|
123
97
|
atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);
|
|
124
98
|
} else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)
|
|
125
|
-
|
|
126
|
-
write_slot_groups(2u, next, next); // bfs-fused is one WORKGROUP per frontier entry
|
|
127
|
-
path = 2u;
|
|
99
|
+
path = 2u; // bfs-fused: one WORKGROUP per frontier entry
|
|
128
100
|
atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
|
|
129
101
|
} else {
|
|
130
|
-
|
|
131
|
-
write_slot(0u, next); // slots 1 and 6 are role 1's
|
|
132
|
-
path = 1u;
|
|
102
|
+
path = 1u; // advance-expand, then role 1 and bfs-contract
|
|
133
103
|
}
|
|
134
104
|
atomicStore(&counters[14], direction);
|
|
135
105
|
atomicStore(&counters[24], path);
|
|
136
106
|
} else if (P.role == 1u) { // the edge queue is filled
|
|
137
|
-
if (
|
|
138
|
-
zero_slot(1u); zero_slot(6u);
|
|
107
|
+
if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count
|
|
139
108
|
return;
|
|
140
109
|
}
|
|
141
110
|
let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);
|
|
142
111
|
atomicStore(&counters[8], clamped);
|
|
143
112
|
if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry
|
|
144
|
-
|
|
145
|
-
zero_slot(1u); write_slot_groups(6u, entries, entries); // one workgroup per frontier entry, as slot 2
|
|
146
|
-
atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry, bfs-contract nothing
|
|
113
|
+
atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing
|
|
147
114
|
atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);
|
|
148
115
|
atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
|
|
149
116
|
} else {
|
|
150
|
-
write_slot(1u, clamped); zero_slot(6u);
|
|
151
117
|
atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)
|
|
152
118
|
}
|
|
153
119
|
} else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes
|
|
@@ -155,54 +121,43 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
155
121
|
atomicStore(&counters[9], 0u);
|
|
156
122
|
atomicStore(&counters[24], 0u);
|
|
157
123
|
if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)
|
|
158
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
159
124
|
return;
|
|
160
125
|
}
|
|
161
126
|
let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped
|
|
162
127
|
let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped
|
|
163
128
|
if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE
|
|
164
129
|
atomicStore(&counters[15], 2u);
|
|
165
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
166
130
|
return;
|
|
167
131
|
}
|
|
168
|
-
zero_slot(2u); zero_slot(5u); zero_slot(6u); // role 3 sizes the relax slots once the piles are deduped
|
|
169
132
|
if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn
|
|
170
133
|
atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter
|
|
171
134
|
atomicStore(&counters[14], 0u); // mode 0
|
|
172
|
-
|
|
173
|
-
atomicStore(&counters[8], nearRaw); // the near dedupe's count word
|
|
135
|
+
atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)
|
|
174
136
|
atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing
|
|
175
|
-
zero_slot(3u); zero_slot(4u);
|
|
176
137
|
atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)
|
|
177
138
|
} else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile
|
|
178
139
|
let threshold = bitcast<f32>(atomicLoad(&counters[22]));
|
|
179
140
|
let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)
|
|
180
141
|
if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED
|
|
181
142
|
atomicStore(&counters[15], 3u);
|
|
182
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
183
143
|
return;
|
|
184
144
|
}
|
|
185
145
|
atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below
|
|
186
146
|
atomicStore(&counters[22], bitcast<u32>(raised));
|
|
187
147
|
atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter
|
|
188
148
|
atomicStore(&counters[14], 1u); // mode 1
|
|
189
|
-
|
|
190
|
-
write_slot(3u, farRaw); write_slot(4u, farRaw); // dedupe-claim, dedupe-filter over the far half
|
|
191
|
-
atomicStore(&counters[9], farRaw); // the far dedupe's count word
|
|
149
|
+
atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)
|
|
192
150
|
atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing
|
|
193
151
|
atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);
|
|
194
152
|
} else { // both piles empty: finished
|
|
195
153
|
atomicStore(&counters[15], 1u);
|
|
196
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
197
154
|
}
|
|
198
|
-
} else if (P.role == 3u) { // the piles are deduped:
|
|
199
|
-
if (atomicLoad(&counters[15]) != 0u) { return; } // role 2
|
|
155
|
+
} else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed
|
|
156
|
+
if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round
|
|
200
157
|
if (atomicLoad(&counters[14]) == 0u) {
|
|
201
|
-
|
|
202
|
-
atomicStore(&counters[1], 0u); // the raw near half restarts
|
|
158
|
+
atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)
|
|
203
159
|
} else {
|
|
204
|
-
|
|
205
|
-
atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far)
|
|
160
|
+
atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)
|
|
206
161
|
}
|
|
207
162
|
}
|
|
208
163
|
}
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* G1, the `grid-cell-key` kernel body (spec 7.7; P4-T8): the finest cell of every node from the state's robust extent,
|
|
3
3
|
* `floor((p - gridMin) * invCellSize)` (a multiply, correctly rounded everywhere: PD-10), linearised when every axis
|
|
4
|
-
* is in [0, G) and the outside pseudo-
|
|
4
|
+
* is in [0, G), and otherwise one of the 2^dim outside pseudo-cells `cells + orthant`, the orthant of the cell about the
|
|
5
|
+
* grid centre (bit a set when `c[a] >= G / 2`; issue #90); `cellVal[i] = i`. The clamp before the floor keeps a
|
|
5
6
|
* far-away or NaN coordinate out of an out-of-range float-to-int conversion. Body only; normative text.
|
|
6
7
|
*/
|
|
7
8
|
export const gridCellKeyWgsl = /* wgsl */ `
|
|
@@ -18,7 +19,8 @@ fn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocatio
|
|
|
18
19
|
let g = i32(P.gridMax);
|
|
19
20
|
var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;
|
|
20
21
|
if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }
|
|
21
|
-
var key = cells;
|
|
22
|
+
var key = cells + select(0u, 1u, c.x >= g / 2) + select(0u, 2u, c.y >= g / 2); // an outside pseudo-cell: its orthant (issue #90)
|
|
23
|
+
if (P.dim == 3u) { key = key + select(0u, 4u, c.z >= g / 2); }
|
|
22
24
|
if (inside) {
|
|
23
25
|
key = u32(c.x) + P.gridMax * u32(c.y);
|
|
24
26
|
if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the pseudo-
|
|
2
|
+
* G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the 2^dim outside pseudo-cells
|
|
3
|
+
* included (issue #90); the
|
|
3
4
|
* mass-weighted position sum of a cell's sorted range in index order (no atomics: deterministic), the largest
|
|
4
5
|
* occupancy into hubCounters[1], and cells above GRID_HUB_CELL entries appended to hubList for G4b (PD-13). Body
|
|
5
6
|
* only; normative text.
|
|
@@ -10,7 +11,7 @@ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.
|
|
|
10
11
|
@compute @workgroup_size(WG)
|
|
11
12
|
fn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
12
13
|
let c = linear_id(wid, lid.x);
|
|
13
|
-
if (c
|
|
14
|
+
if (c >= grid_cells() + select(4u, 8u, P.dim == 3u)) { return; } // cells [0, cells + 2^dim): the pseudo-cells follow the real ones; no barrier follows
|
|
14
15
|
let start = cellStart[c];
|
|
15
16
|
let count = cellStart[c + 1u] - start;
|
|
16
17
|
atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* G5, the `grid-downsample` kernel body (spec 7.7; P4-T9): one dispatch per coarser level; every parent cell is the
|
|
3
3
|
* sum of its 4 (2D) or 8 (3D) children at the level below, read at P.childBase and written at P.parentBase (the
|
|
4
|
-
* pseudo-
|
|
4
|
+
* pseudo-cells, indices cells .. of level 0, are never children). No atomics. Body only; normative text.
|
|
5
5
|
*/
|
|
6
6
|
export const gridDownsampleWgsl = /* wgsl */ `
|
|
7
7
|
@compute @workgroup_size(WG)
|
|
@@ -2,9 +2,11 @@
|
|
|
2
2
|
* G6, the `grid-far-field` kernel body (spec 7.7; P4-T10; D24): per node `i = sortedIdx[t]`, its finest cell
|
|
3
3
|
* recomputed from `pos[i]` and the state (PD-10); for an inside node the coarsest level minus the 3x3 (3x3x3)
|
|
4
4
|
* around its coarsest cell, then at every finer level the 6x6 (6x6x6) block that is the parent's 3x3 minus this
|
|
5
|
-
* level's own 3x3 -- space tiled exactly once, no theta -- plus the
|
|
6
|
-
*
|
|
7
|
-
*
|
|
5
|
+
* level's own 3x3 -- space tiled exactly once, no theta -- plus the centroid of each of the 2^dim outside
|
|
6
|
+
* pseudo-cells, one per orthant about the grid centre (issue #90); for an outside node the coarsest level in full and
|
|
7
|
+
* no pseudo-cell. Every cell term is the per-cell law on the mass-weighted
|
|
8
|
+
* centroid (Gephi Region semantics), softened by `eps^2`: `LAW` 0 (FA2) `d * (k m_i M / d2)` with `|d|^2` first
|
|
9
|
+
* floored at 0.01^2 like K3's pair law (issue #89), `LAW` 1 (FR, 7.20)
|
|
8
10
|
* `d * (k^2 M / d2)` (mass 1 per node, so `M` is the cell's count), `LAW` 2 (coulomb) `d * (-g m_i M / d2^1.5)`
|
|
9
11
|
* (P4-T13, PD-22). `force += f` (K2 wrote it). The loop bounds are `P.levels` and `P.gridMax` from the uniform,
|
|
10
12
|
* not a `LEVELS` override (PD-16, DEP-P4-G). Body only; normative text.
|
|
@@ -18,11 +20,12 @@ fn store_force(i: u32, f: vec3f) {
|
|
|
18
20
|
}
|
|
19
21
|
fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
|
|
20
22
|
fn grid_side(level: u32) -> u32 { return P.gridMax >> level; }
|
|
21
|
-
fn
|
|
23
|
+
fn outside_cells() -> u32 { return select(4u, 8u, P.dim == 3u); } // one pseudo-cell per orthant (issue #90)
|
|
24
|
+
fn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cells at cells ..)
|
|
22
25
|
var base = 0u;
|
|
23
26
|
for (var l = 0u; l < level; l = l + 1u) {
|
|
24
27
|
let s = grid_side(l);
|
|
25
|
-
base = base + s * s * select(1u, s, P.dim == 3u) + select(0u,
|
|
28
|
+
base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, outside_cells(), l == 0u);
|
|
26
29
|
}
|
|
27
30
|
return base;
|
|
28
31
|
}
|
|
@@ -33,7 +36,9 @@ fn cell_at(level: u32, cx: i32, cy: i32, cz: i32) -> u32 {
|
|
|
33
36
|
fn cell_force(pi: vec4f, q: vec4f) -> vec3f { // one far-field term, softened by state.eps (7.7)
|
|
34
37
|
if (q.w <= 0.0) { return vec3f(0.0); } // an empty cell
|
|
35
38
|
let d = pi.xyz - q.xyz / q.w; // to the mass-weighted centroid
|
|
36
|
-
|
|
39
|
+
var d2 = dot(d, d);
|
|
40
|
+
if (LAW == 0u) { d2 = max(d2, FA2_DIST_FLOOR_SQ); } // FA2 alone floors d >= 0.01, as K3 and G7 do (issue #89); FR and coulomb are unfloored (7.20)
|
|
41
|
+
d2 = d2 + S.eps * S.eps;
|
|
37
42
|
if (LAW == 1u) { return d * (P.frK * P.frK * q.w / d2); } // LAW 1 (FR, 7.20): k^2 / d per node, q.w nodes at the centroid
|
|
38
43
|
if (LAW == 2u) { return d * (-P.coulomb * pi.w * q.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb): -g m_i M_cell / d^2
|
|
39
44
|
return d * (P.scalingRatio * pi.w * q.w / d2); // LAW 0 (FA2): |F| = k m_i M_cell / d
|
|
@@ -82,9 +87,11 @@ fn grid_far_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
|
|
|
82
87
|
}
|
|
83
88
|
}
|
|
84
89
|
}
|
|
85
|
-
|
|
90
|
+
for (var o = 0u; o < outside_cells(); o = o + 1u) { // every outside pseudo-cell: one far-field term per orthant
|
|
91
|
+
f = f + cell_force(pi, pyramid[grid_cells() + o]);
|
|
92
|
+
}
|
|
86
93
|
} else {
|
|
87
|
-
for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (
|
|
94
|
+
for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (G7 sums them pair by pair)
|
|
88
95
|
for (var cy = 0; cy < ts; cy = cy + 1) {
|
|
89
96
|
for (var cx = 0; cx < ts; cx = cx + 1) {
|
|
90
97
|
f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
* exact pair law of K3 (`LAW` 0: `|F| = k m_i m_j / d` with the 0.01 floor; `LAW` 1: FR's unfloored `k^2 / d`;
|
|
4
4
|
* `LAW` 2: the unfloored coulomb `-g m_i m_j / d^2`; the antisymmetric coincident kick at the law's magnitude at
|
|
5
5
|
* d = 0.01, PD-22) over the 9 (27) finest
|
|
6
|
-
* cells around its own, or over the outside pseudo-
|
|
6
|
+
* cells around its own, or over the 2^dim outside pseudo-cells (one per orthant, issue #90) for an outside node; a cell above `nearMax` entries
|
|
7
7
|
* is sampled by `nearMax` INDEPENDENT draws with replacement, draw `k` reading the slot
|
|
8
8
|
* `lowbias32(((c ^ (iteration * 0x9E3779B9)) ^ seed) ^ (k * 0x85EBCA6B)) % count` (every slot's inclusion
|
|
9
9
|
* probability is `nearMax / count` whatever its position in the sorted order, so a duplicated draw is counted twice
|
|
@@ -103,7 +103,11 @@ fn grid_near_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocat
|
|
|
103
103
|
}
|
|
104
104
|
}
|
|
105
105
|
} else {
|
|
106
|
-
|
|
106
|
+
var own = grid_cells() + select(0u, 1u, c0.x >= g / 2) + select(0u, 2u, c0.y >= g / 2); // G1's orthant key
|
|
107
|
+
if (P.dim == 3u) { own = own + select(0u, 4u, c0.z >= g / 2); }
|
|
108
|
+
for (var o = grid_cells(); o < grid_cells() + select(4u, 8u, P.dim == 3u); o = o + 1u) { // an outside node: every outside pseudo-cell
|
|
109
|
+
f = f + cell_sum(i, pi, o, o == own);
|
|
110
|
+
}
|
|
107
111
|
}
|
|
108
112
|
}
|
|
109
113
|
// epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it (K3's text)
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The `histogram` kernel body (spec 6 row 5; P4-T3): one atomicAdd per key into the global `hist` (zeroed by a fill
|
|
3
3
|
* dispatch earlier in the pass, PD-4). Order-independent, hence deterministic. A key >= P.bins is not counted (the
|
|
4
|
-
* caller's contract; the grid's keys are always < cells +
|
|
4
|
+
* caller's contract; the grid's keys are always < cells + 2^dim). Body only; normative text.
|
|
5
5
|
*/
|
|
6
6
|
export const histogramWgsl = /* wgsl */ `
|
|
7
7
|
@compute @workgroup_size(WG)
|