@graphty/webgpu-graph-algorithms 0.5.1 → 0.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +459 -58
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-BR7fx3vR.js → context-BXqgCifx.js} +190 -40
- package/dist/chunks/context-BXqgCifx.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/algorithms/components.d.ts.map +1 -1
- package/dist/src/algorithms/components.js +12 -13
- package/dist/src/algorithms/components.js.map +1 -1
- package/dist/src/algorithms/degree.d.ts +6 -8
- package/dist/src/algorithms/degree.d.ts.map +1 -1
- package/dist/src/algorithms/degree.js +58 -35
- package/dist/src/algorithms/degree.js.map +1 -1
- package/dist/src/algorithms/pagerank.d.ts.map +1 -1
- package/dist/src/algorithms/pagerank.js +16 -14
- package/dist/src/algorithms/pagerank.js.map +1 -1
- package/dist/src/algorithms/power-iteration.d.ts +2 -2
- package/dist/src/algorithms/power-iteration.d.ts.map +1 -1
- package/dist/src/algorithms/power-iteration.js +17 -14
- package/dist/src/algorithms/power-iteration.js.map +1 -1
- package/dist/src/constants.d.ts +38 -8
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +38 -8
- package/dist/src/constants.js.map +1 -1
- package/dist/src/errors.d.ts +3 -2
- package/dist/src/errors.d.ts.map +1 -1
- package/dist/src/errors.js +2 -1
- package/dist/src/errors.js.map +1 -1
- package/dist/src/index.d.ts +6 -4
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +8 -3
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/dispatch.d.ts +8 -3
- package/dist/src/kernel/dispatch.d.ts.map +1 -1
- package/dist/src/kernel/dispatch.js +18 -7
- package/dist/src/kernel/dispatch.js.map +1 -1
- package/dist/src/kernel/kernel.d.ts +30 -1
- package/dist/src/kernel/kernel.d.ts.map +1 -1
- package/dist/src/kernel/kernel.js +49 -5
- package/dist/src/kernel/kernel.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +6 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernel/profiler.d.ts +15 -3
- package/dist/src/kernel/profiler.d.ts.map +1 -1
- package/dist/src/kernel/profiler.js +27 -4
- package/dist/src/kernel/profiler.js.map +1 -1
- package/dist/src/kernels.d.ts +17 -7
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +323 -16
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/calibrate.d.ts +51 -0
- package/dist/src/layouts/calibrate.d.ts.map +1 -0
- package/dist/src/layouts/calibrate.js +172 -0
- package/dist/src/layouts/calibrate.js.map +1 -0
- package/dist/src/layouts/force-simulation.d.ts +39 -4
- package/dist/src/layouts/force-simulation.d.ts.map +1 -1
- package/dist/src/layouts/force-simulation.js +71 -19
- package/dist/src/layouts/force-simulation.js.map +1 -1
- package/dist/src/layouts/forceatlas2.d.ts +107 -36
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
- package/dist/src/layouts/forceatlas2.js +296 -100
- package/dist/src/layouts/forceatlas2.js.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.d.ts +73 -27
- package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.js +230 -70
- package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
- package/dist/src/layouts/model-common.d.ts +41 -3
- package/dist/src/layouts/model-common.d.ts.map +1 -1
- package/dist/src/layouts/model-common.js +74 -3
- package/dist/src/layouts/model-common.js.map +1 -1
- package/dist/src/layouts/repulsion-grid.d.ts +152 -0
- package/dist/src/layouts/repulsion-grid.d.ts.map +1 -0
- package/dist/src/layouts/repulsion-grid.js +318 -0
- package/dist/src/layouts/repulsion-grid.js.map +1 -0
- package/dist/src/layouts/spring-electrical.d.ts +75 -30
- package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
- package/dist/src/layouts/spring-electrical.js +231 -74
- package/dist/src/layouts/spring-electrical.js.map +1 -1
- package/dist/src/memory/residency.d.ts +6 -2
- package/dist/src/memory/residency.d.ts.map +1 -1
- package/dist/src/memory/residency.js +84 -14
- package/dist/src/memory/residency.js.map +1 -1
- package/dist/src/primitives/core-shape.d.ts +38 -2
- package/dist/src/primitives/core-shape.d.ts.map +1 -1
- package/dist/src/primitives/core-shape.js +71 -3
- package/dist/src/primitives/core-shape.js.map +1 -1
- package/dist/src/primitives/grid-pyramid.d.ts +71 -0
- package/dist/src/primitives/grid-pyramid.d.ts.map +1 -0
- package/dist/src/primitives/grid-pyramid.js +143 -0
- package/dist/src/primitives/grid-pyramid.js.map +1 -0
- package/dist/src/primitives/grid.d.ts +118 -0
- package/dist/src/primitives/grid.d.ts.map +1 -0
- package/dist/src/primitives/grid.js +225 -0
- package/dist/src/primitives/grid.js.map +1 -0
- package/dist/src/primitives/histogram.d.ts +67 -0
- package/dist/src/primitives/histogram.d.ts.map +1 -0
- package/dist/src/primitives/histogram.js +190 -0
- package/dist/src/primitives/histogram.js.map +1 -0
- package/dist/src/primitives/radix-sort.d.ts +75 -0
- package/dist/src/primitives/radix-sort.d.ts.map +1 -0
- package/dist/src/primitives/radix-sort.js +168 -0
- package/dist/src/primitives/radix-sort.js.map +1 -0
- package/dist/src/primitives/scan.d.ts +44 -0
- package/dist/src/primitives/scan.d.ts.map +1 -0
- package/dist/src/primitives/scan.js +151 -0
- package/dist/src/primitives/scan.js.map +1 -0
- package/dist/src/primitives/segmented-reduce.d.ts +25 -17
- package/dist/src/primitives/segmented-reduce.d.ts.map +1 -1
- package/dist/src/primitives/segmented-reduce.js +166 -47
- package/dist/src/primitives/segmented-reduce.js.map +1 -1
- package/dist/src/primitives/spmv.d.ts +18 -14
- package/dist/src/primitives/spmv.d.ts.map +1 -1
- package/dist/src/primitives/spmv.js +94 -58
- package/dist/src/primitives/spmv.js.map +1 -1
- package/dist/src/primitives/verify.d.ts +49 -0
- package/dist/src/primitives/verify.d.ts.map +1 -0
- package/dist/src/primitives/verify.js +229 -0
- package/dist/src/primitives/verify.js.map +1 -0
- package/dist/src/types/context.d.ts +53 -0
- package/dist/src/types/context.d.ts.map +1 -1
- package/dist/src/types/layout.d.ts +20 -0
- package/dist/src/types/layout.d.ts.map +1 -1
- package/dist/src/wgsl/counting-scatter.wgsl.d.ts +8 -0
- package/dist/src/wgsl/counting-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/counting-scatter.wgsl.js +17 -0
- package/dist/src/wgsl/counting-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-attraction.wgsl.d.ts +23 -11
- package/dist/src/wgsl/fa2-attraction.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-attraction.wgsl.js +98 -20
- package/dist/src/wgsl/fa2-attraction.wgsl.js.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +6 -2
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +22 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +8 -0
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/grid-cell-key.wgsl.js +30 -0
- package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -0
- package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +8 -0
- package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/grid-centroid-hub.wgsl.js +29 -0
- package/dist/src/wgsl/grid-centroid-hub.wgsl.js.map +1 -0
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts +8 -0
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/grid-centroid.wgsl.js +29 -0
- package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -0
- package/dist/src/wgsl/grid-downsample.wgsl.d.ts +7 -0
- package/dist/src/wgsl/grid-downsample.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/grid-downsample.wgsl.js +28 -0
- package/dist/src/wgsl/grid-downsample.wgsl.js.map +1 -0
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts +13 -0
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/grid-far-field.wgsl.js +98 -0
- package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -0
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts +19 -0
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/grid-near-field.wgsl.js +129 -0
- package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -0
- package/dist/src/wgsl/histogram.wgsl.d.ts +7 -0
- package/dist/src/wgsl/histogram.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/histogram.wgsl.js +15 -0
- package/dist/src/wgsl/histogram.wgsl.js.map +1 -0
- package/dist/src/wgsl/indirect-finalize.wgsl.d.ts +8 -0
- package/dist/src/wgsl/indirect-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/indirect-finalize.wgsl.js +26 -0
- package/dist/src/wgsl/indirect-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/radix-hist.wgsl.d.ts +9 -0
- package/dist/src/wgsl/radix-hist.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/radix-hist.wgsl.js +31 -0
- package/dist/src/wgsl/radix-hist.wgsl.js.map +1 -0
- package/dist/src/wgsl/radix-scatter.wgsl.d.ts +9 -0
- package/dist/src/wgsl/radix-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/radix-scatter.wgsl.js +40 -0
- package/dist/src/wgsl/radix-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/scan-add.wgsl.d.ts +6 -0
- package/dist/src/wgsl/scan-add.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/scan-add.wgsl.js +14 -0
- package/dist/src/wgsl/scan-add.wgsl.js.map +1 -0
- package/dist/src/wgsl/scan-block.wgsl.d.ts +8 -0
- package/dist/src/wgsl/scan-block.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/scan-block.wgsl.js +30 -0
- package/dist/src/wgsl/scan-block.wgsl.js.map +1 -0
- package/dist/src/wgsl/segmented-reduce.wgsl.d.ts +22 -8
- package/dist/src/wgsl/segmented-reduce.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/segmented-reduce.wgsl.js +84 -15
- package/dist/src/wgsl/segmented-reduce.wgsl.js.map +1 -1
- package/dist/src/wgsl/spmv-pull.wgsl.d.ts +22 -11
- package/dist/src/wgsl/spmv-pull.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/spmv-pull.wgsl.js +110 -36
- package/dist/src/wgsl/spmv-pull.wgsl.js.map +1 -1
- package/dist/tsconfig.build.tsbuildinfo +1 -1
- package/dist/webgpu-graph-algorithms.js +3815 -1003
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +9 -8
- package/src/algorithms/components.ts +12 -16
- package/src/algorithms/degree.ts +58 -43
- package/src/algorithms/pagerank.ts +20 -18
- package/src/algorithms/power-iteration.ts +19 -18
- package/src/constants.ts +38 -8
- package/src/errors.ts +3 -1
- package/src/index.ts +14 -4
- package/src/kernel/dispatch.ts +18 -7
- package/src/kernel/kernel.ts +59 -5
- package/src/kernel/prelude.ts +9 -0
- package/src/kernel/profiler.ts +28 -4
- package/src/kernels.ts +356 -18
- package/src/layouts/calibrate.ts +187 -0
- package/src/layouts/force-simulation.ts +91 -23
- package/src/layouts/forceatlas2.ts +331 -106
- package/src/layouts/fruchterman-reingold.ts +255 -74
- package/src/layouts/model-common.ts +98 -3
- package/src/layouts/repulsion-grid.ts +451 -0
- package/src/layouts/spring-electrical.ts +257 -78
- package/src/memory/residency.ts +126 -20
- package/src/primitives/core-shape.ts +91 -4
- package/src/primitives/grid-pyramid.ts +221 -0
- package/src/primitives/grid.ts +349 -0
- package/src/primitives/histogram.ts +273 -0
- package/src/primitives/radix-sort.ts +246 -0
- package/src/primitives/scan.ts +197 -0
- package/src/primitives/segmented-reduce.ts +214 -56
- package/src/primitives/spmv.ts +125 -65
- package/src/primitives/verify.ts +249 -0
- package/src/types/context.ts +56 -0
- package/src/types/layout.ts +22 -0
- package/src/wgsl/counting-scatter.wgsl.ts +16 -0
- package/src/wgsl/fa2-attraction.wgsl.ts +98 -20
- package/src/wgsl/fa2-stats-finalize.wgsl.ts +22 -1
- package/src/wgsl/grid-cell-key.wgsl.ts +29 -0
- package/src/wgsl/grid-centroid-hub.wgsl.ts +28 -0
- package/src/wgsl/grid-centroid.wgsl.ts +28 -0
- package/src/wgsl/grid-downsample.wgsl.ts +27 -0
- package/src/wgsl/grid-far-field.wgsl.ts +97 -0
- package/src/wgsl/grid-near-field.wgsl.ts +128 -0
- package/src/wgsl/histogram.wgsl.ts +14 -0
- package/src/wgsl/indirect-finalize.wgsl.ts +25 -0
- package/src/wgsl/radix-hist.wgsl.ts +30 -0
- package/src/wgsl/radix-scatter.wgsl.ts +39 -0
- package/src/wgsl/scan-add.wgsl.ts +13 -0
- package/src/wgsl/scan-block.wgsl.ts +29 -0
- package/src/wgsl/segmented-reduce.wgsl.ts +84 -15
- package/src/wgsl/spmv-pull.wgsl.ts +110 -36
- package/dist/chunks/context-BR7fx3vR.js.map +0 -1
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* G6, the `grid-far-field` kernel body (spec 7.7; P4-T10; D24): per node `i = sortedIdx[t]`, its finest cell
|
|
3
|
+
* recomputed from `pos[i]` and the state (PD-10); for an inside node the coarsest level minus the 3x3 (3x3x3)
|
|
4
|
+
* around its coarsest cell, then at every finer level the 6x6 (6x6x6) block that is the parent's 3x3 minus this
|
|
5
|
+
* level's own 3x3 -- space tiled exactly once, no theta -- plus the outside pseudo-cell's centroid; for an outside
|
|
6
|
+
* node the coarsest level in full and no pseudo-cell. Every cell term is the per-cell law on the mass-weighted
|
|
7
|
+
* centroid (Gephi Region semantics), softened by `eps^2`: `LAW` 0 (FA2) `d * (k m_i M / d2)`, `LAW` 1 (FR, 7.20)
|
|
8
|
+
* `d * (k^2 M / d2)` (mass 1 per node, so `M` is the cell's count), `LAW` 2 (coulomb) `d * (-g m_i M / d2^1.5)`
|
|
9
|
+
* (P4-T13, PD-22). `force += f` (K2 wrote it). The loop bounds are `P.levels` and `P.gridMax` from the uniform,
|
|
10
|
+
* not a `LEVELS` override (PD-16, DEP-P4-G). Body only; normative text.
|
|
11
|
+
*/
|
|
12
|
+
export const gridFarFieldWgsl = /* wgsl */ `
|
|
13
|
+
fn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }
|
|
14
|
+
fn store_force(i: u32, f: vec3f) {
|
|
15
|
+
force[3u * i] = f.x;
|
|
16
|
+
force[3u * i + 1u] = f.y;
|
|
17
|
+
force[3u * i + 2u] = f.z;
|
|
18
|
+
}
|
|
19
|
+
fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
|
|
20
|
+
fn grid_side(level: u32) -> u32 { return P.gridMax >> level; }
|
|
21
|
+
fn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cell at index cells)
|
|
22
|
+
var base = 0u;
|
|
23
|
+
for (var l = 0u; l < level; l = l + 1u) {
|
|
24
|
+
let s = grid_side(l);
|
|
25
|
+
base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, 1u, l == 0u);
|
|
26
|
+
}
|
|
27
|
+
return base;
|
|
28
|
+
}
|
|
29
|
+
fn cell_at(level: u32, cx: i32, cy: i32, cz: i32) -> u32 {
|
|
30
|
+
let s = grid_side(level);
|
|
31
|
+
return level_base(level) + u32(cx) + s * (u32(cy) + select(0u, s * u32(cz), P.dim == 3u));
|
|
32
|
+
}
|
|
33
|
+
fn cell_force(pi: vec4f, q: vec4f) -> vec3f { // one far-field term, softened by state.eps (7.7)
|
|
34
|
+
if (q.w <= 0.0) { return vec3f(0.0); } // an empty cell
|
|
35
|
+
let d = pi.xyz - q.xyz / q.w; // to the mass-weighted centroid
|
|
36
|
+
let d2 = dot(d, d) + S.eps * S.eps;
|
|
37
|
+
if (LAW == 1u) { return d * (P.frK * P.frK * q.w / d2); } // LAW 1 (FR, 7.20): k^2 / d per node, q.w nodes at the centroid
|
|
38
|
+
if (LAW == 2u) { return d * (-P.coulomb * pi.w * q.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb): -g m_i M_cell / d^2
|
|
39
|
+
return d * (P.scalingRatio * pi.w * q.w / d2); // LAW 0 (FA2): |F| = k m_i M_cell / d
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
@compute @workgroup_size(WG)
|
|
43
|
+
fn grid_far_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
44
|
+
let t = linear_id(wid, lid.x);
|
|
45
|
+
if (t >= P.n) { return; } // no barrier follows
|
|
46
|
+
let i = sortedIdx[t]; // sorted order (D24)
|
|
47
|
+
let pi = pos[i];
|
|
48
|
+
let gf = f32(P.gridMax);
|
|
49
|
+
let q = (pi.xyz - S.gridMin.xyz) * S.invCellSize; // PD-10
|
|
50
|
+
var c0 = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));
|
|
51
|
+
if (P.dim == 2u) { c0.z = 0; } // 2D: one z plane; the loops below visit cz = 0 only, so the 3x3 test must see cz - 0
|
|
52
|
+
let g = i32(P.gridMax);
|
|
53
|
+
var inside = c0.x >= 0 && c0.x < g && c0.y >= 0 && c0.y < g;
|
|
54
|
+
if (P.dim == 3u) { inside = inside && c0.z >= 0 && c0.z < g; }
|
|
55
|
+
let top = P.levels - 1u;
|
|
56
|
+
let ts = i32(grid_side(top)); // the coarsest side (4)
|
|
57
|
+
let zTop = select(0, ts - 1, P.dim == 3u); // z ranges: one plane in 2D
|
|
58
|
+
var f = vec3f(0.0);
|
|
59
|
+
if (inside) {
|
|
60
|
+
let ct = c0 / i32(1u << top); // the node's coarsest cell
|
|
61
|
+
for (var cz = 0; cz <= zTop; cz = cz + 1) {
|
|
62
|
+
for (var cy = 0; cy < ts; cy = cy + 1) {
|
|
63
|
+
for (var cx = 0; cx < ts; cx = cx + 1) {
|
|
64
|
+
if (abs(cx - ct.x) <= 1 && abs(cy - ct.y) <= 1 && abs(cz - ct.z) <= 1) { continue; } // the 3x3(x3) is finer levels' work
|
|
65
|
+
f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
for (var l = top; l > 0u; l = l - 1u) { // level l - 1: the parent's 3x3 at level l, refined, minus this level's own 3x3
|
|
70
|
+
let level = l - 1u;
|
|
71
|
+
let cl = c0 / i32(1u << level);
|
|
72
|
+
let cp = cl / 2;
|
|
73
|
+
let side = i32(grid_side(level));
|
|
74
|
+
let zLo = select(0, max(0, 2 * (cp.z - 1)), P.dim == 3u);
|
|
75
|
+
let zHi = select(0, min(side - 1, 2 * (cp.z + 1) + 1), P.dim == 3u);
|
|
76
|
+
for (var cz = zLo; cz <= zHi; cz = cz + 1) {
|
|
77
|
+
for (var cy = max(0, 2 * (cp.y - 1)); cy <= min(side - 1, 2 * (cp.y + 1) + 1); cy = cy + 1) {
|
|
78
|
+
for (var cx = max(0, 2 * (cp.x - 1)); cx <= min(side - 1, 2 * (cp.x + 1) + 1); cx = cx + 1) {
|
|
79
|
+
if (abs(cx - cl.x) <= 1 && abs(cy - cl.y) <= 1 && abs(cz - cl.z) <= 1) { continue; }
|
|
80
|
+
f = f + cell_force(pi, pyramid[cell_at(level, cx, cy, cz)]);
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
f = f + cell_force(pi, pyramid[grid_cells()]); // the outside pseudo-cell as one far-field term
|
|
86
|
+
} else {
|
|
87
|
+
for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (it would include itself)
|
|
88
|
+
for (var cy = 0; cy < ts; cy = cy + 1) {
|
|
89
|
+
for (var cx = 0; cx < ts; cx = cx + 1) {
|
|
90
|
+
f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
store_force(i, load_force(i) + f);
|
|
96
|
+
}
|
|
97
|
+
`;
|
|
98
|
+
//# sourceMappingURL=grid-far-field.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"grid-far-field.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-far-field.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAqF1C,CAAC"}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* G7, the `grid-near-field` kernel body (spec 7.7, 7.6, 7.20; P4-T10, P4-T13; D24): per node `i = sortedIdx[t]`, the
|
|
3
|
+
* exact pair law of K3 (`LAW` 0: `|F| = k m_i m_j / d` with the 0.01 floor; `LAW` 1: FR's unfloored `k^2 / d`;
|
|
4
|
+
* `LAW` 2: the unfloored coulomb `-g m_i m_j / d^2`; the antisymmetric coincident kick at the law's magnitude at
|
|
5
|
+
* d = 0.01, PD-22) over the 9 (27) finest
|
|
6
|
+
* cells around its own, or over the outside pseudo-cell alone for an outside node; a cell above `nearMax` entries
|
|
7
|
+
* is sampled by `nearMax` INDEPENDENT draws with replacement, draw `k` reading the slot
|
|
8
|
+
* `lowbias32(((c ^ (iteration * 0x9E3779B9)) ^ seed) ^ (k * 0x85EBCA6B)) % count` (every slot's inclusion
|
|
9
|
+
* probability is `nearMax / count` whatever its position in the sorted order, so a duplicated draw is counted twice
|
|
10
|
+
* and the node itself, when drawn, is skipped and not replaced), and scaled by `others / sampled` where `sampled` is
|
|
11
|
+
* the realised number of draws that were not the node (the Horvitz-Thompson form of PD-15 / DEP-P4-K: given
|
|
12
|
+
* `sampled = s`, those `s` draws are i.i.d. uniform over the `others` slots, so the expectation of the scaled sum is
|
|
13
|
+
* the exact cell sum whenever `s >= 1`; the G4 record's G4-F2 row carries the measurement); then
|
|
14
|
+
* the fused epilogue of K3 (gravity, `force +=`, the swing / traction workgroup reduction in uniform control flow).
|
|
15
|
+
* The helpers `load_force`, `store_force`, `load_old`, `kick_magnitude`, `gravity_force` and the epilogue are K3's
|
|
16
|
+
* text. Body only; normative text.
|
|
17
|
+
*/
|
|
18
|
+
export declare const gridNearFieldWgsl = "\nfn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }\nfn store_force(i: u32, f: vec3f) {\n force[3u * i] = f.x;\n force[3u * i + 1u] = f.y;\n force[3u * i + 2u] = f.z;\n}\nfn load_old(i: u32) -> vec3f { return vec3f(oldForce[3u * i], oldForce[3u * i + 1u], oldForce[3u * i + 2u]); }\nfn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: centroid (GRAVITY_CENTER 0) or origin (1); regular or strong\n var q = pi.xyz;\n if (GRAVITY_CENTER == 0u) { q = pi.xyz - S.centroid.xyz; }\n if (STRONG_GRAVITY) { return -P.gravity * pi.w * q; }\n let d = length(q);\n if (d > FA2_DIST_FLOOR) { return -P.gravity * pi.w * q / d; }\n return vec3f(0.0);\n}\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\nfn kick_magnitude(mi: f32, mj: f32) -> f32 { // the law's magnitude at d = FA2_DIST_FLOOR (the P5 plan's PD-10)\n if (LAW == 1u) { return P.frK * P.frK / FA2_DIST_FLOOR; }\n if (LAW == 2u) { return -P.coulomb * mi * mj / FA2_DIST_FLOOR_SQ; }\n return P.scalingRatio * mi * mj / FA2_DIST_FLOOR;\n}\nfn pair_force(i: u32, pi: vec4f, jj: u32, o: vec4f) -> vec3f { // the exact pair law of K3 (7.6, 7.20): the floor (FA2 only), the coincident kick\n let d = pi.xyz - o.xyz;\n var d2 = dot(d, d);\n if (d2 < FA2_COINCIDENT_SQ) { return kick_dir(i, jj, P.dim) * kick_magnitude(pi.w, o.w); }\n if (LAW == 0u) { d2 = max(d2, FA2_DIST_FLOOR_SQ); }\n let k = P.scalingRatio * pi.w * o.w;\n if (LAW == 1u) { return d * (P.frK * P.frK / d2); }\n if (LAW == 2u) { return d * (-P.coulomb * pi.w * o.w / (d2 * sqrt(d2))); }\n return d * (k / d2);\n}\nfn cell_sum(i: u32, pi: vec4f, c: u32, own: bool) -> vec3f { // one finest cell: exact below nearMax entries, Horvitz-Thompson above (PD-15)\n let start = cellStart[c];\n let count = cellStart[c + 1u] - start;\n var f = vec3f(0.0);\n if (count <= P.nearMax) {\n for (var k = start; k < start + count; k = k + 1u) {\n let jj = sortedIdx[k];\n if (jj != i) { f = f + pair_force(i, pi, jj, pos[jj]); }\n }\n return f;\n }\n let base = (c ^ (P.iterationIndex * 0x9E3779B9u)) ^ P.seed; // the per-iteration draw seed (7.16)\n var sampled = 0u;\n for (var k = 0u; k < P.nearMax; k = k + 1u) {\n let jj = sortedIdx[start + (lowbias32(base ^ (k * 0x85EBCA6Bu)) % count)]; // draw k: independent inclusion, with replacement (PD-15)\n if (jj == i) { continue; }\n f = f + pair_force(i, pi, jj, pos[jj]);\n sampled = sampled + 1u;\n }\n if (sampled == 0u) { return vec3f(0.0); }\n let others = select(count, count - 1u, own);\n return f * (f32(others) / f32(sampled)); // others / sampled over the realised sample (DEP-P4-K)\n}\n\n@compute @workgroup_size(WG)\nfn grid_near_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let t = linear_id(wid, lid.x);\n let valid = t < P.n;\n var i = 0u;\n var pi = vec4f(0.0);\n var f = vec3f(0.0);\n if (valid) {\n i = sortedIdx[t]; // sorted order (D24)\n pi = pos[i];\n let gf = f32(P.gridMax);\n let q = (pi.xyz - S.gridMin.xyz) * S.invCellSize;\n var c0 = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));\n if (P.dim == 2u) { c0.z = 0; } // 2D: the one z plane\n let g = i32(P.gridMax);\n var inside = c0.x >= 0 && c0.x < g && c0.y >= 0 && c0.y < g;\n if (P.dim == 3u) { inside = inside && c0.z >= 0 && c0.z < g; }\n if (inside) {\n let zr = select(0, 1, P.dim == 3u);\n for (var dz = -zr; dz <= zr; dz = dz + 1) {\n for (var dy = -1; dy <= 1; dy = dy + 1) {\n for (var dx = -1; dx <= 1; dx = dx + 1) {\n let cx = c0.x + dx;\n let cy = c0.y + dy;\n let cz = c0.z + dz;\n if (cx < 0 || cx >= g || cy < 0 || cy >= g || cz < 0 || cz >= g) { continue; }\n let c = u32(cx) + P.gridMax * (u32(cy) + select(0u, P.gridMax * u32(cz), P.dim == 3u));\n f = f + cell_sum(i, pi, c, dx == 0 && dy == 0 && dz == 0);\n }\n }\n }\n } else {\n f = cell_sum(i, pi, grid_cells(), true); // an outside node: the pseudo-cell alone\n }\n }\n // epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it (K3's text)\n var sw = 0.0;\n var tr = 0.0;\n if (valid) {\n f = f + gravity_force(pi);\n let fnew = load_force(i) + f;\n store_force(i, fnew);\n if (SWING_MODE == 1u) { // NetworkX: positions and forces mixed, every node (7.2)\n sw = pi.w * length(pi.xyz - fnew);\n tr = 0.5 * pi.w * length(pi.xyz + fnew);\n } else if (!mask_bit(fixedMask[i >> 5u], i)) { // paper: free nodes only\n let fold = load_old(i);\n sw = pi.w * length(fnew - fold);\n tr = 0.5 * pi.w * length(fnew + fold);\n }\n }\n let tt = wg_reduce_vec4(vec4f(sw, tr, 0.0, 0.0), lid.x, 0u); // uniform control flow: 256 -> 1\n if (lid.x == 0u) { partials[group_id(wid)].swingTraction = tt.xy; }\n}\n";
|
|
19
|
+
//# sourceMappingURL=grid-near-field.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"grid-near-field.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-near-field.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;GAgBG;AACH,eAAO,MAAM,iBAAiB,27KA8G7B,CAAC"}
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* G7, the `grid-near-field` kernel body (spec 7.7, 7.6, 7.20; P4-T10, P4-T13; D24): per node `i = sortedIdx[t]`, the
|
|
3
|
+
* exact pair law of K3 (`LAW` 0: `|F| = k m_i m_j / d` with the 0.01 floor; `LAW` 1: FR's unfloored `k^2 / d`;
|
|
4
|
+
* `LAW` 2: the unfloored coulomb `-g m_i m_j / d^2`; the antisymmetric coincident kick at the law's magnitude at
|
|
5
|
+
* d = 0.01, PD-22) over the 9 (27) finest
|
|
6
|
+
* cells around its own, or over the outside pseudo-cell alone for an outside node; a cell above `nearMax` entries
|
|
7
|
+
* is sampled by `nearMax` INDEPENDENT draws with replacement, draw `k` reading the slot
|
|
8
|
+
* `lowbias32(((c ^ (iteration * 0x9E3779B9)) ^ seed) ^ (k * 0x85EBCA6B)) % count` (every slot's inclusion
|
|
9
|
+
* probability is `nearMax / count` whatever its position in the sorted order, so a duplicated draw is counted twice
|
|
10
|
+
* and the node itself, when drawn, is skipped and not replaced), and scaled by `others / sampled` where `sampled` is
|
|
11
|
+
* the realised number of draws that were not the node (the Horvitz-Thompson form of PD-15 / DEP-P4-K: given
|
|
12
|
+
* `sampled = s`, those `s` draws are i.i.d. uniform over the `others` slots, so the expectation of the scaled sum is
|
|
13
|
+
* the exact cell sum whenever `s >= 1`; the G4 record's G4-F2 row carries the measurement); then
|
|
14
|
+
* the fused epilogue of K3 (gravity, `force +=`, the swing / traction workgroup reduction in uniform control flow).
|
|
15
|
+
* The helpers `load_force`, `store_force`, `load_old`, `kick_magnitude`, `gravity_force` and the epilogue are K3's
|
|
16
|
+
* text. Body only; normative text.
|
|
17
|
+
*/
|
|
18
|
+
export const gridNearFieldWgsl = /* wgsl */ `
|
|
19
|
+
fn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }
|
|
20
|
+
fn store_force(i: u32, f: vec3f) {
|
|
21
|
+
force[3u * i] = f.x;
|
|
22
|
+
force[3u * i + 1u] = f.y;
|
|
23
|
+
force[3u * i + 2u] = f.z;
|
|
24
|
+
}
|
|
25
|
+
fn load_old(i: u32) -> vec3f { return vec3f(oldForce[3u * i], oldForce[3u * i + 1u], oldForce[3u * i + 2u]); }
|
|
26
|
+
fn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: centroid (GRAVITY_CENTER 0) or origin (1); regular or strong
|
|
27
|
+
var q = pi.xyz;
|
|
28
|
+
if (GRAVITY_CENTER == 0u) { q = pi.xyz - S.centroid.xyz; }
|
|
29
|
+
if (STRONG_GRAVITY) { return -P.gravity * pi.w * q; }
|
|
30
|
+
let d = length(q);
|
|
31
|
+
if (d > FA2_DIST_FLOOR) { return -P.gravity * pi.w * q / d; }
|
|
32
|
+
return vec3f(0.0);
|
|
33
|
+
}
|
|
34
|
+
fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
|
|
35
|
+
fn kick_magnitude(mi: f32, mj: f32) -> f32 { // the law's magnitude at d = FA2_DIST_FLOOR (the P5 plan's PD-10)
|
|
36
|
+
if (LAW == 1u) { return P.frK * P.frK / FA2_DIST_FLOOR; }
|
|
37
|
+
if (LAW == 2u) { return -P.coulomb * mi * mj / FA2_DIST_FLOOR_SQ; }
|
|
38
|
+
return P.scalingRatio * mi * mj / FA2_DIST_FLOOR;
|
|
39
|
+
}
|
|
40
|
+
fn pair_force(i: u32, pi: vec4f, jj: u32, o: vec4f) -> vec3f { // the exact pair law of K3 (7.6, 7.20): the floor (FA2 only), the coincident kick
|
|
41
|
+
let d = pi.xyz - o.xyz;
|
|
42
|
+
var d2 = dot(d, d);
|
|
43
|
+
if (d2 < FA2_COINCIDENT_SQ) { return kick_dir(i, jj, P.dim) * kick_magnitude(pi.w, o.w); }
|
|
44
|
+
if (LAW == 0u) { d2 = max(d2, FA2_DIST_FLOOR_SQ); }
|
|
45
|
+
let k = P.scalingRatio * pi.w * o.w;
|
|
46
|
+
if (LAW == 1u) { return d * (P.frK * P.frK / d2); }
|
|
47
|
+
if (LAW == 2u) { return d * (-P.coulomb * pi.w * o.w / (d2 * sqrt(d2))); }
|
|
48
|
+
return d * (k / d2);
|
|
49
|
+
}
|
|
50
|
+
fn cell_sum(i: u32, pi: vec4f, c: u32, own: bool) -> vec3f { // one finest cell: exact below nearMax entries, Horvitz-Thompson above (PD-15)
|
|
51
|
+
let start = cellStart[c];
|
|
52
|
+
let count = cellStart[c + 1u] - start;
|
|
53
|
+
var f = vec3f(0.0);
|
|
54
|
+
if (count <= P.nearMax) {
|
|
55
|
+
for (var k = start; k < start + count; k = k + 1u) {
|
|
56
|
+
let jj = sortedIdx[k];
|
|
57
|
+
if (jj != i) { f = f + pair_force(i, pi, jj, pos[jj]); }
|
|
58
|
+
}
|
|
59
|
+
return f;
|
|
60
|
+
}
|
|
61
|
+
let base = (c ^ (P.iterationIndex * 0x9E3779B9u)) ^ P.seed; // the per-iteration draw seed (7.16)
|
|
62
|
+
var sampled = 0u;
|
|
63
|
+
for (var k = 0u; k < P.nearMax; k = k + 1u) {
|
|
64
|
+
let jj = sortedIdx[start + (lowbias32(base ^ (k * 0x85EBCA6Bu)) % count)]; // draw k: independent inclusion, with replacement (PD-15)
|
|
65
|
+
if (jj == i) { continue; }
|
|
66
|
+
f = f + pair_force(i, pi, jj, pos[jj]);
|
|
67
|
+
sampled = sampled + 1u;
|
|
68
|
+
}
|
|
69
|
+
if (sampled == 0u) { return vec3f(0.0); }
|
|
70
|
+
let others = select(count, count - 1u, own);
|
|
71
|
+
return f * (f32(others) / f32(sampled)); // others / sampled over the realised sample (DEP-P4-K)
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
@compute @workgroup_size(WG)
|
|
75
|
+
fn grid_near_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
76
|
+
let t = linear_id(wid, lid.x);
|
|
77
|
+
let valid = t < P.n;
|
|
78
|
+
var i = 0u;
|
|
79
|
+
var pi = vec4f(0.0);
|
|
80
|
+
var f = vec3f(0.0);
|
|
81
|
+
if (valid) {
|
|
82
|
+
i = sortedIdx[t]; // sorted order (D24)
|
|
83
|
+
pi = pos[i];
|
|
84
|
+
let gf = f32(P.gridMax);
|
|
85
|
+
let q = (pi.xyz - S.gridMin.xyz) * S.invCellSize;
|
|
86
|
+
var c0 = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));
|
|
87
|
+
if (P.dim == 2u) { c0.z = 0; } // 2D: the one z plane
|
|
88
|
+
let g = i32(P.gridMax);
|
|
89
|
+
var inside = c0.x >= 0 && c0.x < g && c0.y >= 0 && c0.y < g;
|
|
90
|
+
if (P.dim == 3u) { inside = inside && c0.z >= 0 && c0.z < g; }
|
|
91
|
+
if (inside) {
|
|
92
|
+
let zr = select(0, 1, P.dim == 3u);
|
|
93
|
+
for (var dz = -zr; dz <= zr; dz = dz + 1) {
|
|
94
|
+
for (var dy = -1; dy <= 1; dy = dy + 1) {
|
|
95
|
+
for (var dx = -1; dx <= 1; dx = dx + 1) {
|
|
96
|
+
let cx = c0.x + dx;
|
|
97
|
+
let cy = c0.y + dy;
|
|
98
|
+
let cz = c0.z + dz;
|
|
99
|
+
if (cx < 0 || cx >= g || cy < 0 || cy >= g || cz < 0 || cz >= g) { continue; }
|
|
100
|
+
let c = u32(cx) + P.gridMax * (u32(cy) + select(0u, P.gridMax * u32(cz), P.dim == 3u));
|
|
101
|
+
f = f + cell_sum(i, pi, c, dx == 0 && dy == 0 && dz == 0);
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
} else {
|
|
106
|
+
f = cell_sum(i, pi, grid_cells(), true); // an outside node: the pseudo-cell alone
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
// epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it (K3's text)
|
|
110
|
+
var sw = 0.0;
|
|
111
|
+
var tr = 0.0;
|
|
112
|
+
if (valid) {
|
|
113
|
+
f = f + gravity_force(pi);
|
|
114
|
+
let fnew = load_force(i) + f;
|
|
115
|
+
store_force(i, fnew);
|
|
116
|
+
if (SWING_MODE == 1u) { // NetworkX: positions and forces mixed, every node (7.2)
|
|
117
|
+
sw = pi.w * length(pi.xyz - fnew);
|
|
118
|
+
tr = 0.5 * pi.w * length(pi.xyz + fnew);
|
|
119
|
+
} else if (!mask_bit(fixedMask[i >> 5u], i)) { // paper: free nodes only
|
|
120
|
+
let fold = load_old(i);
|
|
121
|
+
sw = pi.w * length(fnew - fold);
|
|
122
|
+
tr = 0.5 * pi.w * length(fnew + fold);
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
let tt = wg_reduce_vec4(vec4f(sw, tr, 0.0, 0.0), lid.x, 0u); // uniform control flow: 256 -> 1
|
|
126
|
+
if (lid.x == 0u) { partials[group_id(wid)].swingTraction = tt.xy; }
|
|
127
|
+
}
|
|
128
|
+
`;
|
|
129
|
+
//# sourceMappingURL=grid-near-field.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"grid-near-field.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-near-field.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;GAgBG;AACH,MAAM,CAAC,MAAM,iBAAiB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA8G3C,CAAC"}
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `histogram` kernel body (spec 6 row 5; P4-T3): one atomicAdd per key into the global `hist` (zeroed by a fill
|
|
3
|
+
* dispatch earlier in the pass, PD-4). Order-independent, hence deterministic. A key >= P.bins is not counted (the
|
|
4
|
+
* caller's contract; the grid's keys are always < cells + 1). Body only; normative text.
|
|
5
|
+
*/
|
|
6
|
+
export declare const histogramWgsl = "\n@compute @workgroup_size(WG)\nfn histogram(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n if (i >= P.count) { return; } // no barrier follows\n let k = keys[i];\n if (k < P.bins) { atomicAdd(&hist[k], 1u); } // order-independent: the count is the same whatever the schedule\n}\n";
|
|
7
|
+
//# sourceMappingURL=histogram.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"histogram.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/histogram.wgsl.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH,eAAO,MAAM,aAAa,uaAQzB,CAAC"}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `histogram` kernel body (spec 6 row 5; P4-T3): one atomicAdd per key into the global `hist` (zeroed by a fill
|
|
3
|
+
* dispatch earlier in the pass, PD-4). Order-independent, hence deterministic. A key >= P.bins is not counted (the
|
|
4
|
+
* caller's contract; the grid's keys are always < cells + 1). Body only; normative text.
|
|
5
|
+
*/
|
|
6
|
+
export const histogramWgsl = /* wgsl */ `
|
|
7
|
+
@compute @workgroup_size(WG)
|
|
8
|
+
fn histogram(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
9
|
+
let i = linear_id(wid, lid.x);
|
|
10
|
+
if (i >= P.count) { return; } // no barrier follows
|
|
11
|
+
let k = keys[i];
|
|
12
|
+
if (k < P.bins) { atomicAdd(&hist[k], 1u); } // order-independent: the count is the same whatever the schedule
|
|
13
|
+
}
|
|
14
|
+
`;
|
|
15
|
+
//# sourceMappingURL=histogram.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"histogram.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/histogram.wgsl.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH,MAAM,CAAC,MAAM,aAAa,GAAG,UAAU,CAAC;;;;;;;;CAQvC,CAAC"}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `indirect-finalize` kernel body (spec 5.4; P4-T1): one lane turns the device-side count `counters[P.countIndex]`
|
|
3
|
+
* into the `(x, y, 1)` of an indirect dispatch by plan1d's rule (spec 5.2) and writes it with the count into the
|
|
4
|
+
* 16-byte slot `P.slot` of `args` (PD-2). No barrier follows the early return of the other lanes. Body only (spec 3.5,
|
|
5
|
+
* D9); the text is normative: the P4 sabotage rows are textual edits of it.
|
|
6
|
+
*/
|
|
7
|
+
export declare const indirectFinalizeWgsl = "\n@compute @workgroup_size(WG)\nfn indirect_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n let count = counters[P.countIndex];\n let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap of (count + wg - 1) above 2^32 - wg: plan1d's rule (5.2) for ANY u32 count\n var x = groups;\n var y = 1u;\n if (groups > MAX_WORKGROUPS_PER_DIM) { // the 2D split; y <= 1,025 for any u32 count\n x = MAX_WORKGROUPS_PER_DIM;\n y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;\n }\n let base = 4u * P.slot; // 16-byte slots: (x, y, 1, count) (PD-2)\n args[base] = x;\n args[base + 1u] = y;\n args[base + 2u] = 1u;\n args[base + 3u] = count;\n}\n";
|
|
8
|
+
//# sourceMappingURL=indirect-finalize.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"indirect-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/indirect-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,eAAO,MAAM,oBAAoB,26BAkBhC,CAAC"}
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `indirect-finalize` kernel body (spec 5.4; P4-T1): one lane turns the device-side count `counters[P.countIndex]`
|
|
3
|
+
* into the `(x, y, 1)` of an indirect dispatch by plan1d's rule (spec 5.2) and writes it with the count into the
|
|
4
|
+
* 16-byte slot `P.slot` of `args` (PD-2). No barrier follows the early return of the other lanes. Body only (spec 3.5,
|
|
5
|
+
* D9); the text is normative: the P4 sabotage rows are textual edits of it.
|
|
6
|
+
*/
|
|
7
|
+
export const indirectFinalizeWgsl = /* wgsl */ `
|
|
8
|
+
@compute @workgroup_size(WG)
|
|
9
|
+
fn indirect_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
10
|
+
if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
|
|
11
|
+
let count = counters[P.countIndex];
|
|
12
|
+
let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap of (count + wg - 1) above 2^32 - wg: plan1d's rule (5.2) for ANY u32 count
|
|
13
|
+
var x = groups;
|
|
14
|
+
var y = 1u;
|
|
15
|
+
if (groups > MAX_WORKGROUPS_PER_DIM) { // the 2D split; y <= 1,025 for any u32 count
|
|
16
|
+
x = MAX_WORKGROUPS_PER_DIM;
|
|
17
|
+
y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;
|
|
18
|
+
}
|
|
19
|
+
let base = 4u * P.slot; // 16-byte slots: (x, y, 1, count) (PD-2)
|
|
20
|
+
args[base] = x;
|
|
21
|
+
args[base + 1u] = y;
|
|
22
|
+
args[base + 2u] = 1u;
|
|
23
|
+
args[base + 3u] = count;
|
|
24
|
+
}
|
|
25
|
+
`;
|
|
26
|
+
//# sourceMappingURL=indirect-finalize.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"indirect-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/indirect-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;CAkB9C,CAAC"}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `radix-hist` kernel body (spec 6 row 6; P4-T4): the RADIX_BINS-bin (2^8) digit histogram of one WG-wide block
|
|
3
|
+
* of keys, privatised in workgroup memory (atomics on workgroup memory are order-independent) and written
|
|
4
|
+
* DIGIT-MAJOR, `hist[digit * P.groups + group]`, so one exclusiveScan over the table yields, per digit, the offsets
|
|
5
|
+
* of the workgroups in workgroup order -- what a stable LSD scatter needs. The bitwise operators act on a KEY and a
|
|
6
|
+
* DIGIT, never on an arc index (house rule). Body only; normative text.
|
|
7
|
+
*/
|
|
8
|
+
export declare const radixHistWgsl = "\nvar<workgroup> local: array<atomic<u32>, RADIX_BINS>;\n\n@compute @workgroup_size(WG)\nfn radix_hist(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n for (var b = lid.x; b < RADIX_BINS; b = b + WG) { atomicStore(&local[b], 0u); }\n workgroupBarrier();\n let g = group_id(wid);\n let i = g * WG + lid.x;\n if (i < P.count) {\n let d = (keys[i] >> P.shift) & RADIX_DIGIT_MASK; // the pass's digit\n atomicAdd(&local[d], 1u);\n }\n workgroupBarrier();\n // Above MAX_1D_ITEMS plan1d pads the 2D grid to x * y >= P.groups workgroups; a padding workgroup (g >= P.groups)\n // holds an all-zero table and its digit-major slot b * P.groups + g is digit b + 1's slot of a REAL group, so it\n // must not store. Uniform per workgroup, no barrier inside. Every test size fits 1D (the ladder tops at 2^22); no\n // case reaches this branch, so the guard is proved by reading, not by a run.\n if (g < P.groups) {\n for (var b = lid.x; b < RADIX_BINS; b = b + WG) { hist[b * P.groups + g] = atomicLoad(&local[b]); } // digit-major\n }\n}\n";
|
|
9
|
+
//# sourceMappingURL=radix-hist.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"radix-hist.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/radix-hist.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,eAAO,MAAM,aAAa,+nCAsBzB,CAAC"}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `radix-hist` kernel body (spec 6 row 6; P4-T4): the RADIX_BINS-bin (2^8) digit histogram of one WG-wide block
|
|
3
|
+
* of keys, privatised in workgroup memory (atomics on workgroup memory are order-independent) and written
|
|
4
|
+
* DIGIT-MAJOR, `hist[digit * P.groups + group]`, so one exclusiveScan over the table yields, per digit, the offsets
|
|
5
|
+
* of the workgroups in workgroup order -- what a stable LSD scatter needs. The bitwise operators act on a KEY and a
|
|
6
|
+
* DIGIT, never on an arc index (house rule). Body only; normative text.
|
|
7
|
+
*/
|
|
8
|
+
export const radixHistWgsl = /* wgsl */ `
|
|
9
|
+
var<workgroup> local: array<atomic<u32>, RADIX_BINS>;
|
|
10
|
+
|
|
11
|
+
@compute @workgroup_size(WG)
|
|
12
|
+
fn radix_hist(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
13
|
+
for (var b = lid.x; b < RADIX_BINS; b = b + WG) { atomicStore(&local[b], 0u); }
|
|
14
|
+
workgroupBarrier();
|
|
15
|
+
let g = group_id(wid);
|
|
16
|
+
let i = g * WG + lid.x;
|
|
17
|
+
if (i < P.count) {
|
|
18
|
+
let d = (keys[i] >> P.shift) & RADIX_DIGIT_MASK; // the pass's digit
|
|
19
|
+
atomicAdd(&local[d], 1u);
|
|
20
|
+
}
|
|
21
|
+
workgroupBarrier();
|
|
22
|
+
// Above MAX_1D_ITEMS plan1d pads the 2D grid to x * y >= P.groups workgroups; a padding workgroup (g >= P.groups)
|
|
23
|
+
// holds an all-zero table and its digit-major slot b * P.groups + g is digit b + 1's slot of a REAL group, so it
|
|
24
|
+
// must not store. Uniform per workgroup, no barrier inside. Every test size fits 1D (the ladder tops at 2^22); no
|
|
25
|
+
// case reaches this branch, so the guard is proved by reading, not by a run.
|
|
26
|
+
if (g < P.groups) {
|
|
27
|
+
for (var b = lid.x; b < RADIX_BINS; b = b + WG) { hist[b * P.groups + g] = atomicLoad(&local[b]); } // digit-major
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
`;
|
|
31
|
+
//# sourceMappingURL=radix-hist.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"radix-hist.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/radix-hist.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,MAAM,CAAC,MAAM,aAAa,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;CAsBvC,CAAC"}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `radix-scatter` kernel body (spec 6 row 6; P4-T4): the stable scatter of one LSD pass. Lane 0 ranks the block's
|
|
3
|
+
* keys serially in index order with a RADIX_BINS-entry counter table (PD-5: deterministic and stable, WG steps per
|
|
4
|
+
* workgroup; ponytail: the 8-way split ranking of GraphWaGu is the upgrade if T-6 shows the sort on the critical
|
|
5
|
+
* path), then every lane writes its key and value at `offsets[digit * P.groups + group] + rank`. Body only;
|
|
6
|
+
* normative text.
|
|
7
|
+
*/
|
|
8
|
+
export declare const radixScatterWgsl = "\nvar<workgroup> rank: array<u32, WG>;\nvar<workgroup> cnt: array<u32, RADIX_BINS>;\n\n@compute @workgroup_size(WG)\nfn radix_scatter(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n for (var b = lid.x; b < RADIX_BINS; b = b + WG) { cnt[b] = 0u; }\n let g = group_id(wid);\n let i = g * WG + lid.x;\n var key = 0u;\n var d = 0u;\n if (i < P.count) {\n key = keys[i];\n d = (key >> P.shift) & RADIX_DIGIT_MASK;\n }\n workgroupBarrier();\n if (lid.x == 0u) { // serial stable ranking (PD-5)\n let last = min(WG, P.count - g * WG);\n for (var s = 0u; s < last; s = s + 1u) {\n let ds = (keys[g * WG + s] >> P.shift) & RADIX_DIGIT_MASK;\n rank[s] = cnt[ds];\n cnt[ds] = cnt[ds] + 1u;\n }\n }\n workgroupBarrier();\n if (i < P.count) {\n let dst = offsets[d * P.groups + g] + rank[lid.x];\n keysOut[dst] = key;\n valsOut[dst] = vals[i];\n }\n}\n";
|
|
9
|
+
//# sourceMappingURL=radix-scatter.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"radix-scatter.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/radix-scatter.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,eAAO,MAAM,gBAAgB,iiCA+B5B,CAAC"}
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `radix-scatter` kernel body (spec 6 row 6; P4-T4): the stable scatter of one LSD pass. Lane 0 ranks the block's
|
|
3
|
+
* keys serially in index order with a RADIX_BINS-entry counter table (PD-5: deterministic and stable, WG steps per
|
|
4
|
+
* workgroup; ponytail: the 8-way split ranking of GraphWaGu is the upgrade if T-6 shows the sort on the critical
|
|
5
|
+
* path), then every lane writes its key and value at `offsets[digit * P.groups + group] + rank`. Body only;
|
|
6
|
+
* normative text.
|
|
7
|
+
*/
|
|
8
|
+
export const radixScatterWgsl = /* wgsl */ `
|
|
9
|
+
var<workgroup> rank: array<u32, WG>;
|
|
10
|
+
var<workgroup> cnt: array<u32, RADIX_BINS>;
|
|
11
|
+
|
|
12
|
+
@compute @workgroup_size(WG)
|
|
13
|
+
fn radix_scatter(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
14
|
+
for (var b = lid.x; b < RADIX_BINS; b = b + WG) { cnt[b] = 0u; }
|
|
15
|
+
let g = group_id(wid);
|
|
16
|
+
let i = g * WG + lid.x;
|
|
17
|
+
var key = 0u;
|
|
18
|
+
var d = 0u;
|
|
19
|
+
if (i < P.count) {
|
|
20
|
+
key = keys[i];
|
|
21
|
+
d = (key >> P.shift) & RADIX_DIGIT_MASK;
|
|
22
|
+
}
|
|
23
|
+
workgroupBarrier();
|
|
24
|
+
if (lid.x == 0u) { // serial stable ranking (PD-5)
|
|
25
|
+
let last = min(WG, P.count - g * WG);
|
|
26
|
+
for (var s = 0u; s < last; s = s + 1u) {
|
|
27
|
+
let ds = (keys[g * WG + s] >> P.shift) & RADIX_DIGIT_MASK;
|
|
28
|
+
rank[s] = cnt[ds];
|
|
29
|
+
cnt[ds] = cnt[ds] + 1u;
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
workgroupBarrier();
|
|
33
|
+
if (i < P.count) {
|
|
34
|
+
let dst = offsets[d * P.groups + g] + rank[lid.x];
|
|
35
|
+
keysOut[dst] = key;
|
|
36
|
+
valsOut[dst] = vals[i];
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
`;
|
|
40
|
+
//# sourceMappingURL=radix-scatter.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"radix-scatter.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/radix-scatter.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA+B1C,CAAC"}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `scan-add` kernel body (spec 6 row 2 step (c); P4-T2): adds the exclusive prefix of its block's sum,
|
|
3
|
+
* `blockOffsets[group]`, to every element of the block. Body only; normative text.
|
|
4
|
+
*/
|
|
5
|
+
export declare const scanAddWgsl = "\n@compute @workgroup_size(WG)\nfn scan_add(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let g = group_id(wid);\n let i = g * WG + lid.x;\n if (i >= P.count) { return; } // no barrier follows\n out[i] = out[i] + blockOffsets[g];\n}\n";
|
|
6
|
+
//# sourceMappingURL=scan-add.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"scan-add.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/scan-add.wgsl.ts"],"names":[],"mappings":"AAAA;;;GAGG;AACH,eAAO,MAAM,WAAW,uUAQvB,CAAC"}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `scan-add` kernel body (spec 6 row 2 step (c); P4-T2): adds the exclusive prefix of its block's sum,
|
|
3
|
+
* `blockOffsets[group]`, to every element of the block. Body only; normative text.
|
|
4
|
+
*/
|
|
5
|
+
export const scanAddWgsl = /* wgsl */ `
|
|
6
|
+
@compute @workgroup_size(WG)
|
|
7
|
+
fn scan_add(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
8
|
+
let g = group_id(wid);
|
|
9
|
+
let i = g * WG + lid.x;
|
|
10
|
+
if (i >= P.count) { return; } // no barrier follows
|
|
11
|
+
out[i] = out[i] + blockOffsets[g];
|
|
12
|
+
}
|
|
13
|
+
`;
|
|
14
|
+
//# sourceMappingURL=scan-add.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"scan-add.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/scan-add.wgsl.ts"],"names":[],"mappings":"AAAA;;;GAGG;AACH,MAAM,CAAC,MAAM,WAAW,GAAG,UAAU,CAAC;;;;;;;;CAQrC,CAAC"}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `scan-block` kernel body (spec 6 row 2; P4-T2): an exclusive prefix sum of one WG-wide block of `src` in
|
|
3
|
+
* workgroup memory (Hillis-Steele, log2(WG) rounds of two barriers each, every lane in uniform control flow) into
|
|
4
|
+
* `out`, and the block's inclusive total into `blockSums[group]`. u32 addition is exact in any order, so the
|
|
5
|
+
* output is bitwise the same on every adapter (PD-3). Body only (spec 3.5, D9); normative text.
|
|
6
|
+
*/
|
|
7
|
+
export declare const scanBlockWgsl = "\nvar<workgroup> sh: array<u32, WG>;\n\n@compute @workgroup_size(WG)\nfn scan_block(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let g = group_id(wid);\n let i = g * WG + lid.x;\n var v = 0u;\n if (i < P.count) { v = src[i]; }\n sh[lid.x] = v;\n workgroupBarrier();\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan; uniform: every lane runs every round\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let inclusive = sh[lid.x];\n if (i < P.count) { out[i] = inclusive - v; } // exclusive = inclusive - own value\n if (lid.x == WG - 1u) { blockSums[g] = inclusive; } // the block total (the last lane's inclusive sum)\n}\n";
|
|
8
|
+
//# sourceMappingURL=scan-block.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"scan-block.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/scan-block.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,eAAO,MAAM,aAAa,i4BAsBzB,CAAC"}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `scan-block` kernel body (spec 6 row 2; P4-T2): an exclusive prefix sum of one WG-wide block of `src` in
|
|
3
|
+
* workgroup memory (Hillis-Steele, log2(WG) rounds of two barriers each, every lane in uniform control flow) into
|
|
4
|
+
* `out`, and the block's inclusive total into `blockSums[group]`. u32 addition is exact in any order, so the
|
|
5
|
+
* output is bitwise the same on every adapter (PD-3). Body only (spec 3.5, D9); normative text.
|
|
6
|
+
*/
|
|
7
|
+
export const scanBlockWgsl = /* wgsl */ `
|
|
8
|
+
var<workgroup> sh: array<u32, WG>;
|
|
9
|
+
|
|
10
|
+
@compute @workgroup_size(WG)
|
|
11
|
+
fn scan_block(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
12
|
+
let g = group_id(wid);
|
|
13
|
+
let i = g * WG + lid.x;
|
|
14
|
+
var v = 0u;
|
|
15
|
+
if (i < P.count) { v = src[i]; }
|
|
16
|
+
sh[lid.x] = v;
|
|
17
|
+
workgroupBarrier();
|
|
18
|
+
for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan; uniform: every lane runs every round
|
|
19
|
+
var t = 0u;
|
|
20
|
+
if (lid.x >= s) { t = sh[lid.x - s]; }
|
|
21
|
+
workgroupBarrier();
|
|
22
|
+
sh[lid.x] = sh[lid.x] + t;
|
|
23
|
+
workgroupBarrier();
|
|
24
|
+
}
|
|
25
|
+
let inclusive = sh[lid.x];
|
|
26
|
+
if (i < P.count) { out[i] = inclusive - v; } // exclusive = inclusive - own value
|
|
27
|
+
if (lid.x == WG - 1u) { blockSums[g] = inclusive; } // the block total (the last lane's inclusive sum)
|
|
28
|
+
}
|
|
29
|
+
`;
|
|
30
|
+
//# sourceMappingURL=scan-block.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"scan-block.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/scan-block.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,CAAC,MAAM,aAAa,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;CAsBvC,CAAC"}
|
|
@@ -1,13 +1,27 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* The `segmented-reduce` kernel body (spec 6 row 3; contract 4.5)
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
2
|
+
* The `segmented-reduce` kernel body (spec 6 row 3; contract 4.5) in its three degree tiers (P4-T5, PD-6). Every
|
|
3
|
+
* tier folds the VALUE snippet over a row's arcs inside the bound window [P.arcBase, P.arcEnd) and writes out[i]
|
|
4
|
+
* as f32 (a row with no arcs gets the identity element: 0 for sum, F32_MAX for min, -F32_MAX for max); the row is
|
|
5
|
+
* `perm[row]` under USE_PERM and `row` otherwise, over the rows [P.start, P.end) of the dispatch. TIER 0 is one row
|
|
6
|
+
* per thread (no barrier, so its early return is legal); TIER 1 is 32 lanes per row and WG / 32 rows per workgroup,
|
|
7
|
+
* reduced by a five-step tree in workgroup memory (bitwise the same on every subgroup size); TIER 2 is one row per
|
|
8
|
+
* workgroup through `wg_reduce_f32` of the prelude (the subgroup variant when the feature exists, the workgroup twin
|
|
9
|
+
* otherwise -- hence `needs: ["subgroups"]`). The tier bodies are functions called under `if (TIER == 0u)`, an
|
|
10
|
+
* override, so every barrier is reached in uniform control flow. The body is normative: a sabotage mutation
|
|
11
|
+
* (test/helpers/sabotage.ts) is a textual edit of it, so it is not restyled.
|
|
12
|
+
*
|
|
13
|
+
* TIER 0 folds its row through `row_fold_dense`, a stride-one copy of `row_fold`, because the shader compiler emits
|
|
14
|
+
* `row_fold(i, 0u, 1u)` as a call and leaves the stride in a parameter: the loop then walks the row with a runtime
|
|
15
|
+
* step, which costs the strength-reduced addressing into colIdx / weights and the unrolling that keeps several loads
|
|
16
|
+
* in flight per thread. PageRank's out-weight pass, the only caller that passes no tiers today, runs TIER 0 alone,
|
|
17
|
+
* and under the tiers TIER 0 still folds the low-degree rows, which are most of them. The two folds spell their
|
|
18
|
+
* locals apart (`lo` / `hi` / `k` against `a0` / `a1`) so that each sabotage row names exactly one of them; the
|
|
19
|
+
* VALUE snippet sees the same `row`, `arc`, `nbr` and `weight` in both.
|
|
7
20
|
*/
|
|
8
21
|
/**
|
|
9
|
-
* Entry point `segmented_reduce`; overrides OP (0 sum, 1 min, 2 max) and TIER (0
|
|
10
|
-
* assigning `v` from `row`, `arc`, `nbr`, `weight` (`target`
|
|
22
|
+
* Entry point `segmented_reduce`; overrides OP (0 sum, 1 min, 2 max) and TIER (0 thread-per-row, 1 32-lanes-per-row,
|
|
23
|
+
* 2 workgroup-per-row); snippet slot VALUE: statements assigning `v` from `row`, `arc`, `nbr`, `weight` (`target`
|
|
24
|
+
* is a WGSL reserved word, hence `nbr`).
|
|
11
25
|
*/
|
|
12
|
-
export declare const segmentedReduceWgsl = "\nfn identity() -> f32 { if (OP == 1u) { return F32_MAX; } if (OP == 2u) { return -F32_MAX; } return 0.0; }\nfn comb(a: f32, b: f32) -> f32 { if (OP == 1u) { return min(a, b); } if (OP == 2u) { return max(a, b); } return a + b; }\
|
|
26
|
+
export declare const segmentedReduceWgsl = "\nfn identity() -> f32 { if (OP == 1u) { return F32_MAX; } if (OP == 2u) { return -F32_MAX; } return 0.0; }\nfn comb(a: f32, b: f32) -> f32 { if (OP == 1u) { return min(a, b); } if (OP == 2u) { return max(a, b); } return a + b; }\nfn row_node(row: u32) -> u32 { return select(row, perm[row], USE_PERM); }\nfn row_fold(i: u32, lane: u32, step: u32) -> f32 { // the arcs of row i this lane walks inside the bound window [P.arcBase, P.arcEnd)\n let row = i; // the CSR row the VALUE snippet may name (the node index, under USE_PERM too)\n let a0 = max(rowPtr[i], P.arcBase);\n let a1 = min(rowPtr[i + 1u], P.arcEnd);\n var acc = identity();\n for (var arc = a0 + lane; arc < a1; arc = arc + step) {\n let nbr = colIdx[arc - P.arcBase]; // the neighbour index (`target` is a WGSL reserved word)\n var weight = 1.0;\n if (HAS_WEIGHTS) { weight = weights[arc - P.arcBase]; }\n var v = 0.0;\n //@@VALUE@@\n acc = comb(acc, v);\n }\n return acc;\n}\nfn row_fold_dense(i: u32) -> f32 { // TIER 0's stride-one twin of row_fold (see the header)\n let row = i; // the CSR row the VALUE snippet may name (as in row_fold)\n let lo = max(rowPtr[i], P.arcBase);\n let hi = min(rowPtr[i + 1u], P.arcEnd);\n var acc = identity();\n for (var arc = lo; arc < hi; arc = arc + 1u) {\n let k = arc - P.arcBase; // the window-local index; this walk is contiguous\n let nbr = colIdx[k];\n var weight = 1.0;\n if (HAS_WEIGHTS) { weight = weights[k]; }\n var v = 0.0;\n //@@VALUE@@\n acc = comb(acc, v);\n }\n return acc;\n}\nfn finish(i: u32, acc: f32) { out[i] = select(acc, comb(out[i], acc), P.accumulate == 1u); }\nfn tier0(wid: vec3<u32>, lane: u32) { // TIER 0: one row per thread over [P.start, P.end); no barrier, so the early return is legal (3.5 rule 1)\n let row = linear_id(wid, lane) + P.start;\n if (row >= P.end) { return; }\n let i = row_node(row);\n finish(i, row_fold_dense(i));\n}\n\nvar<workgroup> sh: array<f32, WG>;\n\nfn tiered(wid: vec3<u32>, lid: u32) { // TIER 1: 32 lanes per row, WG / 32 rows per workgroup; TIER 2: WG lanes per row (PD-6)\n let g = group_id(wid);\n var row = P.start + g;\n var lane = lid;\n var step = WG;\n if (TIER == 1u) { row = P.start + g * (WG / 32u) + lid / 32u; lane = lid % 32u; step = 32u; }\n let valid = row < P.end;\n var i = 0u;\n var acc = identity();\n if (valid) { i = row_node(row); acc = row_fold(i, lane, step); }\n if (TIER == 1u) {\n sh[lid] = acc;\n workgroupBarrier();\n for (var s = 16u; s >= 1u; s = s / 2u) { // the five-step tree over each 32-lane group; every lane runs every step\n var t = identity();\n if (lane < s) { t = sh[lid + s]; }\n workgroupBarrier();\n sh[lid] = comb(sh[lid], t);\n workgroupBarrier();\n }\n if (valid && lane == 0u) { finish(i, sh[lid]); }\n }\n if (TIER == 2u) {\n let t = wg_reduce_f32(acc, lid, OP); // the workgroup tree of the prelude (subgroup variant when available)\n if (valid && lid == 0u) { finish(i, t); }\n }\n}\n\n@compute @workgroup_size(WG)\nfn segmented_reduce(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n if (TIER == 0u) { tier0(wid, lid.x); } else { tiered(wid, lid.x); }\n}\n";
|
|
13
27
|
//# sourceMappingURL=segmented-reduce.wgsl.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"segmented-reduce.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/segmented-reduce.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"segmented-reduce.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/segmented-reduce.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAEH;;;;GAIG;AACH,eAAO,MAAM,mBAAmB,gmHA6E/B,CAAC"}
|