@graphty/webgpu-graph-algorithms 0.6.4 → 0.6.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +38 -17
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-Dvq-Cc6v.js → context-DiSr6eiz.js} +45 -33
- package/dist/chunks/context-DiSr6eiz.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/algorithms/bfs.d.ts +11 -7
- package/dist/src/algorithms/bfs.d.ts.map +1 -1
- package/dist/src/algorithms/bfs.js +33 -10
- package/dist/src/algorithms/bfs.js.map +1 -1
- package/dist/src/algorithms/scope.d.ts +3 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +0 -2
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +4 -3
- package/dist/src/algorithms/sssp.d.ts.map +1 -1
- package/dist/src/algorithms/sssp.js +4 -3
- package/dist/src/algorithms/sssp.js.map +1 -1
- package/dist/src/constants.d.ts +33 -2
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +33 -2
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernel/dispatch.d.ts +2 -2
- package/dist/src/kernel/dispatch.d.ts.map +1 -1
- package/dist/src/kernel/kernel.d.ts +1 -1
- package/dist/src/kernel/kernel.js +2 -2
- package/dist/src/kernel/kernel.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +2 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernels.d.ts +15 -11
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +42 -21
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
- package/dist/src/layouts/forceatlas2.js +2 -1
- package/dist/src/layouts/forceatlas2.js.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.js +4 -2
- package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
- package/dist/src/layouts/repulsion-exact.d.ts +16 -0
- package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
- package/dist/src/layouts/repulsion-exact.js +21 -1
- package/dist/src/layouts/repulsion-exact.js.map +1 -1
- package/dist/src/layouts/repulsion-grid.d.ts +1 -1
- package/dist/src/layouts/repulsion-grid.js +1 -1
- package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
- package/dist/src/layouts/spring-electrical.js +6 -2
- package/dist/src/layouts/spring-electrical.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +3 -2
- package/dist/src/primitives/advance.d.ts.map +1 -1
- package/dist/src/primitives/advance.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +34 -38
- package/dist/src/primitives/frontier.d.ts.map +1 -1
- package/dist/src/primitives/frontier.js +24 -32
- package/dist/src/primitives/frontier.js.map +1 -1
- package/dist/src/primitives/grid-pyramid.d.ts +4 -4
- package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
- package/dist/src/primitives/grid-pyramid.js +4 -3
- package/dist/src/primitives/grid-pyramid.js.map +1 -1
- package/dist/src/primitives/grid.d.ts +13 -10
- package/dist/src/primitives/grid.d.ts.map +1 -1
- package/dist/src/primitives/grid.js +10 -7
- package/dist/src/primitives/grid.js.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +44 -49
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.js +62 -107
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
- package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
- package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
- package/dist/src/wgsl/histogram.wgsl.js +1 -1
- package/dist/webgpu-graph-algorithms.js +144 -119
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +3 -3
- package/src/algorithms/bfs.ts +34 -10
- package/src/algorithms/pagerank.ts +19 -5
- package/src/algorithms/power-iteration.ts +8 -2
- package/src/algorithms/scope.ts +3 -10
- package/src/algorithms/sssp.ts +4 -3
- package/src/constants.ts +35 -2
- package/src/kernel/dispatch.ts +2 -2
- package/src/kernel/kernel.ts +2 -2
- package/src/kernel/prelude.ts +2 -0
- package/src/kernels.ts +44 -21
- package/src/layouts/forceatlas2.ts +2 -0
- package/src/layouts/fruchterman-reingold.ts +4 -1
- package/src/layouts/repulsion-exact.ts +29 -1
- package/src/layouts/repulsion-grid.ts +1 -1
- package/src/layouts/spring-electrical.ts +8 -1
- package/src/memory/residency.ts +14 -4
- package/src/primitives/advance.ts +5 -4
- package/src/primitives/frontier.ts +42 -56
- package/src/primitives/grid-pyramid.ts +6 -5
- package/src/primitives/grid.ts +17 -12
- package/src/wgsl/advance-expand.wgsl.ts +4 -3
- package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
- package/src/wgsl/bfs-fused.wgsl.ts +6 -6
- package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
- package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
- package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
- package/src/wgsl/frontier-finalize.wgsl.ts +62 -107
- package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
- package/src/wgsl/grid-centroid.wgsl.ts +3 -2
- package/src/wgsl/grid-downsample.wgsl.ts +1 -1
- package/src/wgsl/grid-far-field.wgsl.ts +15 -8
- package/src/wgsl/grid-near-field.wgsl.ts +6 -2
- package/src/wgsl/histogram.wgsl.ts +1 -1
- package/dist/chunks/context-Dvq-Cc6v.js.map +0 -1
|
@@ -3,8 +3,9 @@
|
|
|
3
3
|
* the caller's pass, the cell keys (G1, `grid-cell-key`), the stable sort by key (G2: `radixSort` at GRID_SORT_BITS,
|
|
4
4
|
* or `countingSortByKey` when the caller asks for the set-deterministic path) and the per-cell histogram with its
|
|
5
5
|
* exclusive scan (G3: the `histogram` kernel over `cellKey` and the `scan` of it; DEP-P4-I names no grid-specific
|
|
6
|
-
* id). `cellHist` and `cellStart` hold `cells + 2` words: every real cell, the outside
|
|
7
|
-
*
|
|
6
|
+
* id). `cellHist` and `cellStart` hold `histWords = cells + 2^dim + 1` words: every real cell, the 2^dim outside
|
|
7
|
+
* pseudo-cells (one per orthant about the grid centre, issue #90) from index `cells`, and one more so
|
|
8
|
+
* `cellStart[histWords - 1] === n` closes the last range. Every zeroing is a `fill` dispatch inside the
|
|
8
9
|
* pass (PD-12), never an encoder clear.
|
|
9
10
|
*
|
|
10
11
|
* The named grid buffers (`cellKey`, `cellVal`, `sortedKey`, `sortedIdx`, `cellHist`, `cellStart`) are the caller's
|
|
@@ -48,8 +49,8 @@ function floorPow2(x) {
|
|
|
48
49
|
* The grid of `n` nodes in `dim` dimensions under the tuning (spec 7.7 geometry table; PD-9): `G = clamp(nextPow2(2 *
|
|
49
50
|
* ceil(n^(1 / dim))), GRID_MIN_SIDE, floorPow2(gridMax))` where `gridMax` is `gridMax2D` or `gridMax3D`, rounded DOWN
|
|
50
51
|
* to a power of two so every level's side is an integer (512 and 128 stay; 100 becomes 64); `levels = log2(G /
|
|
51
|
-
* GRID_COARSEST_SIDE) + 1`. At the caps: 349,
|
|
52
|
-
* pseudo-
|
|
52
|
+
* GRID_COARSEST_SIDE) + 1`. At the caps: 349,524 pyramid cells in 2D, 2,396,744 in 3D (the design's counts plus the
|
|
53
|
+
* 2^dim pseudo-cells).
|
|
53
54
|
* @param n - the node count (>= 0)
|
|
54
55
|
* @param dim - 2 or 3
|
|
55
56
|
* @param tuning - the resolved layout tuning (`gridMax2D`, `gridMax3D`, `deterministic`)
|
|
@@ -65,10 +66,11 @@ export function gridSpecFor(n, dim, tuning) {
|
|
|
65
66
|
levels++;
|
|
66
67
|
}
|
|
67
68
|
const cells = g ** dim;
|
|
69
|
+
const outsideCells = 2 ** dim;
|
|
68
70
|
const levelOffsets = [0];
|
|
69
71
|
let s = g;
|
|
70
72
|
for (let level = 0; level + 1 < levels; level++) {
|
|
71
|
-
levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ?
|
|
73
|
+
levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? outsideCells : 0));
|
|
72
74
|
s /= 2;
|
|
73
75
|
}
|
|
74
76
|
return {
|
|
@@ -76,14 +78,15 @@ export function gridSpecFor(n, dim, tuning) {
|
|
|
76
78
|
g,
|
|
77
79
|
levels,
|
|
78
80
|
cells,
|
|
79
|
-
|
|
81
|
+
outsideCells,
|
|
82
|
+
histWords: cells + outsideCells + 1,
|
|
80
83
|
levelOffsets: Object.freeze(levelOffsets),
|
|
81
84
|
pyramidCells: levelOffsets[levels - 1] + GRID_COARSEST_SIDE ** dim,
|
|
82
85
|
deterministic: tuning.deterministic,
|
|
83
86
|
};
|
|
84
87
|
}
|
|
85
88
|
/**
|
|
86
|
-
* The bytes of the pyramid (spec 7.7: 16 B per cell, every level, the pseudo-
|
|
89
|
+
* The bytes of the pyramid (spec 7.7: 16 B per cell, every level, the pseudo-cells included): 38,347,904 at the 3D cap.
|
|
87
90
|
* @param spec - the grid
|
|
88
91
|
* @returns the byte length
|
|
89
92
|
*/
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"grid.js","sourceRoot":"","sources":["../../../src/primitives/grid.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"grid.js","sourceRoot":"","sources":["../../../src/primitives/grid.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AAEH,OAAO,EAAE,kBAAkB,EAAE,aAAa,EAAE,cAAc,EAAE,MAAM,iBAAiB,CAAC;AACpF,OAAO,EAAE,gBAAgB,EAAE,MAAM,cAAc,CAAC;AAChD,OAAO,EAAE,MAAM,EAAE,MAAM,uBAAuB,CAAC;AAE/C,OAAO,EAAE,UAAU,EAAE,MAAM,eAAe,CAAC;AAG3C,OAAO,EAGH,mBAAmB,EACnB,gBAAgB,GACnB,MAAM,gBAAgB,CAAC;AACxB,OAAO,EAAE,gBAAgB,EAAE,cAAc,EAAyB,MAAM,iBAAiB,CAAC;AAE1F,OAAO,EAAE,WAAW,EAAoB,MAAM,WAAW,CAAC;AAwB1D;;;;GAIG;AACH,SAAS,QAAQ,CAAC,CAAS;IACvB,IAAI,CAAC,GAAG,CAAC,CAAC;IACV,OAAO,CAAC,GAAG,CAAC,EAAE,CAAC;QACX,CAAC,IAAI,CAAC,CAAC;IACX,CAAC;IACD,OAAO,CAAC,CAAC;AACb,CAAC;AAED;;;;GAIG;AACH,SAAS,SAAS,CAAC,CAAS;IACxB,IAAI,CAAC,GAAG,CAAC,CAAC;IACV,OAAO,CAAC,GAAG,CAAC,IAAI,CAAC,EAAE,CAAC;QAChB,CAAC,IAAI,CAAC,CAAC;IACX,CAAC;IACD,OAAO,CAAC,CAAC;AACb,CAAC;AAED;;;;;;;;;;GAUG;AACH,MAAM,UAAU,WAAW,CACvB,CAAS,EACT,GAAU,EACV,MAA+E;IAE/E,MAAM,OAAO,GAAG,GAAG,KAAK,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,SAAS,CAAC,CAAC,CAAC,MAAM,CAAC,SAAS,CAAC;IAChE,MAAM,IAAI,GAAG,GAAG,KAAK,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;IACrD,MAAM,GAAG,GAAG,IAAI,CAAC,GAAG,CAAC,aAAa,EAAE,SAAS,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,OAAO,CAAC,CAAC,CAAC,CAAC;IACrE,MAAM,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,GAAG,EAAE,IAAI,CAAC,GAAG,CAAC,aAAa,EAAE,QAAQ,CAAC,CAAC,GAAG,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,CAAC;IAChF,IAAI,MAAM,GAAG,CAAC,CAAC;IACf,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,kBAAkB,EAAE,CAAC,IAAI,CAAC,EAAE,CAAC;QAC7C,MAAM,EAAE,CAAC;IACb,CAAC;IACD,MAAM,KAAK,GAAG,CAAC,IAAI,GAAG,CAAC;IACvB,MAAM,YAAY,GAAG,CAAC,IAAI,GAAG,CAAC;IAC9B,MAAM,YAAY,GAAa,CAAC,CAAC,CAAC,CAAC;IACnC,IAAI,CAAC,GAAG,CAAC,CAAC;IACV,KAAK,IAAI,KAAK,GAAG,CAAC,EAAE,KAAK,GAAG,CAAC,GAAG,MAAM,EAAE,KAAK,EAAE,EAAE,CAAC;QAC9C,YAAY,CAAC,IAAI,CAAC,YAAY,CAAC,KAAK,CAAC,GAAG,CAAC,IAAI,GAAG,GAAG,CAAC,KAAK,KAAK,CAAC,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;QACrF,CAAC,IAAI,CAAC,CAAC;IACX,CAAC;IACD,OAAO;QACH,GAAG;QACH,CAAC;QACD,MAAM;QACN,KAAK;QACL,YAAY;QACZ,SAAS,EAAE,KAAK,GAAG,YAAY,GAAG,CAAC;QACnC,YAAY,EAAE,MAAM,CAAC,MAAM,CAAC,YAAY,CAAC;QACzC,YAAY,EAAE,YAAY,CAAC,MAAM,GAAG,CAAC,CAAC,GAAG,kBAAkB,IAAI,GAAG;QAClE,aAAa,EAAE,MAAM,CAAC,aAAa;KACtC,CAAC;AACN,CAAC;AAED;;;;GAIG;AACH,MAAM,UAAU,gBAAgB,CAAC,IAAc;IAC3C,OAAO,EAAE,GAAG,IAAI,CAAC,YAAY,CAAC;AAClC,CAAC;AAqED;;;;;;GAMG;AACH,MAAM,CAAC,KAAK,UAAU,gBAAgB,CAAC,KAAkB,EAAE,IAAc;IACrE,MAAM,OAAO,GAAG,MAAM,KAAK,CAAC,SAAS,CAAC,MAAM,CAAC,UAAU,CAAC,eAAe,CAAC,CAAC,CAAC;IAC1E,MAAM,IAAI,GAAa,IAAI,CAAC,aAAa;QACrC,CAAC,CAAC;YACI,IAAI,EAAE,OAAO;YACb,KAAK,EAAE,MAAM,gBAAgB,CAAC,KAAK,CAAC;YACpC,SAAS,EAAE,MAAM,gBAAgB,CAAC,KAAK,CAAC;YACxC,IAAI,EAAE,MAAM,WAAW,CAAC,KAAK,CAAC;SACjC;QACH,CAAC,CAAC,EAAE,IAAI,EAAE,UAAU,EAAE,QAAQ,EAAE,MAAM,mBAAmB,CAAC,KAAK,CAAC,EAAE,CAAC;IACvE,OAAO,IAAI,oBAAoB,CAAC,KAAK,EAAE,IAAI,EAAE,OAAO,EAAE,IAAI,CAAC,CAAC;AAChE,CAAC;AAUD,+EAA+E;AAC/E,MAAM,oBAAoB;IAQtB;;;;;;OAMG;IACH,YAAY,KAAkB,EAAE,IAAc,EAAE,OAAe,EAAE,IAAc;QAVvE,UAAK,GAAiB,IAAI,CAAC;QAC3B,eAAU,GAAG,CAAC,CAAC;QAUnB,IAAI,CAAC,KAAK,GAAG,KAAK,CAAC;QACnB,IAAI,CAAC,IAAI,GAAG,IAAI,CAAC;QACjB,IAAI,CAAC,OAAO,GAAG,OAAO,CAAC;QACvB,IAAI,CAAC,IAAI,GAAG,IAAI,CAAC;IACrB,CAAC;IAED;;;OAGG;IACH,IAAI,cAAc;QACd,OAAO,IAAI,CAAC,UAAU,CAAC;IAC3B,CAAC;IAED;;;OAGG;IACH,IAAI,CAAC,QAA2B;QAC5B,MAAM,QAAQ,GAAG,IAAI,CAAC,KAAK,CAAC,QAAQ,CAAC,OAAO,CAAC,IAAI,GAAG,CAAC,CAAC,CAAC;QACvD,IAAI,QAAQ,GAAG,CAAC,EAAE,CAAC;YACf,MAAM,IAAI,gBAAgB,CAAC,oBAAoB,EAAE,gDAAgD,EAAE;gBAC/F,QAAQ,EAAE,SAAS;gBACnB,KAAK,EAAE,QAAQ,CAAC,OAAO,CAAC,IAAI;gBAC5B,QAAQ,EAAE,CAAC;aACd,CAAC,CAAC;QACP,CAAC;QACD,MAAM,KAAK,GACP,IAAI,CAAC,IAAI,CAAC,IAAI,KAAK,OAAO,CAAC,CAAC,CAAC,cAAc,CAAC,QAAQ,EAAE,IAAI,CAAC,KAAK,CAAC,aAAa,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,SAAS,CAAC;QAC9G,IAAI,CAAC,KAAK,GAAG;YACT,QAAQ;YACR,QAAQ;YACR,QAAQ,EAAE,IAAI,CAAC,OAAO,CAAC,KAAK,EAAE,qBAAqB,CAAC;YACpD,QAAQ,EAAE,IAAI,CAAC,OAAO,CAAC,KAAK,EAAE,qBAAqB,CAAC;SACvD,CAAC;IACN,CAAC;IAED;;;;;;OAMG;IACH,MAAM,CAAC,IAA2B,EAAE,CAAS,EAAE,YAAoB,EAAE,IAAqB;QACtF,MAAM,EAAE,KAAK,EAAE,GAAG,IAAI,CAAC;QACvB,IAAI,KAAK,KAAK,IAAI,EAAE,CAAC;YACjB,MAAM,IAAI,gBAAgB,CAAC,cAAc,EAAE,mCAAmC,EAAE,EAAE,QAAQ,EAAE,MAAM,EAAE,CAAC,CAAC;QAC1G,CAAC;QACD,IAAI,CAAC,MAAM,CAAC,aAAa,CAAC,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,IAAI,CAAC,GAAG,KAAK,CAAC,QAAQ,EAAE,CAAC;YAC1D,MAAM,IAAI,gBAAgB,CAAC,oBAAoB,EAAE,sDAAsD,EAAE;gBACrG,QAAQ,EAAE,GAAG;gBACb,KAAK,EAAE,CAAC;gBACR,QAAQ,EAAE,KAAK,CAAC,QAAQ;aAC3B,CAAC,CAAC;QACP,CAAC;QACD,MAAM,IAAI,GAAG,IAAI,IAAI,IAAI,CAAC;QAC1B,MAAM,CAAC,GAAG,KAAK,CAAC,QAAQ,CAAC;QACzB,MAAM,QAAQ,GAAG,IAAI,CAAC,OAAO,CAAC,IAAI,CAAC;YAC/B,GAAG,EAAE,CAAC,CAAC,GAAG;YACV,CAAC,EAAE,CAAC,CAAC,KAAK;YACV,OAAO,EAAE,CAAC,CAAC,OAAO;YAClB,OAAO,EAAE,CAAC,CAAC,OAAO;YAClB,CAAC,EAAE,CAAC,CAAC,MAAM;SACd,CAAC,CAAC;QACH,IAAI,CAAC,OAAO,CAAC,QAAQ,CAAC,IAAI,EAAE,QAAQ,EAAE,MAAM,CAAC,CAAC,EAAE,IAAI,CAAC,KAAK,CAAC,aAAa,EAAE,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,YAAY,CAAC,CAAC,CAAC;QAC5G,IAAI,CAAC,UAAU,GAAG,CAAC,CAAC;QACpB,IAAI,IAAI,KAAK,IAAI,EAAE,CAAC;YAChB,OAAO;QACX,CAAC;QACD,MAAM,EAAE,SAAS,EAAE,IAAI,EAAE,GAAG,IAAI,CAAC,IAAI,CAAC;QACtC,IAAI,IAAI,CAAC,IAAI,CAAC,IAAI,KAAK,UAAU,EAAE,CAAC;YAChC,MAAM,EAAE,QAAQ,EAAE,GAAG,IAAI,CAAC,IAAI,CAAC;YAC/B,MAAM,OAAO,GAAG,EAAE,IAAI,EAAE,CAAC,CAAC,QAAQ,EAAE,MAAM,EAAE,KAAK,CAAC,QAAQ,EAAE,CAAC;YAC7D,QAAQ,CAAC,MAAM,CAAC,IAAI,EAAE,CAAC,CAAC,OAAO,EAAE,CAAC,EAAE,IAAI,EAAE,OAAO,EAAE,CAAC,CAAC,SAAS,EAAE,CAAC,CAAC,SAAS,CAAC,CAAC;YAC7E,IAAI,CAAC,UAAU,IAAI,QAAQ,CAAC,cAAc,CAAC;YAC3C,OAAO;QACX,CAAC;QACD,MAAM,EAAE,KAAK,EAAE,SAAS,EAAE,IAAI,EAAE,GAAG,IAAI,CAAC,IAAI,CAAC;QAC7C,KAAK,CAAC,MAAM,CAAC,IAAI,EAAE,CAAC,CAAC,OAAO,EAAE,CAAC,CAAC,OAAO,EAAE,CAAC,EAAE,cAAc,EAAE;YACxD,IAAI,EAAE,CAAC,CAAC,SAAS;YACjB,IAAI,EAAE,CAAC,CAAC,SAAS;YACjB,IAAI,EAAE,KAAK,CAAC,QAAQ;YACpB,OAAO,EAAE,KAAK,CAAC,QAAQ;SAC1B,CAAC,CAAC;QACH,IAAI,CAAC,UAAU,IAAI,KAAK,CAAC,cAAc,CAAC;QACxC,IAAI,IAAI,KAAK,IAAI,EAAE,CAAC;YAChB,OAAO;QACX,CAAC;QACD,SAAS,CAAC,MAAM,CAAC,IAAI,EAAE,CAAC,CAAC,OAAO,EAAE,CAAC,EAAE,IAAI,EAAE,CAAC,CAAC,QAAQ,CAAC,CAAC;QACvD,IAAI,CAAC,MAAM,CAAC,IAAI,EAAE,CAAC,CAAC,QAAQ,EAAE,IAAI,EAAE,CAAC,CAAC,SAAS,CAAC,CAAC;QACjD,IAAI,CAAC,UAAU,IAAI,SAAS,CAAC,cAAc,GAAG,IAAI,CAAC,cAAc,CAAC;IACtE,CAAC;IAED;;;;;OAKG;IACK,OAAO,CAAC,KAAa,EAAE,KAAa;QACxC,MAAM,IAAI,GAAG,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,KAAK,CAAC,CAAC;QACpC,MAAM,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,OAAO,CAAC,IAAI,EAAE,KAAK,CAAC,CAAC;QAC/C,OAAO,EAAE,MAAM,EAAE,MAAM,EAAE,CAAC,EAAE,IAAI,EAAE,MAAM,EAAE,IAAI,EAAE,CAAC;IACrD,CAAC;CACJ"}
|
|
@@ -7,13 +7,14 @@
|
|
|
7
7
|
* the twin's two compilations -- and then every invocation strips the range `[0, aggregate)` with a binary search
|
|
8
8
|
* (`upper_bound`) over the scanned degrees to find which entry its arc belongs to. One `atomicAdd` per WORKGROUP
|
|
9
9
|
* reserves the block's span in the queue (`edgeCount`), the same aggregate lands in `edgeCountUnclamped` (the overflow
|
|
10
|
-
* detector, never clamped) and in `frontierDegreeSum` (
|
|
11
|
-
* `
|
|
10
|
+
* detector, never clamped) and in `frontierDegreeSum` (the inspect seam's per-level expansion count, rotated into
|
|
11
|
+
* `prevDegreeSum` by the boundary; Beamer's m_f is `nextDegreeSum`, measured by `bfs-next-degree` -- issue #391);
|
|
12
|
+
* a lane whose queue position is at or past `P.edgeCapacity` writes nothing (the clamp). The queue holds the TARGET vertex of each arc only (PD-24: `parent`
|
|
12
13
|
* comes from the post-pass). Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan call and the
|
|
13
14
|
* `workgroupUniformLoad` sit unconditionally after the guard, and the strip loop is bounded by a uniform value.
|
|
14
15
|
* There is no `TIER` override: a hub row is balanced over all `WG` lanes inside its block, and the small-frontier
|
|
15
16
|
* workgroup-per-row structure is `bfs-fused` (P8-T7). Body only (spec 3.5, D9); the text is normative: the
|
|
16
17
|
* sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
17
18
|
*/
|
|
18
|
-
export declare const advanceExpandWgsl = "\nvar<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan\nvar<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row\nvar<workgroup> base: u32; // the block's reserved span in the edge queue\nvar<workgroup> wcount: u32; // the frontier's length on a two-phase level, 0 on any other\n\n@compute @workgroup_size(WG)\nfn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x == 0u) { wcount = select(0u, atomicLoad(&counters[0]), atomicLoad(&counters[24]) == 1u); } // frontierCount, on the two-phase path only (the path word)\n let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers\n for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries (a direct dispatch of the plan's groups)\n let i = b0 + lid.x; // this lane's frontier entry\n var deg = 0u;\n var start = 0u;\n if (i < count) { // guarded loads into locals (3.5 rule 1)\n let v = frontierIn[i];\n let lo = max(rowPtr[v], P.arcBase); // the row clipped to the bound window (P8-T12)\n let hi = min(rowPtr[v + 1u], P.arcEnd);\n start = lo;\n deg = select(0u, hi - lo, hi > lo);\n }\n let inclusive = wg_scan_u32(deg, lid.x); // the prelude's inclusive scan in LANE order (P8-T1 Step 5); the twin\n sh[lid.x] = inclusive; // sh is monotone in lid.x, the index the binary search walks\n rowStart[lid.x] = start;\n workgroupBarrier();\n let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier\n if (lid.x == 0u) {\n base = atomicAdd(&counters[8], aggregate); // edgeCount: ONE reservation per workgroup, not one per arc\n atomicAdd(&counters[9], aggregate); // edgeCountUnclamped: the overflow detector (PD-23)\n atomicAdd(&counters[2], aggregate); // frontierDegreeSum:
|
|
19
|
+
export declare const advanceExpandWgsl = "\nvar<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan\nvar<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row\nvar<workgroup> base: u32; // the block's reserved span in the edge queue\nvar<workgroup> wcount: u32; // the frontier's length on a two-phase level, 0 on any other\n\n@compute @workgroup_size(WG)\nfn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x == 0u) { wcount = select(0u, atomicLoad(&counters[0]), atomicLoad(&counters[24]) == 1u); } // frontierCount, on the two-phase path only (the path word)\n let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers\n for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries (a direct dispatch of the plan's groups)\n let i = b0 + lid.x; // this lane's frontier entry\n var deg = 0u;\n var start = 0u;\n if (i < count) { // guarded loads into locals (3.5 rule 1)\n let v = frontierIn[i];\n let lo = max(rowPtr[v], P.arcBase); // the row clipped to the bound window (P8-T12)\n let hi = min(rowPtr[v + 1u], P.arcEnd);\n start = lo;\n deg = select(0u, hi - lo, hi > lo);\n }\n let inclusive = wg_scan_u32(deg, lid.x); // the prelude's inclusive scan in LANE order (P8-T1 Step 5); the twin\n sh[lid.x] = inclusive; // sh is monotone in lid.x, the index the binary search walks\n rowStart[lid.x] = start;\n workgroupBarrier();\n let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier\n if (lid.x == 0u) {\n base = atomicAdd(&counters[8], aggregate); // edgeCount: ONE reservation per workgroup, not one per arc\n atomicAdd(&counters[9], aggregate); // edgeCountUnclamped: the overflow detector (PD-23)\n atomicAdd(&counters[2], aggregate); // frontierDegreeSum: what this level expanded (the inspect seam)\n }\n workgroupBarrier();\n for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...\n var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p\n var hi = WG;\n loop {\n if (lo >= hi) { break; }\n let mid = (lo + hi) / 2u;\n if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }\n }\n let k = lo;\n var exclusive = 0u;\n if (k > 0u) { exclusive = sh[k - 1u]; }\n let arc = rowStart[k] + (p - exclusive);\n let q = base + p;\n if (q < P.edgeCapacity) { edgeQueue[q] = colIdx[arc - P.arcBase]; } // the clamp of PD-23; the queue holds the target vertex only (PD-24)\n }\n workgroupBarrier(); // sh, rowStart and base are reused by the next block\n }\n}\n";
|
|
19
20
|
//# sourceMappingURL=advance-expand.wgsl.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"advance-expand.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/advance-expand.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"advance-expand.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/advance-expand.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,eAAO,MAAM,iBAAiB,61GAkD7B,CAAC"}
|
|
@@ -7,8 +7,9 @@
|
|
|
7
7
|
* the twin's two compilations -- and then every invocation strips the range `[0, aggregate)` with a binary search
|
|
8
8
|
* (`upper_bound`) over the scanned degrees to find which entry its arc belongs to. One `atomicAdd` per WORKGROUP
|
|
9
9
|
* reserves the block's span in the queue (`edgeCount`), the same aggregate lands in `edgeCountUnclamped` (the overflow
|
|
10
|
-
* detector, never clamped) and in `frontierDegreeSum` (
|
|
11
|
-
* `
|
|
10
|
+
* detector, never clamped) and in `frontierDegreeSum` (the inspect seam's per-level expansion count, rotated into
|
|
11
|
+
* `prevDegreeSum` by the boundary; Beamer's m_f is `nextDegreeSum`, measured by `bfs-next-degree` -- issue #391);
|
|
12
|
+
* a lane whose queue position is at or past `P.edgeCapacity` writes nothing (the clamp). The queue holds the TARGET vertex of each arc only (PD-24: `parent`
|
|
12
13
|
* comes from the post-pass). Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan call and the
|
|
13
14
|
* `workgroupUniformLoad` sit unconditionally after the guard, and the strip loop is bounded by a uniform value.
|
|
14
15
|
* There is no `TIER` override: a hub row is balanced over all `WG` lanes inside its block, and the small-frontier
|
|
@@ -44,7 +45,7 @@ fn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
|
|
|
44
45
|
if (lid.x == 0u) {
|
|
45
46
|
base = atomicAdd(&counters[8], aggregate); // edgeCount: ONE reservation per workgroup, not one per arc
|
|
46
47
|
atomicAdd(&counters[9], aggregate); // edgeCountUnclamped: the overflow detector (PD-23)
|
|
47
|
-
atomicAdd(&counters[2], aggregate); // frontierDegreeSum:
|
|
48
|
+
atomicAdd(&counters[2], aggregate); // frontierDegreeSum: what this level expanded (the inspect seam)
|
|
48
49
|
}
|
|
49
50
|
workgroupBarrier();
|
|
50
51
|
for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"advance-expand.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/advance-expand.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"advance-expand.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/advance-expand.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,CAAC,MAAM,iBAAiB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAkD3C,CAAC"}
|
|
@@ -11,9 +11,10 @@
|
|
|
11
11
|
* The winners are packed into the output vertex queue by `bfs-contract`'s workgroup scan and one `atomicAdd` per
|
|
12
12
|
* workgroup on `nextFrontierCount`, and claim with a plain `atomicStore`: the list holds every vertex once and the
|
|
13
13
|
* sweep is vertex-parallel, so no two lanes claim one vertex. It adds nothing to `frontierDegreeSum` (a bottom-up
|
|
14
|
-
* level expands nothing), which is why
|
|
15
|
-
*
|
|
16
|
-
*
|
|
14
|
+
* level expands nothing), which is why that word is not Beamer's m_f: `bfs-next-degree` sums the degree of what
|
|
15
|
+
* this sweep CLAIMS into `nextDegreeSum`, so the boundary's test and its `unvisitedDegreeSum` subtraction are exact
|
|
16
|
+
* on a bottom-up level like any other (issue #391; the selector's JSDoc). Uniformity (spec 3.5 rule 1): the
|
|
17
|
+
* guarded walk writes locals, the scan and the reduction run unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
|
|
17
18
|
* test/helpers/sabotage.ts are textual edits of it.
|
|
18
19
|
*/
|
|
19
20
|
export declare const bfsBottomUpWgsl = "\nvar<workgroup> sh: array<u32, WG>;\nvar<workgroup> base: u32;\nvar<workgroup> wcount: u32; // the unvisited list's length on a bottom-up level, 0 on any other\n\n@compute @workgroup_size(WG)\nfn bfs_bottom_up(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let claim = atomicLoad(&counters[11]) + 1u;\n if (lid.x == 0u) { wcount = select(0u, atomicLoad(&counters[7]), atomicLoad(&counters[24]) == 3u); } // unvisitedListLen, on the bottom-up path only (the path word)\n let len = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers\n for (var b0 = group_id(wid) * WG; b0 < len; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries\n let i = b0 + lid.x;\n var won = 0u;\n var v = 0u;\n var reads = 0u;\n if (i < len) { // guarded work into locals\n v = sweepIn[i];\n if (atomicLoad(&depth[v]) == INVALID_INDEX) { // a stale entry, claimed since the rebuild, is skipped\n let end = min(rowPtr[v + 1u], P.arcEnd);\n for (var a = max(rowPtr[v], P.arcBase); a < end; a = a + 1u) { // in-neighbours through the reverse core\n reads = reads + 1u;\n let u = colIdx[a - P.arcBase];\n if (mask_bit(sweepIn[P.bitsBase + (u >> 5u)], u)) { won = 1u; break; } // the early exit: a real break, never a flag\n }\n }\n }\n sh[lid.x] = won;\n workgroupBarrier();\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of won (bfs-contract's)\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let inclusive = sh[lid.x];\n let readsTotal = wg_reduce_u32(reads, lid.x, 0u); // arcsScanned, one atomic per workgroup (the sabotage witness, Step 6)\n if (lid.x == WG - 1u) { base = atomicAdd(&counters[1], inclusive); }\n if (lid.x == 0u) { atomicAdd(&counters[16], readsTotal); }\n workgroupBarrier();\n if (won == 1u) {\n atomicStore(&depth[v], claim); // no claim race: the list holds v once and the sweep is vertex-parallel\n frontierOut[base + inclusive - 1u] = v;\n }\n workgroupBarrier(); // sh and base are reused by the next block\n }\n}\n";
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"bfs-bottom-up.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bfs-bottom-up.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"bfs-bottom-up.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bfs-bottom-up.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;GAkBG;AACH,eAAO,MAAM,eAAe,8pFA+C3B,CAAC"}
|
|
@@ -11,9 +11,10 @@
|
|
|
11
11
|
* The winners are packed into the output vertex queue by `bfs-contract`'s workgroup scan and one `atomicAdd` per
|
|
12
12
|
* workgroup on `nextFrontierCount`, and claim with a plain `atomicStore`: the list holds every vertex once and the
|
|
13
13
|
* sweep is vertex-parallel, so no two lanes claim one vertex. It adds nothing to `frontierDegreeSum` (a bottom-up
|
|
14
|
-
* level expands nothing), which is why
|
|
15
|
-
*
|
|
16
|
-
*
|
|
14
|
+
* level expands nothing), which is why that word is not Beamer's m_f: `bfs-next-degree` sums the degree of what
|
|
15
|
+
* this sweep CLAIMS into `nextDegreeSum`, so the boundary's test and its `unvisitedDegreeSum` subtraction are exact
|
|
16
|
+
* on a bottom-up level like any other (issue #391; the selector's JSDoc). Uniformity (spec 3.5 rule 1): the
|
|
17
|
+
* guarded walk writes locals, the scan and the reduction run unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
|
|
17
18
|
* test/helpers/sabotage.ts are textual edits of it.
|
|
18
19
|
*/
|
|
19
20
|
export const bfsBottomUpWgsl = /* wgsl */ `
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"bfs-bottom-up.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bfs-bottom-up.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"bfs-bottom-up.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bfs-bottom-up.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;GAkBG;AACH,MAAM,CAAC,MAAM,eAAe,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA+CzC,CAAC"}
|
|
@@ -4,15 +4,15 @@
|
|
|
4
4
|
* ONE dispatch, chosen by `frontier-finalize` for a frontier below `P.fusedMax` entries (`path` 2) and for the retry
|
|
5
5
|
* of a level whose edge queue overflowed (`path` 4, PD-23). One WORKGROUP per frontier entry, the workgroups striding
|
|
6
6
|
* the entries by the dispatch's group count (`P.stride`): lane 0 reads the entry's row clipped
|
|
7
|
-
* to the bound arc window, adds its degree to `frontierDegreeSum` (
|
|
8
|
-
* too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim
|
|
9
|
-
* `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
|
|
7
|
+
* to the bound arc window, adds its degree to `frontierDegreeSum` (so the inspect seam's per-level expansion count
|
|
8
|
+
* covers fused levels too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim
|
|
9
|
+
* inline -- `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
|
|
10
10
|
* packing the strip's winners into the output vertex queue by the same Hillis-Steele scan and one `atomicAdd` per
|
|
11
11
|
* strip on `nextFrontierCount`. No edge queue is written or read, which is the whole win for a tiny frontier
|
|
12
12
|
* (Merrill's fleeting iterations) and what makes the overflow retry exact: the partial edge queue is never consulted.
|
|
13
13
|
* On a retry level `advance-expand` has already added the frontier's degree to `frontierDegreeSum`, so that word
|
|
14
|
-
* holds
|
|
15
|
-
*
|
|
14
|
+
* holds twice the level's expanded degree; since issue #391 no decision reads it (Beamer's m_f is `nextDegreeSum`
|
|
15
|
+
* and the boundary subtracts that), so the doubling only reaches the inspect seam's `prevDegreeSum`. Nothing here
|
|
16
16
|
* writes a parent (PD-24: the post-pass does), which is what keeps the kernel at the eight-storage-buffer budget
|
|
17
17
|
* with the four graph slots. Uniformity (spec 3.5 rule 1): the strip loop's bound and the row start are
|
|
18
18
|
* `workgroupUniformLoad`s, so every barrier of the per-strip append is in uniform control flow; the guarded claim
|
|
@@ -21,5 +21,5 @@
|
|
|
21
21
|
* (the sabotage rows need distinct find strings). Body only (spec 3.5, D9); the text is normative: the sabotage rows
|
|
22
22
|
* of test/helpers/sabotage.ts are textual edits of it.
|
|
23
23
|
*/
|
|
24
|
-
export declare const bfsFusedWgsl = "\nvar<workgroup> sh: array<u32, WG>;\nvar<workgroup> wdeg: u32;\nvar<workgroup> wstart: u32;\nvar<workgroup> base: u32;\nvar<workgroup> wcount: u32; // the frontier's length on a fused or retry level, 0 on any other\n\n@compute @workgroup_size(WG)\nfn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let claim = atomicLoad(&counters[11]) + 1u;\n if (lid.x == 0u) {\n let path = atomicLoad(&counters[24]); // 2 the fused level, 4 the overflow retry (PD-23)\n wcount = select(0u, atomicLoad(&counters[0]), path == 2u || path == 4u);\n }\n let count = workgroupUniformLoad(&wcount); // uniform: the entry loop below holds barriers\n for (var g = group_id(wid); g < count; g = g + P.stride) { // one workgroup per frontier entry; P.stride is the dispatch's GROUP count\n if (lid.x == 0u) {\n let u = frontierIn[g];\n let a0 = max(rowPtr[u], P.arcBase);\n let a1 = min(rowPtr[u + 1u], P.arcEnd);\n let d = select(0u, a1 - a0, a1 > a0);\n wdeg = d;\n wstart = a0;\n atomicAdd(&counters[2], d); // frontierDegreeSum, so
|
|
24
|
+
export declare const bfsFusedWgsl = "\nvar<workgroup> sh: array<u32, WG>;\nvar<workgroup> wdeg: u32;\nvar<workgroup> wstart: u32;\nvar<workgroup> base: u32;\nvar<workgroup> wcount: u32; // the frontier's length on a fused or retry level, 0 on any other\n\n@compute @workgroup_size(WG)\nfn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let claim = atomicLoad(&counters[11]) + 1u;\n if (lid.x == 0u) {\n let path = atomicLoad(&counters[24]); // 2 the fused level, 4 the overflow retry (PD-23)\n wcount = select(0u, atomicLoad(&counters[0]), path == 2u || path == 4u);\n }\n let count = workgroupUniformLoad(&wcount); // uniform: the entry loop below holds barriers\n for (var g = group_id(wid); g < count; g = g + P.stride) { // one workgroup per frontier entry; P.stride is the dispatch's GROUP count\n if (lid.x == 0u) {\n let u = frontierIn[g];\n let a0 = max(rowPtr[u], P.arcBase);\n let a1 = min(rowPtr[u + 1u], P.arcEnd);\n let d = select(0u, a1 - a0, a1 > a0);\n wdeg = d;\n wstart = a0;\n atomicAdd(&counters[2], d); // frontierDegreeSum, so the inspect seam sees fused levels too\n }\n let deg = workgroupUniformLoad(&wdeg); // uniform: the loop below may hold barriers\n let start = workgroupUniformLoad(&wstart);\n for (var p0 = 0u; p0 < deg; p0 = p0 + WG) { // strip the row WG arcs at a time\n let p = p0 + lid.x;\n var won = 0u;\n var v = 0u;\n if (p < deg) { // guarded claim into locals\n v = colIdx[start + p - P.arcBase];\n let old = atomicMin(&depth[v], claim);\n won = select(0u, 1u, old == INVALID_INDEX);\n }\n sh[lid.x] = won;\n workgroupBarrier();\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of won (bfs-contract's, verbatim)\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let inclusive = sh[lid.x];\n if (lid.x == WG - 1u) { base = atomicAdd(&counters[1], inclusive); }\n workgroupBarrier();\n if (won == 1u) { frontierOut[base + inclusive - 1u] = v; }\n workgroupBarrier(); // sh and base are reused by the next strip\n }\n }\n}\n";
|
|
25
25
|
//# sourceMappingURL=bfs-fused.wgsl.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"bfs-fused.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bfs-fused.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;GAsBG;AACH,eAAO,MAAM,YAAY,
|
|
1
|
+
{"version":3,"file":"bfs-fused.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bfs-fused.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;GAsBG;AACH,eAAO,MAAM,YAAY,6uFAqDxB,CAAC"}
|
|
@@ -4,15 +4,15 @@
|
|
|
4
4
|
* ONE dispatch, chosen by `frontier-finalize` for a frontier below `P.fusedMax` entries (`path` 2) and for the retry
|
|
5
5
|
* of a level whose edge queue overflowed (`path` 4, PD-23). One WORKGROUP per frontier entry, the workgroups striding
|
|
6
6
|
* the entries by the dispatch's group count (`P.stride`): lane 0 reads the entry's row clipped
|
|
7
|
-
* to the bound arc window, adds its degree to `frontierDegreeSum` (
|
|
8
|
-
* too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim
|
|
9
|
-
* `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
|
|
7
|
+
* to the bound arc window, adds its degree to `frontierDegreeSum` (so the inspect seam's per-level expansion count
|
|
8
|
+
* covers fused levels too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim
|
|
9
|
+
* inline -- `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
|
|
10
10
|
* packing the strip's winners into the output vertex queue by the same Hillis-Steele scan and one `atomicAdd` per
|
|
11
11
|
* strip on `nextFrontierCount`. No edge queue is written or read, which is the whole win for a tiny frontier
|
|
12
12
|
* (Merrill's fleeting iterations) and what makes the overflow retry exact: the partial edge queue is never consulted.
|
|
13
13
|
* On a retry level `advance-expand` has already added the frontier's degree to `frontierDegreeSum`, so that word
|
|
14
|
-
* holds
|
|
15
|
-
*
|
|
14
|
+
* holds twice the level's expanded degree; since issue #391 no decision reads it (Beamer's m_f is `nextDegreeSum`
|
|
15
|
+
* and the boundary subtracts that), so the doubling only reaches the inspect seam's `prevDegreeSum`. Nothing here
|
|
16
16
|
* writes a parent (PD-24: the post-pass does), which is what keeps the kernel at the eight-storage-buffer budget
|
|
17
17
|
* with the four graph slots. Uniformity (spec 3.5 rule 1): the strip loop's bound and the row start are
|
|
18
18
|
* `workgroupUniformLoad`s, so every barrier of the per-strip append is in uniform control flow; the guarded claim
|
|
@@ -44,7 +44,7 @@ fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
|
|
|
44
44
|
let d = select(0u, a1 - a0, a1 > a0);
|
|
45
45
|
wdeg = d;
|
|
46
46
|
wstart = a0;
|
|
47
|
-
atomicAdd(&counters[2], d); // frontierDegreeSum, so
|
|
47
|
+
atomicAdd(&counters[2], d); // frontierDegreeSum, so the inspect seam sees fused levels too
|
|
48
48
|
}
|
|
49
49
|
let deg = workgroupUniformLoad(&wdeg); // uniform: the loop below may hold barriers
|
|
50
50
|
let start = workgroupUniformLoad(&wstart);
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bfs-next-degree` kernel body (design 8.4; issue #391, the amendment to P8-T8's PD-21): Beamer's m_f measured
|
|
3
|
+
* EXACTLY, at the end of every level, as the out-degree sum of the vertices the level just claimed -- the next
|
|
4
|
+
* frontier, which is what the next boundary decides the direction FOR. It grid-strides over the output vertex
|
|
5
|
+
* queue (`nextFrontierCount`, word 1, the claim kernels' append span; `P.stride` the plan's stride), reads each
|
|
6
|
+
* entry's out-degree from the `outDegree` view, reduces the lane sums with the prelude's `wg_reduce_u32` and lands
|
|
7
|
+
* ONE `atomicAdd` per workgroup in `nextDegreeSum` (word 25), which `frontier-finalize` role 0 reads for the
|
|
8
|
+
* switch-into-bottom-up test, subtracts from `unvisitedDegreeSum` and zeroes for the next level. It runs on every
|
|
9
|
+
* path that claims (the path word 24 non-zero: two-phase, fused, bottom-up, the retry) and does nothing on a level
|
|
10
|
+
* past the end.
|
|
11
|
+
*
|
|
12
|
+
* Why a kernel of its own: before it, the test used `frontierDegreeSum` (word 2), which the EXPANSION of the
|
|
13
|
+
* previous frontier accumulates, so the boundary compared the degree of the frontier it had just finished with the
|
|
14
|
+
* unvisited set, one level stale, and a bottom-up level (which expands nothing) left it at 0. On the 1M / 10M R-MAT
|
|
15
|
+
* that misses the switch at the level that matters: the frontier of 46,524 hubs at level 1 has 13.6M out-arcs, the
|
|
16
|
+
* unvisited set 7.3M, and the boundary saw the source's 86,405 instead -- top-down wrote 13.6M edge-queue entries
|
|
17
|
+
* where the bottom-up sweep reads 0.6M. Measuring the next frontier's degree at claim time is Beamer's own m_f, and
|
|
18
|
+
* the same word makes `unvisitedDegreeSum` exact after a bottom-up level too. Uniformity (spec 3.5 rule 1): the
|
|
19
|
+
* loop holds no barrier (its trip count is per lane), and the reduction runs unconditionally after it. Body only
|
|
20
|
+
* (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
21
|
+
*/
|
|
22
|
+
export declare const bfsNextDegreeWgsl = "\n@compute @workgroup_size(WG)\nfn bfs_next_degree(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let count = select(0u, atomicLoad(&counters[1]), atomicLoad(&counters[24]) != 0u); // nextFrontierCount, on a level that claimed (the path word)\n var sum = 0u;\n for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // no barrier inside: the trip count is per lane\n sum = sum + outDegree[frontier[i]];\n }\n let total = wg_reduce_u32(sum, lid.x, 0u); // the prelude's workgroup sum; uniform: after the loop\n if (lid.x == 0u) { atomicAdd(&counters[25], total); } // nextDegreeSum: ONE atomic per workgroup\n}\n";
|
|
23
|
+
//# sourceMappingURL=bfs-next-degree.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"bfs-next-degree.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bfs-next-degree.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,eAAO,MAAM,iBAAiB,4tBAW7B,CAAC"}
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bfs-next-degree` kernel body (design 8.4; issue #391, the amendment to P8-T8's PD-21): Beamer's m_f measured
|
|
3
|
+
* EXACTLY, at the end of every level, as the out-degree sum of the vertices the level just claimed -- the next
|
|
4
|
+
* frontier, which is what the next boundary decides the direction FOR. It grid-strides over the output vertex
|
|
5
|
+
* queue (`nextFrontierCount`, word 1, the claim kernels' append span; `P.stride` the plan's stride), reads each
|
|
6
|
+
* entry's out-degree from the `outDegree` view, reduces the lane sums with the prelude's `wg_reduce_u32` and lands
|
|
7
|
+
* ONE `atomicAdd` per workgroup in `nextDegreeSum` (word 25), which `frontier-finalize` role 0 reads for the
|
|
8
|
+
* switch-into-bottom-up test, subtracts from `unvisitedDegreeSum` and zeroes for the next level. It runs on every
|
|
9
|
+
* path that claims (the path word 24 non-zero: two-phase, fused, bottom-up, the retry) and does nothing on a level
|
|
10
|
+
* past the end.
|
|
11
|
+
*
|
|
12
|
+
* Why a kernel of its own: before it, the test used `frontierDegreeSum` (word 2), which the EXPANSION of the
|
|
13
|
+
* previous frontier accumulates, so the boundary compared the degree of the frontier it had just finished with the
|
|
14
|
+
* unvisited set, one level stale, and a bottom-up level (which expands nothing) left it at 0. On the 1M / 10M R-MAT
|
|
15
|
+
* that misses the switch at the level that matters: the frontier of 46,524 hubs at level 1 has 13.6M out-arcs, the
|
|
16
|
+
* unvisited set 7.3M, and the boundary saw the source's 86,405 instead -- top-down wrote 13.6M edge-queue entries
|
|
17
|
+
* where the bottom-up sweep reads 0.6M. Measuring the next frontier's degree at claim time is Beamer's own m_f, and
|
|
18
|
+
* the same word makes `unvisitedDegreeSum` exact after a bottom-up level too. Uniformity (spec 3.5 rule 1): the
|
|
19
|
+
* loop holds no barrier (its trip count is per lane), and the reduction runs unconditionally after it. Body only
|
|
20
|
+
* (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
21
|
+
*/
|
|
22
|
+
export const bfsNextDegreeWgsl = /* wgsl */ `
|
|
23
|
+
@compute @workgroup_size(WG)
|
|
24
|
+
fn bfs_next_degree(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
25
|
+
let count = select(0u, atomicLoad(&counters[1]), atomicLoad(&counters[24]) != 0u); // nextFrontierCount, on a level that claimed (the path word)
|
|
26
|
+
var sum = 0u;
|
|
27
|
+
for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // no barrier inside: the trip count is per lane
|
|
28
|
+
sum = sum + outDegree[frontier[i]];
|
|
29
|
+
}
|
|
30
|
+
let total = wg_reduce_u32(sum, lid.x, 0u); // the prelude's workgroup sum; uniform: after the loop
|
|
31
|
+
if (lid.x == 0u) { atomicAdd(&counters[25], total); } // nextDegreeSum: ONE atomic per workgroup
|
|
32
|
+
}
|
|
33
|
+
`;
|
|
34
|
+
//# sourceMappingURL=bfs-next-degree.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"bfs-next-degree.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bfs-next-degree.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,MAAM,CAAC,MAAM,iBAAiB,GAAG,UAAU,CAAC;;;;;;;;;;;CAW3C,CAAC"}
|
|
@@ -9,6 +9,9 @@
|
|
|
9
9
|
* `partials[group].swingTraction`. `pos` is `array<vec4f>` with the mass in `.w`; `force` / `oldForce` are stride-3
|
|
10
10
|
* `array<f32>` read through the per-body helpers (4.4 rule 6). Normative text, copied verbatim: the P1-T5 sabotage
|
|
11
11
|
* rows (gravity sign, `k / d2`, the `jj != i` guard, the `.w` mass lane) are textual edits of this string.
|
|
12
|
+
* The j range is split into passes of at most EXACT_TILES_PER_PASS tiles (issue #87: llvmpipe's per-invocation loop
|
|
13
|
+
* budget); an earlier pass adds its partial sum into `force`, the last one runs gravity and the epilogue. A graph of
|
|
14
|
+
* at most 32,768 nodes is one pass, bitwise the single-pass kernel. Record it through recordExactRepulsion.
|
|
12
15
|
*/
|
|
13
|
-
export declare const fa2RepulsionExactWgsl = "\nvar<workgroup> tile: array<vec4f, WG>; // xyz + mass, 4 KiB at WG = 256\n\nfn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }\nfn store_force(i: u32, f: vec3f) {\n force[3u * i] = f.x;\n force[3u * i + 1u] = f.y;\n force[3u * i + 2u] = f.z;\n}\nfn load_old(i: u32) -> vec3f { return vec3f(oldForce[3u * i], oldForce[3u * i + 1u], oldForce[3u * i + 2u]); }\nfn kick_magnitude(mi: f32, mj: f32) -> f32 { // the law's magnitude at d = FA2_DIST_FLOOR (PD-10)\n if (LAW == 1u) { return P.frK * P.frK / FA2_DIST_FLOOR; }\n if (LAW == 2u) { return -P.coulomb * mi * mj / FA2_DIST_FLOOR_SQ; }\n return P.scalingRatio * mi * mj / FA2_DIST_FLOOR;\n}\nfn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: centroid (GRAVITY_CENTER 0) or origin (1); regular or strong\n var q = pi.xyz;\n if (GRAVITY_CENTER == 0u) { q = pi.xyz - S.centroid.xyz; }\n if (STRONG_GRAVITY) { return -P.gravity * pi.w * q; }\n let d = length(q);\n if (d > FA2_DIST_FLOOR) { return -P.gravity * pi.w * q / d; }\n return vec3f(0.0);\n}\n\n@compute @workgroup_size(WG)\nfn repulsion(@builtin(workgroup_id) wid: vec3<u32
|
|
16
|
+
export declare const fa2RepulsionExactWgsl = "\nvar<workgroup> tile: array<vec4f, WG>; // xyz + mass, 4 KiB at WG = 256\n\nfn load_force(i: u32) -> vec3f { return vec3f(force[3u * i], force[3u * i + 1u], force[3u * i + 2u]); }\nfn store_force(i: u32, f: vec3f) {\n force[3u * i] = f.x;\n force[3u * i + 1u] = f.y;\n force[3u * i + 2u] = f.z;\n}\nfn load_old(i: u32) -> vec3f { return vec3f(oldForce[3u * i], oldForce[3u * i + 1u], oldForce[3u * i + 2u]); }\nfn kick_magnitude(mi: f32, mj: f32) -> f32 { // the law's magnitude at d = FA2_DIST_FLOOR (PD-10)\n if (LAW == 1u) { return P.frK * P.frK / FA2_DIST_FLOOR; }\n if (LAW == 2u) { return -P.coulomb * mi * mj / FA2_DIST_FLOOR_SQ; }\n return P.scalingRatio * mi * mj / FA2_DIST_FLOOR;\n}\nfn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: centroid (GRAVITY_CENTER 0) or origin (1); regular or strong\n var q = pi.xyz;\n if (GRAVITY_CENTER == 0u) { q = pi.xyz - S.centroid.xyz; }\n if (STRONG_GRAVITY) { return -P.gravity * pi.w * q; }\n let d = length(q);\n if (d > FA2_DIST_FLOOR) { return -P.gravity * pi.w * q / d; }\n return vec3f(0.0);\n}\n\n@compute @workgroup_size(WG)\nfn repulsion(\n @builtin(workgroup_id) wid: vec3<u32>,\n @builtin(local_invocation_id) lid: vec3<u32>,\n @builtin(num_workgroups) nwg: vec3<u32>,\n) {\n // issue #87: pass p of the tile range is dispatched with p + 1 z slices; only the last slice works, so the pass\n // index needs no uniform. Uniform: keyed on workgroup_id and num_workgroups only.\n if (wid.z + 1u < nwg.z) { return; }\n let i = linear_id(wid, lid.x);\n let valid = i < P.n;\n var pi = vec4f(0.0);\n if (valid) { pi = pos[i]; }\n var f = vec3f(0.0);\n let tiles = (P.n + WG - 1u) / WG;\n let tileBegin = (nwg.z - 1u) * EXACT_TILES_PER_PASS;\n let tileEnd = min(tiles, tileBegin + EXACT_TILES_PER_PASS);\n for (var t = tileBegin; t < tileEnd; t = t + 1u) {\n let j = t * WG + lid.x;\n if (j < P.n) { tile[lid.x] = pos[j]; } else { tile[lid.x] = vec4f(0.0); } // guarded fill; mass 0 marks the pad\n workgroupBarrier(); // uniform: every invocation reaches it\n for (var s = 0u; s < WG; s = s + 1u) {\n let o = tile[s];\n let jj = t * WG + s;\n if (o.w > 0.0 && jj != i) { // mass > 0 for every real node, 0 for the pad\n let d = pi.xyz - o.xyz;\n var d2 = dot(d, d);\n if (d2 < FA2_COINCIDENT_SQ) { // coincident: antisymmetric unit kick of the law's magnitude at d = 0.01 (7.2; PD-10)\n f = f + kick_dir(i, jj, P.dim) * kick_magnitude(pi.w, o.w);\n continue;\n }\n if (LAW == 0u) { d2 = max(d2, FA2_DIST_FLOOR_SQ); } // FA2 alone floors d >= 0.01 (7.2); FR and coulomb are unfloored (7.20)\n let k = P.scalingRatio * pi.w * o.w;\n if (LAW == 0u) { f = f + d * (k / d2); } // LAW 0 (FA2): |F| = k m_i m_j / d along d / d\n if (LAW == 1u) { f = f + d * (P.frK * P.frK / d2); } // LAW 1 (FR, 7.20): |F| = k^2 / d, mass ignored\n if (LAW == 2u) { f = f + d * (-P.coulomb * pi.w * o.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb, ngraph generateQuadTree.js:131-132): |F| = -g m_i m_j / d^2\n }\n }\n workgroupBarrier();\n }\n if (tileEnd < tiles) { // an earlier pass: its partial sum only (uniform: P.n, nwg)\n if (valid) { store_force(i, load_force(i) + f); }\n return;\n }\n // epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it\n var sw = 0.0;\n var tr = 0.0;\n if (valid) {\n f = f + gravity_force(pi);\n let fnew = load_force(i) + f;\n store_force(i, fnew);\n if (SWING_MODE == 1u) { // NetworkX: positions and forces mixed, every node (7.2)\n sw = pi.w * length(pi.xyz - fnew);\n tr = 0.5 * pi.w * length(pi.xyz + fnew);\n } else if (!mask_bit(fixedMask[i >> 5u], i)) { // paper: free nodes only (Gephi ForceAtlas2.java 283-293)\n let fold = load_old(i);\n sw = pi.w * length(fnew - fold);\n tr = 0.5 * pi.w * length(fnew + fold);\n }\n }\n let t = wg_reduce_vec4(vec4f(sw, tr, 0.0, 0.0), lid.x, 0u); // uniform control flow: 256 -> 1\n if (lid.x == 0u) { partials[group_id(wid)].swingTraction = t.xy; }\n}\n";
|
|
14
17
|
//# sourceMappingURL=fa2-repulsion-exact.wgsl.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"fa2-repulsion-exact.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-repulsion-exact.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"fa2-repulsion-exact.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-repulsion-exact.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AACH,eAAO,MAAM,qBAAqB,msJAuFjC,CAAC"}
|
|
@@ -9,6 +9,9 @@
|
|
|
9
9
|
* `partials[group].swingTraction`. `pos` is `array<vec4f>` with the mass in `.w`; `force` / `oldForce` are stride-3
|
|
10
10
|
* `array<f32>` read through the per-body helpers (4.4 rule 6). Normative text, copied verbatim: the P1-T5 sabotage
|
|
11
11
|
* rows (gravity sign, `k / d2`, the `jj != i` guard, the `.w` mass lane) are textual edits of this string.
|
|
12
|
+
* The j range is split into passes of at most EXACT_TILES_PER_PASS tiles (issue #87: llvmpipe's per-invocation loop
|
|
13
|
+
* budget); an earlier pass adds its partial sum into `force`, the last one runs gravity and the epilogue. A graph of
|
|
14
|
+
* at most 32,768 nodes is one pass, bitwise the single-pass kernel. Record it through recordExactRepulsion.
|
|
12
15
|
*/
|
|
13
16
|
export const fa2RepulsionExactWgsl = /* wgsl */ `
|
|
14
17
|
var<workgroup> tile: array<vec4f, WG>; // xyz + mass, 4 KiB at WG = 256
|
|
@@ -35,14 +38,23 @@ fn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: cent
|
|
|
35
38
|
}
|
|
36
39
|
|
|
37
40
|
@compute @workgroup_size(WG)
|
|
38
|
-
fn repulsion(
|
|
41
|
+
fn repulsion(
|
|
42
|
+
@builtin(workgroup_id) wid: vec3<u32>,
|
|
43
|
+
@builtin(local_invocation_id) lid: vec3<u32>,
|
|
44
|
+
@builtin(num_workgroups) nwg: vec3<u32>,
|
|
45
|
+
) {
|
|
46
|
+
// issue #87: pass p of the tile range is dispatched with p + 1 z slices; only the last slice works, so the pass
|
|
47
|
+
// index needs no uniform. Uniform: keyed on workgroup_id and num_workgroups only.
|
|
48
|
+
if (wid.z + 1u < nwg.z) { return; }
|
|
39
49
|
let i = linear_id(wid, lid.x);
|
|
40
50
|
let valid = i < P.n;
|
|
41
51
|
var pi = vec4f(0.0);
|
|
42
52
|
if (valid) { pi = pos[i]; }
|
|
43
53
|
var f = vec3f(0.0);
|
|
44
54
|
let tiles = (P.n + WG - 1u) / WG;
|
|
45
|
-
|
|
55
|
+
let tileBegin = (nwg.z - 1u) * EXACT_TILES_PER_PASS;
|
|
56
|
+
let tileEnd = min(tiles, tileBegin + EXACT_TILES_PER_PASS);
|
|
57
|
+
for (var t = tileBegin; t < tileEnd; t = t + 1u) {
|
|
46
58
|
let j = t * WG + lid.x;
|
|
47
59
|
if (j < P.n) { tile[lid.x] = pos[j]; } else { tile[lid.x] = vec4f(0.0); } // guarded fill; mass 0 marks the pad
|
|
48
60
|
workgroupBarrier(); // uniform: every invocation reaches it
|
|
@@ -65,6 +77,10 @@ fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
|
|
|
65
77
|
}
|
|
66
78
|
workgroupBarrier();
|
|
67
79
|
}
|
|
80
|
+
if (tileEnd < tiles) { // an earlier pass: its partial sum only (uniform: P.n, nwg)
|
|
81
|
+
if (valid) { store_force(i, load_force(i) + f); }
|
|
82
|
+
return;
|
|
83
|
+
}
|
|
68
84
|
// epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it
|
|
69
85
|
var sw = 0.0;
|
|
70
86
|
var tr = 0.0;
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"fa2-repulsion-exact.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-repulsion-exact.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"fa2-repulsion-exact.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-repulsion-exact.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AACH,MAAM,CAAC,MAAM,qBAAqB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAuF/C,CAAC"}
|
|
@@ -21,5 +21,5 @@
|
|
|
21
21
|
* is the target of the K1 sabotage mutations (test/helpers/sabotage.ts, P3-T5); amend the contract before editing.
|
|
22
22
|
*/
|
|
23
23
|
/** The K1 body: entry point `stats_finalize`; calls the reduction helpers (`needs: ["subgroups"]`, contract 4.3). */
|
|
24
|
-
export declare const fa2StatsFinalizeWgsl = "// K1: folds the previous integrate's partials into the state block (spec 7.4); one workgroup\n@compute @workgroup_size(WG)\nfn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n let groups = (P.n + WG - 1u) / WG;\n let fold = (P.flags & FA2_FLAG_FIRST) == 0u; // the first iteration after load() keeps the host-written state\n var sum = vec4f(0.0);\n var lo = vec4f(F32_MAX);\n var hi = vec4f(-F32_MAX);\n var disp = 0.0;\n var free = 0u;\n var ke = 0.0;\n if (fold) {\n for (var g = lid.x; g < groups; g = g + WG) { // sequential per lane in index order: deterministic\n let q = partials[g];\n sum = sum + q.sum;\n lo = min(lo, q.min);\n hi = max(hi, q.max);\n disp = disp + q.dispFree.x;\n free = free + u32(q.dispFree.y);\n ke = ke + q.swingTraction.x;\n }\n }\n let tSum = wg_reduce_vec4(sum, lid.x, 0u);\n let tLo = wg_reduce_vec4(lo, lid.x, 1u);\n let tHi = wg_reduce_vec4(hi, lid.x, 2u);\n let tDisp = wg_reduce_f32(disp, lid.x, 0u);\n let tFree = wg_reduce_u32(free, lid.x, 0u);\n let tKe = wg_reduce_f32(ke, lid.x, 0u);\n if (lid.x == 0u) {\n if (fold) {\n let n = f32(P.n);\n let c = tSum.xyz / n;\n S.centroid = vec4f(c, 0.0);\n S.rmsRadius = sqrt(max(tSum.w, 0.0) / n); // RMS radius about the previous centroid (7.17)\n S.min = vec4f(tLo.xyz, 0.0);\n S.max = vec4f(tHi.xyz, 0.0);\n S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)\n let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)\n S.meanDisplacement = meanDisp;\n S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius)
|
|
24
|
+
export declare const fa2StatsFinalizeWgsl = "// K1: folds the previous integrate's partials into the state block (spec 7.4); one workgroup\n@compute @workgroup_size(WG)\nfn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n let groups = (P.n + WG - 1u) / WG;\n let fold = (P.flags & FA2_FLAG_FIRST) == 0u; // the first iteration after load() keeps the host-written state\n var sum = vec4f(0.0);\n var lo = vec4f(F32_MAX);\n var hi = vec4f(-F32_MAX);\n var disp = 0.0;\n var free = 0u;\n var ke = 0.0;\n if (fold) {\n for (var g = lid.x; g < groups; g = g + WG) { // sequential per lane in index order: deterministic\n let q = partials[g];\n sum = sum + q.sum;\n lo = min(lo, q.min);\n hi = max(hi, q.max);\n disp = disp + q.dispFree.x;\n free = free + u32(q.dispFree.y);\n ke = ke + q.swingTraction.x;\n }\n }\n let tSum = wg_reduce_vec4(sum, lid.x, 0u);\n let tLo = wg_reduce_vec4(lo, lid.x, 1u);\n let tHi = wg_reduce_vec4(hi, lid.x, 2u);\n let tDisp = wg_reduce_f32(disp, lid.x, 0u);\n let tFree = wg_reduce_u32(free, lid.x, 0u);\n let tKe = wg_reduce_f32(ke, lid.x, 0u);\n if (lid.x == 0u) {\n if (fold) {\n let n = f32(P.n);\n let c = tSum.xyz / n;\n S.centroid = vec4f(c, 0.0);\n S.rmsRadius = sqrt(max(tSum.w, 0.0) / n); // RMS radius about the previous centroid (7.17)\n S.min = vec4f(tLo.xyz, 0.0);\n S.max = vec4f(tHi.xyz, 0.0);\n S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)\n let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)\n S.meanDisplacement = meanDisp;\n S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= min(P.settleThreshold * S.rmsRadius, P.settleFloor)); // relative AND absolute (issue #97)\n }\n S.iteration = S.iteration + 1u;\n T[P.iterationIndex].meanDisplacement = S.meanDisplacement;\n T[P.iterationIndex].settledCount = S.settledCount;\n T[P.iterationIndex].iteration = S.iteration;\n if (P.gridMax > 0u) { // the grid tier (7.7): the robust extent, the cell size, eps, last iteration's counts, the hub counter reset (PD-14)\n let cells = P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u);\n if (fold) {\n let box = (S.max.xyz - S.min.xyz) * GRID_BBOX_MARGIN;\n var bboxExtent = max(box.x, box.y);\n if (P.dim == 3u) { bboxExtent = max(bboxExtent, box.z); }\n let extent = max(min(bboxExtent, P.extentFactor * S.rmsRadius), GRID_EXTENT_FLOOR); // min(bbox, extentFactor x rms), floored (7.7)\n let cellSize = extent / f32(P.gridMax);\n S.gridMin = vec4f(S.centroid.xyz - vec3f(0.5 * extent), cellSize); // gridMin.w carries cellSize\n S.invCellSize = 1.0 / cellSize;\n S.eps = 0.25 * cellSize;\n }\n var outside = 0u; // the previous iteration's pseudo-cell counts, one per orthant (issue #90; 0 after load)\n for (var o = 0u; o < select(4u, 8u, P.dim == 3u); o = o + 1u) { outside = outside + cellHist[cells + o]; }\n S.outsideGrid = outside;\n S.maxCellOccupancy = atomicLoad(&hubCounters[1]);\n atomicStore(&hubCounters[0], 0u);\n atomicStore(&hubCounters[1], 0u);\n }\n if (STATS_MODE == 1u) { // FR: this iteration's temperature (7.20) into the state and the trace\n if ((P.flags & FA2_FLAG_ADAPTIVE) != 0u) { // adaptive cooling (Yifan Hu 2005 3.2): tKe is the previous iteration's sum |F|^2 over free nodes\n if (fold) {\n var t = S.temperature;\n if (tKe < S.frEnergy) {\n S.frProgress = S.frProgress + 1u;\n if (S.frProgress >= FR_COOLING_PATIENCE) { S.frProgress = 0u; t = t / FR_COOLING_STEP; }\n } else {\n S.frProgress = 0u;\n t = t * FR_COOLING_STEP;\n }\n S.frEnergy = tKe;\n S.temperature = t;\n }\n } else {\n S.temperature = P.temperature;\n }\n T[P.iterationIndex].modelScalar = S.temperature;\n }\n if (STATS_MODE == 2u) { // spring-electrical: the kinetic energy K5 folded into partials B (PD-4); 0 on the first iteration after load()\n S.kineticEnergy = tKe;\n T[P.iterationIndex].modelScalar = tKe;\n }\n }\n}";
|
|
25
25
|
//# sourceMappingURL=fa2-stats-finalize.wgsl.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"fa2-stats-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH,qHAAqH;AACrH,eAAO,MAAM,oBAAoB,
|
|
1
|
+
{"version":3,"file":"fa2-stats-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH,qHAAqH;AACrH,eAAO,MAAM,oBAAoB,42JAwF/B,CAAC"}
|
|
@@ -60,7 +60,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
60
60
|
S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)
|
|
61
61
|
let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)
|
|
62
62
|
S.meanDisplacement = meanDisp;
|
|
63
|
-
S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);
|
|
63
|
+
S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= min(P.settleThreshold * S.rmsRadius, P.settleFloor)); // relative AND absolute (issue #97)
|
|
64
64
|
}
|
|
65
65
|
S.iteration = S.iteration + 1u;
|
|
66
66
|
T[P.iterationIndex].meanDisplacement = S.meanDisplacement;
|
|
@@ -78,7 +78,9 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
78
78
|
S.invCellSize = 1.0 / cellSize;
|
|
79
79
|
S.eps = 0.25 * cellSize;
|
|
80
80
|
}
|
|
81
|
-
|
|
81
|
+
var outside = 0u; // the previous iteration's pseudo-cell counts, one per orthant (issue #90; 0 after load)
|
|
82
|
+
for (var o = 0u; o < select(4u, 8u, P.dim == 3u); o = o + 1u) { outside = outside + cellHist[cells + o]; }
|
|
83
|
+
S.outsideGrid = outside;
|
|
82
84
|
S.maxCellOccupancy = atomicLoad(&hubCounters[1]);
|
|
83
85
|
atomicStore(&hubCounters[0], 0u);
|
|
84
86
|
atomicStore(&hubCounters[1], 0u);
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"fa2-stats-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH,qHAAqH;AACrH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC
|
|
1
|
+
{"version":3,"file":"fa2-stats-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/fa2-stats-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH,qHAAqH;AACrH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EAwF7C,CAAC"}
|