@graphty/webgpu-graph-algorithms 0.6.5 → 0.6.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +38 -17
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-hzGggHeM.js → context-DiSr6eiz.js} +32 -18
- package/dist/chunks/context-DiSr6eiz.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/algorithms/bfs.d.ts +10 -6
- package/dist/src/algorithms/bfs.d.ts.map +1 -1
- package/dist/src/algorithms/bfs.js +32 -9
- package/dist/src/algorithms/bfs.js.map +1 -1
- package/dist/src/constants.d.ts +33 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +33 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernel/dispatch.d.ts +2 -2
- package/dist/src/kernel/dispatch.d.ts.map +1 -1
- package/dist/src/kernel/kernel.d.ts +1 -1
- package/dist/src/kernel/kernel.js +2 -2
- package/dist/src/kernel/kernel.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +2 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernels.d.ts +10 -7
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +33 -9
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
- package/dist/src/layouts/forceatlas2.js +2 -1
- package/dist/src/layouts/forceatlas2.js.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.js +4 -2
- package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
- package/dist/src/layouts/repulsion-exact.d.ts +16 -0
- package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
- package/dist/src/layouts/repulsion-exact.js +21 -1
- package/dist/src/layouts/repulsion-exact.js.map +1 -1
- package/dist/src/layouts/repulsion-grid.d.ts +1 -1
- package/dist/src/layouts/repulsion-grid.js +1 -1
- package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
- package/dist/src/layouts/spring-electrical.js +6 -2
- package/dist/src/layouts/spring-electrical.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +1 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -1
- package/dist/src/primitives/frontier.js +1 -0
- package/dist/src/primitives/frontier.js.map +1 -1
- package/dist/src/primitives/grid-pyramid.d.ts +4 -4
- package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
- package/dist/src/primitives/grid-pyramid.js +4 -3
- package/dist/src/primitives/grid-pyramid.js.map +1 -1
- package/dist/src/primitives/grid.d.ts +13 -10
- package/dist/src/primitives/grid.d.ts.map +1 -1
- package/dist/src/primitives/grid.js +10 -7
- package/dist/src/primitives/grid.js.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +21 -20
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.js +26 -25
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
- package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
- package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
- package/dist/src/wgsl/histogram.wgsl.js +1 -1
- package/dist/webgpu-graph-algorithms.js +124 -40
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +3 -3
- package/src/algorithms/bfs.ts +33 -9
- package/src/algorithms/pagerank.ts +19 -5
- package/src/algorithms/power-iteration.ts +8 -2
- package/src/constants.ts +35 -0
- package/src/kernel/dispatch.ts +2 -2
- package/src/kernel/kernel.ts +2 -2
- package/src/kernel/prelude.ts +2 -0
- package/src/kernels.ts +35 -9
- package/src/layouts/forceatlas2.ts +2 -0
- package/src/layouts/fruchterman-reingold.ts +4 -1
- package/src/layouts/repulsion-exact.ts +29 -1
- package/src/layouts/repulsion-grid.ts +1 -1
- package/src/layouts/spring-electrical.ts +8 -1
- package/src/memory/residency.ts +14 -4
- package/src/primitives/frontier.ts +2 -0
- package/src/primitives/grid-pyramid.ts +6 -5
- package/src/primitives/grid.ts +17 -12
- package/src/wgsl/advance-expand.wgsl.ts +4 -3
- package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
- package/src/wgsl/bfs-fused.wgsl.ts +6 -6
- package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
- package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
- package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
- package/src/wgsl/frontier-finalize.wgsl.ts +26 -25
- package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
- package/src/wgsl/grid-centroid.wgsl.ts +3 -2
- package/src/wgsl/grid-downsample.wgsl.ts +1 -1
- package/src/wgsl/grid-far-field.wgsl.ts +15 -8
- package/src/wgsl/grid-near-field.wgsl.ts +6 -2
- package/src/wgsl/histogram.wgsl.ts +1 -1
- package/dist/chunks/context-hzGggHeM.js.map +0 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@graphty/webgpu-graph-algorithms",
|
|
3
|
-
"version": "0.6.
|
|
3
|
+
"version": "0.6.6",
|
|
4
4
|
"description": "WebGPU-accelerated graph algorithms and layouts over the @graphty/graph-format snapshot, for Node (Dawn) and browsers",
|
|
5
5
|
"author": "Adam Powers <apowers@ato.ms>",
|
|
6
6
|
"type": "module",
|
|
@@ -93,8 +93,8 @@
|
|
|
93
93
|
"vite": "^7.0.5",
|
|
94
94
|
"vitest": "^3.2.4",
|
|
95
95
|
"webgpu": "0.4.0",
|
|
96
|
-
"@graphty/algorithms": "^2.0.
|
|
97
|
-
"@graphty/layout": "^1.
|
|
96
|
+
"@graphty/algorithms": "^2.0.4",
|
|
97
|
+
"@graphty/layout": "^1.10.0"
|
|
98
98
|
},
|
|
99
99
|
"scripts": {
|
|
100
100
|
"build": "tsc -p tsconfig.build.json",
|
package/src/algorithms/bfs.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Breadth-first search on the device (design 8.4, 3.3 line 807, 9.7; P8-T6 / P8-T7 / P8-T8, the P8 plan's PD-5 /
|
|
3
3
|
* PD-6 / PD-7 / PD-14 / PD-18 / PD-21 / PD-23 / PD-24 / PD-26 / DEP-P8-B): the direction-optimizing traversal over
|
|
4
|
-
* the `Frontier` of P8-T4, every per-level choice made ON THE DEVICE. Every level is
|
|
4
|
+
* the `Frontier` of P8-T4, every per-level choice made ON THE DEVICE. Every level is nine recorded dispatches, all
|
|
5
5
|
* DIRECT (2026-09-25, docs/decisions/G8.md G8-F5: Dawn validates every `dispatchWorkgroupsIndirect` with an internal
|
|
6
6
|
* clamp pass costing about 0.4 ms of device time whether or not it dispatches anything, in Node and in Chromium
|
|
7
7
|
* alike, and that was 97 % of a traversal's wall time) -- `frontier-finalize` role 0 (the level boundary: rotates
|
|
@@ -13,11 +13,15 @@
|
|
|
13
13
|
* per frontier entry with the same claim inline and no edge queue; the fused level and the retry alike), then the
|
|
14
14
|
* bottom-up trio: a `fill` zeroing the frontier bitset, `bfs-bitset-build` setting the frontier's bits, and
|
|
15
15
|
* `bfs-bottom-up` sweeping the unvisited list over the REVERSE core (each unvisited vertex reads its in-neighbours
|
|
16
|
-
* until the first one in the bitset and claims itself)
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
* the
|
|
20
|
-
*
|
|
16
|
+
* until the first one in the bitset and claims itself), and last `bfs-next-degree` (the out-degree sum of whatever
|
|
17
|
+
* the level claimed, into the block's `nextDegreeSum` word: Beamer's m_f for the NEXT boundary, measured on the
|
|
18
|
+
* frontier that boundary decides for -- issue #391, which found the previous proxy, the degree of the frontier just
|
|
19
|
+
* expanded, one level stale and missing the switch at the level holding two thirds of the 1M / 10M R-MAT's arcs).
|
|
20
|
+
* Every kernel is a grid-stride dispatch of a host-planned grid (`planGridStride`) that loops to its count word and
|
|
21
|
+
* reads the path word first, so exactly one path does the level's work and the others cost one uniform load per
|
|
22
|
+
* workgroup. The unvisited set Beamer's test is against (PD-18) is rebuilt exactly once per submit, before the
|
|
23
|
+
* levels, by `bfs-unvisited-flags` plus `compact` over an iota queue, and maintained between rebuilds by subtraction
|
|
24
|
+
* inside the selector (whose JSDoc states the boundary rule; both words are exact at every boundary). The host records `MAX_LEVELS_PER_SUBMIT`
|
|
21
25
|
* levels into ONE command buffer, submits, and reads four bytes, the `done` word (PD-7): a road network has
|
|
22
26
|
* thousands of levels and a per-level `mapAsync` would be slower than the CPU. A level recorded past the end is a
|
|
23
27
|
* no-op (its boundary finds `done` set, writes path 0 and moves no counter), so the loop needs no diameter and
|
|
@@ -79,6 +83,9 @@ import { type AlgorithmScope, algorithmScope } from "./scope.js";
|
|
|
79
83
|
|
|
80
84
|
const ALGORITHM = "breadthFirstSearch";
|
|
81
85
|
|
|
86
|
+
/** Workgroups of the `bfs-next-degree` grid (issue #391): enough to sum n out-degrees by grid stride, few enough that a high-diameter traversal does not pay a full-width reduction per level. */
|
|
87
|
+
const NEXT_DEGREE_MAX_GROUPS = 128;
|
|
88
|
+
|
|
82
89
|
/**
|
|
83
90
|
* Params slots of the ring, COUNTED per run (P8-T12), because `UniformRing.reserve` wraps to slot 0 when a submit's
|
|
84
91
|
* records outrun the ring and silently overwrites a record the submit still reads; the ring's `overruns` counts
|
|
@@ -87,8 +94,9 @@ const ALGORITHM = "breadthFirstSearch";
|
|
|
87
94
|
* `bfs-contract`, the bits `fill`, `bfs-bitset-build`) and four re-issued once per arc window (`advance-expand`,
|
|
88
95
|
* `bfs-fused`, the fused retry, `bfs-bottom-up`). The driver as built writes fewer -- the contract and the bitset
|
|
89
96
|
* build share one window-free record, a window's two fused dispatches share one, the bits fill has one record per
|
|
90
|
-
* submit,
|
|
91
|
-
* `
|
|
97
|
+
* submit, `bfs-next-degree` has one window-free record of its own (issue #391), and a directed snapshot's reverse
|
|
98
|
+
* view is one window whatever the forward core's count -- so at most `4 + 3 x windows` per level, and the bound
|
|
99
|
+
* holds with room. The 16 covers the per-submit rebuild
|
|
92
100
|
* (`bfs-unvisited-flags` and `compact`, whose scan is at most 9 dispatches for any n below 2^32, so 10, plus the
|
|
93
101
|
* bits fill). The result batch flushes in its own submit, so the ring must hold IT too, and its size grows with `n`,
|
|
94
102
|
* not with the cadence: the iota `fill`, the radix sort's four passes of one record plus its scan's `2 x levels - 1`
|
|
@@ -339,6 +347,7 @@ export async function bfsWithTuning(
|
|
|
339
347
|
const bitset = await ctx.pipelines.kernel(kernelSpec("bfs-bitset-build"));
|
|
340
348
|
const bottomUp = await ctx.pipelines.kernel(kernelSpec("bfs-bottom-up", graphOverrides(reverse, null)));
|
|
341
349
|
const unvisited = await ctx.pipelines.kernel(kernelSpec("bfs-unvisited-flags"));
|
|
350
|
+
const nextDegree = await ctx.pipelines.kernel(kernelSpec("bfs-next-degree"));
|
|
342
351
|
const pred = await ctx.pipelines.kernel(kernelSpec("sssp-pred", { ...graphOverrides(core, null), MODE: 1 }));
|
|
343
352
|
const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
|
|
344
353
|
const sort = await prepareRadixSort(scope);
|
|
@@ -352,6 +361,11 @@ export async function bfsWithTuning(
|
|
|
352
361
|
// plan's GROUP count as its stride (planGridStride's cap applies to the groups)
|
|
353
362
|
const levelPlan = planGridStride(Math.max(n, frontier.edgeCapacity), wg, ctx.caps);
|
|
354
363
|
const sweepPlan = planGridStride(n, wg, ctx.caps);
|
|
364
|
+
// `bfs-next-degree` sums at most n words and is dispatched on EVERY level, so its fixed cost -- one
|
|
365
|
+
// workgroup reduction (six barriers) and one atomic per workgroup -- is paid 1,999 times on the
|
|
366
|
+
// 1000 x 1000 grid. A capped grid pays it 128 times a level instead of ceil(n / wg): on the card the
|
|
367
|
+
// grid row went from 436 to 356 ms and neither R-MAT row moved.
|
|
368
|
+
const degreePlan = planGridStride(n, wg, ctx.caps, NEXT_DEGREE_MAX_GROUPS);
|
|
355
369
|
const fusedPlan = planGridStride(n * wg, wg, ctx.caps);
|
|
356
370
|
const bitsPlan = plan1d(bitsWords, wg, ctx.caps);
|
|
357
371
|
const recordFill = (pass: GPUComputePassEncoder, dst: Binding, value: number, mode: 0 | 1): void => {
|
|
@@ -415,7 +429,7 @@ export async function bfsWithTuning(
|
|
|
415
429
|
const bitsParams = scope.params(FILL_PARAMS, { count: bitsWords, value: 0, mode: 0, pad0: 0 });
|
|
416
430
|
const boundBitsFill = fill.bind({ dst: frontierBits, P: bitsParams.binding });
|
|
417
431
|
for (let level = 0; level < levelsPerSubmit; level++) {
|
|
418
|
-
planner.recordFinalize(pass, 0, level, { ...fields, firstOfSubmit: Math.min(level,
|
|
432
|
+
planner.recordFinalize(pass, 0, level, { ...fields, firstOfSubmit: Math.min(level, 1) });
|
|
419
433
|
advance.record(pass, frontier);
|
|
420
434
|
planner.recordFinalize(pass, 1, level, fields);
|
|
421
435
|
// one window-free record serves the contract and the bitset build, which read no arc; the kernels that
|
|
@@ -489,6 +503,16 @@ export async function bfsWithTuning(
|
|
|
489
503
|
});
|
|
490
504
|
bottomUp.dispatch(pass, boundSweep, sweepPlan, [sweepParams.offset]);
|
|
491
505
|
}
|
|
506
|
+
// the next frontier's out-degree sum (issue #391): whichever path claimed, the vertices are in the output
|
|
507
|
+
// queue now, and the next boundary reads word 25 as Beamer's m_f for the frontier it is about to expand
|
|
508
|
+
const degreeParams = scope.params(FRONTIER_PARAMS, { wg, n, stride: degreePlan.stride ?? wg });
|
|
509
|
+
const boundDegree = nextDegree.bind({
|
|
510
|
+
frontier: frontier.output,
|
|
511
|
+
outDegree,
|
|
512
|
+
counters,
|
|
513
|
+
P: degreeParams.binding,
|
|
514
|
+
});
|
|
515
|
+
nextDegree.dispatch(pass, boundDegree, degreePlan, [degreeParams.offset]);
|
|
492
516
|
frontier.swap();
|
|
493
517
|
}
|
|
494
518
|
batch.endPass();
|
|
@@ -34,8 +34,8 @@ import { algorithmScope } from "./scope.js";
|
|
|
34
34
|
|
|
35
35
|
/** Iterations per submit (spec 8.2: k = 8). */
|
|
36
36
|
const PR_BATCH = 8;
|
|
37
|
-
/** Params slots of one batch: per iteration one PrParams (shared by pr-scale and pr-finalize) and one SpmvParams, plus the normaliser's RangeParams; a smaller ring wraps onto a slot the same batch still reads. */
|
|
38
|
-
const RING_SLOTS = 2 * PR_BATCH +
|
|
37
|
+
/** Params slots of one batch: per iteration one PrParams (shared by pr-scale and pr-finalize) and one SpmvParams, plus the normaliser's RangeParams and the last batch's convergence check; a smaller ring wraps onto a slot the same batch still reads. */
|
|
38
|
+
const RING_SLOTS = 2 * PR_BATCH + 2;
|
|
39
39
|
|
|
40
40
|
/**
|
|
41
41
|
* Validates `options.dest` for a score result of `n` elements.
|
|
@@ -175,6 +175,8 @@ async function run(
|
|
|
175
175
|
const groups = groupsOf(scalePlan);
|
|
176
176
|
const partialsBytes = PR_PARTIAL.byteLength * (1 + groups);
|
|
177
177
|
const partials = scope.scratch(partialsBytes, "partials");
|
|
178
|
+
// the header as the last pull left it, before the convergence check of the last batch overwrites danglingMass
|
|
179
|
+
const lastHeader = scope.scratch(PR_PARTIAL.byteLength, "lastHeader");
|
|
178
180
|
if (personalization !== null) {
|
|
179
181
|
uploaded = ctx.residency.array(personalization, `${algorithm}/personalization`);
|
|
180
182
|
}
|
|
@@ -203,17 +205,21 @@ async function run(
|
|
|
203
205
|
const xNormBinding = bindingOf(xNorm, bytes);
|
|
204
206
|
const outWeightSumBinding = bindingOf(outWeightSum, bytes);
|
|
205
207
|
const partialsBinding = bindingOf(partials, partialsBytes);
|
|
208
|
+
const lastHeaderBinding = bindingOf(lastHeader, PR_PARTIAL.byteLength);
|
|
206
209
|
const coefficients = { alpha, beta: 1 - alpha, uniformP: 1 / n };
|
|
207
210
|
let cur = 0;
|
|
208
211
|
let iterationsRun = 0;
|
|
209
212
|
for (;;) {
|
|
210
213
|
const k = Math.min(PR_BATCH, maxIterations - iterationsRun);
|
|
211
214
|
const batch = new CommandBatch(ctx, algorithm);
|
|
212
|
-
|
|
215
|
+
let pass = batch.pass("iterations");
|
|
213
216
|
if (iterationsRun === 0) {
|
|
214
217
|
normaliser.record(pass, weightedCore, outWeightSumBinding);
|
|
215
218
|
}
|
|
216
|
-
|
|
219
|
+
const last = iterationsRun + k === maxIterations;
|
|
220
|
+
// the scale + finalize of iteration i measures the error of iteration i - 1 (PD-9), so the last batch
|
|
221
|
+
// runs them once more, with no pull, to measure the error of iteration maxIterations itself
|
|
222
|
+
for (let i = 0; i < k + (last ? 1 : 0); i++) {
|
|
217
223
|
const params = scope.params(PR_PARAMS, {
|
|
218
224
|
n,
|
|
219
225
|
groups,
|
|
@@ -222,6 +228,10 @@ async function run(
|
|
|
222
228
|
convergeThreshold: tolerance * n,
|
|
223
229
|
});
|
|
224
230
|
const other = 1 - cur;
|
|
231
|
+
if (i === k) {
|
|
232
|
+
batch.copy(partialsBinding, lastHeaderBinding, PR_PARTIAL.byteLength);
|
|
233
|
+
pass = batch.pass("convergence");
|
|
234
|
+
}
|
|
225
235
|
const scaleBound = scale.bind({
|
|
226
236
|
rankIn: rank[cur],
|
|
227
237
|
rankPrev: rank[other],
|
|
@@ -233,6 +243,9 @@ async function run(
|
|
|
233
243
|
scale.dispatch(pass, scaleBound, scalePlan, [params.offset]);
|
|
234
244
|
const finalizeBound = finalize.bind({ partials: partialsBinding, P: params.binding });
|
|
235
245
|
finalize.dispatch(pass, finalizeBound, finalizePlan, [params.offset]);
|
|
246
|
+
if (i === k) {
|
|
247
|
+
break;
|
|
248
|
+
}
|
|
236
249
|
pull.record(
|
|
237
250
|
pass,
|
|
238
251
|
weightedRev,
|
|
@@ -248,6 +261,7 @@ async function run(
|
|
|
248
261
|
}
|
|
249
262
|
batch.endPass();
|
|
250
263
|
const headerRequest = batch.readback(partials, 0, PR_PARTIAL.byteLength);
|
|
264
|
+
const lastHeaderRequest = last ? batch.readback(lastHeader, 0, PR_PARTIAL.byteLength) : headerRequest;
|
|
251
265
|
const scoresRequest = batch.readback(rank[cur].buffer, 0, bytes);
|
|
252
266
|
scope.flush();
|
|
253
267
|
const submitted = batch.submit();
|
|
@@ -271,7 +285,7 @@ async function run(
|
|
|
271
285
|
scores,
|
|
272
286
|
iterations: converged ? firstConverged : iterationsRun,
|
|
273
287
|
converged,
|
|
274
|
-
danglingMass:
|
|
288
|
+
danglingMass: PR_PARTIAL.read(new DataView(back), lastHeaderRequest.offset).danglingMass as number,
|
|
275
289
|
precision: "f32",
|
|
276
290
|
};
|
|
277
291
|
}
|
|
@@ -33,7 +33,7 @@ import { algorithmScope } from "./scope.js";
|
|
|
33
33
|
|
|
34
34
|
/** Iterations per submit (spec 8.2: k = 8). */
|
|
35
35
|
const BATCH = 8;
|
|
36
|
-
/** Params slots of one batch: four blocks per iteration times the batch, plus a margin (the plan's `4 * 8 + 8`). */
|
|
36
|
+
/** Params slots of one batch: four blocks per iteration times the batch, plus a margin (the plan's `4 * 8 + 8`) that also holds the last batch's convergence check. */
|
|
37
37
|
const RING_SLOTS = 4 * BATCH + 8;
|
|
38
38
|
|
|
39
39
|
/**
|
|
@@ -210,7 +210,10 @@ export async function runPowerIteration(
|
|
|
210
210
|
const k = Math.min(BATCH, config.maxIterations - iterationsRun);
|
|
211
211
|
const batch = new CommandBatch(ctx, config.label);
|
|
212
212
|
const pass = batch.pass("iterations");
|
|
213
|
-
|
|
213
|
+
// the scale + finalize of iteration i measures the error of iteration i - 1 (PD-9), so the last batch
|
|
214
|
+
// runs them once more, with no apply and no pull, to measure the error of iteration maxIterations itself
|
|
215
|
+
const last = iterationsRun + k === config.maxIterations;
|
|
216
|
+
for (let i = 0; i < k + (last ? 1 : 0); i++) {
|
|
214
217
|
const iteration = iterationsRun + i + 1;
|
|
215
218
|
const params = scope.params(PR_PARAMS, {
|
|
216
219
|
n,
|
|
@@ -234,6 +237,9 @@ export async function runPowerIteration(
|
|
|
234
237
|
};
|
|
235
238
|
scaleNorm.dispatch(pass, scaleNorm.bind(scaleBindings), scalePlan, [params.offset]);
|
|
236
239
|
finalize.dispatch(pass, finalize.bind({ partials, P: params.binding }), finalizePlan, [params.offset]);
|
|
240
|
+
if (i === k) {
|
|
241
|
+
break;
|
|
242
|
+
}
|
|
237
243
|
if (scaleApply !== null) {
|
|
238
244
|
scaleApply.dispatch(pass, scaleApply.bind(scaleBindings), scalePlan, [params.offset]);
|
|
239
245
|
}
|
package/src/constants.ts
CHANGED
|
@@ -130,6 +130,13 @@ export const FA2_DISTANCE_FLOOR = 0.01;
|
|
|
130
130
|
export const FA2_DISTANCE_FLOOR_SQ = 0.0001;
|
|
131
131
|
/** The coincident threshold `d^2 < 1e-8` of spec 7.2. */
|
|
132
132
|
export const FA2_COINCIDENT_SQ = 1e-8;
|
|
133
|
+
/**
|
|
134
|
+
* The most WG-node tiles one K3 (`fa2-repulsion-exact`) dispatch sums per invocation (issue #87). llvmpipe runs at
|
|
135
|
+
* most 65,535 loop iterations per shader invocation, counted over every loop together, and then quietly breaks
|
|
136
|
+
* out of each loop; one tile costs WG + 2 of them (258 at WG 256), so a single pass lost every node past
|
|
137
|
+
* j = 65,027. 128 tiles is 33,024 iterations, half the budget; a larger exact run records ceil(tiles / 128) passes.
|
|
138
|
+
*/
|
|
139
|
+
export const EXACT_TILES_PER_PASS = 128;
|
|
133
140
|
/** Bits of Fa2Params.flags (contract 4.4). */
|
|
134
141
|
export const FA2_FLAG_FIRST = 1;
|
|
135
142
|
/** Fa2Params.flags bit: the Fruchterman-Reingold temperature is the adaptive one in the state block, not the uniform's (the `cooling: "adaptive"` option). */
|
|
@@ -202,6 +209,34 @@ export const SE_DEFAULTS: Readonly<{
|
|
|
202
209
|
iterationsPerStep: 1,
|
|
203
210
|
maxInFlight: 2,
|
|
204
211
|
});
|
|
212
|
+
/**
|
|
213
|
+
* The absolute settle floor of the shared settle rule (spec 7.17; issue #97): an iteration counts toward `settled` only
|
|
214
|
+
* when its mean displacement is at most `settleThreshold x rmsRadius` AND at most this fraction of the model's length
|
|
215
|
+
* unit -- `springLength` for spring-electrical (scaled by node count, see SETTLE_FLOOR_REFERENCE_NODES), `k` for
|
|
216
|
+
* Fruchterman-Reingold. The relative rule alone reported a spring
|
|
217
|
+
* layout settled while it still grew (6 % over 1,000 iterations at 10k nodes). ForceAtlas2 writes
|
|
218
|
+
* `SETTLE_FLOOR_UNBOUNDED` instead: it does not drift after settling, and its per-iteration jitter grows with n, so any
|
|
219
|
+
* fixed floor only delays or blocks its stop. The values are measured:
|
|
220
|
+
* design/decisions/2026-09-24-settle-rule-has-an-absolute-floor.md.
|
|
221
|
+
*/
|
|
222
|
+
export const SETTLE_FLOOR_FRACTION: Readonly<{ springElectrical: number; fruchtermanReingold: number }> = Object.freeze(
|
|
223
|
+
{
|
|
224
|
+
springElectrical: 3e-3,
|
|
225
|
+
fruchtermanReingold: 2e-3,
|
|
226
|
+
},
|
|
227
|
+
);
|
|
228
|
+
/**
|
|
229
|
+
* The node count at which the spring-electrical floor is exactly `SETTLE_FLOOR_FRACTION.springElectrical x
|
|
230
|
+
* springLength`; at `n` nodes it is that times `(SETTLE_FLOOR_REFERENCE_NODES / n)^(1/4)`. A spring layout's rms radius
|
|
231
|
+
* grows about as n^(1/4) in springLengths (3.4 at 150 nodes, 6.4 at 2,000, 10.4 at 10,000), so the relative half of the
|
|
232
|
+
* rule loosens with size while the floor tightens with it: on a small graph the floor sits above the relative threshold
|
|
233
|
+
* and the relative rule decides alone, as before issue #97; on a large one the floor binds, which is where the relative
|
|
234
|
+
* rule let an expanding layout stop. A fixed floor bound the 150-node story graph too, where the grid tier's jitter sits
|
|
235
|
+
* at the relative threshold, and nearly doubled its settle (427 -> 829 iterations on the RTX 4070 SUPER).
|
|
236
|
+
*/
|
|
237
|
+
export const SETTLE_FLOOR_REFERENCE_NODES = 2000;
|
|
238
|
+
/** The settle floor that never binds: the largest finite f32, 0x1.fffffep+127 (ForceAtlas2's `settleFloor`, issue #97). */
|
|
239
|
+
export const SETTLE_FLOOR_UNBOUNDED = 2 ** 128 - 2 ** 104;
|
|
205
240
|
/** The smallest finest grid side `G` (spec 7.7 geometry table: `clamp(nextPow2(2 n^(1/dim)), 8, gridMax)`; P4 PD-9). */
|
|
206
241
|
export const GRID_MIN_SIDE = 8;
|
|
207
242
|
/** The coarsest pyramid level's side (spec 7.7: "levels (coarsest 4 per axis)", `levels = log2(G / 4) + 1`). */
|
package/src/kernel/dispatch.ts
CHANGED
|
@@ -9,11 +9,11 @@ import { MAX_WORKGROUPS_PER_DIM } from "../constants.js";
|
|
|
9
9
|
import { WebGpuGraphError } from "../errors.js";
|
|
10
10
|
import { type PlanCaps } from "../types/context.js";
|
|
11
11
|
|
|
12
|
-
/** A dispatch shape (spec 5.2). `stride` is the grid-stride step (null for plain 1D / 2D plans). */
|
|
12
|
+
/** A dispatch shape (spec 5.2). `stride` is the grid-stride step (null for plain 1D / 2D plans). `z` is 1 from every planner; only K3's pass split sets it (issue #87). */
|
|
13
13
|
export interface DispatchPlan {
|
|
14
14
|
readonly x: number;
|
|
15
15
|
readonly y: number;
|
|
16
|
-
readonly z:
|
|
16
|
+
readonly z: number;
|
|
17
17
|
readonly items: number;
|
|
18
18
|
readonly stride: number | null;
|
|
19
19
|
}
|
package/src/kernel/kernel.ts
CHANGED
|
@@ -245,7 +245,7 @@ export class Kernel {
|
|
|
245
245
|
}
|
|
246
246
|
|
|
247
247
|
/**
|
|
248
|
-
* setPipeline + setBindGroup for every group (dynamic offsets in dynamicGroups order) + dispatchWorkgroups(plan.x, plan.y,
|
|
248
|
+
* setPipeline + setBindGroup for every group (dynamic offsets in dynamicGroups order) + dispatchWorkgroups(plan.x, plan.y, plan.z); a plan with x === 0 records nothing (spec 5.6).
|
|
249
249
|
* PLAN DECISION: one dynamic offset per dynamic GROUP, replicated over every uniform binding of that group (every
|
|
250
250
|
* P1-P3 kernel has exactly one params uniform per group); absent offsets mean 0; a BoundKernel of another kernel
|
|
251
251
|
* or an offset list of the wrong length is E_INVALID_ARGUMENT.
|
|
@@ -268,7 +268,7 @@ export class Kernel {
|
|
|
268
268
|
return;
|
|
269
269
|
}
|
|
270
270
|
this.setUp(pass, bound, dynamicOffsets);
|
|
271
|
-
pass.dispatchWorkgroups(plan.x, plan.y,
|
|
271
|
+
pass.dispatchWorkgroups(plan.x, plan.y, plan.z);
|
|
272
272
|
}
|
|
273
273
|
|
|
274
274
|
/**
|
package/src/kernel/prelude.ts
CHANGED
|
@@ -12,6 +12,7 @@
|
|
|
12
12
|
import { INVALID_INDEX } from "@graphty/graph-format";
|
|
13
13
|
|
|
14
14
|
import {
|
|
15
|
+
EXACT_TILES_PER_PASS,
|
|
15
16
|
F32_INF_BITS,
|
|
16
17
|
FA2_COINCIDENT_SQ,
|
|
17
18
|
FA2_DISTANCE_FLOOR,
|
|
@@ -54,6 +55,7 @@ const INVALID_INDEX: u32 = ${INVALID_INDEX}u;
|
|
|
54
55
|
const U32_MAX: u32 = ${U32_MAX}u;
|
|
55
56
|
const F32_INF_BITS: u32 = ${F32_INF_BITS}u;
|
|
56
57
|
const MAX_WORKGROUPS_PER_DIM: u32 = ${MAX_WORKGROUPS_PER_DIM}u;
|
|
58
|
+
const EXACT_TILES_PER_PASS: u32 = ${EXACT_TILES_PER_PASS}u;
|
|
57
59
|
const FA2_DIST_FLOOR: f32 = ${wgslF32Literal(FA2_DISTANCE_FLOOR)};
|
|
58
60
|
const FA2_DIST_FLOOR_SQ: f32 = ${wgslF32Literal(FA2_DISTANCE_FLOOR_SQ)};
|
|
59
61
|
const FA2_COINCIDENT_SQ: f32 = ${wgslF32Literal(FA2_COINCIDENT_SQ)};
|
package/src/kernels.ts
CHANGED
|
@@ -28,6 +28,7 @@ import { bfsBitsetBuildWgsl } from "./wgsl/bfs-bitset-build.wgsl.js";
|
|
|
28
28
|
import { bfsBottomUpWgsl } from "./wgsl/bfs-bottom-up.wgsl.js";
|
|
29
29
|
import { bfsContractWgsl } from "./wgsl/bfs-contract.wgsl.js";
|
|
30
30
|
import { bfsFusedWgsl } from "./wgsl/bfs-fused.wgsl.js";
|
|
31
|
+
import { bfsNextDegreeWgsl } from "./wgsl/bfs-next-degree.wgsl.js";
|
|
31
32
|
import { bfsUnvisitedFlagsWgsl } from "./wgsl/bfs-unvisited-flags.wgsl.js";
|
|
32
33
|
import { closenessReduceWgsl } from "./wgsl/closeness-reduce.wgsl.js";
|
|
33
34
|
import { closenessSweepWgsl } from "./wgsl/closeness-sweep.wgsl.js";
|
|
@@ -111,6 +112,7 @@ export type KernelId =
|
|
|
111
112
|
| "bfs-bottom-up"
|
|
112
113
|
| "bfs-bitset-build"
|
|
113
114
|
| "bfs-unvisited-flags"
|
|
115
|
+
| "bfs-next-degree"
|
|
114
116
|
| "sssp-relax"
|
|
115
117
|
| "bf-relax"
|
|
116
118
|
| "closeness-sweep"
|
|
@@ -162,7 +164,7 @@ export const FILL_PARAMS: UniformBlock = UniformBlock.define("FillParams", [
|
|
|
162
164
|
["pad0", "u32"],
|
|
163
165
|
]);
|
|
164
166
|
|
|
165
|
-
/** `Fa2Params` (uniform,
|
|
167
|
+
/** `Fa2Params` (uniform, 144 B; spec 7.3): the per-iteration ForceAtlas2 parameters -- `n` @0, `dim` @4, `flags` @8 (bit 0 = FA2_FLAG_FIRST), `tierStart` @12, `tierEnd` @16, `iterationIndex` @20, `seed` @24, `nearMax` @28, `scalingRatio` @32, `gravity` @36, `jitterTolerance` @40, `scale` @44, `center` @48 (xyz, w 0), `settleThreshold` @64, `extentFactor` @68, `gridMax` @72, `levels` @76, `arcBase` @80 / `arcEnd` @84 (the bound arc window of K2, 0 and arcCount in the layout), `accumulate` @88 (1 combines into `force`: the windowed pattern), `hiEnd` @92 / `midEnd` @124 (the degreeOrder tier boundaries, PD-7; both 0 without a permutation); the P5 model fields (PD-3): `frK` @96 (the FR optimal distance), `temperature` @100 (the FR temperature of this iteration), `springLength` @104, `springCoefficient` @108, `coulomb` @112 (ngraph's `gravity`, negative repels), `dragCoefficient` @116, `timeStep` @120; `settleFloor` @128 (the absolute bound on the mean displacement of a settled iteration, in layout units: issue #97); 144 B. */
|
|
166
168
|
export const FA2_PARAMS: UniformBlock = UniformBlock.define("Fa2Params", [
|
|
167
169
|
["n", "u32"],
|
|
168
170
|
["dim", "u32"],
|
|
@@ -193,6 +195,7 @@ export const FA2_PARAMS: UniformBlock = UniformBlock.define("Fa2Params", [
|
|
|
193
195
|
["dragCoefficient", "f32"],
|
|
194
196
|
["timeStep", "f32"],
|
|
195
197
|
["midEnd", "u32"],
|
|
198
|
+
["settleFloor", "f32"],
|
|
196
199
|
]);
|
|
197
200
|
|
|
198
201
|
/** `Fa2State` (storage, padded to STATE_HEADER_BYTES = 256; spec 7.3): the device-resident controller state the finalize kernels write and the host reads back for stats -- `speed` @0, `speedEfficiency` @4, `swing` @8, `traction` @12, `centroid` @16, `rmsRadius` @32, `radius` @36, `meanDisplacement` @40, `iteration` @44, `min` @48, `max` @64, `gridMin` @80 (P4), `eps` @96 (P4), `settledCount` @100, `outsideGrid` @104 (P4), `maxCellOccupancy` @108 (P4), `temperature` @112 (FR, written by K1 under STATS_MODE 1), `kineticEnergy` @116 (the preset, K1 under STATS_MODE 2), `frEnergy` @120 / `frProgress` @124 (the FR adaptive cooling), `invCellSize` @128 (P4, PD-10: `1 / cellSize`, written by K1 beside `cellSize` in `gridMin.w`; G1 multiplies by it so every key is bitwise reproducible), `reserved0` @132 (f32), `reserved1` @136 (vec2f), `reserved2` .. `reserved8` @144 .. @240. */
|
|
@@ -386,8 +389,11 @@ export const COMPACT_PARAMS: UniformBlock = UniformBlock.define("CompactParams",
|
|
|
386
389
|
* submit), `arcsScanned` @64, `fusedLevels` @68, `twoPhaseLevels` @72, `bottomUpLevels` @76, `farCount` @80,
|
|
387
390
|
* `nextFarCount` @84, `thresholdBits` @88, `deltaBits` @92 (P8-T9), `path` @96 (what the level's kernels run, written
|
|
388
391
|
* by the selector: 0 nothing, 1 two-phase, 2 fused, 3 bottom-up, 4 the fused retry, 5 a near SSSP round, 6 a far
|
|
389
|
-
* one; every level kernel is a direct dispatch that reads it first -- G8-F5)
|
|
390
|
-
*
|
|
392
|
+
* one; every level kernel is a direct dispatch that reads it first -- G8-F5), `nextDegreeSum` @100 (issue #391: the
|
|
393
|
+
* out-degree sum of the vertices the level claimed, accumulated by `bfs-next-degree` at the end of every level and
|
|
394
|
+
* read, subtracted and zeroed by the next boundary -- Beamer's m_f measured on the frontier the boundary decides
|
|
395
|
+
* for, not on the one it has just expanded). The words nothing writes before P8-T8 / P8-T9 are declared now because
|
|
396
|
+
* the byte layout is what the single result copy decodes.
|
|
391
397
|
*/
|
|
392
398
|
export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
393
399
|
"FrontierCounters",
|
|
@@ -417,6 +423,7 @@ export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
|
417
423
|
["thresholdBits", "u32"],
|
|
418
424
|
["deltaBits", "u32"],
|
|
419
425
|
["path", "u32"],
|
|
426
|
+
["nextDegreeSum", "u32"],
|
|
420
427
|
],
|
|
421
428
|
{ layout: "storage" },
|
|
422
429
|
);
|
|
@@ -427,9 +434,9 @@ export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
|
427
434
|
* `alpha` @8, `beta` @12 (Beamer's thresholds, P8-T8), `fusedMax` @16, `edgeCapacity` @20, `maxDepth` @24, `n` @28,
|
|
428
435
|
* `mode` @32 (BFS: 0 auto, 1 top-down only; `sssp-pred`: the PD-27 key rule), `cutoffBits` @36, `arcBase` @40,
|
|
429
436
|
* `arcEnd` @44 (the bound arc window), `predKind` @48 (0 arc, 1 node), `bitsBase` @52, `source` @56, `stride` @60
|
|
430
|
-
* (a grid-stride plan's stride), `firstOfSubmit` @64 (the boundary's index inside its submit, clamped to
|
|
431
|
-
* unvisited-count
|
|
432
|
-
* P8-T9), `pad1` @72, `pad2` @76. The `slotBase` field that once addressed the selector's indirect slots went with
|
|
437
|
+
* (a grid-stride plan's stride), `firstOfSubmit` @64 (the boundary's index inside its submit, clamped to 1: both
|
|
438
|
+
* the unvisited-count and the unvisited-degree-sum subtraction run at >= 1, issue #391), `iteration` @68 (an
|
|
439
|
+
* `sssp-pred` hop pass, P8-T9), `pad1` @72, `pad2` @76. The `slotBase` field that once addressed the selector's indirect slots went with
|
|
433
440
|
* the slots (2026-09-25); `pad2` keeps the block an explicit 80 bytes, the way every block here is padded.
|
|
434
441
|
*/
|
|
435
442
|
export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams", [
|
|
@@ -939,7 +946,7 @@ const RADIX_SCATTER: KernelEntry = {
|
|
|
939
946
|
phase: "P4",
|
|
940
947
|
};
|
|
941
948
|
|
|
942
|
-
/** `grid-cell-key` (G1, spec 7.7; P4-T8, PD-10): the finest cell key of every node, `floor((p - gridMin) * invCellSize)` linearised, or the outside pseudo-cell `G^dim
|
|
949
|
+
/** `grid-cell-key` (G1, spec 7.7; P4-T8, PD-10): the finest cell key of every node, `floor((p - gridMin) * invCellSize)` linearised, or the outside pseudo-cell of its orthant `G^dim + orthant` (issue #90); `cellVal[i] = i`; 4 storage bindings (the state read-only: K1 writes it). */
|
|
943
950
|
const GRID_CELL_KEY: KernelEntry = {
|
|
944
951
|
id: "grid-cell-key",
|
|
945
952
|
body: gridCellKeyWgsl,
|
|
@@ -958,7 +965,7 @@ const GRID_CELL_KEY: KernelEntry = {
|
|
|
958
965
|
phase: "P4",
|
|
959
966
|
};
|
|
960
967
|
|
|
961
|
-
/** `grid-centroid` (G4, spec 7.7; P4-T9, PD-13): thread per finest cell (the pseudo-
|
|
968
|
+
/** `grid-centroid` (G4, spec 7.7; P4-T9, PD-13): thread per finest cell (the 2^dim orthant pseudo-cells included), the serial mass-weighted sum in sorted order into level 0, the occupancy max into `hubCounters[1]`, hub cells (> GRID_HUB_CELL) appended to `hubList`; 6 storage bindings. */
|
|
962
969
|
const GRID_CENTROID: KernelEntry = {
|
|
963
970
|
id: "grid-centroid",
|
|
964
971
|
body: gridCentroidWgsl,
|
|
@@ -1013,7 +1020,7 @@ const GRID_DOWNSAMPLE: KernelEntry = {
|
|
|
1013
1020
|
phase: "P4",
|
|
1014
1021
|
};
|
|
1015
1022
|
|
|
1016
|
-
/** `grid-far-field` (G6, spec 7.7; P4-T10, PD-16, DEP-P4-G): per node in sorted order, the coarsest level minus its 3x3 (3x3x3) and, per finer level, the parent's 3x3 refined minus the level's own 3x3, plus the pseudo-
|
|
1023
|
+
/** `grid-far-field` (G6, spec 7.7; P4-T10, PD-16, DEP-P4-G): per node in sorted order, the coarsest level minus its 3x3 (3x3x3) and, per finer level, the parent's 3x3 refined minus the level's own 3x3, plus the 2^dim orthant pseudo-cells; the FA2 term floored at 0.01 (issue #89); the loop bounds are `P.levels` / `P.gridMax`; LAW 0 FA2 / 1 FR / 2 coulomb per cell (P4-T13, PD-22); 5 storage bindings. */
|
|
1017
1024
|
const GRID_FAR_FIELD: KernelEntry = {
|
|
1018
1025
|
id: "grid-far-field",
|
|
1019
1026
|
body: gridFarFieldWgsl,
|
|
@@ -1261,6 +1268,24 @@ const BFS_UNVISITED_FLAGS: KernelEntry = {
|
|
|
1261
1268
|
phase: "P8",
|
|
1262
1269
|
};
|
|
1263
1270
|
|
|
1271
|
+
/** `bfs-next-degree` (design 8.4; issue #391): Beamer's m_f measured exactly -- once per level, after the claim kernels, grid-striding over the output vertex queue and summing the `outDegree` view over the vertices the level claimed into word 25, one `atomicAdd` per workgroup; 3 storage bindings (`frontier` read-only, the `outDegree` VIEW, the counters block); `needs: ["subgroups"]` for the `wg_reduce_u32` call (a twin kernel). */
|
|
1272
|
+
const BFS_NEXT_DEGREE: KernelEntry = {
|
|
1273
|
+
id: "bfs-next-degree",
|
|
1274
|
+
body: bfsNextDegreeWgsl,
|
|
1275
|
+
entryPoint: "bfs_next_degree",
|
|
1276
|
+
bindings: [
|
|
1277
|
+
decl(1, 0, "frontier", "storage-ro", "array<u32>"),
|
|
1278
|
+
decl(1, 1, "outDegree", "storage-ro", "array<u32>"),
|
|
1279
|
+
decl(1, 2, "counters", "storage", "array<atomic<u32>>"),
|
|
1280
|
+
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1281
|
+
],
|
|
1282
|
+
overrideDecls: [],
|
|
1283
|
+
uniforms: [FRONTIER_PARAMS],
|
|
1284
|
+
needs: ["subgroups"],
|
|
1285
|
+
snippetSlots: [],
|
|
1286
|
+
phase: "P8",
|
|
1287
|
+
};
|
|
1288
|
+
|
|
1264
1289
|
/** `sssp-relax` (design 8.4 "Davidson's near-far", 8.10 "SSSP near-far relax"; P8-T9, PD-9 / PD-20 / DEP-P8-E): one round of the near-far loop -- role 0 relaxes the deduped near pile's whole rows with `atomicMin` on the f32 bit patterns of `dist` and appends each improved vertex to the raw near or far half of `queueOut` (the two halves of ONE buffer at word 0 and word `P.edgeCapacity`), role 1 re-buckets the deduped far pile; 8 storage bindings (the four graph slots with the run's weights bound in the weights slot, `dist` and the counters block as `array<atomic<u32>>`, `queueIn` read-only, `queueOut`) -- exactly at the budget, which is why no `pred` lives here (PD-11) and why the piles' counts, the threshold and the delta are words of the block. */
|
|
1265
1290
|
const SSSP_RELAX: KernelEntry = {
|
|
1266
1291
|
id: "sssp-relax",
|
|
@@ -1392,6 +1417,7 @@ const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze
|
|
|
1392
1417
|
"bfs-bottom-up": BFS_BOTTOM_UP,
|
|
1393
1418
|
"bfs-bitset-build": BFS_BITSET_BUILD,
|
|
1394
1419
|
"bfs-unvisited-flags": BFS_UNVISITED_FLAGS,
|
|
1420
|
+
"bfs-next-degree": BFS_NEXT_DEGREE,
|
|
1395
1421
|
"sssp-relax": SSSP_RELAX,
|
|
1396
1422
|
"bf-relax": BF_RELAX,
|
|
1397
1423
|
"closeness-sweep": CLOSENESS_SWEEP,
|
|
@@ -26,6 +26,7 @@ import {
|
|
|
26
26
|
GRID_EXTENT_FLOOR,
|
|
27
27
|
LAYOUT_TUNING_DEFAULTS,
|
|
28
28
|
MAX_ITERATIONS_PER_STEP,
|
|
29
|
+
SETTLE_FLOOR_UNBOUNDED,
|
|
29
30
|
TRACE_RECORD_BYTES,
|
|
30
31
|
UNIFORM_SLOT_BYTES,
|
|
31
32
|
} from "../constants.js";
|
|
@@ -590,6 +591,7 @@ export class ForceAtlas2Model implements ForceModel<ForceAtlas2Options, ForceAtl
|
|
|
590
591
|
accumulate: 0,
|
|
591
592
|
hiEnd,
|
|
592
593
|
midEnd,
|
|
594
|
+
settleFloor: SETTLE_FLOOR_UNBOUNDED, // ForceAtlas2 does not drift after settling (issue #97)
|
|
593
595
|
};
|
|
594
596
|
}
|
|
595
597
|
|
|
@@ -32,6 +32,7 @@ import {
|
|
|
32
32
|
FR_REHEAT_FRACTION,
|
|
33
33
|
FR_START_TEMPERATURE,
|
|
34
34
|
MAX_ITERATIONS_PER_STEP,
|
|
35
|
+
SETTLE_FLOOR_FRACTION,
|
|
35
36
|
TRACE_RECORD_BYTES,
|
|
36
37
|
UNIFORM_SLOT_BYTES,
|
|
37
38
|
} from "../constants.js";
|
|
@@ -84,6 +85,7 @@ import {
|
|
|
84
85
|
subset,
|
|
85
86
|
vector,
|
|
86
87
|
} from "./model-common.js";
|
|
88
|
+
import { recordExactRepulsion } from "./repulsion-exact.js";
|
|
87
89
|
import { type GridStage, RepulsionGrid, type RepulsionGridOverrides } from "./repulsion-grid.js";
|
|
88
90
|
|
|
89
91
|
// ============================================================ constants
|
|
@@ -587,6 +589,7 @@ export class FruchtermanReingoldModel implements ForceModel<FruchtermanReingoldO
|
|
|
587
589
|
hiEnd,
|
|
588
590
|
midEnd,
|
|
589
591
|
frK: resolved.k ?? 1 / Math.sqrt(n),
|
|
592
|
+
settleFloor: SETTLE_FLOOR_FRACTION.fruchtermanReingold * (resolved.k ?? 1 / Math.sqrt(n)),
|
|
590
593
|
temperature: adaptive ? FR_START_TEMPERATURE : this.temperatureAt(iteration, resolved),
|
|
591
594
|
};
|
|
592
595
|
}
|
|
@@ -640,7 +643,7 @@ export class FruchtermanReingoldModel implements ForceModel<FruchtermanReingoldO
|
|
|
640
643
|
if (stop < 2) {
|
|
641
644
|
return;
|
|
642
645
|
}
|
|
643
|
-
k3
|
|
646
|
+
recordExactRepulsion(k3, pass, k3Bound, bound.plan, bound.n, offset);
|
|
644
647
|
if (stop < STAGE_K5) {
|
|
645
648
|
return;
|
|
646
649
|
}
|
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
* override set is a distinct pipeline and the subgroup twin is selected by the device's features (spec 5.1, D16).
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
|
+
import { EXACT_TILES_PER_PASS } from "../constants.js";
|
|
10
11
|
import { WebGpuGraphError } from "../errors.js";
|
|
11
12
|
import { type DispatchPlan, plan1d } from "../kernel/dispatch.js";
|
|
12
13
|
import { type BoundKernel, type Kernel } from "../kernel/kernel.js";
|
|
@@ -35,6 +36,33 @@ export interface RepulsionExactOverrides {
|
|
|
35
36
|
readonly GRAVITY_CENTER: 0 | 1;
|
|
36
37
|
}
|
|
37
38
|
|
|
39
|
+
/**
|
|
40
|
+
* Records K3 over `n` nodes as ceil(tiles / EXACT_TILES_PER_PASS) dispatches (issue #87: llvmpipe's per-invocation
|
|
41
|
+
* loop budget), pass p with p + 1 z slices of which only the last works (the kernel reads the pass from
|
|
42
|
+
* `num_workgroups.z`). One pass up to 32,768 nodes at WG 256. Every model's exact tier records K3 through here.
|
|
43
|
+
* ponytail: pass p also launches p idle slices, sum p over P passes; negligible against the O(n^2) pass work (31
|
|
44
|
+
* passes at 1M nodes); a per-pass uniform removes them if they ever show in a profile.
|
|
45
|
+
* @param kernel - the compiled `fa2-repulsion-exact`
|
|
46
|
+
* @param pass - the open compute pass
|
|
47
|
+
* @param bound - the kernel's bound groups
|
|
48
|
+
* @param plan - plan1d(n) of the kernel
|
|
49
|
+
* @param n - the node count
|
|
50
|
+
* @param paramsOffset - the dynamic offset of the Fa2Params slot
|
|
51
|
+
*/
|
|
52
|
+
export function recordExactRepulsion(
|
|
53
|
+
kernel: Kernel,
|
|
54
|
+
pass: GPUComputePassEncoder,
|
|
55
|
+
bound: BoundKernel,
|
|
56
|
+
plan: DispatchPlan,
|
|
57
|
+
n: number,
|
|
58
|
+
paramsOffset: number,
|
|
59
|
+
): void {
|
|
60
|
+
const passes = Math.max(1, Math.ceil(Math.ceil(n / kernel.workgroupSize) / EXACT_TILES_PER_PASS));
|
|
61
|
+
for (let z = 1; z <= passes; z++) {
|
|
62
|
+
kernel.dispatch(pass, bound, { ...plan, z }, [paramsOffset]);
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
|
|
38
66
|
/** K3 (tiled all-pairs repulsion + gravity + the swing / traction epilogue) followed by K4 (the one-workgroup speed finalize) (spec 7.6, 7.10). */
|
|
39
67
|
export class RepulsionExact {
|
|
40
68
|
/** The overrides both kernels were compiled with (a frozen copy of the argument of create()). */
|
|
@@ -148,7 +176,7 @@ export class RepulsionExact {
|
|
|
148
176
|
recordRepulsion(pass: GPUComputePassEncoder, n: number, paramsOffset: number): void {
|
|
149
177
|
const bound = this.bound(this.boundRepulsion, "recordRepulsion");
|
|
150
178
|
const plan = plan1d(n, this.repulsion.workgroupSize, this.caps);
|
|
151
|
-
this.repulsion
|
|
179
|
+
recordExactRepulsion(this.repulsion, pass, bound, plan, n, paramsOffset);
|
|
152
180
|
}
|
|
153
181
|
|
|
154
182
|
/**
|
|
@@ -216,7 +216,7 @@ export class RepulsionGrid {
|
|
|
216
216
|
|
|
217
217
|
/**
|
|
218
218
|
* The model-owned buffers of the grid tier (spec 7.3; PD-11): `cellKey` / `cellVal` / `sortedKey` / `sortedIdx`
|
|
219
|
-
* 4n, `cellHist` / `cellStart` 4 (cells + 2) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
|
|
219
|
+
* 4n, `cellHist` / `cellStart` 4 histWords (cells + 2^dim + 1) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
|
|
220
220
|
* indirect slot, `pyramid` 16 B per pyramid cell zeroed. `hubCounters` (16 B, zeroed) is the MODEL's on every
|
|
221
221
|
* tier (PD-14: K1 binds it on the exact tier too). n = 0 reports one node's worth of bytes (spec 3.6).
|
|
222
222
|
* @param n - the node count
|
|
@@ -25,6 +25,8 @@ import {
|
|
|
25
25
|
MAX_ITERATIONS_PER_STEP,
|
|
26
26
|
SE_DEFAULTS,
|
|
27
27
|
SE_SCALE_REFERENCE_NODES,
|
|
28
|
+
SETTLE_FLOOR_FRACTION,
|
|
29
|
+
SETTLE_FLOOR_REFERENCE_NODES,
|
|
28
30
|
TRACE_RECORD_BYTES,
|
|
29
31
|
UNIFORM_SLOT_BYTES,
|
|
30
32
|
} from "../constants.js";
|
|
@@ -77,6 +79,7 @@ import {
|
|
|
77
79
|
subset,
|
|
78
80
|
vector,
|
|
79
81
|
} from "./model-common.js";
|
|
82
|
+
import { recordExactRepulsion } from "./repulsion-exact.js";
|
|
80
83
|
import { type GridStage, RepulsionGrid, type RepulsionGridOverrides } from "./repulsion-grid.js";
|
|
81
84
|
|
|
82
85
|
// ============================================================ constants
|
|
@@ -555,6 +558,10 @@ export class SpringElectricalModel implements ForceModel<SpringElectricalOptions
|
|
|
555
558
|
frK: 0,
|
|
556
559
|
temperature: 0,
|
|
557
560
|
springLength: resolved.springLength,
|
|
561
|
+
settleFloor:
|
|
562
|
+
SETTLE_FLOOR_FRACTION.springElectrical *
|
|
563
|
+
resolved.springLength *
|
|
564
|
+
(SETTLE_FLOOR_REFERENCE_NODES / Math.max(n, 1)) ** 0.25,
|
|
558
565
|
springCoefficient: resolved.springCoefficient ?? SE_DEFAULTS.springCoefficient * springSizeFactor(n),
|
|
559
566
|
coulomb: resolved.gravity ?? SE_DEFAULTS.gravity * springSizeFactor(n),
|
|
560
567
|
dragCoefficient: resolved.dragCoefficient,
|
|
@@ -609,7 +616,7 @@ export class SpringElectricalModel implements ForceModel<SpringElectricalOptions
|
|
|
609
616
|
if (stop < 2) {
|
|
610
617
|
return;
|
|
611
618
|
}
|
|
612
|
-
k3
|
|
619
|
+
recordExactRepulsion(k3, pass, k3Bound, bound.plan, bound.n, offset);
|
|
613
620
|
if (stop < STAGE_K5) {
|
|
614
621
|
return;
|
|
615
622
|
}
|