@graphty/webgpu-graph-algorithms 0.6.4 → 0.6.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +38 -17
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-Dvq-Cc6v.js → context-DiSr6eiz.js} +45 -33
- package/dist/chunks/context-DiSr6eiz.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/algorithms/bfs.d.ts +11 -7
- package/dist/src/algorithms/bfs.d.ts.map +1 -1
- package/dist/src/algorithms/bfs.js +33 -10
- package/dist/src/algorithms/bfs.js.map +1 -1
- package/dist/src/algorithms/scope.d.ts +3 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +0 -2
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +4 -3
- package/dist/src/algorithms/sssp.d.ts.map +1 -1
- package/dist/src/algorithms/sssp.js +4 -3
- package/dist/src/algorithms/sssp.js.map +1 -1
- package/dist/src/constants.d.ts +33 -2
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +33 -2
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernel/dispatch.d.ts +2 -2
- package/dist/src/kernel/dispatch.d.ts.map +1 -1
- package/dist/src/kernel/kernel.d.ts +1 -1
- package/dist/src/kernel/kernel.js +2 -2
- package/dist/src/kernel/kernel.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +2 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernels.d.ts +15 -11
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +42 -21
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
- package/dist/src/layouts/forceatlas2.js +2 -1
- package/dist/src/layouts/forceatlas2.js.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.js +4 -2
- package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
- package/dist/src/layouts/repulsion-exact.d.ts +16 -0
- package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
- package/dist/src/layouts/repulsion-exact.js +21 -1
- package/dist/src/layouts/repulsion-exact.js.map +1 -1
- package/dist/src/layouts/repulsion-grid.d.ts +1 -1
- package/dist/src/layouts/repulsion-grid.js +1 -1
- package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
- package/dist/src/layouts/spring-electrical.js +6 -2
- package/dist/src/layouts/spring-electrical.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +3 -2
- package/dist/src/primitives/advance.d.ts.map +1 -1
- package/dist/src/primitives/advance.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +34 -38
- package/dist/src/primitives/frontier.d.ts.map +1 -1
- package/dist/src/primitives/frontier.js +24 -32
- package/dist/src/primitives/frontier.js.map +1 -1
- package/dist/src/primitives/grid-pyramid.d.ts +4 -4
- package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
- package/dist/src/primitives/grid-pyramid.js +4 -3
- package/dist/src/primitives/grid-pyramid.js.map +1 -1
- package/dist/src/primitives/grid.d.ts +13 -10
- package/dist/src/primitives/grid.d.ts.map +1 -1
- package/dist/src/primitives/grid.js +10 -7
- package/dist/src/primitives/grid.js.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +44 -49
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.js +62 -107
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
- package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
- package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
- package/dist/src/wgsl/histogram.wgsl.js +1 -1
- package/dist/webgpu-graph-algorithms.js +144 -119
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +3 -3
- package/src/algorithms/bfs.ts +34 -10
- package/src/algorithms/pagerank.ts +19 -5
- package/src/algorithms/power-iteration.ts +8 -2
- package/src/algorithms/scope.ts +3 -10
- package/src/algorithms/sssp.ts +4 -3
- package/src/constants.ts +35 -2
- package/src/kernel/dispatch.ts +2 -2
- package/src/kernel/kernel.ts +2 -2
- package/src/kernel/prelude.ts +2 -0
- package/src/kernels.ts +44 -21
- package/src/layouts/forceatlas2.ts +2 -0
- package/src/layouts/fruchterman-reingold.ts +4 -1
- package/src/layouts/repulsion-exact.ts +29 -1
- package/src/layouts/repulsion-grid.ts +1 -1
- package/src/layouts/spring-electrical.ts +8 -1
- package/src/memory/residency.ts +14 -4
- package/src/primitives/advance.ts +5 -4
- package/src/primitives/frontier.ts +42 -56
- package/src/primitives/grid-pyramid.ts +6 -5
- package/src/primitives/grid.ts +17 -12
- package/src/wgsl/advance-expand.wgsl.ts +4 -3
- package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
- package/src/wgsl/bfs-fused.wgsl.ts +6 -6
- package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
- package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
- package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
- package/src/wgsl/frontier-finalize.wgsl.ts +62 -107
- package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
- package/src/wgsl/grid-centroid.wgsl.ts +3 -2
- package/src/wgsl/grid-downsample.wgsl.ts +1 -1
- package/src/wgsl/grid-far-field.wgsl.ts +15 -8
- package/src/wgsl/grid-near-field.wgsl.ts +6 -2
- package/src/wgsl/histogram.wgsl.ts +1 -1
- package/dist/chunks/context-Dvq-Cc6v.js.map +0 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@graphty/webgpu-graph-algorithms",
|
|
3
|
-
"version": "0.6.
|
|
3
|
+
"version": "0.6.6",
|
|
4
4
|
"description": "WebGPU-accelerated graph algorithms and layouts over the @graphty/graph-format snapshot, for Node (Dawn) and browsers",
|
|
5
5
|
"author": "Adam Powers <apowers@ato.ms>",
|
|
6
6
|
"type": "module",
|
|
@@ -93,8 +93,8 @@
|
|
|
93
93
|
"vite": "^7.0.5",
|
|
94
94
|
"vitest": "^3.2.4",
|
|
95
95
|
"webgpu": "0.4.0",
|
|
96
|
-
"@graphty/algorithms": "^2.0.
|
|
97
|
-
"@graphty/layout": "^1.
|
|
96
|
+
"@graphty/algorithms": "^2.0.4",
|
|
97
|
+
"@graphty/layout": "^1.10.0"
|
|
98
98
|
},
|
|
99
99
|
"scripts": {
|
|
100
100
|
"build": "tsc -p tsconfig.build.json",
|
package/src/algorithms/bfs.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Breadth-first search on the device (design 8.4, 3.3 line 807, 9.7; P8-T6 / P8-T7 / P8-T8, the P8 plan's PD-5 /
|
|
3
3
|
* PD-6 / PD-7 / PD-14 / PD-18 / PD-21 / PD-23 / PD-24 / PD-26 / DEP-P8-B): the direction-optimizing traversal over
|
|
4
|
-
* the `Frontier` of P8-T4, every per-level choice made ON THE DEVICE. Every level is
|
|
4
|
+
* the `Frontier` of P8-T4, every per-level choice made ON THE DEVICE. Every level is nine recorded dispatches, all
|
|
5
5
|
* DIRECT (2026-09-25, docs/decisions/G8.md G8-F5: Dawn validates every `dispatchWorkgroupsIndirect` with an internal
|
|
6
6
|
* clamp pass costing about 0.4 ms of device time whether or not it dispatches anything, in Node and in Chromium
|
|
7
7
|
* alike, and that was 97 % of a traversal's wall time) -- `frontier-finalize` role 0 (the level boundary: rotates
|
|
@@ -13,14 +13,18 @@
|
|
|
13
13
|
* per frontier entry with the same claim inline and no edge queue; the fused level and the retry alike), then the
|
|
14
14
|
* bottom-up trio: a `fill` zeroing the frontier bitset, `bfs-bitset-build` setting the frontier's bits, and
|
|
15
15
|
* `bfs-bottom-up` sweeping the unvisited list over the REVERSE core (each unvisited vertex reads its in-neighbours
|
|
16
|
-
* until the first one in the bitset and claims itself)
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
* the
|
|
20
|
-
*
|
|
16
|
+
* until the first one in the bitset and claims itself), and last `bfs-next-degree` (the out-degree sum of whatever
|
|
17
|
+
* the level claimed, into the block's `nextDegreeSum` word: Beamer's m_f for the NEXT boundary, measured on the
|
|
18
|
+
* frontier that boundary decides for -- issue #391, which found the previous proxy, the degree of the frontier just
|
|
19
|
+
* expanded, one level stale and missing the switch at the level holding two thirds of the 1M / 10M R-MAT's arcs).
|
|
20
|
+
* Every kernel is a grid-stride dispatch of a host-planned grid (`planGridStride`) that loops to its count word and
|
|
21
|
+
* reads the path word first, so exactly one path does the level's work and the others cost one uniform load per
|
|
22
|
+
* workgroup. The unvisited set Beamer's test is against (PD-18) is rebuilt exactly once per submit, before the
|
|
23
|
+
* levels, by `bfs-unvisited-flags` plus `compact` over an iota queue, and maintained between rebuilds by subtraction
|
|
24
|
+
* inside the selector (whose JSDoc states the boundary rule; both words are exact at every boundary). The host records `MAX_LEVELS_PER_SUBMIT`
|
|
21
25
|
* levels into ONE command buffer, submits, and reads four bytes, the `done` word (PD-7): a road network has
|
|
22
26
|
* thousands of levels and a per-level `mapAsync` would be slower than the CPU. A level recorded past the end is a
|
|
23
|
-
* no-op (its boundary finds `done` set,
|
|
27
|
+
* no-op (its boundary finds `done` set, writes path 0 and moves no counter), so the loop needs no diameter and
|
|
24
28
|
* ends on `done`; a traversal has at most `n` levels, so more submits than that is E_VALIDATION, never a hang. The
|
|
25
29
|
* thresholds are uniform fields (`FUSED_FRONTIER_MAX`; alpha derived as `max(1, floor(arcCount / n))`, PD-21;
|
|
26
30
|
* `BEAMER_BETA`; `mode 1` pinning top-down -- each unless the tuning says otherwise), so a test forces any path
|
|
@@ -79,6 +83,9 @@ import { type AlgorithmScope, algorithmScope } from "./scope.js";
|
|
|
79
83
|
|
|
80
84
|
const ALGORITHM = "breadthFirstSearch";
|
|
81
85
|
|
|
86
|
+
/** Workgroups of the `bfs-next-degree` grid (issue #391): enough to sum n out-degrees by grid stride, few enough that a high-diameter traversal does not pay a full-width reduction per level. */
|
|
87
|
+
const NEXT_DEGREE_MAX_GROUPS = 128;
|
|
88
|
+
|
|
82
89
|
/**
|
|
83
90
|
* Params slots of the ring, COUNTED per run (P8-T12), because `UniformRing.reserve` wraps to slot 0 when a submit's
|
|
84
91
|
* records outrun the ring and silently overwrites a record the submit still reads; the ring's `overruns` counts
|
|
@@ -87,8 +94,9 @@ const ALGORITHM = "breadthFirstSearch";
|
|
|
87
94
|
* `bfs-contract`, the bits `fill`, `bfs-bitset-build`) and four re-issued once per arc window (`advance-expand`,
|
|
88
95
|
* `bfs-fused`, the fused retry, `bfs-bottom-up`). The driver as built writes fewer -- the contract and the bitset
|
|
89
96
|
* build share one window-free record, a window's two fused dispatches share one, the bits fill has one record per
|
|
90
|
-
* submit,
|
|
91
|
-
* `
|
|
97
|
+
* submit, `bfs-next-degree` has one window-free record of its own (issue #391), and a directed snapshot's reverse
|
|
98
|
+
* view is one window whatever the forward core's count -- so at most `4 + 3 x windows` per level, and the bound
|
|
99
|
+
* holds with room. The 16 covers the per-submit rebuild
|
|
92
100
|
* (`bfs-unvisited-flags` and `compact`, whose scan is at most 9 dispatches for any n below 2^32, so 10, plus the
|
|
93
101
|
* bits fill). The result batch flushes in its own submit, so the ring must hold IT too, and its size grows with `n`,
|
|
94
102
|
* not with the cadence: the iota `fill`, the radix sort's four passes of one record plus its scan's `2 x levels - 1`
|
|
@@ -339,6 +347,7 @@ export async function bfsWithTuning(
|
|
|
339
347
|
const bitset = await ctx.pipelines.kernel(kernelSpec("bfs-bitset-build"));
|
|
340
348
|
const bottomUp = await ctx.pipelines.kernel(kernelSpec("bfs-bottom-up", graphOverrides(reverse, null)));
|
|
341
349
|
const unvisited = await ctx.pipelines.kernel(kernelSpec("bfs-unvisited-flags"));
|
|
350
|
+
const nextDegree = await ctx.pipelines.kernel(kernelSpec("bfs-next-degree"));
|
|
342
351
|
const pred = await ctx.pipelines.kernel(kernelSpec("sssp-pred", { ...graphOverrides(core, null), MODE: 1 }));
|
|
343
352
|
const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
|
|
344
353
|
const sort = await prepareRadixSort(scope);
|
|
@@ -352,6 +361,11 @@ export async function bfsWithTuning(
|
|
|
352
361
|
// plan's GROUP count as its stride (planGridStride's cap applies to the groups)
|
|
353
362
|
const levelPlan = planGridStride(Math.max(n, frontier.edgeCapacity), wg, ctx.caps);
|
|
354
363
|
const sweepPlan = planGridStride(n, wg, ctx.caps);
|
|
364
|
+
// `bfs-next-degree` sums at most n words and is dispatched on EVERY level, so its fixed cost -- one
|
|
365
|
+
// workgroup reduction (six barriers) and one atomic per workgroup -- is paid 1,999 times on the
|
|
366
|
+
// 1000 x 1000 grid. A capped grid pays it 128 times a level instead of ceil(n / wg): on the card the
|
|
367
|
+
// grid row went from 436 to 356 ms and neither R-MAT row moved.
|
|
368
|
+
const degreePlan = planGridStride(n, wg, ctx.caps, NEXT_DEGREE_MAX_GROUPS);
|
|
355
369
|
const fusedPlan = planGridStride(n * wg, wg, ctx.caps);
|
|
356
370
|
const bitsPlan = plan1d(bitsWords, wg, ctx.caps);
|
|
357
371
|
const recordFill = (pass: GPUComputePassEncoder, dst: Binding, value: number, mode: 0 | 1): void => {
|
|
@@ -415,7 +429,7 @@ export async function bfsWithTuning(
|
|
|
415
429
|
const bitsParams = scope.params(FILL_PARAMS, { count: bitsWords, value: 0, mode: 0, pad0: 0 });
|
|
416
430
|
const boundBitsFill = fill.bind({ dst: frontierBits, P: bitsParams.binding });
|
|
417
431
|
for (let level = 0; level < levelsPerSubmit; level++) {
|
|
418
|
-
planner.recordFinalize(pass, 0, level, { ...fields, firstOfSubmit: Math.min(level,
|
|
432
|
+
planner.recordFinalize(pass, 0, level, { ...fields, firstOfSubmit: Math.min(level, 1) });
|
|
419
433
|
advance.record(pass, frontier);
|
|
420
434
|
planner.recordFinalize(pass, 1, level, fields);
|
|
421
435
|
// one window-free record serves the contract and the bitset build, which read no arc; the kernels that
|
|
@@ -489,6 +503,16 @@ export async function bfsWithTuning(
|
|
|
489
503
|
});
|
|
490
504
|
bottomUp.dispatch(pass, boundSweep, sweepPlan, [sweepParams.offset]);
|
|
491
505
|
}
|
|
506
|
+
// the next frontier's out-degree sum (issue #391): whichever path claimed, the vertices are in the output
|
|
507
|
+
// queue now, and the next boundary reads word 25 as Beamer's m_f for the frontier it is about to expand
|
|
508
|
+
const degreeParams = scope.params(FRONTIER_PARAMS, { wg, n, stride: degreePlan.stride ?? wg });
|
|
509
|
+
const boundDegree = nextDegree.bind({
|
|
510
|
+
frontier: frontier.output,
|
|
511
|
+
outDegree,
|
|
512
|
+
counters,
|
|
513
|
+
P: degreeParams.binding,
|
|
514
|
+
});
|
|
515
|
+
nextDegree.dispatch(pass, boundDegree, degreePlan, [degreeParams.offset]);
|
|
492
516
|
frontier.swap();
|
|
493
517
|
}
|
|
494
518
|
batch.endPass();
|
|
@@ -34,8 +34,8 @@ import { algorithmScope } from "./scope.js";
|
|
|
34
34
|
|
|
35
35
|
/** Iterations per submit (spec 8.2: k = 8). */
|
|
36
36
|
const PR_BATCH = 8;
|
|
37
|
-
/** Params slots of one batch: per iteration one PrParams (shared by pr-scale and pr-finalize) and one SpmvParams, plus the normaliser's RangeParams; a smaller ring wraps onto a slot the same batch still reads. */
|
|
38
|
-
const RING_SLOTS = 2 * PR_BATCH +
|
|
37
|
+
/** Params slots of one batch: per iteration one PrParams (shared by pr-scale and pr-finalize) and one SpmvParams, plus the normaliser's RangeParams and the last batch's convergence check; a smaller ring wraps onto a slot the same batch still reads. */
|
|
38
|
+
const RING_SLOTS = 2 * PR_BATCH + 2;
|
|
39
39
|
|
|
40
40
|
/**
|
|
41
41
|
* Validates `options.dest` for a score result of `n` elements.
|
|
@@ -175,6 +175,8 @@ async function run(
|
|
|
175
175
|
const groups = groupsOf(scalePlan);
|
|
176
176
|
const partialsBytes = PR_PARTIAL.byteLength * (1 + groups);
|
|
177
177
|
const partials = scope.scratch(partialsBytes, "partials");
|
|
178
|
+
// the header as the last pull left it, before the convergence check of the last batch overwrites danglingMass
|
|
179
|
+
const lastHeader = scope.scratch(PR_PARTIAL.byteLength, "lastHeader");
|
|
178
180
|
if (personalization !== null) {
|
|
179
181
|
uploaded = ctx.residency.array(personalization, `${algorithm}/personalization`);
|
|
180
182
|
}
|
|
@@ -203,17 +205,21 @@ async function run(
|
|
|
203
205
|
const xNormBinding = bindingOf(xNorm, bytes);
|
|
204
206
|
const outWeightSumBinding = bindingOf(outWeightSum, bytes);
|
|
205
207
|
const partialsBinding = bindingOf(partials, partialsBytes);
|
|
208
|
+
const lastHeaderBinding = bindingOf(lastHeader, PR_PARTIAL.byteLength);
|
|
206
209
|
const coefficients = { alpha, beta: 1 - alpha, uniformP: 1 / n };
|
|
207
210
|
let cur = 0;
|
|
208
211
|
let iterationsRun = 0;
|
|
209
212
|
for (;;) {
|
|
210
213
|
const k = Math.min(PR_BATCH, maxIterations - iterationsRun);
|
|
211
214
|
const batch = new CommandBatch(ctx, algorithm);
|
|
212
|
-
|
|
215
|
+
let pass = batch.pass("iterations");
|
|
213
216
|
if (iterationsRun === 0) {
|
|
214
217
|
normaliser.record(pass, weightedCore, outWeightSumBinding);
|
|
215
218
|
}
|
|
216
|
-
|
|
219
|
+
const last = iterationsRun + k === maxIterations;
|
|
220
|
+
// the scale + finalize of iteration i measures the error of iteration i - 1 (PD-9), so the last batch
|
|
221
|
+
// runs them once more, with no pull, to measure the error of iteration maxIterations itself
|
|
222
|
+
for (let i = 0; i < k + (last ? 1 : 0); i++) {
|
|
217
223
|
const params = scope.params(PR_PARAMS, {
|
|
218
224
|
n,
|
|
219
225
|
groups,
|
|
@@ -222,6 +228,10 @@ async function run(
|
|
|
222
228
|
convergeThreshold: tolerance * n,
|
|
223
229
|
});
|
|
224
230
|
const other = 1 - cur;
|
|
231
|
+
if (i === k) {
|
|
232
|
+
batch.copy(partialsBinding, lastHeaderBinding, PR_PARTIAL.byteLength);
|
|
233
|
+
pass = batch.pass("convergence");
|
|
234
|
+
}
|
|
225
235
|
const scaleBound = scale.bind({
|
|
226
236
|
rankIn: rank[cur],
|
|
227
237
|
rankPrev: rank[other],
|
|
@@ -233,6 +243,9 @@ async function run(
|
|
|
233
243
|
scale.dispatch(pass, scaleBound, scalePlan, [params.offset]);
|
|
234
244
|
const finalizeBound = finalize.bind({ partials: partialsBinding, P: params.binding });
|
|
235
245
|
finalize.dispatch(pass, finalizeBound, finalizePlan, [params.offset]);
|
|
246
|
+
if (i === k) {
|
|
247
|
+
break;
|
|
248
|
+
}
|
|
236
249
|
pull.record(
|
|
237
250
|
pass,
|
|
238
251
|
weightedRev,
|
|
@@ -248,6 +261,7 @@ async function run(
|
|
|
248
261
|
}
|
|
249
262
|
batch.endPass();
|
|
250
263
|
const headerRequest = batch.readback(partials, 0, PR_PARTIAL.byteLength);
|
|
264
|
+
const lastHeaderRequest = last ? batch.readback(lastHeader, 0, PR_PARTIAL.byteLength) : headerRequest;
|
|
251
265
|
const scoresRequest = batch.readback(rank[cur].buffer, 0, bytes);
|
|
252
266
|
scope.flush();
|
|
253
267
|
const submitted = batch.submit();
|
|
@@ -271,7 +285,7 @@ async function run(
|
|
|
271
285
|
scores,
|
|
272
286
|
iterations: converged ? firstConverged : iterationsRun,
|
|
273
287
|
converged,
|
|
274
|
-
danglingMass:
|
|
288
|
+
danglingMass: PR_PARTIAL.read(new DataView(back), lastHeaderRequest.offset).danglingMass as number,
|
|
275
289
|
precision: "f32",
|
|
276
290
|
};
|
|
277
291
|
}
|
|
@@ -33,7 +33,7 @@ import { algorithmScope } from "./scope.js";
|
|
|
33
33
|
|
|
34
34
|
/** Iterations per submit (spec 8.2: k = 8). */
|
|
35
35
|
const BATCH = 8;
|
|
36
|
-
/** Params slots of one batch: four blocks per iteration times the batch, plus a margin (the plan's `4 * 8 + 8`). */
|
|
36
|
+
/** Params slots of one batch: four blocks per iteration times the batch, plus a margin (the plan's `4 * 8 + 8`) that also holds the last batch's convergence check. */
|
|
37
37
|
const RING_SLOTS = 4 * BATCH + 8;
|
|
38
38
|
|
|
39
39
|
/**
|
|
@@ -210,7 +210,10 @@ export async function runPowerIteration(
|
|
|
210
210
|
const k = Math.min(BATCH, config.maxIterations - iterationsRun);
|
|
211
211
|
const batch = new CommandBatch(ctx, config.label);
|
|
212
212
|
const pass = batch.pass("iterations");
|
|
213
|
-
|
|
213
|
+
// the scale + finalize of iteration i measures the error of iteration i - 1 (PD-9), so the last batch
|
|
214
|
+
// runs them once more, with no apply and no pull, to measure the error of iteration maxIterations itself
|
|
215
|
+
const last = iterationsRun + k === config.maxIterations;
|
|
216
|
+
for (let i = 0; i < k + (last ? 1 : 0); i++) {
|
|
214
217
|
const iteration = iterationsRun + i + 1;
|
|
215
218
|
const params = scope.params(PR_PARAMS, {
|
|
216
219
|
n,
|
|
@@ -234,6 +237,9 @@ export async function runPowerIteration(
|
|
|
234
237
|
};
|
|
235
238
|
scaleNorm.dispatch(pass, scaleNorm.bind(scaleBindings), scalePlan, [params.offset]);
|
|
236
239
|
finalize.dispatch(pass, finalize.bind({ partials, P: params.binding }), finalizePlan, [params.offset]);
|
|
240
|
+
if (i === k) {
|
|
241
|
+
break;
|
|
242
|
+
}
|
|
237
243
|
if (scaleApply !== null) {
|
|
238
244
|
scaleApply.dispatch(pass, scaleApply.bind(scaleBindings), scalePlan, [params.offset]);
|
|
239
245
|
}
|
package/src/algorithms/scope.ts
CHANGED
|
@@ -8,12 +8,11 @@
|
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
10
|
import { type GpuContext } from "../context.js";
|
|
11
|
-
import { BufferUsage } from "../device/webgpu-constants.js";
|
|
12
11
|
import { UniformRing } from "../kernel/uniform-ring.js";
|
|
13
|
-
import { type
|
|
12
|
+
import { type ReduceScope } from "../primitives/reduce.js";
|
|
14
13
|
|
|
15
|
-
/** A
|
|
16
|
-
export interface AlgorithmScope extends
|
|
14
|
+
/** A ReduceScope over a context plus the two lifecycle calls an algorithm makes: flush() before submit, dispose() in its finally. (The `indirect()` lease it once added for the frontier's args buffer went with that buffer, 2026-09-25.) */
|
|
15
|
+
export interface AlgorithmScope extends ReduceScope {
|
|
17
16
|
/** queue.writeBuffer of the params slots written since the last flush (called before the batch is submitted). */
|
|
18
17
|
flush(): void;
|
|
19
18
|
/** Destroys the ring and releases every scratch buffer of the lease; idempotent. */
|
|
@@ -39,12 +38,6 @@ export function algorithmScope(ctx: GpuContext, label: string, slots: number): A
|
|
|
39
38
|
pool: ctx.pool,
|
|
40
39
|
workgroupSize: ctx.workgroupSize,
|
|
41
40
|
scratch: (byteLength, scratchLabel) => lease.storage(byteLength, `${label}/${scratchLabel}`),
|
|
42
|
-
indirect: (byteLength, indirectLabel) =>
|
|
43
|
-
lease.acquire(
|
|
44
|
-
byteLength,
|
|
45
|
-
BufferUsage.STORAGE | BufferUsage.INDIRECT | BufferUsage.COPY_DST | BufferUsage.COPY_SRC,
|
|
46
|
-
`${label}/${indirectLabel}`,
|
|
47
|
-
),
|
|
48
41
|
params(block, values) {
|
|
49
42
|
const slot = ring.reserve(1);
|
|
50
43
|
ring.write(slot, block, values);
|
package/src/algorithms/sssp.ts
CHANGED
|
@@ -11,9 +11,10 @@
|
|
|
11
11
|
* delta and dedupes the far half into `farIn` for a pass-through; both empty is `done`), `dedupe-claim` and
|
|
12
12
|
* `dedupe-filter` over each half (direct grid-stride dispatches; role 2 writes the chosen half's raw count into that
|
|
13
13
|
* dedupe's count word -- `edgeCount` for the near half, `edgeCountUnclamped` for the far one, two words SSSP borrows
|
|
14
|
-
* -- and 0 into the other's, so the half not chosen is a no-op), role 3 (restarts the raw half),
|
|
15
|
-
* twice (role 0 over `nearIn`, role 1 over `farIn
|
|
16
|
-
* the other role's dispatch a no-op).
|
|
14
|
+
* -- and 0 into the other's, so the half not chosen is a no-op), role 3 (restarts the raw half the round consumed),
|
|
15
|
+
* and `sssp-relax` twice (role 0 over `nearIn` sized by `frontierCount`, role 1 over `farIn` sized by `farCount`;
|
|
16
|
+
* the block's `path` word, 5 a near round and 6 a far one, makes the other role's dispatch a no-op). Nothing in a
|
|
17
|
+
* round is an indirect dispatch (2026-09-25). The far pile is re-bucketed by the
|
|
17
18
|
* relax kernel's pass-through, not by `compact` (PD-20): a far entry whose settled distance fell below the previous
|
|
18
19
|
* threshold was relaxed in the near band already and is dropped, the rest go back to near or far against the raised
|
|
19
20
|
* threshold. The near pile is ONE pile (no sub-partitions). The host records `MAX_LEVELS_PER_SUBMIT` rounds per
|
package/src/constants.ts
CHANGED
|
@@ -130,6 +130,13 @@ export const FA2_DISTANCE_FLOOR = 0.01;
|
|
|
130
130
|
export const FA2_DISTANCE_FLOOR_SQ = 0.0001;
|
|
131
131
|
/** The coincident threshold `d^2 < 1e-8` of spec 7.2. */
|
|
132
132
|
export const FA2_COINCIDENT_SQ = 1e-8;
|
|
133
|
+
/**
|
|
134
|
+
* The most WG-node tiles one K3 (`fa2-repulsion-exact`) dispatch sums per invocation (issue #87). llvmpipe runs at
|
|
135
|
+
* most 65,535 loop iterations per shader invocation, counted over every loop together, and then quietly breaks
|
|
136
|
+
* out of each loop; one tile costs WG + 2 of them (258 at WG 256), so a single pass lost every node past
|
|
137
|
+
* j = 65,027. 128 tiles is 33,024 iterations, half the budget; a larger exact run records ceil(tiles / 128) passes.
|
|
138
|
+
*/
|
|
139
|
+
export const EXACT_TILES_PER_PASS = 128;
|
|
133
140
|
/** Bits of Fa2Params.flags (contract 4.4). */
|
|
134
141
|
export const FA2_FLAG_FIRST = 1;
|
|
135
142
|
/** Fa2Params.flags bit: the Fruchterman-Reingold temperature is the adaptive one in the state block, not the uniform's (the `cooling: "adaptive"` option). */
|
|
@@ -202,6 +209,34 @@ export const SE_DEFAULTS: Readonly<{
|
|
|
202
209
|
iterationsPerStep: 1,
|
|
203
210
|
maxInFlight: 2,
|
|
204
211
|
});
|
|
212
|
+
/**
|
|
213
|
+
* The absolute settle floor of the shared settle rule (spec 7.17; issue #97): an iteration counts toward `settled` only
|
|
214
|
+
* when its mean displacement is at most `settleThreshold x rmsRadius` AND at most this fraction of the model's length
|
|
215
|
+
* unit -- `springLength` for spring-electrical (scaled by node count, see SETTLE_FLOOR_REFERENCE_NODES), `k` for
|
|
216
|
+
* Fruchterman-Reingold. The relative rule alone reported a spring
|
|
217
|
+
* layout settled while it still grew (6 % over 1,000 iterations at 10k nodes). ForceAtlas2 writes
|
|
218
|
+
* `SETTLE_FLOOR_UNBOUNDED` instead: it does not drift after settling, and its per-iteration jitter grows with n, so any
|
|
219
|
+
* fixed floor only delays or blocks its stop. The values are measured:
|
|
220
|
+
* design/decisions/2026-09-24-settle-rule-has-an-absolute-floor.md.
|
|
221
|
+
*/
|
|
222
|
+
export const SETTLE_FLOOR_FRACTION: Readonly<{ springElectrical: number; fruchtermanReingold: number }> = Object.freeze(
|
|
223
|
+
{
|
|
224
|
+
springElectrical: 3e-3,
|
|
225
|
+
fruchtermanReingold: 2e-3,
|
|
226
|
+
},
|
|
227
|
+
);
|
|
228
|
+
/**
|
|
229
|
+
* The node count at which the spring-electrical floor is exactly `SETTLE_FLOOR_FRACTION.springElectrical x
|
|
230
|
+
* springLength`; at `n` nodes it is that times `(SETTLE_FLOOR_REFERENCE_NODES / n)^(1/4)`. A spring layout's rms radius
|
|
231
|
+
* grows about as n^(1/4) in springLengths (3.4 at 150 nodes, 6.4 at 2,000, 10.4 at 10,000), so the relative half of the
|
|
232
|
+
* rule loosens with size while the floor tightens with it: on a small graph the floor sits above the relative threshold
|
|
233
|
+
* and the relative rule decides alone, as before issue #97; on a large one the floor binds, which is where the relative
|
|
234
|
+
* rule let an expanding layout stop. A fixed floor bound the 150-node story graph too, where the grid tier's jitter sits
|
|
235
|
+
* at the relative threshold, and nearly doubled its settle (427 -> 829 iterations on the RTX 4070 SUPER).
|
|
236
|
+
*/
|
|
237
|
+
export const SETTLE_FLOOR_REFERENCE_NODES = 2000;
|
|
238
|
+
/** The settle floor that never binds: the largest finite f32, 0x1.fffffep+127 (ForceAtlas2's `settleFloor`, issue #97). */
|
|
239
|
+
export const SETTLE_FLOOR_UNBOUNDED = 2 ** 128 - 2 ** 104;
|
|
205
240
|
/** The smallest finest grid side `G` (spec 7.7 geometry table: `clamp(nextPow2(2 n^(1/dim)), 8, gridMax)`; P4 PD-9). */
|
|
206
241
|
export const GRID_MIN_SIDE = 8;
|
|
207
242
|
/** The coarsest pyramid level's side (spec 7.7: "levels (coarsest 4 per axis)", `levels = log2(G / 4) + 1`). */
|
|
@@ -218,8 +253,6 @@ export const GRID_SORT_BITS = 24;
|
|
|
218
253
|
export const MAX_LEVELS_PER_SUBMIT = 32;
|
|
219
254
|
/** Design 8.4 and 6 row 8 (P8): a frontier at most this long runs the fused expand-contract kernel (Merrill's "fleeting iterations"); the default of the `fusedMax` uniform, which a test may set to 0 or `U32_MAX`. */
|
|
220
255
|
export const FUSED_FRONTIER_MAX = 4096;
|
|
221
|
-
/** P8-T4: the indirect dispatch slots `frontier-finalize` writes per level (expand, contract, fused, fill-bits, bitset-build, bottom-up, fused-retry); the args buffer is `MAX_LEVELS_PER_SUBMIT x FRONTIER_CANDIDATES x 16` bytes. */
|
|
222
|
-
export const FRONTIER_CANDIDATES = 7;
|
|
223
256
|
/** Design 8.4 (P8 PD-21): Beamer's beta -- switch back to top-down when `frontierCount * BEAMER_BETA < unvisitedCount` and the frontier is shrinking; alpha is derived from the graph, so it has no constant. */
|
|
224
257
|
export const BEAMER_BETA = 24;
|
|
225
258
|
/** Design 8.4 (P8 PD-22): the near-far split `delta = SSSP_DELTA_FACTOR * avgWeight / avgDegree`, computed on the host from the weight vector the run uses. */
|
package/src/kernel/dispatch.ts
CHANGED
|
@@ -9,11 +9,11 @@ import { MAX_WORKGROUPS_PER_DIM } from "../constants.js";
|
|
|
9
9
|
import { WebGpuGraphError } from "../errors.js";
|
|
10
10
|
import { type PlanCaps } from "../types/context.js";
|
|
11
11
|
|
|
12
|
-
/** A dispatch shape (spec 5.2). `stride` is the grid-stride step (null for plain 1D / 2D plans). */
|
|
12
|
+
/** A dispatch shape (spec 5.2). `stride` is the grid-stride step (null for plain 1D / 2D plans). `z` is 1 from every planner; only K3's pass split sets it (issue #87). */
|
|
13
13
|
export interface DispatchPlan {
|
|
14
14
|
readonly x: number;
|
|
15
15
|
readonly y: number;
|
|
16
|
-
readonly z:
|
|
16
|
+
readonly z: number;
|
|
17
17
|
readonly items: number;
|
|
18
18
|
readonly stride: number | null;
|
|
19
19
|
}
|
package/src/kernel/kernel.ts
CHANGED
|
@@ -245,7 +245,7 @@ export class Kernel {
|
|
|
245
245
|
}
|
|
246
246
|
|
|
247
247
|
/**
|
|
248
|
-
* setPipeline + setBindGroup for every group (dynamic offsets in dynamicGroups order) + dispatchWorkgroups(plan.x, plan.y,
|
|
248
|
+
* setPipeline + setBindGroup for every group (dynamic offsets in dynamicGroups order) + dispatchWorkgroups(plan.x, plan.y, plan.z); a plan with x === 0 records nothing (spec 5.6).
|
|
249
249
|
* PLAN DECISION: one dynamic offset per dynamic GROUP, replicated over every uniform binding of that group (every
|
|
250
250
|
* P1-P3 kernel has exactly one params uniform per group); absent offsets mean 0; a BoundKernel of another kernel
|
|
251
251
|
* or an offset list of the wrong length is E_INVALID_ARGUMENT.
|
|
@@ -268,7 +268,7 @@ export class Kernel {
|
|
|
268
268
|
return;
|
|
269
269
|
}
|
|
270
270
|
this.setUp(pass, bound, dynamicOffsets);
|
|
271
|
-
pass.dispatchWorkgroups(plan.x, plan.y,
|
|
271
|
+
pass.dispatchWorkgroups(plan.x, plan.y, plan.z);
|
|
272
272
|
}
|
|
273
273
|
|
|
274
274
|
/**
|
package/src/kernel/prelude.ts
CHANGED
|
@@ -12,6 +12,7 @@
|
|
|
12
12
|
import { INVALID_INDEX } from "@graphty/graph-format";
|
|
13
13
|
|
|
14
14
|
import {
|
|
15
|
+
EXACT_TILES_PER_PASS,
|
|
15
16
|
F32_INF_BITS,
|
|
16
17
|
FA2_COINCIDENT_SQ,
|
|
17
18
|
FA2_DISTANCE_FLOOR,
|
|
@@ -54,6 +55,7 @@ const INVALID_INDEX: u32 = ${INVALID_INDEX}u;
|
|
|
54
55
|
const U32_MAX: u32 = ${U32_MAX}u;
|
|
55
56
|
const F32_INF_BITS: u32 = ${F32_INF_BITS}u;
|
|
56
57
|
const MAX_WORKGROUPS_PER_DIM: u32 = ${MAX_WORKGROUPS_PER_DIM}u;
|
|
58
|
+
const EXACT_TILES_PER_PASS: u32 = ${EXACT_TILES_PER_PASS}u;
|
|
57
59
|
const FA2_DIST_FLOOR: f32 = ${wgslF32Literal(FA2_DISTANCE_FLOOR)};
|
|
58
60
|
const FA2_DIST_FLOOR_SQ: f32 = ${wgslF32Literal(FA2_DISTANCE_FLOOR_SQ)};
|
|
59
61
|
const FA2_COINCIDENT_SQ: f32 = ${wgslF32Literal(FA2_COINCIDENT_SQ)};
|
package/src/kernels.ts
CHANGED
|
@@ -28,6 +28,7 @@ import { bfsBitsetBuildWgsl } from "./wgsl/bfs-bitset-build.wgsl.js";
|
|
|
28
28
|
import { bfsBottomUpWgsl } from "./wgsl/bfs-bottom-up.wgsl.js";
|
|
29
29
|
import { bfsContractWgsl } from "./wgsl/bfs-contract.wgsl.js";
|
|
30
30
|
import { bfsFusedWgsl } from "./wgsl/bfs-fused.wgsl.js";
|
|
31
|
+
import { bfsNextDegreeWgsl } from "./wgsl/bfs-next-degree.wgsl.js";
|
|
31
32
|
import { bfsUnvisitedFlagsWgsl } from "./wgsl/bfs-unvisited-flags.wgsl.js";
|
|
32
33
|
import { closenessReduceWgsl } from "./wgsl/closeness-reduce.wgsl.js";
|
|
33
34
|
import { closenessSweepWgsl } from "./wgsl/closeness-sweep.wgsl.js";
|
|
@@ -111,6 +112,7 @@ export type KernelId =
|
|
|
111
112
|
| "bfs-bottom-up"
|
|
112
113
|
| "bfs-bitset-build"
|
|
113
114
|
| "bfs-unvisited-flags"
|
|
115
|
+
| "bfs-next-degree"
|
|
114
116
|
| "sssp-relax"
|
|
115
117
|
| "bf-relax"
|
|
116
118
|
| "closeness-sweep"
|
|
@@ -162,7 +164,7 @@ export const FILL_PARAMS: UniformBlock = UniformBlock.define("FillParams", [
|
|
|
162
164
|
["pad0", "u32"],
|
|
163
165
|
]);
|
|
164
166
|
|
|
165
|
-
/** `Fa2Params` (uniform,
|
|
167
|
+
/** `Fa2Params` (uniform, 144 B; spec 7.3): the per-iteration ForceAtlas2 parameters -- `n` @0, `dim` @4, `flags` @8 (bit 0 = FA2_FLAG_FIRST), `tierStart` @12, `tierEnd` @16, `iterationIndex` @20, `seed` @24, `nearMax` @28, `scalingRatio` @32, `gravity` @36, `jitterTolerance` @40, `scale` @44, `center` @48 (xyz, w 0), `settleThreshold` @64, `extentFactor` @68, `gridMax` @72, `levels` @76, `arcBase` @80 / `arcEnd` @84 (the bound arc window of K2, 0 and arcCount in the layout), `accumulate` @88 (1 combines into `force`: the windowed pattern), `hiEnd` @92 / `midEnd` @124 (the degreeOrder tier boundaries, PD-7; both 0 without a permutation); the P5 model fields (PD-3): `frK` @96 (the FR optimal distance), `temperature` @100 (the FR temperature of this iteration), `springLength` @104, `springCoefficient` @108, `coulomb` @112 (ngraph's `gravity`, negative repels), `dragCoefficient` @116, `timeStep` @120; `settleFloor` @128 (the absolute bound on the mean displacement of a settled iteration, in layout units: issue #97); 144 B. */
|
|
166
168
|
export const FA2_PARAMS: UniformBlock = UniformBlock.define("Fa2Params", [
|
|
167
169
|
["n", "u32"],
|
|
168
170
|
["dim", "u32"],
|
|
@@ -193,6 +195,7 @@ export const FA2_PARAMS: UniformBlock = UniformBlock.define("Fa2Params", [
|
|
|
193
195
|
["dragCoefficient", "f32"],
|
|
194
196
|
["timeStep", "f32"],
|
|
195
197
|
["midEnd", "u32"],
|
|
198
|
+
["settleFloor", "f32"],
|
|
196
199
|
]);
|
|
197
200
|
|
|
198
201
|
/** `Fa2State` (storage, padded to STATE_HEADER_BYTES = 256; spec 7.3): the device-resident controller state the finalize kernels write and the host reads back for stats -- `speed` @0, `speedEfficiency` @4, `swing` @8, `traction` @12, `centroid` @16, `rmsRadius` @32, `radius` @36, `meanDisplacement` @40, `iteration` @44, `min` @48, `max` @64, `gridMin` @80 (P4), `eps` @96 (P4), `settledCount` @100, `outsideGrid` @104 (P4), `maxCellOccupancy` @108 (P4), `temperature` @112 (FR, written by K1 under STATS_MODE 1), `kineticEnergy` @116 (the preset, K1 under STATS_MODE 2), `frEnergy` @120 / `frProgress` @124 (the FR adaptive cooling), `invCellSize` @128 (P4, PD-10: `1 / cellSize`, written by K1 beside `cellSize` in `gridMin.w`; G1 multiplies by it so every key is bitwise reproducible), `reserved0` @132 (f32), `reserved1` @136 (vec2f), `reserved2` .. `reserved8` @144 .. @240. */
|
|
@@ -386,8 +389,11 @@ export const COMPACT_PARAMS: UniformBlock = UniformBlock.define("CompactParams",
|
|
|
386
389
|
* submit), `arcsScanned` @64, `fusedLevels` @68, `twoPhaseLevels` @72, `bottomUpLevels` @76, `farCount` @80,
|
|
387
390
|
* `nextFarCount` @84, `thresholdBits` @88, `deltaBits` @92 (P8-T9), `path` @96 (what the level's kernels run, written
|
|
388
391
|
* by the selector: 0 nothing, 1 two-phase, 2 fused, 3 bottom-up, 4 the fused retry, 5 a near SSSP round, 6 a far
|
|
389
|
-
* one; every level kernel is a direct dispatch that reads it first -- G8-F5)
|
|
390
|
-
*
|
|
392
|
+
* one; every level kernel is a direct dispatch that reads it first -- G8-F5), `nextDegreeSum` @100 (issue #391: the
|
|
393
|
+
* out-degree sum of the vertices the level claimed, accumulated by `bfs-next-degree` at the end of every level and
|
|
394
|
+
* read, subtracted and zeroed by the next boundary -- Beamer's m_f measured on the frontier the boundary decides
|
|
395
|
+
* for, not on the one it has just expanded). The words nothing writes before P8-T8 / P8-T9 are declared now because
|
|
396
|
+
* the byte layout is what the single result copy decodes.
|
|
391
397
|
*/
|
|
392
398
|
export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
393
399
|
"FrontierCounters",
|
|
@@ -417,23 +423,24 @@ export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
|
417
423
|
["thresholdBits", "u32"],
|
|
418
424
|
["deltaBits", "u32"],
|
|
419
425
|
["path", "u32"],
|
|
426
|
+
["nextDegreeSum", "u32"],
|
|
420
427
|
],
|
|
421
428
|
{ layout: "storage" },
|
|
422
429
|
);
|
|
423
430
|
|
|
424
431
|
/**
|
|
425
432
|
* `FrontierParams` (uniform, 80 B; P8-T4): the params block every P8 kernel except the three compact / dedupe
|
|
426
|
-
* primitives and `bf-relax` binds -- `role` @0 (the finalize role), `
|
|
427
|
-
* `
|
|
428
|
-
* `
|
|
429
|
-
*
|
|
430
|
-
*
|
|
431
|
-
*
|
|
432
|
-
* `
|
|
433
|
+
* primitives and `bf-relax` binds -- `role` @0 (the finalize role), `wg` @4 (the consumers' workgroup size),
|
|
434
|
+
* `alpha` @8, `beta` @12 (Beamer's thresholds, P8-T8), `fusedMax` @16, `edgeCapacity` @20, `maxDepth` @24, `n` @28,
|
|
435
|
+
* `mode` @32 (BFS: 0 auto, 1 top-down only; `sssp-pred`: the PD-27 key rule), `cutoffBits` @36, `arcBase` @40,
|
|
436
|
+
* `arcEnd` @44 (the bound arc window), `predKind` @48 (0 arc, 1 node), `bitsBase` @52, `source` @56, `stride` @60
|
|
437
|
+
* (a grid-stride plan's stride), `firstOfSubmit` @64 (the boundary's index inside its submit, clamped to 1: both
|
|
438
|
+
* the unvisited-count and the unvisited-degree-sum subtraction run at >= 1, issue #391), `iteration` @68 (an
|
|
439
|
+
* `sssp-pred` hop pass, P8-T9), `pad1` @72, `pad2` @76. The `slotBase` field that once addressed the selector's indirect slots went with
|
|
440
|
+
* the slots (2026-09-25); `pad2` keeps the block an explicit 80 bytes, the way every block here is padded.
|
|
433
441
|
*/
|
|
434
442
|
export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams", [
|
|
435
443
|
["role", "u32"],
|
|
436
|
-
["slotBase", "u32"],
|
|
437
444
|
["wg", "u32"],
|
|
438
445
|
["alpha", "u32"],
|
|
439
446
|
["beta", "u32"],
|
|
@@ -452,6 +459,7 @@ export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams
|
|
|
452
459
|
["firstOfSubmit", "u32"],
|
|
453
460
|
["iteration", "u32"],
|
|
454
461
|
["pad1", "u32"],
|
|
462
|
+
["pad2", "u32"],
|
|
455
463
|
]);
|
|
456
464
|
|
|
457
465
|
/** `BfParams` (uniform, 16 B; P8-T10): `edgeCount` @0 (the logical edges of the `edgeList` view), `stride` @4 (the grid-stride plan's stride), `maxRetries` @8 (PD-12's compare-exchange bound), `cutoffBits` @12 (the f32 bit pattern of the CPU port's `cutoff`, `+Inf` when absent). */
|
|
@@ -938,7 +946,7 @@ const RADIX_SCATTER: KernelEntry = {
|
|
|
938
946
|
phase: "P4",
|
|
939
947
|
};
|
|
940
948
|
|
|
941
|
-
/** `grid-cell-key` (G1, spec 7.7; P4-T8, PD-10): the finest cell key of every node, `floor((p - gridMin) * invCellSize)` linearised, or the outside pseudo-cell `G^dim
|
|
949
|
+
/** `grid-cell-key` (G1, spec 7.7; P4-T8, PD-10): the finest cell key of every node, `floor((p - gridMin) * invCellSize)` linearised, or the outside pseudo-cell of its orthant `G^dim + orthant` (issue #90); `cellVal[i] = i`; 4 storage bindings (the state read-only: K1 writes it). */
|
|
942
950
|
const GRID_CELL_KEY: KernelEntry = {
|
|
943
951
|
id: "grid-cell-key",
|
|
944
952
|
body: gridCellKeyWgsl,
|
|
@@ -957,7 +965,7 @@ const GRID_CELL_KEY: KernelEntry = {
|
|
|
957
965
|
phase: "P4",
|
|
958
966
|
};
|
|
959
967
|
|
|
960
|
-
/** `grid-centroid` (G4, spec 7.7; P4-T9, PD-13): thread per finest cell (the pseudo-
|
|
968
|
+
/** `grid-centroid` (G4, spec 7.7; P4-T9, PD-13): thread per finest cell (the 2^dim orthant pseudo-cells included), the serial mass-weighted sum in sorted order into level 0, the occupancy max into `hubCounters[1]`, hub cells (> GRID_HUB_CELL) appended to `hubList`; 6 storage bindings. */
|
|
961
969
|
const GRID_CENTROID: KernelEntry = {
|
|
962
970
|
id: "grid-centroid",
|
|
963
971
|
body: gridCentroidWgsl,
|
|
@@ -1012,7 +1020,7 @@ const GRID_DOWNSAMPLE: KernelEntry = {
|
|
|
1012
1020
|
phase: "P4",
|
|
1013
1021
|
};
|
|
1014
1022
|
|
|
1015
|
-
/** `grid-far-field` (G6, spec 7.7; P4-T10, PD-16, DEP-P4-G): per node in sorted order, the coarsest level minus its 3x3 (3x3x3) and, per finer level, the parent's 3x3 refined minus the level's own 3x3, plus the pseudo-
|
|
1023
|
+
/** `grid-far-field` (G6, spec 7.7; P4-T10, PD-16, DEP-P4-G): per node in sorted order, the coarsest level minus its 3x3 (3x3x3) and, per finer level, the parent's 3x3 refined minus the level's own 3x3, plus the 2^dim orthant pseudo-cells; the FA2 term floored at 0.01 (issue #89); the loop bounds are `P.levels` / `P.gridMax`; LAW 0 FA2 / 1 FR / 2 coulomb per cell (P4-T13, PD-22); 5 storage bindings. */
|
|
1016
1024
|
const GRID_FAR_FIELD: KernelEntry = {
|
|
1017
1025
|
id: "grid-far-field",
|
|
1018
1026
|
body: gridFarFieldWgsl,
|
|
@@ -1117,16 +1125,12 @@ const DEDUPE_FILTER: KernelEntry = {
|
|
|
1117
1125
|
phase: "P8",
|
|
1118
1126
|
};
|
|
1119
1127
|
|
|
1120
|
-
/** `frontier-finalize` (design 5.4, 8.10 "BFS finalizeArgs"; P8-T4, PD-3): the one-lane level-boundary selector that rotates the counters block and writes the level's
|
|
1128
|
+
/** `frontier-finalize` (design 5.4, 8.10 "BFS finalizeArgs"; P8-T4, PD-3): the one-lane level-boundary selector that rotates the counters block and writes the level's `path` word (role 0), then clamps the edge count or switches the path to the fused retry (role 1); 1 storage binding (the block as `array<atomic<u32>>`). Since 2026-09-25 it writes no indirect slots: every level kernel is a direct dispatch gated by the path word. */
|
|
1121
1129
|
const FRONTIER_FINALIZE: KernelEntry = {
|
|
1122
1130
|
id: "frontier-finalize",
|
|
1123
1131
|
body: frontierFinalizeWgsl,
|
|
1124
1132
|
entryPoint: "frontier_finalize",
|
|
1125
|
-
bindings: [
|
|
1126
|
-
decl(1, 0, "counters", "storage", "array<atomic<u32>>"),
|
|
1127
|
-
decl(1, 1, "args", "storage", "array<u32>"),
|
|
1128
|
-
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1129
|
-
],
|
|
1133
|
+
bindings: [decl(1, 0, "counters", "storage", "array<atomic<u32>>"), decl(2, 0, "P", "uniform", "FrontierParams")],
|
|
1130
1134
|
overrideDecls: [],
|
|
1131
1135
|
uniforms: [FRONTIER_PARAMS],
|
|
1132
1136
|
needs: [],
|
|
@@ -1188,7 +1192,7 @@ const SSSP_PRED: KernelEntry = {
|
|
|
1188
1192
|
phase: "P8",
|
|
1189
1193
|
};
|
|
1190
1194
|
|
|
1191
|
-
/** `bfs-fused` (design 8.4 "the fused variant", 6 row 8 "the workgroup-per-row tier", 8.10 "BFS fused expand-contract"; P8-T7, PD-23): one level's expansion and contraction in one dispatch, one WORKGROUP per frontier entry, every lane stripping the entry's row with `bfs-contract`'s claim inline and no edge queue traffic;
|
|
1195
|
+
/** `bfs-fused` (design 8.4 "the fused variant", 6 row 8 "the workgroup-per-row tier", 8.10 "BFS fused expand-contract"; P8-T7, PD-23): one level's expansion and contraction in one dispatch, one WORKGROUP per frontier entry, every lane stripping the entry's row with `bfs-contract`'s claim inline and no edge queue traffic; a direct grid-stride dispatch that runs when the path word is 2 (a frontier below `P.fusedMax`) or 4 (the overflow retry, role 1's), sized from `frontierCount`; 8 storage bindings (the four graph slots, `frontierIn`, the counters block as `array<atomic<u32>>`, `depth` as `array<atomic<u32>>`, `frontierOut`) -- exactly at the budget, which is why no `parent` lives here (PD-24). */
|
|
1192
1196
|
const BFS_FUSED: KernelEntry = {
|
|
1193
1197
|
id: "bfs-fused",
|
|
1194
1198
|
body: bfsFusedWgsl,
|
|
@@ -1264,6 +1268,24 @@ const BFS_UNVISITED_FLAGS: KernelEntry = {
|
|
|
1264
1268
|
phase: "P8",
|
|
1265
1269
|
};
|
|
1266
1270
|
|
|
1271
|
+
/** `bfs-next-degree` (design 8.4; issue #391): Beamer's m_f measured exactly -- once per level, after the claim kernels, grid-striding over the output vertex queue and summing the `outDegree` view over the vertices the level claimed into word 25, one `atomicAdd` per workgroup; 3 storage bindings (`frontier` read-only, the `outDegree` VIEW, the counters block); `needs: ["subgroups"]` for the `wg_reduce_u32` call (a twin kernel). */
|
|
1272
|
+
const BFS_NEXT_DEGREE: KernelEntry = {
|
|
1273
|
+
id: "bfs-next-degree",
|
|
1274
|
+
body: bfsNextDegreeWgsl,
|
|
1275
|
+
entryPoint: "bfs_next_degree",
|
|
1276
|
+
bindings: [
|
|
1277
|
+
decl(1, 0, "frontier", "storage-ro", "array<u32>"),
|
|
1278
|
+
decl(1, 1, "outDegree", "storage-ro", "array<u32>"),
|
|
1279
|
+
decl(1, 2, "counters", "storage", "array<atomic<u32>>"),
|
|
1280
|
+
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1281
|
+
],
|
|
1282
|
+
overrideDecls: [],
|
|
1283
|
+
uniforms: [FRONTIER_PARAMS],
|
|
1284
|
+
needs: ["subgroups"],
|
|
1285
|
+
snippetSlots: [],
|
|
1286
|
+
phase: "P8",
|
|
1287
|
+
};
|
|
1288
|
+
|
|
1267
1289
|
/** `sssp-relax` (design 8.4 "Davidson's near-far", 8.10 "SSSP near-far relax"; P8-T9, PD-9 / PD-20 / DEP-P8-E): one round of the near-far loop -- role 0 relaxes the deduped near pile's whole rows with `atomicMin` on the f32 bit patterns of `dist` and appends each improved vertex to the raw near or far half of `queueOut` (the two halves of ONE buffer at word 0 and word `P.edgeCapacity`), role 1 re-buckets the deduped far pile; 8 storage bindings (the four graph slots with the run's weights bound in the weights slot, `dist` and the counters block as `array<atomic<u32>>`, `queueIn` read-only, `queueOut`) -- exactly at the budget, which is why no `pred` lives here (PD-11) and why the piles' counts, the threshold and the delta are words of the block. */
|
|
1268
1290
|
const SSSP_RELAX: KernelEntry = {
|
|
1269
1291
|
id: "sssp-relax",
|
|
@@ -1395,6 +1417,7 @@ const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze
|
|
|
1395
1417
|
"bfs-bottom-up": BFS_BOTTOM_UP,
|
|
1396
1418
|
"bfs-bitset-build": BFS_BITSET_BUILD,
|
|
1397
1419
|
"bfs-unvisited-flags": BFS_UNVISITED_FLAGS,
|
|
1420
|
+
"bfs-next-degree": BFS_NEXT_DEGREE,
|
|
1398
1421
|
"sssp-relax": SSSP_RELAX,
|
|
1399
1422
|
"bf-relax": BF_RELAX,
|
|
1400
1423
|
"closeness-sweep": CLOSENESS_SWEEP,
|
|
@@ -26,6 +26,7 @@ import {
|
|
|
26
26
|
GRID_EXTENT_FLOOR,
|
|
27
27
|
LAYOUT_TUNING_DEFAULTS,
|
|
28
28
|
MAX_ITERATIONS_PER_STEP,
|
|
29
|
+
SETTLE_FLOOR_UNBOUNDED,
|
|
29
30
|
TRACE_RECORD_BYTES,
|
|
30
31
|
UNIFORM_SLOT_BYTES,
|
|
31
32
|
} from "../constants.js";
|
|
@@ -590,6 +591,7 @@ export class ForceAtlas2Model implements ForceModel<ForceAtlas2Options, ForceAtl
|
|
|
590
591
|
accumulate: 0,
|
|
591
592
|
hiEnd,
|
|
592
593
|
midEnd,
|
|
594
|
+
settleFloor: SETTLE_FLOOR_UNBOUNDED, // ForceAtlas2 does not drift after settling (issue #97)
|
|
593
595
|
};
|
|
594
596
|
}
|
|
595
597
|
|
|
@@ -32,6 +32,7 @@ import {
|
|
|
32
32
|
FR_REHEAT_FRACTION,
|
|
33
33
|
FR_START_TEMPERATURE,
|
|
34
34
|
MAX_ITERATIONS_PER_STEP,
|
|
35
|
+
SETTLE_FLOOR_FRACTION,
|
|
35
36
|
TRACE_RECORD_BYTES,
|
|
36
37
|
UNIFORM_SLOT_BYTES,
|
|
37
38
|
} from "../constants.js";
|
|
@@ -84,6 +85,7 @@ import {
|
|
|
84
85
|
subset,
|
|
85
86
|
vector,
|
|
86
87
|
} from "./model-common.js";
|
|
88
|
+
import { recordExactRepulsion } from "./repulsion-exact.js";
|
|
87
89
|
import { type GridStage, RepulsionGrid, type RepulsionGridOverrides } from "./repulsion-grid.js";
|
|
88
90
|
|
|
89
91
|
// ============================================================ constants
|
|
@@ -587,6 +589,7 @@ export class FruchtermanReingoldModel implements ForceModel<FruchtermanReingoldO
|
|
|
587
589
|
hiEnd,
|
|
588
590
|
midEnd,
|
|
589
591
|
frK: resolved.k ?? 1 / Math.sqrt(n),
|
|
592
|
+
settleFloor: SETTLE_FLOOR_FRACTION.fruchtermanReingold * (resolved.k ?? 1 / Math.sqrt(n)),
|
|
590
593
|
temperature: adaptive ? FR_START_TEMPERATURE : this.temperatureAt(iteration, resolved),
|
|
591
594
|
};
|
|
592
595
|
}
|
|
@@ -640,7 +643,7 @@ export class FruchtermanReingoldModel implements ForceModel<FruchtermanReingoldO
|
|
|
640
643
|
if (stop < 2) {
|
|
641
644
|
return;
|
|
642
645
|
}
|
|
643
|
-
k3
|
|
646
|
+
recordExactRepulsion(k3, pass, k3Bound, bound.plan, bound.n, offset);
|
|
644
647
|
if (stop < STAGE_K5) {
|
|
645
648
|
return;
|
|
646
649
|
}
|