@graphty/webgpu-graph-algorithms 0.6.4 → 0.6.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +38 -17
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-Dvq-Cc6v.js → context-DiSr6eiz.js} +45 -33
- package/dist/chunks/context-DiSr6eiz.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/algorithms/bfs.d.ts +11 -7
- package/dist/src/algorithms/bfs.d.ts.map +1 -1
- package/dist/src/algorithms/bfs.js +33 -10
- package/dist/src/algorithms/bfs.js.map +1 -1
- package/dist/src/algorithms/scope.d.ts +3 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +0 -2
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +4 -3
- package/dist/src/algorithms/sssp.d.ts.map +1 -1
- package/dist/src/algorithms/sssp.js +4 -3
- package/dist/src/algorithms/sssp.js.map +1 -1
- package/dist/src/constants.d.ts +33 -2
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +33 -2
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernel/dispatch.d.ts +2 -2
- package/dist/src/kernel/dispatch.d.ts.map +1 -1
- package/dist/src/kernel/kernel.d.ts +1 -1
- package/dist/src/kernel/kernel.js +2 -2
- package/dist/src/kernel/kernel.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +2 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernels.d.ts +15 -11
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +42 -21
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
- package/dist/src/layouts/forceatlas2.js +2 -1
- package/dist/src/layouts/forceatlas2.js.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.js +4 -2
- package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
- package/dist/src/layouts/repulsion-exact.d.ts +16 -0
- package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
- package/dist/src/layouts/repulsion-exact.js +21 -1
- package/dist/src/layouts/repulsion-exact.js.map +1 -1
- package/dist/src/layouts/repulsion-grid.d.ts +1 -1
- package/dist/src/layouts/repulsion-grid.js +1 -1
- package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
- package/dist/src/layouts/spring-electrical.js +6 -2
- package/dist/src/layouts/spring-electrical.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +3 -2
- package/dist/src/primitives/advance.d.ts.map +1 -1
- package/dist/src/primitives/advance.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +34 -38
- package/dist/src/primitives/frontier.d.ts.map +1 -1
- package/dist/src/primitives/frontier.js +24 -32
- package/dist/src/primitives/frontier.js.map +1 -1
- package/dist/src/primitives/grid-pyramid.d.ts +4 -4
- package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
- package/dist/src/primitives/grid-pyramid.js +4 -3
- package/dist/src/primitives/grid-pyramid.js.map +1 -1
- package/dist/src/primitives/grid.d.ts +13 -10
- package/dist/src/primitives/grid.d.ts.map +1 -1
- package/dist/src/primitives/grid.js +10 -7
- package/dist/src/primitives/grid.js.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +44 -49
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.js +62 -107
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
- package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
- package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
- package/dist/src/wgsl/histogram.wgsl.js +1 -1
- package/dist/webgpu-graph-algorithms.js +144 -119
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +3 -3
- package/src/algorithms/bfs.ts +34 -10
- package/src/algorithms/pagerank.ts +19 -5
- package/src/algorithms/power-iteration.ts +8 -2
- package/src/algorithms/scope.ts +3 -10
- package/src/algorithms/sssp.ts +4 -3
- package/src/constants.ts +35 -2
- package/src/kernel/dispatch.ts +2 -2
- package/src/kernel/kernel.ts +2 -2
- package/src/kernel/prelude.ts +2 -0
- package/src/kernels.ts +44 -21
- package/src/layouts/forceatlas2.ts +2 -0
- package/src/layouts/fruchterman-reingold.ts +4 -1
- package/src/layouts/repulsion-exact.ts +29 -1
- package/src/layouts/repulsion-grid.ts +1 -1
- package/src/layouts/spring-electrical.ts +8 -1
- package/src/memory/residency.ts +14 -4
- package/src/primitives/advance.ts +5 -4
- package/src/primitives/frontier.ts +42 -56
- package/src/primitives/grid-pyramid.ts +6 -5
- package/src/primitives/grid.ts +17 -12
- package/src/wgsl/advance-expand.wgsl.ts +4 -3
- package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
- package/src/wgsl/bfs-fused.wgsl.ts +6 -6
- package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
- package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
- package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
- package/src/wgsl/frontier-finalize.wgsl.ts +62 -107
- package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
- package/src/wgsl/grid-centroid.wgsl.ts +3 -2
- package/src/wgsl/grid-downsample.wgsl.ts +1 -1
- package/src/wgsl/grid-far-field.wgsl.ts +15 -8
- package/src/wgsl/grid-near-field.wgsl.ts +6 -2
- package/src/wgsl/histogram.wgsl.ts +1 -1
- package/dist/chunks/context-Dvq-Cc6v.js.map +0 -1
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { W as WebGpuGraphError, U as UNIFORM_SLOT_BYTES, B as BufferUsage, M as MAX_WORKGROUPS_PER_DIM, a as WGSL_RESERVED_WORDS, S as STATE_HEADER_BYTES, d as deviceLostError, i as isWebGpuGraphError, b as U32_MAX$2, c as MAX_LEVELS_PER_SUBMIT,
|
|
2
|
-
import {
|
|
1
|
+
import { W as WebGpuGraphError, U as UNIFORM_SLOT_BYTES, B as BufferUsage, M as MAX_WORKGROUPS_PER_DIM, a as WGSL_RESERVED_WORDS, S as STATE_HEADER_BYTES, d as deviceLostError, i as isWebGpuGraphError, b as U32_MAX$2, c as MAX_LEVELS_PER_SUBMIT, R as RADIX_BINS, F as FUSED_FRONTIER_MAX, e as BEAMER_BETA, f as SSSP_DELTA_FACTOR, g as F32_INF_BITS, h as GRID_COARSEST_SIDE, j as GRID_MIN_SIDE, k as GRID_SORT_BITS, l as FA2_DEFAULTS, m as MAX_ITERATIONS_PER_STEP, n as MAX_1D_ITEMS, o as hasErrorCode, p as FA2_FLAG_FIRST, P as PARTIAL_BYTES, E as EXACT_TILES_PER_PASS, I as INDIRECT_ARGS_STRIDE, q as GRID_HUB_CELL, L as LAYOUT_TUNING_DEFAULTS, r as EXACT_MAX_NODES, s as SETTLE_FLOOR_UNBOUNDED, T as TRACE_RECORD_BYTES, t as GRID_BBOX_MARGIN, u as GRID_EXTENT_FLOOR, v as FR_ADAPTIVE_MAX_ITERATIONS, w as FR_START_TEMPERATURE, x as FA2_FLAG_ADAPTIVE, y as SETTLE_FLOOR_FRACTION, z as FR_REHEAT_FRACTION, A as FR_DEFAULTS, C as SE_DEFAULTS, D as SETTLE_FLOOR_REFERENCE_NODES, H as SE_SCALE_REFERENCE_NODES } from "./chunks/context-DiSr6eiz.js";
|
|
2
|
+
import { J, G, K, N, O, Q } from "./chunks/context-DiSr6eiz.js";
|
|
3
3
|
import { renumberPartition, INVALID_INDEX, makeMask, maskTest, expandEdges, fromEdgeArrays } from "@graphty/graph-format";
|
|
4
4
|
class UniformRing {
|
|
5
5
|
/**
|
|
@@ -217,7 +217,7 @@ function planGridStride(items, wg, caps, maxGroups) {
|
|
|
217
217
|
if (items === 0) {
|
|
218
218
|
return { x: 0, y: 1, z: 1, items, stride: null };
|
|
219
219
|
}
|
|
220
|
-
const cap = Math.min(caps.software ? 64 : 4096, perDimension(caps));
|
|
220
|
+
const cap = Math.min(maxGroups ?? (caps.software ? 64 : 4096), perDimension(caps));
|
|
221
221
|
const groups = Math.min(Math.ceil(items / wg), cap);
|
|
222
222
|
return { x: groups, y: 1, z: 1, items, stride: groups * wg };
|
|
223
223
|
}
|
|
@@ -582,7 +582,7 @@ fn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
|
|
|
582
582
|
if (lid.x == 0u) {
|
|
583
583
|
base = atomicAdd(&counters[8], aggregate); // edgeCount: ONE reservation per workgroup, not one per arc
|
|
584
584
|
atomicAdd(&counters[9], aggregate); // edgeCountUnclamped: the overflow detector (PD-23)
|
|
585
|
-
atomicAdd(&counters[2], aggregate); // frontierDegreeSum:
|
|
585
|
+
atomicAdd(&counters[2], aggregate); // frontierDegreeSum: what this level expanded (the inspect seam)
|
|
586
586
|
}
|
|
587
587
|
workgroupBarrier();
|
|
588
588
|
for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
|
|
@@ -773,7 +773,7 @@ fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
|
|
|
773
773
|
let d = select(0u, a1 - a0, a1 > a0);
|
|
774
774
|
wdeg = d;
|
|
775
775
|
wstart = a0;
|
|
776
|
-
atomicAdd(&counters[2], d); // frontierDegreeSum, so
|
|
776
|
+
atomicAdd(&counters[2], d); // frontierDegreeSum, so the inspect seam sees fused levels too
|
|
777
777
|
}
|
|
778
778
|
let deg = workgroupUniformLoad(&wdeg); // uniform: the loop below may hold barriers
|
|
779
779
|
let start = workgroupUniformLoad(&wstart);
|
|
@@ -805,6 +805,21 @@ fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
|
|
|
805
805
|
}
|
|
806
806
|
`
|
|
807
807
|
);
|
|
808
|
+
const bfsNextDegreeWgsl = (
|
|
809
|
+
/* wgsl */
|
|
810
|
+
`
|
|
811
|
+
@compute @workgroup_size(WG)
|
|
812
|
+
fn bfs_next_degree(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
813
|
+
let count = select(0u, atomicLoad(&counters[1]), atomicLoad(&counters[24]) != 0u); // nextFrontierCount, on a level that claimed (the path word)
|
|
814
|
+
var sum = 0u;
|
|
815
|
+
for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // no barrier inside: the trip count is per lane
|
|
816
|
+
sum = sum + outDegree[frontier[i]];
|
|
817
|
+
}
|
|
818
|
+
let total = wg_reduce_u32(sum, lid.x, 0u); // the prelude's workgroup sum; uniform: after the loop
|
|
819
|
+
if (lid.x == 0u) { atomicAdd(&counters[25], total); } // nextDegreeSum: ONE atomic per workgroup
|
|
820
|
+
}
|
|
821
|
+
`
|
|
822
|
+
);
|
|
808
823
|
const bfsUnvisitedFlagsWgsl = (
|
|
809
824
|
/* wgsl */
|
|
810
825
|
`
|
|
@@ -1265,14 +1280,23 @@ fn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: cent
|
|
|
1265
1280
|
}
|
|
1266
1281
|
|
|
1267
1282
|
@compute @workgroup_size(WG)
|
|
1268
|
-
fn repulsion(
|
|
1283
|
+
fn repulsion(
|
|
1284
|
+
@builtin(workgroup_id) wid: vec3<u32>,
|
|
1285
|
+
@builtin(local_invocation_id) lid: vec3<u32>,
|
|
1286
|
+
@builtin(num_workgroups) nwg: vec3<u32>,
|
|
1287
|
+
) {
|
|
1288
|
+
// issue #87: pass p of the tile range is dispatched with p + 1 z slices; only the last slice works, so the pass
|
|
1289
|
+
// index needs no uniform. Uniform: keyed on workgroup_id and num_workgroups only.
|
|
1290
|
+
if (wid.z + 1u < nwg.z) { return; }
|
|
1269
1291
|
let i = linear_id(wid, lid.x);
|
|
1270
1292
|
let valid = i < P.n;
|
|
1271
1293
|
var pi = vec4f(0.0);
|
|
1272
1294
|
if (valid) { pi = pos[i]; }
|
|
1273
1295
|
var f = vec3f(0.0);
|
|
1274
1296
|
let tiles = (P.n + WG - 1u) / WG;
|
|
1275
|
-
|
|
1297
|
+
let tileBegin = (nwg.z - 1u) * EXACT_TILES_PER_PASS;
|
|
1298
|
+
let tileEnd = min(tiles, tileBegin + EXACT_TILES_PER_PASS);
|
|
1299
|
+
for (var t = tileBegin; t < tileEnd; t = t + 1u) {
|
|
1276
1300
|
let j = t * WG + lid.x;
|
|
1277
1301
|
if (j < P.n) { tile[lid.x] = pos[j]; } else { tile[lid.x] = vec4f(0.0); } // guarded fill; mass 0 marks the pad
|
|
1278
1302
|
workgroupBarrier(); // uniform: every invocation reaches it
|
|
@@ -1295,6 +1319,10 @@ fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
|
|
|
1295
1319
|
}
|
|
1296
1320
|
workgroupBarrier();
|
|
1297
1321
|
}
|
|
1322
|
+
if (tileEnd < tiles) { // an earlier pass: its partial sum only (uniform: P.n, nwg)
|
|
1323
|
+
if (valid) { store_force(i, load_force(i) + f); }
|
|
1324
|
+
return;
|
|
1325
|
+
}
|
|
1298
1326
|
// epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it
|
|
1299
1327
|
var sw = 0.0;
|
|
1300
1328
|
var tr = 0.0;
|
|
@@ -1400,7 +1428,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
1400
1428
|
S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)
|
|
1401
1429
|
let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)
|
|
1402
1430
|
S.meanDisplacement = meanDisp;
|
|
1403
|
-
S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);
|
|
1431
|
+
S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= min(P.settleThreshold * S.rmsRadius, P.settleFloor)); // relative AND absolute (issue #97)
|
|
1404
1432
|
}
|
|
1405
1433
|
S.iteration = S.iteration + 1u;
|
|
1406
1434
|
T[P.iterationIndex].meanDisplacement = S.meanDisplacement;
|
|
@@ -1418,7 +1446,9 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
1418
1446
|
S.invCellSize = 1.0 / cellSize;
|
|
1419
1447
|
S.eps = 0.25 * cellSize;
|
|
1420
1448
|
}
|
|
1421
|
-
|
|
1449
|
+
var outside = 0u; // the previous iteration's pseudo-cell counts, one per orthant (issue #90; 0 after load)
|
|
1450
|
+
for (var o = 0u; o < select(4u, 8u, P.dim == 3u); o = o + 1u) { outside = outside + cellHist[cells + o]; }
|
|
1451
|
+
S.outsideGrid = outside;
|
|
1422
1452
|
S.maxCellOccupancy = atomicLoad(&hubCounters[1]);
|
|
1423
1453
|
atomicStore(&hubCounters[0], 0u);
|
|
1424
1454
|
atomicStore(&hubCounters[1], 0u);
|
|
@@ -1475,49 +1505,30 @@ fn fill(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid
|
|
|
1475
1505
|
const frontierFinalizeWgsl = (
|
|
1476
1506
|
/* wgsl */
|
|
1477
1507
|
`
|
|
1478
|
-
fn write_slot_groups(slot: u32, groups: u32, count: u32) { // groups workgroups, split in 2D above the per-dim limit
|
|
1479
|
-
var x = groups;
|
|
1480
|
-
var y = 1u;
|
|
1481
|
-
if (groups > MAX_WORKGROUPS_PER_DIM) {
|
|
1482
|
-
x = MAX_WORKGROUPS_PER_DIM;
|
|
1483
|
-
y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;
|
|
1484
|
-
}
|
|
1485
|
-
let base = 4u * (P.slotBase + slot); // 16-byte slots: (x, y, 1, count)
|
|
1486
|
-
args[base] = x; args[base + 1u] = y; args[base + 2u] = 1u; args[base + 3u] = count;
|
|
1487
|
-
}
|
|
1488
|
-
fn write_slot(slot: u32, count: u32) { // one INVOCATION per entry: ceil(count / wg) workgroups
|
|
1489
|
-
let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap (indirect-finalize's rule)
|
|
1490
|
-
write_slot_groups(slot, groups, count);
|
|
1491
|
-
}
|
|
1492
|
-
fn zero_slot(slot: u32) {
|
|
1493
|
-
let base = 4u * (P.slotBase + slot);
|
|
1494
|
-
args[base] = 0u; args[base + 1u] = 0u; args[base + 2u] = 1u; args[base + 3u] = 0u;
|
|
1495
|
-
}
|
|
1496
|
-
|
|
1497
1508
|
@compute @workgroup_size(WG)
|
|
1498
1509
|
fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
1499
1510
|
if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
|
|
1500
1511
|
if (P.role == 0u) { // the level boundary
|
|
1501
1512
|
if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end
|
|
1502
|
-
|
|
1503
|
-
return;
|
|
1513
|
+
atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)
|
|
1514
|
+
return;
|
|
1504
1515
|
}
|
|
1505
1516
|
let finished = atomicLoad(&counters[0]);
|
|
1506
1517
|
let next = atomicLoad(&counters[1]);
|
|
1507
1518
|
let degSum = atomicLoad(&counters[2]);
|
|
1519
|
+
let nextDeg = atomicLoad(&counters[25]); // deg(F_b), summed by bfs-next-degree when F_b was claimed (issue #391)
|
|
1508
1520
|
atomicStore(&counters[3], finished); // prevFrontierCount
|
|
1509
1521
|
atomicStore(&counters[4], degSum); // prevDegreeSum
|
|
1510
1522
|
atomicStore(&counters[0], next); // the rotation
|
|
1511
1523
|
atomicStore(&counters[1], 0u);
|
|
1512
1524
|
atomicStore(&counters[2], 0u);
|
|
1525
|
+
atomicStore(&counters[25], 0u); // the next level's claims sum from 0
|
|
1513
1526
|
atomicStore(&counters[8], 0u); // edgeCount
|
|
1514
1527
|
atomicStore(&counters[9], 0u); // edgeCountUnclamped
|
|
1515
1528
|
atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount
|
|
1516
|
-
if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped
|
|
1517
|
-
atomicStore(&counters[5], atomicLoad(&counters[5]) - next);
|
|
1518
|
-
|
|
1519
|
-
if (P.firstOfSubmit >= 2u) {
|
|
1520
|
-
atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2
|
|
1529
|
+
if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped (P8-T8, PD-18): F_b was inside the submit's rebuilt sums iff b >= 1
|
|
1530
|
+
atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount, exact
|
|
1531
|
+
atomicStore(&counters[6], atomicLoad(&counters[6]) - nextDeg); // unvisitedDegreeSum, exact (issue #391: no longer one level stale)
|
|
1521
1532
|
}
|
|
1522
1533
|
let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0
|
|
1523
1534
|
atomicStore(&counters[11], level);
|
|
@@ -1527,46 +1538,36 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
1527
1538
|
if (P.mode == 1u) {
|
|
1528
1539
|
direction = 0u; // top-down only (the test seam)
|
|
1529
1540
|
} else if (direction == 0u) {
|
|
1530
|
-
if (
|
|
1541
|
+
if (nextDeg > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing, m_f the degree of the frontier about to be expanded
|
|
1531
1542
|
} else {
|
|
1532
1543
|
if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking
|
|
1533
1544
|
}
|
|
1534
1545
|
if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches
|
|
1535
1546
|
var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)
|
|
1536
1547
|
if (done) {
|
|
1537
|
-
|
|
1548
|
+
path = 0u;
|
|
1538
1549
|
} else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep
|
|
1539
|
-
zero_slot(0u); zero_slot(1u); zero_slot(2u); zero_slot(6u);
|
|
1540
|
-
write_slot(3u, (P.n + 31u) / 32u); write_slot(4u, next); write_slot(5u, atomicLoad(&counters[7]));
|
|
1541
1550
|
path = 3u;
|
|
1542
1551
|
atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);
|
|
1543
1552
|
} else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)
|
|
1544
|
-
|
|
1545
|
-
write_slot_groups(2u, next, next); // bfs-fused is one WORKGROUP per frontier entry
|
|
1546
|
-
path = 2u;
|
|
1553
|
+
path = 2u; // bfs-fused: one WORKGROUP per frontier entry
|
|
1547
1554
|
atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
|
|
1548
1555
|
} else {
|
|
1549
|
-
|
|
1550
|
-
write_slot(0u, next); // slots 1 and 6 are role 1's
|
|
1551
|
-
path = 1u;
|
|
1556
|
+
path = 1u; // advance-expand, then role 1 and bfs-contract
|
|
1552
1557
|
}
|
|
1553
1558
|
atomicStore(&counters[14], direction);
|
|
1554
1559
|
atomicStore(&counters[24], path);
|
|
1555
1560
|
} else if (P.role == 1u) { // the edge queue is filled
|
|
1556
|
-
if (
|
|
1557
|
-
zero_slot(1u); zero_slot(6u);
|
|
1561
|
+
if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count
|
|
1558
1562
|
return;
|
|
1559
1563
|
}
|
|
1560
1564
|
let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);
|
|
1561
1565
|
atomicStore(&counters[8], clamped);
|
|
1562
1566
|
if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry
|
|
1563
|
-
|
|
1564
|
-
zero_slot(1u); write_slot_groups(6u, entries, entries); // one workgroup per frontier entry, as slot 2
|
|
1565
|
-
atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry, bfs-contract nothing
|
|
1567
|
+
atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing
|
|
1566
1568
|
atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);
|
|
1567
1569
|
atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
|
|
1568
1570
|
} else {
|
|
1569
|
-
write_slot(1u, clamped); zero_slot(6u);
|
|
1570
1571
|
atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)
|
|
1571
1572
|
}
|
|
1572
1573
|
} else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes
|
|
@@ -1574,54 +1575,43 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
1574
1575
|
atomicStore(&counters[9], 0u);
|
|
1575
1576
|
atomicStore(&counters[24], 0u);
|
|
1576
1577
|
if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)
|
|
1577
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
1578
1578
|
return;
|
|
1579
1579
|
}
|
|
1580
1580
|
let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped
|
|
1581
1581
|
let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped
|
|
1582
1582
|
if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE
|
|
1583
1583
|
atomicStore(&counters[15], 2u);
|
|
1584
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
1585
1584
|
return;
|
|
1586
1585
|
}
|
|
1587
|
-
zero_slot(2u); zero_slot(5u); zero_slot(6u); // role 3 sizes the relax slots once the piles are deduped
|
|
1588
1586
|
if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn
|
|
1589
1587
|
atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter
|
|
1590
1588
|
atomicStore(&counters[14], 0u); // mode 0
|
|
1591
|
-
|
|
1592
|
-
atomicStore(&counters[8], nearRaw); // the near dedupe's count word
|
|
1589
|
+
atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)
|
|
1593
1590
|
atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing
|
|
1594
|
-
zero_slot(3u); zero_slot(4u);
|
|
1595
1591
|
atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)
|
|
1596
1592
|
} else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile
|
|
1597
1593
|
let threshold = bitcast<f32>(atomicLoad(&counters[22]));
|
|
1598
1594
|
let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)
|
|
1599
1595
|
if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED
|
|
1600
1596
|
atomicStore(&counters[15], 3u);
|
|
1601
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
1602
1597
|
return;
|
|
1603
1598
|
}
|
|
1604
1599
|
atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below
|
|
1605
1600
|
atomicStore(&counters[22], bitcast<u32>(raised));
|
|
1606
1601
|
atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter
|
|
1607
1602
|
atomicStore(&counters[14], 1u); // mode 1
|
|
1608
|
-
|
|
1609
|
-
write_slot(3u, farRaw); write_slot(4u, farRaw); // dedupe-claim, dedupe-filter over the far half
|
|
1610
|
-
atomicStore(&counters[9], farRaw); // the far dedupe's count word
|
|
1603
|
+
atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)
|
|
1611
1604
|
atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing
|
|
1612
1605
|
atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);
|
|
1613
1606
|
} else { // both piles empty: finished
|
|
1614
1607
|
atomicStore(&counters[15], 1u);
|
|
1615
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
1616
1608
|
}
|
|
1617
|
-
} else if (P.role == 3u) { // the piles are deduped:
|
|
1618
|
-
if (atomicLoad(&counters[15]) != 0u) { return; } // role 2
|
|
1609
|
+
} else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed
|
|
1610
|
+
if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round
|
|
1619
1611
|
if (atomicLoad(&counters[14]) == 0u) {
|
|
1620
|
-
|
|
1621
|
-
atomicStore(&counters[1], 0u); // the raw near half restarts
|
|
1612
|
+
atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)
|
|
1622
1613
|
} else {
|
|
1623
|
-
|
|
1624
|
-
atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far)
|
|
1614
|
+
atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)
|
|
1625
1615
|
}
|
|
1626
1616
|
}
|
|
1627
1617
|
}
|
|
@@ -1643,7 +1633,8 @@ fn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocatio
|
|
|
1643
1633
|
let g = i32(P.gridMax);
|
|
1644
1634
|
var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;
|
|
1645
1635
|
if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }
|
|
1646
|
-
var key = cells;
|
|
1636
|
+
var key = cells + select(0u, 1u, c.x >= g / 2) + select(0u, 2u, c.y >= g / 2); // an outside pseudo-cell: its orthant (issue #90)
|
|
1637
|
+
if (P.dim == 3u) { key = key + select(0u, 4u, c.z >= g / 2); }
|
|
1647
1638
|
if (inside) {
|
|
1648
1639
|
key = u32(c.x) + P.gridMax * u32(c.y);
|
|
1649
1640
|
if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }
|
|
@@ -1661,7 +1652,7 @@ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.
|
|
|
1661
1652
|
@compute @workgroup_size(WG)
|
|
1662
1653
|
fn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
1663
1654
|
let c = linear_id(wid, lid.x);
|
|
1664
|
-
if (c
|
|
1655
|
+
if (c >= grid_cells() + select(4u, 8u, P.dim == 3u)) { return; } // cells [0, cells + 2^dim): the pseudo-cells follow the real ones; no barrier follows
|
|
1665
1656
|
let start = cellStart[c];
|
|
1666
1657
|
let count = cellStart[c + 1u] - start;
|
|
1667
1658
|
atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration
|
|
@@ -1739,11 +1730,12 @@ fn store_force(i: u32, f: vec3f) {
|
|
|
1739
1730
|
}
|
|
1740
1731
|
fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
|
|
1741
1732
|
fn grid_side(level: u32) -> u32 { return P.gridMax >> level; }
|
|
1742
|
-
fn
|
|
1733
|
+
fn outside_cells() -> u32 { return select(4u, 8u, P.dim == 3u); } // one pseudo-cell per orthant (issue #90)
|
|
1734
|
+
fn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cells at cells ..)
|
|
1743
1735
|
var base = 0u;
|
|
1744
1736
|
for (var l = 0u; l < level; l = l + 1u) {
|
|
1745
1737
|
let s = grid_side(l);
|
|
1746
|
-
base = base + s * s * select(1u, s, P.dim == 3u) + select(0u,
|
|
1738
|
+
base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, outside_cells(), l == 0u);
|
|
1747
1739
|
}
|
|
1748
1740
|
return base;
|
|
1749
1741
|
}
|
|
@@ -1754,7 +1746,9 @@ fn cell_at(level: u32, cx: i32, cy: i32, cz: i32) -> u32 {
|
|
|
1754
1746
|
fn cell_force(pi: vec4f, q: vec4f) -> vec3f { // one far-field term, softened by state.eps (7.7)
|
|
1755
1747
|
if (q.w <= 0.0) { return vec3f(0.0); } // an empty cell
|
|
1756
1748
|
let d = pi.xyz - q.xyz / q.w; // to the mass-weighted centroid
|
|
1757
|
-
|
|
1749
|
+
var d2 = dot(d, d);
|
|
1750
|
+
if (LAW == 0u) { d2 = max(d2, FA2_DIST_FLOOR_SQ); } // FA2 alone floors d >= 0.01, as K3 and G7 do (issue #89); FR and coulomb are unfloored (7.20)
|
|
1751
|
+
d2 = d2 + S.eps * S.eps;
|
|
1758
1752
|
if (LAW == 1u) { return d * (P.frK * P.frK * q.w / d2); } // LAW 1 (FR, 7.20): k^2 / d per node, q.w nodes at the centroid
|
|
1759
1753
|
if (LAW == 2u) { return d * (-P.coulomb * pi.w * q.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb): -g m_i M_cell / d^2
|
|
1760
1754
|
return d * (P.scalingRatio * pi.w * q.w / d2); // LAW 0 (FA2): |F| = k m_i M_cell / d
|
|
@@ -1803,9 +1797,11 @@ fn grid_far_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
|
|
|
1803
1797
|
}
|
|
1804
1798
|
}
|
|
1805
1799
|
}
|
|
1806
|
-
|
|
1800
|
+
for (var o = 0u; o < outside_cells(); o = o + 1u) { // every outside pseudo-cell: one far-field term per orthant
|
|
1801
|
+
f = f + cell_force(pi, pyramid[grid_cells() + o]);
|
|
1802
|
+
}
|
|
1807
1803
|
} else {
|
|
1808
|
-
for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (
|
|
1804
|
+
for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (G7 sums them pair by pair)
|
|
1809
1805
|
for (var cy = 0; cy < ts; cy = cy + 1) {
|
|
1810
1806
|
for (var cx = 0; cx < ts; cx = cx + 1) {
|
|
1811
1807
|
f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);
|
|
@@ -1907,7 +1903,11 @@ fn grid_near_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocat
|
|
|
1907
1903
|
}
|
|
1908
1904
|
}
|
|
1909
1905
|
} else {
|
|
1910
|
-
|
|
1906
|
+
var own = grid_cells() + select(0u, 1u, c0.x >= g / 2) + select(0u, 2u, c0.y >= g / 2); // G1's orthant key
|
|
1907
|
+
if (P.dim == 3u) { own = own + select(0u, 4u, c0.z >= g / 2); }
|
|
1908
|
+
for (var o = grid_cells(); o < grid_cells() + select(4u, 8u, P.dim == 3u); o = o + 1u) { // an outside node: every outside pseudo-cell
|
|
1909
|
+
f = f + cell_sum(i, pi, o, o == own);
|
|
1910
|
+
}
|
|
1911
1911
|
}
|
|
1912
1912
|
}
|
|
1913
1913
|
// epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it (K3's text)
|
|
@@ -2629,7 +2629,8 @@ const FA2_PARAMS = UniformBlock.define("Fa2Params", [
|
|
|
2629
2629
|
["coulomb", "f32"],
|
|
2630
2630
|
["dragCoefficient", "f32"],
|
|
2631
2631
|
["timeStep", "f32"],
|
|
2632
|
-
["midEnd", "u32"]
|
|
2632
|
+
["midEnd", "u32"],
|
|
2633
|
+
["settleFloor", "f32"]
|
|
2633
2634
|
]);
|
|
2634
2635
|
const FA2_STATE = UniformBlock.define(
|
|
2635
2636
|
"Fa2State",
|
|
@@ -2803,13 +2804,13 @@ const FRONTIER_COUNTERS = UniformBlock.define(
|
|
|
2803
2804
|
["nextFarCount", "u32"],
|
|
2804
2805
|
["thresholdBits", "u32"],
|
|
2805
2806
|
["deltaBits", "u32"],
|
|
2806
|
-
["path", "u32"]
|
|
2807
|
+
["path", "u32"],
|
|
2808
|
+
["nextDegreeSum", "u32"]
|
|
2807
2809
|
],
|
|
2808
2810
|
{ layout: "storage" }
|
|
2809
2811
|
);
|
|
2810
2812
|
const FRONTIER_PARAMS = UniformBlock.define("FrontierParams", [
|
|
2811
2813
|
["role", "u32"],
|
|
2812
|
-
["slotBase", "u32"],
|
|
2813
2814
|
["wg", "u32"],
|
|
2814
2815
|
["alpha", "u32"],
|
|
2815
2816
|
["beta", "u32"],
|
|
@@ -2827,7 +2828,8 @@ const FRONTIER_PARAMS = UniformBlock.define("FrontierParams", [
|
|
|
2827
2828
|
["stride", "u32"],
|
|
2828
2829
|
["firstOfSubmit", "u32"],
|
|
2829
2830
|
["iteration", "u32"],
|
|
2830
|
-
["pad1", "u32"]
|
|
2831
|
+
["pad1", "u32"],
|
|
2832
|
+
["pad2", "u32"]
|
|
2831
2833
|
]);
|
|
2832
2834
|
const BF_PARAMS = UniformBlock.define("BfParams", [
|
|
2833
2835
|
["edgeCount", "u32"],
|
|
@@ -3406,11 +3408,7 @@ const FRONTIER_FINALIZE = {
|
|
|
3406
3408
|
id: "frontier-finalize",
|
|
3407
3409
|
body: frontierFinalizeWgsl,
|
|
3408
3410
|
entryPoint: "frontier_finalize",
|
|
3409
|
-
bindings: [
|
|
3410
|
-
decl(1, 0, "counters", "storage", "array<atomic<u32>>"),
|
|
3411
|
-
decl(1, 1, "args", "storage", "array<u32>"),
|
|
3412
|
-
decl(2, 0, "P", "uniform", "FrontierParams")
|
|
3413
|
-
],
|
|
3411
|
+
bindings: [decl(1, 0, "counters", "storage", "array<atomic<u32>>"), decl(2, 0, "P", "uniform", "FrontierParams")],
|
|
3414
3412
|
overrideDecls: [],
|
|
3415
3413
|
uniforms: [FRONTIER_PARAMS],
|
|
3416
3414
|
needs: [],
|
|
@@ -3533,6 +3531,22 @@ const BFS_UNVISITED_FLAGS = {
|
|
|
3533
3531
|
snippetSlots: [],
|
|
3534
3532
|
phase: "P8"
|
|
3535
3533
|
};
|
|
3534
|
+
const BFS_NEXT_DEGREE = {
|
|
3535
|
+
id: "bfs-next-degree",
|
|
3536
|
+
body: bfsNextDegreeWgsl,
|
|
3537
|
+
entryPoint: "bfs_next_degree",
|
|
3538
|
+
bindings: [
|
|
3539
|
+
decl(1, 0, "frontier", "storage-ro", "array<u32>"),
|
|
3540
|
+
decl(1, 1, "outDegree", "storage-ro", "array<u32>"),
|
|
3541
|
+
decl(1, 2, "counters", "storage", "array<atomic<u32>>"),
|
|
3542
|
+
decl(2, 0, "P", "uniform", "FrontierParams")
|
|
3543
|
+
],
|
|
3544
|
+
overrideDecls: [],
|
|
3545
|
+
uniforms: [FRONTIER_PARAMS],
|
|
3546
|
+
needs: ["subgroups"],
|
|
3547
|
+
snippetSlots: [],
|
|
3548
|
+
phase: "P8"
|
|
3549
|
+
};
|
|
3536
3550
|
const SSSP_RELAX = {
|
|
3537
3551
|
id: "sssp-relax",
|
|
3538
3552
|
body: ssspRelaxWgsl,
|
|
@@ -3644,6 +3658,7 @@ const REGISTRY = Object.freeze({
|
|
|
3644
3658
|
"bfs-bottom-up": BFS_BOTTOM_UP,
|
|
3645
3659
|
"bfs-bitset-build": BFS_BITSET_BUILD,
|
|
3646
3660
|
"bfs-unvisited-flags": BFS_UNVISITED_FLAGS,
|
|
3661
|
+
"bfs-next-degree": BFS_NEXT_DEGREE,
|
|
3647
3662
|
"sssp-relax": SSSP_RELAX,
|
|
3648
3663
|
"bf-relax": BF_RELAX,
|
|
3649
3664
|
"closeness-sweep": CLOSENESS_SWEEP,
|
|
@@ -4486,11 +4501,6 @@ function algorithmScope(ctx, label, slots) {
|
|
|
4486
4501
|
pool: ctx.pool,
|
|
4487
4502
|
workgroupSize: ctx.workgroupSize,
|
|
4488
4503
|
scratch: (byteLength, scratchLabel) => lease.storage(byteLength, `${label}/${scratchLabel}`),
|
|
4489
|
-
indirect: (byteLength, indirectLabel) => lease.acquire(
|
|
4490
|
-
byteLength,
|
|
4491
|
-
BufferUsage.STORAGE | BufferUsage.INDIRECT | BufferUsage.COPY_DST | BufferUsage.COPY_SRC,
|
|
4492
|
-
`${label}/${indirectLabel}`
|
|
4493
|
-
),
|
|
4494
4504
|
params(block, values) {
|
|
4495
4505
|
const slot = ring.reserve(1);
|
|
4496
4506
|
ring.write(slot, block, values);
|
|
@@ -5796,7 +5806,8 @@ const W = Object.freeze({
|
|
|
5796
5806
|
nextFarCount: 21,
|
|
5797
5807
|
thresholdBits: 22,
|
|
5798
5808
|
deltaBits: 23,
|
|
5799
|
-
path: 24
|
|
5809
|
+
path: 24,
|
|
5810
|
+
nextDegreeSum: 25
|
|
5800
5811
|
});
|
|
5801
5812
|
function definedWords(words) {
|
|
5802
5813
|
const out = {};
|
|
@@ -5822,16 +5833,14 @@ class Frontier {
|
|
|
5822
5833
|
* Wraps the leased buffers; use prepareFrontier().
|
|
5823
5834
|
* @param vertices - the two vertex queues
|
|
5824
5835
|
* @param counters - the counters block
|
|
5825
|
-
* @param args - the args buffer
|
|
5826
5836
|
* @param edgeQueue - the edge queue
|
|
5827
5837
|
* @param edgeCapacity - the edge queue's entry count
|
|
5828
5838
|
* @param n - the vertex count
|
|
5829
5839
|
*/
|
|
5830
|
-
constructor(vertices, counters,
|
|
5840
|
+
constructor(vertices, counters, edgeQueue, edgeCapacity, n) {
|
|
5831
5841
|
this.sideIndex = 0;
|
|
5832
5842
|
this.vertices = vertices;
|
|
5833
5843
|
this.counters = counters;
|
|
5834
|
-
this.args = args;
|
|
5835
5844
|
this.edgeQueue = edgeQueue;
|
|
5836
5845
|
this.edgeCapacity = edgeCapacity;
|
|
5837
5846
|
this.n = n;
|
|
@@ -5862,7 +5871,7 @@ class Frontier {
|
|
|
5862
5871
|
this.sideIndex = this.sideIndex === 0 ? 1 : 0;
|
|
5863
5872
|
}
|
|
5864
5873
|
/**
|
|
5865
|
-
* Seeds a traversal: one `queue.writeBuffer` of the whole
|
|
5874
|
+
* Seeds a traversal: one `queue.writeBuffer` of the whole 112-byte block (zero except the caller's words) and one of
|
|
5866
5875
|
* `vertices[0][0] = source`, both ordered before the submit that follows; the source is on side 0 afterwards.
|
|
5867
5876
|
* `frontierCount` is not a word to seed: the first boundary rotates word 1 into it (the BFS seed is
|
|
5868
5877
|
* `{ nextFrontierCount: 1, level: U32_MAX }`). A source outside `[0, n)`, an unknown word or a value that is not a
|
|
@@ -5899,7 +5908,6 @@ async function prepareFrontier(scope, n, arcCount, edgeCapacity) {
|
|
|
5899
5908
|
}
|
|
5900
5909
|
const kernel = await scope.pipelines.kernel(kernelSpec("frontier-finalize"));
|
|
5901
5910
|
const queueBytes = 4 * Math.max(1, n);
|
|
5902
|
-
const argsBytes = MAX_LEVELS_PER_SUBMIT * FRONTIER_CANDIDATES * INDIRECT_ARGS_STRIDE;
|
|
5903
5911
|
const vertices = [
|
|
5904
5912
|
{ buffer: scope.scratch(queueBytes, "frontier/vertices-0"), offset: 0, size: queueBytes, window: null },
|
|
5905
5913
|
{ buffer: scope.scratch(queueBytes, "frontier/vertices-1"), offset: 0, size: queueBytes, window: null }
|
|
@@ -5910,19 +5918,13 @@ async function prepareFrontier(scope, n, arcCount, edgeCapacity) {
|
|
|
5910
5918
|
size: FRONTIER_COUNTERS.byteLength,
|
|
5911
5919
|
window: null
|
|
5912
5920
|
};
|
|
5913
|
-
const args = {
|
|
5914
|
-
buffer: scope.indirect(argsBytes, "frontier/args"),
|
|
5915
|
-
offset: 0,
|
|
5916
|
-
size: argsBytes,
|
|
5917
|
-
window: null
|
|
5918
|
-
};
|
|
5919
5921
|
const edgeQueue = {
|
|
5920
5922
|
buffer: scope.scratch(4 * capacity, "frontier/edge-queue"),
|
|
5921
5923
|
offset: 0,
|
|
5922
5924
|
size: 4 * capacity,
|
|
5923
5925
|
window: null
|
|
5924
5926
|
};
|
|
5925
|
-
const frontier = new Frontier(vertices, counters,
|
|
5927
|
+
const frontier = new Frontier(vertices, counters, edgeQueue, capacity, n);
|
|
5926
5928
|
return new FrontierPlannerImpl(scope, kernel, frontier);
|
|
5927
5929
|
}
|
|
5928
5930
|
class FrontierPlannerImpl {
|
|
@@ -5963,12 +5965,11 @@ class FrontierPlannerImpl {
|
|
|
5963
5965
|
const params = scope.params(FRONTIER_PARAMS, {
|
|
5964
5966
|
...definedWords(fields),
|
|
5965
5967
|
role,
|
|
5966
|
-
slotBase: level * FRONTIER_CANDIDATES,
|
|
5967
5968
|
wg: scope.workgroupSize,
|
|
5968
5969
|
edgeCapacity: frontier.edgeCapacity,
|
|
5969
5970
|
n: frontier.n
|
|
5970
5971
|
});
|
|
5971
|
-
const bound = this.kernel.bind({ counters: frontier.counters,
|
|
5972
|
+
const bound = this.kernel.bind({ counters: frontier.counters, P: params.binding });
|
|
5972
5973
|
this.kernel.dispatch(pass, bound, plan1d(1, scope.workgroupSize, scope.caps), [params.offset]);
|
|
5973
5974
|
}
|
|
5974
5975
|
}
|
|
@@ -6141,6 +6142,7 @@ class RadixSortPlannerImpl {
|
|
|
6141
6142
|
}
|
|
6142
6143
|
}
|
|
6143
6144
|
const ALGORITHM$3 = "breadthFirstSearch";
|
|
6145
|
+
const NEXT_DEGREE_MAX_GROUPS = 128;
|
|
6144
6146
|
function bfsRingSlots(windows, levelsPerSubmit) {
|
|
6145
6147
|
return Math.max((5 + 4 * windows) * levelsPerSubmit + 16, RESULT_BATCH_SLOTS + windows);
|
|
6146
6148
|
}
|
|
@@ -6275,6 +6277,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
|
|
|
6275
6277
|
const bitset = await ctx.pipelines.kernel(kernelSpec("bfs-bitset-build"));
|
|
6276
6278
|
const bottomUp = await ctx.pipelines.kernel(kernelSpec("bfs-bottom-up", graphOverrides(reverse, null)));
|
|
6277
6279
|
const unvisited = await ctx.pipelines.kernel(kernelSpec("bfs-unvisited-flags"));
|
|
6280
|
+
const nextDegree = await ctx.pipelines.kernel(kernelSpec("bfs-next-degree"));
|
|
6278
6281
|
const pred = await ctx.pipelines.kernel(kernelSpec("sssp-pred", { ...graphOverrides(core, null), MODE: 1 }));
|
|
6279
6282
|
const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
|
|
6280
6283
|
const sort = await prepareRadixSort(scope);
|
|
@@ -6284,6 +6287,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
|
|
|
6284
6287
|
const fillPlan = plan1d(n, wg, ctx.caps);
|
|
6285
6288
|
const levelPlan = planGridStride(Math.max(n, frontier.edgeCapacity), wg, ctx.caps);
|
|
6286
6289
|
const sweepPlan = planGridStride(n, wg, ctx.caps);
|
|
6290
|
+
const degreePlan = planGridStride(n, wg, ctx.caps, NEXT_DEGREE_MAX_GROUPS);
|
|
6287
6291
|
const fusedPlan = planGridStride(n * wg, wg, ctx.caps);
|
|
6288
6292
|
const bitsPlan = plan1d(bitsWords, wg, ctx.caps);
|
|
6289
6293
|
const recordFill = (pass2, dst, value, mode) => {
|
|
@@ -6334,7 +6338,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
|
|
|
6334
6338
|
const bitsParams = scope.params(FILL_PARAMS, { count: bitsWords, value: 0, mode: 0, pad0: 0 });
|
|
6335
6339
|
const boundBitsFill = fill.bind({ dst: frontierBits, P: bitsParams.binding });
|
|
6336
6340
|
for (let level2 = 0; level2 < levelsPerSubmit; level2++) {
|
|
6337
|
-
planner.recordFinalize(pass2, 0, level2, { ...fields, firstOfSubmit: Math.min(level2,
|
|
6341
|
+
planner.recordFinalize(pass2, 0, level2, { ...fields, firstOfSubmit: Math.min(level2, 1) });
|
|
6338
6342
|
advance.record(pass2, frontier);
|
|
6339
6343
|
planner.recordFinalize(pass2, 1, level2, fields);
|
|
6340
6344
|
const params = scope.params(FRONTIER_PARAMS, {
|
|
@@ -6400,6 +6404,14 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
|
|
|
6400
6404
|
});
|
|
6401
6405
|
bottomUp.dispatch(pass2, boundSweep, sweepPlan, [sweepParams.offset]);
|
|
6402
6406
|
}
|
|
6407
|
+
const degreeParams = scope.params(FRONTIER_PARAMS, { wg, n, stride: degreePlan.stride ?? wg });
|
|
6408
|
+
const boundDegree = nextDegree.bind({
|
|
6409
|
+
frontier: frontier.output,
|
|
6410
|
+
outDegree,
|
|
6411
|
+
counters,
|
|
6412
|
+
P: degreeParams.binding
|
|
6413
|
+
});
|
|
6414
|
+
nextDegree.dispatch(pass2, boundDegree, degreePlan, [degreeParams.offset]);
|
|
6403
6415
|
frontier.swap();
|
|
6404
6416
|
}
|
|
6405
6417
|
batch.endPass();
|
|
@@ -7563,10 +7575,11 @@ function gridSpecFor(n, dim, tuning) {
|
|
|
7563
7575
|
levels++;
|
|
7564
7576
|
}
|
|
7565
7577
|
const cells = g ** dim;
|
|
7578
|
+
const outsideCells = 2 ** dim;
|
|
7566
7579
|
const levelOffsets = [0];
|
|
7567
7580
|
let s = g;
|
|
7568
7581
|
for (let level = 0; level + 1 < levels; level++) {
|
|
7569
|
-
levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ?
|
|
7582
|
+
levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? outsideCells : 0));
|
|
7570
7583
|
s /= 2;
|
|
7571
7584
|
}
|
|
7572
7585
|
return {
|
|
@@ -7574,7 +7587,8 @@ function gridSpecFor(n, dim, tuning) {
|
|
|
7574
7587
|
g,
|
|
7575
7588
|
levels,
|
|
7576
7589
|
cells,
|
|
7577
|
-
|
|
7590
|
+
outsideCells,
|
|
7591
|
+
histWords: cells + outsideCells + 1,
|
|
7578
7592
|
levelOffsets: Object.freeze(levelOffsets),
|
|
7579
7593
|
pyramidCells: levelOffsets[levels - 1] + GRID_COARSEST_SIDE ** dim,
|
|
7580
7594
|
deterministic: tuning.deterministic
|
|
@@ -9890,6 +9904,12 @@ function subset(merged, defaults) {
|
|
|
9890
9904
|
}
|
|
9891
9905
|
return out;
|
|
9892
9906
|
}
|
|
9907
|
+
function recordExactRepulsion(kernel, pass, bound, plan, n, paramsOffset) {
|
|
9908
|
+
const passes = Math.max(1, Math.ceil(Math.ceil(n / kernel.workgroupSize) / EXACT_TILES_PER_PASS));
|
|
9909
|
+
for (let z = 1; z <= passes; z++) {
|
|
9910
|
+
kernel.dispatch(pass, bound, { ...plan, z }, [paramsOffset]);
|
|
9911
|
+
}
|
|
9912
|
+
}
|
|
9893
9913
|
class RepulsionExact {
|
|
9894
9914
|
/**
|
|
9895
9915
|
* Holds the two compiled kernels; create() is the only caller.
|
|
@@ -9982,7 +10002,7 @@ class RepulsionExact {
|
|
|
9982
10002
|
recordRepulsion(pass, n, paramsOffset) {
|
|
9983
10003
|
const bound = this.bound(this.boundRepulsion, "recordRepulsion");
|
|
9984
10004
|
const plan = plan1d(n, this.repulsion.workgroupSize, this.caps);
|
|
9985
|
-
this.repulsion
|
|
10005
|
+
recordExactRepulsion(this.repulsion, pass, bound, plan, n, paramsOffset);
|
|
9986
10006
|
}
|
|
9987
10007
|
/**
|
|
9988
10008
|
* Records K4 only.
|
|
@@ -10115,7 +10135,8 @@ class GridPyramidPlannerImpl {
|
|
|
10115
10135
|
}
|
|
10116
10136
|
const { centroid, finalize, hub, downsample } = this.kernels;
|
|
10117
10137
|
const one = { x: 1, y: 1, z: 1, items: 1, stride: null };
|
|
10118
|
-
|
|
10138
|
+
const level0 = spec.cells + spec.outsideCells;
|
|
10139
|
+
centroid.dispatch(pass, bound.centroid, plan1d(level0, scope.workgroupSize, scope.caps), [paramsOffset]);
|
|
10119
10140
|
finalize.dispatch(pass, bound.finalize, one, [bound.finalizeOffset]);
|
|
10120
10141
|
hub.dispatchIndirect(pass, bound.hub, bound.hubArgs, 0, [paramsOffset]);
|
|
10121
10142
|
this.dispatches = 3;
|
|
@@ -10188,7 +10209,7 @@ class RepulsionGrid {
|
|
|
10188
10209
|
}
|
|
10189
10210
|
/**
|
|
10190
10211
|
* The model-owned buffers of the grid tier (spec 7.3; PD-11): `cellKey` / `cellVal` / `sortedKey` / `sortedIdx`
|
|
10191
|
-
* 4n, `cellHist` / `cellStart` 4 (cells + 2) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
|
|
10212
|
+
* 4n, `cellHist` / `cellStart` 4 histWords (cells + 2^dim + 1) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
|
|
10192
10213
|
* indirect slot, `pyramid` 16 B per pyramid cell zeroed. `hubCounters` (16 B, zeroed) is the MODEL's on every
|
|
10193
10214
|
* tier (PD-14: K1 binds it on the exact tier too). n = 0 reports one node's worth of bytes (spec 3.6).
|
|
10194
10215
|
* @param n - the node count
|
|
@@ -10784,7 +10805,9 @@ class ForceAtlas2Model {
|
|
|
10784
10805
|
arcEnd: arcCountOf(core),
|
|
10785
10806
|
accumulate: 0,
|
|
10786
10807
|
hiEnd,
|
|
10787
|
-
midEnd
|
|
10808
|
+
midEnd,
|
|
10809
|
+
settleFloor: SETTLE_FLOOR_UNBOUNDED
|
|
10810
|
+
// ForceAtlas2 does not drift after settling (issue #97)
|
|
10788
10811
|
};
|
|
10789
10812
|
}
|
|
10790
10813
|
/**
|
|
@@ -11483,6 +11506,7 @@ class FruchtermanReingoldModel {
|
|
|
11483
11506
|
hiEnd,
|
|
11484
11507
|
midEnd,
|
|
11485
11508
|
frK: resolved.k ?? 1 / Math.sqrt(n),
|
|
11509
|
+
settleFloor: SETTLE_FLOOR_FRACTION.fruchtermanReingold * (resolved.k ?? 1 / Math.sqrt(n)),
|
|
11486
11510
|
temperature: adaptive ? FR_START_TEMPERATURE : this.temperatureAt(iteration, resolved)
|
|
11487
11511
|
};
|
|
11488
11512
|
}
|
|
@@ -11535,7 +11559,7 @@ class FruchtermanReingoldModel {
|
|
|
11535
11559
|
if (stop < 2) {
|
|
11536
11560
|
return;
|
|
11537
11561
|
}
|
|
11538
|
-
k3
|
|
11562
|
+
recordExactRepulsion(k3, pass, k3Bound, bound.plan, bound.n, offset);
|
|
11539
11563
|
if (stop < STAGE_K5$1) {
|
|
11540
11564
|
return;
|
|
11541
11565
|
}
|
|
@@ -12114,6 +12138,7 @@ class SpringElectricalModel {
|
|
|
12114
12138
|
frK: 0,
|
|
12115
12139
|
temperature: 0,
|
|
12116
12140
|
springLength: resolved.springLength,
|
|
12141
|
+
settleFloor: SETTLE_FLOOR_FRACTION.springElectrical * resolved.springLength * (SETTLE_FLOOR_REFERENCE_NODES / Math.max(n, 1)) ** 0.25,
|
|
12117
12142
|
springCoefficient: resolved.springCoefficient ?? SE_DEFAULTS.springCoefficient * springSizeFactor(n),
|
|
12118
12143
|
coulomb: resolved.gravity ?? SE_DEFAULTS.gravity * springSizeFactor(n),
|
|
12119
12144
|
dragCoefficient: resolved.dragCoefficient,
|
|
@@ -12167,7 +12192,7 @@ class SpringElectricalModel {
|
|
|
12167
12192
|
if (stop < 2) {
|
|
12168
12193
|
return;
|
|
12169
12194
|
}
|
|
12170
|
-
k3
|
|
12195
|
+
recordExactRepulsion(k3, pass, k3Bound, bound.plan, bound.n, offset);
|
|
12171
12196
|
if (stop < STAGE_K5) {
|
|
12172
12197
|
return;
|
|
12173
12198
|
}
|
|
@@ -12653,7 +12678,7 @@ async function calibrateLayout(ctx, options) {
|
|
|
12653
12678
|
};
|
|
12654
12679
|
}
|
|
12655
12680
|
export {
|
|
12656
|
-
|
|
12681
|
+
J as ARC_WINDOW_ALIGN,
|
|
12657
12682
|
EXACT_MAX_NODES,
|
|
12658
12683
|
FA2_DEFAULTS,
|
|
12659
12684
|
FR_DEFAULTS,
|
|
@@ -12661,10 +12686,10 @@ export {
|
|
|
12661
12686
|
LAYOUT_TUNING_DEFAULTS,
|
|
12662
12687
|
MAX_1D_ITEMS,
|
|
12663
12688
|
MAX_WORKGROUPS_PER_DIM,
|
|
12664
|
-
|
|
12689
|
+
K as PASSTHROUGH_FORMAT_CODES,
|
|
12665
12690
|
SE_DEFAULTS,
|
|
12666
|
-
|
|
12667
|
-
|
|
12691
|
+
N as STORAGE_ALIGN,
|
|
12692
|
+
O as WORKGROUP_SIZE,
|
|
12668
12693
|
WebGpuGraphError,
|
|
12669
12694
|
bellmanFord,
|
|
12670
12695
|
breadthFirstSearch,
|
|
@@ -12679,7 +12704,7 @@ export {
|
|
|
12679
12704
|
eigenvectorCentrality,
|
|
12680
12705
|
hasErrorCode,
|
|
12681
12706
|
hits,
|
|
12682
|
-
|
|
12707
|
+
Q as isSoftwareAdapter,
|
|
12683
12708
|
isWebGpuGraphError,
|
|
12684
12709
|
katzCentrality,
|
|
12685
12710
|
pageRank,
|