@graphty/webgpu-graph-algorithms 0.6.4 → 0.6.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/README.md +38 -17
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-Dvq-Cc6v.js → context-DiSr6eiz.js} +45 -33
  4. package/dist/chunks/context-DiSr6eiz.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/algorithms/bfs.d.ts +11 -7
  7. package/dist/src/algorithms/bfs.d.ts.map +1 -1
  8. package/dist/src/algorithms/bfs.js +33 -10
  9. package/dist/src/algorithms/bfs.js.map +1 -1
  10. package/dist/src/algorithms/scope.d.ts +3 -3
  11. package/dist/src/algorithms/scope.d.ts.map +1 -1
  12. package/dist/src/algorithms/scope.js +0 -2
  13. package/dist/src/algorithms/scope.js.map +1 -1
  14. package/dist/src/algorithms/sssp.d.ts +4 -3
  15. package/dist/src/algorithms/sssp.d.ts.map +1 -1
  16. package/dist/src/algorithms/sssp.js +4 -3
  17. package/dist/src/algorithms/sssp.js.map +1 -1
  18. package/dist/src/constants.d.ts +33 -2
  19. package/dist/src/constants.d.ts.map +1 -1
  20. package/dist/src/constants.js +33 -2
  21. package/dist/src/constants.js.map +1 -1
  22. package/dist/src/kernel/dispatch.d.ts +2 -2
  23. package/dist/src/kernel/dispatch.d.ts.map +1 -1
  24. package/dist/src/kernel/kernel.d.ts +1 -1
  25. package/dist/src/kernel/kernel.js +2 -2
  26. package/dist/src/kernel/kernel.js.map +1 -1
  27. package/dist/src/kernel/prelude.d.ts.map +1 -1
  28. package/dist/src/kernel/prelude.js +2 -1
  29. package/dist/src/kernel/prelude.js.map +1 -1
  30. package/dist/src/kernels.d.ts +15 -11
  31. package/dist/src/kernels.d.ts.map +1 -1
  32. package/dist/src/kernels.js +42 -21
  33. package/dist/src/kernels.js.map +1 -1
  34. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  35. package/dist/src/layouts/forceatlas2.js +2 -1
  36. package/dist/src/layouts/forceatlas2.js.map +1 -1
  37. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  38. package/dist/src/layouts/fruchterman-reingold.js +4 -2
  39. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  40. package/dist/src/layouts/repulsion-exact.d.ts +16 -0
  41. package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
  42. package/dist/src/layouts/repulsion-exact.js +21 -1
  43. package/dist/src/layouts/repulsion-exact.js.map +1 -1
  44. package/dist/src/layouts/repulsion-grid.d.ts +1 -1
  45. package/dist/src/layouts/repulsion-grid.js +1 -1
  46. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  47. package/dist/src/layouts/spring-electrical.js +6 -2
  48. package/dist/src/layouts/spring-electrical.js.map +1 -1
  49. package/dist/src/primitives/advance.d.ts +3 -2
  50. package/dist/src/primitives/advance.d.ts.map +1 -1
  51. package/dist/src/primitives/advance.js.map +1 -1
  52. package/dist/src/primitives/frontier.d.ts +34 -38
  53. package/dist/src/primitives/frontier.d.ts.map +1 -1
  54. package/dist/src/primitives/frontier.js +24 -32
  55. package/dist/src/primitives/frontier.js.map +1 -1
  56. package/dist/src/primitives/grid-pyramid.d.ts +4 -4
  57. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  58. package/dist/src/primitives/grid-pyramid.js +4 -3
  59. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  60. package/dist/src/primitives/grid.d.ts +13 -10
  61. package/dist/src/primitives/grid.d.ts.map +1 -1
  62. package/dist/src/primitives/grid.js +10 -7
  63. package/dist/src/primitives/grid.js.map +1 -1
  64. package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
  65. package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
  66. package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
  67. package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
  68. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
  69. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
  70. package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
  71. package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
  72. package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
  73. package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
  74. package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
  75. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
  76. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
  77. package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
  78. package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
  79. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
  80. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
  81. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
  82. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
  83. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
  84. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
  85. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
  86. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
  87. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +44 -49
  88. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
  89. package/dist/src/wgsl/frontier-finalize.wgsl.js +62 -107
  90. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
  91. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
  92. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
  93. package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
  94. package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
  95. package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
  96. package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
  97. package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
  98. package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
  99. package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
  100. package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
  101. package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
  102. package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
  103. package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
  104. package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
  105. package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
  106. package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
  107. package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
  108. package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
  109. package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
  110. package/dist/src/wgsl/histogram.wgsl.js +1 -1
  111. package/dist/webgpu-graph-algorithms.js +144 -119
  112. package/dist/webgpu-graph-algorithms.js.map +1 -1
  113. package/package.json +3 -3
  114. package/src/algorithms/bfs.ts +34 -10
  115. package/src/algorithms/pagerank.ts +19 -5
  116. package/src/algorithms/power-iteration.ts +8 -2
  117. package/src/algorithms/scope.ts +3 -10
  118. package/src/algorithms/sssp.ts +4 -3
  119. package/src/constants.ts +35 -2
  120. package/src/kernel/dispatch.ts +2 -2
  121. package/src/kernel/kernel.ts +2 -2
  122. package/src/kernel/prelude.ts +2 -0
  123. package/src/kernels.ts +44 -21
  124. package/src/layouts/forceatlas2.ts +2 -0
  125. package/src/layouts/fruchterman-reingold.ts +4 -1
  126. package/src/layouts/repulsion-exact.ts +29 -1
  127. package/src/layouts/repulsion-grid.ts +1 -1
  128. package/src/layouts/spring-electrical.ts +8 -1
  129. package/src/memory/residency.ts +14 -4
  130. package/src/primitives/advance.ts +5 -4
  131. package/src/primitives/frontier.ts +42 -56
  132. package/src/primitives/grid-pyramid.ts +6 -5
  133. package/src/primitives/grid.ts +17 -12
  134. package/src/wgsl/advance-expand.wgsl.ts +4 -3
  135. package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
  136. package/src/wgsl/bfs-fused.wgsl.ts +6 -6
  137. package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
  138. package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
  139. package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
  140. package/src/wgsl/frontier-finalize.wgsl.ts +62 -107
  141. package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
  142. package/src/wgsl/grid-centroid.wgsl.ts +3 -2
  143. package/src/wgsl/grid-downsample.wgsl.ts +1 -1
  144. package/src/wgsl/grid-far-field.wgsl.ts +15 -8
  145. package/src/wgsl/grid-near-field.wgsl.ts +6 -2
  146. package/src/wgsl/histogram.wgsl.ts +1 -1
  147. package/dist/chunks/context-Dvq-Cc6v.js.map +0 -1
@@ -1,5 +1,5 @@
1
- import { W as WebGpuGraphError, U as UNIFORM_SLOT_BYTES, B as BufferUsage, M as MAX_WORKGROUPS_PER_DIM, a as WGSL_RESERVED_WORDS, S as STATE_HEADER_BYTES, d as deviceLostError, i as isWebGpuGraphError, b as U32_MAX$2, c as MAX_LEVELS_PER_SUBMIT, F as FRONTIER_CANDIDATES, I as INDIRECT_ARGS_STRIDE, R as RADIX_BINS, e as FUSED_FRONTIER_MAX, f as BEAMER_BETA, g as SSSP_DELTA_FACTOR, h as F32_INF_BITS, j as GRID_COARSEST_SIDE, k as GRID_MIN_SIDE, l as GRID_SORT_BITS, m as FA2_DEFAULTS, n as MAX_ITERATIONS_PER_STEP, o as MAX_1D_ITEMS, p as hasErrorCode, q as FA2_FLAG_FIRST, P as PARTIAL_BYTES, r as GRID_HUB_CELL, L as LAYOUT_TUNING_DEFAULTS, E as EXACT_MAX_NODES, T as TRACE_RECORD_BYTES, s as GRID_BBOX_MARGIN, t as GRID_EXTENT_FLOOR, u as FR_ADAPTIVE_MAX_ITERATIONS, v as FR_START_TEMPERATURE, w as FA2_FLAG_ADAPTIVE, x as FR_REHEAT_FRACTION, y as FR_DEFAULTS, z as SE_DEFAULTS, A as SE_SCALE_REFERENCE_NODES } from "./chunks/context-Dvq-Cc6v.js";
2
- import { C, G, D, H, J, K } from "./chunks/context-Dvq-Cc6v.js";
1
+ import { W as WebGpuGraphError, U as UNIFORM_SLOT_BYTES, B as BufferUsage, M as MAX_WORKGROUPS_PER_DIM, a as WGSL_RESERVED_WORDS, S as STATE_HEADER_BYTES, d as deviceLostError, i as isWebGpuGraphError, b as U32_MAX$2, c as MAX_LEVELS_PER_SUBMIT, R as RADIX_BINS, F as FUSED_FRONTIER_MAX, e as BEAMER_BETA, f as SSSP_DELTA_FACTOR, g as F32_INF_BITS, h as GRID_COARSEST_SIDE, j as GRID_MIN_SIDE, k as GRID_SORT_BITS, l as FA2_DEFAULTS, m as MAX_ITERATIONS_PER_STEP, n as MAX_1D_ITEMS, o as hasErrorCode, p as FA2_FLAG_FIRST, P as PARTIAL_BYTES, E as EXACT_TILES_PER_PASS, I as INDIRECT_ARGS_STRIDE, q as GRID_HUB_CELL, L as LAYOUT_TUNING_DEFAULTS, r as EXACT_MAX_NODES, s as SETTLE_FLOOR_UNBOUNDED, T as TRACE_RECORD_BYTES, t as GRID_BBOX_MARGIN, u as GRID_EXTENT_FLOOR, v as FR_ADAPTIVE_MAX_ITERATIONS, w as FR_START_TEMPERATURE, x as FA2_FLAG_ADAPTIVE, y as SETTLE_FLOOR_FRACTION, z as FR_REHEAT_FRACTION, A as FR_DEFAULTS, C as SE_DEFAULTS, D as SETTLE_FLOOR_REFERENCE_NODES, H as SE_SCALE_REFERENCE_NODES } from "./chunks/context-DiSr6eiz.js";
2
+ import { J, G, K, N, O, Q } from "./chunks/context-DiSr6eiz.js";
3
3
  import { renumberPartition, INVALID_INDEX, makeMask, maskTest, expandEdges, fromEdgeArrays } from "@graphty/graph-format";
4
4
  class UniformRing {
5
5
  /**
@@ -217,7 +217,7 @@ function planGridStride(items, wg, caps, maxGroups) {
217
217
  if (items === 0) {
218
218
  return { x: 0, y: 1, z: 1, items, stride: null };
219
219
  }
220
- const cap = Math.min(caps.software ? 64 : 4096, perDimension(caps));
220
+ const cap = Math.min(maxGroups ?? (caps.software ? 64 : 4096), perDimension(caps));
221
221
  const groups = Math.min(Math.ceil(items / wg), cap);
222
222
  return { x: groups, y: 1, z: 1, items, stride: groups * wg };
223
223
  }
@@ -582,7 +582,7 @@ fn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
582
582
  if (lid.x == 0u) {
583
583
  base = atomicAdd(&counters[8], aggregate); // edgeCount: ONE reservation per workgroup, not one per arc
584
584
  atomicAdd(&counters[9], aggregate); // edgeCountUnclamped: the overflow detector (PD-23)
585
- atomicAdd(&counters[2], aggregate); // frontierDegreeSum: Beamer's m_f (P8-T8)
585
+ atomicAdd(&counters[2], aggregate); // frontierDegreeSum: what this level expanded (the inspect seam)
586
586
  }
587
587
  workgroupBarrier();
588
588
  for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
@@ -773,7 +773,7 @@ fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
773
773
  let d = select(0u, a1 - a0, a1 > a0);
774
774
  wdeg = d;
775
775
  wstart = a0;
776
- atomicAdd(&counters[2], d); // frontierDegreeSum, so Beamer's test (P8-T8) sees fused levels too
776
+ atomicAdd(&counters[2], d); // frontierDegreeSum, so the inspect seam sees fused levels too
777
777
  }
778
778
  let deg = workgroupUniformLoad(&wdeg); // uniform: the loop below may hold barriers
779
779
  let start = workgroupUniformLoad(&wstart);
@@ -805,6 +805,21 @@ fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
805
805
  }
806
806
  `
807
807
  );
808
+ const bfsNextDegreeWgsl = (
809
+ /* wgsl */
810
+ `
811
+ @compute @workgroup_size(WG)
812
+ fn bfs_next_degree(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
813
+ let count = select(0u, atomicLoad(&counters[1]), atomicLoad(&counters[24]) != 0u); // nextFrontierCount, on a level that claimed (the path word)
814
+ var sum = 0u;
815
+ for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // no barrier inside: the trip count is per lane
816
+ sum = sum + outDegree[frontier[i]];
817
+ }
818
+ let total = wg_reduce_u32(sum, lid.x, 0u); // the prelude's workgroup sum; uniform: after the loop
819
+ if (lid.x == 0u) { atomicAdd(&counters[25], total); } // nextDegreeSum: ONE atomic per workgroup
820
+ }
821
+ `
822
+ );
808
823
  const bfsUnvisitedFlagsWgsl = (
809
824
  /* wgsl */
810
825
  `
@@ -1265,14 +1280,23 @@ fn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: cent
1265
1280
  }
1266
1281
 
1267
1282
  @compute @workgroup_size(WG)
1268
- fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
1283
+ fn repulsion(
1284
+ @builtin(workgroup_id) wid: vec3<u32>,
1285
+ @builtin(local_invocation_id) lid: vec3<u32>,
1286
+ @builtin(num_workgroups) nwg: vec3<u32>,
1287
+ ) {
1288
+ // issue #87: pass p of the tile range is dispatched with p + 1 z slices; only the last slice works, so the pass
1289
+ // index needs no uniform. Uniform: keyed on workgroup_id and num_workgroups only.
1290
+ if (wid.z + 1u < nwg.z) { return; }
1269
1291
  let i = linear_id(wid, lid.x);
1270
1292
  let valid = i < P.n;
1271
1293
  var pi = vec4f(0.0);
1272
1294
  if (valid) { pi = pos[i]; }
1273
1295
  var f = vec3f(0.0);
1274
1296
  let tiles = (P.n + WG - 1u) / WG;
1275
- for (var t = 0u; t < tiles; t = t + 1u) {
1297
+ let tileBegin = (nwg.z - 1u) * EXACT_TILES_PER_PASS;
1298
+ let tileEnd = min(tiles, tileBegin + EXACT_TILES_PER_PASS);
1299
+ for (var t = tileBegin; t < tileEnd; t = t + 1u) {
1276
1300
  let j = t * WG + lid.x;
1277
1301
  if (j < P.n) { tile[lid.x] = pos[j]; } else { tile[lid.x] = vec4f(0.0); } // guarded fill; mass 0 marks the pad
1278
1302
  workgroupBarrier(); // uniform: every invocation reaches it
@@ -1295,6 +1319,10 @@ fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
1295
1319
  }
1296
1320
  workgroupBarrier();
1297
1321
  }
1322
+ if (tileEnd < tiles) { // an earlier pass: its partial sum only (uniform: P.n, nwg)
1323
+ if (valid) { store_force(i, load_force(i) + f); }
1324
+ return;
1325
+ }
1298
1326
  // epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it
1299
1327
  var sw = 0.0;
1300
1328
  var tr = 0.0;
@@ -1400,7 +1428,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
1400
1428
  S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)
1401
1429
  let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)
1402
1430
  S.meanDisplacement = meanDisp;
1403
- S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);
1431
+ S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= min(P.settleThreshold * S.rmsRadius, P.settleFloor)); // relative AND absolute (issue #97)
1404
1432
  }
1405
1433
  S.iteration = S.iteration + 1u;
1406
1434
  T[P.iterationIndex].meanDisplacement = S.meanDisplacement;
@@ -1418,7 +1446,9 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
1418
1446
  S.invCellSize = 1.0 / cellSize;
1419
1447
  S.eps = 0.25 * cellSize;
1420
1448
  }
1421
- S.outsideGrid = cellHist[cells]; // the previous iteration's pseudo-cell count (0 after load)
1449
+ var outside = 0u; // the previous iteration's pseudo-cell counts, one per orthant (issue #90; 0 after load)
1450
+ for (var o = 0u; o < select(4u, 8u, P.dim == 3u); o = o + 1u) { outside = outside + cellHist[cells + o]; }
1451
+ S.outsideGrid = outside;
1422
1452
  S.maxCellOccupancy = atomicLoad(&hubCounters[1]);
1423
1453
  atomicStore(&hubCounters[0], 0u);
1424
1454
  atomicStore(&hubCounters[1], 0u);
@@ -1475,49 +1505,30 @@ fn fill(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid
1475
1505
  const frontierFinalizeWgsl = (
1476
1506
  /* wgsl */
1477
1507
  `
1478
- fn write_slot_groups(slot: u32, groups: u32, count: u32) { // groups workgroups, split in 2D above the per-dim limit
1479
- var x = groups;
1480
- var y = 1u;
1481
- if (groups > MAX_WORKGROUPS_PER_DIM) {
1482
- x = MAX_WORKGROUPS_PER_DIM;
1483
- y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;
1484
- }
1485
- let base = 4u * (P.slotBase + slot); // 16-byte slots: (x, y, 1, count)
1486
- args[base] = x; args[base + 1u] = y; args[base + 2u] = 1u; args[base + 3u] = count;
1487
- }
1488
- fn write_slot(slot: u32, count: u32) { // one INVOCATION per entry: ceil(count / wg) workgroups
1489
- let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap (indirect-finalize's rule)
1490
- write_slot_groups(slot, groups, count);
1491
- }
1492
- fn zero_slot(slot: u32) {
1493
- let base = 4u * (P.slotBase + slot);
1494
- args[base] = 0u; args[base + 1u] = 0u; args[base + 2u] = 1u; args[base + 3u] = 0u;
1495
- }
1496
-
1497
1508
  @compute @workgroup_size(WG)
1498
1509
  fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
1499
1510
  if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
1500
1511
  if (P.role == 0u) { // the level boundary
1501
1512
  if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end
1502
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); } // this slotBase holds the previous submit's args
1503
- return; // no counter word moves (P8-T6's levels formula reads them)
1513
+ atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)
1514
+ return;
1504
1515
  }
1505
1516
  let finished = atomicLoad(&counters[0]);
1506
1517
  let next = atomicLoad(&counters[1]);
1507
1518
  let degSum = atomicLoad(&counters[2]);
1519
+ let nextDeg = atomicLoad(&counters[25]); // deg(F_b), summed by bfs-next-degree when F_b was claimed (issue #391)
1508
1520
  atomicStore(&counters[3], finished); // prevFrontierCount
1509
1521
  atomicStore(&counters[4], degSum); // prevDegreeSum
1510
1522
  atomicStore(&counters[0], next); // the rotation
1511
1523
  atomicStore(&counters[1], 0u);
1512
1524
  atomicStore(&counters[2], 0u);
1525
+ atomicStore(&counters[25], 0u); // the next level's claims sum from 0
1513
1526
  atomicStore(&counters[8], 0u); // edgeCount
1514
1527
  atomicStore(&counters[9], 0u); // edgeCountUnclamped
1515
1528
  atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount
1516
- if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped to 2 (P8-T8, PD-18)
1517
- atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount (exact): F_b was inside the submit's rebuilt sum iff b >= 1
1518
- }
1519
- if (P.firstOfSubmit >= 2u) {
1520
- atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2
1529
+ if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped (P8-T8, PD-18): F_b was inside the submit's rebuilt sums iff b >= 1
1530
+ atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount, exact
1531
+ atomicStore(&counters[6], atomicLoad(&counters[6]) - nextDeg); // unvisitedDegreeSum, exact (issue #391: no longer one level stale)
1521
1532
  }
1522
1533
  let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0
1523
1534
  atomicStore(&counters[11], level);
@@ -1527,46 +1538,36 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
1527
1538
  if (P.mode == 1u) {
1528
1539
  direction = 0u; // top-down only (the test seam)
1529
1540
  } else if (direction == 0u) {
1530
- if (degSum > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing
1541
+ if (nextDeg > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing, m_f the degree of the frontier about to be expanded
1531
1542
  } else {
1532
1543
  if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking
1533
1544
  }
1534
1545
  if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches
1535
1546
  var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)
1536
1547
  if (done) {
1537
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
1548
+ path = 0u;
1538
1549
  } else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep
1539
- zero_slot(0u); zero_slot(1u); zero_slot(2u); zero_slot(6u);
1540
- write_slot(3u, (P.n + 31u) / 32u); write_slot(4u, next); write_slot(5u, atomicLoad(&counters[7]));
1541
1550
  path = 3u;
1542
1551
  atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);
1543
1552
  } else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)
1544
- zero_slot(0u); zero_slot(1u); zero_slot(3u); zero_slot(4u); zero_slot(5u); zero_slot(6u);
1545
- write_slot_groups(2u, next, next); // bfs-fused is one WORKGROUP per frontier entry
1546
- path = 2u;
1553
+ path = 2u; // bfs-fused: one WORKGROUP per frontier entry
1547
1554
  atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
1548
1555
  } else {
1549
- zero_slot(2u); zero_slot(3u); zero_slot(4u); zero_slot(5u);
1550
- write_slot(0u, next); // slots 1 and 6 are role 1's
1551
- path = 1u;
1556
+ path = 1u; // advance-expand, then role 1 and bfs-contract
1552
1557
  }
1553
1558
  atomicStore(&counters[14], direction);
1554
1559
  atomicStore(&counters[24], path);
1555
1560
  } else if (P.role == 1u) { // the edge queue is filled
1556
- if (args[4u * P.slotBase] == 0u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to size, nothing to count
1557
- zero_slot(1u); zero_slot(6u);
1561
+ if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count
1558
1562
  return;
1559
1563
  }
1560
1564
  let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);
1561
1565
  atomicStore(&counters[8], clamped);
1562
1566
  if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry
1563
- let entries = atomicLoad(&counters[0]);
1564
- zero_slot(1u); write_slot_groups(6u, entries, entries); // one workgroup per frontier entry, as slot 2
1565
- atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry, bfs-contract nothing
1567
+ atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing
1566
1568
  atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);
1567
1569
  atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
1568
1570
  } else {
1569
- write_slot(1u, clamped); zero_slot(6u);
1570
1571
  atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)
1571
1572
  }
1572
1573
  } else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes
@@ -1574,54 +1575,43 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
1574
1575
  atomicStore(&counters[9], 0u);
1575
1576
  atomicStore(&counters[24], 0u);
1576
1577
  if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)
1577
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
1578
1578
  return;
1579
1579
  }
1580
1580
  let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped
1581
1581
  let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped
1582
1582
  if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE
1583
1583
  atomicStore(&counters[15], 2u);
1584
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
1585
1584
  return;
1586
1585
  }
1587
- zero_slot(2u); zero_slot(5u); zero_slot(6u); // role 3 sizes the relax slots once the piles are deduped
1588
1586
  if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn
1589
1587
  atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter
1590
1588
  atomicStore(&counters[14], 0u); // mode 0
1591
- write_slot(0u, nearRaw); write_slot(1u, nearRaw); // dedupe-claim, dedupe-filter over the near half
1592
- atomicStore(&counters[8], nearRaw); // the near dedupe's count word
1589
+ atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)
1593
1590
  atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing
1594
- zero_slot(3u); zero_slot(4u);
1595
1591
  atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)
1596
1592
  } else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile
1597
1593
  let threshold = bitcast<f32>(atomicLoad(&counters[22]));
1598
1594
  let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)
1599
1595
  if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED
1600
1596
  atomicStore(&counters[15], 3u);
1601
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
1602
1597
  return;
1603
1598
  }
1604
1599
  atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below
1605
1600
  atomicStore(&counters[22], bitcast<u32>(raised));
1606
1601
  atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter
1607
1602
  atomicStore(&counters[14], 1u); // mode 1
1608
- zero_slot(0u); zero_slot(1u);
1609
- write_slot(3u, farRaw); write_slot(4u, farRaw); // dedupe-claim, dedupe-filter over the far half
1610
- atomicStore(&counters[9], farRaw); // the far dedupe's count word
1603
+ atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)
1611
1604
  atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing
1612
1605
  atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);
1613
1606
  } else { // both piles empty: finished
1614
1607
  atomicStore(&counters[15], 1u);
1615
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
1616
1608
  }
1617
- } else if (P.role == 3u) { // the piles are deduped: size the relax
1618
- if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 zeroed every slot of the round
1609
+ } else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed
1610
+ if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round
1619
1611
  if (atomicLoad(&counters[14]) == 0u) {
1620
- write_slot(2u, atomicLoad(&counters[0])); // the near round over nearIn
1621
- atomicStore(&counters[1], 0u); // the raw near half restarts
1612
+ atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)
1622
1613
  } else {
1623
- write_slot(5u, atomicLoad(&counters[20])); // the pass-through over farIn
1624
- atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far)
1614
+ atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)
1625
1615
  }
1626
1616
  }
1627
1617
  }
@@ -1643,7 +1633,8 @@ fn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocatio
1643
1633
  let g = i32(P.gridMax);
1644
1634
  var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;
1645
1635
  if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }
1646
- var key = cells; // the outside pseudo-cell (7.7)
1636
+ var key = cells + select(0u, 1u, c.x >= g / 2) + select(0u, 2u, c.y >= g / 2); // an outside pseudo-cell: its orthant (issue #90)
1637
+ if (P.dim == 3u) { key = key + select(0u, 4u, c.z >= g / 2); }
1647
1638
  if (inside) {
1648
1639
  key = u32(c.x) + P.gridMax * u32(c.y);
1649
1640
  if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }
@@ -1661,7 +1652,7 @@ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.
1661
1652
  @compute @workgroup_size(WG)
1662
1653
  fn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
1663
1654
  let c = linear_id(wid, lid.x);
1664
- if (c > grid_cells()) { return; } // cells [0, cells]: the pseudo-cell is index cells; no barrier follows
1655
+ if (c >= grid_cells() + select(4u, 8u, P.dim == 3u)) { return; } // cells [0, cells + 2^dim): the pseudo-cells follow the real ones; no barrier follows
1665
1656
  let start = cellStart[c];
1666
1657
  let count = cellStart[c + 1u] - start;
1667
1658
  atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration
@@ -1739,11 +1730,12 @@ fn store_force(i: u32, f: vec3f) {
1739
1730
  }
1740
1731
  fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
1741
1732
  fn grid_side(level: u32) -> u32 { return P.gridMax >> level; }
1742
- fn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cell at index cells)
1733
+ fn outside_cells() -> u32 { return select(4u, 8u, P.dim == 3u); } // one pseudo-cell per orthant (issue #90)
1734
+ fn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cells at cells ..)
1743
1735
  var base = 0u;
1744
1736
  for (var l = 0u; l < level; l = l + 1u) {
1745
1737
  let s = grid_side(l);
1746
- base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, 1u, l == 0u);
1738
+ base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, outside_cells(), l == 0u);
1747
1739
  }
1748
1740
  return base;
1749
1741
  }
@@ -1754,7 +1746,9 @@ fn cell_at(level: u32, cx: i32, cy: i32, cz: i32) -> u32 {
1754
1746
  fn cell_force(pi: vec4f, q: vec4f) -> vec3f { // one far-field term, softened by state.eps (7.7)
1755
1747
  if (q.w <= 0.0) { return vec3f(0.0); } // an empty cell
1756
1748
  let d = pi.xyz - q.xyz / q.w; // to the mass-weighted centroid
1757
- let d2 = dot(d, d) + S.eps * S.eps;
1749
+ var d2 = dot(d, d);
1750
+ if (LAW == 0u) { d2 = max(d2, FA2_DIST_FLOOR_SQ); } // FA2 alone floors d >= 0.01, as K3 and G7 do (issue #89); FR and coulomb are unfloored (7.20)
1751
+ d2 = d2 + S.eps * S.eps;
1758
1752
  if (LAW == 1u) { return d * (P.frK * P.frK * q.w / d2); } // LAW 1 (FR, 7.20): k^2 / d per node, q.w nodes at the centroid
1759
1753
  if (LAW == 2u) { return d * (-P.coulomb * pi.w * q.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb): -g m_i M_cell / d^2
1760
1754
  return d * (P.scalingRatio * pi.w * q.w / d2); // LAW 0 (FA2): |F| = k m_i M_cell / d
@@ -1803,9 +1797,11 @@ fn grid_far_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
1803
1797
  }
1804
1798
  }
1805
1799
  }
1806
- f = f + cell_force(pi, pyramid[grid_cells()]); // the outside pseudo-cell as one far-field term
1800
+ for (var o = 0u; o < outside_cells(); o = o + 1u) { // every outside pseudo-cell: one far-field term per orthant
1801
+ f = f + cell_force(pi, pyramid[grid_cells() + o]);
1802
+ }
1807
1803
  } else {
1808
- for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (it would include itself)
1804
+ for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (G7 sums them pair by pair)
1809
1805
  for (var cy = 0; cy < ts; cy = cy + 1) {
1810
1806
  for (var cx = 0; cx < ts; cx = cx + 1) {
1811
1807
  f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);
@@ -1907,7 +1903,11 @@ fn grid_near_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocat
1907
1903
  }
1908
1904
  }
1909
1905
  } else {
1910
- f = cell_sum(i, pi, grid_cells(), true); // an outside node: the pseudo-cell alone
1906
+ var own = grid_cells() + select(0u, 1u, c0.x >= g / 2) + select(0u, 2u, c0.y >= g / 2); // G1's orthant key
1907
+ if (P.dim == 3u) { own = own + select(0u, 4u, c0.z >= g / 2); }
1908
+ for (var o = grid_cells(); o < grid_cells() + select(4u, 8u, P.dim == 3u); o = o + 1u) { // an outside node: every outside pseudo-cell
1909
+ f = f + cell_sum(i, pi, o, o == own);
1910
+ }
1911
1911
  }
1912
1912
  }
1913
1913
  // epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it (K3's text)
@@ -2629,7 +2629,8 @@ const FA2_PARAMS = UniformBlock.define("Fa2Params", [
2629
2629
  ["coulomb", "f32"],
2630
2630
  ["dragCoefficient", "f32"],
2631
2631
  ["timeStep", "f32"],
2632
- ["midEnd", "u32"]
2632
+ ["midEnd", "u32"],
2633
+ ["settleFloor", "f32"]
2633
2634
  ]);
2634
2635
  const FA2_STATE = UniformBlock.define(
2635
2636
  "Fa2State",
@@ -2803,13 +2804,13 @@ const FRONTIER_COUNTERS = UniformBlock.define(
2803
2804
  ["nextFarCount", "u32"],
2804
2805
  ["thresholdBits", "u32"],
2805
2806
  ["deltaBits", "u32"],
2806
- ["path", "u32"]
2807
+ ["path", "u32"],
2808
+ ["nextDegreeSum", "u32"]
2807
2809
  ],
2808
2810
  { layout: "storage" }
2809
2811
  );
2810
2812
  const FRONTIER_PARAMS = UniformBlock.define("FrontierParams", [
2811
2813
  ["role", "u32"],
2812
- ["slotBase", "u32"],
2813
2814
  ["wg", "u32"],
2814
2815
  ["alpha", "u32"],
2815
2816
  ["beta", "u32"],
@@ -2827,7 +2828,8 @@ const FRONTIER_PARAMS = UniformBlock.define("FrontierParams", [
2827
2828
  ["stride", "u32"],
2828
2829
  ["firstOfSubmit", "u32"],
2829
2830
  ["iteration", "u32"],
2830
- ["pad1", "u32"]
2831
+ ["pad1", "u32"],
2832
+ ["pad2", "u32"]
2831
2833
  ]);
2832
2834
  const BF_PARAMS = UniformBlock.define("BfParams", [
2833
2835
  ["edgeCount", "u32"],
@@ -3406,11 +3408,7 @@ const FRONTIER_FINALIZE = {
3406
3408
  id: "frontier-finalize",
3407
3409
  body: frontierFinalizeWgsl,
3408
3410
  entryPoint: "frontier_finalize",
3409
- bindings: [
3410
- decl(1, 0, "counters", "storage", "array<atomic<u32>>"),
3411
- decl(1, 1, "args", "storage", "array<u32>"),
3412
- decl(2, 0, "P", "uniform", "FrontierParams")
3413
- ],
3411
+ bindings: [decl(1, 0, "counters", "storage", "array<atomic<u32>>"), decl(2, 0, "P", "uniform", "FrontierParams")],
3414
3412
  overrideDecls: [],
3415
3413
  uniforms: [FRONTIER_PARAMS],
3416
3414
  needs: [],
@@ -3533,6 +3531,22 @@ const BFS_UNVISITED_FLAGS = {
3533
3531
  snippetSlots: [],
3534
3532
  phase: "P8"
3535
3533
  };
3534
+ const BFS_NEXT_DEGREE = {
3535
+ id: "bfs-next-degree",
3536
+ body: bfsNextDegreeWgsl,
3537
+ entryPoint: "bfs_next_degree",
3538
+ bindings: [
3539
+ decl(1, 0, "frontier", "storage-ro", "array<u32>"),
3540
+ decl(1, 1, "outDegree", "storage-ro", "array<u32>"),
3541
+ decl(1, 2, "counters", "storage", "array<atomic<u32>>"),
3542
+ decl(2, 0, "P", "uniform", "FrontierParams")
3543
+ ],
3544
+ overrideDecls: [],
3545
+ uniforms: [FRONTIER_PARAMS],
3546
+ needs: ["subgroups"],
3547
+ snippetSlots: [],
3548
+ phase: "P8"
3549
+ };
3536
3550
  const SSSP_RELAX = {
3537
3551
  id: "sssp-relax",
3538
3552
  body: ssspRelaxWgsl,
@@ -3644,6 +3658,7 @@ const REGISTRY = Object.freeze({
3644
3658
  "bfs-bottom-up": BFS_BOTTOM_UP,
3645
3659
  "bfs-bitset-build": BFS_BITSET_BUILD,
3646
3660
  "bfs-unvisited-flags": BFS_UNVISITED_FLAGS,
3661
+ "bfs-next-degree": BFS_NEXT_DEGREE,
3647
3662
  "sssp-relax": SSSP_RELAX,
3648
3663
  "bf-relax": BF_RELAX,
3649
3664
  "closeness-sweep": CLOSENESS_SWEEP,
@@ -4486,11 +4501,6 @@ function algorithmScope(ctx, label, slots) {
4486
4501
  pool: ctx.pool,
4487
4502
  workgroupSize: ctx.workgroupSize,
4488
4503
  scratch: (byteLength, scratchLabel) => lease.storage(byteLength, `${label}/${scratchLabel}`),
4489
- indirect: (byteLength, indirectLabel) => lease.acquire(
4490
- byteLength,
4491
- BufferUsage.STORAGE | BufferUsage.INDIRECT | BufferUsage.COPY_DST | BufferUsage.COPY_SRC,
4492
- `${label}/${indirectLabel}`
4493
- ),
4494
4504
  params(block, values) {
4495
4505
  const slot = ring.reserve(1);
4496
4506
  ring.write(slot, block, values);
@@ -5796,7 +5806,8 @@ const W = Object.freeze({
5796
5806
  nextFarCount: 21,
5797
5807
  thresholdBits: 22,
5798
5808
  deltaBits: 23,
5799
- path: 24
5809
+ path: 24,
5810
+ nextDegreeSum: 25
5800
5811
  });
5801
5812
  function definedWords(words) {
5802
5813
  const out = {};
@@ -5822,16 +5833,14 @@ class Frontier {
5822
5833
  * Wraps the leased buffers; use prepareFrontier().
5823
5834
  * @param vertices - the two vertex queues
5824
5835
  * @param counters - the counters block
5825
- * @param args - the args buffer
5826
5836
  * @param edgeQueue - the edge queue
5827
5837
  * @param edgeCapacity - the edge queue's entry count
5828
5838
  * @param n - the vertex count
5829
5839
  */
5830
- constructor(vertices, counters, args, edgeQueue, edgeCapacity, n) {
5840
+ constructor(vertices, counters, edgeQueue, edgeCapacity, n) {
5831
5841
  this.sideIndex = 0;
5832
5842
  this.vertices = vertices;
5833
5843
  this.counters = counters;
5834
- this.args = args;
5835
5844
  this.edgeQueue = edgeQueue;
5836
5845
  this.edgeCapacity = edgeCapacity;
5837
5846
  this.n = n;
@@ -5862,7 +5871,7 @@ class Frontier {
5862
5871
  this.sideIndex = this.sideIndex === 0 ? 1 : 0;
5863
5872
  }
5864
5873
  /**
5865
- * Seeds a traversal: one `queue.writeBuffer` of the whole 96-byte block (zero except the caller's words) and one of
5874
+ * Seeds a traversal: one `queue.writeBuffer` of the whole 112-byte block (zero except the caller's words) and one of
5866
5875
  * `vertices[0][0] = source`, both ordered before the submit that follows; the source is on side 0 afterwards.
5867
5876
  * `frontierCount` is not a word to seed: the first boundary rotates word 1 into it (the BFS seed is
5868
5877
  * `{ nextFrontierCount: 1, level: U32_MAX }`). A source outside `[0, n)`, an unknown word or a value that is not a
@@ -5899,7 +5908,6 @@ async function prepareFrontier(scope, n, arcCount, edgeCapacity) {
5899
5908
  }
5900
5909
  const kernel = await scope.pipelines.kernel(kernelSpec("frontier-finalize"));
5901
5910
  const queueBytes = 4 * Math.max(1, n);
5902
- const argsBytes = MAX_LEVELS_PER_SUBMIT * FRONTIER_CANDIDATES * INDIRECT_ARGS_STRIDE;
5903
5911
  const vertices = [
5904
5912
  { buffer: scope.scratch(queueBytes, "frontier/vertices-0"), offset: 0, size: queueBytes, window: null },
5905
5913
  { buffer: scope.scratch(queueBytes, "frontier/vertices-1"), offset: 0, size: queueBytes, window: null }
@@ -5910,19 +5918,13 @@ async function prepareFrontier(scope, n, arcCount, edgeCapacity) {
5910
5918
  size: FRONTIER_COUNTERS.byteLength,
5911
5919
  window: null
5912
5920
  };
5913
- const args = {
5914
- buffer: scope.indirect(argsBytes, "frontier/args"),
5915
- offset: 0,
5916
- size: argsBytes,
5917
- window: null
5918
- };
5919
5921
  const edgeQueue = {
5920
5922
  buffer: scope.scratch(4 * capacity, "frontier/edge-queue"),
5921
5923
  offset: 0,
5922
5924
  size: 4 * capacity,
5923
5925
  window: null
5924
5926
  };
5925
- const frontier = new Frontier(vertices, counters, args, edgeQueue, capacity, n);
5927
+ const frontier = new Frontier(vertices, counters, edgeQueue, capacity, n);
5926
5928
  return new FrontierPlannerImpl(scope, kernel, frontier);
5927
5929
  }
5928
5930
  class FrontierPlannerImpl {
@@ -5963,12 +5965,11 @@ class FrontierPlannerImpl {
5963
5965
  const params = scope.params(FRONTIER_PARAMS, {
5964
5966
  ...definedWords(fields),
5965
5967
  role,
5966
- slotBase: level * FRONTIER_CANDIDATES,
5967
5968
  wg: scope.workgroupSize,
5968
5969
  edgeCapacity: frontier.edgeCapacity,
5969
5970
  n: frontier.n
5970
5971
  });
5971
- const bound = this.kernel.bind({ counters: frontier.counters, args: frontier.args, P: params.binding });
5972
+ const bound = this.kernel.bind({ counters: frontier.counters, P: params.binding });
5972
5973
  this.kernel.dispatch(pass, bound, plan1d(1, scope.workgroupSize, scope.caps), [params.offset]);
5973
5974
  }
5974
5975
  }
@@ -6141,6 +6142,7 @@ class RadixSortPlannerImpl {
6141
6142
  }
6142
6143
  }
6143
6144
  const ALGORITHM$3 = "breadthFirstSearch";
6145
+ const NEXT_DEGREE_MAX_GROUPS = 128;
6144
6146
  function bfsRingSlots(windows, levelsPerSubmit) {
6145
6147
  return Math.max((5 + 4 * windows) * levelsPerSubmit + 16, RESULT_BATCH_SLOTS + windows);
6146
6148
  }
@@ -6275,6 +6277,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6275
6277
  const bitset = await ctx.pipelines.kernel(kernelSpec("bfs-bitset-build"));
6276
6278
  const bottomUp = await ctx.pipelines.kernel(kernelSpec("bfs-bottom-up", graphOverrides(reverse, null)));
6277
6279
  const unvisited = await ctx.pipelines.kernel(kernelSpec("bfs-unvisited-flags"));
6280
+ const nextDegree = await ctx.pipelines.kernel(kernelSpec("bfs-next-degree"));
6278
6281
  const pred = await ctx.pipelines.kernel(kernelSpec("sssp-pred", { ...graphOverrides(core, null), MODE: 1 }));
6279
6282
  const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
6280
6283
  const sort = await prepareRadixSort(scope);
@@ -6284,6 +6287,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6284
6287
  const fillPlan = plan1d(n, wg, ctx.caps);
6285
6288
  const levelPlan = planGridStride(Math.max(n, frontier.edgeCapacity), wg, ctx.caps);
6286
6289
  const sweepPlan = planGridStride(n, wg, ctx.caps);
6290
+ const degreePlan = planGridStride(n, wg, ctx.caps, NEXT_DEGREE_MAX_GROUPS);
6287
6291
  const fusedPlan = planGridStride(n * wg, wg, ctx.caps);
6288
6292
  const bitsPlan = plan1d(bitsWords, wg, ctx.caps);
6289
6293
  const recordFill = (pass2, dst, value, mode) => {
@@ -6334,7 +6338,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6334
6338
  const bitsParams = scope.params(FILL_PARAMS, { count: bitsWords, value: 0, mode: 0, pad0: 0 });
6335
6339
  const boundBitsFill = fill.bind({ dst: frontierBits, P: bitsParams.binding });
6336
6340
  for (let level2 = 0; level2 < levelsPerSubmit; level2++) {
6337
- planner.recordFinalize(pass2, 0, level2, { ...fields, firstOfSubmit: Math.min(level2, 2) });
6341
+ planner.recordFinalize(pass2, 0, level2, { ...fields, firstOfSubmit: Math.min(level2, 1) });
6338
6342
  advance.record(pass2, frontier);
6339
6343
  planner.recordFinalize(pass2, 1, level2, fields);
6340
6344
  const params = scope.params(FRONTIER_PARAMS, {
@@ -6400,6 +6404,14 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6400
6404
  });
6401
6405
  bottomUp.dispatch(pass2, boundSweep, sweepPlan, [sweepParams.offset]);
6402
6406
  }
6407
+ const degreeParams = scope.params(FRONTIER_PARAMS, { wg, n, stride: degreePlan.stride ?? wg });
6408
+ const boundDegree = nextDegree.bind({
6409
+ frontier: frontier.output,
6410
+ outDegree,
6411
+ counters,
6412
+ P: degreeParams.binding
6413
+ });
6414
+ nextDegree.dispatch(pass2, boundDegree, degreePlan, [degreeParams.offset]);
6403
6415
  frontier.swap();
6404
6416
  }
6405
6417
  batch.endPass();
@@ -7563,10 +7575,11 @@ function gridSpecFor(n, dim, tuning) {
7563
7575
  levels++;
7564
7576
  }
7565
7577
  const cells = g ** dim;
7578
+ const outsideCells = 2 ** dim;
7566
7579
  const levelOffsets = [0];
7567
7580
  let s = g;
7568
7581
  for (let level = 0; level + 1 < levels; level++) {
7569
- levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? 1 : 0));
7582
+ levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? outsideCells : 0));
7570
7583
  s /= 2;
7571
7584
  }
7572
7585
  return {
@@ -7574,7 +7587,8 @@ function gridSpecFor(n, dim, tuning) {
7574
7587
  g,
7575
7588
  levels,
7576
7589
  cells,
7577
- histWords: cells + 2,
7590
+ outsideCells,
7591
+ histWords: cells + outsideCells + 1,
7578
7592
  levelOffsets: Object.freeze(levelOffsets),
7579
7593
  pyramidCells: levelOffsets[levels - 1] + GRID_COARSEST_SIDE ** dim,
7580
7594
  deterministic: tuning.deterministic
@@ -9890,6 +9904,12 @@ function subset(merged, defaults) {
9890
9904
  }
9891
9905
  return out;
9892
9906
  }
9907
+ function recordExactRepulsion(kernel, pass, bound, plan, n, paramsOffset) {
9908
+ const passes = Math.max(1, Math.ceil(Math.ceil(n / kernel.workgroupSize) / EXACT_TILES_PER_PASS));
9909
+ for (let z = 1; z <= passes; z++) {
9910
+ kernel.dispatch(pass, bound, { ...plan, z }, [paramsOffset]);
9911
+ }
9912
+ }
9893
9913
  class RepulsionExact {
9894
9914
  /**
9895
9915
  * Holds the two compiled kernels; create() is the only caller.
@@ -9982,7 +10002,7 @@ class RepulsionExact {
9982
10002
  recordRepulsion(pass, n, paramsOffset) {
9983
10003
  const bound = this.bound(this.boundRepulsion, "recordRepulsion");
9984
10004
  const plan = plan1d(n, this.repulsion.workgroupSize, this.caps);
9985
- this.repulsion.dispatch(pass, bound, plan, [paramsOffset]);
10005
+ recordExactRepulsion(this.repulsion, pass, bound, plan, n, paramsOffset);
9986
10006
  }
9987
10007
  /**
9988
10008
  * Records K4 only.
@@ -10115,7 +10135,8 @@ class GridPyramidPlannerImpl {
10115
10135
  }
10116
10136
  const { centroid, finalize, hub, downsample } = this.kernels;
10117
10137
  const one = { x: 1, y: 1, z: 1, items: 1, stride: null };
10118
- centroid.dispatch(pass, bound.centroid, plan1d(spec.cells + 1, scope.workgroupSize, scope.caps), [paramsOffset]);
10138
+ const level0 = spec.cells + spec.outsideCells;
10139
+ centroid.dispatch(pass, bound.centroid, plan1d(level0, scope.workgroupSize, scope.caps), [paramsOffset]);
10119
10140
  finalize.dispatch(pass, bound.finalize, one, [bound.finalizeOffset]);
10120
10141
  hub.dispatchIndirect(pass, bound.hub, bound.hubArgs, 0, [paramsOffset]);
10121
10142
  this.dispatches = 3;
@@ -10188,7 +10209,7 @@ class RepulsionGrid {
10188
10209
  }
10189
10210
  /**
10190
10211
  * The model-owned buffers of the grid tier (spec 7.3; PD-11): `cellKey` / `cellVal` / `sortedKey` / `sortedIdx`
10191
- * 4n, `cellHist` / `cellStart` 4 (cells + 2) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
10212
+ * 4n, `cellHist` / `cellStart` 4 histWords (cells + 2^dim + 1) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
10192
10213
  * indirect slot, `pyramid` 16 B per pyramid cell zeroed. `hubCounters` (16 B, zeroed) is the MODEL's on every
10193
10214
  * tier (PD-14: K1 binds it on the exact tier too). n = 0 reports one node's worth of bytes (spec 3.6).
10194
10215
  * @param n - the node count
@@ -10784,7 +10805,9 @@ class ForceAtlas2Model {
10784
10805
  arcEnd: arcCountOf(core),
10785
10806
  accumulate: 0,
10786
10807
  hiEnd,
10787
- midEnd
10808
+ midEnd,
10809
+ settleFloor: SETTLE_FLOOR_UNBOUNDED
10810
+ // ForceAtlas2 does not drift after settling (issue #97)
10788
10811
  };
10789
10812
  }
10790
10813
  /**
@@ -11483,6 +11506,7 @@ class FruchtermanReingoldModel {
11483
11506
  hiEnd,
11484
11507
  midEnd,
11485
11508
  frK: resolved.k ?? 1 / Math.sqrt(n),
11509
+ settleFloor: SETTLE_FLOOR_FRACTION.fruchtermanReingold * (resolved.k ?? 1 / Math.sqrt(n)),
11486
11510
  temperature: adaptive ? FR_START_TEMPERATURE : this.temperatureAt(iteration, resolved)
11487
11511
  };
11488
11512
  }
@@ -11535,7 +11559,7 @@ class FruchtermanReingoldModel {
11535
11559
  if (stop < 2) {
11536
11560
  return;
11537
11561
  }
11538
- k3.dispatch(pass, k3Bound, bound.plan, [offset]);
11562
+ recordExactRepulsion(k3, pass, k3Bound, bound.plan, bound.n, offset);
11539
11563
  if (stop < STAGE_K5$1) {
11540
11564
  return;
11541
11565
  }
@@ -12114,6 +12138,7 @@ class SpringElectricalModel {
12114
12138
  frK: 0,
12115
12139
  temperature: 0,
12116
12140
  springLength: resolved.springLength,
12141
+ settleFloor: SETTLE_FLOOR_FRACTION.springElectrical * resolved.springLength * (SETTLE_FLOOR_REFERENCE_NODES / Math.max(n, 1)) ** 0.25,
12117
12142
  springCoefficient: resolved.springCoefficient ?? SE_DEFAULTS.springCoefficient * springSizeFactor(n),
12118
12143
  coulomb: resolved.gravity ?? SE_DEFAULTS.gravity * springSizeFactor(n),
12119
12144
  dragCoefficient: resolved.dragCoefficient,
@@ -12167,7 +12192,7 @@ class SpringElectricalModel {
12167
12192
  if (stop < 2) {
12168
12193
  return;
12169
12194
  }
12170
- k3.dispatch(pass, k3Bound, bound.plan, [offset]);
12195
+ recordExactRepulsion(k3, pass, k3Bound, bound.plan, bound.n, offset);
12171
12196
  if (stop < STAGE_K5) {
12172
12197
  return;
12173
12198
  }
@@ -12653,7 +12678,7 @@ async function calibrateLayout(ctx, options) {
12653
12678
  };
12654
12679
  }
12655
12680
  export {
12656
- C as ARC_WINDOW_ALIGN,
12681
+ J as ARC_WINDOW_ALIGN,
12657
12682
  EXACT_MAX_NODES,
12658
12683
  FA2_DEFAULTS,
12659
12684
  FR_DEFAULTS,
@@ -12661,10 +12686,10 @@ export {
12661
12686
  LAYOUT_TUNING_DEFAULTS,
12662
12687
  MAX_1D_ITEMS,
12663
12688
  MAX_WORKGROUPS_PER_DIM,
12664
- D as PASSTHROUGH_FORMAT_CODES,
12689
+ K as PASSTHROUGH_FORMAT_CODES,
12665
12690
  SE_DEFAULTS,
12666
- H as STORAGE_ALIGN,
12667
- J as WORKGROUP_SIZE,
12691
+ N as STORAGE_ALIGN,
12692
+ O as WORKGROUP_SIZE,
12668
12693
  WebGpuGraphError,
12669
12694
  bellmanFord,
12670
12695
  breadthFirstSearch,
@@ -12679,7 +12704,7 @@ export {
12679
12704
  eigenvectorCentrality,
12680
12705
  hasErrorCode,
12681
12706
  hits,
12682
- K as isSoftwareAdapter,
12707
+ Q as isSoftwareAdapter,
12683
12708
  isWebGpuGraphError,
12684
12709
  katzCentrality,
12685
12710
  pageRank,