@graphty/webgpu-graph-algorithms 0.6.5 → 0.6.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/README.md +38 -17
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-hzGggHeM.js → context-Cezi7qpi.js} +46 -22
  4. package/dist/chunks/context-Cezi7qpi.js.map +1 -0
  5. package/dist/node.js +19 -11
  6. package/dist/node.js.map +1 -1
  7. package/dist/src/algorithms/bfs.d.ts +10 -6
  8. package/dist/src/algorithms/bfs.d.ts.map +1 -1
  9. package/dist/src/algorithms/bfs.js +32 -9
  10. package/dist/src/algorithms/bfs.js.map +1 -1
  11. package/dist/src/algorithms/pagerank.d.ts.map +1 -1
  12. package/dist/src/algorithms/pagerank.js +19 -5
  13. package/dist/src/algorithms/pagerank.js.map +1 -1
  14. package/dist/src/algorithms/power-iteration.d.ts.map +1 -1
  15. package/dist/src/algorithms/power-iteration.js +8 -2
  16. package/dist/src/algorithms/power-iteration.js.map +1 -1
  17. package/dist/src/constants.d.ts +33 -0
  18. package/dist/src/constants.d.ts.map +1 -1
  19. package/dist/src/constants.js +33 -0
  20. package/dist/src/constants.js.map +1 -1
  21. package/dist/src/kernel/dispatch.d.ts +2 -2
  22. package/dist/src/kernel/dispatch.d.ts.map +1 -1
  23. package/dist/src/kernel/kernel.d.ts +1 -1
  24. package/dist/src/kernel/kernel.js +2 -2
  25. package/dist/src/kernel/kernel.js.map +1 -1
  26. package/dist/src/kernel/prelude.d.ts.map +1 -1
  27. package/dist/src/kernel/prelude.js +2 -1
  28. package/dist/src/kernel/prelude.js.map +1 -1
  29. package/dist/src/kernels.d.ts +10 -7
  30. package/dist/src/kernels.d.ts.map +1 -1
  31. package/dist/src/kernels.js +33 -9
  32. package/dist/src/kernels.js.map +1 -1
  33. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  34. package/dist/src/layouts/forceatlas2.js +2 -1
  35. package/dist/src/layouts/forceatlas2.js.map +1 -1
  36. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  37. package/dist/src/layouts/fruchterman-reingold.js +4 -2
  38. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  39. package/dist/src/layouts/repulsion-exact.d.ts +16 -0
  40. package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
  41. package/dist/src/layouts/repulsion-exact.js +21 -1
  42. package/dist/src/layouts/repulsion-exact.js.map +1 -1
  43. package/dist/src/layouts/repulsion-grid.d.ts +1 -1
  44. package/dist/src/layouts/repulsion-grid.js +1 -1
  45. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  46. package/dist/src/layouts/spring-electrical.js +6 -2
  47. package/dist/src/layouts/spring-electrical.js.map +1 -1
  48. package/dist/src/memory/residency.js +14 -4
  49. package/dist/src/memory/residency.js.map +1 -1
  50. package/dist/src/node/index.d.ts +13 -8
  51. package/dist/src/node/index.d.ts.map +1 -1
  52. package/dist/src/node/index.js +36 -17
  53. package/dist/src/node/index.js.map +1 -1
  54. package/dist/src/primitives/frontier.d.ts +1 -0
  55. package/dist/src/primitives/frontier.d.ts.map +1 -1
  56. package/dist/src/primitives/frontier.js +1 -0
  57. package/dist/src/primitives/frontier.js.map +1 -1
  58. package/dist/src/primitives/grid-pyramid.d.ts +4 -4
  59. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  60. package/dist/src/primitives/grid-pyramid.js +4 -3
  61. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  62. package/dist/src/primitives/grid.d.ts +13 -10
  63. package/dist/src/primitives/grid.d.ts.map +1 -1
  64. package/dist/src/primitives/grid.js +10 -7
  65. package/dist/src/primitives/grid.js.map +1 -1
  66. package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
  67. package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
  68. package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
  69. package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
  70. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
  71. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
  72. package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
  73. package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
  74. package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
  75. package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
  76. package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
  77. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
  78. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
  79. package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
  80. package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
  81. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
  82. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
  83. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
  84. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
  85. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
  86. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
  87. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
  88. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
  89. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +21 -20
  90. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
  91. package/dist/src/wgsl/frontier-finalize.wgsl.js +26 -25
  92. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
  93. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
  94. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
  95. package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
  96. package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
  97. package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
  98. package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
  99. package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
  100. package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
  101. package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
  102. package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
  103. package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
  104. package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
  105. package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
  106. package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
  107. package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
  108. package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
  109. package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
  110. package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
  111. package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
  112. package/dist/src/wgsl/histogram.wgsl.js +1 -1
  113. package/dist/webgpu-graph-algorithms.js +144 -45
  114. package/dist/webgpu-graph-algorithms.js.map +1 -1
  115. package/package.json +3 -3
  116. package/src/algorithms/bfs.ts +33 -9
  117. package/src/algorithms/pagerank.ts +19 -5
  118. package/src/algorithms/power-iteration.ts +8 -2
  119. package/src/constants.ts +35 -0
  120. package/src/kernel/dispatch.ts +2 -2
  121. package/src/kernel/kernel.ts +2 -2
  122. package/src/kernel/prelude.ts +2 -0
  123. package/src/kernels.ts +35 -9
  124. package/src/layouts/forceatlas2.ts +2 -0
  125. package/src/layouts/fruchterman-reingold.ts +4 -1
  126. package/src/layouts/repulsion-exact.ts +29 -1
  127. package/src/layouts/repulsion-grid.ts +1 -1
  128. package/src/layouts/spring-electrical.ts +8 -1
  129. package/src/memory/residency.ts +14 -4
  130. package/src/node/index.ts +42 -18
  131. package/src/primitives/frontier.ts +2 -0
  132. package/src/primitives/grid-pyramid.ts +6 -5
  133. package/src/primitives/grid.ts +17 -12
  134. package/src/wgsl/advance-expand.wgsl.ts +4 -3
  135. package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
  136. package/src/wgsl/bfs-fused.wgsl.ts +6 -6
  137. package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
  138. package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
  139. package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
  140. package/src/wgsl/frontier-finalize.wgsl.ts +26 -25
  141. package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
  142. package/src/wgsl/grid-centroid.wgsl.ts +3 -2
  143. package/src/wgsl/grid-downsample.wgsl.ts +1 -1
  144. package/src/wgsl/grid-far-field.wgsl.ts +15 -8
  145. package/src/wgsl/grid-near-field.wgsl.ts +6 -2
  146. package/src/wgsl/histogram.wgsl.ts +1 -1
  147. package/dist/chunks/context-hzGggHeM.js.map +0 -1
@@ -1,5 +1,5 @@
1
- import { W as WebGpuGraphError, U as UNIFORM_SLOT_BYTES, B as BufferUsage, M as MAX_WORKGROUPS_PER_DIM, a as WGSL_RESERVED_WORDS, S as STATE_HEADER_BYTES, d as deviceLostError, i as isWebGpuGraphError, b as U32_MAX$2, c as MAX_LEVELS_PER_SUBMIT, R as RADIX_BINS, F as FUSED_FRONTIER_MAX, e as BEAMER_BETA, f as SSSP_DELTA_FACTOR, g as F32_INF_BITS, h as GRID_COARSEST_SIDE, j as GRID_MIN_SIDE, k as GRID_SORT_BITS, l as FA2_DEFAULTS, m as MAX_ITERATIONS_PER_STEP, n as MAX_1D_ITEMS, o as hasErrorCode, p as FA2_FLAG_FIRST, P as PARTIAL_BYTES, I as INDIRECT_ARGS_STRIDE, q as GRID_HUB_CELL, L as LAYOUT_TUNING_DEFAULTS, E as EXACT_MAX_NODES, T as TRACE_RECORD_BYTES, r as GRID_BBOX_MARGIN, s as GRID_EXTENT_FLOOR, t as FR_ADAPTIVE_MAX_ITERATIONS, u as FR_START_TEMPERATURE, v as FA2_FLAG_ADAPTIVE, w as FR_REHEAT_FRACTION, x as FR_DEFAULTS, y as SE_DEFAULTS, z as SE_SCALE_REFERENCE_NODES } from "./chunks/context-hzGggHeM.js";
2
- import { A, G, C, D, H, J } from "./chunks/context-hzGggHeM.js";
1
+ import { W as WebGpuGraphError, U as UNIFORM_SLOT_BYTES, B as BufferUsage, M as MAX_WORKGROUPS_PER_DIM, a as WGSL_RESERVED_WORDS, S as STATE_HEADER_BYTES, d as deviceLostError, i as isWebGpuGraphError, b as U32_MAX$2, c as MAX_LEVELS_PER_SUBMIT, R as RADIX_BINS, F as FUSED_FRONTIER_MAX, e as BEAMER_BETA, f as SSSP_DELTA_FACTOR, g as F32_INF_BITS, h as GRID_COARSEST_SIDE, j as GRID_MIN_SIDE, k as GRID_SORT_BITS, l as FA2_DEFAULTS, m as MAX_ITERATIONS_PER_STEP, n as MAX_1D_ITEMS, o as hasErrorCode, p as FA2_FLAG_FIRST, P as PARTIAL_BYTES, E as EXACT_TILES_PER_PASS, I as INDIRECT_ARGS_STRIDE, q as GRID_HUB_CELL, L as LAYOUT_TUNING_DEFAULTS, r as EXACT_MAX_NODES, s as SETTLE_FLOOR_UNBOUNDED, T as TRACE_RECORD_BYTES, t as GRID_BBOX_MARGIN, u as GRID_EXTENT_FLOOR, v as FR_ADAPTIVE_MAX_ITERATIONS, w as FR_START_TEMPERATURE, x as FA2_FLAG_ADAPTIVE, y as SETTLE_FLOOR_FRACTION, z as FR_REHEAT_FRACTION, A as FR_DEFAULTS, C as SE_DEFAULTS, D as SETTLE_FLOOR_REFERENCE_NODES, H as SE_SCALE_REFERENCE_NODES } from "./chunks/context-Cezi7qpi.js";
2
+ import { J, G, K, N, O, Q } from "./chunks/context-Cezi7qpi.js";
3
3
  import { renumberPartition, INVALID_INDEX, makeMask, maskTest, expandEdges, fromEdgeArrays } from "@graphty/graph-format";
4
4
  class UniformRing {
5
5
  /**
@@ -217,7 +217,7 @@ function planGridStride(items, wg, caps, maxGroups) {
217
217
  if (items === 0) {
218
218
  return { x: 0, y: 1, z: 1, items, stride: null };
219
219
  }
220
- const cap = Math.min(caps.software ? 64 : 4096, perDimension(caps));
220
+ const cap = Math.min(maxGroups ?? (caps.software ? 64 : 4096), perDimension(caps));
221
221
  const groups = Math.min(Math.ceil(items / wg), cap);
222
222
  return { x: groups, y: 1, z: 1, items, stride: groups * wg };
223
223
  }
@@ -582,7 +582,7 @@ fn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
582
582
  if (lid.x == 0u) {
583
583
  base = atomicAdd(&counters[8], aggregate); // edgeCount: ONE reservation per workgroup, not one per arc
584
584
  atomicAdd(&counters[9], aggregate); // edgeCountUnclamped: the overflow detector (PD-23)
585
- atomicAdd(&counters[2], aggregate); // frontierDegreeSum: Beamer's m_f (P8-T8)
585
+ atomicAdd(&counters[2], aggregate); // frontierDegreeSum: what this level expanded (the inspect seam)
586
586
  }
587
587
  workgroupBarrier();
588
588
  for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
@@ -773,7 +773,7 @@ fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
773
773
  let d = select(0u, a1 - a0, a1 > a0);
774
774
  wdeg = d;
775
775
  wstart = a0;
776
- atomicAdd(&counters[2], d); // frontierDegreeSum, so Beamer's test (P8-T8) sees fused levels too
776
+ atomicAdd(&counters[2], d); // frontierDegreeSum, so the inspect seam sees fused levels too
777
777
  }
778
778
  let deg = workgroupUniformLoad(&wdeg); // uniform: the loop below may hold barriers
779
779
  let start = workgroupUniformLoad(&wstart);
@@ -805,6 +805,21 @@ fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
805
805
  }
806
806
  `
807
807
  );
808
+ const bfsNextDegreeWgsl = (
809
+ /* wgsl */
810
+ `
811
+ @compute @workgroup_size(WG)
812
+ fn bfs_next_degree(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
813
+ let count = select(0u, atomicLoad(&counters[1]), atomicLoad(&counters[24]) != 0u); // nextFrontierCount, on a level that claimed (the path word)
814
+ var sum = 0u;
815
+ for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // no barrier inside: the trip count is per lane
816
+ sum = sum + outDegree[frontier[i]];
817
+ }
818
+ let total = wg_reduce_u32(sum, lid.x, 0u); // the prelude's workgroup sum; uniform: after the loop
819
+ if (lid.x == 0u) { atomicAdd(&counters[25], total); } // nextDegreeSum: ONE atomic per workgroup
820
+ }
821
+ `
822
+ );
808
823
  const bfsUnvisitedFlagsWgsl = (
809
824
  /* wgsl */
810
825
  `
@@ -1265,14 +1280,23 @@ fn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: cent
1265
1280
  }
1266
1281
 
1267
1282
  @compute @workgroup_size(WG)
1268
- fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
1283
+ fn repulsion(
1284
+ @builtin(workgroup_id) wid: vec3<u32>,
1285
+ @builtin(local_invocation_id) lid: vec3<u32>,
1286
+ @builtin(num_workgroups) nwg: vec3<u32>,
1287
+ ) {
1288
+ // issue #87: pass p of the tile range is dispatched with p + 1 z slices; only the last slice works, so the pass
1289
+ // index needs no uniform. Uniform: keyed on workgroup_id and num_workgroups only.
1290
+ if (wid.z + 1u < nwg.z) { return; }
1269
1291
  let i = linear_id(wid, lid.x);
1270
1292
  let valid = i < P.n;
1271
1293
  var pi = vec4f(0.0);
1272
1294
  if (valid) { pi = pos[i]; }
1273
1295
  var f = vec3f(0.0);
1274
1296
  let tiles = (P.n + WG - 1u) / WG;
1275
- for (var t = 0u; t < tiles; t = t + 1u) {
1297
+ let tileBegin = (nwg.z - 1u) * EXACT_TILES_PER_PASS;
1298
+ let tileEnd = min(tiles, tileBegin + EXACT_TILES_PER_PASS);
1299
+ for (var t = tileBegin; t < tileEnd; t = t + 1u) {
1276
1300
  let j = t * WG + lid.x;
1277
1301
  if (j < P.n) { tile[lid.x] = pos[j]; } else { tile[lid.x] = vec4f(0.0); } // guarded fill; mass 0 marks the pad
1278
1302
  workgroupBarrier(); // uniform: every invocation reaches it
@@ -1295,6 +1319,10 @@ fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
1295
1319
  }
1296
1320
  workgroupBarrier();
1297
1321
  }
1322
+ if (tileEnd < tiles) { // an earlier pass: its partial sum only (uniform: P.n, nwg)
1323
+ if (valid) { store_force(i, load_force(i) + f); }
1324
+ return;
1325
+ }
1298
1326
  // epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it
1299
1327
  var sw = 0.0;
1300
1328
  var tr = 0.0;
@@ -1400,7 +1428,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
1400
1428
  S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)
1401
1429
  let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)
1402
1430
  S.meanDisplacement = meanDisp;
1403
- S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);
1431
+ S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= min(P.settleThreshold * S.rmsRadius, P.settleFloor)); // relative AND absolute (issue #97)
1404
1432
  }
1405
1433
  S.iteration = S.iteration + 1u;
1406
1434
  T[P.iterationIndex].meanDisplacement = S.meanDisplacement;
@@ -1418,7 +1446,9 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
1418
1446
  S.invCellSize = 1.0 / cellSize;
1419
1447
  S.eps = 0.25 * cellSize;
1420
1448
  }
1421
- S.outsideGrid = cellHist[cells]; // the previous iteration's pseudo-cell count (0 after load)
1449
+ var outside = 0u; // the previous iteration's pseudo-cell counts, one per orthant (issue #90; 0 after load)
1450
+ for (var o = 0u; o < select(4u, 8u, P.dim == 3u); o = o + 1u) { outside = outside + cellHist[cells + o]; }
1451
+ S.outsideGrid = outside;
1422
1452
  S.maxCellOccupancy = atomicLoad(&hubCounters[1]);
1423
1453
  atomicStore(&hubCounters[0], 0u);
1424
1454
  atomicStore(&hubCounters[1], 0u);
@@ -1486,19 +1516,19 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
1486
1516
  let finished = atomicLoad(&counters[0]);
1487
1517
  let next = atomicLoad(&counters[1]);
1488
1518
  let degSum = atomicLoad(&counters[2]);
1519
+ let nextDeg = atomicLoad(&counters[25]); // deg(F_b), summed by bfs-next-degree when F_b was claimed (issue #391)
1489
1520
  atomicStore(&counters[3], finished); // prevFrontierCount
1490
1521
  atomicStore(&counters[4], degSum); // prevDegreeSum
1491
1522
  atomicStore(&counters[0], next); // the rotation
1492
1523
  atomicStore(&counters[1], 0u);
1493
1524
  atomicStore(&counters[2], 0u);
1525
+ atomicStore(&counters[25], 0u); // the next level's claims sum from 0
1494
1526
  atomicStore(&counters[8], 0u); // edgeCount
1495
1527
  atomicStore(&counters[9], 0u); // edgeCountUnclamped
1496
1528
  atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount
1497
- if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped to 2 (P8-T8, PD-18)
1498
- atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount (exact): F_b was inside the submit's rebuilt sum iff b >= 1
1499
- }
1500
- if (P.firstOfSubmit >= 2u) {
1501
- atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2
1529
+ if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped (P8-T8, PD-18): F_b was inside the submit's rebuilt sums iff b >= 1
1530
+ atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount, exact
1531
+ atomicStore(&counters[6], atomicLoad(&counters[6]) - nextDeg); // unvisitedDegreeSum, exact (issue #391: no longer one level stale)
1502
1532
  }
1503
1533
  let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0
1504
1534
  atomicStore(&counters[11], level);
@@ -1508,7 +1538,7 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
1508
1538
  if (P.mode == 1u) {
1509
1539
  direction = 0u; // top-down only (the test seam)
1510
1540
  } else if (direction == 0u) {
1511
- if (degSum > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing
1541
+ if (nextDeg > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing, m_f the degree of the frontier about to be expanded
1512
1542
  } else {
1513
1543
  if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking
1514
1544
  }
@@ -1603,7 +1633,8 @@ fn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocatio
1603
1633
  let g = i32(P.gridMax);
1604
1634
  var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;
1605
1635
  if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }
1606
- var key = cells; // the outside pseudo-cell (7.7)
1636
+ var key = cells + select(0u, 1u, c.x >= g / 2) + select(0u, 2u, c.y >= g / 2); // an outside pseudo-cell: its orthant (issue #90)
1637
+ if (P.dim == 3u) { key = key + select(0u, 4u, c.z >= g / 2); }
1607
1638
  if (inside) {
1608
1639
  key = u32(c.x) + P.gridMax * u32(c.y);
1609
1640
  if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }
@@ -1621,7 +1652,7 @@ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.
1621
1652
  @compute @workgroup_size(WG)
1622
1653
  fn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
1623
1654
  let c = linear_id(wid, lid.x);
1624
- if (c > grid_cells()) { return; } // cells [0, cells]: the pseudo-cell is index cells; no barrier follows
1655
+ if (c >= grid_cells() + select(4u, 8u, P.dim == 3u)) { return; } // cells [0, cells + 2^dim): the pseudo-cells follow the real ones; no barrier follows
1625
1656
  let start = cellStart[c];
1626
1657
  let count = cellStart[c + 1u] - start;
1627
1658
  atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration
@@ -1699,11 +1730,12 @@ fn store_force(i: u32, f: vec3f) {
1699
1730
  }
1700
1731
  fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }
1701
1732
  fn grid_side(level: u32) -> u32 { return P.gridMax >> level; }
1702
- fn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cell at index cells)
1733
+ fn outside_cells() -> u32 { return select(4u, 8u, P.dim == 3u); } // one pseudo-cell per orthant (issue #90)
1734
+ fn level_base(level: u32) -> u32 { // the pyramid index of level L's cell 0 (level 0 carries the pseudo-cells at cells ..)
1703
1735
  var base = 0u;
1704
1736
  for (var l = 0u; l < level; l = l + 1u) {
1705
1737
  let s = grid_side(l);
1706
- base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, 1u, l == 0u);
1738
+ base = base + s * s * select(1u, s, P.dim == 3u) + select(0u, outside_cells(), l == 0u);
1707
1739
  }
1708
1740
  return base;
1709
1741
  }
@@ -1714,7 +1746,9 @@ fn cell_at(level: u32, cx: i32, cy: i32, cz: i32) -> u32 {
1714
1746
  fn cell_force(pi: vec4f, q: vec4f) -> vec3f { // one far-field term, softened by state.eps (7.7)
1715
1747
  if (q.w <= 0.0) { return vec3f(0.0); } // an empty cell
1716
1748
  let d = pi.xyz - q.xyz / q.w; // to the mass-weighted centroid
1717
- let d2 = dot(d, d) + S.eps * S.eps;
1749
+ var d2 = dot(d, d);
1750
+ if (LAW == 0u) { d2 = max(d2, FA2_DIST_FLOOR_SQ); } // FA2 alone floors d >= 0.01, as K3 and G7 do (issue #89); FR and coulomb are unfloored (7.20)
1751
+ d2 = d2 + S.eps * S.eps;
1718
1752
  if (LAW == 1u) { return d * (P.frK * P.frK * q.w / d2); } // LAW 1 (FR, 7.20): k^2 / d per node, q.w nodes at the centroid
1719
1753
  if (LAW == 2u) { return d * (-P.coulomb * pi.w * q.w / (d2 * sqrt(d2))); } // LAW 2 (coulomb): -g m_i M_cell / d^2
1720
1754
  return d * (P.scalingRatio * pi.w * q.w / d2); // LAW 0 (FA2): |F| = k m_i M_cell / d
@@ -1763,9 +1797,11 @@ fn grid_far_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
1763
1797
  }
1764
1798
  }
1765
1799
  }
1766
- f = f + cell_force(pi, pyramid[grid_cells()]); // the outside pseudo-cell as one far-field term
1800
+ for (var o = 0u; o < outside_cells(); o = o + 1u) { // every outside pseudo-cell: one far-field term per orthant
1801
+ f = f + cell_force(pi, pyramid[grid_cells() + o]);
1802
+ }
1767
1803
  } else {
1768
- for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (it would include itself)
1804
+ for (var cz = 0; cz <= zTop; cz = cz + 1) { // an outside node: the coarsest level in full, no pseudo-cell (G7 sums them pair by pair)
1769
1805
  for (var cy = 0; cy < ts; cy = cy + 1) {
1770
1806
  for (var cx = 0; cx < ts; cx = cx + 1) {
1771
1807
  f = f + cell_force(pi, pyramid[cell_at(top, cx, cy, cz)]);
@@ -1867,7 +1903,11 @@ fn grid_near_field(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocat
1867
1903
  }
1868
1904
  }
1869
1905
  } else {
1870
- f = cell_sum(i, pi, grid_cells(), true); // an outside node: the pseudo-cell alone
1906
+ var own = grid_cells() + select(0u, 1u, c0.x >= g / 2) + select(0u, 2u, c0.y >= g / 2); // G1's orthant key
1907
+ if (P.dim == 3u) { own = own + select(0u, 4u, c0.z >= g / 2); }
1908
+ for (var o = grid_cells(); o < grid_cells() + select(4u, 8u, P.dim == 3u); o = o + 1u) { // an outside node: every outside pseudo-cell
1909
+ f = f + cell_sum(i, pi, o, o == own);
1910
+ }
1871
1911
  }
1872
1912
  }
1873
1913
  // epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it (K3's text)
@@ -2589,7 +2629,8 @@ const FA2_PARAMS = UniformBlock.define("Fa2Params", [
2589
2629
  ["coulomb", "f32"],
2590
2630
  ["dragCoefficient", "f32"],
2591
2631
  ["timeStep", "f32"],
2592
- ["midEnd", "u32"]
2632
+ ["midEnd", "u32"],
2633
+ ["settleFloor", "f32"]
2593
2634
  ]);
2594
2635
  const FA2_STATE = UniformBlock.define(
2595
2636
  "Fa2State",
@@ -2763,7 +2804,8 @@ const FRONTIER_COUNTERS = UniformBlock.define(
2763
2804
  ["nextFarCount", "u32"],
2764
2805
  ["thresholdBits", "u32"],
2765
2806
  ["deltaBits", "u32"],
2766
- ["path", "u32"]
2807
+ ["path", "u32"],
2808
+ ["nextDegreeSum", "u32"]
2767
2809
  ],
2768
2810
  { layout: "storage" }
2769
2811
  );
@@ -3489,6 +3531,22 @@ const BFS_UNVISITED_FLAGS = {
3489
3531
  snippetSlots: [],
3490
3532
  phase: "P8"
3491
3533
  };
3534
+ const BFS_NEXT_DEGREE = {
3535
+ id: "bfs-next-degree",
3536
+ body: bfsNextDegreeWgsl,
3537
+ entryPoint: "bfs_next_degree",
3538
+ bindings: [
3539
+ decl(1, 0, "frontier", "storage-ro", "array<u32>"),
3540
+ decl(1, 1, "outDegree", "storage-ro", "array<u32>"),
3541
+ decl(1, 2, "counters", "storage", "array<atomic<u32>>"),
3542
+ decl(2, 0, "P", "uniform", "FrontierParams")
3543
+ ],
3544
+ overrideDecls: [],
3545
+ uniforms: [FRONTIER_PARAMS],
3546
+ needs: ["subgroups"],
3547
+ snippetSlots: [],
3548
+ phase: "P8"
3549
+ };
3492
3550
  const SSSP_RELAX = {
3493
3551
  id: "sssp-relax",
3494
3552
  body: ssspRelaxWgsl,
@@ -3600,6 +3658,7 @@ const REGISTRY = Object.freeze({
3600
3658
  "bfs-bottom-up": BFS_BOTTOM_UP,
3601
3659
  "bfs-bitset-build": BFS_BITSET_BUILD,
3602
3660
  "bfs-unvisited-flags": BFS_UNVISITED_FLAGS,
3661
+ "bfs-next-degree": BFS_NEXT_DEGREE,
3603
3662
  "sssp-relax": SSSP_RELAX,
3604
3663
  "bf-relax": BF_RELAX,
3605
3664
  "closeness-sweep": CLOSENESS_SWEEP,
@@ -5039,7 +5098,7 @@ async function prepareSpmvPull(scope, rev, options) {
5039
5098
  return new SpmvPullPlannerImpl(scope, compiled, perm, options.weights);
5040
5099
  }
5041
5100
  const PR_BATCH = 8;
5042
- const RING_SLOTS$4 = 2 * PR_BATCH + 1;
5101
+ const RING_SLOTS$4 = 2 * PR_BATCH + 2;
5043
5102
  function checkDest$3(dest, n, algorithm) {
5044
5103
  if (dest === void 0) {
5045
5104
  return null;
@@ -5129,6 +5188,7 @@ async function run(ctx, s, personalization, options, algorithm) {
5129
5188
  const groups = groupsOf(scalePlan);
5130
5189
  const partialsBytes = PR_PARTIAL.byteLength * (1 + groups);
5131
5190
  const partials = scope.scratch(partialsBytes, "partials");
5191
+ const lastHeader = scope.scratch(PR_PARTIAL.byteLength, "lastHeader");
5132
5192
  if (personalization !== null) {
5133
5193
  uploaded = ctx.residency.array(personalization, `${algorithm}/personalization`);
5134
5194
  }
@@ -5157,17 +5217,19 @@ async function run(ctx, s, personalization, options, algorithm) {
5157
5217
  const xNormBinding = bindingOf$2(xNorm, bytes);
5158
5218
  const outWeightSumBinding = bindingOf$2(outWeightSum, bytes);
5159
5219
  const partialsBinding = bindingOf$2(partials, partialsBytes);
5220
+ const lastHeaderBinding = bindingOf$2(lastHeader, PR_PARTIAL.byteLength);
5160
5221
  const coefficients = { alpha, beta: 1 - alpha, uniformP: 1 / n };
5161
5222
  let cur = 0;
5162
5223
  let iterationsRun = 0;
5163
5224
  for (; ; ) {
5164
5225
  const k = Math.min(PR_BATCH, maxIterations - iterationsRun);
5165
5226
  const batch = new CommandBatch(ctx, algorithm);
5166
- const pass = batch.pass("iterations");
5227
+ let pass = batch.pass("iterations");
5167
5228
  if (iterationsRun === 0) {
5168
5229
  normaliser.record(pass, weightedCore, outWeightSumBinding);
5169
5230
  }
5170
- for (let i = 0; i < k; i++) {
5231
+ const last = iterationsRun + k === maxIterations;
5232
+ for (let i = 0; i < k + (last ? 1 : 0); i++) {
5171
5233
  const params = scope.params(PR_PARAMS, {
5172
5234
  n,
5173
5235
  groups,
@@ -5176,6 +5238,10 @@ async function run(ctx, s, personalization, options, algorithm) {
5176
5238
  convergeThreshold: tolerance * n
5177
5239
  });
5178
5240
  const other = 1 - cur;
5241
+ if (i === k) {
5242
+ batch.copy(partialsBinding, lastHeaderBinding, PR_PARTIAL.byteLength);
5243
+ pass = batch.pass("convergence");
5244
+ }
5179
5245
  const scaleBound = scale.bind({
5180
5246
  rankIn: rank[cur],
5181
5247
  rankPrev: rank[other],
@@ -5187,6 +5253,9 @@ async function run(ctx, s, personalization, options, algorithm) {
5187
5253
  scale.dispatch(pass, scaleBound, scalePlan, [params.offset]);
5188
5254
  const finalizeBound = finalize.bind({ partials: partialsBinding, P: params.binding });
5189
5255
  finalize.dispatch(pass, finalizeBound, finalizePlan, [params.offset]);
5256
+ if (i === k) {
5257
+ break;
5258
+ }
5190
5259
  pull.record(
5191
5260
  pass,
5192
5261
  weightedRev,
@@ -5202,6 +5271,7 @@ async function run(ctx, s, personalization, options, algorithm) {
5202
5271
  }
5203
5272
  batch.endPass();
5204
5273
  const headerRequest = batch.readback(partials, 0, PR_PARTIAL.byteLength);
5274
+ const lastHeaderRequest = last ? batch.readback(lastHeader, 0, PR_PARTIAL.byteLength) : headerRequest;
5205
5275
  const scoresRequest = batch.readback(rank[cur].buffer, 0, bytes);
5206
5276
  scope.flush();
5207
5277
  const submitted = batch.submit();
@@ -5225,7 +5295,7 @@ async function run(ctx, s, personalization, options, algorithm) {
5225
5295
  scores,
5226
5296
  iterations: converged ? firstConverged : iterationsRun,
5227
5297
  converged,
5228
- danglingMass: folded.danglingMass,
5298
+ danglingMass: PR_PARTIAL.read(new DataView(back), lastHeaderRequest.offset).danglingMass,
5229
5299
  precision: "f32"
5230
5300
  };
5231
5301
  }
@@ -5344,7 +5414,8 @@ async function runPowerIteration(ctx, n, config) {
5344
5414
  const k = Math.min(BATCH, config.maxIterations - iterationsRun);
5345
5415
  const batch = new CommandBatch(ctx, config.label);
5346
5416
  const pass = batch.pass("iterations");
5347
- for (let i = 0; i < k; i++) {
5417
+ const last = iterationsRun + k === config.maxIterations;
5418
+ for (let i = 0; i < k + (last ? 1 : 0); i++) {
5348
5419
  const iteration = iterationsRun + i + 1;
5349
5420
  const params = scope.params(PR_PARAMS, {
5350
5421
  n,
@@ -5366,6 +5437,9 @@ async function runPowerIteration(ctx, n, config) {
5366
5437
  };
5367
5438
  scaleNorm.dispatch(pass, scaleNorm.bind(scaleBindings), scalePlan, [params.offset]);
5368
5439
  finalize.dispatch(pass, finalize.bind({ partials, P: params.binding }), finalizePlan, [params.offset]);
5440
+ if (i === k) {
5441
+ break;
5442
+ }
5369
5443
  if (scaleApply !== null) {
5370
5444
  scaleApply.dispatch(pass, scaleApply.bind(scaleBindings), scalePlan, [params.offset]);
5371
5445
  }
@@ -5747,7 +5821,8 @@ const W = Object.freeze({
5747
5821
  nextFarCount: 21,
5748
5822
  thresholdBits: 22,
5749
5823
  deltaBits: 23,
5750
- path: 24
5824
+ path: 24,
5825
+ nextDegreeSum: 25
5751
5826
  });
5752
5827
  function definedWords(words) {
5753
5828
  const out = {};
@@ -6082,6 +6157,7 @@ class RadixSortPlannerImpl {
6082
6157
  }
6083
6158
  }
6084
6159
  const ALGORITHM$3 = "breadthFirstSearch";
6160
+ const NEXT_DEGREE_MAX_GROUPS = 128;
6085
6161
  function bfsRingSlots(windows, levelsPerSubmit) {
6086
6162
  return Math.max((5 + 4 * windows) * levelsPerSubmit + 16, RESULT_BATCH_SLOTS + windows);
6087
6163
  }
@@ -6216,6 +6292,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6216
6292
  const bitset = await ctx.pipelines.kernel(kernelSpec("bfs-bitset-build"));
6217
6293
  const bottomUp = await ctx.pipelines.kernel(kernelSpec("bfs-bottom-up", graphOverrides(reverse, null)));
6218
6294
  const unvisited = await ctx.pipelines.kernel(kernelSpec("bfs-unvisited-flags"));
6295
+ const nextDegree = await ctx.pipelines.kernel(kernelSpec("bfs-next-degree"));
6219
6296
  const pred = await ctx.pipelines.kernel(kernelSpec("sssp-pred", { ...graphOverrides(core, null), MODE: 1 }));
6220
6297
  const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
6221
6298
  const sort = await prepareRadixSort(scope);
@@ -6225,6 +6302,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6225
6302
  const fillPlan = plan1d(n, wg, ctx.caps);
6226
6303
  const levelPlan = planGridStride(Math.max(n, frontier.edgeCapacity), wg, ctx.caps);
6227
6304
  const sweepPlan = planGridStride(n, wg, ctx.caps);
6305
+ const degreePlan = planGridStride(n, wg, ctx.caps, NEXT_DEGREE_MAX_GROUPS);
6228
6306
  const fusedPlan = planGridStride(n * wg, wg, ctx.caps);
6229
6307
  const bitsPlan = plan1d(bitsWords, wg, ctx.caps);
6230
6308
  const recordFill = (pass2, dst, value, mode) => {
@@ -6275,7 +6353,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6275
6353
  const bitsParams = scope.params(FILL_PARAMS, { count: bitsWords, value: 0, mode: 0, pad0: 0 });
6276
6354
  const boundBitsFill = fill.bind({ dst: frontierBits, P: bitsParams.binding });
6277
6355
  for (let level2 = 0; level2 < levelsPerSubmit; level2++) {
6278
- planner.recordFinalize(pass2, 0, level2, { ...fields, firstOfSubmit: Math.min(level2, 2) });
6356
+ planner.recordFinalize(pass2, 0, level2, { ...fields, firstOfSubmit: Math.min(level2, 1) });
6279
6357
  advance.record(pass2, frontier);
6280
6358
  planner.recordFinalize(pass2, 1, level2, fields);
6281
6359
  const params = scope.params(FRONTIER_PARAMS, {
@@ -6341,6 +6419,14 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6341
6419
  });
6342
6420
  bottomUp.dispatch(pass2, boundSweep, sweepPlan, [sweepParams.offset]);
6343
6421
  }
6422
+ const degreeParams = scope.params(FRONTIER_PARAMS, { wg, n, stride: degreePlan.stride ?? wg });
6423
+ const boundDegree = nextDegree.bind({
6424
+ frontier: frontier.output,
6425
+ outDegree,
6426
+ counters,
6427
+ P: degreeParams.binding
6428
+ });
6429
+ nextDegree.dispatch(pass2, boundDegree, degreePlan, [degreeParams.offset]);
6344
6430
  frontier.swap();
6345
6431
  }
6346
6432
  batch.endPass();
@@ -7504,10 +7590,11 @@ function gridSpecFor(n, dim, tuning) {
7504
7590
  levels++;
7505
7591
  }
7506
7592
  const cells = g ** dim;
7593
+ const outsideCells = 2 ** dim;
7507
7594
  const levelOffsets = [0];
7508
7595
  let s = g;
7509
7596
  for (let level = 0; level + 1 < levels; level++) {
7510
- levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? 1 : 0));
7597
+ levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? outsideCells : 0));
7511
7598
  s /= 2;
7512
7599
  }
7513
7600
  return {
@@ -7515,7 +7602,8 @@ function gridSpecFor(n, dim, tuning) {
7515
7602
  g,
7516
7603
  levels,
7517
7604
  cells,
7518
- histWords: cells + 2,
7605
+ outsideCells,
7606
+ histWords: cells + outsideCells + 1,
7519
7607
  levelOffsets: Object.freeze(levelOffsets),
7520
7608
  pyramidCells: levelOffsets[levels - 1] + GRID_COARSEST_SIDE ** dim,
7521
7609
  deterministic: tuning.deterministic
@@ -9831,6 +9919,12 @@ function subset(merged, defaults) {
9831
9919
  }
9832
9920
  return out;
9833
9921
  }
9922
+ function recordExactRepulsion(kernel, pass, bound, plan, n, paramsOffset) {
9923
+ const passes = Math.max(1, Math.ceil(Math.ceil(n / kernel.workgroupSize) / EXACT_TILES_PER_PASS));
9924
+ for (let z = 1; z <= passes; z++) {
9925
+ kernel.dispatch(pass, bound, { ...plan, z }, [paramsOffset]);
9926
+ }
9927
+ }
9834
9928
  class RepulsionExact {
9835
9929
  /**
9836
9930
  * Holds the two compiled kernels; create() is the only caller.
@@ -9923,7 +10017,7 @@ class RepulsionExact {
9923
10017
  recordRepulsion(pass, n, paramsOffset) {
9924
10018
  const bound = this.bound(this.boundRepulsion, "recordRepulsion");
9925
10019
  const plan = plan1d(n, this.repulsion.workgroupSize, this.caps);
9926
- this.repulsion.dispatch(pass, bound, plan, [paramsOffset]);
10020
+ recordExactRepulsion(this.repulsion, pass, bound, plan, n, paramsOffset);
9927
10021
  }
9928
10022
  /**
9929
10023
  * Records K4 only.
@@ -10056,7 +10150,8 @@ class GridPyramidPlannerImpl {
10056
10150
  }
10057
10151
  const { centroid, finalize, hub, downsample } = this.kernels;
10058
10152
  const one = { x: 1, y: 1, z: 1, items: 1, stride: null };
10059
- centroid.dispatch(pass, bound.centroid, plan1d(spec.cells + 1, scope.workgroupSize, scope.caps), [paramsOffset]);
10153
+ const level0 = spec.cells + spec.outsideCells;
10154
+ centroid.dispatch(pass, bound.centroid, plan1d(level0, scope.workgroupSize, scope.caps), [paramsOffset]);
10060
10155
  finalize.dispatch(pass, bound.finalize, one, [bound.finalizeOffset]);
10061
10156
  hub.dispatchIndirect(pass, bound.hub, bound.hubArgs, 0, [paramsOffset]);
10062
10157
  this.dispatches = 3;
@@ -10129,7 +10224,7 @@ class RepulsionGrid {
10129
10224
  }
10130
10225
  /**
10131
10226
  * The model-owned buffers of the grid tier (spec 7.3; PD-11): `cellKey` / `cellVal` / `sortedKey` / `sortedIdx`
10132
- * 4n, `cellHist` / `cellStart` 4 (cells + 2) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
10227
+ * 4n, `cellHist` / `cellStart` 4 histWords (cells + 2^dim + 1) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
10133
10228
  * indirect slot, `pyramid` 16 B per pyramid cell zeroed. `hubCounters` (16 B, zeroed) is the MODEL's on every
10134
10229
  * tier (PD-14: K1 binds it on the exact tier too). n = 0 reports one node's worth of bytes (spec 3.6).
10135
10230
  * @param n - the node count
@@ -10725,7 +10820,9 @@ class ForceAtlas2Model {
10725
10820
  arcEnd: arcCountOf(core),
10726
10821
  accumulate: 0,
10727
10822
  hiEnd,
10728
- midEnd
10823
+ midEnd,
10824
+ settleFloor: SETTLE_FLOOR_UNBOUNDED
10825
+ // ForceAtlas2 does not drift after settling (issue #97)
10729
10826
  };
10730
10827
  }
10731
10828
  /**
@@ -11424,6 +11521,7 @@ class FruchtermanReingoldModel {
11424
11521
  hiEnd,
11425
11522
  midEnd,
11426
11523
  frK: resolved.k ?? 1 / Math.sqrt(n),
11524
+ settleFloor: SETTLE_FLOOR_FRACTION.fruchtermanReingold * (resolved.k ?? 1 / Math.sqrt(n)),
11427
11525
  temperature: adaptive ? FR_START_TEMPERATURE : this.temperatureAt(iteration, resolved)
11428
11526
  };
11429
11527
  }
@@ -11476,7 +11574,7 @@ class FruchtermanReingoldModel {
11476
11574
  if (stop < 2) {
11477
11575
  return;
11478
11576
  }
11479
- k3.dispatch(pass, k3Bound, bound.plan, [offset]);
11577
+ recordExactRepulsion(k3, pass, k3Bound, bound.plan, bound.n, offset);
11480
11578
  if (stop < STAGE_K5$1) {
11481
11579
  return;
11482
11580
  }
@@ -12055,6 +12153,7 @@ class SpringElectricalModel {
12055
12153
  frK: 0,
12056
12154
  temperature: 0,
12057
12155
  springLength: resolved.springLength,
12156
+ settleFloor: SETTLE_FLOOR_FRACTION.springElectrical * resolved.springLength * (SETTLE_FLOOR_REFERENCE_NODES / Math.max(n, 1)) ** 0.25,
12058
12157
  springCoefficient: resolved.springCoefficient ?? SE_DEFAULTS.springCoefficient * springSizeFactor(n),
12059
12158
  coulomb: resolved.gravity ?? SE_DEFAULTS.gravity * springSizeFactor(n),
12060
12159
  dragCoefficient: resolved.dragCoefficient,
@@ -12108,7 +12207,7 @@ class SpringElectricalModel {
12108
12207
  if (stop < 2) {
12109
12208
  return;
12110
12209
  }
12111
- k3.dispatch(pass, k3Bound, bound.plan, [offset]);
12210
+ recordExactRepulsion(k3, pass, k3Bound, bound.plan, bound.n, offset);
12112
12211
  if (stop < STAGE_K5) {
12113
12212
  return;
12114
12213
  }
@@ -12594,7 +12693,7 @@ async function calibrateLayout(ctx, options) {
12594
12693
  };
12595
12694
  }
12596
12695
  export {
12597
- A as ARC_WINDOW_ALIGN,
12696
+ J as ARC_WINDOW_ALIGN,
12598
12697
  EXACT_MAX_NODES,
12599
12698
  FA2_DEFAULTS,
12600
12699
  FR_DEFAULTS,
@@ -12602,10 +12701,10 @@ export {
12602
12701
  LAYOUT_TUNING_DEFAULTS,
12603
12702
  MAX_1D_ITEMS,
12604
12703
  MAX_WORKGROUPS_PER_DIM,
12605
- C as PASSTHROUGH_FORMAT_CODES,
12704
+ K as PASSTHROUGH_FORMAT_CODES,
12606
12705
  SE_DEFAULTS,
12607
- D as STORAGE_ALIGN,
12608
- H as WORKGROUP_SIZE,
12706
+ N as STORAGE_ALIGN,
12707
+ O as WORKGROUP_SIZE,
12609
12708
  WebGpuGraphError,
12610
12709
  bellmanFord,
12611
12710
  breadthFirstSearch,
@@ -12620,7 +12719,7 @@ export {
12620
12719
  eigenvectorCentrality,
12621
12720
  hasErrorCode,
12622
12721
  hits,
12623
- J as isSoftwareAdapter,
12722
+ Q as isSoftwareAdapter,
12624
12723
  isWebGpuGraphError,
12625
12724
  katzCentrality,
12626
12725
  pageRank,