@graphty/webgpu-graph-algorithms 0.6.5 → 0.6.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. package/README.md +38 -17
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-hzGggHeM.js → context-DiSr6eiz.js} +32 -18
  4. package/dist/chunks/context-DiSr6eiz.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/algorithms/bfs.d.ts +10 -6
  7. package/dist/src/algorithms/bfs.d.ts.map +1 -1
  8. package/dist/src/algorithms/bfs.js +32 -9
  9. package/dist/src/algorithms/bfs.js.map +1 -1
  10. package/dist/src/constants.d.ts +33 -0
  11. package/dist/src/constants.d.ts.map +1 -1
  12. package/dist/src/constants.js +33 -0
  13. package/dist/src/constants.js.map +1 -1
  14. package/dist/src/kernel/dispatch.d.ts +2 -2
  15. package/dist/src/kernel/dispatch.d.ts.map +1 -1
  16. package/dist/src/kernel/kernel.d.ts +1 -1
  17. package/dist/src/kernel/kernel.js +2 -2
  18. package/dist/src/kernel/kernel.js.map +1 -1
  19. package/dist/src/kernel/prelude.d.ts.map +1 -1
  20. package/dist/src/kernel/prelude.js +2 -1
  21. package/dist/src/kernel/prelude.js.map +1 -1
  22. package/dist/src/kernels.d.ts +10 -7
  23. package/dist/src/kernels.d.ts.map +1 -1
  24. package/dist/src/kernels.js +33 -9
  25. package/dist/src/kernels.js.map +1 -1
  26. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  27. package/dist/src/layouts/forceatlas2.js +2 -1
  28. package/dist/src/layouts/forceatlas2.js.map +1 -1
  29. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  30. package/dist/src/layouts/fruchterman-reingold.js +4 -2
  31. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  32. package/dist/src/layouts/repulsion-exact.d.ts +16 -0
  33. package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
  34. package/dist/src/layouts/repulsion-exact.js +21 -1
  35. package/dist/src/layouts/repulsion-exact.js.map +1 -1
  36. package/dist/src/layouts/repulsion-grid.d.ts +1 -1
  37. package/dist/src/layouts/repulsion-grid.js +1 -1
  38. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  39. package/dist/src/layouts/spring-electrical.js +6 -2
  40. package/dist/src/layouts/spring-electrical.js.map +1 -1
  41. package/dist/src/primitives/frontier.d.ts +1 -0
  42. package/dist/src/primitives/frontier.d.ts.map +1 -1
  43. package/dist/src/primitives/frontier.js +1 -0
  44. package/dist/src/primitives/frontier.js.map +1 -1
  45. package/dist/src/primitives/grid-pyramid.d.ts +4 -4
  46. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  47. package/dist/src/primitives/grid-pyramid.js +4 -3
  48. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  49. package/dist/src/primitives/grid.d.ts +13 -10
  50. package/dist/src/primitives/grid.d.ts.map +1 -1
  51. package/dist/src/primitives/grid.js +10 -7
  52. package/dist/src/primitives/grid.js.map +1 -1
  53. package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
  54. package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
  55. package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
  56. package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
  57. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
  58. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
  59. package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
  60. package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
  61. package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
  62. package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
  63. package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
  64. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
  65. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
  66. package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
  67. package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
  68. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
  69. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
  70. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
  71. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
  72. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
  73. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
  74. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
  75. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
  76. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +21 -20
  77. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
  78. package/dist/src/wgsl/frontier-finalize.wgsl.js +26 -25
  79. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
  80. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
  81. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
  82. package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
  83. package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
  84. package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
  85. package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
  86. package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
  87. package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
  88. package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
  89. package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
  90. package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
  91. package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
  92. package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
  93. package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
  94. package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
  95. package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
  96. package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
  97. package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
  98. package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
  99. package/dist/src/wgsl/histogram.wgsl.js +1 -1
  100. package/dist/webgpu-graph-algorithms.js +124 -40
  101. package/dist/webgpu-graph-algorithms.js.map +1 -1
  102. package/package.json +3 -3
  103. package/src/algorithms/bfs.ts +33 -9
  104. package/src/algorithms/pagerank.ts +19 -5
  105. package/src/algorithms/power-iteration.ts +8 -2
  106. package/src/constants.ts +35 -0
  107. package/src/kernel/dispatch.ts +2 -2
  108. package/src/kernel/kernel.ts +2 -2
  109. package/src/kernel/prelude.ts +2 -0
  110. package/src/kernels.ts +35 -9
  111. package/src/layouts/forceatlas2.ts +2 -0
  112. package/src/layouts/fruchterman-reingold.ts +4 -1
  113. package/src/layouts/repulsion-exact.ts +29 -1
  114. package/src/layouts/repulsion-grid.ts +1 -1
  115. package/src/layouts/spring-electrical.ts +8 -1
  116. package/src/memory/residency.ts +14 -4
  117. package/src/primitives/frontier.ts +2 -0
  118. package/src/primitives/grid-pyramid.ts +6 -5
  119. package/src/primitives/grid.ts +17 -12
  120. package/src/wgsl/advance-expand.wgsl.ts +4 -3
  121. package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
  122. package/src/wgsl/bfs-fused.wgsl.ts +6 -6
  123. package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
  124. package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
  125. package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
  126. package/src/wgsl/frontier-finalize.wgsl.ts +26 -25
  127. package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
  128. package/src/wgsl/grid-centroid.wgsl.ts +3 -2
  129. package/src/wgsl/grid-downsample.wgsl.ts +1 -1
  130. package/src/wgsl/grid-far-field.wgsl.ts +15 -8
  131. package/src/wgsl/grid-near-field.wgsl.ts +6 -2
  132. package/src/wgsl/histogram.wgsl.ts +1 -1
  133. package/dist/chunks/context-hzGggHeM.js.map +0 -1
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@graphty/webgpu-graph-algorithms",
3
- "version": "0.6.5",
3
+ "version": "0.6.6",
4
4
  "description": "WebGPU-accelerated graph algorithms and layouts over the @graphty/graph-format snapshot, for Node (Dawn) and browsers",
5
5
  "author": "Adam Powers <apowers@ato.ms>",
6
6
  "type": "module",
@@ -93,8 +93,8 @@
93
93
  "vite": "^7.0.5",
94
94
  "vitest": "^3.2.4",
95
95
  "webgpu": "0.4.0",
96
- "@graphty/algorithms": "^2.0.3",
97
- "@graphty/layout": "^1.9.1"
96
+ "@graphty/algorithms": "^2.0.4",
97
+ "@graphty/layout": "^1.10.0"
98
98
  },
99
99
  "scripts": {
100
100
  "build": "tsc -p tsconfig.build.json",
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * Breadth-first search on the device (design 8.4, 3.3 line 807, 9.7; P8-T6 / P8-T7 / P8-T8, the P8 plan's PD-5 /
3
3
  * PD-6 / PD-7 / PD-14 / PD-18 / PD-21 / PD-23 / PD-24 / PD-26 / DEP-P8-B): the direction-optimizing traversal over
4
- * the `Frontier` of P8-T4, every per-level choice made ON THE DEVICE. Every level is eight recorded dispatches, all
4
+ * the `Frontier` of P8-T4, every per-level choice made ON THE DEVICE. Every level is nine recorded dispatches, all
5
5
  * DIRECT (2026-09-25, docs/decisions/G8.md G8-F5: Dawn validates every `dispatchWorkgroupsIndirect` with an internal
6
6
  * clamp pass costing about 0.4 ms of device time whether or not it dispatches anything, in Node and in Chromium
7
7
  * alike, and that was 97 % of a traversal's wall time) -- `frontier-finalize` role 0 (the level boundary: rotates
@@ -13,11 +13,15 @@
13
13
  * per frontier entry with the same claim inline and no edge queue; the fused level and the retry alike), then the
14
14
  * bottom-up trio: a `fill` zeroing the frontier bitset, `bfs-bitset-build` setting the frontier's bits, and
15
15
  * `bfs-bottom-up` sweeping the unvisited list over the REVERSE core (each unvisited vertex reads its in-neighbours
16
- * until the first one in the bitset and claims itself). Every kernel is a grid-stride dispatch of a host-planned
17
- * grid (`planGridStride`) that loops to its count word and reads the path word first, so exactly one path does the
18
- * level's work and the others cost one uniform load per workgroup. The unvisited set Beamer's test is against (PD-18) is rebuilt exactly once per submit, before
19
- * the levels, by `bfs-unvisited-flags` plus `compact` over an iota queue, and maintained between rebuilds by
20
- * subtraction inside the selector (whose JSDoc states the boundary rule). The host records `MAX_LEVELS_PER_SUBMIT`
16
+ * until the first one in the bitset and claims itself), and last `bfs-next-degree` (the out-degree sum of whatever
17
+ * the level claimed, into the block's `nextDegreeSum` word: Beamer's m_f for the NEXT boundary, measured on the
18
+ * frontier that boundary decides for -- issue #391, which found the previous proxy, the degree of the frontier just
19
+ * expanded, one level stale and missing the switch at the level holding two thirds of the 1M / 10M R-MAT's arcs).
20
+ * Every kernel is a grid-stride dispatch of a host-planned grid (`planGridStride`) that loops to its count word and
21
+ * reads the path word first, so exactly one path does the level's work and the others cost one uniform load per
22
+ * workgroup. The unvisited set Beamer's test is against (PD-18) is rebuilt exactly once per submit, before the
23
+ * levels, by `bfs-unvisited-flags` plus `compact` over an iota queue, and maintained between rebuilds by subtraction
24
+ * inside the selector (whose JSDoc states the boundary rule; both words are exact at every boundary). The host records `MAX_LEVELS_PER_SUBMIT`
21
25
  * levels into ONE command buffer, submits, and reads four bytes, the `done` word (PD-7): a road network has
22
26
  * thousands of levels and a per-level `mapAsync` would be slower than the CPU. A level recorded past the end is a
23
27
  * no-op (its boundary finds `done` set, writes path 0 and moves no counter), so the loop needs no diameter and
@@ -79,6 +83,9 @@ import { type AlgorithmScope, algorithmScope } from "./scope.js";
79
83
 
80
84
  const ALGORITHM = "breadthFirstSearch";
81
85
 
86
+ /** Workgroups of the `bfs-next-degree` grid (issue #391): enough to sum n out-degrees by grid stride, few enough that a high-diameter traversal does not pay a full-width reduction per level. */
87
+ const NEXT_DEGREE_MAX_GROUPS = 128;
88
+
82
89
  /**
83
90
  * Params slots of the ring, COUNTED per run (P8-T12), because `UniformRing.reserve` wraps to slot 0 when a submit's
84
91
  * records outrun the ring and silently overwrites a record the submit still reads; the ring's `overruns` counts
@@ -87,8 +94,9 @@ const ALGORITHM = "breadthFirstSearch";
87
94
  * `bfs-contract`, the bits `fill`, `bfs-bitset-build`) and four re-issued once per arc window (`advance-expand`,
88
95
  * `bfs-fused`, the fused retry, `bfs-bottom-up`). The driver as built writes fewer -- the contract and the bitset
89
96
  * build share one window-free record, a window's two fused dispatches share one, the bits fill has one record per
90
- * submit, and a directed snapshot's reverse view is one window whatever the forward core's count -- so at most
91
- * `3 + 3 x windows` per level, and the bound holds with room. The 16 covers the per-submit rebuild
97
+ * submit, `bfs-next-degree` has one window-free record of its own (issue #391), and a directed snapshot's reverse
98
+ * view is one window whatever the forward core's count -- so at most `4 + 3 x windows` per level, and the bound
99
+ * holds with room. The 16 covers the per-submit rebuild
92
100
  * (`bfs-unvisited-flags` and `compact`, whose scan is at most 9 dispatches for any n below 2^32, so 10, plus the
93
101
  * bits fill). The result batch flushes in its own submit, so the ring must hold IT too, and its size grows with `n`,
94
102
  * not with the cadence: the iota `fill`, the radix sort's four passes of one record plus its scan's `2 x levels - 1`
@@ -339,6 +347,7 @@ export async function bfsWithTuning(
339
347
  const bitset = await ctx.pipelines.kernel(kernelSpec("bfs-bitset-build"));
340
348
  const bottomUp = await ctx.pipelines.kernel(kernelSpec("bfs-bottom-up", graphOverrides(reverse, null)));
341
349
  const unvisited = await ctx.pipelines.kernel(kernelSpec("bfs-unvisited-flags"));
350
+ const nextDegree = await ctx.pipelines.kernel(kernelSpec("bfs-next-degree"));
342
351
  const pred = await ctx.pipelines.kernel(kernelSpec("sssp-pred", { ...graphOverrides(core, null), MODE: 1 }));
343
352
  const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
344
353
  const sort = await prepareRadixSort(scope);
@@ -352,6 +361,11 @@ export async function bfsWithTuning(
352
361
  // plan's GROUP count as its stride (planGridStride's cap applies to the groups)
353
362
  const levelPlan = planGridStride(Math.max(n, frontier.edgeCapacity), wg, ctx.caps);
354
363
  const sweepPlan = planGridStride(n, wg, ctx.caps);
364
+ // `bfs-next-degree` sums at most n words and is dispatched on EVERY level, so its fixed cost -- one
365
+ // workgroup reduction (six barriers) and one atomic per workgroup -- is paid 1,999 times on the
366
+ // 1000 x 1000 grid. A capped grid pays it 128 times a level instead of ceil(n / wg): on the card the
367
+ // grid row went from 436 to 356 ms and neither R-MAT row moved.
368
+ const degreePlan = planGridStride(n, wg, ctx.caps, NEXT_DEGREE_MAX_GROUPS);
355
369
  const fusedPlan = planGridStride(n * wg, wg, ctx.caps);
356
370
  const bitsPlan = plan1d(bitsWords, wg, ctx.caps);
357
371
  const recordFill = (pass: GPUComputePassEncoder, dst: Binding, value: number, mode: 0 | 1): void => {
@@ -415,7 +429,7 @@ export async function bfsWithTuning(
415
429
  const bitsParams = scope.params(FILL_PARAMS, { count: bitsWords, value: 0, mode: 0, pad0: 0 });
416
430
  const boundBitsFill = fill.bind({ dst: frontierBits, P: bitsParams.binding });
417
431
  for (let level = 0; level < levelsPerSubmit; level++) {
418
- planner.recordFinalize(pass, 0, level, { ...fields, firstOfSubmit: Math.min(level, 2) });
432
+ planner.recordFinalize(pass, 0, level, { ...fields, firstOfSubmit: Math.min(level, 1) });
419
433
  advance.record(pass, frontier);
420
434
  planner.recordFinalize(pass, 1, level, fields);
421
435
  // one window-free record serves the contract and the bitset build, which read no arc; the kernels that
@@ -489,6 +503,16 @@ export async function bfsWithTuning(
489
503
  });
490
504
  bottomUp.dispatch(pass, boundSweep, sweepPlan, [sweepParams.offset]);
491
505
  }
506
+ // the next frontier's out-degree sum (issue #391): whichever path claimed, the vertices are in the output
507
+ // queue now, and the next boundary reads word 25 as Beamer's m_f for the frontier it is about to expand
508
+ const degreeParams = scope.params(FRONTIER_PARAMS, { wg, n, stride: degreePlan.stride ?? wg });
509
+ const boundDegree = nextDegree.bind({
510
+ frontier: frontier.output,
511
+ outDegree,
512
+ counters,
513
+ P: degreeParams.binding,
514
+ });
515
+ nextDegree.dispatch(pass, boundDegree, degreePlan, [degreeParams.offset]);
492
516
  frontier.swap();
493
517
  }
494
518
  batch.endPass();
@@ -34,8 +34,8 @@ import { algorithmScope } from "./scope.js";
34
34
 
35
35
  /** Iterations per submit (spec 8.2: k = 8). */
36
36
  const PR_BATCH = 8;
37
- /** Params slots of one batch: per iteration one PrParams (shared by pr-scale and pr-finalize) and one SpmvParams, plus the normaliser's RangeParams; a smaller ring wraps onto a slot the same batch still reads. */
38
- const RING_SLOTS = 2 * PR_BATCH + 1;
37
+ /** Params slots of one batch: per iteration one PrParams (shared by pr-scale and pr-finalize) and one SpmvParams, plus the normaliser's RangeParams and the last batch's convergence check; a smaller ring wraps onto a slot the same batch still reads. */
38
+ const RING_SLOTS = 2 * PR_BATCH + 2;
39
39
 
40
40
  /**
41
41
  * Validates `options.dest` for a score result of `n` elements.
@@ -175,6 +175,8 @@ async function run(
175
175
  const groups = groupsOf(scalePlan);
176
176
  const partialsBytes = PR_PARTIAL.byteLength * (1 + groups);
177
177
  const partials = scope.scratch(partialsBytes, "partials");
178
+ // the header as the last pull left it, before the convergence check of the last batch overwrites danglingMass
179
+ const lastHeader = scope.scratch(PR_PARTIAL.byteLength, "lastHeader");
178
180
  if (personalization !== null) {
179
181
  uploaded = ctx.residency.array(personalization, `${algorithm}/personalization`);
180
182
  }
@@ -203,17 +205,21 @@ async function run(
203
205
  const xNormBinding = bindingOf(xNorm, bytes);
204
206
  const outWeightSumBinding = bindingOf(outWeightSum, bytes);
205
207
  const partialsBinding = bindingOf(partials, partialsBytes);
208
+ const lastHeaderBinding = bindingOf(lastHeader, PR_PARTIAL.byteLength);
206
209
  const coefficients = { alpha, beta: 1 - alpha, uniformP: 1 / n };
207
210
  let cur = 0;
208
211
  let iterationsRun = 0;
209
212
  for (;;) {
210
213
  const k = Math.min(PR_BATCH, maxIterations - iterationsRun);
211
214
  const batch = new CommandBatch(ctx, algorithm);
212
- const pass = batch.pass("iterations");
215
+ let pass = batch.pass("iterations");
213
216
  if (iterationsRun === 0) {
214
217
  normaliser.record(pass, weightedCore, outWeightSumBinding);
215
218
  }
216
- for (let i = 0; i < k; i++) {
219
+ const last = iterationsRun + k === maxIterations;
220
+ // the scale + finalize of iteration i measures the error of iteration i - 1 (PD-9), so the last batch
221
+ // runs them once more, with no pull, to measure the error of iteration maxIterations itself
222
+ for (let i = 0; i < k + (last ? 1 : 0); i++) {
217
223
  const params = scope.params(PR_PARAMS, {
218
224
  n,
219
225
  groups,
@@ -222,6 +228,10 @@ async function run(
222
228
  convergeThreshold: tolerance * n,
223
229
  });
224
230
  const other = 1 - cur;
231
+ if (i === k) {
232
+ batch.copy(partialsBinding, lastHeaderBinding, PR_PARTIAL.byteLength);
233
+ pass = batch.pass("convergence");
234
+ }
225
235
  const scaleBound = scale.bind({
226
236
  rankIn: rank[cur],
227
237
  rankPrev: rank[other],
@@ -233,6 +243,9 @@ async function run(
233
243
  scale.dispatch(pass, scaleBound, scalePlan, [params.offset]);
234
244
  const finalizeBound = finalize.bind({ partials: partialsBinding, P: params.binding });
235
245
  finalize.dispatch(pass, finalizeBound, finalizePlan, [params.offset]);
246
+ if (i === k) {
247
+ break;
248
+ }
236
249
  pull.record(
237
250
  pass,
238
251
  weightedRev,
@@ -248,6 +261,7 @@ async function run(
248
261
  }
249
262
  batch.endPass();
250
263
  const headerRequest = batch.readback(partials, 0, PR_PARTIAL.byteLength);
264
+ const lastHeaderRequest = last ? batch.readback(lastHeader, 0, PR_PARTIAL.byteLength) : headerRequest;
251
265
  const scoresRequest = batch.readback(rank[cur].buffer, 0, bytes);
252
266
  scope.flush();
253
267
  const submitted = batch.submit();
@@ -271,7 +285,7 @@ async function run(
271
285
  scores,
272
286
  iterations: converged ? firstConverged : iterationsRun,
273
287
  converged,
274
- danglingMass: folded.danglingMass as number,
288
+ danglingMass: PR_PARTIAL.read(new DataView(back), lastHeaderRequest.offset).danglingMass as number,
275
289
  precision: "f32",
276
290
  };
277
291
  }
@@ -33,7 +33,7 @@ import { algorithmScope } from "./scope.js";
33
33
 
34
34
  /** Iterations per submit (spec 8.2: k = 8). */
35
35
  const BATCH = 8;
36
- /** Params slots of one batch: four blocks per iteration times the batch, plus a margin (the plan's `4 * 8 + 8`). */
36
+ /** Params slots of one batch: four blocks per iteration times the batch, plus a margin (the plan's `4 * 8 + 8`) that also holds the last batch's convergence check. */
37
37
  const RING_SLOTS = 4 * BATCH + 8;
38
38
 
39
39
  /**
@@ -210,7 +210,10 @@ export async function runPowerIteration(
210
210
  const k = Math.min(BATCH, config.maxIterations - iterationsRun);
211
211
  const batch = new CommandBatch(ctx, config.label);
212
212
  const pass = batch.pass("iterations");
213
- for (let i = 0; i < k; i++) {
213
+ // the scale + finalize of iteration i measures the error of iteration i - 1 (PD-9), so the last batch
214
+ // runs them once more, with no apply and no pull, to measure the error of iteration maxIterations itself
215
+ const last = iterationsRun + k === config.maxIterations;
216
+ for (let i = 0; i < k + (last ? 1 : 0); i++) {
214
217
  const iteration = iterationsRun + i + 1;
215
218
  const params = scope.params(PR_PARAMS, {
216
219
  n,
@@ -234,6 +237,9 @@ export async function runPowerIteration(
234
237
  };
235
238
  scaleNorm.dispatch(pass, scaleNorm.bind(scaleBindings), scalePlan, [params.offset]);
236
239
  finalize.dispatch(pass, finalize.bind({ partials, P: params.binding }), finalizePlan, [params.offset]);
240
+ if (i === k) {
241
+ break;
242
+ }
237
243
  if (scaleApply !== null) {
238
244
  scaleApply.dispatch(pass, scaleApply.bind(scaleBindings), scalePlan, [params.offset]);
239
245
  }
package/src/constants.ts CHANGED
@@ -130,6 +130,13 @@ export const FA2_DISTANCE_FLOOR = 0.01;
130
130
  export const FA2_DISTANCE_FLOOR_SQ = 0.0001;
131
131
  /** The coincident threshold `d^2 < 1e-8` of spec 7.2. */
132
132
  export const FA2_COINCIDENT_SQ = 1e-8;
133
+ /**
134
+ * The most WG-node tiles one K3 (`fa2-repulsion-exact`) dispatch sums per invocation (issue #87). llvmpipe runs at
135
+ * most 65,535 loop iterations per shader invocation, counted over every loop together, and then quietly breaks
136
+ * out of each loop; one tile costs WG + 2 of them (258 at WG 256), so a single pass lost every node past
137
+ * j = 65,027. 128 tiles is 33,024 iterations, half the budget; a larger exact run records ceil(tiles / 128) passes.
138
+ */
139
+ export const EXACT_TILES_PER_PASS = 128;
133
140
  /** Bits of Fa2Params.flags (contract 4.4). */
134
141
  export const FA2_FLAG_FIRST = 1;
135
142
  /** Fa2Params.flags bit: the Fruchterman-Reingold temperature is the adaptive one in the state block, not the uniform's (the `cooling: "adaptive"` option). */
@@ -202,6 +209,34 @@ export const SE_DEFAULTS: Readonly<{
202
209
  iterationsPerStep: 1,
203
210
  maxInFlight: 2,
204
211
  });
212
+ /**
213
+ * The absolute settle floor of the shared settle rule (spec 7.17; issue #97): an iteration counts toward `settled` only
214
+ * when its mean displacement is at most `settleThreshold x rmsRadius` AND at most this fraction of the model's length
215
+ * unit -- `springLength` for spring-electrical (scaled by node count, see SETTLE_FLOOR_REFERENCE_NODES), `k` for
216
+ * Fruchterman-Reingold. The relative rule alone reported a spring
217
+ * layout settled while it still grew (6 % over 1,000 iterations at 10k nodes). ForceAtlas2 writes
218
+ * `SETTLE_FLOOR_UNBOUNDED` instead: it does not drift after settling, and its per-iteration jitter grows with n, so any
219
+ * fixed floor only delays or blocks its stop. The values are measured:
220
+ * design/decisions/2026-09-24-settle-rule-has-an-absolute-floor.md.
221
+ */
222
+ export const SETTLE_FLOOR_FRACTION: Readonly<{ springElectrical: number; fruchtermanReingold: number }> = Object.freeze(
223
+ {
224
+ springElectrical: 3e-3,
225
+ fruchtermanReingold: 2e-3,
226
+ },
227
+ );
228
+ /**
229
+ * The node count at which the spring-electrical floor is exactly `SETTLE_FLOOR_FRACTION.springElectrical x
230
+ * springLength`; at `n` nodes it is that times `(SETTLE_FLOOR_REFERENCE_NODES / n)^(1/4)`. A spring layout's rms radius
231
+ * grows about as n^(1/4) in springLengths (3.4 at 150 nodes, 6.4 at 2,000, 10.4 at 10,000), so the relative half of the
232
+ * rule loosens with size while the floor tightens with it: on a small graph the floor sits above the relative threshold
233
+ * and the relative rule decides alone, as before issue #97; on a large one the floor binds, which is where the relative
234
+ * rule let an expanding layout stop. A fixed floor bound the 150-node story graph too, where the grid tier's jitter sits
235
+ * at the relative threshold, and nearly doubled its settle (427 -> 829 iterations on the RTX 4070 SUPER).
236
+ */
237
+ export const SETTLE_FLOOR_REFERENCE_NODES = 2000;
238
+ /** The settle floor that never binds: the largest finite f32, 0x1.fffffep+127 (ForceAtlas2's `settleFloor`, issue #97). */
239
+ export const SETTLE_FLOOR_UNBOUNDED = 2 ** 128 - 2 ** 104;
205
240
  /** The smallest finest grid side `G` (spec 7.7 geometry table: `clamp(nextPow2(2 n^(1/dim)), 8, gridMax)`; P4 PD-9). */
206
241
  export const GRID_MIN_SIDE = 8;
207
242
  /** The coarsest pyramid level's side (spec 7.7: "levels (coarsest 4 per axis)", `levels = log2(G / 4) + 1`). */
@@ -9,11 +9,11 @@ import { MAX_WORKGROUPS_PER_DIM } from "../constants.js";
9
9
  import { WebGpuGraphError } from "../errors.js";
10
10
  import { type PlanCaps } from "../types/context.js";
11
11
 
12
- /** A dispatch shape (spec 5.2). `stride` is the grid-stride step (null for plain 1D / 2D plans). */
12
+ /** A dispatch shape (spec 5.2). `stride` is the grid-stride step (null for plain 1D / 2D plans). `z` is 1 from every planner; only K3's pass split sets it (issue #87). */
13
13
  export interface DispatchPlan {
14
14
  readonly x: number;
15
15
  readonly y: number;
16
- readonly z: 1;
16
+ readonly z: number;
17
17
  readonly items: number;
18
18
  readonly stride: number | null;
19
19
  }
@@ -245,7 +245,7 @@ export class Kernel {
245
245
  }
246
246
 
247
247
  /**
248
- * setPipeline + setBindGroup for every group (dynamic offsets in dynamicGroups order) + dispatchWorkgroups(plan.x, plan.y, 1); a plan with x === 0 records nothing (spec 5.6).
248
+ * setPipeline + setBindGroup for every group (dynamic offsets in dynamicGroups order) + dispatchWorkgroups(plan.x, plan.y, plan.z); a plan with x === 0 records nothing (spec 5.6).
249
249
  * PLAN DECISION: one dynamic offset per dynamic GROUP, replicated over every uniform binding of that group (every
250
250
  * P1-P3 kernel has exactly one params uniform per group); absent offsets mean 0; a BoundKernel of another kernel
251
251
  * or an offset list of the wrong length is E_INVALID_ARGUMENT.
@@ -268,7 +268,7 @@ export class Kernel {
268
268
  return;
269
269
  }
270
270
  this.setUp(pass, bound, dynamicOffsets);
271
- pass.dispatchWorkgroups(plan.x, plan.y, 1);
271
+ pass.dispatchWorkgroups(plan.x, plan.y, plan.z);
272
272
  }
273
273
 
274
274
  /**
@@ -12,6 +12,7 @@
12
12
  import { INVALID_INDEX } from "@graphty/graph-format";
13
13
 
14
14
  import {
15
+ EXACT_TILES_PER_PASS,
15
16
  F32_INF_BITS,
16
17
  FA2_COINCIDENT_SQ,
17
18
  FA2_DISTANCE_FLOOR,
@@ -54,6 +55,7 @@ const INVALID_INDEX: u32 = ${INVALID_INDEX}u;
54
55
  const U32_MAX: u32 = ${U32_MAX}u;
55
56
  const F32_INF_BITS: u32 = ${F32_INF_BITS}u;
56
57
  const MAX_WORKGROUPS_PER_DIM: u32 = ${MAX_WORKGROUPS_PER_DIM}u;
58
+ const EXACT_TILES_PER_PASS: u32 = ${EXACT_TILES_PER_PASS}u;
57
59
  const FA2_DIST_FLOOR: f32 = ${wgslF32Literal(FA2_DISTANCE_FLOOR)};
58
60
  const FA2_DIST_FLOOR_SQ: f32 = ${wgslF32Literal(FA2_DISTANCE_FLOOR_SQ)};
59
61
  const FA2_COINCIDENT_SQ: f32 = ${wgslF32Literal(FA2_COINCIDENT_SQ)};
package/src/kernels.ts CHANGED
@@ -28,6 +28,7 @@ import { bfsBitsetBuildWgsl } from "./wgsl/bfs-bitset-build.wgsl.js";
28
28
  import { bfsBottomUpWgsl } from "./wgsl/bfs-bottom-up.wgsl.js";
29
29
  import { bfsContractWgsl } from "./wgsl/bfs-contract.wgsl.js";
30
30
  import { bfsFusedWgsl } from "./wgsl/bfs-fused.wgsl.js";
31
+ import { bfsNextDegreeWgsl } from "./wgsl/bfs-next-degree.wgsl.js";
31
32
  import { bfsUnvisitedFlagsWgsl } from "./wgsl/bfs-unvisited-flags.wgsl.js";
32
33
  import { closenessReduceWgsl } from "./wgsl/closeness-reduce.wgsl.js";
33
34
  import { closenessSweepWgsl } from "./wgsl/closeness-sweep.wgsl.js";
@@ -111,6 +112,7 @@ export type KernelId =
111
112
  | "bfs-bottom-up"
112
113
  | "bfs-bitset-build"
113
114
  | "bfs-unvisited-flags"
115
+ | "bfs-next-degree"
114
116
  | "sssp-relax"
115
117
  | "bf-relax"
116
118
  | "closeness-sweep"
@@ -162,7 +164,7 @@ export const FILL_PARAMS: UniformBlock = UniformBlock.define("FillParams", [
162
164
  ["pad0", "u32"],
163
165
  ]);
164
166
 
165
- /** `Fa2Params` (uniform, 128 B; spec 7.3): the per-iteration ForceAtlas2 parameters -- `n` @0, `dim` @4, `flags` @8 (bit 0 = FA2_FLAG_FIRST), `tierStart` @12, `tierEnd` @16, `iterationIndex` @20, `seed` @24, `nearMax` @28, `scalingRatio` @32, `gravity` @36, `jitterTolerance` @40, `scale` @44, `center` @48 (xyz, w 0), `settleThreshold` @64, `extentFactor` @68, `gridMax` @72, `levels` @76, `arcBase` @80 / `arcEnd` @84 (the bound arc window of K2, 0 and arcCount in the layout), `accumulate` @88 (1 combines into `force`: the windowed pattern), `hiEnd` @92 / `midEnd` @124 (the degreeOrder tier boundaries, PD-7; both 0 without a permutation); the P5 model fields (PD-3): `frK` @96 (the FR optimal distance), `temperature` @100 (the FR temperature of this iteration), `springLength` @104, `springCoefficient` @108, `coulomb` @112 (ngraph's `gravity`, negative repels), `dragCoefficient` @116, `timeStep` @120; 128 B. */
167
+ /** `Fa2Params` (uniform, 144 B; spec 7.3): the per-iteration ForceAtlas2 parameters -- `n` @0, `dim` @4, `flags` @8 (bit 0 = FA2_FLAG_FIRST), `tierStart` @12, `tierEnd` @16, `iterationIndex` @20, `seed` @24, `nearMax` @28, `scalingRatio` @32, `gravity` @36, `jitterTolerance` @40, `scale` @44, `center` @48 (xyz, w 0), `settleThreshold` @64, `extentFactor` @68, `gridMax` @72, `levels` @76, `arcBase` @80 / `arcEnd` @84 (the bound arc window of K2, 0 and arcCount in the layout), `accumulate` @88 (1 combines into `force`: the windowed pattern), `hiEnd` @92 / `midEnd` @124 (the degreeOrder tier boundaries, PD-7; both 0 without a permutation); the P5 model fields (PD-3): `frK` @96 (the FR optimal distance), `temperature` @100 (the FR temperature of this iteration), `springLength` @104, `springCoefficient` @108, `coulomb` @112 (ngraph's `gravity`, negative repels), `dragCoefficient` @116, `timeStep` @120; `settleFloor` @128 (the absolute bound on the mean displacement of a settled iteration, in layout units: issue #97); 144 B. */
166
168
  export const FA2_PARAMS: UniformBlock = UniformBlock.define("Fa2Params", [
167
169
  ["n", "u32"],
168
170
  ["dim", "u32"],
@@ -193,6 +195,7 @@ export const FA2_PARAMS: UniformBlock = UniformBlock.define("Fa2Params", [
193
195
  ["dragCoefficient", "f32"],
194
196
  ["timeStep", "f32"],
195
197
  ["midEnd", "u32"],
198
+ ["settleFloor", "f32"],
196
199
  ]);
197
200
 
198
201
  /** `Fa2State` (storage, padded to STATE_HEADER_BYTES = 256; spec 7.3): the device-resident controller state the finalize kernels write and the host reads back for stats -- `speed` @0, `speedEfficiency` @4, `swing` @8, `traction` @12, `centroid` @16, `rmsRadius` @32, `radius` @36, `meanDisplacement` @40, `iteration` @44, `min` @48, `max` @64, `gridMin` @80 (P4), `eps` @96 (P4), `settledCount` @100, `outsideGrid` @104 (P4), `maxCellOccupancy` @108 (P4), `temperature` @112 (FR, written by K1 under STATS_MODE 1), `kineticEnergy` @116 (the preset, K1 under STATS_MODE 2), `frEnergy` @120 / `frProgress` @124 (the FR adaptive cooling), `invCellSize` @128 (P4, PD-10: `1 / cellSize`, written by K1 beside `cellSize` in `gridMin.w`; G1 multiplies by it so every key is bitwise reproducible), `reserved0` @132 (f32), `reserved1` @136 (vec2f), `reserved2` .. `reserved8` @144 .. @240. */
@@ -386,8 +389,11 @@ export const COMPACT_PARAMS: UniformBlock = UniformBlock.define("CompactParams",
386
389
  * submit), `arcsScanned` @64, `fusedLevels` @68, `twoPhaseLevels` @72, `bottomUpLevels` @76, `farCount` @80,
387
390
  * `nextFarCount` @84, `thresholdBits` @88, `deltaBits` @92 (P8-T9), `path` @96 (what the level's kernels run, written
388
391
  * by the selector: 0 nothing, 1 two-phase, 2 fused, 3 bottom-up, 4 the fused retry, 5 a near SSSP round, 6 a far
389
- * one; every level kernel is a direct dispatch that reads it first -- G8-F5). The words nothing writes before
390
- * P8-T8 / P8-T9 are declared now because the byte layout is what the single result copy decodes.
392
+ * one; every level kernel is a direct dispatch that reads it first -- G8-F5), `nextDegreeSum` @100 (issue #391: the
393
+ * out-degree sum of the vertices the level claimed, accumulated by `bfs-next-degree` at the end of every level and
394
+ * read, subtracted and zeroed by the next boundary -- Beamer's m_f measured on the frontier the boundary decides
395
+ * for, not on the one it has just expanded). The words nothing writes before P8-T8 / P8-T9 are declared now because
396
+ * the byte layout is what the single result copy decodes.
391
397
  */
392
398
  export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
393
399
  "FrontierCounters",
@@ -417,6 +423,7 @@ export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
417
423
  ["thresholdBits", "u32"],
418
424
  ["deltaBits", "u32"],
419
425
  ["path", "u32"],
426
+ ["nextDegreeSum", "u32"],
420
427
  ],
421
428
  { layout: "storage" },
422
429
  );
@@ -427,9 +434,9 @@ export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
427
434
  * `alpha` @8, `beta` @12 (Beamer's thresholds, P8-T8), `fusedMax` @16, `edgeCapacity` @20, `maxDepth` @24, `n` @28,
428
435
  * `mode` @32 (BFS: 0 auto, 1 top-down only; `sssp-pred`: the PD-27 key rule), `cutoffBits` @36, `arcBase` @40,
429
436
  * `arcEnd` @44 (the bound arc window), `predKind` @48 (0 arc, 1 node), `bitsBase` @52, `source` @56, `stride` @60
430
- * (a grid-stride plan's stride), `firstOfSubmit` @64 (the boundary's index inside its submit, clamped to 2: the
431
- * unvisited-count subtraction runs at >= 1, the degree-sum one at >= 2), `iteration` @68 (an `sssp-pred` hop pass,
432
- * P8-T9), `pad1` @72, `pad2` @76. The `slotBase` field that once addressed the selector's indirect slots went with
437
+ * (a grid-stride plan's stride), `firstOfSubmit` @64 (the boundary's index inside its submit, clamped to 1: both
438
+ * the unvisited-count and the unvisited-degree-sum subtraction run at >= 1, issue #391), `iteration` @68 (an
439
+ * `sssp-pred` hop pass, P8-T9), `pad1` @72, `pad2` @76. The `slotBase` field that once addressed the selector's indirect slots went with
433
440
  * the slots (2026-09-25); `pad2` keeps the block an explicit 80 bytes, the way every block here is padded.
434
441
  */
435
442
  export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams", [
@@ -939,7 +946,7 @@ const RADIX_SCATTER: KernelEntry = {
939
946
  phase: "P4",
940
947
  };
941
948
 
942
- /** `grid-cell-key` (G1, spec 7.7; P4-T8, PD-10): the finest cell key of every node, `floor((p - gridMin) * invCellSize)` linearised, or the outside pseudo-cell `G^dim`; `cellVal[i] = i`; 4 storage bindings (the state read-only: K1 writes it). */
949
+ /** `grid-cell-key` (G1, spec 7.7; P4-T8, PD-10): the finest cell key of every node, `floor((p - gridMin) * invCellSize)` linearised, or the outside pseudo-cell of its orthant `G^dim + orthant` (issue #90); `cellVal[i] = i`; 4 storage bindings (the state read-only: K1 writes it). */
943
950
  const GRID_CELL_KEY: KernelEntry = {
944
951
  id: "grid-cell-key",
945
952
  body: gridCellKeyWgsl,
@@ -958,7 +965,7 @@ const GRID_CELL_KEY: KernelEntry = {
958
965
  phase: "P4",
959
966
  };
960
967
 
961
- /** `grid-centroid` (G4, spec 7.7; P4-T9, PD-13): thread per finest cell (the pseudo-cell included), the serial mass-weighted sum in sorted order into level 0, the occupancy max into `hubCounters[1]`, hub cells (> GRID_HUB_CELL) appended to `hubList`; 6 storage bindings. */
968
+ /** `grid-centroid` (G4, spec 7.7; P4-T9, PD-13): thread per finest cell (the 2^dim orthant pseudo-cells included), the serial mass-weighted sum in sorted order into level 0, the occupancy max into `hubCounters[1]`, hub cells (> GRID_HUB_CELL) appended to `hubList`; 6 storage bindings. */
962
969
  const GRID_CENTROID: KernelEntry = {
963
970
  id: "grid-centroid",
964
971
  body: gridCentroidWgsl,
@@ -1013,7 +1020,7 @@ const GRID_DOWNSAMPLE: KernelEntry = {
1013
1020
  phase: "P4",
1014
1021
  };
1015
1022
 
1016
- /** `grid-far-field` (G6, spec 7.7; P4-T10, PD-16, DEP-P4-G): per node in sorted order, the coarsest level minus its 3x3 (3x3x3) and, per finer level, the parent's 3x3 refined minus the level's own 3x3, plus the pseudo-cell; the loop bounds are `P.levels` / `P.gridMax`; LAW 0 FA2 / 1 FR / 2 coulomb per cell (P4-T13, PD-22); 5 storage bindings. */
1023
+ /** `grid-far-field` (G6, spec 7.7; P4-T10, PD-16, DEP-P4-G): per node in sorted order, the coarsest level minus its 3x3 (3x3x3) and, per finer level, the parent's 3x3 refined minus the level's own 3x3, plus the 2^dim orthant pseudo-cells; the FA2 term floored at 0.01 (issue #89); the loop bounds are `P.levels` / `P.gridMax`; LAW 0 FA2 / 1 FR / 2 coulomb per cell (P4-T13, PD-22); 5 storage bindings. */
1017
1024
  const GRID_FAR_FIELD: KernelEntry = {
1018
1025
  id: "grid-far-field",
1019
1026
  body: gridFarFieldWgsl,
@@ -1261,6 +1268,24 @@ const BFS_UNVISITED_FLAGS: KernelEntry = {
1261
1268
  phase: "P8",
1262
1269
  };
1263
1270
 
1271
+ /** `bfs-next-degree` (design 8.4; issue #391): Beamer's m_f measured exactly -- once per level, after the claim kernels, grid-striding over the output vertex queue and summing the `outDegree` view over the vertices the level claimed into word 25, one `atomicAdd` per workgroup; 3 storage bindings (`frontier` read-only, the `outDegree` VIEW, the counters block); `needs: ["subgroups"]` for the `wg_reduce_u32` call (a twin kernel). */
1272
+ const BFS_NEXT_DEGREE: KernelEntry = {
1273
+ id: "bfs-next-degree",
1274
+ body: bfsNextDegreeWgsl,
1275
+ entryPoint: "bfs_next_degree",
1276
+ bindings: [
1277
+ decl(1, 0, "frontier", "storage-ro", "array<u32>"),
1278
+ decl(1, 1, "outDegree", "storage-ro", "array<u32>"),
1279
+ decl(1, 2, "counters", "storage", "array<atomic<u32>>"),
1280
+ decl(2, 0, "P", "uniform", "FrontierParams"),
1281
+ ],
1282
+ overrideDecls: [],
1283
+ uniforms: [FRONTIER_PARAMS],
1284
+ needs: ["subgroups"],
1285
+ snippetSlots: [],
1286
+ phase: "P8",
1287
+ };
1288
+
1264
1289
  /** `sssp-relax` (design 8.4 "Davidson's near-far", 8.10 "SSSP near-far relax"; P8-T9, PD-9 / PD-20 / DEP-P8-E): one round of the near-far loop -- role 0 relaxes the deduped near pile's whole rows with `atomicMin` on the f32 bit patterns of `dist` and appends each improved vertex to the raw near or far half of `queueOut` (the two halves of ONE buffer at word 0 and word `P.edgeCapacity`), role 1 re-buckets the deduped far pile; 8 storage bindings (the four graph slots with the run's weights bound in the weights slot, `dist` and the counters block as `array<atomic<u32>>`, `queueIn` read-only, `queueOut`) -- exactly at the budget, which is why no `pred` lives here (PD-11) and why the piles' counts, the threshold and the delta are words of the block. */
1265
1290
  const SSSP_RELAX: KernelEntry = {
1266
1291
  id: "sssp-relax",
@@ -1392,6 +1417,7 @@ const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze
1392
1417
  "bfs-bottom-up": BFS_BOTTOM_UP,
1393
1418
  "bfs-bitset-build": BFS_BITSET_BUILD,
1394
1419
  "bfs-unvisited-flags": BFS_UNVISITED_FLAGS,
1420
+ "bfs-next-degree": BFS_NEXT_DEGREE,
1395
1421
  "sssp-relax": SSSP_RELAX,
1396
1422
  "bf-relax": BF_RELAX,
1397
1423
  "closeness-sweep": CLOSENESS_SWEEP,
@@ -26,6 +26,7 @@ import {
26
26
  GRID_EXTENT_FLOOR,
27
27
  LAYOUT_TUNING_DEFAULTS,
28
28
  MAX_ITERATIONS_PER_STEP,
29
+ SETTLE_FLOOR_UNBOUNDED,
29
30
  TRACE_RECORD_BYTES,
30
31
  UNIFORM_SLOT_BYTES,
31
32
  } from "../constants.js";
@@ -590,6 +591,7 @@ export class ForceAtlas2Model implements ForceModel<ForceAtlas2Options, ForceAtl
590
591
  accumulate: 0,
591
592
  hiEnd,
592
593
  midEnd,
594
+ settleFloor: SETTLE_FLOOR_UNBOUNDED, // ForceAtlas2 does not drift after settling (issue #97)
593
595
  };
594
596
  }
595
597
 
@@ -32,6 +32,7 @@ import {
32
32
  FR_REHEAT_FRACTION,
33
33
  FR_START_TEMPERATURE,
34
34
  MAX_ITERATIONS_PER_STEP,
35
+ SETTLE_FLOOR_FRACTION,
35
36
  TRACE_RECORD_BYTES,
36
37
  UNIFORM_SLOT_BYTES,
37
38
  } from "../constants.js";
@@ -84,6 +85,7 @@ import {
84
85
  subset,
85
86
  vector,
86
87
  } from "./model-common.js";
88
+ import { recordExactRepulsion } from "./repulsion-exact.js";
87
89
  import { type GridStage, RepulsionGrid, type RepulsionGridOverrides } from "./repulsion-grid.js";
88
90
 
89
91
  // ============================================================ constants
@@ -587,6 +589,7 @@ export class FruchtermanReingoldModel implements ForceModel<FruchtermanReingoldO
587
589
  hiEnd,
588
590
  midEnd,
589
591
  frK: resolved.k ?? 1 / Math.sqrt(n),
592
+ settleFloor: SETTLE_FLOOR_FRACTION.fruchtermanReingold * (resolved.k ?? 1 / Math.sqrt(n)),
590
593
  temperature: adaptive ? FR_START_TEMPERATURE : this.temperatureAt(iteration, resolved),
591
594
  };
592
595
  }
@@ -640,7 +643,7 @@ export class FruchtermanReingoldModel implements ForceModel<FruchtermanReingoldO
640
643
  if (stop < 2) {
641
644
  return;
642
645
  }
643
- k3.dispatch(pass, k3Bound, bound.plan, [offset]);
646
+ recordExactRepulsion(k3, pass, k3Bound, bound.plan, bound.n, offset);
644
647
  if (stop < STAGE_K5) {
645
648
  return;
646
649
  }
@@ -7,6 +7,7 @@
7
7
  * override set is a distinct pipeline and the subgroup twin is selected by the device's features (spec 5.1, D16).
8
8
  */
9
9
 
10
+ import { EXACT_TILES_PER_PASS } from "../constants.js";
10
11
  import { WebGpuGraphError } from "../errors.js";
11
12
  import { type DispatchPlan, plan1d } from "../kernel/dispatch.js";
12
13
  import { type BoundKernel, type Kernel } from "../kernel/kernel.js";
@@ -35,6 +36,33 @@ export interface RepulsionExactOverrides {
35
36
  readonly GRAVITY_CENTER: 0 | 1;
36
37
  }
37
38
 
39
+ /**
40
+ * Records K3 over `n` nodes as ceil(tiles / EXACT_TILES_PER_PASS) dispatches (issue #87: llvmpipe's per-invocation
41
+ * loop budget), pass p with p + 1 z slices of which only the last works (the kernel reads the pass from
42
+ * `num_workgroups.z`). One pass up to 32,768 nodes at WG 256. Every model's exact tier records K3 through here.
43
+ * ponytail: pass p also launches p idle slices, sum p over P passes; negligible against the O(n^2) pass work (31
44
+ * passes at 1M nodes); a per-pass uniform removes them if they ever show in a profile.
45
+ * @param kernel - the compiled `fa2-repulsion-exact`
46
+ * @param pass - the open compute pass
47
+ * @param bound - the kernel's bound groups
48
+ * @param plan - plan1d(n) of the kernel
49
+ * @param n - the node count
50
+ * @param paramsOffset - the dynamic offset of the Fa2Params slot
51
+ */
52
+ export function recordExactRepulsion(
53
+ kernel: Kernel,
54
+ pass: GPUComputePassEncoder,
55
+ bound: BoundKernel,
56
+ plan: DispatchPlan,
57
+ n: number,
58
+ paramsOffset: number,
59
+ ): void {
60
+ const passes = Math.max(1, Math.ceil(Math.ceil(n / kernel.workgroupSize) / EXACT_TILES_PER_PASS));
61
+ for (let z = 1; z <= passes; z++) {
62
+ kernel.dispatch(pass, bound, { ...plan, z }, [paramsOffset]);
63
+ }
64
+ }
65
+
38
66
  /** K3 (tiled all-pairs repulsion + gravity + the swing / traction epilogue) followed by K4 (the one-workgroup speed finalize) (spec 7.6, 7.10). */
39
67
  export class RepulsionExact {
40
68
  /** The overrides both kernels were compiled with (a frozen copy of the argument of create()). */
@@ -148,7 +176,7 @@ export class RepulsionExact {
148
176
  recordRepulsion(pass: GPUComputePassEncoder, n: number, paramsOffset: number): void {
149
177
  const bound = this.bound(this.boundRepulsion, "recordRepulsion");
150
178
  const plan = plan1d(n, this.repulsion.workgroupSize, this.caps);
151
- this.repulsion.dispatch(pass, bound, plan, [paramsOffset]);
179
+ recordExactRepulsion(this.repulsion, pass, bound, plan, n, paramsOffset);
152
180
  }
153
181
 
154
182
  /**
@@ -216,7 +216,7 @@ export class RepulsionGrid {
216
216
 
217
217
  /**
218
218
  * The model-owned buffers of the grid tier (spec 7.3; PD-11): `cellKey` / `cellVal` / `sortedKey` / `sortedIdx`
219
- * 4n, `cellHist` / `cellStart` 4 (cells + 2) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
219
+ * 4n, `cellHist` / `cellStart` 4 histWords (cells + 2^dim + 1) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
220
220
  * indirect slot, `pyramid` 16 B per pyramid cell zeroed. `hubCounters` (16 B, zeroed) is the MODEL's on every
221
221
  * tier (PD-14: K1 binds it on the exact tier too). n = 0 reports one node's worth of bytes (spec 3.6).
222
222
  * @param n - the node count
@@ -25,6 +25,8 @@ import {
25
25
  MAX_ITERATIONS_PER_STEP,
26
26
  SE_DEFAULTS,
27
27
  SE_SCALE_REFERENCE_NODES,
28
+ SETTLE_FLOOR_FRACTION,
29
+ SETTLE_FLOOR_REFERENCE_NODES,
28
30
  TRACE_RECORD_BYTES,
29
31
  UNIFORM_SLOT_BYTES,
30
32
  } from "../constants.js";
@@ -77,6 +79,7 @@ import {
77
79
  subset,
78
80
  vector,
79
81
  } from "./model-common.js";
82
+ import { recordExactRepulsion } from "./repulsion-exact.js";
80
83
  import { type GridStage, RepulsionGrid, type RepulsionGridOverrides } from "./repulsion-grid.js";
81
84
 
82
85
  // ============================================================ constants
@@ -555,6 +558,10 @@ export class SpringElectricalModel implements ForceModel<SpringElectricalOptions
555
558
  frK: 0,
556
559
  temperature: 0,
557
560
  springLength: resolved.springLength,
561
+ settleFloor:
562
+ SETTLE_FLOOR_FRACTION.springElectrical *
563
+ resolved.springLength *
564
+ (SETTLE_FLOOR_REFERENCE_NODES / Math.max(n, 1)) ** 0.25,
558
565
  springCoefficient: resolved.springCoefficient ?? SE_DEFAULTS.springCoefficient * springSizeFactor(n),
559
566
  coulomb: resolved.gravity ?? SE_DEFAULTS.gravity * springSizeFactor(n),
560
567
  dragCoefficient: resolved.dragCoefficient,
@@ -609,7 +616,7 @@ export class SpringElectricalModel implements ForceModel<SpringElectricalOptions
609
616
  if (stop < 2) {
610
617
  return;
611
618
  }
612
- k3.dispatch(pass, k3Bound, bound.plan, [offset]);
619
+ recordExactRepulsion(k3, pass, k3Bound, bound.plan, bound.n, offset);
613
620
  if (stop < STAGE_K5) {
614
621
  return;
615
622
  }