@graphty/webgpu-graph-algorithms 0.6.4 → 0.6.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/README.md +38 -17
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-Dvq-Cc6v.js → context-DiSr6eiz.js} +45 -33
  4. package/dist/chunks/context-DiSr6eiz.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/algorithms/bfs.d.ts +11 -7
  7. package/dist/src/algorithms/bfs.d.ts.map +1 -1
  8. package/dist/src/algorithms/bfs.js +33 -10
  9. package/dist/src/algorithms/bfs.js.map +1 -1
  10. package/dist/src/algorithms/scope.d.ts +3 -3
  11. package/dist/src/algorithms/scope.d.ts.map +1 -1
  12. package/dist/src/algorithms/scope.js +0 -2
  13. package/dist/src/algorithms/scope.js.map +1 -1
  14. package/dist/src/algorithms/sssp.d.ts +4 -3
  15. package/dist/src/algorithms/sssp.d.ts.map +1 -1
  16. package/dist/src/algorithms/sssp.js +4 -3
  17. package/dist/src/algorithms/sssp.js.map +1 -1
  18. package/dist/src/constants.d.ts +33 -2
  19. package/dist/src/constants.d.ts.map +1 -1
  20. package/dist/src/constants.js +33 -2
  21. package/dist/src/constants.js.map +1 -1
  22. package/dist/src/kernel/dispatch.d.ts +2 -2
  23. package/dist/src/kernel/dispatch.d.ts.map +1 -1
  24. package/dist/src/kernel/kernel.d.ts +1 -1
  25. package/dist/src/kernel/kernel.js +2 -2
  26. package/dist/src/kernel/kernel.js.map +1 -1
  27. package/dist/src/kernel/prelude.d.ts.map +1 -1
  28. package/dist/src/kernel/prelude.js +2 -1
  29. package/dist/src/kernel/prelude.js.map +1 -1
  30. package/dist/src/kernels.d.ts +15 -11
  31. package/dist/src/kernels.d.ts.map +1 -1
  32. package/dist/src/kernels.js +42 -21
  33. package/dist/src/kernels.js.map +1 -1
  34. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  35. package/dist/src/layouts/forceatlas2.js +2 -1
  36. package/dist/src/layouts/forceatlas2.js.map +1 -1
  37. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  38. package/dist/src/layouts/fruchterman-reingold.js +4 -2
  39. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  40. package/dist/src/layouts/repulsion-exact.d.ts +16 -0
  41. package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
  42. package/dist/src/layouts/repulsion-exact.js +21 -1
  43. package/dist/src/layouts/repulsion-exact.js.map +1 -1
  44. package/dist/src/layouts/repulsion-grid.d.ts +1 -1
  45. package/dist/src/layouts/repulsion-grid.js +1 -1
  46. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  47. package/dist/src/layouts/spring-electrical.js +6 -2
  48. package/dist/src/layouts/spring-electrical.js.map +1 -1
  49. package/dist/src/primitives/advance.d.ts +3 -2
  50. package/dist/src/primitives/advance.d.ts.map +1 -1
  51. package/dist/src/primitives/advance.js.map +1 -1
  52. package/dist/src/primitives/frontier.d.ts +34 -38
  53. package/dist/src/primitives/frontier.d.ts.map +1 -1
  54. package/dist/src/primitives/frontier.js +24 -32
  55. package/dist/src/primitives/frontier.js.map +1 -1
  56. package/dist/src/primitives/grid-pyramid.d.ts +4 -4
  57. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  58. package/dist/src/primitives/grid-pyramid.js +4 -3
  59. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  60. package/dist/src/primitives/grid.d.ts +13 -10
  61. package/dist/src/primitives/grid.d.ts.map +1 -1
  62. package/dist/src/primitives/grid.js +10 -7
  63. package/dist/src/primitives/grid.js.map +1 -1
  64. package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
  65. package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
  66. package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
  67. package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
  68. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
  69. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
  70. package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
  71. package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
  72. package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
  73. package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
  74. package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
  75. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
  76. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
  77. package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
  78. package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
  79. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
  80. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
  81. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
  82. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
  83. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
  84. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
  85. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
  86. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
  87. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +44 -49
  88. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
  89. package/dist/src/wgsl/frontier-finalize.wgsl.js +62 -107
  90. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
  91. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
  92. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
  93. package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
  94. package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
  95. package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
  96. package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
  97. package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
  98. package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
  99. package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
  100. package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
  101. package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
  102. package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
  103. package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
  104. package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
  105. package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
  106. package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
  107. package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
  108. package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
  109. package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
  110. package/dist/src/wgsl/histogram.wgsl.js +1 -1
  111. package/dist/webgpu-graph-algorithms.js +144 -119
  112. package/dist/webgpu-graph-algorithms.js.map +1 -1
  113. package/package.json +3 -3
  114. package/src/algorithms/bfs.ts +34 -10
  115. package/src/algorithms/pagerank.ts +19 -5
  116. package/src/algorithms/power-iteration.ts +8 -2
  117. package/src/algorithms/scope.ts +3 -10
  118. package/src/algorithms/sssp.ts +4 -3
  119. package/src/constants.ts +35 -2
  120. package/src/kernel/dispatch.ts +2 -2
  121. package/src/kernel/kernel.ts +2 -2
  122. package/src/kernel/prelude.ts +2 -0
  123. package/src/kernels.ts +44 -21
  124. package/src/layouts/forceatlas2.ts +2 -0
  125. package/src/layouts/fruchterman-reingold.ts +4 -1
  126. package/src/layouts/repulsion-exact.ts +29 -1
  127. package/src/layouts/repulsion-grid.ts +1 -1
  128. package/src/layouts/spring-electrical.ts +8 -1
  129. package/src/memory/residency.ts +14 -4
  130. package/src/primitives/advance.ts +5 -4
  131. package/src/primitives/frontier.ts +42 -56
  132. package/src/primitives/grid-pyramid.ts +6 -5
  133. package/src/primitives/grid.ts +17 -12
  134. package/src/wgsl/advance-expand.wgsl.ts +4 -3
  135. package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
  136. package/src/wgsl/bfs-fused.wgsl.ts +6 -6
  137. package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
  138. package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
  139. package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
  140. package/src/wgsl/frontier-finalize.wgsl.ts +62 -107
  141. package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
  142. package/src/wgsl/grid-centroid.wgsl.ts +3 -2
  143. package/src/wgsl/grid-downsample.wgsl.ts +1 -1
  144. package/src/wgsl/grid-far-field.wgsl.ts +15 -8
  145. package/src/wgsl/grid-near-field.wgsl.ts +6 -2
  146. package/src/wgsl/histogram.wgsl.ts +1 -1
  147. package/dist/chunks/context-Dvq-Cc6v.js.map +0 -1
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@graphty/webgpu-graph-algorithms",
3
- "version": "0.6.4",
3
+ "version": "0.6.6",
4
4
  "description": "WebGPU-accelerated graph algorithms and layouts over the @graphty/graph-format snapshot, for Node (Dawn) and browsers",
5
5
  "author": "Adam Powers <apowers@ato.ms>",
6
6
  "type": "module",
@@ -93,8 +93,8 @@
93
93
  "vite": "^7.0.5",
94
94
  "vitest": "^3.2.4",
95
95
  "webgpu": "0.4.0",
96
- "@graphty/algorithms": "^2.0.3",
97
- "@graphty/layout": "^1.9.1"
96
+ "@graphty/algorithms": "^2.0.4",
97
+ "@graphty/layout": "^1.10.0"
98
98
  },
99
99
  "scripts": {
100
100
  "build": "tsc -p tsconfig.build.json",
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * Breadth-first search on the device (design 8.4, 3.3 line 807, 9.7; P8-T6 / P8-T7 / P8-T8, the P8 plan's PD-5 /
3
3
  * PD-6 / PD-7 / PD-14 / PD-18 / PD-21 / PD-23 / PD-24 / PD-26 / DEP-P8-B): the direction-optimizing traversal over
4
- * the `Frontier` of P8-T4, every per-level choice made ON THE DEVICE. Every level is eight recorded dispatches, all
4
+ * the `Frontier` of P8-T4, every per-level choice made ON THE DEVICE. Every level is nine recorded dispatches, all
5
5
  * DIRECT (2026-09-25, docs/decisions/G8.md G8-F5: Dawn validates every `dispatchWorkgroupsIndirect` with an internal
6
6
  * clamp pass costing about 0.4 ms of device time whether or not it dispatches anything, in Node and in Chromium
7
7
  * alike, and that was 97 % of a traversal's wall time) -- `frontier-finalize` role 0 (the level boundary: rotates
@@ -13,14 +13,18 @@
13
13
  * per frontier entry with the same claim inline and no edge queue; the fused level and the retry alike), then the
14
14
  * bottom-up trio: a `fill` zeroing the frontier bitset, `bfs-bitset-build` setting the frontier's bits, and
15
15
  * `bfs-bottom-up` sweeping the unvisited list over the REVERSE core (each unvisited vertex reads its in-neighbours
16
- * until the first one in the bitset and claims itself). Every kernel is a grid-stride dispatch of a host-planned
17
- * grid (`planGridStride`) that loops to its count word and reads the path word first, so exactly one path does the
18
- * level's work and the others cost one uniform load per workgroup. The unvisited set Beamer's test is against (PD-18) is rebuilt exactly once per submit, before
19
- * the levels, by `bfs-unvisited-flags` plus `compact` over an iota queue, and maintained between rebuilds by
20
- * subtraction inside the selector (whose JSDoc states the boundary rule). The host records `MAX_LEVELS_PER_SUBMIT`
16
+ * until the first one in the bitset and claims itself), and last `bfs-next-degree` (the out-degree sum of whatever
17
+ * the level claimed, into the block's `nextDegreeSum` word: Beamer's m_f for the NEXT boundary, measured on the
18
+ * frontier that boundary decides for -- issue #391, which found the previous proxy, the degree of the frontier just
19
+ * expanded, one level stale and missing the switch at the level holding two thirds of the 1M / 10M R-MAT's arcs).
20
+ * Every kernel is a grid-stride dispatch of a host-planned grid (`planGridStride`) that loops to its count word and
21
+ * reads the path word first, so exactly one path does the level's work and the others cost one uniform load per
22
+ * workgroup. The unvisited set Beamer's test is against (PD-18) is rebuilt exactly once per submit, before the
23
+ * levels, by `bfs-unvisited-flags` plus `compact` over an iota queue, and maintained between rebuilds by subtraction
24
+ * inside the selector (whose JSDoc states the boundary rule; both words are exact at every boundary). The host records `MAX_LEVELS_PER_SUBMIT`
21
25
  * levels into ONE command buffer, submits, and reads four bytes, the `done` word (PD-7): a road network has
22
26
  * thousands of levels and a per-level `mapAsync` would be slower than the CPU. A level recorded past the end is a
23
- * no-op (its boundary finds `done` set, zeroes its slots and moves no counter), so the loop needs no diameter and
27
+ * no-op (its boundary finds `done` set, writes path 0 and moves no counter), so the loop needs no diameter and
24
28
  * ends on `done`; a traversal has at most `n` levels, so more submits than that is E_VALIDATION, never a hang. The
25
29
  * thresholds are uniform fields (`FUSED_FRONTIER_MAX`; alpha derived as `max(1, floor(arcCount / n))`, PD-21;
26
30
  * `BEAMER_BETA`; `mode 1` pinning top-down -- each unless the tuning says otherwise), so a test forces any path
@@ -79,6 +83,9 @@ import { type AlgorithmScope, algorithmScope } from "./scope.js";
79
83
 
80
84
  const ALGORITHM = "breadthFirstSearch";
81
85
 
86
+ /** Workgroups of the `bfs-next-degree` grid (issue #391): enough to sum n out-degrees by grid stride, few enough that a high-diameter traversal does not pay a full-width reduction per level. */
87
+ const NEXT_DEGREE_MAX_GROUPS = 128;
88
+
82
89
  /**
83
90
  * Params slots of the ring, COUNTED per run (P8-T12), because `UniformRing.reserve` wraps to slot 0 when a submit's
84
91
  * records outrun the ring and silently overwrites a record the submit still reads; the ring's `overruns` counts
@@ -87,8 +94,9 @@ const ALGORITHM = "breadthFirstSearch";
87
94
  * `bfs-contract`, the bits `fill`, `bfs-bitset-build`) and four re-issued once per arc window (`advance-expand`,
88
95
  * `bfs-fused`, the fused retry, `bfs-bottom-up`). The driver as built writes fewer -- the contract and the bitset
89
96
  * build share one window-free record, a window's two fused dispatches share one, the bits fill has one record per
90
- * submit, and a directed snapshot's reverse view is one window whatever the forward core's count -- so at most
91
- * `3 + 3 x windows` per level, and the bound holds with room. The 16 covers the per-submit rebuild
97
+ * submit, `bfs-next-degree` has one window-free record of its own (issue #391), and a directed snapshot's reverse
98
+ * view is one window whatever the forward core's count -- so at most `4 + 3 x windows` per level, and the bound
99
+ * holds with room. The 16 covers the per-submit rebuild
92
100
  * (`bfs-unvisited-flags` and `compact`, whose scan is at most 9 dispatches for any n below 2^32, so 10, plus the
93
101
  * bits fill). The result batch flushes in its own submit, so the ring must hold IT too, and its size grows with `n`,
94
102
  * not with the cadence: the iota `fill`, the radix sort's four passes of one record plus its scan's `2 x levels - 1`
@@ -339,6 +347,7 @@ export async function bfsWithTuning(
339
347
  const bitset = await ctx.pipelines.kernel(kernelSpec("bfs-bitset-build"));
340
348
  const bottomUp = await ctx.pipelines.kernel(kernelSpec("bfs-bottom-up", graphOverrides(reverse, null)));
341
349
  const unvisited = await ctx.pipelines.kernel(kernelSpec("bfs-unvisited-flags"));
350
+ const nextDegree = await ctx.pipelines.kernel(kernelSpec("bfs-next-degree"));
342
351
  const pred = await ctx.pipelines.kernel(kernelSpec("sssp-pred", { ...graphOverrides(core, null), MODE: 1 }));
343
352
  const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
344
353
  const sort = await prepareRadixSort(scope);
@@ -352,6 +361,11 @@ export async function bfsWithTuning(
352
361
  // plan's GROUP count as its stride (planGridStride's cap applies to the groups)
353
362
  const levelPlan = planGridStride(Math.max(n, frontier.edgeCapacity), wg, ctx.caps);
354
363
  const sweepPlan = planGridStride(n, wg, ctx.caps);
364
+ // `bfs-next-degree` sums at most n words and is dispatched on EVERY level, so its fixed cost -- one
365
+ // workgroup reduction (six barriers) and one atomic per workgroup -- is paid 1,999 times on the
366
+ // 1000 x 1000 grid. A capped grid pays it 128 times a level instead of ceil(n / wg): on the card the
367
+ // grid row went from 436 to 356 ms and neither R-MAT row moved.
368
+ const degreePlan = planGridStride(n, wg, ctx.caps, NEXT_DEGREE_MAX_GROUPS);
355
369
  const fusedPlan = planGridStride(n * wg, wg, ctx.caps);
356
370
  const bitsPlan = plan1d(bitsWords, wg, ctx.caps);
357
371
  const recordFill = (pass: GPUComputePassEncoder, dst: Binding, value: number, mode: 0 | 1): void => {
@@ -415,7 +429,7 @@ export async function bfsWithTuning(
415
429
  const bitsParams = scope.params(FILL_PARAMS, { count: bitsWords, value: 0, mode: 0, pad0: 0 });
416
430
  const boundBitsFill = fill.bind({ dst: frontierBits, P: bitsParams.binding });
417
431
  for (let level = 0; level < levelsPerSubmit; level++) {
418
- planner.recordFinalize(pass, 0, level, { ...fields, firstOfSubmit: Math.min(level, 2) });
432
+ planner.recordFinalize(pass, 0, level, { ...fields, firstOfSubmit: Math.min(level, 1) });
419
433
  advance.record(pass, frontier);
420
434
  planner.recordFinalize(pass, 1, level, fields);
421
435
  // one window-free record serves the contract and the bitset build, which read no arc; the kernels that
@@ -489,6 +503,16 @@ export async function bfsWithTuning(
489
503
  });
490
504
  bottomUp.dispatch(pass, boundSweep, sweepPlan, [sweepParams.offset]);
491
505
  }
506
+ // the next frontier's out-degree sum (issue #391): whichever path claimed, the vertices are in the output
507
+ // queue now, and the next boundary reads word 25 as Beamer's m_f for the frontier it is about to expand
508
+ const degreeParams = scope.params(FRONTIER_PARAMS, { wg, n, stride: degreePlan.stride ?? wg });
509
+ const boundDegree = nextDegree.bind({
510
+ frontier: frontier.output,
511
+ outDegree,
512
+ counters,
513
+ P: degreeParams.binding,
514
+ });
515
+ nextDegree.dispatch(pass, boundDegree, degreePlan, [degreeParams.offset]);
492
516
  frontier.swap();
493
517
  }
494
518
  batch.endPass();
@@ -34,8 +34,8 @@ import { algorithmScope } from "./scope.js";
34
34
 
35
35
  /** Iterations per submit (spec 8.2: k = 8). */
36
36
  const PR_BATCH = 8;
37
- /** Params slots of one batch: per iteration one PrParams (shared by pr-scale and pr-finalize) and one SpmvParams, plus the normaliser's RangeParams; a smaller ring wraps onto a slot the same batch still reads. */
38
- const RING_SLOTS = 2 * PR_BATCH + 1;
37
+ /** Params slots of one batch: per iteration one PrParams (shared by pr-scale and pr-finalize) and one SpmvParams, plus the normaliser's RangeParams and the last batch's convergence check; a smaller ring wraps onto a slot the same batch still reads. */
38
+ const RING_SLOTS = 2 * PR_BATCH + 2;
39
39
 
40
40
  /**
41
41
  * Validates `options.dest` for a score result of `n` elements.
@@ -175,6 +175,8 @@ async function run(
175
175
  const groups = groupsOf(scalePlan);
176
176
  const partialsBytes = PR_PARTIAL.byteLength * (1 + groups);
177
177
  const partials = scope.scratch(partialsBytes, "partials");
178
+ // the header as the last pull left it, before the convergence check of the last batch overwrites danglingMass
179
+ const lastHeader = scope.scratch(PR_PARTIAL.byteLength, "lastHeader");
178
180
  if (personalization !== null) {
179
181
  uploaded = ctx.residency.array(personalization, `${algorithm}/personalization`);
180
182
  }
@@ -203,17 +205,21 @@ async function run(
203
205
  const xNormBinding = bindingOf(xNorm, bytes);
204
206
  const outWeightSumBinding = bindingOf(outWeightSum, bytes);
205
207
  const partialsBinding = bindingOf(partials, partialsBytes);
208
+ const lastHeaderBinding = bindingOf(lastHeader, PR_PARTIAL.byteLength);
206
209
  const coefficients = { alpha, beta: 1 - alpha, uniformP: 1 / n };
207
210
  let cur = 0;
208
211
  let iterationsRun = 0;
209
212
  for (;;) {
210
213
  const k = Math.min(PR_BATCH, maxIterations - iterationsRun);
211
214
  const batch = new CommandBatch(ctx, algorithm);
212
- const pass = batch.pass("iterations");
215
+ let pass = batch.pass("iterations");
213
216
  if (iterationsRun === 0) {
214
217
  normaliser.record(pass, weightedCore, outWeightSumBinding);
215
218
  }
216
- for (let i = 0; i < k; i++) {
219
+ const last = iterationsRun + k === maxIterations;
220
+ // the scale + finalize of iteration i measures the error of iteration i - 1 (PD-9), so the last batch
221
+ // runs them once more, with no pull, to measure the error of iteration maxIterations itself
222
+ for (let i = 0; i < k + (last ? 1 : 0); i++) {
217
223
  const params = scope.params(PR_PARAMS, {
218
224
  n,
219
225
  groups,
@@ -222,6 +228,10 @@ async function run(
222
228
  convergeThreshold: tolerance * n,
223
229
  });
224
230
  const other = 1 - cur;
231
+ if (i === k) {
232
+ batch.copy(partialsBinding, lastHeaderBinding, PR_PARTIAL.byteLength);
233
+ pass = batch.pass("convergence");
234
+ }
225
235
  const scaleBound = scale.bind({
226
236
  rankIn: rank[cur],
227
237
  rankPrev: rank[other],
@@ -233,6 +243,9 @@ async function run(
233
243
  scale.dispatch(pass, scaleBound, scalePlan, [params.offset]);
234
244
  const finalizeBound = finalize.bind({ partials: partialsBinding, P: params.binding });
235
245
  finalize.dispatch(pass, finalizeBound, finalizePlan, [params.offset]);
246
+ if (i === k) {
247
+ break;
248
+ }
236
249
  pull.record(
237
250
  pass,
238
251
  weightedRev,
@@ -248,6 +261,7 @@ async function run(
248
261
  }
249
262
  batch.endPass();
250
263
  const headerRequest = batch.readback(partials, 0, PR_PARTIAL.byteLength);
264
+ const lastHeaderRequest = last ? batch.readback(lastHeader, 0, PR_PARTIAL.byteLength) : headerRequest;
251
265
  const scoresRequest = batch.readback(rank[cur].buffer, 0, bytes);
252
266
  scope.flush();
253
267
  const submitted = batch.submit();
@@ -271,7 +285,7 @@ async function run(
271
285
  scores,
272
286
  iterations: converged ? firstConverged : iterationsRun,
273
287
  converged,
274
- danglingMass: folded.danglingMass as number,
288
+ danglingMass: PR_PARTIAL.read(new DataView(back), lastHeaderRequest.offset).danglingMass as number,
275
289
  precision: "f32",
276
290
  };
277
291
  }
@@ -33,7 +33,7 @@ import { algorithmScope } from "./scope.js";
33
33
 
34
34
  /** Iterations per submit (spec 8.2: k = 8). */
35
35
  const BATCH = 8;
36
- /** Params slots of one batch: four blocks per iteration times the batch, plus a margin (the plan's `4 * 8 + 8`). */
36
+ /** Params slots of one batch: four blocks per iteration times the batch, plus a margin (the plan's `4 * 8 + 8`) that also holds the last batch's convergence check. */
37
37
  const RING_SLOTS = 4 * BATCH + 8;
38
38
 
39
39
  /**
@@ -210,7 +210,10 @@ export async function runPowerIteration(
210
210
  const k = Math.min(BATCH, config.maxIterations - iterationsRun);
211
211
  const batch = new CommandBatch(ctx, config.label);
212
212
  const pass = batch.pass("iterations");
213
- for (let i = 0; i < k; i++) {
213
+ // the scale + finalize of iteration i measures the error of iteration i - 1 (PD-9), so the last batch
214
+ // runs them once more, with no apply and no pull, to measure the error of iteration maxIterations itself
215
+ const last = iterationsRun + k === config.maxIterations;
216
+ for (let i = 0; i < k + (last ? 1 : 0); i++) {
214
217
  const iteration = iterationsRun + i + 1;
215
218
  const params = scope.params(PR_PARAMS, {
216
219
  n,
@@ -234,6 +237,9 @@ export async function runPowerIteration(
234
237
  };
235
238
  scaleNorm.dispatch(pass, scaleNorm.bind(scaleBindings), scalePlan, [params.offset]);
236
239
  finalize.dispatch(pass, finalize.bind({ partials, P: params.binding }), finalizePlan, [params.offset]);
240
+ if (i === k) {
241
+ break;
242
+ }
237
243
  if (scaleApply !== null) {
238
244
  scaleApply.dispatch(pass, scaleApply.bind(scaleBindings), scalePlan, [params.offset]);
239
245
  }
@@ -8,12 +8,11 @@
8
8
  */
9
9
 
10
10
  import { type GpuContext } from "../context.js";
11
- import { BufferUsage } from "../device/webgpu-constants.js";
12
11
  import { UniformRing } from "../kernel/uniform-ring.js";
13
- import { type FrontierScope } from "../primitives/frontier.js";
12
+ import { type ReduceScope } from "../primitives/reduce.js";
14
13
 
15
- /** A FrontierScope (a ReduceScope plus `indirect()`, P8-T4) over a context plus the two lifecycle calls an algorithm makes: flush() before submit, dispose() in its finally. */
16
- export interface AlgorithmScope extends FrontierScope {
14
+ /** A ReduceScope over a context plus the two lifecycle calls an algorithm makes: flush() before submit, dispose() in its finally. (The `indirect()` lease it once added for the frontier's args buffer went with that buffer, 2026-09-25.) */
15
+ export interface AlgorithmScope extends ReduceScope {
17
16
  /** queue.writeBuffer of the params slots written since the last flush (called before the batch is submitted). */
18
17
  flush(): void;
19
18
  /** Destroys the ring and releases every scratch buffer of the lease; idempotent. */
@@ -39,12 +38,6 @@ export function algorithmScope(ctx: GpuContext, label: string, slots: number): A
39
38
  pool: ctx.pool,
40
39
  workgroupSize: ctx.workgroupSize,
41
40
  scratch: (byteLength, scratchLabel) => lease.storage(byteLength, `${label}/${scratchLabel}`),
42
- indirect: (byteLength, indirectLabel) =>
43
- lease.acquire(
44
- byteLength,
45
- BufferUsage.STORAGE | BufferUsage.INDIRECT | BufferUsage.COPY_DST | BufferUsage.COPY_SRC,
46
- `${label}/${indirectLabel}`,
47
- ),
48
41
  params(block, values) {
49
42
  const slot = ring.reserve(1);
50
43
  ring.write(slot, block, values);
@@ -11,9 +11,10 @@
11
11
  * delta and dedupes the far half into `farIn` for a pass-through; both empty is `done`), `dedupe-claim` and
12
12
  * `dedupe-filter` over each half (direct grid-stride dispatches; role 2 writes the chosen half's raw count into that
13
13
  * dedupe's count word -- `edgeCount` for the near half, `edgeCountUnclamped` for the far one, two words SSSP borrows
14
- * -- and 0 into the other's, so the half not chosen is a no-op), role 3 (restarts the raw half), and `sssp-relax`
15
- * twice (role 0 over `nearIn`, role 1 over `farIn`; the block's `path` word, 5 a near round and 6 a far one, makes
16
- * the other role's dispatch a no-op). The far pile is re-bucketed by the
14
+ * -- and 0 into the other's, so the half not chosen is a no-op), role 3 (restarts the raw half the round consumed),
15
+ * and `sssp-relax` twice (role 0 over `nearIn` sized by `frontierCount`, role 1 over `farIn` sized by `farCount`;
16
+ * the block's `path` word, 5 a near round and 6 a far one, makes the other role's dispatch a no-op). Nothing in a
17
+ * round is an indirect dispatch (2026-09-25). The far pile is re-bucketed by the
17
18
  * relax kernel's pass-through, not by `compact` (PD-20): a far entry whose settled distance fell below the previous
18
19
  * threshold was relaxed in the near band already and is dropped, the rest go back to near or far against the raised
19
20
  * threshold. The near pile is ONE pile (no sub-partitions). The host records `MAX_LEVELS_PER_SUBMIT` rounds per
package/src/constants.ts CHANGED
@@ -130,6 +130,13 @@ export const FA2_DISTANCE_FLOOR = 0.01;
130
130
  export const FA2_DISTANCE_FLOOR_SQ = 0.0001;
131
131
  /** The coincident threshold `d^2 < 1e-8` of spec 7.2. */
132
132
  export const FA2_COINCIDENT_SQ = 1e-8;
133
+ /**
134
+ * The most WG-node tiles one K3 (`fa2-repulsion-exact`) dispatch sums per invocation (issue #87). llvmpipe runs at
135
+ * most 65,535 loop iterations per shader invocation, counted over every loop together, and then quietly breaks
136
+ * out of each loop; one tile costs WG + 2 of them (258 at WG 256), so a single pass lost every node past
137
+ * j = 65,027. 128 tiles is 33,024 iterations, half the budget; a larger exact run records ceil(tiles / 128) passes.
138
+ */
139
+ export const EXACT_TILES_PER_PASS = 128;
133
140
  /** Bits of Fa2Params.flags (contract 4.4). */
134
141
  export const FA2_FLAG_FIRST = 1;
135
142
  /** Fa2Params.flags bit: the Fruchterman-Reingold temperature is the adaptive one in the state block, not the uniform's (the `cooling: "adaptive"` option). */
@@ -202,6 +209,34 @@ export const SE_DEFAULTS: Readonly<{
202
209
  iterationsPerStep: 1,
203
210
  maxInFlight: 2,
204
211
  });
212
+ /**
213
+ * The absolute settle floor of the shared settle rule (spec 7.17; issue #97): an iteration counts toward `settled` only
214
+ * when its mean displacement is at most `settleThreshold x rmsRadius` AND at most this fraction of the model's length
215
+ * unit -- `springLength` for spring-electrical (scaled by node count, see SETTLE_FLOOR_REFERENCE_NODES), `k` for
216
+ * Fruchterman-Reingold. The relative rule alone reported a spring
217
+ * layout settled while it still grew (6 % over 1,000 iterations at 10k nodes). ForceAtlas2 writes
218
+ * `SETTLE_FLOOR_UNBOUNDED` instead: it does not drift after settling, and its per-iteration jitter grows with n, so any
219
+ * fixed floor only delays or blocks its stop. The values are measured:
220
+ * design/decisions/2026-09-24-settle-rule-has-an-absolute-floor.md.
221
+ */
222
+ export const SETTLE_FLOOR_FRACTION: Readonly<{ springElectrical: number; fruchtermanReingold: number }> = Object.freeze(
223
+ {
224
+ springElectrical: 3e-3,
225
+ fruchtermanReingold: 2e-3,
226
+ },
227
+ );
228
+ /**
229
+ * The node count at which the spring-electrical floor is exactly `SETTLE_FLOOR_FRACTION.springElectrical x
230
+ * springLength`; at `n` nodes it is that times `(SETTLE_FLOOR_REFERENCE_NODES / n)^(1/4)`. A spring layout's rms radius
231
+ * grows about as n^(1/4) in springLengths (3.4 at 150 nodes, 6.4 at 2,000, 10.4 at 10,000), so the relative half of the
232
+ * rule loosens with size while the floor tightens with it: on a small graph the floor sits above the relative threshold
233
+ * and the relative rule decides alone, as before issue #97; on a large one the floor binds, which is where the relative
234
+ * rule let an expanding layout stop. A fixed floor bound the 150-node story graph too, where the grid tier's jitter sits
235
+ * at the relative threshold, and nearly doubled its settle (427 -> 829 iterations on the RTX 4070 SUPER).
236
+ */
237
+ export const SETTLE_FLOOR_REFERENCE_NODES = 2000;
238
+ /** The settle floor that never binds: the largest finite f32, 0x1.fffffep+127 (ForceAtlas2's `settleFloor`, issue #97). */
239
+ export const SETTLE_FLOOR_UNBOUNDED = 2 ** 128 - 2 ** 104;
205
240
  /** The smallest finest grid side `G` (spec 7.7 geometry table: `clamp(nextPow2(2 n^(1/dim)), 8, gridMax)`; P4 PD-9). */
206
241
  export const GRID_MIN_SIDE = 8;
207
242
  /** The coarsest pyramid level's side (spec 7.7: "levels (coarsest 4 per axis)", `levels = log2(G / 4) + 1`). */
@@ -218,8 +253,6 @@ export const GRID_SORT_BITS = 24;
218
253
  export const MAX_LEVELS_PER_SUBMIT = 32;
219
254
  /** Design 8.4 and 6 row 8 (P8): a frontier at most this long runs the fused expand-contract kernel (Merrill's "fleeting iterations"); the default of the `fusedMax` uniform, which a test may set to 0 or `U32_MAX`. */
220
255
  export const FUSED_FRONTIER_MAX = 4096;
221
- /** P8-T4: the indirect dispatch slots `frontier-finalize` writes per level (expand, contract, fused, fill-bits, bitset-build, bottom-up, fused-retry); the args buffer is `MAX_LEVELS_PER_SUBMIT x FRONTIER_CANDIDATES x 16` bytes. */
222
- export const FRONTIER_CANDIDATES = 7;
223
256
  /** Design 8.4 (P8 PD-21): Beamer's beta -- switch back to top-down when `frontierCount * BEAMER_BETA < unvisitedCount` and the frontier is shrinking; alpha is derived from the graph, so it has no constant. */
224
257
  export const BEAMER_BETA = 24;
225
258
  /** Design 8.4 (P8 PD-22): the near-far split `delta = SSSP_DELTA_FACTOR * avgWeight / avgDegree`, computed on the host from the weight vector the run uses. */
@@ -9,11 +9,11 @@ import { MAX_WORKGROUPS_PER_DIM } from "../constants.js";
9
9
  import { WebGpuGraphError } from "../errors.js";
10
10
  import { type PlanCaps } from "../types/context.js";
11
11
 
12
- /** A dispatch shape (spec 5.2). `stride` is the grid-stride step (null for plain 1D / 2D plans). */
12
+ /** A dispatch shape (spec 5.2). `stride` is the grid-stride step (null for plain 1D / 2D plans). `z` is 1 from every planner; only K3's pass split sets it (issue #87). */
13
13
  export interface DispatchPlan {
14
14
  readonly x: number;
15
15
  readonly y: number;
16
- readonly z: 1;
16
+ readonly z: number;
17
17
  readonly items: number;
18
18
  readonly stride: number | null;
19
19
  }
@@ -245,7 +245,7 @@ export class Kernel {
245
245
  }
246
246
 
247
247
  /**
248
- * setPipeline + setBindGroup for every group (dynamic offsets in dynamicGroups order) + dispatchWorkgroups(plan.x, plan.y, 1); a plan with x === 0 records nothing (spec 5.6).
248
+ * setPipeline + setBindGroup for every group (dynamic offsets in dynamicGroups order) + dispatchWorkgroups(plan.x, plan.y, plan.z); a plan with x === 0 records nothing (spec 5.6).
249
249
  * PLAN DECISION: one dynamic offset per dynamic GROUP, replicated over every uniform binding of that group (every
250
250
  * P1-P3 kernel has exactly one params uniform per group); absent offsets mean 0; a BoundKernel of another kernel
251
251
  * or an offset list of the wrong length is E_INVALID_ARGUMENT.
@@ -268,7 +268,7 @@ export class Kernel {
268
268
  return;
269
269
  }
270
270
  this.setUp(pass, bound, dynamicOffsets);
271
- pass.dispatchWorkgroups(plan.x, plan.y, 1);
271
+ pass.dispatchWorkgroups(plan.x, plan.y, plan.z);
272
272
  }
273
273
 
274
274
  /**
@@ -12,6 +12,7 @@
12
12
  import { INVALID_INDEX } from "@graphty/graph-format";
13
13
 
14
14
  import {
15
+ EXACT_TILES_PER_PASS,
15
16
  F32_INF_BITS,
16
17
  FA2_COINCIDENT_SQ,
17
18
  FA2_DISTANCE_FLOOR,
@@ -54,6 +55,7 @@ const INVALID_INDEX: u32 = ${INVALID_INDEX}u;
54
55
  const U32_MAX: u32 = ${U32_MAX}u;
55
56
  const F32_INF_BITS: u32 = ${F32_INF_BITS}u;
56
57
  const MAX_WORKGROUPS_PER_DIM: u32 = ${MAX_WORKGROUPS_PER_DIM}u;
58
+ const EXACT_TILES_PER_PASS: u32 = ${EXACT_TILES_PER_PASS}u;
57
59
  const FA2_DIST_FLOOR: f32 = ${wgslF32Literal(FA2_DISTANCE_FLOOR)};
58
60
  const FA2_DIST_FLOOR_SQ: f32 = ${wgslF32Literal(FA2_DISTANCE_FLOOR_SQ)};
59
61
  const FA2_COINCIDENT_SQ: f32 = ${wgslF32Literal(FA2_COINCIDENT_SQ)};
package/src/kernels.ts CHANGED
@@ -28,6 +28,7 @@ import { bfsBitsetBuildWgsl } from "./wgsl/bfs-bitset-build.wgsl.js";
28
28
  import { bfsBottomUpWgsl } from "./wgsl/bfs-bottom-up.wgsl.js";
29
29
  import { bfsContractWgsl } from "./wgsl/bfs-contract.wgsl.js";
30
30
  import { bfsFusedWgsl } from "./wgsl/bfs-fused.wgsl.js";
31
+ import { bfsNextDegreeWgsl } from "./wgsl/bfs-next-degree.wgsl.js";
31
32
  import { bfsUnvisitedFlagsWgsl } from "./wgsl/bfs-unvisited-flags.wgsl.js";
32
33
  import { closenessReduceWgsl } from "./wgsl/closeness-reduce.wgsl.js";
33
34
  import { closenessSweepWgsl } from "./wgsl/closeness-sweep.wgsl.js";
@@ -111,6 +112,7 @@ export type KernelId =
111
112
  | "bfs-bottom-up"
112
113
  | "bfs-bitset-build"
113
114
  | "bfs-unvisited-flags"
115
+ | "bfs-next-degree"
114
116
  | "sssp-relax"
115
117
  | "bf-relax"
116
118
  | "closeness-sweep"
@@ -162,7 +164,7 @@ export const FILL_PARAMS: UniformBlock = UniformBlock.define("FillParams", [
162
164
  ["pad0", "u32"],
163
165
  ]);
164
166
 
165
- /** `Fa2Params` (uniform, 128 B; spec 7.3): the per-iteration ForceAtlas2 parameters -- `n` @0, `dim` @4, `flags` @8 (bit 0 = FA2_FLAG_FIRST), `tierStart` @12, `tierEnd` @16, `iterationIndex` @20, `seed` @24, `nearMax` @28, `scalingRatio` @32, `gravity` @36, `jitterTolerance` @40, `scale` @44, `center` @48 (xyz, w 0), `settleThreshold` @64, `extentFactor` @68, `gridMax` @72, `levels` @76, `arcBase` @80 / `arcEnd` @84 (the bound arc window of K2, 0 and arcCount in the layout), `accumulate` @88 (1 combines into `force`: the windowed pattern), `hiEnd` @92 / `midEnd` @124 (the degreeOrder tier boundaries, PD-7; both 0 without a permutation); the P5 model fields (PD-3): `frK` @96 (the FR optimal distance), `temperature` @100 (the FR temperature of this iteration), `springLength` @104, `springCoefficient` @108, `coulomb` @112 (ngraph's `gravity`, negative repels), `dragCoefficient` @116, `timeStep` @120; 128 B. */
167
+ /** `Fa2Params` (uniform, 144 B; spec 7.3): the per-iteration ForceAtlas2 parameters -- `n` @0, `dim` @4, `flags` @8 (bit 0 = FA2_FLAG_FIRST), `tierStart` @12, `tierEnd` @16, `iterationIndex` @20, `seed` @24, `nearMax` @28, `scalingRatio` @32, `gravity` @36, `jitterTolerance` @40, `scale` @44, `center` @48 (xyz, w 0), `settleThreshold` @64, `extentFactor` @68, `gridMax` @72, `levels` @76, `arcBase` @80 / `arcEnd` @84 (the bound arc window of K2, 0 and arcCount in the layout), `accumulate` @88 (1 combines into `force`: the windowed pattern), `hiEnd` @92 / `midEnd` @124 (the degreeOrder tier boundaries, PD-7; both 0 without a permutation); the P5 model fields (PD-3): `frK` @96 (the FR optimal distance), `temperature` @100 (the FR temperature of this iteration), `springLength` @104, `springCoefficient` @108, `coulomb` @112 (ngraph's `gravity`, negative repels), `dragCoefficient` @116, `timeStep` @120; `settleFloor` @128 (the absolute bound on the mean displacement of a settled iteration, in layout units: issue #97); 144 B. */
166
168
  export const FA2_PARAMS: UniformBlock = UniformBlock.define("Fa2Params", [
167
169
  ["n", "u32"],
168
170
  ["dim", "u32"],
@@ -193,6 +195,7 @@ export const FA2_PARAMS: UniformBlock = UniformBlock.define("Fa2Params", [
193
195
  ["dragCoefficient", "f32"],
194
196
  ["timeStep", "f32"],
195
197
  ["midEnd", "u32"],
198
+ ["settleFloor", "f32"],
196
199
  ]);
197
200
 
198
201
  /** `Fa2State` (storage, padded to STATE_HEADER_BYTES = 256; spec 7.3): the device-resident controller state the finalize kernels write and the host reads back for stats -- `speed` @0, `speedEfficiency` @4, `swing` @8, `traction` @12, `centroid` @16, `rmsRadius` @32, `radius` @36, `meanDisplacement` @40, `iteration` @44, `min` @48, `max` @64, `gridMin` @80 (P4), `eps` @96 (P4), `settledCount` @100, `outsideGrid` @104 (P4), `maxCellOccupancy` @108 (P4), `temperature` @112 (FR, written by K1 under STATS_MODE 1), `kineticEnergy` @116 (the preset, K1 under STATS_MODE 2), `frEnergy` @120 / `frProgress` @124 (the FR adaptive cooling), `invCellSize` @128 (P4, PD-10: `1 / cellSize`, written by K1 beside `cellSize` in `gridMin.w`; G1 multiplies by it so every key is bitwise reproducible), `reserved0` @132 (f32), `reserved1` @136 (vec2f), `reserved2` .. `reserved8` @144 .. @240. */
@@ -386,8 +389,11 @@ export const COMPACT_PARAMS: UniformBlock = UniformBlock.define("CompactParams",
386
389
  * submit), `arcsScanned` @64, `fusedLevels` @68, `twoPhaseLevels` @72, `bottomUpLevels` @76, `farCount` @80,
387
390
  * `nextFarCount` @84, `thresholdBits` @88, `deltaBits` @92 (P8-T9), `path` @96 (what the level's kernels run, written
388
391
  * by the selector: 0 nothing, 1 two-phase, 2 fused, 3 bottom-up, 4 the fused retry, 5 a near SSSP round, 6 a far
389
- * one; every level kernel is a direct dispatch that reads it first -- G8-F5). The words nothing writes before
390
- * P8-T8 / P8-T9 are declared now because the byte layout is what the single result copy decodes.
392
+ * one; every level kernel is a direct dispatch that reads it first -- G8-F5), `nextDegreeSum` @100 (issue #391: the
393
+ * out-degree sum of the vertices the level claimed, accumulated by `bfs-next-degree` at the end of every level and
394
+ * read, subtracted and zeroed by the next boundary -- Beamer's m_f measured on the frontier the boundary decides
395
+ * for, not on the one it has just expanded). The words nothing writes before P8-T8 / P8-T9 are declared now because
396
+ * the byte layout is what the single result copy decodes.
391
397
  */
392
398
  export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
393
399
  "FrontierCounters",
@@ -417,23 +423,24 @@ export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
417
423
  ["thresholdBits", "u32"],
418
424
  ["deltaBits", "u32"],
419
425
  ["path", "u32"],
426
+ ["nextDegreeSum", "u32"],
420
427
  ],
421
428
  { layout: "storage" },
422
429
  );
423
430
 
424
431
  /**
425
432
  * `FrontierParams` (uniform, 80 B; P8-T4): the params block every P8 kernel except the three compact / dedupe
426
- * primitives and `bf-relax` binds -- `role` @0 (the finalize role), `slotBase` @4 (`level x FRONTIER_CANDIDATES`),
427
- * `wg` @8 (the consumers' workgroup size), `alpha` @12, `beta` @16 (Beamer's thresholds, P8-T8), `fusedMax` @20,
428
- * `edgeCapacity` @24, `maxDepth` @28, `n` @32, `mode` @36 (BFS: 0 auto, 1 top-down only; `sssp-pred`: the PD-27 key
429
- * rule), `cutoffBits` @40, `arcBase` @44, `arcEnd` @48 (the bound arc window), `predKind` @52 (0 arc, 1 node),
430
- * `bitsBase` @56, `source` @60, `stride` @64 (a grid-stride plan's stride), `firstOfSubmit` @68 (the boundary's index
431
- * inside its submit, clamped to 2: the unvisited-count subtraction runs at >= 1, the degree-sum one at >= 2),
432
- * `iteration` @72 (an `sssp-pred` hop pass, P8-T9), `pad1` @76.
433
+ * primitives and `bf-relax` binds -- `role` @0 (the finalize role), `wg` @4 (the consumers' workgroup size),
434
+ * `alpha` @8, `beta` @12 (Beamer's thresholds, P8-T8), `fusedMax` @16, `edgeCapacity` @20, `maxDepth` @24, `n` @28,
435
+ * `mode` @32 (BFS: 0 auto, 1 top-down only; `sssp-pred`: the PD-27 key rule), `cutoffBits` @36, `arcBase` @40,
436
+ * `arcEnd` @44 (the bound arc window), `predKind` @48 (0 arc, 1 node), `bitsBase` @52, `source` @56, `stride` @60
437
+ * (a grid-stride plan's stride), `firstOfSubmit` @64 (the boundary's index inside its submit, clamped to 1: both
438
+ * the unvisited-count and the unvisited-degree-sum subtraction run at >= 1, issue #391), `iteration` @68 (an
439
+ * `sssp-pred` hop pass, P8-T9), `pad1` @72, `pad2` @76. The `slotBase` field that once addressed the selector's indirect slots went with
440
+ * the slots (2026-09-25); `pad2` keeps the block an explicit 80 bytes, the way every block here is padded.
433
441
  */
434
442
  export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams", [
435
443
  ["role", "u32"],
436
- ["slotBase", "u32"],
437
444
  ["wg", "u32"],
438
445
  ["alpha", "u32"],
439
446
  ["beta", "u32"],
@@ -452,6 +459,7 @@ export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams
452
459
  ["firstOfSubmit", "u32"],
453
460
  ["iteration", "u32"],
454
461
  ["pad1", "u32"],
462
+ ["pad2", "u32"],
455
463
  ]);
456
464
 
457
465
  /** `BfParams` (uniform, 16 B; P8-T10): `edgeCount` @0 (the logical edges of the `edgeList` view), `stride` @4 (the grid-stride plan's stride), `maxRetries` @8 (PD-12's compare-exchange bound), `cutoffBits` @12 (the f32 bit pattern of the CPU port's `cutoff`, `+Inf` when absent). */
@@ -938,7 +946,7 @@ const RADIX_SCATTER: KernelEntry = {
938
946
  phase: "P4",
939
947
  };
940
948
 
941
- /** `grid-cell-key` (G1, spec 7.7; P4-T8, PD-10): the finest cell key of every node, `floor((p - gridMin) * invCellSize)` linearised, or the outside pseudo-cell `G^dim`; `cellVal[i] = i`; 4 storage bindings (the state read-only: K1 writes it). */
949
+ /** `grid-cell-key` (G1, spec 7.7; P4-T8, PD-10): the finest cell key of every node, `floor((p - gridMin) * invCellSize)` linearised, or the outside pseudo-cell of its orthant `G^dim + orthant` (issue #90); `cellVal[i] = i`; 4 storage bindings (the state read-only: K1 writes it). */
942
950
  const GRID_CELL_KEY: KernelEntry = {
943
951
  id: "grid-cell-key",
944
952
  body: gridCellKeyWgsl,
@@ -957,7 +965,7 @@ const GRID_CELL_KEY: KernelEntry = {
957
965
  phase: "P4",
958
966
  };
959
967
 
960
- /** `grid-centroid` (G4, spec 7.7; P4-T9, PD-13): thread per finest cell (the pseudo-cell included), the serial mass-weighted sum in sorted order into level 0, the occupancy max into `hubCounters[1]`, hub cells (> GRID_HUB_CELL) appended to `hubList`; 6 storage bindings. */
968
+ /** `grid-centroid` (G4, spec 7.7; P4-T9, PD-13): thread per finest cell (the 2^dim orthant pseudo-cells included), the serial mass-weighted sum in sorted order into level 0, the occupancy max into `hubCounters[1]`, hub cells (> GRID_HUB_CELL) appended to `hubList`; 6 storage bindings. */
961
969
  const GRID_CENTROID: KernelEntry = {
962
970
  id: "grid-centroid",
963
971
  body: gridCentroidWgsl,
@@ -1012,7 +1020,7 @@ const GRID_DOWNSAMPLE: KernelEntry = {
1012
1020
  phase: "P4",
1013
1021
  };
1014
1022
 
1015
- /** `grid-far-field` (G6, spec 7.7; P4-T10, PD-16, DEP-P4-G): per node in sorted order, the coarsest level minus its 3x3 (3x3x3) and, per finer level, the parent's 3x3 refined minus the level's own 3x3, plus the pseudo-cell; the loop bounds are `P.levels` / `P.gridMax`; LAW 0 FA2 / 1 FR / 2 coulomb per cell (P4-T13, PD-22); 5 storage bindings. */
1023
+ /** `grid-far-field` (G6, spec 7.7; P4-T10, PD-16, DEP-P4-G): per node in sorted order, the coarsest level minus its 3x3 (3x3x3) and, per finer level, the parent's 3x3 refined minus the level's own 3x3, plus the 2^dim orthant pseudo-cells; the FA2 term floored at 0.01 (issue #89); the loop bounds are `P.levels` / `P.gridMax`; LAW 0 FA2 / 1 FR / 2 coulomb per cell (P4-T13, PD-22); 5 storage bindings. */
1016
1024
  const GRID_FAR_FIELD: KernelEntry = {
1017
1025
  id: "grid-far-field",
1018
1026
  body: gridFarFieldWgsl,
@@ -1117,16 +1125,12 @@ const DEDUPE_FILTER: KernelEntry = {
1117
1125
  phase: "P8",
1118
1126
  };
1119
1127
 
1120
- /** `frontier-finalize` (design 5.4, 8.10 "BFS finalizeArgs"; P8-T4, PD-3): the one-lane level-boundary selector that rotates the counters block and writes the level's seven indirect slots (role 0), then clamps the edge count and sizes the contract or the fused-retry slot (role 1); 2 storage bindings (the block as `array<atomic<u32>>`, the args). */
1128
+ /** `frontier-finalize` (design 5.4, 8.10 "BFS finalizeArgs"; P8-T4, PD-3): the one-lane level-boundary selector that rotates the counters block and writes the level's `path` word (role 0), then clamps the edge count or switches the path to the fused retry (role 1); 1 storage binding (the block as `array<atomic<u32>>`). Since 2026-09-25 it writes no indirect slots: every level kernel is a direct dispatch gated by the path word. */
1121
1129
  const FRONTIER_FINALIZE: KernelEntry = {
1122
1130
  id: "frontier-finalize",
1123
1131
  body: frontierFinalizeWgsl,
1124
1132
  entryPoint: "frontier_finalize",
1125
- bindings: [
1126
- decl(1, 0, "counters", "storage", "array<atomic<u32>>"),
1127
- decl(1, 1, "args", "storage", "array<u32>"),
1128
- decl(2, 0, "P", "uniform", "FrontierParams"),
1129
- ],
1133
+ bindings: [decl(1, 0, "counters", "storage", "array<atomic<u32>>"), decl(2, 0, "P", "uniform", "FrontierParams")],
1130
1134
  overrideDecls: [],
1131
1135
  uniforms: [FRONTIER_PARAMS],
1132
1136
  needs: [],
@@ -1188,7 +1192,7 @@ const SSSP_PRED: KernelEntry = {
1188
1192
  phase: "P8",
1189
1193
  };
1190
1194
 
1191
- /** `bfs-fused` (design 8.4 "the fused variant", 6 row 8 "the workgroup-per-row tier", 8.10 "BFS fused expand-contract"; P8-T7, PD-23): one level's expansion and contraction in one dispatch, one WORKGROUP per frontier entry, every lane stripping the entry's row with `bfs-contract`'s claim inline and no edge queue traffic; dispatched from `SLOT.fused` (a frontier below `P.fusedMax`) and from `SLOT.fusedRetry` (an overflowed level); 8 storage bindings (the four graph slots, `frontierIn`, the counters block as `array<atomic<u32>>`, `depth` as `array<atomic<u32>>`, `frontierOut`) -- exactly at the budget, which is why no `parent` lives here (PD-24). */
1195
+ /** `bfs-fused` (design 8.4 "the fused variant", 6 row 8 "the workgroup-per-row tier", 8.10 "BFS fused expand-contract"; P8-T7, PD-23): one level's expansion and contraction in one dispatch, one WORKGROUP per frontier entry, every lane stripping the entry's row with `bfs-contract`'s claim inline and no edge queue traffic; a direct grid-stride dispatch that runs when the path word is 2 (a frontier below `P.fusedMax`) or 4 (the overflow retry, role 1's), sized from `frontierCount`; 8 storage bindings (the four graph slots, `frontierIn`, the counters block as `array<atomic<u32>>`, `depth` as `array<atomic<u32>>`, `frontierOut`) -- exactly at the budget, which is why no `parent` lives here (PD-24). */
1192
1196
  const BFS_FUSED: KernelEntry = {
1193
1197
  id: "bfs-fused",
1194
1198
  body: bfsFusedWgsl,
@@ -1264,6 +1268,24 @@ const BFS_UNVISITED_FLAGS: KernelEntry = {
1264
1268
  phase: "P8",
1265
1269
  };
1266
1270
 
1271
+ /** `bfs-next-degree` (design 8.4; issue #391): Beamer's m_f measured exactly -- once per level, after the claim kernels, grid-striding over the output vertex queue and summing the `outDegree` view over the vertices the level claimed into word 25, one `atomicAdd` per workgroup; 3 storage bindings (`frontier` read-only, the `outDegree` VIEW, the counters block); `needs: ["subgroups"]` for the `wg_reduce_u32` call (a twin kernel). */
1272
+ const BFS_NEXT_DEGREE: KernelEntry = {
1273
+ id: "bfs-next-degree",
1274
+ body: bfsNextDegreeWgsl,
1275
+ entryPoint: "bfs_next_degree",
1276
+ bindings: [
1277
+ decl(1, 0, "frontier", "storage-ro", "array<u32>"),
1278
+ decl(1, 1, "outDegree", "storage-ro", "array<u32>"),
1279
+ decl(1, 2, "counters", "storage", "array<atomic<u32>>"),
1280
+ decl(2, 0, "P", "uniform", "FrontierParams"),
1281
+ ],
1282
+ overrideDecls: [],
1283
+ uniforms: [FRONTIER_PARAMS],
1284
+ needs: ["subgroups"],
1285
+ snippetSlots: [],
1286
+ phase: "P8",
1287
+ };
1288
+
1267
1289
  /** `sssp-relax` (design 8.4 "Davidson's near-far", 8.10 "SSSP near-far relax"; P8-T9, PD-9 / PD-20 / DEP-P8-E): one round of the near-far loop -- role 0 relaxes the deduped near pile's whole rows with `atomicMin` on the f32 bit patterns of `dist` and appends each improved vertex to the raw near or far half of `queueOut` (the two halves of ONE buffer at word 0 and word `P.edgeCapacity`), role 1 re-buckets the deduped far pile; 8 storage bindings (the four graph slots with the run's weights bound in the weights slot, `dist` and the counters block as `array<atomic<u32>>`, `queueIn` read-only, `queueOut`) -- exactly at the budget, which is why no `pred` lives here (PD-11) and why the piles' counts, the threshold and the delta are words of the block. */
1268
1290
  const SSSP_RELAX: KernelEntry = {
1269
1291
  id: "sssp-relax",
@@ -1395,6 +1417,7 @@ const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze
1395
1417
  "bfs-bottom-up": BFS_BOTTOM_UP,
1396
1418
  "bfs-bitset-build": BFS_BITSET_BUILD,
1397
1419
  "bfs-unvisited-flags": BFS_UNVISITED_FLAGS,
1420
+ "bfs-next-degree": BFS_NEXT_DEGREE,
1398
1421
  "sssp-relax": SSSP_RELAX,
1399
1422
  "bf-relax": BF_RELAX,
1400
1423
  "closeness-sweep": CLOSENESS_SWEEP,
@@ -26,6 +26,7 @@ import {
26
26
  GRID_EXTENT_FLOOR,
27
27
  LAYOUT_TUNING_DEFAULTS,
28
28
  MAX_ITERATIONS_PER_STEP,
29
+ SETTLE_FLOOR_UNBOUNDED,
29
30
  TRACE_RECORD_BYTES,
30
31
  UNIFORM_SLOT_BYTES,
31
32
  } from "../constants.js";
@@ -590,6 +591,7 @@ export class ForceAtlas2Model implements ForceModel<ForceAtlas2Options, ForceAtl
590
591
  accumulate: 0,
591
592
  hiEnd,
592
593
  midEnd,
594
+ settleFloor: SETTLE_FLOOR_UNBOUNDED, // ForceAtlas2 does not drift after settling (issue #97)
593
595
  };
594
596
  }
595
597
 
@@ -32,6 +32,7 @@ import {
32
32
  FR_REHEAT_FRACTION,
33
33
  FR_START_TEMPERATURE,
34
34
  MAX_ITERATIONS_PER_STEP,
35
+ SETTLE_FLOOR_FRACTION,
35
36
  TRACE_RECORD_BYTES,
36
37
  UNIFORM_SLOT_BYTES,
37
38
  } from "../constants.js";
@@ -84,6 +85,7 @@ import {
84
85
  subset,
85
86
  vector,
86
87
  } from "./model-common.js";
88
+ import { recordExactRepulsion } from "./repulsion-exact.js";
87
89
  import { type GridStage, RepulsionGrid, type RepulsionGridOverrides } from "./repulsion-grid.js";
88
90
 
89
91
  // ============================================================ constants
@@ -587,6 +589,7 @@ export class FruchtermanReingoldModel implements ForceModel<FruchtermanReingoldO
587
589
  hiEnd,
588
590
  midEnd,
589
591
  frK: resolved.k ?? 1 / Math.sqrt(n),
592
+ settleFloor: SETTLE_FLOOR_FRACTION.fruchtermanReingold * (resolved.k ?? 1 / Math.sqrt(n)),
590
593
  temperature: adaptive ? FR_START_TEMPERATURE : this.temperatureAt(iteration, resolved),
591
594
  };
592
595
  }
@@ -640,7 +643,7 @@ export class FruchtermanReingoldModel implements ForceModel<FruchtermanReingoldO
640
643
  if (stop < 2) {
641
644
  return;
642
645
  }
643
- k3.dispatch(pass, k3Bound, bound.plan, [offset]);
646
+ recordExactRepulsion(k3, pass, k3Bound, bound.plan, bound.n, offset);
644
647
  if (stop < STAGE_K5) {
645
648
  return;
646
649
  }