@graphty/webgpu-graph-algorithms 0.6.26 → 0.6.28

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/README.md +56 -6
  2. package/dist/acquire.d.ts +2 -0
  3. package/dist/browser.js +18 -1
  4. package/dist/browser.js.map +1 -1
  5. package/dist/chunks/accelerator-B-FjQwaA.js +19173 -0
  6. package/dist/chunks/accelerator-B-FjQwaA.js.map +1 -0
  7. package/dist/chunks/managed-D_GdQtnu.js +98 -0
  8. package/dist/chunks/managed-D_GdQtnu.js.map +1 -0
  9. package/dist/node.js +18 -1
  10. package/dist/node.js.map +1 -1
  11. package/dist/src/accelerator.d.ts.map +1 -1
  12. package/dist/src/accelerator.js +5 -3
  13. package/dist/src/accelerator.js.map +1 -1
  14. package/dist/src/algorithms/all-pairs.d.ts.map +1 -1
  15. package/dist/src/algorithms/all-pairs.js +72 -47
  16. package/dist/src/algorithms/all-pairs.js.map +1 -1
  17. package/dist/src/algorithms/betweenness.d.ts +1 -1
  18. package/dist/src/algorithms/betweenness.js +2 -2
  19. package/dist/src/algorithms/closeness.d.ts +45 -42
  20. package/dist/src/algorithms/closeness.d.ts.map +1 -1
  21. package/dist/src/algorithms/closeness.js +295 -226
  22. package/dist/src/algorithms/closeness.js.map +1 -1
  23. package/dist/src/browser/index.d.ts +10 -0
  24. package/dist/src/browser/index.d.ts.map +1 -1
  25. package/dist/src/browser/index.js +22 -0
  26. package/dist/src/browser/index.js.map +1 -1
  27. package/dist/src/constants.d.ts +13 -3
  28. package/dist/src/constants.d.ts.map +1 -1
  29. package/dist/src/constants.js +13 -3
  30. package/dist/src/constants.js.map +1 -1
  31. package/dist/src/kernels.d.ts +14 -4
  32. package/dist/src/kernels.d.ts.map +1 -1
  33. package/dist/src/kernels.js +62 -28
  34. package/dist/src/kernels.js.map +1 -1
  35. package/dist/src/layouts/force-simulation.d.ts.map +1 -1
  36. package/dist/src/layouts/force-simulation.js +0 -1
  37. package/dist/src/layouts/force-simulation.js.map +1 -1
  38. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  39. package/dist/src/layouts/forceatlas2.js +0 -1
  40. package/dist/src/layouts/forceatlas2.js.map +1 -1
  41. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  42. package/dist/src/layouts/fruchterman-reingold.js +0 -1
  43. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  44. package/dist/src/layouts/repulsion-grid.d.ts +1 -3
  45. package/dist/src/layouts/repulsion-grid.d.ts.map +1 -1
  46. package/dist/src/layouts/repulsion-grid.js +1 -6
  47. package/dist/src/layouts/repulsion-grid.js.map +1 -1
  48. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  49. package/dist/src/layouts/spring-electrical.js +0 -1
  50. package/dist/src/layouts/spring-electrical.js.map +1 -1
  51. package/dist/src/managed.d.ts +11 -0
  52. package/dist/src/managed.d.ts.map +1 -0
  53. package/dist/src/managed.js +129 -0
  54. package/dist/src/managed.js.map +1 -0
  55. package/dist/src/node/index.d.ts +11 -0
  56. package/dist/src/node/index.d.ts.map +1 -1
  57. package/dist/src/node/index.js +21 -0
  58. package/dist/src/node/index.js.map +1 -1
  59. package/dist/src/primitives/grid-pyramid.d.ts +16 -15
  60. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  61. package/dist/src/primitives/grid-pyramid.js +20 -28
  62. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  63. package/dist/src/types/accelerator.d.ts +2 -0
  64. package/dist/src/types/accelerator.d.ts.map +1 -1
  65. package/dist/src/types/managed.d.ts +81 -0
  66. package/dist/src/types/managed.d.ts.map +1 -0
  67. package/dist/src/types/managed.js +7 -0
  68. package/dist/src/types/managed.js.map +1 -0
  69. package/dist/src/wgsl/bc-forward.wgsl.d.ts +1 -1
  70. package/dist/src/wgsl/bc-forward.wgsl.js +1 -1
  71. package/dist/src/wgsl/closeness-level.wgsl.d.ts +37 -0
  72. package/dist/src/wgsl/closeness-level.wgsl.d.ts.map +1 -0
  73. package/dist/src/wgsl/closeness-level.wgsl.js +204 -0
  74. package/dist/src/wgsl/closeness-level.wgsl.js.map +1 -0
  75. package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts +11 -0
  76. package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts.map +1 -0
  77. package/dist/src/wgsl/closeness-rowsum.wgsl.js +42 -0
  78. package/dist/src/wgsl/closeness-rowsum.wgsl.js.map +1 -0
  79. package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +2 -2
  80. package/dist/src/wgsl/grid-centroid-hub.wgsl.js +2 -2
  81. package/dist/webgpu-graph-algorithms.js +142 -15586
  82. package/dist/webgpu-graph-algorithms.js.map +1 -1
  83. package/package.json +10 -4
  84. package/src/accelerator.ts +5 -3
  85. package/src/algorithms/all-pairs.ts +86 -56
  86. package/src/algorithms/betweenness.ts +2 -2
  87. package/src/algorithms/closeness.ts +353 -256
  88. package/src/browser/index.ts +37 -0
  89. package/src/constants.ts +13 -3
  90. package/src/kernels.ts +65 -36
  91. package/src/layouts/force-simulation.ts +0 -1
  92. package/src/layouts/forceatlas2.ts +0 -1
  93. package/src/layouts/fruchterman-reingold.ts +0 -1
  94. package/src/layouts/repulsion-grid.ts +2 -7
  95. package/src/layouts/spring-electrical.ts +0 -1
  96. package/src/managed.ts +172 -0
  97. package/src/node/index.ts +36 -0
  98. package/src/primitives/grid-pyramid.ts +29 -41
  99. package/src/types/accelerator.ts +2 -0
  100. package/src/types/managed.ts +86 -0
  101. package/src/wgsl/bc-forward.wgsl.ts +1 -1
  102. package/src/wgsl/closeness-level.wgsl.ts +203 -0
  103. package/src/wgsl/closeness-rowsum.wgsl.ts +41 -0
  104. package/src/wgsl/grid-centroid-hub.wgsl.ts +2 -2
  105. package/dist/chunks/context-BZY6SMsM.js +0 -3615
  106. package/dist/chunks/context-BZY6SMsM.js.map +0 -1
  107. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +0 -20
  108. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +0 -1
  109. package/dist/src/wgsl/closeness-reduce.wgsl.js +0 -69
  110. package/dist/src/wgsl/closeness-reduce.wgsl.js.map +0 -1
  111. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +0 -22
  112. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +0 -1
  113. package/dist/src/wgsl/closeness-sweep.wgsl.js +0 -106
  114. package/dist/src/wgsl/closeness-sweep.wgsl.js.map +0 -1
  115. package/src/wgsl/closeness-reduce.wgsl.ts +0 -68
  116. package/src/wgsl/closeness-sweep.wgsl.ts +0 -105
@@ -1,140 +1,153 @@
1
1
  /**
2
- * Closeness centrality on the device (design 8.4, 3.3 line 810, 9.7; P8-T11, the P8 plan's PD-13 / PD-19 / PD-25 /
3
- * DEP-P8-E / DEP-P8-F): `score[s] = 1 / sumDist_s`, with `sumDist_s` the exact sum of the finite distances from `s`
4
- * to every OTHER node -- an unreached node adds nothing -- and `0` when nothing is reached. No reached factor and no
5
- * Wasserman-Faust scaling: this is EXACTLY the legacy default (`normalized: false`) of `closenessCentrality` in
6
- * `@graphty/algorithms`, the number graphty-element's closeness panel shows today, so a future `indexed` port has one
7
- * number to match (the NetworkX form is 33x it on karate and could never have been substituted silently).
2
+ * Closeness centrality on the device (design 8.4, 3.3 line 810, 9.7): `score[s] = 1 / sumDist_s`, with `sumDist_s`
3
+ * the exact sum of the finite distances from `s` to every OTHER node -- an unreached node adds nothing -- and `0` when
4
+ * nothing is reached; or, with `harmonic`, `score[s] = sum of 1 / dist` over the same nodes (a zero distance adds
5
+ * nothing). No reached factor and no Wasserman-Faust scaling: this is EXACTLY the legacy default (`normalized:
6
+ * false`) of `closenessCentrality` in `@graphty/algorithms`, the number graphty-element's closeness panel shows.
8
7
  *
9
- * The unweighted route is ONE bit-parallel multi-source search per batch of 32 sources (`ceil(n / 32)` batches):
10
- * the batch's state is one `bits` buffer of four regions of `bitsBase = roundUp(n, 64)` words (`visited`, two
11
- * frontier regions that swap by the level's parity, `flags`), bit `s` of word `v` meaning "source `s` has reached /
12
- * is at / is next at `v`"; a level is, all host-recorded, `closeness-reduce` role 0 (the boundary: `done` from the
13
- * previous level's compacted count, the level's claims folded into the exact 64-bit per-source sums at `level + 1`),
14
- * `compact` of the flags into the frontier list, two `fill`s zeroing the level's next region and the flags (AFTER
15
- * the compaction that consumed them), and `closeness-sweep` (the block-mapped expansion with the claim inline, a
16
- * direct grid-stride dispatch looping to the list's count). `MAX_LEVELS_PER_SUBMIT` levels per submit and
17
- * one readback per submit (the `done` word and the 512-byte `perSource` block together, so the finished batch needs
18
- * no extra map); the host folds `sumHi x 2^32 + sumLo` into `1 / sum` in f64 and stores f32. The weighted route
19
- * (`weighted` true on a snapshot whose column is not all ones) is one `sssp` per source with the sums reduced on the
20
- * host between calls: design 8.4's own answer, slow and correct. `weighted` defaults to the snapshot's `flags.weighted`;
21
- * `weighted: false` on a weighted snapshot ignores the column by request and sweeps; `weighted: true` over unit
22
- * weights or no column sweeps too (every `sssp` would route to a BFS anyway). `maxIterations` and `tolerance` are the
8
+ * Three routes, chosen from the inputs before any device work:
9
+ *
10
+ * - ALL-PAIRS (an exact run whose `n x n` f32 matrix fits one binding, weighted, or unweighted up to
11
+ * `ALL_PAIRS_MAX_NODES`): `allPairsShortestPath`'s blocked Floyd-Warshall sweep, then `closeness-rowsum` folds each row on
12
+ * the device -- hop counts as exact integers, weighted distances and harmonic reciprocals in f32 -- and only `n`
13
+ * words come back. Weighted scores agree with the one-search-per-source route up to f32 rounding of the sums.
14
+ * - LEVELS (unweighted, every other case): a bit-parallel multi-source breadth-first search, `32 x words` sources per
15
+ * batch (up to `MAX_WORDS` = 8 words, 256 sources), ONE `closeness-level` dispatch per level. Each level chooses on
16
+ * the device between pushing the frontier over the out-arcs (one invocation per arc) and pulling it over the
17
+ * in-arcs (one invocation per node, with an early exit once every source has reached the node), from the arcs the
18
+ * frontier would push against `pullAt`; a graph with a node of more than `PULL_MAX_DEGREE` in-arcs only pushes. The levels of a
19
+ * submit tally their claims per source into one count table, read back once per submit; the host folds the counts
20
+ * into exact sums in f64 (`count x distance`) and harmonic sums (`count / distance`). A level whose predecessor
21
+ * claimed nothing returns at once, so recording `MAX_LEVELS_PER_SUBMIT` levels costs little past a batch's depth.
22
+ * - ONE SEARCH PER SOURCE (weighted above the all-pairs ceiling, or a weighted sampled run): one `sssp` per source,
23
+ * the sums reduced on the host between calls.
24
+ *
25
+ * `weighted` defaults to the snapshot's `flags.weighted`; `weighted: false` on a weighted snapshot ignores the column;
26
+ * `weighted: true` over unit weights or no column is the unweighted problem. `maxIterations` and `tolerance` are the
23
27
  * seam's placeholder keys and an exact traversal has neither, so a defined value is REFUSED before any device work
24
- * (`E_UNSUPPORTED { option }`, the package's rule for an option it does not implement, PD-25); `undefined` is legal.
25
- * `iterations` reports the source batches run (the sources, on the weighted route), `converged` is always true.
26
- * A SAMPLED run (`sources`, issue #426; undirected snapshots only) seeds its batches from the listed sources (the
27
- * reduce's role 2 reads the list the host wrote after the per-node sums in `perSource`), and the sweep also adds each
28
- * claim's distance into a per-node sum (`perNode`), read back with every submit and folded on the host in f64 into
29
- * `1 / sum` per NODE, where the exact run folds per SOURCE: on an undirected graph the distance from a source to a node
30
- * is the distance from the node to the source, which is what the CPU port's sampled closeness sums.
28
+ * (`E_UNSUPPORTED { option }`); `undefined` is legal. `iterations` reports the source batches run (the sources, on the
29
+ * one-search-per-source route; 1 on the all-pairs route), `converged` is always true.
30
+ *
31
+ * A SAMPLED run (`sources`, undirected snapshots only) seeds its batches from the listed sources and adds every
32
+ * claim's distance into a per-node sum, folded on the host into `1 / sum` per NODE: on an undirected graph the
33
+ * distance from a source to a node is the distance from the node to the source, which is what the CPU port's sampled
34
+ * closeness sums. A sampled harmonic run is refused (`E_UNSUPPORTED closenessCentrality.sampledHarmonic`): the
35
+ * per-node reciprocal sums would need a float atomic.
31
36
  *
32
- * Cost, stated so nobody is surprised: closeness is O(n x m) on any device -- at 1M nodes it is 31,250 batches of a
33
- * full multi-source traversal, minutes on the card, and no target in design 10.4 asks for less. `compact.record`
34
- * leases its offsets and the scan's block sums afresh on every call, once per LEVEL here, so the planner is given a
35
- * scope whose `scratch` hands the same buffer back for the same label and size: the dispatches of one pass run in
36
- * order, a level's scan overwrites the previous level's, and the run holds one set instead of `levels x batches`.
37
- * The sweep keeps DEP-P8-E's refusal of a windowed core (`assertWholeCore`): the bit-parallel claim needs the whole
38
- * arc array bound. The tuning entry `closenessWithTuning` (PD-26's shape) is what the tests drive; nothing public
39
- * exposes it.
37
+ * Cost, stated so nobody is surprised: closeness is O(n x m) on any device -- at 1M nodes it is 3,907 batches of a
38
+ * full multi-source traversal. The level kernel binds the whole arc array (`assertWholeCore`) and, on a directed
39
+ * snapshot, the whole reverse adjacency. `closenessWithTuning` is what the tests drive; nothing public exposes it.
40
40
  */
41
41
 
42
- import { type F32, type GraphSnapshot, type U32 } from "@graphty/graph-format";
42
+ import { type F32, type GraphSnapshot } from "@graphty/graph-format";
43
43
 
44
44
  import { MAX_LEVELS_PER_SUBMIT } from "../constants.js";
45
45
  import { type GpuContext } from "../context.js";
46
46
  import { WebGpuGraphError } from "../errors.js";
47
47
  import { CommandBatch } from "../kernel/batch.js";
48
- import { plan1d, planGridStride } from "../kernel/dispatch.js";
49
- import {
50
- FILL_PARAMS,
51
- FRONTIER_COUNTERS,
52
- FRONTIER_PARAMS,
53
- graphBindings,
54
- graphOverrides,
55
- kernelSpec,
56
- } from "../kernels.js";
57
- import { prepareCompact } from "../primitives/compact.js";
58
- import { assertWholeCore } from "../primitives/core-shape.js";
59
- import { W } from "../primitives/frontier.js";
60
- import { type ReduceScope } from "../primitives/reduce.js";
48
+ import { plan1d, plan2d } from "../kernel/dispatch.js";
49
+ import { CLOSENESS_PARAMS, FILL_PARAMS, kernelSpec } from "../kernels.js";
50
+ import { assertWholeCore, coreOfView } from "../primitives/core-shape.js";
61
51
  import { assertDeviceComputes } from "../primitives/verify.js";
62
52
  import { type ClosenessAcceleratorOptions, type HitsOptionsLike } from "../types/accelerator.js";
63
53
  import { type GpuClosenessResult } from "../types/algorithms.js";
64
- import { type Binding } from "../types/memory.js";
65
54
  import { type GpuRunOptions } from "../types/run.js";
55
+ import { allPairsCeiling, DEFAULT_ROUNDS_PER_SUBMIT, sweepAllPairs } from "./all-pairs.js";
66
56
  import { algorithmScope } from "./scope.js";
67
57
  import { aborted, bindingOf, checkDest, sssp } from "./sssp.js";
68
58
 
69
59
  const ALGORITHM = "closenessCentrality";
70
60
 
71
61
  /**
72
- * Design 8.4: 32 sources per `u32` word, one batch per word.
62
+ * The most 32-bit words per node of the level route: `32 x MAX_WORDS` = 256 sources per batch, the size of the
63
+ * kernel's workgroup tally.
73
64
  * @internal
74
65
  */
75
- export const SOURCES_PER_BATCH = 32;
66
+ export const MAX_WORDS = 8;
76
67
 
77
68
  /**
78
- * The `perSource` block of a batch: `newCount[32]` @0, `reached[32]` @32, `sumLo[32]` @64, `sumHi[32]` @96.
79
- * @internal
69
+ * A level pulls when the arcs its frontier would push exceed `words x arcCount / PULL_RATIO`: the push touches every
70
+ * arc whatever the frontier, so it is the cheap step only while few of them claim anything, and the pull's early exit
71
+ * pays once the frontier is large. Measured on the RTX 4070 SUPER (level route, eight words per node) against always
72
+ * pushing and always pulling: uniform random graphs of 4,096 nodes at 64 to 512 arcs per node 9.9 / 13.5 / 19.9 ms
73
+ * against 16.5 / 25.7 / 68.4 pushing and 12.4 / 18.9 / 36.7 pulling; 16,384 nodes at 32 arcs per node 52.7 against
74
+ * 142.7 and 54.1; at 8 arcs per node and below, rings, a path, a grid and a star within noise of the better of the
75
+ * two. 4 and 16 measured within a few percent of 64.
80
76
  */
81
- export const PER_SOURCE_WORDS = 4 * SOURCES_PER_BATCH;
77
+ const PULL_RATIO = 64;
82
78
 
83
79
  /**
84
- * Params slots of the ring, COUNTED (`UniformRing.reserve` wraps silently): per level `compact`'s records (its scan
85
- * is at most four levels for any n below 2^32, so at most 8 records) while the boundary, the finalize, the fill and
86
- * the two sweep records (one per parity) are written once per submit, plus the seed's two records on a batch's first
87
- * submit (the iota fill flushes in its own submit).
80
+ * The most in-arcs a node may have for the level route to pull: a pull invocation walks one node's in-arcs, so its
81
+ * loop count is about `in-degree x (words + 1)`, and llvmpipe silently ends every loop of an invocation past 65,535
82
+ * iterations (`(65,535 - 512) / 9` is 7,225 at eight words). A graph with a larger hub runs every level as a push,
83
+ * one invocation per arc.
84
+ * @internal
88
85
  */
89
- const RING_SLOTS = 8 * MAX_LEVELS_PER_SUBMIT + 16;
86
+ export const PULL_MAX_DEGREE = 4096;
87
+
88
+ /** The control ring at the end of the count table: three slots of four words (`any`, `arcs`, `pull`, pad). */
89
+ const CTRL_WORDS = 12;
90
+
91
+ /** Which route a run took. @internal */
92
+ export type ClosenessRoute = "levels" | "all-pairs" | "per-source";
90
93
 
91
94
  /**
92
- * The knobs the tests need and nothing public offers (PD-26's shape): the submit cadence and the inspect seam.
95
+ * The knobs the tests and the measurements need and nothing public offers.
93
96
  * @internal
94
97
  */
95
98
  export interface ClosenessTuning {
96
- /** Levels recorded per submit on the bit-parallel route (default `MAX_LEVELS_PER_SUBMIT`). */
99
+ /** Levels recorded per submit on the level route (default `MAX_LEVELS_PER_SUBMIT`). */
97
100
  readonly levelsPerSubmit?: number | undefined;
98
- /** The inspect seam: after every BATCH of the bit-parallel route, its first source and a fresh copy of the 128-word `perSource` block as the last submit left it. */
99
- readonly onBatch?: ((batchStart: number, perSource: U32) => void) | undefined;
101
+ /** Words per node on the level route (default: as many as the sources need, at most `MAX_WORDS`). */
102
+ readonly words?: number | undefined;
103
+ /** The frontier arcs above which a level pulls (default `words x arcCount / PULL_RATIO`); 0 pulls whenever the frontier has an arc, `0xFFFFFFFF` never pulls. A graph with a node of more than `PULL_MAX_DEGREE` in-arcs never pulls. */
104
+ readonly pullAt?: number | undefined;
105
+ /** Force a route of an exact run (the all-pairs route still needs the matrix to fit). */
106
+ readonly route?: ClosenessRoute | undefined;
107
+ /** Called with the route the run takes. */
108
+ readonly onRoute?: ((route: ClosenessRoute) => void) | undefined;
109
+ /** Level route, exact run: after every batch, its first source, the exact sum of distances and the nodes reached of each of its sources. */
110
+ readonly onBatch?: ((batchStart: number, sums: Float64Array, reached: Uint32Array) => void) | undefined;
100
111
  }
101
112
 
102
113
  /**
103
- * A scope whose `scratch` hands the SAME buffer back for the same label and size (see the file comment).
104
- * @param scope - the algorithm's scope
105
- * @returns the reusing scope
114
+ * The most nodes an unweighted exact run sends to the all-pairs route: the blocked sweep's `n^3` work beats the level
115
+ * route's `n / 256` batches of a full traversal only on small graphs. Measured on the RTX 4070 SUPER: a ring with
116
+ * chords takes 2.7 ms on the all-pairs route against 3.5 to 4 ms on the level route at 1,024 nodes, 11 against 4.4
117
+ * at 2,048 and 71 against 10 at 4,096; a graph with `n^2 / 12` arcs 0.9 against 1.4 at 512, 2.5 against 2.6 at
118
+ * 1,024, 11 against 6 to 9 at 2,048 and 73 against 24 at 4,096. A deep sparse graph is the exception this does not
119
+ * see: a 4,096-node path takes 70 ms on the all-pairs route and 0.7 s on the level route, one dispatch per level for
120
+ * 4,095 levels.
121
+ * @internal
106
122
  */
107
- function reusingScratch(scope: ReduceScope): ReduceScope {
108
- const held = new Map<string, GPUBuffer>();
109
- return {
110
- ...scope,
111
- scratch: (byteLength, label) => {
112
- const key = `${label}/${byteLength}`;
113
- let buffer = held.get(key);
114
- if (buffer === undefined) {
115
- buffer = scope.scratch(byteLength, label);
116
- held.set(key, buffer);
117
- }
118
- return buffer;
119
- },
120
- };
123
+ export const ALL_PAIRS_MAX_NODES = 1024;
124
+
125
+ /**
126
+ * The score of a sum: `1 / sum`, `0` when nothing was reached.
127
+ * @param sum - the sum of distances
128
+ * @returns the score
129
+ */
130
+ function inverse(sum: number): number {
131
+ return sum === 0 ? 0 : 1 / sum;
121
132
  }
122
133
 
123
134
  /**
124
- * The weighted route: one `sssp` per source, the sums reduced on the host. With `sources` (a sampled run on an
125
- * undirected snapshot) each search adds its distances into the sums of the nodes it reaches instead of its own.
135
+ * The one-search-per-source route: one `sssp` per source, the sums reduced on the host. With `sources` (a sampled run
136
+ * on an undirected snapshot) each search adds its distances into the sums of the nodes it reaches instead of its own.
126
137
  * @param ctx - the context
127
138
  * @param s - the snapshot
128
139
  * @param scores - the destination
129
140
  * @param sources - a sampled run's sources, or null for every node
141
+ * @param harmonic - sum reciprocal distances (exact runs only)
130
142
  * @param options - the run options
131
143
  * @returns the result
132
144
  */
133
- async function weightedRoute(
145
+ async function perSourceRoute(
134
146
  ctx: GpuContext,
135
147
  s: GraphSnapshot,
136
148
  scores: F32,
137
149
  sources: readonly number[] | null,
150
+ harmonic: boolean,
138
151
  options: GpuRunOptions | undefined,
139
152
  ): Promise<GpuClosenessResult> {
140
153
  const n = s.nodeCount;
@@ -150,42 +163,106 @@ async function weightedRoute(
150
163
  for (let v = 0; v < n; v++) {
151
164
  const d = dist[v];
152
165
  if (v !== source && d !== Infinity) {
153
- if (totals === null) {
154
- sum += d;
155
- } else {
166
+ if (totals !== null) {
156
167
  totals[v] += d;
168
+ } else if (!harmonic) {
169
+ sum += d;
170
+ } else if (d > 0) {
171
+ sum += 1 / d;
157
172
  }
158
173
  }
159
174
  }
160
175
  if (totals === null) {
161
- scores[source] = sum === 0 ? 0 : 1 / sum;
176
+ scores[source] = harmonic ? sum : inverse(sum);
162
177
  }
163
178
  options?.onProgress?.(i + 1, count);
164
179
  }
165
- if (totals !== null) {
166
- totals.forEach((sum, v) => {
167
- scores[v] = sum === 0 ? 0 : 1 / sum;
168
- });
169
- }
180
+ totals?.forEach((sum, v) => {
181
+ scores[v] = inverse(sum);
182
+ });
170
183
  return { scores, iterations: count, converged: true, precision: "f32", sourcesUsed: count };
171
184
  }
172
185
 
173
186
  /**
174
- * The bit-parallel route (see the file comment).
187
+ * The all-pairs route (see the file comment). The caller checked that the matrix fits.
188
+ * @param ctx - the context
189
+ * @param s - the snapshot
190
+ * @param scores - the destination
191
+ * @param weighted - sum the weights, or count hops
192
+ * @param harmonic - sum reciprocal distances
193
+ * @param options - the run options
194
+ * @returns the result
195
+ */
196
+ async function allPairsRoute(
197
+ ctx: GpuContext,
198
+ s: GraphSnapshot,
199
+ scores: F32,
200
+ weighted: boolean,
201
+ harmonic: boolean,
202
+ options: GpuRunOptions | undefined,
203
+ ): Promise<GpuClosenessResult> {
204
+ const n = s.nodeCount;
205
+ const scope = algorithmScope(ctx, ALGORITHM, Math.min(DEFAULT_ROUNDS_PER_SUBMIT, Math.ceil(n / 32)) + 3);
206
+ try {
207
+ const matrix = await sweepAllPairs(
208
+ ctx,
209
+ s,
210
+ scope,
211
+ weighted,
212
+ DEFAULT_ROUNDS_PER_SUBMIT,
213
+ { signal: options?.signal },
214
+ ALGORITHM,
215
+ );
216
+ const rowsum = await ctx.pipelines.kernel(kernelSpec("closeness-rowsum"));
217
+ const out = bindingOf(scope.scratch(4 * n, "row-sums"), 4 * n);
218
+ const role = harmonic ? 2 : Number(weighted);
219
+ const params = scope.params(CLOSENESS_PARAMS, { role, n });
220
+ const batch = new CommandBatch(ctx, `${ALGORITHM}/row-sums`);
221
+ rowsum.dispatch(
222
+ batch.pass("row-sums"),
223
+ rowsum.bind({ dist: matrix, out, P: params.binding }),
224
+ plan2d(n, ctx.caps),
225
+ [params.offset],
226
+ );
227
+ batch.endPass();
228
+ const request = batch.readback(out.buffer, out.offset, 4 * n);
229
+ scope.flush();
230
+ const back = await batch.submit().readback;
231
+ ctx.assertReady();
232
+ if (role === 0) {
233
+ new Uint32Array(back, request.offset, n).forEach((sum, v) => {
234
+ scores[v] = inverse(sum);
235
+ });
236
+ } else {
237
+ new Float32Array(back, request.offset, n).forEach((sum, v) => {
238
+ scores[v] = harmonic ? sum : inverse(sum);
239
+ });
240
+ }
241
+ options?.onProgress?.(n, n);
242
+ return { scores, iterations: 1, converged: true, precision: "f32", sourcesUsed: n };
243
+ } finally {
244
+ scope.dispose();
245
+ }
246
+ }
247
+
248
+ /**
249
+ * The level route (see the file comment).
175
250
  * @param ctx - the context
176
251
  * @param s - the snapshot
177
252
  * @param scores - the destination
178
253
  * @param sources - a sampled run's sources, or null for every node
254
+ * @param harmonic - sum reciprocal distances (exact runs only)
179
255
  * @param levelsPerSubmit - the submit cadence
180
256
  * @param options - the run options
181
257
  * @param tuning - the knobs
182
258
  * @returns the result
183
259
  */
184
- async function sweepRoute(
260
+ async function levelRoute(
185
261
  ctx: GpuContext,
186
262
  s: GraphSnapshot,
187
263
  scores: F32,
188
264
  sources: readonly number[] | null,
265
+ harmonic: boolean,
189
266
  levelsPerSubmit: number,
190
267
  options: GpuRunOptions | undefined,
191
268
  tuning: ClosenessTuning,
@@ -195,170 +272,166 @@ async function sweepRoute(
195
272
  if (seedCount === 0) {
196
273
  return { scores, iterations: 0, converged: true, precision: "f32", sourcesUsed: 0 };
197
274
  }
275
+ const limit = Math.min(ctx.caps.limits.maxStorageBufferBindingSize, ctx.caps.limits.maxBufferSize);
198
276
  const core = ctx.residency.core(s);
199
277
  assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM);
200
- const scope = algorithmScope(ctx, ALGORITHM, RING_SLOTS);
278
+ if (s.directed && 4 * s.arcCount > limit) {
279
+ throw new WebGpuGraphError(
280
+ "E_TOO_LARGE",
281
+ `${ALGORITHM}: the reverse adjacency of a directed snapshot (${4 * s.arcCount} bytes) does not fit one binding`,
282
+ { needed: 4 * s.arcCount, limit, path: "closeness.reverse", algorithm: ALGORITHM },
283
+ );
284
+ }
285
+ const reverse = s.directed ? coreOfView(ctx.residency.view(s, "reverse"), s.arcCount) : core;
286
+ // words per node: as many as the sources need, within one binding of four regions (plus a sampled run's per-node
287
+ // sums) and, for a sampled run, so a node's per-batch distance sum (at most 32 words (n - 1)) fits a u32
288
+ const extra = sources === null ? 0 : n;
289
+ const fitWords = Math.floor((limit / 4 - extra) / (4 * n));
290
+ const sumWords = sources === null ? MAX_WORDS : Math.floor(0xffffffff / (32 * Math.max(1, n - 1)));
291
+ const words = tuning.words ?? Math.max(1, Math.min(MAX_WORDS, Math.ceil(seedCount / 32), fitWords, sumWords));
292
+ if (!Number.isInteger(words) || words < 1 || words > MAX_WORDS) {
293
+ throw new WebGpuGraphError(
294
+ "E_INVALID_ARGUMENT",
295
+ `${ALGORITHM}: words must be an integer in [1, ${MAX_WORDS}]`,
296
+ {
297
+ argument: "words",
298
+ value: words,
299
+ expected: `an integer in [1, ${MAX_WORDS}]`,
300
+ },
301
+ );
302
+ }
303
+ const base = n * words;
304
+ const bitsWords = 4 * base + extra;
305
+ if (4 * bitsWords > limit) {
306
+ throw new WebGpuGraphError(
307
+ "E_TOO_LARGE",
308
+ `${ALGORITHM}: ${n} nodes need a ${4 * bitsWords}-byte search state in one binding of at most ${limit} bytes`,
309
+ { needed: 4 * bitsWords, limit, path: "closeness.bits", algorithm: ALGORITHM },
310
+ );
311
+ }
312
+ const lanes = 32 * words;
313
+ const rowWords = levelsPerSubmit * lanes;
314
+ const ctrl = rowWords;
315
+ const sourcesAt = sources === null ? 0 : ctrl + CTRL_WORDS;
316
+ const tableWords = ctrl + CTRL_WORDS + (sources?.length ?? 0);
317
+ const pullAt = tuning.pullAt ?? Math.min(0xffffffff, Math.floor((words * s.arcCount) / PULL_RATIO));
318
+ // slots: the row fill and the levels of one submit, plus the two seed records of a batch's first submit
319
+ const scope = algorithmScope(ctx, ALGORITHM, levelsPerSubmit + 4);
201
320
  try {
202
321
  const wg = ctx.workgroupSize;
203
- const bytes = 4 * n;
204
- // the four regions of the bits buffer, each bitsBase words so its byte offset is 256-aligned and fill and
205
- // compact can bind one alone: visited 0, the frontier pair 1 and 2, flags 3
206
- const bitsBase = Math.ceil(n / 64) * 64;
207
- const regionBytes = 4 * bitsBase;
208
- const bits = bindingOf(scope.scratch(4 * regionBytes, "bits"), 4 * regionBytes);
209
- const region = (index: number): Binding => ({
210
- buffer: bits.buffer,
211
- offset: index * regionBytes,
212
- size: regionBytes,
213
- window: null,
214
- });
215
- const flags = region(3);
216
- const frontierList = bindingOf(scope.scratch(bytes, "frontier-list"), bytes);
217
- const iota = bindingOf(scope.scratch(bytes, "iota"), bytes);
218
- const counters = bindingOf(
219
- scope.scratch(FRONTIER_COUNTERS.byteLength, "counters"),
220
- FRONTIER_COUNTERS.byteLength,
221
- );
222
- const perSourceBytes = 4 * PER_SOURCE_WORDS;
223
- // a sampled run appends the per-node distance sums (bitsBase words) and then its source list
224
- const zeroedWords = PER_SOURCE_WORDS + (sources === null ? 0 : bitsBase);
225
- const perSourceAll = 4 * (zeroedWords + (sources === null ? 0 : sources.length));
226
- const perSource = bindingOf(scope.scratch(perSourceAll, "per-source"), perSourceAll);
322
+ const bits = bindingOf(scope.scratch(4 * bitsWords, "bits"), 4 * bitsWords);
323
+ const table = bindingOf(scope.scratch(4 * tableWords, "table"), 4 * tableWords);
324
+ const rows = { ...table, size: 4 * rowWords };
227
325
  if (sources !== null) {
228
- ctx.device.queue.writeBuffer(
229
- perSource.buffer,
230
- perSource.offset + 4 * zeroedWords,
231
- Uint32Array.from(sources),
232
- );
326
+ ctx.device.queue.writeBuffer(table.buffer, table.offset + 4 * sourcesAt, Uint32Array.from(sources));
233
327
  }
234
- const totals = sources === null ? null : new Float64Array(n);
235
328
  await ctx.allocator.check();
236
- const compact = await prepareCompact(reusingScratch(scope));
237
- const sweep = await ctx.pipelines.kernel(kernelSpec("closeness-sweep", graphOverrides(core, null)));
238
- const reduce = await ctx.pipelines.kernel(kernelSpec("closeness-reduce"));
329
+ const level = await ctx.pipelines.kernel(kernelSpec("closeness-level"));
239
330
  const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
240
- const graph = graphBindings(core, null);
241
- const onePlan = plan1d(1, wg, ctx.caps);
242
- const regionPlan = plan1d(bitsBase, wg, ctx.caps);
243
- const sweepPlan = planGridStride(n, wg, ctx.caps); // the list holds at most n entries; the sweep loops to the count word
244
- const recordFill = (pass: GPUComputePassEncoder, dst: Binding, count: number, mode: 0 | 1): void => {
245
- const params = scope.params(FILL_PARAMS, { count, value: 0, mode, pad0: 0 });
246
- fill.dispatch(pass, fill.bind({ dst, P: params.binding }), plan1d(count, wg, ctx.caps), [params.offset]);
331
+ const state = {
332
+ rowPtr: core.rowPtr,
333
+ colIdx: core.colIdx ?? core.rowPtr,
334
+ inRowPtr: reverse.rowPtr,
335
+ inColIdx: reverse.colIdx ?? reverse.rowPtr,
336
+ bits,
337
+ table,
247
338
  };
248
- const submit = (batch: CommandBatch): ReturnType<CommandBatch["submit"]> => {
249
- scope.flush();
250
- return batch.submit();
339
+ const levelPlan = plan1d(Math.max(n, s.arcCount), wg, ctx.caps);
340
+ const pullOk = s.inDegree().every((d) => d <= PULL_MAX_DEGREE) ? 1 : 0;
341
+ const totals = sources === null ? null : new Float64Array(n);
342
+ const shared = {
343
+ n,
344
+ words,
345
+ base,
346
+ ctrl,
347
+ pullAt,
348
+ perNode: sources === null ? 0 : 1,
349
+ sourcesAt,
350
+ arcCount: s.arcCount,
351
+ pullOk,
251
352
  };
252
353
 
253
- // setup: the iota queue compact reads, once per run
254
- const setup = new CommandBatch(ctx, `${ALGORITHM}/setup`);
255
- recordFill(setup.pass("fill"), iota, n, 1);
256
- setup.endPass();
257
- await submit(setup).readback;
258
- ctx.assertReady();
259
-
260
354
  let batches = 0;
261
- for (let batchStart = 0; batchStart < seedCount; batchStart += SOURCES_PER_BATCH) {
262
- let level = 0;
263
- for (let first = true; ; first = false) {
355
+ for (let batchStart = 0; batchStart < seedCount; batchStart += lanes) {
356
+ const count = Math.min(lanes, seedCount - batchStart);
357
+ const sums = new Float64Array(count);
358
+ const reciprocal = new Float64Array(count);
359
+ const reached = new Uint32Array(count);
360
+ for (let firstLevel = 0, done = false; !done; firstLevel += levelsPerSubmit) {
361
+ if (firstLevel > n + 1) {
362
+ // a batch claims at most n - 1 levels deep, then one level claims nothing
363
+ throw new WebGpuGraphError(
364
+ "E_VALIDATION",
365
+ `${ALGORITHM}: the batch at ${batchStart} still claimed at level ${firstLevel}`,
366
+ { label: ALGORITHM, message: `the batch still claimed at level ${firstLevel}` },
367
+ );
368
+ }
264
369
  const batch = new CommandBatch(ctx, `${ALGORITHM}/levels`);
265
370
  const pass = batch.pass("closeness");
266
- if (first) {
267
- // the batch's seed: the four regions and the block zeroed, then role 1 (the sources' bits, their
268
- // flags, counters[0] = k, level = U32_MAX)
269
- recordFill(pass, bits, 4 * bitsBase, 0);
270
- recordFill(pass, perSource, zeroedWords, 0);
271
- const seed = scope.params(FRONTIER_PARAMS, {
272
- role: sources === null ? 1 : 2,
273
- n: seedCount,
274
- bitsBase,
275
- source: batchStart,
276
- });
277
- reduce.dispatch(pass, reduce.bind({ counters, perSource, bits, P: seed.binding }), onePlan, [
278
- seed.offset,
279
- ]);
371
+ if (firstLevel === 0) {
372
+ const clear = scope.params(CLOSENESS_PARAMS, { ...shared, role: 1, total: bitsWords, count });
373
+ const bound = level.bind({ ...state, P: clear.binding });
374
+ level.dispatch(pass, bound, plan1d(bitsWords, wg, ctx.caps), [clear.offset]);
375
+ const seed = scope.params(CLOSENESS_PARAMS, { ...shared, role: 2, count, source: batchStart });
376
+ level.dispatch(pass, bound, plan1d(count, wg, ctx.caps), [seed.offset]);
280
377
  }
281
- // the records every level of the submit shares (the ring wraps, so they are written per submit)
282
- const boundary = scope.params(FRONTIER_PARAMS, { role: 0, n, bitsBase });
283
- const boundBoundary = reduce.bind({ counters, perSource, bits, P: boundary.binding });
284
- const clear = scope.params(FILL_PARAMS, { count: bitsBase, value: 0, mode: 0, pad0: 0 });
285
- // parity 0 sweeps region 1 into region 2, parity 1 region 2 into region 1: the next region is cleared
286
- const boundClearNext = [region(2), region(1)].map((dst) => fill.bind({ dst, P: clear.binding }));
287
- const boundClearFlags = fill.bind({ dst: flags, P: clear.binding });
288
- const boundSweep = [0, 1].map((mode) => {
289
- const params = scope.params(FRONTIER_PARAMS, {
290
- wg,
291
- n,
292
- bitsBase,
293
- arcBase: 0,
294
- arcEnd: s.arcCount,
295
- mode,
296
- stride: sweepPlan.stride ?? wg,
297
- perNode: sources === null ? 0 : 1,
298
- });
299
- return {
300
- bound: sweep.bind({ ...graph, frontierList, counters, bits, perSource, P: params.binding }),
301
- offset: params.offset,
302
- };
303
- });
304
- for (let k = 0; k < levelsPerSubmit; k++, level++) {
305
- const parity = level % 2;
306
- reduce.dispatch(pass, boundBoundary, onePlan, [boundary.offset]);
307
- compact.record(pass, {
308
- queue: iota,
309
- flags,
310
- count: n,
311
- out: frontierList,
312
- outCount: counters,
313
- outIndex: W.frontierCount,
378
+ const zero = scope.params(FILL_PARAMS, { count: rowWords, value: 0, mode: 0, pad0: 0 });
379
+ fill.dispatch(pass, fill.bind({ dst: rows, P: zero.binding }), plan1d(rowWords, wg, ctx.caps), [
380
+ zero.offset,
381
+ ]);
382
+ let bound: ReturnType<typeof level.bind> | null = null;
383
+ for (let row = 0; row < levelsPerSubmit; row++) {
384
+ const params = scope.params(CLOSENESS_PARAMS, {
385
+ ...shared,
386
+ role: 0,
387
+ count,
388
+ level: firstLevel + row,
389
+ row,
314
390
  });
315
- fill.dispatch(pass, boundClearNext[parity], regionPlan, [clear.offset]);
316
- fill.dispatch(pass, boundClearFlags, regionPlan, [clear.offset]);
317
- sweep.dispatch(pass, boundSweep[parity].bound, sweepPlan, [boundSweep[parity].offset]);
391
+ bound ??= level.bind({ ...state, P: params.binding });
392
+ level.dispatch(pass, bound, levelPlan, [params.offset]);
318
393
  }
319
394
  batch.endPass();
320
- const doneRequest = batch.readback(counters.buffer, counters.offset + 4 * W.done, 4);
321
- const blockRequest = batch.readback(perSource.buffer, perSource.offset, perSourceBytes);
395
+ const rowRequest = batch.readback(table.buffer, table.offset, 4 * rowWords);
322
396
  const nodeRequest =
323
- totals === null ? null : batch.readback(perSource.buffer, perSource.offset + perSourceBytes, 4 * n);
324
- const submitted = submit(batch);
397
+ totals === null ? null : batch.readback(bits.buffer, bits.offset + 16 * base, 4 * n);
398
+ scope.flush();
399
+ const submitted = batch.submit();
325
400
  const back = await submitted.readback;
326
401
  ctx.assertReady();
327
402
  if (options?.signal?.aborted) {
328
403
  throw aborted(ALGORITHM, submitted.id);
329
404
  }
330
- if (new Uint32Array(back, doneRequest.offset, 1)[0] !== 0) {
331
- const block = new Uint32Array(back, blockRequest.offset, PER_SOURCE_WORDS);
332
- if (totals === null || nodeRequest === null) {
333
- const count = Math.min(SOURCES_PER_BATCH, n - batchStart);
334
- for (let i = 0; i < count; i++) {
335
- const sum = block[3 * SOURCES_PER_BATCH + i] * 2 ** 32 + block[2 * SOURCES_PER_BATCH + i];
336
- scores[batchStart + i] = sum === 0 ? 0 : 1 / sum;
337
- }
338
- } else {
339
- // at most 32 (n - 1) per node per batch, so a u32 word never wraps below 134M nodes
340
- const sums = new Uint32Array(back, nodeRequest.offset, n);
341
- for (let v = 0; v < n; v++) {
342
- totals[v] += sums[v];
343
- }
405
+ const counts = new Uint32Array(back, rowRequest.offset, rowWords);
406
+ for (let row = 0; row < levelsPerSubmit && !done; row++) {
407
+ const distance = firstLevel + row + 1;
408
+ let claimed = 0;
409
+ for (let lane = 0; lane < count; lane++) {
410
+ const c = counts[row * lanes + lane];
411
+ sums[lane] += c * distance;
412
+ reciprocal[lane] += c / distance;
413
+ reached[lane] += c;
414
+ claimed += c;
344
415
  }
345
- tuning.onBatch?.(batchStart, block.slice());
346
- break;
416
+ done = claimed === 0;
347
417
  }
348
- if (level > n + 3) {
349
- // a batch claims at most n - 1 levels deep, then one level claims nothing and one is empty
350
- throw new WebGpuGraphError(
351
- "E_VALIDATION",
352
- `${ALGORITHM}: the done flag never rose in ${level} levels of the batch at ${batchStart}`,
353
- { label: ALGORITHM, message: `the done flag never rose in ${level} levels` },
354
- );
418
+ if (done && totals !== null && nodeRequest !== null) {
419
+ new Uint32Array(back, nodeRequest.offset, n).forEach((sum, v) => {
420
+ totals[v] += sum;
421
+ });
422
+ }
423
+ }
424
+ if (sources === null) {
425
+ for (let lane = 0; lane < count; lane++) {
426
+ scores[batchStart + lane] = harmonic ? reciprocal[lane] : inverse(sums[lane]);
355
427
  }
428
+ tuning.onBatch?.(batchStart, sums, reached);
356
429
  }
357
430
  batches += 1;
358
- options?.onProgress?.(Math.min(batchStart + SOURCES_PER_BATCH, seedCount), seedCount);
431
+ options?.onProgress?.(batchStart + count, seedCount);
359
432
  }
360
433
  totals?.forEach((sum, v) => {
361
- scores[v] = sum === 0 ? 0 : 1 / sum;
434
+ scores[v] = inverse(sum);
362
435
  });
363
436
  return { scores, iterations: batches, converged: true, precision: "f32", sourcesUsed: seedCount };
364
437
  } finally {
@@ -398,11 +471,11 @@ function checkSources(s: GraphSnapshot, sources: readonly number[] | undefined):
398
471
  }
399
472
 
400
473
  /**
401
- * Closeness with the test knobs of PD-26's shape; `closenessCentrality` is this with an empty tuning.
474
+ * Closeness with the test knobs; `closenessCentrality` is this with an empty tuning.
402
475
  * @internal
403
476
  * @param ctx - the context whose device runs the kernels
404
477
  * @param s - the snapshot (uploaded through ctx.residency, or found there)
405
- * @param options - `weighted` and a sampled run's `sources` honoured, the placeholder `maxIterations` / `tolerance` refused when defined, plus dest / signal / onProgress
478
+ * @param options - `weighted`, `harmonic` and a sampled run's `sources` honoured, the placeholder `maxIterations` / `tolerance` refused when defined, plus dest / signal / onProgress
406
479
  * @param tuning - the knobs
407
480
  * @returns the scores, the batches run, `converged: true`, `precision: "f32"` and `sourcesUsed`
408
481
  */
@@ -424,6 +497,13 @@ export async function closenessWithTuning(
424
497
  }
425
498
  const n = s.nodeCount;
426
499
  const sources = checkSources(s, options?.sources);
500
+ const harmonic = options?.harmonic === true;
501
+ if (harmonic && sources !== null) {
502
+ throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: harmonic closeness from sampled sources`, {
503
+ feature: "closenessCentrality.sampledHarmonic",
504
+ hint: "run the CPU port, or drop the sources for the exact harmonic score",
505
+ });
506
+ }
427
507
  const levelsPerSubmit = tuning.levelsPerSubmit ?? MAX_LEVELS_PER_SUBMIT;
428
508
  if (!Number.isInteger(levelsPerSubmit) || levelsPerSubmit < 1 || levelsPerSubmit > MAX_LEVELS_PER_SUBMIT) {
429
509
  throw new WebGpuGraphError(
@@ -437,45 +517,62 @@ export async function closenessWithTuning(
437
517
  );
438
518
  }
439
519
  const scores = checkDest(ALGORITHM, options?.dest, n) ?? new Float32Array(n);
440
- const weighted = options?.weighted ?? s.flags.weighted;
520
+ const weighted = (options?.weighted ?? s.flags.weighted) && s.weights !== null && !s.flags.allWeightsOne;
441
521
  if (options?.signal?.aborted) {
442
522
  throw aborted(ALGORITHM);
443
523
  }
444
- if (weighted && s.weights !== null && !s.flags.allWeightsOne) {
445
- if (!s.flags.nonNegativeWeights) {
446
- throw new WebGpuGraphError(
447
- "E_UNSUPPORTED",
448
- `${ALGORITHM}: a negative weight has no shortest-path distance to sum`,
449
- {
450
- feature: "closenessCentrality.negativeWeights",
451
- hint: "pass weighted: false to ignore the column",
452
- },
453
- );
454
- }
455
- if (!s.flags.finiteWeights) {
456
- throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: a NaN or infinite weight has no shortest path`, {
457
- feature: "closenessCentrality.nonFiniteWeights",
458
- });
459
- }
460
- return weightedRoute(ctx, s, scores, sources, options);
524
+ if (weighted && !s.flags.nonNegativeWeights) {
525
+ throw new WebGpuGraphError(
526
+ "E_UNSUPPORTED",
527
+ `${ALGORITHM}: a negative weight has no shortest-path distance to sum`,
528
+ {
529
+ feature: "closenessCentrality.negativeWeights",
530
+ hint: "pass weighted: false to ignore the column",
531
+ },
532
+ );
533
+ }
534
+ if (weighted && !s.flags.finiteWeights) {
535
+ throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: a NaN or infinite weight has no shortest path`, {
536
+ feature: "closenessCentrality.nonFiniteWeights",
537
+ });
538
+ }
539
+ const fits = n > 0 && n <= allPairsCeiling(ctx.caps.limits).maxNodes;
540
+ let route: ClosenessRoute;
541
+ if (sources !== null || !fits) {
542
+ route = weighted ? "per-source" : "levels";
543
+ } else {
544
+ route = tuning.route ?? (weighted || n <= ALL_PAIRS_MAX_NODES ? "all-pairs" : "levels");
545
+ }
546
+ if (route === "levels" && weighted) {
547
+ route = "per-source";
548
+ }
549
+ tuning.onRoute?.(route);
550
+ if (route === "all-pairs") {
551
+ return allPairsRoute(ctx, s, scores, weighted, harmonic, options);
552
+ }
553
+ if (route === "per-source") {
554
+ return perSourceRoute(ctx, s, scores, sources, harmonic, options);
461
555
  }
462
- return sweepRoute(ctx, s, scores, sources, levelsPerSubmit, options, tuning);
556
+ return levelRoute(ctx, s, scores, sources, harmonic, levelsPerSubmit, options, tuning);
463
557
  }
464
558
 
465
559
  /**
466
560
  * Closeness centrality on the device (spec 3.3 line 810, design 8.4, 9.7): `scores[s] = 1 / sumDist_s` over the finite
467
561
  * distances from `s` to every other node, `0` when nothing is reached -- the legacy default of `@graphty/algorithms`'
468
- * `closenessCentrality`, unweighted by one bit-parallel multi-source search per 32 sources, weighted by one `sssp`
469
- * per source; `weighted` defaults to the snapshot's flag, `maxIterations` / `tolerance` are refused when defined
470
- * (PD-25). `iterations` is the source batches run and `converged` is always true.
562
+ * `closenessCentrality` -- or, with `harmonic`, the sum of `1 / dist` over the same nodes. Small and weighted
563
+ * graphs whose all-pairs matrix fits one binding run the all-pairs sweep and a row sum; other unweighted graphs run a
564
+ * bit-parallel multi-source search of up to 256 sources per batch; weighted graphs above the all-pairs ceiling run one
565
+ * `sssp` per source. `weighted` defaults to the snapshot's flag, `maxIterations` / `tolerance` are refused when
566
+ * defined. `iterations` is the source batches run and `converged` is always true.
471
567
  *
472
568
  * SAMPLED (`sources`, node indices, duplicates run twice; undirected snapshots only, E_UNSUPPORTED
473
- * `closenessCentrality.directedSources` otherwise): the batches seed the listed sources instead of every node, and
474
- * each node's score is `1 / sum` of its distances to the sources that reach it (itself excluded), `0` when none does:
475
- * the sampled score of the CPU port, unscaled. `sourcesUsed` is the list's length (`n` exact).
569
+ * `closenessCentrality.directedSources` otherwise; not with `harmonic`, E_UNSUPPORTED
570
+ * `closenessCentrality.sampledHarmonic`): the batches seed the listed sources instead of every node, and each node's
571
+ * score is `1 / sum` of its distances to the sources that reach it (itself excluded), `0` when none does: the sampled
572
+ * score of the CPU port, unscaled. `sourcesUsed` is the list's length (`n` exact).
476
573
  * @param ctx - the context whose device runs the kernels
477
574
  * @param s - the snapshot (uploaded through ctx.residency, or found there)
478
- * @param options - `weighted`, `sources`, plus dest (a Float32Array of length n for `scores`) / signal / onProgress
575
+ * @param options - `weighted`, `harmonic`, `sources`, plus dest (a Float32Array of length n for `scores`) / signal / onProgress
479
576
  * @returns the scores, the batches run, `converged: true`, `precision: "f32"` and `sourcesUsed`
480
577
  */
481
578
  export function closenessCentrality(