@graphty/webgpu-graph-algorithms 0.6.27 → 0.6.28

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/README.md +56 -6
  2. package/dist/acquire.d.ts +2 -0
  3. package/dist/browser.js +18 -1
  4. package/dist/browser.js.map +1 -1
  5. package/dist/chunks/accelerator-B-FjQwaA.js +19173 -0
  6. package/dist/chunks/accelerator-B-FjQwaA.js.map +1 -0
  7. package/dist/chunks/managed-D_GdQtnu.js +98 -0
  8. package/dist/chunks/managed-D_GdQtnu.js.map +1 -0
  9. package/dist/node.js +18 -1
  10. package/dist/node.js.map +1 -1
  11. package/dist/src/accelerator.d.ts.map +1 -1
  12. package/dist/src/accelerator.js +5 -3
  13. package/dist/src/accelerator.js.map +1 -1
  14. package/dist/src/algorithms/all-pairs.d.ts.map +1 -1
  15. package/dist/src/algorithms/all-pairs.js +72 -47
  16. package/dist/src/algorithms/all-pairs.js.map +1 -1
  17. package/dist/src/algorithms/betweenness.d.ts +1 -1
  18. package/dist/src/algorithms/betweenness.js +2 -2
  19. package/dist/src/algorithms/closeness.d.ts +45 -42
  20. package/dist/src/algorithms/closeness.d.ts.map +1 -1
  21. package/dist/src/algorithms/closeness.js +295 -226
  22. package/dist/src/algorithms/closeness.js.map +1 -1
  23. package/dist/src/browser/index.d.ts +10 -0
  24. package/dist/src/browser/index.d.ts.map +1 -1
  25. package/dist/src/browser/index.js +22 -0
  26. package/dist/src/browser/index.js.map +1 -1
  27. package/dist/src/constants.d.ts +13 -3
  28. package/dist/src/constants.d.ts.map +1 -1
  29. package/dist/src/constants.js +13 -3
  30. package/dist/src/constants.js.map +1 -1
  31. package/dist/src/kernels.d.ts +14 -4
  32. package/dist/src/kernels.d.ts.map +1 -1
  33. package/dist/src/kernels.js +62 -28
  34. package/dist/src/kernels.js.map +1 -1
  35. package/dist/src/layouts/force-simulation.d.ts.map +1 -1
  36. package/dist/src/layouts/force-simulation.js +0 -1
  37. package/dist/src/layouts/force-simulation.js.map +1 -1
  38. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  39. package/dist/src/layouts/forceatlas2.js +0 -1
  40. package/dist/src/layouts/forceatlas2.js.map +1 -1
  41. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  42. package/dist/src/layouts/fruchterman-reingold.js +0 -1
  43. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  44. package/dist/src/layouts/repulsion-grid.d.ts +1 -3
  45. package/dist/src/layouts/repulsion-grid.d.ts.map +1 -1
  46. package/dist/src/layouts/repulsion-grid.js +1 -6
  47. package/dist/src/layouts/repulsion-grid.js.map +1 -1
  48. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  49. package/dist/src/layouts/spring-electrical.js +0 -1
  50. package/dist/src/layouts/spring-electrical.js.map +1 -1
  51. package/dist/src/managed.d.ts +11 -0
  52. package/dist/src/managed.d.ts.map +1 -0
  53. package/dist/src/managed.js +129 -0
  54. package/dist/src/managed.js.map +1 -0
  55. package/dist/src/node/index.d.ts +11 -0
  56. package/dist/src/node/index.d.ts.map +1 -1
  57. package/dist/src/node/index.js +21 -0
  58. package/dist/src/node/index.js.map +1 -1
  59. package/dist/src/primitives/grid-pyramid.d.ts +16 -15
  60. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  61. package/dist/src/primitives/grid-pyramid.js +20 -28
  62. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  63. package/dist/src/types/accelerator.d.ts +2 -0
  64. package/dist/src/types/accelerator.d.ts.map +1 -1
  65. package/dist/src/types/managed.d.ts +81 -0
  66. package/dist/src/types/managed.d.ts.map +1 -0
  67. package/dist/src/types/managed.js +7 -0
  68. package/dist/src/types/managed.js.map +1 -0
  69. package/dist/src/wgsl/bc-forward.wgsl.d.ts +1 -1
  70. package/dist/src/wgsl/bc-forward.wgsl.js +1 -1
  71. package/dist/src/wgsl/closeness-level.wgsl.d.ts +37 -0
  72. package/dist/src/wgsl/closeness-level.wgsl.d.ts.map +1 -0
  73. package/dist/src/wgsl/closeness-level.wgsl.js +204 -0
  74. package/dist/src/wgsl/closeness-level.wgsl.js.map +1 -0
  75. package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts +11 -0
  76. package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts.map +1 -0
  77. package/dist/src/wgsl/closeness-rowsum.wgsl.js +42 -0
  78. package/dist/src/wgsl/closeness-rowsum.wgsl.js.map +1 -0
  79. package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +2 -2
  80. package/dist/src/wgsl/grid-centroid-hub.wgsl.js +2 -2
  81. package/dist/webgpu-graph-algorithms.js +142 -15586
  82. package/dist/webgpu-graph-algorithms.js.map +1 -1
  83. package/package.json +10 -4
  84. package/src/accelerator.ts +5 -3
  85. package/src/algorithms/all-pairs.ts +86 -56
  86. package/src/algorithms/betweenness.ts +2 -2
  87. package/src/algorithms/closeness.ts +353 -256
  88. package/src/browser/index.ts +37 -0
  89. package/src/constants.ts +13 -3
  90. package/src/kernels.ts +65 -36
  91. package/src/layouts/force-simulation.ts +0 -1
  92. package/src/layouts/forceatlas2.ts +0 -1
  93. package/src/layouts/fruchterman-reingold.ts +0 -1
  94. package/src/layouts/repulsion-grid.ts +2 -7
  95. package/src/layouts/spring-electrical.ts +0 -1
  96. package/src/managed.ts +172 -0
  97. package/src/node/index.ts +36 -0
  98. package/src/primitives/grid-pyramid.ts +29 -41
  99. package/src/types/accelerator.ts +2 -0
  100. package/src/types/managed.ts +86 -0
  101. package/src/wgsl/bc-forward.wgsl.ts +1 -1
  102. package/src/wgsl/closeness-level.wgsl.ts +203 -0
  103. package/src/wgsl/closeness-rowsum.wgsl.ts +41 -0
  104. package/src/wgsl/grid-centroid-hub.wgsl.ts +2 -2
  105. package/dist/chunks/context-BZY6SMsM.js +0 -3615
  106. package/dist/chunks/context-BZY6SMsM.js.map +0 -1
  107. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +0 -20
  108. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +0 -1
  109. package/dist/src/wgsl/closeness-reduce.wgsl.js +0 -69
  110. package/dist/src/wgsl/closeness-reduce.wgsl.js.map +0 -1
  111. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +0 -22
  112. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +0 -1
  113. package/dist/src/wgsl/closeness-sweep.wgsl.js +0 -106
  114. package/dist/src/wgsl/closeness-sweep.wgsl.js.map +0 -1
  115. package/src/wgsl/closeness-reduce.wgsl.ts +0 -68
  116. package/src/wgsl/closeness-sweep.wgsl.ts +0 -105
@@ -1,103 +1,111 @@
1
1
  /**
2
- * Closeness centrality on the device (design 8.4, 3.3 line 810, 9.7; P8-T11, the P8 plan's PD-13 / PD-19 / PD-25 /
3
- * DEP-P8-E / DEP-P8-F): `score[s] = 1 / sumDist_s`, with `sumDist_s` the exact sum of the finite distances from `s`
4
- * to every OTHER node -- an unreached node adds nothing -- and `0` when nothing is reached. No reached factor and no
5
- * Wasserman-Faust scaling: this is EXACTLY the legacy default (`normalized: false`) of `closenessCentrality` in
6
- * `@graphty/algorithms`, the number graphty-element's closeness panel shows today, so a future `indexed` port has one
7
- * number to match (the NetworkX form is 33x it on karate and could never have been substituted silently).
2
+ * Closeness centrality on the device (design 8.4, 3.3 line 810, 9.7): `score[s] = 1 / sumDist_s`, with `sumDist_s`
3
+ * the exact sum of the finite distances from `s` to every OTHER node -- an unreached node adds nothing -- and `0` when
4
+ * nothing is reached; or, with `harmonic`, `score[s] = sum of 1 / dist` over the same nodes (a zero distance adds
5
+ * nothing). No reached factor and no Wasserman-Faust scaling: this is EXACTLY the legacy default (`normalized:
6
+ * false`) of `closenessCentrality` in `@graphty/algorithms`, the number graphty-element's closeness panel shows.
8
7
  *
9
- * The unweighted route is ONE bit-parallel multi-source search per batch of 32 sources (`ceil(n / 32)` batches):
10
- * the batch's state is one `bits` buffer of four regions of `bitsBase = roundUp(n, 64)` words (`visited`, two
11
- * frontier regions that swap by the level's parity, `flags`), bit `s` of word `v` meaning "source `s` has reached /
12
- * is at / is next at `v`"; a level is, all host-recorded, `closeness-reduce` role 0 (the boundary: `done` from the
13
- * previous level's compacted count, the level's claims folded into the exact 64-bit per-source sums at `level + 1`),
14
- * `compact` of the flags into the frontier list, two `fill`s zeroing the level's next region and the flags (AFTER
15
- * the compaction that consumed them), and `closeness-sweep` (the block-mapped expansion with the claim inline, a
16
- * direct grid-stride dispatch looping to the list's count). `MAX_LEVELS_PER_SUBMIT` levels per submit and
17
- * one readback per submit (the `done` word and the 512-byte `perSource` block together, so the finished batch needs
18
- * no extra map); the host folds `sumHi x 2^32 + sumLo` into `1 / sum` in f64 and stores f32. The weighted route
19
- * (`weighted` true on a snapshot whose column is not all ones) is one `sssp` per source with the sums reduced on the
20
- * host between calls: design 8.4's own answer, slow and correct. `weighted` defaults to the snapshot's `flags.weighted`;
21
- * `weighted: false` on a weighted snapshot ignores the column by request and sweeps; `weighted: true` over unit
22
- * weights or no column sweeps too (every `sssp` would route to a BFS anyway). `maxIterations` and `tolerance` are the
8
+ * Three routes, chosen from the inputs before any device work:
9
+ *
10
+ * - ALL-PAIRS (an exact run whose `n x n` f32 matrix fits one binding, weighted, or unweighted up to
11
+ * `ALL_PAIRS_MAX_NODES`): `allPairsShortestPath`'s blocked Floyd-Warshall sweep, then `closeness-rowsum` folds each row on
12
+ * the device -- hop counts as exact integers, weighted distances and harmonic reciprocals in f32 -- and only `n`
13
+ * words come back. Weighted scores agree with the one-search-per-source route up to f32 rounding of the sums.
14
+ * - LEVELS (unweighted, every other case): a bit-parallel multi-source breadth-first search, `32 x words` sources per
15
+ * batch (up to `MAX_WORDS` = 8 words, 256 sources), ONE `closeness-level` dispatch per level. Each level chooses on
16
+ * the device between pushing the frontier over the out-arcs (one invocation per arc) and pulling it over the
17
+ * in-arcs (one invocation per node, with an early exit once every source has reached the node), from the arcs the
18
+ * frontier would push against `pullAt`; a graph with a node of more than `PULL_MAX_DEGREE` in-arcs only pushes. The levels of a
19
+ * submit tally their claims per source into one count table, read back once per submit; the host folds the counts
20
+ * into exact sums in f64 (`count x distance`) and harmonic sums (`count / distance`). A level whose predecessor
21
+ * claimed nothing returns at once, so recording `MAX_LEVELS_PER_SUBMIT` levels costs little past a batch's depth.
22
+ * - ONE SEARCH PER SOURCE (weighted above the all-pairs ceiling, or a weighted sampled run): one `sssp` per source,
23
+ * the sums reduced on the host between calls.
24
+ *
25
+ * `weighted` defaults to the snapshot's `flags.weighted`; `weighted: false` on a weighted snapshot ignores the column;
26
+ * `weighted: true` over unit weights or no column is the unweighted problem. `maxIterations` and `tolerance` are the
23
27
  * seam's placeholder keys and an exact traversal has neither, so a defined value is REFUSED before any device work
24
- * (`E_UNSUPPORTED { option }`, the package's rule for an option it does not implement, PD-25); `undefined` is legal.
25
- * `iterations` reports the source batches run (the sources, on the weighted route), `converged` is always true.
26
- * A SAMPLED run (`sources`, issue #426; undirected snapshots only) seeds its batches from the listed sources (the
27
- * reduce's role 2 reads the list the host wrote after the per-node sums in `perSource`), and the sweep also adds each
28
- * claim's distance into a per-node sum (`perNode`), read back with every submit and folded on the host in f64 into
29
- * `1 / sum` per NODE, where the exact run folds per SOURCE: on an undirected graph the distance from a source to a node
30
- * is the distance from the node to the source, which is what the CPU port's sampled closeness sums.
28
+ * (`E_UNSUPPORTED { option }`); `undefined` is legal. `iterations` reports the source batches run (the sources, on the
29
+ * one-search-per-source route; 1 on the all-pairs route), `converged` is always true.
31
30
  *
32
- * Cost, stated so nobody is surprised: closeness is O(n x m) on any device -- at 1M nodes it is 31,250 batches of a
33
- * full multi-source traversal, minutes on the card, and no target in design 10.4 asks for less. `compact.record`
34
- * leases its offsets and the scan's block sums afresh on every call, once per LEVEL here, so the planner is given a
35
- * scope whose `scratch` hands the same buffer back for the same label and size: the dispatches of one pass run in
36
- * order, a level's scan overwrites the previous level's, and the run holds one set instead of `levels x batches`.
37
- * The sweep keeps DEP-P8-E's refusal of a windowed core (`assertWholeCore`): the bit-parallel claim needs the whole
38
- * arc array bound. The tuning entry `closenessWithTuning` (PD-26's shape) is what the tests drive; nothing public
39
- * exposes it.
31
+ * A SAMPLED run (`sources`, undirected snapshots only) seeds its batches from the listed sources and adds every
32
+ * claim's distance into a per-node sum, folded on the host into `1 / sum` per NODE: on an undirected graph the
33
+ * distance from a source to a node is the distance from the node to the source, which is what the CPU port's sampled
34
+ * closeness sums. A sampled harmonic run is refused (`E_UNSUPPORTED closenessCentrality.sampledHarmonic`): the
35
+ * per-node reciprocal sums would need a float atomic.
36
+ *
37
+ * Cost, stated so nobody is surprised: closeness is O(n x m) on any device -- at 1M nodes it is 3,907 batches of a
38
+ * full multi-source traversal. The level kernel binds the whole arc array (`assertWholeCore`) and, on a directed
39
+ * snapshot, the whole reverse adjacency. `closenessWithTuning` is what the tests drive; nothing public exposes it.
40
40
  */
41
41
  import { MAX_LEVELS_PER_SUBMIT } from "../constants.js";
42
42
  import { WebGpuGraphError } from "../errors.js";
43
43
  import { CommandBatch } from "../kernel/batch.js";
44
- import { plan1d, planGridStride } from "../kernel/dispatch.js";
45
- import { FILL_PARAMS, FRONTIER_COUNTERS, FRONTIER_PARAMS, graphBindings, graphOverrides, kernelSpec, } from "../kernels.js";
46
- import { prepareCompact } from "../primitives/compact.js";
47
- import { assertWholeCore } from "../primitives/core-shape.js";
48
- import { W } from "../primitives/frontier.js";
44
+ import { plan1d, plan2d } from "../kernel/dispatch.js";
45
+ import { CLOSENESS_PARAMS, FILL_PARAMS, kernelSpec } from "../kernels.js";
46
+ import { assertWholeCore, coreOfView } from "../primitives/core-shape.js";
49
47
  import { assertDeviceComputes } from "../primitives/verify.js";
48
+ import { allPairsCeiling, DEFAULT_ROUNDS_PER_SUBMIT, sweepAllPairs } from "./all-pairs.js";
50
49
  import { algorithmScope } from "./scope.js";
51
50
  import { aborted, bindingOf, checkDest, sssp } from "./sssp.js";
52
51
  const ALGORITHM = "closenessCentrality";
53
52
  /**
54
- * Design 8.4: 32 sources per `u32` word, one batch per word.
53
+ * The most 32-bit words per node of the level route: `32 x MAX_WORDS` = 256 sources per batch, the size of the
54
+ * kernel's workgroup tally.
55
55
  * @internal
56
56
  */
57
- export const SOURCES_PER_BATCH = 32;
57
+ export const MAX_WORDS = 8;
58
+ /**
59
+ * A level pulls when the arcs its frontier would push exceed `words x arcCount / PULL_RATIO`: the push touches every
60
+ * arc whatever the frontier, so it is the cheap step only while few of them claim anything, and the pull's early exit
61
+ * pays once the frontier is large. Measured on the RTX 4070 SUPER (level route, eight words per node) against always
62
+ * pushing and always pulling: uniform random graphs of 4,096 nodes at 64 to 512 arcs per node 9.9 / 13.5 / 19.9 ms
63
+ * against 16.5 / 25.7 / 68.4 pushing and 12.4 / 18.9 / 36.7 pulling; 16,384 nodes at 32 arcs per node 52.7 against
64
+ * 142.7 and 54.1; at 8 arcs per node and below, rings, a path, a grid and a star within noise of the better of the
65
+ * two. 4 and 16 measured within a few percent of 64.
66
+ */
67
+ const PULL_RATIO = 64;
58
68
  /**
59
- * The `perSource` block of a batch: `newCount[32]` @0, `reached[32]` @32, `sumLo[32]` @64, `sumHi[32]` @96.
69
+ * The most in-arcs a node may have for the level route to pull: a pull invocation walks one node's in-arcs, so its
70
+ * loop count is about `in-degree x (words + 1)`, and llvmpipe silently ends every loop of an invocation past 65,535
71
+ * iterations (`(65,535 - 512) / 9` is 7,225 at eight words). A graph with a larger hub runs every level as a push,
72
+ * one invocation per arc.
60
73
  * @internal
61
74
  */
62
- export const PER_SOURCE_WORDS = 4 * SOURCES_PER_BATCH;
75
+ export const PULL_MAX_DEGREE = 4096;
76
+ /** The control ring at the end of the count table: three slots of four words (`any`, `arcs`, `pull`, pad). */
77
+ const CTRL_WORDS = 12;
63
78
  /**
64
- * Params slots of the ring, COUNTED (`UniformRing.reserve` wraps silently): per level `compact`'s records (its scan
65
- * is at most four levels for any n below 2^32, so at most 8 records) while the boundary, the finalize, the fill and
66
- * the two sweep records (one per parity) are written once per submit, plus the seed's two records on a batch's first
67
- * submit (the iota fill flushes in its own submit).
79
+ * The most nodes an unweighted exact run sends to the all-pairs route: the blocked sweep's `n^3` work beats the level
80
+ * route's `n / 256` batches of a full traversal only on small graphs. Measured on the RTX 4070 SUPER: a ring with
81
+ * chords takes 2.7 ms on the all-pairs route against 3.5 to 4 ms on the level route at 1,024 nodes, 11 against 4.4
82
+ * at 2,048 and 71 against 10 at 4,096; a graph with `n^2 / 12` arcs 0.9 against 1.4 at 512, 2.5 against 2.6 at
83
+ * 1,024, 11 against 6 to 9 at 2,048 and 73 against 24 at 4,096. A deep sparse graph is the exception this does not
84
+ * see: a 4,096-node path takes 70 ms on the all-pairs route and 0.7 s on the level route, one dispatch per level for
85
+ * 4,095 levels.
86
+ * @internal
68
87
  */
69
- const RING_SLOTS = 8 * MAX_LEVELS_PER_SUBMIT + 16;
88
+ export const ALL_PAIRS_MAX_NODES = 1024;
70
89
  /**
71
- * A scope whose `scratch` hands the SAME buffer back for the same label and size (see the file comment).
72
- * @param scope - the algorithm's scope
73
- * @returns the reusing scope
90
+ * The score of a sum: `1 / sum`, `0` when nothing was reached.
91
+ * @param sum - the sum of distances
92
+ * @returns the score
74
93
  */
75
- function reusingScratch(scope) {
76
- const held = new Map();
77
- return {
78
- ...scope,
79
- scratch: (byteLength, label) => {
80
- const key = `${label}/${byteLength}`;
81
- let buffer = held.get(key);
82
- if (buffer === undefined) {
83
- buffer = scope.scratch(byteLength, label);
84
- held.set(key, buffer);
85
- }
86
- return buffer;
87
- },
88
- };
94
+ function inverse(sum) {
95
+ return sum === 0 ? 0 : 1 / sum;
89
96
  }
90
97
  /**
91
- * The weighted route: one `sssp` per source, the sums reduced on the host. With `sources` (a sampled run on an
92
- * undirected snapshot) each search adds its distances into the sums of the nodes it reaches instead of its own.
98
+ * The one-search-per-source route: one `sssp` per source, the sums reduced on the host. With `sources` (a sampled run
99
+ * on an undirected snapshot) each search adds its distances into the sums of the nodes it reaches instead of its own.
93
100
  * @param ctx - the context
94
101
  * @param s - the snapshot
95
102
  * @param scores - the destination
96
103
  * @param sources - a sampled run's sources, or null for every node
104
+ * @param harmonic - sum reciprocal distances (exact runs only)
97
105
  * @param options - the run options
98
106
  * @returns the result
99
107
  */
100
- async function weightedRoute(ctx, s, scores, sources, options) {
108
+ async function perSourceRoute(ctx, s, scores, sources, harmonic, options) {
101
109
  const n = s.nodeCount;
102
110
  const count = sources?.length ?? n;
103
111
  const totals = sources === null ? null : new Float64Array(n);
@@ -111,194 +119,230 @@ async function weightedRoute(ctx, s, scores, sources, options) {
111
119
  for (let v = 0; v < n; v++) {
112
120
  const d = dist[v];
113
121
  if (v !== source && d !== Infinity) {
114
- if (totals === null) {
122
+ if (totals !== null) {
123
+ totals[v] += d;
124
+ }
125
+ else if (!harmonic) {
115
126
  sum += d;
116
127
  }
117
- else {
118
- totals[v] += d;
128
+ else if (d > 0) {
129
+ sum += 1 / d;
119
130
  }
120
131
  }
121
132
  }
122
133
  if (totals === null) {
123
- scores[source] = sum === 0 ? 0 : 1 / sum;
134
+ scores[source] = harmonic ? sum : inverse(sum);
124
135
  }
125
136
  options?.onProgress?.(i + 1, count);
126
137
  }
127
- if (totals !== null) {
128
- totals.forEach((sum, v) => {
129
- scores[v] = sum === 0 ? 0 : 1 / sum;
130
- });
131
- }
138
+ totals?.forEach((sum, v) => {
139
+ scores[v] = inverse(sum);
140
+ });
132
141
  return { scores, iterations: count, converged: true, precision: "f32", sourcesUsed: count };
133
142
  }
134
143
  /**
135
- * The bit-parallel route (see the file comment).
144
+ * The all-pairs route (see the file comment). The caller checked that the matrix fits.
145
+ * @param ctx - the context
146
+ * @param s - the snapshot
147
+ * @param scores - the destination
148
+ * @param weighted - sum the weights, or count hops
149
+ * @param harmonic - sum reciprocal distances
150
+ * @param options - the run options
151
+ * @returns the result
152
+ */
153
+ async function allPairsRoute(ctx, s, scores, weighted, harmonic, options) {
154
+ const n = s.nodeCount;
155
+ const scope = algorithmScope(ctx, ALGORITHM, Math.min(DEFAULT_ROUNDS_PER_SUBMIT, Math.ceil(n / 32)) + 3);
156
+ try {
157
+ const matrix = await sweepAllPairs(ctx, s, scope, weighted, DEFAULT_ROUNDS_PER_SUBMIT, { signal: options?.signal }, ALGORITHM);
158
+ const rowsum = await ctx.pipelines.kernel(kernelSpec("closeness-rowsum"));
159
+ const out = bindingOf(scope.scratch(4 * n, "row-sums"), 4 * n);
160
+ const role = harmonic ? 2 : Number(weighted);
161
+ const params = scope.params(CLOSENESS_PARAMS, { role, n });
162
+ const batch = new CommandBatch(ctx, `${ALGORITHM}/row-sums`);
163
+ rowsum.dispatch(batch.pass("row-sums"), rowsum.bind({ dist: matrix, out, P: params.binding }), plan2d(n, ctx.caps), [params.offset]);
164
+ batch.endPass();
165
+ const request = batch.readback(out.buffer, out.offset, 4 * n);
166
+ scope.flush();
167
+ const back = await batch.submit().readback;
168
+ ctx.assertReady();
169
+ if (role === 0) {
170
+ new Uint32Array(back, request.offset, n).forEach((sum, v) => {
171
+ scores[v] = inverse(sum);
172
+ });
173
+ }
174
+ else {
175
+ new Float32Array(back, request.offset, n).forEach((sum, v) => {
176
+ scores[v] = harmonic ? sum : inverse(sum);
177
+ });
178
+ }
179
+ options?.onProgress?.(n, n);
180
+ return { scores, iterations: 1, converged: true, precision: "f32", sourcesUsed: n };
181
+ }
182
+ finally {
183
+ scope.dispose();
184
+ }
185
+ }
186
+ /**
187
+ * The level route (see the file comment).
136
188
  * @param ctx - the context
137
189
  * @param s - the snapshot
138
190
  * @param scores - the destination
139
191
  * @param sources - a sampled run's sources, or null for every node
192
+ * @param harmonic - sum reciprocal distances (exact runs only)
140
193
  * @param levelsPerSubmit - the submit cadence
141
194
  * @param options - the run options
142
195
  * @param tuning - the knobs
143
196
  * @returns the result
144
197
  */
145
- async function sweepRoute(ctx, s, scores, sources, levelsPerSubmit, options, tuning) {
198
+ async function levelRoute(ctx, s, scores, sources, harmonic, levelsPerSubmit, options, tuning) {
146
199
  const n = s.nodeCount;
147
200
  const seedCount = sources?.length ?? n;
148
201
  if (seedCount === 0) {
149
202
  return { scores, iterations: 0, converged: true, precision: "f32", sourcesUsed: 0 };
150
203
  }
204
+ const limit = Math.min(ctx.caps.limits.maxStorageBufferBindingSize, ctx.caps.limits.maxBufferSize);
151
205
  const core = ctx.residency.core(s);
152
206
  assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM);
153
- const scope = algorithmScope(ctx, ALGORITHM, RING_SLOTS);
207
+ if (s.directed && 4 * s.arcCount > limit) {
208
+ throw new WebGpuGraphError("E_TOO_LARGE", `${ALGORITHM}: the reverse adjacency of a directed snapshot (${4 * s.arcCount} bytes) does not fit one binding`, { needed: 4 * s.arcCount, limit, path: "closeness.reverse", algorithm: ALGORITHM });
209
+ }
210
+ const reverse = s.directed ? coreOfView(ctx.residency.view(s, "reverse"), s.arcCount) : core;
211
+ // words per node: as many as the sources need, within one binding of four regions (plus a sampled run's per-node
212
+ // sums) and, for a sampled run, so a node's per-batch distance sum (at most 32 words (n - 1)) fits a u32
213
+ const extra = sources === null ? 0 : n;
214
+ const fitWords = Math.floor((limit / 4 - extra) / (4 * n));
215
+ const sumWords = sources === null ? MAX_WORDS : Math.floor(0xffffffff / (32 * Math.max(1, n - 1)));
216
+ const words = tuning.words ?? Math.max(1, Math.min(MAX_WORDS, Math.ceil(seedCount / 32), fitWords, sumWords));
217
+ if (!Number.isInteger(words) || words < 1 || words > MAX_WORDS) {
218
+ throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM}: words must be an integer in [1, ${MAX_WORDS}]`, {
219
+ argument: "words",
220
+ value: words,
221
+ expected: `an integer in [1, ${MAX_WORDS}]`,
222
+ });
223
+ }
224
+ const base = n * words;
225
+ const bitsWords = 4 * base + extra;
226
+ if (4 * bitsWords > limit) {
227
+ throw new WebGpuGraphError("E_TOO_LARGE", `${ALGORITHM}: ${n} nodes need a ${4 * bitsWords}-byte search state in one binding of at most ${limit} bytes`, { needed: 4 * bitsWords, limit, path: "closeness.bits", algorithm: ALGORITHM });
228
+ }
229
+ const lanes = 32 * words;
230
+ const rowWords = levelsPerSubmit * lanes;
231
+ const ctrl = rowWords;
232
+ const sourcesAt = sources === null ? 0 : ctrl + CTRL_WORDS;
233
+ const tableWords = ctrl + CTRL_WORDS + (sources?.length ?? 0);
234
+ const pullAt = tuning.pullAt ?? Math.min(0xffffffff, Math.floor((words * s.arcCount) / PULL_RATIO));
235
+ // slots: the row fill and the levels of one submit, plus the two seed records of a batch's first submit
236
+ const scope = algorithmScope(ctx, ALGORITHM, levelsPerSubmit + 4);
154
237
  try {
155
238
  const wg = ctx.workgroupSize;
156
- const bytes = 4 * n;
157
- // the four regions of the bits buffer, each bitsBase words so its byte offset is 256-aligned and fill and
158
- // compact can bind one alone: visited 0, the frontier pair 1 and 2, flags 3
159
- const bitsBase = Math.ceil(n / 64) * 64;
160
- const regionBytes = 4 * bitsBase;
161
- const bits = bindingOf(scope.scratch(4 * regionBytes, "bits"), 4 * regionBytes);
162
- const region = (index) => ({
163
- buffer: bits.buffer,
164
- offset: index * regionBytes,
165
- size: regionBytes,
166
- window: null,
167
- });
168
- const flags = region(3);
169
- const frontierList = bindingOf(scope.scratch(bytes, "frontier-list"), bytes);
170
- const iota = bindingOf(scope.scratch(bytes, "iota"), bytes);
171
- const counters = bindingOf(scope.scratch(FRONTIER_COUNTERS.byteLength, "counters"), FRONTIER_COUNTERS.byteLength);
172
- const perSourceBytes = 4 * PER_SOURCE_WORDS;
173
- // a sampled run appends the per-node distance sums (bitsBase words) and then its source list
174
- const zeroedWords = PER_SOURCE_WORDS + (sources === null ? 0 : bitsBase);
175
- const perSourceAll = 4 * (zeroedWords + (sources === null ? 0 : sources.length));
176
- const perSource = bindingOf(scope.scratch(perSourceAll, "per-source"), perSourceAll);
239
+ const bits = bindingOf(scope.scratch(4 * bitsWords, "bits"), 4 * bitsWords);
240
+ const table = bindingOf(scope.scratch(4 * tableWords, "table"), 4 * tableWords);
241
+ const rows = { ...table, size: 4 * rowWords };
177
242
  if (sources !== null) {
178
- ctx.device.queue.writeBuffer(perSource.buffer, perSource.offset + 4 * zeroedWords, Uint32Array.from(sources));
243
+ ctx.device.queue.writeBuffer(table.buffer, table.offset + 4 * sourcesAt, Uint32Array.from(sources));
179
244
  }
180
- const totals = sources === null ? null : new Float64Array(n);
181
245
  await ctx.allocator.check();
182
- const compact = await prepareCompact(reusingScratch(scope));
183
- const sweep = await ctx.pipelines.kernel(kernelSpec("closeness-sweep", graphOverrides(core, null)));
184
- const reduce = await ctx.pipelines.kernel(kernelSpec("closeness-reduce"));
246
+ const level = await ctx.pipelines.kernel(kernelSpec("closeness-level"));
185
247
  const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
186
- const graph = graphBindings(core, null);
187
- const onePlan = plan1d(1, wg, ctx.caps);
188
- const regionPlan = plan1d(bitsBase, wg, ctx.caps);
189
- const sweepPlan = planGridStride(n, wg, ctx.caps); // the list holds at most n entries; the sweep loops to the count word
190
- const recordFill = (pass, dst, count, mode) => {
191
- const params = scope.params(FILL_PARAMS, { count, value: 0, mode, pad0: 0 });
192
- fill.dispatch(pass, fill.bind({ dst, P: params.binding }), plan1d(count, wg, ctx.caps), [params.offset]);
248
+ const state = {
249
+ rowPtr: core.rowPtr,
250
+ colIdx: core.colIdx ?? core.rowPtr,
251
+ inRowPtr: reverse.rowPtr,
252
+ inColIdx: reverse.colIdx ?? reverse.rowPtr,
253
+ bits,
254
+ table,
193
255
  };
194
- const submit = (batch) => {
195
- scope.flush();
196
- return batch.submit();
256
+ const levelPlan = plan1d(Math.max(n, s.arcCount), wg, ctx.caps);
257
+ const pullOk = s.inDegree().every((d) => d <= PULL_MAX_DEGREE) ? 1 : 0;
258
+ const totals = sources === null ? null : new Float64Array(n);
259
+ const shared = {
260
+ n,
261
+ words,
262
+ base,
263
+ ctrl,
264
+ pullAt,
265
+ perNode: sources === null ? 0 : 1,
266
+ sourcesAt,
267
+ arcCount: s.arcCount,
268
+ pullOk,
197
269
  };
198
- // setup: the iota queue compact reads, once per run
199
- const setup = new CommandBatch(ctx, `${ALGORITHM}/setup`);
200
- recordFill(setup.pass("fill"), iota, n, 1);
201
- setup.endPass();
202
- await submit(setup).readback;
203
- ctx.assertReady();
204
270
  let batches = 0;
205
- for (let batchStart = 0; batchStart < seedCount; batchStart += SOURCES_PER_BATCH) {
206
- let level = 0;
207
- for (let first = true;; first = false) {
271
+ for (let batchStart = 0; batchStart < seedCount; batchStart += lanes) {
272
+ const count = Math.min(lanes, seedCount - batchStart);
273
+ const sums = new Float64Array(count);
274
+ const reciprocal = new Float64Array(count);
275
+ const reached = new Uint32Array(count);
276
+ for (let firstLevel = 0, done = false; !done; firstLevel += levelsPerSubmit) {
277
+ if (firstLevel > n + 1) {
278
+ // a batch claims at most n - 1 levels deep, then one level claims nothing
279
+ throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM}: the batch at ${batchStart} still claimed at level ${firstLevel}`, { label: ALGORITHM, message: `the batch still claimed at level ${firstLevel}` });
280
+ }
208
281
  const batch = new CommandBatch(ctx, `${ALGORITHM}/levels`);
209
282
  const pass = batch.pass("closeness");
210
- if (first) {
211
- // the batch's seed: the four regions and the block zeroed, then role 1 (the sources' bits, their
212
- // flags, counters[0] = k, level = U32_MAX)
213
- recordFill(pass, bits, 4 * bitsBase, 0);
214
- recordFill(pass, perSource, zeroedWords, 0);
215
- const seed = scope.params(FRONTIER_PARAMS, {
216
- role: sources === null ? 1 : 2,
217
- n: seedCount,
218
- bitsBase,
219
- source: batchStart,
220
- });
221
- reduce.dispatch(pass, reduce.bind({ counters, perSource, bits, P: seed.binding }), onePlan, [
222
- seed.offset,
223
- ]);
283
+ if (firstLevel === 0) {
284
+ const clear = scope.params(CLOSENESS_PARAMS, { ...shared, role: 1, total: bitsWords, count });
285
+ const bound = level.bind({ ...state, P: clear.binding });
286
+ level.dispatch(pass, bound, plan1d(bitsWords, wg, ctx.caps), [clear.offset]);
287
+ const seed = scope.params(CLOSENESS_PARAMS, { ...shared, role: 2, count, source: batchStart });
288
+ level.dispatch(pass, bound, plan1d(count, wg, ctx.caps), [seed.offset]);
224
289
  }
225
- // the records every level of the submit shares (the ring wraps, so they are written per submit)
226
- const boundary = scope.params(FRONTIER_PARAMS, { role: 0, n, bitsBase });
227
- const boundBoundary = reduce.bind({ counters, perSource, bits, P: boundary.binding });
228
- const clear = scope.params(FILL_PARAMS, { count: bitsBase, value: 0, mode: 0, pad0: 0 });
229
- // parity 0 sweeps region 1 into region 2, parity 1 region 2 into region 1: the next region is cleared
230
- const boundClearNext = [region(2), region(1)].map((dst) => fill.bind({ dst, P: clear.binding }));
231
- const boundClearFlags = fill.bind({ dst: flags, P: clear.binding });
232
- const boundSweep = [0, 1].map((mode) => {
233
- const params = scope.params(FRONTIER_PARAMS, {
234
- wg,
235
- n,
236
- bitsBase,
237
- arcBase: 0,
238
- arcEnd: s.arcCount,
239
- mode,
240
- stride: sweepPlan.stride ?? wg,
241
- perNode: sources === null ? 0 : 1,
242
- });
243
- return {
244
- bound: sweep.bind({ ...graph, frontierList, counters, bits, perSource, P: params.binding }),
245
- offset: params.offset,
246
- };
247
- });
248
- for (let k = 0; k < levelsPerSubmit; k++, level++) {
249
- const parity = level % 2;
250
- reduce.dispatch(pass, boundBoundary, onePlan, [boundary.offset]);
251
- compact.record(pass, {
252
- queue: iota,
253
- flags,
254
- count: n,
255
- out: frontierList,
256
- outCount: counters,
257
- outIndex: W.frontierCount,
290
+ const zero = scope.params(FILL_PARAMS, { count: rowWords, value: 0, mode: 0, pad0: 0 });
291
+ fill.dispatch(pass, fill.bind({ dst: rows, P: zero.binding }), plan1d(rowWords, wg, ctx.caps), [
292
+ zero.offset,
293
+ ]);
294
+ let bound = null;
295
+ for (let row = 0; row < levelsPerSubmit; row++) {
296
+ const params = scope.params(CLOSENESS_PARAMS, {
297
+ ...shared,
298
+ role: 0,
299
+ count,
300
+ level: firstLevel + row,
301
+ row,
258
302
  });
259
- fill.dispatch(pass, boundClearNext[parity], regionPlan, [clear.offset]);
260
- fill.dispatch(pass, boundClearFlags, regionPlan, [clear.offset]);
261
- sweep.dispatch(pass, boundSweep[parity].bound, sweepPlan, [boundSweep[parity].offset]);
303
+ bound ?? (bound = level.bind({ ...state, P: params.binding }));
304
+ level.dispatch(pass, bound, levelPlan, [params.offset]);
262
305
  }
263
306
  batch.endPass();
264
- const doneRequest = batch.readback(counters.buffer, counters.offset + 4 * W.done, 4);
265
- const blockRequest = batch.readback(perSource.buffer, perSource.offset, perSourceBytes);
266
- const nodeRequest = totals === null ? null : batch.readback(perSource.buffer, perSource.offset + perSourceBytes, 4 * n);
267
- const submitted = submit(batch);
307
+ const rowRequest = batch.readback(table.buffer, table.offset, 4 * rowWords);
308
+ const nodeRequest = totals === null ? null : batch.readback(bits.buffer, bits.offset + 16 * base, 4 * n);
309
+ scope.flush();
310
+ const submitted = batch.submit();
268
311
  const back = await submitted.readback;
269
312
  ctx.assertReady();
270
313
  if (options?.signal?.aborted) {
271
314
  throw aborted(ALGORITHM, submitted.id);
272
315
  }
273
- if (new Uint32Array(back, doneRequest.offset, 1)[0] !== 0) {
274
- const block = new Uint32Array(back, blockRequest.offset, PER_SOURCE_WORDS);
275
- if (totals === null || nodeRequest === null) {
276
- const count = Math.min(SOURCES_PER_BATCH, n - batchStart);
277
- for (let i = 0; i < count; i++) {
278
- const sum = block[3 * SOURCES_PER_BATCH + i] * 2 ** 32 + block[2 * SOURCES_PER_BATCH + i];
279
- scores[batchStart + i] = sum === 0 ? 0 : 1 / sum;
280
- }
316
+ const counts = new Uint32Array(back, rowRequest.offset, rowWords);
317
+ for (let row = 0; row < levelsPerSubmit && !done; row++) {
318
+ const distance = firstLevel + row + 1;
319
+ let claimed = 0;
320
+ for (let lane = 0; lane < count; lane++) {
321
+ const c = counts[row * lanes + lane];
322
+ sums[lane] += c * distance;
323
+ reciprocal[lane] += c / distance;
324
+ reached[lane] += c;
325
+ claimed += c;
281
326
  }
282
- else {
283
- // at most 32 (n - 1) per node per batch, so a u32 word never wraps below 134M nodes
284
- const sums = new Uint32Array(back, nodeRequest.offset, n);
285
- for (let v = 0; v < n; v++) {
286
- totals[v] += sums[v];
287
- }
288
- }
289
- tuning.onBatch?.(batchStart, block.slice());
290
- break;
327
+ done = claimed === 0;
328
+ }
329
+ if (done && totals !== null && nodeRequest !== null) {
330
+ new Uint32Array(back, nodeRequest.offset, n).forEach((sum, v) => {
331
+ totals[v] += sum;
332
+ });
291
333
  }
292
- if (level > n + 3) {
293
- // a batch claims at most n - 1 levels deep, then one level claims nothing and one is empty
294
- throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM}: the done flag never rose in ${level} levels of the batch at ${batchStart}`, { label: ALGORITHM, message: `the done flag never rose in ${level} levels` });
334
+ }
335
+ if (sources === null) {
336
+ for (let lane = 0; lane < count; lane++) {
337
+ scores[batchStart + lane] = harmonic ? reciprocal[lane] : inverse(sums[lane]);
295
338
  }
339
+ tuning.onBatch?.(batchStart, sums, reached);
296
340
  }
297
341
  batches += 1;
298
- options?.onProgress?.(Math.min(batchStart + SOURCES_PER_BATCH, seedCount), seedCount);
342
+ options?.onProgress?.(batchStart + count, seedCount);
299
343
  }
300
344
  totals?.forEach((sum, v) => {
301
- scores[v] = sum === 0 ? 0 : 1 / sum;
345
+ scores[v] = inverse(sum);
302
346
  });
303
347
  return { scores, iterations: batches, converged: true, precision: "f32", sourcesUsed: seedCount };
304
348
  }
@@ -337,11 +381,11 @@ function checkSources(s, sources) {
337
381
  return sources;
338
382
  }
339
383
  /**
340
- * Closeness with the test knobs of PD-26's shape; `closenessCentrality` is this with an empty tuning.
384
+ * Closeness with the test knobs; `closenessCentrality` is this with an empty tuning.
341
385
  * @internal
342
386
  * @param ctx - the context whose device runs the kernels
343
387
  * @param s - the snapshot (uploaded through ctx.residency, or found there)
344
- * @param options - `weighted` and a sampled run's `sources` honoured, the placeholder `maxIterations` / `tolerance` refused when defined, plus dest / signal / onProgress
388
+ * @param options - `weighted`, `harmonic` and a sampled run's `sources` honoured, the placeholder `maxIterations` / `tolerance` refused when defined, plus dest / signal / onProgress
345
389
  * @param tuning - the knobs
346
390
  * @returns the scores, the batches run, `converged: true`, `precision: "f32"` and `sourcesUsed`
347
391
  */
@@ -358,6 +402,13 @@ export async function closenessWithTuning(ctx, s, options, tuning) {
358
402
  }
359
403
  const n = s.nodeCount;
360
404
  const sources = checkSources(s, options?.sources);
405
+ const harmonic = options?.harmonic === true;
406
+ if (harmonic && sources !== null) {
407
+ throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: harmonic closeness from sampled sources`, {
408
+ feature: "closenessCentrality.sampledHarmonic",
409
+ hint: "run the CPU port, or drop the sources for the exact harmonic score",
410
+ });
411
+ }
361
412
  const levelsPerSubmit = tuning.levelsPerSubmit ?? MAX_LEVELS_PER_SUBMIT;
362
413
  if (!Number.isInteger(levelsPerSubmit) || levelsPerSubmit < 1 || levelsPerSubmit > MAX_LEVELS_PER_SUBMIT) {
363
414
  throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM}: levelsPerSubmit must be an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`, {
@@ -367,40 +418,58 @@ export async function closenessWithTuning(ctx, s, options, tuning) {
367
418
  });
368
419
  }
369
420
  const scores = checkDest(ALGORITHM, options?.dest, n) ?? new Float32Array(n);
370
- const weighted = options?.weighted ?? s.flags.weighted;
421
+ const weighted = (options?.weighted ?? s.flags.weighted) && s.weights !== null && !s.flags.allWeightsOne;
371
422
  if (options?.signal?.aborted) {
372
423
  throw aborted(ALGORITHM);
373
424
  }
374
- if (weighted && s.weights !== null && !s.flags.allWeightsOne) {
375
- if (!s.flags.nonNegativeWeights) {
376
- throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: a negative weight has no shortest-path distance to sum`, {
377
- feature: "closenessCentrality.negativeWeights",
378
- hint: "pass weighted: false to ignore the column",
379
- });
380
- }
381
- if (!s.flags.finiteWeights) {
382
- throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: a NaN or infinite weight has no shortest path`, {
383
- feature: "closenessCentrality.nonFiniteWeights",
384
- });
385
- }
386
- return weightedRoute(ctx, s, scores, sources, options);
425
+ if (weighted && !s.flags.nonNegativeWeights) {
426
+ throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: a negative weight has no shortest-path distance to sum`, {
427
+ feature: "closenessCentrality.negativeWeights",
428
+ hint: "pass weighted: false to ignore the column",
429
+ });
430
+ }
431
+ if (weighted && !s.flags.finiteWeights) {
432
+ throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: a NaN or infinite weight has no shortest path`, {
433
+ feature: "closenessCentrality.nonFiniteWeights",
434
+ });
435
+ }
436
+ const fits = n > 0 && n <= allPairsCeiling(ctx.caps.limits).maxNodes;
437
+ let route;
438
+ if (sources !== null || !fits) {
439
+ route = weighted ? "per-source" : "levels";
440
+ }
441
+ else {
442
+ route = tuning.route ?? (weighted || n <= ALL_PAIRS_MAX_NODES ? "all-pairs" : "levels");
443
+ }
444
+ if (route === "levels" && weighted) {
445
+ route = "per-source";
446
+ }
447
+ tuning.onRoute?.(route);
448
+ if (route === "all-pairs") {
449
+ return allPairsRoute(ctx, s, scores, weighted, harmonic, options);
450
+ }
451
+ if (route === "per-source") {
452
+ return perSourceRoute(ctx, s, scores, sources, harmonic, options);
387
453
  }
388
- return sweepRoute(ctx, s, scores, sources, levelsPerSubmit, options, tuning);
454
+ return levelRoute(ctx, s, scores, sources, harmonic, levelsPerSubmit, options, tuning);
389
455
  }
390
456
  /**
391
457
  * Closeness centrality on the device (spec 3.3 line 810, design 8.4, 9.7): `scores[s] = 1 / sumDist_s` over the finite
392
458
  * distances from `s` to every other node, `0` when nothing is reached -- the legacy default of `@graphty/algorithms`'
393
- * `closenessCentrality`, unweighted by one bit-parallel multi-source search per 32 sources, weighted by one `sssp`
394
- * per source; `weighted` defaults to the snapshot's flag, `maxIterations` / `tolerance` are refused when defined
395
- * (PD-25). `iterations` is the source batches run and `converged` is always true.
459
+ * `closenessCentrality` -- or, with `harmonic`, the sum of `1 / dist` over the same nodes. Small and weighted
460
+ * graphs whose all-pairs matrix fits one binding run the all-pairs sweep and a row sum; other unweighted graphs run a
461
+ * bit-parallel multi-source search of up to 256 sources per batch; weighted graphs above the all-pairs ceiling run one
462
+ * `sssp` per source. `weighted` defaults to the snapshot's flag, `maxIterations` / `tolerance` are refused when
463
+ * defined. `iterations` is the source batches run and `converged` is always true.
396
464
  *
397
465
  * SAMPLED (`sources`, node indices, duplicates run twice; undirected snapshots only, E_UNSUPPORTED
398
- * `closenessCentrality.directedSources` otherwise): the batches seed the listed sources instead of every node, and
399
- * each node's score is `1 / sum` of its distances to the sources that reach it (itself excluded), `0` when none does:
400
- * the sampled score of the CPU port, unscaled. `sourcesUsed` is the list's length (`n` exact).
466
+ * `closenessCentrality.directedSources` otherwise; not with `harmonic`, E_UNSUPPORTED
467
+ * `closenessCentrality.sampledHarmonic`): the batches seed the listed sources instead of every node, and each node's
468
+ * score is `1 / sum` of its distances to the sources that reach it (itself excluded), `0` when none does: the sampled
469
+ * score of the CPU port, unscaled. `sourcesUsed` is the list's length (`n` exact).
401
470
  * @param ctx - the context whose device runs the kernels
402
471
  * @param s - the snapshot (uploaded through ctx.residency, or found there)
403
- * @param options - `weighted`, `sources`, plus dest (a Float32Array of length n for `scores`) / signal / onProgress
472
+ * @param options - `weighted`, `harmonic`, `sources`, plus dest (a Float32Array of length n for `scores`) / signal / onProgress
404
473
  * @returns the scores, the batches run, `converged: true`, `precision: "f32"` and `sourcesUsed`
405
474
  */
406
475
  export function closenessCentrality(ctx, s, options) {