@graphty/webgpu-graph-algorithms 0.6.3 → 0.6.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. package/README.md +62 -32
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-BXqgCifx.js → context-Dvq-Cc6v.js} +71 -25
  4. package/dist/chunks/context-Dvq-Cc6v.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/accelerator.d.ts +8 -6
  7. package/dist/src/accelerator.d.ts.map +1 -1
  8. package/dist/src/accelerator.js +57 -6
  9. package/dist/src/accelerator.js.map +1 -1
  10. package/dist/src/algorithms/bellman-ford.d.ts +60 -0
  11. package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
  12. package/dist/src/algorithms/bellman-ford.js +301 -0
  13. package/dist/src/algorithms/bellman-ford.js.map +1 -0
  14. package/dist/src/algorithms/bfs.d.ts +67 -0
  15. package/dist/src/algorithms/bfs.d.ts.map +1 -0
  16. package/dist/src/algorithms/bfs.js +534 -0
  17. package/dist/src/algorithms/bfs.js.map +1 -0
  18. package/dist/src/algorithms/closeness.d.ts +53 -0
  19. package/dist/src/algorithms/closeness.d.ts.map +1 -0
  20. package/dist/src/algorithms/closeness.js +323 -0
  21. package/dist/src/algorithms/closeness.js.map +1 -0
  22. package/dist/src/algorithms/scope.d.ts +5 -3
  23. package/dist/src/algorithms/scope.d.ts.map +1 -1
  24. package/dist/src/algorithms/scope.js +3 -0
  25. package/dist/src/algorithms/scope.js.map +1 -1
  26. package/dist/src/algorithms/sssp.d.ts +71 -0
  27. package/dist/src/algorithms/sssp.d.ts.map +1 -0
  28. package/dist/src/algorithms/sssp.js +585 -0
  29. package/dist/src/algorithms/sssp.js.map +1 -0
  30. package/dist/src/constants.d.ts +12 -0
  31. package/dist/src/constants.d.ts.map +1 -1
  32. package/dist/src/constants.js +12 -0
  33. package/dist/src/constants.js.map +1 -1
  34. package/dist/src/index.d.ts +8 -2
  35. package/dist/src/index.d.ts.map +1 -1
  36. package/dist/src/index.js +7 -1
  37. package/dist/src/index.js.map +1 -1
  38. package/dist/src/kernel/prelude.d.ts +4 -4
  39. package/dist/src/kernel/prelude.d.ts.map +1 -1
  40. package/dist/src/kernel/prelude.js +39 -5
  41. package/dist/src/kernel/prelude.js.map +1 -1
  42. package/dist/src/kernel/uniform-ring.d.ts +8 -0
  43. package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
  44. package/dist/src/kernel/uniform-ring.js +13 -0
  45. package/dist/src/kernel/uniform-ring.js.map +1 -1
  46. package/dist/src/kernels.d.ts +44 -4
  47. package/dist/src/kernels.d.ts.map +1 -1
  48. package/dist/src/kernels.js +371 -3
  49. package/dist/src/kernels.js.map +1 -1
  50. package/dist/src/primitives/advance.d.ts +62 -0
  51. package/dist/src/primitives/advance.d.ts.map +1 -0
  52. package/dist/src/primitives/advance.js +95 -0
  53. package/dist/src/primitives/advance.js.map +1 -0
  54. package/dist/src/primitives/compact.d.ts +89 -0
  55. package/dist/src/primitives/compact.d.ts.map +1 -0
  56. package/dist/src/primitives/compact.js +233 -0
  57. package/dist/src/primitives/compact.js.map +1 -0
  58. package/dist/src/primitives/core-shape.d.ts +22 -1
  59. package/dist/src/primitives/core-shape.d.ts.map +1 -1
  60. package/dist/src/primitives/core-shape.js +33 -3
  61. package/dist/src/primitives/core-shape.js.map +1 -1
  62. package/dist/src/primitives/frontier.d.ts +156 -0
  63. package/dist/src/primitives/frontier.d.ts.map +1 -0
  64. package/dist/src/primitives/frontier.js +259 -0
  65. package/dist/src/primitives/frontier.js.map +1 -0
  66. package/dist/src/types/accelerator.d.ts +16 -7
  67. package/dist/src/types/accelerator.d.ts.map +1 -1
  68. package/dist/src/types/traversal.d.ts +53 -0
  69. package/dist/src/types/traversal.d.ts.map +1 -0
  70. package/dist/src/types/traversal.js +10 -0
  71. package/dist/src/types/traversal.js.map +1 -0
  72. package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
  73. package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
  74. package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
  75. package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
  76. package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
  77. package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
  78. package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
  79. package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
  80. package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
  81. package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
  82. package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
  83. package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
  84. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
  85. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
  86. package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
  87. package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
  88. package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
  89. package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
  90. package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
  91. package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
  92. package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
  93. package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
  94. package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
  95. package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
  96. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
  97. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
  98. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
  99. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
  100. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
  101. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
  102. package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
  103. package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
  104. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
  105. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
  106. package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
  107. package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
  108. package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
  109. package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
  110. package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
  111. package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
  112. package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
  113. package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
  114. package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
  115. package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
  116. package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
  117. package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
  118. package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
  119. package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
  120. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +59 -0
  121. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
  122. package/dist/src/wgsl/frontier-finalize.wgsl.js +210 -0
  123. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
  124. package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
  125. package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
  126. package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
  127. package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
  128. package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
  129. package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
  130. package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
  131. package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
  132. package/dist/webgpu-graph-algorithms.js +3207 -377
  133. package/dist/webgpu-graph-algorithms.js.map +1 -1
  134. package/package.json +5 -4
  135. package/src/accelerator.ts +65 -7
  136. package/src/algorithms/bellman-ford.ts +387 -0
  137. package/src/algorithms/bfs.ts +626 -0
  138. package/src/algorithms/closeness.ts +395 -0
  139. package/src/algorithms/scope.ts +13 -3
  140. package/src/algorithms/sssp.ts +767 -0
  141. package/src/constants.ts +12 -0
  142. package/src/index.ts +14 -1
  143. package/src/kernel/prelude.ts +39 -4
  144. package/src/kernel/uniform-ring.ts +14 -0
  145. package/src/kernels.ts +450 -6
  146. package/src/primitives/advance.ts +130 -0
  147. package/src/primitives/compact.ts +323 -0
  148. package/src/primitives/core-shape.ts +41 -3
  149. package/src/primitives/frontier.ts +388 -0
  150. package/src/types/accelerator.ts +18 -5
  151. package/src/types/traversal.ts +56 -0
  152. package/src/wgsl/advance-expand.wgsl.ts +68 -0
  153. package/src/wgsl/bf-relax.wgsl.ts +57 -0
  154. package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
  155. package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
  156. package/src/wgsl/bfs-contract.wgsl.ts +54 -0
  157. package/src/wgsl/bfs-fused.wgsl.ts +77 -0
  158. package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
  159. package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
  160. package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
  161. package/src/wgsl/compact-scatter.wgsl.ts +16 -0
  162. package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
  163. package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
  164. package/src/wgsl/frontier-finalize.wgsl.ts +209 -0
  165. package/src/wgsl/sssp-pred.wgsl.ts +79 -0
  166. package/src/wgsl/sssp-relax.wgsl.ts +71 -0
  167. package/dist/chunks/context-BXqgCifx.js.map +0 -1
@@ -0,0 +1,395 @@
1
+ /**
2
+ * Closeness centrality on the device (design 8.4, 3.3 line 810, 9.7; P8-T11, the P8 plan's PD-13 / PD-19 / PD-25 /
3
+ * DEP-P8-E / DEP-P8-F): `score[s] = 1 / sumDist_s`, with `sumDist_s` the exact sum of the finite distances from `s`
4
+ * to every OTHER node -- an unreached node adds nothing -- and `0` when nothing is reached. No reached factor and no
5
+ * Wasserman-Faust scaling: this is EXACTLY the legacy default (`normalized: false`) of `closenessCentrality` in
6
+ * `@graphty/algorithms`, the number graphty-element's closeness panel shows today, so a future `indexed` port has one
7
+ * number to match (the NetworkX form is 33x it on karate and could never have been substituted silently).
8
+ *
9
+ * The unweighted route is ONE bit-parallel multi-source search per batch of 32 sources (`ceil(n / 32)` batches):
10
+ * the batch's state is one `bits` buffer of four regions of `bitsBase = roundUp(n, 64)` words (`visited`, two
11
+ * frontier regions that swap by the level's parity, `flags`), bit `s` of word `v` meaning "source `s` has reached /
12
+ * is at / is next at `v`"; a level is, all host-recorded, `closeness-reduce` role 0 (the boundary: `done` from the
13
+ * previous level's compacted count, the level's claims folded into the exact 64-bit per-source sums at `level + 1`),
14
+ * `compact` of the flags into the frontier list, two `fill`s zeroing the level's next region and the flags (AFTER
15
+ * the compaction that consumed them), and `closeness-sweep` (the block-mapped expansion with the claim inline, a
16
+ * direct grid-stride dispatch looping to the list's count). `MAX_LEVELS_PER_SUBMIT` levels per submit and
17
+ * one readback per submit (the `done` word and the 512-byte `perSource` block together, so the finished batch needs
18
+ * no extra map); the host folds `sumHi x 2^32 + sumLo` into `1 / sum` in f64 and stores f32. The weighted route
19
+ * (`weighted` true on a snapshot whose column is not all ones) is one `sssp` per source with the sums reduced on the
20
+ * host between calls: design 8.4's own answer, slow and correct. `weighted` defaults to the snapshot's `flags.weighted`;
21
+ * `weighted: false` on a weighted snapshot ignores the column by request and sweeps; `weighted: true` over unit
22
+ * weights or no column sweeps too (every `sssp` would route to a BFS anyway). `maxIterations` and `tolerance` are the
23
+ * seam's placeholder keys and an exact traversal has neither, so a defined value is REFUSED before any device work
24
+ * (`E_UNSUPPORTED { option }`, the package's rule for an option it does not implement, PD-25); `undefined` is legal.
25
+ * `iterations` reports the source batches run (the sources, on the weighted route), `converged` is always true.
26
+ *
27
+ * Cost, stated so nobody is surprised: closeness is O(n x m) on any device -- at 1M nodes it is 31,250 batches of a
28
+ * full multi-source traversal, minutes on the card, and no target in design 10.4 asks for less. `compact.record`
29
+ * leases its offsets and the scan's block sums afresh on every call, once per LEVEL here, so the planner is given a
30
+ * scope whose `scratch` hands the same buffer back for the same label and size: the dispatches of one pass run in
31
+ * order, a level's scan overwrites the previous level's, and the run holds one set instead of `levels x batches`.
32
+ * The sweep keeps DEP-P8-E's refusal of a windowed core (`assertWholeCore`): the bit-parallel claim needs the whole
33
+ * arc array bound. The tuning entry `closenessWithTuning` (PD-26's shape) is what the tests drive; nothing public
34
+ * exposes it.
35
+ */
36
+
37
+ import { type F32, type GraphSnapshot, type U32 } from "@graphty/graph-format";
38
+
39
+ import { MAX_LEVELS_PER_SUBMIT } from "../constants.js";
40
+ import { type GpuContext } from "../context.js";
41
+ import { WebGpuGraphError } from "../errors.js";
42
+ import { CommandBatch } from "../kernel/batch.js";
43
+ import { plan1d, planGridStride } from "../kernel/dispatch.js";
44
+ import {
45
+ FILL_PARAMS,
46
+ FRONTIER_COUNTERS,
47
+ FRONTIER_PARAMS,
48
+ graphBindings,
49
+ graphOverrides,
50
+ kernelSpec,
51
+ } from "../kernels.js";
52
+ import { prepareCompact } from "../primitives/compact.js";
53
+ import { assertWholeCore } from "../primitives/core-shape.js";
54
+ import { W } from "../primitives/frontier.js";
55
+ import { type ReduceScope } from "../primitives/reduce.js";
56
+ import { assertDeviceComputes } from "../primitives/verify.js";
57
+ import { type HitsOptionsLike } from "../types/accelerator.js";
58
+ import { type GpuScoresResult } from "../types/algorithms.js";
59
+ import { type Binding } from "../types/memory.js";
60
+ import { type GpuRunOptions } from "../types/run.js";
61
+ import { algorithmScope } from "./scope.js";
62
+ import { aborted, bindingOf, checkDest, sssp } from "./sssp.js";
63
+
64
+ const ALGORITHM = "closenessCentrality";
65
+
66
+ /**
67
+ * Design 8.4: 32 sources per `u32` word, one batch per word.
68
+ * @internal
69
+ */
70
+ export const SOURCES_PER_BATCH = 32;
71
+
72
+ /**
73
+ * The `perSource` block of a batch: `newCount[32]` @0, `reached[32]` @32, `sumLo[32]` @64, `sumHi[32]` @96.
74
+ * @internal
75
+ */
76
+ export const PER_SOURCE_WORDS = 4 * SOURCES_PER_BATCH;
77
+
78
+ /**
79
+ * Params slots of the ring, COUNTED (`UniformRing.reserve` wraps silently): per level `compact`'s records (its scan
80
+ * is at most four levels for any n below 2^32, so at most 8 records) while the boundary, the finalize, the fill and
81
+ * the two sweep records (one per parity) are written once per submit, plus the seed's two records on a batch's first
82
+ * submit (the iota fill flushes in its own submit).
83
+ */
84
+ const RING_SLOTS = 8 * MAX_LEVELS_PER_SUBMIT + 16;
85
+
86
+ /**
87
+ * The knobs the tests need and nothing public offers (PD-26's shape): the submit cadence and the inspect seam.
88
+ * @internal
89
+ */
90
+ export interface ClosenessTuning {
91
+ /** Levels recorded per submit on the bit-parallel route (default `MAX_LEVELS_PER_SUBMIT`). */
92
+ readonly levelsPerSubmit?: number | undefined;
93
+ /** The inspect seam: after every BATCH of the bit-parallel route, its first source and a fresh copy of the 128-word `perSource` block as the last submit left it. */
94
+ readonly onBatch?: ((batchStart: number, perSource: U32) => void) | undefined;
95
+ }
96
+
97
+ /**
98
+ * A scope whose `scratch` hands the SAME buffer back for the same label and size (see the file comment).
99
+ * @param scope - the algorithm's scope
100
+ * @returns the reusing scope
101
+ */
102
+ function reusingScratch(scope: ReduceScope): ReduceScope {
103
+ const held = new Map<string, GPUBuffer>();
104
+ return {
105
+ ...scope,
106
+ scratch: (byteLength, label) => {
107
+ const key = `${label}/${byteLength}`;
108
+ let buffer = held.get(key);
109
+ if (buffer === undefined) {
110
+ buffer = scope.scratch(byteLength, label);
111
+ held.set(key, buffer);
112
+ }
113
+ return buffer;
114
+ },
115
+ };
116
+ }
117
+
118
+ /**
119
+ * The weighted route: one `sssp` per source, the sums reduced on the host.
120
+ * @param ctx - the context
121
+ * @param s - the snapshot
122
+ * @param scores - the destination
123
+ * @param options - the run options
124
+ * @returns the result
125
+ */
126
+ async function weightedRoute(
127
+ ctx: GpuContext,
128
+ s: GraphSnapshot,
129
+ scores: F32,
130
+ options: GpuRunOptions | undefined,
131
+ ): Promise<GpuScoresResult> {
132
+ const n = s.nodeCount;
133
+ for (let source = 0; source < n; source++) {
134
+ if (options?.signal?.aborted) {
135
+ throw aborted(ALGORITHM);
136
+ }
137
+ const { dist } = await sssp(ctx, s, source, { signal: options?.signal });
138
+ let sum = 0;
139
+ for (let v = 0; v < n; v++) {
140
+ const d = dist[v];
141
+ if (v !== source && d !== Infinity) {
142
+ sum += d;
143
+ }
144
+ }
145
+ scores[source] = sum === 0 ? 0 : 1 / sum;
146
+ options?.onProgress?.(source + 1, n);
147
+ }
148
+ return { scores, iterations: n, converged: true, precision: "f32" };
149
+ }
150
+
151
+ /**
152
+ * The bit-parallel route (see the file comment).
153
+ * @param ctx - the context
154
+ * @param s - the snapshot
155
+ * @param scores - the destination
156
+ * @param levelsPerSubmit - the submit cadence
157
+ * @param options - the run options
158
+ * @param tuning - the knobs
159
+ * @returns the result
160
+ */
161
+ async function sweepRoute(
162
+ ctx: GpuContext,
163
+ s: GraphSnapshot,
164
+ scores: F32,
165
+ levelsPerSubmit: number,
166
+ options: GpuRunOptions | undefined,
167
+ tuning: ClosenessTuning,
168
+ ): Promise<GpuScoresResult> {
169
+ const n = s.nodeCount;
170
+ if (n === 0) {
171
+ return { scores, iterations: 0, converged: true, precision: "f32" };
172
+ }
173
+ const core = ctx.residency.core(s);
174
+ assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM);
175
+ const scope = algorithmScope(ctx, ALGORITHM, RING_SLOTS);
176
+ try {
177
+ const wg = ctx.workgroupSize;
178
+ const bytes = 4 * n;
179
+ // the four regions of the bits buffer, each bitsBase words so its byte offset is 256-aligned and fill and
180
+ // compact can bind one alone: visited 0, the frontier pair 1 and 2, flags 3
181
+ const bitsBase = Math.ceil(n / 64) * 64;
182
+ const regionBytes = 4 * bitsBase;
183
+ const bits = bindingOf(scope.scratch(4 * regionBytes, "bits"), 4 * regionBytes);
184
+ const region = (index: number): Binding => ({
185
+ buffer: bits.buffer,
186
+ offset: index * regionBytes,
187
+ size: regionBytes,
188
+ window: null,
189
+ });
190
+ const flags = region(3);
191
+ const frontierList = bindingOf(scope.scratch(bytes, "frontier-list"), bytes);
192
+ const iota = bindingOf(scope.scratch(bytes, "iota"), bytes);
193
+ const counters = bindingOf(
194
+ scope.scratch(FRONTIER_COUNTERS.byteLength, "counters"),
195
+ FRONTIER_COUNTERS.byteLength,
196
+ );
197
+ const perSourceBytes = 4 * PER_SOURCE_WORDS;
198
+ const perSource = bindingOf(scope.scratch(perSourceBytes, "per-source"), perSourceBytes);
199
+ await ctx.allocator.check();
200
+ const compact = await prepareCompact(reusingScratch(scope));
201
+ const sweep = await ctx.pipelines.kernel(kernelSpec("closeness-sweep", graphOverrides(core, null)));
202
+ const reduce = await ctx.pipelines.kernel(kernelSpec("closeness-reduce"));
203
+ const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
204
+ const graph = graphBindings(core, null);
205
+ const onePlan = plan1d(1, wg, ctx.caps);
206
+ const regionPlan = plan1d(bitsBase, wg, ctx.caps);
207
+ const sweepPlan = planGridStride(n, wg, ctx.caps); // the list holds at most n entries; the sweep loops to the count word
208
+ const recordFill = (pass: GPUComputePassEncoder, dst: Binding, count: number, mode: 0 | 1): void => {
209
+ const params = scope.params(FILL_PARAMS, { count, value: 0, mode, pad0: 0 });
210
+ fill.dispatch(pass, fill.bind({ dst, P: params.binding }), plan1d(count, wg, ctx.caps), [params.offset]);
211
+ };
212
+ const submit = (batch: CommandBatch): ReturnType<CommandBatch["submit"]> => {
213
+ scope.flush();
214
+ return batch.submit();
215
+ };
216
+
217
+ // setup: the iota queue compact reads, once per run
218
+ const setup = new CommandBatch(ctx, `${ALGORITHM}/setup`);
219
+ recordFill(setup.pass("fill"), iota, n, 1);
220
+ setup.endPass();
221
+ await submit(setup).readback;
222
+ ctx.assertReady();
223
+
224
+ let batches = 0;
225
+ for (let batchStart = 0; batchStart < n; batchStart += SOURCES_PER_BATCH) {
226
+ let level = 0;
227
+ for (let first = true; ; first = false) {
228
+ const batch = new CommandBatch(ctx, `${ALGORITHM}/levels`);
229
+ const pass = batch.pass("closeness");
230
+ if (first) {
231
+ // the batch's seed: the four regions and the block zeroed, then role 1 (the sources' bits, their
232
+ // flags, counters[0] = k, level = U32_MAX)
233
+ recordFill(pass, bits, 4 * bitsBase, 0);
234
+ recordFill(pass, perSource, PER_SOURCE_WORDS, 0);
235
+ const seed = scope.params(FRONTIER_PARAMS, { role: 1, n, bitsBase, source: batchStart });
236
+ reduce.dispatch(pass, reduce.bind({ counters, perSource, bits, P: seed.binding }), onePlan, [
237
+ seed.offset,
238
+ ]);
239
+ }
240
+ // the records every level of the submit shares (the ring wraps, so they are written per submit)
241
+ const boundary = scope.params(FRONTIER_PARAMS, { role: 0, n, bitsBase });
242
+ const boundBoundary = reduce.bind({ counters, perSource, bits, P: boundary.binding });
243
+ const clear = scope.params(FILL_PARAMS, { count: bitsBase, value: 0, mode: 0, pad0: 0 });
244
+ // parity 0 sweeps region 1 into region 2, parity 1 region 2 into region 1: the next region is cleared
245
+ const boundClearNext = [region(2), region(1)].map((dst) => fill.bind({ dst, P: clear.binding }));
246
+ const boundClearFlags = fill.bind({ dst: flags, P: clear.binding });
247
+ const boundSweep = [0, 1].map((mode) => {
248
+ const params = scope.params(FRONTIER_PARAMS, {
249
+ wg,
250
+ n,
251
+ bitsBase,
252
+ arcBase: 0,
253
+ arcEnd: s.arcCount,
254
+ mode,
255
+ stride: sweepPlan.stride ?? wg,
256
+ });
257
+ return {
258
+ bound: sweep.bind({ ...graph, frontierList, counters, bits, perSource, P: params.binding }),
259
+ offset: params.offset,
260
+ };
261
+ });
262
+ for (let k = 0; k < levelsPerSubmit; k++, level++) {
263
+ const parity = level % 2;
264
+ reduce.dispatch(pass, boundBoundary, onePlan, [boundary.offset]);
265
+ compact.record(pass, {
266
+ queue: iota,
267
+ flags,
268
+ count: n,
269
+ out: frontierList,
270
+ outCount: counters,
271
+ outIndex: W.frontierCount,
272
+ });
273
+ fill.dispatch(pass, boundClearNext[parity], regionPlan, [clear.offset]);
274
+ fill.dispatch(pass, boundClearFlags, regionPlan, [clear.offset]);
275
+ sweep.dispatch(pass, boundSweep[parity].bound, sweepPlan, [boundSweep[parity].offset]);
276
+ }
277
+ batch.endPass();
278
+ const doneRequest = batch.readback(counters.buffer, counters.offset + 4 * W.done, 4);
279
+ const blockRequest = batch.readback(perSource.buffer, perSource.offset, perSourceBytes);
280
+ const submitted = submit(batch);
281
+ const back = await submitted.readback;
282
+ ctx.assertReady();
283
+ if (options?.signal?.aborted) {
284
+ throw aborted(ALGORITHM, submitted.id);
285
+ }
286
+ if (new Uint32Array(back, doneRequest.offset, 1)[0] !== 0) {
287
+ const block = new Uint32Array(back, blockRequest.offset, PER_SOURCE_WORDS);
288
+ const count = Math.min(SOURCES_PER_BATCH, n - batchStart);
289
+ for (let i = 0; i < count; i++) {
290
+ const sum = block[3 * SOURCES_PER_BATCH + i] * 2 ** 32 + block[2 * SOURCES_PER_BATCH + i];
291
+ scores[batchStart + i] = sum === 0 ? 0 : 1 / sum;
292
+ }
293
+ tuning.onBatch?.(batchStart, block.slice());
294
+ break;
295
+ }
296
+ if (level > n + 3) {
297
+ // a batch claims at most n - 1 levels deep, then one level claims nothing and one is empty
298
+ throw new WebGpuGraphError(
299
+ "E_VALIDATION",
300
+ `${ALGORITHM}: the done flag never rose in ${level} levels of the batch at ${batchStart}`,
301
+ { label: ALGORITHM, message: `the done flag never rose in ${level} levels` },
302
+ );
303
+ }
304
+ }
305
+ batches += 1;
306
+ options?.onProgress?.(Math.min(batchStart + SOURCES_PER_BATCH, n), n);
307
+ }
308
+ return { scores, iterations: batches, converged: true, precision: "f32" };
309
+ } finally {
310
+ scope.dispose();
311
+ }
312
+ }
313
+
314
+ /**
315
+ * Closeness with the test knobs of PD-26's shape; `closenessCentrality` is this with an empty tuning.
316
+ * @internal
317
+ * @param ctx - the context whose device runs the kernels
318
+ * @param s - the snapshot (uploaded through ctx.residency, or found there)
319
+ * @param options - the seam's `HitsOptionsLike` (`weighted` honoured, the other two refused when defined), plus dest / signal / onProgress
320
+ * @param tuning - the knobs
321
+ * @returns the scores, the batches run, `converged: true` and `precision: "f32"`
322
+ */
323
+ export async function closenessWithTuning(
324
+ ctx: GpuContext,
325
+ s: GraphSnapshot,
326
+ options: (HitsOptionsLike & GpuRunOptions) | undefined,
327
+ tuning: ClosenessTuning,
328
+ ): Promise<GpuScoresResult> {
329
+ ctx.assertReady();
330
+ await assertDeviceComputes(ctx);
331
+ for (const key of ["maxIterations", "tolerance"] as const) {
332
+ if (options?.[key] !== undefined) {
333
+ throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: ${key} has no meaning for an exact traversal`, {
334
+ option: key,
335
+ hint: "closeness is an exact traversal; the option has no meaning here",
336
+ });
337
+ }
338
+ }
339
+ const n = s.nodeCount;
340
+ const levelsPerSubmit = tuning.levelsPerSubmit ?? MAX_LEVELS_PER_SUBMIT;
341
+ if (!Number.isInteger(levelsPerSubmit) || levelsPerSubmit < 1 || levelsPerSubmit > MAX_LEVELS_PER_SUBMIT) {
342
+ throw new WebGpuGraphError(
343
+ "E_INVALID_ARGUMENT",
344
+ `${ALGORITHM}: levelsPerSubmit must be an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
345
+ {
346
+ argument: "levelsPerSubmit",
347
+ value: levelsPerSubmit,
348
+ expected: `an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
349
+ },
350
+ );
351
+ }
352
+ const scores = checkDest(ALGORITHM, options?.dest, n) ?? new Float32Array(n);
353
+ const weighted = options?.weighted ?? s.flags.weighted;
354
+ if (options?.signal?.aborted) {
355
+ throw aborted(ALGORITHM);
356
+ }
357
+ if (weighted && s.weights !== null && !s.flags.allWeightsOne) {
358
+ if (!s.flags.nonNegativeWeights) {
359
+ throw new WebGpuGraphError(
360
+ "E_UNSUPPORTED",
361
+ `${ALGORITHM}: a negative weight has no shortest-path distance to sum`,
362
+ {
363
+ feature: "closenessCentrality.negativeWeights",
364
+ hint: "pass weighted: false to ignore the column",
365
+ },
366
+ );
367
+ }
368
+ if (!s.flags.finiteWeights) {
369
+ throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: a NaN or infinite weight has no shortest path`, {
370
+ feature: "closenessCentrality.nonFiniteWeights",
371
+ });
372
+ }
373
+ return weightedRoute(ctx, s, scores, options);
374
+ }
375
+ return sweepRoute(ctx, s, scores, levelsPerSubmit, options, tuning);
376
+ }
377
+
378
+ /**
379
+ * Closeness centrality on the device (spec 3.3 line 810, design 8.4, 9.7): `scores[s] = 1 / sumDist_s` over the finite
380
+ * distances from `s` to every other node, `0` when nothing is reached -- the legacy default of `@graphty/algorithms`'
381
+ * `closenessCentrality`, unweighted by one bit-parallel multi-source search per 32 sources, weighted by one `sssp`
382
+ * per source; `weighted` defaults to the snapshot's flag, `maxIterations` / `tolerance` are refused when defined
383
+ * (PD-25). `iterations` is the source batches run and `converged` is always true.
384
+ * @param ctx - the context whose device runs the kernels
385
+ * @param s - the snapshot (uploaded through ctx.residency, or found there)
386
+ * @param options - the seam's `HitsOptionsLike`, plus dest (a Float32Array of length n for `scores`) / signal / onProgress
387
+ * @returns the scores, the batches run, `converged: true` and `precision: "f32"`
388
+ */
389
+ export function closenessCentrality(
390
+ ctx: GpuContext,
391
+ s: GraphSnapshot,
392
+ options?: HitsOptionsLike & GpuRunOptions,
393
+ ): Promise<GpuScoresResult> {
394
+ return closenessWithTuning(ctx, s, options, {});
395
+ }
@@ -8,15 +8,18 @@
8
8
  */
9
9
 
10
10
  import { type GpuContext } from "../context.js";
11
+ import { BufferUsage } from "../device/webgpu-constants.js";
11
12
  import { UniformRing } from "../kernel/uniform-ring.js";
12
- import { type ReduceScope } from "../primitives/reduce.js";
13
+ import { type FrontierScope } from "../primitives/frontier.js";
13
14
 
14
- /** A ReduceScope over a context plus the two lifecycle calls an algorithm makes: flush() before submit, dispose() in its finally. */
15
- export interface AlgorithmScope extends ReduceScope {
15
+ /** A FrontierScope (a ReduceScope plus `indirect()`, P8-T4) over a context plus the two lifecycle calls an algorithm makes: flush() before submit, dispose() in its finally. */
16
+ export interface AlgorithmScope extends FrontierScope {
16
17
  /** queue.writeBuffer of the params slots written since the last flush (called before the batch is submitted). */
17
18
  flush(): void;
18
19
  /** Destroys the ring and releases every scratch buffer of the lease; idempotent. */
19
20
  dispose(): void;
21
+ /** The ring's `overruns` so far: reservations that wrapped over a record the batch being recorded still reads (0 on a correctly sized ring; P8-T12). */
22
+ ringOverruns(): number;
20
23
  }
21
24
 
22
25
  /**
@@ -36,6 +39,12 @@ export function algorithmScope(ctx: GpuContext, label: string, slots: number): A
36
39
  pool: ctx.pool,
37
40
  workgroupSize: ctx.workgroupSize,
38
41
  scratch: (byteLength, scratchLabel) => lease.storage(byteLength, `${label}/${scratchLabel}`),
42
+ indirect: (byteLength, indirectLabel) =>
43
+ lease.acquire(
44
+ byteLength,
45
+ BufferUsage.STORAGE | BufferUsage.INDIRECT | BufferUsage.COPY_DST | BufferUsage.COPY_SRC,
46
+ `${label}/${indirectLabel}`,
47
+ ),
39
48
  params(block, values) {
40
49
  const slot = ring.reserve(1);
41
50
  ring.write(slot, block, values);
@@ -48,5 +57,6 @@ export function algorithmScope(ctx: GpuContext, label: string, slots: number): A
48
57
  ring.destroy();
49
58
  lease.release();
50
59
  },
60
+ ringOverruns: () => ring.overruns,
51
61
  };
52
62
  }