@graphty/webgpu-graph-algorithms 0.6.14 → 0.6.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. package/README.md +52 -52
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-Bi6AhScG.js → context-oXphO3yj.js} +36 -28
  4. package/dist/chunks/context-oXphO3yj.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/accelerator.d.ts +3 -2
  7. package/dist/src/accelerator.d.ts.map +1 -1
  8. package/dist/src/accelerator.js +48 -4
  9. package/dist/src/accelerator.js.map +1 -1
  10. package/dist/src/algorithms/betweenness.d.ts +70 -0
  11. package/dist/src/algorithms/betweenness.d.ts.map +1 -0
  12. package/dist/src/algorithms/betweenness.js +538 -0
  13. package/dist/src/algorithms/betweenness.js.map +1 -0
  14. package/dist/src/algorithms/closeness.d.ts +15 -5
  15. package/dist/src/algorithms/closeness.d.ts.map +1 -1
  16. package/dist/src/algorithms/closeness.js +112 -26
  17. package/dist/src/algorithms/closeness.js.map +1 -1
  18. package/dist/src/constants.d.ts +8 -0
  19. package/dist/src/constants.d.ts.map +1 -1
  20. package/dist/src/constants.js +8 -0
  21. package/dist/src/constants.js.map +1 -1
  22. package/dist/src/index.d.ts +4 -2
  23. package/dist/src/index.d.ts.map +1 -1
  24. package/dist/src/index.js +1 -0
  25. package/dist/src/index.js.map +1 -1
  26. package/dist/src/kernels.d.ts +12 -6
  27. package/dist/src/kernels.d.ts.map +1 -1
  28. package/dist/src/kernels.js +153 -7
  29. package/dist/src/kernels.js.map +1 -1
  30. package/dist/src/primitives/frontier.d.ts +2 -0
  31. package/dist/src/primitives/frontier.d.ts.map +1 -1
  32. package/dist/src/primitives/frontier.js +2 -0
  33. package/dist/src/primitives/frontier.js.map +1 -1
  34. package/dist/src/types/accelerator.d.ts +11 -7
  35. package/dist/src/types/accelerator.d.ts.map +1 -1
  36. package/dist/src/types/algorithms.d.ts +4 -0
  37. package/dist/src/types/algorithms.d.ts.map +1 -1
  38. package/dist/src/types/betweenness.d.ts +35 -0
  39. package/dist/src/types/betweenness.d.ts.map +1 -0
  40. package/dist/src/types/betweenness.js +7 -0
  41. package/dist/src/types/betweenness.js.map +1 -0
  42. package/dist/src/wgsl/bc-backward.wgsl.d.ts +15 -0
  43. package/dist/src/wgsl/bc-backward.wgsl.d.ts.map +1 -0
  44. package/dist/src/wgsl/bc-backward.wgsl.js +34 -0
  45. package/dist/src/wgsl/bc-backward.wgsl.js.map +1 -0
  46. package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts +12 -0
  47. package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts.map +1 -0
  48. package/dist/src/wgsl/bc-edge-gather.wgsl.js +36 -0
  49. package/dist/src/wgsl/bc-edge-gather.wgsl.js.map +1 -0
  50. package/dist/src/wgsl/bc-finalize.wgsl.d.ts +21 -0
  51. package/dist/src/wgsl/bc-finalize.wgsl.d.ts.map +1 -0
  52. package/dist/src/wgsl/bc-finalize.wgsl.js +47 -0
  53. package/dist/src/wgsl/bc-finalize.wgsl.js.map +1 -0
  54. package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts +15 -0
  55. package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts.map +1 -0
  56. package/dist/src/wgsl/bc-forward-edge.wgsl.js +76 -0
  57. package/dist/src/wgsl/bc-forward-edge.wgsl.js.map +1 -0
  58. package/dist/src/wgsl/bc-forward.wgsl.d.ts +23 -0
  59. package/dist/src/wgsl/bc-forward.wgsl.d.ts.map +1 -0
  60. package/dist/src/wgsl/bc-forward.wgsl.js +106 -0
  61. package/dist/src/wgsl/bc-forward.wgsl.js.map +1 -0
  62. package/dist/src/wgsl/bc-gather.wgsl.d.ts +9 -0
  63. package/dist/src/wgsl/bc-gather.wgsl.d.ts.map +1 -0
  64. package/dist/src/wgsl/bc-gather.wgsl.js +20 -0
  65. package/dist/src/wgsl/bc-gather.wgsl.js.map +1 -0
  66. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +4 -1
  67. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -1
  68. package/dist/src/wgsl/closeness-reduce.wgsl.js +8 -4
  69. package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -1
  70. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +4 -2
  71. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -1
  72. package/dist/src/wgsl/closeness-sweep.wgsl.js +12 -2
  73. package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -1
  74. package/dist/webgpu-graph-algorithms.js +1037 -168
  75. package/dist/webgpu-graph-algorithms.js.map +1 -1
  76. package/package.json +5 -5
  77. package/src/accelerator.ts +75 -7
  78. package/src/algorithms/betweenness.ts +739 -0
  79. package/src/algorithms/closeness.ts +124 -32
  80. package/src/constants.ts +8 -0
  81. package/src/index.ts +8 -0
  82. package/src/kernels.ts +169 -10
  83. package/src/primitives/frontier.ts +4 -0
  84. package/src/types/accelerator.ts +18 -6
  85. package/src/types/algorithms.ts +5 -0
  86. package/src/types/betweenness.ts +38 -0
  87. package/src/wgsl/bc-backward.wgsl.ts +33 -0
  88. package/src/wgsl/bc-edge-gather.wgsl.ts +35 -0
  89. package/src/wgsl/bc-finalize.wgsl.ts +46 -0
  90. package/src/wgsl/bc-forward-edge.wgsl.ts +75 -0
  91. package/src/wgsl/bc-forward.wgsl.ts +105 -0
  92. package/src/wgsl/bc-gather.wgsl.ts +19 -0
  93. package/src/wgsl/closeness-reduce.wgsl.ts +8 -4
  94. package/src/wgsl/closeness-sweep.wgsl.ts +12 -2
  95. package/dist/chunks/context-Bi6AhScG.js.map +0 -1
@@ -0,0 +1,739 @@
1
+ /**
2
+ * Betweenness and edge betweenness on the device (design 8.4 "Betweenness (A7)", 3.3 lines 811-812, 9.7): Brandes'
3
+ * algorithm as McLaughlin-Bader run it, k sources at a time.
4
+ *
5
+ * A source batch keeps four `n x k` arrays -- `depthK`, `sigmaK` (u32 shortest-path counts), `deltaK` (f32
6
+ * dependencies) and the claim log `S` -- plus `ends`, the level boundaries into the log. Every array is indexed by
7
+ * `s * n + v`, and a log entry IS that index, so one u32 names a `(vertex, source)` pair: `4 n k` bytes can never
8
+ * reach 2^32 because each array is one storage binding. The forward pass is a tagged breadth-first search: every
9
+ * level is `bc-finalize` (the boundary: `ends[level + 1] = stackTop`, `done` on an empty level) and then ONE forward
10
+ * dispatch, either `bc-forward` (block-mapped over the level's range of the log) or `bc-forward-edge` (every edge of
11
+ * the `edgeList` view for every source), both claiming, counting and appending with the same rules. Levels are
12
+ * recorded `MAX_LEVELS_PER_SUBMIT` per submit with one small readback (the counters block and the `ends` prefix), and
13
+ * the log's `ends` is known on the host when the forward phase ends, so the backward pass is planned exactly: one
14
+ * `bc-backward` dispatch per level from the deepest to depth 1, each writing `delta[s][w]` once by pulling over the
15
+ * successors; then `bc-gather` adds each vertex's k dependencies into `bc` (and `bc-edge-gather` each arc's k terms
16
+ * into `arcScores`). Nothing accumulates a float through an atomic, so the scores are bitwise reproducible.
17
+ *
18
+ * The batch size k is planned from the device limits at 16 bytes per (node, source) -- the three arrays design 4.7
19
+ * counts plus the 4-byte log entry it omits -- as `min(floor(maxStorageBufferBindingSize / 4n), floor(0.25 x
20
+ * maxBufferSize / 16n), 64)`, at least 1 (`planBatchSize`); a graph whose single source does not fit one binding is
21
+ * E_TOO_LARGE. The forward form is chosen per batch: the first batch runs `bc-forward`; a later one runs the
22
+ * edge-parallel form when the previous batch's level count -- the MAXIMUM depth over its sources, the only depth
23
+ * figure the host has, which over-estimates the median of design 8.4's rule and so errs toward the frontier form --
24
+ * is below `BC_EDGE_PARALLEL_GAMMA x log2(n)`.
25
+ *
26
+ * The host applies the CPU package's convention after the readback (`@graphty/algorithms` betweenness.ts): the vertex
27
+ * scores are halved on an undirected snapshot (the gather counts each unordered pair from both ends); the edge scores
28
+ * are folded with `foldArcs(s, perArc, "sum")` and halved the same way, because an undirected edge's two arcs carry
29
+ * the pairs that cross it in each direction. Over every source both arcs hold the same sum, but over a sample they do
30
+ * not: on the path 0-1-2 with the one source 2, arc 1->0 collects 1 and arc 0->1 collects 0, so keeping either arc
31
+ * alone would lose the pairs crossing the other way. `normalized`
32
+ * divides both by `(n - 1)(n - 2)` directed or half that undirected, when positive. A sampled run (`sources` or `k`)
33
+ * is the UNSCALED sum over the sources run, reported beside `sourcesUsed`. `endpoints: true` is refused: the CPU's
34
+ * endpoints branch (`algorithms/src/algorithms/centrality/betweenness.ts`, `predecessors.length === 0 && w !==
35
+ * source`) can never fire for a vertex on the Brandes stack, so there is no behaviour to be in parity with.
36
+ * Betweenness is breadth-first on both packages; weights are ignored. Parallel edges are distinct shortest paths
37
+ * here (each arc adds to sigma), where the CPU package refuses them or, with `allowParallelEdges`, collapses them to
38
+ * one: on a multigraph the two packages' scores differ.
39
+ */
40
+
41
+ import { type F32, foldArcs, type GraphSnapshot, type U32 } from "@graphty/graph-format";
42
+
43
+ import {
44
+ BC_BACKWARD_LEVELS_PER_SUBMIT,
45
+ BC_BATCH_BUDGET_FRACTION,
46
+ BC_EDGE_PARALLEL_GAMMA,
47
+ BC_MAX_BATCH,
48
+ MAX_LEVELS_PER_SUBMIT,
49
+ } from "../constants.js";
50
+ import { type GpuContext } from "../context.js";
51
+ import { WebGpuGraphError } from "../errors.js";
52
+ import { CommandBatch } from "../kernel/batch.js";
53
+ import { plan1d, planGridStride } from "../kernel/dispatch.js";
54
+ import { type BoundKernel, type Kernel } from "../kernel/kernel.js";
55
+ import { BC_PARAMS, FILL_PARAMS, FRONTIER_COUNTERS, kernelSpec } from "../kernels.js";
56
+ import { assertWholeCore } from "../primitives/core-shape.js";
57
+ import { W } from "../primitives/frontier.js";
58
+ import { assertDeviceComputes } from "../primitives/verify.js";
59
+ import { type BetweennessAcceleratorOptions } from "../types/accelerator.js";
60
+ import { type GpuBetweennessResult, type GpuEdgeScoresResult } from "../types/betweenness.js";
61
+ import { type Binding } from "../types/memory.js";
62
+ import { type GpuRunOptions } from "../types/run.js";
63
+ import { type AlgorithmScope, algorithmScope } from "./scope.js";
64
+ import { aborted, bindingOf, checkDest } from "./sssp.js";
65
+
66
+ const ALGORITHM = "betweennessCentrality";
67
+
68
+ /** Bytes one (node, source) pair holds in a batch: depthK, sigmaK, deltaK and the claim-log entry, 4 each. */
69
+ const BYTES_PER_NODE_SOURCE = 16;
70
+
71
+ /** The seed of the deterministic draw of `k` sources (any fixed value: the draw only has to repeat). */
72
+ const SAMPLE_SEED = 0x9e3779b9;
73
+
74
+ /** Params slots of the ring: a backward submit's levels plus the fills, the seed, the gathers and the per-submit forward records. */
75
+ const RING_SLOTS = BC_BACKWARD_LEVELS_PER_SUBMIT + 16;
76
+
77
+ /** The device limits the batch planner reads. */
78
+ interface BatchLimits {
79
+ readonly maxStorageBufferBindingSize: number;
80
+ readonly maxBufferSize: number;
81
+ }
82
+
83
+ /**
84
+ * Which forward body a batch runs: pinned by the tuning, or chosen per batch.
85
+ * @internal
86
+ */
87
+ export type ForwardForm = "frontier" | "edge";
88
+
89
+ /**
90
+ * What the inspect seam hands the tests after every batch.
91
+ * @internal
92
+ */
93
+ export interface BetweennessBatchReport {
94
+ /** The batch's sources, in tag order. */
95
+ readonly sources: readonly number[];
96
+ /** The forward body the batch ran. */
97
+ readonly forward: ForwardForm;
98
+ /** The non-empty levels (depth 0 .. levels - 1). */
99
+ readonly levels: number;
100
+ /** `ends[0 .. levels + 1]`: the entries at depth L are `S[ends[L] .. ends[L + 1])`. */
101
+ readonly ends: U32;
102
+ /** Whether a u32 path count wrapped in this batch. */
103
+ readonly sigmaOverflow: boolean;
104
+ /** The `n x k` arrays as the batch left them (null unless the tuning asked for them). */
105
+ readonly depthK: U32 | null;
106
+ readonly sigmaK: U32 | null;
107
+ readonly deltaK: F32 | null;
108
+ }
109
+
110
+ /**
111
+ * The knobs the tests need and nothing public offers.
112
+ * @internal
113
+ */
114
+ export interface BetweennessTuning {
115
+ /** Pins the forward body (default: the per-batch choice). */
116
+ readonly forward?: ForwardForm | "auto" | undefined;
117
+ /** Forward levels recorded per submit (default `MAX_LEVELS_PER_SUBMIT`). */
118
+ readonly levelsPerSubmit?: number | undefined;
119
+ /** Faked device limits for the planner (a test's way to make k shrink). */
120
+ readonly limits?: BatchLimits | undefined;
121
+ /** Called after every batch. */
122
+ readonly onBatch?: ((report: BetweennessBatchReport) => void) | undefined;
123
+ /** Read the batch's `depthK`, `sigmaK` and `deltaK` back for `onBatch`. */
124
+ readonly readArrays?: boolean | undefined;
125
+ }
126
+
127
+ /** The raw outcome of a run: the unhalved, unnormalised sums. */
128
+ interface RawBetweenness {
129
+ readonly vertex: F32;
130
+ readonly perArc: F32 | null;
131
+ readonly sourcesUsed: number;
132
+ readonly sigmaOverflow: boolean;
133
+ readonly batches: number;
134
+ }
135
+
136
+ /**
137
+ * The source batch size (design 8.4, 10.1): `min(floor(maxStorageBufferBindingSize / 4n), floor(0.25 x maxBufferSize /
138
+ * 16n), 64, remaining)`, at least 1. Each of the four `n x k` arrays is one binding of `4 n k` bytes, and the batch's
139
+ * four together stay inside a quarter of the largest buffer.
140
+ * @internal
141
+ * @param n - the vertex count (>= 1)
142
+ * @param remaining - the sources still to run (>= 1)
143
+ * @param limits - the device limits
144
+ * @returns k
145
+ * @throws E_TOO_LARGE when one source's `ends` array (`4 (n + 2)` bytes, the largest) exceeds the binding limit
146
+ */
147
+ export function planBatchSize(n: number, remaining: number, limits: BatchLimits): number {
148
+ // the largest single binding of a one-source batch is `ends`, 4 (n + 2) bytes
149
+ const needed = 4 * (n + 2);
150
+ if (needed > limits.maxStorageBufferBindingSize) {
151
+ throw new WebGpuGraphError(
152
+ "E_TOO_LARGE",
153
+ `${ALGORITHM}: one source needs ${needed} bytes in its largest binding at n = ${n}, above maxStorageBufferBindingSize = ${limits.maxStorageBufferBindingSize}; a limit of at least ${needed} admits one source per batch`,
154
+ { needed, limit: limits.maxStorageBufferBindingSize, path: "binding", algorithm: ALGORITHM },
155
+ );
156
+ }
157
+ const kByBinding = Math.floor(limits.maxStorageBufferBindingSize / (4 * n));
158
+ const kByBudget = Math.floor((BC_BATCH_BUDGET_FRACTION * limits.maxBufferSize) / (BYTES_PER_NODE_SOURCE * n));
159
+ return Math.max(1, Math.min(kByBinding, kByBudget, BC_MAX_BATCH, remaining));
160
+ }
161
+
162
+ /**
163
+ * `k` distinct vertices drawn by a partial Fisher-Yates shuffle over a fixed-seed mulberry32 stream: the same call
164
+ * draws the same sources every time.
165
+ * @param n - the vertex count
166
+ * @param k - how many to draw (<= n)
167
+ * @returns the sources
168
+ */
169
+ function drawSources(n: number, k: number): number[] {
170
+ const pool = Array.from({ length: n }, (_, i) => i);
171
+ let state = SAMPLE_SEED;
172
+ for (let i = 0; i < k; i++) {
173
+ state = (state + 0x6d2b79f5) >>> 0;
174
+ let t = Math.imul(state ^ (state >>> 15), state | 1);
175
+ t = (t + Math.imul(t ^ (t >>> 7), t | 61)) ^ t;
176
+ const unit = ((t ^ (t >>> 14)) >>> 0) / 2 ** 32;
177
+ const j = i + Math.floor(unit * (n - i));
178
+ [pool[i], pool[j]] = [pool[j], pool[i]];
179
+ }
180
+ return pool.slice(0, k);
181
+ }
182
+
183
+ /**
184
+ * The E_INVALID_ARGUMENT of a bad option.
185
+ * @param argument - the option
186
+ * @param value - its value
187
+ * @param expected - what it must be
188
+ * @returns the error
189
+ */
190
+ function badArgument(argument: string, value: unknown, expected: string): WebGpuGraphError {
191
+ return new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM}: ${argument} must be ${expected}`, {
192
+ argument,
193
+ value,
194
+ expected,
195
+ });
196
+ }
197
+
198
+ /**
199
+ * The sources a call runs: `sources` as given, else `k` drawn, else every vertex.
200
+ * @param options - the call's options
201
+ * @param n - the vertex count
202
+ * @returns the sources
203
+ */
204
+ function resolveSources(options: BetweennessAcceleratorOptions | undefined, n: number): number[] {
205
+ const k = options?.k;
206
+ if (k !== undefined && (!Number.isInteger(k) || k < 0 || k > n)) {
207
+ throw badArgument("k", k, `an integer in [0, ${n}]`);
208
+ }
209
+ const given = options?.sources;
210
+ if (given !== undefined) {
211
+ for (const v of given) {
212
+ if (!Number.isInteger(v) || v < 0 || v >= n) {
213
+ throw badArgument("sources", v, `node indices in [0, ${n})`);
214
+ }
215
+ }
216
+ if (k !== undefined && k !== given.length) {
217
+ throw badArgument("k", k, `absent or equal to sources.length (${given.length})`);
218
+ }
219
+ return [...given];
220
+ }
221
+ if (k !== undefined) {
222
+ return drawSources(n, k);
223
+ }
224
+ return Array.from({ length: n }, (_, i) => i);
225
+ }
226
+
227
+ /** The device state of a run: the batch arrays sized for the largest batch, the results, the kernels. */
228
+ interface RunState {
229
+ readonly ctx: GpuContext;
230
+ readonly scope: AlgorithmScope;
231
+ readonly n: number;
232
+ readonly S: Binding;
233
+ readonly ends: Binding;
234
+ readonly depthK: Binding;
235
+ readonly sigmaK: Binding;
236
+ readonly deltaK: Binding;
237
+ readonly counters: Binding;
238
+ readonly bc: Binding;
239
+ readonly arcScores: Binding | null;
240
+ readonly rowPtr: Binding;
241
+ readonly colIdx: Binding;
242
+ readonly edgeSrc: Binding | null;
243
+ readonly edgeDst: Binding | null;
244
+ readonly edgeCount: number;
245
+ readonly arcCount: number;
246
+ readonly fill: Kernel;
247
+ readonly finalize: Kernel;
248
+ readonly forward: Kernel;
249
+ readonly forwardEdge: Kernel | null;
250
+ readonly backward: Kernel;
251
+ readonly gather: Kernel;
252
+ readonly edgeGather: Kernel | null;
253
+ /** Each bc kernel bound once per run: its resources never change inside a run, only the params offset does. */
254
+ readonly bound: Map<Kernel, BoundKernel>;
255
+ }
256
+
257
+ /**
258
+ * Records one `fill` of `count` words.
259
+ * @param state - the run
260
+ * @param pass - the pass
261
+ * @param dst - the buffer
262
+ * @param count - the words
263
+ * @param value - the u32 written
264
+ */
265
+ function recordFill(state: RunState, pass: GPUComputePassEncoder, dst: Binding, count: number, value: number): void {
266
+ const { scope, fill, ctx } = state;
267
+ const params = scope.params(FILL_PARAMS, { count, value, mode: 0, pad0: 0 });
268
+ fill.dispatch(pass, fill.bind({ dst, P: params.binding }), plan1d(count, ctx.workgroupSize, ctx.caps), [
269
+ params.offset,
270
+ ]);
271
+ }
272
+
273
+ /**
274
+ * Records a kernel with one BcParams record; the kernel is bound on its first use in the run.
275
+ * @param state - the run
276
+ * @param pass - the pass
277
+ * @param kernel - the kernel
278
+ * @param resources - its storage bindings (the same on every call for one kernel)
279
+ * @param fields - the BcParams fields
280
+ * @param items - the items the grid-stride plan covers
281
+ */
282
+ function recordBc(
283
+ state: RunState,
284
+ pass: GPUComputePassEncoder,
285
+ kernel: Kernel,
286
+ resources: Readonly<Record<string, Binding>>,
287
+ fields: Readonly<Record<string, number>>,
288
+ items: number,
289
+ ): void {
290
+ const { scope, ctx } = state;
291
+ const plan = planGridStride(items, ctx.workgroupSize, ctx.caps);
292
+ const params = scope.params(BC_PARAMS, { ...fields, stride: plan.stride ?? 0 });
293
+ let bound = state.bound.get(kernel);
294
+ if (bound === undefined) {
295
+ bound = kernel.bind({ ...resources, P: params.binding });
296
+ state.bound.set(kernel, bound);
297
+ }
298
+ kernel.dispatch(pass, bound, items === 0 ? plan1d(1, ctx.workgroupSize, ctx.caps) : plan, [params.offset]);
299
+ }
300
+
301
+ /**
302
+ * Submits after flushing the ring and waits for the readback; aborts between submits.
303
+ * @param state - the run
304
+ * @param batch - the batch
305
+ * @param signal - the caller's signal
306
+ * @returns the readback bytes
307
+ */
308
+ async function submit(state: RunState, batch: CommandBatch, signal: AbortSignal | undefined): Promise<ArrayBuffer> {
309
+ state.scope.flush();
310
+ const submitted = batch.submit();
311
+ const back = await submitted.readback;
312
+ state.ctx.assertReady();
313
+ if (signal?.aborted) {
314
+ throw aborted(ALGORITHM, submitted.id);
315
+ }
316
+ return back;
317
+ }
318
+
319
+ /**
320
+ * One source batch: seed, forward levels until an empty one, backward levels, the gathers.
321
+ * @param state - the run
322
+ * @param sources - the batch's sources
323
+ * @param form - the forward body
324
+ * @param levelsPerSubmit - the forward cadence
325
+ * @param tuning - the knobs
326
+ * @param signal - the caller's signal
327
+ * @returns the batch's levels and overflow flag
328
+ */
329
+ async function runBatch(
330
+ state: RunState,
331
+ sources: readonly number[],
332
+ form: ForwardForm,
333
+ levelsPerSubmit: number,
334
+ tuning: BetweennessTuning,
335
+ signal: AbortSignal | undefined,
336
+ ): Promise<{ levels: number; overflow: boolean }> {
337
+ const { ctx, n, S, ends, depthK, sigmaK, deltaK, counters } = state;
338
+ const k = sources.length;
339
+ const words = n * k;
340
+ const seeds = Uint32Array.from(sources, (v, s) => s * n + v);
341
+ ctx.device.queue.writeBuffer(S.buffer, S.offset, seeds);
342
+
343
+ // the forward phase: the first submit also clears the batch's arrays and seeds it
344
+ const forwardFields = { n, k, count: form === "edge" ? state.edgeCount : 0 };
345
+ const forwardItems = form === "edge" ? state.edgeCount : words;
346
+ let recorded = 0;
347
+ let levels = 0;
348
+ let overflow = false;
349
+ let endsWords: U32 | null = null;
350
+ for (let first = true; endsWords === null; first = false) {
351
+ const batch = new CommandBatch(ctx, `${ALGORITHM}/forward`);
352
+ const pass = batch.pass("forward");
353
+ if (first) {
354
+ recordFill(state, pass, depthK, words, 0xffffffff);
355
+ recordFill(state, pass, sigmaK, words, 0);
356
+ recordFill(state, pass, deltaK, words, 0);
357
+ recordBc(state, pass, state.finalize, { counters, ends, S, depthK, sigmaK }, { n, k, role: 1 }, 1);
358
+ }
359
+ for (let level = 0; level < levelsPerSubmit; level++) {
360
+ recordBc(state, pass, state.finalize, { counters, ends, S, depthK, sigmaK }, { n, k, role: 0 }, 1);
361
+ if (form === "edge" && state.forwardEdge !== null && state.edgeSrc !== null && state.edgeDst !== null) {
362
+ recordBc(
363
+ state,
364
+ pass,
365
+ state.forwardEdge,
366
+ { edgeSrc: state.edgeSrc, edgeDst: state.edgeDst, S, ends, counters, depthK, sigmaK },
367
+ forwardFields,
368
+ forwardItems,
369
+ );
370
+ } else {
371
+ recordBc(
372
+ state,
373
+ pass,
374
+ state.forward,
375
+ { rowPtr: state.rowPtr, colIdx: state.colIdx, S, ends, counters, depthK, sigmaK },
376
+ forwardFields,
377
+ forwardItems,
378
+ );
379
+ }
380
+ }
381
+ recorded += levelsPerSubmit;
382
+ batch.endPass();
383
+ const endsCount = Math.min(n + 2, recorded + 1);
384
+ const countersRequest = batch.readback(counters.buffer, counters.offset, FRONTIER_COUNTERS.byteLength);
385
+ const endsRequest = batch.readback(ends.buffer, ends.offset, 4 * endsCount);
386
+ const back = await submit(state, batch, signal);
387
+ const words32 = new Uint32Array(back, countersRequest.offset, FRONTIER_COUNTERS.byteLength / 4);
388
+ if (words32[W.done] !== 0) {
389
+ levels = words32[W.level];
390
+ overflow = words32[W.sigmaOverflow] !== 0;
391
+ endsWords = new Uint32Array(back, endsRequest.offset, endsCount).slice(0, levels + 1);
392
+ } else if (recorded > n + 2) {
393
+ // a batch claims at most n - 1 levels deep, then one level is empty
394
+ throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM}: the done flag never rose in ${recorded} levels`, {
395
+ label: ALGORITHM,
396
+ message: `the done flag never rose in ${recorded} levels`,
397
+ });
398
+ }
399
+ }
400
+
401
+ // the backward phase, planned from ends: depth levels - 1 down to 1, then the gathers
402
+ const backwardLevels: number[] = [];
403
+ for (let level = levels - 1; level >= 1; level--) {
404
+ backwardLevels.push(level);
405
+ }
406
+ let arrays: { depthK: U32; sigmaK: U32; deltaK: F32 } | null = null;
407
+ for (let i = 0; ; i += BC_BACKWARD_LEVELS_PER_SUBMIT) {
408
+ const chunk = backwardLevels.slice(i, i + BC_BACKWARD_LEVELS_PER_SUBMIT);
409
+ const last = i + BC_BACKWARD_LEVELS_PER_SUBMIT >= backwardLevels.length;
410
+ const batch = new CommandBatch(ctx, `${ALGORITHM}/backward`);
411
+ const pass = batch.pass("backward");
412
+ for (const level of chunk) {
413
+ const start = endsWords[level];
414
+ const count = endsWords[level + 1] - start;
415
+ recordBc(
416
+ state,
417
+ pass,
418
+ state.backward,
419
+ { rowPtr: state.rowPtr, colIdx: state.colIdx, S, depthK, sigmaK, deltaK },
420
+ { n, k, start, count },
421
+ count,
422
+ );
423
+ }
424
+ if (last) {
425
+ recordBc(state, pass, state.gather, { deltaK, bc: state.bc }, { n, k }, n);
426
+ if (state.edgeGather !== null && state.arcScores !== null) {
427
+ recordBc(
428
+ state,
429
+ pass,
430
+ state.edgeGather,
431
+ { rowPtr: state.rowPtr, colIdx: state.colIdx, depthK, sigmaK, deltaK, arcScores: state.arcScores },
432
+ { n, k, count: state.arcCount },
433
+ state.arcCount,
434
+ );
435
+ }
436
+ }
437
+ batch.endPass();
438
+ const wantArrays = last && tuning.readArrays === true;
439
+ const requests = wantArrays
440
+ ? [depthK, sigmaK, deltaK].map((b) => batch.readback(b.buffer, b.offset, 4 * words))
441
+ : [];
442
+ const back = await submit(state, batch, signal);
443
+ if (wantArrays) {
444
+ arrays = {
445
+ depthK: new Uint32Array(back, requests[0].offset, words).slice(),
446
+ sigmaK: new Uint32Array(back, requests[1].offset, words).slice(),
447
+ deltaK: new Float32Array(back, requests[2].offset, words).slice(),
448
+ };
449
+ }
450
+ if (last) {
451
+ break;
452
+ }
453
+ }
454
+ tuning.onBatch?.({
455
+ sources,
456
+ forward: form,
457
+ levels,
458
+ ends: endsWords.slice(),
459
+ sigmaOverflow: overflow,
460
+ depthK: arrays?.depthK ?? null,
461
+ sigmaK: arrays?.sigmaK ?? null,
462
+ deltaK: arrays?.deltaK ?? null,
463
+ });
464
+ return { levels, overflow };
465
+ }
466
+
467
+ /**
468
+ * The raw sums of a run over `sources` (see the file comment).
469
+ * @param ctx - the context
470
+ * @param s - the snapshot
471
+ * @param sources - the sources (non-empty, n >= 1)
472
+ * @param withEdges - also accumulate the per-arc scores
473
+ * @param tuning - the knobs
474
+ * @param options - signal / onProgress
475
+ * @returns the raw sums
476
+ */
477
+ async function runRaw(
478
+ ctx: GpuContext,
479
+ s: GraphSnapshot,
480
+ sources: readonly number[],
481
+ withEdges: boolean,
482
+ tuning: BetweennessTuning,
483
+ options: GpuRunOptions | undefined,
484
+ ): Promise<RawBetweenness> {
485
+ const n = s.nodeCount;
486
+ const pinned = tuning.forward ?? "auto";
487
+ const levelsPerSubmit = tuning.levelsPerSubmit ?? MAX_LEVELS_PER_SUBMIT;
488
+ if (!Number.isInteger(levelsPerSubmit) || levelsPerSubmit < 1 || levelsPerSubmit > MAX_LEVELS_PER_SUBMIT) {
489
+ throw badArgument("levelsPerSubmit", levelsPerSubmit, `an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`);
490
+ }
491
+ const limits = tuning.limits ?? ctx.caps.limits;
492
+ const kMax = planBatchSize(n, sources.length, limits);
493
+ const core = ctx.residency.core(s);
494
+ assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM);
495
+ const { edgeCount } = s;
496
+ // the edge-parallel form runs only when pinned, or from the second batch on: a one-batch run never uploads the view
497
+ const mayRunEdge = pinned === "edge" || (pinned === "auto" && sources.length > kMax);
498
+ const edgeView = mayRunEdge && edgeCount > 0 ? ctx.residency.view(s, "edgeList") : null;
499
+ const scope = algorithmScope(ctx, ALGORITHM, RING_SLOTS);
500
+ try {
501
+ const arrayBytes = 4 * n * kMax;
502
+ const lease = (bytes: number, label: string): Binding => bindingOf(scope.scratch(bytes, label), bytes);
503
+ const arcBytes = 4 * Math.max(1, s.arcCount);
504
+ const [fill, finalize, forward, backward, gather] = await Promise.all(
505
+ (["fill", "bc-finalize", "bc-forward", "bc-backward", "bc-gather"] as const).map((id) =>
506
+ ctx.pipelines.kernel(kernelSpec(id)),
507
+ ),
508
+ );
509
+ const forwardEdge =
510
+ edgeView === null
511
+ ? null
512
+ : await ctx.pipelines.kernel(kernelSpec("bc-forward-edge", { UNDIRECTED: !s.directed }));
513
+ const edgeGather = withEdges ? await ctx.pipelines.kernel(kernelSpec("bc-edge-gather")) : null;
514
+ const state: RunState = {
515
+ ctx,
516
+ scope,
517
+ n,
518
+ S: lease(arrayBytes, "S"),
519
+ ends: lease(4 * (n + 2), "ends"),
520
+ depthK: lease(arrayBytes, "depthK"),
521
+ sigmaK: lease(arrayBytes, "sigmaK"),
522
+ deltaK: lease(arrayBytes, "deltaK"),
523
+ counters: lease(FRONTIER_COUNTERS.byteLength, "counters"),
524
+ bc: lease(4 * n, "bc"),
525
+ arcScores: withEdges ? lease(arcBytes, "arc-scores") : null,
526
+ rowPtr: core.rowPtr,
527
+ colIdx: core.colIdx ?? core.rowPtr,
528
+ edgeSrc: edgeView?.bindings.src ?? null,
529
+ edgeDst: edgeView?.bindings.dst ?? null,
530
+ edgeCount,
531
+ arcCount: s.arcCount,
532
+ fill,
533
+ finalize,
534
+ forward,
535
+ forwardEdge,
536
+ backward,
537
+ gather,
538
+ edgeGather,
539
+ bound: new Map(),
540
+ };
541
+ await ctx.allocator.check();
542
+
543
+ // the result accumulators start at 0
544
+ const setup = new CommandBatch(ctx, `${ALGORITHM}/setup`);
545
+ const setupPass = setup.pass("setup");
546
+ recordFill(state, setupPass, state.bc, n, 0);
547
+ if (state.arcScores !== null) {
548
+ recordFill(state, setupPass, state.arcScores, arcBytes / 4, 0);
549
+ }
550
+ setup.endPass();
551
+ await submit(state, setup, options?.signal);
552
+
553
+ let overflow = false;
554
+ let batches = 0;
555
+ let previousLevels = -1;
556
+ for (let start = 0; start < sources.length; ) {
557
+ const k = planBatchSize(n, sources.length - start, limits);
558
+ let form: ForwardForm = pinned === "edge" ? "edge" : "frontier";
559
+ if (pinned === "auto" && previousLevels >= 0) {
560
+ form = previousLevels < BC_EDGE_PARALLEL_GAMMA * Math.log2(n) ? "edge" : "frontier";
561
+ }
562
+ if (state.forwardEdge === null) {
563
+ form = "frontier"; // no edges to run edge-parallel over
564
+ }
565
+ const batch = sources.slice(start, start + k);
566
+ const outcome = await runBatch(state, batch, form, levelsPerSubmit, tuning, options?.signal);
567
+ overflow = overflow || outcome.overflow;
568
+ previousLevels = outcome.levels;
569
+ start += k;
570
+ batches += 1;
571
+ options?.onProgress?.(start, sources.length);
572
+ }
573
+
574
+ const result = new CommandBatch(ctx, `${ALGORITHM}/result`);
575
+ const vertexRequest = result.readback(state.bc.buffer, state.bc.offset, 4 * n);
576
+ const arcRequest =
577
+ state.arcScores === null
578
+ ? null
579
+ : result.readback(state.arcScores.buffer, state.arcScores.offset, 4 * s.arcCount);
580
+ const back = await submit(state, result, options?.signal);
581
+ return {
582
+ vertex: new Float32Array(back, vertexRequest.offset, n).slice(),
583
+ perArc: arcRequest === null ? null : new Float32Array(back, arcRequest.offset, s.arcCount).slice(),
584
+ sourcesUsed: sources.length,
585
+ sigmaOverflow: overflow,
586
+ batches,
587
+ };
588
+ } finally {
589
+ scope.dispose();
590
+ }
591
+ }
592
+
593
+ /**
594
+ * The CPU's normalisation factor: `(n - 1)(n - 2)` directed, half that undirected; 1 when not normalising or when the
595
+ * factor is not positive.
596
+ * @param s - the snapshot
597
+ * @param normalized - the option
598
+ * @returns the divisor
599
+ */
600
+ function normaliser(s: GraphSnapshot, normalized: boolean | undefined): number {
601
+ const n = s.nodeCount;
602
+ const factor = s.directed ? (n - 1) * (n - 2) : ((n - 1) * (n - 2)) / 2;
603
+ return normalized === true && factor > 0 ? factor : 1;
604
+ }
605
+
606
+ /**
607
+ * The checks every entry makes before any device work.
608
+ * @param ctx - the context
609
+ * @param options - the options
610
+ */
611
+ async function precheck(ctx: GpuContext, options: BetweennessAcceleratorOptions | undefined): Promise<void> {
612
+ ctx.assertReady();
613
+ await assertDeviceComputes(ctx);
614
+ if (options?.endpoints === true) {
615
+ throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: endpoints: true is not supported`, {
616
+ feature: "betweenness.endpoints",
617
+ hint: "the CPU endpoints branch in algorithms/src/algorithms/centrality/betweenness.ts (predecessors.length === 0 && w !== source) never fires, so there is no convention to match",
618
+ });
619
+ }
620
+ }
621
+
622
+ /**
623
+ * Vertex betweenness with the test knobs; `betweennessCentrality` is this with an empty tuning.
624
+ * @internal
625
+ * @param ctx - the context whose device runs the kernels
626
+ * @param s - the snapshot
627
+ * @param options - the betweenness options plus dest / signal / onProgress
628
+ * @param tuning - the knobs
629
+ * @returns the scores
630
+ */
631
+ export async function betweennessWithTuning(
632
+ ctx: GpuContext,
633
+ s: GraphSnapshot,
634
+ options: (BetweennessAcceleratorOptions & GpuRunOptions) | undefined,
635
+ tuning: BetweennessTuning,
636
+ ): Promise<GpuBetweennessResult> {
637
+ await precheck(ctx, options);
638
+ const n = s.nodeCount;
639
+ const scores = checkDest(ALGORITHM, options?.dest, n) ?? new Float32Array(n);
640
+ const sources = resolveSources(options, n);
641
+ if (options?.signal?.aborted) {
642
+ throw aborted(ALGORITHM);
643
+ }
644
+ if (n === 0 || sources.length === 0) {
645
+ scores.fill(0);
646
+ return { scores, iterations: 0, converged: true, precision: "f32", sourcesUsed: 0, sigmaOverflow: false };
647
+ }
648
+ const raw = await runRaw(ctx, s, sources, false, tuning, options);
649
+ const divisor = (s.directed ? 1 : 2) * normaliser(s, options?.normalized);
650
+ for (let v = 0; v < n; v++) {
651
+ scores[v] = raw.vertex[v] / divisor;
652
+ }
653
+ return {
654
+ scores,
655
+ iterations: raw.batches,
656
+ converged: true,
657
+ precision: "f32",
658
+ sourcesUsed: raw.sourcesUsed,
659
+ sigmaOverflow: raw.sigmaOverflow,
660
+ };
661
+ }
662
+
663
+ /**
664
+ * Edge betweenness with the test knobs; `edgeBetweennessCentrality` is this with an empty tuning.
665
+ * @internal
666
+ * @param ctx - the context whose device runs the kernels
667
+ * @param s - the snapshot
668
+ * @param options - the betweenness options plus dest (a Float32Array of edgeCount) / signal / onProgress
669
+ * @param tuning - the knobs
670
+ * @param onArcs - called with the per-arc scores before the fold (the tests' pairing check)
671
+ * @returns the per-edge scores
672
+ */
673
+ export async function edgeBetweennessWithTuning(
674
+ ctx: GpuContext,
675
+ s: GraphSnapshot,
676
+ options: (BetweennessAcceleratorOptions & GpuRunOptions) | undefined,
677
+ tuning: BetweennessTuning,
678
+ onArcs?: (perArc: F32) => void,
679
+ ): Promise<GpuEdgeScoresResult> {
680
+ await precheck(ctx, options);
681
+ const n = s.nodeCount;
682
+ const scores = checkDest("edgeBetweennessCentrality", options?.dest, s.edgeCount) ?? new Float32Array(s.edgeCount);
683
+ const sources = resolveSources(options, n);
684
+ if (options?.signal?.aborted) {
685
+ throw aborted(ALGORITHM);
686
+ }
687
+ if (n === 0 || sources.length === 0 || s.arcCount === 0) {
688
+ scores.fill(0);
689
+ return { scores, precision: "f32", sourcesUsed: sources.length, sigmaOverflow: false };
690
+ }
691
+ const raw = await runRaw(ctx, s, sources, true, tuning, options);
692
+ const perArc = raw.perArc ?? new Float32Array(s.arcCount);
693
+ onArcs?.(perArc);
694
+ const folded = foldArcs(s, perArc, "sum");
695
+ const divisor = (s.directed ? 1 : 2) * normaliser(s, options?.normalized);
696
+ for (let e = 0; e < s.edgeCount; e++) {
697
+ scores[e] = folded[e] / divisor;
698
+ }
699
+ return { scores, precision: "f32", sourcesUsed: raw.sourcesUsed, sigmaOverflow: raw.sigmaOverflow };
700
+ }
701
+
702
+ /**
703
+ * Betweenness centrality on the device (spec 3.3 line 811, design 8.4): Brandes' algorithm, k sources per batch.
704
+ * Exact over every vertex by default; `sources` or `k` gives the SAMPLED form, whose scores are the unscaled sum over
705
+ * the sources run (`sourcesUsed` beside them; multiply by `n / sourcesUsed` for the estimator of the full sum). The
706
+ * CPU package's convention: halved on an undirected snapshot, `normalized` divides by `(n - 1)(n - 2)` directed or
707
+ * half that undirected. Weights are ignored (breadth-first on both packages). `endpoints: true` is E_UNSUPPORTED.
708
+ * `sigmaOverflow` is true when some pair has more than 2^32 shortest paths: the scores are then wrong.
709
+ * @param ctx - the context whose device runs the kernels
710
+ * @param s - the snapshot (uploaded through ctx.residency, or found there)
711
+ * @param options - `normalized`, `endpoints`, `sources`, `k`, plus dest (a Float32Array of length n) / signal / onProgress (sources done, sources total)
712
+ * @returns the f32 scores, `iterations` (the source batches run), `converged: true`, `sourcesUsed`, `sigmaOverflow`
713
+ */
714
+ export function betweennessCentrality(
715
+ ctx: GpuContext,
716
+ s: GraphSnapshot,
717
+ options?: BetweennessAcceleratorOptions & GpuRunOptions,
718
+ ): Promise<GpuBetweennessResult> {
719
+ return betweennessWithTuning(ctx, s, options, {});
720
+ }
721
+
722
+ /**
723
+ * Edge betweenness centrality on the device (spec 3.3 line 812, design 8.4): the per-arc terms of the same batches,
724
+ * folded to one score per edge with `foldArcs(s, perArc, "sum")` and halved on an undirected snapshot (the two arcs
725
+ * carry the pairs crossing the edge in each direction, so their sum counts each unordered pair twice; this is the CPU
726
+ * package's number, and it holds for a sample too). `normalized`, sampling, `endpoints` and `sigmaOverflow` as in
727
+ * `betweennessCentrality`.
728
+ * @param ctx - the context whose device runs the kernels
729
+ * @param s - the snapshot (uploaded through ctx.residency, or found there)
730
+ * @param options - as `betweennessCentrality`, dest a Float32Array of length edgeCount
731
+ * @returns the f32 per-edge scores, `sourcesUsed`, `sigmaOverflow`
732
+ */
733
+ export function edgeBetweennessCentrality(
734
+ ctx: GpuContext,
735
+ s: GraphSnapshot,
736
+ options?: BetweennessAcceleratorOptions & GpuRunOptions,
737
+ ): Promise<GpuEdgeScoresResult> {
738
+ return edgeBetweennessWithTuning(ctx, s, options, {});
739
+ }