@graphty/webgpu-graph-algorithms 0.6.13 → 0.6.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +52 -52
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-Bi6AhScG.js → context-oXphO3yj.js} +36 -28
- package/dist/chunks/context-oXphO3yj.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +3 -2
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +48 -4
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/betweenness.d.ts +70 -0
- package/dist/src/algorithms/betweenness.d.ts.map +1 -0
- package/dist/src/algorithms/betweenness.js +538 -0
- package/dist/src/algorithms/betweenness.js.map +1 -0
- package/dist/src/algorithms/closeness.d.ts +15 -5
- package/dist/src/algorithms/closeness.d.ts.map +1 -1
- package/dist/src/algorithms/closeness.js +112 -26
- package/dist/src/algorithms/closeness.js.map +1 -1
- package/dist/src/constants.d.ts +8 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +8 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/index.d.ts +4 -2
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +1 -0
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernels.d.ts +12 -6
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +153 -7
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +2 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -1
- package/dist/src/primitives/frontier.js +2 -0
- package/dist/src/primitives/frontier.js.map +1 -1
- package/dist/src/types/accelerator.d.ts +11 -7
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/algorithms.d.ts +4 -0
- package/dist/src/types/algorithms.d.ts.map +1 -1
- package/dist/src/types/betweenness.d.ts +35 -0
- package/dist/src/types/betweenness.d.ts.map +1 -0
- package/dist/src/types/betweenness.js +7 -0
- package/dist/src/types/betweenness.js.map +1 -0
- package/dist/src/wgsl/bc-backward.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bc-backward.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-backward.wgsl.js +34 -0
- package/dist/src/wgsl/bc-backward.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts +12 -0
- package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-edge-gather.wgsl.js +36 -0
- package/dist/src/wgsl/bc-edge-gather.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-finalize.wgsl.d.ts +21 -0
- package/dist/src/wgsl/bc-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-finalize.wgsl.js +47 -0
- package/dist/src/wgsl/bc-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-forward-edge.wgsl.js +76 -0
- package/dist/src/wgsl/bc-forward-edge.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-forward.wgsl.d.ts +23 -0
- package/dist/src/wgsl/bc-forward.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-forward.wgsl.js +106 -0
- package/dist/src/wgsl/bc-forward.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-gather.wgsl.d.ts +9 -0
- package/dist/src/wgsl/bc-gather.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-gather.wgsl.js +20 -0
- package/dist/src/wgsl/bc-gather.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +4 -1
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/closeness-reduce.wgsl.js +8 -4
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -1
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +4 -2
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/closeness-sweep.wgsl.js +12 -2
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -1
- package/dist/webgpu-graph-algorithms.js +1037 -168
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +5 -5
- package/src/accelerator.ts +75 -7
- package/src/algorithms/betweenness.ts +739 -0
- package/src/algorithms/closeness.ts +124 -32
- package/src/constants.ts +8 -0
- package/src/index.ts +8 -0
- package/src/kernels.ts +169 -10
- package/src/primitives/frontier.ts +4 -0
- package/src/types/accelerator.ts +18 -6
- package/src/types/algorithms.ts +5 -0
- package/src/types/betweenness.ts +38 -0
- package/src/wgsl/bc-backward.wgsl.ts +33 -0
- package/src/wgsl/bc-edge-gather.wgsl.ts +35 -0
- package/src/wgsl/bc-finalize.wgsl.ts +46 -0
- package/src/wgsl/bc-forward-edge.wgsl.ts +75 -0
- package/src/wgsl/bc-forward.wgsl.ts +105 -0
- package/src/wgsl/bc-gather.wgsl.ts +19 -0
- package/src/wgsl/closeness-reduce.wgsl.ts +8 -4
- package/src/wgsl/closeness-sweep.wgsl.ts +12 -2
- package/dist/chunks/context-Bi6AhScG.js.map +0 -1
|
@@ -23,6 +23,11 @@
|
|
|
23
23
|
* seam's placeholder keys and an exact traversal has neither, so a defined value is REFUSED before any device work
|
|
24
24
|
* (`E_UNSUPPORTED { option }`, the package's rule for an option it does not implement, PD-25); `undefined` is legal.
|
|
25
25
|
* `iterations` reports the source batches run (the sources, on the weighted route), `converged` is always true.
|
|
26
|
+
* A SAMPLED run (`sources`, issue #426; undirected snapshots only) seeds its batches from the listed sources (the
|
|
27
|
+
* reduce's role 2 reads the list the host wrote after the per-node sums in `perSource`), and the sweep also adds each
|
|
28
|
+
* claim's distance into a per-node sum (`perNode`), read back with every submit and folded on the host in f64 into
|
|
29
|
+
* `1 / sum` per NODE, where the exact run folds per SOURCE: on an undirected graph the distance from a source to a node
|
|
30
|
+
* is the distance from the node to the source, which is what the CPU port's sampled closeness sums.
|
|
26
31
|
*
|
|
27
32
|
* Cost, stated so nobody is surprised: closeness is O(n x m) on any device -- at 1M nodes it is 31,250 batches of a
|
|
28
33
|
* full multi-source traversal, minutes on the card, and no target in design 10.4 asks for less. `compact.record`
|
|
@@ -54,8 +59,8 @@ import { assertWholeCore } from "../primitives/core-shape.js";
|
|
|
54
59
|
import { W } from "../primitives/frontier.js";
|
|
55
60
|
import { type ReduceScope } from "../primitives/reduce.js";
|
|
56
61
|
import { assertDeviceComputes } from "../primitives/verify.js";
|
|
57
|
-
import { type HitsOptionsLike } from "../types/accelerator.js";
|
|
58
|
-
import { type
|
|
62
|
+
import { type ClosenessAcceleratorOptions, type HitsOptionsLike } from "../types/accelerator.js";
|
|
63
|
+
import { type GpuClosenessResult } from "../types/algorithms.js";
|
|
59
64
|
import { type Binding } from "../types/memory.js";
|
|
60
65
|
import { type GpuRunOptions } from "../types/run.js";
|
|
61
66
|
import { algorithmScope } from "./scope.js";
|
|
@@ -116,10 +121,12 @@ function reusingScratch(scope: ReduceScope): ReduceScope {
|
|
|
116
121
|
}
|
|
117
122
|
|
|
118
123
|
/**
|
|
119
|
-
* The weighted route: one `sssp` per source, the sums reduced on the host.
|
|
124
|
+
* The weighted route: one `sssp` per source, the sums reduced on the host. With `sources` (a sampled run on an
|
|
125
|
+
* undirected snapshot) each search adds its distances into the sums of the nodes it reaches instead of its own.
|
|
120
126
|
* @param ctx - the context
|
|
121
127
|
* @param s - the snapshot
|
|
122
128
|
* @param scores - the destination
|
|
129
|
+
* @param sources - a sampled run's sources, or null for every node
|
|
123
130
|
* @param options - the run options
|
|
124
131
|
* @returns the result
|
|
125
132
|
*/
|
|
@@ -127,25 +134,40 @@ async function weightedRoute(
|
|
|
127
134
|
ctx: GpuContext,
|
|
128
135
|
s: GraphSnapshot,
|
|
129
136
|
scores: F32,
|
|
137
|
+
sources: readonly number[] | null,
|
|
130
138
|
options: GpuRunOptions | undefined,
|
|
131
|
-
): Promise<
|
|
139
|
+
): Promise<GpuClosenessResult> {
|
|
132
140
|
const n = s.nodeCount;
|
|
133
|
-
|
|
141
|
+
const count = sources?.length ?? n;
|
|
142
|
+
const totals = sources === null ? null : new Float64Array(n);
|
|
143
|
+
for (let i = 0; i < count; i++) {
|
|
134
144
|
if (options?.signal?.aborted) {
|
|
135
145
|
throw aborted(ALGORITHM);
|
|
136
146
|
}
|
|
147
|
+
const source = sources === null ? i : sources[i];
|
|
137
148
|
const { dist } = await sssp(ctx, s, source, { signal: options?.signal });
|
|
138
149
|
let sum = 0;
|
|
139
150
|
for (let v = 0; v < n; v++) {
|
|
140
151
|
const d = dist[v];
|
|
141
152
|
if (v !== source && d !== Infinity) {
|
|
142
|
-
|
|
153
|
+
if (totals === null) {
|
|
154
|
+
sum += d;
|
|
155
|
+
} else {
|
|
156
|
+
totals[v] += d;
|
|
157
|
+
}
|
|
143
158
|
}
|
|
144
159
|
}
|
|
145
|
-
|
|
146
|
-
|
|
160
|
+
if (totals === null) {
|
|
161
|
+
scores[source] = sum === 0 ? 0 : 1 / sum;
|
|
162
|
+
}
|
|
163
|
+
options?.onProgress?.(i + 1, count);
|
|
164
|
+
}
|
|
165
|
+
if (totals !== null) {
|
|
166
|
+
totals.forEach((sum, v) => {
|
|
167
|
+
scores[v] = sum === 0 ? 0 : 1 / sum;
|
|
168
|
+
});
|
|
147
169
|
}
|
|
148
|
-
return { scores, iterations:
|
|
170
|
+
return { scores, iterations: count, converged: true, precision: "f32", sourcesUsed: count };
|
|
149
171
|
}
|
|
150
172
|
|
|
151
173
|
/**
|
|
@@ -153,6 +175,7 @@ async function weightedRoute(
|
|
|
153
175
|
* @param ctx - the context
|
|
154
176
|
* @param s - the snapshot
|
|
155
177
|
* @param scores - the destination
|
|
178
|
+
* @param sources - a sampled run's sources, or null for every node
|
|
156
179
|
* @param levelsPerSubmit - the submit cadence
|
|
157
180
|
* @param options - the run options
|
|
158
181
|
* @param tuning - the knobs
|
|
@@ -162,13 +185,15 @@ async function sweepRoute(
|
|
|
162
185
|
ctx: GpuContext,
|
|
163
186
|
s: GraphSnapshot,
|
|
164
187
|
scores: F32,
|
|
188
|
+
sources: readonly number[] | null,
|
|
165
189
|
levelsPerSubmit: number,
|
|
166
190
|
options: GpuRunOptions | undefined,
|
|
167
191
|
tuning: ClosenessTuning,
|
|
168
|
-
): Promise<
|
|
192
|
+
): Promise<GpuClosenessResult> {
|
|
169
193
|
const n = s.nodeCount;
|
|
170
|
-
|
|
171
|
-
|
|
194
|
+
const seedCount = sources?.length ?? n;
|
|
195
|
+
if (seedCount === 0) {
|
|
196
|
+
return { scores, iterations: 0, converged: true, precision: "f32", sourcesUsed: 0 };
|
|
172
197
|
}
|
|
173
198
|
const core = ctx.residency.core(s);
|
|
174
199
|
assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM);
|
|
@@ -195,7 +220,18 @@ async function sweepRoute(
|
|
|
195
220
|
FRONTIER_COUNTERS.byteLength,
|
|
196
221
|
);
|
|
197
222
|
const perSourceBytes = 4 * PER_SOURCE_WORDS;
|
|
198
|
-
|
|
223
|
+
// a sampled run appends the per-node distance sums (bitsBase words) and then its source list
|
|
224
|
+
const zeroedWords = PER_SOURCE_WORDS + (sources === null ? 0 : bitsBase);
|
|
225
|
+
const perSourceAll = 4 * (zeroedWords + (sources === null ? 0 : sources.length));
|
|
226
|
+
const perSource = bindingOf(scope.scratch(perSourceAll, "per-source"), perSourceAll);
|
|
227
|
+
if (sources !== null) {
|
|
228
|
+
ctx.device.queue.writeBuffer(
|
|
229
|
+
perSource.buffer,
|
|
230
|
+
perSource.offset + 4 * zeroedWords,
|
|
231
|
+
Uint32Array.from(sources),
|
|
232
|
+
);
|
|
233
|
+
}
|
|
234
|
+
const totals = sources === null ? null : new Float64Array(n);
|
|
199
235
|
await ctx.allocator.check();
|
|
200
236
|
const compact = await prepareCompact(reusingScratch(scope));
|
|
201
237
|
const sweep = await ctx.pipelines.kernel(kernelSpec("closeness-sweep", graphOverrides(core, null)));
|
|
@@ -222,7 +258,7 @@ async function sweepRoute(
|
|
|
222
258
|
ctx.assertReady();
|
|
223
259
|
|
|
224
260
|
let batches = 0;
|
|
225
|
-
for (let batchStart = 0; batchStart <
|
|
261
|
+
for (let batchStart = 0; batchStart < seedCount; batchStart += SOURCES_PER_BATCH) {
|
|
226
262
|
let level = 0;
|
|
227
263
|
for (let first = true; ; first = false) {
|
|
228
264
|
const batch = new CommandBatch(ctx, `${ALGORITHM}/levels`);
|
|
@@ -231,8 +267,13 @@ async function sweepRoute(
|
|
|
231
267
|
// the batch's seed: the four regions and the block zeroed, then role 1 (the sources' bits, their
|
|
232
268
|
// flags, counters[0] = k, level = U32_MAX)
|
|
233
269
|
recordFill(pass, bits, 4 * bitsBase, 0);
|
|
234
|
-
recordFill(pass, perSource,
|
|
235
|
-
const seed = scope.params(FRONTIER_PARAMS, {
|
|
270
|
+
recordFill(pass, perSource, zeroedWords, 0);
|
|
271
|
+
const seed = scope.params(FRONTIER_PARAMS, {
|
|
272
|
+
role: sources === null ? 1 : 2,
|
|
273
|
+
n: seedCount,
|
|
274
|
+
bitsBase,
|
|
275
|
+
source: batchStart,
|
|
276
|
+
});
|
|
236
277
|
reduce.dispatch(pass, reduce.bind({ counters, perSource, bits, P: seed.binding }), onePlan, [
|
|
237
278
|
seed.offset,
|
|
238
279
|
]);
|
|
@@ -253,6 +294,7 @@ async function sweepRoute(
|
|
|
253
294
|
arcEnd: s.arcCount,
|
|
254
295
|
mode,
|
|
255
296
|
stride: sweepPlan.stride ?? wg,
|
|
297
|
+
perNode: sources === null ? 0 : 1,
|
|
256
298
|
});
|
|
257
299
|
return {
|
|
258
300
|
bound: sweep.bind({ ...graph, frontierList, counters, bits, perSource, P: params.binding }),
|
|
@@ -277,6 +319,8 @@ async function sweepRoute(
|
|
|
277
319
|
batch.endPass();
|
|
278
320
|
const doneRequest = batch.readback(counters.buffer, counters.offset + 4 * W.done, 4);
|
|
279
321
|
const blockRequest = batch.readback(perSource.buffer, perSource.offset, perSourceBytes);
|
|
322
|
+
const nodeRequest =
|
|
323
|
+
totals === null ? null : batch.readback(perSource.buffer, perSource.offset + perSourceBytes, 4 * n);
|
|
280
324
|
const submitted = submit(batch);
|
|
281
325
|
const back = await submitted.readback;
|
|
282
326
|
ctx.assertReady();
|
|
@@ -285,10 +329,18 @@ async function sweepRoute(
|
|
|
285
329
|
}
|
|
286
330
|
if (new Uint32Array(back, doneRequest.offset, 1)[0] !== 0) {
|
|
287
331
|
const block = new Uint32Array(back, blockRequest.offset, PER_SOURCE_WORDS);
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
332
|
+
if (totals === null || nodeRequest === null) {
|
|
333
|
+
const count = Math.min(SOURCES_PER_BATCH, n - batchStart);
|
|
334
|
+
for (let i = 0; i < count; i++) {
|
|
335
|
+
const sum = block[3 * SOURCES_PER_BATCH + i] * 2 ** 32 + block[2 * SOURCES_PER_BATCH + i];
|
|
336
|
+
scores[batchStart + i] = sum === 0 ? 0 : 1 / sum;
|
|
337
|
+
}
|
|
338
|
+
} else {
|
|
339
|
+
// at most 32 (n - 1) per node per batch, so a u32 word never wraps below 134M nodes
|
|
340
|
+
const sums = new Uint32Array(back, nodeRequest.offset, n);
|
|
341
|
+
for (let v = 0; v < n; v++) {
|
|
342
|
+
totals[v] += sums[v];
|
|
343
|
+
}
|
|
292
344
|
}
|
|
293
345
|
tuning.onBatch?.(batchStart, block.slice());
|
|
294
346
|
break;
|
|
@@ -303,29 +355,63 @@ async function sweepRoute(
|
|
|
303
355
|
}
|
|
304
356
|
}
|
|
305
357
|
batches += 1;
|
|
306
|
-
options?.onProgress?.(Math.min(batchStart + SOURCES_PER_BATCH,
|
|
358
|
+
options?.onProgress?.(Math.min(batchStart + SOURCES_PER_BATCH, seedCount), seedCount);
|
|
307
359
|
}
|
|
308
|
-
|
|
360
|
+
totals?.forEach((sum, v) => {
|
|
361
|
+
scores[v] = sum === 0 ? 0 : 1 / sum;
|
|
362
|
+
});
|
|
363
|
+
return { scores, iterations: batches, converged: true, precision: "f32", sourcesUsed: seedCount };
|
|
309
364
|
} finally {
|
|
310
365
|
scope.dispose();
|
|
311
366
|
}
|
|
312
367
|
}
|
|
313
368
|
|
|
369
|
+
/**
|
|
370
|
+
* A sampled run's sources, checked: node indices of `s`, on an undirected snapshot only.
|
|
371
|
+
* @param s - the snapshot
|
|
372
|
+
* @param sources - the caller's list, or undefined for every node
|
|
373
|
+
* @returns the list, or null for every node
|
|
374
|
+
* @throws WebGpuGraphError E_INVALID_ARGUMENT for an index outside the snapshot, E_UNSUPPORTED on a directed snapshot
|
|
375
|
+
*/
|
|
376
|
+
function checkSources(s: GraphSnapshot, sources: readonly number[] | undefined): readonly number[] | null {
|
|
377
|
+
if (sources === undefined) {
|
|
378
|
+
return null;
|
|
379
|
+
}
|
|
380
|
+
if (s.directed) {
|
|
381
|
+
// a search FROM a source measures distance to the nodes it reaches, which is the distance FROM them to the
|
|
382
|
+
// source only when every edge runs both ways
|
|
383
|
+
throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: sampled sources need an undirected snapshot`, {
|
|
384
|
+
feature: "closenessCentrality.directedSources",
|
|
385
|
+
hint: "run the CPU port, which searches the in-arcs",
|
|
386
|
+
});
|
|
387
|
+
}
|
|
388
|
+
for (const v of sources) {
|
|
389
|
+
if (!Number.isInteger(v) || v < 0 || v >= s.nodeCount) {
|
|
390
|
+
throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM}: a source is not a node index`, {
|
|
391
|
+
argument: "sources",
|
|
392
|
+
value: v,
|
|
393
|
+
expected: `an integer in [0, ${s.nodeCount})`,
|
|
394
|
+
});
|
|
395
|
+
}
|
|
396
|
+
}
|
|
397
|
+
return sources;
|
|
398
|
+
}
|
|
399
|
+
|
|
314
400
|
/**
|
|
315
401
|
* Closeness with the test knobs of PD-26's shape; `closenessCentrality` is this with an empty tuning.
|
|
316
402
|
* @internal
|
|
317
403
|
* @param ctx - the context whose device runs the kernels
|
|
318
404
|
* @param s - the snapshot (uploaded through ctx.residency, or found there)
|
|
319
|
-
* @param options -
|
|
405
|
+
* @param options - `weighted` and a sampled run's `sources` honoured, the placeholder `maxIterations` / `tolerance` refused when defined, plus dest / signal / onProgress
|
|
320
406
|
* @param tuning - the knobs
|
|
321
|
-
* @returns the scores, the batches run, `converged: true
|
|
407
|
+
* @returns the scores, the batches run, `converged: true`, `precision: "f32"` and `sourcesUsed`
|
|
322
408
|
*/
|
|
323
409
|
export async function closenessWithTuning(
|
|
324
410
|
ctx: GpuContext,
|
|
325
411
|
s: GraphSnapshot,
|
|
326
|
-
options: (HitsOptionsLike & GpuRunOptions) | undefined,
|
|
412
|
+
options: (ClosenessAcceleratorOptions & HitsOptionsLike & GpuRunOptions) | undefined,
|
|
327
413
|
tuning: ClosenessTuning,
|
|
328
|
-
): Promise<
|
|
414
|
+
): Promise<GpuClosenessResult> {
|
|
329
415
|
ctx.assertReady();
|
|
330
416
|
await assertDeviceComputes(ctx);
|
|
331
417
|
for (const key of ["maxIterations", "tolerance"] as const) {
|
|
@@ -337,6 +423,7 @@ export async function closenessWithTuning(
|
|
|
337
423
|
}
|
|
338
424
|
}
|
|
339
425
|
const n = s.nodeCount;
|
|
426
|
+
const sources = checkSources(s, options?.sources);
|
|
340
427
|
const levelsPerSubmit = tuning.levelsPerSubmit ?? MAX_LEVELS_PER_SUBMIT;
|
|
341
428
|
if (!Number.isInteger(levelsPerSubmit) || levelsPerSubmit < 1 || levelsPerSubmit > MAX_LEVELS_PER_SUBMIT) {
|
|
342
429
|
throw new WebGpuGraphError(
|
|
@@ -370,9 +457,9 @@ export async function closenessWithTuning(
|
|
|
370
457
|
feature: "closenessCentrality.nonFiniteWeights",
|
|
371
458
|
});
|
|
372
459
|
}
|
|
373
|
-
return weightedRoute(ctx, s, scores, options);
|
|
460
|
+
return weightedRoute(ctx, s, scores, sources, options);
|
|
374
461
|
}
|
|
375
|
-
return sweepRoute(ctx, s, scores, levelsPerSubmit, options, tuning);
|
|
462
|
+
return sweepRoute(ctx, s, scores, sources, levelsPerSubmit, options, tuning);
|
|
376
463
|
}
|
|
377
464
|
|
|
378
465
|
/**
|
|
@@ -381,15 +468,20 @@ export async function closenessWithTuning(
|
|
|
381
468
|
* `closenessCentrality`, unweighted by one bit-parallel multi-source search per 32 sources, weighted by one `sssp`
|
|
382
469
|
* per source; `weighted` defaults to the snapshot's flag, `maxIterations` / `tolerance` are refused when defined
|
|
383
470
|
* (PD-25). `iterations` is the source batches run and `converged` is always true.
|
|
471
|
+
*
|
|
472
|
+
* SAMPLED (`sources`, node indices, duplicates run twice; undirected snapshots only, E_UNSUPPORTED
|
|
473
|
+
* `closenessCentrality.directedSources` otherwise): the batches seed the listed sources instead of every node, and
|
|
474
|
+
* each node's score is `1 / sum` of its distances to the sources that reach it (itself excluded), `0` when none does:
|
|
475
|
+
* the sampled score of the CPU port, unscaled. `sourcesUsed` is the list's length (`n` exact).
|
|
384
476
|
* @param ctx - the context whose device runs the kernels
|
|
385
477
|
* @param s - the snapshot (uploaded through ctx.residency, or found there)
|
|
386
|
-
* @param options -
|
|
387
|
-
* @returns the scores, the batches run, `converged: true
|
|
478
|
+
* @param options - `weighted`, `sources`, plus dest (a Float32Array of length n for `scores`) / signal / onProgress
|
|
479
|
+
* @returns the scores, the batches run, `converged: true`, `precision: "f32"` and `sourcesUsed`
|
|
388
480
|
*/
|
|
389
481
|
export function closenessCentrality(
|
|
390
482
|
ctx: GpuContext,
|
|
391
483
|
s: GraphSnapshot,
|
|
392
|
-
options?: HitsOptionsLike & GpuRunOptions,
|
|
393
|
-
): Promise<
|
|
484
|
+
options?: ClosenessAcceleratorOptions & HitsOptionsLike & GpuRunOptions,
|
|
485
|
+
): Promise<GpuClosenessResult> {
|
|
394
486
|
return closenessWithTuning(ctx, s, options, {});
|
|
395
487
|
}
|
package/src/constants.ts
CHANGED
|
@@ -259,3 +259,11 @@ export const BEAMER_BETA = 24;
|
|
|
259
259
|
export const SSSP_DELTA_FACTOR = 32;
|
|
260
260
|
/** The bit pattern of +Infinity, the unreached sentinel of `dist` (P8 PD-9); interpolated into the prelude as `F32_INF_BITS` so no body types the literal. */
|
|
261
261
|
export const F32_INF_BITS = 0x7f800000;
|
|
262
|
+
/** Design 8.4 "k planned from maxBufferSize and a 25% budget": the share of `maxBufferSize` one betweenness source batch may hold. WebGPU exposes no device memory size, so this is a fraction of the largest buffer, not a memory measurement. */
|
|
263
|
+
export const BC_BATCH_BUDGET_FRACTION = 0.25;
|
|
264
|
+
/** Design 10.1's betweenness column: the most sources one betweenness batch runs together. */
|
|
265
|
+
export const BC_MAX_BATCH = 64;
|
|
266
|
+
/** Design 8.4 (McLaughlin-Bader): a betweenness batch runs the edge-parallel forward pass when the previous batch's level count is below `BC_EDGE_PARALLEL_GAMMA * log2(n)`. The design names the rule and no value; 2 is unmeasured and a benchmark run re-fixes it. */
|
|
267
|
+
export const BC_EDGE_PARALLEL_GAMMA = 2;
|
|
268
|
+
/** Backward-pass levels recorded per submit: each level is one dispatch with its own parameter record, so this bounds the uniform ring. */
|
|
269
|
+
export const BC_BACKWARD_LEVELS_PER_SUBMIT = 64;
|
package/src/index.ts
CHANGED
|
@@ -57,6 +57,7 @@ export { eigenvectorCentrality, hits, katzCentrality } from "./algorithms/spectr
|
|
|
57
57
|
|
|
58
58
|
// ==================== algorithms (P8: the frontier family, spec 3.3 lines 807-810, 8.4; the seam's option types in)
|
|
59
59
|
export { bellmanFord } from "./algorithms/bellman-ford.js";
|
|
60
|
+
export { betweennessCentrality, edgeBetweennessCentrality } from "./algorithms/betweenness.js";
|
|
60
61
|
export { breadthFirstSearch } from "./algorithms/bfs.js";
|
|
61
62
|
export { closenessCentrality } from "./algorithms/closeness.js";
|
|
62
63
|
export { sssp } from "./algorithms/sssp.js";
|
|
@@ -79,6 +80,8 @@ export type {
|
|
|
79
80
|
BetweennessAcceleratorOptions,
|
|
80
81
|
BfsOptions,
|
|
81
82
|
BfsResultLike,
|
|
83
|
+
ClosenessAcceleratorOptions,
|
|
84
|
+
ClosenessResultLike,
|
|
82
85
|
CommunityResultLike,
|
|
83
86
|
CorenessResultLike,
|
|
84
87
|
EdgeScoresResultLike,
|
|
@@ -99,10 +102,15 @@ export type {
|
|
|
99
102
|
// BfsOptions / SsspOptions / HitsOptionsLike above (P8 PD-19)
|
|
100
103
|
export type { GpuBellmanFordResult, GpuBfsResult, GpuSsspResult } from "./types/traversal.js";
|
|
101
104
|
|
|
105
|
+
// ==================== types: the betweenness results (spec 3.3 lines 833-834); the option type is the seam's
|
|
106
|
+
// BetweennessAcceleratorOptions above
|
|
107
|
+
export type { GpuBetweennessResult, GpuEdgeScoresResult } from "./types/betweenness.js";
|
|
108
|
+
|
|
102
109
|
// ==================== types: the P7 algorithm results and option records (spec 3.3 lines 815-828, 9.7)
|
|
103
110
|
export type {
|
|
104
111
|
ComponentsOptions,
|
|
105
112
|
EigenvectorOptions,
|
|
113
|
+
GpuClosenessResult,
|
|
106
114
|
GpuHitsResult,
|
|
107
115
|
GpuLabelResult,
|
|
108
116
|
GpuPageRankResult,
|
package/src/kernels.ts
CHANGED
|
@@ -12,7 +12,8 @@
|
|
|
12
12
|
* dedupe-claim and dedupe-filter; P8-T4 adds frontier-finalize with the FrontierCounters and FrontierParams blocks;
|
|
13
13
|
* P8-T5 adds advance-expand; P8-T6 adds bfs-contract and sssp-pred; P8-T7 adds bfs-fused; P8-T8 adds bfs-bottom-up,
|
|
14
14
|
* bfs-bitset-build and bfs-unvisited-flags; P8-T9 adds sssp-relax; P8-T10 adds bf-relax with the BfParams and BfFlags
|
|
15
|
-
* blocks; P8-T11 adds closeness-sweep and closeness-reduce
|
|
15
|
+
* blocks; P8-T11 adds closeness-sweep and closeness-reduce; P9 (betweenness) adds bc-finalize, bc-forward,
|
|
16
|
+
* bc-backward, bc-gather, bc-edge-gather and bc-forward-edge with the BcParams block. This file is the only importer of src/wgsl/** (spec 3.2;
|
|
16
17
|
* test/layers.test.ts).
|
|
17
18
|
*/
|
|
18
19
|
|
|
@@ -23,6 +24,12 @@ import { type BindingDecl, type OverrideDecl, type WgslModuleSpec } from "./kern
|
|
|
23
24
|
import { type CoreBinding } from "./memory/residency.js";
|
|
24
25
|
import { type Binding } from "./types/memory.js";
|
|
25
26
|
import { advanceExpandWgsl } from "./wgsl/advance-expand.wgsl.js";
|
|
27
|
+
import { bcBackwardWgsl } from "./wgsl/bc-backward.wgsl.js";
|
|
28
|
+
import { bcEdgeGatherWgsl } from "./wgsl/bc-edge-gather.wgsl.js";
|
|
29
|
+
import { bcFinalizeWgsl } from "./wgsl/bc-finalize.wgsl.js";
|
|
30
|
+
import { bcForwardWgsl } from "./wgsl/bc-forward.wgsl.js";
|
|
31
|
+
import { bcForwardEdgeWgsl } from "./wgsl/bc-forward-edge.wgsl.js";
|
|
32
|
+
import { bcGatherWgsl } from "./wgsl/bc-gather.wgsl.js";
|
|
26
33
|
import { bfRelaxWgsl } from "./wgsl/bf-relax.wgsl.js";
|
|
27
34
|
import { bfsBitsetBuildWgsl } from "./wgsl/bfs-bitset-build.wgsl.js";
|
|
28
35
|
import { bfsBottomUpWgsl } from "./wgsl/bfs-bottom-up.wgsl.js";
|
|
@@ -69,7 +76,7 @@ import { wccLinkEdgesWgsl } from "./wgsl/wcc-link-edges.wgsl.js";
|
|
|
69
76
|
import { wccLinkSampleWgsl } from "./wgsl/wcc-link-sample.wgsl.js";
|
|
70
77
|
import { wccSampleWgsl } from "./wgsl/wcc-sample.wgsl.js";
|
|
71
78
|
|
|
72
|
-
/** Every module id of P1-P4, P7 and
|
|
79
|
+
/** Every module id of P1-P4, P7, P8 and P9 (later ids are appended, never renamed). */
|
|
73
80
|
export type KernelId =
|
|
74
81
|
| "degree"
|
|
75
82
|
| "reduce"
|
|
@@ -116,7 +123,13 @@ export type KernelId =
|
|
|
116
123
|
| "sssp-relax"
|
|
117
124
|
| "bf-relax"
|
|
118
125
|
| "closeness-sweep"
|
|
119
|
-
| "closeness-reduce"
|
|
126
|
+
| "closeness-reduce"
|
|
127
|
+
| "bc-finalize"
|
|
128
|
+
| "bc-forward"
|
|
129
|
+
| "bc-backward"
|
|
130
|
+
| "bc-gather"
|
|
131
|
+
| "bc-edge-gather"
|
|
132
|
+
| "bc-forward-edge";
|
|
120
133
|
|
|
121
134
|
/** One registry entry: everything of a WgslModuleSpec except the per-variant overrides and snippets. */
|
|
122
135
|
export interface KernelEntry {
|
|
@@ -131,7 +144,7 @@ export interface KernelEntry {
|
|
|
131
144
|
/** The snippet marker names the body carries (segmented-reduce: ["VALUE"]). */
|
|
132
145
|
readonly snippetSlots: readonly string[];
|
|
133
146
|
/** The phase the entry landed in (documentation and the compile-matrix filter). */
|
|
134
|
-
readonly phase: "P1" | "P2" | "P3" | "P4" | "P7" | "P8";
|
|
147
|
+
readonly phase: "P1" | "P2" | "P3" | "P4" | "P7" | "P8" | "P9";
|
|
135
148
|
}
|
|
136
149
|
|
|
137
150
|
// ---- the generated blocks (spec 5.3; contract 3.10.2): field order = byte order, offsets in the JSDoc
|
|
@@ -392,7 +405,9 @@ export const COMPACT_PARAMS: UniformBlock = UniformBlock.define("CompactParams",
|
|
|
392
405
|
* one; every level kernel is a direct dispatch that reads it first -- G8-F5), `nextDegreeSum` @100 (issue #391: the
|
|
393
406
|
* out-degree sum of the vertices the level claimed, accumulated by `bfs-next-degree` at the end of every level and
|
|
394
407
|
* read, subtracted and zeroed by the next boundary -- Beamer's m_f measured on the frontier the boundary decides
|
|
395
|
-
* for, not on the one it has just expanded)
|
|
408
|
+
* for, not on the one it has just expanded); `stackTop` @104 (betweenness: the append cursor of the claim log, which
|
|
409
|
+
* `bc-finalize` also writes into `ends` at every level boundary) and `sigmaOverflow` @108 (betweenness: 1 once a u32
|
|
410
|
+
* path count wrapped) are APPENDED so every earlier word keeps its byte offset. The words nothing writes before P8-T8 / P8-T9 are declared now because
|
|
396
411
|
* the byte layout is what the single result copy decodes.
|
|
397
412
|
*/
|
|
398
413
|
export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
@@ -424,6 +439,8 @@ export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
|
424
439
|
["deltaBits", "u32"],
|
|
425
440
|
["path", "u32"],
|
|
426
441
|
["nextDegreeSum", "u32"],
|
|
442
|
+
["stackTop", "u32"],
|
|
443
|
+
["sigmaOverflow", "u32"],
|
|
427
444
|
],
|
|
428
445
|
{ layout: "storage" },
|
|
429
446
|
);
|
|
@@ -436,7 +453,8 @@ export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
|
436
453
|
* `arcEnd` @44 (the bound arc window), `predKind` @48 (0 arc, 1 node), `bitsBase` @52, `source` @56, `stride` @60
|
|
437
454
|
* (a grid-stride plan's stride), `firstOfSubmit` @64 (the boundary's index inside its submit, clamped to 1: both
|
|
438
455
|
* the unvisited-count and the unvisited-degree-sum subtraction run at >= 1, issue #391), `iteration` @68 (an
|
|
439
|
-
* `sssp-pred` hop pass, P8-T9), `
|
|
456
|
+
* `sssp-pred` hop pass, P8-T9), `perNode` @72 (`closeness-sweep`: 1 also folds every claim into the per-node
|
|
457
|
+
* distance sums of a sampled run), `pad2` @76. The `slotBase` field that once addressed the selector's indirect slots went with
|
|
440
458
|
* the slots (2026-09-25); `pad2` keeps the block an explicit 80 bytes, the way every block here is padded.
|
|
441
459
|
*/
|
|
442
460
|
export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams", [
|
|
@@ -458,10 +476,22 @@ export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams
|
|
|
458
476
|
["stride", "u32"],
|
|
459
477
|
["firstOfSubmit", "u32"],
|
|
460
478
|
["iteration", "u32"],
|
|
461
|
-
["
|
|
479
|
+
["perNode", "u32"],
|
|
462
480
|
["pad2", "u32"],
|
|
463
481
|
]);
|
|
464
482
|
|
|
483
|
+
/** `BcParams` (uniform, 32 B; betweenness): `n` @0, `k` @4 (the batch's sources), `start` @8 (the first log index of a backward level), `count` @12 (a backward level's entries, or the edge count of `bc-forward-edge`), `stride` @16 (a grid-stride plan's stride), `role` @20 (`bc-finalize`: 0 the level boundary, 1 the seed), `pad0` @24, `pad1` @28. */
|
|
484
|
+
export const BC_PARAMS: UniformBlock = UniformBlock.define("BcParams", [
|
|
485
|
+
["n", "u32"],
|
|
486
|
+
["k", "u32"],
|
|
487
|
+
["start", "u32"],
|
|
488
|
+
["count", "u32"],
|
|
489
|
+
["stride", "u32"],
|
|
490
|
+
["role", "u32"],
|
|
491
|
+
["pad0", "u32"],
|
|
492
|
+
["pad1", "u32"],
|
|
493
|
+
]);
|
|
494
|
+
|
|
465
495
|
/** `BfParams` (uniform, 16 B; P8-T10): `edgeCount` @0 (the logical edges of the `edgeList` view), `stride` @4 (the grid-stride plan's stride), `maxRetries` @8 (PD-12's compare-exchange bound), `cutoffBits` @12 (the f32 bit pattern of the CPU port's `cutoff`, `+Inf` when absent). */
|
|
466
496
|
export const BF_PARAMS: UniformBlock = UniformBlock.define("BfParams", [
|
|
467
497
|
["edgeCount", "u32"],
|
|
@@ -1345,7 +1375,7 @@ const CLOSENESS_SWEEP: KernelEntry = {
|
|
|
1345
1375
|
phase: "P8",
|
|
1346
1376
|
};
|
|
1347
1377
|
|
|
1348
|
-
/** `closeness-reduce` (design 8.4, 9.7; P8-T11, PD-13): the one-lane bookkeeping of the sweep -- role 0 the level boundary (`done` from the previous level's compacted count, `newCount` folded into `reached` and the 64-bit `sum` at `level + 1` with the 16-bit split product and the carry, `level` advanced), role 1 the seed of a batch (the sources' bits into `visited` and the level-0 frontier region, their flags, `counters[0] = k`, `level = U32_MAX`); 3 storage bindings (`counters` and `perSource` as `array<atomic<u32>>`, `bits` plain: one lane writes the seed). */
|
|
1378
|
+
/** `closeness-reduce` (design 8.4, 9.7; P8-T11, PD-13): the one-lane bookkeeping of the sweep -- role 0 the level boundary (`done` from the previous level's compacted count, `newCount` folded into `reached` and the 64-bit `sum` at `level + 1` with the 16-bit split product and the carry, `level` advanced), role 1 the seed of a batch (the sources' bits into `visited` and the level-0 frontier region, their flags, `counters[0] = k`, `level = U32_MAX`), role 2 the same seed from a sampled run's source list; 3 storage bindings (`counters` and `perSource` as `array<atomic<u32>>`, `bits` plain: one lane writes the seed). */
|
|
1349
1379
|
const CLOSENESS_REDUCE: KernelEntry = {
|
|
1350
1380
|
id: "closeness-reduce",
|
|
1351
1381
|
body: closenessReduceWgsl,
|
|
@@ -1363,6 +1393,129 @@ const CLOSENESS_REDUCE: KernelEntry = {
|
|
|
1363
1393
|
phase: "P8",
|
|
1364
1394
|
};
|
|
1365
1395
|
|
|
1396
|
+
/** `bc-finalize` (design 8.4, 5.4): the one-lane bookkeeping of a betweenness batch -- role 1 seeds it (depth 0 and one path for the k seed entries of the claim log, `stackTop = k`, `level = U32_MAX`), role 0 is the level boundary (`ends[level + 1] = stackTop`, `frontierCount`, `done` on an empty level); 5 storage bindings (the counters block as `array<atomic<u32>>`, `ends`, `S` read-only, `depthK` and `sigmaK` plain: one lane writes the seed). The design's finalize row has 2; `ends` is the third (the level boundary), and the seed's `S`, `depthK` and `sigmaK` make it five. */
|
|
1397
|
+
const BC_FINALIZE: KernelEntry = {
|
|
1398
|
+
id: "bc-finalize",
|
|
1399
|
+
body: bcFinalizeWgsl,
|
|
1400
|
+
entryPoint: "bc_finalize",
|
|
1401
|
+
bindings: [
|
|
1402
|
+
decl(1, 0, "counters", "storage", "array<atomic<u32>>"),
|
|
1403
|
+
decl(1, 1, "ends", "storage", "array<u32>"),
|
|
1404
|
+
decl(1, 2, "S", "storage-ro", "array<u32>"),
|
|
1405
|
+
decl(1, 3, "depthK", "storage", "array<u32>"),
|
|
1406
|
+
decl(1, 4, "sigmaK", "storage", "array<u32>"),
|
|
1407
|
+
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1408
|
+
],
|
|
1409
|
+
overrideDecls: [],
|
|
1410
|
+
uniforms: [BC_PARAMS],
|
|
1411
|
+
needs: [],
|
|
1412
|
+
snippetSlots: [],
|
|
1413
|
+
phase: "P9",
|
|
1414
|
+
};
|
|
1415
|
+
|
|
1416
|
+
/** `bc-forward` (design 8.4, 8.10 "BC forward (tagged)", 16.1): one level of the tagged multi-source BFS -- the block-mapped strip over the level's range of the claim log with the claim, the path count and the overflow report inline, the winners appended to the log; 7 storage bindings (`rowPtr`, `colIdx`, `S` read-write -- the level being read and the appends are ranges of ONE binding --, `ends` read-only, the counters block, `depthK` and `sigmaK` as `array<atomic<u32>>`); the inlined Hillis-Steele scan, so `needs: []`. */
|
|
1417
|
+
const BC_FORWARD: KernelEntry = {
|
|
1418
|
+
id: "bc-forward",
|
|
1419
|
+
body: bcForwardWgsl,
|
|
1420
|
+
entryPoint: "bc_forward",
|
|
1421
|
+
bindings: [
|
|
1422
|
+
decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
|
|
1423
|
+
decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
|
|
1424
|
+
decl(1, 2, "S", "storage", "array<u32>"),
|
|
1425
|
+
decl(1, 3, "ends", "storage-ro", "array<u32>"),
|
|
1426
|
+
decl(1, 4, "counters", "storage", "array<atomic<u32>>"),
|
|
1427
|
+
decl(1, 5, "depthK", "storage", "array<atomic<u32>>"),
|
|
1428
|
+
decl(1, 6, "sigmaK", "storage", "array<atomic<u32>>"),
|
|
1429
|
+
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1430
|
+
],
|
|
1431
|
+
overrideDecls: [],
|
|
1432
|
+
uniforms: [BC_PARAMS],
|
|
1433
|
+
needs: [],
|
|
1434
|
+
snippetSlots: [],
|
|
1435
|
+
phase: "P9",
|
|
1436
|
+
};
|
|
1437
|
+
|
|
1438
|
+
/** `bc-backward` (design 8.4, 8.10 "BC backward (successor pull)"): one level of the dependency accumulation, one invocation per log entry of a host-planned range, each pulling over its successors and writing its delta once; 6 storage bindings (`rowPtr`, `colIdx`, `S`, `depthK`, `sigmaK` read-only, `deltaK`). */
|
|
1439
|
+
const BC_BACKWARD: KernelEntry = {
|
|
1440
|
+
id: "bc-backward",
|
|
1441
|
+
body: bcBackwardWgsl,
|
|
1442
|
+
entryPoint: "bc_backward",
|
|
1443
|
+
bindings: [
|
|
1444
|
+
decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
|
|
1445
|
+
decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
|
|
1446
|
+
decl(1, 2, "S", "storage-ro", "array<u32>"),
|
|
1447
|
+
decl(1, 3, "depthK", "storage-ro", "array<u32>"),
|
|
1448
|
+
decl(1, 4, "sigmaK", "storage-ro", "array<u32>"),
|
|
1449
|
+
decl(1, 5, "deltaK", "storage", "array<f32>"),
|
|
1450
|
+
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1451
|
+
],
|
|
1452
|
+
overrideDecls: [],
|
|
1453
|
+
uniforms: [BC_PARAMS],
|
|
1454
|
+
needs: [],
|
|
1455
|
+
snippetSlots: [],
|
|
1456
|
+
phase: "P9",
|
|
1457
|
+
};
|
|
1458
|
+
|
|
1459
|
+
/** `bc-gather` (design 8.4, 8.10 "BC gather"): `bc[w] += sum over s of delta[s][w]`, one invocation per vertex, no atomic; 2 storage bindings (`deltaK` read-only, `bc`). */
|
|
1460
|
+
const BC_GATHER: KernelEntry = {
|
|
1461
|
+
id: "bc-gather",
|
|
1462
|
+
body: bcGatherWgsl,
|
|
1463
|
+
entryPoint: "bc_gather",
|
|
1464
|
+
bindings: [
|
|
1465
|
+
decl(1, 0, "deltaK", "storage-ro", "array<f32>"),
|
|
1466
|
+
decl(1, 1, "bc", "storage", "array<f32>"),
|
|
1467
|
+
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1468
|
+
],
|
|
1469
|
+
overrideDecls: [],
|
|
1470
|
+
uniforms: [BC_PARAMS],
|
|
1471
|
+
needs: [],
|
|
1472
|
+
snippetSlots: [],
|
|
1473
|
+
phase: "P9",
|
|
1474
|
+
};
|
|
1475
|
+
|
|
1476
|
+
/** `bc-edge-gather` (design 8.4 "edge BC accumulates per arc from the same n x k deltas"): the per-arc twin of `bc-gather`, one invocation per arc (its row found by an upper-bound search over `rowPtr`) adding the arc's term over the batch's sources; 6 storage bindings (`rowPtr`, `colIdx`, `depthK`, `sigmaK`, `deltaK` read-only, `arcScores`). */
|
|
1477
|
+
const BC_EDGE_GATHER: KernelEntry = {
|
|
1478
|
+
id: "bc-edge-gather",
|
|
1479
|
+
body: bcEdgeGatherWgsl,
|
|
1480
|
+
entryPoint: "bc_edge_gather",
|
|
1481
|
+
bindings: [
|
|
1482
|
+
decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
|
|
1483
|
+
decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
|
|
1484
|
+
decl(1, 2, "depthK", "storage-ro", "array<u32>"),
|
|
1485
|
+
decl(1, 3, "sigmaK", "storage-ro", "array<u32>"),
|
|
1486
|
+
decl(1, 4, "deltaK", "storage-ro", "array<f32>"),
|
|
1487
|
+
decl(1, 5, "arcScores", "storage", "array<f32>"),
|
|
1488
|
+
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1489
|
+
],
|
|
1490
|
+
overrideDecls: [],
|
|
1491
|
+
uniforms: [BC_PARAMS],
|
|
1492
|
+
needs: [],
|
|
1493
|
+
snippetSlots: [],
|
|
1494
|
+
phase: "P9",
|
|
1495
|
+
};
|
|
1496
|
+
|
|
1497
|
+
/** `bc-forward-edge` (design 8.4 "the edge-parallel form", 8.8 row 7): one forward level edge-parallel over the `edgeList` view for every source of the batch, with `bc-forward`'s claim, count and overflow report, appending to the same claim log; `UNDIRECTED` relaxes both directions of every edge; 7 storage bindings (`edgeSrc`, `edgeDst`, `S` read-write, `ends` read-only, the counters block, `depthK` and `sigmaK` as `array<atomic<u32>>`). */
|
|
1498
|
+
const BC_FORWARD_EDGE: KernelEntry = {
|
|
1499
|
+
id: "bc-forward-edge",
|
|
1500
|
+
body: bcForwardEdgeWgsl,
|
|
1501
|
+
entryPoint: "bc_forward_edge",
|
|
1502
|
+
bindings: [
|
|
1503
|
+
decl(1, 0, "edgeSrc", "storage-ro", "array<u32>"),
|
|
1504
|
+
decl(1, 1, "edgeDst", "storage-ro", "array<u32>"),
|
|
1505
|
+
decl(1, 2, "S", "storage", "array<u32>"),
|
|
1506
|
+
decl(1, 3, "ends", "storage-ro", "array<u32>"),
|
|
1507
|
+
decl(1, 4, "counters", "storage", "array<atomic<u32>>"),
|
|
1508
|
+
decl(1, 5, "depthK", "storage", "array<atomic<u32>>"),
|
|
1509
|
+
decl(1, 6, "sigmaK", "storage", "array<atomic<u32>>"),
|
|
1510
|
+
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1511
|
+
],
|
|
1512
|
+
overrideDecls: [{ name: "UNDIRECTED", type: "bool", default: false }],
|
|
1513
|
+
uniforms: [BC_PARAMS],
|
|
1514
|
+
needs: [],
|
|
1515
|
+
snippetSlots: [],
|
|
1516
|
+
phase: "P9",
|
|
1517
|
+
};
|
|
1518
|
+
|
|
1366
1519
|
/**
|
|
1367
1520
|
* The entries by id, in dispatch order. PLAN DECISION: `KernelId` is declared in full (contract 3.10) while the
|
|
1368
1521
|
* entries landed phase by phase, so the table is built as a Partial record and exported below through the
|
|
@@ -1372,8 +1525,8 @@ const CLOSENESS_REDUCE: KernelEntry = {
|
|
|
1372
1525
|
* and `"fa2-to-scene"`; M8b-T3 landed the seven P7 entries and P4 its thirteen; P8-T3 landed the three compact /
|
|
1373
1526
|
* dedupe entries, P8-T4 `"frontier-finalize"`, P8-T5 `"advance-expand"`, P8-T6 `"bfs-contract"` and `"sssp-pred"` and
|
|
1374
1527
|
* P8-T7 `"bfs-fused"`, P8-T8 `"bfs-bottom-up"`, `"bfs-bitset-build"` and `"bfs-unvisited-flags"`, P8-T9
|
|
1375
|
-
* `"sssp-relax"`, P8-T10 `"bf-relax"` and P8-T11 `"closeness-sweep"` and `"closeness-reduce"`,
|
|
1376
|
-
* `KernelId` is present and the assertion is exact.
|
|
1528
|
+
* `"sssp-relax"`, P8-T10 `"bf-relax"` and P8-T11 `"closeness-sweep"` and `"closeness-reduce"`, and betweenness the
|
|
1529
|
+
* six `"bc-*"` entries, so every member of `KernelId` is present and the assertion is exact.
|
|
1377
1530
|
*/
|
|
1378
1531
|
const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze({
|
|
1379
1532
|
degree: DEGREE,
|
|
@@ -1422,6 +1575,12 @@ const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze
|
|
|
1422
1575
|
"bf-relax": BF_RELAX,
|
|
1423
1576
|
"closeness-sweep": CLOSENESS_SWEEP,
|
|
1424
1577
|
"closeness-reduce": CLOSENESS_REDUCE,
|
|
1578
|
+
"bc-finalize": BC_FINALIZE,
|
|
1579
|
+
"bc-forward": BC_FORWARD,
|
|
1580
|
+
"bc-backward": BC_BACKWARD,
|
|
1581
|
+
"bc-gather": BC_GATHER,
|
|
1582
|
+
"bc-edge-gather": BC_EDGE_GATHER,
|
|
1583
|
+
"bc-forward-edge": BC_FORWARD_EDGE,
|
|
1425
1584
|
});
|
|
1426
1585
|
|
|
1427
1586
|
/** THE registry (spec 3.5): every entry, keyed by id. */
|
|
@@ -72,6 +72,8 @@ export const W: Readonly<{
|
|
|
72
72
|
deltaBits: 23;
|
|
73
73
|
path: 24;
|
|
74
74
|
nextDegreeSum: 25;
|
|
75
|
+
stackTop: 26;
|
|
76
|
+
sigmaOverflow: 27;
|
|
75
77
|
}> = Object.freeze({
|
|
76
78
|
frontierCount: 0,
|
|
77
79
|
nextFrontierCount: 1,
|
|
@@ -99,6 +101,8 @@ export const W: Readonly<{
|
|
|
99
101
|
deltaBits: 23,
|
|
100
102
|
path: 24,
|
|
101
103
|
nextDegreeSum: 25,
|
|
104
|
+
stackTop: 26,
|
|
105
|
+
sigmaOverflow: 27,
|
|
102
106
|
});
|
|
103
107
|
|
|
104
108
|
/** The words a `reset` seeds (every other word is zeroed). */
|