@graphty/webgpu-graph-algorithms 0.6.24 → 0.6.26
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +1 -1
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-B40Z6lV_.js → context-BZY6SMsM.js} +41 -35
- package/dist/chunks/context-BZY6SMsM.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/algorithms/betweenness.d.ts +24 -9
- package/dist/src/algorithms/betweenness.d.ts.map +1 -1
- package/dist/src/algorithms/betweenness.js +105 -39
- package/dist/src/algorithms/betweenness.js.map +1 -1
- package/dist/src/constants.d.ts +11 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +11 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +4 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernels.d.ts +4 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +43 -13
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/seed.d.ts +3 -1
- package/dist/src/layouts/seed.d.ts.map +1 -1
- package/dist/src/layouts/seed.js +10 -1
- package/dist/src/layouts/seed.js.map +1 -1
- package/dist/src/types/betweenness.d.ts +5 -2
- package/dist/src/types/betweenness.d.ts.map +1 -1
- package/dist/src/wgsl/bc-backward.wgsl.d.ts +5 -2
- package/dist/src/wgsl/bc-backward.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bc-backward.wgsl.js +12 -3
- package/dist/src/wgsl/bc-backward.wgsl.js.map +1 -1
- package/dist/src/wgsl/bc-count.wgsl.d.ts +22 -0
- package/dist/src/wgsl/bc-count.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-count.wgsl.js +46 -0
- package/dist/src/wgsl/bc-count.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts +3 -2
- package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bc-edge-gather.wgsl.js +10 -2
- package/dist/src/wgsl/bc-edge-gather.wgsl.js.map +1 -1
- package/dist/src/wgsl/bc-finalize.wgsl.d.ts +4 -3
- package/dist/src/wgsl/bc-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bc-finalize.wgsl.js +5 -3
- package/dist/src/wgsl/bc-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts +2 -2
- package/dist/src/wgsl/bc-forward-edge.wgsl.js +2 -2
- package/dist/src/wgsl/bc-forward.wgsl.d.ts +4 -2
- package/dist/src/wgsl/bc-forward.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bc-forward.wgsl.js +4 -2
- package/dist/src/wgsl/bc-forward.wgsl.js.map +1 -1
- package/dist/webgpu-graph-algorithms.js +176 -47
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +3 -3
- package/src/algorithms/betweenness.ts +155 -48
- package/src/constants.ts +11 -0
- package/src/kernel/prelude.ts +4 -0
- package/src/kernels.ts +45 -13
- package/src/layouts/seed.ts +10 -1
- package/src/types/betweenness.ts +5 -2
- package/src/wgsl/bc-backward.wgsl.ts +12 -3
- package/src/wgsl/bc-count.wgsl.ts +45 -0
- package/src/wgsl/bc-edge-gather.wgsl.ts +10 -2
- package/src/wgsl/bc-finalize.wgsl.ts +5 -3
- package/src/wgsl/bc-forward-edge.wgsl.ts +2 -2
- package/src/wgsl/bc-forward.wgsl.ts +4 -2
- package/dist/chunks/context-B40Z6lV_.js.map +0 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@graphty/webgpu-graph-algorithms",
|
|
3
|
-
"version": "0.6.
|
|
3
|
+
"version": "0.6.26",
|
|
4
4
|
"description": "WebGPU-accelerated graph algorithms and layouts over the @graphty/graph-format snapshot, for Node (Dawn) and browsers",
|
|
5
5
|
"author": "Adam Powers <apowers@ato.ms>",
|
|
6
6
|
"type": "module",
|
|
@@ -95,8 +95,8 @@
|
|
|
95
95
|
"vite": "^7.0.5",
|
|
96
96
|
"vitest": "4.1.11",
|
|
97
97
|
"webgpu": "0.4.0",
|
|
98
|
-
"@graphty/
|
|
99
|
-
"@graphty/
|
|
98
|
+
"@graphty/layout": "^2.0.9",
|
|
99
|
+
"@graphty/algorithms": "^3.1.8"
|
|
100
100
|
},
|
|
101
101
|
"scripts": {
|
|
102
102
|
"build": "node -e \"require('fs').rmSync('dist',{recursive:true,force:true})\" && tsc -p tsconfig.build.json",
|
|
@@ -2,18 +2,31 @@
|
|
|
2
2
|
* Betweenness and edge betweenness on the device (design 8.4 "Betweenness (A7)", 3.3 lines 811-812, 9.7): Brandes'
|
|
3
3
|
* algorithm as McLaughlin-Bader run it, k sources at a time.
|
|
4
4
|
*
|
|
5
|
-
* A source batch keeps four `n x k` arrays -- `depthK`, `sigmaK` (u32 shortest-path counts
|
|
6
|
-
* dependencies) and the claim log `S` -- plus `ends`, the level boundaries into the log
|
|
5
|
+
* A source batch keeps four `n x k` arrays -- `depthK`, `sigmaK` (u32 shortest-path counts, or the f32 bits of the
|
|
6
|
+
* rescaled ones), `deltaK` (f32 dependencies) and the claim log `S` -- plus `ends`, the level boundaries into the log,
|
|
7
|
+
* and `levelMax`, the f32 bits of the largest rescaled count at each depth. Every array is indexed by
|
|
7
8
|
* `s * n + v`, and a log entry IS that index, so one u32 names a `(vertex, source)` pair: `4 n k` bytes can never
|
|
8
9
|
* reach 2^32 because each array is one storage binding. The forward pass is a tagged breadth-first search: every
|
|
9
10
|
* level is `bc-finalize` (the boundary: `ends[level + 1] = stackTop`, `done` on an empty level) and then ONE forward
|
|
10
11
|
* dispatch, either `bc-forward` (block-mapped over the level's range of the log) or `bc-forward-edge` (every edge of
|
|
11
|
-
* the `edgeList` view for every source), both claiming, counting and appending with the same rules.
|
|
12
|
-
* recorded `MAX_LEVELS_PER_SUBMIT` per submit with one small readback (the counters block and the `ends`
|
|
13
|
-
* the log's `ends` is known on the host when the forward phase ends, so the backward pass is planned
|
|
14
|
-
* `bc-backward` dispatch per level from the deepest to depth 1, each writing `delta[s][w]` once by pulling
|
|
15
|
-
* successors; then `bc-gather` adds each vertex's k dependencies into `bc` (and `bc-edge-gather` each arc's k
|
|
16
|
-
* into `arcScores`). Nothing accumulates a float through an atomic, so the scores are bitwise reproducible.
|
|
12
|
+
* the `edgeList` view for every source), both claiming, counting and appending with the same rules.
|
|
13
|
+
* Levels are recorded `MAX_LEVELS_PER_SUBMIT` per submit with one small readback (the counters block and the `ends`
|
|
14
|
+
* prefix), and the log's `ends` is known on the host when the forward phase ends, so the backward pass is planned
|
|
15
|
+
* exactly: one `bc-backward` dispatch per level from the deepest to depth 1, each writing `delta[s][w]` once by pulling
|
|
16
|
+
* over the successors; then `bc-gather` adds each vertex's k dependencies into `bc` (and `bc-edge-gather` each arc's k
|
|
17
|
+
* terms into `arcScores`). Nothing accumulates a float through an atomic, so the scores are bitwise reproducible.
|
|
18
|
+
*
|
|
19
|
+
* Path counts that outgrow u32 (issue #719): a batch counts exactly in u32 first, and the forward kernels report a
|
|
20
|
+
* wrap. A 19 x 19 grid already has C(36, 18), about 9.1e9, corner-to-corner paths, and f32 counts would only move the
|
|
21
|
+
* cliff to a 68 x 68 grid. So a batch whose counts wrapped is abandoned before its backward pass and run again with
|
|
22
|
+
* every kernel's `SCALED` override: the forward bodies only claim, and after each level `bc-count` pulls the new
|
|
23
|
+
* depth's counts over the in-arcs (the `reverse` view; the forward core when undirected) as f32 divided by the power of
|
|
24
|
+
* two that keeps the previous depth's largest count at 2^BC_SIGMA_EXPONENT_CAP; `bc-backward` and `bc-edge-gather` undo
|
|
25
|
+
* that one step in the ratio Brandes' recursion reads, so the scores are what unbounded counts give. The rest of the
|
|
26
|
+
* run stays scaled. A batch that never wraps runs exactly as before, bit for bit. In the scaled form `sigmaOverflow`
|
|
27
|
+
* rises only when the counts at ONE depth span more than f32's exponent range (about 2^226; measured from a corner of a
|
|
28
|
+
* grid, 235 x 235 raises it and 230 x 230 does not); the `@graphty/algorithms` dispatcher then throws instead of
|
|
29
|
+
* returning the scores.
|
|
17
30
|
*
|
|
18
31
|
* The batch size k is planned from the device limits at 16 bytes per (node, source) -- the three arrays design 4.7
|
|
19
32
|
* counts plus the 4-byte log entry it omits -- as `min(floor(maxStorageBufferBindingSize / 4n), floor(0.25 x
|
|
@@ -43,6 +56,7 @@ import { type F32, foldArcs, type GraphSnapshot, type U32 } from "@graphty/graph
|
|
|
43
56
|
import {
|
|
44
57
|
BC_BACKWARD_LEVELS_PER_SUBMIT,
|
|
45
58
|
BC_BATCH_BUDGET_FRACTION,
|
|
59
|
+
BC_COUNT_MAX_GROUPS,
|
|
46
60
|
BC_EDGE_PARALLEL_GAMMA,
|
|
47
61
|
BC_MAX_BATCH,
|
|
48
62
|
MAX_LEVELS_PER_SUBMIT,
|
|
@@ -53,6 +67,7 @@ import { CommandBatch } from "../kernel/batch.js";
|
|
|
53
67
|
import { plan1d, planGridStride } from "../kernel/dispatch.js";
|
|
54
68
|
import { type BoundKernel, type Kernel } from "../kernel/kernel.js";
|
|
55
69
|
import { BC_PARAMS, FILL_PARAMS, FRONTIER_COUNTERS, kernelSpec } from "../kernels.js";
|
|
70
|
+
import { type CoreBinding } from "../memory/residency.js";
|
|
56
71
|
import { assertWholeCore } from "../primitives/core-shape.js";
|
|
57
72
|
import { W } from "../primitives/frontier.js";
|
|
58
73
|
import { assertDeviceComputes } from "../primitives/verify.js";
|
|
@@ -60,6 +75,7 @@ import { type BetweennessAcceleratorOptions } from "../types/accelerator.js";
|
|
|
60
75
|
import { type GpuBetweennessResult, type GpuEdgeScoresResult } from "../types/betweenness.js";
|
|
61
76
|
import { type Binding } from "../types/memory.js";
|
|
62
77
|
import { type GpuRunOptions } from "../types/run.js";
|
|
78
|
+
import { reverseOf } from "./power-iteration.js";
|
|
63
79
|
import { type AlgorithmScope, algorithmScope } from "./scope.js";
|
|
64
80
|
import { aborted, bindingOf, checkDest } from "./sssp.js";
|
|
65
81
|
|
|
@@ -71,8 +87,8 @@ const BYTES_PER_NODE_SOURCE = 16;
|
|
|
71
87
|
/** The seed of the deterministic draw of `k` sources (any fixed value: the draw only has to repeat). */
|
|
72
88
|
const SAMPLE_SEED = 0x9e3779b9;
|
|
73
89
|
|
|
74
|
-
/** Params slots of the ring: a backward submit's levels plus the fills, the seed
|
|
75
|
-
const RING_SLOTS = BC_BACKWARD_LEVELS_PER_SUBMIT + 16;
|
|
90
|
+
/** Params slots of the ring: a forward submit's records per level (boundary, forward and, when scaled, count) or a backward submit's levels, plus the fills, the seed and the gathers. */
|
|
91
|
+
const RING_SLOTS = Math.max(3 * MAX_LEVELS_PER_SUBMIT, BC_BACKWARD_LEVELS_PER_SUBMIT) + 16;
|
|
76
92
|
|
|
77
93
|
/** The device limits the batch planner reads. */
|
|
78
94
|
interface BatchLimits {
|
|
@@ -99,9 +115,11 @@ export interface BetweennessBatchReport {
|
|
|
99
115
|
readonly levels: number;
|
|
100
116
|
/** `ends[0 .. levels + 1]`: the entries at depth L are `S[ends[L] .. ends[L + 1])`. */
|
|
101
117
|
readonly ends: U32;
|
|
102
|
-
/** Whether
|
|
118
|
+
/** Whether the batch ran with rescaled f32 counts (its u32 counts wrapped on the first try). */
|
|
119
|
+
readonly scaled: boolean;
|
|
120
|
+
/** Whether a rescaled path count left f32's normal range in this batch. */
|
|
103
121
|
readonly sigmaOverflow: boolean;
|
|
104
|
-
/** The `n x k` arrays as the batch left them (null unless the tuning asked for them). */
|
|
122
|
+
/** The `n x k` arrays as the batch left them (null unless the tuning asked for them); `sigmaK` as stored: u32 counts, or the f32 bits of the scaled ones. */
|
|
105
123
|
readonly depthK: U32 | null;
|
|
106
124
|
readonly sigmaK: U32 | null;
|
|
107
125
|
readonly deltaK: F32 | null;
|
|
@@ -231,6 +249,7 @@ interface RunState {
|
|
|
231
249
|
readonly n: number;
|
|
232
250
|
readonly S: Binding;
|
|
233
251
|
readonly ends: Binding;
|
|
252
|
+
readonly levelMax: Binding;
|
|
234
253
|
readonly depthK: Binding;
|
|
235
254
|
readonly sigmaK: Binding;
|
|
236
255
|
readonly deltaK: Binding;
|
|
@@ -244,14 +263,63 @@ interface RunState {
|
|
|
244
263
|
readonly edgeCount: number;
|
|
245
264
|
readonly arcCount: number;
|
|
246
265
|
readonly fill: Kernel;
|
|
266
|
+
readonly gather: Kernel;
|
|
267
|
+
/** The u32 kernels every batch tries first. */
|
|
268
|
+
readonly exact: ModeKernels;
|
|
269
|
+
/** The `SCALED` kernels, compiled when a batch's counts first wrap. */
|
|
270
|
+
scaled: ModeKernels | null;
|
|
271
|
+
/** Each bc kernel bound once per run: its resources never change inside a run, only the params offset does. */
|
|
272
|
+
readonly bound: Map<Kernel, BoundKernel>;
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
/** The kernels of one counting mode; `count` and `reverse` (the in-arcs it pulls over) only when scaled. */
|
|
276
|
+
interface ModeKernels {
|
|
247
277
|
readonly finalize: Kernel;
|
|
248
278
|
readonly forward: Kernel;
|
|
249
279
|
readonly forwardEdge: Kernel | null;
|
|
250
280
|
readonly backward: Kernel;
|
|
251
|
-
readonly gather: Kernel;
|
|
252
281
|
readonly edgeGather: Kernel | null;
|
|
253
|
-
|
|
254
|
-
|
|
282
|
+
readonly count: { readonly kernel: Kernel; readonly rowPtr: Binding; readonly colIdx: Binding } | null;
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
/**
|
|
286
|
+
* Compiles the kernels of one counting mode.
|
|
287
|
+
* @param ctx - the context
|
|
288
|
+
* @param s - the snapshot
|
|
289
|
+
* @param core - its forward core (the in-arcs of an undirected snapshot)
|
|
290
|
+
* @param withForwardEdge - also the edge-parallel forward body
|
|
291
|
+
* @param withEdges - also the per-arc gather
|
|
292
|
+
* @param scaled - the `SCALED` variants plus `bc-count`
|
|
293
|
+
* @returns the kernels
|
|
294
|
+
*/
|
|
295
|
+
async function modeKernels(
|
|
296
|
+
ctx: GpuContext,
|
|
297
|
+
s: GraphSnapshot,
|
|
298
|
+
core: CoreBinding,
|
|
299
|
+
withForwardEdge: boolean,
|
|
300
|
+
withEdges: boolean,
|
|
301
|
+
scaled: boolean,
|
|
302
|
+
): Promise<ModeKernels> {
|
|
303
|
+
const SCALED = scaled;
|
|
304
|
+
const [finalize, forward, backward] = await Promise.all(
|
|
305
|
+
(["bc-finalize", "bc-forward", "bc-backward"] as const).map((id) =>
|
|
306
|
+
ctx.pipelines.kernel(kernelSpec(id, { SCALED })),
|
|
307
|
+
),
|
|
308
|
+
);
|
|
309
|
+
const forwardEdge = withForwardEdge
|
|
310
|
+
? await ctx.pipelines.kernel(kernelSpec("bc-forward-edge", { UNDIRECTED: !s.directed, SCALED }))
|
|
311
|
+
: null;
|
|
312
|
+
const edgeGather = withEdges ? await ctx.pipelines.kernel(kernelSpec("bc-edge-gather", { SCALED })) : null;
|
|
313
|
+
let count: ModeKernels["count"] = null;
|
|
314
|
+
if (scaled) {
|
|
315
|
+
const reverse = s.directed ? reverseOf(ctx, s) : core;
|
|
316
|
+
count = {
|
|
317
|
+
kernel: await ctx.pipelines.kernel(kernelSpec("bc-count")),
|
|
318
|
+
rowPtr: reverse.rowPtr,
|
|
319
|
+
colIdx: reverse.colIdx ?? reverse.rowPtr,
|
|
320
|
+
};
|
|
321
|
+
}
|
|
322
|
+
return { finalize, forward, forwardEdge, backward, edgeGather, count };
|
|
255
323
|
}
|
|
256
324
|
|
|
257
325
|
/**
|
|
@@ -278,6 +346,7 @@ function recordFill(state: RunState, pass: GPUComputePassEncoder, dst: Binding,
|
|
|
278
346
|
* @param resources - its storage bindings (the same on every call for one kernel)
|
|
279
347
|
* @param fields - the BcParams fields
|
|
280
348
|
* @param items - the items the grid-stride plan covers
|
|
349
|
+
* @param maxGroups - the most workgroups the plan launches (default: the planner's cap)
|
|
281
350
|
*/
|
|
282
351
|
function recordBc(
|
|
283
352
|
state: RunState,
|
|
@@ -286,9 +355,10 @@ function recordBc(
|
|
|
286
355
|
resources: Readonly<Record<string, Binding>>,
|
|
287
356
|
fields: Readonly<Record<string, number>>,
|
|
288
357
|
items: number,
|
|
358
|
+
maxGroups?: number,
|
|
289
359
|
): void {
|
|
290
360
|
const { scope, ctx } = state;
|
|
291
|
-
const plan = planGridStride(items, ctx.workgroupSize, ctx.caps);
|
|
361
|
+
const plan = planGridStride(items, ctx.workgroupSize, ctx.caps, maxGroups);
|
|
292
362
|
const params = scope.params(BC_PARAMS, { ...fields, stride: plan.stride ?? 0 });
|
|
293
363
|
let bound = state.bound.get(kernel);
|
|
294
364
|
if (bound === undefined) {
|
|
@@ -317,24 +387,27 @@ async function submit(state: RunState, batch: CommandBatch, signal: AbortSignal
|
|
|
317
387
|
}
|
|
318
388
|
|
|
319
389
|
/**
|
|
320
|
-
* One source batch: seed, forward levels until an empty one, backward levels, the gathers.
|
|
390
|
+
* One source batch: seed, forward levels until an empty one, backward levels, the gathers. In the exact mode a batch
|
|
391
|
+
* whose u32 counts wrapped stops after its forward phase (`wrapped`), before anything reaches the scores.
|
|
321
392
|
* @param state - the run
|
|
393
|
+
* @param mode - the kernels of the counting mode
|
|
322
394
|
* @param sources - the batch's sources
|
|
323
395
|
* @param form - the forward body
|
|
324
396
|
* @param levelsPerSubmit - the forward cadence
|
|
325
397
|
* @param tuning - the knobs
|
|
326
398
|
* @param signal - the caller's signal
|
|
327
|
-
* @returns the batch's levels and overflow flag
|
|
399
|
+
* @returns the batch's levels, whether its exact counts wrapped, and the scaled form's overflow flag
|
|
328
400
|
*/
|
|
329
401
|
async function runBatch(
|
|
330
402
|
state: RunState,
|
|
403
|
+
mode: ModeKernels,
|
|
331
404
|
sources: readonly number[],
|
|
332
405
|
form: ForwardForm,
|
|
333
406
|
levelsPerSubmit: number,
|
|
334
407
|
tuning: BetweennessTuning,
|
|
335
408
|
signal: AbortSignal | undefined,
|
|
336
|
-
): Promise<{ levels: number; overflow: boolean }> {
|
|
337
|
-
const { ctx, n, S, ends, depthK, sigmaK, deltaK, counters } = state;
|
|
409
|
+
): Promise<{ levels: number; wrapped: boolean; overflow: boolean }> {
|
|
410
|
+
const { ctx, n, S, ends, levelMax, depthK, sigmaK, deltaK, counters } = state;
|
|
338
411
|
const k = sources.length;
|
|
339
412
|
const words = n * k;
|
|
340
413
|
const seeds = Uint32Array.from(sources, (v, s) => s * n + v);
|
|
@@ -354,15 +427,18 @@ async function runBatch(
|
|
|
354
427
|
recordFill(state, pass, depthK, words, 0xffffffff);
|
|
355
428
|
recordFill(state, pass, sigmaK, words, 0);
|
|
356
429
|
recordFill(state, pass, deltaK, words, 0);
|
|
357
|
-
|
|
430
|
+
if (mode.count !== null) {
|
|
431
|
+
recordFill(state, pass, levelMax, n + 2, 0);
|
|
432
|
+
}
|
|
433
|
+
recordBc(state, pass, mode.finalize, { counters, ends, S, depthK, sigmaK, levelMax }, { n, k, role: 1 }, 1);
|
|
358
434
|
}
|
|
359
435
|
for (let level = 0; level < levelsPerSubmit; level++) {
|
|
360
|
-
recordBc(state, pass,
|
|
361
|
-
if (form === "edge" &&
|
|
436
|
+
recordBc(state, pass, mode.finalize, { counters, ends, S, depthK, sigmaK, levelMax }, { n, k, role: 0 }, 1);
|
|
437
|
+
if (form === "edge" && mode.forwardEdge !== null && state.edgeSrc !== null && state.edgeDst !== null) {
|
|
362
438
|
recordBc(
|
|
363
439
|
state,
|
|
364
440
|
pass,
|
|
365
|
-
|
|
441
|
+
mode.forwardEdge,
|
|
366
442
|
{ edgeSrc: state.edgeSrc, edgeDst: state.edgeDst, S, ends, counters, depthK, sigmaK },
|
|
367
443
|
forwardFields,
|
|
368
444
|
forwardItems,
|
|
@@ -371,12 +447,24 @@ async function runBatch(
|
|
|
371
447
|
recordBc(
|
|
372
448
|
state,
|
|
373
449
|
pass,
|
|
374
|
-
|
|
450
|
+
mode.forward,
|
|
375
451
|
{ rowPtr: state.rowPtr, colIdx: state.colIdx, S, ends, counters, depthK, sigmaK },
|
|
376
452
|
forwardFields,
|
|
377
453
|
forwardItems,
|
|
378
454
|
);
|
|
379
455
|
}
|
|
456
|
+
if (mode.count !== null) {
|
|
457
|
+
const { kernel, rowPtr, colIdx } = mode.count;
|
|
458
|
+
recordBc(
|
|
459
|
+
state,
|
|
460
|
+
pass,
|
|
461
|
+
kernel,
|
|
462
|
+
{ rowPtr, colIdx, S, ends, counters, depthK, sigmaK, levelMax },
|
|
463
|
+
{ n, k },
|
|
464
|
+
words,
|
|
465
|
+
BC_COUNT_MAX_GROUPS,
|
|
466
|
+
);
|
|
467
|
+
}
|
|
380
468
|
}
|
|
381
469
|
recorded += levelsPerSubmit;
|
|
382
470
|
batch.endPass();
|
|
@@ -389,6 +477,9 @@ async function runBatch(
|
|
|
389
477
|
levels = words32[W.level];
|
|
390
478
|
overflow = words32[W.sigmaOverflow] !== 0;
|
|
391
479
|
endsWords = new Uint32Array(back, endsRequest.offset, endsCount).slice(0, levels + 1);
|
|
480
|
+
if (overflow && mode.count === null) {
|
|
481
|
+
return { levels, wrapped: true, overflow: false };
|
|
482
|
+
}
|
|
392
483
|
} else if (recorded > n + 2) {
|
|
393
484
|
// a batch claims at most n - 1 levels deep, then one level is empty
|
|
394
485
|
throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM}: the done flag never rose in ${recorded} levels`, {
|
|
@@ -415,20 +506,28 @@ async function runBatch(
|
|
|
415
506
|
recordBc(
|
|
416
507
|
state,
|
|
417
508
|
pass,
|
|
418
|
-
|
|
419
|
-
{ rowPtr: state.rowPtr, colIdx: state.colIdx, S, depthK, sigmaK, deltaK },
|
|
509
|
+
mode.backward,
|
|
510
|
+
{ rowPtr: state.rowPtr, colIdx: state.colIdx, S, depthK, sigmaK, deltaK, levelMax },
|
|
420
511
|
{ n, k, start, count },
|
|
421
512
|
count,
|
|
422
513
|
);
|
|
423
514
|
}
|
|
424
515
|
if (last) {
|
|
425
516
|
recordBc(state, pass, state.gather, { deltaK, bc: state.bc }, { n, k }, n);
|
|
426
|
-
if (
|
|
517
|
+
if (mode.edgeGather !== null && state.arcScores !== null) {
|
|
427
518
|
recordBc(
|
|
428
519
|
state,
|
|
429
520
|
pass,
|
|
430
|
-
|
|
431
|
-
{
|
|
521
|
+
mode.edgeGather,
|
|
522
|
+
{
|
|
523
|
+
rowPtr: state.rowPtr,
|
|
524
|
+
colIdx: state.colIdx,
|
|
525
|
+
depthK,
|
|
526
|
+
sigmaK,
|
|
527
|
+
deltaK,
|
|
528
|
+
arcScores: state.arcScores,
|
|
529
|
+
levelMax,
|
|
530
|
+
},
|
|
432
531
|
{ n, k, count: state.arcCount },
|
|
433
532
|
state.arcCount,
|
|
434
533
|
);
|
|
@@ -456,12 +555,13 @@ async function runBatch(
|
|
|
456
555
|
forward: form,
|
|
457
556
|
levels,
|
|
458
557
|
ends: endsWords.slice(),
|
|
558
|
+
scaled: mode.count !== null,
|
|
459
559
|
sigmaOverflow: overflow,
|
|
460
560
|
depthK: arrays?.depthK ?? null,
|
|
461
561
|
sigmaK: arrays?.sigmaK ?? null,
|
|
462
562
|
deltaK: arrays?.deltaK ?? null,
|
|
463
563
|
});
|
|
464
|
-
return { levels, overflow };
|
|
564
|
+
return { levels, wrapped: false, overflow };
|
|
465
565
|
}
|
|
466
566
|
|
|
467
567
|
/**
|
|
@@ -501,22 +601,17 @@ async function runRaw(
|
|
|
501
601
|
const arrayBytes = 4 * n * kMax;
|
|
502
602
|
const lease = (bytes: number, label: string): Binding => bindingOf(scope.scratch(bytes, label), bytes);
|
|
503
603
|
const arcBytes = 4 * Math.max(1, s.arcCount);
|
|
504
|
-
const [fill,
|
|
505
|
-
(["fill", "bc-
|
|
506
|
-
ctx.pipelines.kernel(kernelSpec(id)),
|
|
507
|
-
),
|
|
604
|
+
const [fill, gather] = await Promise.all(
|
|
605
|
+
(["fill", "bc-gather"] as const).map((id) => ctx.pipelines.kernel(kernelSpec(id))),
|
|
508
606
|
);
|
|
509
|
-
const
|
|
510
|
-
edgeView === null
|
|
511
|
-
? null
|
|
512
|
-
: await ctx.pipelines.kernel(kernelSpec("bc-forward-edge", { UNDIRECTED: !s.directed }));
|
|
513
|
-
const edgeGather = withEdges ? await ctx.pipelines.kernel(kernelSpec("bc-edge-gather")) : null;
|
|
607
|
+
const exact = await modeKernels(ctx, s, core, edgeView !== null, withEdges, false);
|
|
514
608
|
const state: RunState = {
|
|
515
609
|
ctx,
|
|
516
610
|
scope,
|
|
517
611
|
n,
|
|
518
612
|
S: lease(arrayBytes, "S"),
|
|
519
613
|
ends: lease(4 * (n + 2), "ends"),
|
|
614
|
+
levelMax: lease(4 * (n + 2), "level-max"),
|
|
520
615
|
depthK: lease(arrayBytes, "depthK"),
|
|
521
616
|
sigmaK: lease(arrayBytes, "sigmaK"),
|
|
522
617
|
deltaK: lease(arrayBytes, "deltaK"),
|
|
@@ -530,12 +625,9 @@ async function runRaw(
|
|
|
530
625
|
edgeCount,
|
|
531
626
|
arcCount: s.arcCount,
|
|
532
627
|
fill,
|
|
533
|
-
finalize,
|
|
534
|
-
forward,
|
|
535
|
-
forwardEdge,
|
|
536
|
-
backward,
|
|
537
628
|
gather,
|
|
538
|
-
|
|
629
|
+
exact,
|
|
630
|
+
scaled: null,
|
|
539
631
|
bound: new Map(),
|
|
540
632
|
};
|
|
541
633
|
await ctx.allocator.check();
|
|
@@ -559,11 +651,24 @@ async function runRaw(
|
|
|
559
651
|
if (pinned === "auto" && previousLevels >= 0) {
|
|
560
652
|
form = previousLevels < BC_EDGE_PARALLEL_GAMMA * Math.log2(n) ? "edge" : "frontier";
|
|
561
653
|
}
|
|
562
|
-
if (state.forwardEdge === null) {
|
|
654
|
+
if (state.exact.forwardEdge === null) {
|
|
563
655
|
form = "frontier"; // no edges to run edge-parallel over
|
|
564
656
|
}
|
|
565
657
|
const batch = sources.slice(start, start + k);
|
|
566
|
-
|
|
658
|
+
let outcome = await runBatch(
|
|
659
|
+
state,
|
|
660
|
+
state.scaled ?? state.exact,
|
|
661
|
+
batch,
|
|
662
|
+
form,
|
|
663
|
+
levelsPerSubmit,
|
|
664
|
+
tuning,
|
|
665
|
+
options?.signal,
|
|
666
|
+
);
|
|
667
|
+
if (outcome.wrapped) {
|
|
668
|
+
// the u32 counts wrapped: this batch, and every later one, counts rescaled f32 instead
|
|
669
|
+
state.scaled ??= await modeKernels(ctx, s, core, edgeView !== null, withEdges, true);
|
|
670
|
+
outcome = await runBatch(state, state.scaled, batch, form, levelsPerSubmit, tuning, options?.signal);
|
|
671
|
+
}
|
|
567
672
|
overflow = overflow || outcome.overflow;
|
|
568
673
|
previousLevels = outcome.levels;
|
|
569
674
|
start += k;
|
|
@@ -705,7 +810,9 @@ export async function edgeBetweennessWithTuning(
|
|
|
705
810
|
* the sources run (`sourcesUsed` beside them; multiply by `n / sourcesUsed` for the estimator of the full sum). The
|
|
706
811
|
* CPU package's convention: halved on an undirected snapshot, `normalized` divides by `(n - 1)(n - 2)` directed or
|
|
707
812
|
* half that undirected. Weights are ignored (breadth-first on both packages). `endpoints: true` is E_UNSUPPORTED.
|
|
708
|
-
*
|
|
813
|
+
* The path counts are f32 rescaled per depth, so a lattice's astronomically many shortest paths are counted exactly
|
|
814
|
+
* enough; `sigmaOverflow` is true only when the counts at one depth spread wider than f32's exponent range (about
|
|
815
|
+
* 2^226): the scores are then wrong, and the `@graphty/algorithms` dispatcher throws instead of returning them.
|
|
709
816
|
* @param ctx - the context whose device runs the kernels
|
|
710
817
|
* @param s - the snapshot (uploaded through ctx.residency, or found there)
|
|
711
818
|
* @param options - `normalized`, `endpoints`, `sources`, `k`, plus dest (a Float32Array of length n) / signal / onProgress (sources done, sources total)
|
package/src/constants.ts
CHANGED
|
@@ -265,6 +265,17 @@ export const BC_BATCH_BUDGET_FRACTION = 0.25;
|
|
|
265
265
|
export const BC_MAX_BATCH = 64;
|
|
266
266
|
/** Design 8.4 (McLaughlin-Bader): a betweenness batch runs the edge-parallel forward pass when the previous batch's level count is below `BC_EDGE_PARALLEL_GAMMA * log2(n)`. The design names the rule and no value; 2 is unmeasured and a benchmark run re-fixes it. */
|
|
267
267
|
export const BC_EDGE_PARALLEL_GAMMA = 2;
|
|
268
|
+
/**
|
|
269
|
+
* Betweenness path counts that wrapped u32 are recounted as f32, rescaled per depth by a power of two (`bc-count`):
|
|
270
|
+
* when the largest count at one depth has a binary exponent above this cap, the next depth's counts are scaled down so
|
|
271
|
+
* that largest count lands at 2^cap. A count then never exceeds 2^cap x 2 x (in-degree), below f32's 2^128 for any
|
|
272
|
+
* in-degree a binding can hold (2^25), and the counts at one depth may span 2^(cap + 126) before the smallest leaves
|
|
273
|
+
* f32's normal range, which is when the overflow flag rises (measured: a corner of a 235 x 235 grid, not of a 230 x 230
|
|
274
|
+
* one). Powers of two scale without rounding, so the scores do not depend on the cap.
|
|
275
|
+
*/
|
|
276
|
+
export const BC_SIGMA_EXPONENT_CAP = 100;
|
|
277
|
+
/** The most workgroups one `bc-count` dispatch launches (it grid-strides over the level): sized for the largest level, a dispatch of n x k lanes costs every level a mostly idle launch, which doubled a 100 x 100 grid's run on an RTX 4070 SUPER; 256 measured 1.3 s against 2.0 s there. */
|
|
278
|
+
export const BC_COUNT_MAX_GROUPS = 256;
|
|
268
279
|
/** Backward-pass levels recorded per submit: each level is one dispatch with its own parameter record, so this bounds the uniform ring. */
|
|
269
280
|
export const BC_BACKWARD_LEVELS_PER_SUBMIT = 64;
|
|
270
281
|
/**
|
package/src/kernel/prelude.ts
CHANGED
|
@@ -13,6 +13,7 @@ import { INVALID_INDEX } from "@graphty/graph-format";
|
|
|
13
13
|
|
|
14
14
|
import {
|
|
15
15
|
APSP_TILE,
|
|
16
|
+
BC_SIGMA_EXPONENT_CAP,
|
|
16
17
|
EXACT_TILES_PER_PASS,
|
|
17
18
|
F32_INF_BITS,
|
|
18
19
|
F32_SIGN_BIT,
|
|
@@ -76,6 +77,7 @@ const RADIX_DIGIT_MASK: u32 = ${RADIX_BINS - 1}u;
|
|
|
76
77
|
const APSP_TILE: u32 = ${APSP_TILE}u;
|
|
77
78
|
const GROUP_HASH_LOAD_FACTOR: u32 = ${GROUP_HASH_LOAD_FACTOR}u;
|
|
78
79
|
const TRIANGLE_BINARY_SEARCH_RATIO: u32 = ${TRIANGLE_BINARY_SEARCH_RATIO}u;
|
|
80
|
+
const BC_SIGMA_EXPONENT_CAP: i32 = ${BC_SIGMA_EXPONENT_CAP}i;
|
|
79
81
|
const F32_MAX: f32 = 0x1.fffffep+127;
|
|
80
82
|
override WG: u32 = ${WORKGROUP_SIZE}u;
|
|
81
83
|
override USE_PERM: bool = false;
|
|
@@ -85,6 +87,8 @@ override SUBGROUP_MAX: u32 = 0u;
|
|
|
85
87
|
|
|
86
88
|
fn linear_id(wid: vec3<u32>, lid: u32) -> u32 { return (wid.x + wid.y * MAX_WORKGROUPS_PER_DIM) * WG + lid; }
|
|
87
89
|
fn group_id(wid: vec3<u32>) -> u32 { return wid.x + wid.y * MAX_WORKGROUPS_PER_DIM; }
|
|
90
|
+
// betweenness: the power of two the counts of depth L + 1 are divided by, from the f32 bits of depth L's largest count
|
|
91
|
+
fn sigma_shift(maxBits: u32) -> i32 { return max(0i, i32((maxBits >> 23u) & 0xffu) - 127i - BC_SIGMA_EXPONENT_CAP); }
|
|
88
92
|
fn lowbias32(x0: u32) -> u32 {
|
|
89
93
|
var x = x0;
|
|
90
94
|
x = x ^ (x >> 16u);
|
package/src/kernels.ts
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
* P8-T5 adds advance-expand; P8-T6 adds bfs-contract and sssp-pred; P8-T7 adds bfs-fused; P8-T8 adds bfs-bottom-up,
|
|
14
14
|
* bfs-bitset-build and bfs-unvisited-flags; P8-T9 adds sssp-relax; P8-T10 adds bf-relax with the BfParams and BfFlags
|
|
15
15
|
* blocks; P8-T11 adds closeness-sweep and closeness-reduce; P9 (betweenness) adds bc-finalize, bc-forward,
|
|
16
|
-
* bc-backward, bc-gather, bc-edge-gather
|
|
16
|
+
* bc-backward, bc-gather, bc-edge-gather, bc-forward-edge and bc-count (issue #719) with the BcParams block; all-pairs shortest paths
|
|
17
17
|
* (design 8.7) adds apsp-init and apsp-fw with the ApspParams block. P11 (the structure and community phase, plan
|
|
18
18
|
* design/webgpu/plans/2026-09-23-webgpu-p11-structure-and-community.md) adds the graph build on the device (coo-emit,
|
|
19
19
|
* run-flags, coo-scatter), the per-row group-by-key (group-by-key-row), label propagation's step (lpa-step) and
|
|
@@ -32,6 +32,7 @@ import { advanceExpandWgsl } from "./wgsl/advance-expand.wgsl.js";
|
|
|
32
32
|
import { apspFwWgsl } from "./wgsl/apsp-fw.wgsl.js";
|
|
33
33
|
import { apspInitWgsl } from "./wgsl/apsp-init.wgsl.js";
|
|
34
34
|
import { bcBackwardWgsl } from "./wgsl/bc-backward.wgsl.js";
|
|
35
|
+
import { bcCountWgsl } from "./wgsl/bc-count.wgsl.js";
|
|
35
36
|
import { bcEdgeGatherWgsl } from "./wgsl/bc-edge-gather.wgsl.js";
|
|
36
37
|
import { bcFinalizeWgsl } from "./wgsl/bc-finalize.wgsl.js";
|
|
37
38
|
import { bcForwardWgsl } from "./wgsl/bc-forward.wgsl.js";
|
|
@@ -144,6 +145,7 @@ export type KernelId =
|
|
|
144
145
|
| "bc-forward"
|
|
145
146
|
| "bc-backward"
|
|
146
147
|
| "bc-gather"
|
|
148
|
+
| "bc-count"
|
|
147
149
|
| "bc-edge-gather"
|
|
148
150
|
| "bc-forward-edge"
|
|
149
151
|
| "apsp-init"
|
|
@@ -433,8 +435,8 @@ export const COMPACT_PARAMS: UniformBlock = UniformBlock.define("CompactParams",
|
|
|
433
435
|
* out-degree sum of the vertices the level claimed, accumulated by `bfs-next-degree` at the end of every level and
|
|
434
436
|
* read, subtracted and zeroed by the next boundary -- Beamer's m_f measured on the frontier the boundary decides
|
|
435
437
|
* for, not on the one it has just expanded); `stackTop` @104 (betweenness: the append cursor of the claim log, which
|
|
436
|
-
* `bc-finalize` also writes into `ends` at every level boundary) and `sigmaOverflow` @108 (betweenness: 1 once a
|
|
437
|
-
* path count
|
|
438
|
+
* `bc-finalize` also writes into `ends` at every level boundary) and `sigmaOverflow` @108 (betweenness: 1 once a
|
|
439
|
+
* rescaled path count left f32's normal range) are APPENDED so every earlier word keeps its byte offset. The words nothing writes before P8-T8 / P8-T9 are declared now because
|
|
438
440
|
* the byte layout is what the single result copy decodes.
|
|
439
441
|
*/
|
|
440
442
|
export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
@@ -1460,7 +1462,7 @@ const CLOSENESS_REDUCE: KernelEntry = {
|
|
|
1460
1462
|
phase: "P8",
|
|
1461
1463
|
};
|
|
1462
1464
|
|
|
1463
|
-
/** `bc-finalize` (design 8.4, 5.4): the one-lane bookkeeping of a betweenness batch -- role 1 seeds it (depth 0 and one path for the k seed entries of the claim log, `stackTop = k`, `level = U32_MAX`), role 0 is the level boundary (`ends[level + 1] = stackTop`, `frontierCount`, `done` on an empty level);
|
|
1465
|
+
/** `bc-finalize` (design 8.4, 5.4): the one-lane bookkeeping of a betweenness batch -- role 1 seeds it (depth 0 and one path for the k seed entries of the claim log, `stackTop = k`, `level = U32_MAX`), role 0 is the level boundary (`ends[level + 1] = stackTop`, `frontierCount`, `done` on an empty level); 6 storage bindings (the counters block as `array<atomic<u32>>`, `ends`, `S` read-only, `depthK`, `sigmaK` and `levelMax` plain: one lane writes the seed; `SCALED` seeds f32 bits). The design's finalize row has 2; `ends` is the third (the level boundary), and the seed's `S`, `depthK`, `sigmaK` and `levelMax` make it six. */
|
|
1464
1466
|
const BC_FINALIZE: KernelEntry = {
|
|
1465
1467
|
id: "bc-finalize",
|
|
1466
1468
|
body: bcFinalizeWgsl,
|
|
@@ -1471,16 +1473,17 @@ const BC_FINALIZE: KernelEntry = {
|
|
|
1471
1473
|
decl(1, 2, "S", "storage-ro", "array<u32>"),
|
|
1472
1474
|
decl(1, 3, "depthK", "storage", "array<u32>"),
|
|
1473
1475
|
decl(1, 4, "sigmaK", "storage", "array<u32>"),
|
|
1476
|
+
decl(1, 5, "levelMax", "storage", "array<u32>"),
|
|
1474
1477
|
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1475
1478
|
],
|
|
1476
|
-
overrideDecls: [],
|
|
1479
|
+
overrideDecls: [{ name: "SCALED", type: "bool", default: false }],
|
|
1477
1480
|
uniforms: [BC_PARAMS],
|
|
1478
1481
|
needs: [],
|
|
1479
1482
|
snippetSlots: [],
|
|
1480
1483
|
phase: "P9",
|
|
1481
1484
|
};
|
|
1482
1485
|
|
|
1483
|
-
/** `bc-forward` (design 8.4, 8.10 "BC forward (tagged)", 16.1): one level of the tagged multi-source BFS -- the block-mapped strip over the level's range of the claim log with the claim, the path count and the overflow report inline, the winners appended to the log; 7 storage bindings (`rowPtr`, `colIdx`, `S` read-write -- the level being read and the appends are ranges of ONE binding --, `ends` read-only, the counters block, `depthK` and `sigmaK` as `array<atomic<u32>>`); the inlined Hillis-Steele scan, so `needs: []`. */
|
|
1486
|
+
/** `bc-forward` (design 8.4, 8.10 "BC forward (tagged)", 16.1): one level of the tagged multi-source BFS -- the block-mapped strip over the level's range of the claim log with the claim, the path count and the overflow report inline (`SCALED`: the claim only, `bc-count` counts), the winners appended to the log; 7 storage bindings (`rowPtr`, `colIdx`, `S` read-write -- the level being read and the appends are ranges of ONE binding --, `ends` read-only, the counters block, `depthK` and `sigmaK` as `array<atomic<u32>>`); the inlined Hillis-Steele scan, so `needs: []`. */
|
|
1484
1487
|
const BC_FORWARD: KernelEntry = {
|
|
1485
1488
|
id: "bc-forward",
|
|
1486
1489
|
body: bcForwardWgsl,
|
|
@@ -1495,14 +1498,14 @@ const BC_FORWARD: KernelEntry = {
|
|
|
1495
1498
|
decl(1, 6, "sigmaK", "storage", "array<atomic<u32>>"),
|
|
1496
1499
|
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1497
1500
|
],
|
|
1498
|
-
overrideDecls: [],
|
|
1501
|
+
overrideDecls: [{ name: "SCALED", type: "bool", default: false }],
|
|
1499
1502
|
uniforms: [BC_PARAMS],
|
|
1500
1503
|
needs: [],
|
|
1501
1504
|
snippetSlots: [],
|
|
1502
1505
|
phase: "P9",
|
|
1503
1506
|
};
|
|
1504
1507
|
|
|
1505
|
-
/** `bc-backward` (design 8.4, 8.10 "BC backward (successor pull)"): one level of the dependency accumulation, one invocation per log entry of a host-planned range, each pulling over its successors and writing its delta once;
|
|
1508
|
+
/** `bc-backward` (design 8.4, 8.10 "BC backward (successor pull)"): one level of the dependency accumulation, one invocation per log entry of a host-planned range, each pulling over its successors and writing its delta once; 7 storage bindings (`rowPtr`, `colIdx`, `S`, `depthK`, `sigmaK` read-only, `deltaK`, `levelMax` read-only: the depth scale of `bc-count`, read under `SCALED`). */
|
|
1506
1509
|
const BC_BACKWARD: KernelEntry = {
|
|
1507
1510
|
id: "bc-backward",
|
|
1508
1511
|
body: bcBackwardWgsl,
|
|
@@ -1514,9 +1517,10 @@ const BC_BACKWARD: KernelEntry = {
|
|
|
1514
1517
|
decl(1, 3, "depthK", "storage-ro", "array<u32>"),
|
|
1515
1518
|
decl(1, 4, "sigmaK", "storage-ro", "array<u32>"),
|
|
1516
1519
|
decl(1, 5, "deltaK", "storage", "array<f32>"),
|
|
1520
|
+
decl(1, 6, "levelMax", "storage-ro", "array<u32>"),
|
|
1517
1521
|
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1518
1522
|
],
|
|
1519
|
-
overrideDecls: [],
|
|
1523
|
+
overrideDecls: [{ name: "SCALED", type: "bool", default: false }],
|
|
1520
1524
|
uniforms: [BC_PARAMS],
|
|
1521
1525
|
needs: [],
|
|
1522
1526
|
snippetSlots: [],
|
|
@@ -1540,7 +1544,7 @@ const BC_GATHER: KernelEntry = {
|
|
|
1540
1544
|
phase: "P9",
|
|
1541
1545
|
};
|
|
1542
1546
|
|
|
1543
|
-
/** `bc-edge-gather` (design 8.4 "edge BC accumulates per arc from the same n x k deltas"): the per-arc twin of `bc-gather`, one invocation per arc (its row found by an upper-bound search over `rowPtr`) adding the arc's term over the batch's sources;
|
|
1547
|
+
/** `bc-edge-gather` (design 8.4 "edge BC accumulates per arc from the same n x k deltas"): the per-arc twin of `bc-gather`, one invocation per arc (its row found by an upper-bound search over `rowPtr`) adding the arc's term over the batch's sources; 7 storage bindings (`rowPtr`, `colIdx`, `depthK`, `sigmaK`, `deltaK` read-only, `arcScores`, `levelMax` read-only). */
|
|
1544
1548
|
const BC_EDGE_GATHER: KernelEntry = {
|
|
1545
1549
|
id: "bc-edge-gather",
|
|
1546
1550
|
body: bcEdgeGatherWgsl,
|
|
@@ -1552,16 +1556,17 @@ const BC_EDGE_GATHER: KernelEntry = {
|
|
|
1552
1556
|
decl(1, 3, "sigmaK", "storage-ro", "array<u32>"),
|
|
1553
1557
|
decl(1, 4, "deltaK", "storage-ro", "array<f32>"),
|
|
1554
1558
|
decl(1, 5, "arcScores", "storage", "array<f32>"),
|
|
1559
|
+
decl(1, 6, "levelMax", "storage-ro", "array<u32>"),
|
|
1555
1560
|
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1556
1561
|
],
|
|
1557
|
-
overrideDecls: [],
|
|
1562
|
+
overrideDecls: [{ name: "SCALED", type: "bool", default: false }],
|
|
1558
1563
|
uniforms: [BC_PARAMS],
|
|
1559
1564
|
needs: [],
|
|
1560
1565
|
snippetSlots: [],
|
|
1561
1566
|
phase: "P9",
|
|
1562
1567
|
};
|
|
1563
1568
|
|
|
1564
|
-
/** `bc-forward-edge` (design 8.4 "the edge-parallel form", 8.8 row 7): one forward level edge-parallel over the `edgeList` view for every source of the batch, with `bc-forward`'s claim, count and overflow report, appending to the same claim log; `UNDIRECTED` relaxes both directions of every edge; 7 storage bindings (`edgeSrc`, `edgeDst`, `S` read-write, `ends` read-only, the counters block, `depthK` and `sigmaK` as `array<atomic<u32>>`). */
|
|
1569
|
+
/** `bc-forward-edge` (design 8.4 "the edge-parallel form", 8.8 row 7): one forward level edge-parallel over the `edgeList` view for every source of the batch, with `bc-forward`'s claim, count and overflow report (`SCALED`: the claim only), appending to the same claim log; `UNDIRECTED` relaxes both directions of every edge; 7 storage bindings (`edgeSrc`, `edgeDst`, `S` read-write, `ends` read-only, the counters block, `depthK` and `sigmaK` as `array<atomic<u32>>`). */
|
|
1565
1570
|
const BC_FORWARD_EDGE: KernelEntry = {
|
|
1566
1571
|
id: "bc-forward-edge",
|
|
1567
1572
|
body: bcForwardEdgeWgsl,
|
|
@@ -1576,7 +1581,33 @@ const BC_FORWARD_EDGE: KernelEntry = {
|
|
|
1576
1581
|
decl(1, 6, "sigmaK", "storage", "array<atomic<u32>>"),
|
|
1577
1582
|
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1578
1583
|
],
|
|
1579
|
-
overrideDecls: [
|
|
1584
|
+
overrideDecls: [
|
|
1585
|
+
{ name: "UNDIRECTED", type: "bool", default: false },
|
|
1586
|
+
{ name: "SCALED", type: "bool", default: false },
|
|
1587
|
+
],
|
|
1588
|
+
uniforms: [BC_PARAMS],
|
|
1589
|
+
needs: [],
|
|
1590
|
+
snippetSlots: [],
|
|
1591
|
+
phase: "P9",
|
|
1592
|
+
};
|
|
1593
|
+
|
|
1594
|
+
/** `bc-count` (design 8.4, issue #719): in a batch rerun because its u32 counts wrapped, the path counts of the depth a forward level just claimed, pulled over the in-arcs in CSR order as f32 bits and rescaled per depth by a power of two, the overflow flag raised when a count leaves f32's normal range; 8 storage bindings (the REVERSE core's `rowPtr` and `colIdx`, `S` and `ends` read-only, the counters block and `levelMax` as `array<atomic<u32>>`, `depthK` read-only, `sigmaK` read-write). */
|
|
1595
|
+
const BC_COUNT: KernelEntry = {
|
|
1596
|
+
id: "bc-count",
|
|
1597
|
+
body: bcCountWgsl,
|
|
1598
|
+
entryPoint: "bc_count",
|
|
1599
|
+
bindings: [
|
|
1600
|
+
decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
|
|
1601
|
+
decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
|
|
1602
|
+
decl(1, 2, "S", "storage-ro", "array<u32>"),
|
|
1603
|
+
decl(1, 3, "ends", "storage-ro", "array<u32>"),
|
|
1604
|
+
decl(1, 4, "counters", "storage", "array<atomic<u32>>"),
|
|
1605
|
+
decl(1, 5, "depthK", "storage-ro", "array<u32>"),
|
|
1606
|
+
decl(1, 6, "sigmaK", "storage", "array<u32>"),
|
|
1607
|
+
decl(1, 7, "levelMax", "storage", "array<atomic<u32>>"),
|
|
1608
|
+
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1609
|
+
],
|
|
1610
|
+
overrideDecls: [],
|
|
1580
1611
|
uniforms: [BC_PARAMS],
|
|
1581
1612
|
needs: [],
|
|
1582
1613
|
snippetSlots: [],
|
|
@@ -1871,6 +1902,7 @@ const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze
|
|
|
1871
1902
|
"bc-forward": BC_FORWARD,
|
|
1872
1903
|
"bc-backward": BC_BACKWARD,
|
|
1873
1904
|
"bc-gather": BC_GATHER,
|
|
1905
|
+
"bc-count": BC_COUNT,
|
|
1874
1906
|
"bc-edge-gather": BC_EDGE_GATHER,
|
|
1875
1907
|
"bc-forward-edge": BC_FORWARD_EDGE,
|
|
1876
1908
|
"apsp-init": APSP_INIT,
|