@graphty/webgpu-graph-algorithms 0.6.2 → 0.6.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +62 -32
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-BXqgCifx.js → context-Dvq-Cc6v.js} +71 -25
- package/dist/chunks/context-Dvq-Cc6v.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +8 -6
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +57 -6
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/bellman-ford.d.ts +60 -0
- package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
- package/dist/src/algorithms/bellman-ford.js +301 -0
- package/dist/src/algorithms/bellman-ford.js.map +1 -0
- package/dist/src/algorithms/bfs.d.ts +67 -0
- package/dist/src/algorithms/bfs.d.ts.map +1 -0
- package/dist/src/algorithms/bfs.js +534 -0
- package/dist/src/algorithms/bfs.js.map +1 -0
- package/dist/src/algorithms/closeness.d.ts +53 -0
- package/dist/src/algorithms/closeness.d.ts.map +1 -0
- package/dist/src/algorithms/closeness.js +323 -0
- package/dist/src/algorithms/closeness.js.map +1 -0
- package/dist/src/algorithms/scope.d.ts +5 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +3 -0
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +71 -0
- package/dist/src/algorithms/sssp.d.ts.map +1 -0
- package/dist/src/algorithms/sssp.js +585 -0
- package/dist/src/algorithms/sssp.js.map +1 -0
- package/dist/src/constants.d.ts +12 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +12 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/index.d.ts +8 -2
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +7 -1
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts +4 -4
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +39 -5
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernel/uniform-ring.d.ts +8 -0
- package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
- package/dist/src/kernel/uniform-ring.js +13 -0
- package/dist/src/kernel/uniform-ring.js.map +1 -1
- package/dist/src/kernels.d.ts +44 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +371 -3
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +62 -0
- package/dist/src/primitives/advance.d.ts.map +1 -0
- package/dist/src/primitives/advance.js +95 -0
- package/dist/src/primitives/advance.js.map +1 -0
- package/dist/src/primitives/compact.d.ts +89 -0
- package/dist/src/primitives/compact.d.ts.map +1 -0
- package/dist/src/primitives/compact.js +233 -0
- package/dist/src/primitives/compact.js.map +1 -0
- package/dist/src/primitives/core-shape.d.ts +22 -1
- package/dist/src/primitives/core-shape.d.ts.map +1 -1
- package/dist/src/primitives/core-shape.js +33 -3
- package/dist/src/primitives/core-shape.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +156 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -0
- package/dist/src/primitives/frontier.js +259 -0
- package/dist/src/primitives/frontier.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +16 -7
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/traversal.d.ts +53 -0
- package/dist/src/types/traversal.d.ts.map +1 -0
- package/dist/src/types/traversal.js +10 -0
- package/dist/src/types/traversal.js.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
- package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +59 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js +210 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
- package/dist/webgpu-graph-algorithms.js +3207 -377
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +5 -4
- package/src/accelerator.ts +65 -7
- package/src/algorithms/bellman-ford.ts +387 -0
- package/src/algorithms/bfs.ts +626 -0
- package/src/algorithms/closeness.ts +395 -0
- package/src/algorithms/scope.ts +13 -3
- package/src/algorithms/sssp.ts +767 -0
- package/src/constants.ts +12 -0
- package/src/index.ts +14 -1
- package/src/kernel/prelude.ts +39 -4
- package/src/kernel/uniform-ring.ts +14 -0
- package/src/kernels.ts +450 -6
- package/src/primitives/advance.ts +130 -0
- package/src/primitives/compact.ts +323 -0
- package/src/primitives/core-shape.ts +41 -3
- package/src/primitives/frontier.ts +388 -0
- package/src/types/accelerator.ts +18 -5
- package/src/types/traversal.ts +56 -0
- package/src/wgsl/advance-expand.wgsl.ts +68 -0
- package/src/wgsl/bf-relax.wgsl.ts +57 -0
- package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
- package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
- package/src/wgsl/bfs-contract.wgsl.ts +54 -0
- package/src/wgsl/bfs-fused.wgsl.ts +77 -0
- package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
- package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
- package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
- package/src/wgsl/compact-scatter.wgsl.ts +16 -0
- package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
- package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
- package/src/wgsl/frontier-finalize.wgsl.ts +209 -0
- package/src/wgsl/sssp-pred.wgsl.ts +79 -0
- package/src/wgsl/sssp-relax.wgsl.ts +71 -0
- package/dist/chunks/context-BXqgCifx.js.map +0 -1
|
@@ -0,0 +1,395 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Closeness centrality on the device (design 8.4, 3.3 line 810, 9.7; P8-T11, the P8 plan's PD-13 / PD-19 / PD-25 /
|
|
3
|
+
* DEP-P8-E / DEP-P8-F): `score[s] = 1 / sumDist_s`, with `sumDist_s` the exact sum of the finite distances from `s`
|
|
4
|
+
* to every OTHER node -- an unreached node adds nothing -- and `0` when nothing is reached. No reached factor and no
|
|
5
|
+
* Wasserman-Faust scaling: this is EXACTLY the legacy default (`normalized: false`) of `closenessCentrality` in
|
|
6
|
+
* `@graphty/algorithms`, the number graphty-element's closeness panel shows today, so a future `indexed` port has one
|
|
7
|
+
* number to match (the NetworkX form is 33x it on karate and could never have been substituted silently).
|
|
8
|
+
*
|
|
9
|
+
* The unweighted route is ONE bit-parallel multi-source search per batch of 32 sources (`ceil(n / 32)` batches):
|
|
10
|
+
* the batch's state is one `bits` buffer of four regions of `bitsBase = roundUp(n, 64)` words (`visited`, two
|
|
11
|
+
* frontier regions that swap by the level's parity, `flags`), bit `s` of word `v` meaning "source `s` has reached /
|
|
12
|
+
* is at / is next at `v`"; a level is, all host-recorded, `closeness-reduce` role 0 (the boundary: `done` from the
|
|
13
|
+
* previous level's compacted count, the level's claims folded into the exact 64-bit per-source sums at `level + 1`),
|
|
14
|
+
* `compact` of the flags into the frontier list, two `fill`s zeroing the level's next region and the flags (AFTER
|
|
15
|
+
* the compaction that consumed them), and `closeness-sweep` (the block-mapped expansion with the claim inline, a
|
|
16
|
+
* direct grid-stride dispatch looping to the list's count). `MAX_LEVELS_PER_SUBMIT` levels per submit and
|
|
17
|
+
* one readback per submit (the `done` word and the 512-byte `perSource` block together, so the finished batch needs
|
|
18
|
+
* no extra map); the host folds `sumHi x 2^32 + sumLo` into `1 / sum` in f64 and stores f32. The weighted route
|
|
19
|
+
* (`weighted` true on a snapshot whose column is not all ones) is one `sssp` per source with the sums reduced on the
|
|
20
|
+
* host between calls: design 8.4's own answer, slow and correct. `weighted` defaults to the snapshot's `flags.weighted`;
|
|
21
|
+
* `weighted: false` on a weighted snapshot ignores the column by request and sweeps; `weighted: true` over unit
|
|
22
|
+
* weights or no column sweeps too (every `sssp` would route to a BFS anyway). `maxIterations` and `tolerance` are the
|
|
23
|
+
* seam's placeholder keys and an exact traversal has neither, so a defined value is REFUSED before any device work
|
|
24
|
+
* (`E_UNSUPPORTED { option }`, the package's rule for an option it does not implement, PD-25); `undefined` is legal.
|
|
25
|
+
* `iterations` reports the source batches run (the sources, on the weighted route), `converged` is always true.
|
|
26
|
+
*
|
|
27
|
+
* Cost, stated so nobody is surprised: closeness is O(n x m) on any device -- at 1M nodes it is 31,250 batches of a
|
|
28
|
+
* full multi-source traversal, minutes on the card, and no target in design 10.4 asks for less. `compact.record`
|
|
29
|
+
* leases its offsets and the scan's block sums afresh on every call, once per LEVEL here, so the planner is given a
|
|
30
|
+
* scope whose `scratch` hands the same buffer back for the same label and size: the dispatches of one pass run in
|
|
31
|
+
* order, a level's scan overwrites the previous level's, and the run holds one set instead of `levels x batches`.
|
|
32
|
+
* The sweep keeps DEP-P8-E's refusal of a windowed core (`assertWholeCore`): the bit-parallel claim needs the whole
|
|
33
|
+
* arc array bound. The tuning entry `closenessWithTuning` (PD-26's shape) is what the tests drive; nothing public
|
|
34
|
+
* exposes it.
|
|
35
|
+
*/
|
|
36
|
+
|
|
37
|
+
import { type F32, type GraphSnapshot, type U32 } from "@graphty/graph-format";
|
|
38
|
+
|
|
39
|
+
import { MAX_LEVELS_PER_SUBMIT } from "../constants.js";
|
|
40
|
+
import { type GpuContext } from "../context.js";
|
|
41
|
+
import { WebGpuGraphError } from "../errors.js";
|
|
42
|
+
import { CommandBatch } from "../kernel/batch.js";
|
|
43
|
+
import { plan1d, planGridStride } from "../kernel/dispatch.js";
|
|
44
|
+
import {
|
|
45
|
+
FILL_PARAMS,
|
|
46
|
+
FRONTIER_COUNTERS,
|
|
47
|
+
FRONTIER_PARAMS,
|
|
48
|
+
graphBindings,
|
|
49
|
+
graphOverrides,
|
|
50
|
+
kernelSpec,
|
|
51
|
+
} from "../kernels.js";
|
|
52
|
+
import { prepareCompact } from "../primitives/compact.js";
|
|
53
|
+
import { assertWholeCore } from "../primitives/core-shape.js";
|
|
54
|
+
import { W } from "../primitives/frontier.js";
|
|
55
|
+
import { type ReduceScope } from "../primitives/reduce.js";
|
|
56
|
+
import { assertDeviceComputes } from "../primitives/verify.js";
|
|
57
|
+
import { type HitsOptionsLike } from "../types/accelerator.js";
|
|
58
|
+
import { type GpuScoresResult } from "../types/algorithms.js";
|
|
59
|
+
import { type Binding } from "../types/memory.js";
|
|
60
|
+
import { type GpuRunOptions } from "../types/run.js";
|
|
61
|
+
import { algorithmScope } from "./scope.js";
|
|
62
|
+
import { aborted, bindingOf, checkDest, sssp } from "./sssp.js";
|
|
63
|
+
|
|
64
|
+
const ALGORITHM = "closenessCentrality";
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* Design 8.4: 32 sources per `u32` word, one batch per word.
|
|
68
|
+
* @internal
|
|
69
|
+
*/
|
|
70
|
+
export const SOURCES_PER_BATCH = 32;
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* The `perSource` block of a batch: `newCount[32]` @0, `reached[32]` @32, `sumLo[32]` @64, `sumHi[32]` @96.
|
|
74
|
+
* @internal
|
|
75
|
+
*/
|
|
76
|
+
export const PER_SOURCE_WORDS = 4 * SOURCES_PER_BATCH;
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Params slots of the ring, COUNTED (`UniformRing.reserve` wraps silently): per level `compact`'s records (its scan
|
|
80
|
+
* is at most four levels for any n below 2^32, so at most 8 records) while the boundary, the finalize, the fill and
|
|
81
|
+
* the two sweep records (one per parity) are written once per submit, plus the seed's two records on a batch's first
|
|
82
|
+
* submit (the iota fill flushes in its own submit).
|
|
83
|
+
*/
|
|
84
|
+
const RING_SLOTS = 8 * MAX_LEVELS_PER_SUBMIT + 16;
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* The knobs the tests need and nothing public offers (PD-26's shape): the submit cadence and the inspect seam.
|
|
88
|
+
* @internal
|
|
89
|
+
*/
|
|
90
|
+
export interface ClosenessTuning {
|
|
91
|
+
/** Levels recorded per submit on the bit-parallel route (default `MAX_LEVELS_PER_SUBMIT`). */
|
|
92
|
+
readonly levelsPerSubmit?: number | undefined;
|
|
93
|
+
/** The inspect seam: after every BATCH of the bit-parallel route, its first source and a fresh copy of the 128-word `perSource` block as the last submit left it. */
|
|
94
|
+
readonly onBatch?: ((batchStart: number, perSource: U32) => void) | undefined;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
/**
|
|
98
|
+
* A scope whose `scratch` hands the SAME buffer back for the same label and size (see the file comment).
|
|
99
|
+
* @param scope - the algorithm's scope
|
|
100
|
+
* @returns the reusing scope
|
|
101
|
+
*/
|
|
102
|
+
function reusingScratch(scope: ReduceScope): ReduceScope {
|
|
103
|
+
const held = new Map<string, GPUBuffer>();
|
|
104
|
+
return {
|
|
105
|
+
...scope,
|
|
106
|
+
scratch: (byteLength, label) => {
|
|
107
|
+
const key = `${label}/${byteLength}`;
|
|
108
|
+
let buffer = held.get(key);
|
|
109
|
+
if (buffer === undefined) {
|
|
110
|
+
buffer = scope.scratch(byteLength, label);
|
|
111
|
+
held.set(key, buffer);
|
|
112
|
+
}
|
|
113
|
+
return buffer;
|
|
114
|
+
},
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* The weighted route: one `sssp` per source, the sums reduced on the host.
|
|
120
|
+
* @param ctx - the context
|
|
121
|
+
* @param s - the snapshot
|
|
122
|
+
* @param scores - the destination
|
|
123
|
+
* @param options - the run options
|
|
124
|
+
* @returns the result
|
|
125
|
+
*/
|
|
126
|
+
async function weightedRoute(
|
|
127
|
+
ctx: GpuContext,
|
|
128
|
+
s: GraphSnapshot,
|
|
129
|
+
scores: F32,
|
|
130
|
+
options: GpuRunOptions | undefined,
|
|
131
|
+
): Promise<GpuScoresResult> {
|
|
132
|
+
const n = s.nodeCount;
|
|
133
|
+
for (let source = 0; source < n; source++) {
|
|
134
|
+
if (options?.signal?.aborted) {
|
|
135
|
+
throw aborted(ALGORITHM);
|
|
136
|
+
}
|
|
137
|
+
const { dist } = await sssp(ctx, s, source, { signal: options?.signal });
|
|
138
|
+
let sum = 0;
|
|
139
|
+
for (let v = 0; v < n; v++) {
|
|
140
|
+
const d = dist[v];
|
|
141
|
+
if (v !== source && d !== Infinity) {
|
|
142
|
+
sum += d;
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
scores[source] = sum === 0 ? 0 : 1 / sum;
|
|
146
|
+
options?.onProgress?.(source + 1, n);
|
|
147
|
+
}
|
|
148
|
+
return { scores, iterations: n, converged: true, precision: "f32" };
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
/**
|
|
152
|
+
* The bit-parallel route (see the file comment).
|
|
153
|
+
* @param ctx - the context
|
|
154
|
+
* @param s - the snapshot
|
|
155
|
+
* @param scores - the destination
|
|
156
|
+
* @param levelsPerSubmit - the submit cadence
|
|
157
|
+
* @param options - the run options
|
|
158
|
+
* @param tuning - the knobs
|
|
159
|
+
* @returns the result
|
|
160
|
+
*/
|
|
161
|
+
async function sweepRoute(
|
|
162
|
+
ctx: GpuContext,
|
|
163
|
+
s: GraphSnapshot,
|
|
164
|
+
scores: F32,
|
|
165
|
+
levelsPerSubmit: number,
|
|
166
|
+
options: GpuRunOptions | undefined,
|
|
167
|
+
tuning: ClosenessTuning,
|
|
168
|
+
): Promise<GpuScoresResult> {
|
|
169
|
+
const n = s.nodeCount;
|
|
170
|
+
if (n === 0) {
|
|
171
|
+
return { scores, iterations: 0, converged: true, precision: "f32" };
|
|
172
|
+
}
|
|
173
|
+
const core = ctx.residency.core(s);
|
|
174
|
+
assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM);
|
|
175
|
+
const scope = algorithmScope(ctx, ALGORITHM, RING_SLOTS);
|
|
176
|
+
try {
|
|
177
|
+
const wg = ctx.workgroupSize;
|
|
178
|
+
const bytes = 4 * n;
|
|
179
|
+
// the four regions of the bits buffer, each bitsBase words so its byte offset is 256-aligned and fill and
|
|
180
|
+
// compact can bind one alone: visited 0, the frontier pair 1 and 2, flags 3
|
|
181
|
+
const bitsBase = Math.ceil(n / 64) * 64;
|
|
182
|
+
const regionBytes = 4 * bitsBase;
|
|
183
|
+
const bits = bindingOf(scope.scratch(4 * regionBytes, "bits"), 4 * regionBytes);
|
|
184
|
+
const region = (index: number): Binding => ({
|
|
185
|
+
buffer: bits.buffer,
|
|
186
|
+
offset: index * regionBytes,
|
|
187
|
+
size: regionBytes,
|
|
188
|
+
window: null,
|
|
189
|
+
});
|
|
190
|
+
const flags = region(3);
|
|
191
|
+
const frontierList = bindingOf(scope.scratch(bytes, "frontier-list"), bytes);
|
|
192
|
+
const iota = bindingOf(scope.scratch(bytes, "iota"), bytes);
|
|
193
|
+
const counters = bindingOf(
|
|
194
|
+
scope.scratch(FRONTIER_COUNTERS.byteLength, "counters"),
|
|
195
|
+
FRONTIER_COUNTERS.byteLength,
|
|
196
|
+
);
|
|
197
|
+
const perSourceBytes = 4 * PER_SOURCE_WORDS;
|
|
198
|
+
const perSource = bindingOf(scope.scratch(perSourceBytes, "per-source"), perSourceBytes);
|
|
199
|
+
await ctx.allocator.check();
|
|
200
|
+
const compact = await prepareCompact(reusingScratch(scope));
|
|
201
|
+
const sweep = await ctx.pipelines.kernel(kernelSpec("closeness-sweep", graphOverrides(core, null)));
|
|
202
|
+
const reduce = await ctx.pipelines.kernel(kernelSpec("closeness-reduce"));
|
|
203
|
+
const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
|
|
204
|
+
const graph = graphBindings(core, null);
|
|
205
|
+
const onePlan = plan1d(1, wg, ctx.caps);
|
|
206
|
+
const regionPlan = plan1d(bitsBase, wg, ctx.caps);
|
|
207
|
+
const sweepPlan = planGridStride(n, wg, ctx.caps); // the list holds at most n entries; the sweep loops to the count word
|
|
208
|
+
const recordFill = (pass: GPUComputePassEncoder, dst: Binding, count: number, mode: 0 | 1): void => {
|
|
209
|
+
const params = scope.params(FILL_PARAMS, { count, value: 0, mode, pad0: 0 });
|
|
210
|
+
fill.dispatch(pass, fill.bind({ dst, P: params.binding }), plan1d(count, wg, ctx.caps), [params.offset]);
|
|
211
|
+
};
|
|
212
|
+
const submit = (batch: CommandBatch): ReturnType<CommandBatch["submit"]> => {
|
|
213
|
+
scope.flush();
|
|
214
|
+
return batch.submit();
|
|
215
|
+
};
|
|
216
|
+
|
|
217
|
+
// setup: the iota queue compact reads, once per run
|
|
218
|
+
const setup = new CommandBatch(ctx, `${ALGORITHM}/setup`);
|
|
219
|
+
recordFill(setup.pass("fill"), iota, n, 1);
|
|
220
|
+
setup.endPass();
|
|
221
|
+
await submit(setup).readback;
|
|
222
|
+
ctx.assertReady();
|
|
223
|
+
|
|
224
|
+
let batches = 0;
|
|
225
|
+
for (let batchStart = 0; batchStart < n; batchStart += SOURCES_PER_BATCH) {
|
|
226
|
+
let level = 0;
|
|
227
|
+
for (let first = true; ; first = false) {
|
|
228
|
+
const batch = new CommandBatch(ctx, `${ALGORITHM}/levels`);
|
|
229
|
+
const pass = batch.pass("closeness");
|
|
230
|
+
if (first) {
|
|
231
|
+
// the batch's seed: the four regions and the block zeroed, then role 1 (the sources' bits, their
|
|
232
|
+
// flags, counters[0] = k, level = U32_MAX)
|
|
233
|
+
recordFill(pass, bits, 4 * bitsBase, 0);
|
|
234
|
+
recordFill(pass, perSource, PER_SOURCE_WORDS, 0);
|
|
235
|
+
const seed = scope.params(FRONTIER_PARAMS, { role: 1, n, bitsBase, source: batchStart });
|
|
236
|
+
reduce.dispatch(pass, reduce.bind({ counters, perSource, bits, P: seed.binding }), onePlan, [
|
|
237
|
+
seed.offset,
|
|
238
|
+
]);
|
|
239
|
+
}
|
|
240
|
+
// the records every level of the submit shares (the ring wraps, so they are written per submit)
|
|
241
|
+
const boundary = scope.params(FRONTIER_PARAMS, { role: 0, n, bitsBase });
|
|
242
|
+
const boundBoundary = reduce.bind({ counters, perSource, bits, P: boundary.binding });
|
|
243
|
+
const clear = scope.params(FILL_PARAMS, { count: bitsBase, value: 0, mode: 0, pad0: 0 });
|
|
244
|
+
// parity 0 sweeps region 1 into region 2, parity 1 region 2 into region 1: the next region is cleared
|
|
245
|
+
const boundClearNext = [region(2), region(1)].map((dst) => fill.bind({ dst, P: clear.binding }));
|
|
246
|
+
const boundClearFlags = fill.bind({ dst: flags, P: clear.binding });
|
|
247
|
+
const boundSweep = [0, 1].map((mode) => {
|
|
248
|
+
const params = scope.params(FRONTIER_PARAMS, {
|
|
249
|
+
wg,
|
|
250
|
+
n,
|
|
251
|
+
bitsBase,
|
|
252
|
+
arcBase: 0,
|
|
253
|
+
arcEnd: s.arcCount,
|
|
254
|
+
mode,
|
|
255
|
+
stride: sweepPlan.stride ?? wg,
|
|
256
|
+
});
|
|
257
|
+
return {
|
|
258
|
+
bound: sweep.bind({ ...graph, frontierList, counters, bits, perSource, P: params.binding }),
|
|
259
|
+
offset: params.offset,
|
|
260
|
+
};
|
|
261
|
+
});
|
|
262
|
+
for (let k = 0; k < levelsPerSubmit; k++, level++) {
|
|
263
|
+
const parity = level % 2;
|
|
264
|
+
reduce.dispatch(pass, boundBoundary, onePlan, [boundary.offset]);
|
|
265
|
+
compact.record(pass, {
|
|
266
|
+
queue: iota,
|
|
267
|
+
flags,
|
|
268
|
+
count: n,
|
|
269
|
+
out: frontierList,
|
|
270
|
+
outCount: counters,
|
|
271
|
+
outIndex: W.frontierCount,
|
|
272
|
+
});
|
|
273
|
+
fill.dispatch(pass, boundClearNext[parity], regionPlan, [clear.offset]);
|
|
274
|
+
fill.dispatch(pass, boundClearFlags, regionPlan, [clear.offset]);
|
|
275
|
+
sweep.dispatch(pass, boundSweep[parity].bound, sweepPlan, [boundSweep[parity].offset]);
|
|
276
|
+
}
|
|
277
|
+
batch.endPass();
|
|
278
|
+
const doneRequest = batch.readback(counters.buffer, counters.offset + 4 * W.done, 4);
|
|
279
|
+
const blockRequest = batch.readback(perSource.buffer, perSource.offset, perSourceBytes);
|
|
280
|
+
const submitted = submit(batch);
|
|
281
|
+
const back = await submitted.readback;
|
|
282
|
+
ctx.assertReady();
|
|
283
|
+
if (options?.signal?.aborted) {
|
|
284
|
+
throw aborted(ALGORITHM, submitted.id);
|
|
285
|
+
}
|
|
286
|
+
if (new Uint32Array(back, doneRequest.offset, 1)[0] !== 0) {
|
|
287
|
+
const block = new Uint32Array(back, blockRequest.offset, PER_SOURCE_WORDS);
|
|
288
|
+
const count = Math.min(SOURCES_PER_BATCH, n - batchStart);
|
|
289
|
+
for (let i = 0; i < count; i++) {
|
|
290
|
+
const sum = block[3 * SOURCES_PER_BATCH + i] * 2 ** 32 + block[2 * SOURCES_PER_BATCH + i];
|
|
291
|
+
scores[batchStart + i] = sum === 0 ? 0 : 1 / sum;
|
|
292
|
+
}
|
|
293
|
+
tuning.onBatch?.(batchStart, block.slice());
|
|
294
|
+
break;
|
|
295
|
+
}
|
|
296
|
+
if (level > n + 3) {
|
|
297
|
+
// a batch claims at most n - 1 levels deep, then one level claims nothing and one is empty
|
|
298
|
+
throw new WebGpuGraphError(
|
|
299
|
+
"E_VALIDATION",
|
|
300
|
+
`${ALGORITHM}: the done flag never rose in ${level} levels of the batch at ${batchStart}`,
|
|
301
|
+
{ label: ALGORITHM, message: `the done flag never rose in ${level} levels` },
|
|
302
|
+
);
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
batches += 1;
|
|
306
|
+
options?.onProgress?.(Math.min(batchStart + SOURCES_PER_BATCH, n), n);
|
|
307
|
+
}
|
|
308
|
+
return { scores, iterations: batches, converged: true, precision: "f32" };
|
|
309
|
+
} finally {
|
|
310
|
+
scope.dispose();
|
|
311
|
+
}
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
/**
|
|
315
|
+
* Closeness with the test knobs of PD-26's shape; `closenessCentrality` is this with an empty tuning.
|
|
316
|
+
* @internal
|
|
317
|
+
* @param ctx - the context whose device runs the kernels
|
|
318
|
+
* @param s - the snapshot (uploaded through ctx.residency, or found there)
|
|
319
|
+
* @param options - the seam's `HitsOptionsLike` (`weighted` honoured, the other two refused when defined), plus dest / signal / onProgress
|
|
320
|
+
* @param tuning - the knobs
|
|
321
|
+
* @returns the scores, the batches run, `converged: true` and `precision: "f32"`
|
|
322
|
+
*/
|
|
323
|
+
export async function closenessWithTuning(
|
|
324
|
+
ctx: GpuContext,
|
|
325
|
+
s: GraphSnapshot,
|
|
326
|
+
options: (HitsOptionsLike & GpuRunOptions) | undefined,
|
|
327
|
+
tuning: ClosenessTuning,
|
|
328
|
+
): Promise<GpuScoresResult> {
|
|
329
|
+
ctx.assertReady();
|
|
330
|
+
await assertDeviceComputes(ctx);
|
|
331
|
+
for (const key of ["maxIterations", "tolerance"] as const) {
|
|
332
|
+
if (options?.[key] !== undefined) {
|
|
333
|
+
throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: ${key} has no meaning for an exact traversal`, {
|
|
334
|
+
option: key,
|
|
335
|
+
hint: "closeness is an exact traversal; the option has no meaning here",
|
|
336
|
+
});
|
|
337
|
+
}
|
|
338
|
+
}
|
|
339
|
+
const n = s.nodeCount;
|
|
340
|
+
const levelsPerSubmit = tuning.levelsPerSubmit ?? MAX_LEVELS_PER_SUBMIT;
|
|
341
|
+
if (!Number.isInteger(levelsPerSubmit) || levelsPerSubmit < 1 || levelsPerSubmit > MAX_LEVELS_PER_SUBMIT) {
|
|
342
|
+
throw new WebGpuGraphError(
|
|
343
|
+
"E_INVALID_ARGUMENT",
|
|
344
|
+
`${ALGORITHM}: levelsPerSubmit must be an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
|
|
345
|
+
{
|
|
346
|
+
argument: "levelsPerSubmit",
|
|
347
|
+
value: levelsPerSubmit,
|
|
348
|
+
expected: `an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
|
|
349
|
+
},
|
|
350
|
+
);
|
|
351
|
+
}
|
|
352
|
+
const scores = checkDest(ALGORITHM, options?.dest, n) ?? new Float32Array(n);
|
|
353
|
+
const weighted = options?.weighted ?? s.flags.weighted;
|
|
354
|
+
if (options?.signal?.aborted) {
|
|
355
|
+
throw aborted(ALGORITHM);
|
|
356
|
+
}
|
|
357
|
+
if (weighted && s.weights !== null && !s.flags.allWeightsOne) {
|
|
358
|
+
if (!s.flags.nonNegativeWeights) {
|
|
359
|
+
throw new WebGpuGraphError(
|
|
360
|
+
"E_UNSUPPORTED",
|
|
361
|
+
`${ALGORITHM}: a negative weight has no shortest-path distance to sum`,
|
|
362
|
+
{
|
|
363
|
+
feature: "closenessCentrality.negativeWeights",
|
|
364
|
+
hint: "pass weighted: false to ignore the column",
|
|
365
|
+
},
|
|
366
|
+
);
|
|
367
|
+
}
|
|
368
|
+
if (!s.flags.finiteWeights) {
|
|
369
|
+
throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: a NaN or infinite weight has no shortest path`, {
|
|
370
|
+
feature: "closenessCentrality.nonFiniteWeights",
|
|
371
|
+
});
|
|
372
|
+
}
|
|
373
|
+
return weightedRoute(ctx, s, scores, options);
|
|
374
|
+
}
|
|
375
|
+
return sweepRoute(ctx, s, scores, levelsPerSubmit, options, tuning);
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
/**
|
|
379
|
+
* Closeness centrality on the device (spec 3.3 line 810, design 8.4, 9.7): `scores[s] = 1 / sumDist_s` over the finite
|
|
380
|
+
* distances from `s` to every other node, `0` when nothing is reached -- the legacy default of `@graphty/algorithms`'
|
|
381
|
+
* `closenessCentrality`, unweighted by one bit-parallel multi-source search per 32 sources, weighted by one `sssp`
|
|
382
|
+
* per source; `weighted` defaults to the snapshot's flag, `maxIterations` / `tolerance` are refused when defined
|
|
383
|
+
* (PD-25). `iterations` is the source batches run and `converged` is always true.
|
|
384
|
+
* @param ctx - the context whose device runs the kernels
|
|
385
|
+
* @param s - the snapshot (uploaded through ctx.residency, or found there)
|
|
386
|
+
* @param options - the seam's `HitsOptionsLike`, plus dest (a Float32Array of length n for `scores`) / signal / onProgress
|
|
387
|
+
* @returns the scores, the batches run, `converged: true` and `precision: "f32"`
|
|
388
|
+
*/
|
|
389
|
+
export function closenessCentrality(
|
|
390
|
+
ctx: GpuContext,
|
|
391
|
+
s: GraphSnapshot,
|
|
392
|
+
options?: HitsOptionsLike & GpuRunOptions,
|
|
393
|
+
): Promise<GpuScoresResult> {
|
|
394
|
+
return closenessWithTuning(ctx, s, options, {});
|
|
395
|
+
}
|
package/src/algorithms/scope.ts
CHANGED
|
@@ -8,15 +8,18 @@
|
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
10
|
import { type GpuContext } from "../context.js";
|
|
11
|
+
import { BufferUsage } from "../device/webgpu-constants.js";
|
|
11
12
|
import { UniformRing } from "../kernel/uniform-ring.js";
|
|
12
|
-
import { type
|
|
13
|
+
import { type FrontierScope } from "../primitives/frontier.js";
|
|
13
14
|
|
|
14
|
-
/** A ReduceScope over a context plus the two lifecycle calls an algorithm makes: flush() before submit, dispose() in its finally. */
|
|
15
|
-
export interface AlgorithmScope extends
|
|
15
|
+
/** A FrontierScope (a ReduceScope plus `indirect()`, P8-T4) over a context plus the two lifecycle calls an algorithm makes: flush() before submit, dispose() in its finally. */
|
|
16
|
+
export interface AlgorithmScope extends FrontierScope {
|
|
16
17
|
/** queue.writeBuffer of the params slots written since the last flush (called before the batch is submitted). */
|
|
17
18
|
flush(): void;
|
|
18
19
|
/** Destroys the ring and releases every scratch buffer of the lease; idempotent. */
|
|
19
20
|
dispose(): void;
|
|
21
|
+
/** The ring's `overruns` so far: reservations that wrapped over a record the batch being recorded still reads (0 on a correctly sized ring; P8-T12). */
|
|
22
|
+
ringOverruns(): number;
|
|
20
23
|
}
|
|
21
24
|
|
|
22
25
|
/**
|
|
@@ -36,6 +39,12 @@ export function algorithmScope(ctx: GpuContext, label: string, slots: number): A
|
|
|
36
39
|
pool: ctx.pool,
|
|
37
40
|
workgroupSize: ctx.workgroupSize,
|
|
38
41
|
scratch: (byteLength, scratchLabel) => lease.storage(byteLength, `${label}/${scratchLabel}`),
|
|
42
|
+
indirect: (byteLength, indirectLabel) =>
|
|
43
|
+
lease.acquire(
|
|
44
|
+
byteLength,
|
|
45
|
+
BufferUsage.STORAGE | BufferUsage.INDIRECT | BufferUsage.COPY_DST | BufferUsage.COPY_SRC,
|
|
46
|
+
`${label}/${indirectLabel}`,
|
|
47
|
+
),
|
|
39
48
|
params(block, values) {
|
|
40
49
|
const slot = ring.reserve(1);
|
|
41
50
|
ring.write(slot, block, values);
|
|
@@ -48,5 +57,6 @@ export function algorithmScope(ctx: GpuContext, label: string, slots: number): A
|
|
|
48
57
|
ring.destroy();
|
|
49
58
|
lease.release();
|
|
50
59
|
},
|
|
60
|
+
ringOverruns: () => ring.overruns,
|
|
51
61
|
};
|
|
52
62
|
}
|