@graphty/webgpu-graph-algorithms 0.6.21 → 0.6.23
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-VIvatQOo.js → context-B40Z6lV_.js} +39 -30
- package/dist/chunks/context-B40Z6lV_.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +1 -1
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +20 -1
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/mst.d.ts +50 -0
- package/dist/src/algorithms/mst.d.ts.map +1 -0
- package/dist/src/algorithms/mst.js +245 -0
- package/dist/src/algorithms/mst.js.map +1 -0
- package/dist/src/constants.d.ts +6 -1
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +6 -1
- package/dist/src/constants.js.map +1 -1
- package/dist/src/index.d.ts +3 -2
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +1 -0
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +7 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernels.d.ts +5 -2
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +58 -2
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/types/accelerator.d.ts +7 -5
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/structure.d.ts +17 -0
- package/dist/src/types/structure.d.ts.map +1 -1
- package/dist/src/wgsl/mst-best.wgsl.d.ts +13 -0
- package/dist/src/wgsl/mst-best.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/mst-best.wgsl.js +32 -0
- package/dist/src/wgsl/mst-best.wgsl.js.map +1 -0
- package/dist/src/wgsl/mst-link.wgsl.d.ts +14 -0
- package/dist/src/wgsl/mst-link.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/mst-link.wgsl.js +39 -0
- package/dist/src/wgsl/mst-link.wgsl.js.map +1 -0
- package/dist/webgpu-graph-algorithms.js +473 -162
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +3 -3
- package/src/accelerator.ts +22 -2
- package/src/algorithms/mst.ts +269 -0
- package/src/constants.ts +6 -1
- package/src/index.ts +3 -1
- package/src/kernel/prelude.ts +7 -0
- package/src/kernels.ts +64 -3
- package/src/types/accelerator.ts +7 -3
- package/src/types/structure.ts +18 -0
- package/src/wgsl/mst-best.wgsl.ts +31 -0
- package/src/wgsl/mst-link.wgsl.ts +38 -0
- package/dist/chunks/context-VIvatQOo.js.map +0 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@graphty/webgpu-graph-algorithms",
|
|
3
|
-
"version": "0.6.
|
|
3
|
+
"version": "0.6.23",
|
|
4
4
|
"description": "WebGPU-accelerated graph algorithms and layouts over the @graphty/graph-format snapshot, for Node (Dawn) and browsers",
|
|
5
5
|
"author": "Adam Powers <apowers@ato.ms>",
|
|
6
6
|
"type": "module",
|
|
@@ -95,8 +95,8 @@
|
|
|
95
95
|
"vite": "^7.0.5",
|
|
96
96
|
"vitest": "4.1.11",
|
|
97
97
|
"webgpu": "0.4.0",
|
|
98
|
-
"@graphty/
|
|
99
|
-
"@graphty/
|
|
98
|
+
"@graphty/layout": "^2.0.6",
|
|
99
|
+
"@graphty/algorithms": "^3.1.6"
|
|
100
100
|
},
|
|
101
101
|
"scripts": {
|
|
102
102
|
"build": "node -e \"require('fs').rmSync('dist',{recursive:true,force:true})\" && tsc -p tsconfig.build.json",
|
package/src/accelerator.ts
CHANGED
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
* both models were green, P5 PD-19), P7's seven algorithm members (spec 8.2, 8.3; M8b-T8, PD-14) and P8's four
|
|
9
9
|
* traversal members (spec 8.4; P8-T13 PD-16, PD-19: `breadthFirstSearch`, `sssp`, `bellmanFord`,
|
|
10
10
|
* `closenessCentrality`, each taking the seam's own option type), `allPairsShortestPath` (design 8.7), P11's
|
|
11
|
-
* `triangleCount` and `labelPropagation`, and nothing else: the CPU-side dispatchers
|
|
11
|
+
* `triangleCount` and `labelPropagation`, Boruvka's `minimumSpanningTree`, and nothing else: the CPU-side dispatchers
|
|
12
12
|
* (`accelerated()`, `createSimulation()`) test `acc.betweennessCentrality !== undefined` /
|
|
13
13
|
* `acc.fruchtermanReingold !== undefined` and route to the CPU when the member is absent (spec 2.4 row "method
|
|
14
14
|
* missing"), so a method the GPU does not implement must not exist here -- never a throwing stub. The remaining
|
|
@@ -24,6 +24,7 @@ import { breadthFirstSearch } from "./algorithms/bfs.js";
|
|
|
24
24
|
import { closenessCentrality } from "./algorithms/closeness.js";
|
|
25
25
|
import { connectedComponents } from "./algorithms/components.js";
|
|
26
26
|
import { labelPropagation } from "./algorithms/label-propagation.js";
|
|
27
|
+
import { minimumSpanningTree } from "./algorithms/mst.js";
|
|
27
28
|
import { pageRank, personalizedPageRank } from "./algorithms/pagerank.js";
|
|
28
29
|
import { eigenvectorCentrality, hits, katzCentrality } from "./algorithms/spectral.js";
|
|
29
30
|
import { sssp } from "./algorithms/sssp.js";
|
|
@@ -40,6 +41,7 @@ import {
|
|
|
40
41
|
type ClosenessAcceleratorOptions,
|
|
41
42
|
type GpuAccelerator,
|
|
42
43
|
type HitsOptionsLike,
|
|
44
|
+
type MstOptions,
|
|
43
45
|
type SsspOptions,
|
|
44
46
|
} from "./types/accelerator.js";
|
|
45
47
|
import {
|
|
@@ -68,7 +70,7 @@ import {
|
|
|
68
70
|
type FruchtermanReingoldOptions,
|
|
69
71
|
type SpringElectricalOptions,
|
|
70
72
|
} from "./types/options.js";
|
|
71
|
-
import { type GpuTriangleResult } from "./types/structure.js";
|
|
73
|
+
import { type GpuMstResult, type GpuTriangleResult } from "./types/structure.js";
|
|
72
74
|
import { type GpuBellmanFordResult, type GpuBfsResult, type GpuSsspResult } from "./types/traversal.js";
|
|
73
75
|
|
|
74
76
|
/** The `algorithms` record of AcceleratorOptions (spec 3.3), named for the copy helpers. */
|
|
@@ -412,6 +414,24 @@ export function createAccelerator(ctx: GpuContext, options?: AcceleratorOptions)
|
|
|
412
414
|
}
|
|
413
415
|
return await labelPropagation(ctx, gs, { maxIterations: o?.maxIterations, weighted: o?.weighted });
|
|
414
416
|
},
|
|
417
|
+
/**
|
|
418
|
+
* Boruvka's minimum spanning forest (design 8.5; P11-T4): the forest of the total edge order (weight, then
|
|
419
|
+
* edge index), which is the one `kruskalMST` accepts. The seam's per-arc `weights` override is refused when
|
|
420
|
+
* defined: the forest runs over the snapshot's own edge weights.
|
|
421
|
+
* @param gs - the snapshot
|
|
422
|
+
* @param o - the seam's `MstOptions`; `weights` refused when defined
|
|
423
|
+
* @returns the forest's logical edge indices and its f64 total weight
|
|
424
|
+
*/
|
|
425
|
+
async minimumSpanningTree(gs: GraphSnapshot, o?: MstOptions): Promise<GpuMstResult> {
|
|
426
|
+
ctx.assertReady();
|
|
427
|
+
if (o?.weights !== undefined) {
|
|
428
|
+
throw new WebGpuGraphError("E_UNSUPPORTED", "minimumSpanningTree: weights is not supported", {
|
|
429
|
+
option: "weights",
|
|
430
|
+
hint: "the spanning forest runs over the snapshot's own edge weights",
|
|
431
|
+
});
|
|
432
|
+
}
|
|
433
|
+
return await minimumSpanningTree(ctx, gs);
|
|
434
|
+
},
|
|
415
435
|
/**
|
|
416
436
|
* Destroys every device buffer recorded for the snapshot (spec 4.5); delegates to ctx.release.
|
|
417
437
|
* @param s - the snapshot the app is done with
|
|
@@ -0,0 +1,269 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Boruvka's minimum spanning forest (design 8.5, 3.3 line 808; the P11 plan's P11-T4) over the snapshot's edge list,
|
|
3
|
+
* each logical edge once, so a directed snapshot is spanned as its underlying undirected multigraph -- what
|
|
4
|
+
* `@graphty/algorithms`' `kruskalMST` does. Every vertex starts as its own component (`comp[v] = v`); each round:
|
|
5
|
+
*
|
|
6
|
+
* 1. `fill` resets every component's `bestKey` and `bestEdge` to `U32_MAX`;
|
|
7
|
+
* 2. `mst-best` twice (PD-7): the per-component minimum of the TOTAL edge order -- the order-preserving key of the
|
|
8
|
+
* weight first (PD-6: negative weights order below positive ones, -0 equals +0), the edge index among ties -- over
|
|
9
|
+
* the edges whose endpoints lie in different components, so self-loops and edges inside a component never compete;
|
|
10
|
+
* 3. `mst-link`: each root with a best edge hooks onto the component at its other end and records the edge, the lower
|
|
11
|
+
* root of a two-cycle keeping its label so the edge is recorded once;
|
|
12
|
+
* 4. `wcc-compress` (PD-7: the Afforest kernel, unchanged) until every vertex points at its root: a walk of at most
|
|
13
|
+
* COMPRESS_STEPS per dispatch divides every depth by COMPRESS_STEPS, so `compressPasses(n)` dispatches suffice.
|
|
14
|
+
*
|
|
15
|
+
* Because the order is total the forest is unique: it is the one Kruskal accepts when its sort breaks ties by edge
|
|
16
|
+
* index, which `kruskalMST` does. So the edge SET equals the CPU's on every graph, tied weights included; only the
|
|
17
|
+
* order of `edges` differs.
|
|
18
|
+
*
|
|
19
|
+
* BORUVKA_ROUNDS_PER_SUBMIT rounds are recorded per submit with one readback of their recorded-edge counts; a round
|
|
20
|
+
* that records nothing ends the run (rounds after it inside the same submit change nothing). The round count is at
|
|
21
|
+
* most ceil(log2 n) + 1, because every round at least halves the components that still have an outgoing edge;
|
|
22
|
+
* MAX_ROUNDS is the bound a device bug cannot pass, never a fallback. The forest is read back once at the end.
|
|
23
|
+
*
|
|
24
|
+
* PLAN DECISION (P11-T4 Step 3 appends each edge through an `atomicAdd` on the counters block): a root records its
|
|
25
|
+
* edge in its own word of `treeEdge`, because a root hooks at most once in a run. The result is then deterministic
|
|
26
|
+
* (the edges in root order, round by round) with no host sort, where an atomic append is ordered by the schedule.
|
|
27
|
+
*
|
|
28
|
+
* `totalWeight` is summed on the host in f64 over the snapshot's own edge weights, so it differs from
|
|
29
|
+
* `kruskalMST`'s only by the order of the summation.
|
|
30
|
+
*/
|
|
31
|
+
|
|
32
|
+
import { type GraphSnapshot, INVALID_INDEX, type U32 } from "@graphty/graph-format";
|
|
33
|
+
|
|
34
|
+
import { BORUVKA_ROUNDS_PER_SUBMIT, U32_MAX } from "../constants.js";
|
|
35
|
+
import { type GpuContext } from "../context.js";
|
|
36
|
+
import { WebGpuGraphError } from "../errors.js";
|
|
37
|
+
import { CommandBatch } from "../kernel/batch.js";
|
|
38
|
+
import { plan1d, planGridStride } from "../kernel/dispatch.js";
|
|
39
|
+
import { FILL_PARAMS, kernelSpec, MST_PARAMS, WCC_PARAMS } from "../kernels.js";
|
|
40
|
+
import { assertDeviceComputes } from "../primitives/verify.js";
|
|
41
|
+
import { type Binding } from "../types/memory.js";
|
|
42
|
+
import { type GpuRunOptions } from "../types/run.js";
|
|
43
|
+
import { type GpuMstResult } from "../types/structure.js";
|
|
44
|
+
import { algorithmScope } from "./scope.js";
|
|
45
|
+
|
|
46
|
+
const ALGORITHM = "minimumSpanningTree";
|
|
47
|
+
/** The bound of one compress walk (the `maxSteps` of `wcc-compress`), as connected components uses. */
|
|
48
|
+
const COMPRESS_STEPS = 1024;
|
|
49
|
+
/** A round count no correct run reaches: at most ceil(log2 n) + 1 <= 33 rounds for any u32 node count. */
|
|
50
|
+
const MAX_ROUNDS = 64;
|
|
51
|
+
/** Params slots of the largest batch: the first one's two initial fills, the reset fill, the shared best-pass and compress blocks, and one link block per round. */
|
|
52
|
+
const RING_SLOTS = 5 + BORUVKA_ROUNDS_PER_SUBMIT;
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* The `wcc-compress` dispatches that flatten any forest of `n` vertices: each one divides every depth by
|
|
56
|
+
* COMPRESS_STEPS (a vertex lands at least COMPRESS_STEPS ancestors up, or on its root).
|
|
57
|
+
* @param n - the vertex count
|
|
58
|
+
* @returns the dispatch count, at least 1
|
|
59
|
+
*/
|
|
60
|
+
export function compressPasses(n: number): number {
|
|
61
|
+
let passes = 1;
|
|
62
|
+
for (let reach = COMPRESS_STEPS; reach < n; reach *= COMPRESS_STEPS) {
|
|
63
|
+
passes++;
|
|
64
|
+
}
|
|
65
|
+
return passes;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* The forest from the per-root `treeEdge` words, with the host-side f64 total.
|
|
70
|
+
* @param s - the snapshot
|
|
71
|
+
* @param treeEdge - one word per vertex, the edge its component recorded or INVALID_INDEX
|
|
72
|
+
* @param expected - the edge count the rounds reported
|
|
73
|
+
* @returns the result
|
|
74
|
+
*/
|
|
75
|
+
function resultOf(s: GraphSnapshot, treeEdge: Uint32Array, expected: number): GpuMstResult {
|
|
76
|
+
const edges: U32 = new Uint32Array(expected);
|
|
77
|
+
const { weights } = s.edgeList();
|
|
78
|
+
let taken = 0;
|
|
79
|
+
let totalWeight = 0;
|
|
80
|
+
for (const e of treeEdge) {
|
|
81
|
+
if (e === INVALID_INDEX) {
|
|
82
|
+
continue;
|
|
83
|
+
}
|
|
84
|
+
if (e >= s.edgeCount || taken === expected) {
|
|
85
|
+
throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM}: the device recorded an edge it did not count`, {
|
|
86
|
+
label: `${ALGORITHM}/treeEdge`,
|
|
87
|
+
message: `edge ${e} at forest position ${taken} of ${expected}`,
|
|
88
|
+
});
|
|
89
|
+
}
|
|
90
|
+
edges[taken++] = e;
|
|
91
|
+
totalWeight += weights === null ? 1 : weights[e];
|
|
92
|
+
}
|
|
93
|
+
if (taken !== expected) {
|
|
94
|
+
throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM}: ${taken} recorded edges, ${expected} counted`, {
|
|
95
|
+
label: `${ALGORITHM}/treeEdge`,
|
|
96
|
+
message: "the per-round counts disagree with the recorded forest",
|
|
97
|
+
});
|
|
98
|
+
}
|
|
99
|
+
return { edges, totalWeight };
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Boruvka's minimum spanning forest on the device (see the file header).
|
|
104
|
+
* @param ctx - the context whose device runs the kernels
|
|
105
|
+
* @param s - the snapshot (its edge list is uploaded through ctx.residency, or found there)
|
|
106
|
+
* @param options - signal, onProgress (`dest` is refused: the forest's length is not known before the run)
|
|
107
|
+
* @returns the forest's logical edge indices and its total weight
|
|
108
|
+
*/
|
|
109
|
+
export async function minimumSpanningTree(
|
|
110
|
+
ctx: GpuContext,
|
|
111
|
+
s: GraphSnapshot,
|
|
112
|
+
options?: GpuRunOptions,
|
|
113
|
+
): Promise<GpuMstResult> {
|
|
114
|
+
ctx.assertReady();
|
|
115
|
+
await assertDeviceComputes(ctx);
|
|
116
|
+
if (options?.dest !== undefined) {
|
|
117
|
+
throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: dest is not supported`, {
|
|
118
|
+
option: "dest",
|
|
119
|
+
hint: "the forest's edge count is known only after the run",
|
|
120
|
+
});
|
|
121
|
+
}
|
|
122
|
+
if (options?.signal?.aborted) {
|
|
123
|
+
throw new WebGpuGraphError("E_ABORTED", `${ALGORITHM}: the signal was aborted before any work started`, {});
|
|
124
|
+
}
|
|
125
|
+
const n = s.nodeCount;
|
|
126
|
+
const m = s.edgeCount;
|
|
127
|
+
if (n === 0 || m === 0) {
|
|
128
|
+
options?.onProgress?.(1, 1);
|
|
129
|
+
return { edges: new Uint32Array(0), totalWeight: 0 };
|
|
130
|
+
}
|
|
131
|
+
// core() is what records (or re-records, after a release) the snapshot in the residency; only the edge list is read
|
|
132
|
+
ctx.residency.core(s, ["rowPtr"]);
|
|
133
|
+
const edges = ctx.residency.view(s, "edgeList");
|
|
134
|
+
const edgeWeight = edges.bindings.weights ?? null;
|
|
135
|
+
const scope = algorithmScope(ctx, ALGORITHM, RING_SLOTS);
|
|
136
|
+
try {
|
|
137
|
+
const wg = ctx.workgroupSize;
|
|
138
|
+
const words = (count: number, label: string): Binding => {
|
|
139
|
+
const size = 4 * Math.max(1, count);
|
|
140
|
+
return { buffer: scope.scratch(size, label), offset: 0, size, window: null };
|
|
141
|
+
};
|
|
142
|
+
const comp = words(n, "comp");
|
|
143
|
+
const bestKey = words(n, "bestKey");
|
|
144
|
+
const bestEdge = words(n, "bestEdge");
|
|
145
|
+
const treeEdge = words(n, "treeEdge");
|
|
146
|
+
const counters = words(BORUVKA_ROUNDS_PER_SUBMIT, "counters");
|
|
147
|
+
await ctx.allocator.check();
|
|
148
|
+
const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
|
|
149
|
+
const weighted = edgeWeight !== null;
|
|
150
|
+
const minKey = await ctx.pipelines.kernel(kernelSpec("mst-best", { PASS: 0, WEIGHTED: weighted }));
|
|
151
|
+
const minEdge = await ctx.pipelines.kernel(kernelSpec("mst-best", { PASS: 1, WEIGHTED: weighted }));
|
|
152
|
+
const link = await ctx.pipelines.kernel(kernelSpec("mst-link"));
|
|
153
|
+
const compress = await ctx.pipelines.kernel(kernelSpec("wcc-compress"));
|
|
154
|
+
const nodePlan = plan1d(n, wg, ctx.caps);
|
|
155
|
+
const edgePlan = plan1d(m, wg, ctx.caps);
|
|
156
|
+
const compressPlan = planGridStride(n, wg, ctx.caps);
|
|
157
|
+
const passes = compressPasses(n);
|
|
158
|
+
const { queue } = ctx.device;
|
|
159
|
+
|
|
160
|
+
let batch = new CommandBatch(ctx, `${ALGORITHM}/rounds`);
|
|
161
|
+
let first = true;
|
|
162
|
+
let rounds = 0;
|
|
163
|
+
let recorded = 0;
|
|
164
|
+
for (;;) {
|
|
165
|
+
queue.writeBuffer(counters.buffer, 0, new Uint32Array(BORUVKA_ROUNDS_PER_SUBMIT));
|
|
166
|
+
const pass = batch.pass("rounds");
|
|
167
|
+
const fillFor = (value: number, mode: number): { binding: Binding; offset: number } =>
|
|
168
|
+
scope.params(FILL_PARAMS, { count: n, value, mode, pad0: 0 });
|
|
169
|
+
if (first) {
|
|
170
|
+
const iota = fillFor(0, 1);
|
|
171
|
+
fill.dispatch(pass, fill.bind({ dst: comp, P: iota.binding }), nodePlan, [iota.offset]);
|
|
172
|
+
const none = fillFor(INVALID_INDEX, 0);
|
|
173
|
+
fill.dispatch(pass, fill.bind({ dst: treeEdge, P: none.binding }), nodePlan, [none.offset]);
|
|
174
|
+
}
|
|
175
|
+
const reset = fillFor(U32_MAX, 0);
|
|
176
|
+
const edgeParams = scope.params(MST_PARAMS, { count: m, counterIndex: 0, pad0: 0, pad1: 0 });
|
|
177
|
+
const compressParams = scope.params(WCC_PARAMS, {
|
|
178
|
+
n,
|
|
179
|
+
items: n,
|
|
180
|
+
stride: compressPlan.stride ?? n,
|
|
181
|
+
r: 0,
|
|
182
|
+
flagIndex: 0,
|
|
183
|
+
giant: U32_MAX,
|
|
184
|
+
maxSteps: COMPRESS_STEPS,
|
|
185
|
+
pad0: 0,
|
|
186
|
+
});
|
|
187
|
+
const graph = {
|
|
188
|
+
edgeSrc: edges.bindings.src,
|
|
189
|
+
edgeDst: edges.bindings.dst,
|
|
190
|
+
edgeWeight: edgeWeight ?? edges.bindings.src,
|
|
191
|
+
};
|
|
192
|
+
for (let i = 0; i < BORUVKA_ROUNDS_PER_SUBMIT; i++) {
|
|
193
|
+
for (const dst of [bestKey, bestEdge]) {
|
|
194
|
+
fill.dispatch(pass, fill.bind({ dst, P: reset.binding }), nodePlan, [reset.offset]);
|
|
195
|
+
}
|
|
196
|
+
for (const kernel of [minKey, minEdge]) {
|
|
197
|
+
kernel.dispatch(
|
|
198
|
+
pass,
|
|
199
|
+
kernel.bind({ ...graph, comp, bestKey, bestEdge, P: edgeParams.binding }),
|
|
200
|
+
edgePlan,
|
|
201
|
+
[edgeParams.offset],
|
|
202
|
+
);
|
|
203
|
+
}
|
|
204
|
+
const linkParams = scope.params(MST_PARAMS, { count: n, counterIndex: i, pad0: 0, pad1: 0 });
|
|
205
|
+
link.dispatch(
|
|
206
|
+
pass,
|
|
207
|
+
link.bind({
|
|
208
|
+
edgeSrc: graph.edgeSrc,
|
|
209
|
+
edgeDst: graph.edgeDst,
|
|
210
|
+
bestEdge,
|
|
211
|
+
comp,
|
|
212
|
+
treeEdge,
|
|
213
|
+
counters,
|
|
214
|
+
P: linkParams.binding,
|
|
215
|
+
}),
|
|
216
|
+
nodePlan,
|
|
217
|
+
[linkParams.offset],
|
|
218
|
+
);
|
|
219
|
+
for (let j = 0; j < passes; j++) {
|
|
220
|
+
compress.dispatch(pass, compress.bind({ comp, P: compressParams.binding }), compressPlan, [
|
|
221
|
+
compressParams.offset,
|
|
222
|
+
]);
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
batch.endPass();
|
|
226
|
+
const countsRequest = batch.readback(counters.buffer, 0, 4 * BORUVKA_ROUNDS_PER_SUBMIT);
|
|
227
|
+
scope.flush();
|
|
228
|
+
const submitted = batch.submit();
|
|
229
|
+
const bytes = await submitted.readback;
|
|
230
|
+
ctx.assertReady();
|
|
231
|
+
first = false;
|
|
232
|
+
if (options?.signal?.aborted) {
|
|
233
|
+
throw new WebGpuGraphError("E_ABORTED", `${ALGORITHM}: the signal was aborted`, {
|
|
234
|
+
batchId: submitted.id,
|
|
235
|
+
});
|
|
236
|
+
}
|
|
237
|
+
const counts = new Uint32Array(bytes, countsRequest.offset, BORUVKA_ROUNDS_PER_SUBMIT);
|
|
238
|
+
let settled = false;
|
|
239
|
+
for (const count of counts) {
|
|
240
|
+
recorded += count;
|
|
241
|
+
settled ||= count === 0;
|
|
242
|
+
}
|
|
243
|
+
rounds += BORUVKA_ROUNDS_PER_SUBMIT;
|
|
244
|
+
if (recorded > n - 1) {
|
|
245
|
+
throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM}: ${recorded} forest edges on ${n} nodes`, {
|
|
246
|
+
label: `${ALGORITHM}/counters`,
|
|
247
|
+
message: "a spanning forest has at most n - 1 edges",
|
|
248
|
+
});
|
|
249
|
+
}
|
|
250
|
+
if (settled) {
|
|
251
|
+
break;
|
|
252
|
+
}
|
|
253
|
+
if (rounds >= MAX_ROUNDS) {
|
|
254
|
+
throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM}: still merging after ${rounds} rounds`, {
|
|
255
|
+
label: `${ALGORITHM}/rounds`,
|
|
256
|
+
message: "Boruvka halves the components every round; the device did not",
|
|
257
|
+
});
|
|
258
|
+
}
|
|
259
|
+
batch = new CommandBatch(ctx, `${ALGORITHM}/rounds`);
|
|
260
|
+
}
|
|
261
|
+
const raw = new Uint32Array(n);
|
|
262
|
+
await ctx.readback.read(treeEdge.buffer, 4 * n, raw);
|
|
263
|
+
ctx.assertReady();
|
|
264
|
+
options?.onProgress?.(1, 1);
|
|
265
|
+
return resultOf(s, raw, recorded);
|
|
266
|
+
} finally {
|
|
267
|
+
scope.dispose();
|
|
268
|
+
}
|
|
269
|
+
}
|
package/src/constants.ts
CHANGED
|
@@ -295,9 +295,14 @@ export const LABEL_PROP_PASSES_PER_SUBMIT = 8;
|
|
|
295
295
|
* per submit. Each readback is a device-to-host synchronisation that costs about 2 ms in Chromium, and at one per
|
|
296
296
|
* round the syncs are 61 % of the 100,000-node call; four rounds per submit amortise them over O(log n) rounds, moving
|
|
297
297
|
* the Chromium crossover from 6,000 to 4,600 nodes (design/decisions/2026-09-26-which-algorithms-earn-the-gpu.md).
|
|
298
|
-
*
|
|
298
|
+
* `src/algorithms/mst.ts` reads it.
|
|
299
299
|
*/
|
|
300
300
|
export const BORUVKA_ROUNDS_PER_SUBMIT = 4;
|
|
301
|
+
/**
|
|
302
|
+
* The sign bit of an f32 bit pattern; interpolated into the prelude as `F32_SIGN_BIT` for `order_key`, the
|
|
303
|
+
* order-preserving u32 key of an f32 that Boruvka's `atomicMin` over edge weights runs on (P11 PD-6).
|
|
304
|
+
*/
|
|
305
|
+
export const F32_SIGN_BIT = 0x80000000;
|
|
301
306
|
/** The per-row group-by-key (design 8.6): a row of at most this many arcs is grouped by one thread in registers; a longer row by a workgroup over a global open-addressing region. */
|
|
302
307
|
export const GROUP_ROW_THREAD_MAX = 32;
|
|
303
308
|
/** The largest row the thread tier accepts when a caller forces the tier: its pairwise scan is about d^2 / 2 loop steps, and llvmpipe stops every loop of an invocation after 65,535 steps in total. */
|
package/src/index.ts
CHANGED
|
@@ -67,6 +67,7 @@ export { sssp } from "./algorithms/sssp.js";
|
|
|
67
67
|
export { allPairsShortestPath } from "./algorithms/all-pairs.js";
|
|
68
68
|
// ==================== algorithms (P11: structure and community, design 3.3 lines 806-807, 8.5, 8.6)
|
|
69
69
|
export { labelPropagation } from "./algorithms/label-propagation.js";
|
|
70
|
+
export { minimumSpanningTree } from "./algorithms/mst.js";
|
|
70
71
|
export { triangleCount } from "./algorithms/triangles.js";
|
|
71
72
|
|
|
72
73
|
// ==================== layouts and the accelerator (P3; the two P5 factories; P4's calibrateLayout, spec 2.2)
|
|
@@ -98,6 +99,7 @@ export type {
|
|
|
98
99
|
LabelResultLike,
|
|
99
100
|
LayoutAccelerator,
|
|
100
101
|
LayoutSimulation,
|
|
102
|
+
MstOptions,
|
|
101
103
|
MstResultLike,
|
|
102
104
|
PageRankResultLike,
|
|
103
105
|
ScoresResultLike,
|
|
@@ -108,7 +110,7 @@ export type {
|
|
|
108
110
|
// ==================== types: the P8 traversal results (spec 3.3 lines 830-832, 9.7); the option types are the seam's
|
|
109
111
|
// BfsOptions / SsspOptions / HitsOptionsLike above (P8 PD-19)
|
|
110
112
|
export type { LabelPropagationOptions } from "./types/community.js";
|
|
111
|
-
export type { GpuTriangleResult } from "./types/structure.js";
|
|
113
|
+
export type { GpuMstResult, GpuTriangleResult } from "./types/structure.js";
|
|
112
114
|
export type { GpuBellmanFordResult, GpuBfsResult, GpuSsspResult } from "./types/traversal.js";
|
|
113
115
|
|
|
114
116
|
// ==================== types: the betweenness results (spec 3.3 lines 833-834); the option type is the seam's
|
package/src/kernel/prelude.ts
CHANGED
|
@@ -15,6 +15,7 @@ import {
|
|
|
15
15
|
APSP_TILE,
|
|
16
16
|
EXACT_TILES_PER_PASS,
|
|
17
17
|
F32_INF_BITS,
|
|
18
|
+
F32_SIGN_BIT,
|
|
18
19
|
FA2_COINCIDENT_SQ,
|
|
19
20
|
FA2_DISTANCE_FLOOR,
|
|
20
21
|
FA2_DISTANCE_FLOOR_SQ,
|
|
@@ -57,6 +58,7 @@ export const PRELUDE_WGSL: string = /* wgsl */ `// ---- prelude: constants, stan
|
|
|
57
58
|
const INVALID_INDEX: u32 = ${INVALID_INDEX}u;
|
|
58
59
|
const U32_MAX: u32 = ${U32_MAX}u;
|
|
59
60
|
const F32_INF_BITS: u32 = ${F32_INF_BITS}u;
|
|
61
|
+
const F32_SIGN_BIT: u32 = ${F32_SIGN_BIT}u;
|
|
60
62
|
const MAX_WORKGROUPS_PER_DIM: u32 = ${MAX_WORKGROUPS_PER_DIM}u;
|
|
61
63
|
const EXACT_TILES_PER_PASS: u32 = ${EXACT_TILES_PER_PASS}u;
|
|
62
64
|
const FA2_DIST_FLOOR: f32 = ${wgslF32Literal(FA2_DISTANCE_FLOOR)};
|
|
@@ -95,6 +97,11 @@ fn lowbias32(x0: u32) -> u32 {
|
|
|
95
97
|
fn mask_bit(w: u32, i: u32) -> bool { return ((w >> (i & 31u)) & 1u) == 1u; }
|
|
96
98
|
fn unpack_u8(w: u32, i: u32) -> u32 { return (w >> (8u * (i & 3u))) & 0xFFu; }
|
|
97
99
|
fn pair_hash(i: u32, j: u32) -> u32 { return lowbias32((min(i, j) * 0x9E3779B9u) ^ max(i, j)); }
|
|
100
|
+
fn order_key(w: f32) -> u32 {
|
|
101
|
+
let raw = bitcast<u32>(w);
|
|
102
|
+
let bits = select(raw, 0u, raw == F32_SIGN_BIT);
|
|
103
|
+
return select(bits | F32_SIGN_BIT, ~bits, (bits & F32_SIGN_BIT) != 0u);
|
|
104
|
+
}
|
|
98
105
|
fn hash_unit(h: u32) -> f32 { return f32(h >> 8u) * (1.0 / 16777216.0); }
|
|
99
106
|
fn hash_dir(h: u32, dim: u32) -> vec3f {
|
|
100
107
|
let phi = 6.283185307179586 * hash_unit(h);
|
package/src/kernels.ts
CHANGED
|
@@ -17,7 +17,8 @@
|
|
|
17
17
|
* (design 8.7) adds apsp-init and apsp-fw with the ApspParams block. P11 (the structure and community phase, plan
|
|
18
18
|
* design/webgpu/plans/2026-09-23-webgpu-p11-structure-and-community.md) adds the graph build on the device (coo-emit,
|
|
19
19
|
* run-flags, coo-scatter), the per-row group-by-key (group-by-key-row), label propagation's step (lpa-step) and
|
|
20
|
-
* triangle counting (orient-flags, tri-intersect)
|
|
20
|
+
* triangle counting (orient-flags, tri-intersect), and Boruvka's minimum spanning tree adds mst-best and mst-link
|
|
21
|
+
* with the MstParams block. This file is the only importer of src/wgsl/** (spec 3.2;
|
|
21
22
|
* test/layers.test.ts).
|
|
22
23
|
*/
|
|
23
24
|
|
|
@@ -70,6 +71,8 @@ import { groupByKeyRowWgsl } from "./wgsl/group-by-key-row.wgsl.js";
|
|
|
70
71
|
import { histogramWgsl } from "./wgsl/histogram.wgsl.js";
|
|
71
72
|
import { indirectFinalizeWgsl } from "./wgsl/indirect-finalize.wgsl.js";
|
|
72
73
|
import { lpaStepWgsl } from "./wgsl/lpa-step.wgsl.js";
|
|
74
|
+
import { mstBestWgsl } from "./wgsl/mst-best.wgsl.js";
|
|
75
|
+
import { mstLinkWgsl } from "./wgsl/mst-link.wgsl.js";
|
|
73
76
|
import { orientFlagsWgsl } from "./wgsl/orient-flags.wgsl.js";
|
|
74
77
|
import { prFinalizeWgsl } from "./wgsl/pr-finalize.wgsl.js";
|
|
75
78
|
import { prScaleWgsl } from "./wgsl/pr-scale.wgsl.js";
|
|
@@ -151,7 +154,9 @@ export type KernelId =
|
|
|
151
154
|
| "orient-flags"
|
|
152
155
|
| "tri-intersect"
|
|
153
156
|
| "group-by-key-row"
|
|
154
|
-
| "lpa-step"
|
|
157
|
+
| "lpa-step"
|
|
158
|
+
| "mst-best"
|
|
159
|
+
| "mst-link";
|
|
155
160
|
|
|
156
161
|
/** One registry entry: everything of a WgslModuleSpec except the per-variant overrides and snippets. */
|
|
157
162
|
export interface KernelEntry {
|
|
@@ -566,6 +571,14 @@ export const LPA_PARAMS: UniformBlock = UniformBlock.define("LpaParams", [
|
|
|
566
571
|
["pad0", "u32"],
|
|
567
572
|
]);
|
|
568
573
|
|
|
574
|
+
/** `MstParams` (uniform, 16 B; P11): `count` @0 (the edges of `mst-best`, the vertices of `mst-link`), `counterIndex` @4 (the word of `counters` that receives the round's recorded edges), `pad0` @8, `pad1` @12. */
|
|
575
|
+
export const MST_PARAMS: UniformBlock = UniformBlock.define("MstParams", [
|
|
576
|
+
["count", "u32"],
|
|
577
|
+
["counterIndex", "u32"],
|
|
578
|
+
["pad0", "u32"],
|
|
579
|
+
["pad1", "u32"],
|
|
580
|
+
]);
|
|
581
|
+
|
|
569
582
|
// ---- the entries (contract 3.10.1; group 0 = graph, 1 = state, 2 = params, 3 = cold)
|
|
570
583
|
|
|
571
584
|
/**
|
|
@@ -1747,6 +1760,51 @@ const LPA_STEP: KernelEntry = {
|
|
|
1747
1760
|
phase: "P11",
|
|
1748
1761
|
};
|
|
1749
1762
|
|
|
1763
|
+
/** `mst-best` (design 8.5; P11-T4, PD-7): one edge per invocation offers itself to both endpoint components; PASS 0 takes the minimum `order_key` of the weight, PASS 1 the minimum edge index among that key's edges; 6 storage bindings. */
|
|
1764
|
+
const MST_BEST: KernelEntry = {
|
|
1765
|
+
id: "mst-best",
|
|
1766
|
+
body: mstBestWgsl,
|
|
1767
|
+
entryPoint: "mst_best",
|
|
1768
|
+
bindings: [
|
|
1769
|
+
decl(1, 0, "edgeSrc", "storage-ro", "array<u32>"),
|
|
1770
|
+
decl(1, 1, "edgeDst", "storage-ro", "array<u32>"),
|
|
1771
|
+
decl(1, 2, "edgeWeight", "storage-ro", "array<f32>"),
|
|
1772
|
+
decl(1, 3, "comp", "storage-ro", "array<u32>"),
|
|
1773
|
+
decl(1, 4, "bestKey", "storage", "array<atomic<u32>>"),
|
|
1774
|
+
decl(1, 5, "bestEdge", "storage", "array<atomic<u32>>"),
|
|
1775
|
+
decl(2, 0, "P", "uniform", "MstParams"),
|
|
1776
|
+
],
|
|
1777
|
+
overrideDecls: [
|
|
1778
|
+
{ name: "PASS", type: "u32", default: 0 },
|
|
1779
|
+
{ name: "WEIGHTED", type: "bool", default: false },
|
|
1780
|
+
],
|
|
1781
|
+
uniforms: [MST_PARAMS],
|
|
1782
|
+
needs: [],
|
|
1783
|
+
snippetSlots: [],
|
|
1784
|
+
phase: "P11",
|
|
1785
|
+
};
|
|
1786
|
+
|
|
1787
|
+
/** `mst-link` (design 8.5; P11-T4): a component root hooks onto the component its best edge reaches and records the edge, the lower root of a two-cycle staying; one atomic per workgroup counts the recorded edges; 6 storage bindings. */
|
|
1788
|
+
const MST_LINK: KernelEntry = {
|
|
1789
|
+
id: "mst-link",
|
|
1790
|
+
body: mstLinkWgsl,
|
|
1791
|
+
entryPoint: "mst_link",
|
|
1792
|
+
bindings: [
|
|
1793
|
+
decl(1, 0, "edgeSrc", "storage-ro", "array<u32>"),
|
|
1794
|
+
decl(1, 1, "edgeDst", "storage-ro", "array<u32>"),
|
|
1795
|
+
decl(1, 2, "bestEdge", "storage-ro", "array<u32>"),
|
|
1796
|
+
decl(1, 3, "comp", "storage", "array<atomic<u32>>"),
|
|
1797
|
+
decl(1, 4, "treeEdge", "storage", "array<u32>"),
|
|
1798
|
+
decl(1, 5, "counters", "storage", "array<atomic<u32>>"),
|
|
1799
|
+
decl(2, 0, "P", "uniform", "MstParams"),
|
|
1800
|
+
],
|
|
1801
|
+
overrideDecls: [],
|
|
1802
|
+
uniforms: [MST_PARAMS],
|
|
1803
|
+
needs: [],
|
|
1804
|
+
snippetSlots: [],
|
|
1805
|
+
phase: "P11",
|
|
1806
|
+
};
|
|
1807
|
+
|
|
1750
1808
|
/**
|
|
1751
1809
|
* The entries by id, in dispatch order. PLAN DECISION: `KernelId` is declared in full (contract 3.10) while the
|
|
1752
1810
|
* entries landed phase by phase, so the table is built as a Partial record and exported below through the
|
|
@@ -1758,7 +1816,8 @@ const LPA_STEP: KernelEntry = {
|
|
|
1758
1816
|
* P8-T7 `"bfs-fused"`, P8-T8 `"bfs-bottom-up"`, `"bfs-bitset-build"` and `"bfs-unvisited-flags"`, P8-T9
|
|
1759
1817
|
* `"sssp-relax"`, P8-T10 `"bf-relax"` and P8-T11 `"closeness-sweep"` and `"closeness-reduce"`, betweenness the
|
|
1760
1818
|
* six `"bc-*"` entries, all-pairs shortest paths `"apsp-init"` and `"apsp-fw"`, and P11 its seven (the graph
|
|
1761
|
-
* build, the group-by-key, label propagation's step and triangle counting)
|
|
1819
|
+
* build, the group-by-key, label propagation's step and triangle counting) plus Boruvka's `"mst-best"` and
|
|
1820
|
+
* `"mst-link"`, so every member of `KernelId`
|
|
1762
1821
|
* is present and the assertion is exact.
|
|
1763
1822
|
*/
|
|
1764
1823
|
const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze({
|
|
@@ -1823,6 +1882,8 @@ const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze
|
|
|
1823
1882
|
"tri-intersect": TRI_INTERSECT,
|
|
1824
1883
|
"group-by-key-row": GROUP_BY_KEY_ROW,
|
|
1825
1884
|
"lpa-step": LPA_STEP,
|
|
1885
|
+
"mst-best": MST_BEST,
|
|
1886
|
+
"mst-link": MST_LINK,
|
|
1826
1887
|
});
|
|
1827
1888
|
|
|
1828
1889
|
/** THE registry (spec 3.5): every entry, keyed by id. */
|
package/src/types/accelerator.ts
CHANGED
|
@@ -20,6 +20,7 @@ import type {
|
|
|
20
20
|
HitsOptionsLike,
|
|
21
21
|
HitsResultLike,
|
|
22
22
|
LabelResultLike,
|
|
23
|
+
MstOptions,
|
|
23
24
|
MstResultLike,
|
|
24
25
|
PageRankResultLike,
|
|
25
26
|
ScoresResultLike,
|
|
@@ -52,7 +53,7 @@ import type {
|
|
|
52
53
|
SpringElectricalStats,
|
|
53
54
|
} from "./layout.js";
|
|
54
55
|
import type { ForceAtlas2Options, FruchtermanReingoldOptions, SpringElectricalOptions } from "./options.js";
|
|
55
|
-
import type { GpuTriangleResult } from "./structure.js";
|
|
56
|
+
import type { GpuMstResult, GpuTriangleResult } from "./structure.js";
|
|
56
57
|
import type { GpuBellmanFordResult, GpuBfsResult, GpuSsspResult } from "./traversal.js";
|
|
57
58
|
|
|
58
59
|
// ---- the real @graphty/layout interfaces (spec 9.3, D27): imported at W1b, re-exported so the package's public
|
|
@@ -80,6 +81,7 @@ export type {
|
|
|
80
81
|
HitsOptionsLike,
|
|
81
82
|
HitsResultLike,
|
|
82
83
|
LabelResultLike,
|
|
84
|
+
MstOptions,
|
|
83
85
|
MstResultLike,
|
|
84
86
|
PageRankResultLike,
|
|
85
87
|
ScoresResultLike,
|
|
@@ -120,8 +122,9 @@ export interface AcceleratorOptions {
|
|
|
120
122
|
* betweenness members take the seam's `BetweennessAcceleratorOptions`. `allPairsShortestPath` (design 8.7) takes the
|
|
121
123
|
* seam's `SsspOptions` too and refuses both of its keys. P11 adds `triangleCount` (its result carries `coefficient`
|
|
122
124
|
* and `transitivity` beyond the seam's `{ perNode, total }`, which a wider object satisfies) and `labelPropagation`
|
|
123
|
-
* (the seam's `HitsOptionsLike`: `maxIterations` and `weighted` honoured, `tolerance` refused)
|
|
124
|
-
*
|
|
125
|
+
* (the seam's `HitsOptionsLike`: `maxIterations` and `weighted` honoured, `tolerance` refused), and
|
|
126
|
+
* `minimumSpanningTree` (Boruvka; the seam's `MstOptions`, whose per-arc `weights` override is refused). Later phases
|
|
127
|
+
* add one member per shipped algorithm.
|
|
125
128
|
* Exported: implemented by src/accelerator.ts (P3-T3); re-exported from src/index.ts at P3-T3.
|
|
126
129
|
* @public
|
|
127
130
|
*/
|
|
@@ -156,6 +159,7 @@ export interface GpuAccelerator extends AlgorithmAccelerator, LayoutAccelerator
|
|
|
156
159
|
allPairsShortestPath(s: GraphSnapshot, options?: SsspOptions): Promise<GpuApspResult>;
|
|
157
160
|
triangleCount(s: GraphSnapshot): Promise<GpuTriangleResult>;
|
|
158
161
|
labelPropagation(s: GraphSnapshot, options?: HitsOptionsLike): Promise<GpuLabelResult>;
|
|
162
|
+
minimumSpanningTree(s: GraphSnapshot, options?: MstOptions): Promise<GpuMstResult>;
|
|
159
163
|
release(s: GraphSnapshot): void;
|
|
160
164
|
dispose(): void;
|
|
161
165
|
}
|
package/src/types/structure.ts
CHANGED
|
@@ -26,3 +26,21 @@ export interface GpuTriangleResult {
|
|
|
26
26
|
/** The graph's transitivity, `3 x triangles / connected triples`; 0 when the graph has no connected triple. */
|
|
27
27
|
readonly transitivity: number;
|
|
28
28
|
}
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Boruvka's minimum spanning forest (design 3.3 line 808, 8.5): the result shape of design 3.3, `{ edges,
|
|
32
|
+
* totalWeight }`, which is the shape `@graphty/algorithms`' `kruskalMST` returns.
|
|
33
|
+
* @public
|
|
34
|
+
*/
|
|
35
|
+
export interface GpuMstResult {
|
|
36
|
+
/**
|
|
37
|
+
* The LOGICAL edge indices of the forest: one tree per connected component, so `nodeCount - components` edges.
|
|
38
|
+
* The forest is the one of the total edge order (weight, then edge index): on distinct weights it is THE minimum
|
|
39
|
+
* spanning forest, and on tied weights it is exactly the forest `kruskalMST` accepts, whose sort breaks ties by
|
|
40
|
+
* edge index too. The order of the array is not Kruskal's acceptance order: it is deterministic, grouped by the
|
|
41
|
+
* Boruvka round that chose each edge.
|
|
42
|
+
*/
|
|
43
|
+
readonly edges: U32;
|
|
44
|
+
/** The sum of the forest's edge weights (1 per edge on an unweighted snapshot), summed in f64 on the host. */
|
|
45
|
+
readonly totalWeight: number;
|
|
46
|
+
}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `mst-best` kernel body (design 8.5; the P11 plan's P11-T4, PD-6 and PD-7): one logical edge per invocation of
|
|
3
|
+
* a Boruvka round. An edge whose endpoints lie in different components (`comp` holds every vertex's root) offers
|
|
4
|
+
* itself to both components. `atomicMin` exists for u32 only and no atomic compares one word and writes another, so
|
|
5
|
+
* the per-component minimum of the total edge order (weight, then edge index) is TWO passes of this one body, chosen
|
|
6
|
+
* by the `PASS` override: pass 0 takes the minimum `order_key` of the weight into `bestKey`, pass 1 the minimum edge
|
|
7
|
+
* index among the edges that hold that key into `bestEdge`. `order_key` (the prelude) orders negative weights below
|
|
8
|
+
* positive ones and -0 equal to +0, which a raw bit pattern does not. Without `WEIGHTED` every edge weighs 1 and
|
|
9
|
+
* the index alone decides, as in Kruskal. Body only (spec 3.5, D9); normative text: a sabotage mutation is a textual
|
|
10
|
+
* edit of it.
|
|
11
|
+
*/
|
|
12
|
+
export const mstBestWgsl = /* wgsl */ `
|
|
13
|
+
@compute @workgroup_size(WG)
|
|
14
|
+
fn mst_best(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
15
|
+
let e = linear_id(wid, lid.x);
|
|
16
|
+
if (e >= P.count) { return; }
|
|
17
|
+
let cu = comp[edgeSrc[e]];
|
|
18
|
+
let cv = comp[edgeDst[e]];
|
|
19
|
+
if (cu == cv) { return; } // inside one component, or a self-loop
|
|
20
|
+
var w = 1.0;
|
|
21
|
+
if (WEIGHTED) { w = edgeWeight[e]; }
|
|
22
|
+
let k = order_key(w);
|
|
23
|
+
if (PASS == 0u) {
|
|
24
|
+
atomicMin(&bestKey[cu], k);
|
|
25
|
+
atomicMin(&bestKey[cv], k);
|
|
26
|
+
} else {
|
|
27
|
+
if (k == atomicLoad(&bestKey[cu])) { atomicMin(&bestEdge[cu], e); }
|
|
28
|
+
if (k == atomicLoad(&bestKey[cv])) { atomicMin(&bestEdge[cv], e); }
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
`;
|