@graphty/webgpu-graph-algorithms 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +68 -25
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-E6iKaeuJ.js → context-CRbw2Wyo.js} +178 -19
- package/dist/chunks/{context-E6iKaeuJ.js.map → context-CRbw2Wyo.js.map} +1 -1
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +9 -6
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +85 -6
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/components.d.ts +30 -0
- package/dist/src/algorithms/components.d.ts.map +1 -0
- package/dist/src/algorithms/components.js +300 -0
- package/dist/src/algorithms/components.js.map +1 -0
- package/dist/src/algorithms/pagerank.d.ts +39 -0
- package/dist/src/algorithms/pagerank.d.ts.map +1 -0
- package/dist/src/algorithms/pagerank.js +298 -0
- package/dist/src/algorithms/pagerank.js.map +1 -0
- package/dist/src/algorithms/power-iteration.d.ts +109 -0
- package/dist/src/algorithms/power-iteration.d.ts.map +1 -0
- package/dist/src/algorithms/power-iteration.js +206 -0
- package/dist/src/algorithms/power-iteration.js.map +1 -0
- package/dist/src/algorithms/scope.d.ts +26 -0
- package/dist/src/algorithms/scope.d.ts.map +1 -0
- package/dist/src/algorithms/scope.js +41 -0
- package/dist/src/algorithms/scope.js.map +1 -0
- package/dist/src/algorithms/spectral.d.ts +50 -0
- package/dist/src/algorithms/spectral.d.ts.map +1 -0
- package/dist/src/algorithms/spectral.js +247 -0
- package/dist/src/algorithms/spectral.js.map +1 -0
- package/dist/src/index.d.ts +4 -0
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +4 -0
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/dispatch.d.ts +4 -1
- package/dist/src/kernel/dispatch.d.ts.map +1 -1
- package/dist/src/kernel/dispatch.js +12 -5
- package/dist/src/kernel/dispatch.js.map +1 -1
- package/dist/src/kernels.d.ts +20 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +172 -2
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/memory/residency.d.ts.map +1 -1
- package/dist/src/memory/residency.js +164 -11
- package/dist/src/memory/residency.js.map +1 -1
- package/dist/src/primitives/core-shape.d.ts +41 -0
- package/dist/src/primitives/core-shape.d.ts.map +1 -0
- package/dist/src/primitives/core-shape.js +89 -0
- package/dist/src/primitives/core-shape.js.map +1 -0
- package/dist/src/primitives/segmented-reduce.d.ts.map +1 -1
- package/dist/src/primitives/segmented-reduce.js +4 -30
- package/dist/src/primitives/segmented-reduce.js.map +1 -1
- package/dist/src/primitives/spmv.d.ts +56 -0
- package/dist/src/primitives/spmv.d.ts.map +1 -0
- package/dist/src/primitives/spmv.js +101 -0
- package/dist/src/primitives/spmv.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +14 -2
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/algorithms.d.ts +73 -0
- package/dist/src/types/algorithms.d.ts.map +1 -0
- package/dist/src/types/algorithms.js +17 -0
- package/dist/src/types/algorithms.js.map +1 -0
- package/dist/src/wgsl/pr-finalize.wgsl.d.ts +11 -0
- package/dist/src/wgsl/pr-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/pr-finalize.wgsl.js +36 -0
- package/dist/src/wgsl/pr-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/pr-scale.wgsl.d.ts +14 -0
- package/dist/src/wgsl/pr-scale.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/pr-scale.wgsl.js +48 -0
- package/dist/src/wgsl/pr-scale.wgsl.js.map +1 -0
- package/dist/src/wgsl/spmv-pull.wgsl.d.ts +15 -0
- package/dist/src/wgsl/spmv-pull.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/spmv-pull.wgsl.js +47 -0
- package/dist/src/wgsl/spmv-pull.wgsl.js.map +1 -0
- package/dist/src/wgsl/wcc-compress.wgsl.d.ts +9 -0
- package/dist/src/wgsl/wcc-compress.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/wcc-compress.wgsl.js +26 -0
- package/dist/src/wgsl/wcc-compress.wgsl.js.map +1 -0
- package/dist/src/wgsl/wcc-link-edges.wgsl.d.ts +13 -0
- package/dist/src/wgsl/wcc-link-edges.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/wcc-link-edges.wgsl.js +46 -0
- package/dist/src/wgsl/wcc-link-edges.wgsl.js.map +1 -0
- package/dist/src/wgsl/wcc-link-sample.wgsl.d.ts +11 -0
- package/dist/src/wgsl/wcc-link-sample.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/wcc-link-sample.wgsl.js +45 -0
- package/dist/src/wgsl/wcc-link-sample.wgsl.js.map +1 -0
- package/dist/src/wgsl/wcc-sample.wgsl.d.ts +10 -0
- package/dist/src/wgsl/wcc-sample.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/wcc-sample.wgsl.js +18 -0
- package/dist/src/wgsl/wcc-sample.wgsl.js.map +1 -0
- package/dist/tsconfig.build.tsbuildinfo +1 -1
- package/dist/webgpu-graph-algorithms.js +1550 -29
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +3 -3
- package/src/accelerator.ts +101 -7
- package/src/algorithms/components.ts +348 -0
- package/src/algorithms/pagerank.ts +343 -0
- package/src/algorithms/power-iteration.ts +278 -0
- package/src/algorithms/scope.ts +52 -0
- package/src/algorithms/spectral.ts +300 -0
- package/src/index.ts +18 -0
- package/src/kernel/dispatch.ts +12 -5
- package/src/kernels.ts +206 -5
- package/src/memory/residency.ts +200 -11
- package/src/primitives/core-shape.ts +103 -0
- package/src/primitives/segmented-reduce.ts +4 -36
- package/src/primitives/spmv.ts +155 -0
- package/src/types/accelerator.ts +28 -2
- package/src/types/algorithms.ts +82 -0
- package/src/wgsl/pr-finalize.wgsl.ts +36 -0
- package/src/wgsl/pr-scale.wgsl.ts +48 -0
- package/src/wgsl/spmv-pull.wgsl.ts +47 -0
- package/src/wgsl/wcc-compress.wgsl.ts +26 -0
- package/src/wgsl/wcc-link-edges.wgsl.ts +46 -0
- package/src/wgsl/wcc-link-sample.wgsl.ts +45 -0
- package/src/wgsl/wcc-sample.wgsl.ts +18 -0
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The result and option records of the P7 algorithms (spec 3.3 lines 815-828, 9.7). The `Gpu*Result` shapes are the
|
|
3
|
+
* design's verbatim; the option records are this package's own, spelled MEMBER FOR MEMBER as the CPU seam spells
|
|
4
|
+
* them so one object literal satisfies both sides (the same D27 mirror rule src/types/options.ts follows for the
|
|
5
|
+
* layout options). Types only: this file imports nothing at runtime.
|
|
6
|
+
*
|
|
7
|
+
* The CPU counterparts, when phase M8a lands them (plan 2026-09-19-webgpu-m8a-algorithms-seam, Task M8a-T8):
|
|
8
|
+
* `PageRankOptions` here is `IndexedPageRankOptions` there (`{ dampingFactor?, maxIterations?, tolerance?,
|
|
9
|
+
* weighted? }`); `HitsOptions`, `EigenvectorOptions` and `KatzOptions` here all correspond to the ONE
|
|
10
|
+
* `HitsOptionsLike` there (`{ maxIterations?, tolerance?, weighted? }`), which is why every one of them carries
|
|
11
|
+
* those three members and `KatzOptions` adds `alpha` / `beta` on top; `ComponentsOptions` has no CPU counterpart at
|
|
12
|
+
* all, because `AlgorithmAccelerator.connectedComponents?(s: GraphSnapshot): Promise<LabelResultLike>` declares no
|
|
13
|
+
* options parameter. `weighted`, never `weight`: that is the member name graph-format design 14.2 fixes at
|
|
14
|
+
* `design/graph-format/graph-format-design.md:3892` and the one M8a ports.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import type { F32, U32 } from "@graphty/graph-format";
|
|
18
|
+
|
|
19
|
+
/** Spec 3.3 line 815: every score result carries `precision` so a consumer can label GPU scores (Q-24). */
|
|
20
|
+
export interface GpuScoresResult {
|
|
21
|
+
readonly scores: F32;
|
|
22
|
+
readonly iterations: number;
|
|
23
|
+
readonly converged: boolean;
|
|
24
|
+
readonly precision: "f32";
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/** Spec 3.3 line 816: `iterations` is the first iteration whose L1 delta fell below the tolerance (8.2), not the batch boundary. */
|
|
28
|
+
export interface GpuPageRankResult extends GpuScoresResult {
|
|
29
|
+
readonly danglingMass: number;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/** Spec 3.3 line 817. */
|
|
33
|
+
export interface GpuHitsResult {
|
|
34
|
+
readonly hubs: F32;
|
|
35
|
+
readonly authorities: F32;
|
|
36
|
+
readonly iterations: number;
|
|
37
|
+
readonly converged: boolean;
|
|
38
|
+
readonly precision: "f32";
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** Spec 3.3 line 818: labels dense 0..count-1 in first-seen order (renumberPartition); groups() is index-aligned. */
|
|
42
|
+
export interface GpuLabelResult {
|
|
43
|
+
readonly labels: U32;
|
|
44
|
+
readonly count: number;
|
|
45
|
+
groups(): U32[];
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/** The CPU seam's IndexedPageRankOptions, member for member (M8a Task M8a-T8; graph-format design 14.2 :3892). */
|
|
49
|
+
export interface PageRankOptions {
|
|
50
|
+
readonly dampingFactor?: number | undefined;
|
|
51
|
+
readonly maxIterations?: number | undefined;
|
|
52
|
+
readonly tolerance?: number | undefined;
|
|
53
|
+
readonly weighted?: boolean | undefined;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/** The CPU seam's HitsOptionsLike, member for member (M8a Task M8a-T8). */
|
|
57
|
+
export interface HitsOptions {
|
|
58
|
+
readonly maxIterations?: number | undefined;
|
|
59
|
+
readonly tolerance?: number | undefined;
|
|
60
|
+
readonly weighted?: boolean | undefined;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/** The CPU seam's HitsOptionsLike again: `eigenvectorCentrality` takes that same shape on the CPU side. */
|
|
64
|
+
export interface EigenvectorOptions {
|
|
65
|
+
readonly maxIterations?: number | undefined;
|
|
66
|
+
readonly tolerance?: number | undefined;
|
|
67
|
+
readonly weighted?: boolean | undefined;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** HitsOptionsLike plus Katz's own two: `alpha` is the attenuation and `beta` the constant term. */
|
|
71
|
+
export interface KatzOptions {
|
|
72
|
+
readonly alpha?: number | undefined;
|
|
73
|
+
readonly beta?: number | undefined;
|
|
74
|
+
readonly maxIterations?: number | undefined;
|
|
75
|
+
readonly tolerance?: number | undefined;
|
|
76
|
+
readonly weighted?: boolean | undefined;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/** GPU-only (spec 3.3 line 797: `renumber: true` by default, Q-12); the CPU seam's connectedComponents takes none. */
|
|
80
|
+
export interface ComponentsOptions {
|
|
81
|
+
readonly renumber?: boolean | undefined;
|
|
82
|
+
}
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `pr-finalize` kernel body (spec 8.2 dispatch (b)): ONE workgroup folds the `P.groups` per-workgroup partials
|
|
3
|
+
* into the header at `partials[0]` -- a STORAGE region, never a uniform, read by the next dispatch of the same
|
|
4
|
+
* pass -- and records `firstConverged` the first time the delta falls below `P.convergeThreshold`. NORM_MODE 2
|
|
5
|
+
* stores the square root of the folded norm (the L2 case). The recorded iteration is `P.iteration - 1u` because
|
|
6
|
+
* the delta a scale pass produces at iteration i is `|x(i-1) - x(i-2)|`, the error of iteration i - 1 (PD-9).
|
|
7
|
+
* The body is normative: a sabotage mutation is a textual edit of it, so it is not restyled.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
/** Entry point `pr_finalize`; override NORM_MODE (2 takes the square root of the folded norm, every other value stores it as folded). */
|
|
11
|
+
export const prFinalizeWgsl = /* wgsl */ `
|
|
12
|
+
@compute @workgroup_size(WG)
|
|
13
|
+
fn pr_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
14
|
+
var d = 0.0;
|
|
15
|
+
var e = 0.0;
|
|
16
|
+
var m = 0.0;
|
|
17
|
+
for (var g = lid.x; g < P.groups; g = g + WG) {
|
|
18
|
+
d = d + partials[1u + g].danglingMass;
|
|
19
|
+
e = e + partials[1u + g].delta;
|
|
20
|
+
m = m + partials[1u + g].norm;
|
|
21
|
+
}
|
|
22
|
+
let folded = wg_reduce_vec4(vec4f(d, e, m, 0.0), lid.x, 0u);
|
|
23
|
+
if (lid.x == 0u) {
|
|
24
|
+
partials[0].danglingMass = folded.x;
|
|
25
|
+
partials[0].delta = folded.y;
|
|
26
|
+
var norm = folded.z;
|
|
27
|
+
if (NORM_MODE == 2u) { norm = sqrt(max(0.0, folded.z)); }
|
|
28
|
+
partials[0].norm = norm;
|
|
29
|
+
partials[0].iteration = P.iteration;
|
|
30
|
+
let unset = partials[0].firstConverged == U32_MAX;
|
|
31
|
+
if (P.trackConvergence == 1u && P.iteration >= 2u && folded.y < P.convergeThreshold && unset) {
|
|
32
|
+
partials[0].firstConverged = P.iteration - 1u;
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
`;
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `pr-scale` kernel body (spec 8.2 dispatch (a)): one invocation per node writes `xNorm[u]` and contributes a
|
|
3
|
+
* per-workgroup partial of the dangling mass, the L1 delta `|rankIn - rankPrev|` and, for the spectral modes, the
|
|
4
|
+
* norm term. NORM_MODE selects the divisor: 0 PageRank (`rankIn[u] / outWeightSum[u]`, 0 and a dangling
|
|
5
|
+
* contribution when the sum is not positive); 1 and 2 are NORM PASSES that write no xNorm and only accumulate
|
|
6
|
+
* `abs(x)` (L1) or `x * x` (L2); 3 divides by the scalar `partials[0].norm` the previous dispatch folded; 4 is the
|
|
7
|
+
* identity (Katz). The body is normative: a sabotage mutation is a textual edit of it, so it is not restyled.
|
|
8
|
+
*
|
|
9
|
+
* The guard is named `inRange`, never `active`: `active` is a WGSL reserved word (spec 16.2) and the composer
|
|
10
|
+
* rejects it before a device is touched.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
/** Entry point `pr_scale`; override NORM_MODE (0 PageRank, 1 L1 norm pass, 2 L2 norm pass, 3 scale by partials[0].norm, 4 identity). */
|
|
14
|
+
export const prScaleWgsl = /* wgsl */ `
|
|
15
|
+
@compute @workgroup_size(WG)
|
|
16
|
+
fn pr_scale(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
17
|
+
let u = linear_id(wid, lid.x);
|
|
18
|
+
let inRange = u < P.n;
|
|
19
|
+
var x = 0.0;
|
|
20
|
+
var prev = 0.0;
|
|
21
|
+
if (inRange) { x = rankIn[u]; prev = rankPrev[u]; }
|
|
22
|
+
var dangling = 0.0;
|
|
23
|
+
var delta = 0.0;
|
|
24
|
+
var normTerm = 0.0;
|
|
25
|
+
if (inRange) {
|
|
26
|
+
delta = abs(x - prev);
|
|
27
|
+
if (NORM_MODE == 0u) {
|
|
28
|
+
let divisor = outWeightSum[u];
|
|
29
|
+
if (divisor <= 0.0) { dangling = x; xNorm[u] = 0.0; } else { xNorm[u] = x / divisor; }
|
|
30
|
+
}
|
|
31
|
+
if (NORM_MODE == 1u) { normTerm = abs(x); }
|
|
32
|
+
if (NORM_MODE == 2u) { normTerm = x * x; }
|
|
33
|
+
if (NORM_MODE == 3u) {
|
|
34
|
+
var scale = partials[0].norm;
|
|
35
|
+
if (scale <= 0.0) { scale = 1.0; }
|
|
36
|
+
xNorm[u] = x / scale;
|
|
37
|
+
}
|
|
38
|
+
if (NORM_MODE == 4u) { xNorm[u] = x; }
|
|
39
|
+
}
|
|
40
|
+
let folded = wg_reduce_vec4(vec4f(dangling, delta, normTerm, 0.0), lid.x, 0u);
|
|
41
|
+
if (lid.x == 0u) {
|
|
42
|
+
let slot = 1u + group_id(wid);
|
|
43
|
+
partials[slot].danglingMass = folded.x;
|
|
44
|
+
partials[slot].delta = folded.y;
|
|
45
|
+
partials[slot].norm = folded.z;
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
`;
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `spmv-pull` kernel body (spec 6 row 9, 8.2; PD-1 of the M8b plan): one invocation per row of the REVERSE
|
|
3
|
+
* adjacency, grid-stride over `[0, P.n)`, folding `weight * xNorm[nbr]` over the row's in-arcs in chunks of 64
|
|
4
|
+
* terms (a two-level f32 sum: the chunk absorbs the rounding of 64 terms, the row total the rounding of the chunk
|
|
5
|
+
* count, so a 10,000-arc hub row loses about 200 rounding steps instead of 10,000; Kahan compensation is not used
|
|
6
|
+
* because Metal's shader compiler folds `((acc + term) - acc) - term` to zero whatever hides it) and writing
|
|
7
|
+
* `rankOut[v] = beta * pv + alpha * (sum + danglingMass * pv)`, where `pv` is `personalization[v]` when
|
|
8
|
+
* HAS_PERSONALIZATION and the uniform `P.uniformP` otherwise. PageRank sets alpha to the
|
|
9
|
+
* damping factor, beta to `1 - alpha` and USE_DANGLING; HITS and eigenvector set alpha 1, beta 0, uniformP 0; Katz
|
|
10
|
+
* sets alpha to the attenuation, beta to its constant and uniformP 1. The body is normative: a sabotage mutation
|
|
11
|
+
* (test/helpers/sabotage.ts) is a textual edit of it, so it is not restyled.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
/** Entry point `spmv_pull`; overrides HAS_PERSONALIZATION and USE_DANGLING plus the standard USE_PERM / HAS_WEIGHTS. */
|
|
15
|
+
export const spmvPullWgsl = /* wgsl */ `
|
|
16
|
+
@compute @workgroup_size(WG)
|
|
17
|
+
fn spmv_pull(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
18
|
+
var dangling = 0.0;
|
|
19
|
+
if (USE_DANGLING) { dangling = partials[0].danglingMass; }
|
|
20
|
+
let first = linear_id(wid, lid.x);
|
|
21
|
+
for (var row = first; row < P.n; row = row + P.stride) {
|
|
22
|
+
let v = select(row, perm[row], USE_PERM);
|
|
23
|
+
let a0 = max(rowPtr[v], P.arcBase);
|
|
24
|
+
let a1 = min(rowPtr[v + 1u], P.arcEnd);
|
|
25
|
+
var acc = 0.0;
|
|
26
|
+
var chunk = 0.0;
|
|
27
|
+
var inChunk = 0u;
|
|
28
|
+
for (var arc = a0; arc < a1; arc = arc + 1u) {
|
|
29
|
+
let nbr = colIdx[arc - P.arcBase]; // \`target\` is a WGSL reserved word (spec 16.2)
|
|
30
|
+
var weight = 1.0;
|
|
31
|
+
if (HAS_WEIGHTS) { weight = weights[arc - P.arcBase]; }
|
|
32
|
+
// two-level sum: 64 terms into chunk, chunk into acc (see the header; no compensation, no select)
|
|
33
|
+
chunk = chunk + (weight * xNorm[nbr]);
|
|
34
|
+
inChunk = inChunk + 1u;
|
|
35
|
+
if (inChunk == 64u) {
|
|
36
|
+
acc = acc + chunk;
|
|
37
|
+
chunk = 0.0;
|
|
38
|
+
inChunk = 0u;
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
acc = acc + chunk;
|
|
42
|
+
var pv = P.uniformP;
|
|
43
|
+
if (HAS_PERSONALIZATION) { pv = personalization[v]; }
|
|
44
|
+
rankOut[v] = (P.beta * pv) + (P.alpha * (acc + (dangling * pv)));
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
`;
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `wcc-compress` kernel body (spec 8.3): pointer jumping to the root, reading through `atomicLoad` on the same
|
|
3
|
+
* `array<atomic<u32>>` because WGSL forbids mixing atomic and plain access to one element. The walk is bounded by
|
|
4
|
+
* `P.maxSteps`; a walk that runs out leaves a shorter path, which the next round finishes. The body is normative:
|
|
5
|
+
* a sabotage mutation is a textual edit of it, so it is not restyled.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
/** Entry point `wcc_compress`; no overrides. */
|
|
9
|
+
export const wccCompressWgsl = /* wgsl */ `
|
|
10
|
+
@compute @workgroup_size(WG)
|
|
11
|
+
fn wcc_compress(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
12
|
+
let first = linear_id(wid, lid.x);
|
|
13
|
+
for (var v = first; v < P.items; v = v + P.stride) {
|
|
14
|
+
var root = atomicLoad(&comp[v]);
|
|
15
|
+
var steps = 0u;
|
|
16
|
+
loop {
|
|
17
|
+
let parent = atomicLoad(&comp[root]);
|
|
18
|
+
if (parent == root) { break; }
|
|
19
|
+
if (steps >= P.maxSteps) { break; }
|
|
20
|
+
steps = steps + 1u;
|
|
21
|
+
root = parent;
|
|
22
|
+
}
|
|
23
|
+
atomicStore(&comp[v], root);
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
`;
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `wcc-link-edges` kernel body (spec 8.3): the each-edge-once link round of Afforest, correct for directed and
|
|
3
|
+
* undirected input alike because `edgeList()` yields every logical edge once in declared orientation (design 10.1).
|
|
4
|
+
* `link_pair` is the same GAP `Link` transcription as wcc-link-sample (each module is composed alone, so the helper
|
|
5
|
+
* is copied, not shared): all-u32 CAS on `comp`, an `array<atomic<u32>>` because WGSL forbids mixing atomic and
|
|
6
|
+
* plain access to one element, with a bounded retry loop (PD-5) and the changed flag at `P.flagIndex` inside the
|
|
7
|
+
* same array (PD-4). The `P.giant` guard is GAP's "skip the vertices already in the giant component" and is a pure
|
|
8
|
+
* optimisation -- linking two vertices already in one component is a no-op. The body is normative: a sabotage
|
|
9
|
+
* mutation is a textual edit of it, so it is not restyled.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
/** Entry point `wcc_link_edges`; no overrides. */
|
|
13
|
+
export const wccLinkEdgesWgsl = /* wgsl */ `
|
|
14
|
+
fn link_pair(a: u32, b: u32) {
|
|
15
|
+
var p1 = atomicLoad(&comp[a]);
|
|
16
|
+
var p2 = atomicLoad(&comp[b]);
|
|
17
|
+
var steps = 0u;
|
|
18
|
+
loop {
|
|
19
|
+
if (p1 == p2) { break; }
|
|
20
|
+
if (steps >= P.maxSteps) { atomicStore(&comp[P.flagIndex], 1u); break; }
|
|
21
|
+
steps = steps + 1u;
|
|
22
|
+
let hi = max(p1, p2);
|
|
23
|
+
let lo = min(p1, p2);
|
|
24
|
+
let pHigh = atomicLoad(&comp[hi]);
|
|
25
|
+
if (pHigh == lo) { break; }
|
|
26
|
+
if (pHigh == hi) {
|
|
27
|
+
let swapped = atomicCompareExchangeWeak(&comp[hi], hi, lo);
|
|
28
|
+
if (swapped.exchanged) { atomicStore(&comp[P.flagIndex], 1u); break; }
|
|
29
|
+
}
|
|
30
|
+
p1 = atomicLoad(&comp[atomicLoad(&comp[hi])]);
|
|
31
|
+
p2 = atomicLoad(&comp[lo]);
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
@compute @workgroup_size(WG)
|
|
36
|
+
fn wcc_link_edges(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
37
|
+
let first = linear_id(wid, lid.x);
|
|
38
|
+
for (var e = first; e < P.items; e = e + P.stride) {
|
|
39
|
+
let u = edgeSrc[e];
|
|
40
|
+
let v = edgeDst[e];
|
|
41
|
+
if (u == v) { continue; }
|
|
42
|
+
if (atomicLoad(&comp[u]) == P.giant && atomicLoad(&comp[v]) == P.giant) { continue; }
|
|
43
|
+
link_pair(u, v);
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
`;
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `wcc-link-sample` kernel body (spec 8.3): one of Afforest's sampled link rounds -- every vertex links its
|
|
3
|
+
* r-th neighbour, `colIdx[rowPtr[v] + P.r]`, when it has one. `link_pair` is GAP's `Link` (gapbs/cc.cc lines
|
|
4
|
+
* 40-150) transcribed for WGSL: all-u32 CAS on `comp`, which is `array<atomic<u32>>` because WGSL forbids mixing
|
|
5
|
+
* atomic and plain access to one element, with a bounded retry loop (PD-5). The changed flag is the word at
|
|
6
|
+
* `P.flagIndex` inside the same array (PD-4). The body is normative: a sabotage mutation is a textual edit of it,
|
|
7
|
+
* so it is not restyled.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
/** Entry point `wcc_link_sample`; standard USE_PERM / HAS_WEIGHTS only (the body reads neither weights nor a permutation beyond the row select). */
|
|
11
|
+
export const wccLinkSampleWgsl = /* wgsl */ `
|
|
12
|
+
fn link_pair(a: u32, b: u32) {
|
|
13
|
+
var p1 = atomicLoad(&comp[a]);
|
|
14
|
+
var p2 = atomicLoad(&comp[b]);
|
|
15
|
+
var steps = 0u;
|
|
16
|
+
loop {
|
|
17
|
+
if (p1 == p2) { break; }
|
|
18
|
+
if (steps >= P.maxSteps) { atomicStore(&comp[P.flagIndex], 1u); break; }
|
|
19
|
+
steps = steps + 1u;
|
|
20
|
+
let hi = max(p1, p2);
|
|
21
|
+
let lo = min(p1, p2);
|
|
22
|
+
let pHigh = atomicLoad(&comp[hi]);
|
|
23
|
+
if (pHigh == lo) { break; }
|
|
24
|
+
if (pHigh == hi) {
|
|
25
|
+
let swapped = atomicCompareExchangeWeak(&comp[hi], hi, lo);
|
|
26
|
+
if (swapped.exchanged) { atomicStore(&comp[P.flagIndex], 1u); break; }
|
|
27
|
+
}
|
|
28
|
+
p1 = atomicLoad(&comp[atomicLoad(&comp[hi])]);
|
|
29
|
+
p2 = atomicLoad(&comp[lo]);
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
@compute @workgroup_size(WG)
|
|
34
|
+
fn wcc_link_sample(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
35
|
+
let first = linear_id(wid, lid.x);
|
|
36
|
+
for (var row = first; row < P.items; row = row + P.stride) {
|
|
37
|
+
let v = select(row, perm[row], USE_PERM);
|
|
38
|
+
let a0 = rowPtr[v];
|
|
39
|
+
let a1 = rowPtr[v + 1u];
|
|
40
|
+
if (a0 + P.r < a1) {
|
|
41
|
+
link_pair(v, colIdx[a0 + P.r]);
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
`;
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `wcc-sample` kernel body (spec 8.3 "a 1,024-entry histogram readback to find the giant component"): writes
|
|
3
|
+
* the component label of `P.items` pseudo-randomly chosen vertices into `hist`, which the host reads back and takes
|
|
4
|
+
* the mode of (PD-12: GAP's SampleFrequentElement counts on the host too, and a device histogram over component
|
|
5
|
+
* ids would return a bucket, not an id). The sampler uses the prelude's `lowbias32` and `%`, never a bitwise
|
|
6
|
+
* operator on an index.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
/** Entry point `wcc_sample`; no overrides. */
|
|
10
|
+
export const wccSampleWgsl = /* wgsl */ `
|
|
11
|
+
@compute @workgroup_size(WG)
|
|
12
|
+
fn wcc_sample(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
13
|
+
let i = linear_id(wid, lid.x);
|
|
14
|
+
if (i >= P.items) { return; }
|
|
15
|
+
let v = lowbias32(i + P.r) % P.n;
|
|
16
|
+
hist[i] = atomicLoad(&comp[v]);
|
|
17
|
+
}
|
|
18
|
+
`;
|