@graphty/webgpu-graph-algorithms 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +68 -25
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-E6iKaeuJ.js → context-CRbw2Wyo.js} +178 -19
- package/dist/chunks/{context-E6iKaeuJ.js.map → context-CRbw2Wyo.js.map} +1 -1
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +9 -6
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +85 -6
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/components.d.ts +30 -0
- package/dist/src/algorithms/components.d.ts.map +1 -0
- package/dist/src/algorithms/components.js +300 -0
- package/dist/src/algorithms/components.js.map +1 -0
- package/dist/src/algorithms/pagerank.d.ts +39 -0
- package/dist/src/algorithms/pagerank.d.ts.map +1 -0
- package/dist/src/algorithms/pagerank.js +298 -0
- package/dist/src/algorithms/pagerank.js.map +1 -0
- package/dist/src/algorithms/power-iteration.d.ts +109 -0
- package/dist/src/algorithms/power-iteration.d.ts.map +1 -0
- package/dist/src/algorithms/power-iteration.js +206 -0
- package/dist/src/algorithms/power-iteration.js.map +1 -0
- package/dist/src/algorithms/scope.d.ts +26 -0
- package/dist/src/algorithms/scope.d.ts.map +1 -0
- package/dist/src/algorithms/scope.js +41 -0
- package/dist/src/algorithms/scope.js.map +1 -0
- package/dist/src/algorithms/spectral.d.ts +50 -0
- package/dist/src/algorithms/spectral.d.ts.map +1 -0
- package/dist/src/algorithms/spectral.js +247 -0
- package/dist/src/algorithms/spectral.js.map +1 -0
- package/dist/src/index.d.ts +4 -0
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +4 -0
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/dispatch.d.ts +4 -1
- package/dist/src/kernel/dispatch.d.ts.map +1 -1
- package/dist/src/kernel/dispatch.js +12 -5
- package/dist/src/kernel/dispatch.js.map +1 -1
- package/dist/src/kernels.d.ts +20 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +172 -2
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/memory/residency.d.ts.map +1 -1
- package/dist/src/memory/residency.js +164 -11
- package/dist/src/memory/residency.js.map +1 -1
- package/dist/src/primitives/core-shape.d.ts +41 -0
- package/dist/src/primitives/core-shape.d.ts.map +1 -0
- package/dist/src/primitives/core-shape.js +89 -0
- package/dist/src/primitives/core-shape.js.map +1 -0
- package/dist/src/primitives/segmented-reduce.d.ts.map +1 -1
- package/dist/src/primitives/segmented-reduce.js +4 -30
- package/dist/src/primitives/segmented-reduce.js.map +1 -1
- package/dist/src/primitives/spmv.d.ts +56 -0
- package/dist/src/primitives/spmv.d.ts.map +1 -0
- package/dist/src/primitives/spmv.js +101 -0
- package/dist/src/primitives/spmv.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +14 -2
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/algorithms.d.ts +73 -0
- package/dist/src/types/algorithms.d.ts.map +1 -0
- package/dist/src/types/algorithms.js +17 -0
- package/dist/src/types/algorithms.js.map +1 -0
- package/dist/src/wgsl/pr-finalize.wgsl.d.ts +11 -0
- package/dist/src/wgsl/pr-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/pr-finalize.wgsl.js +36 -0
- package/dist/src/wgsl/pr-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/pr-scale.wgsl.d.ts +14 -0
- package/dist/src/wgsl/pr-scale.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/pr-scale.wgsl.js +48 -0
- package/dist/src/wgsl/pr-scale.wgsl.js.map +1 -0
- package/dist/src/wgsl/spmv-pull.wgsl.d.ts +15 -0
- package/dist/src/wgsl/spmv-pull.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/spmv-pull.wgsl.js +47 -0
- package/dist/src/wgsl/spmv-pull.wgsl.js.map +1 -0
- package/dist/src/wgsl/wcc-compress.wgsl.d.ts +9 -0
- package/dist/src/wgsl/wcc-compress.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/wcc-compress.wgsl.js +26 -0
- package/dist/src/wgsl/wcc-compress.wgsl.js.map +1 -0
- package/dist/src/wgsl/wcc-link-edges.wgsl.d.ts +13 -0
- package/dist/src/wgsl/wcc-link-edges.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/wcc-link-edges.wgsl.js +46 -0
- package/dist/src/wgsl/wcc-link-edges.wgsl.js.map +1 -0
- package/dist/src/wgsl/wcc-link-sample.wgsl.d.ts +11 -0
- package/dist/src/wgsl/wcc-link-sample.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/wcc-link-sample.wgsl.js +45 -0
- package/dist/src/wgsl/wcc-link-sample.wgsl.js.map +1 -0
- package/dist/src/wgsl/wcc-sample.wgsl.d.ts +10 -0
- package/dist/src/wgsl/wcc-sample.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/wcc-sample.wgsl.js +18 -0
- package/dist/src/wgsl/wcc-sample.wgsl.js.map +1 -0
- package/dist/tsconfig.build.tsbuildinfo +1 -1
- package/dist/webgpu-graph-algorithms.js +1550 -29
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +3 -3
- package/src/accelerator.ts +101 -7
- package/src/algorithms/components.ts +348 -0
- package/src/algorithms/pagerank.ts +343 -0
- package/src/algorithms/power-iteration.ts +278 -0
- package/src/algorithms/scope.ts +52 -0
- package/src/algorithms/spectral.ts +300 -0
- package/src/index.ts +18 -0
- package/src/kernel/dispatch.ts +12 -5
- package/src/kernels.ts +206 -5
- package/src/memory/residency.ts +200 -11
- package/src/primitives/core-shape.ts +103 -0
- package/src/primitives/segmented-reduce.ts +4 -36
- package/src/primitives/spmv.ts +155 -0
- package/src/types/accelerator.ts +28 -2
- package/src/types/algorithms.ts +82 -0
- package/src/wgsl/pr-finalize.wgsl.ts +36 -0
- package/src/wgsl/pr-scale.wgsl.ts +48 -0
- package/src/wgsl/spmv-pull.wgsl.ts +47 -0
- package/src/wgsl/wcc-compress.wgsl.ts +26 -0
- package/src/wgsl/wcc-link-edges.wgsl.ts +46 -0
- package/src/wgsl/wcc-link-sample.wgsl.ts +45 -0
- package/src/wgsl/wcc-sample.wgsl.ts +18 -0
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `pr-finalize` kernel body (spec 8.2 dispatch (b)): ONE workgroup folds the `P.groups` per-workgroup partials
|
|
3
|
+
* into the header at `partials[0]` -- a STORAGE region, never a uniform, read by the next dispatch of the same
|
|
4
|
+
* pass -- and records `firstConverged` the first time the delta falls below `P.convergeThreshold`. NORM_MODE 2
|
|
5
|
+
* stores the square root of the folded norm (the L2 case). The recorded iteration is `P.iteration - 1u` because
|
|
6
|
+
* the delta a scale pass produces at iteration i is `|x(i-1) - x(i-2)|`, the error of iteration i - 1 (PD-9).
|
|
7
|
+
* The body is normative: a sabotage mutation is a textual edit of it, so it is not restyled.
|
|
8
|
+
*/
|
|
9
|
+
/** Entry point `pr_finalize`; override NORM_MODE (2 takes the square root of the folded norm, every other value stores it as folded). */
|
|
10
|
+
export const prFinalizeWgsl = /* wgsl */ `
|
|
11
|
+
@compute @workgroup_size(WG)
|
|
12
|
+
fn pr_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
13
|
+
var d = 0.0;
|
|
14
|
+
var e = 0.0;
|
|
15
|
+
var m = 0.0;
|
|
16
|
+
for (var g = lid.x; g < P.groups; g = g + WG) {
|
|
17
|
+
d = d + partials[1u + g].danglingMass;
|
|
18
|
+
e = e + partials[1u + g].delta;
|
|
19
|
+
m = m + partials[1u + g].norm;
|
|
20
|
+
}
|
|
21
|
+
let folded = wg_reduce_vec4(vec4f(d, e, m, 0.0), lid.x, 0u);
|
|
22
|
+
if (lid.x == 0u) {
|
|
23
|
+
partials[0].danglingMass = folded.x;
|
|
24
|
+
partials[0].delta = folded.y;
|
|
25
|
+
var norm = folded.z;
|
|
26
|
+
if (NORM_MODE == 2u) { norm = sqrt(max(0.0, folded.z)); }
|
|
27
|
+
partials[0].norm = norm;
|
|
28
|
+
partials[0].iteration = P.iteration;
|
|
29
|
+
let unset = partials[0].firstConverged == U32_MAX;
|
|
30
|
+
if (P.trackConvergence == 1u && P.iteration >= 2u && folded.y < P.convergeThreshold && unset) {
|
|
31
|
+
partials[0].firstConverged = P.iteration - 1u;
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
`;
|
|
36
|
+
//# sourceMappingURL=pr-finalize.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"pr-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/pr-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAEH,yIAAyI;AACzI,MAAM,CAAC,MAAM,cAAc,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;CAyBxC,CAAC"}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `pr-scale` kernel body (spec 8.2 dispatch (a)): one invocation per node writes `xNorm[u]` and contributes a
|
|
3
|
+
* per-workgroup partial of the dangling mass, the L1 delta `|rankIn - rankPrev|` and, for the spectral modes, the
|
|
4
|
+
* norm term. NORM_MODE selects the divisor: 0 PageRank (`rankIn[u] / outWeightSum[u]`, 0 and a dangling
|
|
5
|
+
* contribution when the sum is not positive); 1 and 2 are NORM PASSES that write no xNorm and only accumulate
|
|
6
|
+
* `abs(x)` (L1) or `x * x` (L2); 3 divides by the scalar `partials[0].norm` the previous dispatch folded; 4 is the
|
|
7
|
+
* identity (Katz). The body is normative: a sabotage mutation is a textual edit of it, so it is not restyled.
|
|
8
|
+
*
|
|
9
|
+
* The guard is named `inRange`, never `active`: `active` is a WGSL reserved word (spec 16.2) and the composer
|
|
10
|
+
* rejects it before a device is touched.
|
|
11
|
+
*/
|
|
12
|
+
/** Entry point `pr_scale`; override NORM_MODE (0 PageRank, 1 L1 norm pass, 2 L2 norm pass, 3 scale by partials[0].norm, 4 identity). */
|
|
13
|
+
export declare const prScaleWgsl = "\n@compute @workgroup_size(WG)\nfn pr_scale(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let u = linear_id(wid, lid.x);\n let inRange = u < P.n;\n var x = 0.0;\n var prev = 0.0;\n if (inRange) { x = rankIn[u]; prev = rankPrev[u]; }\n var dangling = 0.0;\n var delta = 0.0;\n var normTerm = 0.0;\n if (inRange) {\n delta = abs(x - prev);\n if (NORM_MODE == 0u) {\n let divisor = outWeightSum[u];\n if (divisor <= 0.0) { dangling = x; xNorm[u] = 0.0; } else { xNorm[u] = x / divisor; }\n }\n if (NORM_MODE == 1u) { normTerm = abs(x); }\n if (NORM_MODE == 2u) { normTerm = x * x; }\n if (NORM_MODE == 3u) {\n var scale = partials[0].norm;\n if (scale <= 0.0) { scale = 1.0; }\n xNorm[u] = x / scale;\n }\n if (NORM_MODE == 4u) { xNorm[u] = x; }\n }\n let folded = wg_reduce_vec4(vec4f(dangling, delta, normTerm, 0.0), lid.x, 0u);\n if (lid.x == 0u) {\n let slot = 1u + group_id(wid);\n partials[slot].danglingMass = folded.x;\n partials[slot].delta = folded.y;\n partials[slot].norm = folded.z;\n }\n}\n";
|
|
14
|
+
//# sourceMappingURL=pr-scale.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"pr-scale.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/pr-scale.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AAEH,wIAAwI;AACxI,eAAO,MAAM,WAAW,2sCAkCvB,CAAC"}
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `pr-scale` kernel body (spec 8.2 dispatch (a)): one invocation per node writes `xNorm[u]` and contributes a
|
|
3
|
+
* per-workgroup partial of the dangling mass, the L1 delta `|rankIn - rankPrev|` and, for the spectral modes, the
|
|
4
|
+
* norm term. NORM_MODE selects the divisor: 0 PageRank (`rankIn[u] / outWeightSum[u]`, 0 and a dangling
|
|
5
|
+
* contribution when the sum is not positive); 1 and 2 are NORM PASSES that write no xNorm and only accumulate
|
|
6
|
+
* `abs(x)` (L1) or `x * x` (L2); 3 divides by the scalar `partials[0].norm` the previous dispatch folded; 4 is the
|
|
7
|
+
* identity (Katz). The body is normative: a sabotage mutation is a textual edit of it, so it is not restyled.
|
|
8
|
+
*
|
|
9
|
+
* The guard is named `inRange`, never `active`: `active` is a WGSL reserved word (spec 16.2) and the composer
|
|
10
|
+
* rejects it before a device is touched.
|
|
11
|
+
*/
|
|
12
|
+
/** Entry point `pr_scale`; override NORM_MODE (0 PageRank, 1 L1 norm pass, 2 L2 norm pass, 3 scale by partials[0].norm, 4 identity). */
|
|
13
|
+
export const prScaleWgsl = /* wgsl */ `
|
|
14
|
+
@compute @workgroup_size(WG)
|
|
15
|
+
fn pr_scale(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
16
|
+
let u = linear_id(wid, lid.x);
|
|
17
|
+
let inRange = u < P.n;
|
|
18
|
+
var x = 0.0;
|
|
19
|
+
var prev = 0.0;
|
|
20
|
+
if (inRange) { x = rankIn[u]; prev = rankPrev[u]; }
|
|
21
|
+
var dangling = 0.0;
|
|
22
|
+
var delta = 0.0;
|
|
23
|
+
var normTerm = 0.0;
|
|
24
|
+
if (inRange) {
|
|
25
|
+
delta = abs(x - prev);
|
|
26
|
+
if (NORM_MODE == 0u) {
|
|
27
|
+
let divisor = outWeightSum[u];
|
|
28
|
+
if (divisor <= 0.0) { dangling = x; xNorm[u] = 0.0; } else { xNorm[u] = x / divisor; }
|
|
29
|
+
}
|
|
30
|
+
if (NORM_MODE == 1u) { normTerm = abs(x); }
|
|
31
|
+
if (NORM_MODE == 2u) { normTerm = x * x; }
|
|
32
|
+
if (NORM_MODE == 3u) {
|
|
33
|
+
var scale = partials[0].norm;
|
|
34
|
+
if (scale <= 0.0) { scale = 1.0; }
|
|
35
|
+
xNorm[u] = x / scale;
|
|
36
|
+
}
|
|
37
|
+
if (NORM_MODE == 4u) { xNorm[u] = x; }
|
|
38
|
+
}
|
|
39
|
+
let folded = wg_reduce_vec4(vec4f(dangling, delta, normTerm, 0.0), lid.x, 0u);
|
|
40
|
+
if (lid.x == 0u) {
|
|
41
|
+
let slot = 1u + group_id(wid);
|
|
42
|
+
partials[slot].danglingMass = folded.x;
|
|
43
|
+
partials[slot].delta = folded.y;
|
|
44
|
+
partials[slot].norm = folded.z;
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
`;
|
|
48
|
+
//# sourceMappingURL=pr-scale.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"pr-scale.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/pr-scale.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AAEH,wIAAwI;AACxI,MAAM,CAAC,MAAM,WAAW,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAkCrC,CAAC"}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `spmv-pull` kernel body (spec 6 row 9, 8.2; PD-1 of the M8b plan): one invocation per row of the REVERSE
|
|
3
|
+
* adjacency, grid-stride over `[0, P.n)`, folding `weight * xNorm[nbr]` over the row's in-arcs in chunks of 64
|
|
4
|
+
* terms (a two-level f32 sum: the chunk absorbs the rounding of 64 terms, the row total the rounding of the chunk
|
|
5
|
+
* count, so a 10,000-arc hub row loses about 200 rounding steps instead of 10,000; Kahan compensation is not used
|
|
6
|
+
* because Metal's shader compiler folds `((acc + term) - acc) - term` to zero whatever hides it) and writing
|
|
7
|
+
* `rankOut[v] = beta * pv + alpha * (sum + danglingMass * pv)`, where `pv` is `personalization[v]` when
|
|
8
|
+
* HAS_PERSONALIZATION and the uniform `P.uniformP` otherwise. PageRank sets alpha to the
|
|
9
|
+
* damping factor, beta to `1 - alpha` and USE_DANGLING; HITS and eigenvector set alpha 1, beta 0, uniformP 0; Katz
|
|
10
|
+
* sets alpha to the attenuation, beta to its constant and uniformP 1. The body is normative: a sabotage mutation
|
|
11
|
+
* (test/helpers/sabotage.ts) is a textual edit of it, so it is not restyled.
|
|
12
|
+
*/
|
|
13
|
+
/** Entry point `spmv_pull`; overrides HAS_PERSONALIZATION and USE_DANGLING plus the standard USE_PERM / HAS_WEIGHTS. */
|
|
14
|
+
export declare const spmvPullWgsl = "\n@compute @workgroup_size(WG)\nfn spmv_pull(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n var dangling = 0.0;\n if (USE_DANGLING) { dangling = partials[0].danglingMass; }\n let first = linear_id(wid, lid.x);\n for (var row = first; row < P.n; row = row + P.stride) {\n let v = select(row, perm[row], USE_PERM);\n let a0 = max(rowPtr[v], P.arcBase);\n let a1 = min(rowPtr[v + 1u], P.arcEnd);\n var acc = 0.0;\n var chunk = 0.0;\n var inChunk = 0u;\n for (var arc = a0; arc < a1; arc = arc + 1u) {\n let nbr = colIdx[arc - P.arcBase]; // `target` is a WGSL reserved word (spec 16.2)\n var weight = 1.0;\n if (HAS_WEIGHTS) { weight = weights[arc - P.arcBase]; }\n // two-level sum: 64 terms into chunk, chunk into acc (see the header; no compensation, no select)\n chunk = chunk + (weight * xNorm[nbr]);\n inChunk = inChunk + 1u;\n if (inChunk == 64u) {\n acc = acc + chunk;\n chunk = 0.0;\n inChunk = 0u;\n }\n }\n acc = acc + chunk;\n var pv = P.uniformP;\n if (HAS_PERSONALIZATION) { pv = personalization[v]; }\n rankOut[v] = (P.beta * pv) + (P.alpha * (acc + (dangling * pv)));\n }\n}\n";
|
|
15
|
+
//# sourceMappingURL=spmv-pull.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"spmv-pull.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/spmv-pull.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;GAWG;AAEH,wHAAwH;AACxH,eAAO,MAAM,YAAY,s2CAgCxB,CAAC"}
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `spmv-pull` kernel body (spec 6 row 9, 8.2; PD-1 of the M8b plan): one invocation per row of the REVERSE
|
|
3
|
+
* adjacency, grid-stride over `[0, P.n)`, folding `weight * xNorm[nbr]` over the row's in-arcs in chunks of 64
|
|
4
|
+
* terms (a two-level f32 sum: the chunk absorbs the rounding of 64 terms, the row total the rounding of the chunk
|
|
5
|
+
* count, so a 10,000-arc hub row loses about 200 rounding steps instead of 10,000; Kahan compensation is not used
|
|
6
|
+
* because Metal's shader compiler folds `((acc + term) - acc) - term` to zero whatever hides it) and writing
|
|
7
|
+
* `rankOut[v] = beta * pv + alpha * (sum + danglingMass * pv)`, where `pv` is `personalization[v]` when
|
|
8
|
+
* HAS_PERSONALIZATION and the uniform `P.uniformP` otherwise. PageRank sets alpha to the
|
|
9
|
+
* damping factor, beta to `1 - alpha` and USE_DANGLING; HITS and eigenvector set alpha 1, beta 0, uniformP 0; Katz
|
|
10
|
+
* sets alpha to the attenuation, beta to its constant and uniformP 1. The body is normative: a sabotage mutation
|
|
11
|
+
* (test/helpers/sabotage.ts) is a textual edit of it, so it is not restyled.
|
|
12
|
+
*/
|
|
13
|
+
/** Entry point `spmv_pull`; overrides HAS_PERSONALIZATION and USE_DANGLING plus the standard USE_PERM / HAS_WEIGHTS. */
|
|
14
|
+
export const spmvPullWgsl = /* wgsl */ `
|
|
15
|
+
@compute @workgroup_size(WG)
|
|
16
|
+
fn spmv_pull(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
17
|
+
var dangling = 0.0;
|
|
18
|
+
if (USE_DANGLING) { dangling = partials[0].danglingMass; }
|
|
19
|
+
let first = linear_id(wid, lid.x);
|
|
20
|
+
for (var row = first; row < P.n; row = row + P.stride) {
|
|
21
|
+
let v = select(row, perm[row], USE_PERM);
|
|
22
|
+
let a0 = max(rowPtr[v], P.arcBase);
|
|
23
|
+
let a1 = min(rowPtr[v + 1u], P.arcEnd);
|
|
24
|
+
var acc = 0.0;
|
|
25
|
+
var chunk = 0.0;
|
|
26
|
+
var inChunk = 0u;
|
|
27
|
+
for (var arc = a0; arc < a1; arc = arc + 1u) {
|
|
28
|
+
let nbr = colIdx[arc - P.arcBase]; // \`target\` is a WGSL reserved word (spec 16.2)
|
|
29
|
+
var weight = 1.0;
|
|
30
|
+
if (HAS_WEIGHTS) { weight = weights[arc - P.arcBase]; }
|
|
31
|
+
// two-level sum: 64 terms into chunk, chunk into acc (see the header; no compensation, no select)
|
|
32
|
+
chunk = chunk + (weight * xNorm[nbr]);
|
|
33
|
+
inChunk = inChunk + 1u;
|
|
34
|
+
if (inChunk == 64u) {
|
|
35
|
+
acc = acc + chunk;
|
|
36
|
+
chunk = 0.0;
|
|
37
|
+
inChunk = 0u;
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
acc = acc + chunk;
|
|
41
|
+
var pv = P.uniformP;
|
|
42
|
+
if (HAS_PERSONALIZATION) { pv = personalization[v]; }
|
|
43
|
+
rankOut[v] = (P.beta * pv) + (P.alpha * (acc + (dangling * pv)));
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
`;
|
|
47
|
+
//# sourceMappingURL=spmv-pull.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"spmv-pull.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/spmv-pull.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;GAWG;AAEH,wHAAwH;AACxH,MAAM,CAAC,MAAM,YAAY,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAgCtC,CAAC"}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `wcc-compress` kernel body (spec 8.3): pointer jumping to the root, reading through `atomicLoad` on the same
|
|
3
|
+
* `array<atomic<u32>>` because WGSL forbids mixing atomic and plain access to one element. The walk is bounded by
|
|
4
|
+
* `P.maxSteps`; a walk that runs out leaves a shorter path, which the next round finishes. The body is normative:
|
|
5
|
+
* a sabotage mutation is a textual edit of it, so it is not restyled.
|
|
6
|
+
*/
|
|
7
|
+
/** Entry point `wcc_compress`; no overrides. */
|
|
8
|
+
export declare const wccCompressWgsl = "\n@compute @workgroup_size(WG)\nfn wcc_compress(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let first = linear_id(wid, lid.x);\n for (var v = first; v < P.items; v = v + P.stride) {\n var root = atomicLoad(&comp[v]);\n var steps = 0u;\n loop {\n let parent = atomicLoad(&comp[root]);\n if (parent == root) { break; }\n if (steps >= P.maxSteps) { break; }\n steps = steps + 1u;\n root = parent;\n }\n atomicStore(&comp[v], root);\n }\n}\n";
|
|
9
|
+
//# sourceMappingURL=wcc-compress.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"wcc-compress.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/wcc-compress.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,gDAAgD;AAChD,eAAO,MAAM,eAAe,0kBAiB3B,CAAC"}
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `wcc-compress` kernel body (spec 8.3): pointer jumping to the root, reading through `atomicLoad` on the same
|
|
3
|
+
* `array<atomic<u32>>` because WGSL forbids mixing atomic and plain access to one element. The walk is bounded by
|
|
4
|
+
* `P.maxSteps`; a walk that runs out leaves a shorter path, which the next round finishes. The body is normative:
|
|
5
|
+
* a sabotage mutation is a textual edit of it, so it is not restyled.
|
|
6
|
+
*/
|
|
7
|
+
/** Entry point `wcc_compress`; no overrides. */
|
|
8
|
+
export const wccCompressWgsl = /* wgsl */ `
|
|
9
|
+
@compute @workgroup_size(WG)
|
|
10
|
+
fn wcc_compress(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
11
|
+
let first = linear_id(wid, lid.x);
|
|
12
|
+
for (var v = first; v < P.items; v = v + P.stride) {
|
|
13
|
+
var root = atomicLoad(&comp[v]);
|
|
14
|
+
var steps = 0u;
|
|
15
|
+
loop {
|
|
16
|
+
let parent = atomicLoad(&comp[root]);
|
|
17
|
+
if (parent == root) { break; }
|
|
18
|
+
if (steps >= P.maxSteps) { break; }
|
|
19
|
+
steps = steps + 1u;
|
|
20
|
+
root = parent;
|
|
21
|
+
}
|
|
22
|
+
atomicStore(&comp[v], root);
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
`;
|
|
26
|
+
//# sourceMappingURL=wcc-compress.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"wcc-compress.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/wcc-compress.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,gDAAgD;AAChD,MAAM,CAAC,MAAM,eAAe,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;CAiBzC,CAAC"}
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `wcc-link-edges` kernel body (spec 8.3): the each-edge-once link round of Afforest, correct for directed and
|
|
3
|
+
* undirected input alike because `edgeList()` yields every logical edge once in declared orientation (design 10.1).
|
|
4
|
+
* `link_pair` is the same GAP `Link` transcription as wcc-link-sample (each module is composed alone, so the helper
|
|
5
|
+
* is copied, not shared): all-u32 CAS on `comp`, an `array<atomic<u32>>` because WGSL forbids mixing atomic and
|
|
6
|
+
* plain access to one element, with a bounded retry loop (PD-5) and the changed flag at `P.flagIndex` inside the
|
|
7
|
+
* same array (PD-4). The `P.giant` guard is GAP's "skip the vertices already in the giant component" and is a pure
|
|
8
|
+
* optimisation -- linking two vertices already in one component is a no-op. The body is normative: a sabotage
|
|
9
|
+
* mutation is a textual edit of it, so it is not restyled.
|
|
10
|
+
*/
|
|
11
|
+
/** Entry point `wcc_link_edges`; no overrides. */
|
|
12
|
+
export declare const wccLinkEdgesWgsl = "\nfn link_pair(a: u32, b: u32) {\n var p1 = atomicLoad(&comp[a]);\n var p2 = atomicLoad(&comp[b]);\n var steps = 0u;\n loop {\n if (p1 == p2) { break; }\n if (steps >= P.maxSteps) { atomicStore(&comp[P.flagIndex], 1u); break; }\n steps = steps + 1u;\n let hi = max(p1, p2);\n let lo = min(p1, p2);\n let pHigh = atomicLoad(&comp[hi]);\n if (pHigh == lo) { break; }\n if (pHigh == hi) {\n let swapped = atomicCompareExchangeWeak(&comp[hi], hi, lo);\n if (swapped.exchanged) { atomicStore(&comp[P.flagIndex], 1u); break; }\n }\n p1 = atomicLoad(&comp[atomicLoad(&comp[hi])]);\n p2 = atomicLoad(&comp[lo]);\n }\n}\n\n@compute @workgroup_size(WG)\nfn wcc_link_edges(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let first = linear_id(wid, lid.x);\n for (var e = first; e < P.items; e = e + P.stride) {\n let u = edgeSrc[e];\n let v = edgeDst[e];\n if (u == v) { continue; }\n if (atomicLoad(&comp[u]) == P.giant && atomicLoad(&comp[v]) == P.giant) { continue; }\n link_pair(u, v);\n }\n}\n";
|
|
13
|
+
//# sourceMappingURL=wcc-link-edges.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"wcc-link-edges.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/wcc-link-edges.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AAEH,kDAAkD;AAClD,eAAO,MAAM,gBAAgB,uqCAiC5B,CAAC"}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `wcc-link-edges` kernel body (spec 8.3): the each-edge-once link round of Afforest, correct for directed and
|
|
3
|
+
* undirected input alike because `edgeList()` yields every logical edge once in declared orientation (design 10.1).
|
|
4
|
+
* `link_pair` is the same GAP `Link` transcription as wcc-link-sample (each module is composed alone, so the helper
|
|
5
|
+
* is copied, not shared): all-u32 CAS on `comp`, an `array<atomic<u32>>` because WGSL forbids mixing atomic and
|
|
6
|
+
* plain access to one element, with a bounded retry loop (PD-5) and the changed flag at `P.flagIndex` inside the
|
|
7
|
+
* same array (PD-4). The `P.giant` guard is GAP's "skip the vertices already in the giant component" and is a pure
|
|
8
|
+
* optimisation -- linking two vertices already in one component is a no-op. The body is normative: a sabotage
|
|
9
|
+
* mutation is a textual edit of it, so it is not restyled.
|
|
10
|
+
*/
|
|
11
|
+
/** Entry point `wcc_link_edges`; no overrides. */
|
|
12
|
+
export const wccLinkEdgesWgsl = /* wgsl */ `
|
|
13
|
+
fn link_pair(a: u32, b: u32) {
|
|
14
|
+
var p1 = atomicLoad(&comp[a]);
|
|
15
|
+
var p2 = atomicLoad(&comp[b]);
|
|
16
|
+
var steps = 0u;
|
|
17
|
+
loop {
|
|
18
|
+
if (p1 == p2) { break; }
|
|
19
|
+
if (steps >= P.maxSteps) { atomicStore(&comp[P.flagIndex], 1u); break; }
|
|
20
|
+
steps = steps + 1u;
|
|
21
|
+
let hi = max(p1, p2);
|
|
22
|
+
let lo = min(p1, p2);
|
|
23
|
+
let pHigh = atomicLoad(&comp[hi]);
|
|
24
|
+
if (pHigh == lo) { break; }
|
|
25
|
+
if (pHigh == hi) {
|
|
26
|
+
let swapped = atomicCompareExchangeWeak(&comp[hi], hi, lo);
|
|
27
|
+
if (swapped.exchanged) { atomicStore(&comp[P.flagIndex], 1u); break; }
|
|
28
|
+
}
|
|
29
|
+
p1 = atomicLoad(&comp[atomicLoad(&comp[hi])]);
|
|
30
|
+
p2 = atomicLoad(&comp[lo]);
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
@compute @workgroup_size(WG)
|
|
35
|
+
fn wcc_link_edges(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
36
|
+
let first = linear_id(wid, lid.x);
|
|
37
|
+
for (var e = first; e < P.items; e = e + P.stride) {
|
|
38
|
+
let u = edgeSrc[e];
|
|
39
|
+
let v = edgeDst[e];
|
|
40
|
+
if (u == v) { continue; }
|
|
41
|
+
if (atomicLoad(&comp[u]) == P.giant && atomicLoad(&comp[v]) == P.giant) { continue; }
|
|
42
|
+
link_pair(u, v);
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
`;
|
|
46
|
+
//# sourceMappingURL=wcc-link-edges.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"wcc-link-edges.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/wcc-link-edges.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AAEH,kDAAkD;AAClD,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAiC1C,CAAC"}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `wcc-link-sample` kernel body (spec 8.3): one of Afforest's sampled link rounds -- every vertex links its
|
|
3
|
+
* r-th neighbour, `colIdx[rowPtr[v] + P.r]`, when it has one. `link_pair` is GAP's `Link` (gapbs/cc.cc lines
|
|
4
|
+
* 40-150) transcribed for WGSL: all-u32 CAS on `comp`, which is `array<atomic<u32>>` because WGSL forbids mixing
|
|
5
|
+
* atomic and plain access to one element, with a bounded retry loop (PD-5). The changed flag is the word at
|
|
6
|
+
* `P.flagIndex` inside the same array (PD-4). The body is normative: a sabotage mutation is a textual edit of it,
|
|
7
|
+
* so it is not restyled.
|
|
8
|
+
*/
|
|
9
|
+
/** Entry point `wcc_link_sample`; standard USE_PERM / HAS_WEIGHTS only (the body reads neither weights nor a permutation beyond the row select). */
|
|
10
|
+
export declare const wccLinkSampleWgsl = "\nfn link_pair(a: u32, b: u32) {\n var p1 = atomicLoad(&comp[a]);\n var p2 = atomicLoad(&comp[b]);\n var steps = 0u;\n loop {\n if (p1 == p2) { break; }\n if (steps >= P.maxSteps) { atomicStore(&comp[P.flagIndex], 1u); break; }\n steps = steps + 1u;\n let hi = max(p1, p2);\n let lo = min(p1, p2);\n let pHigh = atomicLoad(&comp[hi]);\n if (pHigh == lo) { break; }\n if (pHigh == hi) {\n let swapped = atomicCompareExchangeWeak(&comp[hi], hi, lo);\n if (swapped.exchanged) { atomicStore(&comp[P.flagIndex], 1u); break; }\n }\n p1 = atomicLoad(&comp[atomicLoad(&comp[hi])]);\n p2 = atomicLoad(&comp[lo]);\n }\n}\n\n@compute @workgroup_size(WG)\nfn wcc_link_sample(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let first = linear_id(wid, lid.x);\n for (var row = first; row < P.items; row = row + P.stride) {\n let v = select(row, perm[row], USE_PERM);\n let a0 = rowPtr[v];\n let a1 = rowPtr[v + 1u];\n if (a0 + P.r < a1) {\n link_pair(v, colIdx[a0 + P.r]);\n }\n }\n}\n";
|
|
11
|
+
//# sourceMappingURL=wcc-link-sample.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"wcc-link-sample.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/wcc-link-sample.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAEH,oJAAoJ;AACpJ,eAAO,MAAM,iBAAiB,kqCAkC7B,CAAC"}
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `wcc-link-sample` kernel body (spec 8.3): one of Afforest's sampled link rounds -- every vertex links its
|
|
3
|
+
* r-th neighbour, `colIdx[rowPtr[v] + P.r]`, when it has one. `link_pair` is GAP's `Link` (gapbs/cc.cc lines
|
|
4
|
+
* 40-150) transcribed for WGSL: all-u32 CAS on `comp`, which is `array<atomic<u32>>` because WGSL forbids mixing
|
|
5
|
+
* atomic and plain access to one element, with a bounded retry loop (PD-5). The changed flag is the word at
|
|
6
|
+
* `P.flagIndex` inside the same array (PD-4). The body is normative: a sabotage mutation is a textual edit of it,
|
|
7
|
+
* so it is not restyled.
|
|
8
|
+
*/
|
|
9
|
+
/** Entry point `wcc_link_sample`; standard USE_PERM / HAS_WEIGHTS only (the body reads neither weights nor a permutation beyond the row select). */
|
|
10
|
+
export const wccLinkSampleWgsl = /* wgsl */ `
|
|
11
|
+
fn link_pair(a: u32, b: u32) {
|
|
12
|
+
var p1 = atomicLoad(&comp[a]);
|
|
13
|
+
var p2 = atomicLoad(&comp[b]);
|
|
14
|
+
var steps = 0u;
|
|
15
|
+
loop {
|
|
16
|
+
if (p1 == p2) { break; }
|
|
17
|
+
if (steps >= P.maxSteps) { atomicStore(&comp[P.flagIndex], 1u); break; }
|
|
18
|
+
steps = steps + 1u;
|
|
19
|
+
let hi = max(p1, p2);
|
|
20
|
+
let lo = min(p1, p2);
|
|
21
|
+
let pHigh = atomicLoad(&comp[hi]);
|
|
22
|
+
if (pHigh == lo) { break; }
|
|
23
|
+
if (pHigh == hi) {
|
|
24
|
+
let swapped = atomicCompareExchangeWeak(&comp[hi], hi, lo);
|
|
25
|
+
if (swapped.exchanged) { atomicStore(&comp[P.flagIndex], 1u); break; }
|
|
26
|
+
}
|
|
27
|
+
p1 = atomicLoad(&comp[atomicLoad(&comp[hi])]);
|
|
28
|
+
p2 = atomicLoad(&comp[lo]);
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
@compute @workgroup_size(WG)
|
|
33
|
+
fn wcc_link_sample(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
34
|
+
let first = linear_id(wid, lid.x);
|
|
35
|
+
for (var row = first; row < P.items; row = row + P.stride) {
|
|
36
|
+
let v = select(row, perm[row], USE_PERM);
|
|
37
|
+
let a0 = rowPtr[v];
|
|
38
|
+
let a1 = rowPtr[v + 1u];
|
|
39
|
+
if (a0 + P.r < a1) {
|
|
40
|
+
link_pair(v, colIdx[a0 + P.r]);
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
`;
|
|
45
|
+
//# sourceMappingURL=wcc-link-sample.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"wcc-link-sample.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/wcc-link-sample.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAEH,oJAAoJ;AACpJ,MAAM,CAAC,MAAM,iBAAiB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAkC3C,CAAC"}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `wcc-sample` kernel body (spec 8.3 "a 1,024-entry histogram readback to find the giant component"): writes
|
|
3
|
+
* the component label of `P.items` pseudo-randomly chosen vertices into `hist`, which the host reads back and takes
|
|
4
|
+
* the mode of (PD-12: GAP's SampleFrequentElement counts on the host too, and a device histogram over component
|
|
5
|
+
* ids would return a bucket, not an id). The sampler uses the prelude's `lowbias32` and `%`, never a bitwise
|
|
6
|
+
* operator on an index.
|
|
7
|
+
*/
|
|
8
|
+
/** Entry point `wcc_sample`; no overrides. */
|
|
9
|
+
export declare const wccSampleWgsl = "\n@compute @workgroup_size(WG)\nfn wcc_sample(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n if (i >= P.items) { return; }\n let v = lowbias32(i + P.r) % P.n;\n hist[i] = atomicLoad(&comp[v]);\n}\n";
|
|
10
|
+
//# sourceMappingURL=wcc-sample.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"wcc-sample.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/wcc-sample.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AAEH,8CAA8C;AAC9C,eAAO,MAAM,aAAa,iSAQzB,CAAC"}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `wcc-sample` kernel body (spec 8.3 "a 1,024-entry histogram readback to find the giant component"): writes
|
|
3
|
+
* the component label of `P.items` pseudo-randomly chosen vertices into `hist`, which the host reads back and takes
|
|
4
|
+
* the mode of (PD-12: GAP's SampleFrequentElement counts on the host too, and a device histogram over component
|
|
5
|
+
* ids would return a bucket, not an id). The sampler uses the prelude's `lowbias32` and `%`, never a bitwise
|
|
6
|
+
* operator on an index.
|
|
7
|
+
*/
|
|
8
|
+
/** Entry point `wcc_sample`; no overrides. */
|
|
9
|
+
export const wccSampleWgsl = /* wgsl */ `
|
|
10
|
+
@compute @workgroup_size(WG)
|
|
11
|
+
fn wcc_sample(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
12
|
+
let i = linear_id(wid, lid.x);
|
|
13
|
+
if (i >= P.items) { return; }
|
|
14
|
+
let v = lowbias32(i + P.r) % P.n;
|
|
15
|
+
hist[i] = atomicLoad(&comp[v]);
|
|
16
|
+
}
|
|
17
|
+
`;
|
|
18
|
+
//# sourceMappingURL=wcc-sample.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"wcc-sample.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/wcc-sample.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AAEH,8CAA8C;AAC9C,MAAM,CAAC,MAAM,aAAa,GAAG,UAAU,CAAC;;;;;;;;CAQvC,CAAC"}
|