@graphty/webgpu-graph-algorithms 0.6.2 → 0.6.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +62 -32
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-BXqgCifx.js → context-Dvq-Cc6v.js} +71 -25
- package/dist/chunks/context-Dvq-Cc6v.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +8 -6
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +57 -6
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/bellman-ford.d.ts +60 -0
- package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
- package/dist/src/algorithms/bellman-ford.js +301 -0
- package/dist/src/algorithms/bellman-ford.js.map +1 -0
- package/dist/src/algorithms/bfs.d.ts +67 -0
- package/dist/src/algorithms/bfs.d.ts.map +1 -0
- package/dist/src/algorithms/bfs.js +534 -0
- package/dist/src/algorithms/bfs.js.map +1 -0
- package/dist/src/algorithms/closeness.d.ts +53 -0
- package/dist/src/algorithms/closeness.d.ts.map +1 -0
- package/dist/src/algorithms/closeness.js +323 -0
- package/dist/src/algorithms/closeness.js.map +1 -0
- package/dist/src/algorithms/scope.d.ts +5 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +3 -0
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +71 -0
- package/dist/src/algorithms/sssp.d.ts.map +1 -0
- package/dist/src/algorithms/sssp.js +585 -0
- package/dist/src/algorithms/sssp.js.map +1 -0
- package/dist/src/constants.d.ts +12 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +12 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/index.d.ts +8 -2
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +7 -1
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts +4 -4
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +39 -5
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernel/uniform-ring.d.ts +8 -0
- package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
- package/dist/src/kernel/uniform-ring.js +13 -0
- package/dist/src/kernel/uniform-ring.js.map +1 -1
- package/dist/src/kernels.d.ts +44 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +371 -3
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +62 -0
- package/dist/src/primitives/advance.d.ts.map +1 -0
- package/dist/src/primitives/advance.js +95 -0
- package/dist/src/primitives/advance.js.map +1 -0
- package/dist/src/primitives/compact.d.ts +89 -0
- package/dist/src/primitives/compact.d.ts.map +1 -0
- package/dist/src/primitives/compact.js +233 -0
- package/dist/src/primitives/compact.js.map +1 -0
- package/dist/src/primitives/core-shape.d.ts +22 -1
- package/dist/src/primitives/core-shape.d.ts.map +1 -1
- package/dist/src/primitives/core-shape.js +33 -3
- package/dist/src/primitives/core-shape.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +156 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -0
- package/dist/src/primitives/frontier.js +259 -0
- package/dist/src/primitives/frontier.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +16 -7
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/traversal.d.ts +53 -0
- package/dist/src/types/traversal.d.ts.map +1 -0
- package/dist/src/types/traversal.js +10 -0
- package/dist/src/types/traversal.js.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
- package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +59 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js +210 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
- package/dist/webgpu-graph-algorithms.js +3207 -377
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +5 -4
- package/src/accelerator.ts +65 -7
- package/src/algorithms/bellman-ford.ts +387 -0
- package/src/algorithms/bfs.ts +626 -0
- package/src/algorithms/closeness.ts +395 -0
- package/src/algorithms/scope.ts +13 -3
- package/src/algorithms/sssp.ts +767 -0
- package/src/constants.ts +12 -0
- package/src/index.ts +14 -1
- package/src/kernel/prelude.ts +39 -4
- package/src/kernel/uniform-ring.ts +14 -0
- package/src/kernels.ts +450 -6
- package/src/primitives/advance.ts +130 -0
- package/src/primitives/compact.ts +323 -0
- package/src/primitives/core-shape.ts +41 -3
- package/src/primitives/frontier.ts +388 -0
- package/src/types/accelerator.ts +18 -5
- package/src/types/traversal.ts +56 -0
- package/src/wgsl/advance-expand.wgsl.ts +68 -0
- package/src/wgsl/bf-relax.wgsl.ts +57 -0
- package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
- package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
- package/src/wgsl/bfs-contract.wgsl.ts +54 -0
- package/src/wgsl/bfs-fused.wgsl.ts +77 -0
- package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
- package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
- package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
- package/src/wgsl/compact-scatter.wgsl.ts +16 -0
- package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
- package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
- package/src/wgsl/frontier-finalize.wgsl.ts +209 -0
- package/src/wgsl/sssp-pred.wgsl.ts +79 -0
- package/src/wgsl/sssp-relax.wgsl.ts +71 -0
- package/dist/chunks/context-BXqgCifx.js.map +0 -1
package/src/kernels.ts
CHANGED
|
@@ -8,7 +8,12 @@
|
|
|
8
8
|
* segmented-reduce; P3-T2 adds fa2-stats-finalize (K1), fa2-attraction (K2), fa2-integrate (K5) and
|
|
9
9
|
* fa2-to-scene; M8b-T3 adds the seven P7 entries: spmv-pull, pr-scale, pr-finalize, wcc-link-sample,
|
|
10
10
|
* wcc-link-edges, wcc-compress and wcc-sample. P4-T1 adds indirect-finalize; the other P4 entries follow, one task
|
|
11
|
-
* each (the P4 plan, PD-1).
|
|
11
|
+
* each (the P4 plan, PD-1). P8-T3 opens the P8 run of nine appends (the P8 plan, PD-2) with compact-scatter,
|
|
12
|
+
* dedupe-claim and dedupe-filter; P8-T4 adds frontier-finalize with the FrontierCounters and FrontierParams blocks;
|
|
13
|
+
* P8-T5 adds advance-expand; P8-T6 adds bfs-contract and sssp-pred; P8-T7 adds bfs-fused; P8-T8 adds bfs-bottom-up,
|
|
14
|
+
* bfs-bitset-build and bfs-unvisited-flags; P8-T9 adds sssp-relax; P8-T10 adds bf-relax with the BfParams and BfFlags
|
|
15
|
+
* blocks; P8-T11 adds closeness-sweep and closeness-reduce. This file is the only importer of src/wgsl/** (spec 3.2;
|
|
16
|
+
* test/layers.test.ts).
|
|
12
17
|
*/
|
|
13
18
|
|
|
14
19
|
import { STATE_HEADER_BYTES } from "./constants.js";
|
|
@@ -17,7 +22,19 @@ import { UniformBlock } from "./kernel/struct-block.js";
|
|
|
17
22
|
import { type BindingDecl, type OverrideDecl, type WgslModuleSpec } from "./kernel/wgsl.js";
|
|
18
23
|
import { type CoreBinding } from "./memory/residency.js";
|
|
19
24
|
import { type Binding } from "./types/memory.js";
|
|
25
|
+
import { advanceExpandWgsl } from "./wgsl/advance-expand.wgsl.js";
|
|
26
|
+
import { bfRelaxWgsl } from "./wgsl/bf-relax.wgsl.js";
|
|
27
|
+
import { bfsBitsetBuildWgsl } from "./wgsl/bfs-bitset-build.wgsl.js";
|
|
28
|
+
import { bfsBottomUpWgsl } from "./wgsl/bfs-bottom-up.wgsl.js";
|
|
29
|
+
import { bfsContractWgsl } from "./wgsl/bfs-contract.wgsl.js";
|
|
30
|
+
import { bfsFusedWgsl } from "./wgsl/bfs-fused.wgsl.js";
|
|
31
|
+
import { bfsUnvisitedFlagsWgsl } from "./wgsl/bfs-unvisited-flags.wgsl.js";
|
|
32
|
+
import { closenessReduceWgsl } from "./wgsl/closeness-reduce.wgsl.js";
|
|
33
|
+
import { closenessSweepWgsl } from "./wgsl/closeness-sweep.wgsl.js";
|
|
34
|
+
import { compactScatterWgsl } from "./wgsl/compact-scatter.wgsl.js";
|
|
20
35
|
import { countingScatterWgsl } from "./wgsl/counting-scatter.wgsl.js";
|
|
36
|
+
import { dedupeClaimWgsl } from "./wgsl/dedupe-claim.wgsl.js";
|
|
37
|
+
import { dedupeFilterWgsl } from "./wgsl/dedupe-filter.wgsl.js";
|
|
21
38
|
import { degreeWgsl } from "./wgsl/degree.wgsl.js";
|
|
22
39
|
import { fa2AttractionWgsl } from "./wgsl/fa2-attraction.wgsl.js";
|
|
23
40
|
import { fa2IntegrateWgsl } from "./wgsl/fa2-integrate.wgsl.js";
|
|
@@ -26,6 +43,7 @@ import { fa2SpeedFinalizeWgsl } from "./wgsl/fa2-speed-finalize.wgsl.js";
|
|
|
26
43
|
import { fa2StatsFinalizeWgsl } from "./wgsl/fa2-stats-finalize.wgsl.js";
|
|
27
44
|
import { fa2ToSceneWgsl } from "./wgsl/fa2-to-scene.wgsl.js";
|
|
28
45
|
import { fillWgsl } from "./wgsl/fill.wgsl.js";
|
|
46
|
+
import { frontierFinalizeWgsl } from "./wgsl/frontier-finalize.wgsl.js";
|
|
29
47
|
import { gridCellKeyWgsl } from "./wgsl/grid-cell-key.wgsl.js";
|
|
30
48
|
import { gridCentroidWgsl } from "./wgsl/grid-centroid.wgsl.js";
|
|
31
49
|
import { gridCentroidHubWgsl } from "./wgsl/grid-centroid-hub.wgsl.js";
|
|
@@ -43,12 +61,14 @@ import { scanAddWgsl } from "./wgsl/scan-add.wgsl.js";
|
|
|
43
61
|
import { scanBlockWgsl } from "./wgsl/scan-block.wgsl.js";
|
|
44
62
|
import { segmentedReduceWgsl } from "./wgsl/segmented-reduce.wgsl.js";
|
|
45
63
|
import { spmvPullWgsl } from "./wgsl/spmv-pull.wgsl.js";
|
|
64
|
+
import { ssspPredWgsl } from "./wgsl/sssp-pred.wgsl.js";
|
|
65
|
+
import { ssspRelaxWgsl } from "./wgsl/sssp-relax.wgsl.js";
|
|
46
66
|
import { wccCompressWgsl } from "./wgsl/wcc-compress.wgsl.js";
|
|
47
67
|
import { wccLinkEdgesWgsl } from "./wgsl/wcc-link-edges.wgsl.js";
|
|
48
68
|
import { wccLinkSampleWgsl } from "./wgsl/wcc-link-sample.wgsl.js";
|
|
49
69
|
import { wccSampleWgsl } from "./wgsl/wcc-sample.wgsl.js";
|
|
50
70
|
|
|
51
|
-
/** Every module id of P1-
|
|
71
|
+
/** Every module id of P1-P4, P7 and P8 (later ids are appended, never renamed). */
|
|
52
72
|
export type KernelId =
|
|
53
73
|
| "degree"
|
|
54
74
|
| "reduce"
|
|
@@ -79,7 +99,22 @@ export type KernelId =
|
|
|
79
99
|
| "grid-centroid-hub"
|
|
80
100
|
| "grid-downsample"
|
|
81
101
|
| "grid-far-field"
|
|
82
|
-
| "grid-near-field"
|
|
102
|
+
| "grid-near-field"
|
|
103
|
+
| "compact-scatter"
|
|
104
|
+
| "dedupe-claim"
|
|
105
|
+
| "dedupe-filter"
|
|
106
|
+
| "frontier-finalize"
|
|
107
|
+
| "advance-expand"
|
|
108
|
+
| "bfs-contract"
|
|
109
|
+
| "sssp-pred"
|
|
110
|
+
| "bfs-fused"
|
|
111
|
+
| "bfs-bottom-up"
|
|
112
|
+
| "bfs-bitset-build"
|
|
113
|
+
| "bfs-unvisited-flags"
|
|
114
|
+
| "sssp-relax"
|
|
115
|
+
| "bf-relax"
|
|
116
|
+
| "closeness-sweep"
|
|
117
|
+
| "closeness-reduce";
|
|
83
118
|
|
|
84
119
|
/** One registry entry: everything of a WgslModuleSpec except the per-variant overrides and snippets. */
|
|
85
120
|
export interface KernelEntry {
|
|
@@ -94,7 +129,7 @@ export interface KernelEntry {
|
|
|
94
129
|
/** The snippet marker names the body carries (segmented-reduce: ["VALUE"]). */
|
|
95
130
|
readonly snippetSlots: readonly string[];
|
|
96
131
|
/** The phase the entry landed in (documentation and the compile-matrix filter). */
|
|
97
|
-
readonly phase: "P1" | "P2" | "P3" | "P4" | "P7";
|
|
132
|
+
readonly phase: "P1" | "P2" | "P3" | "P4" | "P7" | "P8";
|
|
98
133
|
}
|
|
99
134
|
|
|
100
135
|
// ---- the generated blocks (spec 5.3; contract 3.10.2): field order = byte order, offsets in the JSDoc
|
|
@@ -329,6 +364,116 @@ export const RADIX_PARAMS: UniformBlock = UniformBlock.define("RadixParams", [
|
|
|
329
364
|
["pad0", "u32"],
|
|
330
365
|
]);
|
|
331
366
|
|
|
367
|
+
/** `CompactParams` (uniform, 16 B; spec 6 row 4, P8-T3): `count` @0 (the entries, or the capacity a device count is clamped to), `outIndex` @4 (the word of `outCount` that receives the output count: the block is bound whole because a four-byte word is never 256-aligned), `countIndex` @8 (the word of the counters block holding the entry count, or `U32_MAX` for a host-known count; `compact-scatter` ignores it), `pad0` @12. */
|
|
368
|
+
export const COMPACT_PARAMS: UniformBlock = UniformBlock.define("CompactParams", [
|
|
369
|
+
["count", "u32"],
|
|
370
|
+
["outIndex", "u32"],
|
|
371
|
+
["countIndex", "u32"],
|
|
372
|
+
["stride", "u32"],
|
|
373
|
+
]);
|
|
374
|
+
|
|
375
|
+
/**
|
|
376
|
+
* `FrontierCounters` (storage, 112 B; design 6 row 7, P8-T4, PD-8): EVERY counter of the frontier family is a word of
|
|
377
|
+
* this one block, byte offset 4 x index, because a four-byte word is never a legal storage-binding offset and the
|
|
378
|
+
* device-side selector must reach every count it acts on through one binding; every kernel binds it as
|
|
379
|
+
* `array<atomic<u32>>` and indexes by the `W` record of src/primitives/frontier.ts, and the host decodes the result
|
|
380
|
+
* copy with `FRONTIER_COUNTERS.read`. `frontierCount` @0 (role 0 rotates it in from word 1; never seeded),
|
|
381
|
+
* `nextFrontierCount` @4 (the claim kernels' append span; the BFS seed is 1), `frontierDegreeSum` @8,
|
|
382
|
+
* `prevFrontierCount` @12, `prevDegreeSum` @16, `unvisitedCount` @20, `unvisitedDegreeSum` @24,
|
|
383
|
+
* `unvisitedListLen` @28 (P8-T8), `edgeCount` @32 (clamped by role 1), `edgeCountUnclamped` @36 (the overflow
|
|
384
|
+
* detector, PD-23), `overflowLevels` @40, `level` @44 (the current level; the seed is U32_MAX so the first boundary
|
|
385
|
+
* lands on 0), `visitedCount` @48, `switches` @52, `direction` @56, `done` @60 (the four bytes the host reads per
|
|
386
|
+
* submit), `arcsScanned` @64, `fusedLevels` @68, `twoPhaseLevels` @72, `bottomUpLevels` @76, `farCount` @80,
|
|
387
|
+
* `nextFarCount` @84, `thresholdBits` @88, `deltaBits` @92 (P8-T9), `path` @96 (what the level's kernels run, written
|
|
388
|
+
* by the selector: 0 nothing, 1 two-phase, 2 fused, 3 bottom-up, 4 the fused retry, 5 a near SSSP round, 6 a far
|
|
389
|
+
* one; every level kernel is a direct dispatch that reads it first -- G8-F5). The words nothing writes before
|
|
390
|
+
* P8-T8 / P8-T9 are declared now because the byte layout is what the single result copy decodes.
|
|
391
|
+
*/
|
|
392
|
+
export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
393
|
+
"FrontierCounters",
|
|
394
|
+
[
|
|
395
|
+
["frontierCount", "u32"],
|
|
396
|
+
["nextFrontierCount", "u32"],
|
|
397
|
+
["frontierDegreeSum", "u32"],
|
|
398
|
+
["prevFrontierCount", "u32"],
|
|
399
|
+
["prevDegreeSum", "u32"],
|
|
400
|
+
["unvisitedCount", "u32"],
|
|
401
|
+
["unvisitedDegreeSum", "u32"],
|
|
402
|
+
["unvisitedListLen", "u32"],
|
|
403
|
+
["edgeCount", "u32"],
|
|
404
|
+
["edgeCountUnclamped", "u32"],
|
|
405
|
+
["overflowLevels", "u32"],
|
|
406
|
+
["level", "u32"],
|
|
407
|
+
["visitedCount", "u32"],
|
|
408
|
+
["switches", "u32"],
|
|
409
|
+
["direction", "u32"],
|
|
410
|
+
["done", "u32"],
|
|
411
|
+
["arcsScanned", "u32"],
|
|
412
|
+
["fusedLevels", "u32"],
|
|
413
|
+
["twoPhaseLevels", "u32"],
|
|
414
|
+
["bottomUpLevels", "u32"],
|
|
415
|
+
["farCount", "u32"],
|
|
416
|
+
["nextFarCount", "u32"],
|
|
417
|
+
["thresholdBits", "u32"],
|
|
418
|
+
["deltaBits", "u32"],
|
|
419
|
+
["path", "u32"],
|
|
420
|
+
],
|
|
421
|
+
{ layout: "storage" },
|
|
422
|
+
);
|
|
423
|
+
|
|
424
|
+
/**
|
|
425
|
+
* `FrontierParams` (uniform, 80 B; P8-T4): the params block every P8 kernel except the three compact / dedupe
|
|
426
|
+
* primitives and `bf-relax` binds -- `role` @0 (the finalize role), `slotBase` @4 (`level x FRONTIER_CANDIDATES`),
|
|
427
|
+
* `wg` @8 (the consumers' workgroup size), `alpha` @12, `beta` @16 (Beamer's thresholds, P8-T8), `fusedMax` @20,
|
|
428
|
+
* `edgeCapacity` @24, `maxDepth` @28, `n` @32, `mode` @36 (BFS: 0 auto, 1 top-down only; `sssp-pred`: the PD-27 key
|
|
429
|
+
* rule), `cutoffBits` @40, `arcBase` @44, `arcEnd` @48 (the bound arc window), `predKind` @52 (0 arc, 1 node),
|
|
430
|
+
* `bitsBase` @56, `source` @60, `stride` @64 (a grid-stride plan's stride), `firstOfSubmit` @68 (the boundary's index
|
|
431
|
+
* inside its submit, clamped to 2: the unvisited-count subtraction runs at >= 1, the degree-sum one at >= 2),
|
|
432
|
+
* `iteration` @72 (an `sssp-pred` hop pass, P8-T9), `pad1` @76.
|
|
433
|
+
*/
|
|
434
|
+
export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams", [
|
|
435
|
+
["role", "u32"],
|
|
436
|
+
["slotBase", "u32"],
|
|
437
|
+
["wg", "u32"],
|
|
438
|
+
["alpha", "u32"],
|
|
439
|
+
["beta", "u32"],
|
|
440
|
+
["fusedMax", "u32"],
|
|
441
|
+
["edgeCapacity", "u32"],
|
|
442
|
+
["maxDepth", "u32"],
|
|
443
|
+
["n", "u32"],
|
|
444
|
+
["mode", "u32"],
|
|
445
|
+
["cutoffBits", "u32"],
|
|
446
|
+
["arcBase", "u32"],
|
|
447
|
+
["arcEnd", "u32"],
|
|
448
|
+
["predKind", "u32"],
|
|
449
|
+
["bitsBase", "u32"],
|
|
450
|
+
["source", "u32"],
|
|
451
|
+
["stride", "u32"],
|
|
452
|
+
["firstOfSubmit", "u32"],
|
|
453
|
+
["iteration", "u32"],
|
|
454
|
+
["pad1", "u32"],
|
|
455
|
+
]);
|
|
456
|
+
|
|
457
|
+
/** `BfParams` (uniform, 16 B; P8-T10): `edgeCount` @0 (the logical edges of the `edgeList` view), `stride` @4 (the grid-stride plan's stride), `maxRetries` @8 (PD-12's compare-exchange bound), `cutoffBits` @12 (the f32 bit pattern of the CPU port's `cutoff`, `+Inf` when absent). */
|
|
458
|
+
export const BF_PARAMS: UniformBlock = UniformBlock.define("BfParams", [
|
|
459
|
+
["edgeCount", "u32"],
|
|
460
|
+
["stride", "u32"],
|
|
461
|
+
["maxRetries", "u32"],
|
|
462
|
+
["cutoffBits", "u32"],
|
|
463
|
+
]);
|
|
464
|
+
|
|
465
|
+
/** `BfFlags` (storage, 16 B; P8-T10): the two words `bf-relax` raises and the host reads back after every batch of rounds -- `changed` @0 (some exchange succeeded), `retryExhausted` @4 (some lane hit `maxRetries`, PD-12), `pad0` @8, `pad1` @12. Bound by the kernel as `array<atomic<u32>>`; the block is the host's decoder. */
|
|
466
|
+
export const BF_FLAGS: UniformBlock = UniformBlock.define(
|
|
467
|
+
"BfFlags",
|
|
468
|
+
[
|
|
469
|
+
["changed", "u32"],
|
|
470
|
+
["retryExhausted", "u32"],
|
|
471
|
+
["pad0", "u32"],
|
|
472
|
+
["pad1", "u32"],
|
|
473
|
+
],
|
|
474
|
+
{ layout: "storage" },
|
|
475
|
+
);
|
|
476
|
+
|
|
332
477
|
// ---- the entries (contract 3.10.1; group 0 = graph, 1 = state, 2 = params, 3 = cold)
|
|
333
478
|
|
|
334
479
|
/**
|
|
@@ -915,14 +1060,298 @@ const GRID_NEAR_FIELD: KernelEntry = {
|
|
|
915
1060
|
phase: "P4",
|
|
916
1061
|
};
|
|
917
1062
|
|
|
1063
|
+
/** `compact-scatter` (spec 6 row 4; P8-T3): the scatter of `compact` after the flags' exclusive scan, plus the total into `outCount[P.outIndex]`; 5 storage bindings; order-preserving, so bitwise reproducible. */
|
|
1064
|
+
const COMPACT_SCATTER: KernelEntry = {
|
|
1065
|
+
id: "compact-scatter",
|
|
1066
|
+
body: compactScatterWgsl,
|
|
1067
|
+
entryPoint: "compact_scatter",
|
|
1068
|
+
bindings: [
|
|
1069
|
+
decl(1, 0, "queue", "storage-ro", "array<u32>"),
|
|
1070
|
+
decl(1, 1, "flags", "storage-ro", "array<u32>"),
|
|
1071
|
+
decl(1, 2, "offsets", "storage-ro", "array<u32>"),
|
|
1072
|
+
decl(1, 3, "out", "storage", "array<u32>"),
|
|
1073
|
+
decl(1, 4, "outCount", "storage", "array<u32>"),
|
|
1074
|
+
decl(2, 0, "P", "uniform", "CompactParams"),
|
|
1075
|
+
],
|
|
1076
|
+
overrideDecls: [],
|
|
1077
|
+
uniforms: [COMPACT_PARAMS],
|
|
1078
|
+
needs: [],
|
|
1079
|
+
snippetSlots: [],
|
|
1080
|
+
phase: "P8",
|
|
1081
|
+
};
|
|
1082
|
+
|
|
1083
|
+
/** `dedupe-claim` (spec 6 row 4; P8-T3): `atomicStore(&owner[queue[i]], i)` for every entry, the count from `P.count` or the device word `counters[P.countIndex]`; 3 storage bindings (`counters` is `array<atomic<u32>>` and therefore read-write, although only loaded: WGSL admits an atomic only in a read-write storage buffer). */
|
|
1084
|
+
const DEDUPE_CLAIM: KernelEntry = {
|
|
1085
|
+
id: "dedupe-claim",
|
|
1086
|
+
body: dedupeClaimWgsl,
|
|
1087
|
+
entryPoint: "dedupe_claim",
|
|
1088
|
+
bindings: [
|
|
1089
|
+
decl(1, 0, "queue", "storage-ro", "array<u32>"),
|
|
1090
|
+
decl(1, 1, "owner", "storage", "array<atomic<u32>>"),
|
|
1091
|
+
decl(1, 2, "counters", "storage", "array<atomic<u32>>"),
|
|
1092
|
+
decl(2, 0, "P", "uniform", "CompactParams"),
|
|
1093
|
+
],
|
|
1094
|
+
overrideDecls: [],
|
|
1095
|
+
uniforms: [COMPACT_PARAMS],
|
|
1096
|
+
needs: [],
|
|
1097
|
+
snippetSlots: [],
|
|
1098
|
+
phase: "P8",
|
|
1099
|
+
};
|
|
1100
|
+
|
|
1101
|
+
/** `dedupe-filter` (spec 6 row 4; P8-T3): a separate dispatch keeping the entries that still own their vertex, packed by a workgroup scan and one `atomicAdd` per workgroup on `outCount[P.outIndex]`, which is also the block the count word `P.countIndex` is read from; 4 storage bindings; set-deterministic. */
|
|
1102
|
+
const DEDUPE_FILTER: KernelEntry = {
|
|
1103
|
+
id: "dedupe-filter",
|
|
1104
|
+
body: dedupeFilterWgsl,
|
|
1105
|
+
entryPoint: "dedupe_filter",
|
|
1106
|
+
bindings: [
|
|
1107
|
+
decl(1, 0, "queue", "storage-ro", "array<u32>"),
|
|
1108
|
+
decl(1, 1, "owner", "storage", "array<atomic<u32>>"),
|
|
1109
|
+
decl(1, 2, "out", "storage", "array<u32>"),
|
|
1110
|
+
decl(1, 3, "outCount", "storage", "array<atomic<u32>>"),
|
|
1111
|
+
decl(2, 0, "P", "uniform", "CompactParams"),
|
|
1112
|
+
],
|
|
1113
|
+
overrideDecls: [],
|
|
1114
|
+
uniforms: [COMPACT_PARAMS],
|
|
1115
|
+
needs: [],
|
|
1116
|
+
snippetSlots: [],
|
|
1117
|
+
phase: "P8",
|
|
1118
|
+
};
|
|
1119
|
+
|
|
1120
|
+
/** `frontier-finalize` (design 5.4, 8.10 "BFS finalizeArgs"; P8-T4, PD-3): the one-lane level-boundary selector that rotates the counters block and writes the level's seven indirect slots (role 0), then clamps the edge count and sizes the contract or the fused-retry slot (role 1); 2 storage bindings (the block as `array<atomic<u32>>`, the args). */
|
|
1121
|
+
const FRONTIER_FINALIZE: KernelEntry = {
|
|
1122
|
+
id: "frontier-finalize",
|
|
1123
|
+
body: frontierFinalizeWgsl,
|
|
1124
|
+
entryPoint: "frontier_finalize",
|
|
1125
|
+
bindings: [
|
|
1126
|
+
decl(1, 0, "counters", "storage", "array<atomic<u32>>"),
|
|
1127
|
+
decl(1, 1, "args", "storage", "array<u32>"),
|
|
1128
|
+
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1129
|
+
],
|
|
1130
|
+
overrideDecls: [],
|
|
1131
|
+
uniforms: [FRONTIER_PARAMS],
|
|
1132
|
+
needs: [],
|
|
1133
|
+
snippetSlots: [],
|
|
1134
|
+
phase: "P8",
|
|
1135
|
+
};
|
|
1136
|
+
|
|
1137
|
+
/** `advance-expand` (design 6 row 8, 8.10 "BFS expand"; P8-T5): the block-mapped expansion of the frontier into the edge queue -- each workgroup scans its entries' degrees with the prelude's `wg_scan_u32` (the twin axis: `needs: ["subgroups"]` is what makes the subgroup compilation differ) and strips the aggregate by binary search, one `atomicAdd` per workgroup reserving its span; 7 storage bindings (the four graph slots, `frontierIn`, the counters block as `array<atomic<u32>>`, `edgeQueue`); no `TIER` override (the workgroup-per-row structure is `bfs-fused`). */
|
|
1138
|
+
const ADVANCE_EXPAND: KernelEntry = {
|
|
1139
|
+
id: "advance-expand",
|
|
1140
|
+
body: advanceExpandWgsl,
|
|
1141
|
+
entryPoint: "advance_expand",
|
|
1142
|
+
bindings: GRAPH_SLOTS.concat(
|
|
1143
|
+
decl(1, 0, "frontierIn", "storage-ro", "array<u32>"),
|
|
1144
|
+
decl(1, 1, "counters", "storage", "array<atomic<u32>>"),
|
|
1145
|
+
decl(1, 2, "edgeQueue", "storage", "array<u32>"),
|
|
1146
|
+
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1147
|
+
),
|
|
1148
|
+
overrideDecls: [],
|
|
1149
|
+
uniforms: [FRONTIER_PARAMS],
|
|
1150
|
+
needs: ["subgroups"],
|
|
1151
|
+
snippetSlots: [],
|
|
1152
|
+
phase: "P8",
|
|
1153
|
+
};
|
|
1154
|
+
|
|
1155
|
+
/** `bfs-contract` (design 8.4, 8.10 "BFS contract"; P8-T6, PD-6): the contraction of the edge queue -- `atomicMin(&depth[v], level + 1)` with the invocation that observes `INVALID_INDEX` the unique winner, packed into the output vertex queue by a workgroup scan and one `atomicAdd` per workgroup on `nextFrontierCount`; 4 storage bindings (no `owner`: the claim already dedupes, DEP-P8-B; no `parent`: the post-pass writes it, PD-24; both counts are words of `counters`). */
|
|
1156
|
+
const BFS_CONTRACT: KernelEntry = {
|
|
1157
|
+
id: "bfs-contract",
|
|
1158
|
+
body: bfsContractWgsl,
|
|
1159
|
+
entryPoint: "bfs_contract",
|
|
1160
|
+
bindings: [
|
|
1161
|
+
decl(1, 0, "edgeQueue", "storage-ro", "array<u32>"),
|
|
1162
|
+
decl(1, 1, "counters", "storage", "array<atomic<u32>>"),
|
|
1163
|
+
decl(1, 2, "depth", "storage", "array<atomic<u32>>"),
|
|
1164
|
+
decl(1, 3, "frontierOut", "storage", "array<u32>"),
|
|
1165
|
+
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1166
|
+
],
|
|
1167
|
+
overrideDecls: [],
|
|
1168
|
+
uniforms: [FRONTIER_PARAMS],
|
|
1169
|
+
needs: [],
|
|
1170
|
+
snippetSlots: [],
|
|
1171
|
+
phase: "P8",
|
|
1172
|
+
};
|
|
1173
|
+
|
|
1174
|
+
/** `sssp-pred` (design 8.10 "SSSP predecessor pass"; P8-T6 / P8-T9, PD-24 / PD-27): the one post-pass over the settled distances, grid-striding every row -- `MODE 1` reads u32 depths and writes BFS `parent`, `MODE 0` reads f32 bit patterns and writes `predArc` under PD-27's key, `P.predKind` choosing the node or the arc index; 6 storage bindings (the four graph slots, `dist` bound plain across dispatches, `pred` as `array<atomic<u32>>`, which in `MODE 0` also carries the hop counts and the two flag words in its upper regions). */
|
|
1175
|
+
const SSSP_PRED: KernelEntry = {
|
|
1176
|
+
id: "sssp-pred",
|
|
1177
|
+
body: ssspPredWgsl,
|
|
1178
|
+
entryPoint: "sssp_pred",
|
|
1179
|
+
bindings: GRAPH_SLOTS.concat(
|
|
1180
|
+
decl(1, 0, "dist", "storage-ro", "array<u32>"),
|
|
1181
|
+
decl(1, 1, "pred", "storage", "array<atomic<u32>>"),
|
|
1182
|
+
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1183
|
+
),
|
|
1184
|
+
overrideDecls: [{ name: "MODE", type: "u32", default: 0 }],
|
|
1185
|
+
uniforms: [FRONTIER_PARAMS],
|
|
1186
|
+
needs: [],
|
|
1187
|
+
snippetSlots: [],
|
|
1188
|
+
phase: "P8",
|
|
1189
|
+
};
|
|
1190
|
+
|
|
1191
|
+
/** `bfs-fused` (design 8.4 "the fused variant", 6 row 8 "the workgroup-per-row tier", 8.10 "BFS fused expand-contract"; P8-T7, PD-23): one level's expansion and contraction in one dispatch, one WORKGROUP per frontier entry, every lane stripping the entry's row with `bfs-contract`'s claim inline and no edge queue traffic; dispatched from `SLOT.fused` (a frontier below `P.fusedMax`) and from `SLOT.fusedRetry` (an overflowed level); 8 storage bindings (the four graph slots, `frontierIn`, the counters block as `array<atomic<u32>>`, `depth` as `array<atomic<u32>>`, `frontierOut`) -- exactly at the budget, which is why no `parent` lives here (PD-24). */
|
|
1192
|
+
const BFS_FUSED: KernelEntry = {
|
|
1193
|
+
id: "bfs-fused",
|
|
1194
|
+
body: bfsFusedWgsl,
|
|
1195
|
+
entryPoint: "bfs_fused",
|
|
1196
|
+
bindings: GRAPH_SLOTS.concat(
|
|
1197
|
+
decl(1, 0, "frontierIn", "storage-ro", "array<u32>"),
|
|
1198
|
+
decl(1, 1, "counters", "storage", "array<atomic<u32>>"),
|
|
1199
|
+
decl(1, 2, "depth", "storage", "array<atomic<u32>>"),
|
|
1200
|
+
decl(1, 3, "frontierOut", "storage", "array<u32>"),
|
|
1201
|
+
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1202
|
+
),
|
|
1203
|
+
overrideDecls: [],
|
|
1204
|
+
uniforms: [FRONTIER_PARAMS],
|
|
1205
|
+
needs: [],
|
|
1206
|
+
snippetSlots: [],
|
|
1207
|
+
phase: "P8",
|
|
1208
|
+
};
|
|
1209
|
+
|
|
1210
|
+
/** `bfs-bottom-up` (design 8.4 "the bottom-up sweep", 8.10 "BFS bottom-up"; P8-T8, PD-18 / PD-21): one invocation per entry of the unvisited list, walking its in-neighbours through the REVERSE core bound in group 0 until the first one whose bit is set in the frontier bitset (the early exit `arcsScanned` witnesses), claiming with a plain `atomicStore` and packing the winners into the output vertex queue by `bfs-contract`'s scan; 8 storage bindings (the four graph slots of the reverse core, `sweepIn` read-only -- the unvisited list at word 0 and the bitset at `P.bitsBase`, one buffer -- the counters block, `depth` and `frontierOut` as `array<atomic<u32>>` / `array<u32>`); `needs: ["subgroups"]` for the `wg_reduce_u32` call (a twin kernel). */
|
|
1211
|
+
const BFS_BOTTOM_UP: KernelEntry = {
|
|
1212
|
+
id: "bfs-bottom-up",
|
|
1213
|
+
body: bfsBottomUpWgsl,
|
|
1214
|
+
entryPoint: "bfs_bottom_up",
|
|
1215
|
+
bindings: GRAPH_SLOTS.concat(
|
|
1216
|
+
decl(1, 0, "sweepIn", "storage-ro", "array<u32>"),
|
|
1217
|
+
decl(1, 1, "counters", "storage", "array<atomic<u32>>"),
|
|
1218
|
+
decl(1, 2, "depth", "storage", "array<atomic<u32>>"),
|
|
1219
|
+
decl(1, 3, "frontierOut", "storage", "array<u32>"),
|
|
1220
|
+
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1221
|
+
),
|
|
1222
|
+
overrideDecls: [],
|
|
1223
|
+
uniforms: [FRONTIER_PARAMS],
|
|
1224
|
+
needs: ["subgroups"],
|
|
1225
|
+
snippetSlots: [],
|
|
1226
|
+
phase: "P8",
|
|
1227
|
+
};
|
|
1228
|
+
|
|
1229
|
+
/** `bfs-bitset-build` (design 8.4 "the bitset frontier"; P8-T8): the vertex-list-to-bitset hand-off of a bottom-up level -- one `atomicOr` per entry of the input frontier into the `ceil(n / 32)`-word bitset at `P.bitsBase` of the `sweepIn` buffer (bound whole as `bits`, read-write); 3 storage bindings. */
|
|
1230
|
+
const BFS_BITSET_BUILD: KernelEntry = {
|
|
1231
|
+
id: "bfs-bitset-build",
|
|
1232
|
+
body: bfsBitsetBuildWgsl,
|
|
1233
|
+
entryPoint: "bfs_bitset_build",
|
|
1234
|
+
bindings: [
|
|
1235
|
+
decl(1, 0, "frontierIn", "storage-ro", "array<u32>"),
|
|
1236
|
+
decl(1, 1, "counters", "storage", "array<atomic<u32>>"),
|
|
1237
|
+
decl(1, 2, "bits", "storage", "array<atomic<u32>>"),
|
|
1238
|
+
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1239
|
+
],
|
|
1240
|
+
overrideDecls: [],
|
|
1241
|
+
uniforms: [FRONTIER_PARAMS],
|
|
1242
|
+
needs: [],
|
|
1243
|
+
snippetSlots: [],
|
|
1244
|
+
phase: "P8",
|
|
1245
|
+
};
|
|
1246
|
+
|
|
1247
|
+
/** `bfs-unvisited-flags` (design 8.4; P8-T8, PD-18): the unvisited set's producer, once per submit -- grid-striding over the vertices, counting the unclaimed ones and their OUT-degree sum into words 5 and 6 and flagging those with a non-zero IN-degree (word 7, what the sweep iterates) for `compact`; 5 storage bindings (the `outDegree` and `inDegree` VIEWS rather than the graph group, `depth` read-only, `flags`, the counters block); `needs: ["subgroups"]` for the three `wg_reduce_u32` calls (a twin kernel). */
|
|
1248
|
+
const BFS_UNVISITED_FLAGS: KernelEntry = {
|
|
1249
|
+
id: "bfs-unvisited-flags",
|
|
1250
|
+
body: bfsUnvisitedFlagsWgsl,
|
|
1251
|
+
entryPoint: "bfs_unvisited_flags",
|
|
1252
|
+
bindings: [
|
|
1253
|
+
decl(1, 0, "outDegree", "storage-ro", "array<u32>"),
|
|
1254
|
+
decl(1, 1, "inDegree", "storage-ro", "array<u32>"),
|
|
1255
|
+
decl(1, 2, "depth", "storage-ro", "array<u32>"),
|
|
1256
|
+
decl(1, 3, "flags", "storage", "array<u32>"),
|
|
1257
|
+
decl(1, 4, "counters", "storage", "array<atomic<u32>>"),
|
|
1258
|
+
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1259
|
+
],
|
|
1260
|
+
overrideDecls: [],
|
|
1261
|
+
uniforms: [FRONTIER_PARAMS],
|
|
1262
|
+
needs: ["subgroups"],
|
|
1263
|
+
snippetSlots: [],
|
|
1264
|
+
phase: "P8",
|
|
1265
|
+
};
|
|
1266
|
+
|
|
1267
|
+
/** `sssp-relax` (design 8.4 "Davidson's near-far", 8.10 "SSSP near-far relax"; P8-T9, PD-9 / PD-20 / DEP-P8-E): one round of the near-far loop -- role 0 relaxes the deduped near pile's whole rows with `atomicMin` on the f32 bit patterns of `dist` and appends each improved vertex to the raw near or far half of `queueOut` (the two halves of ONE buffer at word 0 and word `P.edgeCapacity`), role 1 re-buckets the deduped far pile; 8 storage bindings (the four graph slots with the run's weights bound in the weights slot, `dist` and the counters block as `array<atomic<u32>>`, `queueIn` read-only, `queueOut`) -- exactly at the budget, which is why no `pred` lives here (PD-11) and why the piles' counts, the threshold and the delta are words of the block. */
|
|
1268
|
+
const SSSP_RELAX: KernelEntry = {
|
|
1269
|
+
id: "sssp-relax",
|
|
1270
|
+
body: ssspRelaxWgsl,
|
|
1271
|
+
entryPoint: "sssp_relax",
|
|
1272
|
+
bindings: GRAPH_SLOTS.concat(
|
|
1273
|
+
decl(1, 0, "dist", "storage", "array<atomic<u32>>"),
|
|
1274
|
+
decl(1, 1, "counters", "storage", "array<atomic<u32>>"),
|
|
1275
|
+
decl(1, 2, "queueIn", "storage-ro", "array<u32>"),
|
|
1276
|
+
decl(1, 3, "queueOut", "storage", "array<u32>"),
|
|
1277
|
+
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1278
|
+
),
|
|
1279
|
+
overrideDecls: [],
|
|
1280
|
+
uniforms: [FRONTIER_PARAMS],
|
|
1281
|
+
needs: [],
|
|
1282
|
+
snippetSlots: [],
|
|
1283
|
+
phase: "P8",
|
|
1284
|
+
};
|
|
1285
|
+
|
|
1286
|
+
/** `bf-relax` (design 8.4 "Bellman-Ford"; P8-T10, PD-12 / DEP-P8-E): one edge-parallel relaxation round over the `edgeList` view with a bounded compare-exchange on the f32 bit patterns of `dist` (negative distances reverse the bit-pattern order, so no `atomicMin`); `UNDIRECTED` relaxes the other direction of every edge too; 6 storage bindings (`edgeSrc`, `edgeDst`, `edgeToArc` -- the core's segment, or an iota scratch on a directed identity snapshot --, the run's arc-indexed `weights`, `dist` and the `BfFlags` block as `array<atomic<u32>>`); no graph group. */
|
|
1287
|
+
const BF_RELAX: KernelEntry = {
|
|
1288
|
+
id: "bf-relax",
|
|
1289
|
+
body: bfRelaxWgsl,
|
|
1290
|
+
entryPoint: "bf_relax",
|
|
1291
|
+
bindings: [
|
|
1292
|
+
decl(1, 0, "edgeSrc", "storage-ro", "array<u32>"),
|
|
1293
|
+
decl(1, 1, "edgeDst", "storage-ro", "array<u32>"),
|
|
1294
|
+
decl(1, 2, "edgeToArc", "storage-ro", "array<u32>"),
|
|
1295
|
+
decl(1, 3, "weights", "storage-ro", "array<f32>"),
|
|
1296
|
+
decl(1, 4, "dist", "storage", "array<atomic<u32>>"),
|
|
1297
|
+
decl(1, 5, "flags", "storage", "array<atomic<u32>>"),
|
|
1298
|
+
decl(2, 0, "P", "uniform", "BfParams"),
|
|
1299
|
+
],
|
|
1300
|
+
overrideDecls: [{ name: "UNDIRECTED", type: "bool", default: false }],
|
|
1301
|
+
uniforms: [BF_PARAMS],
|
|
1302
|
+
needs: [],
|
|
1303
|
+
snippetSlots: [],
|
|
1304
|
+
phase: "P8",
|
|
1305
|
+
};
|
|
1306
|
+
|
|
1307
|
+
/** `closeness-sweep` (design 8.4 "32 sources per u32 word"; P8-T11, PD-13 / DEP-P8-E): one level of the bit-parallel multi-source BFS -- `advance-expand`'s block-mapped strip over the compacted frontier list with the claim inline (`atomicOr` on the visited word of the four-region `bits` buffer, the won bits into the level's next region and the flags region, one workgroup-memory tally per source flushed by one `atomicAdd` per source per workgroup into `perSource`); 8 storage bindings (the four graph slots, `frontierList` read-only, `counters`, `bits` and `perSource` as `array<atomic<u32>>`) -- exactly at the budget; the inlined Hillis-Steele scan, so `needs: []`. */
|
|
1308
|
+
const CLOSENESS_SWEEP: KernelEntry = {
|
|
1309
|
+
id: "closeness-sweep",
|
|
1310
|
+
body: closenessSweepWgsl,
|
|
1311
|
+
entryPoint: "closeness_sweep",
|
|
1312
|
+
bindings: GRAPH_SLOTS.concat(
|
|
1313
|
+
decl(1, 0, "frontierList", "storage-ro", "array<u32>"),
|
|
1314
|
+
decl(1, 1, "counters", "storage", "array<atomic<u32>>"),
|
|
1315
|
+
decl(1, 2, "bits", "storage", "array<atomic<u32>>"),
|
|
1316
|
+
decl(1, 3, "perSource", "storage", "array<atomic<u32>>"),
|
|
1317
|
+
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1318
|
+
),
|
|
1319
|
+
overrideDecls: [],
|
|
1320
|
+
uniforms: [FRONTIER_PARAMS],
|
|
1321
|
+
needs: [],
|
|
1322
|
+
snippetSlots: [],
|
|
1323
|
+
phase: "P8",
|
|
1324
|
+
};
|
|
1325
|
+
|
|
1326
|
+
/** `closeness-reduce` (design 8.4, 9.7; P8-T11, PD-13): the one-lane bookkeeping of the sweep -- role 0 the level boundary (`done` from the previous level's compacted count, `newCount` folded into `reached` and the 64-bit `sum` at `level + 1` with the 16-bit split product and the carry, `level` advanced), role 1 the seed of a batch (the sources' bits into `visited` and the level-0 frontier region, their flags, `counters[0] = k`, `level = U32_MAX`); 3 storage bindings (`counters` and `perSource` as `array<atomic<u32>>`, `bits` plain: one lane writes the seed). */
|
|
1327
|
+
const CLOSENESS_REDUCE: KernelEntry = {
|
|
1328
|
+
id: "closeness-reduce",
|
|
1329
|
+
body: closenessReduceWgsl,
|
|
1330
|
+
entryPoint: "closeness_reduce",
|
|
1331
|
+
bindings: [
|
|
1332
|
+
decl(1, 0, "counters", "storage", "array<atomic<u32>>"),
|
|
1333
|
+
decl(1, 1, "perSource", "storage", "array<atomic<u32>>"),
|
|
1334
|
+
decl(1, 2, "bits", "storage", "array<u32>"),
|
|
1335
|
+
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1336
|
+
],
|
|
1337
|
+
overrideDecls: [],
|
|
1338
|
+
uniforms: [FRONTIER_PARAMS],
|
|
1339
|
+
needs: [],
|
|
1340
|
+
snippetSlots: [],
|
|
1341
|
+
phase: "P8",
|
|
1342
|
+
};
|
|
1343
|
+
|
|
918
1344
|
/**
|
|
919
1345
|
* The entries by id, in dispatch order. PLAN DECISION: `KernelId` is declared in full (contract 3.10) while the
|
|
920
1346
|
* entries landed phase by phase, so the table is built as a Partial record and exported below through the
|
|
921
1347
|
* contract's `Readonly<Record<KernelId, KernelEntry>>` type by one assertion; the runtime membership check of
|
|
922
1348
|
* `entryOf` is the E_INVALID_ARGUMENT the contract documents for a JS caller's unknown id. P1-T4 landed the five
|
|
923
1349
|
* P1 entries, P2-T2 `"segmented-reduce"`, and P3-T2 `"fa2-stats-finalize"`, `"fa2-attraction"`, `"fa2-integrate"`
|
|
924
|
-
* and `"fa2-to-scene"`; M8b-T3 landed the seven P7 entries
|
|
925
|
-
*
|
|
1350
|
+
* and `"fa2-to-scene"`; M8b-T3 landed the seven P7 entries and P4 its thirteen; P8-T3 landed the three compact /
|
|
1351
|
+
* dedupe entries, P8-T4 `"frontier-finalize"`, P8-T5 `"advance-expand"`, P8-T6 `"bfs-contract"` and `"sssp-pred"` and
|
|
1352
|
+
* P8-T7 `"bfs-fused"`, P8-T8 `"bfs-bottom-up"`, `"bfs-bitset-build"` and `"bfs-unvisited-flags"`, P8-T9
|
|
1353
|
+
* `"sssp-relax"`, P8-T10 `"bf-relax"` and P8-T11 `"closeness-sweep"` and `"closeness-reduce"`, so every member of
|
|
1354
|
+
* `KernelId` is present and the assertion is exact.
|
|
926
1355
|
*/
|
|
927
1356
|
const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze({
|
|
928
1357
|
degree: DEGREE,
|
|
@@ -955,6 +1384,21 @@ const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze
|
|
|
955
1384
|
"grid-downsample": GRID_DOWNSAMPLE,
|
|
956
1385
|
"grid-far-field": GRID_FAR_FIELD,
|
|
957
1386
|
"grid-near-field": GRID_NEAR_FIELD,
|
|
1387
|
+
"compact-scatter": COMPACT_SCATTER,
|
|
1388
|
+
"dedupe-claim": DEDUPE_CLAIM,
|
|
1389
|
+
"dedupe-filter": DEDUPE_FILTER,
|
|
1390
|
+
"frontier-finalize": FRONTIER_FINALIZE,
|
|
1391
|
+
"advance-expand": ADVANCE_EXPAND,
|
|
1392
|
+
"bfs-contract": BFS_CONTRACT,
|
|
1393
|
+
"sssp-pred": SSSP_PRED,
|
|
1394
|
+
"bfs-fused": BFS_FUSED,
|
|
1395
|
+
"bfs-bottom-up": BFS_BOTTOM_UP,
|
|
1396
|
+
"bfs-bitset-build": BFS_BITSET_BUILD,
|
|
1397
|
+
"bfs-unvisited-flags": BFS_UNVISITED_FLAGS,
|
|
1398
|
+
"sssp-relax": SSSP_RELAX,
|
|
1399
|
+
"bf-relax": BF_RELAX,
|
|
1400
|
+
"closeness-sweep": CLOSENESS_SWEEP,
|
|
1401
|
+
"closeness-reduce": CLOSENESS_REDUCE,
|
|
958
1402
|
});
|
|
959
1403
|
|
|
960
1404
|
/** THE registry (spec 3.5): every entry, keyed by id. */
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `advance` primitive of design 6 row 8 (P8-T5 / P8-T12; the P8 plan's PD-1, PD-23, DEP-P8-A, DEP-P8-E): the
|
|
3
|
+
* block-mapped expansion of a `Frontier`'s input queue into its edge queue by the `advance-expand` kernel, a DIRECT
|
|
4
|
+
* grid-stride dispatch over a grid planned for n entries that loops to `frontierCount` and runs only when the `path`
|
|
5
|
+
* word `frontier-finalize` role 0 wrote earlier in the same pass says the level is two-phase; the host never knows
|
|
6
|
+
* the frontier's size. The kernel leaves `edgeCount` (clamped later by role 1),
|
|
7
|
+
* `edgeCountUnclamped` (the overflow detector) and `frontierDegreeSum` in the counters block.
|
|
8
|
+
*
|
|
9
|
+
* Design 6 row 8 names three structures -- block-mapped, a WORKGROUP_PER_ROW tier for rows above 1,024 arcs, and a
|
|
10
|
+
* subgroup tier. Here they are: the block-mapped body is the large-frontier kernel and already balances a hub row
|
|
11
|
+
* over all `WG` lanes of its block (a 10,000-degree entry costs its block 40 strips, never one lane 10,000 reads);
|
|
12
|
+
* the workgroup-per-row structure IS `bfs-fused` (P8-T7), one workgroup per frontier entry, which the selector
|
|
13
|
+
* chooses for a small frontier such as a hub-only level, so `advance-expand` declares no `TIER` override and the tier
|
|
14
|
+
* choice is the fused / two-phase choice; and the subgroup tier is not a second body but the same body compiled
|
|
15
|
+
* against the prelude's subgroup helper block (`wg_scan_u32` becomes `subgroupExclusiveAdd` / `subgroupAdd` plus the
|
|
16
|
+
* elected-lane slot carry), which `needs: ["subgroups"]` in the registry entry switches on when the device has the
|
|
17
|
+
* feature. Both forms return each lane's prefix in LANE order, so the arcs land at the same queue positions; only
|
|
18
|
+
* the order ACROSS workgroups (the `atomicAdd` reservation order) is schedule-dependent, which is why a test compares
|
|
19
|
+
* sorted queues.
|
|
20
|
+
*
|
|
21
|
+
* A windowed core (spec 4.2: `colIdx` above `maxStorageBufferBindingSize`, bound as arc windows) is executed, not
|
|
22
|
+
* refused (P8-T12 lifts DEP-P8-E for the frontier family): one dispatch per window, the frontier and its count the
|
|
23
|
+
* same in every window, each dispatch binding that window's `colIdx` and
|
|
24
|
+
* carrying its owned arc range in `arcBase` / `arcEnd` (`coreWindows`: the ranges partition the arcs, which P4's
|
|
25
|
+
* overlapping window bounds do not). The body clips every row to the range, so a row straddling two windows emits
|
|
26
|
+
* each arc exactly once, and the per-window dispatches accumulate into the same edge queue through the same
|
|
27
|
+
* `edgeCount` reservation while `edgeCountUnclamped` and `frontierDegreeSum` add up across windows to the level's
|
|
28
|
+
* totals. `src/primitives/**` never imports `src/context.ts`.
|
|
29
|
+
*/
|
|
30
|
+
|
|
31
|
+
import { WebGpuGraphError } from "../errors.js";
|
|
32
|
+
import { planGridStride } from "../kernel/dispatch.js";
|
|
33
|
+
import { type Kernel } from "../kernel/kernel.js";
|
|
34
|
+
import { FRONTIER_PARAMS, graphBindings, graphOverrides, kernelSpec } from "../kernels.js";
|
|
35
|
+
import { type CoreBinding } from "../memory/residency.js";
|
|
36
|
+
import { type CoreWindow, coreWindows, rowCountOf } from "./core-shape.js";
|
|
37
|
+
import { type Frontier, type FrontierScope } from "./frontier.js";
|
|
38
|
+
|
|
39
|
+
/** A prepared advance (design 6 row 8): records one expansion of a frontier per level. */
|
|
40
|
+
export interface AdvancePlanner {
|
|
41
|
+
/** The compiled `advance-expand` kernel (so `bfs-fused` can share its bind groups' shape). */
|
|
42
|
+
readonly kernel: Kernel;
|
|
43
|
+
/** The arc windows every level dispatches over: one over the whole core, or the windowed core's (P8-T12). */
|
|
44
|
+
readonly windows: readonly CoreWindow[];
|
|
45
|
+
/**
|
|
46
|
+
* Records the expansion of `frontier.input` into `frontier.edgeQueue` as a DIRECT grid-stride dispatch per window
|
|
47
|
+
* (`planGridStride(n)`: the frontier holds at most n entries; the kernel loops to the count word and runs only
|
|
48
|
+
* when the selector's path word says the level is two-phase): one `FrontierParams` record per window
|
|
49
|
+
* (`edgeCapacity`, `n`, `wg`, `stride` and the window's arc range) and the kernel bound to the window's graph
|
|
50
|
+
* arrays, the input queue, the counters block and the edge queue (`Kernel.bind` reuses the bind groups of the
|
|
51
|
+
* same buffers and ranges, so the two sides of a frontier cost two bind-group sets per window). A frontier
|
|
52
|
+
* whose `n` is not the core's row count is E_INVALID_ARGUMENT before anything is recorded.
|
|
53
|
+
* @param pass - the compute pass
|
|
54
|
+
* @param frontier - the queue to expand (its `input` side)
|
|
55
|
+
*/
|
|
56
|
+
record(pass: GPUComputePassEncoder, frontier: Frontier): void;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Compiles `advance-expand` for the core's weights pattern (no permutation: the frontier names rows directly) so
|
|
61
|
+
* `record` is synchronous, and resolves the core's windows once. The planner lives exactly as long as the scope:
|
|
62
|
+
* never use it after the scope's dispose().
|
|
63
|
+
* @param scope - the caller's scope (pipelines, params)
|
|
64
|
+
* @param core - the resident core arrays of the snapshot the frontier walks (windowed or not)
|
|
65
|
+
* @returns the planner
|
|
66
|
+
*/
|
|
67
|
+
export async function prepareAdvance(scope: FrontierScope, core: CoreBinding): Promise<AdvancePlanner> {
|
|
68
|
+
const kernel = await scope.pipelines.kernel(kernelSpec("advance-expand", graphOverrides(core, null)));
|
|
69
|
+
return new AdvancePlannerImpl(scope, core, kernel);
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/** The planner: the compiled kernel over one core's windows. */
|
|
73
|
+
class AdvancePlannerImpl implements AdvancePlanner {
|
|
74
|
+
readonly kernel: Kernel;
|
|
75
|
+
readonly windows: readonly CoreWindow[];
|
|
76
|
+
private readonly scope: FrontierScope;
|
|
77
|
+
private readonly n: number;
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* Wraps the resolved kernel; use prepareAdvance().
|
|
81
|
+
* @param scope - the caller's scope
|
|
82
|
+
* @param core - the core the kernel was compiled for
|
|
83
|
+
* @param kernel - the `advance-expand` kernel
|
|
84
|
+
*/
|
|
85
|
+
constructor(scope: FrontierScope, core: CoreBinding, kernel: Kernel) {
|
|
86
|
+
this.scope = scope;
|
|
87
|
+
this.kernel = kernel;
|
|
88
|
+
this.windows = coreWindows(core);
|
|
89
|
+
this.n = rowCountOf(core, "advance");
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* Records one direct expansion per window (see the interface).
|
|
94
|
+
* @param pass - the compute pass
|
|
95
|
+
* @param frontier - the queue to expand
|
|
96
|
+
*/
|
|
97
|
+
record(pass: GPUComputePassEncoder, frontier: Frontier): void {
|
|
98
|
+
if (frontier.n !== this.n) {
|
|
99
|
+
throw new WebGpuGraphError(
|
|
100
|
+
"E_INVALID_ARGUMENT",
|
|
101
|
+
`advance: the frontier holds ${frontier.n} vertices, the core ${this.n} rows`,
|
|
102
|
+
{
|
|
103
|
+
argument: "frontier",
|
|
104
|
+
value: frontier.n,
|
|
105
|
+
expected: this.n,
|
|
106
|
+
},
|
|
107
|
+
);
|
|
108
|
+
}
|
|
109
|
+
const { scope } = this;
|
|
110
|
+
const plan = planGridStride(frontier.n, scope.workgroupSize, scope.caps); // the frontier holds at most n entries
|
|
111
|
+
for (const w of this.windows) {
|
|
112
|
+
const params = scope.params(FRONTIER_PARAMS, {
|
|
113
|
+
wg: scope.workgroupSize,
|
|
114
|
+
edgeCapacity: frontier.edgeCapacity,
|
|
115
|
+
n: frontier.n,
|
|
116
|
+
arcBase: w.arcBase,
|
|
117
|
+
arcEnd: w.arcEnd,
|
|
118
|
+
stride: plan.stride ?? scope.workgroupSize,
|
|
119
|
+
});
|
|
120
|
+
const bound = this.kernel.bind({
|
|
121
|
+
...graphBindings(w.core, null),
|
|
122
|
+
frontierIn: frontier.input,
|
|
123
|
+
counters: frontier.counters,
|
|
124
|
+
edgeQueue: frontier.edgeQueue,
|
|
125
|
+
P: params.binding,
|
|
126
|
+
});
|
|
127
|
+
this.kernel.dispatch(pass, bound, plan, [params.offset]);
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
}
|