@graphty/webgpu-graph-algorithms 0.6.3 → 0.6.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +62 -32
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-BXqgCifx.js → context-hzGggHeM.js} +68 -24
- package/dist/chunks/context-hzGggHeM.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +8 -6
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +57 -6
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/bellman-ford.d.ts +60 -0
- package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
- package/dist/src/algorithms/bellman-ford.js +301 -0
- package/dist/src/algorithms/bellman-ford.js.map +1 -0
- package/dist/src/algorithms/bfs.d.ts +67 -0
- package/dist/src/algorithms/bfs.d.ts.map +1 -0
- package/dist/src/algorithms/bfs.js +534 -0
- package/dist/src/algorithms/bfs.js.map +1 -0
- package/dist/src/algorithms/closeness.d.ts +53 -0
- package/dist/src/algorithms/closeness.d.ts.map +1 -0
- package/dist/src/algorithms/closeness.js +323 -0
- package/dist/src/algorithms/closeness.js.map +1 -0
- package/dist/src/algorithms/scope.d.ts +3 -1
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +1 -0
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +72 -0
- package/dist/src/algorithms/sssp.d.ts.map +1 -0
- package/dist/src/algorithms/sssp.js +586 -0
- package/dist/src/algorithms/sssp.js.map +1 -0
- package/dist/src/constants.d.ts +10 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +10 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/index.d.ts +8 -2
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +7 -1
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts +4 -4
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +39 -5
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernel/uniform-ring.d.ts +8 -0
- package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
- package/dist/src/kernel/uniform-ring.js +13 -0
- package/dist/src/kernel/uniform-ring.js.map +1 -1
- package/dist/src/kernels.d.ts +45 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +368 -3
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +63 -0
- package/dist/src/primitives/advance.d.ts.map +1 -0
- package/dist/src/primitives/advance.js +95 -0
- package/dist/src/primitives/advance.js.map +1 -0
- package/dist/src/primitives/compact.d.ts +89 -0
- package/dist/src/primitives/compact.d.ts.map +1 -0
- package/dist/src/primitives/compact.js +233 -0
- package/dist/src/primitives/compact.js.map +1 -0
- package/dist/src/primitives/core-shape.d.ts +22 -1
- package/dist/src/primitives/core-shape.d.ts.map +1 -1
- package/dist/src/primitives/core-shape.js +33 -3
- package/dist/src/primitives/core-shape.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +151 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -0
- package/dist/src/primitives/frontier.js +250 -0
- package/dist/src/primitives/frontier.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +16 -7
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/traversal.d.ts +53 -0
- package/dist/src/types/traversal.d.ts.map +1 -0
- package/dist/src/types/traversal.js +10 -0
- package/dist/src/types/traversal.js.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
- package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +53 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js +164 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
- package/dist/webgpu-graph-algorithms.js +3155 -384
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +5 -4
- package/src/accelerator.ts +65 -7
- package/src/algorithms/bellman-ford.ts +387 -0
- package/src/algorithms/bfs.ts +626 -0
- package/src/algorithms/closeness.ts +395 -0
- package/src/algorithms/scope.ts +4 -1
- package/src/algorithms/sssp.ts +768 -0
- package/src/constants.ts +10 -0
- package/src/index.ts +14 -1
- package/src/kernel/prelude.ts +39 -4
- package/src/kernel/uniform-ring.ts +14 -0
- package/src/kernels.ts +447 -6
- package/src/primitives/advance.ts +131 -0
- package/src/primitives/compact.ts +323 -0
- package/src/primitives/core-shape.ts +41 -3
- package/src/primitives/frontier.ts +372 -0
- package/src/types/accelerator.ts +18 -5
- package/src/types/traversal.ts +56 -0
- package/src/wgsl/advance-expand.wgsl.ts +68 -0
- package/src/wgsl/bf-relax.wgsl.ts +57 -0
- package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
- package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
- package/src/wgsl/bfs-contract.wgsl.ts +54 -0
- package/src/wgsl/bfs-fused.wgsl.ts +77 -0
- package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
- package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
- package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
- package/src/wgsl/compact-scatter.wgsl.ts +16 -0
- package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
- package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
- package/src/wgsl/frontier-finalize.wgsl.ts +163 -0
- package/src/wgsl/sssp-pred.wgsl.ts +79 -0
- package/src/wgsl/sssp-relax.wgsl.ts +71 -0
- package/dist/chunks/context-BXqgCifx.js.map +0 -1
|
@@ -0,0 +1,372 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `Frontier` of design 6 row 7 and the device-side dispatch selector of design 5.4 (P8-T4; the P8 plan's PD-1,
|
|
3
|
+
* PD-3, PD-8, PD-23, DEP-P8-A, DEP-P8-C). A traversal's per-level state is two n-slot vertex queues, ONE 25-word
|
|
4
|
+
* counters block (every counter of the phase is a word of it: a four-byte word is never a legal storage-binding
|
|
5
|
+
* offset, and the selector must reach every count it acts on through one binding) and the edge queue. The host
|
|
6
|
+
* never reads a counter inside a submit: `frontier-finalize`, one lane, runs at the START of every level (role 0:
|
|
7
|
+
* rotates `nextFrontierCount` into `frontierCount`, advances `level`, decides `done`, chooses the level's path) and
|
|
8
|
+
* again once the edge queue is filled (role 1: clamps `edgeCount`, or on an overflow -- `edgeCountUnclamped >
|
|
9
|
+
* edgeCapacity` -- switches the path to the fused retry, PD-23). The choice is the block's `path` word (`W.path`,
|
|
10
|
+
* word 24); every level kernel is a direct grid-stride dispatch that reads it first and does nothing unless the
|
|
11
|
+
* word names it (design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md: the seven indirect slots the
|
|
12
|
+
* selector once wrote per level cost about 0.4 ms of Dawn validation each and were 97 % of a traversal's wall
|
|
13
|
+
* time; they and their args buffer are gone). The host records up to `MAX_LEVELS_PER_SUBMIT` levels per submit and
|
|
14
|
+
* reads `done` (four bytes) once per submit; a boundary that finds `done` set writes path 0 and moves no counter,
|
|
15
|
+
* so the recorded levels past the end are no-ops and the counters freeze at the finishing boundary's values.
|
|
16
|
+
*
|
|
17
|
+
* The seed: `frontierCount` is NEVER seeded, because the first boundary rotates it out unread. A BFS driver calls
|
|
18
|
+
* `reset(queue, source, { nextFrontierCount: 1, level: U32_MAX })`: the first boundary rotates the 1 in, adds it into
|
|
19
|
+
* `visitedCount`, and wraps `level` to 0, so the level-0 expansion claims the source's neighbours at `level + 1 == 1`.
|
|
20
|
+
* `reset` is a queue write, ordered before the submit that follows, and it puts the source on side 0.
|
|
21
|
+
*
|
|
22
|
+
* Every buffer comes from the caller's ONE lease (`scope.scratch`) so the algorithm's dispose()
|
|
23
|
+
* releases them together (design 4.4). The two vertex queues are two BUFFERS, never two ranges of one: `Kernel.bind`
|
|
24
|
+
* rejects one buffer bound read-only and read-write in one dispatch even for disjoint ranges. `src/primitives/**`
|
|
25
|
+
* never imports `src/context.ts`.
|
|
26
|
+
*/
|
|
27
|
+
|
|
28
|
+
import { MAX_LEVELS_PER_SUBMIT, U32_MAX } from "../constants.js";
|
|
29
|
+
import { WebGpuGraphError } from "../errors.js";
|
|
30
|
+
import { plan1d } from "../kernel/dispatch.js";
|
|
31
|
+
import { type Kernel } from "../kernel/kernel.js";
|
|
32
|
+
import { FRONTIER_COUNTERS, FRONTIER_PARAMS, kernelSpec } from "../kernels.js";
|
|
33
|
+
import { type Binding } from "../types/memory.js";
|
|
34
|
+
import { type ReduceScope } from "./reduce.js";
|
|
35
|
+
|
|
36
|
+
/** The values of the `path` word (`W.path`): what a level's kernels run; every level kernel reads it first. */
|
|
37
|
+
export const PATH: Readonly<{
|
|
38
|
+
none: 0;
|
|
39
|
+
twoPhase: 1;
|
|
40
|
+
fused: 2;
|
|
41
|
+
bottomUp: 3;
|
|
42
|
+
fusedRetry: 4;
|
|
43
|
+
near: 5;
|
|
44
|
+
far: 6;
|
|
45
|
+
}> = Object.freeze({ none: 0, twoPhase: 1, fused: 2, bottomUp: 3, fusedRetry: 4, near: 5, far: 6 });
|
|
46
|
+
|
|
47
|
+
/** The words of the counters block (`FRONTIER_COUNTERS`), by index: byte offset 4 x word; no driver types a number. */
|
|
48
|
+
export const W: Readonly<{
|
|
49
|
+
frontierCount: 0;
|
|
50
|
+
nextFrontierCount: 1;
|
|
51
|
+
frontierDegreeSum: 2;
|
|
52
|
+
prevFrontierCount: 3;
|
|
53
|
+
prevDegreeSum: 4;
|
|
54
|
+
unvisitedCount: 5;
|
|
55
|
+
unvisitedDegreeSum: 6;
|
|
56
|
+
unvisitedListLen: 7;
|
|
57
|
+
edgeCount: 8;
|
|
58
|
+
edgeCountUnclamped: 9;
|
|
59
|
+
overflowLevels: 10;
|
|
60
|
+
level: 11;
|
|
61
|
+
visitedCount: 12;
|
|
62
|
+
switches: 13;
|
|
63
|
+
direction: 14;
|
|
64
|
+
done: 15;
|
|
65
|
+
arcsScanned: 16;
|
|
66
|
+
fusedLevels: 17;
|
|
67
|
+
twoPhaseLevels: 18;
|
|
68
|
+
bottomUpLevels: 19;
|
|
69
|
+
farCount: 20;
|
|
70
|
+
nextFarCount: 21;
|
|
71
|
+
thresholdBits: 22;
|
|
72
|
+
deltaBits: 23;
|
|
73
|
+
path: 24;
|
|
74
|
+
}> = Object.freeze({
|
|
75
|
+
frontierCount: 0,
|
|
76
|
+
nextFrontierCount: 1,
|
|
77
|
+
frontierDegreeSum: 2,
|
|
78
|
+
prevFrontierCount: 3,
|
|
79
|
+
prevDegreeSum: 4,
|
|
80
|
+
unvisitedCount: 5,
|
|
81
|
+
unvisitedDegreeSum: 6,
|
|
82
|
+
unvisitedListLen: 7,
|
|
83
|
+
edgeCount: 8,
|
|
84
|
+
edgeCountUnclamped: 9,
|
|
85
|
+
overflowLevels: 10,
|
|
86
|
+
level: 11,
|
|
87
|
+
visitedCount: 12,
|
|
88
|
+
switches: 13,
|
|
89
|
+
direction: 14,
|
|
90
|
+
done: 15,
|
|
91
|
+
arcsScanned: 16,
|
|
92
|
+
fusedLevels: 17,
|
|
93
|
+
twoPhaseLevels: 18,
|
|
94
|
+
bottomUpLevels: 19,
|
|
95
|
+
farCount: 20,
|
|
96
|
+
nextFarCount: 21,
|
|
97
|
+
thresholdBits: 22,
|
|
98
|
+
deltaBits: 23,
|
|
99
|
+
path: 24,
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
/** The words a `reset` seeds (every other word is zeroed). */
|
|
103
|
+
export type FrontierSeed = Readonly<Partial<Record<keyof typeof W, number>>>;
|
|
104
|
+
|
|
105
|
+
/** The `FrontierParams` fields a caller passes to `recordFinalize`; the planner fills `role`, `wg`, `edgeCapacity` and `n` itself. A missing field is written as 0. */
|
|
106
|
+
export type FrontierFinalizeFields = Readonly<
|
|
107
|
+
Partial<
|
|
108
|
+
Record<
|
|
109
|
+
| "alpha"
|
|
110
|
+
| "beta"
|
|
111
|
+
| "fusedMax"
|
|
112
|
+
| "maxDepth"
|
|
113
|
+
| "mode"
|
|
114
|
+
| "cutoffBits"
|
|
115
|
+
| "arcBase"
|
|
116
|
+
| "arcEnd"
|
|
117
|
+
| "predKind"
|
|
118
|
+
| "bitsBase"
|
|
119
|
+
| "source"
|
|
120
|
+
| "stride"
|
|
121
|
+
| "firstOfSubmit"
|
|
122
|
+
| "iteration",
|
|
123
|
+
number
|
|
124
|
+
>
|
|
125
|
+
>
|
|
126
|
+
>;
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* The given words of a partial record (a key spelled with an undefined value is dropped, so the block writer sees
|
|
130
|
+
* only numbers).
|
|
131
|
+
* @param words - the partial record
|
|
132
|
+
* @returns the defined entries
|
|
133
|
+
*/
|
|
134
|
+
function definedWords(words: Readonly<Partial<Record<string, number>>>): Record<string, number> {
|
|
135
|
+
const out: Record<string, number> = {};
|
|
136
|
+
for (const [name, value] of Object.entries(words)) {
|
|
137
|
+
if (value !== undefined) {
|
|
138
|
+
out[name] = value;
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
return out;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/** The largest role `frontier-finalize` knows: 0 and 1 land here (P8-T4), 2 and 3 are the SSSP piles' (P8-T9). */
|
|
145
|
+
const MAX_ROLE = 3;
|
|
146
|
+
|
|
147
|
+
/**
|
|
148
|
+
* The E_INVALID_ARGUMENT of a value that is not a non-negative integer.
|
|
149
|
+
* @param argument - the argument name
|
|
150
|
+
* @param value - the value
|
|
151
|
+
*/
|
|
152
|
+
function assertCount(argument: string, value: number): void {
|
|
153
|
+
if (!Number.isSafeInteger(value) || value < 0) {
|
|
154
|
+
throw new WebGpuGraphError("E_INVALID_ARGUMENT", `frontier: ${argument} must be a non-negative integer`, {
|
|
155
|
+
argument,
|
|
156
|
+
value,
|
|
157
|
+
expected: "a non-negative integer",
|
|
158
|
+
});
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/** The frontier queue of design 6 row 7: two vertex queues, the counters block and the edge queue, all leased by the caller's scope. */
|
|
163
|
+
export class Frontier {
|
|
164
|
+
/** The two n-slot u32 vertex queues (`vertices[side]` is the input of the current level). */
|
|
165
|
+
readonly vertices: readonly [Binding, Binding];
|
|
166
|
+
/** The `FrontierCounters` block, 112 B, bound by every kernel as `array<atomic<u32>>`. */
|
|
167
|
+
readonly counters: Binding;
|
|
168
|
+
/** The edge queue: `edgeCapacity` entries of u32 (the target vertex of an arc). */
|
|
169
|
+
readonly edgeQueue: Binding;
|
|
170
|
+
/** How many entries the edge queue holds; role 1 clamps `edgeCount` to it and detects an overflow above it. */
|
|
171
|
+
readonly edgeCapacity: number;
|
|
172
|
+
/** The vertex count (a `reset` source is below it). */
|
|
173
|
+
readonly n: number;
|
|
174
|
+
private sideIndex: 0 | 1 = 0;
|
|
175
|
+
|
|
176
|
+
/**
|
|
177
|
+
* Wraps the leased buffers; use prepareFrontier().
|
|
178
|
+
* @param vertices - the two vertex queues
|
|
179
|
+
* @param counters - the counters block
|
|
180
|
+
* @param edgeQueue - the edge queue
|
|
181
|
+
* @param edgeCapacity - the edge queue's entry count
|
|
182
|
+
* @param n - the vertex count
|
|
183
|
+
*/
|
|
184
|
+
constructor(
|
|
185
|
+
vertices: readonly [Binding, Binding],
|
|
186
|
+
counters: Binding,
|
|
187
|
+
edgeQueue: Binding,
|
|
188
|
+
edgeCapacity: number,
|
|
189
|
+
n: number,
|
|
190
|
+
) {
|
|
191
|
+
this.vertices = vertices;
|
|
192
|
+
this.counters = counters;
|
|
193
|
+
this.edgeQueue = edgeQueue;
|
|
194
|
+
this.edgeCapacity = edgeCapacity;
|
|
195
|
+
this.n = n;
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
/**
|
|
199
|
+
* Which vertex queue is the input of the current level (a host-side index between two cached bind-group sets).
|
|
200
|
+
* @returns 0 or 1
|
|
201
|
+
*/
|
|
202
|
+
get side(): 0 | 1 {
|
|
203
|
+
return this.sideIndex;
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
/**
|
|
207
|
+
* The vertex queue the current level expands from.
|
|
208
|
+
* @returns `vertices[side]`
|
|
209
|
+
*/
|
|
210
|
+
get input(): Binding {
|
|
211
|
+
return this.sideIndex === 0 ? this.vertices[0] : this.vertices[1];
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
/**
|
|
215
|
+
* The vertex queue the current level's claims append to.
|
|
216
|
+
* @returns `vertices[1 - side]`
|
|
217
|
+
*/
|
|
218
|
+
get output(): Binding {
|
|
219
|
+
return this.sideIndex === 0 ? this.vertices[1] : this.vertices[0];
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
/** Flips the two vertex queues; the counts rotate inside the block, so nothing else moves. */
|
|
223
|
+
swap(): void {
|
|
224
|
+
this.sideIndex = this.sideIndex === 0 ? 1 : 0;
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
/**
|
|
228
|
+
* Seeds a traversal: one `queue.writeBuffer` of the whole 112-byte block (zero except the caller's words) and one of
|
|
229
|
+
* `vertices[0][0] = source`, both ordered before the submit that follows; the source is on side 0 afterwards.
|
|
230
|
+
* `frontierCount` is not a word to seed: the first boundary rotates word 1 into it (the BFS seed is
|
|
231
|
+
* `{ nextFrontierCount: 1, level: U32_MAX }`). A source outside `[0, n)`, an unknown word or a value that is not a
|
|
232
|
+
* u32 is E_INVALID_ARGUMENT before anything is written.
|
|
233
|
+
* @param queue - the device queue
|
|
234
|
+
* @param source - the source vertex
|
|
235
|
+
* @param seed - the words to seed
|
|
236
|
+
*/
|
|
237
|
+
reset(queue: GPUQueue, source: number, seed: FrontierSeed): void {
|
|
238
|
+
if (!Number.isInteger(source) || source < 0 || source >= this.n) {
|
|
239
|
+
throw new WebGpuGraphError("E_INVALID_ARGUMENT", `frontier: source ${source} is outside [0, ${this.n})`, {
|
|
240
|
+
argument: "source",
|
|
241
|
+
value: source,
|
|
242
|
+
expected: `an integer in [0, ${this.n})`,
|
|
243
|
+
});
|
|
244
|
+
}
|
|
245
|
+
const bytes = new ArrayBuffer(FRONTIER_COUNTERS.byteLength);
|
|
246
|
+
FRONTIER_COUNTERS.write(new DataView(bytes), definedWords(seed));
|
|
247
|
+
this.sideIndex = 0;
|
|
248
|
+
queue.writeBuffer(this.counters.buffer, this.counters.offset, bytes);
|
|
249
|
+
queue.writeBuffer(this.vertices[0].buffer, this.vertices[0].offset, Uint32Array.of(source));
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/** A prepared frontier (design 6 row 7): the leased queue and the recorded selector dispatches. */
|
|
254
|
+
export interface FrontierPlanner {
|
|
255
|
+
/** The queue the planner's selector rotates and chooses the path of. */
|
|
256
|
+
readonly frontier: Frontier;
|
|
257
|
+
/**
|
|
258
|
+
* Records one `frontier-finalize` dispatch (one workgroup) in `role` for `level` of the current submit: one
|
|
259
|
+
* `FrontierParams` record with `wg`, `edgeCapacity` and `n` filled by the planner and every other field from
|
|
260
|
+
* `fields`. A level outside `[0, MAX_LEVELS_PER_SUBMIT)` (the selector addresses nothing by level any more, but
|
|
261
|
+
* the host's submit cadence still is the bound) or a role outside `[0, 3]` is E_INVALID_ARGUMENT before
|
|
262
|
+
* anything is recorded.
|
|
263
|
+
* @param pass - the compute pass
|
|
264
|
+
* @param role - 0 the level boundary, 1 the edge-queue role (2 and 3 are P8-T9's)
|
|
265
|
+
* @param level - the level inside the submit
|
|
266
|
+
* @param fields - the tuning and window words
|
|
267
|
+
*/
|
|
268
|
+
recordFinalize(pass: GPUComputePassEncoder, role: number, level: number, fields: FrontierFinalizeFields): void;
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
/**
|
|
272
|
+
* Leases the frontier's buffers from the scope and compiles the selector so `recordFinalize` is synchronous. The
|
|
273
|
+
* edge capacity defaults to `max(1, min(arcCount, floor(maxStorageBufferBindingSize / 4)))` (never a zero-length
|
|
274
|
+
* buffer: the one-node graph has no arcs); a test passes a small one to force the overflow path. The planner lives
|
|
275
|
+
* exactly as long as the scope: never use it after the scope's dispose().
|
|
276
|
+
* @param scope - the caller's scope (device, caps, cache, scratch, params)
|
|
277
|
+
* @param n - the vertex count
|
|
278
|
+
* @param arcCount - the arc count (the natural edge-queue size)
|
|
279
|
+
* @param edgeCapacity - the edge queue's entry count, when the caller chooses it (an integer >= 1)
|
|
280
|
+
* @returns the planner
|
|
281
|
+
*/
|
|
282
|
+
export async function prepareFrontier(
|
|
283
|
+
scope: ReduceScope,
|
|
284
|
+
n: number,
|
|
285
|
+
arcCount: number,
|
|
286
|
+
edgeCapacity?: number,
|
|
287
|
+
): Promise<FrontierPlanner> {
|
|
288
|
+
assertCount("n", n);
|
|
289
|
+
assertCount("arcCount", arcCount);
|
|
290
|
+
const capacity =
|
|
291
|
+
edgeCapacity ?? Math.max(1, Math.min(arcCount, Math.floor(scope.caps.limits.maxStorageBufferBindingSize / 4)));
|
|
292
|
+
if (!Number.isSafeInteger(capacity) || capacity < 1 || capacity > U32_MAX) {
|
|
293
|
+
throw new WebGpuGraphError("E_INVALID_ARGUMENT", "frontier: edgeCapacity must be an integer >= 1", {
|
|
294
|
+
argument: "edgeCapacity",
|
|
295
|
+
value: capacity,
|
|
296
|
+
expected: "an integer in [1, 2^32)",
|
|
297
|
+
});
|
|
298
|
+
}
|
|
299
|
+
const kernel = await scope.pipelines.kernel(kernelSpec("frontier-finalize"));
|
|
300
|
+
const queueBytes = 4 * Math.max(1, n);
|
|
301
|
+
const vertices: readonly [Binding, Binding] = [
|
|
302
|
+
{ buffer: scope.scratch(queueBytes, "frontier/vertices-0"), offset: 0, size: queueBytes, window: null },
|
|
303
|
+
{ buffer: scope.scratch(queueBytes, "frontier/vertices-1"), offset: 0, size: queueBytes, window: null },
|
|
304
|
+
];
|
|
305
|
+
const counters: Binding = {
|
|
306
|
+
buffer: scope.scratch(FRONTIER_COUNTERS.byteLength, "frontier/counters"),
|
|
307
|
+
offset: 0,
|
|
308
|
+
size: FRONTIER_COUNTERS.byteLength,
|
|
309
|
+
window: null,
|
|
310
|
+
};
|
|
311
|
+
const edgeQueue: Binding = {
|
|
312
|
+
buffer: scope.scratch(4 * capacity, "frontier/edge-queue"),
|
|
313
|
+
offset: 0,
|
|
314
|
+
size: 4 * capacity,
|
|
315
|
+
window: null,
|
|
316
|
+
};
|
|
317
|
+
const frontier = new Frontier(vertices, counters, edgeQueue, capacity, n);
|
|
318
|
+
return new FrontierPlannerImpl(scope, kernel, frontier);
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
/** The planner: the selector kernel bound once to the frontier's block. */
|
|
322
|
+
class FrontierPlannerImpl implements FrontierPlanner {
|
|
323
|
+
readonly frontier: Frontier;
|
|
324
|
+
private readonly scope: ReduceScope;
|
|
325
|
+
private readonly kernel: Kernel;
|
|
326
|
+
|
|
327
|
+
/**
|
|
328
|
+
* Wraps the resolved kernel; use prepareFrontier().
|
|
329
|
+
* @param scope - the caller's scope
|
|
330
|
+
* @param kernel - the `frontier-finalize` kernel
|
|
331
|
+
* @param frontier - the leased queue
|
|
332
|
+
*/
|
|
333
|
+
constructor(scope: ReduceScope, kernel: Kernel, frontier: Frontier) {
|
|
334
|
+
this.scope = scope;
|
|
335
|
+
this.kernel = kernel;
|
|
336
|
+
this.frontier = frontier;
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
/**
|
|
340
|
+
* Records one selector dispatch (see the interface).
|
|
341
|
+
* @param pass - the compute pass
|
|
342
|
+
* @param role - the role
|
|
343
|
+
* @param level - the level inside the submit
|
|
344
|
+
* @param fields - the caller's fields
|
|
345
|
+
*/
|
|
346
|
+
recordFinalize(pass: GPUComputePassEncoder, role: number, level: number, fields: FrontierFinalizeFields): void {
|
|
347
|
+
if (!Number.isInteger(level) || level < 0 || level >= MAX_LEVELS_PER_SUBMIT) {
|
|
348
|
+
throw new WebGpuGraphError("E_INVALID_ARGUMENT", `frontier: level ${level} is outside the submit`, {
|
|
349
|
+
argument: "level",
|
|
350
|
+
value: level,
|
|
351
|
+
expected: `an integer in [0, ${MAX_LEVELS_PER_SUBMIT})`,
|
|
352
|
+
});
|
|
353
|
+
}
|
|
354
|
+
if (!Number.isInteger(role) || role < 0 || role > MAX_ROLE) {
|
|
355
|
+
throw new WebGpuGraphError("E_INVALID_ARGUMENT", `frontier: role ${role} is not a finalize role`, {
|
|
356
|
+
argument: "role",
|
|
357
|
+
value: role,
|
|
358
|
+
expected: `an integer in [0, ${MAX_ROLE}]`,
|
|
359
|
+
});
|
|
360
|
+
}
|
|
361
|
+
const { scope, frontier } = this;
|
|
362
|
+
const params = scope.params(FRONTIER_PARAMS, {
|
|
363
|
+
...definedWords(fields),
|
|
364
|
+
role,
|
|
365
|
+
wg: scope.workgroupSize,
|
|
366
|
+
edgeCapacity: frontier.edgeCapacity,
|
|
367
|
+
n: frontier.n,
|
|
368
|
+
});
|
|
369
|
+
const bound = this.kernel.bind({ counters: frontier.counters, P: params.binding });
|
|
370
|
+
this.kernel.dispatch(pass, bound, plan1d(1, scope.workgroupSize, scope.caps), [params.offset]);
|
|
371
|
+
}
|
|
372
|
+
}
|
package/src/types/accelerator.ts
CHANGED
|
@@ -10,6 +10,7 @@ import type {
|
|
|
10
10
|
ApspResultLike,
|
|
11
11
|
BellmanFordResultLike,
|
|
12
12
|
BetweennessAcceleratorOptions,
|
|
13
|
+
BfsOptions,
|
|
13
14
|
BfsResultLike,
|
|
14
15
|
CommunityResultLike,
|
|
15
16
|
CorenessResultLike,
|
|
@@ -20,6 +21,7 @@ import type {
|
|
|
20
21
|
MstResultLike,
|
|
21
22
|
PageRankResultLike,
|
|
22
23
|
ScoresResultLike,
|
|
24
|
+
SsspOptions,
|
|
23
25
|
SsspResultLike,
|
|
24
26
|
} from "@graphty/algorithms";
|
|
25
27
|
import type { F32, F64, GraphSnapshot } from "@graphty/graph-format";
|
|
@@ -45,6 +47,7 @@ import type {
|
|
|
45
47
|
SpringElectricalStats,
|
|
46
48
|
} from "./layout.js";
|
|
47
49
|
import type { ForceAtlas2Options, FruchtermanReingoldOptions, SpringElectricalOptions } from "./options.js";
|
|
50
|
+
import type { GpuBellmanFordResult, GpuBfsResult, GpuSsspResult } from "./traversal.js";
|
|
48
51
|
|
|
49
52
|
// ---- the real @graphty/layout interfaces (spec 9.3, D27): imported at W1b, re-exported so the package's public
|
|
50
53
|
// surface is unchanged and src/types/layout.ts keeps resolving them from here. `export type`, never a bare
|
|
@@ -61,6 +64,7 @@ export type {
|
|
|
61
64
|
ApspResultLike,
|
|
62
65
|
BellmanFordResultLike,
|
|
63
66
|
BetweennessAcceleratorOptions,
|
|
67
|
+
BfsOptions,
|
|
64
68
|
BfsResultLike,
|
|
65
69
|
CommunityResultLike,
|
|
66
70
|
CorenessResultLike,
|
|
@@ -71,6 +75,7 @@ export type {
|
|
|
71
75
|
MstResultLike,
|
|
72
76
|
PageRankResultLike,
|
|
73
77
|
ScoresResultLike,
|
|
78
|
+
SsspOptions,
|
|
74
79
|
SsspResultLike,
|
|
75
80
|
};
|
|
76
81
|
|
|
@@ -95,11 +100,15 @@ export interface AcceleratorOptions {
|
|
|
95
100
|
* The injectable object (spec 3.3): P3's forceAtlas2, release and dispose, P5's fruchtermanReingold and
|
|
96
101
|
* springElectrical (the two other optional members of the real LayoutAccelerator, spec 9.3; the CPU option types in,
|
|
97
102
|
* the GPU simulations out), plus P7's seven algorithm members
|
|
98
|
-
* (spec 8.2, 8.3; M8b-T8)
|
|
99
|
-
* mirrors (spec 9.7: `precision` is an extra field, `F32` is
|
|
100
|
-
* `weaklyConnectedComponents` are the same algorithm (spec 3.3: WCC
|
|
101
|
-
* the mirror declares; their options parameter stays OPTIONAL, because
|
|
102
|
-
* REQUIRED parameter would stop the member satisfying it.
|
|
103
|
+
* (spec 8.2, 8.3; M8b-T8) and P8's four traversal members (spec 8.4; P8-T13 PD-16), non-optional here and returning
|
|
104
|
+
* the `Gpu*Result` shapes, which satisfy the `*ResultLike` mirrors (spec 9.7: `precision` is an extra field, `F32` is
|
|
105
|
+
* a `NumericVector`). `connectedComponents` and `weaklyConnectedComponents` are the same algorithm (spec 3.3: WCC
|
|
106
|
+
* semantics on directed input) under both names the mirror declares; their options parameter stays OPTIONAL, because
|
|
107
|
+
* the mirror declares none and an extra REQUIRED parameter would stop the member satisfying it. The four traversals
|
|
108
|
+
* take the seam's OWN option types (PD-19: `BfsOptions`, `SsspOptions` for both `sssp` and `bellmanFord`,
|
|
109
|
+
* `HitsOptionsLike` for `closenessCentrality`), so a key the CPU dispatcher forwards is exactly a key the GPU reads;
|
|
110
|
+
* `test/types/conformance.test-d.ts` holds each parameter EQUAL to the seam's, not merely assignable. Later phases add
|
|
111
|
+
* one member per shipped algorithm.
|
|
103
112
|
* Exported: implemented by src/accelerator.ts (P3-T3); re-exported from src/index.ts at P3-T3.
|
|
104
113
|
* @public
|
|
105
114
|
*/
|
|
@@ -123,6 +132,10 @@ export interface GpuAccelerator extends AlgorithmAccelerator, LayoutAccelerator
|
|
|
123
132
|
katzCentrality(s: GraphSnapshot, options?: KatzOptions): Promise<GpuScoresResult>;
|
|
124
133
|
connectedComponents(s: GraphSnapshot, options?: ComponentsOptions): Promise<GpuLabelResult>;
|
|
125
134
|
weaklyConnectedComponents(s: GraphSnapshot, options?: ComponentsOptions): Promise<GpuLabelResult>;
|
|
135
|
+
breadthFirstSearch(s: GraphSnapshot, source: number, options?: BfsOptions): Promise<GpuBfsResult>;
|
|
136
|
+
sssp(s: GraphSnapshot, source: number, options?: SsspOptions): Promise<GpuSsspResult>;
|
|
137
|
+
bellmanFord(s: GraphSnapshot, source: number, options?: SsspOptions): Promise<GpuBellmanFordResult>;
|
|
138
|
+
closenessCentrality(s: GraphSnapshot, options?: HitsOptionsLike): Promise<GpuScoresResult>;
|
|
126
139
|
release(s: GraphSnapshot): void;
|
|
127
140
|
dispose(): void;
|
|
128
141
|
}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The result records of the P8 frontier family (design 3.3 lines 830-832, 9.7), verbatim. The OPTION types are the
|
|
3
|
+
* seam's own -- `BfsOptions` and `SsspOptions` from `@graphty/algorithms`, re-exported through
|
|
4
|
+
* `src/types/accelerator.ts` beside `HitsOptionsLike` -- and are deliberately NOT redeclared here (P8 PD-19): a
|
|
5
|
+
* member the CPU dispatcher forwards options into must read exactly the keys the CPU port reads. Every sentinel is
|
|
6
|
+
* spelled on its field, because a consumer who reads `parent[root] === 4294967295` and guesses is the failure this
|
|
7
|
+
* file prevents. Types only: this file imports nothing at runtime.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import type { F32, U32 } from "@graphty/graph-format";
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* Design 3.3 line 830: what `breadthFirstSearch` returns. Satisfies the seam's `BfsResultLike` (`depth`, `parent`,
|
|
14
|
+
* `order`, `visitedCount`) with `levels` and `switches` on top. Every array is bitwise reproducible (PD-14).
|
|
15
|
+
*/
|
|
16
|
+
export interface GpuBfsResult {
|
|
17
|
+
/** Per node: its hop count from the source; `INVALID_INDEX` (4294967295) = unreached. */
|
|
18
|
+
readonly depth: U32;
|
|
19
|
+
/**
|
|
20
|
+
* Per node: the SMALLEST node index `u` with `depth[u] + 1 == depth[v]` and an arc `u -> v` (PD-24: a post-pass over
|
|
21
|
+
* the settled depths, so the chain to the source strictly decreases `depth`); `INVALID_INDEX` for the source and
|
|
22
|
+
* for an unreached node.
|
|
23
|
+
*/
|
|
24
|
+
readonly parent: U32;
|
|
25
|
+
/** Length `visitedCount`: the reached nodes grouped by depth, ascending by node index within a depth (PD-14). */
|
|
26
|
+
readonly order: U32;
|
|
27
|
+
/** How many nodes were reached, the source included. */
|
|
28
|
+
readonly visitedCount: number;
|
|
29
|
+
/** `max depth + 1`; 1 for a source with no out-arcs. */
|
|
30
|
+
readonly levels: number;
|
|
31
|
+
/** Direction changes of the direction-optimizing search (a device counter, design 8.4); 0 on a top-down-only run. */
|
|
32
|
+
readonly switches: number;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Design 3.3 line 831: what `sssp` returns. Satisfies the seam's `SsspResultLike` (`dist: NumericVector` admits `F32`;
|
|
37
|
+
* `predArc: U32`). `dist` is bitwise reproducible (PD-9) and `predArc` is a function of it alone (PD-27).
|
|
38
|
+
*/
|
|
39
|
+
export interface GpuSsspResult {
|
|
40
|
+
/** Per node: its shortest distance from the source in f32; `+Infinity` = unreached, which includes beyond `cutoff`. */
|
|
41
|
+
readonly dist: F32;
|
|
42
|
+
/**
|
|
43
|
+
* Per node `v`: a TIGHT arc `a` into `v` (`fround(dist[src(a)] + w(a)) == dist[v]`) whose source sits one PD-27 key
|
|
44
|
+
* step below `v`, the smallest such arc index; the chain `v -> src(a) -> ...` always ends at the source.
|
|
45
|
+
* `INVALID_INDEX` (4294967295) for the source and for an unreached node.
|
|
46
|
+
*/
|
|
47
|
+
readonly predArc: U32;
|
|
48
|
+
/** How many nodes have a finite `dist`, the source included. */
|
|
49
|
+
readonly reachedCount: number;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** Design 3.3 line 832: what `bellmanFord` returns. Satisfies the seam's `BellmanFordResultLike`. */
|
|
53
|
+
export interface GpuBellmanFordResult extends GpuSsspResult {
|
|
54
|
+
/** When true, `dist` and `predArc` are the values of the last round, not shortest paths. */
|
|
55
|
+
readonly hasNegativeCycle: boolean;
|
|
56
|
+
}
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `advance-expand` kernel body (design 6 row 8, 8.10 "BFS expand"; P8-T5, the P8 plan's PD-23): Gunrock's
|
|
3
|
+
* block-mapped expansion of a vertex frontier into the edge queue. Each workgroup loads up to `WG` frontier entries,
|
|
4
|
+
* reads their degrees clipped to the bound arc window `[P.arcBase, P.arcEnd)` (so the window-aware form of P8-T12
|
|
5
|
+
* changes nothing here), runs the prelude's INCLUSIVE workgroup scan `wg_scan_u32` over those degrees -- the subgroup
|
|
6
|
+
* form when the device has the feature, the Hillis-Steele form otherwise, which is the ONLY text that differs between
|
|
7
|
+
* the twin's two compilations -- and then every invocation strips the range `[0, aggregate)` with a binary search
|
|
8
|
+
* (`upper_bound`) over the scanned degrees to find which entry its arc belongs to. One `atomicAdd` per WORKGROUP
|
|
9
|
+
* reserves the block's span in the queue (`edgeCount`), the same aggregate lands in `edgeCountUnclamped` (the overflow
|
|
10
|
+
* detector, never clamped) and in `frontierDegreeSum` (Beamer's m_f); a lane whose queue position is at or past
|
|
11
|
+
* `P.edgeCapacity` writes nothing (the clamp). The queue holds the TARGET vertex of each arc only (PD-24: `parent`
|
|
12
|
+
* comes from the post-pass). Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan call and the
|
|
13
|
+
* `workgroupUniformLoad` sit unconditionally after the guard, and the strip loop is bounded by a uniform value.
|
|
14
|
+
* There is no `TIER` override: a hub row is balanced over all `WG` lanes inside its block, and the small-frontier
|
|
15
|
+
* workgroup-per-row structure is `bfs-fused` (P8-T7). Body only (spec 3.5, D9); the text is normative: the
|
|
16
|
+
* sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
17
|
+
*/
|
|
18
|
+
export const advanceExpandWgsl = /* wgsl */ `
|
|
19
|
+
var<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan
|
|
20
|
+
var<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row
|
|
21
|
+
var<workgroup> base: u32; // the block's reserved span in the edge queue
|
|
22
|
+
var<workgroup> wcount: u32; // the frontier's length on a two-phase level, 0 on any other
|
|
23
|
+
|
|
24
|
+
@compute @workgroup_size(WG)
|
|
25
|
+
fn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
26
|
+
if (lid.x == 0u) { wcount = select(0u, atomicLoad(&counters[0]), atomicLoad(&counters[24]) == 1u); } // frontierCount, on the two-phase path only (the path word)
|
|
27
|
+
let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
|
|
28
|
+
for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries (a direct dispatch of the plan's groups)
|
|
29
|
+
let i = b0 + lid.x; // this lane's frontier entry
|
|
30
|
+
var deg = 0u;
|
|
31
|
+
var start = 0u;
|
|
32
|
+
if (i < count) { // guarded loads into locals (3.5 rule 1)
|
|
33
|
+
let v = frontierIn[i];
|
|
34
|
+
let lo = max(rowPtr[v], P.arcBase); // the row clipped to the bound window (P8-T12)
|
|
35
|
+
let hi = min(rowPtr[v + 1u], P.arcEnd);
|
|
36
|
+
start = lo;
|
|
37
|
+
deg = select(0u, hi - lo, hi > lo);
|
|
38
|
+
}
|
|
39
|
+
let inclusive = wg_scan_u32(deg, lid.x); // the prelude's inclusive scan in LANE order (P8-T1 Step 5); the twin
|
|
40
|
+
sh[lid.x] = inclusive; // sh is monotone in lid.x, the index the binary search walks
|
|
41
|
+
rowStart[lid.x] = start;
|
|
42
|
+
workgroupBarrier();
|
|
43
|
+
let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier
|
|
44
|
+
if (lid.x == 0u) {
|
|
45
|
+
base = atomicAdd(&counters[8], aggregate); // edgeCount: ONE reservation per workgroup, not one per arc
|
|
46
|
+
atomicAdd(&counters[9], aggregate); // edgeCountUnclamped: the overflow detector (PD-23)
|
|
47
|
+
atomicAdd(&counters[2], aggregate); // frontierDegreeSum: Beamer's m_f (P8-T8)
|
|
48
|
+
}
|
|
49
|
+
workgroupBarrier();
|
|
50
|
+
for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
|
|
51
|
+
var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p
|
|
52
|
+
var hi = WG;
|
|
53
|
+
loop {
|
|
54
|
+
if (lo >= hi) { break; }
|
|
55
|
+
let mid = (lo + hi) / 2u;
|
|
56
|
+
if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }
|
|
57
|
+
}
|
|
58
|
+
let k = lo;
|
|
59
|
+
var exclusive = 0u;
|
|
60
|
+
if (k > 0u) { exclusive = sh[k - 1u]; }
|
|
61
|
+
let arc = rowStart[k] + (p - exclusive);
|
|
62
|
+
let q = base + p;
|
|
63
|
+
if (q < P.edgeCapacity) { edgeQueue[q] = colIdx[arc - P.arcBase]; } // the clamp of PD-23; the queue holds the target vertex only (PD-24)
|
|
64
|
+
}
|
|
65
|
+
workgroupBarrier(); // sh, rowStart and base are reused by the next block
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
`;
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bf-relax` kernel body (design 8.4 "Bellman-Ford"; P8-T10, the P8 plan's PD-12 / PD-22 / PD-27 / DEP-P8-E):
|
|
3
|
+
* one edge-parallel relaxation round over the `edgeList()` view -- every logical edge once, in declared orientation,
|
|
4
|
+
* and under `UNDIRECTED` the other direction too, because an undirected edge has ONE weight (read through its
|
|
5
|
+
* forward arc, `weights[edgeToArc[e]]`, from the run's arc-indexed vector: the snapshot's column or the uploaded
|
|
6
|
+
* override). `dist` is `array<atomic<u32>>` holding the f32 bit patterns (`F32_INF_BITS` = unreached, tested as the
|
|
7
|
+
* pattern: `bitcast<f32>(F32_INF_BITS)` is a const-expression Tint rejects). `atomicMin` on the patterns is `min` on
|
|
8
|
+
* the values for NON-NEGATIVE floats only (PD-9); with a negative distance the bit-pattern order reverses, so the
|
|
9
|
+
* claim is a compare-exchange loop on the pattern: read, compute, compare as floats, try to exchange. WGSL lets
|
|
10
|
+
* `atomicCompareExchangeWeak` fail spuriously, so the loop is BOUNDED (PD-12: `P.maxRetries`, the driver's
|
|
11
|
+
* `MAX_RETRIES`; contention is per vertex, not global) and a lane that exhausts it sets `flags[1]`
|
|
12
|
+
* (`retryExhausted`): the driver treats the round as changed and runs on, so the lost update is retried by the next
|
|
13
|
+
* round, which examines every edge anyway; a bound hit in the decision round is E_VALIDATION, never a guess. A
|
|
14
|
+
* successful exchange sets `flags[0]` (`changed`); the driver stops when a batch of rounds changed nothing, and a
|
|
15
|
+
* change in the round after `n - 1` is the negative cycle. `P.cutoffBits` is the CPU port's `dv <= cutoff` guard
|
|
16
|
+
* (`+Inf` when absent; with negative weights a negative cutoff legitimately relaxes). Grid-strided over
|
|
17
|
+
* `planGridStride(edgeCount)` with `P.stride` the plan's stride; no barrier anywhere, so the loops may be per lane.
|
|
18
|
+
* The whole arc array is bound (never windowed, DEP-P8-E). Body only (spec 3.5, D9); the text is normative: the
|
|
19
|
+
* sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
20
|
+
*/
|
|
21
|
+
export const bfRelaxWgsl = /* wgsl */ `
|
|
22
|
+
fn relax(v: u32, nd: f32) {
|
|
23
|
+
var cur = atomicLoad(&dist[v]);
|
|
24
|
+
var tries = 0u;
|
|
25
|
+
loop {
|
|
26
|
+
if (!(nd < bitcast<f32>(cur))) { break; } // no improvement; +Inf is greater than every finite nd
|
|
27
|
+
let r = atomicCompareExchangeWeak(&dist[v], cur, bitcast<u32>(nd));
|
|
28
|
+
if (r.exchanged) { atomicStore(&flags[0], 1u); break; } // changed
|
|
29
|
+
cur = r.old_value;
|
|
30
|
+
tries = tries + 1u;
|
|
31
|
+
if (tries >= P.maxRetries) { atomicStore(&flags[1], 1u); break; } // retryExhausted (PD-12)
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
@compute @workgroup_size(WG)
|
|
36
|
+
fn bf_relax(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
37
|
+
let first = linear_id(wid, lid.x);
|
|
38
|
+
let cutoff = bitcast<f32>(P.cutoffBits);
|
|
39
|
+
for (var e = first; e < P.edgeCount; e = e + P.stride) { // each logical edge once (edgeList)
|
|
40
|
+
let u = edgeSrc[e];
|
|
41
|
+
let v = edgeDst[e];
|
|
42
|
+
let w = weights[edgeToArc[e]]; // the edge's weight through its forward arc
|
|
43
|
+
let du = atomicLoad(&dist[u]);
|
|
44
|
+
if (du != F32_INF_BITS) { // unreached is tested as the bit pattern, as sssp-pred does
|
|
45
|
+
let nd = bitcast<f32>(du) + w;
|
|
46
|
+
if (nd <= cutoff) { relax(v, nd); }
|
|
47
|
+
}
|
|
48
|
+
if (UNDIRECTED) { // the other direction of an undirected edge
|
|
49
|
+
let dv = atomicLoad(&dist[v]);
|
|
50
|
+
if (dv != F32_INF_BITS) {
|
|
51
|
+
let nd = bitcast<f32>(dv) + w;
|
|
52
|
+
if (nd <= cutoff) { relax(u, nd); }
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
`;
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bfs-bitset-build` kernel body (design 8.4 "the bitset frontier"; P8-T8): the vertex-list-to-bitset hand-off
|
|
3
|
+
* of a bottom-up level. One invocation per entry of the input frontier (`frontierCount`, word 0), each `atomicOr`ing
|
|
4
|
+
* its vertex's bit into the `ceil(n / 32)`-word bitset at `P.bitsBase` inside the `sweepIn` buffer (the region a
|
|
5
|
+
* `fill` zeroed just before, on every level). The design's "bulk non-atomic path when the frontier is >= 40% of
|
|
6
|
+
* n" iterates WORDS of a frontier that is already a bitset; this frontier is a vertex list, so a word-owning store
|
|
7
|
+
* has nothing to iterate and `atomicOr` per vertex is the whole kernel. The sweep appends what it claims as a plain
|
|
8
|
+
* vertex list, exactly as the contract does, and writes no second bitset: the next level's bits come from this
|
|
9
|
+
* kernel running over that list, so one representation of a frontier (a vertex list plus a count) serves the whole
|
|
10
|
+
* phase and the hand-off in either direction is free. The early return keys on a counter and `local_invocation_id`
|
|
11
|
+
* with no barrier after it (spec 3.5 rule 1). Body only (spec 3.5, D9); the text is normative: the sabotage rows of
|
|
12
|
+
* test/helpers/sabotage.ts are textual edits of it.
|
|
13
|
+
*/
|
|
14
|
+
export const bfsBitsetBuildWgsl = /* wgsl */ `
|
|
15
|
+
@compute @workgroup_size(WG)
|
|
16
|
+
fn bfs_bitset_build(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
17
|
+
let count = select(0u, atomicLoad(&counters[0]), atomicLoad(&counters[24]) == 3u); // frontierCount, on the bottom-up path only (the path word)
|
|
18
|
+
for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // grid-stride; no barrier anywhere
|
|
19
|
+
let v = frontierIn[i];
|
|
20
|
+
atomicOr(&bits[P.bitsBase + (v >> 5u)], 1u << (v & 31u));
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
`;
|