@graphty/webgpu-graph-algorithms 0.6.3 → 0.6.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +62 -32
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-BXqgCifx.js → context-hzGggHeM.js} +68 -24
- package/dist/chunks/context-hzGggHeM.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +8 -6
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +57 -6
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/bellman-ford.d.ts +60 -0
- package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
- package/dist/src/algorithms/bellman-ford.js +301 -0
- package/dist/src/algorithms/bellman-ford.js.map +1 -0
- package/dist/src/algorithms/bfs.d.ts +67 -0
- package/dist/src/algorithms/bfs.d.ts.map +1 -0
- package/dist/src/algorithms/bfs.js +534 -0
- package/dist/src/algorithms/bfs.js.map +1 -0
- package/dist/src/algorithms/closeness.d.ts +53 -0
- package/dist/src/algorithms/closeness.d.ts.map +1 -0
- package/dist/src/algorithms/closeness.js +323 -0
- package/dist/src/algorithms/closeness.js.map +1 -0
- package/dist/src/algorithms/scope.d.ts +3 -1
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +1 -0
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +72 -0
- package/dist/src/algorithms/sssp.d.ts.map +1 -0
- package/dist/src/algorithms/sssp.js +586 -0
- package/dist/src/algorithms/sssp.js.map +1 -0
- package/dist/src/constants.d.ts +10 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +10 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/index.d.ts +8 -2
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +7 -1
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts +4 -4
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +39 -5
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernel/uniform-ring.d.ts +8 -0
- package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
- package/dist/src/kernel/uniform-ring.js +13 -0
- package/dist/src/kernel/uniform-ring.js.map +1 -1
- package/dist/src/kernels.d.ts +45 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +368 -3
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +63 -0
- package/dist/src/primitives/advance.d.ts.map +1 -0
- package/dist/src/primitives/advance.js +95 -0
- package/dist/src/primitives/advance.js.map +1 -0
- package/dist/src/primitives/compact.d.ts +89 -0
- package/dist/src/primitives/compact.d.ts.map +1 -0
- package/dist/src/primitives/compact.js +233 -0
- package/dist/src/primitives/compact.js.map +1 -0
- package/dist/src/primitives/core-shape.d.ts +22 -1
- package/dist/src/primitives/core-shape.d.ts.map +1 -1
- package/dist/src/primitives/core-shape.js +33 -3
- package/dist/src/primitives/core-shape.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +151 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -0
- package/dist/src/primitives/frontier.js +250 -0
- package/dist/src/primitives/frontier.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +16 -7
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/traversal.d.ts +53 -0
- package/dist/src/types/traversal.d.ts.map +1 -0
- package/dist/src/types/traversal.js +10 -0
- package/dist/src/types/traversal.js.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
- package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +53 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js +164 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
- package/dist/webgpu-graph-algorithms.js +3155 -384
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +5 -4
- package/src/accelerator.ts +65 -7
- package/src/algorithms/bellman-ford.ts +387 -0
- package/src/algorithms/bfs.ts +626 -0
- package/src/algorithms/closeness.ts +395 -0
- package/src/algorithms/scope.ts +4 -1
- package/src/algorithms/sssp.ts +768 -0
- package/src/constants.ts +10 -0
- package/src/index.ts +14 -1
- package/src/kernel/prelude.ts +39 -4
- package/src/kernel/uniform-ring.ts +14 -0
- package/src/kernels.ts +447 -6
- package/src/primitives/advance.ts +131 -0
- package/src/primitives/compact.ts +323 -0
- package/src/primitives/core-shape.ts +41 -3
- package/src/primitives/frontier.ts +372 -0
- package/src/types/accelerator.ts +18 -5
- package/src/types/traversal.ts +56 -0
- package/src/wgsl/advance-expand.wgsl.ts +68 -0
- package/src/wgsl/bf-relax.wgsl.ts +57 -0
- package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
- package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
- package/src/wgsl/bfs-contract.wgsl.ts +54 -0
- package/src/wgsl/bfs-fused.wgsl.ts +77 -0
- package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
- package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
- package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
- package/src/wgsl/compact-scatter.wgsl.ts +16 -0
- package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
- package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
- package/src/wgsl/frontier-finalize.wgsl.ts +163 -0
- package/src/wgsl/sssp-pred.wgsl.ts +79 -0
- package/src/wgsl/sssp-relax.wgsl.ts +71 -0
- package/dist/chunks/context-BXqgCifx.js.map +0 -1
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `frontier-finalize` kernel body (design 5.4, 6 row 7; P8-T4, the P8 plan's PD-3 / PD-23 / DEP-P8-C): the
|
|
3
|
+
* device-side selector of the frontier family. One workgroup, one lane, no barrier after the early return (spec 3.5
|
|
4
|
+
* rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's counts become known at
|
|
5
|
+
* two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`, advances
|
|
6
|
+
* `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`) and chooses the path; role 1 runs once the
|
|
7
|
+
* edge queue is filled -- clamps `edgeCount` to `P.edgeCapacity`, or, when `edgeCountUnclamped` exceeds the
|
|
8
|
+
* capacity, switches the path to the fused retry over `frontierCount` (PD-23). The decision is ONE word of the
|
|
9
|
+
* counters block, `path` (word 24): 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3 bottom-up,
|
|
10
|
+
* 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2). Every level kernel is a direct grid-stride
|
|
11
|
+
* dispatch that reads that word first and runs only when it names it (decision record
|
|
12
|
+
* design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md: Dawn's validation of an indirect dispatch cost
|
|
13
|
+
* about 0.4 ms of device time each, and the seven indirect slots this kernel once wrote per level were 97 % of a
|
|
14
|
+
* traversal's wall time; the slots and their args buffer are gone). Two rules: a boundary that finds `done` set
|
|
15
|
+
* moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when role 0
|
|
16
|
+
* chose one, which it reads from the path word (a storage write of one dispatch is visible to the next of the same
|
|
17
|
+
* pass).
|
|
18
|
+
*
|
|
19
|
+
* Roles 2 and 3 are the SSSP round boundary of the near-far loop (P8-T9, PD-20), over the same block read in its
|
|
20
|
+
* SSSP sense (word 1 the raw near half's appends, 21 the raw far half's, 0 and 20 the deduped pile counts, 22 the
|
|
21
|
+
* threshold, 23 the delta, 4 the previous threshold, 14 the round's mode): role 2 finds a non-empty raw near half
|
|
22
|
+
* and sizes the near dedupe (count word 1, output word 0; the dedupe's count in word 8) in mode 0; finds it empty
|
|
23
|
+
* and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
|
|
24
|
+
* unchanged sets `done 3`, the host's E_UNSUPPORTED), remembers the previous threshold in word 4 and sizes the far
|
|
25
|
+
* dedupe (count word 21, output word 20; the dedupe's count in word 9) in mode 1; finds both empty and sets
|
|
26
|
+
* `done 1`; and finds a raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in
|
|
27
|
+
* `level` when it picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds
|
|
28
|
+
* dispatched; a boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed and restarts the
|
|
29
|
+
* raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
|
|
30
|
+
* to 0); the relax kernels size themselves from words 0 and 20.
|
|
31
|
+
*
|
|
32
|
+
* Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a switch can be counted at
|
|
33
|
+
* the done boundary too, which the host model of the tests mirrors): top-down switches to bottom-up when
|
|
34
|
+
* `frontierDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's `max(1, floor(arcCount / n))`
|
|
35
|
+
* unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up switches back when
|
|
36
|
+
* `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no admitted device
|
|
37
|
+
* reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the direction at 0. Every
|
|
38
|
+
* change is counted in `switches`, the previous direction is word 14. The two unvisited words the test reads are
|
|
39
|
+
* rebuilt exactly once per submit by `bfs-unvisited-flags` (PD-18) and maintained here by subtraction: the count is
|
|
40
|
+
* subtracted from the SECOND boundary of a submit on and the degree sum from the THIRD on, because a boundary may
|
|
41
|
+
* only subtract what the submit's rebuild counted, and the frontier whose degree sum the second boundary holds was
|
|
42
|
+
* claimed before the rebuild ran (the rebuild counts the vertices unclaimed when it runs; the frontier rotated in at
|
|
43
|
+
* boundary 0 was claimed by the previous submit's last contract, so it was never in the sum; boundary b subtracts
|
|
44
|
+
* `next = |F_b|`, inside the sum iff b >= 1, and `degSum = deg(F_{b-1})`, inside it iff b >= 2). The degree sum is
|
|
45
|
+
* the "unvisited degree estimate" of the design rather than an exact count for two reasons: it is one level stale
|
|
46
|
+
* (a frontier's degree sum is only known once it has been expanded), and a bottom-up level expands nothing, so the
|
|
47
|
+
* word stops falling while bottom-up runs and overstates the set afterwards. The bias is one-directional -- an
|
|
48
|
+
* overstated m_u makes the switch INTO bottom-up harder, never easier -- and the next submit's rebuild makes it
|
|
49
|
+
* exact again. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
|
|
50
|
+
* textual edits of it.
|
|
51
|
+
*/
|
|
52
|
+
export const frontierFinalizeWgsl = /* wgsl */ `
|
|
53
|
+
@compute @workgroup_size(WG)
|
|
54
|
+
fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
55
|
+
if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
|
|
56
|
+
if (P.role == 0u) { // the level boundary
|
|
57
|
+
if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end
|
|
58
|
+
atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)
|
|
59
|
+
return;
|
|
60
|
+
}
|
|
61
|
+
let finished = atomicLoad(&counters[0]);
|
|
62
|
+
let next = atomicLoad(&counters[1]);
|
|
63
|
+
let degSum = atomicLoad(&counters[2]);
|
|
64
|
+
atomicStore(&counters[3], finished); // prevFrontierCount
|
|
65
|
+
atomicStore(&counters[4], degSum); // prevDegreeSum
|
|
66
|
+
atomicStore(&counters[0], next); // the rotation
|
|
67
|
+
atomicStore(&counters[1], 0u);
|
|
68
|
+
atomicStore(&counters[2], 0u);
|
|
69
|
+
atomicStore(&counters[8], 0u); // edgeCount
|
|
70
|
+
atomicStore(&counters[9], 0u); // edgeCountUnclamped
|
|
71
|
+
atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount
|
|
72
|
+
if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped to 2 (P8-T8, PD-18)
|
|
73
|
+
atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount (exact): F_b was inside the submit's rebuilt sum iff b >= 1
|
|
74
|
+
}
|
|
75
|
+
if (P.firstOfSubmit >= 2u) {
|
|
76
|
+
atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2
|
|
77
|
+
}
|
|
78
|
+
let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0
|
|
79
|
+
atomicStore(&counters[11], level);
|
|
80
|
+
let done = (next == 0u) || (level >= P.maxDepth);
|
|
81
|
+
atomicStore(&counters[15], select(0u, 1u, done));
|
|
82
|
+
var direction = atomicLoad(&counters[14]);
|
|
83
|
+
if (P.mode == 1u) {
|
|
84
|
+
direction = 0u; // top-down only (the test seam)
|
|
85
|
+
} else if (direction == 0u) {
|
|
86
|
+
if (degSum > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing
|
|
87
|
+
} else {
|
|
88
|
+
if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking
|
|
89
|
+
}
|
|
90
|
+
if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches
|
|
91
|
+
var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)
|
|
92
|
+
if (done) {
|
|
93
|
+
path = 0u;
|
|
94
|
+
} else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep
|
|
95
|
+
path = 3u;
|
|
96
|
+
atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);
|
|
97
|
+
} else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)
|
|
98
|
+
path = 2u; // bfs-fused: one WORKGROUP per frontier entry
|
|
99
|
+
atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
|
|
100
|
+
} else {
|
|
101
|
+
path = 1u; // advance-expand, then role 1 and bfs-contract
|
|
102
|
+
}
|
|
103
|
+
atomicStore(&counters[14], direction);
|
|
104
|
+
atomicStore(&counters[24], path);
|
|
105
|
+
} else if (P.role == 1u) { // the edge queue is filled
|
|
106
|
+
if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count
|
|
107
|
+
return;
|
|
108
|
+
}
|
|
109
|
+
let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);
|
|
110
|
+
atomicStore(&counters[8], clamped);
|
|
111
|
+
if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry
|
|
112
|
+
atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing
|
|
113
|
+
atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);
|
|
114
|
+
atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
|
|
115
|
+
} else {
|
|
116
|
+
atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)
|
|
117
|
+
}
|
|
118
|
+
} else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes
|
|
119
|
+
atomicStore(&counters[8], 0u); // the dedupe counts (words 8 and 9, the SSSP sense) and the path word: nothing unless a pile is chosen below
|
|
120
|
+
atomicStore(&counters[9], 0u);
|
|
121
|
+
atomicStore(&counters[24], 0u);
|
|
122
|
+
if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)
|
|
123
|
+
return;
|
|
124
|
+
}
|
|
125
|
+
let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped
|
|
126
|
+
let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped
|
|
127
|
+
if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE
|
|
128
|
+
atomicStore(&counters[15], 2u);
|
|
129
|
+
return;
|
|
130
|
+
}
|
|
131
|
+
if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn
|
|
132
|
+
atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter
|
|
133
|
+
atomicStore(&counters[14], 0u); // mode 0
|
|
134
|
+
atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)
|
|
135
|
+
atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing
|
|
136
|
+
atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)
|
|
137
|
+
} else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile
|
|
138
|
+
let threshold = bitcast<f32>(atomicLoad(&counters[22]));
|
|
139
|
+
let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)
|
|
140
|
+
if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED
|
|
141
|
+
atomicStore(&counters[15], 3u);
|
|
142
|
+
return;
|
|
143
|
+
}
|
|
144
|
+
atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below
|
|
145
|
+
atomicStore(&counters[22], bitcast<u32>(raised));
|
|
146
|
+
atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter
|
|
147
|
+
atomicStore(&counters[14], 1u); // mode 1
|
|
148
|
+
atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)
|
|
149
|
+
atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing
|
|
150
|
+
atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);
|
|
151
|
+
} else { // both piles empty: finished
|
|
152
|
+
atomicStore(&counters[15], 1u);
|
|
153
|
+
}
|
|
154
|
+
} else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed
|
|
155
|
+
if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round
|
|
156
|
+
if (atomicLoad(&counters[14]) == 0u) {
|
|
157
|
+
atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)
|
|
158
|
+
} else {
|
|
159
|
+
atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
`;
|
|
164
|
+
//# sourceMappingURL=frontier-finalize.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"frontier-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkDG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA+G9C,CAAC"}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `sssp-pred` kernel body (design 8.10 "SSSP predecessor pass"; P8-T6 and P8-T9, the P8 plan's PD-11 / PD-24 /
|
|
3
|
+
* PD-27): the ONE post-pass over the settled distances that produces `parent` for `breadthFirstSearch` and
|
|
4
|
+
* `predArc` for `sssp` and `bellmanFord`, so no claim kernel writes a predecessor and every predecessor array is a
|
|
5
|
+
* function of the settled `depth` / `dist` alone (bitwise reproducible, PD-14). `MODE` (a pipeline override) picks
|
|
6
|
+
* the distance type: 1 = u32 depths (`dist` is the BFS depth array, a tight arc is `depth[u] + 1 == depth[v]`, every
|
|
7
|
+
* tight arc is admitted, and the chain is acyclic because depth strictly decreases along it); 0 = f32 bit patterns
|
|
8
|
+
* (a tight arc is `bitcast<u32>(bitcast<f32>(dist[u]) + w) == dist[v]`, one f32 add compared as the bits PD-9
|
|
9
|
+
* stores). `P.predKind` picks what is written: 0 the arc index, 1 the source node index. The winner is the SMALLEST
|
|
10
|
+
* admitted value through `atomicMin` on `pred[v]`, which the driver fills with `INVALID_INDEX` first; the source
|
|
11
|
+
* keeps `INVALID_INDEX` whatever attains it.
|
|
12
|
+
*
|
|
13
|
+
* In `MODE 0` the body runs PD-27's three roles over a `pred` buffer of `2 x hb + 64` words, `hb = roundUp(n, 64)`:
|
|
14
|
+
* `pred[0, n)` the arcs, `pred[hb, hb + n)` the hop counts, `pred[2 hb]` the changed word, `pred[2 hb + 1]` the orphan
|
|
15
|
+
* word. Role 0 (the roots pass, `P.mode 0` only) marks every node with a tight in-arc from a strictly smaller
|
|
16
|
+
* distance as a root (`hops 0`); role 1 (a hop pass, `P.iteration` inside its batch) lowers `hops[v]` to
|
|
17
|
+
* `hops[u] + 1` along every plateau arc (`P.mode 0`: equal distance) or every tight arc (`P.mode 1`: bellmanFord's
|
|
18
|
+
* tight-subgraph rule) and records the pass index in the changed word when it lowered one, returning at its first
|
|
19
|
+
* line when the previous pass changed nothing; role 2 (the predecessor pass) admits the smallest tight arc whose
|
|
20
|
+
* source sits exactly one key step below `v` and counts a reached non-source node the key never reached as an
|
|
21
|
+
* orphan (only bellmanFord can make one). `MODE 1` reads none of `P.role`, `P.mode` and `P.iteration` and never
|
|
22
|
+
* touches `pred` beyond word `n - 1`. The body STRIDES over the rows (`planGridStride(n)`, `P.stride` the plan's
|
|
23
|
+
* stride): a per-invocation body under that plan would leave every row above the dispatch cap at `INVALID_INDEX`,
|
|
24
|
+
* silently. No barrier anywhere, so the early return is legal. Body only (spec 3.5, D9); the text is normative: the
|
|
25
|
+
* sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
26
|
+
*/
|
|
27
|
+
export declare const ssspPredWgsl = "\n@compute @workgroup_size(WG)\nfn sssp_pred(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let first = linear_id(wid, lid.x);\n let hb = ((P.n + 63u) / 64u) * 64u; // MODE 0 only: hops[v] is pred[hb + v]; pred[2 hb] is the changed word, pred[2 hb + 1] the orphan word (PD-27)\n if (P.role == 1u && P.iteration != 0u && atomicLoad(&pred[2u * hb]) < P.iteration) { return; } // the previous hop pass changed nothing: converged (no barrier anywhere, so the early return is legal)\n for (var u = first; u < P.n; u = u + P.stride) { // grid-stride over the rows (planGridStride(n)); no barrier anywhere\n let du = dist[u];\n var unreached = false;\n if (MODE == 1u) { unreached = du == INVALID_INDEX; } else { unreached = du == F32_INF_BITS; }\n if (unreached) { continue; }\n var hu = 0u; // u's hop count (MODE 0, roles 1 and 2)\n if (MODE == 0u && P.role != 0u) {\n hu = atomicLoad(&pred[hb + u]);\n if (hu == INVALID_INDEX) { // the key has not reached u yet\n if (P.role == 2u && u != P.source) { atomicAdd(&pred[2u * hb + 1u], 1u); } // an orphan: only bellmanFord can make one (P8-T10 Step 3)\n continue;\n }\n }\n let end = min(rowPtr[u + 1u], P.arcEnd);\n for (var a = max(rowPtr[u], P.arcBase); a < end; a = a + 1u) {\n let v = colIdx[a - P.arcBase];\n if (v == P.source) { continue; } // the source keeps INVALID_INDEX whatever attains it (a zero-weight arc could)\n let dv = dist[v];\n var tight = false; // the arc explains dist[v]\n var below = false; // and its source sits at a strictly smaller distance\n if (MODE == 1u) {\n tight = (du + 1u) == dv; // depth mode: BFS parent, one depth down\n } else {\n let w = select(1.0, weights[a - P.arcBase], HAS_WEIGHTS);\n tight = (dv != F32_INF_BITS) && (bitcast<u32>(bitcast<f32>(du) + w) == dv); // one f32 add, compared as the bit pattern PD-9 stores; never into an unreached v (an overflowed sum is +Inf too)\n below = bitcast<f32>(du) < bitcast<f32>(dv);\n }\n if (!tight) { continue; }\n var admit = false; // this arc is one key step below v\n if (MODE == 1u) {\n admit = true;\n } else if (P.role == 0u) { // the roots pass (the plateau rule only; bellmanFord seeds the source alone)\n if (P.mode == 0u && below) { atomicMin(&pred[hb + v], 0u); }\n } else if (P.role == 1u) { // a hop pass: one hop along a plateau arc (mode 0) or along any tight arc (mode 1)\n if (P.mode == 1u || du == dv) {\n let old = atomicMin(&pred[hb + v], hu + 1u);\n if (hu + 1u < old) { atomicMax(&pred[2u * hb], P.iteration + 1u); } // this pass changed something\n }\n } else { // the predecessor pass: the smallest tight arc one key step below v\n let hv = atomicLoad(&pred[hb + v]);\n if (P.mode == 1u) { admit = hu + 1u == hv; } else if (hv == 0u) { admit = below; } else { admit = (du == dv) && (hu + 1u == hv); }\n }\n if (admit) { atomicMin(&pred[v], select(a, u, P.predKind == 1u)); }\n }\n }\n}\n";
|
|
28
|
+
//# sourceMappingURL=sssp-pred.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"sssp-pred.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/sssp-pred.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AACH,eAAO,MAAM,YAAY,muHAoDxB,CAAC"}
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `sssp-pred` kernel body (design 8.10 "SSSP predecessor pass"; P8-T6 and P8-T9, the P8 plan's PD-11 / PD-24 /
|
|
3
|
+
* PD-27): the ONE post-pass over the settled distances that produces `parent` for `breadthFirstSearch` and
|
|
4
|
+
* `predArc` for `sssp` and `bellmanFord`, so no claim kernel writes a predecessor and every predecessor array is a
|
|
5
|
+
* function of the settled `depth` / `dist` alone (bitwise reproducible, PD-14). `MODE` (a pipeline override) picks
|
|
6
|
+
* the distance type: 1 = u32 depths (`dist` is the BFS depth array, a tight arc is `depth[u] + 1 == depth[v]`, every
|
|
7
|
+
* tight arc is admitted, and the chain is acyclic because depth strictly decreases along it); 0 = f32 bit patterns
|
|
8
|
+
* (a tight arc is `bitcast<u32>(bitcast<f32>(dist[u]) + w) == dist[v]`, one f32 add compared as the bits PD-9
|
|
9
|
+
* stores). `P.predKind` picks what is written: 0 the arc index, 1 the source node index. The winner is the SMALLEST
|
|
10
|
+
* admitted value through `atomicMin` on `pred[v]`, which the driver fills with `INVALID_INDEX` first; the source
|
|
11
|
+
* keeps `INVALID_INDEX` whatever attains it.
|
|
12
|
+
*
|
|
13
|
+
* In `MODE 0` the body runs PD-27's three roles over a `pred` buffer of `2 x hb + 64` words, `hb = roundUp(n, 64)`:
|
|
14
|
+
* `pred[0, n)` the arcs, `pred[hb, hb + n)` the hop counts, `pred[2 hb]` the changed word, `pred[2 hb + 1]` the orphan
|
|
15
|
+
* word. Role 0 (the roots pass, `P.mode 0` only) marks every node with a tight in-arc from a strictly smaller
|
|
16
|
+
* distance as a root (`hops 0`); role 1 (a hop pass, `P.iteration` inside its batch) lowers `hops[v]` to
|
|
17
|
+
* `hops[u] + 1` along every plateau arc (`P.mode 0`: equal distance) or every tight arc (`P.mode 1`: bellmanFord's
|
|
18
|
+
* tight-subgraph rule) and records the pass index in the changed word when it lowered one, returning at its first
|
|
19
|
+
* line when the previous pass changed nothing; role 2 (the predecessor pass) admits the smallest tight arc whose
|
|
20
|
+
* source sits exactly one key step below `v` and counts a reached non-source node the key never reached as an
|
|
21
|
+
* orphan (only bellmanFord can make one). `MODE 1` reads none of `P.role`, `P.mode` and `P.iteration` and never
|
|
22
|
+
* touches `pred` beyond word `n - 1`. The body STRIDES over the rows (`planGridStride(n)`, `P.stride` the plan's
|
|
23
|
+
* stride): a per-invocation body under that plan would leave every row above the dispatch cap at `INVALID_INDEX`,
|
|
24
|
+
* silently. No barrier anywhere, so the early return is legal. Body only (spec 3.5, D9); the text is normative: the
|
|
25
|
+
* sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
26
|
+
*/
|
|
27
|
+
export const ssspPredWgsl = /* wgsl */ `
|
|
28
|
+
@compute @workgroup_size(WG)
|
|
29
|
+
fn sssp_pred(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
30
|
+
let first = linear_id(wid, lid.x);
|
|
31
|
+
let hb = ((P.n + 63u) / 64u) * 64u; // MODE 0 only: hops[v] is pred[hb + v]; pred[2 hb] is the changed word, pred[2 hb + 1] the orphan word (PD-27)
|
|
32
|
+
if (P.role == 1u && P.iteration != 0u && atomicLoad(&pred[2u * hb]) < P.iteration) { return; } // the previous hop pass changed nothing: converged (no barrier anywhere, so the early return is legal)
|
|
33
|
+
for (var u = first; u < P.n; u = u + P.stride) { // grid-stride over the rows (planGridStride(n)); no barrier anywhere
|
|
34
|
+
let du = dist[u];
|
|
35
|
+
var unreached = false;
|
|
36
|
+
if (MODE == 1u) { unreached = du == INVALID_INDEX; } else { unreached = du == F32_INF_BITS; }
|
|
37
|
+
if (unreached) { continue; }
|
|
38
|
+
var hu = 0u; // u's hop count (MODE 0, roles 1 and 2)
|
|
39
|
+
if (MODE == 0u && P.role != 0u) {
|
|
40
|
+
hu = atomicLoad(&pred[hb + u]);
|
|
41
|
+
if (hu == INVALID_INDEX) { // the key has not reached u yet
|
|
42
|
+
if (P.role == 2u && u != P.source) { atomicAdd(&pred[2u * hb + 1u], 1u); } // an orphan: only bellmanFord can make one (P8-T10 Step 3)
|
|
43
|
+
continue;
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
let end = min(rowPtr[u + 1u], P.arcEnd);
|
|
47
|
+
for (var a = max(rowPtr[u], P.arcBase); a < end; a = a + 1u) {
|
|
48
|
+
let v = colIdx[a - P.arcBase];
|
|
49
|
+
if (v == P.source) { continue; } // the source keeps INVALID_INDEX whatever attains it (a zero-weight arc could)
|
|
50
|
+
let dv = dist[v];
|
|
51
|
+
var tight = false; // the arc explains dist[v]
|
|
52
|
+
var below = false; // and its source sits at a strictly smaller distance
|
|
53
|
+
if (MODE == 1u) {
|
|
54
|
+
tight = (du + 1u) == dv; // depth mode: BFS parent, one depth down
|
|
55
|
+
} else {
|
|
56
|
+
let w = select(1.0, weights[a - P.arcBase], HAS_WEIGHTS);
|
|
57
|
+
tight = (dv != F32_INF_BITS) && (bitcast<u32>(bitcast<f32>(du) + w) == dv); // one f32 add, compared as the bit pattern PD-9 stores; never into an unreached v (an overflowed sum is +Inf too)
|
|
58
|
+
below = bitcast<f32>(du) < bitcast<f32>(dv);
|
|
59
|
+
}
|
|
60
|
+
if (!tight) { continue; }
|
|
61
|
+
var admit = false; // this arc is one key step below v
|
|
62
|
+
if (MODE == 1u) {
|
|
63
|
+
admit = true;
|
|
64
|
+
} else if (P.role == 0u) { // the roots pass (the plateau rule only; bellmanFord seeds the source alone)
|
|
65
|
+
if (P.mode == 0u && below) { atomicMin(&pred[hb + v], 0u); }
|
|
66
|
+
} else if (P.role == 1u) { // a hop pass: one hop along a plateau arc (mode 0) or along any tight arc (mode 1)
|
|
67
|
+
if (P.mode == 1u || du == dv) {
|
|
68
|
+
let old = atomicMin(&pred[hb + v], hu + 1u);
|
|
69
|
+
if (hu + 1u < old) { atomicMax(&pred[2u * hb], P.iteration + 1u); } // this pass changed something
|
|
70
|
+
}
|
|
71
|
+
} else { // the predecessor pass: the smallest tight arc one key step below v
|
|
72
|
+
let hv = atomicLoad(&pred[hb + v]);
|
|
73
|
+
if (P.mode == 1u) { admit = hu + 1u == hv; } else if (hv == 0u) { admit = below; } else { admit = (du == dv) && (hu + 1u == hv); }
|
|
74
|
+
}
|
|
75
|
+
if (admit) { atomicMin(&pred[v], select(a, u, P.predKind == 1u)); }
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
`;
|
|
80
|
+
//# sourceMappingURL=sssp-pred.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"sssp-pred.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/sssp-pred.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AACH,MAAM,CAAC,MAAM,YAAY,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAoDtC,CAAC"}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `sssp-relax` kernel body (design 8.4 "Davidson's near-far", 8.10 "SSSP near-far relax"; P8-T9, the P8 plan's
|
|
3
|
+
* PD-9 / PD-20 / PD-22 / DEP-P8-E): one round of the near-far shortest-path loop. `dist` is `array<atomic<u32>>`
|
|
4
|
+
* holding the IEEE-754 bit patterns of the f32 distances (`F32_INF_BITS` = unreached), and `atomicMin` on the
|
|
5
|
+
* patterns IS `min` on the values, because for non-negative floats the unsigned bit order is the value order
|
|
6
|
+
* (`+0` is `0x00000000`, every sum of non-negatives under round-to-nearest is `+0` or positive, never `-0`), which
|
|
7
|
+
* the driver's host scan of the weight vector guarantees -- the whole trick, and it needs no float atomic (PD-9).
|
|
8
|
+
* Every candidate is ONE f32 add, `dist[u] + w`, so the settled value is the minimum of a fixed set of f32
|
|
9
|
+
* numbers: order-independent, bitwise reproducible, and bitwise equal to the f32 Dijkstra oracle.
|
|
10
|
+
*
|
|
11
|
+
* Two roles over one body, chosen by `P.role`. Role 0 (the near round) reads the deduped near pile `queueIn`
|
|
12
|
+
* (`counters[0]` entries, at most `P.n`), relaxes every arc of every entry's WHOLE row (never windowed, DEP-P8-E),
|
|
13
|
+
* skips a candidate above `P.cutoffBits` (the CPU port's `dv <= cutoff` guard as `nd > cutoff`), and when its
|
|
14
|
+
* `atomicMin` improved `v` -- this lane alone observed a larger old value, so this lane alone owns the append --
|
|
15
|
+
* appends `v` to the raw near half of `queueOut` (word 0, count word 1) when `nd` is below the threshold
|
|
16
|
+
* (`counters[22]`), else to the raw far half (word `P.edgeCapacity`, count word 21). Role 1 (the pass-through,
|
|
17
|
+
* PD-20) re-buckets the deduped far pile (`counters[20]` entries): an entry whose settled distance fell below the
|
|
18
|
+
* PREVIOUS threshold (`counters[4]`) was appended to near at that improvement and relaxed there, so it is dropped;
|
|
19
|
+
* the rest go back to near or far against the threshold the boundary just raised. The count words are unclamped
|
|
20
|
+
* (the write is guarded by the capacity; `frontier-finalize` role 2 detects a pile above it). The appends are per
|
|
21
|
+
* improving relaxation, one atomic each: inside a per-lane arc loop no workgroup aggregation is possible without the
|
|
22
|
+
* uniform-strip structure of `bfs-fused`, and Davidson's kernel appends per thread too. Grid-strided under a direct
|
|
23
|
+
* dispatch of `planGridStride(n)`: `P.stride` is the plan's, a deduped pile of at most `n` entries is covered in a
|
|
24
|
+
* few trips per lane and `i + stride` never wraps (`U32_MAX` would); the block's `path` word (5 a near round, 6 a
|
|
25
|
+
* far one) makes the other role's dispatch a no-op. No barrier anywhere: the loops may
|
|
26
|
+
* be per lane. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
|
|
27
|
+
* textual edits of it.
|
|
28
|
+
*/
|
|
29
|
+
export declare const ssspRelaxWgsl = "\n@compute @workgroup_size(WG)\nfn sssp_relax(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let first = linear_id(wid, lid.x);\n let mine = select(5u, 6u, P.role == 1u); // the path word role 2 wrote: 5 a near round, 6 a far pass-through\n let chosen = atomicLoad(&counters[24]) == mine; // the other role's dispatch of the round is a no-op\n let count = select(0u, min(atomicLoad(&counters[select(0u, 20u, P.role == 1u)]), P.n), chosen); // the deduped near or far pile\n let threshold = bitcast<f32>(atomicLoad(&counters[22]));\n let cutoff = bitcast<f32>(P.cutoffBits);\n for (var i = first; i < count; i = i + P.stride) { // no barrier anywhere: the loops may be per lane\n let u = queueIn[i];\n let du = bitcast<f32>(atomicLoad(&dist[u]));\n if (P.role == 1u) { // the pass-through (PD-20): re-bucket a far entry\n if (du < bitcast<f32>(atomicLoad(&counters[4]))) { continue; } // below the previous threshold: relaxed in an earlier bucket\n if (du < threshold) {\n let q = atomicAdd(&counters[1], 1u);\n if (q < P.edgeCapacity) { queueOut[q] = u; }\n } else {\n let q = atomicAdd(&counters[21], 1u);\n if (q < P.edgeCapacity) { queueOut[P.edgeCapacity + q] = u; }\n }\n continue;\n }\n let end = rowPtr[u + 1u];\n for (var a = rowPtr[u]; a < end; a = a + 1u) { // the whole row: never windowed (DEP-P8-E)\n let v = colIdx[a];\n let nd = du + select(1.0, weights[a], HAS_WEIGHTS); // ONE f32 add (PD-9)\n if (nd > cutoff) { continue; } // SsspOptions.cutoff: the CPU port's dv <= cutoff\n let bits = bitcast<u32>(nd);\n let old = atomicMin(&dist[v], bits); // exact on non-negative floats\n if (bits < old) { // this lane improved v, so it owns the append\n if (nd < threshold) {\n let q = atomicAdd(&counters[1], 1u);\n if (q < P.edgeCapacity) { queueOut[q] = v; }\n } else {\n let q = atomicAdd(&counters[21], 1u);\n if (q < P.edgeCapacity) { queueOut[P.edgeCapacity + q] = v; }\n }\n }\n }\n }\n}\n";
|
|
30
|
+
//# sourceMappingURL=sssp-relax.wgsl.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"sssp-relax.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/sssp-relax.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;GA2BG;AACH,eAAO,MAAM,aAAa,kgFA0CzB,CAAC"}
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `sssp-relax` kernel body (design 8.4 "Davidson's near-far", 8.10 "SSSP near-far relax"; P8-T9, the P8 plan's
|
|
3
|
+
* PD-9 / PD-20 / PD-22 / DEP-P8-E): one round of the near-far shortest-path loop. `dist` is `array<atomic<u32>>`
|
|
4
|
+
* holding the IEEE-754 bit patterns of the f32 distances (`F32_INF_BITS` = unreached), and `atomicMin` on the
|
|
5
|
+
* patterns IS `min` on the values, because for non-negative floats the unsigned bit order is the value order
|
|
6
|
+
* (`+0` is `0x00000000`, every sum of non-negatives under round-to-nearest is `+0` or positive, never `-0`), which
|
|
7
|
+
* the driver's host scan of the weight vector guarantees -- the whole trick, and it needs no float atomic (PD-9).
|
|
8
|
+
* Every candidate is ONE f32 add, `dist[u] + w`, so the settled value is the minimum of a fixed set of f32
|
|
9
|
+
* numbers: order-independent, bitwise reproducible, and bitwise equal to the f32 Dijkstra oracle.
|
|
10
|
+
*
|
|
11
|
+
* Two roles over one body, chosen by `P.role`. Role 0 (the near round) reads the deduped near pile `queueIn`
|
|
12
|
+
* (`counters[0]` entries, at most `P.n`), relaxes every arc of every entry's WHOLE row (never windowed, DEP-P8-E),
|
|
13
|
+
* skips a candidate above `P.cutoffBits` (the CPU port's `dv <= cutoff` guard as `nd > cutoff`), and when its
|
|
14
|
+
* `atomicMin` improved `v` -- this lane alone observed a larger old value, so this lane alone owns the append --
|
|
15
|
+
* appends `v` to the raw near half of `queueOut` (word 0, count word 1) when `nd` is below the threshold
|
|
16
|
+
* (`counters[22]`), else to the raw far half (word `P.edgeCapacity`, count word 21). Role 1 (the pass-through,
|
|
17
|
+
* PD-20) re-buckets the deduped far pile (`counters[20]` entries): an entry whose settled distance fell below the
|
|
18
|
+
* PREVIOUS threshold (`counters[4]`) was appended to near at that improvement and relaxed there, so it is dropped;
|
|
19
|
+
* the rest go back to near or far against the threshold the boundary just raised. The count words are unclamped
|
|
20
|
+
* (the write is guarded by the capacity; `frontier-finalize` role 2 detects a pile above it). The appends are per
|
|
21
|
+
* improving relaxation, one atomic each: inside a per-lane arc loop no workgroup aggregation is possible without the
|
|
22
|
+
* uniform-strip structure of `bfs-fused`, and Davidson's kernel appends per thread too. Grid-strided under a direct
|
|
23
|
+
* dispatch of `planGridStride(n)`: `P.stride` is the plan's, a deduped pile of at most `n` entries is covered in a
|
|
24
|
+
* few trips per lane and `i + stride` never wraps (`U32_MAX` would); the block's `path` word (5 a near round, 6 a
|
|
25
|
+
* far one) makes the other role's dispatch a no-op. No barrier anywhere: the loops may
|
|
26
|
+
* be per lane. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
|
|
27
|
+
* textual edits of it.
|
|
28
|
+
*/
|
|
29
|
+
export const ssspRelaxWgsl = /* wgsl */ `
|
|
30
|
+
@compute @workgroup_size(WG)
|
|
31
|
+
fn sssp_relax(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
32
|
+
let first = linear_id(wid, lid.x);
|
|
33
|
+
let mine = select(5u, 6u, P.role == 1u); // the path word role 2 wrote: 5 a near round, 6 a far pass-through
|
|
34
|
+
let chosen = atomicLoad(&counters[24]) == mine; // the other role's dispatch of the round is a no-op
|
|
35
|
+
let count = select(0u, min(atomicLoad(&counters[select(0u, 20u, P.role == 1u)]), P.n), chosen); // the deduped near or far pile
|
|
36
|
+
let threshold = bitcast<f32>(atomicLoad(&counters[22]));
|
|
37
|
+
let cutoff = bitcast<f32>(P.cutoffBits);
|
|
38
|
+
for (var i = first; i < count; i = i + P.stride) { // no barrier anywhere: the loops may be per lane
|
|
39
|
+
let u = queueIn[i];
|
|
40
|
+
let du = bitcast<f32>(atomicLoad(&dist[u]));
|
|
41
|
+
if (P.role == 1u) { // the pass-through (PD-20): re-bucket a far entry
|
|
42
|
+
if (du < bitcast<f32>(atomicLoad(&counters[4]))) { continue; } // below the previous threshold: relaxed in an earlier bucket
|
|
43
|
+
if (du < threshold) {
|
|
44
|
+
let q = atomicAdd(&counters[1], 1u);
|
|
45
|
+
if (q < P.edgeCapacity) { queueOut[q] = u; }
|
|
46
|
+
} else {
|
|
47
|
+
let q = atomicAdd(&counters[21], 1u);
|
|
48
|
+
if (q < P.edgeCapacity) { queueOut[P.edgeCapacity + q] = u; }
|
|
49
|
+
}
|
|
50
|
+
continue;
|
|
51
|
+
}
|
|
52
|
+
let end = rowPtr[u + 1u];
|
|
53
|
+
for (var a = rowPtr[u]; a < end; a = a + 1u) { // the whole row: never windowed (DEP-P8-E)
|
|
54
|
+
let v = colIdx[a];
|
|
55
|
+
let nd = du + select(1.0, weights[a], HAS_WEIGHTS); // ONE f32 add (PD-9)
|
|
56
|
+
if (nd > cutoff) { continue; } // SsspOptions.cutoff: the CPU port's dv <= cutoff
|
|
57
|
+
let bits = bitcast<u32>(nd);
|
|
58
|
+
let old = atomicMin(&dist[v], bits); // exact on non-negative floats
|
|
59
|
+
if (bits < old) { // this lane improved v, so it owns the append
|
|
60
|
+
if (nd < threshold) {
|
|
61
|
+
let q = atomicAdd(&counters[1], 1u);
|
|
62
|
+
if (q < P.edgeCapacity) { queueOut[q] = v; }
|
|
63
|
+
} else {
|
|
64
|
+
let q = atomicAdd(&counters[21], 1u);
|
|
65
|
+
if (q < P.edgeCapacity) { queueOut[P.edgeCapacity + q] = v; }
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
`;
|
|
72
|
+
//# sourceMappingURL=sssp-relax.wgsl.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"sssp-relax.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/sssp-relax.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;GA2BG;AACH,MAAM,CAAC,MAAM,aAAa,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA0CvC,CAAC"}
|