@graphty/webgpu-graph-algorithms 0.6.4 → 0.6.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-Dvq-Cc6v.js → context-hzGggHeM.js} +28 -30
- package/dist/chunks/context-hzGggHeM.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/algorithms/bfs.d.ts +1 -1
- package/dist/src/algorithms/bfs.js +1 -1
- package/dist/src/algorithms/scope.d.ts +3 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +0 -2
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +4 -3
- package/dist/src/algorithms/sssp.d.ts.map +1 -1
- package/dist/src/algorithms/sssp.js +4 -3
- package/dist/src/algorithms/sssp.js.map +1 -1
- package/dist/src/constants.d.ts +0 -2
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +0 -2
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernels.d.ts +8 -7
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +12 -15
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +3 -2
- package/dist/src/primitives/advance.d.ts.map +1 -1
- package/dist/src/primitives/advance.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +33 -38
- package/dist/src/primitives/frontier.d.ts.map +1 -1
- package/dist/src/primitives/frontier.js +23 -32
- package/dist/src/primitives/frontier.js.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +24 -30
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.js +36 -82
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
- package/dist/webgpu-graph-algorithms.js +27 -86
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +1 -1
- package/src/algorithms/bfs.ts +1 -1
- package/src/algorithms/scope.ts +3 -10
- package/src/algorithms/sssp.ts +4 -3
- package/src/constants.ts +0 -2
- package/src/kernels.ts +12 -15
- package/src/primitives/advance.ts +5 -4
- package/src/primitives/frontier.ts +40 -56
- package/src/wgsl/frontier-finalize.wgsl.ts +36 -82
- package/dist/chunks/context-Dvq-Cc6v.js.map +0 -1
|
@@ -1,31 +1,33 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The `frontier-finalize` kernel body (design 5.4, 6 row 7; P8-T4, the P8 plan's PD-3 / PD-23 / DEP-P8-C): the
|
|
3
3
|
* device-side selector of the frontier family. One workgroup, one lane, no barrier after the early return (spec 3.5
|
|
4
|
-
* rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
4
|
+
* rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's counts become known at
|
|
5
|
+
* two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`, advances
|
|
6
|
+
* `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`) and chooses the path; role 1 runs once the
|
|
7
|
+
* edge queue is filled -- clamps `edgeCount` to `P.edgeCapacity`, or, when `edgeCountUnclamped` exceeds the
|
|
8
|
+
* capacity, switches the path to the fused retry over `frontierCount` (PD-23). The decision is ONE word of the
|
|
9
|
+
* counters block, `path` (word 24): 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3 bottom-up,
|
|
10
|
+
* 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2). Every level kernel is a direct grid-stride
|
|
11
|
+
* dispatch that reads that word first and runs only when it names it (decision record
|
|
12
|
+
* design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md: Dawn's validation of an indirect dispatch cost
|
|
13
|
+
* about 0.4 ms of device time each, and the seven indirect slots this kernel once wrote per level were 97 % of a
|
|
14
|
+
* traversal's wall time; the slots and their args buffer are gone). Two rules: a boundary that finds `done` set
|
|
15
|
+
* moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when role 0
|
|
16
|
+
* chose one, which it reads from the path word (a storage write of one dispatch is visible to the next of the same
|
|
17
|
+
* pass).
|
|
16
18
|
*
|
|
17
19
|
* Roles 2 and 3 are the SSSP round boundary of the near-far loop (P8-T9, PD-20), over the same block read in its
|
|
18
20
|
* SSSP sense (word 1 the raw near half's appends, 21 the raw far half's, 0 and 20 the deduped pile counts, 22 the
|
|
19
|
-
* threshold, 23 the delta, 4 the previous threshold, 14 the round's mode)
|
|
20
|
-
*
|
|
21
|
-
*
|
|
21
|
+
* threshold, 23 the delta, 4 the previous threshold, 14 the round's mode): role 2 finds a non-empty raw near half
|
|
22
|
+
* and sizes the near dedupe (count word 1, output word 0; the dedupe's count in word 8) in mode 0; finds it empty
|
|
23
|
+
* and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
|
|
22
24
|
* unchanged sets `done 3`, the host's E_UNSUPPORTED), remembers the previous threshold in word 4 and sizes the far
|
|
23
|
-
* dedupe (
|
|
24
|
-
* raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in
|
|
25
|
-
* picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds
|
|
26
|
-
* boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed
|
|
27
|
-
*
|
|
28
|
-
*
|
|
25
|
+
* dedupe (count word 21, output word 20; the dedupe's count in word 9) in mode 1; finds both empty and sets
|
|
26
|
+
* `done 1`; and finds a raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in
|
|
27
|
+
* `level` when it picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds
|
|
28
|
+
* dispatched; a boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed and restarts the
|
|
29
|
+
* raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
|
|
30
|
+
* to 0); the relax kernels size themselves from words 0 and 20.
|
|
29
31
|
*
|
|
30
32
|
* Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a switch can be counted at
|
|
31
33
|
* the done boundary too, which the host model of the tests mirrors): top-down switches to bottom-up when
|
|
@@ -46,42 +48,15 @@
|
|
|
46
48
|
* overstated m_u makes the switch INTO bottom-up harder, never easier -- and the next submit's rebuild makes it
|
|
47
49
|
* exact again. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
|
|
48
50
|
* textual edits of it.
|
|
49
|
-
*
|
|
50
|
-
* Since 2026-09-25 nothing dispatches FROM the slots (G8-F5: Dawn's validation of an indirect dispatch cost about
|
|
51
|
-
* 0.4 ms of device time each, whether or not it dispatched anything, and the seven slots of thirty-two recorded
|
|
52
|
-
* levels were 97 % of a traversal's wall time). Every level kernel is a direct grid-stride dispatch that reads the
|
|
53
|
-
* `path` word (24) this kernel writes -- 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3
|
|
54
|
-
* bottom-up, 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2) -- and the SSSP dedupes read
|
|
55
|
-
* their counts from words 8 and 9, which role 2 writes. The slots stay as the selector's recorded decision, read
|
|
56
|
-
* back by the frontier tests; deleting them with those tests is the follow-up.
|
|
57
51
|
*/
|
|
58
52
|
export const frontierFinalizeWgsl = /* wgsl */ `
|
|
59
|
-
fn write_slot_groups(slot: u32, groups: u32, count: u32) { // groups workgroups, split in 2D above the per-dim limit
|
|
60
|
-
var x = groups;
|
|
61
|
-
var y = 1u;
|
|
62
|
-
if (groups > MAX_WORKGROUPS_PER_DIM) {
|
|
63
|
-
x = MAX_WORKGROUPS_PER_DIM;
|
|
64
|
-
y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;
|
|
65
|
-
}
|
|
66
|
-
let base = 4u * (P.slotBase + slot); // 16-byte slots: (x, y, 1, count)
|
|
67
|
-
args[base] = x; args[base + 1u] = y; args[base + 2u] = 1u; args[base + 3u] = count;
|
|
68
|
-
}
|
|
69
|
-
fn write_slot(slot: u32, count: u32) { // one INVOCATION per entry: ceil(count / wg) workgroups
|
|
70
|
-
let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap (indirect-finalize's rule)
|
|
71
|
-
write_slot_groups(slot, groups, count);
|
|
72
|
-
}
|
|
73
|
-
fn zero_slot(slot: u32) {
|
|
74
|
-
let base = 4u * (P.slotBase + slot);
|
|
75
|
-
args[base] = 0u; args[base + 1u] = 0u; args[base + 2u] = 1u; args[base + 3u] = 0u;
|
|
76
|
-
}
|
|
77
|
-
|
|
78
53
|
@compute @workgroup_size(WG)
|
|
79
54
|
fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
80
55
|
if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
|
|
81
56
|
if (P.role == 0u) { // the level boundary
|
|
82
57
|
if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end
|
|
83
|
-
|
|
84
|
-
return;
|
|
58
|
+
atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)
|
|
59
|
+
return;
|
|
85
60
|
}
|
|
86
61
|
let finished = atomicLoad(&counters[0]);
|
|
87
62
|
let next = atomicLoad(&counters[1]);
|
|
@@ -115,39 +90,29 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
115
90
|
if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches
|
|
116
91
|
var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)
|
|
117
92
|
if (done) {
|
|
118
|
-
|
|
93
|
+
path = 0u;
|
|
119
94
|
} else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep
|
|
120
|
-
zero_slot(0u); zero_slot(1u); zero_slot(2u); zero_slot(6u);
|
|
121
|
-
write_slot(3u, (P.n + 31u) / 32u); write_slot(4u, next); write_slot(5u, atomicLoad(&counters[7]));
|
|
122
95
|
path = 3u;
|
|
123
96
|
atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);
|
|
124
97
|
} else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)
|
|
125
|
-
|
|
126
|
-
write_slot_groups(2u, next, next); // bfs-fused is one WORKGROUP per frontier entry
|
|
127
|
-
path = 2u;
|
|
98
|
+
path = 2u; // bfs-fused: one WORKGROUP per frontier entry
|
|
128
99
|
atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
|
|
129
100
|
} else {
|
|
130
|
-
|
|
131
|
-
write_slot(0u, next); // slots 1 and 6 are role 1's
|
|
132
|
-
path = 1u;
|
|
101
|
+
path = 1u; // advance-expand, then role 1 and bfs-contract
|
|
133
102
|
}
|
|
134
103
|
atomicStore(&counters[14], direction);
|
|
135
104
|
atomicStore(&counters[24], path);
|
|
136
105
|
} else if (P.role == 1u) { // the edge queue is filled
|
|
137
|
-
if (
|
|
138
|
-
zero_slot(1u); zero_slot(6u);
|
|
106
|
+
if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count
|
|
139
107
|
return;
|
|
140
108
|
}
|
|
141
109
|
let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);
|
|
142
110
|
atomicStore(&counters[8], clamped);
|
|
143
111
|
if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry
|
|
144
|
-
|
|
145
|
-
zero_slot(1u); write_slot_groups(6u, entries, entries); // one workgroup per frontier entry, as slot 2
|
|
146
|
-
atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry, bfs-contract nothing
|
|
112
|
+
atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing
|
|
147
113
|
atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);
|
|
148
114
|
atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
|
|
149
115
|
} else {
|
|
150
|
-
write_slot(1u, clamped); zero_slot(6u);
|
|
151
116
|
atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)
|
|
152
117
|
}
|
|
153
118
|
} else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes
|
|
@@ -155,54 +120,43 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
155
120
|
atomicStore(&counters[9], 0u);
|
|
156
121
|
atomicStore(&counters[24], 0u);
|
|
157
122
|
if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)
|
|
158
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
159
123
|
return;
|
|
160
124
|
}
|
|
161
125
|
let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped
|
|
162
126
|
let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped
|
|
163
127
|
if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE
|
|
164
128
|
atomicStore(&counters[15], 2u);
|
|
165
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
166
129
|
return;
|
|
167
130
|
}
|
|
168
|
-
zero_slot(2u); zero_slot(5u); zero_slot(6u); // role 3 sizes the relax slots once the piles are deduped
|
|
169
131
|
if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn
|
|
170
132
|
atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter
|
|
171
133
|
atomicStore(&counters[14], 0u); // mode 0
|
|
172
|
-
|
|
173
|
-
atomicStore(&counters[8], nearRaw); // the near dedupe's count word
|
|
134
|
+
atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)
|
|
174
135
|
atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing
|
|
175
|
-
zero_slot(3u); zero_slot(4u);
|
|
176
136
|
atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)
|
|
177
137
|
} else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile
|
|
178
138
|
let threshold = bitcast<f32>(atomicLoad(&counters[22]));
|
|
179
139
|
let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)
|
|
180
140
|
if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED
|
|
181
141
|
atomicStore(&counters[15], 3u);
|
|
182
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
183
142
|
return;
|
|
184
143
|
}
|
|
185
144
|
atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below
|
|
186
145
|
atomicStore(&counters[22], bitcast<u32>(raised));
|
|
187
146
|
atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter
|
|
188
147
|
atomicStore(&counters[14], 1u); // mode 1
|
|
189
|
-
|
|
190
|
-
write_slot(3u, farRaw); write_slot(4u, farRaw); // dedupe-claim, dedupe-filter over the far half
|
|
191
|
-
atomicStore(&counters[9], farRaw); // the far dedupe's count word
|
|
148
|
+
atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)
|
|
192
149
|
atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing
|
|
193
150
|
atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);
|
|
194
151
|
} else { // both piles empty: finished
|
|
195
152
|
atomicStore(&counters[15], 1u);
|
|
196
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
197
153
|
}
|
|
198
|
-
} else if (P.role == 3u) { // the piles are deduped:
|
|
199
|
-
if (atomicLoad(&counters[15]) != 0u) { return; } // role 2
|
|
154
|
+
} else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed
|
|
155
|
+
if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round
|
|
200
156
|
if (atomicLoad(&counters[14]) == 0u) {
|
|
201
|
-
|
|
202
|
-
atomicStore(&counters[1], 0u); // the raw near half restarts
|
|
157
|
+
atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)
|
|
203
158
|
} else {
|
|
204
|
-
|
|
205
|
-
atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far)
|
|
159
|
+
atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)
|
|
206
160
|
}
|
|
207
161
|
}
|
|
208
162
|
}
|