@graphty/webgpu-graph-algorithms 0.6.4 → 0.6.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-Dvq-Cc6v.js → context-hzGggHeM.js} +28 -30
- package/dist/chunks/context-hzGggHeM.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/algorithms/bfs.d.ts +1 -1
- package/dist/src/algorithms/bfs.js +1 -1
- package/dist/src/algorithms/scope.d.ts +3 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +0 -2
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +4 -3
- package/dist/src/algorithms/sssp.d.ts.map +1 -1
- package/dist/src/algorithms/sssp.js +4 -3
- package/dist/src/algorithms/sssp.js.map +1 -1
- package/dist/src/constants.d.ts +0 -2
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +0 -2
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernels.d.ts +8 -7
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +12 -15
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +3 -2
- package/dist/src/primitives/advance.d.ts.map +1 -1
- package/dist/src/primitives/advance.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +33 -38
- package/dist/src/primitives/frontier.d.ts.map +1 -1
- package/dist/src/primitives/frontier.js +23 -32
- package/dist/src/primitives/frontier.js.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +24 -30
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.js +36 -82
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
- package/dist/webgpu-graph-algorithms.js +27 -86
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +1 -1
- package/src/algorithms/bfs.ts +1 -1
- package/src/algorithms/scope.ts +3 -10
- package/src/algorithms/sssp.ts +4 -3
- package/src/constants.ts +0 -2
- package/src/kernels.ts +12 -15
- package/src/primitives/advance.ts +5 -4
- package/src/primitives/frontier.ts +40 -56
- package/src/wgsl/frontier-finalize.wgsl.ts +36 -82
- package/dist/chunks/context-Dvq-Cc6v.js.map +0 -1
|
@@ -1,31 +1,33 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The `frontier-finalize` kernel body (design 5.4, 6 row 7; P8-T4, the P8 plan's PD-3 / PD-23 / DEP-P8-C): the
|
|
3
3
|
* device-side selector of the frontier family. One workgroup, one lane, no barrier after the early return (spec 3.5
|
|
4
|
-
* rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
4
|
+
* rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's counts become known at
|
|
5
|
+
* two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`, advances
|
|
6
|
+
* `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`) and chooses the path; role 1 runs once the
|
|
7
|
+
* edge queue is filled -- clamps `edgeCount` to `P.edgeCapacity`, or, when `edgeCountUnclamped` exceeds the
|
|
8
|
+
* capacity, switches the path to the fused retry over `frontierCount` (PD-23). The decision is ONE word of the
|
|
9
|
+
* counters block, `path` (word 24): 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3 bottom-up,
|
|
10
|
+
* 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2). Every level kernel is a direct grid-stride
|
|
11
|
+
* dispatch that reads that word first and runs only when it names it (decision record
|
|
12
|
+
* design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md: Dawn's validation of an indirect dispatch cost
|
|
13
|
+
* about 0.4 ms of device time each, and the seven indirect slots this kernel once wrote per level were 97 % of a
|
|
14
|
+
* traversal's wall time; the slots and their args buffer are gone). Two rules: a boundary that finds `done` set
|
|
15
|
+
* moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when role 0
|
|
16
|
+
* chose one, which it reads from the path word (a storage write of one dispatch is visible to the next of the same
|
|
17
|
+
* pass).
|
|
16
18
|
*
|
|
17
19
|
* Roles 2 and 3 are the SSSP round boundary of the near-far loop (P8-T9, PD-20), over the same block read in its
|
|
18
20
|
* SSSP sense (word 1 the raw near half's appends, 21 the raw far half's, 0 and 20 the deduped pile counts, 22 the
|
|
19
|
-
* threshold, 23 the delta, 4 the previous threshold, 14 the round's mode)
|
|
20
|
-
*
|
|
21
|
-
*
|
|
21
|
+
* threshold, 23 the delta, 4 the previous threshold, 14 the round's mode): role 2 finds a non-empty raw near half
|
|
22
|
+
* and sizes the near dedupe (count word 1, output word 0; the dedupe's count in word 8) in mode 0; finds it empty
|
|
23
|
+
* and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
|
|
22
24
|
* unchanged sets `done 3`, the host's E_UNSUPPORTED), remembers the previous threshold in word 4 and sizes the far
|
|
23
|
-
* dedupe (
|
|
24
|
-
* raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in
|
|
25
|
-
* picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds
|
|
26
|
-
* boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed
|
|
27
|
-
*
|
|
28
|
-
*
|
|
25
|
+
* dedupe (count word 21, output word 20; the dedupe's count in word 9) in mode 1; finds both empty and sets
|
|
26
|
+
* `done 1`; and finds a raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in
|
|
27
|
+
* `level` when it picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds
|
|
28
|
+
* dispatched; a boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed and restarts the
|
|
29
|
+
* raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
|
|
30
|
+
* to 0); the relax kernels size themselves from words 0 and 20.
|
|
29
31
|
*
|
|
30
32
|
* Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a switch can be counted at
|
|
31
33
|
* the done boundary too, which the host model of the tests mirrors): top-down switches to bottom-up when
|
|
@@ -46,42 +48,15 @@
|
|
|
46
48
|
* overstated m_u makes the switch INTO bottom-up harder, never easier -- and the next submit's rebuild makes it
|
|
47
49
|
* exact again. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
|
|
48
50
|
* textual edits of it.
|
|
49
|
-
*
|
|
50
|
-
* Since 2026-09-25 nothing dispatches FROM the slots (G8-F5: Dawn's validation of an indirect dispatch cost about
|
|
51
|
-
* 0.4 ms of device time each, whether or not it dispatched anything, and the seven slots of thirty-two recorded
|
|
52
|
-
* levels were 97 % of a traversal's wall time). Every level kernel is a direct grid-stride dispatch that reads the
|
|
53
|
-
* `path` word (24) this kernel writes -- 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3
|
|
54
|
-
* bottom-up, 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2) -- and the SSSP dedupes read
|
|
55
|
-
* their counts from words 8 and 9, which role 2 writes. The slots stay as the selector's recorded decision, read
|
|
56
|
-
* back by the frontier tests; deleting them with those tests is the follow-up.
|
|
57
51
|
*/
|
|
58
52
|
export const frontierFinalizeWgsl = /* wgsl */ `
|
|
59
|
-
fn write_slot_groups(slot: u32, groups: u32, count: u32) { // groups workgroups, split in 2D above the per-dim limit
|
|
60
|
-
var x = groups;
|
|
61
|
-
var y = 1u;
|
|
62
|
-
if (groups > MAX_WORKGROUPS_PER_DIM) {
|
|
63
|
-
x = MAX_WORKGROUPS_PER_DIM;
|
|
64
|
-
y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;
|
|
65
|
-
}
|
|
66
|
-
let base = 4u * (P.slotBase + slot); // 16-byte slots: (x, y, 1, count)
|
|
67
|
-
args[base] = x; args[base + 1u] = y; args[base + 2u] = 1u; args[base + 3u] = count;
|
|
68
|
-
}
|
|
69
|
-
fn write_slot(slot: u32, count: u32) { // one INVOCATION per entry: ceil(count / wg) workgroups
|
|
70
|
-
let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap (indirect-finalize's rule)
|
|
71
|
-
write_slot_groups(slot, groups, count);
|
|
72
|
-
}
|
|
73
|
-
fn zero_slot(slot: u32) {
|
|
74
|
-
let base = 4u * (P.slotBase + slot);
|
|
75
|
-
args[base] = 0u; args[base + 1u] = 0u; args[base + 2u] = 1u; args[base + 3u] = 0u;
|
|
76
|
-
}
|
|
77
|
-
|
|
78
53
|
@compute @workgroup_size(WG)
|
|
79
54
|
fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
80
55
|
if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
|
|
81
56
|
if (P.role == 0u) { // the level boundary
|
|
82
57
|
if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end
|
|
83
|
-
|
|
84
|
-
return;
|
|
58
|
+
atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)
|
|
59
|
+
return;
|
|
85
60
|
}
|
|
86
61
|
let finished = atomicLoad(&counters[0]);
|
|
87
62
|
let next = atomicLoad(&counters[1]);
|
|
@@ -115,39 +90,29 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
115
90
|
if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches
|
|
116
91
|
var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)
|
|
117
92
|
if (done) {
|
|
118
|
-
|
|
93
|
+
path = 0u;
|
|
119
94
|
} else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep
|
|
120
|
-
zero_slot(0u); zero_slot(1u); zero_slot(2u); zero_slot(6u);
|
|
121
|
-
write_slot(3u, (P.n + 31u) / 32u); write_slot(4u, next); write_slot(5u, atomicLoad(&counters[7]));
|
|
122
95
|
path = 3u;
|
|
123
96
|
atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);
|
|
124
97
|
} else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)
|
|
125
|
-
|
|
126
|
-
write_slot_groups(2u, next, next); // bfs-fused is one WORKGROUP per frontier entry
|
|
127
|
-
path = 2u;
|
|
98
|
+
path = 2u; // bfs-fused: one WORKGROUP per frontier entry
|
|
128
99
|
atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
|
|
129
100
|
} else {
|
|
130
|
-
|
|
131
|
-
write_slot(0u, next); // slots 1 and 6 are role 1's
|
|
132
|
-
path = 1u;
|
|
101
|
+
path = 1u; // advance-expand, then role 1 and bfs-contract
|
|
133
102
|
}
|
|
134
103
|
atomicStore(&counters[14], direction);
|
|
135
104
|
atomicStore(&counters[24], path);
|
|
136
105
|
} else if (P.role == 1u) { // the edge queue is filled
|
|
137
|
-
if (
|
|
138
|
-
zero_slot(1u); zero_slot(6u);
|
|
106
|
+
if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count
|
|
139
107
|
return;
|
|
140
108
|
}
|
|
141
109
|
let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);
|
|
142
110
|
atomicStore(&counters[8], clamped);
|
|
143
111
|
if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry
|
|
144
|
-
|
|
145
|
-
zero_slot(1u); write_slot_groups(6u, entries, entries); // one workgroup per frontier entry, as slot 2
|
|
146
|
-
atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry, bfs-contract nothing
|
|
112
|
+
atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing
|
|
147
113
|
atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);
|
|
148
114
|
atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
|
|
149
115
|
} else {
|
|
150
|
-
write_slot(1u, clamped); zero_slot(6u);
|
|
151
116
|
atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)
|
|
152
117
|
}
|
|
153
118
|
} else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes
|
|
@@ -155,54 +120,43 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
155
120
|
atomicStore(&counters[9], 0u);
|
|
156
121
|
atomicStore(&counters[24], 0u);
|
|
157
122
|
if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)
|
|
158
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
159
123
|
return;
|
|
160
124
|
}
|
|
161
125
|
let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped
|
|
162
126
|
let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped
|
|
163
127
|
if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE
|
|
164
128
|
atomicStore(&counters[15], 2u);
|
|
165
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
166
129
|
return;
|
|
167
130
|
}
|
|
168
|
-
zero_slot(2u); zero_slot(5u); zero_slot(6u); // role 3 sizes the relax slots once the piles are deduped
|
|
169
131
|
if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn
|
|
170
132
|
atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter
|
|
171
133
|
atomicStore(&counters[14], 0u); // mode 0
|
|
172
|
-
|
|
173
|
-
atomicStore(&counters[8], nearRaw); // the near dedupe's count word
|
|
134
|
+
atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)
|
|
174
135
|
atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing
|
|
175
|
-
zero_slot(3u); zero_slot(4u);
|
|
176
136
|
atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)
|
|
177
137
|
} else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile
|
|
178
138
|
let threshold = bitcast<f32>(atomicLoad(&counters[22]));
|
|
179
139
|
let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)
|
|
180
140
|
if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED
|
|
181
141
|
atomicStore(&counters[15], 3u);
|
|
182
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
183
142
|
return;
|
|
184
143
|
}
|
|
185
144
|
atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below
|
|
186
145
|
atomicStore(&counters[22], bitcast<u32>(raised));
|
|
187
146
|
atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter
|
|
188
147
|
atomicStore(&counters[14], 1u); // mode 1
|
|
189
|
-
|
|
190
|
-
write_slot(3u, farRaw); write_slot(4u, farRaw); // dedupe-claim, dedupe-filter over the far half
|
|
191
|
-
atomicStore(&counters[9], farRaw); // the far dedupe's count word
|
|
148
|
+
atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)
|
|
192
149
|
atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing
|
|
193
150
|
atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);
|
|
194
151
|
} else { // both piles empty: finished
|
|
195
152
|
atomicStore(&counters[15], 1u);
|
|
196
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
197
153
|
}
|
|
198
|
-
} else if (P.role == 3u) { // the piles are deduped:
|
|
199
|
-
if (atomicLoad(&counters[15]) != 0u) { return; } // role 2
|
|
154
|
+
} else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed
|
|
155
|
+
if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round
|
|
200
156
|
if (atomicLoad(&counters[14]) == 0u) {
|
|
201
|
-
|
|
202
|
-
atomicStore(&counters[1], 0u); // the raw near half restarts
|
|
157
|
+
atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)
|
|
203
158
|
} else {
|
|
204
|
-
|
|
205
|
-
atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far)
|
|
159
|
+
atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)
|
|
206
160
|
}
|
|
207
161
|
}
|
|
208
162
|
}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"frontier-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"frontier-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkDG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA+G9C,CAAC"}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { W as WebGpuGraphError, U as UNIFORM_SLOT_BYTES, B as BufferUsage, M as MAX_WORKGROUPS_PER_DIM, a as WGSL_RESERVED_WORDS, S as STATE_HEADER_BYTES, d as deviceLostError, i as isWebGpuGraphError, b as U32_MAX$2, c as MAX_LEVELS_PER_SUBMIT,
|
|
2
|
-
import {
|
|
1
|
+
import { W as WebGpuGraphError, U as UNIFORM_SLOT_BYTES, B as BufferUsage, M as MAX_WORKGROUPS_PER_DIM, a as WGSL_RESERVED_WORDS, S as STATE_HEADER_BYTES, d as deviceLostError, i as isWebGpuGraphError, b as U32_MAX$2, c as MAX_LEVELS_PER_SUBMIT, R as RADIX_BINS, F as FUSED_FRONTIER_MAX, e as BEAMER_BETA, f as SSSP_DELTA_FACTOR, g as F32_INF_BITS, h as GRID_COARSEST_SIDE, j as GRID_MIN_SIDE, k as GRID_SORT_BITS, l as FA2_DEFAULTS, m as MAX_ITERATIONS_PER_STEP, n as MAX_1D_ITEMS, o as hasErrorCode, p as FA2_FLAG_FIRST, P as PARTIAL_BYTES, I as INDIRECT_ARGS_STRIDE, q as GRID_HUB_CELL, L as LAYOUT_TUNING_DEFAULTS, E as EXACT_MAX_NODES, T as TRACE_RECORD_BYTES, r as GRID_BBOX_MARGIN, s as GRID_EXTENT_FLOOR, t as FR_ADAPTIVE_MAX_ITERATIONS, u as FR_START_TEMPERATURE, v as FA2_FLAG_ADAPTIVE, w as FR_REHEAT_FRACTION, x as FR_DEFAULTS, y as SE_DEFAULTS, z as SE_SCALE_REFERENCE_NODES } from "./chunks/context-hzGggHeM.js";
|
|
2
|
+
import { A, G, C, D, H, J } from "./chunks/context-hzGggHeM.js";
|
|
3
3
|
import { renumberPartition, INVALID_INDEX, makeMask, maskTest, expandEdges, fromEdgeArrays } from "@graphty/graph-format";
|
|
4
4
|
class UniformRing {
|
|
5
5
|
/**
|
|
@@ -1475,32 +1475,13 @@ fn fill(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid
|
|
|
1475
1475
|
const frontierFinalizeWgsl = (
|
|
1476
1476
|
/* wgsl */
|
|
1477
1477
|
`
|
|
1478
|
-
fn write_slot_groups(slot: u32, groups: u32, count: u32) { // groups workgroups, split in 2D above the per-dim limit
|
|
1479
|
-
var x = groups;
|
|
1480
|
-
var y = 1u;
|
|
1481
|
-
if (groups > MAX_WORKGROUPS_PER_DIM) {
|
|
1482
|
-
x = MAX_WORKGROUPS_PER_DIM;
|
|
1483
|
-
y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;
|
|
1484
|
-
}
|
|
1485
|
-
let base = 4u * (P.slotBase + slot); // 16-byte slots: (x, y, 1, count)
|
|
1486
|
-
args[base] = x; args[base + 1u] = y; args[base + 2u] = 1u; args[base + 3u] = count;
|
|
1487
|
-
}
|
|
1488
|
-
fn write_slot(slot: u32, count: u32) { // one INVOCATION per entry: ceil(count / wg) workgroups
|
|
1489
|
-
let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap (indirect-finalize's rule)
|
|
1490
|
-
write_slot_groups(slot, groups, count);
|
|
1491
|
-
}
|
|
1492
|
-
fn zero_slot(slot: u32) {
|
|
1493
|
-
let base = 4u * (P.slotBase + slot);
|
|
1494
|
-
args[base] = 0u; args[base + 1u] = 0u; args[base + 2u] = 1u; args[base + 3u] = 0u;
|
|
1495
|
-
}
|
|
1496
|
-
|
|
1497
1478
|
@compute @workgroup_size(WG)
|
|
1498
1479
|
fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
1499
1480
|
if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
|
|
1500
1481
|
if (P.role == 0u) { // the level boundary
|
|
1501
1482
|
if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end
|
|
1502
|
-
|
|
1503
|
-
return;
|
|
1483
|
+
atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)
|
|
1484
|
+
return;
|
|
1504
1485
|
}
|
|
1505
1486
|
let finished = atomicLoad(&counters[0]);
|
|
1506
1487
|
let next = atomicLoad(&counters[1]);
|
|
@@ -1534,39 +1515,29 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
1534
1515
|
if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches
|
|
1535
1516
|
var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)
|
|
1536
1517
|
if (done) {
|
|
1537
|
-
|
|
1518
|
+
path = 0u;
|
|
1538
1519
|
} else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep
|
|
1539
|
-
zero_slot(0u); zero_slot(1u); zero_slot(2u); zero_slot(6u);
|
|
1540
|
-
write_slot(3u, (P.n + 31u) / 32u); write_slot(4u, next); write_slot(5u, atomicLoad(&counters[7]));
|
|
1541
1520
|
path = 3u;
|
|
1542
1521
|
atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);
|
|
1543
1522
|
} else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)
|
|
1544
|
-
|
|
1545
|
-
write_slot_groups(2u, next, next); // bfs-fused is one WORKGROUP per frontier entry
|
|
1546
|
-
path = 2u;
|
|
1523
|
+
path = 2u; // bfs-fused: one WORKGROUP per frontier entry
|
|
1547
1524
|
atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
|
|
1548
1525
|
} else {
|
|
1549
|
-
|
|
1550
|
-
write_slot(0u, next); // slots 1 and 6 are role 1's
|
|
1551
|
-
path = 1u;
|
|
1526
|
+
path = 1u; // advance-expand, then role 1 and bfs-contract
|
|
1552
1527
|
}
|
|
1553
1528
|
atomicStore(&counters[14], direction);
|
|
1554
1529
|
atomicStore(&counters[24], path);
|
|
1555
1530
|
} else if (P.role == 1u) { // the edge queue is filled
|
|
1556
|
-
if (
|
|
1557
|
-
zero_slot(1u); zero_slot(6u);
|
|
1531
|
+
if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count
|
|
1558
1532
|
return;
|
|
1559
1533
|
}
|
|
1560
1534
|
let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);
|
|
1561
1535
|
atomicStore(&counters[8], clamped);
|
|
1562
1536
|
if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry
|
|
1563
|
-
|
|
1564
|
-
zero_slot(1u); write_slot_groups(6u, entries, entries); // one workgroup per frontier entry, as slot 2
|
|
1565
|
-
atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry, bfs-contract nothing
|
|
1537
|
+
atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing
|
|
1566
1538
|
atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);
|
|
1567
1539
|
atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
|
|
1568
1540
|
} else {
|
|
1569
|
-
write_slot(1u, clamped); zero_slot(6u);
|
|
1570
1541
|
atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)
|
|
1571
1542
|
}
|
|
1572
1543
|
} else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes
|
|
@@ -1574,54 +1545,43 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
1574
1545
|
atomicStore(&counters[9], 0u);
|
|
1575
1546
|
atomicStore(&counters[24], 0u);
|
|
1576
1547
|
if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)
|
|
1577
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
1578
1548
|
return;
|
|
1579
1549
|
}
|
|
1580
1550
|
let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped
|
|
1581
1551
|
let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped
|
|
1582
1552
|
if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE
|
|
1583
1553
|
atomicStore(&counters[15], 2u);
|
|
1584
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
1585
1554
|
return;
|
|
1586
1555
|
}
|
|
1587
|
-
zero_slot(2u); zero_slot(5u); zero_slot(6u); // role 3 sizes the relax slots once the piles are deduped
|
|
1588
1556
|
if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn
|
|
1589
1557
|
atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter
|
|
1590
1558
|
atomicStore(&counters[14], 0u); // mode 0
|
|
1591
|
-
|
|
1592
|
-
atomicStore(&counters[8], nearRaw); // the near dedupe's count word
|
|
1559
|
+
atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)
|
|
1593
1560
|
atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing
|
|
1594
|
-
zero_slot(3u); zero_slot(4u);
|
|
1595
1561
|
atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)
|
|
1596
1562
|
} else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile
|
|
1597
1563
|
let threshold = bitcast<f32>(atomicLoad(&counters[22]));
|
|
1598
1564
|
let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)
|
|
1599
1565
|
if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED
|
|
1600
1566
|
atomicStore(&counters[15], 3u);
|
|
1601
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
1602
1567
|
return;
|
|
1603
1568
|
}
|
|
1604
1569
|
atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below
|
|
1605
1570
|
atomicStore(&counters[22], bitcast<u32>(raised));
|
|
1606
1571
|
atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter
|
|
1607
1572
|
atomicStore(&counters[14], 1u); // mode 1
|
|
1608
|
-
|
|
1609
|
-
write_slot(3u, farRaw); write_slot(4u, farRaw); // dedupe-claim, dedupe-filter over the far half
|
|
1610
|
-
atomicStore(&counters[9], farRaw); // the far dedupe's count word
|
|
1573
|
+
atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)
|
|
1611
1574
|
atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing
|
|
1612
1575
|
atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);
|
|
1613
1576
|
} else { // both piles empty: finished
|
|
1614
1577
|
atomicStore(&counters[15], 1u);
|
|
1615
|
-
for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
|
|
1616
1578
|
}
|
|
1617
|
-
} else if (P.role == 3u) { // the piles are deduped:
|
|
1618
|
-
if (atomicLoad(&counters[15]) != 0u) { return; } // role 2
|
|
1579
|
+
} else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed
|
|
1580
|
+
if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round
|
|
1619
1581
|
if (atomicLoad(&counters[14]) == 0u) {
|
|
1620
|
-
|
|
1621
|
-
atomicStore(&counters[1], 0u); // the raw near half restarts
|
|
1582
|
+
atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)
|
|
1622
1583
|
} else {
|
|
1623
|
-
|
|
1624
|
-
atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far)
|
|
1584
|
+
atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)
|
|
1625
1585
|
}
|
|
1626
1586
|
}
|
|
1627
1587
|
}
|
|
@@ -2809,7 +2769,6 @@ const FRONTIER_COUNTERS = UniformBlock.define(
|
|
|
2809
2769
|
);
|
|
2810
2770
|
const FRONTIER_PARAMS = UniformBlock.define("FrontierParams", [
|
|
2811
2771
|
["role", "u32"],
|
|
2812
|
-
["slotBase", "u32"],
|
|
2813
2772
|
["wg", "u32"],
|
|
2814
2773
|
["alpha", "u32"],
|
|
2815
2774
|
["beta", "u32"],
|
|
@@ -2827,7 +2786,8 @@ const FRONTIER_PARAMS = UniformBlock.define("FrontierParams", [
|
|
|
2827
2786
|
["stride", "u32"],
|
|
2828
2787
|
["firstOfSubmit", "u32"],
|
|
2829
2788
|
["iteration", "u32"],
|
|
2830
|
-
["pad1", "u32"]
|
|
2789
|
+
["pad1", "u32"],
|
|
2790
|
+
["pad2", "u32"]
|
|
2831
2791
|
]);
|
|
2832
2792
|
const BF_PARAMS = UniformBlock.define("BfParams", [
|
|
2833
2793
|
["edgeCount", "u32"],
|
|
@@ -3406,11 +3366,7 @@ const FRONTIER_FINALIZE = {
|
|
|
3406
3366
|
id: "frontier-finalize",
|
|
3407
3367
|
body: frontierFinalizeWgsl,
|
|
3408
3368
|
entryPoint: "frontier_finalize",
|
|
3409
|
-
bindings: [
|
|
3410
|
-
decl(1, 0, "counters", "storage", "array<atomic<u32>>"),
|
|
3411
|
-
decl(1, 1, "args", "storage", "array<u32>"),
|
|
3412
|
-
decl(2, 0, "P", "uniform", "FrontierParams")
|
|
3413
|
-
],
|
|
3369
|
+
bindings: [decl(1, 0, "counters", "storage", "array<atomic<u32>>"), decl(2, 0, "P", "uniform", "FrontierParams")],
|
|
3414
3370
|
overrideDecls: [],
|
|
3415
3371
|
uniforms: [FRONTIER_PARAMS],
|
|
3416
3372
|
needs: [],
|
|
@@ -4486,11 +4442,6 @@ function algorithmScope(ctx, label, slots) {
|
|
|
4486
4442
|
pool: ctx.pool,
|
|
4487
4443
|
workgroupSize: ctx.workgroupSize,
|
|
4488
4444
|
scratch: (byteLength, scratchLabel) => lease.storage(byteLength, `${label}/${scratchLabel}`),
|
|
4489
|
-
indirect: (byteLength, indirectLabel) => lease.acquire(
|
|
4490
|
-
byteLength,
|
|
4491
|
-
BufferUsage.STORAGE | BufferUsage.INDIRECT | BufferUsage.COPY_DST | BufferUsage.COPY_SRC,
|
|
4492
|
-
`${label}/${indirectLabel}`
|
|
4493
|
-
),
|
|
4494
4445
|
params(block, values) {
|
|
4495
4446
|
const slot = ring.reserve(1);
|
|
4496
4447
|
ring.write(slot, block, values);
|
|
@@ -5822,16 +5773,14 @@ class Frontier {
|
|
|
5822
5773
|
* Wraps the leased buffers; use prepareFrontier().
|
|
5823
5774
|
* @param vertices - the two vertex queues
|
|
5824
5775
|
* @param counters - the counters block
|
|
5825
|
-
* @param args - the args buffer
|
|
5826
5776
|
* @param edgeQueue - the edge queue
|
|
5827
5777
|
* @param edgeCapacity - the edge queue's entry count
|
|
5828
5778
|
* @param n - the vertex count
|
|
5829
5779
|
*/
|
|
5830
|
-
constructor(vertices, counters,
|
|
5780
|
+
constructor(vertices, counters, edgeQueue, edgeCapacity, n) {
|
|
5831
5781
|
this.sideIndex = 0;
|
|
5832
5782
|
this.vertices = vertices;
|
|
5833
5783
|
this.counters = counters;
|
|
5834
|
-
this.args = args;
|
|
5835
5784
|
this.edgeQueue = edgeQueue;
|
|
5836
5785
|
this.edgeCapacity = edgeCapacity;
|
|
5837
5786
|
this.n = n;
|
|
@@ -5862,7 +5811,7 @@ class Frontier {
|
|
|
5862
5811
|
this.sideIndex = this.sideIndex === 0 ? 1 : 0;
|
|
5863
5812
|
}
|
|
5864
5813
|
/**
|
|
5865
|
-
* Seeds a traversal: one `queue.writeBuffer` of the whole
|
|
5814
|
+
* Seeds a traversal: one `queue.writeBuffer` of the whole 112-byte block (zero except the caller's words) and one of
|
|
5866
5815
|
* `vertices[0][0] = source`, both ordered before the submit that follows; the source is on side 0 afterwards.
|
|
5867
5816
|
* `frontierCount` is not a word to seed: the first boundary rotates word 1 into it (the BFS seed is
|
|
5868
5817
|
* `{ nextFrontierCount: 1, level: U32_MAX }`). A source outside `[0, n)`, an unknown word or a value that is not a
|
|
@@ -5899,7 +5848,6 @@ async function prepareFrontier(scope, n, arcCount, edgeCapacity) {
|
|
|
5899
5848
|
}
|
|
5900
5849
|
const kernel = await scope.pipelines.kernel(kernelSpec("frontier-finalize"));
|
|
5901
5850
|
const queueBytes = 4 * Math.max(1, n);
|
|
5902
|
-
const argsBytes = MAX_LEVELS_PER_SUBMIT * FRONTIER_CANDIDATES * INDIRECT_ARGS_STRIDE;
|
|
5903
5851
|
const vertices = [
|
|
5904
5852
|
{ buffer: scope.scratch(queueBytes, "frontier/vertices-0"), offset: 0, size: queueBytes, window: null },
|
|
5905
5853
|
{ buffer: scope.scratch(queueBytes, "frontier/vertices-1"), offset: 0, size: queueBytes, window: null }
|
|
@@ -5910,19 +5858,13 @@ async function prepareFrontier(scope, n, arcCount, edgeCapacity) {
|
|
|
5910
5858
|
size: FRONTIER_COUNTERS.byteLength,
|
|
5911
5859
|
window: null
|
|
5912
5860
|
};
|
|
5913
|
-
const args = {
|
|
5914
|
-
buffer: scope.indirect(argsBytes, "frontier/args"),
|
|
5915
|
-
offset: 0,
|
|
5916
|
-
size: argsBytes,
|
|
5917
|
-
window: null
|
|
5918
|
-
};
|
|
5919
5861
|
const edgeQueue = {
|
|
5920
5862
|
buffer: scope.scratch(4 * capacity, "frontier/edge-queue"),
|
|
5921
5863
|
offset: 0,
|
|
5922
5864
|
size: 4 * capacity,
|
|
5923
5865
|
window: null
|
|
5924
5866
|
};
|
|
5925
|
-
const frontier = new Frontier(vertices, counters,
|
|
5867
|
+
const frontier = new Frontier(vertices, counters, edgeQueue, capacity, n);
|
|
5926
5868
|
return new FrontierPlannerImpl(scope, kernel, frontier);
|
|
5927
5869
|
}
|
|
5928
5870
|
class FrontierPlannerImpl {
|
|
@@ -5963,12 +5905,11 @@ class FrontierPlannerImpl {
|
|
|
5963
5905
|
const params = scope.params(FRONTIER_PARAMS, {
|
|
5964
5906
|
...definedWords(fields),
|
|
5965
5907
|
role,
|
|
5966
|
-
slotBase: level * FRONTIER_CANDIDATES,
|
|
5967
5908
|
wg: scope.workgroupSize,
|
|
5968
5909
|
edgeCapacity: frontier.edgeCapacity,
|
|
5969
5910
|
n: frontier.n
|
|
5970
5911
|
});
|
|
5971
|
-
const bound = this.kernel.bind({ counters: frontier.counters,
|
|
5912
|
+
const bound = this.kernel.bind({ counters: frontier.counters, P: params.binding });
|
|
5972
5913
|
this.kernel.dispatch(pass, bound, plan1d(1, scope.workgroupSize, scope.caps), [params.offset]);
|
|
5973
5914
|
}
|
|
5974
5915
|
}
|
|
@@ -12653,7 +12594,7 @@ async function calibrateLayout(ctx, options) {
|
|
|
12653
12594
|
};
|
|
12654
12595
|
}
|
|
12655
12596
|
export {
|
|
12656
|
-
|
|
12597
|
+
A as ARC_WINDOW_ALIGN,
|
|
12657
12598
|
EXACT_MAX_NODES,
|
|
12658
12599
|
FA2_DEFAULTS,
|
|
12659
12600
|
FR_DEFAULTS,
|
|
@@ -12661,10 +12602,10 @@ export {
|
|
|
12661
12602
|
LAYOUT_TUNING_DEFAULTS,
|
|
12662
12603
|
MAX_1D_ITEMS,
|
|
12663
12604
|
MAX_WORKGROUPS_PER_DIM,
|
|
12664
|
-
|
|
12605
|
+
C as PASSTHROUGH_FORMAT_CODES,
|
|
12665
12606
|
SE_DEFAULTS,
|
|
12666
|
-
|
|
12667
|
-
|
|
12607
|
+
D as STORAGE_ALIGN,
|
|
12608
|
+
H as WORKGROUP_SIZE,
|
|
12668
12609
|
WebGpuGraphError,
|
|
12669
12610
|
bellmanFord,
|
|
12670
12611
|
breadthFirstSearch,
|
|
@@ -12679,7 +12620,7 @@ export {
|
|
|
12679
12620
|
eigenvectorCentrality,
|
|
12680
12621
|
hasErrorCode,
|
|
12681
12622
|
hits,
|
|
12682
|
-
|
|
12623
|
+
J as isSoftwareAdapter,
|
|
12683
12624
|
isWebGpuGraphError,
|
|
12684
12625
|
katzCentrality,
|
|
12685
12626
|
pageRank,
|