@graphty/webgpu-graph-algorithms 0.6.4 → 0.6.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/dist/browser.js +1 -1
  2. package/dist/chunks/{context-Dvq-Cc6v.js → context-hzGggHeM.js} +28 -30
  3. package/dist/chunks/context-hzGggHeM.js.map +1 -0
  4. package/dist/node.js +1 -1
  5. package/dist/src/algorithms/bfs.d.ts +1 -1
  6. package/dist/src/algorithms/bfs.js +1 -1
  7. package/dist/src/algorithms/scope.d.ts +3 -3
  8. package/dist/src/algorithms/scope.d.ts.map +1 -1
  9. package/dist/src/algorithms/scope.js +0 -2
  10. package/dist/src/algorithms/scope.js.map +1 -1
  11. package/dist/src/algorithms/sssp.d.ts +4 -3
  12. package/dist/src/algorithms/sssp.d.ts.map +1 -1
  13. package/dist/src/algorithms/sssp.js +4 -3
  14. package/dist/src/algorithms/sssp.js.map +1 -1
  15. package/dist/src/constants.d.ts +0 -2
  16. package/dist/src/constants.d.ts.map +1 -1
  17. package/dist/src/constants.js +0 -2
  18. package/dist/src/constants.js.map +1 -1
  19. package/dist/src/kernels.d.ts +8 -7
  20. package/dist/src/kernels.d.ts.map +1 -1
  21. package/dist/src/kernels.js +12 -15
  22. package/dist/src/kernels.js.map +1 -1
  23. package/dist/src/primitives/advance.d.ts +3 -2
  24. package/dist/src/primitives/advance.d.ts.map +1 -1
  25. package/dist/src/primitives/advance.js.map +1 -1
  26. package/dist/src/primitives/frontier.d.ts +33 -38
  27. package/dist/src/primitives/frontier.d.ts.map +1 -1
  28. package/dist/src/primitives/frontier.js +23 -32
  29. package/dist/src/primitives/frontier.js.map +1 -1
  30. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +24 -30
  31. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
  32. package/dist/src/wgsl/frontier-finalize.wgsl.js +36 -82
  33. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
  34. package/dist/webgpu-graph-algorithms.js +27 -86
  35. package/dist/webgpu-graph-algorithms.js.map +1 -1
  36. package/package.json +1 -1
  37. package/src/algorithms/bfs.ts +1 -1
  38. package/src/algorithms/scope.ts +3 -10
  39. package/src/algorithms/sssp.ts +4 -3
  40. package/src/constants.ts +0 -2
  41. package/src/kernels.ts +12 -15
  42. package/src/primitives/advance.ts +5 -4
  43. package/src/primitives/frontier.ts +40 -56
  44. package/src/wgsl/frontier-finalize.wgsl.ts +36 -82
  45. package/dist/chunks/context-Dvq-Cc6v.js.map +0 -1
@@ -1,31 +1,33 @@
1
1
  /**
2
2
  * The `frontier-finalize` kernel body (design 5.4, 6 row 7; P8-T4, the P8 plan's PD-3 / PD-23 / DEP-P8-C): the
3
3
  * device-side selector of the frontier family. One workgroup, one lane, no barrier after the early return (spec 3.5
4
- * rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's dispatch sizes become
5
- * known at two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`,
6
- * advances `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`), chooses the path and writes this
7
- * level's seven 16-byte slots at `P.slotBase`; role 1 runs once the edge queue is filled -- clamps `edgeCount` to
8
- * `P.edgeCapacity`, sizes the contract slot, or, when `edgeCountUnclamped` exceeds the capacity, zeroes it and sizes
9
- * the fused-retry slot from `frontierCount` instead (PD-23). Two rules: a boundary that finds `done` set zeroes its
10
- * slots and moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when
11
- * role 0 chose one, which it reads from slot 0's `x` (a storage write of one dispatch is visible to the next of the
12
- * same pass). The `(x, y)` arithmetic is `indirect-finalize`'s verbatim (P4), so a count above 2^32 - wg cannot wrap
13
- * and a group count above MAX_WORKGROUPS_PER_DIM splits in 2D; slots 2 and 6 are sized one WORKGROUP per entry for
14
- * `bfs-fused`; slots 3, 4 and 5 (the bits fill over `ceil(n / 32)` words, the bitset build over the frontier, the sweep
15
- * over the unvisited list) are the bottom-up level's.
4
+ * rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's counts become known at
5
+ * two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`, advances
6
+ * `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`) and chooses the path; role 1 runs once the
7
+ * edge queue is filled -- clamps `edgeCount` to `P.edgeCapacity`, or, when `edgeCountUnclamped` exceeds the
8
+ * capacity, switches the path to the fused retry over `frontierCount` (PD-23). The decision is ONE word of the
9
+ * counters block, `path` (word 24): 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3 bottom-up,
10
+ * 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2). Every level kernel is a direct grid-stride
11
+ * dispatch that reads that word first and runs only when it names it (decision record
12
+ * design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md: Dawn's validation of an indirect dispatch cost
13
+ * about 0.4 ms of device time each, and the seven indirect slots this kernel once wrote per level were 97 % of a
14
+ * traversal's wall time; the slots and their args buffer are gone). Two rules: a boundary that finds `done` set
15
+ * moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when role 0
16
+ * chose one, which it reads from the path word (a storage write of one dispatch is visible to the next of the same
17
+ * pass).
16
18
  *
17
19
  * Roles 2 and 3 are the SSSP round boundary of the near-far loop (P8-T9, PD-20), over the same block read in its
18
20
  * SSSP sense (word 1 the raw near half's appends, 21 the raw far half's, 0 and 20 the deduped pile counts, 22 the
19
- * threshold, 23 the delta, 4 the previous threshold, 14 the round's mode) and the same seven slots: role 2 finds a
20
- * non-empty raw near half and sizes the near dedupe (slots 0 and 1, count word 1, output word 0) in mode 0; finds it
21
- * empty and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
21
+ * threshold, 23 the delta, 4 the previous threshold, 14 the round's mode): role 2 finds a non-empty raw near half
22
+ * and sizes the near dedupe (count word 1, output word 0; the dedupe's count in word 8) in mode 0; finds it empty
23
+ * and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
22
24
  * unchanged sets `done 3`, the host's E_UNSUPPORTED), remembers the previous threshold in word 4 and sizes the far
23
- * dedupe (slots 3 and 4, count word 21, output word 20) in mode 1; finds both empty and sets `done 1`; and finds a
24
- * raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in `level` when it
25
- * picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds dispatched; a
26
- * boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed: mode 0 sizes slot 2 (the
27
- * relax over nearIn) from word 0 and restarts the raw near half (word 1 to 0); mode 1 sizes slot 5 (the pass-through
28
- * over farIn) from word 20 and restarts the raw far half (word 21 to 0).
25
+ * dedupe (count word 21, output word 20; the dedupe's count in word 9) in mode 1; finds both empty and sets
26
+ * `done 1`; and finds a raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in
27
+ * `level` when it picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds
28
+ * dispatched; a boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed and restarts the
29
+ * raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
30
+ * to 0); the relax kernels size themselves from words 0 and 20.
29
31
  *
30
32
  * Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a switch can be counted at
31
33
  * the done boundary too, which the host model of the tests mirrors): top-down switches to bottom-up when
@@ -46,42 +48,15 @@
46
48
  * overstated m_u makes the switch INTO bottom-up harder, never easier -- and the next submit's rebuild makes it
47
49
  * exact again. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
48
50
  * textual edits of it.
49
- *
50
- * Since 2026-09-25 nothing dispatches FROM the slots (G8-F5: Dawn's validation of an indirect dispatch cost about
51
- * 0.4 ms of device time each, whether or not it dispatched anything, and the seven slots of thirty-two recorded
52
- * levels were 97 % of a traversal's wall time). Every level kernel is a direct grid-stride dispatch that reads the
53
- * `path` word (24) this kernel writes -- 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3
54
- * bottom-up, 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2) -- and the SSSP dedupes read
55
- * their counts from words 8 and 9, which role 2 writes. The slots stay as the selector's recorded decision, read
56
- * back by the frontier tests; deleting them with those tests is the follow-up.
57
51
  */
58
52
  export const frontierFinalizeWgsl = /* wgsl */ `
59
- fn write_slot_groups(slot: u32, groups: u32, count: u32) { // groups workgroups, split in 2D above the per-dim limit
60
- var x = groups;
61
- var y = 1u;
62
- if (groups > MAX_WORKGROUPS_PER_DIM) {
63
- x = MAX_WORKGROUPS_PER_DIM;
64
- y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;
65
- }
66
- let base = 4u * (P.slotBase + slot); // 16-byte slots: (x, y, 1, count)
67
- args[base] = x; args[base + 1u] = y; args[base + 2u] = 1u; args[base + 3u] = count;
68
- }
69
- fn write_slot(slot: u32, count: u32) { // one INVOCATION per entry: ceil(count / wg) workgroups
70
- let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap (indirect-finalize's rule)
71
- write_slot_groups(slot, groups, count);
72
- }
73
- fn zero_slot(slot: u32) {
74
- let base = 4u * (P.slotBase + slot);
75
- args[base] = 0u; args[base + 1u] = 0u; args[base + 2u] = 1u; args[base + 3u] = 0u;
76
- }
77
-
78
53
  @compute @workgroup_size(WG)
79
54
  fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
80
55
  if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
81
56
  if (P.role == 0u) { // the level boundary
82
57
  if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end
83
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); } // this slotBase holds the previous submit's args
84
- return; // no counter word moves (P8-T6's levels formula reads them)
58
+ atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)
59
+ return;
85
60
  }
86
61
  let finished = atomicLoad(&counters[0]);
87
62
  let next = atomicLoad(&counters[1]);
@@ -115,39 +90,29 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
115
90
  if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches
116
91
  var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)
117
92
  if (done) {
118
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
93
+ path = 0u;
119
94
  } else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep
120
- zero_slot(0u); zero_slot(1u); zero_slot(2u); zero_slot(6u);
121
- write_slot(3u, (P.n + 31u) / 32u); write_slot(4u, next); write_slot(5u, atomicLoad(&counters[7]));
122
95
  path = 3u;
123
96
  atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);
124
97
  } else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)
125
- zero_slot(0u); zero_slot(1u); zero_slot(3u); zero_slot(4u); zero_slot(5u); zero_slot(6u);
126
- write_slot_groups(2u, next, next); // bfs-fused is one WORKGROUP per frontier entry
127
- path = 2u;
98
+ path = 2u; // bfs-fused: one WORKGROUP per frontier entry
128
99
  atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
129
100
  } else {
130
- zero_slot(2u); zero_slot(3u); zero_slot(4u); zero_slot(5u);
131
- write_slot(0u, next); // slots 1 and 6 are role 1's
132
- path = 1u;
101
+ path = 1u; // advance-expand, then role 1 and bfs-contract
133
102
  }
134
103
  atomicStore(&counters[14], direction);
135
104
  atomicStore(&counters[24], path);
136
105
  } else if (P.role == 1u) { // the edge queue is filled
137
- if (args[4u * P.slotBase] == 0u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to size, nothing to count
138
- zero_slot(1u); zero_slot(6u);
106
+ if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count
139
107
  return;
140
108
  }
141
109
  let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);
142
110
  atomicStore(&counters[8], clamped);
143
111
  if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry
144
- let entries = atomicLoad(&counters[0]);
145
- zero_slot(1u); write_slot_groups(6u, entries, entries); // one workgroup per frontier entry, as slot 2
146
- atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry, bfs-contract nothing
112
+ atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing
147
113
  atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);
148
114
  atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
149
115
  } else {
150
- write_slot(1u, clamped); zero_slot(6u);
151
116
  atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)
152
117
  }
153
118
  } else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes
@@ -155,54 +120,43 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
155
120
  atomicStore(&counters[9], 0u);
156
121
  atomicStore(&counters[24], 0u);
157
122
  if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)
158
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
159
123
  return;
160
124
  }
161
125
  let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped
162
126
  let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped
163
127
  if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE
164
128
  atomicStore(&counters[15], 2u);
165
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
166
129
  return;
167
130
  }
168
- zero_slot(2u); zero_slot(5u); zero_slot(6u); // role 3 sizes the relax slots once the piles are deduped
169
131
  if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn
170
132
  atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter
171
133
  atomicStore(&counters[14], 0u); // mode 0
172
- write_slot(0u, nearRaw); write_slot(1u, nearRaw); // dedupe-claim, dedupe-filter over the near half
173
- atomicStore(&counters[8], nearRaw); // the near dedupe's count word
134
+ atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)
174
135
  atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing
175
- zero_slot(3u); zero_slot(4u);
176
136
  atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)
177
137
  } else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile
178
138
  let threshold = bitcast<f32>(atomicLoad(&counters[22]));
179
139
  let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)
180
140
  if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED
181
141
  atomicStore(&counters[15], 3u);
182
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
183
142
  return;
184
143
  }
185
144
  atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below
186
145
  atomicStore(&counters[22], bitcast<u32>(raised));
187
146
  atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter
188
147
  atomicStore(&counters[14], 1u); // mode 1
189
- zero_slot(0u); zero_slot(1u);
190
- write_slot(3u, farRaw); write_slot(4u, farRaw); // dedupe-claim, dedupe-filter over the far half
191
- atomicStore(&counters[9], farRaw); // the far dedupe's count word
148
+ atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)
192
149
  atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing
193
150
  atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);
194
151
  } else { // both piles empty: finished
195
152
  atomicStore(&counters[15], 1u);
196
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
197
153
  }
198
- } else if (P.role == 3u) { // the piles are deduped: size the relax
199
- if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 zeroed every slot of the round
154
+ } else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed
155
+ if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round
200
156
  if (atomicLoad(&counters[14]) == 0u) {
201
- write_slot(2u, atomicLoad(&counters[0])); // the near round over nearIn
202
- atomicStore(&counters[1], 0u); // the raw near half restarts
157
+ atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)
203
158
  } else {
204
- write_slot(5u, atomicLoad(&counters[20])); // the pass-through over farIn
205
- atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far)
159
+ atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)
206
160
  }
207
161
  }
208
162
  }