@graphty/webgpu-graph-algorithms 0.6.4 → 0.6.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/dist/browser.js +1 -1
  2. package/dist/chunks/{context-Dvq-Cc6v.js → context-hzGggHeM.js} +28 -30
  3. package/dist/chunks/context-hzGggHeM.js.map +1 -0
  4. package/dist/node.js +1 -1
  5. package/dist/src/algorithms/bfs.d.ts +1 -1
  6. package/dist/src/algorithms/bfs.js +1 -1
  7. package/dist/src/algorithms/scope.d.ts +3 -3
  8. package/dist/src/algorithms/scope.d.ts.map +1 -1
  9. package/dist/src/algorithms/scope.js +0 -2
  10. package/dist/src/algorithms/scope.js.map +1 -1
  11. package/dist/src/algorithms/sssp.d.ts +4 -3
  12. package/dist/src/algorithms/sssp.d.ts.map +1 -1
  13. package/dist/src/algorithms/sssp.js +4 -3
  14. package/dist/src/algorithms/sssp.js.map +1 -1
  15. package/dist/src/constants.d.ts +0 -2
  16. package/dist/src/constants.d.ts.map +1 -1
  17. package/dist/src/constants.js +0 -2
  18. package/dist/src/constants.js.map +1 -1
  19. package/dist/src/kernels.d.ts +8 -7
  20. package/dist/src/kernels.d.ts.map +1 -1
  21. package/dist/src/kernels.js +12 -15
  22. package/dist/src/kernels.js.map +1 -1
  23. package/dist/src/primitives/advance.d.ts +3 -2
  24. package/dist/src/primitives/advance.d.ts.map +1 -1
  25. package/dist/src/primitives/advance.js.map +1 -1
  26. package/dist/src/primitives/frontier.d.ts +33 -38
  27. package/dist/src/primitives/frontier.d.ts.map +1 -1
  28. package/dist/src/primitives/frontier.js +23 -32
  29. package/dist/src/primitives/frontier.js.map +1 -1
  30. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +24 -30
  31. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
  32. package/dist/src/wgsl/frontier-finalize.wgsl.js +36 -82
  33. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
  34. package/dist/webgpu-graph-algorithms.js +27 -86
  35. package/dist/webgpu-graph-algorithms.js.map +1 -1
  36. package/package.json +1 -1
  37. package/src/algorithms/bfs.ts +1 -1
  38. package/src/algorithms/scope.ts +3 -10
  39. package/src/algorithms/sssp.ts +4 -3
  40. package/src/constants.ts +0 -2
  41. package/src/kernels.ts +12 -15
  42. package/src/primitives/advance.ts +5 -4
  43. package/src/primitives/frontier.ts +40 -56
  44. package/src/wgsl/frontier-finalize.wgsl.ts +36 -82
  45. package/dist/chunks/context-Dvq-Cc6v.js.map +0 -1
@@ -1,31 +1,33 @@
1
1
  /**
2
2
  * The `frontier-finalize` kernel body (design 5.4, 6 row 7; P8-T4, the P8 plan's PD-3 / PD-23 / DEP-P8-C): the
3
3
  * device-side selector of the frontier family. One workgroup, one lane, no barrier after the early return (spec 3.5
4
- * rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's dispatch sizes become
5
- * known at two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`,
6
- * advances `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`), chooses the path and writes this
7
- * level's seven 16-byte slots at `P.slotBase`; role 1 runs once the edge queue is filled -- clamps `edgeCount` to
8
- * `P.edgeCapacity`, sizes the contract slot, or, when `edgeCountUnclamped` exceeds the capacity, zeroes it and sizes
9
- * the fused-retry slot from `frontierCount` instead (PD-23). Two rules: a boundary that finds `done` set zeroes its
10
- * slots and moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when
11
- * role 0 chose one, which it reads from slot 0's `x` (a storage write of one dispatch is visible to the next of the
12
- * same pass). The `(x, y)` arithmetic is `indirect-finalize`'s verbatim (P4), so a count above 2^32 - wg cannot wrap
13
- * and a group count above MAX_WORKGROUPS_PER_DIM splits in 2D; slots 2 and 6 are sized one WORKGROUP per entry for
14
- * `bfs-fused`; slots 3, 4 and 5 (the bits fill over `ceil(n / 32)` words, the bitset build over the frontier, the sweep
15
- * over the unvisited list) are the bottom-up level's.
4
+ * rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's counts become known at
5
+ * two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`, advances
6
+ * `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`) and chooses the path; role 1 runs once the
7
+ * edge queue is filled -- clamps `edgeCount` to `P.edgeCapacity`, or, when `edgeCountUnclamped` exceeds the
8
+ * capacity, switches the path to the fused retry over `frontierCount` (PD-23). The decision is ONE word of the
9
+ * counters block, `path` (word 24): 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3 bottom-up,
10
+ * 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2). Every level kernel is a direct grid-stride
11
+ * dispatch that reads that word first and runs only when it names it (decision record
12
+ * design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md: Dawn's validation of an indirect dispatch cost
13
+ * about 0.4 ms of device time each, and the seven indirect slots this kernel once wrote per level were 97 % of a
14
+ * traversal's wall time; the slots and their args buffer are gone). Two rules: a boundary that finds `done` set
15
+ * moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when role 0
16
+ * chose one, which it reads from the path word (a storage write of one dispatch is visible to the next of the same
17
+ * pass).
16
18
  *
17
19
  * Roles 2 and 3 are the SSSP round boundary of the near-far loop (P8-T9, PD-20), over the same block read in its
18
20
  * SSSP sense (word 1 the raw near half's appends, 21 the raw far half's, 0 and 20 the deduped pile counts, 22 the
19
- * threshold, 23 the delta, 4 the previous threshold, 14 the round's mode) and the same seven slots: role 2 finds a
20
- * non-empty raw near half and sizes the near dedupe (slots 0 and 1, count word 1, output word 0) in mode 0; finds it
21
- * empty and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
21
+ * threshold, 23 the delta, 4 the previous threshold, 14 the round's mode): role 2 finds a non-empty raw near half
22
+ * and sizes the near dedupe (count word 1, output word 0; the dedupe's count in word 8) in mode 0; finds it empty
23
+ * and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
22
24
  * unchanged sets `done 3`, the host's E_UNSUPPORTED), remembers the previous threshold in word 4 and sizes the far
23
- * dedupe (slots 3 and 4, count word 21, output word 20) in mode 1; finds both empty and sets `done 1`; and finds a
24
- * raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in `level` when it
25
- * picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds dispatched; a
26
- * boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed: mode 0 sizes slot 2 (the
27
- * relax over nearIn) from word 0 and restarts the raw near half (word 1 to 0); mode 1 sizes slot 5 (the pass-through
28
- * over farIn) from word 20 and restarts the raw far half (word 21 to 0).
25
+ * dedupe (count word 21, output word 20; the dedupe's count in word 9) in mode 1; finds both empty and sets
26
+ * `done 1`; and finds a raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in
27
+ * `level` when it picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds
28
+ * dispatched; a boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed and restarts the
29
+ * raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
30
+ * to 0); the relax kernels size themselves from words 0 and 20.
29
31
  *
30
32
  * Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a switch can be counted at
31
33
  * the done boundary too, which the host model of the tests mirrors): top-down switches to bottom-up when
@@ -46,42 +48,15 @@
46
48
  * overstated m_u makes the switch INTO bottom-up harder, never easier -- and the next submit's rebuild makes it
47
49
  * exact again. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
48
50
  * textual edits of it.
49
- *
50
- * Since 2026-09-25 nothing dispatches FROM the slots (G8-F5: Dawn's validation of an indirect dispatch cost about
51
- * 0.4 ms of device time each, whether or not it dispatched anything, and the seven slots of thirty-two recorded
52
- * levels were 97 % of a traversal's wall time). Every level kernel is a direct grid-stride dispatch that reads the
53
- * `path` word (24) this kernel writes -- 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3
54
- * bottom-up, 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2) -- and the SSSP dedupes read
55
- * their counts from words 8 and 9, which role 2 writes. The slots stay as the selector's recorded decision, read
56
- * back by the frontier tests; deleting them with those tests is the follow-up.
57
51
  */
58
52
  export const frontierFinalizeWgsl = /* wgsl */ `
59
- fn write_slot_groups(slot: u32, groups: u32, count: u32) { // groups workgroups, split in 2D above the per-dim limit
60
- var x = groups;
61
- var y = 1u;
62
- if (groups > MAX_WORKGROUPS_PER_DIM) {
63
- x = MAX_WORKGROUPS_PER_DIM;
64
- y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;
65
- }
66
- let base = 4u * (P.slotBase + slot); // 16-byte slots: (x, y, 1, count)
67
- args[base] = x; args[base + 1u] = y; args[base + 2u] = 1u; args[base + 3u] = count;
68
- }
69
- fn write_slot(slot: u32, count: u32) { // one INVOCATION per entry: ceil(count / wg) workgroups
70
- let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap (indirect-finalize's rule)
71
- write_slot_groups(slot, groups, count);
72
- }
73
- fn zero_slot(slot: u32) {
74
- let base = 4u * (P.slotBase + slot);
75
- args[base] = 0u; args[base + 1u] = 0u; args[base + 2u] = 1u; args[base + 3u] = 0u;
76
- }
77
-
78
53
  @compute @workgroup_size(WG)
79
54
  fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
80
55
  if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
81
56
  if (P.role == 0u) { // the level boundary
82
57
  if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end
83
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); } // this slotBase holds the previous submit's args
84
- return; // no counter word moves (P8-T6's levels formula reads them)
58
+ atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)
59
+ return;
85
60
  }
86
61
  let finished = atomicLoad(&counters[0]);
87
62
  let next = atomicLoad(&counters[1]);
@@ -115,39 +90,29 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
115
90
  if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches
116
91
  var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)
117
92
  if (done) {
118
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
93
+ path = 0u;
119
94
  } else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep
120
- zero_slot(0u); zero_slot(1u); zero_slot(2u); zero_slot(6u);
121
- write_slot(3u, (P.n + 31u) / 32u); write_slot(4u, next); write_slot(5u, atomicLoad(&counters[7]));
122
95
  path = 3u;
123
96
  atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);
124
97
  } else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)
125
- zero_slot(0u); zero_slot(1u); zero_slot(3u); zero_slot(4u); zero_slot(5u); zero_slot(6u);
126
- write_slot_groups(2u, next, next); // bfs-fused is one WORKGROUP per frontier entry
127
- path = 2u;
98
+ path = 2u; // bfs-fused: one WORKGROUP per frontier entry
128
99
  atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
129
100
  } else {
130
- zero_slot(2u); zero_slot(3u); zero_slot(4u); zero_slot(5u);
131
- write_slot(0u, next); // slots 1 and 6 are role 1's
132
- path = 1u;
101
+ path = 1u; // advance-expand, then role 1 and bfs-contract
133
102
  }
134
103
  atomicStore(&counters[14], direction);
135
104
  atomicStore(&counters[24], path);
136
105
  } else if (P.role == 1u) { // the edge queue is filled
137
- if (args[4u * P.slotBase] == 0u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to size, nothing to count
138
- zero_slot(1u); zero_slot(6u);
106
+ if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count
139
107
  return;
140
108
  }
141
109
  let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);
142
110
  atomicStore(&counters[8], clamped);
143
111
  if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry
144
- let entries = atomicLoad(&counters[0]);
145
- zero_slot(1u); write_slot_groups(6u, entries, entries); // one workgroup per frontier entry, as slot 2
146
- atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry, bfs-contract nothing
112
+ atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing
147
113
  atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);
148
114
  atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
149
115
  } else {
150
- write_slot(1u, clamped); zero_slot(6u);
151
116
  atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)
152
117
  }
153
118
  } else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes
@@ -155,54 +120,43 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
155
120
  atomicStore(&counters[9], 0u);
156
121
  atomicStore(&counters[24], 0u);
157
122
  if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)
158
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
159
123
  return;
160
124
  }
161
125
  let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped
162
126
  let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped
163
127
  if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE
164
128
  atomicStore(&counters[15], 2u);
165
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
166
129
  return;
167
130
  }
168
- zero_slot(2u); zero_slot(5u); zero_slot(6u); // role 3 sizes the relax slots once the piles are deduped
169
131
  if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn
170
132
  atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter
171
133
  atomicStore(&counters[14], 0u); // mode 0
172
- write_slot(0u, nearRaw); write_slot(1u, nearRaw); // dedupe-claim, dedupe-filter over the near half
173
- atomicStore(&counters[8], nearRaw); // the near dedupe's count word
134
+ atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)
174
135
  atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing
175
- zero_slot(3u); zero_slot(4u);
176
136
  atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)
177
137
  } else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile
178
138
  let threshold = bitcast<f32>(atomicLoad(&counters[22]));
179
139
  let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)
180
140
  if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED
181
141
  atomicStore(&counters[15], 3u);
182
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
183
142
  return;
184
143
  }
185
144
  atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below
186
145
  atomicStore(&counters[22], bitcast<u32>(raised));
187
146
  atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter
188
147
  atomicStore(&counters[14], 1u); // mode 1
189
- zero_slot(0u); zero_slot(1u);
190
- write_slot(3u, farRaw); write_slot(4u, farRaw); // dedupe-claim, dedupe-filter over the far half
191
- atomicStore(&counters[9], farRaw); // the far dedupe's count word
148
+ atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)
192
149
  atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing
193
150
  atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);
194
151
  } else { // both piles empty: finished
195
152
  atomicStore(&counters[15], 1u);
196
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
197
153
  }
198
- } else if (P.role == 3u) { // the piles are deduped: size the relax
199
- if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 zeroed every slot of the round
154
+ } else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed
155
+ if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round
200
156
  if (atomicLoad(&counters[14]) == 0u) {
201
- write_slot(2u, atomicLoad(&counters[0])); // the near round over nearIn
202
- atomicStore(&counters[1], 0u); // the raw near half restarts
157
+ atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)
203
158
  } else {
204
- write_slot(5u, atomicLoad(&counters[20])); // the pass-through over farIn
205
- atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far)
159
+ atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)
206
160
  }
207
161
  }
208
162
  }
@@ -1 +1 @@
1
- {"version":3,"file":"frontier-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAwDG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAuJ9C,CAAC"}
1
+ {"version":3,"file":"frontier-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkDG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA+G9C,CAAC"}
@@ -1,5 +1,5 @@
1
- import { W as WebGpuGraphError, U as UNIFORM_SLOT_BYTES, B as BufferUsage, M as MAX_WORKGROUPS_PER_DIM, a as WGSL_RESERVED_WORDS, S as STATE_HEADER_BYTES, d as deviceLostError, i as isWebGpuGraphError, b as U32_MAX$2, c as MAX_LEVELS_PER_SUBMIT, F as FRONTIER_CANDIDATES, I as INDIRECT_ARGS_STRIDE, R as RADIX_BINS, e as FUSED_FRONTIER_MAX, f as BEAMER_BETA, g as SSSP_DELTA_FACTOR, h as F32_INF_BITS, j as GRID_COARSEST_SIDE, k as GRID_MIN_SIDE, l as GRID_SORT_BITS, m as FA2_DEFAULTS, n as MAX_ITERATIONS_PER_STEP, o as MAX_1D_ITEMS, p as hasErrorCode, q as FA2_FLAG_FIRST, P as PARTIAL_BYTES, r as GRID_HUB_CELL, L as LAYOUT_TUNING_DEFAULTS, E as EXACT_MAX_NODES, T as TRACE_RECORD_BYTES, s as GRID_BBOX_MARGIN, t as GRID_EXTENT_FLOOR, u as FR_ADAPTIVE_MAX_ITERATIONS, v as FR_START_TEMPERATURE, w as FA2_FLAG_ADAPTIVE, x as FR_REHEAT_FRACTION, y as FR_DEFAULTS, z as SE_DEFAULTS, A as SE_SCALE_REFERENCE_NODES } from "./chunks/context-Dvq-Cc6v.js";
2
- import { C, G, D, H, J, K } from "./chunks/context-Dvq-Cc6v.js";
1
+ import { W as WebGpuGraphError, U as UNIFORM_SLOT_BYTES, B as BufferUsage, M as MAX_WORKGROUPS_PER_DIM, a as WGSL_RESERVED_WORDS, S as STATE_HEADER_BYTES, d as deviceLostError, i as isWebGpuGraphError, b as U32_MAX$2, c as MAX_LEVELS_PER_SUBMIT, R as RADIX_BINS, F as FUSED_FRONTIER_MAX, e as BEAMER_BETA, f as SSSP_DELTA_FACTOR, g as F32_INF_BITS, h as GRID_COARSEST_SIDE, j as GRID_MIN_SIDE, k as GRID_SORT_BITS, l as FA2_DEFAULTS, m as MAX_ITERATIONS_PER_STEP, n as MAX_1D_ITEMS, o as hasErrorCode, p as FA2_FLAG_FIRST, P as PARTIAL_BYTES, I as INDIRECT_ARGS_STRIDE, q as GRID_HUB_CELL, L as LAYOUT_TUNING_DEFAULTS, E as EXACT_MAX_NODES, T as TRACE_RECORD_BYTES, r as GRID_BBOX_MARGIN, s as GRID_EXTENT_FLOOR, t as FR_ADAPTIVE_MAX_ITERATIONS, u as FR_START_TEMPERATURE, v as FA2_FLAG_ADAPTIVE, w as FR_REHEAT_FRACTION, x as FR_DEFAULTS, y as SE_DEFAULTS, z as SE_SCALE_REFERENCE_NODES } from "./chunks/context-hzGggHeM.js";
2
+ import { A, G, C, D, H, J } from "./chunks/context-hzGggHeM.js";
3
3
  import { renumberPartition, INVALID_INDEX, makeMask, maskTest, expandEdges, fromEdgeArrays } from "@graphty/graph-format";
4
4
  class UniformRing {
5
5
  /**
@@ -1475,32 +1475,13 @@ fn fill(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid
1475
1475
  const frontierFinalizeWgsl = (
1476
1476
  /* wgsl */
1477
1477
  `
1478
- fn write_slot_groups(slot: u32, groups: u32, count: u32) { // groups workgroups, split in 2D above the per-dim limit
1479
- var x = groups;
1480
- var y = 1u;
1481
- if (groups > MAX_WORKGROUPS_PER_DIM) {
1482
- x = MAX_WORKGROUPS_PER_DIM;
1483
- y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;
1484
- }
1485
- let base = 4u * (P.slotBase + slot); // 16-byte slots: (x, y, 1, count)
1486
- args[base] = x; args[base + 1u] = y; args[base + 2u] = 1u; args[base + 3u] = count;
1487
- }
1488
- fn write_slot(slot: u32, count: u32) { // one INVOCATION per entry: ceil(count / wg) workgroups
1489
- let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap (indirect-finalize's rule)
1490
- write_slot_groups(slot, groups, count);
1491
- }
1492
- fn zero_slot(slot: u32) {
1493
- let base = 4u * (P.slotBase + slot);
1494
- args[base] = 0u; args[base + 1u] = 0u; args[base + 2u] = 1u; args[base + 3u] = 0u;
1495
- }
1496
-
1497
1478
  @compute @workgroup_size(WG)
1498
1479
  fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
1499
1480
  if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
1500
1481
  if (P.role == 0u) { // the level boundary
1501
1482
  if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end
1502
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); } // this slotBase holds the previous submit's args
1503
- return; // no counter word moves (P8-T6's levels formula reads them)
1483
+ atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)
1484
+ return;
1504
1485
  }
1505
1486
  let finished = atomicLoad(&counters[0]);
1506
1487
  let next = atomicLoad(&counters[1]);
@@ -1534,39 +1515,29 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
1534
1515
  if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches
1535
1516
  var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)
1536
1517
  if (done) {
1537
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
1518
+ path = 0u;
1538
1519
  } else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep
1539
- zero_slot(0u); zero_slot(1u); zero_slot(2u); zero_slot(6u);
1540
- write_slot(3u, (P.n + 31u) / 32u); write_slot(4u, next); write_slot(5u, atomicLoad(&counters[7]));
1541
1520
  path = 3u;
1542
1521
  atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);
1543
1522
  } else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)
1544
- zero_slot(0u); zero_slot(1u); zero_slot(3u); zero_slot(4u); zero_slot(5u); zero_slot(6u);
1545
- write_slot_groups(2u, next, next); // bfs-fused is one WORKGROUP per frontier entry
1546
- path = 2u;
1523
+ path = 2u; // bfs-fused: one WORKGROUP per frontier entry
1547
1524
  atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
1548
1525
  } else {
1549
- zero_slot(2u); zero_slot(3u); zero_slot(4u); zero_slot(5u);
1550
- write_slot(0u, next); // slots 1 and 6 are role 1's
1551
- path = 1u;
1526
+ path = 1u; // advance-expand, then role 1 and bfs-contract
1552
1527
  }
1553
1528
  atomicStore(&counters[14], direction);
1554
1529
  atomicStore(&counters[24], path);
1555
1530
  } else if (P.role == 1u) { // the edge queue is filled
1556
- if (args[4u * P.slotBase] == 0u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to size, nothing to count
1557
- zero_slot(1u); zero_slot(6u);
1531
+ if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count
1558
1532
  return;
1559
1533
  }
1560
1534
  let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);
1561
1535
  atomicStore(&counters[8], clamped);
1562
1536
  if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry
1563
- let entries = atomicLoad(&counters[0]);
1564
- zero_slot(1u); write_slot_groups(6u, entries, entries); // one workgroup per frontier entry, as slot 2
1565
- atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry, bfs-contract nothing
1537
+ atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing
1566
1538
  atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);
1567
1539
  atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
1568
1540
  } else {
1569
- write_slot(1u, clamped); zero_slot(6u);
1570
1541
  atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)
1571
1542
  }
1572
1543
  } else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes
@@ -1574,54 +1545,43 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
1574
1545
  atomicStore(&counters[9], 0u);
1575
1546
  atomicStore(&counters[24], 0u);
1576
1547
  if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)
1577
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
1578
1548
  return;
1579
1549
  }
1580
1550
  let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped
1581
1551
  let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped
1582
1552
  if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE
1583
1553
  atomicStore(&counters[15], 2u);
1584
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
1585
1554
  return;
1586
1555
  }
1587
- zero_slot(2u); zero_slot(5u); zero_slot(6u); // role 3 sizes the relax slots once the piles are deduped
1588
1556
  if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn
1589
1557
  atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter
1590
1558
  atomicStore(&counters[14], 0u); // mode 0
1591
- write_slot(0u, nearRaw); write_slot(1u, nearRaw); // dedupe-claim, dedupe-filter over the near half
1592
- atomicStore(&counters[8], nearRaw); // the near dedupe's count word
1559
+ atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)
1593
1560
  atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing
1594
- zero_slot(3u); zero_slot(4u);
1595
1561
  atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)
1596
1562
  } else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile
1597
1563
  let threshold = bitcast<f32>(atomicLoad(&counters[22]));
1598
1564
  let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)
1599
1565
  if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED
1600
1566
  atomicStore(&counters[15], 3u);
1601
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
1602
1567
  return;
1603
1568
  }
1604
1569
  atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below
1605
1570
  atomicStore(&counters[22], bitcast<u32>(raised));
1606
1571
  atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter
1607
1572
  atomicStore(&counters[14], 1u); // mode 1
1608
- zero_slot(0u); zero_slot(1u);
1609
- write_slot(3u, farRaw); write_slot(4u, farRaw); // dedupe-claim, dedupe-filter over the far half
1610
- atomicStore(&counters[9], farRaw); // the far dedupe's count word
1573
+ atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)
1611
1574
  atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing
1612
1575
  atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);
1613
1576
  } else { // both piles empty: finished
1614
1577
  atomicStore(&counters[15], 1u);
1615
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
1616
1578
  }
1617
- } else if (P.role == 3u) { // the piles are deduped: size the relax
1618
- if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 zeroed every slot of the round
1579
+ } else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed
1580
+ if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round
1619
1581
  if (atomicLoad(&counters[14]) == 0u) {
1620
- write_slot(2u, atomicLoad(&counters[0])); // the near round over nearIn
1621
- atomicStore(&counters[1], 0u); // the raw near half restarts
1582
+ atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)
1622
1583
  } else {
1623
- write_slot(5u, atomicLoad(&counters[20])); // the pass-through over farIn
1624
- atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far)
1584
+ atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)
1625
1585
  }
1626
1586
  }
1627
1587
  }
@@ -2809,7 +2769,6 @@ const FRONTIER_COUNTERS = UniformBlock.define(
2809
2769
  );
2810
2770
  const FRONTIER_PARAMS = UniformBlock.define("FrontierParams", [
2811
2771
  ["role", "u32"],
2812
- ["slotBase", "u32"],
2813
2772
  ["wg", "u32"],
2814
2773
  ["alpha", "u32"],
2815
2774
  ["beta", "u32"],
@@ -2827,7 +2786,8 @@ const FRONTIER_PARAMS = UniformBlock.define("FrontierParams", [
2827
2786
  ["stride", "u32"],
2828
2787
  ["firstOfSubmit", "u32"],
2829
2788
  ["iteration", "u32"],
2830
- ["pad1", "u32"]
2789
+ ["pad1", "u32"],
2790
+ ["pad2", "u32"]
2831
2791
  ]);
2832
2792
  const BF_PARAMS = UniformBlock.define("BfParams", [
2833
2793
  ["edgeCount", "u32"],
@@ -3406,11 +3366,7 @@ const FRONTIER_FINALIZE = {
3406
3366
  id: "frontier-finalize",
3407
3367
  body: frontierFinalizeWgsl,
3408
3368
  entryPoint: "frontier_finalize",
3409
- bindings: [
3410
- decl(1, 0, "counters", "storage", "array<atomic<u32>>"),
3411
- decl(1, 1, "args", "storage", "array<u32>"),
3412
- decl(2, 0, "P", "uniform", "FrontierParams")
3413
- ],
3369
+ bindings: [decl(1, 0, "counters", "storage", "array<atomic<u32>>"), decl(2, 0, "P", "uniform", "FrontierParams")],
3414
3370
  overrideDecls: [],
3415
3371
  uniforms: [FRONTIER_PARAMS],
3416
3372
  needs: [],
@@ -4486,11 +4442,6 @@ function algorithmScope(ctx, label, slots) {
4486
4442
  pool: ctx.pool,
4487
4443
  workgroupSize: ctx.workgroupSize,
4488
4444
  scratch: (byteLength, scratchLabel) => lease.storage(byteLength, `${label}/${scratchLabel}`),
4489
- indirect: (byteLength, indirectLabel) => lease.acquire(
4490
- byteLength,
4491
- BufferUsage.STORAGE | BufferUsage.INDIRECT | BufferUsage.COPY_DST | BufferUsage.COPY_SRC,
4492
- `${label}/${indirectLabel}`
4493
- ),
4494
4445
  params(block, values) {
4495
4446
  const slot = ring.reserve(1);
4496
4447
  ring.write(slot, block, values);
@@ -5822,16 +5773,14 @@ class Frontier {
5822
5773
  * Wraps the leased buffers; use prepareFrontier().
5823
5774
  * @param vertices - the two vertex queues
5824
5775
  * @param counters - the counters block
5825
- * @param args - the args buffer
5826
5776
  * @param edgeQueue - the edge queue
5827
5777
  * @param edgeCapacity - the edge queue's entry count
5828
5778
  * @param n - the vertex count
5829
5779
  */
5830
- constructor(vertices, counters, args, edgeQueue, edgeCapacity, n) {
5780
+ constructor(vertices, counters, edgeQueue, edgeCapacity, n) {
5831
5781
  this.sideIndex = 0;
5832
5782
  this.vertices = vertices;
5833
5783
  this.counters = counters;
5834
- this.args = args;
5835
5784
  this.edgeQueue = edgeQueue;
5836
5785
  this.edgeCapacity = edgeCapacity;
5837
5786
  this.n = n;
@@ -5862,7 +5811,7 @@ class Frontier {
5862
5811
  this.sideIndex = this.sideIndex === 0 ? 1 : 0;
5863
5812
  }
5864
5813
  /**
5865
- * Seeds a traversal: one `queue.writeBuffer` of the whole 96-byte block (zero except the caller's words) and one of
5814
+ * Seeds a traversal: one `queue.writeBuffer` of the whole 112-byte block (zero except the caller's words) and one of
5866
5815
  * `vertices[0][0] = source`, both ordered before the submit that follows; the source is on side 0 afterwards.
5867
5816
  * `frontierCount` is not a word to seed: the first boundary rotates word 1 into it (the BFS seed is
5868
5817
  * `{ nextFrontierCount: 1, level: U32_MAX }`). A source outside `[0, n)`, an unknown word or a value that is not a
@@ -5899,7 +5848,6 @@ async function prepareFrontier(scope, n, arcCount, edgeCapacity) {
5899
5848
  }
5900
5849
  const kernel = await scope.pipelines.kernel(kernelSpec("frontier-finalize"));
5901
5850
  const queueBytes = 4 * Math.max(1, n);
5902
- const argsBytes = MAX_LEVELS_PER_SUBMIT * FRONTIER_CANDIDATES * INDIRECT_ARGS_STRIDE;
5903
5851
  const vertices = [
5904
5852
  { buffer: scope.scratch(queueBytes, "frontier/vertices-0"), offset: 0, size: queueBytes, window: null },
5905
5853
  { buffer: scope.scratch(queueBytes, "frontier/vertices-1"), offset: 0, size: queueBytes, window: null }
@@ -5910,19 +5858,13 @@ async function prepareFrontier(scope, n, arcCount, edgeCapacity) {
5910
5858
  size: FRONTIER_COUNTERS.byteLength,
5911
5859
  window: null
5912
5860
  };
5913
- const args = {
5914
- buffer: scope.indirect(argsBytes, "frontier/args"),
5915
- offset: 0,
5916
- size: argsBytes,
5917
- window: null
5918
- };
5919
5861
  const edgeQueue = {
5920
5862
  buffer: scope.scratch(4 * capacity, "frontier/edge-queue"),
5921
5863
  offset: 0,
5922
5864
  size: 4 * capacity,
5923
5865
  window: null
5924
5866
  };
5925
- const frontier = new Frontier(vertices, counters, args, edgeQueue, capacity, n);
5867
+ const frontier = new Frontier(vertices, counters, edgeQueue, capacity, n);
5926
5868
  return new FrontierPlannerImpl(scope, kernel, frontier);
5927
5869
  }
5928
5870
  class FrontierPlannerImpl {
@@ -5963,12 +5905,11 @@ class FrontierPlannerImpl {
5963
5905
  const params = scope.params(FRONTIER_PARAMS, {
5964
5906
  ...definedWords(fields),
5965
5907
  role,
5966
- slotBase: level * FRONTIER_CANDIDATES,
5967
5908
  wg: scope.workgroupSize,
5968
5909
  edgeCapacity: frontier.edgeCapacity,
5969
5910
  n: frontier.n
5970
5911
  });
5971
- const bound = this.kernel.bind({ counters: frontier.counters, args: frontier.args, P: params.binding });
5912
+ const bound = this.kernel.bind({ counters: frontier.counters, P: params.binding });
5972
5913
  this.kernel.dispatch(pass, bound, plan1d(1, scope.workgroupSize, scope.caps), [params.offset]);
5973
5914
  }
5974
5915
  }
@@ -12653,7 +12594,7 @@ async function calibrateLayout(ctx, options) {
12653
12594
  };
12654
12595
  }
12655
12596
  export {
12656
- C as ARC_WINDOW_ALIGN,
12597
+ A as ARC_WINDOW_ALIGN,
12657
12598
  EXACT_MAX_NODES,
12658
12599
  FA2_DEFAULTS,
12659
12600
  FR_DEFAULTS,
@@ -12661,10 +12602,10 @@ export {
12661
12602
  LAYOUT_TUNING_DEFAULTS,
12662
12603
  MAX_1D_ITEMS,
12663
12604
  MAX_WORKGROUPS_PER_DIM,
12664
- D as PASSTHROUGH_FORMAT_CODES,
12605
+ C as PASSTHROUGH_FORMAT_CODES,
12665
12606
  SE_DEFAULTS,
12666
- H as STORAGE_ALIGN,
12667
- J as WORKGROUP_SIZE,
12607
+ D as STORAGE_ALIGN,
12608
+ H as WORKGROUP_SIZE,
12668
12609
  WebGpuGraphError,
12669
12610
  bellmanFord,
12670
12611
  breadthFirstSearch,
@@ -12679,7 +12620,7 @@ export {
12679
12620
  eigenvectorCentrality,
12680
12621
  hasErrorCode,
12681
12622
  hits,
12682
- K as isSoftwareAdapter,
12623
+ J as isSoftwareAdapter,
12683
12624
  isWebGpuGraphError,
12684
12625
  katzCentrality,
12685
12626
  pageRank,