@graphty/webgpu-graph-algorithms 0.6.2 → 0.6.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. package/README.md +62 -32
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-BXqgCifx.js → context-Dvq-Cc6v.js} +71 -25
  4. package/dist/chunks/context-Dvq-Cc6v.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/accelerator.d.ts +8 -6
  7. package/dist/src/accelerator.d.ts.map +1 -1
  8. package/dist/src/accelerator.js +57 -6
  9. package/dist/src/accelerator.js.map +1 -1
  10. package/dist/src/algorithms/bellman-ford.d.ts +60 -0
  11. package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
  12. package/dist/src/algorithms/bellman-ford.js +301 -0
  13. package/dist/src/algorithms/bellman-ford.js.map +1 -0
  14. package/dist/src/algorithms/bfs.d.ts +67 -0
  15. package/dist/src/algorithms/bfs.d.ts.map +1 -0
  16. package/dist/src/algorithms/bfs.js +534 -0
  17. package/dist/src/algorithms/bfs.js.map +1 -0
  18. package/dist/src/algorithms/closeness.d.ts +53 -0
  19. package/dist/src/algorithms/closeness.d.ts.map +1 -0
  20. package/dist/src/algorithms/closeness.js +323 -0
  21. package/dist/src/algorithms/closeness.js.map +1 -0
  22. package/dist/src/algorithms/scope.d.ts +5 -3
  23. package/dist/src/algorithms/scope.d.ts.map +1 -1
  24. package/dist/src/algorithms/scope.js +3 -0
  25. package/dist/src/algorithms/scope.js.map +1 -1
  26. package/dist/src/algorithms/sssp.d.ts +71 -0
  27. package/dist/src/algorithms/sssp.d.ts.map +1 -0
  28. package/dist/src/algorithms/sssp.js +585 -0
  29. package/dist/src/algorithms/sssp.js.map +1 -0
  30. package/dist/src/constants.d.ts +12 -0
  31. package/dist/src/constants.d.ts.map +1 -1
  32. package/dist/src/constants.js +12 -0
  33. package/dist/src/constants.js.map +1 -1
  34. package/dist/src/index.d.ts +8 -2
  35. package/dist/src/index.d.ts.map +1 -1
  36. package/dist/src/index.js +7 -1
  37. package/dist/src/index.js.map +1 -1
  38. package/dist/src/kernel/prelude.d.ts +4 -4
  39. package/dist/src/kernel/prelude.d.ts.map +1 -1
  40. package/dist/src/kernel/prelude.js +39 -5
  41. package/dist/src/kernel/prelude.js.map +1 -1
  42. package/dist/src/kernel/uniform-ring.d.ts +8 -0
  43. package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
  44. package/dist/src/kernel/uniform-ring.js +13 -0
  45. package/dist/src/kernel/uniform-ring.js.map +1 -1
  46. package/dist/src/kernels.d.ts +44 -4
  47. package/dist/src/kernels.d.ts.map +1 -1
  48. package/dist/src/kernels.js +371 -3
  49. package/dist/src/kernels.js.map +1 -1
  50. package/dist/src/primitives/advance.d.ts +62 -0
  51. package/dist/src/primitives/advance.d.ts.map +1 -0
  52. package/dist/src/primitives/advance.js +95 -0
  53. package/dist/src/primitives/advance.js.map +1 -0
  54. package/dist/src/primitives/compact.d.ts +89 -0
  55. package/dist/src/primitives/compact.d.ts.map +1 -0
  56. package/dist/src/primitives/compact.js +233 -0
  57. package/dist/src/primitives/compact.js.map +1 -0
  58. package/dist/src/primitives/core-shape.d.ts +22 -1
  59. package/dist/src/primitives/core-shape.d.ts.map +1 -1
  60. package/dist/src/primitives/core-shape.js +33 -3
  61. package/dist/src/primitives/core-shape.js.map +1 -1
  62. package/dist/src/primitives/frontier.d.ts +156 -0
  63. package/dist/src/primitives/frontier.d.ts.map +1 -0
  64. package/dist/src/primitives/frontier.js +259 -0
  65. package/dist/src/primitives/frontier.js.map +1 -0
  66. package/dist/src/types/accelerator.d.ts +16 -7
  67. package/dist/src/types/accelerator.d.ts.map +1 -1
  68. package/dist/src/types/traversal.d.ts +53 -0
  69. package/dist/src/types/traversal.d.ts.map +1 -0
  70. package/dist/src/types/traversal.js +10 -0
  71. package/dist/src/types/traversal.js.map +1 -0
  72. package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
  73. package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
  74. package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
  75. package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
  76. package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
  77. package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
  78. package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
  79. package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
  80. package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
  81. package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
  82. package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
  83. package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
  84. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
  85. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
  86. package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
  87. package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
  88. package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
  89. package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
  90. package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
  91. package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
  92. package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
  93. package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
  94. package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
  95. package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
  96. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
  97. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
  98. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
  99. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
  100. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
  101. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
  102. package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
  103. package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
  104. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
  105. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
  106. package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
  107. package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
  108. package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
  109. package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
  110. package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
  111. package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
  112. package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
  113. package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
  114. package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
  115. package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
  116. package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
  117. package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
  118. package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
  119. package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
  120. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +59 -0
  121. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
  122. package/dist/src/wgsl/frontier-finalize.wgsl.js +210 -0
  123. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
  124. package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
  125. package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
  126. package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
  127. package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
  128. package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
  129. package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
  130. package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
  131. package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
  132. package/dist/webgpu-graph-algorithms.js +3207 -377
  133. package/dist/webgpu-graph-algorithms.js.map +1 -1
  134. package/package.json +5 -4
  135. package/src/accelerator.ts +65 -7
  136. package/src/algorithms/bellman-ford.ts +387 -0
  137. package/src/algorithms/bfs.ts +626 -0
  138. package/src/algorithms/closeness.ts +395 -0
  139. package/src/algorithms/scope.ts +13 -3
  140. package/src/algorithms/sssp.ts +767 -0
  141. package/src/constants.ts +12 -0
  142. package/src/index.ts +14 -1
  143. package/src/kernel/prelude.ts +39 -4
  144. package/src/kernel/uniform-ring.ts +14 -0
  145. package/src/kernels.ts +450 -6
  146. package/src/primitives/advance.ts +130 -0
  147. package/src/primitives/compact.ts +323 -0
  148. package/src/primitives/core-shape.ts +41 -3
  149. package/src/primitives/frontier.ts +388 -0
  150. package/src/types/accelerator.ts +18 -5
  151. package/src/types/traversal.ts +56 -0
  152. package/src/wgsl/advance-expand.wgsl.ts +68 -0
  153. package/src/wgsl/bf-relax.wgsl.ts +57 -0
  154. package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
  155. package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
  156. package/src/wgsl/bfs-contract.wgsl.ts +54 -0
  157. package/src/wgsl/bfs-fused.wgsl.ts +77 -0
  158. package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
  159. package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
  160. package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
  161. package/src/wgsl/compact-scatter.wgsl.ts +16 -0
  162. package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
  163. package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
  164. package/src/wgsl/frontier-finalize.wgsl.ts +209 -0
  165. package/src/wgsl/sssp-pred.wgsl.ts +79 -0
  166. package/src/wgsl/sssp-relax.wgsl.ts +71 -0
  167. package/dist/chunks/context-BXqgCifx.js.map +0 -1
@@ -0,0 +1,209 @@
1
+ /**
2
+ * The `frontier-finalize` kernel body (design 5.4, 6 row 7; P8-T4, the P8 plan's PD-3 / PD-23 / DEP-P8-C): the
3
+ * device-side selector of the frontier family. One workgroup, one lane, no barrier after the early return (spec 3.5
4
+ * rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's dispatch sizes become
5
+ * known at two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`,
6
+ * advances `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`), chooses the path and writes this
7
+ * level's seven 16-byte slots at `P.slotBase`; role 1 runs once the edge queue is filled -- clamps `edgeCount` to
8
+ * `P.edgeCapacity`, sizes the contract slot, or, when `edgeCountUnclamped` exceeds the capacity, zeroes it and sizes
9
+ * the fused-retry slot from `frontierCount` instead (PD-23). Two rules: a boundary that finds `done` set zeroes its
10
+ * slots and moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when
11
+ * role 0 chose one, which it reads from slot 0's `x` (a storage write of one dispatch is visible to the next of the
12
+ * same pass). The `(x, y)` arithmetic is `indirect-finalize`'s verbatim (P4), so a count above 2^32 - wg cannot wrap
13
+ * and a group count above MAX_WORKGROUPS_PER_DIM splits in 2D; slots 2 and 6 are sized one WORKGROUP per entry for
14
+ * `bfs-fused`; slots 3, 4 and 5 (the bits fill over `ceil(n / 32)` words, the bitset build over the frontier, the sweep
15
+ * over the unvisited list) are the bottom-up level's.
16
+ *
17
+ * Roles 2 and 3 are the SSSP round boundary of the near-far loop (P8-T9, PD-20), over the same block read in its
18
+ * SSSP sense (word 1 the raw near half's appends, 21 the raw far half's, 0 and 20 the deduped pile counts, 22 the
19
+ * threshold, 23 the delta, 4 the previous threshold, 14 the round's mode) and the same seven slots: role 2 finds a
20
+ * non-empty raw near half and sizes the near dedupe (slots 0 and 1, count word 1, output word 0) in mode 0; finds it
21
+ * empty and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
22
+ * unchanged sets `done 3`, the host's E_UNSUPPORTED), remembers the previous threshold in word 4 and sizes the far
23
+ * dedupe (slots 3 and 4, count word 21, output word 20) in mode 1; finds both empty and sets `done 1`; and finds a
24
+ * raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in `level` when it
25
+ * picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds dispatched; a
26
+ * boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed: mode 0 sizes slot 2 (the
27
+ * relax over nearIn) from word 0 and restarts the raw near half (word 1 to 0); mode 1 sizes slot 5 (the pass-through
28
+ * over farIn) from word 20 and restarts the raw far half (word 21 to 0).
29
+ *
30
+ * Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a switch can be counted at
31
+ * the done boundary too, which the host model of the tests mirrors): top-down switches to bottom-up when
32
+ * `frontierDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's `max(1, floor(arcCount / n))`
33
+ * unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up switches back when
34
+ * `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no admitted device
35
+ * reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the direction at 0. Every
36
+ * change is counted in `switches`, the previous direction is word 14. The two unvisited words the test reads are
37
+ * rebuilt exactly once per submit by `bfs-unvisited-flags` (PD-18) and maintained here by subtraction: the count is
38
+ * subtracted from the SECOND boundary of a submit on and the degree sum from the THIRD on, because a boundary may
39
+ * only subtract what the submit's rebuild counted, and the frontier whose degree sum the second boundary holds was
40
+ * claimed before the rebuild ran (the rebuild counts the vertices unclaimed when it runs; the frontier rotated in at
41
+ * boundary 0 was claimed by the previous submit's last contract, so it was never in the sum; boundary b subtracts
42
+ * `next = |F_b|`, inside the sum iff b >= 1, and `degSum = deg(F_{b-1})`, inside it iff b >= 2). The degree sum is
43
+ * the "unvisited degree estimate" of the design rather than an exact count for two reasons: it is one level stale
44
+ * (a frontier's degree sum is only known once it has been expanded), and a bottom-up level expands nothing, so the
45
+ * word stops falling while bottom-up runs and overstates the set afterwards. The bias is one-directional -- an
46
+ * overstated m_u makes the switch INTO bottom-up harder, never easier -- and the next submit's rebuild makes it
47
+ * exact again. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
48
+ * textual edits of it.
49
+ *
50
+ * Since 2026-09-25 nothing dispatches FROM the slots (G8-F5: Dawn's validation of an indirect dispatch cost about
51
+ * 0.4 ms of device time each, whether or not it dispatched anything, and the seven slots of thirty-two recorded
52
+ * levels were 97 % of a traversal's wall time). Every level kernel is a direct grid-stride dispatch that reads the
53
+ * `path` word (24) this kernel writes -- 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3
54
+ * bottom-up, 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2) -- and the SSSP dedupes read
55
+ * their counts from words 8 and 9, which role 2 writes. The slots stay as the selector's recorded decision, read
56
+ * back by the frontier tests; deleting them with those tests is the follow-up.
57
+ */
58
+ export const frontierFinalizeWgsl = /* wgsl */ `
59
+ fn write_slot_groups(slot: u32, groups: u32, count: u32) { // groups workgroups, split in 2D above the per-dim limit
60
+ var x = groups;
61
+ var y = 1u;
62
+ if (groups > MAX_WORKGROUPS_PER_DIM) {
63
+ x = MAX_WORKGROUPS_PER_DIM;
64
+ y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;
65
+ }
66
+ let base = 4u * (P.slotBase + slot); // 16-byte slots: (x, y, 1, count)
67
+ args[base] = x; args[base + 1u] = y; args[base + 2u] = 1u; args[base + 3u] = count;
68
+ }
69
+ fn write_slot(slot: u32, count: u32) { // one INVOCATION per entry: ceil(count / wg) workgroups
70
+ let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap (indirect-finalize's rule)
71
+ write_slot_groups(slot, groups, count);
72
+ }
73
+ fn zero_slot(slot: u32) {
74
+ let base = 4u * (P.slotBase + slot);
75
+ args[base] = 0u; args[base + 1u] = 0u; args[base + 2u] = 1u; args[base + 3u] = 0u;
76
+ }
77
+
78
+ @compute @workgroup_size(WG)
79
+ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
80
+ if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
81
+ if (P.role == 0u) { // the level boundary
82
+ if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end
83
+ for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); } // this slotBase holds the previous submit's args
84
+ return; // no counter word moves (P8-T6's levels formula reads them)
85
+ }
86
+ let finished = atomicLoad(&counters[0]);
87
+ let next = atomicLoad(&counters[1]);
88
+ let degSum = atomicLoad(&counters[2]);
89
+ atomicStore(&counters[3], finished); // prevFrontierCount
90
+ atomicStore(&counters[4], degSum); // prevDegreeSum
91
+ atomicStore(&counters[0], next); // the rotation
92
+ atomicStore(&counters[1], 0u);
93
+ atomicStore(&counters[2], 0u);
94
+ atomicStore(&counters[8], 0u); // edgeCount
95
+ atomicStore(&counters[9], 0u); // edgeCountUnclamped
96
+ atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount
97
+ if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped to 2 (P8-T8, PD-18)
98
+ atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount (exact): F_b was inside the submit's rebuilt sum iff b >= 1
99
+ }
100
+ if (P.firstOfSubmit >= 2u) {
101
+ atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2
102
+ }
103
+ let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0
104
+ atomicStore(&counters[11], level);
105
+ let done = (next == 0u) || (level >= P.maxDepth);
106
+ atomicStore(&counters[15], select(0u, 1u, done));
107
+ var direction = atomicLoad(&counters[14]);
108
+ if (P.mode == 1u) {
109
+ direction = 0u; // top-down only (the test seam)
110
+ } else if (direction == 0u) {
111
+ if (degSum > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing
112
+ } else {
113
+ if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking
114
+ }
115
+ if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches
116
+ var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)
117
+ if (done) {
118
+ for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
119
+ } else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep
120
+ zero_slot(0u); zero_slot(1u); zero_slot(2u); zero_slot(6u);
121
+ write_slot(3u, (P.n + 31u) / 32u); write_slot(4u, next); write_slot(5u, atomicLoad(&counters[7]));
122
+ path = 3u;
123
+ atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);
124
+ } else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)
125
+ zero_slot(0u); zero_slot(1u); zero_slot(3u); zero_slot(4u); zero_slot(5u); zero_slot(6u);
126
+ write_slot_groups(2u, next, next); // bfs-fused is one WORKGROUP per frontier entry
127
+ path = 2u;
128
+ atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
129
+ } else {
130
+ zero_slot(2u); zero_slot(3u); zero_slot(4u); zero_slot(5u);
131
+ write_slot(0u, next); // slots 1 and 6 are role 1's
132
+ path = 1u;
133
+ }
134
+ atomicStore(&counters[14], direction);
135
+ atomicStore(&counters[24], path);
136
+ } else if (P.role == 1u) { // the edge queue is filled
137
+ if (args[4u * P.slotBase] == 0u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to size, nothing to count
138
+ zero_slot(1u); zero_slot(6u);
139
+ return;
140
+ }
141
+ let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);
142
+ atomicStore(&counters[8], clamped);
143
+ if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry
144
+ let entries = atomicLoad(&counters[0]);
145
+ zero_slot(1u); write_slot_groups(6u, entries, entries); // one workgroup per frontier entry, as slot 2
146
+ atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry, bfs-contract nothing
147
+ atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);
148
+ atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
149
+ } else {
150
+ write_slot(1u, clamped); zero_slot(6u);
151
+ atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)
152
+ }
153
+ } else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes
154
+ atomicStore(&counters[8], 0u); // the dedupe counts (words 8 and 9, the SSSP sense) and the path word: nothing unless a pile is chosen below
155
+ atomicStore(&counters[9], 0u);
156
+ atomicStore(&counters[24], 0u);
157
+ if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)
158
+ for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
159
+ return;
160
+ }
161
+ let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped
162
+ let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped
163
+ if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE
164
+ atomicStore(&counters[15], 2u);
165
+ for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
166
+ return;
167
+ }
168
+ zero_slot(2u); zero_slot(5u); zero_slot(6u); // role 3 sizes the relax slots once the piles are deduped
169
+ if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn
170
+ atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter
171
+ atomicStore(&counters[14], 0u); // mode 0
172
+ write_slot(0u, nearRaw); write_slot(1u, nearRaw); // dedupe-claim, dedupe-filter over the near half
173
+ atomicStore(&counters[8], nearRaw); // the near dedupe's count word
174
+ atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing
175
+ zero_slot(3u); zero_slot(4u);
176
+ atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)
177
+ } else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile
178
+ let threshold = bitcast<f32>(atomicLoad(&counters[22]));
179
+ let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)
180
+ if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED
181
+ atomicStore(&counters[15], 3u);
182
+ for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
183
+ return;
184
+ }
185
+ atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below
186
+ atomicStore(&counters[22], bitcast<u32>(raised));
187
+ atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter
188
+ atomicStore(&counters[14], 1u); // mode 1
189
+ zero_slot(0u); zero_slot(1u);
190
+ write_slot(3u, farRaw); write_slot(4u, farRaw); // dedupe-claim, dedupe-filter over the far half
191
+ atomicStore(&counters[9], farRaw); // the far dedupe's count word
192
+ atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing
193
+ atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);
194
+ } else { // both piles empty: finished
195
+ atomicStore(&counters[15], 1u);
196
+ for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
197
+ }
198
+ } else if (P.role == 3u) { // the piles are deduped: size the relax
199
+ if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 zeroed every slot of the round
200
+ if (atomicLoad(&counters[14]) == 0u) {
201
+ write_slot(2u, atomicLoad(&counters[0])); // the near round over nearIn
202
+ atomicStore(&counters[1], 0u); // the raw near half restarts
203
+ } else {
204
+ write_slot(5u, atomicLoad(&counters[20])); // the pass-through over farIn
205
+ atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far)
206
+ }
207
+ }
208
+ }
209
+ `;
@@ -0,0 +1,79 @@
1
+ /**
2
+ * The `sssp-pred` kernel body (design 8.10 "SSSP predecessor pass"; P8-T6 and P8-T9, the P8 plan's PD-11 / PD-24 /
3
+ * PD-27): the ONE post-pass over the settled distances that produces `parent` for `breadthFirstSearch` and
4
+ * `predArc` for `sssp` and `bellmanFord`, so no claim kernel writes a predecessor and every predecessor array is a
5
+ * function of the settled `depth` / `dist` alone (bitwise reproducible, PD-14). `MODE` (a pipeline override) picks
6
+ * the distance type: 1 = u32 depths (`dist` is the BFS depth array, a tight arc is `depth[u] + 1 == depth[v]`, every
7
+ * tight arc is admitted, and the chain is acyclic because depth strictly decreases along it); 0 = f32 bit patterns
8
+ * (a tight arc is `bitcast<u32>(bitcast<f32>(dist[u]) + w) == dist[v]`, one f32 add compared as the bits PD-9
9
+ * stores). `P.predKind` picks what is written: 0 the arc index, 1 the source node index. The winner is the SMALLEST
10
+ * admitted value through `atomicMin` on `pred[v]`, which the driver fills with `INVALID_INDEX` first; the source
11
+ * keeps `INVALID_INDEX` whatever attains it.
12
+ *
13
+ * In `MODE 0` the body runs PD-27's three roles over a `pred` buffer of `2 x hb + 64` words, `hb = roundUp(n, 64)`:
14
+ * `pred[0, n)` the arcs, `pred[hb, hb + n)` the hop counts, `pred[2 hb]` the changed word, `pred[2 hb + 1]` the orphan
15
+ * word. Role 0 (the roots pass, `P.mode 0` only) marks every node with a tight in-arc from a strictly smaller
16
+ * distance as a root (`hops 0`); role 1 (a hop pass, `P.iteration` inside its batch) lowers `hops[v]` to
17
+ * `hops[u] + 1` along every plateau arc (`P.mode 0`: equal distance) or every tight arc (`P.mode 1`: bellmanFord's
18
+ * tight-subgraph rule) and records the pass index in the changed word when it lowered one, returning at its first
19
+ * line when the previous pass changed nothing; role 2 (the predecessor pass) admits the smallest tight arc whose
20
+ * source sits exactly one key step below `v` and counts a reached non-source node the key never reached as an
21
+ * orphan (only bellmanFord can make one). `MODE 1` reads none of `P.role`, `P.mode` and `P.iteration` and never
22
+ * touches `pred` beyond word `n - 1`. The body STRIDES over the rows (`planGridStride(n)`, `P.stride` the plan's
23
+ * stride): a per-invocation body under that plan would leave every row above the dispatch cap at `INVALID_INDEX`,
24
+ * silently. No barrier anywhere, so the early return is legal. Body only (spec 3.5, D9); the text is normative: the
25
+ * sabotage rows of test/helpers/sabotage.ts are textual edits of it.
26
+ */
27
+ export const ssspPredWgsl = /* wgsl */ `
28
+ @compute @workgroup_size(WG)
29
+ fn sssp_pred(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
30
+ let first = linear_id(wid, lid.x);
31
+ let hb = ((P.n + 63u) / 64u) * 64u; // MODE 0 only: hops[v] is pred[hb + v]; pred[2 hb] is the changed word, pred[2 hb + 1] the orphan word (PD-27)
32
+ if (P.role == 1u && P.iteration != 0u && atomicLoad(&pred[2u * hb]) < P.iteration) { return; } // the previous hop pass changed nothing: converged (no barrier anywhere, so the early return is legal)
33
+ for (var u = first; u < P.n; u = u + P.stride) { // grid-stride over the rows (planGridStride(n)); no barrier anywhere
34
+ let du = dist[u];
35
+ var unreached = false;
36
+ if (MODE == 1u) { unreached = du == INVALID_INDEX; } else { unreached = du == F32_INF_BITS; }
37
+ if (unreached) { continue; }
38
+ var hu = 0u; // u's hop count (MODE 0, roles 1 and 2)
39
+ if (MODE == 0u && P.role != 0u) {
40
+ hu = atomicLoad(&pred[hb + u]);
41
+ if (hu == INVALID_INDEX) { // the key has not reached u yet
42
+ if (P.role == 2u && u != P.source) { atomicAdd(&pred[2u * hb + 1u], 1u); } // an orphan: only bellmanFord can make one (P8-T10 Step 3)
43
+ continue;
44
+ }
45
+ }
46
+ let end = min(rowPtr[u + 1u], P.arcEnd);
47
+ for (var a = max(rowPtr[u], P.arcBase); a < end; a = a + 1u) {
48
+ let v = colIdx[a - P.arcBase];
49
+ if (v == P.source) { continue; } // the source keeps INVALID_INDEX whatever attains it (a zero-weight arc could)
50
+ let dv = dist[v];
51
+ var tight = false; // the arc explains dist[v]
52
+ var below = false; // and its source sits at a strictly smaller distance
53
+ if (MODE == 1u) {
54
+ tight = (du + 1u) == dv; // depth mode: BFS parent, one depth down
55
+ } else {
56
+ let w = select(1.0, weights[a - P.arcBase], HAS_WEIGHTS);
57
+ tight = (dv != F32_INF_BITS) && (bitcast<u32>(bitcast<f32>(du) + w) == dv); // one f32 add, compared as the bit pattern PD-9 stores; never into an unreached v (an overflowed sum is +Inf too)
58
+ below = bitcast<f32>(du) < bitcast<f32>(dv);
59
+ }
60
+ if (!tight) { continue; }
61
+ var admit = false; // this arc is one key step below v
62
+ if (MODE == 1u) {
63
+ admit = true;
64
+ } else if (P.role == 0u) { // the roots pass (the plateau rule only; bellmanFord seeds the source alone)
65
+ if (P.mode == 0u && below) { atomicMin(&pred[hb + v], 0u); }
66
+ } else if (P.role == 1u) { // a hop pass: one hop along a plateau arc (mode 0) or along any tight arc (mode 1)
67
+ if (P.mode == 1u || du == dv) {
68
+ let old = atomicMin(&pred[hb + v], hu + 1u);
69
+ if (hu + 1u < old) { atomicMax(&pred[2u * hb], P.iteration + 1u); } // this pass changed something
70
+ }
71
+ } else { // the predecessor pass: the smallest tight arc one key step below v
72
+ let hv = atomicLoad(&pred[hb + v]);
73
+ if (P.mode == 1u) { admit = hu + 1u == hv; } else if (hv == 0u) { admit = below; } else { admit = (du == dv) && (hu + 1u == hv); }
74
+ }
75
+ if (admit) { atomicMin(&pred[v], select(a, u, P.predKind == 1u)); }
76
+ }
77
+ }
78
+ }
79
+ `;
@@ -0,0 +1,71 @@
1
+ /**
2
+ * The `sssp-relax` kernel body (design 8.4 "Davidson's near-far", 8.10 "SSSP near-far relax"; P8-T9, the P8 plan's
3
+ * PD-9 / PD-20 / PD-22 / DEP-P8-E): one round of the near-far shortest-path loop. `dist` is `array<atomic<u32>>`
4
+ * holding the IEEE-754 bit patterns of the f32 distances (`F32_INF_BITS` = unreached), and `atomicMin` on the
5
+ * patterns IS `min` on the values, because for non-negative floats the unsigned bit order is the value order
6
+ * (`+0` is `0x00000000`, every sum of non-negatives under round-to-nearest is `+0` or positive, never `-0`), which
7
+ * the driver's host scan of the weight vector guarantees -- the whole trick, and it needs no float atomic (PD-9).
8
+ * Every candidate is ONE f32 add, `dist[u] + w`, so the settled value is the minimum of a fixed set of f32
9
+ * numbers: order-independent, bitwise reproducible, and bitwise equal to the f32 Dijkstra oracle.
10
+ *
11
+ * Two roles over one body, chosen by `P.role`. Role 0 (the near round) reads the deduped near pile `queueIn`
12
+ * (`counters[0]` entries, at most `P.n`), relaxes every arc of every entry's WHOLE row (never windowed, DEP-P8-E),
13
+ * skips a candidate above `P.cutoffBits` (the CPU port's `dv <= cutoff` guard as `nd > cutoff`), and when its
14
+ * `atomicMin` improved `v` -- this lane alone observed a larger old value, so this lane alone owns the append --
15
+ * appends `v` to the raw near half of `queueOut` (word 0, count word 1) when `nd` is below the threshold
16
+ * (`counters[22]`), else to the raw far half (word `P.edgeCapacity`, count word 21). Role 1 (the pass-through,
17
+ * PD-20) re-buckets the deduped far pile (`counters[20]` entries): an entry whose settled distance fell below the
18
+ * PREVIOUS threshold (`counters[4]`) was appended to near at that improvement and relaxed there, so it is dropped;
19
+ * the rest go back to near or far against the threshold the boundary just raised. The count words are unclamped
20
+ * (the write is guarded by the capacity; `frontier-finalize` role 2 detects a pile above it). The appends are per
21
+ * improving relaxation, one atomic each: inside a per-lane arc loop no workgroup aggregation is possible without the
22
+ * uniform-strip structure of `bfs-fused`, and Davidson's kernel appends per thread too. Grid-strided under a direct
23
+ * dispatch of `planGridStride(n)`: `P.stride` is the plan's, a deduped pile of at most `n` entries is covered in a
24
+ * few trips per lane and `i + stride` never wraps (`U32_MAX` would); the block's `path` word (5 a near round, 6 a
25
+ * far one) makes the other role's dispatch a no-op. No barrier anywhere: the loops may
26
+ * be per lane. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
27
+ * textual edits of it.
28
+ */
29
+ export const ssspRelaxWgsl = /* wgsl */ `
30
+ @compute @workgroup_size(WG)
31
+ fn sssp_relax(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
32
+ let first = linear_id(wid, lid.x);
33
+ let mine = select(5u, 6u, P.role == 1u); // the path word role 2 wrote: 5 a near round, 6 a far pass-through
34
+ let chosen = atomicLoad(&counters[24]) == mine; // the other role's dispatch of the round is a no-op
35
+ let count = select(0u, min(atomicLoad(&counters[select(0u, 20u, P.role == 1u)]), P.n), chosen); // the deduped near or far pile
36
+ let threshold = bitcast<f32>(atomicLoad(&counters[22]));
37
+ let cutoff = bitcast<f32>(P.cutoffBits);
38
+ for (var i = first; i < count; i = i + P.stride) { // no barrier anywhere: the loops may be per lane
39
+ let u = queueIn[i];
40
+ let du = bitcast<f32>(atomicLoad(&dist[u]));
41
+ if (P.role == 1u) { // the pass-through (PD-20): re-bucket a far entry
42
+ if (du < bitcast<f32>(atomicLoad(&counters[4]))) { continue; } // below the previous threshold: relaxed in an earlier bucket
43
+ if (du < threshold) {
44
+ let q = atomicAdd(&counters[1], 1u);
45
+ if (q < P.edgeCapacity) { queueOut[q] = u; }
46
+ } else {
47
+ let q = atomicAdd(&counters[21], 1u);
48
+ if (q < P.edgeCapacity) { queueOut[P.edgeCapacity + q] = u; }
49
+ }
50
+ continue;
51
+ }
52
+ let end = rowPtr[u + 1u];
53
+ for (var a = rowPtr[u]; a < end; a = a + 1u) { // the whole row: never windowed (DEP-P8-E)
54
+ let v = colIdx[a];
55
+ let nd = du + select(1.0, weights[a], HAS_WEIGHTS); // ONE f32 add (PD-9)
56
+ if (nd > cutoff) { continue; } // SsspOptions.cutoff: the CPU port's dv <= cutoff
57
+ let bits = bitcast<u32>(nd);
58
+ let old = atomicMin(&dist[v], bits); // exact on non-negative floats
59
+ if (bits < old) { // this lane improved v, so it owns the append
60
+ if (nd < threshold) {
61
+ let q = atomicAdd(&counters[1], 1u);
62
+ if (q < P.edgeCapacity) { queueOut[q] = v; }
63
+ } else {
64
+ let q = atomicAdd(&counters[21], 1u);
65
+ if (q < P.edgeCapacity) { queueOut[P.edgeCapacity + q] = v; }
66
+ }
67
+ }
68
+ }
69
+ }
70
+ }
71
+ `;