@graphty/webgpu-graph-algorithms 0.6.3 → 0.6.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. package/README.md +62 -32
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-BXqgCifx.js → context-hzGggHeM.js} +68 -24
  4. package/dist/chunks/context-hzGggHeM.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/accelerator.d.ts +8 -6
  7. package/dist/src/accelerator.d.ts.map +1 -1
  8. package/dist/src/accelerator.js +57 -6
  9. package/dist/src/accelerator.js.map +1 -1
  10. package/dist/src/algorithms/bellman-ford.d.ts +60 -0
  11. package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
  12. package/dist/src/algorithms/bellman-ford.js +301 -0
  13. package/dist/src/algorithms/bellman-ford.js.map +1 -0
  14. package/dist/src/algorithms/bfs.d.ts +67 -0
  15. package/dist/src/algorithms/bfs.d.ts.map +1 -0
  16. package/dist/src/algorithms/bfs.js +534 -0
  17. package/dist/src/algorithms/bfs.js.map +1 -0
  18. package/dist/src/algorithms/closeness.d.ts +53 -0
  19. package/dist/src/algorithms/closeness.d.ts.map +1 -0
  20. package/dist/src/algorithms/closeness.js +323 -0
  21. package/dist/src/algorithms/closeness.js.map +1 -0
  22. package/dist/src/algorithms/scope.d.ts +3 -1
  23. package/dist/src/algorithms/scope.d.ts.map +1 -1
  24. package/dist/src/algorithms/scope.js +1 -0
  25. package/dist/src/algorithms/scope.js.map +1 -1
  26. package/dist/src/algorithms/sssp.d.ts +72 -0
  27. package/dist/src/algorithms/sssp.d.ts.map +1 -0
  28. package/dist/src/algorithms/sssp.js +586 -0
  29. package/dist/src/algorithms/sssp.js.map +1 -0
  30. package/dist/src/constants.d.ts +10 -0
  31. package/dist/src/constants.d.ts.map +1 -1
  32. package/dist/src/constants.js +10 -0
  33. package/dist/src/constants.js.map +1 -1
  34. package/dist/src/index.d.ts +8 -2
  35. package/dist/src/index.d.ts.map +1 -1
  36. package/dist/src/index.js +7 -1
  37. package/dist/src/index.js.map +1 -1
  38. package/dist/src/kernel/prelude.d.ts +4 -4
  39. package/dist/src/kernel/prelude.d.ts.map +1 -1
  40. package/dist/src/kernel/prelude.js +39 -5
  41. package/dist/src/kernel/prelude.js.map +1 -1
  42. package/dist/src/kernel/uniform-ring.d.ts +8 -0
  43. package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
  44. package/dist/src/kernel/uniform-ring.js +13 -0
  45. package/dist/src/kernel/uniform-ring.js.map +1 -1
  46. package/dist/src/kernels.d.ts +45 -4
  47. package/dist/src/kernels.d.ts.map +1 -1
  48. package/dist/src/kernels.js +368 -3
  49. package/dist/src/kernels.js.map +1 -1
  50. package/dist/src/primitives/advance.d.ts +63 -0
  51. package/dist/src/primitives/advance.d.ts.map +1 -0
  52. package/dist/src/primitives/advance.js +95 -0
  53. package/dist/src/primitives/advance.js.map +1 -0
  54. package/dist/src/primitives/compact.d.ts +89 -0
  55. package/dist/src/primitives/compact.d.ts.map +1 -0
  56. package/dist/src/primitives/compact.js +233 -0
  57. package/dist/src/primitives/compact.js.map +1 -0
  58. package/dist/src/primitives/core-shape.d.ts +22 -1
  59. package/dist/src/primitives/core-shape.d.ts.map +1 -1
  60. package/dist/src/primitives/core-shape.js +33 -3
  61. package/dist/src/primitives/core-shape.js.map +1 -1
  62. package/dist/src/primitives/frontier.d.ts +151 -0
  63. package/dist/src/primitives/frontier.d.ts.map +1 -0
  64. package/dist/src/primitives/frontier.js +250 -0
  65. package/dist/src/primitives/frontier.js.map +1 -0
  66. package/dist/src/types/accelerator.d.ts +16 -7
  67. package/dist/src/types/accelerator.d.ts.map +1 -1
  68. package/dist/src/types/traversal.d.ts +53 -0
  69. package/dist/src/types/traversal.d.ts.map +1 -0
  70. package/dist/src/types/traversal.js +10 -0
  71. package/dist/src/types/traversal.js.map +1 -0
  72. package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
  73. package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
  74. package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
  75. package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
  76. package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
  77. package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
  78. package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
  79. package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
  80. package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
  81. package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
  82. package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
  83. package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
  84. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
  85. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
  86. package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
  87. package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
  88. package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
  89. package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
  90. package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
  91. package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
  92. package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
  93. package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
  94. package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
  95. package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
  96. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
  97. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
  98. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
  99. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
  100. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
  101. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
  102. package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
  103. package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
  104. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
  105. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
  106. package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
  107. package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
  108. package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
  109. package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
  110. package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
  111. package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
  112. package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
  113. package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
  114. package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
  115. package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
  116. package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
  117. package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
  118. package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
  119. package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
  120. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +53 -0
  121. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
  122. package/dist/src/wgsl/frontier-finalize.wgsl.js +164 -0
  123. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
  124. package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
  125. package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
  126. package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
  127. package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
  128. package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
  129. package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
  130. package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
  131. package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
  132. package/dist/webgpu-graph-algorithms.js +3155 -384
  133. package/dist/webgpu-graph-algorithms.js.map +1 -1
  134. package/package.json +5 -4
  135. package/src/accelerator.ts +65 -7
  136. package/src/algorithms/bellman-ford.ts +387 -0
  137. package/src/algorithms/bfs.ts +626 -0
  138. package/src/algorithms/closeness.ts +395 -0
  139. package/src/algorithms/scope.ts +4 -1
  140. package/src/algorithms/sssp.ts +768 -0
  141. package/src/constants.ts +10 -0
  142. package/src/index.ts +14 -1
  143. package/src/kernel/prelude.ts +39 -4
  144. package/src/kernel/uniform-ring.ts +14 -0
  145. package/src/kernels.ts +447 -6
  146. package/src/primitives/advance.ts +131 -0
  147. package/src/primitives/compact.ts +323 -0
  148. package/src/primitives/core-shape.ts +41 -3
  149. package/src/primitives/frontier.ts +372 -0
  150. package/src/types/accelerator.ts +18 -5
  151. package/src/types/traversal.ts +56 -0
  152. package/src/wgsl/advance-expand.wgsl.ts +68 -0
  153. package/src/wgsl/bf-relax.wgsl.ts +57 -0
  154. package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
  155. package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
  156. package/src/wgsl/bfs-contract.wgsl.ts +54 -0
  157. package/src/wgsl/bfs-fused.wgsl.ts +77 -0
  158. package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
  159. package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
  160. package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
  161. package/src/wgsl/compact-scatter.wgsl.ts +16 -0
  162. package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
  163. package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
  164. package/src/wgsl/frontier-finalize.wgsl.ts +163 -0
  165. package/src/wgsl/sssp-pred.wgsl.ts +79 -0
  166. package/src/wgsl/sssp-relax.wgsl.ts +71 -0
  167. package/dist/chunks/context-BXqgCifx.js.map +0 -1
@@ -0,0 +1,164 @@
1
+ /**
2
+ * The `frontier-finalize` kernel body (design 5.4, 6 row 7; P8-T4, the P8 plan's PD-3 / PD-23 / DEP-P8-C): the
3
+ * device-side selector of the frontier family. One workgroup, one lane, no barrier after the early return (spec 3.5
4
+ * rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's counts become known at
5
+ * two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`, advances
6
+ * `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`) and chooses the path; role 1 runs once the
7
+ * edge queue is filled -- clamps `edgeCount` to `P.edgeCapacity`, or, when `edgeCountUnclamped` exceeds the
8
+ * capacity, switches the path to the fused retry over `frontierCount` (PD-23). The decision is ONE word of the
9
+ * counters block, `path` (word 24): 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3 bottom-up,
10
+ * 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2). Every level kernel is a direct grid-stride
11
+ * dispatch that reads that word first and runs only when it names it (decision record
12
+ * design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md: Dawn's validation of an indirect dispatch cost
13
+ * about 0.4 ms of device time each, and the seven indirect slots this kernel once wrote per level were 97 % of a
14
+ * traversal's wall time; the slots and their args buffer are gone). Two rules: a boundary that finds `done` set
15
+ * moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when role 0
16
+ * chose one, which it reads from the path word (a storage write of one dispatch is visible to the next of the same
17
+ * pass).
18
+ *
19
+ * Roles 2 and 3 are the SSSP round boundary of the near-far loop (P8-T9, PD-20), over the same block read in its
20
+ * SSSP sense (word 1 the raw near half's appends, 21 the raw far half's, 0 and 20 the deduped pile counts, 22 the
21
+ * threshold, 23 the delta, 4 the previous threshold, 14 the round's mode): role 2 finds a non-empty raw near half
22
+ * and sizes the near dedupe (count word 1, output word 0; the dedupe's count in word 8) in mode 0; finds it empty
23
+ * and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
24
+ * unchanged sets `done 3`, the host's E_UNSUPPORTED), remembers the previous threshold in word 4 and sizes the far
25
+ * dedupe (count word 21, output word 20; the dedupe's count in word 9) in mode 1; finds both empty and sets
26
+ * `done 1`; and finds a raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in
27
+ * `level` when it picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds
28
+ * dispatched; a boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed and restarts the
29
+ * raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
30
+ * to 0); the relax kernels size themselves from words 0 and 20.
31
+ *
32
+ * Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a switch can be counted at
33
+ * the done boundary too, which the host model of the tests mirrors): top-down switches to bottom-up when
34
+ * `frontierDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's `max(1, floor(arcCount / n))`
35
+ * unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up switches back when
36
+ * `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no admitted device
37
+ * reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the direction at 0. Every
38
+ * change is counted in `switches`, the previous direction is word 14. The two unvisited words the test reads are
39
+ * rebuilt exactly once per submit by `bfs-unvisited-flags` (PD-18) and maintained here by subtraction: the count is
40
+ * subtracted from the SECOND boundary of a submit on and the degree sum from the THIRD on, because a boundary may
41
+ * only subtract what the submit's rebuild counted, and the frontier whose degree sum the second boundary holds was
42
+ * claimed before the rebuild ran (the rebuild counts the vertices unclaimed when it runs; the frontier rotated in at
43
+ * boundary 0 was claimed by the previous submit's last contract, so it was never in the sum; boundary b subtracts
44
+ * `next = |F_b|`, inside the sum iff b >= 1, and `degSum = deg(F_{b-1})`, inside it iff b >= 2). The degree sum is
45
+ * the "unvisited degree estimate" of the design rather than an exact count for two reasons: it is one level stale
46
+ * (a frontier's degree sum is only known once it has been expanded), and a bottom-up level expands nothing, so the
47
+ * word stops falling while bottom-up runs and overstates the set afterwards. The bias is one-directional -- an
48
+ * overstated m_u makes the switch INTO bottom-up harder, never easier -- and the next submit's rebuild makes it
49
+ * exact again. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
50
+ * textual edits of it.
51
+ */
52
+ export const frontierFinalizeWgsl = /* wgsl */ `
53
+ @compute @workgroup_size(WG)
54
+ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
55
+ if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
56
+ if (P.role == 0u) { // the level boundary
57
+ if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end
58
+ atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)
59
+ return;
60
+ }
61
+ let finished = atomicLoad(&counters[0]);
62
+ let next = atomicLoad(&counters[1]);
63
+ let degSum = atomicLoad(&counters[2]);
64
+ atomicStore(&counters[3], finished); // prevFrontierCount
65
+ atomicStore(&counters[4], degSum); // prevDegreeSum
66
+ atomicStore(&counters[0], next); // the rotation
67
+ atomicStore(&counters[1], 0u);
68
+ atomicStore(&counters[2], 0u);
69
+ atomicStore(&counters[8], 0u); // edgeCount
70
+ atomicStore(&counters[9], 0u); // edgeCountUnclamped
71
+ atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount
72
+ if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped to 2 (P8-T8, PD-18)
73
+ atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount (exact): F_b was inside the submit's rebuilt sum iff b >= 1
74
+ }
75
+ if (P.firstOfSubmit >= 2u) {
76
+ atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2
77
+ }
78
+ let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0
79
+ atomicStore(&counters[11], level);
80
+ let done = (next == 0u) || (level >= P.maxDepth);
81
+ atomicStore(&counters[15], select(0u, 1u, done));
82
+ var direction = atomicLoad(&counters[14]);
83
+ if (P.mode == 1u) {
84
+ direction = 0u; // top-down only (the test seam)
85
+ } else if (direction == 0u) {
86
+ if (degSum > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing
87
+ } else {
88
+ if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking
89
+ }
90
+ if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches
91
+ var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)
92
+ if (done) {
93
+ path = 0u;
94
+ } else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep
95
+ path = 3u;
96
+ atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);
97
+ } else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)
98
+ path = 2u; // bfs-fused: one WORKGROUP per frontier entry
99
+ atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
100
+ } else {
101
+ path = 1u; // advance-expand, then role 1 and bfs-contract
102
+ }
103
+ atomicStore(&counters[14], direction);
104
+ atomicStore(&counters[24], path);
105
+ } else if (P.role == 1u) { // the edge queue is filled
106
+ if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count
107
+ return;
108
+ }
109
+ let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);
110
+ atomicStore(&counters[8], clamped);
111
+ if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry
112
+ atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing
113
+ atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);
114
+ atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
115
+ } else {
116
+ atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)
117
+ }
118
+ } else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes
119
+ atomicStore(&counters[8], 0u); // the dedupe counts (words 8 and 9, the SSSP sense) and the path word: nothing unless a pile is chosen below
120
+ atomicStore(&counters[9], 0u);
121
+ atomicStore(&counters[24], 0u);
122
+ if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)
123
+ return;
124
+ }
125
+ let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped
126
+ let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped
127
+ if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE
128
+ atomicStore(&counters[15], 2u);
129
+ return;
130
+ }
131
+ if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn
132
+ atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter
133
+ atomicStore(&counters[14], 0u); // mode 0
134
+ atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)
135
+ atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing
136
+ atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)
137
+ } else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile
138
+ let threshold = bitcast<f32>(atomicLoad(&counters[22]));
139
+ let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)
140
+ if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED
141
+ atomicStore(&counters[15], 3u);
142
+ return;
143
+ }
144
+ atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below
145
+ atomicStore(&counters[22], bitcast<u32>(raised));
146
+ atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter
147
+ atomicStore(&counters[14], 1u); // mode 1
148
+ atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)
149
+ atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing
150
+ atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);
151
+ } else { // both piles empty: finished
152
+ atomicStore(&counters[15], 1u);
153
+ }
154
+ } else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed
155
+ if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round
156
+ if (atomicLoad(&counters[14]) == 0u) {
157
+ atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)
158
+ } else {
159
+ atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)
160
+ }
161
+ }
162
+ }
163
+ `;
164
+ //# sourceMappingURL=frontier-finalize.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"frontier-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkDG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA+G9C,CAAC"}
@@ -0,0 +1,28 @@
1
+ /**
2
+ * The `sssp-pred` kernel body (design 8.10 "SSSP predecessor pass"; P8-T6 and P8-T9, the P8 plan's PD-11 / PD-24 /
3
+ * PD-27): the ONE post-pass over the settled distances that produces `parent` for `breadthFirstSearch` and
4
+ * `predArc` for `sssp` and `bellmanFord`, so no claim kernel writes a predecessor and every predecessor array is a
5
+ * function of the settled `depth` / `dist` alone (bitwise reproducible, PD-14). `MODE` (a pipeline override) picks
6
+ * the distance type: 1 = u32 depths (`dist` is the BFS depth array, a tight arc is `depth[u] + 1 == depth[v]`, every
7
+ * tight arc is admitted, and the chain is acyclic because depth strictly decreases along it); 0 = f32 bit patterns
8
+ * (a tight arc is `bitcast<u32>(bitcast<f32>(dist[u]) + w) == dist[v]`, one f32 add compared as the bits PD-9
9
+ * stores). `P.predKind` picks what is written: 0 the arc index, 1 the source node index. The winner is the SMALLEST
10
+ * admitted value through `atomicMin` on `pred[v]`, which the driver fills with `INVALID_INDEX` first; the source
11
+ * keeps `INVALID_INDEX` whatever attains it.
12
+ *
13
+ * In `MODE 0` the body runs PD-27's three roles over a `pred` buffer of `2 x hb + 64` words, `hb = roundUp(n, 64)`:
14
+ * `pred[0, n)` the arcs, `pred[hb, hb + n)` the hop counts, `pred[2 hb]` the changed word, `pred[2 hb + 1]` the orphan
15
+ * word. Role 0 (the roots pass, `P.mode 0` only) marks every node with a tight in-arc from a strictly smaller
16
+ * distance as a root (`hops 0`); role 1 (a hop pass, `P.iteration` inside its batch) lowers `hops[v]` to
17
+ * `hops[u] + 1` along every plateau arc (`P.mode 0`: equal distance) or every tight arc (`P.mode 1`: bellmanFord's
18
+ * tight-subgraph rule) and records the pass index in the changed word when it lowered one, returning at its first
19
+ * line when the previous pass changed nothing; role 2 (the predecessor pass) admits the smallest tight arc whose
20
+ * source sits exactly one key step below `v` and counts a reached non-source node the key never reached as an
21
+ * orphan (only bellmanFord can make one). `MODE 1` reads none of `P.role`, `P.mode` and `P.iteration` and never
22
+ * touches `pred` beyond word `n - 1`. The body STRIDES over the rows (`planGridStride(n)`, `P.stride` the plan's
23
+ * stride): a per-invocation body under that plan would leave every row above the dispatch cap at `INVALID_INDEX`,
24
+ * silently. No barrier anywhere, so the early return is legal. Body only (spec 3.5, D9); the text is normative: the
25
+ * sabotage rows of test/helpers/sabotage.ts are textual edits of it.
26
+ */
27
+ export declare const ssspPredWgsl = "\n@compute @workgroup_size(WG)\nfn sssp_pred(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let first = linear_id(wid, lid.x);\n let hb = ((P.n + 63u) / 64u) * 64u; // MODE 0 only: hops[v] is pred[hb + v]; pred[2 hb] is the changed word, pred[2 hb + 1] the orphan word (PD-27)\n if (P.role == 1u && P.iteration != 0u && atomicLoad(&pred[2u * hb]) < P.iteration) { return; } // the previous hop pass changed nothing: converged (no barrier anywhere, so the early return is legal)\n for (var u = first; u < P.n; u = u + P.stride) { // grid-stride over the rows (planGridStride(n)); no barrier anywhere\n let du = dist[u];\n var unreached = false;\n if (MODE == 1u) { unreached = du == INVALID_INDEX; } else { unreached = du == F32_INF_BITS; }\n if (unreached) { continue; }\n var hu = 0u; // u's hop count (MODE 0, roles 1 and 2)\n if (MODE == 0u && P.role != 0u) {\n hu = atomicLoad(&pred[hb + u]);\n if (hu == INVALID_INDEX) { // the key has not reached u yet\n if (P.role == 2u && u != P.source) { atomicAdd(&pred[2u * hb + 1u], 1u); } // an orphan: only bellmanFord can make one (P8-T10 Step 3)\n continue;\n }\n }\n let end = min(rowPtr[u + 1u], P.arcEnd);\n for (var a = max(rowPtr[u], P.arcBase); a < end; a = a + 1u) {\n let v = colIdx[a - P.arcBase];\n if (v == P.source) { continue; } // the source keeps INVALID_INDEX whatever attains it (a zero-weight arc could)\n let dv = dist[v];\n var tight = false; // the arc explains dist[v]\n var below = false; // and its source sits at a strictly smaller distance\n if (MODE == 1u) {\n tight = (du + 1u) == dv; // depth mode: BFS parent, one depth down\n } else {\n let w = select(1.0, weights[a - P.arcBase], HAS_WEIGHTS);\n tight = (dv != F32_INF_BITS) && (bitcast<u32>(bitcast<f32>(du) + w) == dv); // one f32 add, compared as the bit pattern PD-9 stores; never into an unreached v (an overflowed sum is +Inf too)\n below = bitcast<f32>(du) < bitcast<f32>(dv);\n }\n if (!tight) { continue; }\n var admit = false; // this arc is one key step below v\n if (MODE == 1u) {\n admit = true;\n } else if (P.role == 0u) { // the roots pass (the plateau rule only; bellmanFord seeds the source alone)\n if (P.mode == 0u && below) { atomicMin(&pred[hb + v], 0u); }\n } else if (P.role == 1u) { // a hop pass: one hop along a plateau arc (mode 0) or along any tight arc (mode 1)\n if (P.mode == 1u || du == dv) {\n let old = atomicMin(&pred[hb + v], hu + 1u);\n if (hu + 1u < old) { atomicMax(&pred[2u * hb], P.iteration + 1u); } // this pass changed something\n }\n } else { // the predecessor pass: the smallest tight arc one key step below v\n let hv = atomicLoad(&pred[hb + v]);\n if (P.mode == 1u) { admit = hu + 1u == hv; } else if (hv == 0u) { admit = below; } else { admit = (du == dv) && (hu + 1u == hv); }\n }\n if (admit) { atomicMin(&pred[v], select(a, u, P.predKind == 1u)); }\n }\n }\n}\n";
28
+ //# sourceMappingURL=sssp-pred.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"sssp-pred.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/sssp-pred.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AACH,eAAO,MAAM,YAAY,muHAoDxB,CAAC"}
@@ -0,0 +1,80 @@
1
+ /**
2
+ * The `sssp-pred` kernel body (design 8.10 "SSSP predecessor pass"; P8-T6 and P8-T9, the P8 plan's PD-11 / PD-24 /
3
+ * PD-27): the ONE post-pass over the settled distances that produces `parent` for `breadthFirstSearch` and
4
+ * `predArc` for `sssp` and `bellmanFord`, so no claim kernel writes a predecessor and every predecessor array is a
5
+ * function of the settled `depth` / `dist` alone (bitwise reproducible, PD-14). `MODE` (a pipeline override) picks
6
+ * the distance type: 1 = u32 depths (`dist` is the BFS depth array, a tight arc is `depth[u] + 1 == depth[v]`, every
7
+ * tight arc is admitted, and the chain is acyclic because depth strictly decreases along it); 0 = f32 bit patterns
8
+ * (a tight arc is `bitcast<u32>(bitcast<f32>(dist[u]) + w) == dist[v]`, one f32 add compared as the bits PD-9
9
+ * stores). `P.predKind` picks what is written: 0 the arc index, 1 the source node index. The winner is the SMALLEST
10
+ * admitted value through `atomicMin` on `pred[v]`, which the driver fills with `INVALID_INDEX` first; the source
11
+ * keeps `INVALID_INDEX` whatever attains it.
12
+ *
13
+ * In `MODE 0` the body runs PD-27's three roles over a `pred` buffer of `2 x hb + 64` words, `hb = roundUp(n, 64)`:
14
+ * `pred[0, n)` the arcs, `pred[hb, hb + n)` the hop counts, `pred[2 hb]` the changed word, `pred[2 hb + 1]` the orphan
15
+ * word. Role 0 (the roots pass, `P.mode 0` only) marks every node with a tight in-arc from a strictly smaller
16
+ * distance as a root (`hops 0`); role 1 (a hop pass, `P.iteration` inside its batch) lowers `hops[v]` to
17
+ * `hops[u] + 1` along every plateau arc (`P.mode 0`: equal distance) or every tight arc (`P.mode 1`: bellmanFord's
18
+ * tight-subgraph rule) and records the pass index in the changed word when it lowered one, returning at its first
19
+ * line when the previous pass changed nothing; role 2 (the predecessor pass) admits the smallest tight arc whose
20
+ * source sits exactly one key step below `v` and counts a reached non-source node the key never reached as an
21
+ * orphan (only bellmanFord can make one). `MODE 1` reads none of `P.role`, `P.mode` and `P.iteration` and never
22
+ * touches `pred` beyond word `n - 1`. The body STRIDES over the rows (`planGridStride(n)`, `P.stride` the plan's
23
+ * stride): a per-invocation body under that plan would leave every row above the dispatch cap at `INVALID_INDEX`,
24
+ * silently. No barrier anywhere, so the early return is legal. Body only (spec 3.5, D9); the text is normative: the
25
+ * sabotage rows of test/helpers/sabotage.ts are textual edits of it.
26
+ */
27
+ export const ssspPredWgsl = /* wgsl */ `
28
+ @compute @workgroup_size(WG)
29
+ fn sssp_pred(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
30
+ let first = linear_id(wid, lid.x);
31
+ let hb = ((P.n + 63u) / 64u) * 64u; // MODE 0 only: hops[v] is pred[hb + v]; pred[2 hb] is the changed word, pred[2 hb + 1] the orphan word (PD-27)
32
+ if (P.role == 1u && P.iteration != 0u && atomicLoad(&pred[2u * hb]) < P.iteration) { return; } // the previous hop pass changed nothing: converged (no barrier anywhere, so the early return is legal)
33
+ for (var u = first; u < P.n; u = u + P.stride) { // grid-stride over the rows (planGridStride(n)); no barrier anywhere
34
+ let du = dist[u];
35
+ var unreached = false;
36
+ if (MODE == 1u) { unreached = du == INVALID_INDEX; } else { unreached = du == F32_INF_BITS; }
37
+ if (unreached) { continue; }
38
+ var hu = 0u; // u's hop count (MODE 0, roles 1 and 2)
39
+ if (MODE == 0u && P.role != 0u) {
40
+ hu = atomicLoad(&pred[hb + u]);
41
+ if (hu == INVALID_INDEX) { // the key has not reached u yet
42
+ if (P.role == 2u && u != P.source) { atomicAdd(&pred[2u * hb + 1u], 1u); } // an orphan: only bellmanFord can make one (P8-T10 Step 3)
43
+ continue;
44
+ }
45
+ }
46
+ let end = min(rowPtr[u + 1u], P.arcEnd);
47
+ for (var a = max(rowPtr[u], P.arcBase); a < end; a = a + 1u) {
48
+ let v = colIdx[a - P.arcBase];
49
+ if (v == P.source) { continue; } // the source keeps INVALID_INDEX whatever attains it (a zero-weight arc could)
50
+ let dv = dist[v];
51
+ var tight = false; // the arc explains dist[v]
52
+ var below = false; // and its source sits at a strictly smaller distance
53
+ if (MODE == 1u) {
54
+ tight = (du + 1u) == dv; // depth mode: BFS parent, one depth down
55
+ } else {
56
+ let w = select(1.0, weights[a - P.arcBase], HAS_WEIGHTS);
57
+ tight = (dv != F32_INF_BITS) && (bitcast<u32>(bitcast<f32>(du) + w) == dv); // one f32 add, compared as the bit pattern PD-9 stores; never into an unreached v (an overflowed sum is +Inf too)
58
+ below = bitcast<f32>(du) < bitcast<f32>(dv);
59
+ }
60
+ if (!tight) { continue; }
61
+ var admit = false; // this arc is one key step below v
62
+ if (MODE == 1u) {
63
+ admit = true;
64
+ } else if (P.role == 0u) { // the roots pass (the plateau rule only; bellmanFord seeds the source alone)
65
+ if (P.mode == 0u && below) { atomicMin(&pred[hb + v], 0u); }
66
+ } else if (P.role == 1u) { // a hop pass: one hop along a plateau arc (mode 0) or along any tight arc (mode 1)
67
+ if (P.mode == 1u || du == dv) {
68
+ let old = atomicMin(&pred[hb + v], hu + 1u);
69
+ if (hu + 1u < old) { atomicMax(&pred[2u * hb], P.iteration + 1u); } // this pass changed something
70
+ }
71
+ } else { // the predecessor pass: the smallest tight arc one key step below v
72
+ let hv = atomicLoad(&pred[hb + v]);
73
+ if (P.mode == 1u) { admit = hu + 1u == hv; } else if (hv == 0u) { admit = below; } else { admit = (du == dv) && (hu + 1u == hv); }
74
+ }
75
+ if (admit) { atomicMin(&pred[v], select(a, u, P.predKind == 1u)); }
76
+ }
77
+ }
78
+ }
79
+ `;
80
+ //# sourceMappingURL=sssp-pred.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"sssp-pred.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/sssp-pred.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AACH,MAAM,CAAC,MAAM,YAAY,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAoDtC,CAAC"}
@@ -0,0 +1,30 @@
1
+ /**
2
+ * The `sssp-relax` kernel body (design 8.4 "Davidson's near-far", 8.10 "SSSP near-far relax"; P8-T9, the P8 plan's
3
+ * PD-9 / PD-20 / PD-22 / DEP-P8-E): one round of the near-far shortest-path loop. `dist` is `array<atomic<u32>>`
4
+ * holding the IEEE-754 bit patterns of the f32 distances (`F32_INF_BITS` = unreached), and `atomicMin` on the
5
+ * patterns IS `min` on the values, because for non-negative floats the unsigned bit order is the value order
6
+ * (`+0` is `0x00000000`, every sum of non-negatives under round-to-nearest is `+0` or positive, never `-0`), which
7
+ * the driver's host scan of the weight vector guarantees -- the whole trick, and it needs no float atomic (PD-9).
8
+ * Every candidate is ONE f32 add, `dist[u] + w`, so the settled value is the minimum of a fixed set of f32
9
+ * numbers: order-independent, bitwise reproducible, and bitwise equal to the f32 Dijkstra oracle.
10
+ *
11
+ * Two roles over one body, chosen by `P.role`. Role 0 (the near round) reads the deduped near pile `queueIn`
12
+ * (`counters[0]` entries, at most `P.n`), relaxes every arc of every entry's WHOLE row (never windowed, DEP-P8-E),
13
+ * skips a candidate above `P.cutoffBits` (the CPU port's `dv <= cutoff` guard as `nd > cutoff`), and when its
14
+ * `atomicMin` improved `v` -- this lane alone observed a larger old value, so this lane alone owns the append --
15
+ * appends `v` to the raw near half of `queueOut` (word 0, count word 1) when `nd` is below the threshold
16
+ * (`counters[22]`), else to the raw far half (word `P.edgeCapacity`, count word 21). Role 1 (the pass-through,
17
+ * PD-20) re-buckets the deduped far pile (`counters[20]` entries): an entry whose settled distance fell below the
18
+ * PREVIOUS threshold (`counters[4]`) was appended to near at that improvement and relaxed there, so it is dropped;
19
+ * the rest go back to near or far against the threshold the boundary just raised. The count words are unclamped
20
+ * (the write is guarded by the capacity; `frontier-finalize` role 2 detects a pile above it). The appends are per
21
+ * improving relaxation, one atomic each: inside a per-lane arc loop no workgroup aggregation is possible without the
22
+ * uniform-strip structure of `bfs-fused`, and Davidson's kernel appends per thread too. Grid-strided under a direct
23
+ * dispatch of `planGridStride(n)`: `P.stride` is the plan's, a deduped pile of at most `n` entries is covered in a
24
+ * few trips per lane and `i + stride` never wraps (`U32_MAX` would); the block's `path` word (5 a near round, 6 a
25
+ * far one) makes the other role's dispatch a no-op. No barrier anywhere: the loops may
26
+ * be per lane. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
27
+ * textual edits of it.
28
+ */
29
+ export declare const ssspRelaxWgsl = "\n@compute @workgroup_size(WG)\nfn sssp_relax(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let first = linear_id(wid, lid.x);\n let mine = select(5u, 6u, P.role == 1u); // the path word role 2 wrote: 5 a near round, 6 a far pass-through\n let chosen = atomicLoad(&counters[24]) == mine; // the other role's dispatch of the round is a no-op\n let count = select(0u, min(atomicLoad(&counters[select(0u, 20u, P.role == 1u)]), P.n), chosen); // the deduped near or far pile\n let threshold = bitcast<f32>(atomicLoad(&counters[22]));\n let cutoff = bitcast<f32>(P.cutoffBits);\n for (var i = first; i < count; i = i + P.stride) { // no barrier anywhere: the loops may be per lane\n let u = queueIn[i];\n let du = bitcast<f32>(atomicLoad(&dist[u]));\n if (P.role == 1u) { // the pass-through (PD-20): re-bucket a far entry\n if (du < bitcast<f32>(atomicLoad(&counters[4]))) { continue; } // below the previous threshold: relaxed in an earlier bucket\n if (du < threshold) {\n let q = atomicAdd(&counters[1], 1u);\n if (q < P.edgeCapacity) { queueOut[q] = u; }\n } else {\n let q = atomicAdd(&counters[21], 1u);\n if (q < P.edgeCapacity) { queueOut[P.edgeCapacity + q] = u; }\n }\n continue;\n }\n let end = rowPtr[u + 1u];\n for (var a = rowPtr[u]; a < end; a = a + 1u) { // the whole row: never windowed (DEP-P8-E)\n let v = colIdx[a];\n let nd = du + select(1.0, weights[a], HAS_WEIGHTS); // ONE f32 add (PD-9)\n if (nd > cutoff) { continue; } // SsspOptions.cutoff: the CPU port's dv <= cutoff\n let bits = bitcast<u32>(nd);\n let old = atomicMin(&dist[v], bits); // exact on non-negative floats\n if (bits < old) { // this lane improved v, so it owns the append\n if (nd < threshold) {\n let q = atomicAdd(&counters[1], 1u);\n if (q < P.edgeCapacity) { queueOut[q] = v; }\n } else {\n let q = atomicAdd(&counters[21], 1u);\n if (q < P.edgeCapacity) { queueOut[P.edgeCapacity + q] = v; }\n }\n }\n }\n }\n}\n";
30
+ //# sourceMappingURL=sssp-relax.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"sssp-relax.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/sssp-relax.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;GA2BG;AACH,eAAO,MAAM,aAAa,kgFA0CzB,CAAC"}
@@ -0,0 +1,72 @@
1
+ /**
2
+ * The `sssp-relax` kernel body (design 8.4 "Davidson's near-far", 8.10 "SSSP near-far relax"; P8-T9, the P8 plan's
3
+ * PD-9 / PD-20 / PD-22 / DEP-P8-E): one round of the near-far shortest-path loop. `dist` is `array<atomic<u32>>`
4
+ * holding the IEEE-754 bit patterns of the f32 distances (`F32_INF_BITS` = unreached), and `atomicMin` on the
5
+ * patterns IS `min` on the values, because for non-negative floats the unsigned bit order is the value order
6
+ * (`+0` is `0x00000000`, every sum of non-negatives under round-to-nearest is `+0` or positive, never `-0`), which
7
+ * the driver's host scan of the weight vector guarantees -- the whole trick, and it needs no float atomic (PD-9).
8
+ * Every candidate is ONE f32 add, `dist[u] + w`, so the settled value is the minimum of a fixed set of f32
9
+ * numbers: order-independent, bitwise reproducible, and bitwise equal to the f32 Dijkstra oracle.
10
+ *
11
+ * Two roles over one body, chosen by `P.role`. Role 0 (the near round) reads the deduped near pile `queueIn`
12
+ * (`counters[0]` entries, at most `P.n`), relaxes every arc of every entry's WHOLE row (never windowed, DEP-P8-E),
13
+ * skips a candidate above `P.cutoffBits` (the CPU port's `dv <= cutoff` guard as `nd > cutoff`), and when its
14
+ * `atomicMin` improved `v` -- this lane alone observed a larger old value, so this lane alone owns the append --
15
+ * appends `v` to the raw near half of `queueOut` (word 0, count word 1) when `nd` is below the threshold
16
+ * (`counters[22]`), else to the raw far half (word `P.edgeCapacity`, count word 21). Role 1 (the pass-through,
17
+ * PD-20) re-buckets the deduped far pile (`counters[20]` entries): an entry whose settled distance fell below the
18
+ * PREVIOUS threshold (`counters[4]`) was appended to near at that improvement and relaxed there, so it is dropped;
19
+ * the rest go back to near or far against the threshold the boundary just raised. The count words are unclamped
20
+ * (the write is guarded by the capacity; `frontier-finalize` role 2 detects a pile above it). The appends are per
21
+ * improving relaxation, one atomic each: inside a per-lane arc loop no workgroup aggregation is possible without the
22
+ * uniform-strip structure of `bfs-fused`, and Davidson's kernel appends per thread too. Grid-strided under a direct
23
+ * dispatch of `planGridStride(n)`: `P.stride` is the plan's, a deduped pile of at most `n` entries is covered in a
24
+ * few trips per lane and `i + stride` never wraps (`U32_MAX` would); the block's `path` word (5 a near round, 6 a
25
+ * far one) makes the other role's dispatch a no-op. No barrier anywhere: the loops may
26
+ * be per lane. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
27
+ * textual edits of it.
28
+ */
29
+ export const ssspRelaxWgsl = /* wgsl */ `
30
+ @compute @workgroup_size(WG)
31
+ fn sssp_relax(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
32
+ let first = linear_id(wid, lid.x);
33
+ let mine = select(5u, 6u, P.role == 1u); // the path word role 2 wrote: 5 a near round, 6 a far pass-through
34
+ let chosen = atomicLoad(&counters[24]) == mine; // the other role's dispatch of the round is a no-op
35
+ let count = select(0u, min(atomicLoad(&counters[select(0u, 20u, P.role == 1u)]), P.n), chosen); // the deduped near or far pile
36
+ let threshold = bitcast<f32>(atomicLoad(&counters[22]));
37
+ let cutoff = bitcast<f32>(P.cutoffBits);
38
+ for (var i = first; i < count; i = i + P.stride) { // no barrier anywhere: the loops may be per lane
39
+ let u = queueIn[i];
40
+ let du = bitcast<f32>(atomicLoad(&dist[u]));
41
+ if (P.role == 1u) { // the pass-through (PD-20): re-bucket a far entry
42
+ if (du < bitcast<f32>(atomicLoad(&counters[4]))) { continue; } // below the previous threshold: relaxed in an earlier bucket
43
+ if (du < threshold) {
44
+ let q = atomicAdd(&counters[1], 1u);
45
+ if (q < P.edgeCapacity) { queueOut[q] = u; }
46
+ } else {
47
+ let q = atomicAdd(&counters[21], 1u);
48
+ if (q < P.edgeCapacity) { queueOut[P.edgeCapacity + q] = u; }
49
+ }
50
+ continue;
51
+ }
52
+ let end = rowPtr[u + 1u];
53
+ for (var a = rowPtr[u]; a < end; a = a + 1u) { // the whole row: never windowed (DEP-P8-E)
54
+ let v = colIdx[a];
55
+ let nd = du + select(1.0, weights[a], HAS_WEIGHTS); // ONE f32 add (PD-9)
56
+ if (nd > cutoff) { continue; } // SsspOptions.cutoff: the CPU port's dv <= cutoff
57
+ let bits = bitcast<u32>(nd);
58
+ let old = atomicMin(&dist[v], bits); // exact on non-negative floats
59
+ if (bits < old) { // this lane improved v, so it owns the append
60
+ if (nd < threshold) {
61
+ let q = atomicAdd(&counters[1], 1u);
62
+ if (q < P.edgeCapacity) { queueOut[q] = v; }
63
+ } else {
64
+ let q = atomicAdd(&counters[21], 1u);
65
+ if (q < P.edgeCapacity) { queueOut[P.edgeCapacity + q] = v; }
66
+ }
67
+ }
68
+ }
69
+ }
70
+ }
71
+ `;
72
+ //# sourceMappingURL=sssp-relax.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"sssp-relax.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/sssp-relax.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;GA2BG;AACH,MAAM,CAAC,MAAM,aAAa,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA0CvC,CAAC"}