@graphty/webgpu-graph-algorithms 0.6.3 → 0.6.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. package/README.md +62 -32
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-BXqgCifx.js → context-Dvq-Cc6v.js} +71 -25
  4. package/dist/chunks/context-Dvq-Cc6v.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/accelerator.d.ts +8 -6
  7. package/dist/src/accelerator.d.ts.map +1 -1
  8. package/dist/src/accelerator.js +57 -6
  9. package/dist/src/accelerator.js.map +1 -1
  10. package/dist/src/algorithms/bellman-ford.d.ts +60 -0
  11. package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
  12. package/dist/src/algorithms/bellman-ford.js +301 -0
  13. package/dist/src/algorithms/bellman-ford.js.map +1 -0
  14. package/dist/src/algorithms/bfs.d.ts +67 -0
  15. package/dist/src/algorithms/bfs.d.ts.map +1 -0
  16. package/dist/src/algorithms/bfs.js +534 -0
  17. package/dist/src/algorithms/bfs.js.map +1 -0
  18. package/dist/src/algorithms/closeness.d.ts +53 -0
  19. package/dist/src/algorithms/closeness.d.ts.map +1 -0
  20. package/dist/src/algorithms/closeness.js +323 -0
  21. package/dist/src/algorithms/closeness.js.map +1 -0
  22. package/dist/src/algorithms/scope.d.ts +5 -3
  23. package/dist/src/algorithms/scope.d.ts.map +1 -1
  24. package/dist/src/algorithms/scope.js +3 -0
  25. package/dist/src/algorithms/scope.js.map +1 -1
  26. package/dist/src/algorithms/sssp.d.ts +71 -0
  27. package/dist/src/algorithms/sssp.d.ts.map +1 -0
  28. package/dist/src/algorithms/sssp.js +585 -0
  29. package/dist/src/algorithms/sssp.js.map +1 -0
  30. package/dist/src/constants.d.ts +12 -0
  31. package/dist/src/constants.d.ts.map +1 -1
  32. package/dist/src/constants.js +12 -0
  33. package/dist/src/constants.js.map +1 -1
  34. package/dist/src/index.d.ts +8 -2
  35. package/dist/src/index.d.ts.map +1 -1
  36. package/dist/src/index.js +7 -1
  37. package/dist/src/index.js.map +1 -1
  38. package/dist/src/kernel/prelude.d.ts +4 -4
  39. package/dist/src/kernel/prelude.d.ts.map +1 -1
  40. package/dist/src/kernel/prelude.js +39 -5
  41. package/dist/src/kernel/prelude.js.map +1 -1
  42. package/dist/src/kernel/uniform-ring.d.ts +8 -0
  43. package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
  44. package/dist/src/kernel/uniform-ring.js +13 -0
  45. package/dist/src/kernel/uniform-ring.js.map +1 -1
  46. package/dist/src/kernels.d.ts +44 -4
  47. package/dist/src/kernels.d.ts.map +1 -1
  48. package/dist/src/kernels.js +371 -3
  49. package/dist/src/kernels.js.map +1 -1
  50. package/dist/src/primitives/advance.d.ts +62 -0
  51. package/dist/src/primitives/advance.d.ts.map +1 -0
  52. package/dist/src/primitives/advance.js +95 -0
  53. package/dist/src/primitives/advance.js.map +1 -0
  54. package/dist/src/primitives/compact.d.ts +89 -0
  55. package/dist/src/primitives/compact.d.ts.map +1 -0
  56. package/dist/src/primitives/compact.js +233 -0
  57. package/dist/src/primitives/compact.js.map +1 -0
  58. package/dist/src/primitives/core-shape.d.ts +22 -1
  59. package/dist/src/primitives/core-shape.d.ts.map +1 -1
  60. package/dist/src/primitives/core-shape.js +33 -3
  61. package/dist/src/primitives/core-shape.js.map +1 -1
  62. package/dist/src/primitives/frontier.d.ts +156 -0
  63. package/dist/src/primitives/frontier.d.ts.map +1 -0
  64. package/dist/src/primitives/frontier.js +259 -0
  65. package/dist/src/primitives/frontier.js.map +1 -0
  66. package/dist/src/types/accelerator.d.ts +16 -7
  67. package/dist/src/types/accelerator.d.ts.map +1 -1
  68. package/dist/src/types/traversal.d.ts +53 -0
  69. package/dist/src/types/traversal.d.ts.map +1 -0
  70. package/dist/src/types/traversal.js +10 -0
  71. package/dist/src/types/traversal.js.map +1 -0
  72. package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
  73. package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
  74. package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
  75. package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
  76. package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
  77. package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
  78. package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
  79. package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
  80. package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
  81. package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
  82. package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
  83. package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
  84. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
  85. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
  86. package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
  87. package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
  88. package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
  89. package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
  90. package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
  91. package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
  92. package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
  93. package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
  94. package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
  95. package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
  96. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
  97. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
  98. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
  99. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
  100. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
  101. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
  102. package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
  103. package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
  104. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
  105. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
  106. package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
  107. package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
  108. package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
  109. package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
  110. package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
  111. package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
  112. package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
  113. package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
  114. package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
  115. package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
  116. package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
  117. package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
  118. package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
  119. package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
  120. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +59 -0
  121. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
  122. package/dist/src/wgsl/frontier-finalize.wgsl.js +210 -0
  123. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
  124. package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
  125. package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
  126. package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
  127. package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
  128. package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
  129. package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
  130. package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
  131. package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
  132. package/dist/webgpu-graph-algorithms.js +3207 -377
  133. package/dist/webgpu-graph-algorithms.js.map +1 -1
  134. package/package.json +5 -4
  135. package/src/accelerator.ts +65 -7
  136. package/src/algorithms/bellman-ford.ts +387 -0
  137. package/src/algorithms/bfs.ts +626 -0
  138. package/src/algorithms/closeness.ts +395 -0
  139. package/src/algorithms/scope.ts +13 -3
  140. package/src/algorithms/sssp.ts +767 -0
  141. package/src/constants.ts +12 -0
  142. package/src/index.ts +14 -1
  143. package/src/kernel/prelude.ts +39 -4
  144. package/src/kernel/uniform-ring.ts +14 -0
  145. package/src/kernels.ts +450 -6
  146. package/src/primitives/advance.ts +130 -0
  147. package/src/primitives/compact.ts +323 -0
  148. package/src/primitives/core-shape.ts +41 -3
  149. package/src/primitives/frontier.ts +388 -0
  150. package/src/types/accelerator.ts +18 -5
  151. package/src/types/traversal.ts +56 -0
  152. package/src/wgsl/advance-expand.wgsl.ts +68 -0
  153. package/src/wgsl/bf-relax.wgsl.ts +57 -0
  154. package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
  155. package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
  156. package/src/wgsl/bfs-contract.wgsl.ts +54 -0
  157. package/src/wgsl/bfs-fused.wgsl.ts +77 -0
  158. package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
  159. package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
  160. package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
  161. package/src/wgsl/compact-scatter.wgsl.ts +16 -0
  162. package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
  163. package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
  164. package/src/wgsl/frontier-finalize.wgsl.ts +209 -0
  165. package/src/wgsl/sssp-pred.wgsl.ts +79 -0
  166. package/src/wgsl/sssp-relax.wgsl.ts +71 -0
  167. package/dist/chunks/context-BXqgCifx.js.map +0 -1
@@ -0,0 +1,69 @@
1
+ /**
2
+ * The `advance-expand` kernel body (design 6 row 8, 8.10 "BFS expand"; P8-T5, the P8 plan's PD-23): Gunrock's
3
+ * block-mapped expansion of a vertex frontier into the edge queue. Each workgroup loads up to `WG` frontier entries,
4
+ * reads their degrees clipped to the bound arc window `[P.arcBase, P.arcEnd)` (so the window-aware form of P8-T12
5
+ * changes nothing here), runs the prelude's INCLUSIVE workgroup scan `wg_scan_u32` over those degrees -- the subgroup
6
+ * form when the device has the feature, the Hillis-Steele form otherwise, which is the ONLY text that differs between
7
+ * the twin's two compilations -- and then every invocation strips the range `[0, aggregate)` with a binary search
8
+ * (`upper_bound`) over the scanned degrees to find which entry its arc belongs to. One `atomicAdd` per WORKGROUP
9
+ * reserves the block's span in the queue (`edgeCount`), the same aggregate lands in `edgeCountUnclamped` (the overflow
10
+ * detector, never clamped) and in `frontierDegreeSum` (Beamer's m_f); a lane whose queue position is at or past
11
+ * `P.edgeCapacity` writes nothing (the clamp). The queue holds the TARGET vertex of each arc only (PD-24: `parent`
12
+ * comes from the post-pass). Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan call and the
13
+ * `workgroupUniformLoad` sit unconditionally after the guard, and the strip loop is bounded by a uniform value.
14
+ * There is no `TIER` override: a hub row is balanced over all `WG` lanes inside its block, and the small-frontier
15
+ * workgroup-per-row structure is `bfs-fused` (P8-T7). Body only (spec 3.5, D9); the text is normative: the
16
+ * sabotage rows of test/helpers/sabotage.ts are textual edits of it.
17
+ */
18
+ export const advanceExpandWgsl = /* wgsl */ `
19
+ var<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan
20
+ var<workgroup> rowStart: array<u32, WG>; // the first bound arc of each entry's row
21
+ var<workgroup> base: u32; // the block's reserved span in the edge queue
22
+ var<workgroup> wcount: u32; // the frontier's length on a two-phase level, 0 on any other
23
+
24
+ @compute @workgroup_size(WG)
25
+ fn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
26
+ if (lid.x == 0u) { wcount = select(0u, atomicLoad(&counters[0]), atomicLoad(&counters[24]) == 1u); } // frontierCount, on the two-phase path only (the path word)
27
+ let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
28
+ for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries (a direct dispatch of the plan's groups)
29
+ let i = b0 + lid.x; // this lane's frontier entry
30
+ var deg = 0u;
31
+ var start = 0u;
32
+ if (i < count) { // guarded loads into locals (3.5 rule 1)
33
+ let v = frontierIn[i];
34
+ let lo = max(rowPtr[v], P.arcBase); // the row clipped to the bound window (P8-T12)
35
+ let hi = min(rowPtr[v + 1u], P.arcEnd);
36
+ start = lo;
37
+ deg = select(0u, hi - lo, hi > lo);
38
+ }
39
+ let inclusive = wg_scan_u32(deg, lid.x); // the prelude's inclusive scan in LANE order (P8-T1 Step 5); the twin
40
+ sh[lid.x] = inclusive; // sh is monotone in lid.x, the index the binary search walks
41
+ rowStart[lid.x] = start;
42
+ workgroupBarrier();
43
+ let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier
44
+ if (lid.x == 0u) {
45
+ base = atomicAdd(&counters[8], aggregate); // edgeCount: ONE reservation per workgroup, not one per arc
46
+ atomicAdd(&counters[9], aggregate); // edgeCountUnclamped: the overflow detector (PD-23)
47
+ atomicAdd(&counters[2], aggregate); // frontierDegreeSum: Beamer's m_f (P8-T8)
48
+ }
49
+ workgroupBarrier();
50
+ for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
51
+ var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p
52
+ var hi = WG;
53
+ loop {
54
+ if (lo >= hi) { break; }
55
+ let mid = (lo + hi) / 2u;
56
+ if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }
57
+ }
58
+ let k = lo;
59
+ var exclusive = 0u;
60
+ if (k > 0u) { exclusive = sh[k - 1u]; }
61
+ let arc = rowStart[k] + (p - exclusive);
62
+ let q = base + p;
63
+ if (q < P.edgeCapacity) { edgeQueue[q] = colIdx[arc - P.arcBase]; } // the clamp of PD-23; the queue holds the target vertex only (PD-24)
64
+ }
65
+ workgroupBarrier(); // sh, rowStart and base are reused by the next block
66
+ }
67
+ }
68
+ `;
69
+ //# sourceMappingURL=advance-expand.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"advance-expand.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/advance-expand.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;GAgBG;AACH,MAAM,CAAC,MAAM,iBAAiB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAkD3C,CAAC"}
@@ -0,0 +1,22 @@
1
+ /**
2
+ * The `bf-relax` kernel body (design 8.4 "Bellman-Ford"; P8-T10, the P8 plan's PD-12 / PD-22 / PD-27 / DEP-P8-E):
3
+ * one edge-parallel relaxation round over the `edgeList()` view -- every logical edge once, in declared orientation,
4
+ * and under `UNDIRECTED` the other direction too, because an undirected edge has ONE weight (read through its
5
+ * forward arc, `weights[edgeToArc[e]]`, from the run's arc-indexed vector: the snapshot's column or the uploaded
6
+ * override). `dist` is `array<atomic<u32>>` holding the f32 bit patterns (`F32_INF_BITS` = unreached, tested as the
7
+ * pattern: `bitcast<f32>(F32_INF_BITS)` is a const-expression Tint rejects). `atomicMin` on the patterns is `min` on
8
+ * the values for NON-NEGATIVE floats only (PD-9); with a negative distance the bit-pattern order reverses, so the
9
+ * claim is a compare-exchange loop on the pattern: read, compute, compare as floats, try to exchange. WGSL lets
10
+ * `atomicCompareExchangeWeak` fail spuriously, so the loop is BOUNDED (PD-12: `P.maxRetries`, the driver's
11
+ * `MAX_RETRIES`; contention is per vertex, not global) and a lane that exhausts it sets `flags[1]`
12
+ * (`retryExhausted`): the driver treats the round as changed and runs on, so the lost update is retried by the next
13
+ * round, which examines every edge anyway; a bound hit in the decision round is E_VALIDATION, never a guess. A
14
+ * successful exchange sets `flags[0]` (`changed`); the driver stops when a batch of rounds changed nothing, and a
15
+ * change in the round after `n - 1` is the negative cycle. `P.cutoffBits` is the CPU port's `dv <= cutoff` guard
16
+ * (`+Inf` when absent; with negative weights a negative cutoff legitimately relaxes). Grid-strided over
17
+ * `planGridStride(edgeCount)` with `P.stride` the plan's stride; no barrier anywhere, so the loops may be per lane.
18
+ * The whole arc array is bound (never windowed, DEP-P8-E). Body only (spec 3.5, D9); the text is normative: the
19
+ * sabotage rows of test/helpers/sabotage.ts are textual edits of it.
20
+ */
21
+ export declare const bfRelaxWgsl = "\nfn relax(v: u32, nd: f32) {\n var cur = atomicLoad(&dist[v]);\n var tries = 0u;\n loop {\n if (!(nd < bitcast<f32>(cur))) { break; } // no improvement; +Inf is greater than every finite nd\n let r = atomicCompareExchangeWeak(&dist[v], cur, bitcast<u32>(nd));\n if (r.exchanged) { atomicStore(&flags[0], 1u); break; } // changed\n cur = r.old_value;\n tries = tries + 1u;\n if (tries >= P.maxRetries) { atomicStore(&flags[1], 1u); break; } // retryExhausted (PD-12)\n }\n}\n\n@compute @workgroup_size(WG)\nfn bf_relax(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let first = linear_id(wid, lid.x);\n let cutoff = bitcast<f32>(P.cutoffBits);\n for (var e = first; e < P.edgeCount; e = e + P.stride) { // each logical edge once (edgeList)\n let u = edgeSrc[e];\n let v = edgeDst[e];\n let w = weights[edgeToArc[e]]; // the edge's weight through its forward arc\n let du = atomicLoad(&dist[u]);\n if (du != F32_INF_BITS) { // unreached is tested as the bit pattern, as sssp-pred does\n let nd = bitcast<f32>(du) + w;\n if (nd <= cutoff) { relax(v, nd); }\n }\n if (UNDIRECTED) { // the other direction of an undirected edge\n let dv = atomicLoad(&dist[v]);\n if (dv != F32_INF_BITS) {\n let nd = bitcast<f32>(dv) + w;\n if (nd <= cutoff) { relax(u, nd); }\n }\n }\n }\n}\n";
22
+ //# sourceMappingURL=bf-relax.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bf-relax.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bf-relax.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AACH,eAAO,MAAM,WAAW,8oDAoCvB,CAAC"}
@@ -0,0 +1,58 @@
1
+ /**
2
+ * The `bf-relax` kernel body (design 8.4 "Bellman-Ford"; P8-T10, the P8 plan's PD-12 / PD-22 / PD-27 / DEP-P8-E):
3
+ * one edge-parallel relaxation round over the `edgeList()` view -- every logical edge once, in declared orientation,
4
+ * and under `UNDIRECTED` the other direction too, because an undirected edge has ONE weight (read through its
5
+ * forward arc, `weights[edgeToArc[e]]`, from the run's arc-indexed vector: the snapshot's column or the uploaded
6
+ * override). `dist` is `array<atomic<u32>>` holding the f32 bit patterns (`F32_INF_BITS` = unreached, tested as the
7
+ * pattern: `bitcast<f32>(F32_INF_BITS)` is a const-expression Tint rejects). `atomicMin` on the patterns is `min` on
8
+ * the values for NON-NEGATIVE floats only (PD-9); with a negative distance the bit-pattern order reverses, so the
9
+ * claim is a compare-exchange loop on the pattern: read, compute, compare as floats, try to exchange. WGSL lets
10
+ * `atomicCompareExchangeWeak` fail spuriously, so the loop is BOUNDED (PD-12: `P.maxRetries`, the driver's
11
+ * `MAX_RETRIES`; contention is per vertex, not global) and a lane that exhausts it sets `flags[1]`
12
+ * (`retryExhausted`): the driver treats the round as changed and runs on, so the lost update is retried by the next
13
+ * round, which examines every edge anyway; a bound hit in the decision round is E_VALIDATION, never a guess. A
14
+ * successful exchange sets `flags[0]` (`changed`); the driver stops when a batch of rounds changed nothing, and a
15
+ * change in the round after `n - 1` is the negative cycle. `P.cutoffBits` is the CPU port's `dv <= cutoff` guard
16
+ * (`+Inf` when absent; with negative weights a negative cutoff legitimately relaxes). Grid-strided over
17
+ * `planGridStride(edgeCount)` with `P.stride` the plan's stride; no barrier anywhere, so the loops may be per lane.
18
+ * The whole arc array is bound (never windowed, DEP-P8-E). Body only (spec 3.5, D9); the text is normative: the
19
+ * sabotage rows of test/helpers/sabotage.ts are textual edits of it.
20
+ */
21
+ export const bfRelaxWgsl = /* wgsl */ `
22
+ fn relax(v: u32, nd: f32) {
23
+ var cur = atomicLoad(&dist[v]);
24
+ var tries = 0u;
25
+ loop {
26
+ if (!(nd < bitcast<f32>(cur))) { break; } // no improvement; +Inf is greater than every finite nd
27
+ let r = atomicCompareExchangeWeak(&dist[v], cur, bitcast<u32>(nd));
28
+ if (r.exchanged) { atomicStore(&flags[0], 1u); break; } // changed
29
+ cur = r.old_value;
30
+ tries = tries + 1u;
31
+ if (tries >= P.maxRetries) { atomicStore(&flags[1], 1u); break; } // retryExhausted (PD-12)
32
+ }
33
+ }
34
+
35
+ @compute @workgroup_size(WG)
36
+ fn bf_relax(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
37
+ let first = linear_id(wid, lid.x);
38
+ let cutoff = bitcast<f32>(P.cutoffBits);
39
+ for (var e = first; e < P.edgeCount; e = e + P.stride) { // each logical edge once (edgeList)
40
+ let u = edgeSrc[e];
41
+ let v = edgeDst[e];
42
+ let w = weights[edgeToArc[e]]; // the edge's weight through its forward arc
43
+ let du = atomicLoad(&dist[u]);
44
+ if (du != F32_INF_BITS) { // unreached is tested as the bit pattern, as sssp-pred does
45
+ let nd = bitcast<f32>(du) + w;
46
+ if (nd <= cutoff) { relax(v, nd); }
47
+ }
48
+ if (UNDIRECTED) { // the other direction of an undirected edge
49
+ let dv = atomicLoad(&dist[v]);
50
+ if (dv != F32_INF_BITS) {
51
+ let nd = bitcast<f32>(dv) + w;
52
+ if (nd <= cutoff) { relax(u, nd); }
53
+ }
54
+ }
55
+ }
56
+ }
57
+ `;
58
+ //# sourceMappingURL=bf-relax.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bf-relax.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bf-relax.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AACH,MAAM,CAAC,MAAM,WAAW,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAoCrC,CAAC"}
@@ -0,0 +1,15 @@
1
+ /**
2
+ * The `bfs-bitset-build` kernel body (design 8.4 "the bitset frontier"; P8-T8): the vertex-list-to-bitset hand-off
3
+ * of a bottom-up level. One invocation per entry of the input frontier (`frontierCount`, word 0), each `atomicOr`ing
4
+ * its vertex's bit into the `ceil(n / 32)`-word bitset at `P.bitsBase` inside the `sweepIn` buffer (the region a
5
+ * `fill` zeroed just before, on every level). The design's "bulk non-atomic path when the frontier is >= 40% of
6
+ * n" iterates WORDS of a frontier that is already a bitset; this frontier is a vertex list, so a word-owning store
7
+ * has nothing to iterate and `atomicOr` per vertex is the whole kernel. The sweep appends what it claims as a plain
8
+ * vertex list, exactly as the contract does, and writes no second bitset: the next level's bits come from this
9
+ * kernel running over that list, so one representation of a frontier (a vertex list plus a count) serves the whole
10
+ * phase and the hand-off in either direction is free. The early return keys on a counter and `local_invocation_id`
11
+ * with no barrier after it (spec 3.5 rule 1). Body only (spec 3.5, D9); the text is normative: the sabotage rows of
12
+ * test/helpers/sabotage.ts are textual edits of it.
13
+ */
14
+ export declare const bfsBitsetBuildWgsl = "\n@compute @workgroup_size(WG)\nfn bfs_bitset_build(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let count = select(0u, atomicLoad(&counters[0]), atomicLoad(&counters[24]) == 3u); // frontierCount, on the bottom-up path only (the path word)\n for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // grid-stride; no barrier anywhere\n let v = frontierIn[i];\n atomicOr(&bits[P.bitsBase + (v >> 5u)], 1u << (v & 31u));\n }\n}\n";
15
+ //# sourceMappingURL=bfs-bitset-build.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bfs-bitset-build.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bfs-bitset-build.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AACH,eAAO,MAAM,kBAAkB,mgBAS9B,CAAC"}
@@ -0,0 +1,24 @@
1
+ /**
2
+ * The `bfs-bitset-build` kernel body (design 8.4 "the bitset frontier"; P8-T8): the vertex-list-to-bitset hand-off
3
+ * of a bottom-up level. One invocation per entry of the input frontier (`frontierCount`, word 0), each `atomicOr`ing
4
+ * its vertex's bit into the `ceil(n / 32)`-word bitset at `P.bitsBase` inside the `sweepIn` buffer (the region a
5
+ * `fill` zeroed just before, on every level). The design's "bulk non-atomic path when the frontier is >= 40% of
6
+ * n" iterates WORDS of a frontier that is already a bitset; this frontier is a vertex list, so a word-owning store
7
+ * has nothing to iterate and `atomicOr` per vertex is the whole kernel. The sweep appends what it claims as a plain
8
+ * vertex list, exactly as the contract does, and writes no second bitset: the next level's bits come from this
9
+ * kernel running over that list, so one representation of a frontier (a vertex list plus a count) serves the whole
10
+ * phase and the hand-off in either direction is free. The early return keys on a counter and `local_invocation_id`
11
+ * with no barrier after it (spec 3.5 rule 1). Body only (spec 3.5, D9); the text is normative: the sabotage rows of
12
+ * test/helpers/sabotage.ts are textual edits of it.
13
+ */
14
+ export const bfsBitsetBuildWgsl = /* wgsl */ `
15
+ @compute @workgroup_size(WG)
16
+ fn bfs_bitset_build(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
17
+ let count = select(0u, atomicLoad(&counters[0]), atomicLoad(&counters[24]) == 3u); // frontierCount, on the bottom-up path only (the path word)
18
+ for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // grid-stride; no barrier anywhere
19
+ let v = frontierIn[i];
20
+ atomicOr(&bits[P.bitsBase + (v >> 5u)], 1u << (v & 31u));
21
+ }
22
+ }
23
+ `;
24
+ //# sourceMappingURL=bfs-bitset-build.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bfs-bitset-build.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bfs-bitset-build.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,UAAU,CAAC;;;;;;;;;CAS5C,CAAC"}
@@ -0,0 +1,20 @@
1
+ /**
2
+ * The `bfs-bottom-up` kernel body (design 8.4 "the bottom-up sweep"; P8-T8, the P8 plan's PD-18 / PD-21): one
3
+ * invocation per entry of the unvisited list (`unvisitedListLen`, word 7; the list is the first region of the
4
+ * read-only `sweepIn` buffer, the frontier bitset its second at `P.bitsBase`), over the REVERSE core bound in group 0
5
+ * (`coreOfView(residency.view(s, "reverse"))`: the forward arrays on an undirected snapshot, which P7's residency
6
+ * aliases at zero upload cost). A lane whose vertex is still at `INVALID_INDEX` -- the list is up to 32 levels
7
+ * stale, and an entry claimed since the rebuild is skipped on that test -- walks its in-neighbours until the FIRST
8
+ * one whose bit is set and breaks: that early exit is the whole point of the direction (a vertex with one frontier
9
+ * neighbour among thousands costs one read), and `arcsScanned` (word 16, one `atomicAdd` per workgroup of the lanes'
10
+ * reads through `wg_reduce_u32`) is the counter the sabotage row that removes the exit is pinned to, never a timing.
11
+ * The winners are packed into the output vertex queue by `bfs-contract`'s workgroup scan and one `atomicAdd` per
12
+ * workgroup on `nextFrontierCount`, and claim with a plain `atomicStore`: the list holds every vertex once and the
13
+ * sweep is vertex-parallel, so no two lanes claim one vertex. It adds nothing to `frontierDegreeSum` (a bottom-up
14
+ * level expands nothing), which is why `unvisitedDegreeSum` stops falling while bottom-up runs (the selector's
15
+ * JSDoc). Uniformity (spec 3.5 rule 1): the guarded walk writes locals, the scan and the reduction run
16
+ * unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
17
+ * test/helpers/sabotage.ts are textual edits of it.
18
+ */
19
+ export declare const bfsBottomUpWgsl = "\nvar<workgroup> sh: array<u32, WG>;\nvar<workgroup> base: u32;\nvar<workgroup> wcount: u32; // the unvisited list's length on a bottom-up level, 0 on any other\n\n@compute @workgroup_size(WG)\nfn bfs_bottom_up(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let claim = atomicLoad(&counters[11]) + 1u;\n if (lid.x == 0u) { wcount = select(0u, atomicLoad(&counters[7]), atomicLoad(&counters[24]) == 3u); } // unvisitedListLen, on the bottom-up path only (the path word)\n let len = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers\n for (var b0 = group_id(wid) * WG; b0 < len; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries\n let i = b0 + lid.x;\n var won = 0u;\n var v = 0u;\n var reads = 0u;\n if (i < len) { // guarded work into locals\n v = sweepIn[i];\n if (atomicLoad(&depth[v]) == INVALID_INDEX) { // a stale entry, claimed since the rebuild, is skipped\n let end = min(rowPtr[v + 1u], P.arcEnd);\n for (var a = max(rowPtr[v], P.arcBase); a < end; a = a + 1u) { // in-neighbours through the reverse core\n reads = reads + 1u;\n let u = colIdx[a - P.arcBase];\n if (mask_bit(sweepIn[P.bitsBase + (u >> 5u)], u)) { won = 1u; break; } // the early exit: a real break, never a flag\n }\n }\n }\n sh[lid.x] = won;\n workgroupBarrier();\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of won (bfs-contract's)\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let inclusive = sh[lid.x];\n let readsTotal = wg_reduce_u32(reads, lid.x, 0u); // arcsScanned, one atomic per workgroup (the sabotage witness, Step 6)\n if (lid.x == WG - 1u) { base = atomicAdd(&counters[1], inclusive); }\n if (lid.x == 0u) { atomicAdd(&counters[16], readsTotal); }\n workgroupBarrier();\n if (won == 1u) {\n atomicStore(&depth[v], claim); // no claim race: the list holds v once and the sweep is vertex-parallel\n frontierOut[base + inclusive - 1u] = v;\n }\n workgroupBarrier(); // sh and base are reused by the next block\n }\n}\n";
20
+ //# sourceMappingURL=bfs-bottom-up.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bfs-bottom-up.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bfs-bottom-up.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,eAAO,MAAM,eAAe,8pFA+C3B,CAAC"}
@@ -0,0 +1,67 @@
1
+ /**
2
+ * The `bfs-bottom-up` kernel body (design 8.4 "the bottom-up sweep"; P8-T8, the P8 plan's PD-18 / PD-21): one
3
+ * invocation per entry of the unvisited list (`unvisitedListLen`, word 7; the list is the first region of the
4
+ * read-only `sweepIn` buffer, the frontier bitset its second at `P.bitsBase`), over the REVERSE core bound in group 0
5
+ * (`coreOfView(residency.view(s, "reverse"))`: the forward arrays on an undirected snapshot, which P7's residency
6
+ * aliases at zero upload cost). A lane whose vertex is still at `INVALID_INDEX` -- the list is up to 32 levels
7
+ * stale, and an entry claimed since the rebuild is skipped on that test -- walks its in-neighbours until the FIRST
8
+ * one whose bit is set and breaks: that early exit is the whole point of the direction (a vertex with one frontier
9
+ * neighbour among thousands costs one read), and `arcsScanned` (word 16, one `atomicAdd` per workgroup of the lanes'
10
+ * reads through `wg_reduce_u32`) is the counter the sabotage row that removes the exit is pinned to, never a timing.
11
+ * The winners are packed into the output vertex queue by `bfs-contract`'s workgroup scan and one `atomicAdd` per
12
+ * workgroup on `nextFrontierCount`, and claim with a plain `atomicStore`: the list holds every vertex once and the
13
+ * sweep is vertex-parallel, so no two lanes claim one vertex. It adds nothing to `frontierDegreeSum` (a bottom-up
14
+ * level expands nothing), which is why `unvisitedDegreeSum` stops falling while bottom-up runs (the selector's
15
+ * JSDoc). Uniformity (spec 3.5 rule 1): the guarded walk writes locals, the scan and the reduction run
16
+ * unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
17
+ * test/helpers/sabotage.ts are textual edits of it.
18
+ */
19
+ export const bfsBottomUpWgsl = /* wgsl */ `
20
+ var<workgroup> sh: array<u32, WG>;
21
+ var<workgroup> base: u32;
22
+ var<workgroup> wcount: u32; // the unvisited list's length on a bottom-up level, 0 on any other
23
+
24
+ @compute @workgroup_size(WG)
25
+ fn bfs_bottom_up(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
26
+ let claim = atomicLoad(&counters[11]) + 1u;
27
+ if (lid.x == 0u) { wcount = select(0u, atomicLoad(&counters[7]), atomicLoad(&counters[24]) == 3u); } // unvisitedListLen, on the bottom-up path only (the path word)
28
+ let len = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
29
+ for (var b0 = group_id(wid) * WG; b0 < len; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
30
+ let i = b0 + lid.x;
31
+ var won = 0u;
32
+ var v = 0u;
33
+ var reads = 0u;
34
+ if (i < len) { // guarded work into locals
35
+ v = sweepIn[i];
36
+ if (atomicLoad(&depth[v]) == INVALID_INDEX) { // a stale entry, claimed since the rebuild, is skipped
37
+ let end = min(rowPtr[v + 1u], P.arcEnd);
38
+ for (var a = max(rowPtr[v], P.arcBase); a < end; a = a + 1u) { // in-neighbours through the reverse core
39
+ reads = reads + 1u;
40
+ let u = colIdx[a - P.arcBase];
41
+ if (mask_bit(sweepIn[P.bitsBase + (u >> 5u)], u)) { won = 1u; break; } // the early exit: a real break, never a flag
42
+ }
43
+ }
44
+ }
45
+ sh[lid.x] = won;
46
+ workgroupBarrier();
47
+ for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of won (bfs-contract's)
48
+ var t = 0u;
49
+ if (lid.x >= s) { t = sh[lid.x - s]; }
50
+ workgroupBarrier();
51
+ sh[lid.x] = sh[lid.x] + t;
52
+ workgroupBarrier();
53
+ }
54
+ let inclusive = sh[lid.x];
55
+ let readsTotal = wg_reduce_u32(reads, lid.x, 0u); // arcsScanned, one atomic per workgroup (the sabotage witness, Step 6)
56
+ if (lid.x == WG - 1u) { base = atomicAdd(&counters[1], inclusive); }
57
+ if (lid.x == 0u) { atomicAdd(&counters[16], readsTotal); }
58
+ workgroupBarrier();
59
+ if (won == 1u) {
60
+ atomicStore(&depth[v], claim); // no claim race: the list holds v once and the sweep is vertex-parallel
61
+ frontierOut[base + inclusive - 1u] = v;
62
+ }
63
+ workgroupBarrier(); // sh and base are reused by the next block
64
+ }
65
+ }
66
+ `;
67
+ //# sourceMappingURL=bfs-bottom-up.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bfs-bottom-up.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bfs-bottom-up.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,CAAC,MAAM,eAAe,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA+CzC,CAAC"}
@@ -0,0 +1,20 @@
1
+ /**
2
+ * The `bfs-contract` kernel body (design 8.4, 8.10 "BFS contract"; P8-T6, the P8 plan's PD-5 / PD-6 / PD-24 /
3
+ * DEP-P8-B): the contraction phase of the two-phase top-down level. One invocation per edge-queue entry (the target
4
+ * vertex `advance-expand` wrote), a direct grid-stride dispatch looping to the `edgeCount` `frontier-finalize` role 1
5
+ * clamped, and only when the block's `path` word says the level is two-phase. The claim is `atomicMin(&depth[v], level + 1)` and the invocation that
6
+ * observes `INVALID_INDEX` is the unique winner (PD-6): `atomicMin` is one read-modify-write, so only the first
7
+ * claimant of this level can see the sentinel; a second claimant of the same level observes `claim`, and a vertex
8
+ * already at a smaller depth returns that depth and keeps it. Design 8.4 says why this is not
9
+ * `atomicCompareExchangeWeak`: WGSL 17.8.5 lets it fail spuriously, so a compare-exchange claim needs a retry loop
10
+ * and a bounded loop could leave a vertex unclaimed for its level; `atomicMin` has no such failure mode. The winners
11
+ * are packed by a Hillis-Steele inclusive scan of the workgroup's `won` flags and ONE `atomicAdd` per workgroup on
12
+ * `nextFrontierCount` reserves the block's span in the output vertex queue. The edge queue holds duplicates (a vertex
13
+ * with three frontier neighbours appears three times); exactly one claimant wins, exactly one appends, so the next
14
+ * frontier is duplicate-free by construction and no ownership dedupe follows (DEP-P8-B). Nothing here writes a
15
+ * parent (PD-24: the post-pass does). Uniformity (spec 3.5 rule 1): the guarded work writes locals, every barrier is
16
+ * reached unconditionally. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
17
+ * test/helpers/sabotage.ts are textual edits of it.
18
+ */
19
+ export declare const bfsContractWgsl = "\nvar<workgroup> sh: array<u32, WG>;\nvar<workgroup> base: u32;\nvar<workgroup> wcount: u32; // the clamped edge count on a two-phase level, 0 on any other\n\n@compute @workgroup_size(WG)\nfn bfs_contract(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x == 0u) { wcount = select(0u, atomicLoad(&counters[8]), atomicLoad(&counters[24]) == 1u); } // edgeCount, clamped by role 1; the two-phase path only (the path word)\n let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers\n let claim = atomicLoad(&counters[11]) + 1u; // the depth this level assigns\n for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries\n let i = b0 + lid.x;\n var won = 0u;\n var v = 0u;\n if (i < count) { // guarded work into locals\n v = edgeQueue[i];\n let old = atomicMin(&depth[v], claim);\n won = select(0u, 1u, old == INVALID_INDEX); // PD-6: the unique winner\n }\n sh[lid.x] = won;\n workgroupBarrier();\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of won\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let inclusive = sh[lid.x];\n if (lid.x == WG - 1u) { base = atomicAdd(&counters[1], inclusive); } // nextFrontierCount: one reservation per workgroup\n workgroupBarrier();\n if (won == 1u) { frontierOut[base + inclusive - 1u] = v; }\n workgroupBarrier(); // sh and base are reused by the next block\n }\n}\n";
20
+ //# sourceMappingURL=bfs-contract.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bfs-contract.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bfs-contract.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,eAAO,MAAM,eAAe,67DAmC3B,CAAC"}
@@ -0,0 +1,55 @@
1
+ /**
2
+ * The `bfs-contract` kernel body (design 8.4, 8.10 "BFS contract"; P8-T6, the P8 plan's PD-5 / PD-6 / PD-24 /
3
+ * DEP-P8-B): the contraction phase of the two-phase top-down level. One invocation per edge-queue entry (the target
4
+ * vertex `advance-expand` wrote), a direct grid-stride dispatch looping to the `edgeCount` `frontier-finalize` role 1
5
+ * clamped, and only when the block's `path` word says the level is two-phase. The claim is `atomicMin(&depth[v], level + 1)` and the invocation that
6
+ * observes `INVALID_INDEX` is the unique winner (PD-6): `atomicMin` is one read-modify-write, so only the first
7
+ * claimant of this level can see the sentinel; a second claimant of the same level observes `claim`, and a vertex
8
+ * already at a smaller depth returns that depth and keeps it. Design 8.4 says why this is not
9
+ * `atomicCompareExchangeWeak`: WGSL 17.8.5 lets it fail spuriously, so a compare-exchange claim needs a retry loop
10
+ * and a bounded loop could leave a vertex unclaimed for its level; `atomicMin` has no such failure mode. The winners
11
+ * are packed by a Hillis-Steele inclusive scan of the workgroup's `won` flags and ONE `atomicAdd` per workgroup on
12
+ * `nextFrontierCount` reserves the block's span in the output vertex queue. The edge queue holds duplicates (a vertex
13
+ * with three frontier neighbours appears three times); exactly one claimant wins, exactly one appends, so the next
14
+ * frontier is duplicate-free by construction and no ownership dedupe follows (DEP-P8-B). Nothing here writes a
15
+ * parent (PD-24: the post-pass does). Uniformity (spec 3.5 rule 1): the guarded work writes locals, every barrier is
16
+ * reached unconditionally. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
17
+ * test/helpers/sabotage.ts are textual edits of it.
18
+ */
19
+ export const bfsContractWgsl = /* wgsl */ `
20
+ var<workgroup> sh: array<u32, WG>;
21
+ var<workgroup> base: u32;
22
+ var<workgroup> wcount: u32; // the clamped edge count on a two-phase level, 0 on any other
23
+
24
+ @compute @workgroup_size(WG)
25
+ fn bfs_contract(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
26
+ if (lid.x == 0u) { wcount = select(0u, atomicLoad(&counters[8]), atomicLoad(&counters[24]) == 1u); } // edgeCount, clamped by role 1; the two-phase path only (the path word)
27
+ let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
28
+ let claim = atomicLoad(&counters[11]) + 1u; // the depth this level assigns
29
+ for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
30
+ let i = b0 + lid.x;
31
+ var won = 0u;
32
+ var v = 0u;
33
+ if (i < count) { // guarded work into locals
34
+ v = edgeQueue[i];
35
+ let old = atomicMin(&depth[v], claim);
36
+ won = select(0u, 1u, old == INVALID_INDEX); // PD-6: the unique winner
37
+ }
38
+ sh[lid.x] = won;
39
+ workgroupBarrier();
40
+ for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of won
41
+ var t = 0u;
42
+ if (lid.x >= s) { t = sh[lid.x - s]; }
43
+ workgroupBarrier();
44
+ sh[lid.x] = sh[lid.x] + t;
45
+ workgroupBarrier();
46
+ }
47
+ let inclusive = sh[lid.x];
48
+ if (lid.x == WG - 1u) { base = atomicAdd(&counters[1], inclusive); } // nextFrontierCount: one reservation per workgroup
49
+ workgroupBarrier();
50
+ if (won == 1u) { frontierOut[base + inclusive - 1u] = v; }
51
+ workgroupBarrier(); // sh and base are reused by the next block
52
+ }
53
+ }
54
+ `;
55
+ //# sourceMappingURL=bfs-contract.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bfs-contract.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bfs-contract.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,CAAC,MAAM,eAAe,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAmCzC,CAAC"}
@@ -0,0 +1,25 @@
1
+ /**
2
+ * The `bfs-fused` kernel body (design 8.4 "the fused variant", 6 row 8 "the workgroup-per-row tier", 8.10 "BFS
3
+ * fused expand-contract"; P8-T7, the P8 plan's PD-23 / PD-24): the expansion and the contraction of one level in
4
+ * ONE dispatch, chosen by `frontier-finalize` for a frontier below `P.fusedMax` entries (`path` 2) and for the retry
5
+ * of a level whose edge queue overflowed (`path` 4, PD-23). One WORKGROUP per frontier entry, the workgroups striding
6
+ * the entries by the dispatch's group count (`P.stride`): lane 0 reads the entry's row clipped
7
+ * to the bound arc window, adds its degree to `frontierDegreeSum` (Beamer's m_f, so P8-T8's test sees fused levels
8
+ * too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim inline --
9
+ * `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
10
+ * packing the strip's winners into the output vertex queue by the same Hillis-Steele scan and one `atomicAdd` per
11
+ * strip on `nextFrontierCount`. No edge queue is written or read, which is the whole win for a tiny frontier
12
+ * (Merrill's fleeting iterations) and what makes the overflow retry exact: the partial edge queue is never consulted.
13
+ * On a retry level `advance-expand` has already added the frontier's degree to `frontierDegreeSum`, so that word
14
+ * holds 2 x m_f for the level and the next boundary's Beamer test and degree-sum subtraction see the doubled value;
15
+ * reachable only with a faked capacity or an absurd graph, accepted and said here rather than guarded. Nothing here
16
+ * writes a parent (PD-24: the post-pass does), which is what keeps the kernel at the eight-storage-buffer budget
17
+ * with the four graph slots. Uniformity (spec 3.5 rule 1): the strip loop's bound and the row start are
18
+ * `workgroupUniformLoad`s, so every barrier of the per-strip append is in uniform control flow; the guarded claim
19
+ * writes locals; a trailing barrier protects `sh` and `base` before the next strip reuses them. The claim line and
20
+ * the scan are `bfs-contract`'s verbatim so a reader can diff the two bodies; the scan's comment differs on purpose
21
+ * (the sabotage rows need distinct find strings). Body only (spec 3.5, D9); the text is normative: the sabotage rows
22
+ * of test/helpers/sabotage.ts are textual edits of it.
23
+ */
24
+ export declare const bfsFusedWgsl = "\nvar<workgroup> sh: array<u32, WG>;\nvar<workgroup> wdeg: u32;\nvar<workgroup> wstart: u32;\nvar<workgroup> base: u32;\nvar<workgroup> wcount: u32; // the frontier's length on a fused or retry level, 0 on any other\n\n@compute @workgroup_size(WG)\nfn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let claim = atomicLoad(&counters[11]) + 1u;\n if (lid.x == 0u) {\n let path = atomicLoad(&counters[24]); // 2 the fused level, 4 the overflow retry (PD-23)\n wcount = select(0u, atomicLoad(&counters[0]), path == 2u || path == 4u);\n }\n let count = workgroupUniformLoad(&wcount); // uniform: the entry loop below holds barriers\n for (var g = group_id(wid); g < count; g = g + P.stride) { // one workgroup per frontier entry; P.stride is the dispatch's GROUP count\n if (lid.x == 0u) {\n let u = frontierIn[g];\n let a0 = max(rowPtr[u], P.arcBase);\n let a1 = min(rowPtr[u + 1u], P.arcEnd);\n let d = select(0u, a1 - a0, a1 > a0);\n wdeg = d;\n wstart = a0;\n atomicAdd(&counters[2], d); // frontierDegreeSum, so Beamer's test (P8-T8) sees fused levels too\n }\n let deg = workgroupUniformLoad(&wdeg); // uniform: the loop below may hold barriers\n let start = workgroupUniformLoad(&wstart);\n for (var p0 = 0u; p0 < deg; p0 = p0 + WG) { // strip the row WG arcs at a time\n let p = p0 + lid.x;\n var won = 0u;\n var v = 0u;\n if (p < deg) { // guarded claim into locals\n v = colIdx[start + p - P.arcBase];\n let old = atomicMin(&depth[v], claim);\n won = select(0u, 1u, old == INVALID_INDEX);\n }\n sh[lid.x] = won;\n workgroupBarrier();\n for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of won (bfs-contract's, verbatim)\n var t = 0u;\n if (lid.x >= s) { t = sh[lid.x - s]; }\n workgroupBarrier();\n sh[lid.x] = sh[lid.x] + t;\n workgroupBarrier();\n }\n let inclusive = sh[lid.x];\n if (lid.x == WG - 1u) { base = atomicAdd(&counters[1], inclusive); }\n workgroupBarrier();\n if (won == 1u) { frontierOut[base + inclusive - 1u] = v; }\n workgroupBarrier(); // sh and base are reused by the next strip\n }\n }\n}\n";
25
+ //# sourceMappingURL=bfs-fused.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bfs-fused.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bfs-fused.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;GAsBG;AACH,eAAO,MAAM,YAAY,kvFAqDxB,CAAC"}
@@ -0,0 +1,78 @@
1
+ /**
2
+ * The `bfs-fused` kernel body (design 8.4 "the fused variant", 6 row 8 "the workgroup-per-row tier", 8.10 "BFS
3
+ * fused expand-contract"; P8-T7, the P8 plan's PD-23 / PD-24): the expansion and the contraction of one level in
4
+ * ONE dispatch, chosen by `frontier-finalize` for a frontier below `P.fusedMax` entries (`path` 2) and for the retry
5
+ * of a level whose edge queue overflowed (`path` 4, PD-23). One WORKGROUP per frontier entry, the workgroups striding
6
+ * the entries by the dispatch's group count (`P.stride`): lane 0 reads the entry's row clipped
7
+ * to the bound arc window, adds its degree to `frontierDegreeSum` (Beamer's m_f, so P8-T8's test sees fused levels
8
+ * too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim inline --
9
+ * `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
10
+ * packing the strip's winners into the output vertex queue by the same Hillis-Steele scan and one `atomicAdd` per
11
+ * strip on `nextFrontierCount`. No edge queue is written or read, which is the whole win for a tiny frontier
12
+ * (Merrill's fleeting iterations) and what makes the overflow retry exact: the partial edge queue is never consulted.
13
+ * On a retry level `advance-expand` has already added the frontier's degree to `frontierDegreeSum`, so that word
14
+ * holds 2 x m_f for the level and the next boundary's Beamer test and degree-sum subtraction see the doubled value;
15
+ * reachable only with a faked capacity or an absurd graph, accepted and said here rather than guarded. Nothing here
16
+ * writes a parent (PD-24: the post-pass does), which is what keeps the kernel at the eight-storage-buffer budget
17
+ * with the four graph slots. Uniformity (spec 3.5 rule 1): the strip loop's bound and the row start are
18
+ * `workgroupUniformLoad`s, so every barrier of the per-strip append is in uniform control flow; the guarded claim
19
+ * writes locals; a trailing barrier protects `sh` and `base` before the next strip reuses them. The claim line and
20
+ * the scan are `bfs-contract`'s verbatim so a reader can diff the two bodies; the scan's comment differs on purpose
21
+ * (the sabotage rows need distinct find strings). Body only (spec 3.5, D9); the text is normative: the sabotage rows
22
+ * of test/helpers/sabotage.ts are textual edits of it.
23
+ */
24
+ export const bfsFusedWgsl = /* wgsl */ `
25
+ var<workgroup> sh: array<u32, WG>;
26
+ var<workgroup> wdeg: u32;
27
+ var<workgroup> wstart: u32;
28
+ var<workgroup> base: u32;
29
+ var<workgroup> wcount: u32; // the frontier's length on a fused or retry level, 0 on any other
30
+
31
+ @compute @workgroup_size(WG)
32
+ fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
33
+ let claim = atomicLoad(&counters[11]) + 1u;
34
+ if (lid.x == 0u) {
35
+ let path = atomicLoad(&counters[24]); // 2 the fused level, 4 the overflow retry (PD-23)
36
+ wcount = select(0u, atomicLoad(&counters[0]), path == 2u || path == 4u);
37
+ }
38
+ let count = workgroupUniformLoad(&wcount); // uniform: the entry loop below holds barriers
39
+ for (var g = group_id(wid); g < count; g = g + P.stride) { // one workgroup per frontier entry; P.stride is the dispatch's GROUP count
40
+ if (lid.x == 0u) {
41
+ let u = frontierIn[g];
42
+ let a0 = max(rowPtr[u], P.arcBase);
43
+ let a1 = min(rowPtr[u + 1u], P.arcEnd);
44
+ let d = select(0u, a1 - a0, a1 > a0);
45
+ wdeg = d;
46
+ wstart = a0;
47
+ atomicAdd(&counters[2], d); // frontierDegreeSum, so Beamer's test (P8-T8) sees fused levels too
48
+ }
49
+ let deg = workgroupUniformLoad(&wdeg); // uniform: the loop below may hold barriers
50
+ let start = workgroupUniformLoad(&wstart);
51
+ for (var p0 = 0u; p0 < deg; p0 = p0 + WG) { // strip the row WG arcs at a time
52
+ let p = p0 + lid.x;
53
+ var won = 0u;
54
+ var v = 0u;
55
+ if (p < deg) { // guarded claim into locals
56
+ v = colIdx[start + p - P.arcBase];
57
+ let old = atomicMin(&depth[v], claim);
58
+ won = select(0u, 1u, old == INVALID_INDEX);
59
+ }
60
+ sh[lid.x] = won;
61
+ workgroupBarrier();
62
+ for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of won (bfs-contract's, verbatim)
63
+ var t = 0u;
64
+ if (lid.x >= s) { t = sh[lid.x - s]; }
65
+ workgroupBarrier();
66
+ sh[lid.x] = sh[lid.x] + t;
67
+ workgroupBarrier();
68
+ }
69
+ let inclusive = sh[lid.x];
70
+ if (lid.x == WG - 1u) { base = atomicAdd(&counters[1], inclusive); }
71
+ workgroupBarrier();
72
+ if (won == 1u) { frontierOut[base + inclusive - 1u] = v; }
73
+ workgroupBarrier(); // sh and base are reused by the next strip
74
+ }
75
+ }
76
+ }
77
+ `;
78
+ //# sourceMappingURL=bfs-fused.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bfs-fused.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bfs-fused.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;GAsBG;AACH,MAAM,CAAC,MAAM,YAAY,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAqDtC,CAAC"}
@@ -0,0 +1,18 @@
1
+ /**
2
+ * The `bfs-unvisited-flags` kernel body (design 8.4; P8-T8, the P8 plan's PD-18): the producer of the unvisited set
3
+ * Beamer's test is against, run ONCE per submit before the levels. It grid-strides over the vertices
4
+ * (`planGridStride(n)`, `P.stride` the plan's stride) and, per lane, counts the vertices still at `INVALID_INDEX`
5
+ * (`unvisitedCount`, word 5), sums their OUT-degrees (`unvisitedDegreeSum`, word 6: Beamer's m_u counts the edges
6
+ * top-down would examine), and flags the unvisited vertices with a non-zero IN-degree (`unvisitedListLen`, word 7:
7
+ * what the bottom-up sweep iterates, since a vertex nobody points at can never be claimed by it); the three lane
8
+ * sums are reduced by the prelude's `wg_reduce_u32` (its sum code) and ONE `atomicAdd` per word per workgroup lands
9
+ * them in the counters block. `compact` over an iota queue then turns `flags` into the unvisited list. The list is
10
+ * up to `MAX_LEVELS_PER_SUBMIT` levels stale by the time the sweep reads it: it holds vertices claimed since the
11
+ * rebuild, which the sweep skips on the `depth == INVALID_INDEX` test it makes anyway, so staleness costs a few
12
+ * wasted reads and never a wrong depth. Between rebuilds `frontier-finalize` maintains words 5 and 6 by subtraction
13
+ * (its JSDoc states the boundary rule). Uniformity (spec 3.5 rule 1): the loop holds no barrier, and the three
14
+ * reductions run unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
15
+ * test/helpers/sabotage.ts are textual edits of it.
16
+ */
17
+ export declare const bfsUnvisitedFlagsWgsl = "\n@compute @workgroup_size(WG)\nfn bfs_unvisited_flags(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let first = linear_id(wid, lid.x);\n var cnt = 0u;\n var degSum = 0u;\n var len = 0u;\n for (var v = first; v < P.n; v = v + P.stride) { // no barrier inside: the trip count is per lane\n let unv = depth[v] == INVALID_INDEX;\n let listed = unv && (inDegree[v] != 0u);\n flags[v] = select(0u, 1u, listed);\n cnt = cnt + select(0u, 1u, unv);\n degSum = degSum + select(0u, outDegree[v], unv); // the OUT-degree: Beamer's m_u counts the edges top-down would examine\n len = len + select(0u, 1u, listed);\n }\n let c = wg_reduce_u32(cnt, lid.x, 0u); // the prelude's workgroup sum (combine_u's sum code); uniform: after the loop\n let d = wg_reduce_u32(degSum, lid.x, 0u);\n let l = wg_reduce_u32(len, lid.x, 0u);\n if (lid.x == 0u) { // ONE atomic per word per workgroup\n atomicAdd(&counters[5], c);\n atomicAdd(&counters[6], d);\n atomicAdd(&counters[7], l);\n }\n}\n";
18
+ //# sourceMappingURL=bfs-unvisited-flags.wgsl.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bfs-unvisited-flags.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/bfs-unvisited-flags.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AACH,eAAO,MAAM,qBAAqB,2rCAwBjC,CAAC"}
@@ -0,0 +1,42 @@
1
+ /**
2
+ * The `bfs-unvisited-flags` kernel body (design 8.4; P8-T8, the P8 plan's PD-18): the producer of the unvisited set
3
+ * Beamer's test is against, run ONCE per submit before the levels. It grid-strides over the vertices
4
+ * (`planGridStride(n)`, `P.stride` the plan's stride) and, per lane, counts the vertices still at `INVALID_INDEX`
5
+ * (`unvisitedCount`, word 5), sums their OUT-degrees (`unvisitedDegreeSum`, word 6: Beamer's m_u counts the edges
6
+ * top-down would examine), and flags the unvisited vertices with a non-zero IN-degree (`unvisitedListLen`, word 7:
7
+ * what the bottom-up sweep iterates, since a vertex nobody points at can never be claimed by it); the three lane
8
+ * sums are reduced by the prelude's `wg_reduce_u32` (its sum code) and ONE `atomicAdd` per word per workgroup lands
9
+ * them in the counters block. `compact` over an iota queue then turns `flags` into the unvisited list. The list is
10
+ * up to `MAX_LEVELS_PER_SUBMIT` levels stale by the time the sweep reads it: it holds vertices claimed since the
11
+ * rebuild, which the sweep skips on the `depth == INVALID_INDEX` test it makes anyway, so staleness costs a few
12
+ * wasted reads and never a wrong depth. Between rebuilds `frontier-finalize` maintains words 5 and 6 by subtraction
13
+ * (its JSDoc states the boundary rule). Uniformity (spec 3.5 rule 1): the loop holds no barrier, and the three
14
+ * reductions run unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
15
+ * test/helpers/sabotage.ts are textual edits of it.
16
+ */
17
+ export const bfsUnvisitedFlagsWgsl = /* wgsl */ `
18
+ @compute @workgroup_size(WG)
19
+ fn bfs_unvisited_flags(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
20
+ let first = linear_id(wid, lid.x);
21
+ var cnt = 0u;
22
+ var degSum = 0u;
23
+ var len = 0u;
24
+ for (var v = first; v < P.n; v = v + P.stride) { // no barrier inside: the trip count is per lane
25
+ let unv = depth[v] == INVALID_INDEX;
26
+ let listed = unv && (inDegree[v] != 0u);
27
+ flags[v] = select(0u, 1u, listed);
28
+ cnt = cnt + select(0u, 1u, unv);
29
+ degSum = degSum + select(0u, outDegree[v], unv); // the OUT-degree: Beamer's m_u counts the edges top-down would examine
30
+ len = len + select(0u, 1u, listed);
31
+ }
32
+ let c = wg_reduce_u32(cnt, lid.x, 0u); // the prelude's workgroup sum (combine_u's sum code); uniform: after the loop
33
+ let d = wg_reduce_u32(degSum, lid.x, 0u);
34
+ let l = wg_reduce_u32(len, lid.x, 0u);
35
+ if (lid.x == 0u) { // ONE atomic per word per workgroup
36
+ atomicAdd(&counters[5], c);
37
+ atomicAdd(&counters[6], d);
38
+ atomicAdd(&counters[7], l);
39
+ }
40
+ }
41
+ `;
42
+ //# sourceMappingURL=bfs-unvisited-flags.wgsl.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bfs-unvisited-flags.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/bfs-unvisited-flags.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AACH,MAAM,CAAC,MAAM,qBAAqB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;CAwB/C,CAAC"}