@graphty/webgpu-graph-algorithms 0.6.4 → 0.6.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/README.md +38 -17
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-Dvq-Cc6v.js → context-DiSr6eiz.js} +45 -33
  4. package/dist/chunks/context-DiSr6eiz.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/algorithms/bfs.d.ts +11 -7
  7. package/dist/src/algorithms/bfs.d.ts.map +1 -1
  8. package/dist/src/algorithms/bfs.js +33 -10
  9. package/dist/src/algorithms/bfs.js.map +1 -1
  10. package/dist/src/algorithms/scope.d.ts +3 -3
  11. package/dist/src/algorithms/scope.d.ts.map +1 -1
  12. package/dist/src/algorithms/scope.js +0 -2
  13. package/dist/src/algorithms/scope.js.map +1 -1
  14. package/dist/src/algorithms/sssp.d.ts +4 -3
  15. package/dist/src/algorithms/sssp.d.ts.map +1 -1
  16. package/dist/src/algorithms/sssp.js +4 -3
  17. package/dist/src/algorithms/sssp.js.map +1 -1
  18. package/dist/src/constants.d.ts +33 -2
  19. package/dist/src/constants.d.ts.map +1 -1
  20. package/dist/src/constants.js +33 -2
  21. package/dist/src/constants.js.map +1 -1
  22. package/dist/src/kernel/dispatch.d.ts +2 -2
  23. package/dist/src/kernel/dispatch.d.ts.map +1 -1
  24. package/dist/src/kernel/kernel.d.ts +1 -1
  25. package/dist/src/kernel/kernel.js +2 -2
  26. package/dist/src/kernel/kernel.js.map +1 -1
  27. package/dist/src/kernel/prelude.d.ts.map +1 -1
  28. package/dist/src/kernel/prelude.js +2 -1
  29. package/dist/src/kernel/prelude.js.map +1 -1
  30. package/dist/src/kernels.d.ts +15 -11
  31. package/dist/src/kernels.d.ts.map +1 -1
  32. package/dist/src/kernels.js +42 -21
  33. package/dist/src/kernels.js.map +1 -1
  34. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  35. package/dist/src/layouts/forceatlas2.js +2 -1
  36. package/dist/src/layouts/forceatlas2.js.map +1 -1
  37. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  38. package/dist/src/layouts/fruchterman-reingold.js +4 -2
  39. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  40. package/dist/src/layouts/repulsion-exact.d.ts +16 -0
  41. package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
  42. package/dist/src/layouts/repulsion-exact.js +21 -1
  43. package/dist/src/layouts/repulsion-exact.js.map +1 -1
  44. package/dist/src/layouts/repulsion-grid.d.ts +1 -1
  45. package/dist/src/layouts/repulsion-grid.js +1 -1
  46. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  47. package/dist/src/layouts/spring-electrical.js +6 -2
  48. package/dist/src/layouts/spring-electrical.js.map +1 -1
  49. package/dist/src/primitives/advance.d.ts +3 -2
  50. package/dist/src/primitives/advance.d.ts.map +1 -1
  51. package/dist/src/primitives/advance.js.map +1 -1
  52. package/dist/src/primitives/frontier.d.ts +34 -38
  53. package/dist/src/primitives/frontier.d.ts.map +1 -1
  54. package/dist/src/primitives/frontier.js +24 -32
  55. package/dist/src/primitives/frontier.js.map +1 -1
  56. package/dist/src/primitives/grid-pyramid.d.ts +4 -4
  57. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  58. package/dist/src/primitives/grid-pyramid.js +4 -3
  59. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  60. package/dist/src/primitives/grid.d.ts +13 -10
  61. package/dist/src/primitives/grid.d.ts.map +1 -1
  62. package/dist/src/primitives/grid.js +10 -7
  63. package/dist/src/primitives/grid.js.map +1 -1
  64. package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
  65. package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
  66. package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
  67. package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
  68. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
  69. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
  70. package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
  71. package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
  72. package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
  73. package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
  74. package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
  75. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
  76. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
  77. package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
  78. package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
  79. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
  80. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
  81. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
  82. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
  83. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
  84. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
  85. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
  86. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
  87. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +44 -49
  88. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
  89. package/dist/src/wgsl/frontier-finalize.wgsl.js +62 -107
  90. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
  91. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
  92. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
  93. package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
  94. package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
  95. package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
  96. package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
  97. package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
  98. package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
  99. package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
  100. package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
  101. package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
  102. package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
  103. package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
  104. package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
  105. package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
  106. package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
  107. package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
  108. package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
  109. package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
  110. package/dist/src/wgsl/histogram.wgsl.js +1 -1
  111. package/dist/webgpu-graph-algorithms.js +144 -119
  112. package/dist/webgpu-graph-algorithms.js.map +1 -1
  113. package/package.json +3 -3
  114. package/src/algorithms/bfs.ts +34 -10
  115. package/src/algorithms/pagerank.ts +19 -5
  116. package/src/algorithms/power-iteration.ts +8 -2
  117. package/src/algorithms/scope.ts +3 -10
  118. package/src/algorithms/sssp.ts +4 -3
  119. package/src/constants.ts +35 -2
  120. package/src/kernel/dispatch.ts +2 -2
  121. package/src/kernel/kernel.ts +2 -2
  122. package/src/kernel/prelude.ts +2 -0
  123. package/src/kernels.ts +44 -21
  124. package/src/layouts/forceatlas2.ts +2 -0
  125. package/src/layouts/fruchterman-reingold.ts +4 -1
  126. package/src/layouts/repulsion-exact.ts +29 -1
  127. package/src/layouts/repulsion-grid.ts +1 -1
  128. package/src/layouts/spring-electrical.ts +8 -1
  129. package/src/memory/residency.ts +14 -4
  130. package/src/primitives/advance.ts +5 -4
  131. package/src/primitives/frontier.ts +42 -56
  132. package/src/primitives/grid-pyramid.ts +6 -5
  133. package/src/primitives/grid.ts +17 -12
  134. package/src/wgsl/advance-expand.wgsl.ts +4 -3
  135. package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
  136. package/src/wgsl/bfs-fused.wgsl.ts +6 -6
  137. package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
  138. package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
  139. package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
  140. package/src/wgsl/frontier-finalize.wgsl.ts +62 -107
  141. package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
  142. package/src/wgsl/grid-centroid.wgsl.ts +3 -2
  143. package/src/wgsl/grid-downsample.wgsl.ts +1 -1
  144. package/src/wgsl/grid-far-field.wgsl.ts +15 -8
  145. package/src/wgsl/grid-near-field.wgsl.ts +6 -2
  146. package/src/wgsl/histogram.wgsl.ts +1 -1
  147. package/dist/chunks/context-Dvq-Cc6v.js.map +0 -1
@@ -1,59 +1,54 @@
1
1
  /**
2
2
  * The `frontier-finalize` kernel body (design 5.4, 6 row 7; P8-T4, the P8 plan's PD-3 / PD-23 / DEP-P8-C): the
3
3
  * device-side selector of the frontier family. One workgroup, one lane, no barrier after the early return (spec 3.5
4
- * rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's dispatch sizes become
5
- * known at two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`,
6
- * advances `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`), chooses the path and writes this
7
- * level's seven 16-byte slots at `P.slotBase`; role 1 runs once the edge queue is filled -- clamps `edgeCount` to
8
- * `P.edgeCapacity`, sizes the contract slot, or, when `edgeCountUnclamped` exceeds the capacity, zeroes it and sizes
9
- * the fused-retry slot from `frontierCount` instead (PD-23). Two rules: a boundary that finds `done` set zeroes its
10
- * slots and moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when
11
- * role 0 chose one, which it reads from slot 0's `x` (a storage write of one dispatch is visible to the next of the
12
- * same pass). The `(x, y)` arithmetic is `indirect-finalize`'s verbatim (P4), so a count above 2^32 - wg cannot wrap
13
- * and a group count above MAX_WORKGROUPS_PER_DIM splits in 2D; slots 2 and 6 are sized one WORKGROUP per entry for
14
- * `bfs-fused`; slots 3, 4 and 5 (the bits fill over `ceil(n / 32)` words, the bitset build over the frontier, the sweep
15
- * over the unvisited list) are the bottom-up level's.
4
+ * rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's counts become known at
5
+ * two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`, advances
6
+ * `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`) and chooses the path; role 1 runs once the
7
+ * edge queue is filled -- clamps `edgeCount` to `P.edgeCapacity`, or, when `edgeCountUnclamped` exceeds the
8
+ * capacity, switches the path to the fused retry over `frontierCount` (PD-23). The decision is ONE word of the
9
+ * counters block, `path` (word 24): 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3 bottom-up,
10
+ * 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2). Every level kernel is a direct grid-stride
11
+ * dispatch that reads that word first and runs only when it names it (decision record
12
+ * design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md: Dawn's validation of an indirect dispatch cost
13
+ * about 0.4 ms of device time each, and the seven indirect slots this kernel once wrote per level were 97 % of a
14
+ * traversal's wall time; the slots and their args buffer are gone). Two rules: a boundary that finds `done` set
15
+ * moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when role 0
16
+ * chose one, which it reads from the path word (a storage write of one dispatch is visible to the next of the same
17
+ * pass).
16
18
  *
17
19
  * Roles 2 and 3 are the SSSP round boundary of the near-far loop (P8-T9, PD-20), over the same block read in its
18
20
  * SSSP sense (word 1 the raw near half's appends, 21 the raw far half's, 0 and 20 the deduped pile counts, 22 the
19
- * threshold, 23 the delta, 4 the previous threshold, 14 the round's mode) and the same seven slots: role 2 finds a
20
- * non-empty raw near half and sizes the near dedupe (slots 0 and 1, count word 1, output word 0) in mode 0; finds it
21
- * empty and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
21
+ * threshold, 23 the delta, 4 the previous threshold, 14 the round's mode): role 2 finds a non-empty raw near half
22
+ * and sizes the near dedupe (count word 1, output word 0; the dedupe's count in word 8) in mode 0; finds it empty
23
+ * and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
22
24
  * unchanged sets `done 3`, the host's E_UNSUPPORTED), remembers the previous threshold in word 4 and sizes the far
23
- * dedupe (slots 3 and 4, count word 21, output word 20) in mode 1; finds both empty and sets `done 1`; and finds a
24
- * raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in `level` when it
25
- * picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds dispatched; a
26
- * boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed: mode 0 sizes slot 2 (the
27
- * relax over nearIn) from word 0 and restarts the raw near half (word 1 to 0); mode 1 sizes slot 5 (the pass-through
28
- * over farIn) from word 20 and restarts the raw far half (word 21 to 0).
25
+ * dedupe (count word 21, output word 20; the dedupe's count in word 9) in mode 1; finds both empty and sets
26
+ * `done 1`; and finds a raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in
27
+ * `level` when it picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds
28
+ * dispatched; a boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed and restarts the
29
+ * raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
30
+ * to 0); the relax kernels size themselves from words 0 and 20.
29
31
  *
30
- * Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a switch can be counted at
31
- * the done boundary too, which the host model of the tests mirrors): top-down switches to bottom-up when
32
- * `frontierDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's `max(1, floor(arcCount / n))`
33
- * unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up switches back when
34
- * `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no admitted device
35
- * reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the direction at 0. Every
36
- * change is counted in `switches`, the previous direction is word 14. The two unvisited words the test reads are
37
- * rebuilt exactly once per submit by `bfs-unvisited-flags` (PD-18) and maintained here by subtraction: the count is
38
- * subtracted from the SECOND boundary of a submit on and the degree sum from the THIRD on, because a boundary may
39
- * only subtract what the submit's rebuild counted, and the frontier whose degree sum the second boundary holds was
40
- * claimed before the rebuild ran (the rebuild counts the vertices unclaimed when it runs; the frontier rotated in at
41
- * boundary 0 was claimed by the previous submit's last contract, so it was never in the sum; boundary b subtracts
42
- * `next = |F_b|`, inside the sum iff b >= 1, and `degSum = deg(F_{b-1})`, inside it iff b >= 2). The degree sum is
43
- * the "unvisited degree estimate" of the design rather than an exact count for two reasons: it is one level stale
44
- * (a frontier's degree sum is only known once it has been expanded), and a bottom-up level expands nothing, so the
45
- * word stops falling while bottom-up runs and overstates the set afterwards. The bias is one-directional -- an
46
- * overstated m_u makes the switch INTO bottom-up harder, never easier -- and the next submit's rebuild makes it
47
- * exact again. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
48
- * textual edits of it.
49
- *
50
- * Since 2026-09-25 nothing dispatches FROM the slots (G8-F5: Dawn's validation of an indirect dispatch cost about
51
- * 0.4 ms of device time each, whether or not it dispatched anything, and the seven slots of thirty-two recorded
52
- * levels were 97 % of a traversal's wall time). Every level kernel is a direct grid-stride dispatch that reads the
53
- * `path` word (24) this kernel writes -- 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3
54
- * bottom-up, 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2) -- and the SSSP dedupes read
55
- * their counts from words 8 and 9, which role 2 writes. The slots stay as the selector's recorded decision, read
56
- * back by the frontier tests; deleting them with those tests is the follow-up.
32
+ * Beamer's test (P8-T8, PD-21; amended for issue #391), evaluated at every boundary BEFORE the `done` branch (so a
33
+ * switch can be counted at the done boundary too, which the host model of the tests mirrors): top-down switches to
34
+ * bottom-up when `nextDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's
35
+ * `max(1, floor(arcCount / n))` unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up
36
+ * switches back when `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no
37
+ * admitted device reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the
38
+ * direction at 0. Every change is counted in `switches`, the previous direction is word 14. `nextDegreeSum` (word
39
+ * 25) is Beamer's m_f measured EXACTLY: `bfs-next-degree` sums the out-degrees of the vertices a level claims at
40
+ * the end of that level, so the boundary that rotates them in as `next` compares the degree of the frontier it is
41
+ * about to expand -- not, as before the amendment, `frontierDegreeSum` (word 2), the degree of the frontier the
42
+ * previous level EXPANDED, one level stale and 0 after a bottom-up level, which on the 1M / 10M R-MAT missed the
43
+ * switch at the level holding 13.6M of the 21M arcs. Word 2 is still accumulated by the expansion and rotated into
44
+ * word 4 for the inspect seam. The two unvisited words are rebuilt exactly once per submit by `bfs-unvisited-flags`
45
+ * (PD-18) and maintained here by subtraction from the SECOND boundary of a submit on, because a boundary may only
46
+ * subtract what the submit's rebuild counted: the rebuild counts the vertices unclaimed when it runs, the frontier
47
+ * rotated in at boundary 0 was claimed by the previous submit's last level, so it was never in the sums, and
48
+ * boundary b subtracts `next = |F_b|` and `nextDegreeSum = deg(F_b)`, both inside the sums iff b >= 1. Both words
49
+ * are therefore exact at every boundary, bottom-up levels included (the sweep's claims are summed like any other).
50
+ * Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits
51
+ * of it.
57
52
  */
58
- export declare const frontierFinalizeWgsl = "\nfn write_slot_groups(slot: u32, groups: u32, count: u32) { // groups workgroups, split in 2D above the per-dim limit\n var x = groups;\n var y = 1u;\n if (groups > MAX_WORKGROUPS_PER_DIM) {\n x = MAX_WORKGROUPS_PER_DIM;\n y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;\n }\n let base = 4u * (P.slotBase + slot); // 16-byte slots: (x, y, 1, count)\n args[base] = x; args[base + 1u] = y; args[base + 2u] = 1u; args[base + 3u] = count;\n}\nfn write_slot(slot: u32, count: u32) { // one INVOCATION per entry: ceil(count / wg) workgroups\n let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap (indirect-finalize's rule)\n write_slot_groups(slot, groups, count);\n}\nfn zero_slot(slot: u32) {\n let base = 4u * (P.slotBase + slot);\n args[base] = 0u; args[base + 1u] = 0u; args[base + 2u] = 1u; args[base + 3u] = 0u;\n}\n\n@compute @workgroup_size(WG)\nfn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role == 0u) { // the level boundary\n if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end\n for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); } // this slotBase holds the previous submit's args\n return; // no counter word moves (P8-T6's levels formula reads them)\n }\n let finished = atomicLoad(&counters[0]);\n let next = atomicLoad(&counters[1]);\n let degSum = atomicLoad(&counters[2]);\n atomicStore(&counters[3], finished); // prevFrontierCount\n atomicStore(&counters[4], degSum); // prevDegreeSum\n atomicStore(&counters[0], next); // the rotation\n atomicStore(&counters[1], 0u);\n atomicStore(&counters[2], 0u);\n atomicStore(&counters[8], 0u); // edgeCount\n atomicStore(&counters[9], 0u); // edgeCountUnclamped\n atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount\n if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped to 2 (P8-T8, PD-18)\n atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount (exact): F_b was inside the submit's rebuilt sum iff b >= 1\n }\n if (P.firstOfSubmit >= 2u) {\n atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2\n }\n let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0\n atomicStore(&counters[11], level);\n let done = (next == 0u) || (level >= P.maxDepth);\n atomicStore(&counters[15], select(0u, 1u, done));\n var direction = atomicLoad(&counters[14]);\n if (P.mode == 1u) {\n direction = 0u; // top-down only (the test seam)\n } else if (direction == 0u) {\n if (degSum > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing\n } else {\n if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking\n }\n if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches\n var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)\n if (done) {\n for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }\n } else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep\n zero_slot(0u); zero_slot(1u); zero_slot(2u); zero_slot(6u);\n write_slot(3u, (P.n + 31u) / 32u); write_slot(4u, next); write_slot(5u, atomicLoad(&counters[7]));\n path = 3u;\n atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);\n } else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)\n zero_slot(0u); zero_slot(1u); zero_slot(3u); zero_slot(4u); zero_slot(5u); zero_slot(6u);\n write_slot_groups(2u, next, next); // bfs-fused is one WORKGROUP per frontier entry\n path = 2u;\n atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);\n } else {\n zero_slot(2u); zero_slot(3u); zero_slot(4u); zero_slot(5u);\n write_slot(0u, next); // slots 1 and 6 are role 1's\n path = 1u;\n }\n atomicStore(&counters[14], direction);\n atomicStore(&counters[24], path);\n } else if (P.role == 1u) { // the edge queue is filled\n if (args[4u * P.slotBase] == 0u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to size, nothing to count\n zero_slot(1u); zero_slot(6u);\n return;\n }\n let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);\n atomicStore(&counters[8], clamped);\n if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry\n let entries = atomicLoad(&counters[0]);\n zero_slot(1u); write_slot_groups(6u, entries, entries); // one workgroup per frontier entry, as slot 2\n atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry, bfs-contract nothing\n atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);\n atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);\n } else {\n write_slot(1u, clamped); zero_slot(6u);\n atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)\n }\n } else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes\n atomicStore(&counters[8], 0u); // the dedupe counts (words 8 and 9, the SSSP sense) and the path word: nothing unless a pile is chosen below\n atomicStore(&counters[9], 0u);\n atomicStore(&counters[24], 0u);\n if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)\n for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }\n return;\n }\n let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped\n let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped\n if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE\n atomicStore(&counters[15], 2u);\n for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }\n return;\n }\n zero_slot(2u); zero_slot(5u); zero_slot(6u); // role 3 sizes the relax slots once the piles are deduped\n if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn\n atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter\n atomicStore(&counters[14], 0u); // mode 0\n write_slot(0u, nearRaw); write_slot(1u, nearRaw); // dedupe-claim, dedupe-filter over the near half\n atomicStore(&counters[8], nearRaw); // the near dedupe's count word\n atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing\n zero_slot(3u); zero_slot(4u);\n atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)\n } else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile\n let threshold = bitcast<f32>(atomicLoad(&counters[22]));\n let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)\n if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED\n atomicStore(&counters[15], 3u);\n for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }\n return;\n }\n atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below\n atomicStore(&counters[22], bitcast<u32>(raised));\n atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter\n atomicStore(&counters[14], 1u); // mode 1\n zero_slot(0u); zero_slot(1u);\n write_slot(3u, farRaw); write_slot(4u, farRaw); // dedupe-claim, dedupe-filter over the far half\n atomicStore(&counters[9], farRaw); // the far dedupe's count word\n atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing\n atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);\n } else { // both piles empty: finished\n atomicStore(&counters[15], 1u);\n for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }\n }\n } else if (P.role == 3u) { // the piles are deduped: size the relax\n if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 zeroed every slot of the round\n if (atomicLoad(&counters[14]) == 0u) {\n write_slot(2u, atomicLoad(&counters[0])); // the near round over nearIn\n atomicStore(&counters[1], 0u); // the raw near half restarts\n } else {\n write_slot(5u, atomicLoad(&counters[20])); // the pass-through over farIn\n atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far)\n }\n }\n}\n";
53
+ export declare const frontierFinalizeWgsl = "\n@compute @workgroup_size(WG)\nfn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {\n if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)\n if (P.role == 0u) { // the level boundary\n if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end\n atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)\n return;\n }\n let finished = atomicLoad(&counters[0]);\n let next = atomicLoad(&counters[1]);\n let degSum = atomicLoad(&counters[2]);\n let nextDeg = atomicLoad(&counters[25]); // deg(F_b), summed by bfs-next-degree when F_b was claimed (issue #391)\n atomicStore(&counters[3], finished); // prevFrontierCount\n atomicStore(&counters[4], degSum); // prevDegreeSum\n atomicStore(&counters[0], next); // the rotation\n atomicStore(&counters[1], 0u);\n atomicStore(&counters[2], 0u);\n atomicStore(&counters[25], 0u); // the next level's claims sum from 0\n atomicStore(&counters[8], 0u); // edgeCount\n atomicStore(&counters[9], 0u); // edgeCountUnclamped\n atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount\n if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped (P8-T8, PD-18): F_b was inside the submit's rebuilt sums iff b >= 1\n atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount, exact\n atomicStore(&counters[6], atomicLoad(&counters[6]) - nextDeg); // unvisitedDegreeSum, exact (issue #391: no longer one level stale)\n }\n let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0\n atomicStore(&counters[11], level);\n let done = (next == 0u) || (level >= P.maxDepth);\n atomicStore(&counters[15], select(0u, 1u, done));\n var direction = atomicLoad(&counters[14]);\n if (P.mode == 1u) {\n direction = 0u; // top-down only (the test seam)\n } else if (direction == 0u) {\n if (nextDeg > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing, m_f the degree of the frontier about to be expanded\n } else {\n if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking\n }\n if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches\n var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)\n if (done) {\n path = 0u;\n } else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep\n path = 3u;\n atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);\n } else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)\n path = 2u; // bfs-fused: one WORKGROUP per frontier entry\n atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);\n } else {\n path = 1u; // advance-expand, then role 1 and bfs-contract\n }\n atomicStore(&counters[14], direction);\n atomicStore(&counters[24], path);\n } else if (P.role == 1u) { // the edge queue is filled\n if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count\n return;\n }\n let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);\n atomicStore(&counters[8], clamped);\n if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry\n atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing\n atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);\n atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);\n } else {\n atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)\n }\n } else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes\n atomicStore(&counters[8], 0u); // the dedupe counts (words 8 and 9, the SSSP sense) and the path word: nothing unless a pile is chosen below\n atomicStore(&counters[9], 0u);\n atomicStore(&counters[24], 0u);\n if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)\n return;\n }\n let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped\n let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped\n if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE\n atomicStore(&counters[15], 2u);\n return;\n }\n if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn\n atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter\n atomicStore(&counters[14], 0u); // mode 0\n atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)\n atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing\n atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)\n } else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile\n let threshold = bitcast<f32>(atomicLoad(&counters[22]));\n let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)\n if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED\n atomicStore(&counters[15], 3u);\n return;\n }\n atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below\n atomicStore(&counters[22], bitcast<u32>(raised));\n atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter\n atomicStore(&counters[14], 1u); // mode 1\n atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)\n atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing\n atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);\n } else { // both piles empty: finished\n atomicStore(&counters[15], 1u);\n }\n } else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed\n if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round\n if (atomicLoad(&counters[14]) == 0u) {\n atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)\n } else {\n atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)\n }\n }\n}\n";
59
54
  //# sourceMappingURL=frontier-finalize.wgsl.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"frontier-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAwDG;AACH,eAAO,MAAM,oBAAoB,0mWAuJhC,CAAC"}
1
+ {"version":3,"file":"frontier-finalize.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAmDG;AACH,eAAO,MAAM,oBAAoB,s5RA+GhC,CAAC"}
@@ -1,104 +1,80 @@
1
1
  /**
2
2
  * The `frontier-finalize` kernel body (design 5.4, 6 row 7; P8-T4, the P8 plan's PD-3 / PD-23 / DEP-P8-C): the
3
3
  * device-side selector of the frontier family. One workgroup, one lane, no barrier after the early return (spec 3.5
4
- * rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's dispatch sizes become
5
- * known at two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`,
6
- * advances `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`), chooses the path and writes this
7
- * level's seven 16-byte slots at `P.slotBase`; role 1 runs once the edge queue is filled -- clamps `edgeCount` to
8
- * `P.edgeCapacity`, sizes the contract slot, or, when `edgeCountUnclamped` exceeds the capacity, zeroes it and sizes
9
- * the fused-retry slot from `frontierCount` instead (PD-23). Two rules: a boundary that finds `done` set zeroes its
10
- * slots and moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when
11
- * role 0 chose one, which it reads from slot 0's `x` (a storage write of one dispatch is visible to the next of the
12
- * same pass). The `(x, y)` arithmetic is `indirect-finalize`'s verbatim (P4), so a count above 2^32 - wg cannot wrap
13
- * and a group count above MAX_WORKGROUPS_PER_DIM splits in 2D; slots 2 and 6 are sized one WORKGROUP per entry for
14
- * `bfs-fused`; slots 3, 4 and 5 (the bits fill over `ceil(n / 32)` words, the bitset build over the frontier, the sweep
15
- * over the unvisited list) are the bottom-up level's.
4
+ * rule 1). It is recorded TWICE per level, in two roles chosen by `P.role`, because a level's counts become known at
5
+ * two moments: role 0 runs at the START of a level -- rotates `nextFrontierCount` into `frontierCount`, advances
6
+ * `level`, decides `done` (an empty frontier, or `level >= P.maxDepth`) and chooses the path; role 1 runs once the
7
+ * edge queue is filled -- clamps `edgeCount` to `P.edgeCapacity`, or, when `edgeCountUnclamped` exceeds the
8
+ * capacity, switches the path to the fused retry over `frontierCount` (PD-23). The decision is ONE word of the
9
+ * counters block, `path` (word 24): 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3 bottom-up,
10
+ * 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2). Every level kernel is a direct grid-stride
11
+ * dispatch that reads that word first and runs only when it names it (decision record
12
+ * design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md: Dawn's validation of an indirect dispatch cost
13
+ * about 0.4 ms of device time each, and the seven indirect slots this kernel once wrote per level were 97 % of a
14
+ * traversal's wall time; the slots and their args buffer are gone). Two rules: a boundary that finds `done` set
15
+ * moves no counter word (the host records levels past the end); role 1 counts a two-phase level only when role 0
16
+ * chose one, which it reads from the path word (a storage write of one dispatch is visible to the next of the same
17
+ * pass).
16
18
  *
17
19
  * Roles 2 and 3 are the SSSP round boundary of the near-far loop (P8-T9, PD-20), over the same block read in its
18
20
  * SSSP sense (word 1 the raw near half's appends, 21 the raw far half's, 0 and 20 the deduped pile counts, 22 the
19
- * threshold, 23 the delta, 4 the previous threshold, 14 the round's mode) and the same seven slots: role 2 finds a
20
- * non-empty raw near half and sizes the near dedupe (slots 0 and 1, count word 1, output word 0) in mode 0; finds it
21
- * empty and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
21
+ * threshold, 23 the delta, 4 the previous threshold, 14 the round's mode): role 2 finds a non-empty raw near half
22
+ * and sizes the near dedupe (count word 1, output word 0; the dedupe's count in word 8) in mode 0; finds it empty
23
+ * and the far half not, raises the threshold by the delta (one f32 add; an add that returns the threshold
22
24
  * unchanged sets `done 3`, the host's E_UNSUPPORTED), remembers the previous threshold in word 4 and sizes the far
23
- * dedupe (slots 3 and 4, count word 21, output word 20) in mode 1; finds both empty and sets `done 1`; and finds a
24
- * raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in `level` when it
25
- * picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds dispatched; a
26
- * boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed: mode 0 sizes slot 2 (the
27
- * relax over nearIn) from word 0 and restarts the raw near half (word 1 to 0); mode 1 sizes slot 5 (the pass-through
28
- * over farIn) from word 20 and restarts the raw far half (word 21 to 0).
25
+ * dedupe (count word 21, output word 20; the dedupe's count in word 9) in mode 1; finds both empty and sets
26
+ * `done 1`; and finds a raw half above the capacity and sets `done 2` (the host's E_TOO_LARGE). It counts a round in
27
+ * `level` when it picks a mode and NOT at the done boundary, so `level` at the end is the number of relax rounds
28
+ * dispatched; a boundary that finds `done` set obeys rule 1. Role 3 runs once the dedupe has landed and restarts the
29
+ * raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
30
+ * to 0); the relax kernels size themselves from words 0 and 20.
29
31
  *
30
- * Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a switch can be counted at
31
- * the done boundary too, which the host model of the tests mirrors): top-down switches to bottom-up when
32
- * `frontierDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's `max(1, floor(arcCount / n))`
33
- * unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up switches back when
34
- * `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no admitted device
35
- * reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the direction at 0. Every
36
- * change is counted in `switches`, the previous direction is word 14. The two unvisited words the test reads are
37
- * rebuilt exactly once per submit by `bfs-unvisited-flags` (PD-18) and maintained here by subtraction: the count is
38
- * subtracted from the SECOND boundary of a submit on and the degree sum from the THIRD on, because a boundary may
39
- * only subtract what the submit's rebuild counted, and the frontier whose degree sum the second boundary holds was
40
- * claimed before the rebuild ran (the rebuild counts the vertices unclaimed when it runs; the frontier rotated in at
41
- * boundary 0 was claimed by the previous submit's last contract, so it was never in the sum; boundary b subtracts
42
- * `next = |F_b|`, inside the sum iff b >= 1, and `degSum = deg(F_{b-1})`, inside it iff b >= 2). The degree sum is
43
- * the "unvisited degree estimate" of the design rather than an exact count for two reasons: it is one level stale
44
- * (a frontier's degree sum is only known once it has been expanded), and a bottom-up level expands nothing, so the
45
- * word stops falling while bottom-up runs and overstates the set afterwards. The bias is one-directional -- an
46
- * overstated m_u makes the switch INTO bottom-up harder, never easier -- and the next submit's rebuild makes it
47
- * exact again. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
48
- * textual edits of it.
49
- *
50
- * Since 2026-09-25 nothing dispatches FROM the slots (G8-F5: Dawn's validation of an indirect dispatch cost about
51
- * 0.4 ms of device time each, whether or not it dispatched anything, and the seven slots of thirty-two recorded
52
- * levels were 97 % of a traversal's wall time). Every level kernel is a direct grid-stride dispatch that reads the
53
- * `path` word (24) this kernel writes -- 0 nothing (done, or a level past the end), 1 two-phase, 2 fused, 3
54
- * bottom-up, 4 the fused retry (role 1), 5 a near SSSP round, 6 a far one (role 2) -- and the SSSP dedupes read
55
- * their counts from words 8 and 9, which role 2 writes. The slots stay as the selector's recorded decision, read
56
- * back by the frontier tests; deleting them with those tests is the follow-up.
32
+ * Beamer's test (P8-T8, PD-21; amended for issue #391), evaluated at every boundary BEFORE the `done` branch (so a
33
+ * switch can be counted at the done boundary too, which the host model of the tests mirrors): top-down switches to
34
+ * bottom-up when `nextDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's
35
+ * `max(1, floor(arcCount / n))` unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up
36
+ * switches back when `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no
37
+ * admitted device reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the
38
+ * direction at 0. Every change is counted in `switches`, the previous direction is word 14. `nextDegreeSum` (word
39
+ * 25) is Beamer's m_f measured EXACTLY: `bfs-next-degree` sums the out-degrees of the vertices a level claims at
40
+ * the end of that level, so the boundary that rotates them in as `next` compares the degree of the frontier it is
41
+ * about to expand -- not, as before the amendment, `frontierDegreeSum` (word 2), the degree of the frontier the
42
+ * previous level EXPANDED, one level stale and 0 after a bottom-up level, which on the 1M / 10M R-MAT missed the
43
+ * switch at the level holding 13.6M of the 21M arcs. Word 2 is still accumulated by the expansion and rotated into
44
+ * word 4 for the inspect seam. The two unvisited words are rebuilt exactly once per submit by `bfs-unvisited-flags`
45
+ * (PD-18) and maintained here by subtraction from the SECOND boundary of a submit on, because a boundary may only
46
+ * subtract what the submit's rebuild counted: the rebuild counts the vertices unclaimed when it runs, the frontier
47
+ * rotated in at boundary 0 was claimed by the previous submit's last level, so it was never in the sums, and
48
+ * boundary b subtracts `next = |F_b|` and `nextDegreeSum = deg(F_b)`, both inside the sums iff b >= 1. Both words
49
+ * are therefore exact at every boundary, bottom-up levels included (the sweep's claims are summed like any other).
50
+ * Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits
51
+ * of it.
57
52
  */
58
53
  export const frontierFinalizeWgsl = /* wgsl */ `
59
- fn write_slot_groups(slot: u32, groups: u32, count: u32) { // groups workgroups, split in 2D above the per-dim limit
60
- var x = groups;
61
- var y = 1u;
62
- if (groups > MAX_WORKGROUPS_PER_DIM) {
63
- x = MAX_WORKGROUPS_PER_DIM;
64
- y = (groups + MAX_WORKGROUPS_PER_DIM - 1u) / MAX_WORKGROUPS_PER_DIM;
65
- }
66
- let base = 4u * (P.slotBase + slot); // 16-byte slots: (x, y, 1, count)
67
- args[base] = x; args[base + 1u] = y; args[base + 2u] = 1u; args[base + 3u] = count;
68
- }
69
- fn write_slot(slot: u32, count: u32) { // one INVOCATION per entry: ceil(count / wg) workgroups
70
- let groups = count / P.wg + select(0u, 1u, count % P.wg != 0u); // ceil(count / wg) without the u32 wrap (indirect-finalize's rule)
71
- write_slot_groups(slot, groups, count);
72
- }
73
- fn zero_slot(slot: u32) {
74
- let base = 4u * (P.slotBase + slot);
75
- args[base] = 0u; args[base + 1u] = 0u; args[base + 2u] = 1u; args[base + 3u] = 0u;
76
- }
77
-
78
54
  @compute @workgroup_size(WG)
79
55
  fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
80
56
  if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
81
57
  if (P.role == 0u) { // the level boundary
82
58
  if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op level the host recorded past the end
83
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); } // this slotBase holds the previous submit's args
84
- return; // no counter word moves (P8-T6's levels formula reads them)
59
+ atomicStore(&counters[24], 0u); // the path word is the only word that moves (P8-T6's levels formula reads the rest)
60
+ return;
85
61
  }
86
62
  let finished = atomicLoad(&counters[0]);
87
63
  let next = atomicLoad(&counters[1]);
88
64
  let degSum = atomicLoad(&counters[2]);
65
+ let nextDeg = atomicLoad(&counters[25]); // deg(F_b), summed by bfs-next-degree when F_b was claimed (issue #391)
89
66
  atomicStore(&counters[3], finished); // prevFrontierCount
90
67
  atomicStore(&counters[4], degSum); // prevDegreeSum
91
68
  atomicStore(&counters[0], next); // the rotation
92
69
  atomicStore(&counters[1], 0u);
93
70
  atomicStore(&counters[2], 0u);
71
+ atomicStore(&counters[25], 0u); // the next level's claims sum from 0
94
72
  atomicStore(&counters[8], 0u); // edgeCount
95
73
  atomicStore(&counters[9], 0u); // edgeCountUnclamped
96
74
  atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount
97
- if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped to 2 (P8-T8, PD-18)
98
- atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount (exact): F_b was inside the submit's rebuilt sum iff b >= 1
99
- }
100
- if (P.firstOfSubmit >= 2u) {
101
- atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2
75
+ if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped (P8-T8, PD-18): F_b was inside the submit's rebuilt sums iff b >= 1
76
+ atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount, exact
77
+ atomicStore(&counters[6], atomicLoad(&counters[6]) - nextDeg); // unvisitedDegreeSum, exact (issue #391: no longer one level stale)
102
78
  }
103
79
  let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0
104
80
  atomicStore(&counters[11], level);
@@ -108,46 +84,36 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
108
84
  if (P.mode == 1u) {
109
85
  direction = 0u; // top-down only (the test seam)
110
86
  } else if (direction == 0u) {
111
- if (degSum > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing
87
+ if (nextDeg > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing, m_f the degree of the frontier about to be expanded
112
88
  } else {
113
89
  if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking
114
90
  }
115
91
  if (direction != atomicLoad(&counters[14])) { atomicStore(&counters[13], atomicLoad(&counters[13]) + 1u); } // switches
116
92
  var path = 0u; // word 24: what the level's kernels run (0 nothing, 1 two-phase, 2 fused, 3 bottom-up; role 1 writes 4 for the retry)
117
93
  if (done) {
118
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
94
+ path = 0u;
119
95
  } else if (direction == 1u) { // the bottom-up level (P8-T8): the bits fill, the bitset build, the sweep
120
- zero_slot(0u); zero_slot(1u); zero_slot(2u); zero_slot(6u);
121
- write_slot(3u, (P.n + 31u) / 32u); write_slot(4u, next); write_slot(5u, atomicLoad(&counters[7]));
122
96
  path = 3u;
123
97
  atomicStore(&counters[19], atomicLoad(&counters[19]) + 1u);
124
98
  } else if (next < P.fusedMax) { // P8-T7 makes this branch reachable (fusedMax is 0 until then)
125
- zero_slot(0u); zero_slot(1u); zero_slot(3u); zero_slot(4u); zero_slot(5u); zero_slot(6u);
126
- write_slot_groups(2u, next, next); // bfs-fused is one WORKGROUP per frontier entry
127
- path = 2u;
99
+ path = 2u; // bfs-fused: one WORKGROUP per frontier entry
128
100
  atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
129
101
  } else {
130
- zero_slot(2u); zero_slot(3u); zero_slot(4u); zero_slot(5u);
131
- write_slot(0u, next); // slots 1 and 6 are role 1's
132
- path = 1u;
102
+ path = 1u; // advance-expand, then role 1 and bfs-contract
133
103
  }
134
104
  atomicStore(&counters[14], direction);
135
105
  atomicStore(&counters[24], path);
136
106
  } else if (P.role == 1u) { // the edge queue is filled
137
- if (args[4u * P.slotBase] == 0u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to size, nothing to count
138
- zero_slot(1u); zero_slot(6u);
107
+ if (atomicLoad(&counters[24]) != 1u) { // role 0 did not choose the two-phase path (done, fused or bottom-up): nothing to clamp, nothing to count
139
108
  return;
140
109
  }
141
110
  let clamped = min(atomicLoad(&counters[8]), P.edgeCapacity);
142
111
  atomicStore(&counters[8], clamped);
143
112
  if (atomicLoad(&counters[9]) > P.edgeCapacity) { // PD-23: the fused retry
144
- let entries = atomicLoad(&counters[0]);
145
- zero_slot(1u); write_slot_groups(6u, entries, entries); // one workgroup per frontier entry, as slot 2
146
- atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry, bfs-contract nothing
113
+ atomicStore(&counters[24], 4u); // the path word: bfs-fused runs the retry over frontierCount, bfs-contract nothing
147
114
  atomicStore(&counters[10], atomicLoad(&counters[10]) + 1u);
148
115
  atomicStore(&counters[17], atomicLoad(&counters[17]) + 1u);
149
116
  } else {
150
- write_slot(1u, clamped); zero_slot(6u);
151
117
  atomicStore(&counters[18], atomicLoad(&counters[18]) + 1u); // twoPhaseLevels counts the CHOICE role 0 made, even for zero edges (P8-T7 Step 4's invariant)
152
118
  }
153
119
  } else if (P.role == 2u) { // the SSSP round boundary (P8-T9, PD-20): which pile this round relaxes
@@ -155,54 +121,43 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
155
121
  atomicStore(&counters[9], 0u);
156
122
  atomicStore(&counters[24], 0u);
157
123
  if (atomicLoad(&counters[15]) != 0u) { // done already: a no-op round the host recorded past the end (rule 1)
158
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
159
124
  return;
160
125
  }
161
126
  let nearRaw = atomicLoad(&counters[1]); // the raw near half's appends, unclamped
162
127
  let farRaw = atomicLoad(&counters[21]); // the raw far half's appends, unclamped
163
128
  if (nearRaw > P.edgeCapacity || farRaw > P.edgeCapacity) { // a pile overflowed its half: the host raises E_TOO_LARGE
164
129
  atomicStore(&counters[15], 2u);
165
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
166
130
  return;
167
131
  }
168
- zero_slot(2u); zero_slot(5u); zero_slot(6u); // role 3 sizes the relax slots once the piles are deduped
169
132
  if (nearRaw != 0u) { // a near round: dedupe the near half into nearIn
170
133
  atomicStore(&counters[0], 0u); // the deduped near count, accumulated by dedupe-filter
171
134
  atomicStore(&counters[14], 0u); // mode 0
172
- write_slot(0u, nearRaw); write_slot(1u, nearRaw); // dedupe-claim, dedupe-filter over the near half
173
- atomicStore(&counters[8], nearRaw); // the near dedupe's count word
135
+ atomicStore(&counters[8], nearRaw); // the near dedupe's count word (dedupe-claim, dedupe-filter over the near half)
174
136
  atomicStore(&counters[24], 5u); // the path word: sssp-relax role 0 runs, role 1 nothing
175
- zero_slot(3u); zero_slot(4u);
176
137
  atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u); // rounds dispatched (the done boundary is not counted)
177
138
  } else if (farRaw != 0u) { // the near pile is empty: raise the threshold and re-bucket the far pile
178
139
  let threshold = bitcast<f32>(atomicLoad(&counters[22]));
179
140
  let raised = threshold + bitcast<f32>(atomicLoad(&counters[23])); // ONE f32 add on the bit patterns (PD-9)
180
141
  if (raised == threshold) { // the delta is below the threshold's ulp: the host raises E_UNSUPPORTED
181
142
  atomicStore(&counters[15], 3u);
182
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
183
143
  return;
184
144
  }
185
145
  atomicStore(&counters[4], atomicLoad(&counters[22])); // prevThresholdBits: what the pass-through drops below
186
146
  atomicStore(&counters[22], bitcast<u32>(raised));
187
147
  atomicStore(&counters[20], 0u); // the deduped far count, accumulated by dedupe-filter
188
148
  atomicStore(&counters[14], 1u); // mode 1
189
- zero_slot(0u); zero_slot(1u);
190
- write_slot(3u, farRaw); write_slot(4u, farRaw); // dedupe-claim, dedupe-filter over the far half
191
- atomicStore(&counters[9], farRaw); // the far dedupe's count word
149
+ atomicStore(&counters[9], farRaw); // the far dedupe's count word (dedupe-claim, dedupe-filter over the far half)
192
150
  atomicStore(&counters[24], 6u); // the path word: sssp-relax role 1 runs, role 0 nothing
193
151
  atomicStore(&counters[11], atomicLoad(&counters[11]) + 1u);
194
152
  } else { // both piles empty: finished
195
153
  atomicStore(&counters[15], 1u);
196
- for (var s = 0u; s < 7u; s = s + 1u) { zero_slot(s); }
197
154
  }
198
- } else if (P.role == 3u) { // the piles are deduped: size the relax
199
- if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 zeroed every slot of the round
155
+ } else if (P.role == 3u) { // the piles are deduped: restart the raw half the round consumed
156
+ if (atomicLoad(&counters[15]) != 0u) { return; } // role 2 chose no pile this round
200
157
  if (atomicLoad(&counters[14]) == 0u) {
201
- write_slot(2u, atomicLoad(&counters[0])); // the near round over nearIn
202
- atomicStore(&counters[1], 0u); // the raw near half restarts
158
+ atomicStore(&counters[1], 0u); // the raw near half restarts (sssp-relax role 0 sizes itself from word 0)
203
159
  } else {
204
- write_slot(5u, atomicLoad(&counters[20])); // the pass-through over farIn
205
- atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far)
160
+ atomicStore(&counters[21], 0u); // the raw far half restarts (the pass-through re-appends what stays far; role 1 sizes itself from word 20)
206
161
  }
207
162
  }
208
163
  }
@@ -1 +1 @@
1
- {"version":3,"file":"frontier-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAwDG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAuJ9C,CAAC"}
1
+ {"version":3,"file":"frontier-finalize.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/frontier-finalize.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAmDG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA+G9C,CAAC"}
@@ -1,8 +1,9 @@
1
1
  /**
2
2
  * G1, the `grid-cell-key` kernel body (spec 7.7; P4-T8): the finest cell of every node from the state's robust extent,
3
3
  * `floor((p - gridMin) * invCellSize)` (a multiply, correctly rounded everywhere: PD-10), linearised when every axis
4
- * is in [0, G) and the outside pseudo-cell `cells` otherwise; `cellVal[i] = i`. The clamp before the floor keeps a
4
+ * is in [0, G), and otherwise one of the 2^dim outside pseudo-cells `cells + orthant`, the orthant of the cell about the
5
+ * grid centre (bit a set when `c[a] >= G / 2`; issue #90); `cellVal[i] = i`. The clamp before the floor keeps a
5
6
  * far-away or NaN coordinate out of an out-of-range float-to-int conversion. Body only; normative text.
6
7
  */
7
- export declare const gridCellKeyWgsl = "\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\n\n@compute @workgroup_size(WG)\nfn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n if (i >= P.n) { return; } // no barrier follows\n let cells = grid_cells();\n let gf = f32(P.gridMax);\n let q = (pos[i].xyz - S.gridMin.xyz) * S.invCellSize; // PD-10: never a division\n let c = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));\n let g = i32(P.gridMax);\n var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;\n if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }\n var key = cells; // the outside pseudo-cell (7.7)\n if (inside) {\n key = u32(c.x) + P.gridMax * u32(c.y);\n if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }\n }\n cellKey[i] = key;\n cellVal[i] = i;\n}\n";
8
+ export declare const gridCellKeyWgsl = "\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\n\n@compute @workgroup_size(WG)\nfn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let i = linear_id(wid, lid.x);\n if (i >= P.n) { return; } // no barrier follows\n let cells = grid_cells();\n let gf = f32(P.gridMax);\n let q = (pos[i].xyz - S.gridMin.xyz) * S.invCellSize; // PD-10: never a division\n let c = vec3<i32>(floor(clamp(q, vec3f(-1.0), vec3f(gf + 1.0))));\n let g = i32(P.gridMax);\n var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;\n if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }\n var key = cells + select(0u, 1u, c.x >= g / 2) + select(0u, 2u, c.y >= g / 2); // an outside pseudo-cell: its orthant (issue #90)\n if (P.dim == 3u) { key = key + select(0u, 4u, c.z >= g / 2); }\n if (inside) {\n key = u32(c.x) + P.gridMax * u32(c.y);\n if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }\n }\n cellKey[i] = key;\n cellVal[i] = i;\n}\n";
8
9
  //# sourceMappingURL=grid-cell-key.wgsl.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"grid-cell-key.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-cell-key.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,eAAO,MAAM,eAAe,uhCAsB3B,CAAC"}
1
+ {"version":3,"file":"grid-cell-key.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-cell-key.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,eAAO,MAAM,eAAe,+nCAuB3B,CAAC"}
@@ -1,7 +1,8 @@
1
1
  /**
2
2
  * G1, the `grid-cell-key` kernel body (spec 7.7; P4-T8): the finest cell of every node from the state's robust extent,
3
3
  * `floor((p - gridMin) * invCellSize)` (a multiply, correctly rounded everywhere: PD-10), linearised when every axis
4
- * is in [0, G) and the outside pseudo-cell `cells` otherwise; `cellVal[i] = i`. The clamp before the floor keeps a
4
+ * is in [0, G), and otherwise one of the 2^dim outside pseudo-cells `cells + orthant`, the orthant of the cell about the
5
+ * grid centre (bit a set when `c[a] >= G / 2`; issue #90); `cellVal[i] = i`. The clamp before the floor keeps a
5
6
  * far-away or NaN coordinate out of an out-of-range float-to-int conversion. Body only; normative text.
6
7
  */
7
8
  export const gridCellKeyWgsl = /* wgsl */ `
@@ -18,7 +19,8 @@ fn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocatio
18
19
  let g = i32(P.gridMax);
19
20
  var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;
20
21
  if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }
21
- var key = cells; // the outside pseudo-cell (7.7)
22
+ var key = cells + select(0u, 1u, c.x >= g / 2) + select(0u, 2u, c.y >= g / 2); // an outside pseudo-cell: its orthant (issue #90)
23
+ if (P.dim == 3u) { key = key + select(0u, 4u, c.z >= g / 2); }
22
24
  if (inside) {
23
25
  key = u32(c.x) + P.gridMax * u32(c.y);
24
26
  if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }
@@ -1 +1 @@
1
- {"version":3,"file":"grid-cell-key.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-cell-key.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,CAAC,MAAM,eAAe,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;CAsBzC,CAAC"}
1
+ {"version":3,"file":"grid-cell-key.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-cell-key.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,MAAM,CAAC,MAAM,eAAe,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;;;CAuBzC,CAAC"}
@@ -1,8 +1,9 @@
1
1
  /**
2
- * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the pseudo-cell included; the
2
+ * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the 2^dim outside pseudo-cells
3
+ * included (issue #90); the
3
4
  * mass-weighted position sum of a cell's sorted range in index order (no atomics: deterministic), the largest
4
5
  * occupancy into hubCounters[1], and cells above GRID_HUB_CELL entries appended to hubList for G4b (PD-13). Body
5
6
  * only; normative text.
6
7
  */
7
- export declare const gridCentroidWgsl = "\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\n\n@compute @workgroup_size(WG)\nfn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let c = linear_id(wid, lid.x);\n if (c > grid_cells()) { return; } // cells [0, cells]: the pseudo-cell is index cells; no barrier follows\n let start = cellStart[c];\n let count = cellStart[c + 1u] - start;\n atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration\n if (count > GRID_HUB_CELL) { // a hub cell: G4b sums it (PD-13)\n hubList[atomicAdd(&hubCounters[0], 1u)] = c;\n return;\n }\n var acc = vec4f(0.0);\n for (var k = start; k < start + count; k = k + 1u) { // sorted order: deterministic\n let p = pos[sortedIdx[k]];\n acc = acc + vec4f(p.xyz * p.w, p.w); // (sum m x, sum m y, sum m z, sum m)\n }\n pyramid[c] = acc;\n}\n";
8
+ export declare const gridCentroidWgsl = "\nfn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.dim == 3u); }\n\n@compute @workgroup_size(WG)\nfn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let c = linear_id(wid, lid.x);\n if (c >= grid_cells() + select(4u, 8u, P.dim == 3u)) { return; } // cells [0, cells + 2^dim): the pseudo-cells follow the real ones; no barrier follows\n let start = cellStart[c];\n let count = cellStart[c + 1u] - start;\n atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration\n if (count > GRID_HUB_CELL) { // a hub cell: G4b sums it (PD-13)\n hubList[atomicAdd(&hubCounters[0], 1u)] = c;\n return;\n }\n var acc = vec4f(0.0);\n for (var k = start; k < start + count; k = k + 1u) { // sorted order: deterministic\n let p = pos[sortedIdx[k]];\n acc = acc + vec4f(p.xyz * p.w, p.w); // (sum m x, sum m y, sum m z, sum m)\n }\n pyramid[c] = acc;\n}\n";
8
9
  //# sourceMappingURL=grid-centroid.wgsl.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"grid-centroid.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,eAAO,MAAM,gBAAgB,+jCAqB5B,CAAC"}
1
+ {"version":3,"file":"grid-centroid.wgsl.d.ts","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,eAAO,MAAM,gBAAgB,klCAqB5B,CAAC"}
@@ -1,5 +1,6 @@
1
1
  /**
2
- * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the pseudo-cell included; the
2
+ * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the 2^dim outside pseudo-cells
3
+ * included (issue #90); the
3
4
  * mass-weighted position sum of a cell's sorted range in index order (no atomics: deterministic), the largest
4
5
  * occupancy into hubCounters[1], and cells above GRID_HUB_CELL entries appended to hubList for G4b (PD-13). Body
5
6
  * only; normative text.
@@ -10,7 +11,7 @@ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.
10
11
  @compute @workgroup_size(WG)
11
12
  fn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
12
13
  let c = linear_id(wid, lid.x);
13
- if (c > grid_cells()) { return; } // cells [0, cells]: the pseudo-cell is index cells; no barrier follows
14
+ if (c >= grid_cells() + select(4u, 8u, P.dim == 3u)) { return; } // cells [0, cells + 2^dim): the pseudo-cells follow the real ones; no barrier follows
14
15
  let start = cellStart[c];
15
16
  let count = cellStart[c + 1u] - start;
16
17
  atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration
@@ -1 +1 @@
1
- {"version":3,"file":"grid-centroid.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;CAqB1C,CAAC"}
1
+ {"version":3,"file":"grid-centroid.wgsl.js","sourceRoot":"","sources":["../../../src/wgsl/grid-centroid.wgsl.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,UAAU,CAAC;;;;;;;;;;;;;;;;;;;;;CAqB1C,CAAC"}
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * G5, the `grid-downsample` kernel body (spec 7.7; P4-T9): one dispatch per coarser level; every parent cell is the
3
3
  * sum of its 4 (2D) or 8 (3D) children at the level below, read at P.childBase and written at P.parentBase (the
4
- * pseudo-cell, index cells of level 0, is never a child). No atomics. Body only; normative text.
4
+ * pseudo-cells, indices cells .. of level 0, are never children). No atomics. Body only; normative text.
5
5
  */
6
6
  export declare const gridDownsampleWgsl = "\n@compute @workgroup_size(WG)\nfn grid_downsample(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {\n let pc = linear_id(wid, lid.x); // the parent cell inside its level\n if (pc >= P.parentCells) { return; } // no barrier follows\n let side = P.parentSide;\n let cs = 2u * side; // the child level's side\n let px = pc % side;\n let py = (pc / side) % side;\n let pz = pc / (side * side);\n var acc = vec4f(0.0);\n for (var dz = 0u; dz < P.depth; dz = dz + 1u) {\n for (var dy = 0u; dy < 2u; dy = dy + 1u) {\n for (var dx = 0u; dx < 2u; dx = dx + 1u) {\n let child = (2u * px + dx) + cs * ((2u * py + dy) + cs * (2u * pz + dz));\n acc = acc + pyramid[P.childBase + child];\n }\n }\n }\n pyramid[P.parentBase + pc] = acc;\n}\n";
7
7
  //# sourceMappingURL=grid-downsample.wgsl.d.ts.map