@graphty/webgpu-graph-algorithms 0.6.3 → 0.6.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. package/README.md +62 -32
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-BXqgCifx.js → context-hzGggHeM.js} +68 -24
  4. package/dist/chunks/context-hzGggHeM.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/accelerator.d.ts +8 -6
  7. package/dist/src/accelerator.d.ts.map +1 -1
  8. package/dist/src/accelerator.js +57 -6
  9. package/dist/src/accelerator.js.map +1 -1
  10. package/dist/src/algorithms/bellman-ford.d.ts +60 -0
  11. package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
  12. package/dist/src/algorithms/bellman-ford.js +301 -0
  13. package/dist/src/algorithms/bellman-ford.js.map +1 -0
  14. package/dist/src/algorithms/bfs.d.ts +67 -0
  15. package/dist/src/algorithms/bfs.d.ts.map +1 -0
  16. package/dist/src/algorithms/bfs.js +534 -0
  17. package/dist/src/algorithms/bfs.js.map +1 -0
  18. package/dist/src/algorithms/closeness.d.ts +53 -0
  19. package/dist/src/algorithms/closeness.d.ts.map +1 -0
  20. package/dist/src/algorithms/closeness.js +323 -0
  21. package/dist/src/algorithms/closeness.js.map +1 -0
  22. package/dist/src/algorithms/scope.d.ts +3 -1
  23. package/dist/src/algorithms/scope.d.ts.map +1 -1
  24. package/dist/src/algorithms/scope.js +1 -0
  25. package/dist/src/algorithms/scope.js.map +1 -1
  26. package/dist/src/algorithms/sssp.d.ts +72 -0
  27. package/dist/src/algorithms/sssp.d.ts.map +1 -0
  28. package/dist/src/algorithms/sssp.js +586 -0
  29. package/dist/src/algorithms/sssp.js.map +1 -0
  30. package/dist/src/constants.d.ts +10 -0
  31. package/dist/src/constants.d.ts.map +1 -1
  32. package/dist/src/constants.js +10 -0
  33. package/dist/src/constants.js.map +1 -1
  34. package/dist/src/index.d.ts +8 -2
  35. package/dist/src/index.d.ts.map +1 -1
  36. package/dist/src/index.js +7 -1
  37. package/dist/src/index.js.map +1 -1
  38. package/dist/src/kernel/prelude.d.ts +4 -4
  39. package/dist/src/kernel/prelude.d.ts.map +1 -1
  40. package/dist/src/kernel/prelude.js +39 -5
  41. package/dist/src/kernel/prelude.js.map +1 -1
  42. package/dist/src/kernel/uniform-ring.d.ts +8 -0
  43. package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
  44. package/dist/src/kernel/uniform-ring.js +13 -0
  45. package/dist/src/kernel/uniform-ring.js.map +1 -1
  46. package/dist/src/kernels.d.ts +45 -4
  47. package/dist/src/kernels.d.ts.map +1 -1
  48. package/dist/src/kernels.js +368 -3
  49. package/dist/src/kernels.js.map +1 -1
  50. package/dist/src/primitives/advance.d.ts +63 -0
  51. package/dist/src/primitives/advance.d.ts.map +1 -0
  52. package/dist/src/primitives/advance.js +95 -0
  53. package/dist/src/primitives/advance.js.map +1 -0
  54. package/dist/src/primitives/compact.d.ts +89 -0
  55. package/dist/src/primitives/compact.d.ts.map +1 -0
  56. package/dist/src/primitives/compact.js +233 -0
  57. package/dist/src/primitives/compact.js.map +1 -0
  58. package/dist/src/primitives/core-shape.d.ts +22 -1
  59. package/dist/src/primitives/core-shape.d.ts.map +1 -1
  60. package/dist/src/primitives/core-shape.js +33 -3
  61. package/dist/src/primitives/core-shape.js.map +1 -1
  62. package/dist/src/primitives/frontier.d.ts +151 -0
  63. package/dist/src/primitives/frontier.d.ts.map +1 -0
  64. package/dist/src/primitives/frontier.js +250 -0
  65. package/dist/src/primitives/frontier.js.map +1 -0
  66. package/dist/src/types/accelerator.d.ts +16 -7
  67. package/dist/src/types/accelerator.d.ts.map +1 -1
  68. package/dist/src/types/traversal.d.ts +53 -0
  69. package/dist/src/types/traversal.d.ts.map +1 -0
  70. package/dist/src/types/traversal.js +10 -0
  71. package/dist/src/types/traversal.js.map +1 -0
  72. package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
  73. package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
  74. package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
  75. package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
  76. package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
  77. package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
  78. package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
  79. package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
  80. package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
  81. package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
  82. package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
  83. package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
  84. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
  85. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
  86. package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
  87. package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
  88. package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
  89. package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
  90. package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
  91. package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
  92. package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
  93. package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
  94. package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
  95. package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
  96. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
  97. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
  98. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
  99. package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
  100. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
  101. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
  102. package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
  103. package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
  104. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
  105. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
  106. package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
  107. package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
  108. package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
  109. package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
  110. package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
  111. package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
  112. package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
  113. package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
  114. package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
  115. package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
  116. package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
  117. package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
  118. package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
  119. package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
  120. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +53 -0
  121. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
  122. package/dist/src/wgsl/frontier-finalize.wgsl.js +164 -0
  123. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
  124. package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
  125. package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
  126. package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
  127. package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
  128. package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
  129. package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
  130. package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
  131. package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
  132. package/dist/webgpu-graph-algorithms.js +3155 -384
  133. package/dist/webgpu-graph-algorithms.js.map +1 -1
  134. package/package.json +5 -4
  135. package/src/accelerator.ts +65 -7
  136. package/src/algorithms/bellman-ford.ts +387 -0
  137. package/src/algorithms/bfs.ts +626 -0
  138. package/src/algorithms/closeness.ts +395 -0
  139. package/src/algorithms/scope.ts +4 -1
  140. package/src/algorithms/sssp.ts +768 -0
  141. package/src/constants.ts +10 -0
  142. package/src/index.ts +14 -1
  143. package/src/kernel/prelude.ts +39 -4
  144. package/src/kernel/uniform-ring.ts +14 -0
  145. package/src/kernels.ts +447 -6
  146. package/src/primitives/advance.ts +131 -0
  147. package/src/primitives/compact.ts +323 -0
  148. package/src/primitives/core-shape.ts +41 -3
  149. package/src/primitives/frontier.ts +372 -0
  150. package/src/types/accelerator.ts +18 -5
  151. package/src/types/traversal.ts +56 -0
  152. package/src/wgsl/advance-expand.wgsl.ts +68 -0
  153. package/src/wgsl/bf-relax.wgsl.ts +57 -0
  154. package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
  155. package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
  156. package/src/wgsl/bfs-contract.wgsl.ts +54 -0
  157. package/src/wgsl/bfs-fused.wgsl.ts +77 -0
  158. package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
  159. package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
  160. package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
  161. package/src/wgsl/compact-scatter.wgsl.ts +16 -0
  162. package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
  163. package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
  164. package/src/wgsl/frontier-finalize.wgsl.ts +163 -0
  165. package/src/wgsl/sssp-pred.wgsl.ts +79 -0
  166. package/src/wgsl/sssp-relax.wgsl.ts +71 -0
  167. package/dist/chunks/context-BXqgCifx.js.map +0 -1
@@ -0,0 +1,626 @@
1
+ /**
2
+ * Breadth-first search on the device (design 8.4, 3.3 line 807, 9.7; P8-T6 / P8-T7 / P8-T8, the P8 plan's PD-5 /
3
+ * PD-6 / PD-7 / PD-14 / PD-18 / PD-21 / PD-23 / PD-24 / PD-26 / DEP-P8-B): the direction-optimizing traversal over
4
+ * the `Frontier` of P8-T4, every per-level choice made ON THE DEVICE. Every level is eight recorded dispatches, all
5
+ * DIRECT (2026-09-25, docs/decisions/G8.md G8-F5: Dawn validates every `dispatchWorkgroupsIndirect` with an internal
6
+ * clamp pass costing about 0.4 ms of device time whether or not it dispatches anything, in Node and in Chromium
7
+ * alike, and that was 97 % of a traversal's wall time) -- `frontier-finalize` role 0 (the level boundary: rotates
8
+ * the counts, advances `level`, decides `done`, evaluates Beamer's test and CHOOSES, writing the choice into the
9
+ * block's `path` word: 3 bottom-up, 2 fused for a top-down frontier below `fusedMax` entries, 1 two-phase for any
10
+ * other, 0 done), `advance-expand` (the frontier's rows into the edge queue), `frontier-finalize` role 1 (clamps
11
+ * `edgeCount`, or, when the edge queue overflowed, sets `path` 4 for the fused retry, PD-23), `bfs-contract` (the
12
+ * claim `atomicMin(&depth[v], level + 1)`, the winners packed into the next vertex queue), `bfs-fused` (one workgroup
13
+ * per frontier entry with the same claim inline and no edge queue; the fused level and the retry alike), then the
14
+ * bottom-up trio: a `fill` zeroing the frontier bitset, `bfs-bitset-build` setting the frontier's bits, and
15
+ * `bfs-bottom-up` sweeping the unvisited list over the REVERSE core (each unvisited vertex reads its in-neighbours
16
+ * until the first one in the bitset and claims itself). Every kernel is a grid-stride dispatch of a host-planned
17
+ * grid (`planGridStride`) that loops to its count word and reads the path word first, so exactly one path does the
18
+ * level's work and the others cost one uniform load per workgroup. The unvisited set Beamer's test is against (PD-18) is rebuilt exactly once per submit, before
19
+ * the levels, by `bfs-unvisited-flags` plus `compact` over an iota queue, and maintained between rebuilds by
20
+ * subtraction inside the selector (whose JSDoc states the boundary rule). The host records `MAX_LEVELS_PER_SUBMIT`
21
+ * levels into ONE command buffer, submits, and reads four bytes, the `done` word (PD-7): a road network has
22
+ * thousands of levels and a per-level `mapAsync` would be slower than the CPU. A level recorded past the end is a
23
+ * no-op (its boundary finds `done` set, writes path 0 and moves no counter), so the loop needs no diameter and
24
+ * ends on `done`; a traversal has at most `n` levels, so more submits than that is E_VALIDATION, never a hang. The
25
+ * thresholds are uniform fields (`FUSED_FRONTIER_MAX`; alpha derived as `max(1, floor(arcCount / n))`, PD-21;
26
+ * `BEAMER_BETA`; `mode 1` pinning top-down -- each unless the tuning says otherwise), so a test forces any path
27
+ * without recompiling; the block's `fusedLevels` / `twoPhaseLevels` / `bottomUpLevels` / `overflowLevels` /
28
+ * `switches` words record which path each level took.
29
+ *
30
+ * There is no dedupe (DEP-P8-B, PD-5): the edge queue holds duplicates -- a vertex with three frontier neighbours
31
+ * appears three times -- but for one level exactly one invocation observes `INVALID_INDEX` at `depth[v]` and
32
+ * exactly that one appends `v`, so the next vertex frontier is duplicate-free by construction and an ownership
33
+ * dedupe would have nothing to remove. `parent` is not written by the claim (PD-24): `sssp-pred` in `MODE 1` runs
34
+ * once over the settled depths and writes the SMALLEST `u` with `depth[u] + 1 == depth[v]` and an arc `u -> v`, so
35
+ * it is bitwise reproducible where a claim-time parent would be a scheduling accident. `order` (PD-14) is one
36
+ * stable radix sort of the node indices by `depth`: grouped by level, ascending by index within a level,
37
+ * `INVALID_INDEX` last, and the first `visitedCount` values are the result. The result batch stages every array
38
+ * into one staging slot, so a traversal maps `ceil(levels / 32) + 1` buffers in all.
39
+ *
40
+ * `maxDepth` reaches the device as a `u32` field of `FrontierParams`, and the uniform writer refuses anything that
41
+ * is not an integer in `[0, 2^32 - 1]`; the host therefore normalises it first, to exactly the CPU port's rule
42
+ * (`algorithms/src/indexed/bfs.ts`: a node at depth `d >= maxDepth` is reached and not expanded, unbounded when
43
+ * absent). The tuning entry `bfsWithTuning` (PD-26) is what the tests drive; nothing public exposes it.
44
+ *
45
+ * A core whose `colIdx` exceeds `maxStorageBufferBindingSize` is bound as arc windows and EXECUTED (P8-T12, DEP-P8-E
46
+ * lifted for the frontier family): `advance-expand`, `bfs-fused` and `sssp-pred` run once per window of the
47
+ * forward core and `bfs-bottom-up` once per window of the reverse core, every window its own dispatch with its own
48
+ * `FrontierParams` record carrying the window's owned arc range (`coreWindows`), and the claims, being `atomicMin`s
49
+ * and an `INVALID_INDEX` entry test, are idempotent across windows. The ring is sized per run by `bfsRingSlots`.
50
+ */
51
+
52
+ import { type GraphSnapshot, INVALID_INDEX, type U32 } from "@graphty/graph-format";
53
+
54
+ import { BEAMER_BETA, FUSED_FRONTIER_MAX, MAX_LEVELS_PER_SUBMIT, U32_MAX } from "../constants.js";
55
+ import { type GpuContext } from "../context.js";
56
+ import { WebGpuGraphError } from "../errors.js";
57
+ import { CommandBatch } from "../kernel/batch.js";
58
+ import { type DispatchPlan, plan1d, planGridStride } from "../kernel/dispatch.js";
59
+ import { type UniformValues } from "../kernel/struct-block.js";
60
+ import {
61
+ FILL_PARAMS,
62
+ FRONTIER_COUNTERS,
63
+ FRONTIER_PARAMS,
64
+ graphBindings,
65
+ graphOverrides,
66
+ kernelSpec,
67
+ } from "../kernels.js";
68
+ import { prepareAdvance } from "../primitives/advance.js";
69
+ import { prepareCompact } from "../primitives/compact.js";
70
+ import { coreOfView, coreWindows } from "../primitives/core-shape.js";
71
+ import { type FrontierFinalizeFields, prepareFrontier, W } from "../primitives/frontier.js";
72
+ import { prepareRadixSort, radixHistBytes } from "../primitives/radix-sort.js";
73
+ import { assertDeviceComputes } from "../primitives/verify.js";
74
+ import { type BfsOptions } from "../types/accelerator.js";
75
+ import { type Binding } from "../types/memory.js";
76
+ import { type GpuRunOptions } from "../types/run.js";
77
+ import { type GpuBfsResult } from "../types/traversal.js";
78
+ import { type AlgorithmScope, algorithmScope } from "./scope.js";
79
+
80
+ const ALGORITHM = "breadthFirstSearch";
81
+
82
+ /**
83
+ * Params slots of the ring, COUNTED per run (P8-T12), because `UniformRing.reserve` wraps to slot 0 when a submit's
84
+ * records outrun the ring and silently overwrites a record the submit still reads; the ring's `overruns` counts
85
+ * exactly that reuse, `AlgorithmScope.ringOverruns()` exposes it, and the faked-limit test holds it at 0 for the
86
+ * whole run. The bound per level is `5 + 4 x windows`: five recorded once per level (`frontier-finalize` twice,
87
+ * `bfs-contract`, the bits `fill`, `bfs-bitset-build`) and four re-issued once per arc window (`advance-expand`,
88
+ * `bfs-fused`, the fused retry, `bfs-bottom-up`). The driver as built writes fewer -- the contract and the bitset
89
+ * build share one window-free record, a window's two fused dispatches share one, the bits fill has one record per
90
+ * submit, and a directed snapshot's reverse view is one window whatever the forward core's count -- so at most
91
+ * `3 + 3 x windows` per level, and the bound holds with room. The 16 covers the per-submit rebuild
92
+ * (`bfs-unvisited-flags` and `compact`, whose scan is at most 9 dispatches for any n below 2^32, so 10, plus the
93
+ * bits fill). The result batch flushes in its own submit, so the ring must hold IT too, and its size grows with `n`,
94
+ * not with the cadence: the iota `fill`, the radix sort's four passes of one record plus its scan's `2 x levels - 1`
95
+ * (the table is `256 x ceil(n / WG)` words and a level covers `WG` of them, so four levels for any n below 2^32 at
96
+ * the package's WG of 256), the `pred` `fill` and `sssp-pred` once per window -- `2 + 4 x 8 + windows = 34 + windows`
97
+ * at most, which a small cadence undercuts (built 2026-09-25: `(5 + 4) x 1 + 16 = 25` slots against the 27 a
98
+ * 262,143-node result batch records wrapped over the radix sort's own records and returned a wrong `order`, silently
99
+ * -- the plan's `19 + windows` count had no scan level in it). The slot count is therefore the larger of the two
100
+ * batches. At one window and `MAX_LEVELS_PER_SUBMIT` it is P8-T6's 304; test/device/constants.test.ts pins the
101
+ * arithmetic.
102
+ * @internal
103
+ * @param windows - the forward core's arc windows (1 when it is not windowed)
104
+ * @param levelsPerSubmit - the levels recorded per submit
105
+ * @returns the ring's slot count
106
+ */
107
+ export function bfsRingSlots(windows: number, levelsPerSubmit: number): number {
108
+ return Math.max((5 + 4 * windows) * levelsPerSubmit + 16, RESULT_BATCH_SLOTS + windows);
109
+ }
110
+
111
+ /** The result batch's records before its per-window `sssp-pred` dispatches: two `fill`s and the 32-bit radix sort's four passes of at most eight records each (see `bfsRingSlots`). */
112
+ const RESULT_BATCH_SLOTS = 2 + 4 * 8;
113
+
114
+ /**
115
+ * The knobs the tests need and nothing public offers (PD-26): the per-level candidate rule, the edge queue's size,
116
+ * the submit cadence, the predecessor kind, and the two inspect seams.
117
+ * @internal
118
+ */
119
+ export interface BfsTuning {
120
+ /** `"auto"` (default) lets the selector choose per level by Beamer's test; `"top-down"` disables the bottom-up candidate (`mode 1`). */
121
+ readonly direction?: "auto" | "top-down" | undefined;
122
+ /** Beamer's alpha (`max(1, floor(arcCount / n))` when absent, PD-21): top-down switches to bottom-up when the frontier's degree sum exceeds the unvisited degree sum divided by it and the frontier is growing. */
123
+ readonly alpha?: number | undefined;
124
+ /** Beamer's beta (`BEAMER_BETA` when absent): bottom-up switches back when `next * beta < unvisitedCount` and the frontier is shrinking; 0 makes that half of the test true whenever anything is unvisited. */
125
+ readonly beta?: number | undefined;
126
+ /** The fused expand-contract threshold (`FUSED_FRONTIER_MAX` when absent): a level whose frontier is below it takes the fused path; 0 never fuses, `U32_MAX` always does. */
127
+ readonly fusedMax?: number | undefined;
128
+ /** The edge queue's entry count; a test fakes a small one to force the overflow path (PD-23). */
129
+ readonly edgeCapacity?: number | undefined;
130
+ /** Levels recorded per submit (default `MAX_LEVELS_PER_SUBMIT`); 1 hands `onLevel` the block after every level. */
131
+ readonly levelsPerSubmit?: number | undefined;
132
+ /** What the post-pass writes into `parent`: 1 (default) the node index, 0 the arc index (P8-T9's unit-weight route). */
133
+ readonly predKind?: 0 | 1 | undefined;
134
+ /** The inspect seam (design 11.9 item 2): after every SUBMIT, the index of the last level recorded, the whole counters block as the submit left it, the vertices the submit's last level claimed (the next level's input queue, `nextFrontierCount` long), and the total `compact` wrote when it rebuilt the unvisited list at the top of the submit (the independent count the block's `unvisitedListLen` must equal, P8-T8). */
135
+ readonly onLevel?:
136
+ | ((level: number, counters: UniformValues, frontier: U32, compactCount: number) => void)
137
+ | undefined;
138
+ /** Fires once, right after `algorithmScope(...)`, so a test can hold the scope and read its ring counters after the run. */
139
+ readonly onScope?: ((scope: AlgorithmScope) => void) | undefined;
140
+ }
141
+
142
+ /**
143
+ * The CPU port's `maxDepth` as a `u32`: `U32_MAX` (no cap) when absent, `NaN` (`d >= NaN` never holds) or at or
144
+ * above `U32_MAX` (which covers `Infinity`); otherwise `max(0, ceil(maxDepth))` -- `2.5` behaves as 3 because
145
+ * `d >= 2.5` first holds at `d == 3`, and a negative value gives 0, the source alone, because `0 >= -1` already holds.
146
+ * @param maxDepth - the caller's option
147
+ * @returns the value the uniform carries
148
+ */
149
+ function normaliseMaxDepth(maxDepth: number | undefined): number {
150
+ if (maxDepth === undefined || Number.isNaN(maxDepth) || maxDepth >= U32_MAX) {
151
+ return U32_MAX;
152
+ }
153
+ return Math.max(0, Math.ceil(maxDepth));
154
+ }
155
+
156
+ /**
157
+ * Validates `options.dest` for a depth result of `n` elements.
158
+ * @param dest - the caller's destination array, if any
159
+ * @param n - the node count
160
+ * @returns the destination as a U32, or null when none was given
161
+ */
162
+ function checkDest(dest: Float32Array | Uint32Array | undefined, n: number): U32 | null {
163
+ if (dest === undefined) {
164
+ return null;
165
+ }
166
+ if (dest instanceof Uint32Array && dest.length === n && dest.buffer instanceof ArrayBuffer) {
167
+ return dest as U32;
168
+ }
169
+ throw new WebGpuGraphError(
170
+ "E_INVALID_ARGUMENT",
171
+ `${ALGORITHM}: dest must be a Uint32Array of length ${n} over an ArrayBuffer`,
172
+ {
173
+ argument: "dest",
174
+ value: `${dest.constructor.name}(${dest.length})`,
175
+ expected: `Uint32Array(${n}) over an ArrayBuffer`,
176
+ },
177
+ );
178
+ }
179
+
180
+ /**
181
+ * Whole-buffer binding of a scratch buffer over its first `size` bytes.
182
+ * @param buffer - the buffer
183
+ * @param size - the bound byte length
184
+ * @returns the binding
185
+ */
186
+ function bindingOf(buffer: GPUBuffer, size: number): Binding {
187
+ return { buffer, offset: 0, size, window: null };
188
+ }
189
+
190
+ /**
191
+ * A degree view's one array on the device (the `outDegree` / `inDegree` views of P7, one u32 per vertex).
192
+ * @param ctx - the context
193
+ * @param s - the snapshot
194
+ * @param name - the view
195
+ * @returns the binding
196
+ */
197
+ function degreeView(ctx: GpuContext, s: GraphSnapshot, name: "outDegree" | "inDegree"): Binding {
198
+ const { bindings }: { readonly bindings: Readonly<Partial<Record<"outDegree" | "inDegree", Binding>>> } =
199
+ ctx.residency.view(s, name);
200
+ const { [name]: binding } = bindings;
201
+ if (binding === undefined) {
202
+ throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM}: the ${name} view has no ${name} binding`, {
203
+ label: `${ALGORITHM}/${name}`,
204
+ message: `the ${name} view has no ${name} binding`,
205
+ });
206
+ }
207
+ return binding;
208
+ }
209
+
210
+ /**
211
+ * The E_ABORTED error of a signal.
212
+ * @param batchId - the last submitted batch, when one exists
213
+ * @returns the error
214
+ */
215
+ function aborted(batchId?: number): WebGpuGraphError {
216
+ return new WebGpuGraphError(
217
+ "E_ABORTED",
218
+ `${ALGORITHM}: the signal was aborted`,
219
+ batchId === undefined ? {} : { batchId },
220
+ );
221
+ }
222
+
223
+ /**
224
+ * One scalar word of a decoded block (every FrontierCounters field is a u32, so anything else is a decoder bug).
225
+ * @param block - the decoded block
226
+ * @param name - the field
227
+ * @returns the word
228
+ */
229
+ function wordOf(block: UniformValues, name: string): number {
230
+ const value = block[name];
231
+ if (typeof value !== "number") {
232
+ throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM}: counters.${name} did not decode to a number`, {
233
+ label: `${ALGORITHM}/counters`,
234
+ message: `the field ${name} did not decode to a number`,
235
+ });
236
+ }
237
+ return value;
238
+ }
239
+
240
+ /**
241
+ * Breadth-first search with the test knobs of PD-26; `breadthFirstSearch` is this with an empty tuning.
242
+ * @internal
243
+ * @param ctx - the context whose device runs the kernels
244
+ * @param s - the snapshot (uploaded through ctx.residency, or found there)
245
+ * @param source - the source node index
246
+ * @param options - `maxDepth`, plus dest / signal / onProgress
247
+ * @param tuning - the knobs
248
+ * @returns the depths, parents, order, visited count, level count and switches
249
+ */
250
+ export async function bfsWithTuning(
251
+ ctx: GpuContext,
252
+ s: GraphSnapshot,
253
+ source: number,
254
+ options: (BfsOptions & GpuRunOptions) | undefined,
255
+ tuning: BfsTuning,
256
+ ): Promise<GpuBfsResult> {
257
+ ctx.assertReady();
258
+ await assertDeviceComputes(ctx);
259
+ const n = s.nodeCount;
260
+ if (!Number.isInteger(source) || source < 0 || source >= n) {
261
+ throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM}: source ${source} is outside [0, ${n})`, {
262
+ argument: "source",
263
+ value: source,
264
+ expected: `an integer in [0, ${n})`,
265
+ });
266
+ }
267
+ const levelsPerSubmit = tuning.levelsPerSubmit ?? MAX_LEVELS_PER_SUBMIT;
268
+ if (!Number.isInteger(levelsPerSubmit) || levelsPerSubmit < 1 || levelsPerSubmit > MAX_LEVELS_PER_SUBMIT) {
269
+ throw new WebGpuGraphError(
270
+ "E_INVALID_ARGUMENT",
271
+ `${ALGORITHM}: levelsPerSubmit must be an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
272
+ {
273
+ argument: "levelsPerSubmit",
274
+ value: levelsPerSubmit,
275
+ expected: `an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
276
+ },
277
+ );
278
+ }
279
+ const dest = checkDest(options?.dest, n);
280
+ const maxDepth = normaliseMaxDepth(options?.maxDepth);
281
+ const predKind = tuning.predKind ?? 1;
282
+ if (options?.signal?.aborted) {
283
+ throw aborted();
284
+ }
285
+ const core = ctx.residency.core(s);
286
+ // a windowed core (colIdx above the binding limit) is executed, not refused (P8-T12 lifts DEP-P8-E): every
287
+ // frontier-walking kernel is dispatched once per arc window with the window's owned range in its record
288
+ const forward = coreWindows(core);
289
+ // the bottom-up sweep walks in-neighbours over the reverse core: on an undirected snapshot the forward arrays
290
+ // themselves (graph-format invariant I7; P7's residency aliases them, so a windowed core's windows ARE the
291
+ // reverse's), on a directed one the reverse VIEW, which spec 4.3 never windows -- so a directed snapshot whose
292
+ // reverse adjacency exceeds one binding is refused here, as the residency refuses the undirected view, instead
293
+ // of failing at the device's bind-group validation
294
+ if (s.directed && 4 * s.arcCount > ctx.caps.limits.maxStorageBufferBindingSize) {
295
+ throw new WebGpuGraphError(
296
+ "E_TOO_LARGE",
297
+ `${ALGORITHM}: the reverse adjacency of a directed snapshot (${4 * s.arcCount} bytes) needs arc windows, which no view executes (spec 4.3); the bottom-up sweep binds it whole`,
298
+ {
299
+ needed: 4 * s.arcCount,
300
+ limit: ctx.caps.limits.maxStorageBufferBindingSize,
301
+ path: "windowed",
302
+ algorithm: ALGORITHM,
303
+ },
304
+ );
305
+ }
306
+ const reverse = s.directed ? coreOfView(ctx.residency.view(s, "reverse"), s.arcCount) : core;
307
+ const backward = coreWindows(reverse);
308
+ // the two degree views the unvisited rebuild reads (P8-T8)
309
+ const outDegree = degreeView(ctx, s, "outDegree");
310
+ const inDegree = degreeView(ctx, s, "inDegree");
311
+ const scope = algorithmScope(ctx, ALGORITHM, bfsRingSlots(forward.length, levelsPerSubmit));
312
+ tuning.onScope?.(scope);
313
+ try {
314
+ const bytes = 4 * n;
315
+ const wg = ctx.workgroupSize;
316
+ const depth = bindingOf(scope.scratch(bytes, "depth"), bytes);
317
+ // the sweep's input, one buffer in two regions: the unvisited list at word 0 and the frontier bitset at
318
+ // word bitsBase = roundUp(n, 64), so the bits region's byte offset is 256-aligned and fill can bind it alone
319
+ const bitsBase = Math.ceil(n / 64) * 64;
320
+ const bitsWords = Math.ceil(n / 32);
321
+ const sweepBytes = 4 * (bitsBase + bitsWords);
322
+ const sweepIn = bindingOf(scope.scratch(sweepBytes, "sweep-in"), sweepBytes);
323
+ const unvisitedList: Binding = { buffer: sweepIn.buffer, offset: 0, size: bytes, window: null };
324
+ const frontierBits: Binding = {
325
+ buffer: sweepIn.buffer,
326
+ offset: 4 * bitsBase,
327
+ size: 4 * bitsWords,
328
+ window: null,
329
+ };
330
+ const flags = bindingOf(scope.scratch(bytes, "unvisited-flags"), bytes);
331
+ const iota = bindingOf(scope.scratch(bytes, "iota"), bytes);
332
+ const compactCount = bindingOf(scope.scratch(4, "compact-count"), 4);
333
+ await ctx.allocator.check();
334
+ const planner = await prepareFrontier(scope, n, s.arcCount, tuning.edgeCapacity);
335
+ const advance = await prepareAdvance(scope, core);
336
+ const compact = await prepareCompact(scope);
337
+ const contract = await ctx.pipelines.kernel(kernelSpec("bfs-contract"));
338
+ const fused = await ctx.pipelines.kernel(kernelSpec("bfs-fused", graphOverrides(core, null)));
339
+ const bitset = await ctx.pipelines.kernel(kernelSpec("bfs-bitset-build"));
340
+ const bottomUp = await ctx.pipelines.kernel(kernelSpec("bfs-bottom-up", graphOverrides(reverse, null)));
341
+ const unvisited = await ctx.pipelines.kernel(kernelSpec("bfs-unvisited-flags"));
342
+ const pred = await ctx.pipelines.kernel(kernelSpec("sssp-pred", { ...graphOverrides(core, null), MODE: 1 }));
343
+ const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
344
+ const sort = await prepareRadixSort(scope);
345
+ const { frontier } = planner;
346
+ const { counters } = frontier;
347
+ const { queue } = ctx.device;
348
+ const fillPlan = plan1d(n, wg, ctx.caps);
349
+ // the level kernels are DIRECT grid-stride dispatches gated by the selector's path word (word 24): the plan
350
+ // bounds the grid, the kernel loops to its count word. The contract walks the edge queue and the bitset build
351
+ // the frontier from one record, so one plan covers both; bfs-fused is one workgroup per entry and takes the
352
+ // plan's GROUP count as its stride (planGridStride's cap applies to the groups)
353
+ const levelPlan = planGridStride(Math.max(n, frontier.edgeCapacity), wg, ctx.caps);
354
+ const sweepPlan = planGridStride(n, wg, ctx.caps);
355
+ const fusedPlan = planGridStride(n * wg, wg, ctx.caps);
356
+ const bitsPlan = plan1d(bitsWords, wg, ctx.caps);
357
+ const recordFill = (pass: GPUComputePassEncoder, dst: Binding, value: number, mode: 0 | 1): void => {
358
+ const params = scope.params(FILL_PARAMS, { count: n, value, mode, pad0: 0 });
359
+ fill.dispatch(pass, fill.bind({ dst, P: params.binding }), fillPlan, [params.offset]);
360
+ };
361
+ const flagsPlan: DispatchPlan = planGridStride(n, wg, ctx.caps);
362
+ /**
363
+ * The rebuild of the unvisited set (PD-18), recorded at the top of every submit before its levels.
364
+ * @param pass - the compute pass
365
+ */
366
+ const recordRebuild = (pass: GPUComputePassEncoder): void => {
367
+ const params = scope.params(FRONTIER_PARAMS, { wg, n, stride: flagsPlan.stride ?? n });
368
+ const bound = unvisited.bind({ outDegree, inDegree, depth, flags, counters, P: params.binding });
369
+ unvisited.dispatch(pass, bound, flagsPlan, [params.offset]);
370
+ compact.record(pass, {
371
+ queue: iota,
372
+ flags,
373
+ count: n,
374
+ out: unvisitedList,
375
+ outCount: compactCount,
376
+ outIndex: 0,
377
+ });
378
+ };
379
+ const submit = (batch: CommandBatch): ReturnType<CommandBatch["submit"]> => {
380
+ scope.flush();
381
+ return batch.submit();
382
+ };
383
+
384
+ // setup: depth = INVALID_INDEX everywhere and the iota queue compact reads; then the source at 0 and the
385
+ // seeded block, both queue writes ordered before the first level submit (the seed is rotated in by the
386
+ // first boundary, P8-T4)
387
+ const setup = new CommandBatch(ctx, `${ALGORITHM}/setup`);
388
+ const setupPass = setup.pass("fill");
389
+ recordFill(setupPass, depth, INVALID_INDEX, 0);
390
+ recordFill(setupPass, iota, 0, 1);
391
+ setup.endPass();
392
+ await submit(setup).readback;
393
+ ctx.assertReady();
394
+ queue.writeBuffer(depth.buffer, depth.offset + 4 * source, Uint32Array.of(0));
395
+ frontier.reset(queue, source, { nextFrontierCount: 1, level: U32_MAX });
396
+
397
+ // the levels: MAX_LEVELS_PER_SUBMIT per submit, four bytes back (PD-7); Beamer's test chooses the direction
398
+ // per level (mode 0; mode 1 pins top-down), with the fused path below fusedMax (P8-T7)
399
+ const fields: FrontierFinalizeFields = {
400
+ mode: tuning.direction === "top-down" ? 1 : 0,
401
+ alpha: tuning.alpha ?? Math.max(1, Math.floor(s.arcCount / n)),
402
+ beta: tuning.beta ?? BEAMER_BETA,
403
+ fusedMax: tuning.fusedMax ?? FUSED_FRONTIER_MAX,
404
+ maxDepth,
405
+ };
406
+ let levelsRecorded = 0;
407
+ let submits = 0;
408
+ for (;;) {
409
+ // the rebuild (PD-18): the three unvisited words zeroed by a queue write ordered before this submit,
410
+ // then the flags kernel and compact at the top of the pass, unconditionally, never per level
411
+ queue.writeBuffer(counters.buffer, counters.offset + 4 * W.unvisitedCount, new Uint32Array(3));
412
+ const batch = new CommandBatch(ctx, `${ALGORITHM}/levels`);
413
+ const pass = batch.pass("bfs");
414
+ recordRebuild(pass);
415
+ const bitsParams = scope.params(FILL_PARAMS, { count: bitsWords, value: 0, mode: 0, pad0: 0 });
416
+ const boundBitsFill = fill.bind({ dst: frontierBits, P: bitsParams.binding });
417
+ for (let level = 0; level < levelsPerSubmit; level++) {
418
+ planner.recordFinalize(pass, 0, level, { ...fields, firstOfSubmit: Math.min(level, 2) });
419
+ advance.record(pass, frontier);
420
+ planner.recordFinalize(pass, 1, level, fields);
421
+ // one window-free record serves the contract and the bitset build, which read no arc; the kernels that
422
+ // walk arcs (the fused dispatch, the sweep) get one record per window below (P8-T12)
423
+ const params = scope.params(FRONTIER_PARAMS, {
424
+ wg,
425
+ n,
426
+ edgeCapacity: frontier.edgeCapacity,
427
+ arcBase: 0,
428
+ arcEnd: s.arcCount,
429
+ bitsBase,
430
+ stride: levelPlan.stride ?? wg,
431
+ });
432
+ const boundContract = contract.bind({
433
+ edgeQueue: frontier.edgeQueue,
434
+ counters,
435
+ depth,
436
+ frontierOut: frontier.output,
437
+ P: params.binding,
438
+ });
439
+ contract.dispatch(pass, boundContract, levelPlan, [params.offset]);
440
+ // the fused path (role 0's choice) and the overflow retry (role 1's, PD-23): ONE dispatch per window that
441
+ // the path word turns into the level's work or into nothing (the claim is idempotent across windows)
442
+ for (const w of forward) {
443
+ const fusedParams = scope.params(FRONTIER_PARAMS, {
444
+ wg,
445
+ n,
446
+ edgeCapacity: frontier.edgeCapacity,
447
+ arcBase: w.arcBase,
448
+ arcEnd: w.arcEnd,
449
+ stride: fusedPlan.x * fusedPlan.y,
450
+ });
451
+ const boundFused = fused.bind({
452
+ ...graphBindings(w.core, null),
453
+ frontierIn: frontier.input,
454
+ counters,
455
+ depth,
456
+ frontierOut: frontier.output,
457
+ P: fusedParams.binding,
458
+ });
459
+ fused.dispatch(pass, boundFused, fusedPlan, [fusedParams.offset]);
460
+ }
461
+ // the bottom-up level (role 0's choice under Beamer's test, P8-T8): the bitset zeroed (every level: a
462
+ // top-down level zeroes bits nobody reads, cheaper than a gate), the frontier's bits set, the unvisited
463
+ // list swept over the reverse core
464
+ fill.dispatch(pass, boundBitsFill, bitsPlan, [bitsParams.offset]);
465
+ const boundBitset = bitset.bind({
466
+ frontierIn: frontier.input,
467
+ counters,
468
+ bits: sweepIn,
469
+ P: params.binding,
470
+ });
471
+ bitset.dispatch(pass, boundBitset, levelPlan, [params.offset]);
472
+ // the sweep once per window of the REVERSE core (an entry claimed in one window is skipped in the next)
473
+ for (const w of backward) {
474
+ const sweepParams = scope.params(FRONTIER_PARAMS, {
475
+ wg,
476
+ n,
477
+ arcBase: w.arcBase,
478
+ arcEnd: w.arcEnd,
479
+ bitsBase,
480
+ stride: sweepPlan.stride ?? wg,
481
+ });
482
+ const boundSweep = bottomUp.bind({
483
+ ...graphBindings(w.core, null),
484
+ sweepIn,
485
+ counters,
486
+ depth,
487
+ frontierOut: frontier.output,
488
+ P: sweepParams.binding,
489
+ });
490
+ bottomUp.dispatch(pass, boundSweep, sweepPlan, [sweepParams.offset]);
491
+ }
492
+ frontier.swap();
493
+ }
494
+ batch.endPass();
495
+ const doneRequest = batch.readback(counters.buffer, counters.offset + 4 * W.done, 4);
496
+ const inspect =
497
+ tuning.onLevel === undefined
498
+ ? null
499
+ : {
500
+ block: batch.readback(counters.buffer, counters.offset, FRONTIER_COUNTERS.byteLength),
501
+ frontier: batch.readback(frontier.input.buffer, frontier.input.offset, frontier.input.size),
502
+ count: batch.readback(compactCount.buffer, compactCount.offset, 4),
503
+ };
504
+ const submitted = submit(batch);
505
+ const back = await submitted.readback;
506
+ levelsRecorded += levelsPerSubmit;
507
+ submits += 1;
508
+ ctx.assertReady();
509
+ if (options?.signal?.aborted) {
510
+ throw aborted(submitted.id);
511
+ }
512
+ options?.onProgress?.(Math.min(levelsRecorded, n), n);
513
+ if (inspect !== null && tuning.onLevel !== undefined) {
514
+ const block = FRONTIER_COUNTERS.read(new DataView(back), inspect.block.offset);
515
+ const claimed = new Uint32Array(back, inspect.frontier.offset, wordOf(block, "nextFrontierCount"));
516
+ const rebuilt = new Uint32Array(back, inspect.count.offset, 1)[0];
517
+ tuning.onLevel(levelsRecorded - 1, block, claimed.slice(), rebuilt);
518
+ }
519
+ if (new Uint32Array(back, doneRequest.offset, 1)[0] !== 0) {
520
+ break;
521
+ }
522
+ if (submits > n + 1) {
523
+ throw new WebGpuGraphError(
524
+ "E_VALIDATION",
525
+ `${ALGORITHM}: the done flag never rose in ${submits} submits (a traversal has at most ${n} levels)`,
526
+ { label: ALGORITHM, message: `the done flag never rose in ${submits} submits` },
527
+ );
528
+ }
529
+ }
530
+
531
+ // the result batch: order by one stable radix sort of the node indices by depth (PD-14), parent by the
532
+ // post-pass in depth mode (PD-24), then every array through ONE staging slot
533
+ const keys = bindingOf(scope.scratch(bytes, "order/keys"), bytes);
534
+ const vals = bindingOf(scope.scratch(bytes, "order/vals"), bytes);
535
+ const histBytes = radixHistBytes(n, wg);
536
+ const scratch = {
537
+ keys: bindingOf(scope.scratch(bytes, "order/keys-scratch"), bytes),
538
+ vals: bindingOf(scope.scratch(bytes, "order/vals-scratch"), bytes),
539
+ hist: bindingOf(scope.scratch(histBytes, "order/hist"), histBytes),
540
+ offsets: bindingOf(scope.scratch(histBytes, "order/offsets"), histBytes),
541
+ };
542
+ const parent = bindingOf(scope.scratch(bytes, "parent"), bytes);
543
+ const result = new CommandBatch(ctx, `${ALGORITHM}/result`);
544
+ result.copy(depth, keys, bytes);
545
+ const pass = result.pass("result");
546
+ recordFill(pass, vals, 0, 1);
547
+ const sorted = sort.record(pass, keys, vals, n, 32, scratch);
548
+ recordFill(pass, parent, INVALID_INDEX, 0);
549
+ // the post-pass once per arc window (its atomicMin admits the same smallest parent whichever window holds the arc)
550
+ const predPlan = planGridStride(n, wg, ctx.caps);
551
+ for (const w of forward) {
552
+ const predParams = scope.params(FRONTIER_PARAMS, {
553
+ wg,
554
+ n,
555
+ arcBase: w.arcBase,
556
+ arcEnd: w.arcEnd,
557
+ predKind,
558
+ source,
559
+ stride: predPlan.stride ?? n,
560
+ });
561
+ const predBound = pred.bind({
562
+ ...graphBindings(w.core, null),
563
+ dist: depth,
564
+ pred: parent,
565
+ P: predParams.binding,
566
+ });
567
+ pred.dispatch(pass, predBound, predPlan, [predParams.offset]);
568
+ }
569
+ result.endPass();
570
+ const depthRequest = result.readback(depth.buffer, depth.offset, bytes);
571
+ const parentRequest = result.readback(parent.buffer, parent.offset, bytes);
572
+ const orderRequest = result.readback(sorted.vals.buffer, sorted.vals.offset, bytes);
573
+ const blockRequest = result.readback(counters.buffer, counters.offset, FRONTIER_COUNTERS.byteLength);
574
+ const back = await submit(result).readback;
575
+ ctx.assertReady();
576
+ const block = FRONTIER_COUNTERS.read(new DataView(back), blockRequest.offset);
577
+ const visitedCount = wordOf(block, "visitedCount");
578
+ if (visitedCount > n) {
579
+ // every vertex is claimed at most once, so a count above n is a kernel bug (a claim that lets two
580
+ // same-level claimants append), never a result to return
581
+ throw new WebGpuGraphError(
582
+ "E_VALIDATION",
583
+ `${ALGORITHM}: visitedCount ${visitedCount} exceeds the ${n} vertices (a duplicate claim)`,
584
+ {
585
+ label: `${ALGORITHM}/visitedCount`,
586
+ message: `the device counted ${visitedCount} visits of ${n} vertices`,
587
+ },
588
+ );
589
+ }
590
+ const level = wordOf(block, "level");
591
+ // done by an empty frontier (the level word already counts the boundary that found it), or by maxDepth with
592
+ // a reached-but-unexpanded level
593
+ const levels = wordOf(block, "frontierCount") === 0 ? level : level + 1;
594
+ const depthOut = dest ?? new Uint32Array(n);
595
+ depthOut.set(new Uint32Array(back, depthRequest.offset, n));
596
+ return {
597
+ depth: depthOut,
598
+ parent: new Uint32Array(back, parentRequest.offset, n).slice(),
599
+ order: new Uint32Array(back, orderRequest.offset, visitedCount).slice(),
600
+ visitedCount,
601
+ levels,
602
+ switches: wordOf(block, "switches"),
603
+ };
604
+ } finally {
605
+ scope.dispose();
606
+ }
607
+ }
608
+
609
+ /**
610
+ * Breadth-first search on the device (spec 3.3 line 807, design 8.4, 9.7): `depth` exact, `parent` the smallest
611
+ * predecessor one depth down (PD-24), `order` grouped by depth and ascending by index within a depth (PD-14), all
612
+ * bitwise reproducible; `maxDepth` as the CPU port reads it (a node at the cap is reached and not expanded).
613
+ * @param ctx - the context whose device runs the kernels
614
+ * @param s - the snapshot (uploaded through ctx.residency, or found there)
615
+ * @param source - the source node index (E_INVALID_ARGUMENT outside `[0, n)`, so the empty graph refuses every source)
616
+ * @param options - `maxDepth`, plus dest (a Uint32Array of length n for `depth`) / signal / onProgress
617
+ * @returns the depths, parents, order, visited count, level count and switches (the direction changes Beamer's test made on the device)
618
+ */
619
+ export function breadthFirstSearch(
620
+ ctx: GpuContext,
621
+ s: GraphSnapshot,
622
+ source: number,
623
+ options?: BfsOptions & GpuRunOptions,
624
+ ): Promise<GpuBfsResult> {
625
+ return bfsWithTuning(ctx, s, source, options, {});
626
+ }