@graphty/webgpu-graph-algorithms 0.6.3 → 0.6.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +62 -32
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-BXqgCifx.js → context-hzGggHeM.js} +68 -24
- package/dist/chunks/context-hzGggHeM.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +8 -6
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +57 -6
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/bellman-ford.d.ts +60 -0
- package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
- package/dist/src/algorithms/bellman-ford.js +301 -0
- package/dist/src/algorithms/bellman-ford.js.map +1 -0
- package/dist/src/algorithms/bfs.d.ts +67 -0
- package/dist/src/algorithms/bfs.d.ts.map +1 -0
- package/dist/src/algorithms/bfs.js +534 -0
- package/dist/src/algorithms/bfs.js.map +1 -0
- package/dist/src/algorithms/closeness.d.ts +53 -0
- package/dist/src/algorithms/closeness.d.ts.map +1 -0
- package/dist/src/algorithms/closeness.js +323 -0
- package/dist/src/algorithms/closeness.js.map +1 -0
- package/dist/src/algorithms/scope.d.ts +3 -1
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +1 -0
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +72 -0
- package/dist/src/algorithms/sssp.d.ts.map +1 -0
- package/dist/src/algorithms/sssp.js +586 -0
- package/dist/src/algorithms/sssp.js.map +1 -0
- package/dist/src/constants.d.ts +10 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +10 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/index.d.ts +8 -2
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +7 -1
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts +4 -4
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +39 -5
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernel/uniform-ring.d.ts +8 -0
- package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
- package/dist/src/kernel/uniform-ring.js +13 -0
- package/dist/src/kernel/uniform-ring.js.map +1 -1
- package/dist/src/kernels.d.ts +45 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +368 -3
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +63 -0
- package/dist/src/primitives/advance.d.ts.map +1 -0
- package/dist/src/primitives/advance.js +95 -0
- package/dist/src/primitives/advance.js.map +1 -0
- package/dist/src/primitives/compact.d.ts +89 -0
- package/dist/src/primitives/compact.d.ts.map +1 -0
- package/dist/src/primitives/compact.js +233 -0
- package/dist/src/primitives/compact.js.map +1 -0
- package/dist/src/primitives/core-shape.d.ts +22 -1
- package/dist/src/primitives/core-shape.d.ts.map +1 -1
- package/dist/src/primitives/core-shape.js +33 -3
- package/dist/src/primitives/core-shape.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +151 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -0
- package/dist/src/primitives/frontier.js +250 -0
- package/dist/src/primitives/frontier.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +16 -7
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/traversal.d.ts +53 -0
- package/dist/src/types/traversal.d.ts.map +1 -0
- package/dist/src/types/traversal.js +10 -0
- package/dist/src/types/traversal.js.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
- package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +53 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js +164 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
- package/dist/webgpu-graph-algorithms.js +3155 -384
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +5 -4
- package/src/accelerator.ts +65 -7
- package/src/algorithms/bellman-ford.ts +387 -0
- package/src/algorithms/bfs.ts +626 -0
- package/src/algorithms/closeness.ts +395 -0
- package/src/algorithms/scope.ts +4 -1
- package/src/algorithms/sssp.ts +768 -0
- package/src/constants.ts +10 -0
- package/src/index.ts +14 -1
- package/src/kernel/prelude.ts +39 -4
- package/src/kernel/uniform-ring.ts +14 -0
- package/src/kernels.ts +447 -6
- package/src/primitives/advance.ts +131 -0
- package/src/primitives/compact.ts +323 -0
- package/src/primitives/core-shape.ts +41 -3
- package/src/primitives/frontier.ts +372 -0
- package/src/types/accelerator.ts +18 -5
- package/src/types/traversal.ts +56 -0
- package/src/wgsl/advance-expand.wgsl.ts +68 -0
- package/src/wgsl/bf-relax.wgsl.ts +57 -0
- package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
- package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
- package/src/wgsl/bfs-contract.wgsl.ts +54 -0
- package/src/wgsl/bfs-fused.wgsl.ts +77 -0
- package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
- package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
- package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
- package/src/wgsl/compact-scatter.wgsl.ts +16 -0
- package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
- package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
- package/src/wgsl/frontier-finalize.wgsl.ts +163 -0
- package/src/wgsl/sssp-pred.wgsl.ts +79 -0
- package/src/wgsl/sssp-relax.wgsl.ts +71 -0
- package/dist/chunks/context-BXqgCifx.js.map +0 -1
|
@@ -0,0 +1,626 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Breadth-first search on the device (design 8.4, 3.3 line 807, 9.7; P8-T6 / P8-T7 / P8-T8, the P8 plan's PD-5 /
|
|
3
|
+
* PD-6 / PD-7 / PD-14 / PD-18 / PD-21 / PD-23 / PD-24 / PD-26 / DEP-P8-B): the direction-optimizing traversal over
|
|
4
|
+
* the `Frontier` of P8-T4, every per-level choice made ON THE DEVICE. Every level is eight recorded dispatches, all
|
|
5
|
+
* DIRECT (2026-09-25, docs/decisions/G8.md G8-F5: Dawn validates every `dispatchWorkgroupsIndirect` with an internal
|
|
6
|
+
* clamp pass costing about 0.4 ms of device time whether or not it dispatches anything, in Node and in Chromium
|
|
7
|
+
* alike, and that was 97 % of a traversal's wall time) -- `frontier-finalize` role 0 (the level boundary: rotates
|
|
8
|
+
* the counts, advances `level`, decides `done`, evaluates Beamer's test and CHOOSES, writing the choice into the
|
|
9
|
+
* block's `path` word: 3 bottom-up, 2 fused for a top-down frontier below `fusedMax` entries, 1 two-phase for any
|
|
10
|
+
* other, 0 done), `advance-expand` (the frontier's rows into the edge queue), `frontier-finalize` role 1 (clamps
|
|
11
|
+
* `edgeCount`, or, when the edge queue overflowed, sets `path` 4 for the fused retry, PD-23), `bfs-contract` (the
|
|
12
|
+
* claim `atomicMin(&depth[v], level + 1)`, the winners packed into the next vertex queue), `bfs-fused` (one workgroup
|
|
13
|
+
* per frontier entry with the same claim inline and no edge queue; the fused level and the retry alike), then the
|
|
14
|
+
* bottom-up trio: a `fill` zeroing the frontier bitset, `bfs-bitset-build` setting the frontier's bits, and
|
|
15
|
+
* `bfs-bottom-up` sweeping the unvisited list over the REVERSE core (each unvisited vertex reads its in-neighbours
|
|
16
|
+
* until the first one in the bitset and claims itself). Every kernel is a grid-stride dispatch of a host-planned
|
|
17
|
+
* grid (`planGridStride`) that loops to its count word and reads the path word first, so exactly one path does the
|
|
18
|
+
* level's work and the others cost one uniform load per workgroup. The unvisited set Beamer's test is against (PD-18) is rebuilt exactly once per submit, before
|
|
19
|
+
* the levels, by `bfs-unvisited-flags` plus `compact` over an iota queue, and maintained between rebuilds by
|
|
20
|
+
* subtraction inside the selector (whose JSDoc states the boundary rule). The host records `MAX_LEVELS_PER_SUBMIT`
|
|
21
|
+
* levels into ONE command buffer, submits, and reads four bytes, the `done` word (PD-7): a road network has
|
|
22
|
+
* thousands of levels and a per-level `mapAsync` would be slower than the CPU. A level recorded past the end is a
|
|
23
|
+
* no-op (its boundary finds `done` set, writes path 0 and moves no counter), so the loop needs no diameter and
|
|
24
|
+
* ends on `done`; a traversal has at most `n` levels, so more submits than that is E_VALIDATION, never a hang. The
|
|
25
|
+
* thresholds are uniform fields (`FUSED_FRONTIER_MAX`; alpha derived as `max(1, floor(arcCount / n))`, PD-21;
|
|
26
|
+
* `BEAMER_BETA`; `mode 1` pinning top-down -- each unless the tuning says otherwise), so a test forces any path
|
|
27
|
+
* without recompiling; the block's `fusedLevels` / `twoPhaseLevels` / `bottomUpLevels` / `overflowLevels` /
|
|
28
|
+
* `switches` words record which path each level took.
|
|
29
|
+
*
|
|
30
|
+
* There is no dedupe (DEP-P8-B, PD-5): the edge queue holds duplicates -- a vertex with three frontier neighbours
|
|
31
|
+
* appears three times -- but for one level exactly one invocation observes `INVALID_INDEX` at `depth[v]` and
|
|
32
|
+
* exactly that one appends `v`, so the next vertex frontier is duplicate-free by construction and an ownership
|
|
33
|
+
* dedupe would have nothing to remove. `parent` is not written by the claim (PD-24): `sssp-pred` in `MODE 1` runs
|
|
34
|
+
* once over the settled depths and writes the SMALLEST `u` with `depth[u] + 1 == depth[v]` and an arc `u -> v`, so
|
|
35
|
+
* it is bitwise reproducible where a claim-time parent would be a scheduling accident. `order` (PD-14) is one
|
|
36
|
+
* stable radix sort of the node indices by `depth`: grouped by level, ascending by index within a level,
|
|
37
|
+
* `INVALID_INDEX` last, and the first `visitedCount` values are the result. The result batch stages every array
|
|
38
|
+
* into one staging slot, so a traversal maps `ceil(levels / 32) + 1` buffers in all.
|
|
39
|
+
*
|
|
40
|
+
* `maxDepth` reaches the device as a `u32` field of `FrontierParams`, and the uniform writer refuses anything that
|
|
41
|
+
* is not an integer in `[0, 2^32 - 1]`; the host therefore normalises it first, to exactly the CPU port's rule
|
|
42
|
+
* (`algorithms/src/indexed/bfs.ts`: a node at depth `d >= maxDepth` is reached and not expanded, unbounded when
|
|
43
|
+
* absent). The tuning entry `bfsWithTuning` (PD-26) is what the tests drive; nothing public exposes it.
|
|
44
|
+
*
|
|
45
|
+
* A core whose `colIdx` exceeds `maxStorageBufferBindingSize` is bound as arc windows and EXECUTED (P8-T12, DEP-P8-E
|
|
46
|
+
* lifted for the frontier family): `advance-expand`, `bfs-fused` and `sssp-pred` run once per window of the
|
|
47
|
+
* forward core and `bfs-bottom-up` once per window of the reverse core, every window its own dispatch with its own
|
|
48
|
+
* `FrontierParams` record carrying the window's owned arc range (`coreWindows`), and the claims, being `atomicMin`s
|
|
49
|
+
* and an `INVALID_INDEX` entry test, are idempotent across windows. The ring is sized per run by `bfsRingSlots`.
|
|
50
|
+
*/
|
|
51
|
+
|
|
52
|
+
import { type GraphSnapshot, INVALID_INDEX, type U32 } from "@graphty/graph-format";
|
|
53
|
+
|
|
54
|
+
import { BEAMER_BETA, FUSED_FRONTIER_MAX, MAX_LEVELS_PER_SUBMIT, U32_MAX } from "../constants.js";
|
|
55
|
+
import { type GpuContext } from "../context.js";
|
|
56
|
+
import { WebGpuGraphError } from "../errors.js";
|
|
57
|
+
import { CommandBatch } from "../kernel/batch.js";
|
|
58
|
+
import { type DispatchPlan, plan1d, planGridStride } from "../kernel/dispatch.js";
|
|
59
|
+
import { type UniformValues } from "../kernel/struct-block.js";
|
|
60
|
+
import {
|
|
61
|
+
FILL_PARAMS,
|
|
62
|
+
FRONTIER_COUNTERS,
|
|
63
|
+
FRONTIER_PARAMS,
|
|
64
|
+
graphBindings,
|
|
65
|
+
graphOverrides,
|
|
66
|
+
kernelSpec,
|
|
67
|
+
} from "../kernels.js";
|
|
68
|
+
import { prepareAdvance } from "../primitives/advance.js";
|
|
69
|
+
import { prepareCompact } from "../primitives/compact.js";
|
|
70
|
+
import { coreOfView, coreWindows } from "../primitives/core-shape.js";
|
|
71
|
+
import { type FrontierFinalizeFields, prepareFrontier, W } from "../primitives/frontier.js";
|
|
72
|
+
import { prepareRadixSort, radixHistBytes } from "../primitives/radix-sort.js";
|
|
73
|
+
import { assertDeviceComputes } from "../primitives/verify.js";
|
|
74
|
+
import { type BfsOptions } from "../types/accelerator.js";
|
|
75
|
+
import { type Binding } from "../types/memory.js";
|
|
76
|
+
import { type GpuRunOptions } from "../types/run.js";
|
|
77
|
+
import { type GpuBfsResult } from "../types/traversal.js";
|
|
78
|
+
import { type AlgorithmScope, algorithmScope } from "./scope.js";
|
|
79
|
+
|
|
80
|
+
const ALGORITHM = "breadthFirstSearch";
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Params slots of the ring, COUNTED per run (P8-T12), because `UniformRing.reserve` wraps to slot 0 when a submit's
|
|
84
|
+
* records outrun the ring and silently overwrites a record the submit still reads; the ring's `overruns` counts
|
|
85
|
+
* exactly that reuse, `AlgorithmScope.ringOverruns()` exposes it, and the faked-limit test holds it at 0 for the
|
|
86
|
+
* whole run. The bound per level is `5 + 4 x windows`: five recorded once per level (`frontier-finalize` twice,
|
|
87
|
+
* `bfs-contract`, the bits `fill`, `bfs-bitset-build`) and four re-issued once per arc window (`advance-expand`,
|
|
88
|
+
* `bfs-fused`, the fused retry, `bfs-bottom-up`). The driver as built writes fewer -- the contract and the bitset
|
|
89
|
+
* build share one window-free record, a window's two fused dispatches share one, the bits fill has one record per
|
|
90
|
+
* submit, and a directed snapshot's reverse view is one window whatever the forward core's count -- so at most
|
|
91
|
+
* `3 + 3 x windows` per level, and the bound holds with room. The 16 covers the per-submit rebuild
|
|
92
|
+
* (`bfs-unvisited-flags` and `compact`, whose scan is at most 9 dispatches for any n below 2^32, so 10, plus the
|
|
93
|
+
* bits fill). The result batch flushes in its own submit, so the ring must hold IT too, and its size grows with `n`,
|
|
94
|
+
* not with the cadence: the iota `fill`, the radix sort's four passes of one record plus its scan's `2 x levels - 1`
|
|
95
|
+
* (the table is `256 x ceil(n / WG)` words and a level covers `WG` of them, so four levels for any n below 2^32 at
|
|
96
|
+
* the package's WG of 256), the `pred` `fill` and `sssp-pred` once per window -- `2 + 4 x 8 + windows = 34 + windows`
|
|
97
|
+
* at most, which a small cadence undercuts (built 2026-09-25: `(5 + 4) x 1 + 16 = 25` slots against the 27 a
|
|
98
|
+
* 262,143-node result batch records wrapped over the radix sort's own records and returned a wrong `order`, silently
|
|
99
|
+
* -- the plan's `19 + windows` count had no scan level in it). The slot count is therefore the larger of the two
|
|
100
|
+
* batches. At one window and `MAX_LEVELS_PER_SUBMIT` it is P8-T6's 304; test/device/constants.test.ts pins the
|
|
101
|
+
* arithmetic.
|
|
102
|
+
* @internal
|
|
103
|
+
* @param windows - the forward core's arc windows (1 when it is not windowed)
|
|
104
|
+
* @param levelsPerSubmit - the levels recorded per submit
|
|
105
|
+
* @returns the ring's slot count
|
|
106
|
+
*/
|
|
107
|
+
export function bfsRingSlots(windows: number, levelsPerSubmit: number): number {
|
|
108
|
+
return Math.max((5 + 4 * windows) * levelsPerSubmit + 16, RESULT_BATCH_SLOTS + windows);
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/** The result batch's records before its per-window `sssp-pred` dispatches: two `fill`s and the 32-bit radix sort's four passes of at most eight records each (see `bfsRingSlots`). */
|
|
112
|
+
const RESULT_BATCH_SLOTS = 2 + 4 * 8;
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* The knobs the tests need and nothing public offers (PD-26): the per-level candidate rule, the edge queue's size,
|
|
116
|
+
* the submit cadence, the predecessor kind, and the two inspect seams.
|
|
117
|
+
* @internal
|
|
118
|
+
*/
|
|
119
|
+
export interface BfsTuning {
|
|
120
|
+
/** `"auto"` (default) lets the selector choose per level by Beamer's test; `"top-down"` disables the bottom-up candidate (`mode 1`). */
|
|
121
|
+
readonly direction?: "auto" | "top-down" | undefined;
|
|
122
|
+
/** Beamer's alpha (`max(1, floor(arcCount / n))` when absent, PD-21): top-down switches to bottom-up when the frontier's degree sum exceeds the unvisited degree sum divided by it and the frontier is growing. */
|
|
123
|
+
readonly alpha?: number | undefined;
|
|
124
|
+
/** Beamer's beta (`BEAMER_BETA` when absent): bottom-up switches back when `next * beta < unvisitedCount` and the frontier is shrinking; 0 makes that half of the test true whenever anything is unvisited. */
|
|
125
|
+
readonly beta?: number | undefined;
|
|
126
|
+
/** The fused expand-contract threshold (`FUSED_FRONTIER_MAX` when absent): a level whose frontier is below it takes the fused path; 0 never fuses, `U32_MAX` always does. */
|
|
127
|
+
readonly fusedMax?: number | undefined;
|
|
128
|
+
/** The edge queue's entry count; a test fakes a small one to force the overflow path (PD-23). */
|
|
129
|
+
readonly edgeCapacity?: number | undefined;
|
|
130
|
+
/** Levels recorded per submit (default `MAX_LEVELS_PER_SUBMIT`); 1 hands `onLevel` the block after every level. */
|
|
131
|
+
readonly levelsPerSubmit?: number | undefined;
|
|
132
|
+
/** What the post-pass writes into `parent`: 1 (default) the node index, 0 the arc index (P8-T9's unit-weight route). */
|
|
133
|
+
readonly predKind?: 0 | 1 | undefined;
|
|
134
|
+
/** The inspect seam (design 11.9 item 2): after every SUBMIT, the index of the last level recorded, the whole counters block as the submit left it, the vertices the submit's last level claimed (the next level's input queue, `nextFrontierCount` long), and the total `compact` wrote when it rebuilt the unvisited list at the top of the submit (the independent count the block's `unvisitedListLen` must equal, P8-T8). */
|
|
135
|
+
readonly onLevel?:
|
|
136
|
+
| ((level: number, counters: UniformValues, frontier: U32, compactCount: number) => void)
|
|
137
|
+
| undefined;
|
|
138
|
+
/** Fires once, right after `algorithmScope(...)`, so a test can hold the scope and read its ring counters after the run. */
|
|
139
|
+
readonly onScope?: ((scope: AlgorithmScope) => void) | undefined;
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
/**
|
|
143
|
+
* The CPU port's `maxDepth` as a `u32`: `U32_MAX` (no cap) when absent, `NaN` (`d >= NaN` never holds) or at or
|
|
144
|
+
* above `U32_MAX` (which covers `Infinity`); otherwise `max(0, ceil(maxDepth))` -- `2.5` behaves as 3 because
|
|
145
|
+
* `d >= 2.5` first holds at `d == 3`, and a negative value gives 0, the source alone, because `0 >= -1` already holds.
|
|
146
|
+
* @param maxDepth - the caller's option
|
|
147
|
+
* @returns the value the uniform carries
|
|
148
|
+
*/
|
|
149
|
+
function normaliseMaxDepth(maxDepth: number | undefined): number {
|
|
150
|
+
if (maxDepth === undefined || Number.isNaN(maxDepth) || maxDepth >= U32_MAX) {
|
|
151
|
+
return U32_MAX;
|
|
152
|
+
}
|
|
153
|
+
return Math.max(0, Math.ceil(maxDepth));
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* Validates `options.dest` for a depth result of `n` elements.
|
|
158
|
+
* @param dest - the caller's destination array, if any
|
|
159
|
+
* @param n - the node count
|
|
160
|
+
* @returns the destination as a U32, or null when none was given
|
|
161
|
+
*/
|
|
162
|
+
function checkDest(dest: Float32Array | Uint32Array | undefined, n: number): U32 | null {
|
|
163
|
+
if (dest === undefined) {
|
|
164
|
+
return null;
|
|
165
|
+
}
|
|
166
|
+
if (dest instanceof Uint32Array && dest.length === n && dest.buffer instanceof ArrayBuffer) {
|
|
167
|
+
return dest as U32;
|
|
168
|
+
}
|
|
169
|
+
throw new WebGpuGraphError(
|
|
170
|
+
"E_INVALID_ARGUMENT",
|
|
171
|
+
`${ALGORITHM}: dest must be a Uint32Array of length ${n} over an ArrayBuffer`,
|
|
172
|
+
{
|
|
173
|
+
argument: "dest",
|
|
174
|
+
value: `${dest.constructor.name}(${dest.length})`,
|
|
175
|
+
expected: `Uint32Array(${n}) over an ArrayBuffer`,
|
|
176
|
+
},
|
|
177
|
+
);
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/**
|
|
181
|
+
* Whole-buffer binding of a scratch buffer over its first `size` bytes.
|
|
182
|
+
* @param buffer - the buffer
|
|
183
|
+
* @param size - the bound byte length
|
|
184
|
+
* @returns the binding
|
|
185
|
+
*/
|
|
186
|
+
function bindingOf(buffer: GPUBuffer, size: number): Binding {
|
|
187
|
+
return { buffer, offset: 0, size, window: null };
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/**
|
|
191
|
+
* A degree view's one array on the device (the `outDegree` / `inDegree` views of P7, one u32 per vertex).
|
|
192
|
+
* @param ctx - the context
|
|
193
|
+
* @param s - the snapshot
|
|
194
|
+
* @param name - the view
|
|
195
|
+
* @returns the binding
|
|
196
|
+
*/
|
|
197
|
+
function degreeView(ctx: GpuContext, s: GraphSnapshot, name: "outDegree" | "inDegree"): Binding {
|
|
198
|
+
const { bindings }: { readonly bindings: Readonly<Partial<Record<"outDegree" | "inDegree", Binding>>> } =
|
|
199
|
+
ctx.residency.view(s, name);
|
|
200
|
+
const { [name]: binding } = bindings;
|
|
201
|
+
if (binding === undefined) {
|
|
202
|
+
throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM}: the ${name} view has no ${name} binding`, {
|
|
203
|
+
label: `${ALGORITHM}/${name}`,
|
|
204
|
+
message: `the ${name} view has no ${name} binding`,
|
|
205
|
+
});
|
|
206
|
+
}
|
|
207
|
+
return binding;
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
/**
|
|
211
|
+
* The E_ABORTED error of a signal.
|
|
212
|
+
* @param batchId - the last submitted batch, when one exists
|
|
213
|
+
* @returns the error
|
|
214
|
+
*/
|
|
215
|
+
function aborted(batchId?: number): WebGpuGraphError {
|
|
216
|
+
return new WebGpuGraphError(
|
|
217
|
+
"E_ABORTED",
|
|
218
|
+
`${ALGORITHM}: the signal was aborted`,
|
|
219
|
+
batchId === undefined ? {} : { batchId },
|
|
220
|
+
);
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
/**
|
|
224
|
+
* One scalar word of a decoded block (every FrontierCounters field is a u32, so anything else is a decoder bug).
|
|
225
|
+
* @param block - the decoded block
|
|
226
|
+
* @param name - the field
|
|
227
|
+
* @returns the word
|
|
228
|
+
*/
|
|
229
|
+
function wordOf(block: UniformValues, name: string): number {
|
|
230
|
+
const value = block[name];
|
|
231
|
+
if (typeof value !== "number") {
|
|
232
|
+
throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM}: counters.${name} did not decode to a number`, {
|
|
233
|
+
label: `${ALGORITHM}/counters`,
|
|
234
|
+
message: `the field ${name} did not decode to a number`,
|
|
235
|
+
});
|
|
236
|
+
}
|
|
237
|
+
return value;
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
/**
|
|
241
|
+
* Breadth-first search with the test knobs of PD-26; `breadthFirstSearch` is this with an empty tuning.
|
|
242
|
+
* @internal
|
|
243
|
+
* @param ctx - the context whose device runs the kernels
|
|
244
|
+
* @param s - the snapshot (uploaded through ctx.residency, or found there)
|
|
245
|
+
* @param source - the source node index
|
|
246
|
+
* @param options - `maxDepth`, plus dest / signal / onProgress
|
|
247
|
+
* @param tuning - the knobs
|
|
248
|
+
* @returns the depths, parents, order, visited count, level count and switches
|
|
249
|
+
*/
|
|
250
|
+
export async function bfsWithTuning(
|
|
251
|
+
ctx: GpuContext,
|
|
252
|
+
s: GraphSnapshot,
|
|
253
|
+
source: number,
|
|
254
|
+
options: (BfsOptions & GpuRunOptions) | undefined,
|
|
255
|
+
tuning: BfsTuning,
|
|
256
|
+
): Promise<GpuBfsResult> {
|
|
257
|
+
ctx.assertReady();
|
|
258
|
+
await assertDeviceComputes(ctx);
|
|
259
|
+
const n = s.nodeCount;
|
|
260
|
+
if (!Number.isInteger(source) || source < 0 || source >= n) {
|
|
261
|
+
throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM}: source ${source} is outside [0, ${n})`, {
|
|
262
|
+
argument: "source",
|
|
263
|
+
value: source,
|
|
264
|
+
expected: `an integer in [0, ${n})`,
|
|
265
|
+
});
|
|
266
|
+
}
|
|
267
|
+
const levelsPerSubmit = tuning.levelsPerSubmit ?? MAX_LEVELS_PER_SUBMIT;
|
|
268
|
+
if (!Number.isInteger(levelsPerSubmit) || levelsPerSubmit < 1 || levelsPerSubmit > MAX_LEVELS_PER_SUBMIT) {
|
|
269
|
+
throw new WebGpuGraphError(
|
|
270
|
+
"E_INVALID_ARGUMENT",
|
|
271
|
+
`${ALGORITHM}: levelsPerSubmit must be an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
|
|
272
|
+
{
|
|
273
|
+
argument: "levelsPerSubmit",
|
|
274
|
+
value: levelsPerSubmit,
|
|
275
|
+
expected: `an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
|
|
276
|
+
},
|
|
277
|
+
);
|
|
278
|
+
}
|
|
279
|
+
const dest = checkDest(options?.dest, n);
|
|
280
|
+
const maxDepth = normaliseMaxDepth(options?.maxDepth);
|
|
281
|
+
const predKind = tuning.predKind ?? 1;
|
|
282
|
+
if (options?.signal?.aborted) {
|
|
283
|
+
throw aborted();
|
|
284
|
+
}
|
|
285
|
+
const core = ctx.residency.core(s);
|
|
286
|
+
// a windowed core (colIdx above the binding limit) is executed, not refused (P8-T12 lifts DEP-P8-E): every
|
|
287
|
+
// frontier-walking kernel is dispatched once per arc window with the window's owned range in its record
|
|
288
|
+
const forward = coreWindows(core);
|
|
289
|
+
// the bottom-up sweep walks in-neighbours over the reverse core: on an undirected snapshot the forward arrays
|
|
290
|
+
// themselves (graph-format invariant I7; P7's residency aliases them, so a windowed core's windows ARE the
|
|
291
|
+
// reverse's), on a directed one the reverse VIEW, which spec 4.3 never windows -- so a directed snapshot whose
|
|
292
|
+
// reverse adjacency exceeds one binding is refused here, as the residency refuses the undirected view, instead
|
|
293
|
+
// of failing at the device's bind-group validation
|
|
294
|
+
if (s.directed && 4 * s.arcCount > ctx.caps.limits.maxStorageBufferBindingSize) {
|
|
295
|
+
throw new WebGpuGraphError(
|
|
296
|
+
"E_TOO_LARGE",
|
|
297
|
+
`${ALGORITHM}: the reverse adjacency of a directed snapshot (${4 * s.arcCount} bytes) needs arc windows, which no view executes (spec 4.3); the bottom-up sweep binds it whole`,
|
|
298
|
+
{
|
|
299
|
+
needed: 4 * s.arcCount,
|
|
300
|
+
limit: ctx.caps.limits.maxStorageBufferBindingSize,
|
|
301
|
+
path: "windowed",
|
|
302
|
+
algorithm: ALGORITHM,
|
|
303
|
+
},
|
|
304
|
+
);
|
|
305
|
+
}
|
|
306
|
+
const reverse = s.directed ? coreOfView(ctx.residency.view(s, "reverse"), s.arcCount) : core;
|
|
307
|
+
const backward = coreWindows(reverse);
|
|
308
|
+
// the two degree views the unvisited rebuild reads (P8-T8)
|
|
309
|
+
const outDegree = degreeView(ctx, s, "outDegree");
|
|
310
|
+
const inDegree = degreeView(ctx, s, "inDegree");
|
|
311
|
+
const scope = algorithmScope(ctx, ALGORITHM, bfsRingSlots(forward.length, levelsPerSubmit));
|
|
312
|
+
tuning.onScope?.(scope);
|
|
313
|
+
try {
|
|
314
|
+
const bytes = 4 * n;
|
|
315
|
+
const wg = ctx.workgroupSize;
|
|
316
|
+
const depth = bindingOf(scope.scratch(bytes, "depth"), bytes);
|
|
317
|
+
// the sweep's input, one buffer in two regions: the unvisited list at word 0 and the frontier bitset at
|
|
318
|
+
// word bitsBase = roundUp(n, 64), so the bits region's byte offset is 256-aligned and fill can bind it alone
|
|
319
|
+
const bitsBase = Math.ceil(n / 64) * 64;
|
|
320
|
+
const bitsWords = Math.ceil(n / 32);
|
|
321
|
+
const sweepBytes = 4 * (bitsBase + bitsWords);
|
|
322
|
+
const sweepIn = bindingOf(scope.scratch(sweepBytes, "sweep-in"), sweepBytes);
|
|
323
|
+
const unvisitedList: Binding = { buffer: sweepIn.buffer, offset: 0, size: bytes, window: null };
|
|
324
|
+
const frontierBits: Binding = {
|
|
325
|
+
buffer: sweepIn.buffer,
|
|
326
|
+
offset: 4 * bitsBase,
|
|
327
|
+
size: 4 * bitsWords,
|
|
328
|
+
window: null,
|
|
329
|
+
};
|
|
330
|
+
const flags = bindingOf(scope.scratch(bytes, "unvisited-flags"), bytes);
|
|
331
|
+
const iota = bindingOf(scope.scratch(bytes, "iota"), bytes);
|
|
332
|
+
const compactCount = bindingOf(scope.scratch(4, "compact-count"), 4);
|
|
333
|
+
await ctx.allocator.check();
|
|
334
|
+
const planner = await prepareFrontier(scope, n, s.arcCount, tuning.edgeCapacity);
|
|
335
|
+
const advance = await prepareAdvance(scope, core);
|
|
336
|
+
const compact = await prepareCompact(scope);
|
|
337
|
+
const contract = await ctx.pipelines.kernel(kernelSpec("bfs-contract"));
|
|
338
|
+
const fused = await ctx.pipelines.kernel(kernelSpec("bfs-fused", graphOverrides(core, null)));
|
|
339
|
+
const bitset = await ctx.pipelines.kernel(kernelSpec("bfs-bitset-build"));
|
|
340
|
+
const bottomUp = await ctx.pipelines.kernel(kernelSpec("bfs-bottom-up", graphOverrides(reverse, null)));
|
|
341
|
+
const unvisited = await ctx.pipelines.kernel(kernelSpec("bfs-unvisited-flags"));
|
|
342
|
+
const pred = await ctx.pipelines.kernel(kernelSpec("sssp-pred", { ...graphOverrides(core, null), MODE: 1 }));
|
|
343
|
+
const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
|
|
344
|
+
const sort = await prepareRadixSort(scope);
|
|
345
|
+
const { frontier } = planner;
|
|
346
|
+
const { counters } = frontier;
|
|
347
|
+
const { queue } = ctx.device;
|
|
348
|
+
const fillPlan = plan1d(n, wg, ctx.caps);
|
|
349
|
+
// the level kernels are DIRECT grid-stride dispatches gated by the selector's path word (word 24): the plan
|
|
350
|
+
// bounds the grid, the kernel loops to its count word. The contract walks the edge queue and the bitset build
|
|
351
|
+
// the frontier from one record, so one plan covers both; bfs-fused is one workgroup per entry and takes the
|
|
352
|
+
// plan's GROUP count as its stride (planGridStride's cap applies to the groups)
|
|
353
|
+
const levelPlan = planGridStride(Math.max(n, frontier.edgeCapacity), wg, ctx.caps);
|
|
354
|
+
const sweepPlan = planGridStride(n, wg, ctx.caps);
|
|
355
|
+
const fusedPlan = planGridStride(n * wg, wg, ctx.caps);
|
|
356
|
+
const bitsPlan = plan1d(bitsWords, wg, ctx.caps);
|
|
357
|
+
const recordFill = (pass: GPUComputePassEncoder, dst: Binding, value: number, mode: 0 | 1): void => {
|
|
358
|
+
const params = scope.params(FILL_PARAMS, { count: n, value, mode, pad0: 0 });
|
|
359
|
+
fill.dispatch(pass, fill.bind({ dst, P: params.binding }), fillPlan, [params.offset]);
|
|
360
|
+
};
|
|
361
|
+
const flagsPlan: DispatchPlan = planGridStride(n, wg, ctx.caps);
|
|
362
|
+
/**
|
|
363
|
+
* The rebuild of the unvisited set (PD-18), recorded at the top of every submit before its levels.
|
|
364
|
+
* @param pass - the compute pass
|
|
365
|
+
*/
|
|
366
|
+
const recordRebuild = (pass: GPUComputePassEncoder): void => {
|
|
367
|
+
const params = scope.params(FRONTIER_PARAMS, { wg, n, stride: flagsPlan.stride ?? n });
|
|
368
|
+
const bound = unvisited.bind({ outDegree, inDegree, depth, flags, counters, P: params.binding });
|
|
369
|
+
unvisited.dispatch(pass, bound, flagsPlan, [params.offset]);
|
|
370
|
+
compact.record(pass, {
|
|
371
|
+
queue: iota,
|
|
372
|
+
flags,
|
|
373
|
+
count: n,
|
|
374
|
+
out: unvisitedList,
|
|
375
|
+
outCount: compactCount,
|
|
376
|
+
outIndex: 0,
|
|
377
|
+
});
|
|
378
|
+
};
|
|
379
|
+
const submit = (batch: CommandBatch): ReturnType<CommandBatch["submit"]> => {
|
|
380
|
+
scope.flush();
|
|
381
|
+
return batch.submit();
|
|
382
|
+
};
|
|
383
|
+
|
|
384
|
+
// setup: depth = INVALID_INDEX everywhere and the iota queue compact reads; then the source at 0 and the
|
|
385
|
+
// seeded block, both queue writes ordered before the first level submit (the seed is rotated in by the
|
|
386
|
+
// first boundary, P8-T4)
|
|
387
|
+
const setup = new CommandBatch(ctx, `${ALGORITHM}/setup`);
|
|
388
|
+
const setupPass = setup.pass("fill");
|
|
389
|
+
recordFill(setupPass, depth, INVALID_INDEX, 0);
|
|
390
|
+
recordFill(setupPass, iota, 0, 1);
|
|
391
|
+
setup.endPass();
|
|
392
|
+
await submit(setup).readback;
|
|
393
|
+
ctx.assertReady();
|
|
394
|
+
queue.writeBuffer(depth.buffer, depth.offset + 4 * source, Uint32Array.of(0));
|
|
395
|
+
frontier.reset(queue, source, { nextFrontierCount: 1, level: U32_MAX });
|
|
396
|
+
|
|
397
|
+
// the levels: MAX_LEVELS_PER_SUBMIT per submit, four bytes back (PD-7); Beamer's test chooses the direction
|
|
398
|
+
// per level (mode 0; mode 1 pins top-down), with the fused path below fusedMax (P8-T7)
|
|
399
|
+
const fields: FrontierFinalizeFields = {
|
|
400
|
+
mode: tuning.direction === "top-down" ? 1 : 0,
|
|
401
|
+
alpha: tuning.alpha ?? Math.max(1, Math.floor(s.arcCount / n)),
|
|
402
|
+
beta: tuning.beta ?? BEAMER_BETA,
|
|
403
|
+
fusedMax: tuning.fusedMax ?? FUSED_FRONTIER_MAX,
|
|
404
|
+
maxDepth,
|
|
405
|
+
};
|
|
406
|
+
let levelsRecorded = 0;
|
|
407
|
+
let submits = 0;
|
|
408
|
+
for (;;) {
|
|
409
|
+
// the rebuild (PD-18): the three unvisited words zeroed by a queue write ordered before this submit,
|
|
410
|
+
// then the flags kernel and compact at the top of the pass, unconditionally, never per level
|
|
411
|
+
queue.writeBuffer(counters.buffer, counters.offset + 4 * W.unvisitedCount, new Uint32Array(3));
|
|
412
|
+
const batch = new CommandBatch(ctx, `${ALGORITHM}/levels`);
|
|
413
|
+
const pass = batch.pass("bfs");
|
|
414
|
+
recordRebuild(pass);
|
|
415
|
+
const bitsParams = scope.params(FILL_PARAMS, { count: bitsWords, value: 0, mode: 0, pad0: 0 });
|
|
416
|
+
const boundBitsFill = fill.bind({ dst: frontierBits, P: bitsParams.binding });
|
|
417
|
+
for (let level = 0; level < levelsPerSubmit; level++) {
|
|
418
|
+
planner.recordFinalize(pass, 0, level, { ...fields, firstOfSubmit: Math.min(level, 2) });
|
|
419
|
+
advance.record(pass, frontier);
|
|
420
|
+
planner.recordFinalize(pass, 1, level, fields);
|
|
421
|
+
// one window-free record serves the contract and the bitset build, which read no arc; the kernels that
|
|
422
|
+
// walk arcs (the fused dispatch, the sweep) get one record per window below (P8-T12)
|
|
423
|
+
const params = scope.params(FRONTIER_PARAMS, {
|
|
424
|
+
wg,
|
|
425
|
+
n,
|
|
426
|
+
edgeCapacity: frontier.edgeCapacity,
|
|
427
|
+
arcBase: 0,
|
|
428
|
+
arcEnd: s.arcCount,
|
|
429
|
+
bitsBase,
|
|
430
|
+
stride: levelPlan.stride ?? wg,
|
|
431
|
+
});
|
|
432
|
+
const boundContract = contract.bind({
|
|
433
|
+
edgeQueue: frontier.edgeQueue,
|
|
434
|
+
counters,
|
|
435
|
+
depth,
|
|
436
|
+
frontierOut: frontier.output,
|
|
437
|
+
P: params.binding,
|
|
438
|
+
});
|
|
439
|
+
contract.dispatch(pass, boundContract, levelPlan, [params.offset]);
|
|
440
|
+
// the fused path (role 0's choice) and the overflow retry (role 1's, PD-23): ONE dispatch per window that
|
|
441
|
+
// the path word turns into the level's work or into nothing (the claim is idempotent across windows)
|
|
442
|
+
for (const w of forward) {
|
|
443
|
+
const fusedParams = scope.params(FRONTIER_PARAMS, {
|
|
444
|
+
wg,
|
|
445
|
+
n,
|
|
446
|
+
edgeCapacity: frontier.edgeCapacity,
|
|
447
|
+
arcBase: w.arcBase,
|
|
448
|
+
arcEnd: w.arcEnd,
|
|
449
|
+
stride: fusedPlan.x * fusedPlan.y,
|
|
450
|
+
});
|
|
451
|
+
const boundFused = fused.bind({
|
|
452
|
+
...graphBindings(w.core, null),
|
|
453
|
+
frontierIn: frontier.input,
|
|
454
|
+
counters,
|
|
455
|
+
depth,
|
|
456
|
+
frontierOut: frontier.output,
|
|
457
|
+
P: fusedParams.binding,
|
|
458
|
+
});
|
|
459
|
+
fused.dispatch(pass, boundFused, fusedPlan, [fusedParams.offset]);
|
|
460
|
+
}
|
|
461
|
+
// the bottom-up level (role 0's choice under Beamer's test, P8-T8): the bitset zeroed (every level: a
|
|
462
|
+
// top-down level zeroes bits nobody reads, cheaper than a gate), the frontier's bits set, the unvisited
|
|
463
|
+
// list swept over the reverse core
|
|
464
|
+
fill.dispatch(pass, boundBitsFill, bitsPlan, [bitsParams.offset]);
|
|
465
|
+
const boundBitset = bitset.bind({
|
|
466
|
+
frontierIn: frontier.input,
|
|
467
|
+
counters,
|
|
468
|
+
bits: sweepIn,
|
|
469
|
+
P: params.binding,
|
|
470
|
+
});
|
|
471
|
+
bitset.dispatch(pass, boundBitset, levelPlan, [params.offset]);
|
|
472
|
+
// the sweep once per window of the REVERSE core (an entry claimed in one window is skipped in the next)
|
|
473
|
+
for (const w of backward) {
|
|
474
|
+
const sweepParams = scope.params(FRONTIER_PARAMS, {
|
|
475
|
+
wg,
|
|
476
|
+
n,
|
|
477
|
+
arcBase: w.arcBase,
|
|
478
|
+
arcEnd: w.arcEnd,
|
|
479
|
+
bitsBase,
|
|
480
|
+
stride: sweepPlan.stride ?? wg,
|
|
481
|
+
});
|
|
482
|
+
const boundSweep = bottomUp.bind({
|
|
483
|
+
...graphBindings(w.core, null),
|
|
484
|
+
sweepIn,
|
|
485
|
+
counters,
|
|
486
|
+
depth,
|
|
487
|
+
frontierOut: frontier.output,
|
|
488
|
+
P: sweepParams.binding,
|
|
489
|
+
});
|
|
490
|
+
bottomUp.dispatch(pass, boundSweep, sweepPlan, [sweepParams.offset]);
|
|
491
|
+
}
|
|
492
|
+
frontier.swap();
|
|
493
|
+
}
|
|
494
|
+
batch.endPass();
|
|
495
|
+
const doneRequest = batch.readback(counters.buffer, counters.offset + 4 * W.done, 4);
|
|
496
|
+
const inspect =
|
|
497
|
+
tuning.onLevel === undefined
|
|
498
|
+
? null
|
|
499
|
+
: {
|
|
500
|
+
block: batch.readback(counters.buffer, counters.offset, FRONTIER_COUNTERS.byteLength),
|
|
501
|
+
frontier: batch.readback(frontier.input.buffer, frontier.input.offset, frontier.input.size),
|
|
502
|
+
count: batch.readback(compactCount.buffer, compactCount.offset, 4),
|
|
503
|
+
};
|
|
504
|
+
const submitted = submit(batch);
|
|
505
|
+
const back = await submitted.readback;
|
|
506
|
+
levelsRecorded += levelsPerSubmit;
|
|
507
|
+
submits += 1;
|
|
508
|
+
ctx.assertReady();
|
|
509
|
+
if (options?.signal?.aborted) {
|
|
510
|
+
throw aborted(submitted.id);
|
|
511
|
+
}
|
|
512
|
+
options?.onProgress?.(Math.min(levelsRecorded, n), n);
|
|
513
|
+
if (inspect !== null && tuning.onLevel !== undefined) {
|
|
514
|
+
const block = FRONTIER_COUNTERS.read(new DataView(back), inspect.block.offset);
|
|
515
|
+
const claimed = new Uint32Array(back, inspect.frontier.offset, wordOf(block, "nextFrontierCount"));
|
|
516
|
+
const rebuilt = new Uint32Array(back, inspect.count.offset, 1)[0];
|
|
517
|
+
tuning.onLevel(levelsRecorded - 1, block, claimed.slice(), rebuilt);
|
|
518
|
+
}
|
|
519
|
+
if (new Uint32Array(back, doneRequest.offset, 1)[0] !== 0) {
|
|
520
|
+
break;
|
|
521
|
+
}
|
|
522
|
+
if (submits > n + 1) {
|
|
523
|
+
throw new WebGpuGraphError(
|
|
524
|
+
"E_VALIDATION",
|
|
525
|
+
`${ALGORITHM}: the done flag never rose in ${submits} submits (a traversal has at most ${n} levels)`,
|
|
526
|
+
{ label: ALGORITHM, message: `the done flag never rose in ${submits} submits` },
|
|
527
|
+
);
|
|
528
|
+
}
|
|
529
|
+
}
|
|
530
|
+
|
|
531
|
+
// the result batch: order by one stable radix sort of the node indices by depth (PD-14), parent by the
|
|
532
|
+
// post-pass in depth mode (PD-24), then every array through ONE staging slot
|
|
533
|
+
const keys = bindingOf(scope.scratch(bytes, "order/keys"), bytes);
|
|
534
|
+
const vals = bindingOf(scope.scratch(bytes, "order/vals"), bytes);
|
|
535
|
+
const histBytes = radixHistBytes(n, wg);
|
|
536
|
+
const scratch = {
|
|
537
|
+
keys: bindingOf(scope.scratch(bytes, "order/keys-scratch"), bytes),
|
|
538
|
+
vals: bindingOf(scope.scratch(bytes, "order/vals-scratch"), bytes),
|
|
539
|
+
hist: bindingOf(scope.scratch(histBytes, "order/hist"), histBytes),
|
|
540
|
+
offsets: bindingOf(scope.scratch(histBytes, "order/offsets"), histBytes),
|
|
541
|
+
};
|
|
542
|
+
const parent = bindingOf(scope.scratch(bytes, "parent"), bytes);
|
|
543
|
+
const result = new CommandBatch(ctx, `${ALGORITHM}/result`);
|
|
544
|
+
result.copy(depth, keys, bytes);
|
|
545
|
+
const pass = result.pass("result");
|
|
546
|
+
recordFill(pass, vals, 0, 1);
|
|
547
|
+
const sorted = sort.record(pass, keys, vals, n, 32, scratch);
|
|
548
|
+
recordFill(pass, parent, INVALID_INDEX, 0);
|
|
549
|
+
// the post-pass once per arc window (its atomicMin admits the same smallest parent whichever window holds the arc)
|
|
550
|
+
const predPlan = planGridStride(n, wg, ctx.caps);
|
|
551
|
+
for (const w of forward) {
|
|
552
|
+
const predParams = scope.params(FRONTIER_PARAMS, {
|
|
553
|
+
wg,
|
|
554
|
+
n,
|
|
555
|
+
arcBase: w.arcBase,
|
|
556
|
+
arcEnd: w.arcEnd,
|
|
557
|
+
predKind,
|
|
558
|
+
source,
|
|
559
|
+
stride: predPlan.stride ?? n,
|
|
560
|
+
});
|
|
561
|
+
const predBound = pred.bind({
|
|
562
|
+
...graphBindings(w.core, null),
|
|
563
|
+
dist: depth,
|
|
564
|
+
pred: parent,
|
|
565
|
+
P: predParams.binding,
|
|
566
|
+
});
|
|
567
|
+
pred.dispatch(pass, predBound, predPlan, [predParams.offset]);
|
|
568
|
+
}
|
|
569
|
+
result.endPass();
|
|
570
|
+
const depthRequest = result.readback(depth.buffer, depth.offset, bytes);
|
|
571
|
+
const parentRequest = result.readback(parent.buffer, parent.offset, bytes);
|
|
572
|
+
const orderRequest = result.readback(sorted.vals.buffer, sorted.vals.offset, bytes);
|
|
573
|
+
const blockRequest = result.readback(counters.buffer, counters.offset, FRONTIER_COUNTERS.byteLength);
|
|
574
|
+
const back = await submit(result).readback;
|
|
575
|
+
ctx.assertReady();
|
|
576
|
+
const block = FRONTIER_COUNTERS.read(new DataView(back), blockRequest.offset);
|
|
577
|
+
const visitedCount = wordOf(block, "visitedCount");
|
|
578
|
+
if (visitedCount > n) {
|
|
579
|
+
// every vertex is claimed at most once, so a count above n is a kernel bug (a claim that lets two
|
|
580
|
+
// same-level claimants append), never a result to return
|
|
581
|
+
throw new WebGpuGraphError(
|
|
582
|
+
"E_VALIDATION",
|
|
583
|
+
`${ALGORITHM}: visitedCount ${visitedCount} exceeds the ${n} vertices (a duplicate claim)`,
|
|
584
|
+
{
|
|
585
|
+
label: `${ALGORITHM}/visitedCount`,
|
|
586
|
+
message: `the device counted ${visitedCount} visits of ${n} vertices`,
|
|
587
|
+
},
|
|
588
|
+
);
|
|
589
|
+
}
|
|
590
|
+
const level = wordOf(block, "level");
|
|
591
|
+
// done by an empty frontier (the level word already counts the boundary that found it), or by maxDepth with
|
|
592
|
+
// a reached-but-unexpanded level
|
|
593
|
+
const levels = wordOf(block, "frontierCount") === 0 ? level : level + 1;
|
|
594
|
+
const depthOut = dest ?? new Uint32Array(n);
|
|
595
|
+
depthOut.set(new Uint32Array(back, depthRequest.offset, n));
|
|
596
|
+
return {
|
|
597
|
+
depth: depthOut,
|
|
598
|
+
parent: new Uint32Array(back, parentRequest.offset, n).slice(),
|
|
599
|
+
order: new Uint32Array(back, orderRequest.offset, visitedCount).slice(),
|
|
600
|
+
visitedCount,
|
|
601
|
+
levels,
|
|
602
|
+
switches: wordOf(block, "switches"),
|
|
603
|
+
};
|
|
604
|
+
} finally {
|
|
605
|
+
scope.dispose();
|
|
606
|
+
}
|
|
607
|
+
}
|
|
608
|
+
|
|
609
|
+
/**
|
|
610
|
+
* Breadth-first search on the device (spec 3.3 line 807, design 8.4, 9.7): `depth` exact, `parent` the smallest
|
|
611
|
+
* predecessor one depth down (PD-24), `order` grouped by depth and ascending by index within a depth (PD-14), all
|
|
612
|
+
* bitwise reproducible; `maxDepth` as the CPU port reads it (a node at the cap is reached and not expanded).
|
|
613
|
+
* @param ctx - the context whose device runs the kernels
|
|
614
|
+
* @param s - the snapshot (uploaded through ctx.residency, or found there)
|
|
615
|
+
* @param source - the source node index (E_INVALID_ARGUMENT outside `[0, n)`, so the empty graph refuses every source)
|
|
616
|
+
* @param options - `maxDepth`, plus dest (a Uint32Array of length n for `depth`) / signal / onProgress
|
|
617
|
+
* @returns the depths, parents, order, visited count, level count and switches (the direction changes Beamer's test made on the device)
|
|
618
|
+
*/
|
|
619
|
+
export function breadthFirstSearch(
|
|
620
|
+
ctx: GpuContext,
|
|
621
|
+
s: GraphSnapshot,
|
|
622
|
+
source: number,
|
|
623
|
+
options?: BfsOptions & GpuRunOptions,
|
|
624
|
+
): Promise<GpuBfsResult> {
|
|
625
|
+
return bfsWithTuning(ctx, s, source, options, {});
|
|
626
|
+
}
|