@graphty/webgpu-graph-algorithms 0.6.2 → 0.6.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +62 -32
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-BXqgCifx.js → context-Dvq-Cc6v.js} +71 -25
- package/dist/chunks/context-Dvq-Cc6v.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +8 -6
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +57 -6
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/bellman-ford.d.ts +60 -0
- package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
- package/dist/src/algorithms/bellman-ford.js +301 -0
- package/dist/src/algorithms/bellman-ford.js.map +1 -0
- package/dist/src/algorithms/bfs.d.ts +67 -0
- package/dist/src/algorithms/bfs.d.ts.map +1 -0
- package/dist/src/algorithms/bfs.js +534 -0
- package/dist/src/algorithms/bfs.js.map +1 -0
- package/dist/src/algorithms/closeness.d.ts +53 -0
- package/dist/src/algorithms/closeness.d.ts.map +1 -0
- package/dist/src/algorithms/closeness.js +323 -0
- package/dist/src/algorithms/closeness.js.map +1 -0
- package/dist/src/algorithms/scope.d.ts +5 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +3 -0
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +71 -0
- package/dist/src/algorithms/sssp.d.ts.map +1 -0
- package/dist/src/algorithms/sssp.js +585 -0
- package/dist/src/algorithms/sssp.js.map +1 -0
- package/dist/src/constants.d.ts +12 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +12 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/index.d.ts +8 -2
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +7 -1
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts +4 -4
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +39 -5
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernel/uniform-ring.d.ts +8 -0
- package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
- package/dist/src/kernel/uniform-ring.js +13 -0
- package/dist/src/kernel/uniform-ring.js.map +1 -1
- package/dist/src/kernels.d.ts +44 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +371 -3
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +62 -0
- package/dist/src/primitives/advance.d.ts.map +1 -0
- package/dist/src/primitives/advance.js +95 -0
- package/dist/src/primitives/advance.js.map +1 -0
- package/dist/src/primitives/compact.d.ts +89 -0
- package/dist/src/primitives/compact.d.ts.map +1 -0
- package/dist/src/primitives/compact.js +233 -0
- package/dist/src/primitives/compact.js.map +1 -0
- package/dist/src/primitives/core-shape.d.ts +22 -1
- package/dist/src/primitives/core-shape.d.ts.map +1 -1
- package/dist/src/primitives/core-shape.js +33 -3
- package/dist/src/primitives/core-shape.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +156 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -0
- package/dist/src/primitives/frontier.js +259 -0
- package/dist/src/primitives/frontier.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +16 -7
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/traversal.d.ts +53 -0
- package/dist/src/types/traversal.d.ts.map +1 -0
- package/dist/src/types/traversal.js +10 -0
- package/dist/src/types/traversal.js.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
- package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +59 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js +210 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
- package/dist/webgpu-graph-algorithms.js +3207 -377
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +5 -4
- package/src/accelerator.ts +65 -7
- package/src/algorithms/bellman-ford.ts +387 -0
- package/src/algorithms/bfs.ts +626 -0
- package/src/algorithms/closeness.ts +395 -0
- package/src/algorithms/scope.ts +13 -3
- package/src/algorithms/sssp.ts +767 -0
- package/src/constants.ts +12 -0
- package/src/index.ts +14 -1
- package/src/kernel/prelude.ts +39 -4
- package/src/kernel/uniform-ring.ts +14 -0
- package/src/kernels.ts +450 -6
- package/src/primitives/advance.ts +130 -0
- package/src/primitives/compact.ts +323 -0
- package/src/primitives/core-shape.ts +41 -3
- package/src/primitives/frontier.ts +388 -0
- package/src/types/accelerator.ts +18 -5
- package/src/types/traversal.ts +56 -0
- package/src/wgsl/advance-expand.wgsl.ts +68 -0
- package/src/wgsl/bf-relax.wgsl.ts +57 -0
- package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
- package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
- package/src/wgsl/bfs-contract.wgsl.ts +54 -0
- package/src/wgsl/bfs-fused.wgsl.ts +77 -0
- package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
- package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
- package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
- package/src/wgsl/compact-scatter.wgsl.ts +16 -0
- package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
- package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
- package/src/wgsl/frontier-finalize.wgsl.ts +209 -0
- package/src/wgsl/sssp-pred.wgsl.ts +79 -0
- package/src/wgsl/sssp-relax.wgsl.ts +71 -0
- package/dist/chunks/context-BXqgCifx.js.map +0 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@graphty/webgpu-graph-algorithms",
|
|
3
|
-
"version": "0.6.
|
|
3
|
+
"version": "0.6.4",
|
|
4
4
|
"description": "WebGPU-accelerated graph algorithms and layouts over the @graphty/graph-format snapshot, for Node (Dawn) and browsers",
|
|
5
5
|
"author": "Adam Powers <apowers@ato.ms>",
|
|
6
6
|
"type": "module",
|
|
@@ -61,7 +61,7 @@
|
|
|
61
61
|
"homepage": "https://github.com/graphty-org/graphty-monorepo/tree/master/webgpu-graph-algorithms#readme",
|
|
62
62
|
"dependencies": {
|
|
63
63
|
"@webgpu/types": "^0.1.72",
|
|
64
|
-
"@graphty/graph-format": "^1.0.
|
|
64
|
+
"@graphty/graph-format": "^1.0.5"
|
|
65
65
|
},
|
|
66
66
|
"peerDependencies": {
|
|
67
67
|
"@graphty/algorithms": "^1.0.0 || ^2.0.0",
|
|
@@ -93,8 +93,8 @@
|
|
|
93
93
|
"vite": "^7.0.5",
|
|
94
94
|
"vitest": "^3.2.4",
|
|
95
95
|
"webgpu": "0.4.0",
|
|
96
|
-
"@graphty/
|
|
97
|
-
"@graphty/
|
|
96
|
+
"@graphty/algorithms": "^2.0.3",
|
|
97
|
+
"@graphty/layout": "^1.9.1"
|
|
98
98
|
},
|
|
99
99
|
"scripts": {
|
|
100
100
|
"build": "tsc -p tsconfig.build.json",
|
|
@@ -116,6 +116,7 @@
|
|
|
116
116
|
"coverage:preview": "npx serve coverage -p ${PORT:?start it through servherd, which sets PORT}",
|
|
117
117
|
"bench": "tsx benchmarks/run.ts",
|
|
118
118
|
"bench:compare": "node scripts/bench-compare.js",
|
|
119
|
+
"bench:append": "node scripts/bench-append-session.js",
|
|
119
120
|
"benchmark": "tsx benchmarks/run.ts",
|
|
120
121
|
"gpu:report": "node scripts/gpu-report.js",
|
|
121
122
|
"ready:commit": "npm run build:all && npm run lint && npm run test:node"
|
package/src/accelerator.ts
CHANGED
|
@@ -5,23 +5,35 @@
|
|
|
5
5
|
* `@graphty/algorithms` (W1b, algorithms half), `import type`d by src/types/accelerator.ts, so the object built
|
|
6
6
|
* here is checked against the CPU packages' own contracts. It carries P3's `forceAtlas2`, `release` and `dispose`,
|
|
7
7
|
* P5's `fruchtermanReingold` and `springElectrical` (the two other layout members of spec 9.3, landed together once
|
|
8
|
-
* both models were green, P5 PD-19)
|
|
9
|
-
*
|
|
10
|
-
*
|
|
8
|
+
* both models were green, P5 PD-19), P7's seven algorithm members (spec 8.2, 8.3; M8b-T8, PD-14) and P8's four
|
|
9
|
+
* traversal members (spec 8.4; P8-T13 PD-16, PD-19: `breadthFirstSearch`, `sssp`, `bellmanFord`,
|
|
10
|
+
* `closenessCentrality`, each taking the seam's own option type) and nothing else: the CPU-side dispatchers
|
|
11
|
+
* (`accelerated()`, `createSimulation()`) test `acc.betweennessCentrality !== undefined` /
|
|
12
|
+
* `acc.fruchtermanReingold !== undefined` and route to the CPU when the member is absent (spec 2.4 row "method
|
|
11
13
|
* missing"), so a method the GPU does not implement must not exist here -- never a throwing stub. The remaining
|
|
12
|
-
* algorithm members arrive one per shipped algorithm from
|
|
14
|
+
* algorithm members arrive one per shipped algorithm from P9.
|
|
13
15
|
*/
|
|
14
16
|
|
|
15
17
|
import { type F32, type F64, type GraphSnapshot } from "@graphty/graph-format";
|
|
16
18
|
|
|
19
|
+
import { bellmanFord } from "./algorithms/bellman-ford.js";
|
|
20
|
+
import { breadthFirstSearch } from "./algorithms/bfs.js";
|
|
21
|
+
import { closenessCentrality } from "./algorithms/closeness.js";
|
|
17
22
|
import { connectedComponents } from "./algorithms/components.js";
|
|
18
23
|
import { pageRank, personalizedPageRank } from "./algorithms/pagerank.js";
|
|
19
24
|
import { eigenvectorCentrality, hits, katzCentrality } from "./algorithms/spectral.js";
|
|
25
|
+
import { sssp } from "./algorithms/sssp.js";
|
|
20
26
|
import { type GpuContext } from "./context.js";
|
|
21
27
|
import { createForceAtlas2 } from "./layouts/forceatlas2.js";
|
|
22
28
|
import { createFruchtermanReingold } from "./layouts/fruchterman-reingold.js";
|
|
23
29
|
import { createSpringElectrical } from "./layouts/spring-electrical.js";
|
|
24
|
-
import {
|
|
30
|
+
import {
|
|
31
|
+
type AcceleratorOptions,
|
|
32
|
+
type BfsOptions,
|
|
33
|
+
type GpuAccelerator,
|
|
34
|
+
type HitsOptionsLike,
|
|
35
|
+
type SsspOptions,
|
|
36
|
+
} from "./types/accelerator.js";
|
|
25
37
|
import {
|
|
26
38
|
type ComponentsOptions,
|
|
27
39
|
type EigenvectorOptions,
|
|
@@ -45,6 +57,7 @@ import {
|
|
|
45
57
|
type FruchtermanReingoldOptions,
|
|
46
58
|
type SpringElectricalOptions,
|
|
47
59
|
} from "./types/options.js";
|
|
60
|
+
import { type GpuBellmanFordResult, type GpuBfsResult, type GpuSsspResult } from "./types/traversal.js";
|
|
48
61
|
|
|
49
62
|
/** The `algorithms` record of AcceleratorOptions (spec 3.3), named for the copy helpers. */
|
|
50
63
|
type AlgorithmDefaults = NonNullable<AcceleratorOptions["algorithms"]>;
|
|
@@ -104,8 +117,8 @@ function freezeOptions(options: AcceleratorOptions | undefined): Readonly<Accele
|
|
|
104
117
|
/**
|
|
105
118
|
* Spec 3.3 createAccelerator, verbatim: the object implementing AlgorithmAccelerator & LayoutAccelerator
|
|
106
119
|
* structurally; P3's forceAtlas2, release and dispose, P5's fruchtermanReingold and springElectrical (the same
|
|
107
|
-
* `{ ...o, ...options.layout }` shape as forceAtlas2) plus P7's seven algorithm members
|
|
108
|
-
* its algorithm with `ctx.assertReady()` first. The accelerator's algorithm defaults are not consulted by any of
|
|
120
|
+
* `{ ...o, ...options.layout }` shape as forceAtlas2) plus P7's seven algorithm members and P8's four traversal
|
|
121
|
+
* members, each a delegation to its algorithm with `ctx.assertReady()` first. The accelerator's algorithm defaults are not consulted by any of
|
|
109
122
|
* them: only `betweenness` has any, and it belongs to P9. One per call (the app creates one and injects
|
|
110
123
|
* it, spec 2.4); `kind` is "webgpu"; `options` is a frozen deep copy; `forceAtlas2(o)` is
|
|
111
124
|
* `createForceAtlas2(ctx, { ...o, ...options.layout })`, so the GPU tuning given here wins over anything the
|
|
@@ -231,6 +244,51 @@ export function createAccelerator(ctx: GpuContext, options?: AcceleratorOptions)
|
|
|
231
244
|
ctx.assertReady();
|
|
232
245
|
return await connectedComponents(ctx, gs, o);
|
|
233
246
|
},
|
|
247
|
+
/**
|
|
248
|
+
* Breadth-first search on the device (spec 8.4; P8-T13). The result is bitwise reproducible (P8 PD-14).
|
|
249
|
+
* @param gs - the snapshot
|
|
250
|
+
* @param source - the source node index
|
|
251
|
+
* @param o - the seam's `BfsOptions` (`maxDepth`)
|
|
252
|
+
* @returns depth, parent, the level-grouped order, visitedCount, levels and switches (spec 3.3 line 830)
|
|
253
|
+
*/
|
|
254
|
+
async breadthFirstSearch(gs: GraphSnapshot, source: number, o?: BfsOptions): Promise<GpuBfsResult> {
|
|
255
|
+
ctx.assertReady();
|
|
256
|
+
return await breadthFirstSearch(ctx, gs, source, o);
|
|
257
|
+
},
|
|
258
|
+
/**
|
|
259
|
+
* Single-source shortest paths on the device (spec 8.4; P8-T13): the near-far queue over f32 distances, or the
|
|
260
|
+
* breadth-first route when every weight is one.
|
|
261
|
+
* @param gs - the snapshot
|
|
262
|
+
* @param source - the source node index
|
|
263
|
+
* @param o - the seam's `SsspOptions` (`cutoff`, `weights`)
|
|
264
|
+
* @returns dist, predArc and reachedCount (spec 3.3 line 831)
|
|
265
|
+
*/
|
|
266
|
+
async sssp(gs: GraphSnapshot, source: number, o?: SsspOptions): Promise<GpuSsspResult> {
|
|
267
|
+
ctx.assertReady();
|
|
268
|
+
return await sssp(ctx, gs, source, o);
|
|
269
|
+
},
|
|
270
|
+
/**
|
|
271
|
+
* Bellman-Ford on the device with negative-cycle detection (spec 8.4; P8-T13).
|
|
272
|
+
* @param gs - the snapshot
|
|
273
|
+
* @param source - the source node index
|
|
274
|
+
* @param o - the seam's `SsspOptions` (`cutoff`, `weights`)
|
|
275
|
+
* @returns dist, predArc, reachedCount and hasNegativeCycle (spec 3.3 line 832)
|
|
276
|
+
*/
|
|
277
|
+
async bellmanFord(gs: GraphSnapshot, source: number, o?: SsspOptions): Promise<GpuBellmanFordResult> {
|
|
278
|
+
ctx.assertReady();
|
|
279
|
+
return await bellmanFord(ctx, gs, source, o);
|
|
280
|
+
},
|
|
281
|
+
/**
|
|
282
|
+
* Closeness centrality on the device (spec 8.4; P8-T13): the bit-parallel multi-source sweep, or one `sssp`
|
|
283
|
+
* per source when `weighted`. `maxIterations` / `tolerance` are refused when defined (P8 PD-25).
|
|
284
|
+
* @param gs - the snapshot
|
|
285
|
+
* @param o - the seam's placeholder `HitsOptionsLike` (`weighted`)
|
|
286
|
+
* @returns the f32 scores with `precision: "f32"` (spec 9.7)
|
|
287
|
+
*/
|
|
288
|
+
async closenessCentrality(gs: GraphSnapshot, o?: HitsOptionsLike): Promise<GpuScoresResult> {
|
|
289
|
+
ctx.assertReady();
|
|
290
|
+
return await closenessCentrality(ctx, gs, o);
|
|
291
|
+
},
|
|
234
292
|
/**
|
|
235
293
|
* Destroys every device buffer recorded for the snapshot (spec 4.5); delegates to ctx.release.
|
|
236
294
|
* @param s - the snapshot the app is done with
|
|
@@ -0,0 +1,387 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Bellman-Ford with negative-cycle detection on the device (design 8.4, 3.3 line 809, 9.7; P8-T10, the P8 plan's
|
|
3
|
+
* PD-12 / PD-19 / PD-22 / PD-27 / DEP-P8-E): the edge-parallel relaxation of `bf-relax` over the `edgeList()` view
|
|
4
|
+
* -- every logical edge once, both directions on an undirected snapshot -- in batches of `ROUNDS_PER_BATCH` rounds
|
|
5
|
+
* with the `BfFlags` block (`changed`, `retryExhausted`) zeroed before each batch and read back after it, one
|
|
6
|
+
* `mapAsync` per batch. The loop stops when a batch changed nothing (a reachable negative cycle changes something
|
|
7
|
+
* every round, so nothing has been missed) or when `n - 1` rounds have run, in which case ONE more round runs and
|
|
8
|
+
* `changed` after it IS `hasNegativeCycle`; the last batch is clamped so the reported round count never exceeds
|
|
9
|
+
* `n - 1`. `dist` holds f32 bit patterns and the claim is a BOUNDED compare-exchange (PD-12: `MAX_RETRIES`, because
|
|
10
|
+
* with a negative distance the bit-pattern order reverses and `atomicMin` is wrong, and WGSL lets the exchange fail
|
|
11
|
+
* spuriously): a lane that exhausts the bound raises `retryExhausted`, which the driver treats as changed and runs
|
|
12
|
+
* on -- the lost update is retried by the next round, which examines every edge anyway -- and tallies for the tests;
|
|
13
|
+
* raised in the decision round it is E_VALIDATION, never a guess.
|
|
14
|
+
*
|
|
15
|
+
* The routing is P8-T9's, by the RUN's weight vector (PD-22): `options.weights` when given (`arcCount` long or
|
|
16
|
+
* E_INVALID_ARGUMENT; narrowed to `Float32Array` when it is not one -- a `U32` or `I32` value above 2^24, or the bits
|
|
17
|
+
* of an `F64` value below the f32 ulp, is rounded silently, the package-wide f32 caveat and not a refusal), else the
|
|
18
|
+
* snapshot's column, else none; a run whose vector is all ones, or has none, is the unit-weight BFS of P8-T6 with a
|
|
19
|
+
* depth cap derived from `cutoff` (there is nothing negative to relax) and the flag false; a non-finite weight is
|
|
20
|
+
* E_UNSUPPORTED `bellmanFord.nonFiniteWeights`; `cutoff: NaN` is E_INVALID_ARGUMENT (PD-19); every other cutoff goes
|
|
21
|
+
* to the kernel's `nd <= cutoff` unchanged, a negative one yielding the source alone only when no negative weight
|
|
22
|
+
* can satisfy it. One rule the kernel's shape forces: on an UNDIRECTED snapshot the kernel reads ONE weight per
|
|
23
|
+
* logical edge (its forward arc's) and applies it in both directions, so a caller-supplied override whose two arcs
|
|
24
|
+
* of one edge differ means the caller wanted a directed graph, and the driver REFUSES it (E_UNSUPPORTED
|
|
25
|
+
* `bellmanFord.asymmetricUndirectedWeights`, before any device work) rather than half-honour it; the snapshot's own
|
|
26
|
+
* column never trips it. `edgeToArc` is absent from the residency on a directed snapshot built from a sorted edge
|
|
27
|
+
* list (`flags.arcToEdgeIsIdentity`): edge `e` IS arc `e` there, and the driver binds an `edgeCount`-word iota
|
|
28
|
+
* scratch in that slot (the `fill` kernel's mode 1) so the kernel reads `weights[edgeToArc[e]]` unchanged.
|
|
29
|
+
*
|
|
30
|
+
* `predArc` is P8-T9's predecessor pass under PD-27's tight-subgraph rule (`P.mode 1`): the key is the hop count
|
|
31
|
+
* from the source over every tight arc (`fround(dist[u] + w) == dist[v]`; a tight arc with a negative weight steps to
|
|
32
|
+
* a LARGER distance, so `dist` is no key here), `hops[source] = 0` the only seed and no roots pass, so the chain
|
|
33
|
+
* strictly decreases the key and ends at the source. An ORPHAN -- a reached node the tight subgraph never reaches --
|
|
34
|
+
* can arise only when f32 rounding let a cycle improve a distance once (`s -> a` at 2^24, `a -> b` at +1, `b -> a` at
|
|
35
|
+
* -1: `dist[a]` ends at `2^24 - 1` and the source's arc is no longer tight; exact arithmetic has no such run), and the
|
|
36
|
+
* driver refuses it as E_UNSUPPORTED `bellmanFord.roundedCycle` rather than return a chain that ends short of the
|
|
37
|
+
* source (the seam's walker would hand `arcSourceIn` an `INVALID_INDEX`). With `hasNegativeCycle` the result carries
|
|
38
|
+
* the last round's `dist` and `predArc` and says so: the pass still runs, its chains are still acyclic, a node the
|
|
39
|
+
* tight subgraph does not reach keeps `INVALID_INDEX`, and the orphan refusal is skipped. The tuning entry
|
|
40
|
+
* `bellmanFordWithTuning` (PD-26's shape) is what the tests drive; nothing public exposes it.
|
|
41
|
+
*/
|
|
42
|
+
|
|
43
|
+
import { type F32, type GraphSnapshot, INVALID_INDEX } from "@graphty/graph-format";
|
|
44
|
+
|
|
45
|
+
import { F32_INF_BITS, MAX_LEVELS_PER_SUBMIT } from "../constants.js";
|
|
46
|
+
import { type GpuContext } from "../context.js";
|
|
47
|
+
import { WebGpuGraphError } from "../errors.js";
|
|
48
|
+
import { CommandBatch } from "../kernel/batch.js";
|
|
49
|
+
import { plan1d, planGridStride } from "../kernel/dispatch.js";
|
|
50
|
+
import { BF_FLAGS, BF_PARAMS, FILL_PARAMS, graphBindings, graphOverrides, kernelSpec } from "../kernels.js";
|
|
51
|
+
import { assertWholeCore } from "../primitives/core-shape.js";
|
|
52
|
+
import { assertDeviceComputes } from "../primitives/verify.js";
|
|
53
|
+
import { type SsspOptions } from "../types/accelerator.js";
|
|
54
|
+
import { type Binding } from "../types/memory.js";
|
|
55
|
+
import { type GpuRunOptions } from "../types/run.js";
|
|
56
|
+
import { type GpuBellmanFordResult } from "../types/traversal.js";
|
|
57
|
+
import { algorithmScope } from "./scope.js";
|
|
58
|
+
import {
|
|
59
|
+
aborted,
|
|
60
|
+
assertSource,
|
|
61
|
+
bindingOf,
|
|
62
|
+
bitsOf,
|
|
63
|
+
checkDest,
|
|
64
|
+
normaliseCutoff,
|
|
65
|
+
predBufferWords,
|
|
66
|
+
predecessorPass,
|
|
67
|
+
resolveWeights,
|
|
68
|
+
unitWeightRoute,
|
|
69
|
+
type WeightVector,
|
|
70
|
+
} from "./sssp.js";
|
|
71
|
+
|
|
72
|
+
const ALGORITHM = "bellmanFord";
|
|
73
|
+
|
|
74
|
+
/** Design 8.4: the changed flag is read every 8 rounds (one `mapAsync` per batch). */
|
|
75
|
+
const ROUNDS_PER_BATCH = 8;
|
|
76
|
+
|
|
77
|
+
/** PD-12: the compare-exchange retry bound per lane (contention is per vertex, not global); `components.ts`'s `MAX_STEPS` idiom. */
|
|
78
|
+
const MAX_RETRIES = 16;
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Params slots of the ring, COUNTED (`UniformRing.reserve` wraps silently): a batch of rounds shares ONE `BfParams`
|
|
82
|
+
* record across its dispatches; the setup batch is three `fill`s; the predecessor batch is `MAX_LEVELS_PER_SUBMIT`
|
|
83
|
+
* hop passes, one `fill` and the predecessor pass (34).
|
|
84
|
+
*/
|
|
85
|
+
const RING_SLOTS = MAX_LEVELS_PER_SUBMIT + 16;
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* The knobs the tests need and nothing public offers (PD-26's shape): the retry bound and the batch cadence.
|
|
89
|
+
* @internal
|
|
90
|
+
*/
|
|
91
|
+
export interface BellmanFordTuning {
|
|
92
|
+
/** The compare-exchange retry bound per lane (default `MAX_RETRIES`); 1 makes every failed exchange exhaust it. */
|
|
93
|
+
readonly maxRetries?: number | undefined;
|
|
94
|
+
/** Rounds recorded per batch (default `ROUNDS_PER_BATCH`), each batch one readback of the flags. */
|
|
95
|
+
readonly roundsPerBatch?: number | undefined;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* The tuning entry's answer: the result plus what the tests pin.
|
|
100
|
+
* @internal
|
|
101
|
+
*/
|
|
102
|
+
export interface BellmanFordRun {
|
|
103
|
+
readonly result: GpuBellmanFordResult;
|
|
104
|
+
/** The rounds dispatched before the decision round (at most `n - 1`; 0 on the unit-weight route). */
|
|
105
|
+
readonly rounds: number;
|
|
106
|
+
/** The batches in which some lane exhausted the retry bound (PD-12); 0 on every fixture of the suite. */
|
|
107
|
+
readonly retryExhaustedRounds: number;
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* The undirected rule (see the file comment): every arc's weight equals its edge's forward arc's, else E_UNSUPPORTED.
|
|
112
|
+
* @param s - the snapshot (undirected)
|
|
113
|
+
* @param vector - the caller's override, narrowed
|
|
114
|
+
*/
|
|
115
|
+
function assertSymmetric(s: GraphSnapshot, vector: F32): void {
|
|
116
|
+
const { arcToEdge, edgeToArc } = s;
|
|
117
|
+
for (let a = 0; a < s.arcCount; a++) {
|
|
118
|
+
const forward = edgeToArc[arcToEdge[a]];
|
|
119
|
+
if (vector[a] !== vector[forward]) {
|
|
120
|
+
throw new WebGpuGraphError(
|
|
121
|
+
"E_UNSUPPORTED",
|
|
122
|
+
`${ALGORITHM}: weights[${a}] = ${vector[a]} differs from weights[${forward}] = ${vector[forward]}, the forward arc of the same undirected edge; the kernel reads one weight per edge`,
|
|
123
|
+
{ feature: "bellmanFord.asymmetricUndirectedWeights", hint: "use a directed snapshot" },
|
|
124
|
+
);
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/**
|
|
130
|
+
* Bellman-Ford with the test knobs of PD-26's shape; `bellmanFord` is this with an empty tuning.
|
|
131
|
+
* @internal
|
|
132
|
+
* @param ctx - the context whose device runs the kernels
|
|
133
|
+
* @param s - the snapshot (uploaded through ctx.residency, or found there)
|
|
134
|
+
* @param source - the source node index
|
|
135
|
+
* @param options - `cutoff` and `weights`, plus dest / signal / onProgress
|
|
136
|
+
* @param tuning - the knobs
|
|
137
|
+
* @returns the result, the rounds run and the retry-exhausted tally
|
|
138
|
+
*/
|
|
139
|
+
export async function bellmanFordWithTuning(
|
|
140
|
+
ctx: GpuContext,
|
|
141
|
+
s: GraphSnapshot,
|
|
142
|
+
source: number,
|
|
143
|
+
options: (SsspOptions & GpuRunOptions) | undefined,
|
|
144
|
+
tuning: BellmanFordTuning,
|
|
145
|
+
): Promise<BellmanFordRun> {
|
|
146
|
+
ctx.assertReady();
|
|
147
|
+
await assertDeviceComputes(ctx);
|
|
148
|
+
const n = s.nodeCount;
|
|
149
|
+
assertSource(ALGORITHM, source, n);
|
|
150
|
+
const maxRetries = tuning.maxRetries ?? MAX_RETRIES;
|
|
151
|
+
if (!Number.isInteger(maxRetries) || maxRetries < 1) {
|
|
152
|
+
throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM}: maxRetries must be an integer >= 1`, {
|
|
153
|
+
argument: "maxRetries",
|
|
154
|
+
value: maxRetries,
|
|
155
|
+
expected: "an integer >= 1",
|
|
156
|
+
});
|
|
157
|
+
}
|
|
158
|
+
const roundsPerBatch = tuning.roundsPerBatch ?? ROUNDS_PER_BATCH;
|
|
159
|
+
if (!Number.isInteger(roundsPerBatch) || roundsPerBatch < 1 || roundsPerBatch > MAX_LEVELS_PER_SUBMIT) {
|
|
160
|
+
throw new WebGpuGraphError(
|
|
161
|
+
"E_INVALID_ARGUMENT",
|
|
162
|
+
`${ALGORITHM}: roundsPerBatch must be an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
|
|
163
|
+
{
|
|
164
|
+
argument: "roundsPerBatch",
|
|
165
|
+
value: roundsPerBatch,
|
|
166
|
+
expected: `an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
|
|
167
|
+
},
|
|
168
|
+
);
|
|
169
|
+
}
|
|
170
|
+
const dest = checkDest(ALGORITHM, options?.dest, n);
|
|
171
|
+
const vector: WeightVector | null = resolveWeights(ALGORITHM, s, options?.weights);
|
|
172
|
+
const cutoff = normaliseCutoff(ALGORITHM, options?.cutoff);
|
|
173
|
+
if (options?.signal?.aborted) {
|
|
174
|
+
throw aborted(ALGORITHM);
|
|
175
|
+
}
|
|
176
|
+
if (vector === null || vector.allOne) {
|
|
177
|
+
const unit = await unitWeightRoute(ctx, s, source, cutoff, dest, options);
|
|
178
|
+
return { result: { ...unit, hasNegativeCycle: false }, rounds: 0, retryExhaustedRounds: 0 };
|
|
179
|
+
}
|
|
180
|
+
if (!vector.finite) {
|
|
181
|
+
throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: a NaN or infinite weight has no shortest path`, {
|
|
182
|
+
feature: "bellmanFord.nonFiniteWeights",
|
|
183
|
+
});
|
|
184
|
+
}
|
|
185
|
+
if (!s.directed && vector.override !== null) {
|
|
186
|
+
assertSymmetric(s, vector.override);
|
|
187
|
+
}
|
|
188
|
+
const { arcCount } = s;
|
|
189
|
+
const core = ctx.residency.core(s, ["rowPtr", "colIdx", "weights", "edgeToArc"]);
|
|
190
|
+
assertWholeCore(core, arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM);
|
|
191
|
+
const edges = ctx.residency.view(s, "edgeList");
|
|
192
|
+
const edgeCount = edges.scalars.edgeCount[0];
|
|
193
|
+
const scope = algorithmScope(ctx, ALGORITHM, RING_SLOTS);
|
|
194
|
+
try {
|
|
195
|
+
const wg = ctx.workgroupSize;
|
|
196
|
+
const bytes = 4 * n;
|
|
197
|
+
const dist = bindingOf(scope.scratch(bytes, "dist"), bytes);
|
|
198
|
+
const predWords = predBufferWords(n);
|
|
199
|
+
const pred = bindingOf(scope.scratch(4 * predWords, "pred"), 4 * predWords);
|
|
200
|
+
const flags = bindingOf(scope.scratch(BF_FLAGS.byteLength, "flags"), BF_FLAGS.byteLength);
|
|
201
|
+
// the directed identity snapshot has no edgeToArc segment: edge e IS arc e, so an iota stands in the slot
|
|
202
|
+
let iota: Binding | null = null;
|
|
203
|
+
let { edgeToArc } = core;
|
|
204
|
+
if (edgeToArc === null) {
|
|
205
|
+
iota = bindingOf(scope.scratch(4 * edgeCount, "iota"), 4 * edgeCount);
|
|
206
|
+
edgeToArc = iota;
|
|
207
|
+
}
|
|
208
|
+
const { queue } = ctx.device;
|
|
209
|
+
let weightsBinding: Binding | undefined;
|
|
210
|
+
if (vector.override !== null) {
|
|
211
|
+
const uploaded = scope.scratch(4 * arcCount, "weights");
|
|
212
|
+
queue.writeBuffer(uploaded, 0, vector.override);
|
|
213
|
+
weightsBinding = bindingOf(uploaded, 4 * arcCount);
|
|
214
|
+
}
|
|
215
|
+
await ctx.allocator.check();
|
|
216
|
+
const overrides = graphOverrides(core, null, weightsBinding);
|
|
217
|
+
const relax = await ctx.pipelines.kernel(kernelSpec("bf-relax", { UNDIRECTED: !s.directed }));
|
|
218
|
+
const predKernel = await ctx.pipelines.kernel(kernelSpec("sssp-pred", { ...overrides, MODE: 0 }));
|
|
219
|
+
const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
|
|
220
|
+
const graph = graphBindings(core, null, weightsBinding);
|
|
221
|
+
const recordFill = (
|
|
222
|
+
pass: GPUComputePassEncoder,
|
|
223
|
+
dst: Binding,
|
|
224
|
+
count: number,
|
|
225
|
+
value: number,
|
|
226
|
+
mode = 0,
|
|
227
|
+
): void => {
|
|
228
|
+
const params = scope.params(FILL_PARAMS, { count, value, mode, pad0: 0 });
|
|
229
|
+
fill.dispatch(pass, fill.bind({ dst, P: params.binding }), plan1d(count, wg, ctx.caps), [params.offset]);
|
|
230
|
+
};
|
|
231
|
+
const submit = (batch: CommandBatch): ReturnType<CommandBatch["submit"]> => {
|
|
232
|
+
scope.flush();
|
|
233
|
+
return batch.submit();
|
|
234
|
+
};
|
|
235
|
+
|
|
236
|
+
// setup: dist = +Inf everywhere, the pred buffer INVALID_INDEX, the iota when needed; then the source at 0
|
|
237
|
+
const setup = new CommandBatch(ctx, `${ALGORITHM}/setup`);
|
|
238
|
+
const setupPass = setup.pass("fill");
|
|
239
|
+
recordFill(setupPass, dist, n, F32_INF_BITS);
|
|
240
|
+
recordFill(setupPass, pred, predWords, INVALID_INDEX);
|
|
241
|
+
if (iota !== null) {
|
|
242
|
+
recordFill(setupPass, iota, edgeCount, 0, 1);
|
|
243
|
+
}
|
|
244
|
+
setup.endPass();
|
|
245
|
+
await submit(setup).readback;
|
|
246
|
+
ctx.assertReady();
|
|
247
|
+
queue.writeBuffer(dist.buffer, dist.offset + 4 * source, Uint32Array.of(0));
|
|
248
|
+
|
|
249
|
+
// the rounds: roundsPerBatch per batch, the flags zeroed before and read after; the last batch clamped to
|
|
250
|
+
// n - 1 rounds in all, then the decision round
|
|
251
|
+
const edgePlan = planGridStride(edgeCount, wg, ctx.caps);
|
|
252
|
+
const relaxBindings = {
|
|
253
|
+
edgeSrc: edges.bindings.src,
|
|
254
|
+
edgeDst: edges.bindings.dst,
|
|
255
|
+
edgeToArc,
|
|
256
|
+
weights: graph.weights,
|
|
257
|
+
dist,
|
|
258
|
+
flags,
|
|
259
|
+
};
|
|
260
|
+
const relaxFields = {
|
|
261
|
+
edgeCount,
|
|
262
|
+
stride: edgePlan.stride ?? edgeCount,
|
|
263
|
+
maxRetries,
|
|
264
|
+
cutoffBits: bitsOf(cutoff),
|
|
265
|
+
};
|
|
266
|
+
const zero = new Uint32Array(BF_FLAGS.byteLength / 4);
|
|
267
|
+
const runRounds = async (
|
|
268
|
+
count: number,
|
|
269
|
+
label: string,
|
|
270
|
+
): Promise<{ readonly changed: number; readonly retryExhausted: number; readonly id: number }> => {
|
|
271
|
+
queue.writeBuffer(flags.buffer, flags.offset, zero);
|
|
272
|
+
const batch = new CommandBatch(ctx, `${ALGORITHM}/${label}`);
|
|
273
|
+
const pass = batch.pass("relax");
|
|
274
|
+
const params = scope.params(BF_PARAMS, relaxFields);
|
|
275
|
+
const bound = relax.bind({ ...relaxBindings, P: params.binding });
|
|
276
|
+
for (let round = 0; round < count; round++) {
|
|
277
|
+
relax.dispatch(pass, bound, edgePlan, [params.offset]);
|
|
278
|
+
}
|
|
279
|
+
batch.endPass();
|
|
280
|
+
const request = batch.readback(flags.buffer, flags.offset, BF_FLAGS.byteLength);
|
|
281
|
+
const submitted = submit(batch);
|
|
282
|
+
const back = await submitted.readback;
|
|
283
|
+
ctx.assertReady();
|
|
284
|
+
const block = BF_FLAGS.read(new DataView(back), request.offset);
|
|
285
|
+
return { changed: Number(block.changed), retryExhausted: Number(block.retryExhausted), id: submitted.id };
|
|
286
|
+
};
|
|
287
|
+
let rounds = 0;
|
|
288
|
+
let retryExhaustedRounds = 0;
|
|
289
|
+
let hasNegativeCycle = false;
|
|
290
|
+
for (;;) {
|
|
291
|
+
const remaining = n - 1 - rounds;
|
|
292
|
+
if (remaining <= 0) {
|
|
293
|
+
// the decision round: a change after n - 1 rounds is a negative cycle reachable from the source
|
|
294
|
+
const decision = await runRounds(1, "decision");
|
|
295
|
+
if (decision.retryExhausted !== 0) {
|
|
296
|
+
throw new WebGpuGraphError(
|
|
297
|
+
"E_VALIDATION",
|
|
298
|
+
`${ALGORITHM}: a lane exhausted the ${maxRetries}-retry compare-exchange bound in the decision round, so its change is not a verdict`,
|
|
299
|
+
{
|
|
300
|
+
label: `${ALGORITHM}/retry`,
|
|
301
|
+
message: "retryExhausted in the decision round",
|
|
302
|
+
batchId: decision.id,
|
|
303
|
+
},
|
|
304
|
+
);
|
|
305
|
+
}
|
|
306
|
+
hasNegativeCycle = decision.changed !== 0;
|
|
307
|
+
break;
|
|
308
|
+
}
|
|
309
|
+
const count = Math.min(roundsPerBatch, remaining);
|
|
310
|
+
const batch = await runRounds(count, "rounds");
|
|
311
|
+
rounds += count;
|
|
312
|
+
if (batch.retryExhausted !== 0) {
|
|
313
|
+
retryExhaustedRounds += 1;
|
|
314
|
+
}
|
|
315
|
+
if (options?.signal?.aborted) {
|
|
316
|
+
throw aborted(ALGORITHM, batch.id);
|
|
317
|
+
}
|
|
318
|
+
options?.onProgress?.(rounds, n);
|
|
319
|
+
if (batch.changed === 0 && batch.retryExhausted === 0) {
|
|
320
|
+
break;
|
|
321
|
+
}
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
// the predecessor pass (PD-11, PD-27) under the tight-subgraph rule; the flags block was read after the
|
|
325
|
+
// decision round, so the batch reads dist, the arcs and the pass's two words
|
|
326
|
+
const passed = await predecessorPass({
|
|
327
|
+
algorithm: ALGORITHM,
|
|
328
|
+
ctx,
|
|
329
|
+
scope,
|
|
330
|
+
predKernel,
|
|
331
|
+
recordFill,
|
|
332
|
+
graph,
|
|
333
|
+
dist,
|
|
334
|
+
pred,
|
|
335
|
+
n,
|
|
336
|
+
arcCount,
|
|
337
|
+
source,
|
|
338
|
+
mode: 1,
|
|
339
|
+
});
|
|
340
|
+
if (passed.orphans !== 0 && !hasNegativeCycle) {
|
|
341
|
+
throw new WebGpuGraphError(
|
|
342
|
+
"E_UNSUPPORTED",
|
|
343
|
+
`${ALGORITHM}: ${passed.orphans} reached node(s) the tight subgraph never reaches (a cycle of weights below one f32 ulp relaxed once)`,
|
|
344
|
+
{
|
|
345
|
+
feature: "bellmanFord.roundedCycle",
|
|
346
|
+
hint: "a cycle of weights below one f32 ulp relaxed once at a distance above 2^24; scale the weights or shorten the distances",
|
|
347
|
+
},
|
|
348
|
+
);
|
|
349
|
+
}
|
|
350
|
+
const distOut = dest ?? new Float32Array(n);
|
|
351
|
+
distOut.set(passed.dist);
|
|
352
|
+
let reachedCount = 0;
|
|
353
|
+
for (const d of distOut) {
|
|
354
|
+
if (d !== Infinity) {
|
|
355
|
+
reachedCount += 1;
|
|
356
|
+
}
|
|
357
|
+
}
|
|
358
|
+
return {
|
|
359
|
+
result: { dist: distOut, predArc: passed.predArc, reachedCount, hasNegativeCycle },
|
|
360
|
+
rounds,
|
|
361
|
+
retryExhaustedRounds,
|
|
362
|
+
};
|
|
363
|
+
} finally {
|
|
364
|
+
scope.dispose();
|
|
365
|
+
}
|
|
366
|
+
}
|
|
367
|
+
|
|
368
|
+
/**
|
|
369
|
+
* Bellman-Ford on the device (spec 3.3 line 809, design 8.4, 9.7): `dist` in f32 (bitwise the f32 Bellman-Ford
|
|
370
|
+
* fixed point, which is `sssp`'s where the weights are non-negative), `predArc` a tight arc one hop of the tight
|
|
371
|
+
* subgraph below each reached node (the chain always ends at the source), `reachedCount`, and `hasNegativeCycle` --
|
|
372
|
+
* true when a negative cycle is reachable from the source, in which case `dist` and `predArc` are the last round's
|
|
373
|
+
* values, not shortest paths; `cutoff` and `weights` as `sssp` reads them (the seam's `SsspOptions`).
|
|
374
|
+
* @param ctx - the context whose device runs the kernels
|
|
375
|
+
* @param s - the snapshot (uploaded through ctx.residency, or found there)
|
|
376
|
+
* @param source - the source node index (E_INVALID_ARGUMENT outside `[0, n)`)
|
|
377
|
+
* @param options - `cutoff` and `weights`, plus dest (a Float32Array of length n for `dist`) / signal / onProgress
|
|
378
|
+
* @returns the distances, the predecessor arcs, the reached count and the negative-cycle flag
|
|
379
|
+
*/
|
|
380
|
+
export async function bellmanFord(
|
|
381
|
+
ctx: GpuContext,
|
|
382
|
+
s: GraphSnapshot,
|
|
383
|
+
source: number,
|
|
384
|
+
options?: SsspOptions & GpuRunOptions,
|
|
385
|
+
): Promise<GpuBellmanFordResult> {
|
|
386
|
+
return (await bellmanFordWithTuning(ctx, s, source, options, {})).result;
|
|
387
|
+
}
|