@graphty/webgpu-graph-algorithms 0.6.3 → 0.6.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +62 -32
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-BXqgCifx.js → context-Dvq-Cc6v.js} +71 -25
- package/dist/chunks/context-Dvq-Cc6v.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +8 -6
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +57 -6
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/bellman-ford.d.ts +60 -0
- package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
- package/dist/src/algorithms/bellman-ford.js +301 -0
- package/dist/src/algorithms/bellman-ford.js.map +1 -0
- package/dist/src/algorithms/bfs.d.ts +67 -0
- package/dist/src/algorithms/bfs.d.ts.map +1 -0
- package/dist/src/algorithms/bfs.js +534 -0
- package/dist/src/algorithms/bfs.js.map +1 -0
- package/dist/src/algorithms/closeness.d.ts +53 -0
- package/dist/src/algorithms/closeness.d.ts.map +1 -0
- package/dist/src/algorithms/closeness.js +323 -0
- package/dist/src/algorithms/closeness.js.map +1 -0
- package/dist/src/algorithms/scope.d.ts +5 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +3 -0
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +71 -0
- package/dist/src/algorithms/sssp.d.ts.map +1 -0
- package/dist/src/algorithms/sssp.js +585 -0
- package/dist/src/algorithms/sssp.js.map +1 -0
- package/dist/src/constants.d.ts +12 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +12 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/index.d.ts +8 -2
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +7 -1
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts +4 -4
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +39 -5
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernel/uniform-ring.d.ts +8 -0
- package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
- package/dist/src/kernel/uniform-ring.js +13 -0
- package/dist/src/kernel/uniform-ring.js.map +1 -1
- package/dist/src/kernels.d.ts +44 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +371 -3
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +62 -0
- package/dist/src/primitives/advance.d.ts.map +1 -0
- package/dist/src/primitives/advance.js +95 -0
- package/dist/src/primitives/advance.js.map +1 -0
- package/dist/src/primitives/compact.d.ts +89 -0
- package/dist/src/primitives/compact.d.ts.map +1 -0
- package/dist/src/primitives/compact.js +233 -0
- package/dist/src/primitives/compact.js.map +1 -0
- package/dist/src/primitives/core-shape.d.ts +22 -1
- package/dist/src/primitives/core-shape.d.ts.map +1 -1
- package/dist/src/primitives/core-shape.js +33 -3
- package/dist/src/primitives/core-shape.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +156 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -0
- package/dist/src/primitives/frontier.js +259 -0
- package/dist/src/primitives/frontier.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +16 -7
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/traversal.d.ts +53 -0
- package/dist/src/types/traversal.d.ts.map +1 -0
- package/dist/src/types/traversal.js +10 -0
- package/dist/src/types/traversal.js.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
- package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +59 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js +210 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
- package/dist/webgpu-graph-algorithms.js +3207 -377
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +5 -4
- package/src/accelerator.ts +65 -7
- package/src/algorithms/bellman-ford.ts +387 -0
- package/src/algorithms/bfs.ts +626 -0
- package/src/algorithms/closeness.ts +395 -0
- package/src/algorithms/scope.ts +13 -3
- package/src/algorithms/sssp.ts +767 -0
- package/src/constants.ts +12 -0
- package/src/index.ts +14 -1
- package/src/kernel/prelude.ts +39 -4
- package/src/kernel/uniform-ring.ts +14 -0
- package/src/kernels.ts +450 -6
- package/src/primitives/advance.ts +130 -0
- package/src/primitives/compact.ts +323 -0
- package/src/primitives/core-shape.ts +41 -3
- package/src/primitives/frontier.ts +388 -0
- package/src/types/accelerator.ts +18 -5
- package/src/types/traversal.ts +56 -0
- package/src/wgsl/advance-expand.wgsl.ts +68 -0
- package/src/wgsl/bf-relax.wgsl.ts +57 -0
- package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
- package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
- package/src/wgsl/bfs-contract.wgsl.ts +54 -0
- package/src/wgsl/bfs-fused.wgsl.ts +77 -0
- package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
- package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
- package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
- package/src/wgsl/compact-scatter.wgsl.ts +16 -0
- package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
- package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
- package/src/wgsl/frontier-finalize.wgsl.ts +209 -0
- package/src/wgsl/sssp-pred.wgsl.ts +79 -0
- package/src/wgsl/sssp-relax.wgsl.ts +71 -0
- package/dist/chunks/context-BXqgCifx.js.map +0 -1
package/README.md
CHANGED
|
@@ -3,11 +3,11 @@
|
|
|
3
3
|
WebGPU-accelerated graph algorithms and layouts over the `@graphty/graph-format` snapshot, for Node
|
|
4
4
|
(Dawn, through the `webgpu` npm package) and browsers (Chromium). One code base, three entry points:
|
|
5
5
|
|
|
6
|
-
| Entry | Import | What it gives you
|
|
7
|
-
| ------------------------------------------ | ------------- |
|
|
6
|
+
| Entry | Import | What it gives you |
|
|
7
|
+
| ------------------------------------------ | ------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
8
8
|
| `@graphty/webgpu-graph-algorithms` | the core | the layouts (`createForceAtlas2`, `createFruchtermanReingold`, `createSpringElectrical`, `seedPositions`), the algorithms (`pageRank`, `personalizedPageRank`, `hits`, `eigenvectorCentrality`, `katzCentrality`, `connectedComponents`, `degree`), `createAccelerator`, `calibrateLayout`, `verifyDevice`, `GpuContext`, `WebGpuGraphError`, `isSoftwareAdapter`, the constants (`EXACT_MAX_NODES`, `FA2_DEFAULTS`, `FR_DEFAULTS`, `SE_DEFAULTS`, `LAYOUT_TUNING_DEFAULTS`, ...) and the option / stats / accelerator types |
|
|
9
|
-
| `@graphty/webgpu-graph-algorithms/node` | Node only | `createNodeGpuContext`, `probeNodeWebGpu`, `createNodeGpu` (Dawn), `dawnFlags`
|
|
10
|
-
| `@graphty/webgpu-graph-algorithms/browser` | browsers only | `probeBrowserWebGpu`, `requestGpuContext`
|
|
9
|
+
| `@graphty/webgpu-graph-algorithms/node` | Node only | `createNodeGpuContext`, `probeNodeWebGpu`, `createNodeGpu` (Dawn), `dawnFlags` |
|
|
10
|
+
| `@graphty/webgpu-graph-algorithms/browser` | browsers only | `probeBrowserWebGpu`, `requestGpuContext` |
|
|
11
11
|
|
|
12
12
|
**Status: three force layouts and six algorithms, on the exact and the grid repulsion tiers.**
|
|
13
13
|
ForceAtlas2, Fruchterman-Reingold and ngraph's spring-electrical preset run from Node (`run()`) and from a
|
|
@@ -650,7 +650,8 @@ pnpm run bench # every group; ap
|
|
|
650
650
|
pnpm exec tsx benchmarks/run.ts upload roundtrip layout-exact # selected groups; --no-save, --runs N, --allow-software
|
|
651
651
|
pnpm exec tsx benchmarks/layout-run.ts --nodes 100000 --edges 1000000 # the end-to-end layout driver (exit 1 on a bad result)
|
|
652
652
|
pnpm run gpu:report > gpu-report.json # the adapter report with a 10 s nvidia-smi sample
|
|
653
|
-
pnpm run bench:compare # the last out session vs benchmarks/results/<runner-class>.json (
|
|
653
|
+
pnpm run bench:compare # the last out session vs benchmarks/results/<runner-class>.json (1.35x and 2.5 ms over the pinned best fails)
|
|
654
|
+
pnpm run bench:append benchmarks/out/<class>.json benchmarks/results/<class>.json # append the last out session to a baseline (refuses software / incomplete / duplicate sessions)
|
|
654
655
|
```
|
|
655
656
|
|
|
656
657
|
The runner class is `<vendor>-<architecture>-driver<major>` (`scripts/runner-class.js`; `GRAPHTY_RUNNER_CLASS` overrides it,
|
|
@@ -667,7 +668,12 @@ SM clock at its idle 210 MHz under sparse sub-millisecond dispatches and the ker
|
|
|
667
668
|
at 10k and at 100k with `repulsion: "exact"`, the same two rows per model and rung as `layout-exact`, tagged `fr` /
|
|
668
669
|
`se`), `layout-grid` (T-6 and T-7: `step(1)` of the grid tier on the grid ladder 32k / 65k / 100k / 262k / 1M in 2D
|
|
669
670
|
and in 3D, the same two rows per rung tagged `grid` with the dimension, plus the `fa2-attraction` pass of the 1M 2D
|
|
670
|
-
iteration from the profiler, and the exact ladder's 1k / 4k / 8k / 16k rungs in 2D for the crossover re-check)
|
|
671
|
+
iteration from the profiler, and the exact ladder's 1k / 4k / 8k / 16k rungs in 2D for the crossover re-check),
|
|
672
|
+
`attraction-scale` (no target: the `fa2-attraction` pass across a working-set ladder, the G4-F16 diagnostic) and `bfs`
|
|
673
|
+
(T-10: `breadthFirstSearch` direction-optimizing and top-down from node 0 of the undirected RMAT tiers 100k / 1M and
|
|
674
|
+
1M / 10M, `sssp` on the same tiers with random weights in [0.1, 10), and `breadthFirstSearch` from a corner of the
|
|
675
|
+
1000 x 1000 grid, all wall end to end including the upload). A baseline session must carry every group:
|
|
676
|
+
`pnpm run bench:append` refuses one that does not. The Chromium numbers of T-5 (10k on the exact tier, 100k on the grid tier) come from the
|
|
671
677
|
`bench`-tagged browser test (`GRAPHTY_BROWSER_GPU=nvidia node scripts/run-browser-project.js`), which appends its
|
|
672
678
|
session through the Vitest commands bridge. `exactMaxNodes` is re-fixed from the ladder by the rule of plan section 7.8
|
|
673
679
|
(the largest rung under 4 ms per iteration and not slower than the grid tier at the same n -- the `layout-grid` 2D rows,
|
|
@@ -686,19 +692,22 @@ are the T-table of plan section 10.4. A missed target is re-fixed by a recorded
|
|
|
686
692
|
|
|
687
693
|
Measured on nvidia-lovelace-driver580 (NVIDIA: 580.173.02 580.173.2.0), session 2026-09-20T01:06:48.656Z, medians of 5 runs; Chromium: nvidia / lovelace (nvidia-lovelace-driver0, the description is redacted by Chromium), session 2026-09-16T02:18:11.896Z. The T-5 (100k), T-6 and T-7 rows are from the later session 2026-09-21T06:15:17.027Z (the file's last session, the current baseline, the one the crossover re-check reads; its Chromium session 2026-09-21T05:48:14.190Z), whose other rows are within 1.25x of this table's.
|
|
688
694
|
|
|
689
|
-
| Id | What
|
|
690
|
-
| ---- |
|
|
691
|
-
| T-1 | Upload of the 100k / 1M weighted hot prefix (16.4 MB); 1M / 10M (164 MB)
|
|
692
|
-
| T-2 | `degree` + 400 KB readback at 100k (core resident), Node
|
|
693
|
-
| T-3 | Empty submit + 4-byte `readU32` round trip, Dawn
|
|
694
|
-
| T-4 | ForceAtlas2 exact tier, GPU time per iteration (profiler) at 10k; at 16k
|
|
695
|
-
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 10k, Chromium (Node in brackets)
|
|
696
|
-
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 100k on the grid tier, Chromium (Node in brackets)
|
|
697
|
-
| T-6 | ForceAtlas2 grid tier, GPU time per iteration (profiler) at 100k 2D; at 1M 2D; at 100k 3D
|
|
698
|
-
| T-7 | Attraction gather (the `fa2-attraction` pass of the grid tier), GPU time per iteration (profiler) at 1M / 10M
|
|
699
|
-
| T-8 | PageRank, 100 iterations, wall end to end including the upload, at 100k / 1M; at 1M / 10M
|
|
700
|
-
| T-9 | Weakly connected components (Afforest), wall end to end including the upload and the label readback, at 1M / 10M (100k / 1M in brackets)
|
|
701
|
-
| T-14 | Fruchterman-Reingold exact tier, GPU time per iteration (profiler) at 10k; at 100k (`repulsion: "exact"`)
|
|
695
|
+
| Id | What | Target | Measured |
|
|
696
|
+
| ---- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
|
697
|
+
| T-1 | Upload of the 100k / 1M weighted hot prefix (16.4 MB); 1M / 10M (164 MB) | <= 10 ms; <= 100 ms | 5.980 ms; 125.738 ms |
|
|
698
|
+
| T-2 | `degree` + 400 KB readback at 100k (core resident), Node | <= 2 ms | 0.870 ms |
|
|
699
|
+
| T-3 | Empty submit + 4-byte `readU32` round trip, Dawn | <= 0.1 ms | 0.181 ms |
|
|
700
|
+
| T-4 | ForceAtlas2 exact tier, GPU time per iteration (profiler) at 10k; at 16k | <= 1 ms; <= 2 ms | 0.586 ms; 1.053 ms |
|
|
701
|
+
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 10k, Chromium (Node in brackets) | <= 6 ms | 2.400 ms (0.726 ms) |
|
|
702
|
+
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 100k on the grid tier, Chromium (Node in brackets) | <= 12 ms | 3.700 ms (2.087 ms) |
|
|
703
|
+
| T-6 | ForceAtlas2 grid tier, GPU time per iteration (profiler) at 100k 2D; at 1M 2D; at 100k 3D | <= 10 ms; <= 100 ms; <= 20 ms | 0.634 ms; 5.414 ms; 1.306 ms |
|
|
704
|
+
| T-7 | Attraction gather (the `fa2-attraction` pass of the grid tier), GPU time per iteration (profiler) at 1M / 10M | <= 15 ms | 1.493 ms |
|
|
705
|
+
| T-8 | PageRank, 100 iterations, wall end to end including the upload, at 100k / 1M; at 1M / 10M | <= 150 ms; <= 1.5 s | 17.204 ms; 198.645 ms |
|
|
706
|
+
| T-9 | Weakly connected components (Afforest), wall end to end including the upload and the label readback, at 1M / 10M (100k / 1M in brackets) | <= 100 ms | 145.970 ms (12.613 ms) |
|
|
707
|
+
| T-14 | Fruchterman-Reingold exact tier, GPU time per iteration (profiler) at 10k; at 100k (`repulsion: "exact"`) | recorded | 0.617 ms; 16.367 ms |
|
|
708
|
+
| T-10 | BFS direction-optimizing (`breadthFirstSearch`), wall end to end including the upload, from node 0 of the undirected RMAT at 1M / 10M (577,681 nodes reached, 6 levels; 100k / 1M in brackets; top-down 233.064 ms / 97.259 ms) | <= 100 ms | 237.388 ms (101.183 ms) in the baseline session; 156.560 ms (16.731 ms) re-measured 2026-09-25 under load after the direct-dispatch change, top-down 88.606 ms -- see below and G8-F5 / G8-F21 of `docs/decisions/G8.md` |
|
|
709
|
+
| T-10 | BFS on the 1000 x 1000 grid from a corner (1,999 levels; 64 `mapAsync` calls = ceil(levels / 32) + 1), wall including the upload | <= 1.5 s | 5680.192 ms in the baseline session; 427.235 ms re-measured 2026-09-25 under load after the direct-dispatch change, MET -- see below and G8-F5 of `docs/decisions/G8.md` |
|
|
710
|
+
| T-10 | `sssp` (the near-far queue), wall end to end including the upload, random f32 weights in [0.1, 10), at 1M / 10M (100k / 1M in brackets) | recorded | 310.351 ms (118.064 ms); 271.040 ms (50.871 ms) re-measured 2026-09-25 under load |
|
|
702
711
|
|
|
703
712
|
Three rows miss their target in this session: the 1M / 10M upload (125.7 ms against 100 ms, the open owner decision of
|
|
704
713
|
`docs/decisions/G1.md` section 7), the empty-submit round trip (0.181 ms against 0.1 ms: the row is measured after the
|
|
@@ -712,6 +721,26 @@ four iterations, which would time one pull and call it a hundred. The T-14 row i
|
|
|
712
721
|
session 2026-09-20T19:25:37.311Z (the file's last session, the current baseline), whose other rows are within 1.08x of
|
|
713
722
|
this table's; the spring-electrical preset measures 0.648 ms at 10k and 18.154 ms at 100k in the same session.
|
|
714
723
|
|
|
724
|
+
The three T-10 rows are the `bfs` group of session 2026-09-25T10:19:11.536Z (the file's last session, the current
|
|
725
|
+
baseline; `webgpu` 0.4.0, driver 580.173.02, load average 1.7-2.1, medians of 5). Both T-10 targets are MISSED on the
|
|
726
|
+
reference card, by 2.4x and 3.8x, and the cost is not in the kernels: a traversal costs about 2.85 ms per LEVEL
|
|
727
|
+
whatever the level's size or the submit cadence (the 200 x 200 grid, 399 levels, at 32, 8 and 1 levels per submit
|
|
728
|
+
alike) plus about 80 ms per CALL (a karate BFS, 34 nodes and 3 levels, takes 90 ms), and with Dawn's `skip_validation`
|
|
729
|
+
toggle the same karate call takes 3.0 ms and the 399-level grid 27 ms. About 97 % of the wall time is Dawn's own
|
|
730
|
+
validation of `dispatchWorkgroupsIndirect`, which the frontier design pays nine times a level; the baseline records
|
|
731
|
+
the default runtime because that is what a consumer gets, and the toggle is unsafe. The decision -- re-fix T-10 to
|
|
732
|
+
the class, fold the per-level indirect dispatches, or dispatch directly where the host already knows a count -- is
|
|
733
|
+
recorded in `docs/decisions/G8.md`.
|
|
734
|
+
|
|
735
|
+
That decision was taken on 2026-09-25, before the merge: every level kernel is now a DIRECT grid-stride dispatch
|
|
736
|
+
gated by a path word the device-side selector writes, and no traversal driver dispatches indirectly
|
|
737
|
+
(`design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md`; the measurement of Dawn's per-indirect-dispatch
|
|
738
|
+
cost, about 0.4 ms each in Node and Chromium alike, is `design/webgpu/dawn-indirect-dispatch-validation-cost.md`).
|
|
739
|
+
Re-measured on the same card with `--no-save --runs 3` at load average about 20 (so an upper bound, and not appended
|
|
740
|
+
to the baseline): a 34-node BFS 7.2 ms (from 89.3 with the core resident), the grid row 427 ms (MET), the 1M / 10M
|
|
741
|
+
auto row 156.6 ms and the top-down row 88.6 ms. The auto row now misses only because the direction-optimizing
|
|
742
|
+
choice itself is slower than top-down on that graph, which the validation cost had hidden (G8-F21).
|
|
743
|
+
|
|
715
744
|
The exact curve (the `layout-exact` group: 2D, E = 10n, seeded G(n, m), one simulation per rung; ms / iteration from the profiler):
|
|
716
745
|
|
|
717
746
|
| n | ms / iteration | step(1) wall (ms) | pairs / s |
|
|
@@ -741,19 +770,20 @@ the grid tier's is 2.8x at 100k 2D and 5.7x at 1M 2D while the attraction pass a
|
|
|
741
770
|
|
|
742
771
|
Measured on gpu-linux-t4 (NVIDIA: 580.126.20 580.126.20.0), session 2026-09-20T02:29:33.210Z (run 35483512705), medians of 5 runs; Chromium: nvidia / turing (nvidia-turing-driver0, the description is redacted by Chromium), session 2026-09-20T02:28:10.676Z. The T-5 (100k), T-6 and T-7 rows are from the later session 2026-09-22T20:21:53.814Z (GPU lane run 35775999450 on commit 1d2d4dcf, the file's last session and the current baseline of this class; its Chromium session 2026-09-22T20:19:30.117Z).
|
|
743
772
|
|
|
744
|
-
| Id | What | Target
|
|
745
|
-
| ---- | ---------------------------------------------------------------------------------------------------------------------------------------- |
|
|
746
|
-
| T-1 | Upload of the 100k / 1M weighted hot prefix (16.4 MB); 1M / 10M (164 MB) | <= 10 ms; <= 100 ms
|
|
747
|
-
| T-2 | `degree` + 400 KB readback at 100k (core resident), Node | <= 2 ms
|
|
748
|
-
| T-3 | Empty submit + 4-byte `readU32` round trip, Dawn | <= 0.1 ms
|
|
749
|
-
| T-4 | ForceAtlas2 exact tier, GPU time per iteration (profiler) at 10k; at 16k | <= 1 ms; <= 2 ms
|
|
750
|
-
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 10k, Chromium (Node in brackets) | <= 6 ms
|
|
751
|
-
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 100k on the grid tier, Chromium (Node in brackets) | <= 12 ms
|
|
752
|
-
| T-6 | ForceAtlas2 grid tier, GPU time per iteration (profiler) at 100k 2D; at 1M 2D; at 100k 3D | <= 10 ms; <= 100 ms; <= 20 ms
|
|
753
|
-
| T-7 | Attraction gather (the `fa2-attraction` pass of the grid tier), GPU time per iteration (profiler) at 1M / 10M | <= 15 ms
|
|
754
|
-
| T-8 | PageRank, 100 iterations, wall end to end including the upload, at 100k / 1M; at 1M / 10M | <= 150 ms; <= 1.5 s
|
|
755
|
-
| T-9 | Weakly connected components (Afforest), wall end to end including the upload and the label readback, at 1M / 10M (100k / 1M in brackets) | <= 100 ms
|
|
756
|
-
| T-14 | Fruchterman-Reingold exact tier, GPU time per iteration (profiler) at 10k; at 100k (`repulsion: "exact"`) | recorded
|
|
773
|
+
| Id | What | Target | Measured |
|
|
774
|
+
| ---- | ---------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
775
|
+
| T-1 | Upload of the 100k / 1M weighted hot prefix (16.4 MB); 1M / 10M (164 MB) | <= 10 ms; <= 100 ms | 14.338 ms; 256.418 ms |
|
|
776
|
+
| T-2 | `degree` + 400 KB readback at 100k (core resident), Node | <= 2 ms | 2.957 ms |
|
|
777
|
+
| T-3 | Empty submit + 4-byte `readU32` round trip, Dawn | <= 0.1 ms | 1.296 ms |
|
|
778
|
+
| T-4 | ForceAtlas2 exact tier, GPU time per iteration (profiler) at 10k; at 16k | <= 1 ms; <= 2 ms | 0.972 ms; 1.953 ms |
|
|
779
|
+
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 10k, Chromium (Node in brackets) | <= 6 ms | 2.600 ms (1.359 ms) |
|
|
780
|
+
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 100k on the grid tier, Chromium (Node in brackets) | <= 12 ms | 8.100 ms (5.314 ms) |
|
|
781
|
+
| T-6 | ForceAtlas2 grid tier, GPU time per iteration (profiler) at 100k 2D; at 1M 2D; at 100k 3D | <= 10 ms; <= 100 ms; <= 20 ms | 1.790 ms; 31.057 ms; 3.954 ms |
|
|
782
|
+
| T-7 | Attraction gather (the `fa2-attraction` pass of the grid tier), GPU time per iteration (profiler) at 1M / 10M | <= 15 ms | 18.165 ms -- MISSED: 21 % over the target on this card (1.493 ms on the dev box; G4-F16 of docs/decisions/G4.md) |
|
|
783
|
+
| T-8 | PageRank, 100 iterations, wall end to end including the upload, at 100k / 1M; at 1M / 10M | <= 150 ms; <= 1.5 s | 45.461 ms; 1092.799 ms |
|
|
784
|
+
| T-9 | Weakly connected components (Afforest), wall end to end including the upload and the label readback, at 1M / 10M (100k / 1M in brackets) | <= 100 ms | 293.079 ms (28.544 ms) |
|
|
785
|
+
| T-14 | Fruchterman-Reingold exact tier, GPU time per iteration (profiler) at 10k; at 100k (`repulsion: "exact"`) | recorded | 0.942 ms; 54.232 ms (the spring preset 1.051 ms; 60.706 ms) |
|
|
786
|
+
| T-10 | BFS direction-optimizing at 1M / 10M RMAT; the 1000 x 1000 grid; `sssp` at 1M / 10M (the `bfs` group) | recorded (the contract is the dev box's) | not yet recorded: the lane has not run a branch that carries the `bfs` group; the first `gpu.yml` run of it is appended with `pnpm run bench:append` (G8-F3 of `docs/decisions/G8.md`) |
|
|
757
787
|
|
|
758
788
|
The exact curve (the `layout-exact` group: 2D, E = 10n, seeded G(n, m), one simulation per rung; ms / iteration from the profiler):
|
|
759
789
|
|
package/dist/browser.js
CHANGED
|
@@ -88,6 +88,12 @@ const GRID_HUB_CELL = 1024;
|
|
|
88
88
|
const GRID_EXTENT_FLOOR = 1e-6;
|
|
89
89
|
const GRID_BBOX_MARGIN = 1.01;
|
|
90
90
|
const GRID_SORT_BITS = 24;
|
|
91
|
+
const MAX_LEVELS_PER_SUBMIT = 32;
|
|
92
|
+
const FUSED_FRONTIER_MAX = 4096;
|
|
93
|
+
const FRONTIER_CANDIDATES = 7;
|
|
94
|
+
const BEAMER_BETA = 24;
|
|
95
|
+
const SSSP_DELTA_FACTOR = 32;
|
|
96
|
+
const F32_INF_BITS = 2139095040;
|
|
91
97
|
const PASSTHROUGH_FORMAT_CODES = Object.freeze(["E_GPU_INELIGIBLE", "E_UNKNOWN_NODE", "E_UNKNOWN_COLUMN", "E_COLUMN_LENGTH"]);
|
|
92
98
|
const EMPTY_DETAILS = Object.freeze({});
|
|
93
99
|
class WebGpuGraphError extends Error {
|
|
@@ -883,6 +889,7 @@ const PRELUDE_WGSL = (
|
|
|
883
889
|
`// ---- prelude: constants, standard overrides, helpers (every module receives this text first)
|
|
884
890
|
const INVALID_INDEX: u32 = ${INVALID_INDEX}u;
|
|
885
891
|
const U32_MAX: u32 = ${U32_MAX}u;
|
|
892
|
+
const F32_INF_BITS: u32 = ${F32_INF_BITS}u;
|
|
886
893
|
const MAX_WORKGROUPS_PER_DIM: u32 = ${MAX_WORKGROUPS_PER_DIM}u;
|
|
887
894
|
const FA2_DIST_FLOOR: f32 = ${wgslF32Literal(FA2_DISTANCE_FLOOR)};
|
|
888
895
|
const FA2_DIST_FLOOR_SQ: f32 = ${wgslF32Literal(FA2_DISTANCE_FLOOR_SQ)};
|
|
@@ -969,6 +976,21 @@ fn wg_reduce_u32(v: u32, lid: u32, op: u32) -> u32 {
|
|
|
969
976
|
workgroupBarrier();
|
|
970
977
|
return total;
|
|
971
978
|
}
|
|
979
|
+
fn wg_scan_u32(v: u32, lid: u32) -> u32 {
|
|
980
|
+
workgroupBarrier();
|
|
981
|
+
wg_scratch_u[lid] = v;
|
|
982
|
+
workgroupBarrier();
|
|
983
|
+
for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan (scan-block's body); uniform: every lane runs every round
|
|
984
|
+
var t = 0u;
|
|
985
|
+
if (lid >= s) { t = wg_scratch_u[lid - s]; }
|
|
986
|
+
workgroupBarrier();
|
|
987
|
+
wg_scratch_u[lid] = wg_scratch_u[lid] + t;
|
|
988
|
+
workgroupBarrier();
|
|
989
|
+
}
|
|
990
|
+
let inclusive = wg_scratch_u[lid];
|
|
991
|
+
workgroupBarrier(); // the scratch is free for the next call
|
|
992
|
+
return inclusive;
|
|
993
|
+
}
|
|
972
994
|
fn wg_reduce_f32(v: f32, lid: u32, op: u32) -> f32 { return wg_reduce_vec4(vec4f(v, 0.0, 0.0, 0.0), lid, op).x; }`
|
|
973
995
|
);
|
|
974
996
|
const REDUCE_HELPERS_SUBGROUP_WGSL = (
|
|
@@ -1048,9 +1070,27 @@ fn wg_reduce_u32(v: u32, lid: u32, op: u32) -> u32 {
|
|
|
1048
1070
|
workgroupBarrier();
|
|
1049
1071
|
return total;
|
|
1050
1072
|
}
|
|
1073
|
+
fn wg_scan_u32(v: u32, lid: u32) -> u32 {
|
|
1074
|
+
workgroupBarrier();
|
|
1075
|
+
if (lid == 0u) { atomicStore(&sg_counter, 0u); }
|
|
1076
|
+
workgroupBarrier();
|
|
1077
|
+
let inSub = subgroupExclusiveAdd(v) + v; // the lane's inclusive sum within its subgroup
|
|
1078
|
+
let total = subgroupAdd(v);
|
|
1079
|
+
let key = subgroupMin(lid); // the subgroup's identity: its smallest local id (D16)
|
|
1080
|
+
var slot = 0u;
|
|
1081
|
+
if (subgroupElect()) { slot = atomicAdd(&sg_counter, 1u); }
|
|
1082
|
+
slot = subgroupBroadcast(slot, 0u);
|
|
1083
|
+
if (subgroupElect()) { sg_val_u[slot] = total; sg_key[slot] = key; }
|
|
1084
|
+
workgroupBarrier();
|
|
1085
|
+
let count = atomicLoad(&sg_counter);
|
|
1086
|
+
var carry = 0u; // the totals of every subgroup whose key is below this lane's: LANE order, whatever order the slots were taken in
|
|
1087
|
+
for (var k = 0u; k < count; k = k + 1u) { if (sg_key[k] < key) { carry = carry + sg_val_u[k]; } }
|
|
1088
|
+
workgroupBarrier(); // the slots are free for the next call
|
|
1089
|
+
return inSub + carry;
|
|
1090
|
+
}
|
|
1051
1091
|
fn wg_reduce_f32(v: f32, lid: u32, op: u32) -> f32 { return wg_reduce_vec4(vec4f(v, 0.0, 0.0, 0.0), lid, op).x; }`
|
|
1052
1092
|
);
|
|
1053
|
-
const REDUCE_HELPER_NAMES = ["wg_reduce_f32", "wg_reduce_u32", "wg_reduce_vec4"];
|
|
1093
|
+
const REDUCE_HELPER_NAMES = ["wg_reduce_f32", "wg_reduce_u32", "wg_reduce_vec4", "wg_scan_u32"];
|
|
1054
1094
|
const WGSL_RESERVED_WORDS = Object.freeze(
|
|
1055
1095
|
`NULL Self abstract active alignas alignof as asm asm_fragment async attribute auto await become cast catch class
|
|
1056
1096
|
co_await co_return co_yield coherent column_major common compile compile_fragment concept const_cast consteval
|
|
@@ -3444,12 +3484,17 @@ class GpuContext {
|
|
|
3444
3484
|
}
|
|
3445
3485
|
}
|
|
3446
3486
|
export {
|
|
3447
|
-
|
|
3487
|
+
SE_SCALE_REFERENCE_NODES as A,
|
|
3448
3488
|
BufferUsage as B,
|
|
3489
|
+
ARC_WINDOW_ALIGN as C,
|
|
3490
|
+
PASSTHROUGH_FORMAT_CODES as D,
|
|
3449
3491
|
EXACT_MAX_NODES as E,
|
|
3450
|
-
|
|
3492
|
+
FRONTIER_CANDIDATES as F,
|
|
3451
3493
|
GpuContext as G,
|
|
3494
|
+
STORAGE_ALIGN as H,
|
|
3452
3495
|
INDIRECT_ARGS_STRIDE as I,
|
|
3496
|
+
WORKGROUP_SIZE as J,
|
|
3497
|
+
isSoftwareAdapter as K,
|
|
3453
3498
|
LAYOUT_TUNING_DEFAULTS as L,
|
|
3454
3499
|
MAX_WORKGROUPS_PER_DIM as M,
|
|
3455
3500
|
PARTIAL_BYTES as P,
|
|
@@ -3460,28 +3505,29 @@ export {
|
|
|
3460
3505
|
WebGpuGraphError as W,
|
|
3461
3506
|
WGSL_RESERVED_WORDS as a,
|
|
3462
3507
|
U32_MAX as b,
|
|
3463
|
-
|
|
3508
|
+
MAX_LEVELS_PER_SUBMIT as c,
|
|
3464
3509
|
deviceLostError as d,
|
|
3465
|
-
|
|
3466
|
-
|
|
3467
|
-
|
|
3468
|
-
|
|
3510
|
+
FUSED_FRONTIER_MAX as e,
|
|
3511
|
+
BEAMER_BETA as f,
|
|
3512
|
+
SSSP_DELTA_FACTOR as g,
|
|
3513
|
+
F32_INF_BITS as h,
|
|
3469
3514
|
isWebGpuGraphError as i,
|
|
3470
|
-
|
|
3471
|
-
|
|
3472
|
-
|
|
3473
|
-
|
|
3474
|
-
|
|
3475
|
-
|
|
3476
|
-
|
|
3477
|
-
|
|
3478
|
-
|
|
3479
|
-
|
|
3480
|
-
|
|
3481
|
-
|
|
3482
|
-
|
|
3483
|
-
|
|
3484
|
-
|
|
3485
|
-
|
|
3515
|
+
GRID_COARSEST_SIDE as j,
|
|
3516
|
+
GRID_MIN_SIDE as k,
|
|
3517
|
+
GRID_SORT_BITS as l,
|
|
3518
|
+
FA2_DEFAULTS as m,
|
|
3519
|
+
MAX_ITERATIONS_PER_STEP as n,
|
|
3520
|
+
MAX_1D_ITEMS as o,
|
|
3521
|
+
hasErrorCode as p,
|
|
3522
|
+
FA2_FLAG_FIRST as q,
|
|
3523
|
+
GRID_HUB_CELL as r,
|
|
3524
|
+
GRID_BBOX_MARGIN as s,
|
|
3525
|
+
GRID_EXTENT_FLOOR as t,
|
|
3526
|
+
FR_ADAPTIVE_MAX_ITERATIONS as u,
|
|
3527
|
+
FR_START_TEMPERATURE as v,
|
|
3528
|
+
FA2_FLAG_ADAPTIVE as w,
|
|
3529
|
+
FR_REHEAT_FRACTION as x,
|
|
3530
|
+
FR_DEFAULTS as y,
|
|
3531
|
+
SE_DEFAULTS as z
|
|
3486
3532
|
};
|
|
3487
|
-
//# sourceMappingURL=context-
|
|
3533
|
+
//# sourceMappingURL=context-Dvq-Cc6v.js.map
|