@graphty/webgpu-graph-algorithms 0.6.9 → 0.6.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +104 -53
- package/package.json +11 -8
package/README.md
CHANGED
|
@@ -683,43 +683,65 @@ measures the same crossover on a consumer's own device.
|
|
|
683
683
|
|
|
684
684
|
## Performance
|
|
685
685
|
|
|
686
|
-
|
|
687
|
-
|
|
688
|
-
|
|
689
|
-
|
|
686
|
+
The targets are the T-table of plan section 10.4, kept in `benchmarks/results/targets.json` with one entry per
|
|
687
|
+
benchmark row. They bind the reference card, the Tesla T4 of the GPU lane: on that class `bench:compare` fails the lane
|
|
688
|
+
when a row misses its target, unless the row is one of the class's known misses, which carry the reason and the
|
|
689
|
+
decision record beside the target. Every other runner class reports met or missed and does not gate. The two tables
|
|
690
|
+
below are generated from that file and the checked-in results (`pnpm run bench:readme`, which a node test checks), so
|
|
691
|
+
they are never edited by hand. A missed target is re-fixed by a recorded owner decision in `docs/decisions/G<n>.md`,
|
|
692
|
+
never relaxed silently.
|
|
690
693
|
|
|
691
694
|
### The dev box (nvidia-lovelace-driver580)
|
|
692
695
|
|
|
693
|
-
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
|
|
|
698
|
-
|
|
|
699
|
-
| T-
|
|
700
|
-
| T-
|
|
701
|
-
| T-
|
|
702
|
-
| T-
|
|
703
|
-
| T-
|
|
704
|
-
| T-
|
|
705
|
-
| T-
|
|
706
|
-
| T-
|
|
707
|
-
| T-
|
|
708
|
-
| T-
|
|
709
|
-
| T-
|
|
710
|
-
| T-
|
|
711
|
-
|
|
712
|
-
|
|
713
|
-
`
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
|
|
721
|
-
|
|
722
|
-
|
|
696
|
+
<!-- targets-table:nvidia-lovelace-driver580 -->
|
|
697
|
+
|
|
698
|
+
Generated by `pnpm run bench:readme` from `benchmarks/results/targets.json`. Each figure is the best median the row has recorded, the baseline `bench:compare` pins: Node rows across the 7 session(s) of `benchmarks/results/nvidia-lovelace-driver580.json`, Chromium rows across the 3 of `benchmarks/results/nvidia-lovelace-driver0.json`.
|
|
699
|
+
|
|
700
|
+
| Id | What | Target | Measured | Status |
|
|
701
|
+
| ---- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------- | ----------- | -------------- |
|
|
702
|
+
| T-1 | Upload of the 100k / 1M weighted hot prefix (16.4 MB), Node | <= 10 ms | 5.734 ms | met |
|
|
703
|
+
| T-1 | Upload of the 1M / 10M weighted hot prefix (164 MB), Node | <= 100 ms | 122.254 ms | missed (noisy) |
|
|
704
|
+
| T-2 | `degree` + 400 KB readback at 100k (core resident), Node | <= 2 ms | 0.838 ms | met |
|
|
705
|
+
| T-3 | Empty submit + 4-byte `readU32` round trip, Dawn in Node | <= 0.1 ms | 0.062 ms | met |
|
|
706
|
+
| T-3 | Empty submit + 4-byte `readU32` round trip, Chromium (the <= 0.1 ms target is Dawn's) | recorded | 0.170 ms | recorded |
|
|
707
|
+
| T-4 | ForceAtlas2 exact tier, GPU time per iteration (profiler) at 10k | <= 1 ms | 0.586 ms | met |
|
|
708
|
+
| T-4 | ForceAtlas2 exact tier, GPU time per iteration (profiler) at 16k | <= 2 ms | 1.052 ms | met |
|
|
709
|
+
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 10k, Chromium | <= 6 ms | 2.400 ms | met |
|
|
710
|
+
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 10k, Node | recorded | 0.713 ms | recorded |
|
|
711
|
+
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 100k on the grid tier, Chromium | <= 12 ms | 3.700 ms | met |
|
|
712
|
+
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 100k on the grid tier, Node | recorded | 1.769 ms | recorded |
|
|
713
|
+
| T-6 | ForceAtlas2 grid tier, GPU time per iteration (profiler) at 100k 2D | <= 10 ms | 0.574 ms | met |
|
|
714
|
+
| T-6 | ForceAtlas2 grid tier, GPU time per iteration (profiler) at 1M 2D | <= 100 ms | 5.303 ms | met |
|
|
715
|
+
| T-6 | ForceAtlas2 grid tier, GPU time per iteration (profiler) at 100k 3D | <= 20 ms | 1.279 ms | met |
|
|
716
|
+
| T-7 | Attraction gather (the `fa2-attraction` pass of the grid tier), GPU time per iteration (profiler) at 1M / 10M | <= 15 ms | 1.475 ms | met |
|
|
717
|
+
| T-8 | PageRank, 100 iterations, wall end to end including the upload, at 100k / 1M | <= 150 ms | 17.204 ms | met |
|
|
718
|
+
| T-8 | PageRank, 100 iterations, wall end to end including the upload, at 1M / 10M | <= 1500 ms | 197.154 ms | met |
|
|
719
|
+
| T-9 | Weakly connected components (Afforest), wall end to end including the upload and the label readback, at 1M / 10M | <= 100 ms | 140.984 ms | missed |
|
|
720
|
+
| T-9 | Weakly connected components (Afforest), wall end to end including the upload and the label readback, at 100k / 1M | recorded | 12.613 ms | recorded |
|
|
721
|
+
| T-14 | Fruchterman-Reingold exact tier, GPU time per iteration (profiler) at 10k | recorded | 0.566 ms | recorded |
|
|
722
|
+
| T-14 | Fruchterman-Reingold exact tier, GPU time per iteration (profiler) at 100k (`repulsion: "exact"`) | recorded | 16.322 ms | recorded |
|
|
723
|
+
| T-14 | Spring-electrical preset, GPU time per iteration (profiler) at 10k | recorded | 0.648 ms | recorded |
|
|
724
|
+
| T-14 | Spring-electrical preset, GPU time per iteration (profiler) at 100k (`repulsion: "exact"`) | recorded | 18.154 ms | recorded |
|
|
725
|
+
| T-10 | BFS direction-optimizing (`breadthFirstSearch`), wall end to end including the upload, from node 0 of the undirected RMAT at 1M / 10M (a contract on the dev box only) | <= 100 ms | 237.388 ms | missed |
|
|
726
|
+
| T-10 | BFS top-down, the same traversal at 1M / 10M | recorded | 233.064 ms | recorded |
|
|
727
|
+
| T-10 | BFS direction-optimizing, the same traversal at 100k / 1M | recorded | 101.183 ms | recorded |
|
|
728
|
+
| T-10 | BFS on the 1000 x 1000 grid from a corner (1,999 levels), wall including the upload (a contract on the dev box only) | <= 1500 ms | 5680.192 ms | missed |
|
|
729
|
+
| T-10 | `sssp` (the near-far queue), wall end to end including the upload, random f32 weights in [0.1, 10), at 1M / 10M | recorded | 310.351 ms | recorded |
|
|
730
|
+
| T-10 | `sssp`, the same run at 100k / 1M | recorded | 118.064 ms | recorded |
|
|
731
|
+
|
|
732
|
+
<!-- /targets-table:nvidia-lovelace-driver580 -->
|
|
733
|
+
|
|
734
|
+
Why the misses miss. The 1M / 10M upload is the open owner decision of `docs/decisions/G1.md` section 7. The
|
|
735
|
+
empty-submit round trip in Node meets its target at the working clock, but a session that measures it right after the
|
|
736
|
+
`upload` group reads about 0.15-0.18 ms, because the group's CPU-heavy setup lets the SM clock fall to its idle state
|
|
737
|
+
(finding G3-F2 of `docs/decisions/G3.md` section 10). WCC at 1M / 10M is wall end to end from a released core, so it
|
|
738
|
+
carries the same 164 MB upload T-1 times; with the core resident the same call takes 11-16 ms, and the T-9 target is
|
|
739
|
+
therefore below the T-1 upload it includes, an owner decision for `docs/decisions/G7.md`. The `pagerank` rows run all
|
|
740
|
+
100 iterations (`tolerance: 0`): at the NetworkX tolerance of 1e-6 the seeded G(n, m) input converges from the uniform
|
|
741
|
+
start in one to four iterations, which would time one pull and call it a hundred. The Chromium round trip (T-3, issue
|
|
742
|
+
#278) is what every interactive frame of a browser layout pays; its figure is the mean of batches of 20 round trips,
|
|
743
|
+
because Chromium quantises `performance.now()` to 100 us, and it was measured on 2026-09-27 at a load average near 70,
|
|
744
|
+
so it is an upper bound.
|
|
723
745
|
|
|
724
746
|
The three T-10 rows are the `bfs` group of session 2026-09-25T10:19:11.536Z (the file's last session, the current
|
|
725
747
|
baseline; `webgpu` 0.4.0, driver 580.173.02, load average 1.7-2.1, medians of 5). Both T-10 targets are MISSED on the
|
|
@@ -783,28 +805,57 @@ GPU time per iteration in the last batch).
|
|
|
783
805
|
The first run of the GPU lane (`gpu.yml`, graphty-monorepo run 35316416067, 2026-09-18) on a machine.dev T4 -- one Tesla
|
|
784
806
|
T4 (16 GB), 4 vCPU of a Xeon Platinum 8259CL, driver 580.126.20 -- wrote this baseline; `scripts/bench-compare.js` fails
|
|
785
807
|
a later run of the lane whose median AND minimum both exceed 1.35x the best figures this file has ever held, by at
|
|
786
|
-
least 2.5 ms
|
|
787
|
-
T-4, T-5 and T-
|
|
788
|
-
against 15 ms), which is the class difference of a datacentre card behind a cloud vCPU (host-side copies and submit
|
|
808
|
+
least 2.5 ms, or whose row misses a target that is not one of the known misses listed under the table. The T-table
|
|
809
|
+
targets were set on the dev box; the T4 meets T-4, T-5, T-6 and T-8 and misses T-1 (both uploads), T-2, T-3, T-7 (the
|
|
810
|
+
1M attraction pass of the grid tier, 18.165 ms against 15 ms) and T-9, which is the class difference of a datacentre card behind a cloud vCPU (host-side copies and submit
|
|
789
811
|
latency), not a regression: the exact tier's `ms / iteration` is 1.7x the RTX 4070 SUPER's at 10k and 3.0x at 65k, and
|
|
790
812
|
the grid tier's is 2.8x at 100k 2D and 5.7x at 1M 2D while the attraction pass alone is 12.2x.
|
|
791
813
|
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
|
797
|
-
|
|
|
798
|
-
| T-
|
|
799
|
-
| T-
|
|
800
|
-
| T-
|
|
801
|
-
| T-
|
|
802
|
-
| T-
|
|
803
|
-
| T-
|
|
804
|
-
| T-
|
|
805
|
-
| T-
|
|
806
|
-
| T-
|
|
807
|
-
| T-
|
|
814
|
+
<!-- targets-table:gpu-linux-t4 -->
|
|
815
|
+
|
|
816
|
+
Generated by `pnpm run bench:readme` from `benchmarks/results/targets.json`. Each figure is the best median the row has recorded, the baseline `bench:compare` pins: Node rows across the 4 session(s) of `benchmarks/results/gpu-linux-t4.json`, Chromium rows across the 2 of `benchmarks/results/nvidia-turing-driver0.json`.
|
|
817
|
+
|
|
818
|
+
| Id | What | Target | Measured | Status |
|
|
819
|
+
| ---- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------- | ---------------- | -------------- |
|
|
820
|
+
| T-1 | Upload of the 100k / 1M weighted hot prefix (16.4 MB), Node | <= 10 ms | 14.338 ms | missed (known) |
|
|
821
|
+
| T-1 | Upload of the 1M / 10M weighted hot prefix (164 MB), Node | <= 100 ms | 256.418 ms | missed (known) |
|
|
822
|
+
| T-2 | `degree` + 400 KB readback at 100k (core resident), Node | <= 2 ms | 2.292 ms | missed (known) |
|
|
823
|
+
| T-3 | Empty submit + 4-byte `readU32` round trip, Dawn in Node | <= 0.1 ms | 0.155 ms | missed (known) |
|
|
824
|
+
| T-3 | Empty submit + 4-byte `readU32` round trip, Chromium (the <= 0.1 ms target is Dawn's) | recorded | not yet measured | - |
|
|
825
|
+
| T-4 | ForceAtlas2 exact tier, GPU time per iteration (profiler) at 10k | <= 1 ms | 0.972 ms | met |
|
|
826
|
+
| T-4 | ForceAtlas2 exact tier, GPU time per iteration (profiler) at 16k | <= 2 ms | 1.844 ms | met |
|
|
827
|
+
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 10k, Chromium | <= 6 ms | 2.600 ms | met |
|
|
828
|
+
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 10k, Node | recorded | 1.357 ms | recorded |
|
|
829
|
+
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 100k on the grid tier, Chromium | <= 12 ms | 6.450 ms | met |
|
|
830
|
+
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 100k on the grid tier, Node | recorded | 5.314 ms | recorded |
|
|
831
|
+
| T-6 | ForceAtlas2 grid tier, GPU time per iteration (profiler) at 100k 2D | <= 10 ms | 1.790 ms | met |
|
|
832
|
+
| T-6 | ForceAtlas2 grid tier, GPU time per iteration (profiler) at 1M 2D | <= 100 ms | 31.057 ms | met |
|
|
833
|
+
| T-6 | ForceAtlas2 grid tier, GPU time per iteration (profiler) at 100k 3D | <= 20 ms | 3.954 ms | met |
|
|
834
|
+
| T-7 | Attraction gather (the `fa2-attraction` pass of the grid tier), GPU time per iteration (profiler) at 1M / 10M | <= 15 ms | 18.165 ms | missed (known) |
|
|
835
|
+
| T-8 | PageRank, 100 iterations, wall end to end including the upload, at 100k / 1M | <= 150 ms | 45.461 ms | met |
|
|
836
|
+
| T-8 | PageRank, 100 iterations, wall end to end including the upload, at 1M / 10M | <= 1500 ms | 1092.799 ms | met |
|
|
837
|
+
| T-9 | Weakly connected components (Afforest), wall end to end including the upload and the label readback, at 1M / 10M | <= 100 ms | 290.695 ms | missed (known) |
|
|
838
|
+
| T-9 | Weakly connected components (Afforest), wall end to end including the upload and the label readback, at 100k / 1M | recorded | 28.475 ms | recorded |
|
|
839
|
+
| T-14 | Fruchterman-Reingold exact tier, GPU time per iteration (profiler) at 10k | recorded | 0.942 ms | recorded |
|
|
840
|
+
| T-14 | Fruchterman-Reingold exact tier, GPU time per iteration (profiler) at 100k (`repulsion: "exact"`) | recorded | 52.591 ms | recorded |
|
|
841
|
+
| T-14 | Spring-electrical preset, GPU time per iteration (profiler) at 10k | recorded | 1.051 ms | recorded |
|
|
842
|
+
| T-14 | Spring-electrical preset, GPU time per iteration (profiler) at 100k (`repulsion: "exact"`) | recorded | 59.260 ms | recorded |
|
|
843
|
+
| T-10 | BFS direction-optimizing (`breadthFirstSearch`), wall end to end including the upload, from node 0 of the undirected RMAT at 1M / 10M (a contract on the dev box only) | <= 100 ms | not yet measured | - |
|
|
844
|
+
| T-10 | BFS top-down, the same traversal at 1M / 10M | recorded | not yet measured | - |
|
|
845
|
+
| T-10 | BFS direction-optimizing, the same traversal at 100k / 1M | recorded | not yet measured | - |
|
|
846
|
+
| T-10 | BFS on the 1000 x 1000 grid from a corner (1,999 levels), wall including the upload (a contract on the dev box only) | <= 1500 ms | not yet measured | - |
|
|
847
|
+
| T-10 | `sssp` (the near-far queue), wall end to end including the upload, random f32 weights in [0.1, 10), at 1M / 10M | recorded | not yet measured | - |
|
|
848
|
+
| T-10 | `sssp`, the same run at 100k / 1M | recorded | not yet measured | - |
|
|
849
|
+
|
|
850
|
+
Known misses on this class, which `bench:compare` reports without failing:
|
|
851
|
+
|
|
852
|
+
- T-1: host-side copy cost of a datacentre card behind a cloud vCPU (the README's CI-lane section); the 1M / 10M upload misses on the dev box too (docs/decisions/G1.md section 7).
|
|
853
|
+
- T-2: readback and submit latency of a datacentre card behind a cloud vCPU (the README's CI-lane section); met on the dev box.
|
|
854
|
+
- T-3: submit latency of the cloud runner; on the dev box the row misses too whenever the SM clock has dropped to idle (G3-F2 of docs/decisions/G3.md section 10).
|
|
855
|
+
- T-7: the pass's 16 MiB working set does not fit the T4's 4 MiB L2, and two T4 instances read 13.8 and 18.2 ms on the same work (G4-F16 and G4-F21 of docs/decisions/G4.md).
|
|
856
|
+
- T-9: the row carries the same 164 MB upload T-1 times, which alone takes longer than the 100 ms target; it misses on the dev box too (docs/decisions/G7.md).
|
|
857
|
+
|
|
858
|
+
<!-- /targets-table:gpu-linux-t4 -->
|
|
808
859
|
|
|
809
860
|
The exact curve (the `layout-exact` group: 2D, E = 10n, seeded G(n, m), one simulation per rung; ms / iteration from the profiler):
|
|
810
861
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@graphty/webgpu-graph-algorithms",
|
|
3
|
-
"version": "0.6.
|
|
3
|
+
"version": "0.6.10",
|
|
4
4
|
"description": "WebGPU-accelerated graph algorithms and layouts over the @graphty/graph-format snapshot, for Node (Dawn) and browsers",
|
|
5
5
|
"author": "Adam Powers <apowers@ato.ms>",
|
|
6
6
|
"type": "module",
|
|
@@ -61,7 +61,7 @@
|
|
|
61
61
|
"homepage": "https://github.com/graphty-org/graphty-monorepo/tree/master/webgpu-graph-algorithms#readme",
|
|
62
62
|
"dependencies": {
|
|
63
63
|
"@webgpu/types": "^0.1.72",
|
|
64
|
-
"@graphty/graph-format": "^1.0
|
|
64
|
+
"@graphty/graph-format": "^1.1.0"
|
|
65
65
|
},
|
|
66
66
|
"peerDependencies": {
|
|
67
67
|
"@graphty/algorithms": "^1.0.0 || ^2.0.0",
|
|
@@ -81,20 +81,22 @@
|
|
|
81
81
|
}
|
|
82
82
|
},
|
|
83
83
|
"devDependencies": {
|
|
84
|
-
"@vitest/browser": "
|
|
85
|
-
"@vitest/coverage-v8": "
|
|
86
|
-
"@vitest/ui": "
|
|
84
|
+
"@vitest/browser-playwright": "4.1.11",
|
|
85
|
+
"@vitest/coverage-v8": "4.1.11",
|
|
86
|
+
"@vitest/ui": "4.1.11",
|
|
87
|
+
"eslint": "^9.29.0",
|
|
87
88
|
"fast-check": "^4.2.0",
|
|
88
89
|
"ngraph.forcelayout": "^3.3.1",
|
|
89
90
|
"ngraph.graph": "^20.0.1",
|
|
90
91
|
"playwright": "^1.54.1",
|
|
91
92
|
"tsx": "^4.20.3",
|
|
92
93
|
"typescript": "^5.9.3",
|
|
94
|
+
"typescript-eslint": "^8.34.1",
|
|
93
95
|
"vite": "^7.0.5",
|
|
94
|
-
"vitest": "
|
|
96
|
+
"vitest": "4.1.11",
|
|
95
97
|
"webgpu": "0.4.0",
|
|
96
|
-
"@graphty/
|
|
97
|
-
"@graphty/
|
|
98
|
+
"@graphty/algorithms": "^2.1.0",
|
|
99
|
+
"@graphty/layout": "^1.10.3"
|
|
98
100
|
},
|
|
99
101
|
"scripts": {
|
|
100
102
|
"build": "node -e \"require('fs').rmSync('dist',{recursive:true,force:true})\" && tsc -p tsconfig.build.json",
|
|
@@ -118,6 +120,7 @@
|
|
|
118
120
|
"bench:ab": "node scripts/bench-ab.js",
|
|
119
121
|
"bench:compare": "node scripts/bench-compare.js",
|
|
120
122
|
"bench:append": "node scripts/bench-append-session.js",
|
|
123
|
+
"bench:readme": "node scripts/bench-readme-tables.js",
|
|
121
124
|
"benchmark": "tsx benchmarks/run.ts",
|
|
122
125
|
"gpu:report": "node scripts/gpu-report.js",
|
|
123
126
|
"ready:commit": "npm run build:all && npm run lint && npm run test:node"
|