@graphty/webgpu-graph-algorithms 0.0.0 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +378 -23
- package/dist/browser.d.ts +1 -0
- package/dist/browser.js +32 -0
- package/dist/browser.js.map +1 -0
- package/dist/chunks/context-E6iKaeuJ.js +3136 -0
- package/dist/chunks/context-E6iKaeuJ.js.map +1 -0
- package/dist/node.d.ts +1 -0
- package/dist/node.js +131 -0
- package/dist/node.js.map +1 -0
- package/dist/src/accelerator.d.ts +26 -0
- package/dist/src/accelerator.d.ts.map +1 -0
- package/dist/src/accelerator.js +101 -0
- package/dist/src/accelerator.js.map +1 -0
- package/dist/src/algorithms/degree.d.ts +35 -0
- package/dist/src/algorithms/degree.d.ts.map +1 -0
- package/dist/src/algorithms/degree.js +119 -0
- package/dist/src/algorithms/degree.js.map +1 -0
- package/dist/src/browser/index.d.ts +23 -0
- package/dist/src/browser/index.d.ts.map +1 -0
- package/dist/src/browser/index.js +48 -0
- package/dist/src/browser/index.js.map +1 -0
- package/dist/src/constants.d.ts +92 -0
- package/dist/src/constants.d.ts.map +1 -0
- package/dist/src/constants.js +92 -0
- package/dist/src/constants.js.map +1 -0
- package/dist/src/context.d.ts +84 -0
- package/dist/src/context.d.ts.map +1 -0
- package/dist/src/context.js +304 -0
- package/dist/src/context.js.map +1 -0
- package/dist/src/device/acquire.d.ts +57 -0
- package/dist/src/device/acquire.d.ts.map +1 -0
- package/dist/src/device/acquire.js +232 -0
- package/dist/src/device/acquire.js.map +1 -0
- package/dist/src/device/caps.d.ts +43 -0
- package/dist/src/device/caps.d.ts.map +1 -0
- package/dist/src/device/caps.js +104 -0
- package/dist/src/device/caps.js.map +1 -0
- package/dist/src/device/error-scope.d.ts +75 -0
- package/dist/src/device/error-scope.d.ts.map +1 -0
- package/dist/src/device/error-scope.js +152 -0
- package/dist/src/device/error-scope.js.map +1 -0
- package/dist/src/device/lost.d.ts +51 -0
- package/dist/src/device/lost.d.ts.map +1 -0
- package/dist/src/device/lost.js +130 -0
- package/dist/src/device/lost.js.map +1 -0
- package/dist/src/device/webgpu-constants.d.ts +31 -0
- package/dist/src/device/webgpu-constants.d.ts.map +1 -0
- package/dist/src/device/webgpu-constants.js +31 -0
- package/dist/src/device/webgpu-constants.js.map +1 -0
- package/dist/src/errors.d.ts +56 -0
- package/dist/src/errors.d.ts.map +1 -0
- package/dist/src/errors.js +57 -0
- package/dist/src/errors.js.map +1 -0
- package/dist/src/index.d.ts +29 -0
- package/dist/src/index.d.ts.map +1 -0
- package/dist/src/index.js +27 -0
- package/dist/src/index.js.map +1 -0
- package/dist/src/kernel/batch.d.ts +116 -0
- package/dist/src/kernel/batch.d.ts.map +1 -0
- package/dist/src/kernel/batch.js +335 -0
- package/dist/src/kernel/batch.js.map +1 -0
- package/dist/src/kernel/dispatch.d.ts +59 -0
- package/dist/src/kernel/dispatch.d.ts.map +1 -0
- package/dist/src/kernel/dispatch.js +139 -0
- package/dist/src/kernel/dispatch.js.map +1 -0
- package/dist/src/kernel/kernel.d.ts +84 -0
- package/dist/src/kernel/kernel.d.ts.map +1 -0
- package/dist/src/kernel/kernel.js +239 -0
- package/dist/src/kernel/kernel.js.map +1 -0
- package/dist/src/kernel/pipeline-cache.d.ts +90 -0
- package/dist/src/kernel/pipeline-cache.d.ts.map +1 -0
- package/dist/src/kernel/pipeline-cache.js +251 -0
- package/dist/src/kernel/pipeline-cache.js.map +1 -0
- package/dist/src/kernel/prelude.d.ts +35 -0
- package/dist/src/kernel/prelude.d.ts.map +1 -0
- package/dist/src/kernel/prelude.js +211 -0
- package/dist/src/kernel/prelude.js.map +1 -0
- package/dist/src/kernel/profiler.d.ts +64 -0
- package/dist/src/kernel/profiler.d.ts.map +1 -0
- package/dist/src/kernel/profiler.js +120 -0
- package/dist/src/kernel/profiler.js.map +1 -0
- package/dist/src/kernel/struct-block.d.ts +122 -0
- package/dist/src/kernel/struct-block.d.ts.map +1 -0
- package/dist/src/kernel/struct-block.js +353 -0
- package/dist/src/kernel/struct-block.js.map +1 -0
- package/dist/src/kernel/uniform-ring.d.ts +70 -0
- package/dist/src/kernel/uniform-ring.d.ts.map +1 -0
- package/dist/src/kernel/uniform-ring.js +146 -0
- package/dist/src/kernel/uniform-ring.js.map +1 -0
- package/dist/src/kernel/wgsl.d.ts +88 -0
- package/dist/src/kernel/wgsl.d.ts.map +1 -0
- package/dist/src/kernel/wgsl.js +390 -0
- package/dist/src/kernel/wgsl.js.map +1 -0
- package/dist/src/kernels.d.ts +81 -0
- package/dist/src/kernels.d.ts.map +1 -0
- package/dist/src/kernels.js +417 -0
- package/dist/src/kernels.js.map +1 -0
- package/dist/src/layouts/force-simulation.d.ts +498 -0
- package/dist/src/layouts/force-simulation.d.ts.map +1 -0
- package/dist/src/layouts/force-simulation.js +1650 -0
- package/dist/src/layouts/force-simulation.js.map +1 -0
- package/dist/src/layouts/forceatlas2.d.ts +210 -0
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -0
- package/dist/src/layouts/forceatlas2.js +759 -0
- package/dist/src/layouts/forceatlas2.js.map +1 -0
- package/dist/src/layouts/inputs.d.ts +40 -0
- package/dist/src/layouts/inputs.d.ts.map +1 -0
- package/dist/src/layouts/inputs.js +185 -0
- package/dist/src/layouts/inputs.js.map +1 -0
- package/dist/src/layouts/repulsion-exact.d.ts +85 -0
- package/dist/src/layouts/repulsion-exact.d.ts.map +1 -0
- package/dist/src/layouts/repulsion-exact.js +134 -0
- package/dist/src/layouts/repulsion-exact.js.map +1 -0
- package/dist/src/layouts/seed.d.ts +56 -0
- package/dist/src/layouts/seed.d.ts.map +1 -0
- package/dist/src/layouts/seed.js +173 -0
- package/dist/src/layouts/seed.js.map +1 -0
- package/dist/src/memory/buffer-pool.d.ts +73 -0
- package/dist/src/memory/buffer-pool.d.ts.map +1 -0
- package/dist/src/memory/buffer-pool.js +170 -0
- package/dist/src/memory/buffer-pool.js.map +1 -0
- package/dist/src/memory/lease.d.ts +53 -0
- package/dist/src/memory/lease.d.ts.map +1 -0
- package/dist/src/memory/lease.js +85 -0
- package/dist/src/memory/lease.js.map +1 -0
- package/dist/src/memory/readback.d.ts +143 -0
- package/dist/src/memory/readback.d.ts.map +1 -0
- package/dist/src/memory/readback.js +375 -0
- package/dist/src/memory/readback.js.map +1 -0
- package/dist/src/memory/residency.d.ts +83 -0
- package/dist/src/memory/residency.d.ts.map +1 -0
- package/dist/src/memory/residency.js +573 -0
- package/dist/src/memory/residency.js.map +1 -0
- package/dist/src/memory/upload-plan.d.ts +101 -0
- package/dist/src/memory/upload-plan.d.ts.map +1 -0
- package/dist/src/memory/upload-plan.js +265 -0
- package/dist/src/memory/upload-plan.js.map +1 -0
- package/dist/src/node/index.d.ts +64 -0
- package/dist/src/node/index.d.ts.map +1 -0
- package/dist/src/node/index.js +183 -0
- package/dist/src/node/index.js.map +1 -0
- package/dist/src/primitives/reduce.d.ts +57 -0
- package/dist/src/primitives/reduce.d.ts.map +1 -0
- package/dist/src/primitives/reduce.js +161 -0
- package/dist/src/primitives/reduce.js.map +1 -0
- package/dist/src/primitives/segmented-reduce.d.ts +38 -0
- package/dist/src/primitives/segmented-reduce.d.ts.map +1 -0
- package/dist/src/primitives/segmented-reduce.js +211 -0
- package/dist/src/primitives/segmented-reduce.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +209 -0
- package/dist/src/types/accelerator.d.ts.map +1 -0
- package/dist/src/types/accelerator.js +8 -0
- package/dist/src/types/accelerator.js.map +1 -0
- package/dist/src/types/context.d.ts +114 -0
- package/dist/src/types/context.d.ts.map +1 -0
- package/dist/src/types/context.js +7 -0
- package/dist/src/types/context.js.map +1 -0
- package/dist/src/types/layout.d.ts +95 -0
- package/dist/src/types/layout.d.ts.map +1 -0
- package/dist/src/types/layout.js +6 -0
- package/dist/src/types/layout.js.map +1 -0
- package/dist/src/types/memory.d.ts +22 -0
- package/dist/src/types/memory.d.ts.map +1 -0
- package/dist/src/types/memory.js +7 -0
- package/dist/src/types/memory.js.map +1 -0
- package/dist/src/types/options.d.ts +78 -0
- package/dist/src/types/options.d.ts.map +1 -0
- package/dist/src/types/options.js +7 -0
- package/dist/src/types/options.js.map +1 -0
- package/dist/src/types/run.d.ts +13 -0
- package/dist/src/types/run.d.ts.map +1 -0
- package/dist/src/types/run.js +6 -0
- package/dist/src/types/run.js.map +1 -0
- package/dist/src/wgsl/degree.wgsl.d.ts +10 -0
- package/dist/src/wgsl/degree.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/degree.wgsl.js +25 -0
- package/dist/src/wgsl/degree.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-attraction.wgsl.d.ts +12 -0
- package/dist/src/wgsl/fa2-attraction.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/fa2-attraction.wgsl.js +37 -0
- package/dist/src/wgsl/fa2-attraction.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-integrate.wgsl.d.ts +13 -0
- package/dist/src/wgsl/fa2-integrate.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/fa2-integrate.wgsl.js +69 -0
- package/dist/src/wgsl/fa2-integrate.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +12 -0
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +79 -0
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-speed-finalize.wgsl.d.ts +15 -0
- package/dist/src/wgsl/fa2-speed-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/fa2-speed-finalize.wgsl.js +54 -0
- package/dist/src/wgsl/fa2-speed-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +14 -0
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +57 -0
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-to-scene.wgsl.d.ts +11 -0
- package/dist/src/wgsl/fa2-to-scene.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/fa2-to-scene.wgsl.js +19 -0
- package/dist/src/wgsl/fa2-to-scene.wgsl.js.map +1 -0
- package/dist/src/wgsl/fill.wgsl.d.ts +7 -0
- package/dist/src/wgsl/fill.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/fill.wgsl.js +14 -0
- package/dist/src/wgsl/fill.wgsl.js.map +1 -0
- package/dist/src/wgsl/reduce.wgsl.d.ts +10 -0
- package/dist/src/wgsl/reduce.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/reduce.wgsl.js +63 -0
- package/dist/src/wgsl/reduce.wgsl.js.map +1 -0
- package/dist/src/wgsl/segmented-reduce.wgsl.d.ts +13 -0
- package/dist/src/wgsl/segmented-reduce.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/segmented-reduce.wgsl.js +35 -0
- package/dist/src/wgsl/segmented-reduce.wgsl.js.map +1 -0
- package/dist/tsconfig.build.tsbuildinfo +1 -0
- package/dist/webgpu-graph-algorithms.d.ts +1 -0
- package/dist/webgpu-graph-algorithms.js +4454 -0
- package/dist/webgpu-graph-algorithms.js.map +1 -0
- package/package.json +108 -17
- package/src/accelerator.ts +117 -0
- package/src/algorithms/degree.ts +142 -0
- package/src/browser/index.ts +57 -0
- package/src/constants.ts +116 -0
- package/src/context.ts +399 -0
- package/src/device/acquire.ts +256 -0
- package/src/device/caps.ts +122 -0
- package/src/device/error-scope.ts +171 -0
- package/src/device/lost.ts +142 -0
- package/src/device/webgpu-constants.ts +44 -0
- package/src/errors.ts +94 -0
- package/src/index.ts +102 -0
- package/src/kernel/batch.ts +427 -0
- package/src/kernel/dispatch.ts +162 -0
- package/src/kernel/kernel.ts +311 -0
- package/src/kernel/pipeline-cache.ts +288 -0
- package/src/kernel/prelude.ts +229 -0
- package/src/kernel/profiler.ts +148 -0
- package/src/kernel/struct-block.ts +439 -0
- package/src/kernel/uniform-ring.ts +184 -0
- package/src/kernel/wgsl.ts +490 -0
- package/src/kernels.ts +511 -0
- package/src/layouts/force-simulation.ts +2111 -0
- package/src/layouts/forceatlas2.ts +942 -0
- package/src/layouts/inputs.ts +252 -0
- package/src/layouts/repulsion-exact.ts +183 -0
- package/src/layouts/seed.ts +198 -0
- package/src/memory/buffer-pool.ts +204 -0
- package/src/memory/lease.ts +93 -0
- package/src/memory/readback.ts +429 -0
- package/src/memory/residency.ts +753 -0
- package/src/memory/upload-plan.ts +350 -0
- package/src/node/index.ts +230 -0
- package/src/primitives/reduce.ts +233 -0
- package/src/primitives/segmented-reduce.ts +270 -0
- package/src/types/accelerator.ts +236 -0
- package/src/types/context.ts +135 -0
- package/src/types/layout.ts +103 -0
- package/src/types/memory.ts +23 -0
- package/src/types/options.ts +84 -0
- package/src/types/run.ts +13 -0
- package/src/wgsl/degree.wgsl.ts +24 -0
- package/src/wgsl/fa2-attraction.wgsl.ts +37 -0
- package/src/wgsl/fa2-integrate.wgsl.ts +69 -0
- package/src/wgsl/fa2-repulsion-exact.wgsl.ts +78 -0
- package/src/wgsl/fa2-speed-finalize.wgsl.ts +53 -0
- package/src/wgsl/fa2-stats-finalize.wgsl.ts +57 -0
- package/src/wgsl/fa2-to-scene.wgsl.ts +19 -0
- package/src/wgsl/fill.wgsl.ts +13 -0
- package/src/wgsl/reduce.wgsl.ts +62 -0
- package/src/wgsl/segmented-reduce.wgsl.ts +35 -0
package/README.md
CHANGED
|
@@ -1,36 +1,391 @@
|
|
|
1
1
|
# @graphty/webgpu-graph-algorithms
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
WebGPU-accelerated graph algorithms and layouts over the `@graphty/graph-format` snapshot, for Node
|
|
4
|
+
(Dawn, through the `webgpu` npm package) and browsers (Chromium). One code base, three entry points:
|
|
4
5
|
|
|
5
|
-
|
|
6
|
-
|
|
6
|
+
| Entry | Import | What it gives you |
|
|
7
|
+
| ------------------------------------------ | ------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
8
|
+
| `@graphty/webgpu-graph-algorithms` | the core | `createForceAtlas2`, `createAccelerator`, `GpuContext`, `degree`, `seedPositions`, `WebGpuGraphError`, `isSoftwareAdapter`, the constants (`EXACT_MAX_NODES`, `FA2_DEFAULTS`, `LAYOUT_TUNING_DEFAULTS`, ...) and the option / stats / accelerator types |
|
|
9
|
+
| `@graphty/webgpu-graph-algorithms/node` | Node only | `createNodeGpuContext`, `probeNodeWebGpu`, `createNodeGpu` (Dawn), `dawnFlags` |
|
|
10
|
+
| `@graphty/webgpu-graph-algorithms/browser` | browsers only | `probeBrowserWebGpu`, `requestGpuContext` |
|
|
7
11
|
|
|
8
|
-
|
|
9
|
-
|
|
12
|
+
**Status: phase P3 (ForceAtlas2, exact tier).** The GPU ForceAtlas2 is usable from Node (`run()`) and from a
|
|
13
|
+
browser frame loop (`step()` once per frame) up to `exactMaxNodes` = 32768 nodes with the default
|
|
14
|
+
`repulsion: "auto"`, and at any size with `repulsion: "exact"` (all pairs, O(n^2) per iteration: 18.971 ms per
|
|
15
|
+
iteration at 100k nodes / 1M edges on an RTX 4070 SUPER). The grid tier for 10^5-10^6 nodes (P4),
|
|
16
|
+
Fruchterman-Reingold (P5) and the algorithms (P7+) follow the phase plan of `design/webgpu/webgpu-acceleration-plan.md` (monorepo root)
|
|
17
|
+
section 13 and the interface contract `design/webgpu/plans/2026-09-14-webgpu-p0-p3-interfaces.md`; the gate
|
|
18
|
+
record of this phase is `docs/decisions/G3.md`. There is no CPU fallback anywhere in this package: when no
|
|
19
|
+
adapter or device exists it throws `WebGpuGraphError`.
|
|
10
20
|
|
|
11
|
-
##
|
|
21
|
+
## Install
|
|
12
22
|
|
|
13
|
-
|
|
23
|
+
```bash
|
|
24
|
+
npm install @graphty/webgpu-graph-algorithms @graphty/graph-format
|
|
25
|
+
# Node only: Dawn is an OPTIONAL peer dependency (browser consumers never install it)
|
|
26
|
+
npm install webgpu@0.4.0
|
|
27
|
+
```
|
|
14
28
|
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
29
|
+
`webgpu@0.4.0` is the last Linux binary linking against glibc <= 2.34 (Ubuntu 22.04 ships 2.35;
|
|
30
|
+
`webgpu@0.6.x` needs glibc 2.38). The peer range `>=0.4.0 <1.0.0` admits the newer builds on a newer
|
|
31
|
+
glibc; a Node consumer that forgets the package gets `E_NO_WEBGPU` with the message
|
|
32
|
+
"install the optional peer dependency webgpu@0.4.0". `@graphty/algorithms` and `@graphty/layout` are
|
|
33
|
+
optional peer dependencies too: the accelerator types mirror their interfaces (structurally until the
|
|
34
|
+
package moves into the monorepo), so a consumer that type-checks against this package installs them; a
|
|
35
|
+
consumer that never touches the accelerator types does not need them.
|
|
19
36
|
|
|
20
|
-
##
|
|
37
|
+
## ForceAtlas2 from Node
|
|
21
38
|
|
|
22
|
-
|
|
39
|
+
```ts
|
|
40
|
+
import { fromEdgeArrays } from "@graphty/graph-format";
|
|
41
|
+
import { createForceAtlas2 } from "@graphty/webgpu-graph-algorithms";
|
|
42
|
+
import { createNodeGpuContext } from "@graphty/webgpu-graph-algorithms/node";
|
|
23
43
|
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
- Should not be installed as a dependency
|
|
27
|
-
- Exists only for administrative purposes
|
|
44
|
+
const ctx = await createNodeGpuContext(); // Dawn; { adapter: "llvmpipe" } selects Mesa's software adapter
|
|
45
|
+
const snapshot = fromEdgeArrays({ directed: false, nodeCount, src, dst, weights }); // undirected: pass toUndirected().snapshot otherwise
|
|
28
46
|
|
|
29
|
-
The
|
|
30
|
-
|
|
31
|
-
|
|
47
|
+
// The owner's stride-3 array: x, y, z per node in scene units. NaN rows are seeded by the CPU port's LCG
|
|
48
|
+
// (seed 42 here) inside [-1, 1) x scale + center; finite rows are kept as the starting layout.
|
|
49
|
+
const positions = new Float32Array(3 * snapshot.nodeCount).fill(Number.NaN);
|
|
32
50
|
|
|
33
|
-
|
|
51
|
+
const sim = createForceAtlas2(ctx, { seed: 42, dim: 2, maxIter: 200, gravity: 1, scalingRatio: 2 });
|
|
52
|
+
sim.load(snapshot, positions); // uploads the CSR core, seeds the NaN rows, compiles the kernels
|
|
53
|
+
const stats = await sim.run({ batch: 8 }); // 8 iterations per submit until settled or maxIter
|
|
54
|
+
console.log(sim.iterationsDone, sim.settled, stats.speed, stats.rmsRadius, stats.layoutRadius);
|
|
55
|
+
// positions now holds the layout; stats.trace holds the last batch's per-iteration controller values
|
|
34
56
|
|
|
35
|
-
|
|
36
|
-
|
|
57
|
+
sim.dispose(); // destroys the simulation's buffers
|
|
58
|
+
ctx.release(snapshot); // destroys the snapshot's buffers (nothing is freed by GC)
|
|
59
|
+
ctx.dispose(); // destroys the device and lets the process exit
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
Sizes above `exactMaxNodes` need `repulsion: "exact"` until the grid tier lands (P4); with the default
|
|
63
|
+
`"auto"` the simulation rejects `load()` with `E_UNSUPPORTED { feature: "repulsion.grid" }`. The same run
|
|
64
|
+
from the command line, with a verification of the result:
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
pnpm exec tsx benchmarks/layout-run.ts --nodes 100000 --edges 1000000 --iterations 100 --batch 8
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
## ForceAtlas2 in a browser frame loop
|
|
71
|
+
|
|
72
|
+
The simulation never blocks a frame: `step()` submits one batch and returns a promise that resolves when
|
|
73
|
+
that batch's positions have been copied into your array; while `maxInFlight` batches are in flight the call
|
|
74
|
+
coalesces (nothing is queued, the oldest pending batch's promise is returned). Attach `.catch` once per
|
|
75
|
+
distinct promise and draw whatever the array holds -- it lags the GPU by at most one batch.
|
|
76
|
+
|
|
77
|
+
```ts
|
|
78
|
+
import { createAccelerator } from "@graphty/webgpu-graph-algorithms";
|
|
79
|
+
import { probeBrowserWebGpu, requestGpuContext } from "@graphty/webgpu-graph-algorithms/browser";
|
|
80
|
+
|
|
81
|
+
const probe = await probeBrowserWebGpu(); // never throws: { ok, code, reason, adapter, summary }
|
|
82
|
+
if (!probe.ok) {
|
|
83
|
+
throw new Error(`${probe.code}: ${probe.reason ?? ""}`); // E_NO_WEBGPU, E_NO_ADAPTER or E_SOFTWARE_ONLY
|
|
84
|
+
}
|
|
85
|
+
const ctx = await requestGpuContext({ adapter: probe.adapter ?? undefined });
|
|
86
|
+
const acc = createAccelerator(ctx, { layout: { exactMaxNodes: 32768 } }); // the tuning every simulation inherits
|
|
87
|
+
const sim = acc.forceAtlas2({ seed: 1, iterationsPerStep: 1, maxInFlight: 2 });
|
|
88
|
+
sim.load(snapshot, positions); // positions: your stride-3 Float32Array; NaN rows are seeded
|
|
89
|
+
|
|
90
|
+
let last: Promise<void> | null = null;
|
|
91
|
+
function frame(): void {
|
|
92
|
+
if (!sim.settled) {
|
|
93
|
+
const p = sim.step(); // never awaited
|
|
94
|
+
if (p !== last) {
|
|
95
|
+
last = p;
|
|
96
|
+
p.catch((error: unknown) => {
|
|
97
|
+
console.error(error); // E_DEVICE_LOST, E_VALIDATION, ...; the simulation is disposed on device loss
|
|
98
|
+
});
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
draw(positions);
|
|
102
|
+
requestAnimationFrame(frame);
|
|
103
|
+
}
|
|
104
|
+
requestAnimationFrame(frame);
|
|
105
|
+
|
|
106
|
+
// interaction
|
|
107
|
+
sim.setPosition(i, x, y, 0); // a drag: written to the device now, visible in the next readback, reheats
|
|
108
|
+
sim.setFixed(mask); // pins: a graph-format NodeMask (ceil(n / 32) words); an unpin reheats, a pin does not
|
|
109
|
+
sim.reheat(); // "play again" after the layout settled (nothing is reset but the counters)
|
|
110
|
+
await sim.flush(); // pause: resolves when every submitted batch has landed in `positions`
|
|
111
|
+
sim.dispose(); // stop; then ctx.release(snapshot) and ctx.dispose() when the graph goes away
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
`stats` and `settled` describe the last completed batch. Topology changes go through `load(next,
|
|
115
|
+
remappedPositions)` on the same simulation: in-flight readbacks of the old graph are discarded, new (NaN)
|
|
116
|
+
rows are seeded inside the current bounding box, the fixed mask and the position overrides are cleared (the
|
|
117
|
+
caller re-issues its pins with a mask over the new index space), and the simulation reheats.
|
|
118
|
+
|
|
119
|
+
## Options
|
|
120
|
+
|
|
121
|
+
`createForceAtlas2(ctx, options)` and `accelerator.forceAtlas2(options)` take `ForceAtlas2Options` (the same
|
|
122
|
+
names and defaults as the CPU port in `@graphty/layout`) plus the GPU tuning:
|
|
123
|
+
|
|
124
|
+
| Option | Default | Meaning |
|
|
125
|
+
| --------------------------------- | ---------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
126
|
+
| `dim` | `2` | `2` or `3`; in 2D every readback writes `z = center[2]` whatever was uploaded |
|
|
127
|
+
| `scale`, `center` | `1`, `[0, 0, 0]` | scene units = layout units x `scale` + `center` |
|
|
128
|
+
| `seed` | `null` | the LCG seed of the NaN rows (the CPU port's generator, bit for bit); `0` / `null` draws a random seed |
|
|
129
|
+
| `maxIter` | `100` | `run()` stops and `settled` turns true after this many iterations since `load()` / `reheat()` |
|
|
130
|
+
| `jitterTolerance` | `1` | the speed controller's tolerance |
|
|
131
|
+
| `scalingRatio` | `2` | the repulsion constant |
|
|
132
|
+
| `gravity` | `1` | toward the centroid (`compat: "paper"`) or the origin (`compat: "networkx"`) |
|
|
133
|
+
| `strongGravity` | `false` | gravity proportional to the distance |
|
|
134
|
+
| `distributedAction` | `false` | attraction divided by the source's mass |
|
|
135
|
+
| `linlog` | `false` | logarithmic attraction |
|
|
136
|
+
| `nodeMass` | `null` | `null`: the node column with the role `mass` when present, else `outDegree + 1`; a `Float32Array` of length n; a numeric node column name (a `Record` is `E_UNSUPPORTED`) |
|
|
137
|
+
| `nodeSize` | `null` | `E_UNSUPPORTED` when non-null (no overlap prevention) |
|
|
138
|
+
| `weight` | unset | `true`: the snapshot's arc weights; an edge column name: that numeric column; unset / `false` / `null`: every edge weighs 1 |
|
|
139
|
+
| `dissuadeHubs` | `false` | accepted and ignored |
|
|
140
|
+
| `settleThreshold`, `settleWindow` | `0.001`, `10` | settled when the mean displacement stayed below the threshold for `settleWindow` iterations |
|
|
141
|
+
| `iterationsPerStep` | `1` | iterations per `step()` (the Node `run()` uses its own `batch`) |
|
|
142
|
+
| `maxInFlight` | `2` | batches in flight before `step()` coalesces; `1` for the strictest freshness |
|
|
143
|
+
|
|
144
|
+
GPU tuning (`GpuLayoutTuning`; also the `layout` field of `createAccelerator`'s options, inherited by every
|
|
145
|
+
simulation the accelerator creates):
|
|
146
|
+
|
|
147
|
+
| Option | Default | Meaning |
|
|
148
|
+
| --------------------------------------------------- | ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------ |
|
|
149
|
+
| `repulsion` | `"auto"` | `"exact"` at any n; `"auto"` = exact iff n <= `exactMaxNodes`; `"grid"` is `E_UNSUPPORTED` until P4 |
|
|
150
|
+
| `exactMaxNodes` | `32768` | the crossover, measured on the RTX 4070 SUPER (`docs/decisions/G3.md`); pass your own for another GPU (P4's `calibrateLayout` measures it) |
|
|
151
|
+
| `deterministic` | `true` | fixed summation order (the exact tier is always deterministic) |
|
|
152
|
+
| `compat` | `"paper"` | `"networkx"` reproduces NetworkX 3.4's `forceatlas2_layout` (gravity toward the origin, its accumulated swing / traction) |
|
|
153
|
+
| `nearMax`, `gridMax2D`, `gridMax3D`, `extentFactor` | `64`, `512`, `128`, `6` | stored for the grid tier (P4) |
|
|
154
|
+
|
|
155
|
+
Every range error is `E_INVALID_ARGUMENT` (`gravity < 0`, `scalingRatio <= 0`, `jitterTolerance <= 0`,
|
|
156
|
+
`maxIter < 1`, `settleWindow < 1`, `maxInFlight < 1`, `iterationsPerStep < 1`, `dim` not 2 or 3,
|
|
157
|
+
`scale <= 0`). `setParams(patch)` changes the options of a live simulation; a force-law change (`linlog`,
|
|
158
|
+
`strongGravity`, `distributedAction`) recompiles and resets the speed controller, a numeric tweak only changes
|
|
159
|
+
the next batch's parameters; `dim` and `maxInFlight` cannot change after creation.
|
|
160
|
+
|
|
161
|
+
## Stats
|
|
162
|
+
|
|
163
|
+
`sim.stats` (`ForceAtlas2Stats`) after every completed batch: `iteration`, `swing`, `traction`, `speed`,
|
|
164
|
+
`speedEfficiency`, `meanDisplacement`, `rmsRadius`, `layoutRadius` (max |p - centroid|), `centroid`,
|
|
165
|
+
`repulsionTier` (`"exact"`), `msPerIteration` (the GPU time per iteration when `timestamp-query` was granted,
|
|
166
|
+
else the batch's wall time divided by its iteration count), the grid fields (`null` on the exact tier) and
|
|
167
|
+
`trace`: one `{ swing, traction, speed, speedEfficiency, meanDisplacement, settledCount }` record per
|
|
168
|
+
iteration of the last batch.
|
|
169
|
+
|
|
170
|
+
## Acquisition
|
|
171
|
+
|
|
172
|
+
### Node
|
|
173
|
+
|
|
174
|
+
```ts
|
|
175
|
+
import { hasErrorCode } from "@graphty/webgpu-graph-algorithms";
|
|
176
|
+
import { createNodeGpu, createNodeGpuContext, probeNodeWebGpu } from "@graphty/webgpu-graph-algorithms/node";
|
|
177
|
+
|
|
178
|
+
const probe = await probeNodeWebGpu(); // never throws; probe.summary has vendor / architecture / software / limits
|
|
179
|
+
console.log(probe.ok, probe.code, probe.summary?.software);
|
|
180
|
+
try {
|
|
181
|
+
const ctx = await createNodeGpuContext({ rejectSoftware: true }); // Dawn + GpuContext.create
|
|
182
|
+
console.log(ctx.caps.vendor, ctx.caps.architecture, ctx.caps.software ? "software" : "hardware");
|
|
183
|
+
ctx.dispose();
|
|
184
|
+
} catch (error) {
|
|
185
|
+
if (hasErrorCode(error, "E_NO_WEBGPU") || hasErrorCode(error, "E_SOFTWARE_ONLY")) {
|
|
186
|
+
console.error((error as Error).message);
|
|
187
|
+
process.exit(1);
|
|
188
|
+
}
|
|
189
|
+
throw error;
|
|
190
|
+
}
|
|
191
|
+
const handle = await createNodeGpu({ software: false }); // the bare Dawn handle: import("webgpu"), globals, create(flags)
|
|
192
|
+
handle.dispose();
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
Options of `createNodeGpu` / `createNodeGpuContext`: `adapter` (Dawn `adapter=<substring>`, e.g. `"llvmpipe"`
|
|
196
|
+
or `"4070"`), `backend` (`"vulkan"` | `"null"` | ...), `dawnFeatures` (Dawn toggles), `software` (shorthand
|
|
197
|
+
for `adapter=llvmpipe`, Linux / Mesa specific), `installGlobals` (default `true`: `GPUBufferUsage` and friends
|
|
198
|
+
on `globalThis`), plus the `GpuContext.create` options (`powerPreference`, `rejectSoftware`, `limits`
|
|
199
|
+
(`"raise"` by default), `optionalFeatures` (`["subgroups", "timestamp-query"]` by default), `label`,
|
|
200
|
+
`onError`).
|
|
201
|
+
|
|
202
|
+
On a machine where the NVIDIA Vulkan driver cannot find `libEGL.so.1` (a container without `libegl1`),
|
|
203
|
+
Dawn silently lists only llvmpipe; see `docs/HEADLESS_GPU_REPORT.md` appendix D for the `LD_LIBRARY_PATH`
|
|
204
|
+
recipe.
|
|
205
|
+
|
|
206
|
+
### Browser
|
|
207
|
+
|
|
208
|
+
```ts
|
|
209
|
+
import { probeBrowserWebGpu, requestGpuContext } from "@graphty/webgpu-graph-algorithms/browser";
|
|
210
|
+
|
|
211
|
+
const probe = await probeBrowserWebGpu(); // never throws: { ok, code, reason, adapter, summary }
|
|
212
|
+
if (probe.ok) {
|
|
213
|
+
const ctx = await requestGpuContext({ adapter: probe.adapter ?? undefined });
|
|
214
|
+
// ... createAccelerator(ctx), createForceAtlas2(ctx, ...), degree(ctx, snapshot)
|
|
215
|
+
ctx.dispose();
|
|
216
|
+
}
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
Chromium only until Firefox / WebKit ship `subgroups` and `timestamp-query` on Linux CI. The `./node`
|
|
220
|
+
subpath is never reachable from browser code: nothing in the core or the browser entry imports it, and
|
|
221
|
+
`sideEffects: false` lets bundlers drop what you do not use.
|
|
222
|
+
|
|
223
|
+
## The `degree` diagnostic
|
|
224
|
+
|
|
225
|
+
`degree(ctx, snapshot)` is the walking-skeleton kernel kept public as a diagnostic: it runs the row-walking gather with the
|
|
226
|
+
package's dummy-binding pattern and returns a `Uint32Array` equal to `snapshot.outDegree()`. If it disagrees, nothing else
|
|
227
|
+
will work; if it agrees, the device, the upload path, the pipeline cache and the readback ring all do.
|
|
228
|
+
|
|
229
|
+
```ts
|
|
230
|
+
import { fromEdgeArrays } from "@graphty/graph-format";
|
|
231
|
+
import { degree } from "@graphty/webgpu-graph-algorithms";
|
|
232
|
+
import { createNodeGpuContext } from "@graphty/webgpu-graph-algorithms/node";
|
|
233
|
+
|
|
234
|
+
const ctx = await createNodeGpuContext(); // Dawn through the webgpu package; { adapter: "llvmpipe" } selects the software adapter
|
|
235
|
+
const s = fromEdgeArrays({ directed: false, nodeCount: 3, src: new Uint32Array([0, 1]), dst: new Uint32Array([1, 2]) });
|
|
236
|
+
const out = await degree(ctx, s); // Uint32Array [1, 2, 1] === s.outDegree()
|
|
237
|
+
ctx.release(s); // destroys the snapshot's buffers (nothing is freed by GC)
|
|
238
|
+
ctx.dispose(); // destroys the device and lets the process exit
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
In a browser, `import { requestGpuContext } from "@graphty/webgpu-graph-algorithms/browser"` and `const ctx = await
|
|
242
|
+
requestGpuContext();` replace the first import and line; the rest is identical.
|
|
243
|
+
|
|
244
|
+
## Errors
|
|
245
|
+
|
|
246
|
+
Every condition the package detects itself is a `WebGpuGraphError` with a stable `code`
|
|
247
|
+
(`E_NO_WEBGPU`, `E_NO_ADAPTER`, `E_NO_DEVICE`, `E_SOFTWARE_ONLY`, `E_DEVICE_LOST`, `E_DISPOSED`,
|
|
248
|
+
`E_VALIDATION`, `E_SHADER_COMPILE`, `E_OUT_OF_MEMORY`, `E_TOO_LARGE`, `E_UNSUPPORTED`,
|
|
249
|
+
`E_INVALID_ARGUMENT`, `E_SNAPSHOT`, `E_RELEASED`, `E_NOT_LOADED`, `E_ABORTED`) and frozen `details`.
|
|
250
|
+
`isWebGpuGraphError(x)` and `hasErrorCode(x, code)` are structural brand checks, so they survive two copies
|
|
251
|
+
of the package. The graph-format codes `E_GPU_INELIGIBLE`, `E_UNKNOWN_NODE`, `E_UNKNOWN_COLUMN` and
|
|
252
|
+
`E_COLUMN_LENGTH` pass through unchanged (`PASSTHROUGH_FORMAT_CODES`). A simulation whose snapshot was
|
|
253
|
+
released rejects its next `step()` with `E_RELEASED`; a lost device disposes every simulation and rejects
|
|
254
|
+
every pending `step()` with `E_DEVICE_LOST` (create a new context from a fresh adapter and `load()` again).
|
|
255
|
+
|
|
256
|
+
## Benchmarks
|
|
257
|
+
|
|
258
|
+
```bash
|
|
259
|
+
pnpm run bench # every group; appends to benchmarks/out/<runner-class>.json
|
|
260
|
+
pnpm exec tsx benchmarks/run.ts upload roundtrip layout-exact # selected groups; --no-save, --runs N, --allow-software
|
|
261
|
+
pnpm exec tsx benchmarks/layout-run.ts --nodes 100000 --edges 1000000 # the end-to-end layout driver (exit 1 on a bad result)
|
|
262
|
+
pnpm run gpu:report > gpu-report.json # the adapter report with a 10 s nvidia-smi sample
|
|
263
|
+
pnpm run bench:compare # the last out session vs benchmarks/results/<runner-class>.json (> 3x fails)
|
|
264
|
+
```
|
|
265
|
+
|
|
266
|
+
The runner class is `<vendor>-<architecture>-driver<major>` (`scripts/runner-class.js`; `GRAPHTY_RUNNER_CLASS` overrides it,
|
|
267
|
+
which the GPU lane sets to `gpu-linux-t4`). Software adapters never time anything: `pnpm run bench` prints
|
|
268
|
+
`software adapter: nothing timed` on lavapipe unless `--allow-software` is given, and a software session is never a
|
|
269
|
+
baseline. The checked-in baselines under `benchmarks/results/` are written by the owner on the dev box (the RTX 4070 SUPER,
|
|
270
|
+
runner class `nvidia-lovelace-driver580` -- Dawn spells the architecture `lovelace`) and, for the GPU lane, from the
|
|
271
|
+
lane's own artifact; `bench:compare` skips when the card was not quiet during the report's sample. Groups: `upload` (T-1),
|
|
272
|
+
`roundtrip` (T-2, T-3), `layout-exact` (T-4 and the Node side of T-5: `step(1)` on the exact ladder 1k / 4k / 8k / 16k /
|
|
273
|
+
32k / 65k and at 10k, two rows per rung -- the wall time of one iteration with its readback, and the GPU time per iteration
|
|
274
|
+
the profiler reports; every rung starts with an untimed clock warm-up burst, because NVIDIA's power management leaves the
|
|
275
|
+
SM clock at its idle 210 MHz under sparse sub-millisecond dispatches and the kernels then measure 4-15x slower). The
|
|
276
|
+
Chromium number of T-5 comes from the `bench`-tagged browser test (`GRAPHTY_BROWSER_GPU=nvidia node
|
|
277
|
+
scripts/run-browser-project.js`), which appends its session through the Vitest commands bridge. `exactMaxNodes` is
|
|
278
|
+
re-fixed from the ladder by the rule of plan section 7.8 (the largest rung under 4 ms per iteration, rounded down to a
|
|
279
|
+
power of two; `benchmarks/layout-exact.bench.ts` `exactMaxNodesFromLadder`).
|
|
280
|
+
|
|
281
|
+
## Performance
|
|
282
|
+
|
|
283
|
+
Regenerated from the last session of each baseline under `benchmarks/results/` (`nvidia-lovelace-driver580.json`, the
|
|
284
|
+
dev box; `gpu-linux-t4.json`, the CI lane) by the procedure recorded in `docs/decisions/G3.md` appendix A; the targets
|
|
285
|
+
are the T-table of plan section 10.4. A missed target is re-fixed by a recorded owner decision in
|
|
286
|
+
`docs/decisions/G<n>.md`, never relaxed silently.
|
|
287
|
+
|
|
288
|
+
### The dev box (nvidia-lovelace-driver580)
|
|
289
|
+
|
|
290
|
+
Measured on nvidia-lovelace-driver580 (NVIDIA: 580.173.02 580.173.2.0), session 2026-09-16T02:07:45.933Z, medians of 5 runs; Chromium: nvidia / lovelace (nvidia-lovelace-driver0, the description is redacted by Chromium), session 2026-09-16T02:18:11.896Z.
|
|
291
|
+
|
|
292
|
+
| Id | What | Target | Measured |
|
|
293
|
+
| --- | -------------------------------------------------------------------------------------------- | ------------------- | -------------------- |
|
|
294
|
+
| T-1 | Upload of the 100k / 1M weighted hot prefix (16.4 MB); 1M / 10M (164 MB) | <= 10 ms; <= 100 ms | 6.032 ms; 127.531 ms |
|
|
295
|
+
| T-2 | `degree` + 400 KB readback at 100k (core resident), Node | <= 2 ms | 0.878 ms |
|
|
296
|
+
| T-3 | Empty submit + 4-byte `readU32` round trip, Dawn | <= 0.1 ms | 0.170 ms |
|
|
297
|
+
| T-4 | ForceAtlas2 exact tier, GPU time per iteration (profiler) at 10k; at 16k | <= 1 ms; <= 2 ms | 0.586 ms; 1.052 ms |
|
|
298
|
+
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 10k, Chromium (Node in brackets) | <= 6 ms | 2.400 ms (0.738 ms) |
|
|
299
|
+
|
|
300
|
+
Two rows miss their target in this session: the 1M / 10M upload (127.5 ms against 100 ms, the open owner decision of
|
|
301
|
+
`docs/decisions/G1.md` section 7) and the empty-submit round trip (0.170 ms against 0.1 ms: the row is measured after the
|
|
302
|
+
`upload` group, whose CPU-heavy setup lets the SM clock fall to its idle state; the same row measures 0.041-0.074 ms at
|
|
303
|
+
the working clock -- finding G3-F2 of `docs/decisions/G3.md` section 10).
|
|
304
|
+
|
|
305
|
+
The exact curve (the `layout-exact` group: 2D, E = 10n, seeded G(n, m), one simulation per rung; ms / iteration from the profiler):
|
|
306
|
+
|
|
307
|
+
| n | ms / iteration | step(1) wall (ms) | pairs / s |
|
|
308
|
+
| ----- | -------------- | ----------------- | --------- |
|
|
309
|
+
| 1024 | 0.096 | 0.228 | 1.09e+10 |
|
|
310
|
+
| 4096 | 0.255 | 0.415 | 6.58e+10 |
|
|
311
|
+
| 8192 | 0.478 | 0.630 | 1.40e+11 |
|
|
312
|
+
| 10000 | 0.586 | 0.738 | 1.71e+11 |
|
|
313
|
+
| 16384 | 1.052 | 1.226 | 2.55e+11 |
|
|
314
|
+
| 32768 | 2.560 | 2.787 | 4.19e+11 |
|
|
315
|
+
| 65536 | 8.405 | 8.862 | 5.11e+11 |
|
|
316
|
+
|
|
317
|
+
The end-to-end run of `benchmarks/layout-run.ts --nodes 100000 --edges 1000000` (the exact tier at 100k, 100
|
|
318
|
+
iterations, batches of 8) takes 18.971 ms per iteration on the same card, uploads and readbacks included (16.975 ms of
|
|
319
|
+
GPU time per iteration in the last batch).
|
|
320
|
+
|
|
321
|
+
### The CI lane (gpu-linux-t4)
|
|
322
|
+
|
|
323
|
+
The first run of the GPU lane (`gpu.yml`, graphty-monorepo run 35316416067, 2026-09-18) on a machine.dev T4 -- one Tesla
|
|
324
|
+
T4 (16 GB), 4 vCPU of a Xeon Platinum 8259CL, driver 580.126.20 -- wrote this baseline; `scripts/bench-compare.js` fails
|
|
325
|
+
a later run of the lane whose median exceeds 3x these figures. The T-table targets were set on the dev box; the T4 meets
|
|
326
|
+
T-4 and T-5 and misses T-1 (both uploads), T-2 and T-3, which is the class difference of a datacentre card behind a
|
|
327
|
+
cloud vCPU (host-side copies and submit latency), not a regression: the exact tier's `ms / iteration` is 1.7x the
|
|
328
|
+
RTX 4070 SUPER's at 10k and 3.0x at 65k.
|
|
329
|
+
|
|
330
|
+
Measured on gpu-linux-t4 (NVIDIA: 580.126.20 580.126.20.0), session 2026-09-18T07:03:17.146Z, medians of 5 runs; Chromium: nvidia / turing (nvidia-turing-driver0, the description is redacted by Chromium), session 2026-09-18T07:02:07.834Z.
|
|
331
|
+
|
|
332
|
+
| Id | What | Target | Measured |
|
|
333
|
+
| --- | -------------------------------------------------------------------------------------------- | ------------------- | --------------------- |
|
|
334
|
+
| T-1 | Upload of the 100k / 1M weighted hot prefix (16.4 MB); 1M / 10M (164 MB) | <= 10 ms; <= 100 ms | 14.458 ms; 257.954 ms |
|
|
335
|
+
| T-2 | `degree` + 400 KB readback at 100k (core resident), Node | <= 2 ms | 2.298 ms |
|
|
336
|
+
| T-3 | Empty submit + 4-byte `readU32` round trip, Dawn | <= 0.1 ms | 0.171 ms |
|
|
337
|
+
| T-4 | ForceAtlas2 exact tier, GPU time per iteration (profiler) at 10k; at 16k | <= 1 ms; <= 2 ms | 0.977 ms; 1.879 ms |
|
|
338
|
+
| T-5 | ForceAtlas2 per-frame cost, `step(1)` + the 12n readback at 10k, Chromium (Node in brackets) | <= 6 ms | 2.600 ms (1.357 ms) |
|
|
339
|
+
|
|
340
|
+
The exact curve (the `layout-exact` group: 2D, E = 10n, seeded G(n, m), one simulation per rung; ms / iteration from the profiler):
|
|
341
|
+
|
|
342
|
+
| n | ms / iteration | step(1) wall (ms) | pairs / s |
|
|
343
|
+
| ----- | -------------- | ----------------- | --------- |
|
|
344
|
+
| 1024 | 0.334 | 0.977 | 3.13e+9 |
|
|
345
|
+
| 4096 | 0.426 | 0.881 | 3.93e+10 |
|
|
346
|
+
| 8192 | 0.801 | 1.184 | 8.38e+10 |
|
|
347
|
+
| 10000 | 0.977 | 1.357 | 1.02e+11 |
|
|
348
|
+
| 16384 | 1.879 | 2.317 | 1.43e+11 |
|
|
349
|
+
| 32768 | 6.534 | 7.241 | 1.64e+11 |
|
|
350
|
+
| 65536 | 24.883 | 25.955 | 1.73e+11 |
|
|
351
|
+
|
|
352
|
+
## Development
|
|
353
|
+
|
|
354
|
+
```bash
|
|
355
|
+
pnpm install # at the monorepo root
|
|
356
|
+
cd webgpu-graph-algorithms
|
|
357
|
+
pnpm run build:all # tsc + the vite bundle + the d.ts shims
|
|
358
|
+
pnpm run lint # eslint + tsc --noEmit + the strict-consumer compile
|
|
359
|
+
pnpm exec vitest run --project=node # the node suite on the default adapter
|
|
360
|
+
pnpm run coverage # the node suite with the 80 / 80 / 75 / 80 thresholds
|
|
361
|
+
node scripts/run-browser-project.js # the browser smoke suite (SwiftShader by default)
|
|
362
|
+
node scripts/gpu-report.js # the adapter report and the policy verdict (after build)
|
|
363
|
+
cd .. && pnpm exec knip # unused files / exports / dependencies
|
|
364
|
+
```
|
|
365
|
+
|
|
366
|
+
Environment variables of the test harness (plan section 12.2):
|
|
367
|
+
|
|
368
|
+
| Variable | Default lane (GitHub, software) | GPU lane (NVIDIA T4) | Local (dev box) |
|
|
369
|
+
| ----------------------------------------- | ------------------------------- | -------------------- | ------------------------------------------------------------------- |
|
|
370
|
+
| `GRAPHTY_GPU_ADAPTER` | `llvmpipe` | unset | unset (NVIDIA) or `llvmpipe` to mirror CI |
|
|
371
|
+
| `GRAPHTY_GPU_REQUIRE` | `any` | `nvidia` | unset (skip with a printed reason) or `hardware` |
|
|
372
|
+
| `GRAPHTY_BROWSER_GPU` | `swiftshader` | `nvidia` | `nvidia` (needs the libEGL tree) |
|
|
373
|
+
| `GRAPHTY_GPU_NO_SUBGROUPS` | `1` in a second pass | `1` in a second pass | unset |
|
|
374
|
+
| `GRAPHTY_GPU_INSPECT` | unset | unset | `1` to enable `sim.inspect(name)` / `debugRunStages` in test builds |
|
|
375
|
+
| `GRAPHTY_NOISE_FLOOR_WRITE` | unset | unset | `1` to (re)write this adapter's noise fixtures |
|
|
376
|
+
| `GRAPHTY_DAWN_FEATURES` | unset | unset | optional Dawn toggles |
|
|
377
|
+
| `GRAPHTY_EGL_LIB_DIR` / `LD_LIBRARY_PATH` | -- | unset | the extracted libEGL tree |
|
|
378
|
+
| `VK_DRIVER_FILES` | the lavapipe ICD | unset | unset |
|
|
379
|
+
| `XDG_RUNTIME_DIR` | `/tmp` | `/tmp` | `/tmp` |
|
|
380
|
+
|
|
381
|
+
`GRAPHTY_GPU_REQUIRE` is the one policy switch: unset skips tests that need an adapter (with the reason
|
|
382
|
+
printed), `any` fails when no adapter exists, `hardware` additionally rejects lavapipe / SwiftShader, a
|
|
383
|
+
vendor name (`nvidia`) additionally requires that vendor. A wrong result is never a skip. Every f32 tolerance
|
|
384
|
+
of the layout tests is derived from `benchmarks/results/noise-floor.json` (the measured summation noise across
|
|
385
|
+
the subgroup twins, lavapipe, SwiftShader and NVIDIA; a tolerance is at most 10x its floor), every kernel ships
|
|
386
|
+
with sabotage mutations that must fail its test by 10x that tolerance, and every kernel result is compared
|
|
387
|
+
twice for bitwise determinism before it is compared to an oracle (plan section 11.9).
|
|
388
|
+
|
|
389
|
+
## License
|
|
390
|
+
|
|
391
|
+
MIT
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export * from "./src/browser/index.js";
|
package/dist/browser.js
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
import { G as GpuContext, W as WebGpuGraphError } from "./chunks/context-E6iKaeuJ.js";
|
|
2
|
+
function navigatorGpu() {
|
|
3
|
+
if (typeof navigator === "undefined") {
|
|
4
|
+
return void 0;
|
|
5
|
+
}
|
|
6
|
+
const candidate = navigator;
|
|
7
|
+
return candidate.gpu;
|
|
8
|
+
}
|
|
9
|
+
function probeBrowserWebGpu(options) {
|
|
10
|
+
return GpuContext.probe({
|
|
11
|
+
gpu: navigatorGpu(),
|
|
12
|
+
powerPreference: options?.powerPreference ?? "high-performance",
|
|
13
|
+
rejectSoftware: options?.rejectSoftware
|
|
14
|
+
});
|
|
15
|
+
}
|
|
16
|
+
function requestGpuContext(options) {
|
|
17
|
+
const gpu = navigatorGpu();
|
|
18
|
+
if (gpu === void 0 && options?.adapter === void 0) {
|
|
19
|
+
return Promise.reject(
|
|
20
|
+
new WebGpuGraphError("E_NO_WEBGPU", "navigator.gpu is undefined: this browser or context has no WebGPU", {
|
|
21
|
+
reason: "navigator.gpu is undefined",
|
|
22
|
+
hint: "WebGPU needs a supporting browser and a secure context (https or localhost)"
|
|
23
|
+
})
|
|
24
|
+
);
|
|
25
|
+
}
|
|
26
|
+
return GpuContext.create({ powerPreference: "high-performance", ...options, gpu, runtime: "browser" });
|
|
27
|
+
}
|
|
28
|
+
export {
|
|
29
|
+
probeBrowserWebGpu,
|
|
30
|
+
requestGpuContext
|
|
31
|
+
};
|
|
32
|
+
//# sourceMappingURL=browser.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"browser.js","sources":["../src/browser/index.ts"],"sourcesContent":["/// <reference types=\"@webgpu/types\" preserve=\"true\" />\n/**\n * The ./browser entry (spec 2.1, 2.3, 3.4; contract 3.6): the only directory of the package allowed to read\n * navigator.gpu. probeBrowserWebGpu never throws; requestGpuContext tags the context runtime \"browser\".\n */\n\nimport { GpuContext } from \"../context.js\";\nimport { WebGpuGraphError } from \"../errors.js\";\nimport type { GpuContextOptions, ProbeResult } from \"../types/context.js\";\n\n/** Options of the browser helpers (spec 3.4); a type alias, not an empty `extends` interface, which strictTypeChecked's no-empty-object-type (allowInterfaces \"never\") reports. */\nexport type BrowserGpuOptions = Omit<GpuContextOptions, \"gpu\" | \"device\" | \"runtime\">;\n\n/**\n * navigator.gpu, or undefined when the runtime has no navigator or no WebGPU (Node 22 has a navigator\n * without gpu; browsers without WebGPU have navigator.gpu undefined).\n * @returns the GPU object or undefined\n */\nfunction navigatorGpu(): GPU | undefined {\n if (typeof navigator === \"undefined\") {\n return undefined;\n }\n const candidate: { readonly gpu?: GPU | undefined } = navigator;\n return candidate.gpu;\n}\n\n/**\n * navigator.gpu absent -> { ok: false, code: \"E_NO_WEBGPU\" }; else GpuContext.probe. Never throws.\n * @param options - the power preference and the software policy\n * @returns the probe result\n */\nexport function probeBrowserWebGpu(options?: BrowserGpuOptions): Promise<ProbeResult> {\n return GpuContext.probe({\n gpu: navigatorGpu(),\n powerPreference: options?.powerPreference ?? \"high-performance\",\n rejectSoftware: options?.rejectSoftware,\n });\n}\n\n/**\n * GpuContext.create({ gpu: navigator.gpu, powerPreference: \"high-performance\", runtime: \"browser\", ...options });\n * pass `{ adapter: probe.adapter }` to reuse the probed adapter; create() honours rejectSoftware.\n * @param options - see BrowserGpuOptions\n * @returns the context\n */\nexport function requestGpuContext(options?: BrowserGpuOptions): Promise<GpuContext> {\n const gpu = navigatorGpu();\n if (gpu === undefined && options?.adapter === undefined) {\n return Promise.reject(\n new WebGpuGraphError(\"E_NO_WEBGPU\", \"navigator.gpu is undefined: this browser or context has no WebGPU\", {\n reason: \"navigator.gpu is undefined\",\n hint: \"WebGPU needs a supporting browser and a secure context (https or localhost)\",\n }),\n );\n }\n return GpuContext.create({ powerPreference: \"high-performance\", ...options, gpu, runtime: \"browser\" });\n}\n"],"names":[],"mappings":";AAkBA,SAAS,eAAgC;AACrC,MAAI,OAAO,cAAc,aAAa;AAClC,WAAO;AAAA,EACX;AACA,QAAM,YAAgD;AACtD,SAAO,UAAU;AACrB;AAOO,SAAS,mBAAmB,SAAmD;AAClF,SAAO,WAAW,MAAM;AAAA,IACpB,KAAK,aAAA;AAAA,IACL,iBAAiB,SAAS,mBAAmB;AAAA,IAC7C,gBAAgB,SAAS;AAAA,EAAA,CAC5B;AACL;AAQO,SAAS,kBAAkB,SAAkD;AAChF,QAAM,MAAM,aAAA;AACZ,MAAI,QAAQ,UAAa,SAAS,YAAY,QAAW;AACrD,WAAO,QAAQ;AAAA,MACX,IAAI,iBAAiB,eAAe,qEAAqE;AAAA,QACrG,QAAQ;AAAA,QACR,MAAM;AAAA,MAAA,CACT;AAAA,IAAA;AAAA,EAET;AACA,SAAO,WAAW,OAAO,EAAE,iBAAiB,oBAAoB,GAAG,SAAS,KAAK,SAAS,WAAW;AACzG;"}
|