@graphty/webgpu-graph-algorithms 0.6.27 → 0.6.28
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +56 -6
- package/dist/acquire.d.ts +2 -0
- package/dist/browser.js +18 -1
- package/dist/browser.js.map +1 -1
- package/dist/chunks/accelerator-B-FjQwaA.js +19173 -0
- package/dist/chunks/accelerator-B-FjQwaA.js.map +1 -0
- package/dist/chunks/managed-D_GdQtnu.js +98 -0
- package/dist/chunks/managed-D_GdQtnu.js.map +1 -0
- package/dist/node.js +18 -1
- package/dist/node.js.map +1 -1
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +5 -3
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/all-pairs.d.ts.map +1 -1
- package/dist/src/algorithms/all-pairs.js +72 -47
- package/dist/src/algorithms/all-pairs.js.map +1 -1
- package/dist/src/algorithms/betweenness.d.ts +1 -1
- package/dist/src/algorithms/betweenness.js +2 -2
- package/dist/src/algorithms/closeness.d.ts +45 -42
- package/dist/src/algorithms/closeness.d.ts.map +1 -1
- package/dist/src/algorithms/closeness.js +295 -226
- package/dist/src/algorithms/closeness.js.map +1 -1
- package/dist/src/browser/index.d.ts +10 -0
- package/dist/src/browser/index.d.ts.map +1 -1
- package/dist/src/browser/index.js +22 -0
- package/dist/src/browser/index.js.map +1 -1
- package/dist/src/constants.d.ts +13 -3
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +13 -3
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernels.d.ts +14 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +62 -28
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/force-simulation.d.ts.map +1 -1
- package/dist/src/layouts/force-simulation.js +0 -1
- package/dist/src/layouts/force-simulation.js.map +1 -1
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
- package/dist/src/layouts/forceatlas2.js +0 -1
- package/dist/src/layouts/forceatlas2.js.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.js +0 -1
- package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
- package/dist/src/layouts/repulsion-grid.d.ts +1 -3
- package/dist/src/layouts/repulsion-grid.d.ts.map +1 -1
- package/dist/src/layouts/repulsion-grid.js +1 -6
- package/dist/src/layouts/repulsion-grid.js.map +1 -1
- package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
- package/dist/src/layouts/spring-electrical.js +0 -1
- package/dist/src/layouts/spring-electrical.js.map +1 -1
- package/dist/src/managed.d.ts +11 -0
- package/dist/src/managed.d.ts.map +1 -0
- package/dist/src/managed.js +129 -0
- package/dist/src/managed.js.map +1 -0
- package/dist/src/node/index.d.ts +11 -0
- package/dist/src/node/index.d.ts.map +1 -1
- package/dist/src/node/index.js +21 -0
- package/dist/src/node/index.js.map +1 -1
- package/dist/src/primitives/grid-pyramid.d.ts +16 -15
- package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
- package/dist/src/primitives/grid-pyramid.js +20 -28
- package/dist/src/primitives/grid-pyramid.js.map +1 -1
- package/dist/src/types/accelerator.d.ts +2 -0
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/managed.d.ts +81 -0
- package/dist/src/types/managed.d.ts.map +1 -0
- package/dist/src/types/managed.js +7 -0
- package/dist/src/types/managed.js.map +1 -0
- package/dist/src/wgsl/bc-forward.wgsl.d.ts +1 -1
- package/dist/src/wgsl/bc-forward.wgsl.js +1 -1
- package/dist/src/wgsl/closeness-level.wgsl.d.ts +37 -0
- package/dist/src/wgsl/closeness-level.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-level.wgsl.js +204 -0
- package/dist/src/wgsl/closeness-level.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts +11 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.js +42 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.js.map +1 -0
- package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +2 -2
- package/dist/src/wgsl/grid-centroid-hub.wgsl.js +2 -2
- package/dist/webgpu-graph-algorithms.js +142 -15586
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +10 -4
- package/src/accelerator.ts +5 -3
- package/src/algorithms/all-pairs.ts +86 -56
- package/src/algorithms/betweenness.ts +2 -2
- package/src/algorithms/closeness.ts +353 -256
- package/src/browser/index.ts +37 -0
- package/src/constants.ts +13 -3
- package/src/kernels.ts +65 -36
- package/src/layouts/force-simulation.ts +0 -1
- package/src/layouts/forceatlas2.ts +0 -1
- package/src/layouts/fruchterman-reingold.ts +0 -1
- package/src/layouts/repulsion-grid.ts +2 -7
- package/src/layouts/spring-electrical.ts +0 -1
- package/src/managed.ts +172 -0
- package/src/node/index.ts +36 -0
- package/src/primitives/grid-pyramid.ts +29 -41
- package/src/types/accelerator.ts +2 -0
- package/src/types/managed.ts +86 -0
- package/src/wgsl/bc-forward.wgsl.ts +1 -1
- package/src/wgsl/closeness-level.wgsl.ts +203 -0
- package/src/wgsl/closeness-rowsum.wgsl.ts +41 -0
- package/src/wgsl/grid-centroid-hub.wgsl.ts +2 -2
- package/dist/chunks/context-BZY6SMsM.js +0 -3615
- package/dist/chunks/context-BZY6SMsM.js.map +0 -1
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +0 -20
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +0 -1
- package/dist/src/wgsl/closeness-reduce.wgsl.js +0 -69
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +0 -1
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +0 -22
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +0 -1
- package/dist/src/wgsl/closeness-sweep.wgsl.js +0 -106
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +0 -1
- package/src/wgsl/closeness-reduce.wgsl.ts +0 -68
- package/src/wgsl/closeness-sweep.wgsl.ts +0 -105
package/src/browser/index.ts
CHANGED
|
@@ -6,7 +6,17 @@
|
|
|
6
6
|
|
|
7
7
|
import { GpuContext } from "../context.js";
|
|
8
8
|
import { WebGpuGraphError } from "../errors.js";
|
|
9
|
+
import { manageAccelerator } from "../managed.js";
|
|
9
10
|
import type { GpuContextOptions, ProbeResult } from "../types/context.js";
|
|
11
|
+
import type { AcquireAcceleratorOptions, ManagedAccelerator } from "../types/managed.js";
|
|
12
|
+
|
|
13
|
+
export type {
|
|
14
|
+
AcceleratorDeclined,
|
|
15
|
+
AcceleratorReady,
|
|
16
|
+
AcquireAcceleratorOptions,
|
|
17
|
+
AcquireResult,
|
|
18
|
+
ManagedAccelerator,
|
|
19
|
+
} from "../types/managed.js";
|
|
10
20
|
|
|
11
21
|
/** Options of the browser helpers (spec 3.4); a type alias, not an empty `extends` interface, which strictTypeChecked's no-empty-object-type (allowInterfaces "never") reports. */
|
|
12
22
|
export type BrowserGpuOptions = Omit<GpuContextOptions, "gpu" | "device" | "runtime">;
|
|
@@ -55,3 +65,30 @@ export function requestGpuContext(options?: BrowserGpuOptions): Promise<GpuConte
|
|
|
55
65
|
}
|
|
56
66
|
return GpuContext.create({ powerPreference: "high-performance", ...options, gpu, runtime: "browser" });
|
|
57
67
|
}
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* The managed accelerator on navigator.gpu: probe, context, device self-check and accelerator on the first
|
|
71
|
+
* `current()`, a new device after a loss, disposal by `dispose()`. A software adapter is declined unless
|
|
72
|
+
* `acceptSoftware` is set. The same function as the `./acquire` subpath resolves to in a browser.
|
|
73
|
+
* @param options - see AcquireAcceleratorOptions; `adapter` is ignored here
|
|
74
|
+
* @returns the handle; nothing is probed until `current()` is called
|
|
75
|
+
*/
|
|
76
|
+
export function acquireAccelerator(options?: AcquireAcceleratorOptions): ManagedAccelerator {
|
|
77
|
+
return manageAccelerator(
|
|
78
|
+
{
|
|
79
|
+
probe: (o, rejectSoftware) => probeBrowserWebGpu({ powerPreference: o.powerPreference, rejectSoftware }),
|
|
80
|
+
open: (o, probe, rejectSoftware) =>
|
|
81
|
+
requestGpuContext({
|
|
82
|
+
adapter: probe.adapter ?? undefined,
|
|
83
|
+
powerPreference: o.powerPreference,
|
|
84
|
+
rejectSoftware,
|
|
85
|
+
warnUnreleasedSnapshots: o.warnUnreleasedSnapshots,
|
|
86
|
+
}),
|
|
87
|
+
noWebGpuFix: () =>
|
|
88
|
+
(globalThis as { isSecureContext?: boolean }).isSecureContext === false
|
|
89
|
+
? "serve the page over https or from localhost: WebGPU needs a secure context"
|
|
90
|
+
: null,
|
|
91
|
+
},
|
|
92
|
+
options,
|
|
93
|
+
);
|
|
94
|
+
}
|
package/src/constants.ts
CHANGED
|
@@ -43,7 +43,9 @@ export const UNIFORM_SLOT_BYTES = 256;
|
|
|
43
43
|
* unbiasedness item) the same day by its re-scope to the whole-field ratio, and the far field's accuracy on those
|
|
44
44
|
* fixtures is the follow-up -- because "auto" is the default every consumer sees and the exact tier is the accurate
|
|
45
45
|
* one; the rule's answer on the dev box (1024) is recorded, not shipped, until that work closes and the value is
|
|
46
|
-
* re-fixed.
|
|
46
|
+
* re-fixed. Re-measured on 2026-10-02 after the grid tier's hub-cell centroid became a direct dispatch (issue #732):
|
|
47
|
+
* the rule still answers 1024 (exact 0.26 against grid 0.18 ms per iteration at 4,096 nodes), because the indirect
|
|
48
|
+
* dispatch's cost was Dawn's validation pass, which the profiler's per-pass times never included. A
|
|
47
49
|
* consumer whose GPU differs (integrated, Apple, T4) passes its own value through
|
|
48
50
|
* createAccelerator(ctx, { layout: { exactMaxNodes } }); calibrateLayout(ctx) measures it.
|
|
49
51
|
*/
|
|
@@ -261,8 +263,16 @@ export const SSSP_DELTA_FACTOR = 32;
|
|
|
261
263
|
export const F32_INF_BITS = 0x7f800000;
|
|
262
264
|
/** Design 8.4 "k planned from maxBufferSize and a 25% budget": the share of `maxBufferSize` one betweenness source batch may hold. WebGPU exposes no device memory size, so this is a fraction of the largest buffer, not a memory measurement. */
|
|
263
265
|
export const BC_BATCH_BUDGET_FRACTION = 0.25;
|
|
264
|
-
/**
|
|
265
|
-
|
|
266
|
+
/**
|
|
267
|
+
* The most sources one betweenness batch runs together. 256 rather than design 10.1's 64 (issue #733): a batch's
|
|
268
|
+
* cost is dominated by its per-level submits and readbacks, not by its width, so a quarter of the batches runs an
|
|
269
|
+
* exact call on random graphs of 1k / 2k / 4k nodes (4 edges per node, RTX 4070 SUPER) 2.8x / 2.5x / 1.8x faster
|
|
270
|
+
* with the frontier forward form and 1.9x / 1.6x / 1.5x with the per-batch choice, with bit-identical vertex and
|
|
271
|
+
* edge scores (the gathers add every source's dependency in source order whatever the batching). The memory budget
|
|
272
|
+
* still decides k above about 16k nodes at default limits, so large graphs are unchanged. Forward levels per submit
|
|
273
|
+
* stay at `MAX_LEVELS_PER_SUBMIT`: the parameter ring is sized for it.
|
|
274
|
+
*/
|
|
275
|
+
export const BC_MAX_BATCH = 256;
|
|
266
276
|
/** Design 8.4 (McLaughlin-Bader): a betweenness batch runs the edge-parallel forward pass when the previous batch's level count is below `BC_EDGE_PARALLEL_GAMMA * log2(n)`. The design names the rule and no value; 2 is unmeasured and a benchmark run re-fixes it. */
|
|
267
277
|
export const BC_EDGE_PARALLEL_GAMMA = 2;
|
|
268
278
|
/**
|
package/src/kernels.ts
CHANGED
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
* dedupe-claim and dedupe-filter; P8-T4 adds frontier-finalize with the FrontierCounters and FrontierParams blocks;
|
|
13
13
|
* P8-T5 adds advance-expand; P8-T6 adds bfs-contract and sssp-pred; P8-T7 adds bfs-fused; P8-T8 adds bfs-bottom-up,
|
|
14
14
|
* bfs-bitset-build and bfs-unvisited-flags; P8-T9 adds sssp-relax; P8-T10 adds bf-relax with the BfParams and BfFlags
|
|
15
|
-
* blocks;
|
|
15
|
+
* blocks; closeness adds closeness-level and closeness-rowsum with the ClosenessParams block; P9 (betweenness) adds bc-finalize, bc-forward,
|
|
16
16
|
* bc-backward, bc-gather, bc-edge-gather, bc-forward-edge and bc-count (issue #719) with the BcParams block; all-pairs shortest paths
|
|
17
17
|
* (design 8.7) adds apsp-init and apsp-fw with the ApspParams block. P11 (the structure and community phase, plan
|
|
18
18
|
* design/webgpu/plans/2026-09-23-webgpu-p11-structure-and-community.md) adds the graph build on the device (coo-emit,
|
|
@@ -45,8 +45,8 @@ import { bfsContractWgsl } from "./wgsl/bfs-contract.wgsl.js";
|
|
|
45
45
|
import { bfsFusedWgsl } from "./wgsl/bfs-fused.wgsl.js";
|
|
46
46
|
import { bfsNextDegreeWgsl } from "./wgsl/bfs-next-degree.wgsl.js";
|
|
47
47
|
import { bfsUnvisitedFlagsWgsl } from "./wgsl/bfs-unvisited-flags.wgsl.js";
|
|
48
|
-
import {
|
|
49
|
-
import {
|
|
48
|
+
import { closenessLevelWgsl } from "./wgsl/closeness-level.wgsl.js";
|
|
49
|
+
import { closenessRowsumWgsl } from "./wgsl/closeness-rowsum.wgsl.js";
|
|
50
50
|
import { compactScatterWgsl } from "./wgsl/compact-scatter.wgsl.js";
|
|
51
51
|
import { cooEmitWgsl } from "./wgsl/coo-emit.wgsl.js";
|
|
52
52
|
import { cooScatterWgsl } from "./wgsl/coo-scatter.wgsl.js";
|
|
@@ -139,8 +139,8 @@ export type KernelId =
|
|
|
139
139
|
| "bfs-next-degree"
|
|
140
140
|
| "sssp-relax"
|
|
141
141
|
| "bf-relax"
|
|
142
|
-
| "closeness-
|
|
143
|
-
| "closeness-
|
|
142
|
+
| "closeness-level"
|
|
143
|
+
| "closeness-rowsum"
|
|
144
144
|
| "bc-finalize"
|
|
145
145
|
| "bc-forward"
|
|
146
146
|
| "bc-backward"
|
|
@@ -482,8 +482,7 @@ export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
|
482
482
|
* `arcEnd` @44 (the bound arc window), `predKind` @48 (0 arc, 1 node), `bitsBase` @52, `source` @56, `stride` @60
|
|
483
483
|
* (a grid-stride plan's stride), `firstOfSubmit` @64 (the boundary's index inside its submit, clamped to 1: both
|
|
484
484
|
* the unvisited-count and the unvisited-degree-sum subtraction run at >= 1, issue #391), `iteration` @68 (an
|
|
485
|
-
* `sssp-pred` hop pass, P8-T9), `
|
|
486
|
-
* distance sums of a sampled run), `pad2` @76. The `slotBase` field that once addressed the selector's indirect slots went with
|
|
485
|
+
* `sssp-pred` hop pass, P8-T9), `pad3` @72, `pad2` @76. The `slotBase` field that once addressed the selector's indirect slots went with
|
|
487
486
|
* the slots (2026-09-25); `pad2` keeps the block an explicit 80 bytes, the way every block here is padded.
|
|
488
487
|
*/
|
|
489
488
|
export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams", [
|
|
@@ -505,7 +504,7 @@ export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams
|
|
|
505
504
|
["stride", "u32"],
|
|
506
505
|
["firstOfSubmit", "u32"],
|
|
507
506
|
["iteration", "u32"],
|
|
508
|
-
["
|
|
507
|
+
["pad3", "u32"],
|
|
509
508
|
["pad2", "u32"],
|
|
510
509
|
]);
|
|
511
510
|
|
|
@@ -521,6 +520,35 @@ export const BC_PARAMS: UniformBlock = UniformBlock.define("BcParams", [
|
|
|
521
520
|
["pad1", "u32"],
|
|
522
521
|
]);
|
|
523
522
|
|
|
523
|
+
/**
|
|
524
|
+
* `ClosenessParams` (uniform, 64 B): the params of `closeness-level` and `closeness-rowsum` -- `role` @0 (level: 0 a
|
|
525
|
+
* level, 1 the seed's clear, 2 the seed's sources; rowsum: 0 integer, 1 f32, 2 harmonic), `n` @4, `words` @8 (32-bit
|
|
526
|
+
* words per node, so `32 x words` sources per batch), `base` @12 (the words of one `bits` region), `total` @16 (the
|
|
527
|
+
* words the clear zeroes), `level` @20, `row` @24 (the level's row of the submit's count table), `ctrl` @28 (the word
|
|
528
|
+
* of the control ring in `table`), `count` @32 (the batch's sources), `source` @36 (its first source, or its first
|
|
529
|
+
* word of the source list), `sourcesAt` @40 (the word of a sampled run's source list in `table`, 0 for the exact
|
|
530
|
+
* run), `pullAt` @44 (the frontier arcs above which a level pulls), `perNode` @48 (1: a sampled run's per-node
|
|
531
|
+
* distance sums), `arcCount` @52 (the arcs a push level covers), `pullOk` @56 (1: a level may pull), `pad0` @60.
|
|
532
|
+
*/
|
|
533
|
+
export const CLOSENESS_PARAMS: UniformBlock = UniformBlock.define("ClosenessParams", [
|
|
534
|
+
["role", "u32"],
|
|
535
|
+
["n", "u32"],
|
|
536
|
+
["words", "u32"],
|
|
537
|
+
["base", "u32"],
|
|
538
|
+
["total", "u32"],
|
|
539
|
+
["level", "u32"],
|
|
540
|
+
["row", "u32"],
|
|
541
|
+
["ctrl", "u32"],
|
|
542
|
+
["count", "u32"],
|
|
543
|
+
["source", "u32"],
|
|
544
|
+
["sourcesAt", "u32"],
|
|
545
|
+
["pullAt", "u32"],
|
|
546
|
+
["perNode", "u32"],
|
|
547
|
+
["arcCount", "u32"],
|
|
548
|
+
["pullOk", "u32"],
|
|
549
|
+
["pad0", "u32"],
|
|
550
|
+
]);
|
|
551
|
+
|
|
524
552
|
/** `BfParams` (uniform, 16 B; P8-T10): `edgeCount` @0 (the logical edges of the `edgeList` view), `stride` @4 (the grid-stride plan's stride), `maxRetries` @8 (PD-12's compare-exchange bound), `cutoffBits` @12 (the f32 bit pattern of the CPU port's `cutoff`, `+Inf` when absent). */
|
|
525
553
|
export const BF_PARAMS: UniformBlock = UniformBlock.define("BfParams", [
|
|
526
554
|
["edgeCount", "u32"],
|
|
@@ -1085,7 +1113,7 @@ const GRID_CENTROID: KernelEntry = {
|
|
|
1085
1113
|
phase: "P4",
|
|
1086
1114
|
};
|
|
1087
1115
|
|
|
1088
|
-
/** `grid-centroid-hub` (G4b, spec 7.7; P4-T9, PD-13, DEP-P4-L): one workgroup per
|
|
1116
|
+
/** `grid-centroid-hub` (G4b, spec 7.7; P4-T9, PD-13, DEP-P4-L): one workgroup per word of `hubList`, dispatched directly, a WG-strided sum through `wg_reduce_vec4` guarded by `h < hubCount[0]`; 6 storage bindings (`hubCount` is a read-only view of `hubCounters`). */
|
|
1089
1117
|
const GRID_CENTROID_HUB: KernelEntry = {
|
|
1090
1118
|
id: "grid-centroid-hub",
|
|
1091
1119
|
body: gridCentroidHubWgsl,
|
|
@@ -1425,41 +1453,42 @@ const BF_RELAX: KernelEntry = {
|
|
|
1425
1453
|
phase: "P8",
|
|
1426
1454
|
};
|
|
1427
1455
|
|
|
1428
|
-
/** `closeness-
|
|
1429
|
-
const
|
|
1430
|
-
id: "closeness-
|
|
1431
|
-
body:
|
|
1432
|
-
entryPoint: "
|
|
1433
|
-
bindings:
|
|
1434
|
-
decl(1, 0, "
|
|
1435
|
-
decl(1, 1, "
|
|
1436
|
-
decl(1, 2, "
|
|
1437
|
-
decl(1, 3, "
|
|
1438
|
-
decl(
|
|
1439
|
-
|
|
1456
|
+
/** `closeness-level` (design 8.4 "32 sources per u32 word"): one level of closeness's bit-parallel multi-source BFS in one dispatch, `32 x P.words` sources per batch -- role 0 the level (push over the out-arcs or pull over the in-arcs, push one invocation per arc, pull one per node, chosen per level on the device from the frontier's arcs against `P.pullAt`, the pull with an early exit once every source has reached the node), role 1 and role 2 the batch's seed; 6 storage bindings (the forward and the reverse CSR as group-1 state, `bits` and `table` as `array<atomic<u32>>`). */
|
|
1457
|
+
const CLOSENESS_LEVEL: KernelEntry = {
|
|
1458
|
+
id: "closeness-level",
|
|
1459
|
+
body: closenessLevelWgsl,
|
|
1460
|
+
entryPoint: "closeness_level",
|
|
1461
|
+
bindings: [
|
|
1462
|
+
decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
|
|
1463
|
+
decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
|
|
1464
|
+
decl(1, 2, "inRowPtr", "storage-ro", "array<u32>"),
|
|
1465
|
+
decl(1, 3, "inColIdx", "storage-ro", "array<u32>"),
|
|
1466
|
+
decl(1, 4, "bits", "storage", "array<atomic<u32>>"),
|
|
1467
|
+
decl(1, 5, "table", "storage", "array<atomic<u32>>"),
|
|
1468
|
+
decl(2, 0, "P", "uniform", "ClosenessParams"),
|
|
1469
|
+
],
|
|
1440
1470
|
overrideDecls: [],
|
|
1441
|
-
uniforms: [
|
|
1471
|
+
uniforms: [CLOSENESS_PARAMS],
|
|
1442
1472
|
needs: [],
|
|
1443
1473
|
snippetSlots: [],
|
|
1444
1474
|
phase: "P8",
|
|
1445
1475
|
};
|
|
1446
1476
|
|
|
1447
|
-
/** `closeness-
|
|
1448
|
-
const
|
|
1449
|
-
id: "closeness-
|
|
1450
|
-
body:
|
|
1451
|
-
entryPoint: "
|
|
1477
|
+
/** `closeness-rowsum` (design 8.7): closeness from a finished all-pairs matrix, one workgroup per row -- the off-diagonal finite entries summed as integers (`P.role` 0), as f32 (1) or as reciprocals (2, harmonic), by a tree reduction; 2 storage bindings (`dist` read-only, `out`). */
|
|
1478
|
+
const CLOSENESS_ROWSUM: KernelEntry = {
|
|
1479
|
+
id: "closeness-rowsum",
|
|
1480
|
+
body: closenessRowsumWgsl,
|
|
1481
|
+
entryPoint: "closeness_rowsum",
|
|
1452
1482
|
bindings: [
|
|
1453
|
-
decl(1, 0, "
|
|
1454
|
-
decl(1, 1, "
|
|
1455
|
-
decl(
|
|
1456
|
-
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1483
|
+
decl(1, 0, "dist", "storage-ro", "array<f32>"),
|
|
1484
|
+
decl(1, 1, "out", "storage", "array<u32>"),
|
|
1485
|
+
decl(2, 0, "P", "uniform", "ClosenessParams"),
|
|
1457
1486
|
],
|
|
1458
1487
|
overrideDecls: [],
|
|
1459
|
-
uniforms: [
|
|
1488
|
+
uniforms: [CLOSENESS_PARAMS],
|
|
1460
1489
|
needs: [],
|
|
1461
1490
|
snippetSlots: [],
|
|
1462
|
-
phase: "
|
|
1491
|
+
phase: "P9",
|
|
1463
1492
|
};
|
|
1464
1493
|
|
|
1465
1494
|
/** `bc-finalize` (design 8.4, 5.4): the one-lane bookkeeping of a betweenness batch -- role 1 seeds it (depth 0 and one path for the k seed entries of the claim log, `stackTop = k`, `level = U32_MAX`), role 0 is the level boundary (`ends[level + 1] = stackTop`, `frontierCount`, `done` on an empty level); 6 storage bindings (the counters block as `array<atomic<u32>>`, `ends`, `S` read-only, `depthK`, `sigmaK` and `levelMax` plain: one lane writes the seed; `SCALED` seeds f32 bits). The design's finalize row has 2; `ends` is the third (the level boundary), and the seed's `S`, `depthK`, `sigmaK` and `levelMax` make it six. */
|
|
@@ -1845,7 +1874,7 @@ const MST_LINK: KernelEntry = {
|
|
|
1845
1874
|
* and `"fa2-to-scene"`; M8b-T3 landed the seven P7 entries and P4 its thirteen; P8-T3 landed the three compact /
|
|
1846
1875
|
* dedupe entries, P8-T4 `"frontier-finalize"`, P8-T5 `"advance-expand"`, P8-T6 `"bfs-contract"` and `"sssp-pred"` and
|
|
1847
1876
|
* P8-T7 `"bfs-fused"`, P8-T8 `"bfs-bottom-up"`, `"bfs-bitset-build"` and `"bfs-unvisited-flags"`, P8-T9
|
|
1848
|
-
* `"sssp-relax"`, P8-T10 `"bf-relax"
|
|
1877
|
+
* `"sssp-relax"`, P8-T10 `"bf-relax"`, closeness `"closeness-level"` and `"closeness-rowsum"`, betweenness the
|
|
1849
1878
|
* six `"bc-*"` entries, all-pairs shortest paths `"apsp-init"` and `"apsp-fw"`, and P11 its seven (the graph
|
|
1850
1879
|
* build, the group-by-key, label propagation's step and triangle counting) plus Boruvka's `"mst-best"` and
|
|
1851
1880
|
* `"mst-link"`, so every member of `KernelId`
|
|
@@ -1896,8 +1925,8 @@ const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze
|
|
|
1896
1925
|
"bfs-next-degree": BFS_NEXT_DEGREE,
|
|
1897
1926
|
"sssp-relax": SSSP_RELAX,
|
|
1898
1927
|
"bf-relax": BF_RELAX,
|
|
1899
|
-
"closeness-
|
|
1900
|
-
"closeness-
|
|
1928
|
+
"closeness-level": CLOSENESS_LEVEL,
|
|
1929
|
+
"closeness-rowsum": CLOSENESS_ROWSUM,
|
|
1901
1930
|
"bc-finalize": BC_FINALIZE,
|
|
1902
1931
|
"bc-forward": BC_FORWARD,
|
|
1903
1932
|
"bc-backward": BC_BACKWARD,
|
|
@@ -520,7 +520,6 @@ export class ForceAtlas2Model implements ForceModel<ForceAtlas2Options, ForceAtl
|
|
|
520
520
|
cellStart: resources.buffer("cellStart"),
|
|
521
521
|
hubList: resources.buffer("hubList"),
|
|
522
522
|
hubCounters,
|
|
523
|
-
hubArgs: resources.buffer("hubArgs"),
|
|
524
523
|
pyramid: resources.buffer("pyramid"),
|
|
525
524
|
});
|
|
526
525
|
const wg = k1.workgroupSize;
|
|
@@ -513,7 +513,6 @@ export class FruchtermanReingoldModel implements ForceModel<FruchtermanReingoldO
|
|
|
513
513
|
cellStart: resources.buffer("cellStart"),
|
|
514
514
|
hubList: resources.buffer("hubList"),
|
|
515
515
|
hubCounters,
|
|
516
|
-
hubArgs: resources.buffer("hubArgs"),
|
|
517
516
|
pyramid: resources.buffer("pyramid"),
|
|
518
517
|
});
|
|
519
518
|
const wg = k1.workgroupSize;
|
|
@@ -15,7 +15,7 @@ import { GRID_HUB_CELL } from "../constants.js";
|
|
|
15
15
|
import { BufferUsage } from "../device/webgpu-constants.js";
|
|
16
16
|
import { WebGpuGraphError } from "../errors.js";
|
|
17
17
|
import { plan1d } from "../kernel/dispatch.js";
|
|
18
|
-
import { type BoundKernel,
|
|
18
|
+
import { type BoundKernel, type Kernel } from "../kernel/kernel.js";
|
|
19
19
|
import { type PipelineCache } from "../kernel/pipeline-cache.js";
|
|
20
20
|
import { type UniformBlock, type UniformValues } from "../kernel/struct-block.js";
|
|
21
21
|
import { type WgslModuleSpec } from "../kernel/wgsl.js";
|
|
@@ -56,7 +56,6 @@ export interface RepulsionGridResources extends RepulsionExactResources {
|
|
|
56
56
|
readonly cellStart: Binding;
|
|
57
57
|
readonly hubList: Binding;
|
|
58
58
|
readonly hubCounters: Binding;
|
|
59
|
-
readonly hubArgs: Binding;
|
|
60
59
|
readonly pyramid: Binding;
|
|
61
60
|
}
|
|
62
61
|
|
|
@@ -216,8 +215,7 @@ export class RepulsionGrid {
|
|
|
216
215
|
|
|
217
216
|
/**
|
|
218
217
|
* The model-owned buffers of the grid tier (spec 7.3; PD-11): `cellKey` / `cellVal` / `sortedKey` / `sortedIdx`
|
|
219
|
-
* 4n, `cellHist` / `cellStart` 4 histWords (cells + 2^dim + 1) zeroed, `hubList` one word per possible hub cell, `
|
|
220
|
-
* indirect slot, `pyramid` 16 B per pyramid cell zeroed. `hubCounters` (16 B, zeroed) is the MODEL's on every
|
|
218
|
+
* 4n, `cellHist` / `cellStart` 4 histWords (cells + 2^dim + 1) zeroed, `hubList` one word per possible hub cell, `pyramid` 16 B per pyramid cell zeroed. `hubCounters` (16 B, zeroed) is the MODEL's on every
|
|
221
219
|
* tier (PD-14: K1 binds it on the exact tier too). n = 0 reports one node's worth of bytes (spec 3.6).
|
|
222
220
|
* @param n - the node count
|
|
223
221
|
* @param spec - the grid
|
|
@@ -238,7 +236,6 @@ export class RepulsionGrid {
|
|
|
238
236
|
usage: STORAGE_RW,
|
|
239
237
|
zero: false,
|
|
240
238
|
},
|
|
241
|
-
{ name: "hubArgs", byteLength: INDIRECT_ARGS_STRIDE, usage: STORAGE_RW | BufferUsage.INDIRECT, zero: true },
|
|
242
239
|
{ name: "pyramid", byteLength: gridPyramidBytes(spec), usage: STORAGE_RW, zero: true },
|
|
243
240
|
];
|
|
244
241
|
}
|
|
@@ -261,7 +258,6 @@ export class RepulsionGrid {
|
|
|
261
258
|
kernelSpec("histogram"),
|
|
262
259
|
kernelSpec("fill"),
|
|
263
260
|
kernelSpec("grid-centroid"),
|
|
264
|
-
kernelSpec("indirect-finalize"),
|
|
265
261
|
kernelSpec("grid-centroid-hub"),
|
|
266
262
|
kernelSpec("grid-downsample"),
|
|
267
263
|
farFieldSpec(overrides),
|
|
@@ -348,7 +344,6 @@ export class RepulsionGrid {
|
|
|
348
344
|
pyramid: r.pyramid,
|
|
349
345
|
hubList: r.hubList,
|
|
350
346
|
hubCounters: r.hubCounters,
|
|
351
|
-
hubArgs: r.hubArgs,
|
|
352
347
|
});
|
|
353
348
|
this.bound = {
|
|
354
349
|
far: this.far.bind({
|
|
@@ -483,7 +483,6 @@ export class SpringElectricalModel implements ForceModel<SpringElectricalOptions
|
|
|
483
483
|
cellStart: resources.buffer("cellStart"),
|
|
484
484
|
hubList: resources.buffer("hubList"),
|
|
485
485
|
hubCounters,
|
|
486
|
-
hubArgs: resources.buffer("hubArgs"),
|
|
487
486
|
pyramid: resources.buffer("pyramid"),
|
|
488
487
|
});
|
|
489
488
|
const wg = k1.workgroupSize;
|
package/src/managed.ts
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The managed accelerator behind `acquireAccelerator` (the `./acquire` subpath, and the same function on `./browser`
|
|
3
|
+
* and `./node`): probe, context, device self-check, accelerator, re-acquisition after device loss, disposal. The
|
|
4
|
+
* runtime-specific half (how to probe and how to open a context) comes from the entry that calls `manageAccelerator`;
|
|
5
|
+
* everything else is here once, so no consumer writes it.
|
|
6
|
+
*
|
|
7
|
+
* Every decline is decided before any work runs: a caller with a CPU implementation runs it and reports the reason.
|
|
8
|
+
* A failure of work that already started on the device is never caught here; it reaches the caller of that work.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { createAccelerator } from "./accelerator.js";
|
|
12
|
+
import type { GpuContext } from "./context.js";
|
|
13
|
+
import { isWebGpuGraphError, WebGpuGraphError } from "./errors.js";
|
|
14
|
+
import { verifyDevice } from "./primitives/verify.js";
|
|
15
|
+
import type { GpuAccelerator } from "./types/accelerator.js";
|
|
16
|
+
import type { DeviceCheck, ProbeResult } from "./types/context.js";
|
|
17
|
+
import type {
|
|
18
|
+
AcceleratorDeclined,
|
|
19
|
+
AcquireAcceleratorOptions,
|
|
20
|
+
AcquireResult,
|
|
21
|
+
ManagedAccelerator,
|
|
22
|
+
} from "./types/managed.js";
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* How one runtime finds a device; supplied by the browser and Node entries.
|
|
26
|
+
* @internal
|
|
27
|
+
*/
|
|
28
|
+
export interface AcceleratorPlatform {
|
|
29
|
+
/** Probes without creating a device; never throws. */
|
|
30
|
+
probe(options: AcquireAcceleratorOptions, rejectSoftware: boolean): Promise<ProbeResult>;
|
|
31
|
+
/** Opens a context on the probed adapter. */
|
|
32
|
+
open(options: AcquireAcceleratorOptions, probe: ProbeResult, rejectSoftware: boolean): Promise<GpuContext>;
|
|
33
|
+
/** What the user can do about E_NO_WEBGPU on this runtime, or null. */
|
|
34
|
+
noWebGpuFix(): string | null;
|
|
35
|
+
/** Test seam: replaces `verifyDevice`. */
|
|
36
|
+
readonly verify?: ((ctx: GpuContext) => Promise<DeviceCheck>) | undefined;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/** The fix of a declined software adapter, on every runtime. */
|
|
40
|
+
const SOFTWARE_FIX = "pass acceptSoftware: true to use the software adapter (usually slower than the CPU)";
|
|
41
|
+
|
|
42
|
+
const DECLINE_CODES: ReadonlySet<string> = new Set(["E_NO_WEBGPU", "E_NO_ADAPTER", "E_SOFTWARE_ONLY"]);
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* A decline record.
|
|
46
|
+
* @param platform - the runtime, for the E_NO_WEBGPU fix
|
|
47
|
+
* @param code - the decline code
|
|
48
|
+
* @param reason - the reason in words
|
|
49
|
+
* @param rest - the adapter and the check, when known
|
|
50
|
+
* @returns the record
|
|
51
|
+
*/
|
|
52
|
+
function declined(
|
|
53
|
+
platform: AcceleratorPlatform,
|
|
54
|
+
code: AcceleratorDeclined["code"],
|
|
55
|
+
reason: string,
|
|
56
|
+
rest: Partial<Pick<AcceleratorDeclined, "adapter" | "check">> = {},
|
|
57
|
+
): AcceleratorDeclined {
|
|
58
|
+
let fix: string | null = null;
|
|
59
|
+
if (code === "E_SOFTWARE_ONLY") {
|
|
60
|
+
fix = SOFTWARE_FIX;
|
|
61
|
+
} else if (code === "E_NO_WEBGPU") {
|
|
62
|
+
fix = platform.noWebGpuFix();
|
|
63
|
+
}
|
|
64
|
+
return { ok: false, code, reason, fix, adapter: rest.adapter ?? null, check: rest.check ?? null };
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* One acquisition: probe, open, self-check, accelerator. The context is disposed on every path that does not hand
|
|
69
|
+
* it over.
|
|
70
|
+
* @param platform - the runtime
|
|
71
|
+
* @param options - the caller's options
|
|
72
|
+
* @returns the accelerator, or why there is none
|
|
73
|
+
*/
|
|
74
|
+
async function acquireOnce(platform: AcceleratorPlatform, options: AcquireAcceleratorOptions): Promise<AcquireResult> {
|
|
75
|
+
const rejectSoftware = options.acceptSoftware !== true;
|
|
76
|
+
const probe = await platform.probe(options, rejectSoftware);
|
|
77
|
+
if (!probe.ok || probe.code !== "OK") {
|
|
78
|
+
const code = probe.code === "OK" ? "E_NO_ADAPTER" : probe.code;
|
|
79
|
+
return declined(platform, code, probe.reason ?? code, { adapter: probe.summary });
|
|
80
|
+
}
|
|
81
|
+
let ctx: GpuContext;
|
|
82
|
+
try {
|
|
83
|
+
ctx = await platform.open(options, probe, rejectSoftware);
|
|
84
|
+
} catch (err) {
|
|
85
|
+
// the adapter can change between the probe and the device request (a GPU process restart)
|
|
86
|
+
if (isWebGpuGraphError(err) && DECLINE_CODES.has(err.code)) {
|
|
87
|
+
return declined(platform, err.code as AcceleratorDeclined["code"], err.message, { adapter: probe.summary });
|
|
88
|
+
}
|
|
89
|
+
throw err;
|
|
90
|
+
}
|
|
91
|
+
let accelerator: GpuAccelerator;
|
|
92
|
+
try {
|
|
93
|
+
const check = await (platform.verify ?? verifyDevice)(ctx);
|
|
94
|
+
if (check.mismatch !== null) {
|
|
95
|
+
ctx.dispose();
|
|
96
|
+
const where = check.mismatch.poison
|
|
97
|
+
? `${check.mismatch.where} was never written`
|
|
98
|
+
: `${check.mismatch.where} came back as ${String(check.mismatch.actual)} where ${String(check.mismatch.expected)} was required`;
|
|
99
|
+
return declined(
|
|
100
|
+
platform,
|
|
101
|
+
"E_DEVICE_INCORRECT",
|
|
102
|
+
`the ${check.vendor} device computed a known prefix sum incorrectly: ${where}`,
|
|
103
|
+
{ adapter: probe.summary, check },
|
|
104
|
+
);
|
|
105
|
+
}
|
|
106
|
+
accelerator = createAccelerator(ctx, options.accelerator);
|
|
107
|
+
} catch (err) {
|
|
108
|
+
ctx.dispose();
|
|
109
|
+
throw err;
|
|
110
|
+
}
|
|
111
|
+
return { ok: true, code: "OK", accelerator };
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* The managed accelerator over one runtime.
|
|
116
|
+
* @param platform - how this runtime probes and opens a context
|
|
117
|
+
* @param options - the caller's options
|
|
118
|
+
* @returns the handle
|
|
119
|
+
* @internal
|
|
120
|
+
*/
|
|
121
|
+
export function manageAccelerator(
|
|
122
|
+
platform: AcceleratorPlatform,
|
|
123
|
+
options: AcquireAcceleratorOptions = {},
|
|
124
|
+
): ManagedAccelerator {
|
|
125
|
+
let pending: Promise<AcquireResult> | null = null;
|
|
126
|
+
let ready: GpuAccelerator | null = null;
|
|
127
|
+
let disposed = false;
|
|
128
|
+
const disposedError = (): WebGpuGraphError =>
|
|
129
|
+
new WebGpuGraphError("E_DISPOSED", "the managed accelerator was disposed", { label: "acquireAccelerator" });
|
|
130
|
+
return {
|
|
131
|
+
current(): Promise<AcquireResult> {
|
|
132
|
+
if (disposed) {
|
|
133
|
+
return Promise.reject(disposedError());
|
|
134
|
+
}
|
|
135
|
+
if (ready !== null && ready.ctx.state !== "ready") {
|
|
136
|
+
// lost, or disposed by its holder: the next answer is a new device
|
|
137
|
+
ready = null;
|
|
138
|
+
pending = null;
|
|
139
|
+
}
|
|
140
|
+
if (pending === null) {
|
|
141
|
+
const attempt = acquireOnce(platform, options).then(
|
|
142
|
+
(result) => {
|
|
143
|
+
if (!result.ok) {
|
|
144
|
+
return result;
|
|
145
|
+
}
|
|
146
|
+
if (disposed) {
|
|
147
|
+
result.accelerator.dispose();
|
|
148
|
+
throw disposedError();
|
|
149
|
+
}
|
|
150
|
+
ready = result.accelerator;
|
|
151
|
+
return result;
|
|
152
|
+
},
|
|
153
|
+
(err: unknown) => {
|
|
154
|
+
// not a decline: the next call tries again rather than remembering a transient failure
|
|
155
|
+
if (pending === attempt) {
|
|
156
|
+
pending = null;
|
|
157
|
+
}
|
|
158
|
+
throw err;
|
|
159
|
+
},
|
|
160
|
+
);
|
|
161
|
+
pending = attempt;
|
|
162
|
+
}
|
|
163
|
+
return pending;
|
|
164
|
+
},
|
|
165
|
+
dispose(): void {
|
|
166
|
+
disposed = true;
|
|
167
|
+
ready?.dispose();
|
|
168
|
+
ready = null;
|
|
169
|
+
pending = null;
|
|
170
|
+
},
|
|
171
|
+
};
|
|
172
|
+
}
|
package/src/node/index.ts
CHANGED
|
@@ -11,7 +11,17 @@
|
|
|
11
11
|
|
|
12
12
|
import { GpuContext } from "../context.js";
|
|
13
13
|
import { WebGpuGraphError } from "../errors.js";
|
|
14
|
+
import { manageAccelerator } from "../managed.js";
|
|
14
15
|
import type { GpuContextOptions, ProbeResult } from "../types/context.js";
|
|
16
|
+
import type { AcquireAcceleratorOptions, ManagedAccelerator } from "../types/managed.js";
|
|
17
|
+
|
|
18
|
+
export type {
|
|
19
|
+
AcceleratorDeclined,
|
|
20
|
+
AcceleratorReady,
|
|
21
|
+
AcquireAcceleratorOptions,
|
|
22
|
+
AcquireResult,
|
|
23
|
+
ManagedAccelerator,
|
|
24
|
+
} from "../types/managed.js";
|
|
15
25
|
|
|
16
26
|
/** Options of the Node helpers (spec 3.4). */
|
|
17
27
|
export interface NodeGpuOptions extends Omit<GpuContextOptions, "gpu" | "adapter" | "device" | "runtime"> {
|
|
@@ -252,3 +262,29 @@ export async function probeNodeWebGpu(options?: NodeGpuOptions): Promise<ProbeRe
|
|
|
252
262
|
handle.dispose();
|
|
253
263
|
}
|
|
254
264
|
}
|
|
265
|
+
|
|
266
|
+
/**
|
|
267
|
+
* The managed accelerator on Dawn: probe, context, device self-check and accelerator on the first `current()`, a
|
|
268
|
+
* new device after a loss, disposal by `dispose()`. Without the optional `webgpu` package `current()` declines with
|
|
269
|
+
* E_NO_WEBGPU and the install command as its fix; a software adapter is declined unless `acceptSoftware` is set.
|
|
270
|
+
* The same function as the `./acquire` subpath resolves to under Node.
|
|
271
|
+
* @param options - see AcquireAcceleratorOptions; `adapter` picks the Dawn adapter by name
|
|
272
|
+
* @returns the handle; nothing is loaded or probed until `current()` is called
|
|
273
|
+
*/
|
|
274
|
+
export function acquireAccelerator(options?: AcquireAcceleratorOptions): ManagedAccelerator {
|
|
275
|
+
return manageAccelerator(
|
|
276
|
+
{
|
|
277
|
+
probe: (o, rejectSoftware) =>
|
|
278
|
+
probeNodeWebGpu({ adapter: o.adapter, powerPreference: o.powerPreference, rejectSoftware }),
|
|
279
|
+
open: (o, _probe, rejectSoftware) =>
|
|
280
|
+
createNodeGpuContext({
|
|
281
|
+
adapter: o.adapter,
|
|
282
|
+
powerPreference: o.powerPreference,
|
|
283
|
+
rejectSoftware,
|
|
284
|
+
warnUnreleasedSnapshots: o.warnUnreleasedSnapshots,
|
|
285
|
+
}),
|
|
286
|
+
noWebGpuFix: () => INSTALL_HINT,
|
|
287
|
+
},
|
|
288
|
+
options,
|
|
289
|
+
);
|
|
290
|
+
}
|