@graphty/webgpu-graph-algorithms 0.6.26 → 0.6.28
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +56 -6
- package/dist/acquire.d.ts +2 -0
- package/dist/browser.js +18 -1
- package/dist/browser.js.map +1 -1
- package/dist/chunks/accelerator-B-FjQwaA.js +19173 -0
- package/dist/chunks/accelerator-B-FjQwaA.js.map +1 -0
- package/dist/chunks/managed-D_GdQtnu.js +98 -0
- package/dist/chunks/managed-D_GdQtnu.js.map +1 -0
- package/dist/node.js +18 -1
- package/dist/node.js.map +1 -1
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +5 -3
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/all-pairs.d.ts.map +1 -1
- package/dist/src/algorithms/all-pairs.js +72 -47
- package/dist/src/algorithms/all-pairs.js.map +1 -1
- package/dist/src/algorithms/betweenness.d.ts +1 -1
- package/dist/src/algorithms/betweenness.js +2 -2
- package/dist/src/algorithms/closeness.d.ts +45 -42
- package/dist/src/algorithms/closeness.d.ts.map +1 -1
- package/dist/src/algorithms/closeness.js +295 -226
- package/dist/src/algorithms/closeness.js.map +1 -1
- package/dist/src/browser/index.d.ts +10 -0
- package/dist/src/browser/index.d.ts.map +1 -1
- package/dist/src/browser/index.js +22 -0
- package/dist/src/browser/index.js.map +1 -1
- package/dist/src/constants.d.ts +13 -3
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +13 -3
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernels.d.ts +14 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +62 -28
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/force-simulation.d.ts.map +1 -1
- package/dist/src/layouts/force-simulation.js +0 -1
- package/dist/src/layouts/force-simulation.js.map +1 -1
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
- package/dist/src/layouts/forceatlas2.js +0 -1
- package/dist/src/layouts/forceatlas2.js.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.js +0 -1
- package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
- package/dist/src/layouts/repulsion-grid.d.ts +1 -3
- package/dist/src/layouts/repulsion-grid.d.ts.map +1 -1
- package/dist/src/layouts/repulsion-grid.js +1 -6
- package/dist/src/layouts/repulsion-grid.js.map +1 -1
- package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
- package/dist/src/layouts/spring-electrical.js +0 -1
- package/dist/src/layouts/spring-electrical.js.map +1 -1
- package/dist/src/managed.d.ts +11 -0
- package/dist/src/managed.d.ts.map +1 -0
- package/dist/src/managed.js +129 -0
- package/dist/src/managed.js.map +1 -0
- package/dist/src/node/index.d.ts +11 -0
- package/dist/src/node/index.d.ts.map +1 -1
- package/dist/src/node/index.js +21 -0
- package/dist/src/node/index.js.map +1 -1
- package/dist/src/primitives/grid-pyramid.d.ts +16 -15
- package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
- package/dist/src/primitives/grid-pyramid.js +20 -28
- package/dist/src/primitives/grid-pyramid.js.map +1 -1
- package/dist/src/types/accelerator.d.ts +2 -0
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/managed.d.ts +81 -0
- package/dist/src/types/managed.d.ts.map +1 -0
- package/dist/src/types/managed.js +7 -0
- package/dist/src/types/managed.js.map +1 -0
- package/dist/src/wgsl/bc-forward.wgsl.d.ts +1 -1
- package/dist/src/wgsl/bc-forward.wgsl.js +1 -1
- package/dist/src/wgsl/closeness-level.wgsl.d.ts +37 -0
- package/dist/src/wgsl/closeness-level.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-level.wgsl.js +204 -0
- package/dist/src/wgsl/closeness-level.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts +11 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.js +42 -0
- package/dist/src/wgsl/closeness-rowsum.wgsl.js.map +1 -0
- package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +2 -2
- package/dist/src/wgsl/grid-centroid-hub.wgsl.js +2 -2
- package/dist/webgpu-graph-algorithms.js +142 -15586
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +10 -4
- package/src/accelerator.ts +5 -3
- package/src/algorithms/all-pairs.ts +86 -56
- package/src/algorithms/betweenness.ts +2 -2
- package/src/algorithms/closeness.ts +353 -256
- package/src/browser/index.ts +37 -0
- package/src/constants.ts +13 -3
- package/src/kernels.ts +65 -36
- package/src/layouts/force-simulation.ts +0 -1
- package/src/layouts/forceatlas2.ts +0 -1
- package/src/layouts/fruchterman-reingold.ts +0 -1
- package/src/layouts/repulsion-grid.ts +2 -7
- package/src/layouts/spring-electrical.ts +0 -1
- package/src/managed.ts +172 -0
- package/src/node/index.ts +36 -0
- package/src/primitives/grid-pyramid.ts +29 -41
- package/src/types/accelerator.ts +2 -0
- package/src/types/managed.ts +86 -0
- package/src/wgsl/bc-forward.wgsl.ts +1 -1
- package/src/wgsl/closeness-level.wgsl.ts +203 -0
- package/src/wgsl/closeness-rowsum.wgsl.ts +41 -0
- package/src/wgsl/grid-centroid-hub.wgsl.ts +2 -2
- package/dist/chunks/context-BZY6SMsM.js +0 -3615
- package/dist/chunks/context-BZY6SMsM.js.map +0 -1
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +0 -20
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +0 -1
- package/dist/src/wgsl/closeness-reduce.wgsl.js +0 -69
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +0 -1
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +0 -22
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +0 -1
- package/dist/src/wgsl/closeness-sweep.wgsl.js +0 -106
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +0 -1
- package/src/wgsl/closeness-reduce.wgsl.ts +0 -68
- package/src/wgsl/closeness-sweep.wgsl.ts +0 -105
|
@@ -1,23 +1,26 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The grid pyramid (spec 6 row 12, 7.7 G4-G5; P4-T9): the planner that records, into the caller's pass, the finest
|
|
3
3
|
* centroids (G4, `grid-centroid`: thread per cell over `cells + outsideCells`, the pseudo-cells included), the hub-cell
|
|
4
|
-
* completion (
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
4
|
+
* completion (G4b: `grid-centroid-hub`, one workgroup per hub cell; PD-13) and one `grid-downsample` dispatch per
|
|
5
|
+
* coarser level (G5). G4b is a DIRECT dispatch of one workgroup per word of `hubList` -- the most hub cells the graph
|
|
6
|
+
* can hold -- and the kernel's `h < hubCount[0]` guard idles the workgroups past the count. It was an indirect
|
|
7
|
+
* dispatch from a finalize-written args slot until issue #732: Dawn validates every indirect dispatch with a hidden
|
|
8
|
+
* pass that cost 0.3 to 0.9 ms of device time per iteration, several times the whole grid tier's own work at 10k
|
|
9
|
+
* nodes (the frontier kernels' lesson, docs/decisions/G8.md G8-F5). The idle workgroups cost far less: `hubList` holds
|
|
10
|
+
* `ceil(n / GRID_HUB_CELL)` words, so 977 workgroups at 1M nodes, each reading one word. Level 0 holds `[sum m x, sum m y, sum m z, sum m]` per cell; every parent is the sum of its 2^dim
|
|
8
11
|
* children; the pseudo-cells (indices `cells ..` of level 0) are never children. No atomics touch the sums (design 6 row 12:
|
|
9
12
|
* bitwise reproducible); the only atomics are the hub append and the occupancy max.
|
|
10
13
|
*
|
|
11
|
-
* The named grid buffers (`pyramid`, `hubList`, `hubCounters
|
|
12
|
-
* `BufferSpec`s, so `inspect(name)` reaches them); the static params of
|
|
14
|
+
* The named grid buffers (`pyramid`, `hubList`, `hubCounters`) are the caller's (the model's
|
|
15
|
+
* `BufferSpec`s, so `inspect(name)` reaches them); the static params of every level are written
|
|
13
16
|
* ONCE at bind() through the scope's params writer (PD-11), so record() writes no uniform. `src/primitives/**` never
|
|
14
17
|
* imports `src/context.ts`.
|
|
15
18
|
*/
|
|
16
19
|
|
|
17
20
|
import { WebGpuGraphError } from "../errors.js";
|
|
18
|
-
import {
|
|
21
|
+
import { plan1d, plan2d } from "../kernel/dispatch.js";
|
|
19
22
|
import { type BoundKernel, type Kernel } from "../kernel/kernel.js";
|
|
20
|
-
import { GRID_LEVEL_PARAMS,
|
|
23
|
+
import { GRID_LEVEL_PARAMS, kernelSpec } from "../kernels.js";
|
|
21
24
|
import { type Binding } from "../types/memory.js";
|
|
22
25
|
import { type GridSpec } from "./grid.js";
|
|
23
26
|
import { type ReduceScope } from "./reduce.js";
|
|
@@ -38,39 +41,37 @@ export interface GridPyramidBindings {
|
|
|
38
41
|
readonly cellStart: Binding;
|
|
39
42
|
/** `pyramidCells` vec4f: every level, level 0 first with the 2^dim orthant pseudo-cells from index `cells`. */
|
|
40
43
|
readonly pyramid: Binding;
|
|
41
|
-
/** The hub cells' indices, appended by G4 (at least one word; at most `floor(n / (GRID_HUB_CELL + 1))` are ever written, so `ceil(n / GRID_HUB_CELL)` words always suffice). */
|
|
44
|
+
/** The hub cells' indices, appended by G4 (at least one word; at most `floor(n / (GRID_HUB_CELL + 1))` are ever written, so `ceil(n / GRID_HUB_CELL)` words always suffice). Its word count is G4b's workgroup count. */
|
|
42
45
|
readonly hubList: Binding;
|
|
43
46
|
/** Two u32 (bound whole, at least 8 bytes): `[0]` the hub count, `[1]` the largest cell occupancy; the caller zeroes both before every build (K1, T10). */
|
|
44
47
|
readonly hubCounters: Binding;
|
|
45
|
-
/** One 16-byte indirect args slot with the INDIRECT usage: G4a writes it, G4b dispatches from it. */
|
|
46
|
-
readonly hubArgs: Binding;
|
|
47
48
|
}
|
|
48
49
|
|
|
49
|
-
/** Where `record()` stops: after level 0 is whole (G4
|
|
50
|
+
/** Where `record()` stops: after level 0 is whole (G4 and G4b) or after every coarser level (G5, the default). */
|
|
50
51
|
export type GridPyramidStage = "G4" | "G5";
|
|
51
52
|
|
|
52
53
|
/** A prepared pyramid build (spec 7.7 G4-G5): binds once per load, records the stages of one iteration into a pass. */
|
|
53
54
|
export interface GridPyramidPlanner {
|
|
54
55
|
/**
|
|
55
|
-
* Binds the named buffers and writes the static params of
|
|
56
|
+
* Binds the named buffers and writes the static params of every level (PD-11); called once
|
|
56
57
|
* per load (a second call rebinds and writes fresh params, so it belongs to a reload, never to an iteration).
|
|
57
58
|
* @param bindings - the buffers
|
|
58
59
|
*/
|
|
59
60
|
bind(bindings: GridPyramidBindings): void;
|
|
60
61
|
/**
|
|
61
|
-
* Records G4,
|
|
62
|
-
* stops after G4b (level 0 is whole:
|
|
62
|
+
* Records G4, G4b and then G5 for every coarser level at the `Fa2Params` slot `paramsOffset`; `upTo: "G4"`
|
|
63
|
+
* stops after G4b (level 0 is whole: G4b is G4's completion).
|
|
63
64
|
* @param pass - the compute pass
|
|
64
65
|
* @param paramsOffset - the dynamic offset of this iteration's `Fa2Params`
|
|
65
66
|
* @param upTo - the last stage to record (default "G5")
|
|
66
67
|
*/
|
|
67
68
|
record(pass: GPUComputePassEncoder, paramsOffset: number, upTo?: GridPyramidStage): void;
|
|
68
|
-
/** Dispatches the last record() issued:
|
|
69
|
+
/** Dispatches the last record() issued: 2 after `upTo: "G4"`, `2 + (levels - 1)` for a full record. */
|
|
69
70
|
readonly lastDispatches: number;
|
|
70
71
|
}
|
|
71
72
|
|
|
72
73
|
/**
|
|
73
|
-
* Prepares the pyramid's pipelines over a scope (G4,
|
|
74
|
+
* Prepares the pyramid's pipelines over a scope (G4, G4b and G5; compiles once) so bind() and record()
|
|
74
75
|
* are synchronous. The planner lives exactly as long as the scope.
|
|
75
76
|
* @param scope - the caller's scope (device, caps, cache, scratch, params)
|
|
76
77
|
* @param spec - the grid
|
|
@@ -78,31 +79,28 @@ export interface GridPyramidPlanner {
|
|
|
78
79
|
*/
|
|
79
80
|
export async function preparePyramid(scope: ReduceScope, spec: GridSpec): Promise<GridPyramidPlanner> {
|
|
80
81
|
const centroid = await scope.pipelines.kernel(kernelSpec("grid-centroid"));
|
|
81
|
-
const finalize = await scope.pipelines.kernel(kernelSpec("indirect-finalize"));
|
|
82
82
|
const hub = await scope.pipelines.kernel(kernelSpec("grid-centroid-hub"));
|
|
83
83
|
const downsample = await scope.pipelines.kernel(kernelSpec("grid-downsample"));
|
|
84
|
-
return new GridPyramidPlannerImpl(scope, spec, { centroid,
|
|
84
|
+
return new GridPyramidPlannerImpl(scope, spec, { centroid, hub, downsample });
|
|
85
85
|
}
|
|
86
86
|
|
|
87
|
-
/** The
|
|
87
|
+
/** The three kernels of the build. */
|
|
88
88
|
interface Kernels {
|
|
89
89
|
readonly centroid: Kernel;
|
|
90
|
-
readonly finalize: Kernel;
|
|
91
90
|
readonly hub: Kernel;
|
|
92
91
|
readonly downsample: Kernel;
|
|
93
92
|
}
|
|
94
93
|
|
|
95
94
|
/** What bind() prepared: the bound groups and, per coarser level, its bound group with its params offset. */
|
|
96
95
|
interface Bound {
|
|
97
|
-
readonly hubArgs: Binding;
|
|
98
96
|
readonly centroid: BoundKernel;
|
|
99
|
-
readonly finalize: BoundKernel;
|
|
100
|
-
readonly finalizeOffset: number;
|
|
101
97
|
readonly hub: BoundKernel;
|
|
98
|
+
/** G4b's workgroups: one per word of `hubList`. */
|
|
99
|
+
readonly hubSlots: number;
|
|
102
100
|
readonly levels: readonly { readonly bound: BoundKernel; readonly offset: number; readonly parentCells: number }[];
|
|
103
101
|
}
|
|
104
102
|
|
|
105
|
-
/** The planner: G4,
|
|
103
|
+
/** The planner: G4, G4b and G5 over one scope. */
|
|
106
104
|
class GridPyramidPlannerImpl implements GridPyramidPlanner {
|
|
107
105
|
private readonly scope: ReduceScope;
|
|
108
106
|
private readonly spec: GridSpec;
|
|
@@ -114,7 +112,7 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
|
|
|
114
112
|
* Wraps the resolved kernels; use preparePyramid().
|
|
115
113
|
* @param scope - the caller's scope
|
|
116
114
|
* @param spec - the grid
|
|
117
|
-
* @param kernels - the
|
|
115
|
+
* @param kernels - the three kernels
|
|
118
116
|
*/
|
|
119
117
|
constructor(scope: ReduceScope, spec: GridSpec, kernels: Kernels) {
|
|
120
118
|
this.scope = scope;
|
|
@@ -135,15 +133,9 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
|
|
|
135
133
|
* @param bindings - the buffers
|
|
136
134
|
*/
|
|
137
135
|
bind(bindings: GridPyramidBindings): void {
|
|
138
|
-
const { centroid,
|
|
136
|
+
const { centroid, hub, downsample } = this.kernels;
|
|
139
137
|
const { spec, scope } = this;
|
|
140
138
|
const b = bindings;
|
|
141
|
-
const finalizeParams = scope.params(INDIRECT_PARAMS, {
|
|
142
|
-
countIndex: 0,
|
|
143
|
-
wg: 1, // the finalize plans ceil(count / wg) workgroups over ITEMS; G4b's item is a hub cell, one workgroup each
|
|
144
|
-
slot: 0,
|
|
145
|
-
pad0: 0,
|
|
146
|
-
});
|
|
147
139
|
const levels: { readonly bound: BoundKernel; readonly offset: number; readonly parentCells: number }[] = [];
|
|
148
140
|
let parentSide = spec.g;
|
|
149
141
|
for (let level = 0; level + 1 < spec.levels; level++) {
|
|
@@ -166,7 +158,6 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
|
|
|
166
158
|
});
|
|
167
159
|
}
|
|
168
160
|
this.bound = {
|
|
169
|
-
hubArgs: b.hubArgs,
|
|
170
161
|
centroid: centroid.bind({
|
|
171
162
|
sortedIdx: b.sortedIdx,
|
|
172
163
|
cellStart: b.cellStart,
|
|
@@ -176,8 +167,6 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
|
|
|
176
167
|
hubCounters: b.hubCounters,
|
|
177
168
|
P: b.params,
|
|
178
169
|
}),
|
|
179
|
-
finalize: finalize.bind({ counters: b.hubCounters, args: b.hubArgs, P: finalizeParams.binding }),
|
|
180
|
-
finalizeOffset: finalizeParams.offset,
|
|
181
170
|
hub: hub.bind({
|
|
182
171
|
sortedIdx: b.sortedIdx,
|
|
183
172
|
cellStart: b.cellStart,
|
|
@@ -187,6 +176,7 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
|
|
|
187
176
|
hubCount: b.hubCounters,
|
|
188
177
|
P: b.params,
|
|
189
178
|
}),
|
|
179
|
+
hubSlots: Math.floor(b.hubList.size / 4),
|
|
190
180
|
levels,
|
|
191
181
|
};
|
|
192
182
|
}
|
|
@@ -202,13 +192,11 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
|
|
|
202
192
|
if (bound === null) {
|
|
203
193
|
throw new WebGpuGraphError("E_NOT_LOADED", "gridPyramid: record() before bind()", { argument: "bind" });
|
|
204
194
|
}
|
|
205
|
-
const { centroid,
|
|
206
|
-
const one: DispatchPlan = { x: 1, y: 1, z: 1, items: 1, stride: null };
|
|
195
|
+
const { centroid, hub, downsample } = this.kernels;
|
|
207
196
|
const level0 = spec.cells + spec.outsideCells;
|
|
208
197
|
centroid.dispatch(pass, bound.centroid, plan1d(level0, scope.workgroupSize, scope.caps), [paramsOffset]);
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
this.dispatches = 3;
|
|
198
|
+
hub.dispatch(pass, bound.hub, plan2d(bound.hubSlots, scope.caps), [paramsOffset]);
|
|
199
|
+
this.dispatches = 2;
|
|
212
200
|
if ((upTo ?? "G5") === "G4") {
|
|
213
201
|
return;
|
|
214
202
|
}
|
package/src/types/accelerator.ts
CHANGED
|
@@ -130,6 +130,8 @@ export interface AcceleratorOptions {
|
|
|
130
130
|
*/
|
|
131
131
|
export interface GpuAccelerator extends AlgorithmAccelerator, LayoutAccelerator {
|
|
132
132
|
readonly kind: "webgpu";
|
|
133
|
+
/** `closenessCentrality` honours `harmonic` on an exact run, so the CPU dispatcher sends harmonic closeness here. */
|
|
134
|
+
readonly harmonicCloseness: true;
|
|
133
135
|
readonly ctx: GpuContext;
|
|
134
136
|
readonly options: Readonly<AcceleratorOptions>;
|
|
135
137
|
forceAtlas2(options?: ForceAtlas2Options): GpuLayoutSimulation<ForceAtlas2Options, ForceAtlas2Stats>;
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The types of `acquireAccelerator` (the `./acquire` subpath, and the same function on `./browser` and `./node`):
|
|
3
|
+
* one call that finds a device, checks that it computes correctly and hands back an accelerator, or says why it
|
|
4
|
+
* did not. Types only.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import type { AcceleratorOptions, GpuAccelerator } from "./accelerator.js";
|
|
8
|
+
import type { AdapterSummary, DeviceCheck } from "./context.js";
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Options of `acquireAccelerator`.
|
|
12
|
+
* @public
|
|
13
|
+
*/
|
|
14
|
+
export interface AcquireAcceleratorOptions {
|
|
15
|
+
/**
|
|
16
|
+
* Accept a software adapter (llvmpipe, SwiftShader, WARP). Default false: a software adapter is usually slower
|
|
17
|
+
* than a CPU implementation, so it is declined with `E_SOFTWARE_ONLY`.
|
|
18
|
+
*/
|
|
19
|
+
readonly acceptSoftware?: boolean | undefined;
|
|
20
|
+
/** The adapter power preference; default "high-performance". */
|
|
21
|
+
readonly powerPreference?: GPUPowerPreference | undefined;
|
|
22
|
+
/** Passed to `createAccelerator` (layout tuning, betweenness defaults). */
|
|
23
|
+
readonly accelerator?: AcceleratorOptions | undefined;
|
|
24
|
+
/** Warn once when more than this many snapshots stay resident on the device (the context's default when absent). */
|
|
25
|
+
readonly warnUnreleasedSnapshots?: number | undefined;
|
|
26
|
+
/** Node only: a substring of the Dawn adapter name to pick (`"llvmpipe"`, `"4070"`). Ignored in a browser. */
|
|
27
|
+
readonly adapter?: string | undefined;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* A verified accelerator.
|
|
32
|
+
* @public
|
|
33
|
+
*/
|
|
34
|
+
export interface AcceleratorReady {
|
|
35
|
+
readonly ok: true;
|
|
36
|
+
readonly code: "OK";
|
|
37
|
+
/** The accelerator. Its `ctx.lost` resolves when the device is lost; the handle then acquires a new one. */
|
|
38
|
+
readonly accelerator: GpuAccelerator;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Why no accelerator was handed over. Every code here is decided before any work runs, so a caller that has a CPU
|
|
43
|
+
* implementation runs it and reports the reason; that is detection, not a fallback.
|
|
44
|
+
* @public
|
|
45
|
+
*/
|
|
46
|
+
export interface AcceleratorDeclined {
|
|
47
|
+
readonly ok: false;
|
|
48
|
+
/**
|
|
49
|
+
* E_NO_WEBGPU: this runtime has no WebGPU (in Node: the optional `webgpu` package is missing);
|
|
50
|
+
* E_NO_ADAPTER: WebGPU exists but no adapter answered; E_SOFTWARE_ONLY: the adapter is a software renderer and
|
|
51
|
+
* `acceptSoftware` was not set; E_DEVICE_INCORRECT: the device got a known answer wrong in the self-check.
|
|
52
|
+
*/
|
|
53
|
+
readonly code: "E_NO_WEBGPU" | "E_NO_ADAPTER" | "E_SOFTWARE_ONLY" | "E_DEVICE_INCORRECT";
|
|
54
|
+
/** The reason in words. */
|
|
55
|
+
readonly reason: string;
|
|
56
|
+
/** What the user can change to get the GPU, or null when nothing they can do would help. */
|
|
57
|
+
readonly fix: string | null;
|
|
58
|
+
/** The adapter that was found and declined, when one was. */
|
|
59
|
+
readonly adapter: AdapterSummary | null;
|
|
60
|
+
/** The self-check record of an E_DEVICE_INCORRECT decline (what disagreed); null otherwise. */
|
|
61
|
+
readonly check: DeviceCheck | null;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* What `ManagedAccelerator.current()` resolves to.
|
|
66
|
+
* @public
|
|
67
|
+
*/
|
|
68
|
+
export type AcquireResult = AcceleratorReady | AcceleratorDeclined;
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* The handle `acquireAccelerator` returns. It owns the device: probing, the device self-check, re-acquiring after
|
|
72
|
+
* device loss, and disposal.
|
|
73
|
+
* @public
|
|
74
|
+
*/
|
|
75
|
+
export interface ManagedAccelerator {
|
|
76
|
+
/**
|
|
77
|
+
* The accelerator, or why there is none. The first call acquires (probe, context, self-check); later calls
|
|
78
|
+
* return the same answer until the device is lost or the accelerator is disposed, after which the next call
|
|
79
|
+
* acquires a new device. Concurrent calls share one acquisition. A decline is remembered for the life of the
|
|
80
|
+
* handle. Rejects only for a failure that is not a decline (a device request that failed, a check that could not
|
|
81
|
+
* run), and with E_DISPOSED after `dispose()`.
|
|
82
|
+
*/
|
|
83
|
+
current(): Promise<AcquireResult>;
|
|
84
|
+
/** Disposes the current accelerator and its device; idempotent. */
|
|
85
|
+
dispose(): void;
|
|
86
|
+
}
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
* The `bc-forward` kernel body (design 8.4 "forward pass = BFS with sigma as array<atomic<u32>>", 8.10 "BC forward
|
|
3
3
|
* (tagged)", 16.1): one level of the tagged multi-source breadth-first search of a betweenness batch. The level's
|
|
4
4
|
* frontier is the range `S[ends[level] .. ends[level + 1])` of the claim log, every entry a packed `s * n + u`; the
|
|
5
|
-
* expansion is `
|
|
5
|
+
* expansion is `advance-expand`'s block-mapped strip (each workgroup loads up to `WG` entries, scans their degrees
|
|
6
6
|
* in workgroup memory, and every lane strips the aggregate by an upper-bound binary search), fused with the claim, so
|
|
7
7
|
* no edge queue exists.
|
|
8
8
|
*
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `closeness-level` kernel body: one level of the bit-parallel multi-source breadth-first search of closeness, in
|
|
3
|
+
* ONE dispatch, plus the two seed roles of a batch. A batch runs `32 x P.words` sources at once: bit `b` of word `j`
|
|
4
|
+
* of node `v` is source lane `32 j + b`. The `bits` buffer holds five regions of `P.base` words, `P.words` words per
|
|
5
|
+
* node: `visited` at 0, three frontier regions at `P.base`, `2 P.base` and `3 P.base` that rotate by the level
|
|
6
|
+
* (level `L` reads region `1 + L % 3`, writes region `1 + (L + 1) % 3` and zeroes region `1 + (L + 2) % 3`, the
|
|
7
|
+
* frontier of level `L - 1` that nothing reads any more, so no fill runs between levels; a stale bit could never
|
|
8
|
+
* claim anything, since its node's neighbours were claimed the level after, so the clear only keeps a later frontier
|
|
9
|
+
* from re-walking old nodes), then a sampled run's
|
|
10
|
+
* per-node distance sums at `4 P.base`. The `table` buffer holds the per-level claim counts of the submit (row `P.row`
|
|
11
|
+
* at `P.row x 32 P.words`, one word per source lane), then a ring of three control slots of four words at `P.ctrl`
|
|
12
|
+
* (`any` @0: the level claimed something; `arcs` @1: the out-degree of every (node, word) that joined the next
|
|
13
|
+
* frontier; `pull` @2: `arcs` passed `P.pullAt`), then a sampled run's source list at `P.sourcesAt`.
|
|
14
|
+
*
|
|
15
|
+
* Role 0 (one invocation per node and per arc, `max(n, arcs)` in all): a level whose predecessor claimed nothing
|
|
16
|
+
* returns at once (one uniform load per workgroup), so the host can record more levels than the batch needs.
|
|
17
|
+
* Otherwise the level chooses its step from the slot its predecessor filled, the same way for every invocation:
|
|
18
|
+
* - PUSH (the frontier is cheap to expand): one invocation per out-arc `(u, x)` -- `u` found by a binary search of
|
|
19
|
+
* `rowPtr` -- claims the sources at `u` that `x` has not seen with `atomicOr` on `x`'s visited word; the bits it
|
|
20
|
+
* won (`fresh`) join the next frontier;
|
|
21
|
+
* - PULL (the frontier's arcs pass `P.pullAt`, and `P.pullOk`): one invocation per node not yet reached by every
|
|
22
|
+
* source walks its IN-arcs, ORs the neighbours' frontier words, and keeps the bits it had not seen; it stops at
|
|
23
|
+
* the first in-arc after which every source has reached it (the early exit). Only the owner writes a node's
|
|
24
|
+
* words. The host allows the pull only when no node has more than a few thousand in-arcs, so no
|
|
25
|
+
* invocation of either step loops more than a few tens of thousands of times (llvmpipe silently ends every loop
|
|
26
|
+
* of an invocation past 65,535 iterations).
|
|
27
|
+
* Every claimed bit is tallied per source lane in workgroup memory and flushed with one global `atomicAdd` per lane
|
|
28
|
+
* per workgroup into the level's row; the host turns the counts into exact sums (`count x (L + 1)`) and harmonic
|
|
29
|
+
* sums (`count / (L + 1)`). Role 1 (one invocation per word): zeroes the regions, sets every dead lane of a partial
|
|
30
|
+
* batch as already visited (so the early exit and the "reached by every source" test see a full word), and zeroes the
|
|
31
|
+
* control ring. Role 2 (one invocation per lane): seeds lane `b`'s source -- `P.source + b`, or word `P.source + b` of
|
|
32
|
+
* the source list -- into `visited` and level 0's frontier with `atomicOr` (a node listed twice carries both bits),
|
|
33
|
+
* and fills the control slot level 0 reads. Body only; the text is normative: the sabotage rows of
|
|
34
|
+
* test/helpers/sabotage.ts are textual edits of it.
|
|
35
|
+
*/
|
|
36
|
+
export const closenessLevelWgsl = /* wgsl */ `
|
|
37
|
+
const max_words: u32 = 8u;
|
|
38
|
+
var<workgroup> tally: array<atomic<u32>, 32u * max_words>; // 32 x max_words: this workgroup's claims per source lane
|
|
39
|
+
var<workgroup> wlive: u32;
|
|
40
|
+
var<workgroup> wpull: u32;
|
|
41
|
+
var<workgroup> wany: atomic<u32>;
|
|
42
|
+
var<workgroup> warcs: atomic<u32>;
|
|
43
|
+
var<workgroup> wover: atomic<u32>;
|
|
44
|
+
|
|
45
|
+
fn dead_lanes(j: u32) -> u32 { // the lanes of word j at or past the batch's source count
|
|
46
|
+
let first = 32u * j;
|
|
47
|
+
if (P.count >= first + 32u) { return 0u; }
|
|
48
|
+
if (P.count <= first) { return U32_MAX; }
|
|
49
|
+
return ~((1u << (P.count - first)) - 1u);
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
fn add_arcs(slot: u32, value: u32) { // the arcs total of a control slot; pull once it passes P.pullAt
|
|
53
|
+
if (value == 0u) { return; }
|
|
54
|
+
let before = atomicAdd(&table[slot + 1u], value);
|
|
55
|
+
if (value > P.pullAt || before + value > P.pullAt || before + value < before) {
|
|
56
|
+
atomicStore(&table[slot + 2u], 1u);
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
fn record(j: u32, node: u32, fresh: u32, dist: u32) { // the claims of one word: tallied per lane, a sampled run's sums
|
|
61
|
+
if (P.perNode == 1u) { atomicAdd(&bits[4u * P.base + node], countOneBits(fresh) * dist); }
|
|
62
|
+
var b = fresh;
|
|
63
|
+
loop {
|
|
64
|
+
if (b == 0u) { break; }
|
|
65
|
+
atomicAdd(&tally[32u * j + firstTrailingBit(b)], 1u);
|
|
66
|
+
b = b & (b - 1u);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
@compute @workgroup_size(WG)
|
|
71
|
+
fn closeness_level(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
72
|
+
let i = linear_id(wid, lid.x);
|
|
73
|
+
let W = P.words;
|
|
74
|
+
if (P.role == 1u) { // seed, part 1: one invocation per word
|
|
75
|
+
if (i == 0u) {
|
|
76
|
+
for (var k = 0u; k < 12u; k = k + 1u) { atomicStore(&table[P.ctrl + k], 0u); }
|
|
77
|
+
}
|
|
78
|
+
if (i < P.total) { atomicStore(&bits[i], select(0u, dead_lanes(i % W), i < P.base)); }
|
|
79
|
+
return; // uniform: P.role is
|
|
80
|
+
}
|
|
81
|
+
if (P.role == 2u) { // seed, part 2: one invocation per source lane
|
|
82
|
+
if (i < P.count) {
|
|
83
|
+
var v = P.source + i;
|
|
84
|
+
if (P.sourcesAt != 0u) { v = atomicLoad(&table[P.sourcesAt + P.source + i]); }
|
|
85
|
+
let w = v * W + i / 32u;
|
|
86
|
+
let bit = 1u << (i % 32u);
|
|
87
|
+
atomicOr(&bits[w], bit); // visited
|
|
88
|
+
let before = atomicOr(&bits[P.base + w], bit); // level 0's frontier (region 1)
|
|
89
|
+
let slot = P.ctrl + 8u; // the slot of "level -1", which level 0 reads
|
|
90
|
+
atomicStore(&table[slot], 1u);
|
|
91
|
+
if (before == 0u) { add_arcs(slot, rowPtr[v + 1u] - rowPtr[v]); }
|
|
92
|
+
}
|
|
93
|
+
return;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
// role 0: one level
|
|
97
|
+
let L = P.level;
|
|
98
|
+
let prev = P.ctrl + 4u * ((L + 2u) % 3u);
|
|
99
|
+
let cur = P.ctrl + 4u * (L % 3u);
|
|
100
|
+
if (lid.x == 0u) {
|
|
101
|
+
wlive = atomicLoad(&table[prev]);
|
|
102
|
+
wpull = select(0u, atomicLoad(&table[prev + 2u]), P.pullOk == 1u);
|
|
103
|
+
atomicStore(&wany, 0u);
|
|
104
|
+
atomicStore(&warcs, 0u);
|
|
105
|
+
atomicStore(&wover, 0u);
|
|
106
|
+
}
|
|
107
|
+
for (var k = lid.x; k < 32u * W; k = k + WG) { atomicStore(&tally[k], 0u); }
|
|
108
|
+
let live = workgroupUniformLoad(&wlive); // uniform; includes a barrier
|
|
109
|
+
if (live == 0u) { return; } // the previous level claimed nothing
|
|
110
|
+
let pull = workgroupUniformLoad(&wpull) == 1u;
|
|
111
|
+
if (i == 0u) { // the slot of level L + 1 held level L - 2's
|
|
112
|
+
let nxt = P.ctrl + 4u * ((L + 1u) % 3u);
|
|
113
|
+
atomicStore(&table[nxt], 0u);
|
|
114
|
+
atomicStore(&table[nxt + 1u], 0u);
|
|
115
|
+
atomicStore(&table[nxt + 2u], 0u);
|
|
116
|
+
}
|
|
117
|
+
let frontierBase = P.base * (1u + L % 3u);
|
|
118
|
+
let nextBase = P.base * (1u + (L + 1u) % 3u);
|
|
119
|
+
let staleBase = P.base * (1u + (L + 2u) % 3u);
|
|
120
|
+
let dist = L + 1u;
|
|
121
|
+
var arcs = 0u;
|
|
122
|
+
if (i < P.n) {
|
|
123
|
+
for (var j = 0u; j < W; j = j + 1u) { atomicStore(&bits[staleBase + i * W + j], 0u); }
|
|
124
|
+
}
|
|
125
|
+
if (pull) {
|
|
126
|
+
if (i < P.n) { // one invocation per node: its in-arcs
|
|
127
|
+
var vis: array<u32, max_words>;
|
|
128
|
+
var unseen = 0u;
|
|
129
|
+
for (var j = 0u; j < W; j = j + 1u) {
|
|
130
|
+
vis[j] = atomicLoad(&bits[i * W + j]);
|
|
131
|
+
unseen = unseen | ~vis[j];
|
|
132
|
+
}
|
|
133
|
+
if (unseen != 0u) { // some source has not reached this node yet
|
|
134
|
+
var acc: array<u32, max_words>;
|
|
135
|
+
let end = inRowPtr[i + 1u];
|
|
136
|
+
for (var a = inRowPtr[i]; a < end; a = a + 1u) {
|
|
137
|
+
let u = inColIdx[a];
|
|
138
|
+
var missing = 0u;
|
|
139
|
+
for (var j = 0u; j < W; j = j + 1u) {
|
|
140
|
+
acc[j] = acc[j] | atomicLoad(&bits[frontierBase + u * W + j]);
|
|
141
|
+
missing = missing | ~(acc[j] | vis[j]);
|
|
142
|
+
}
|
|
143
|
+
if (missing == 0u) { break; } // every source has reached it: the early exit
|
|
144
|
+
}
|
|
145
|
+
let degree = rowPtr[i + 1u] - rowPtr[i];
|
|
146
|
+
for (var j = 0u; j < W; j = j + 1u) {
|
|
147
|
+
let fresh = acc[j] & ~vis[j];
|
|
148
|
+
if (fresh != 0u) {
|
|
149
|
+
atomicStore(&bits[i * W + j], vis[j] | fresh);
|
|
150
|
+
atomicStore(&bits[nextBase + i * W + j], fresh);
|
|
151
|
+
arcs = arcs + degree;
|
|
152
|
+
record(j, i, fresh, dist);
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
} else if (i < P.arcCount) { // one invocation per arc: no row is walked whole
|
|
158
|
+
var lo = 0u; // the arc's source: the last row starting at or before it
|
|
159
|
+
var hi = P.n - 1u;
|
|
160
|
+
loop {
|
|
161
|
+
if (lo >= hi) { break; }
|
|
162
|
+
let mid = (lo + hi + 1u) / 2u;
|
|
163
|
+
if (rowPtr[mid] <= i) { lo = mid; } else { hi = mid - 1u; }
|
|
164
|
+
}
|
|
165
|
+
let u = lo;
|
|
166
|
+
let x = colIdx[i];
|
|
167
|
+
for (var j = 0u; j < W; j = j + 1u) {
|
|
168
|
+
let mask = atomicLoad(&bits[frontierBase + u * W + j]) & ~atomicLoad(&bits[x * W + j]); // at u, not yet at x
|
|
169
|
+
if (mask == 0u) { continue; }
|
|
170
|
+
let fresh = mask & ~atomicOr(&bits[x * W + j], mask); // the claims this invocation won
|
|
171
|
+
if (fresh == 0u) { continue; }
|
|
172
|
+
if (atomicOr(&bits[nextBase + x * W + j], fresh) == 0u) {
|
|
173
|
+
arcs = arcs + (rowPtr[x + 1u] - rowPtr[x]);
|
|
174
|
+
}
|
|
175
|
+
record(j, x, fresh, dist);
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
if (arcs > P.pullAt) {
|
|
179
|
+
atomicStore(&wover, 1u);
|
|
180
|
+
} else if (arcs != 0u) {
|
|
181
|
+
let before = atomicAdd(&warcs, arcs);
|
|
182
|
+
if (before + arcs > P.pullAt || before + arcs < before) { atomicStore(&wover, 1u); }
|
|
183
|
+
}
|
|
184
|
+
workgroupBarrier();
|
|
185
|
+
let row = P.row * 32u * W;
|
|
186
|
+
for (var k = lid.x; k < 32u * W; k = k + WG) { // ONE global atomic per source lane per workgroup
|
|
187
|
+
let c = atomicLoad(&tally[k]);
|
|
188
|
+
if (c != 0u) {
|
|
189
|
+
atomicAdd(&table[row + k], c);
|
|
190
|
+
atomicStore(&wany, 1u);
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
workgroupBarrier();
|
|
194
|
+
if (lid.x == 0u) {
|
|
195
|
+
if (atomicLoad(&wany) != 0u) { atomicStore(&table[cur], 1u); }
|
|
196
|
+
if (atomicLoad(&wover) != 0u) {
|
|
197
|
+
atomicStore(&table[cur + 2u], 1u);
|
|
198
|
+
} else {
|
|
199
|
+
add_arcs(cur, atomicLoad(&warcs));
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
`;
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `closeness-rowsum` kernel body: closeness from a finished all-pairs distance matrix, one workgroup per row.
|
|
3
|
+
* Every lane walks its strided columns of row `r` (the distances FROM `r`), skips the diagonal and every unreachable
|
|
4
|
+
* entry (`+Infinity`, compared by its bit pattern because WGSL lets a compiler assume no infinities), and adds, by
|
|
5
|
+
* `P.role`: 0 the hop count as an integer (exact: a row of at most 23,170 hops below 23,170 sums below 2^32), 1 the
|
|
6
|
+
* f32 distance, 2 its reciprocal (harmonic closeness; a zero distance adds nothing, as in the CPU port). A tree reduction in workgroup memory folds the lanes and lane
|
|
7
|
+
* 0 writes `out[r]` -- the integer, or the f32 bit pattern. The host turns the row into the score. Body only; the text
|
|
8
|
+
* is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
9
|
+
*/
|
|
10
|
+
export const closenessRowsumWgsl = /* wgsl */ `
|
|
11
|
+
var<workgroup> partial: array<u32, WG>;
|
|
12
|
+
|
|
13
|
+
@compute @workgroup_size(WG)
|
|
14
|
+
fn closeness_rowsum(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
15
|
+
let row = group_id(wid); // uniform: one workgroup per row
|
|
16
|
+
if (row >= P.n) { return; }
|
|
17
|
+
var whole = 0u;
|
|
18
|
+
var real = 0.0;
|
|
19
|
+
for (var c = lid.x; c < P.n; c = c + WG) {
|
|
20
|
+
let d = dist[row * P.n + c];
|
|
21
|
+
if (c == row || bitcast<u32>(d) == F32_INF_BITS) { continue; }
|
|
22
|
+
if (P.role == 0u) {
|
|
23
|
+
whole = whole + u32(d);
|
|
24
|
+
} else if (P.role == 1u) {
|
|
25
|
+
real = real + d;
|
|
26
|
+
} else if (d > 0.0) {
|
|
27
|
+
real = real + 1.0 / d; // a zero distance adds nothing, as on the CPU
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
partial[lid.x] = select(whole, bitcast<u32>(real), P.role != 0u);
|
|
31
|
+
for (var s = WG / 2u; s > 0u; s = s / 2u) {
|
|
32
|
+
workgroupBarrier();
|
|
33
|
+
if (lid.x < s) {
|
|
34
|
+
let a = partial[lid.x];
|
|
35
|
+
let b = partial[lid.x + s];
|
|
36
|
+
partial[lid.x] = select(a + b, bitcast<u32>(bitcast<f32>(a) + bitcast<f32>(b)), P.role != 0u);
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
if (lid.x == 0u) { out[row] = partial[0]; }
|
|
40
|
+
}
|
|
41
|
+
`;
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per
|
|
3
|
-
*
|
|
2
|
+
* G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per word of hubList, dispatched
|
|
3
|
+
* directly (issue #732), so the workgroups past hubCounters[0] idle; a WG-strided mass-weighted sum reduced by the
|
|
4
4
|
* prelude's tree. The work is guarded by `valid`, never an early return, so the reduction is uniform (PD-13). Body
|
|
5
5
|
* only; normative text.
|
|
6
6
|
*/
|