@graphty/webgpu-graph-algorithms 0.5.1 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +98 -52
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-BR7fx3vR.js → context-BXqgCifx.js} +190 -40
- package/dist/chunks/context-BXqgCifx.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/algorithms/components.d.ts.map +1 -1
- package/dist/src/algorithms/components.js +12 -13
- package/dist/src/algorithms/components.js.map +1 -1
- package/dist/src/algorithms/degree.d.ts +6 -8
- package/dist/src/algorithms/degree.d.ts.map +1 -1
- package/dist/src/algorithms/degree.js +58 -35
- package/dist/src/algorithms/degree.js.map +1 -1
- package/dist/src/algorithms/pagerank.d.ts.map +1 -1
- package/dist/src/algorithms/pagerank.js +16 -14
- package/dist/src/algorithms/pagerank.js.map +1 -1
- package/dist/src/algorithms/power-iteration.d.ts +2 -2
- package/dist/src/algorithms/power-iteration.d.ts.map +1 -1
- package/dist/src/algorithms/power-iteration.js +17 -14
- package/dist/src/algorithms/power-iteration.js.map +1 -1
- package/dist/src/constants.d.ts +38 -8
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +38 -8
- package/dist/src/constants.js.map +1 -1
- package/dist/src/errors.d.ts +3 -2
- package/dist/src/errors.d.ts.map +1 -1
- package/dist/src/errors.js +2 -1
- package/dist/src/errors.js.map +1 -1
- package/dist/src/index.d.ts +6 -4
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +8 -3
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/dispatch.d.ts +8 -3
- package/dist/src/kernel/dispatch.d.ts.map +1 -1
- package/dist/src/kernel/dispatch.js +18 -7
- package/dist/src/kernel/dispatch.js.map +1 -1
- package/dist/src/kernel/kernel.d.ts +30 -1
- package/dist/src/kernel/kernel.d.ts.map +1 -1
- package/dist/src/kernel/kernel.js +49 -5
- package/dist/src/kernel/kernel.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +6 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernel/profiler.d.ts +15 -3
- package/dist/src/kernel/profiler.d.ts.map +1 -1
- package/dist/src/kernel/profiler.js +27 -4
- package/dist/src/kernel/profiler.js.map +1 -1
- package/dist/src/kernels.d.ts +17 -7
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +323 -16
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/calibrate.d.ts +51 -0
- package/dist/src/layouts/calibrate.d.ts.map +1 -0
- package/dist/src/layouts/calibrate.js +172 -0
- package/dist/src/layouts/calibrate.js.map +1 -0
- package/dist/src/layouts/force-simulation.d.ts +39 -4
- package/dist/src/layouts/force-simulation.d.ts.map +1 -1
- package/dist/src/layouts/force-simulation.js +71 -19
- package/dist/src/layouts/force-simulation.js.map +1 -1
- package/dist/src/layouts/forceatlas2.d.ts +107 -36
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
- package/dist/src/layouts/forceatlas2.js +296 -100
- package/dist/src/layouts/forceatlas2.js.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.d.ts +73 -27
- package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.js +230 -70
- package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
- package/dist/src/layouts/model-common.d.ts +41 -3
- package/dist/src/layouts/model-common.d.ts.map +1 -1
- package/dist/src/layouts/model-common.js +74 -3
- package/dist/src/layouts/model-common.js.map +1 -1
- package/dist/src/layouts/repulsion-grid.d.ts +152 -0
- package/dist/src/layouts/repulsion-grid.d.ts.map +1 -0
- package/dist/src/layouts/repulsion-grid.js +318 -0
- package/dist/src/layouts/repulsion-grid.js.map +1 -0
- package/dist/src/layouts/spring-electrical.d.ts +75 -30
- package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
- package/dist/src/layouts/spring-electrical.js +231 -74
- package/dist/src/layouts/spring-electrical.js.map +1 -1
- package/dist/src/memory/residency.d.ts +6 -2
- package/dist/src/memory/residency.d.ts.map +1 -1
- package/dist/src/memory/residency.js +84 -14
- package/dist/src/memory/residency.js.map +1 -1
- package/dist/src/primitives/core-shape.d.ts +38 -2
- package/dist/src/primitives/core-shape.d.ts.map +1 -1
- package/dist/src/primitives/core-shape.js +71 -3
- package/dist/src/primitives/core-shape.js.map +1 -1
- package/dist/src/primitives/grid-pyramid.d.ts +71 -0
- package/dist/src/primitives/grid-pyramid.d.ts.map +1 -0
- package/dist/src/primitives/grid-pyramid.js +143 -0
- package/dist/src/primitives/grid-pyramid.js.map +1 -0
- package/dist/src/primitives/grid.d.ts +118 -0
- package/dist/src/primitives/grid.d.ts.map +1 -0
- package/dist/src/primitives/grid.js +225 -0
- package/dist/src/primitives/grid.js.map +1 -0
- package/dist/src/primitives/histogram.d.ts +67 -0
- package/dist/src/primitives/histogram.d.ts.map +1 -0
- package/dist/src/primitives/histogram.js +190 -0
- package/dist/src/primitives/histogram.js.map +1 -0
- package/dist/src/primitives/radix-sort.d.ts +75 -0
- package/dist/src/primitives/radix-sort.d.ts.map +1 -0
- package/dist/src/primitives/radix-sort.js +168 -0
- package/dist/src/primitives/radix-sort.js.map +1 -0
- package/dist/src/primitives/scan.d.ts +44 -0
- package/dist/src/primitives/scan.d.ts.map +1 -0
- package/dist/src/primitives/scan.js +151 -0
- package/dist/src/primitives/scan.js.map +1 -0
- package/dist/src/primitives/segmented-reduce.d.ts +25 -17
- package/dist/src/primitives/segmented-reduce.d.ts.map +1 -1
- package/dist/src/primitives/segmented-reduce.js +166 -47
- package/dist/src/primitives/segmented-reduce.js.map +1 -1
- package/dist/src/primitives/spmv.d.ts +18 -14
- package/dist/src/primitives/spmv.d.ts.map +1 -1
- package/dist/src/primitives/spmv.js +94 -58
- package/dist/src/primitives/spmv.js.map +1 -1
- package/dist/src/primitives/verify.d.ts +49 -0
- package/dist/src/primitives/verify.d.ts.map +1 -0
- package/dist/src/primitives/verify.js +229 -0
- package/dist/src/primitives/verify.js.map +1 -0
- package/dist/src/types/context.d.ts +53 -0
- package/dist/src/types/context.d.ts.map +1 -1
- package/dist/src/types/layout.d.ts +20 -0
- package/dist/src/types/layout.d.ts.map +1 -1
- package/dist/src/wgsl/counting-scatter.wgsl.d.ts +8 -0
- package/dist/src/wgsl/counting-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/counting-scatter.wgsl.js +17 -0
- package/dist/src/wgsl/counting-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-attraction.wgsl.d.ts +23 -11
- package/dist/src/wgsl/fa2-attraction.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-attraction.wgsl.js +98 -20
- package/dist/src/wgsl/fa2-attraction.wgsl.js.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +6 -2
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +22 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +8 -0
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/grid-cell-key.wgsl.js +30 -0
- package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -0
- package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +8 -0
- package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/grid-centroid-hub.wgsl.js +29 -0
- package/dist/src/wgsl/grid-centroid-hub.wgsl.js.map +1 -0
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts +8 -0
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/grid-centroid.wgsl.js +29 -0
- package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -0
- package/dist/src/wgsl/grid-downsample.wgsl.d.ts +7 -0
- package/dist/src/wgsl/grid-downsample.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/grid-downsample.wgsl.js +28 -0
- package/dist/src/wgsl/grid-downsample.wgsl.js.map +1 -0
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts +13 -0
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/grid-far-field.wgsl.js +98 -0
- package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -0
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts +19 -0
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/grid-near-field.wgsl.js +129 -0
- package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -0
- package/dist/src/wgsl/histogram.wgsl.d.ts +7 -0
- package/dist/src/wgsl/histogram.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/histogram.wgsl.js +15 -0
- package/dist/src/wgsl/histogram.wgsl.js.map +1 -0
- package/dist/src/wgsl/indirect-finalize.wgsl.d.ts +8 -0
- package/dist/src/wgsl/indirect-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/indirect-finalize.wgsl.js +26 -0
- package/dist/src/wgsl/indirect-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/radix-hist.wgsl.d.ts +9 -0
- package/dist/src/wgsl/radix-hist.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/radix-hist.wgsl.js +31 -0
- package/dist/src/wgsl/radix-hist.wgsl.js.map +1 -0
- package/dist/src/wgsl/radix-scatter.wgsl.d.ts +9 -0
- package/dist/src/wgsl/radix-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/radix-scatter.wgsl.js +40 -0
- package/dist/src/wgsl/radix-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/scan-add.wgsl.d.ts +6 -0
- package/dist/src/wgsl/scan-add.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/scan-add.wgsl.js +14 -0
- package/dist/src/wgsl/scan-add.wgsl.js.map +1 -0
- package/dist/src/wgsl/scan-block.wgsl.d.ts +8 -0
- package/dist/src/wgsl/scan-block.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/scan-block.wgsl.js +30 -0
- package/dist/src/wgsl/scan-block.wgsl.js.map +1 -0
- package/dist/src/wgsl/segmented-reduce.wgsl.d.ts +22 -8
- package/dist/src/wgsl/segmented-reduce.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/segmented-reduce.wgsl.js +84 -15
- package/dist/src/wgsl/segmented-reduce.wgsl.js.map +1 -1
- package/dist/src/wgsl/spmv-pull.wgsl.d.ts +22 -11
- package/dist/src/wgsl/spmv-pull.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/spmv-pull.wgsl.js +110 -36
- package/dist/src/wgsl/spmv-pull.wgsl.js.map +1 -1
- package/dist/tsconfig.build.tsbuildinfo +1 -1
- package/dist/webgpu-graph-algorithms.js +3815 -1003
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +5 -4
- package/src/algorithms/components.ts +12 -16
- package/src/algorithms/degree.ts +58 -43
- package/src/algorithms/pagerank.ts +20 -18
- package/src/algorithms/power-iteration.ts +19 -18
- package/src/constants.ts +38 -8
- package/src/errors.ts +3 -1
- package/src/index.ts +14 -4
- package/src/kernel/dispatch.ts +18 -7
- package/src/kernel/kernel.ts +59 -5
- package/src/kernel/prelude.ts +9 -0
- package/src/kernel/profiler.ts +28 -4
- package/src/kernels.ts +356 -18
- package/src/layouts/calibrate.ts +187 -0
- package/src/layouts/force-simulation.ts +91 -23
- package/src/layouts/forceatlas2.ts +331 -106
- package/src/layouts/fruchterman-reingold.ts +255 -74
- package/src/layouts/model-common.ts +98 -3
- package/src/layouts/repulsion-grid.ts +451 -0
- package/src/layouts/spring-electrical.ts +257 -78
- package/src/memory/residency.ts +126 -20
- package/src/primitives/core-shape.ts +91 -4
- package/src/primitives/grid-pyramid.ts +221 -0
- package/src/primitives/grid.ts +349 -0
- package/src/primitives/histogram.ts +273 -0
- package/src/primitives/radix-sort.ts +246 -0
- package/src/primitives/scan.ts +197 -0
- package/src/primitives/segmented-reduce.ts +214 -56
- package/src/primitives/spmv.ts +125 -65
- package/src/primitives/verify.ts +249 -0
- package/src/types/context.ts +56 -0
- package/src/types/layout.ts +22 -0
- package/src/wgsl/counting-scatter.wgsl.ts +16 -0
- package/src/wgsl/fa2-attraction.wgsl.ts +98 -20
- package/src/wgsl/fa2-stats-finalize.wgsl.ts +22 -1
- package/src/wgsl/grid-cell-key.wgsl.ts +29 -0
- package/src/wgsl/grid-centroid-hub.wgsl.ts +28 -0
- package/src/wgsl/grid-centroid.wgsl.ts +28 -0
- package/src/wgsl/grid-downsample.wgsl.ts +27 -0
- package/src/wgsl/grid-far-field.wgsl.ts +97 -0
- package/src/wgsl/grid-near-field.wgsl.ts +128 -0
- package/src/wgsl/histogram.wgsl.ts +14 -0
- package/src/wgsl/indirect-finalize.wgsl.ts +25 -0
- package/src/wgsl/radix-hist.wgsl.ts +30 -0
- package/src/wgsl/radix-scatter.wgsl.ts +39 -0
- package/src/wgsl/scan-add.wgsl.ts +13 -0
- package/src/wgsl/scan-block.wgsl.ts +29 -0
- package/src/wgsl/segmented-reduce.wgsl.ts +84 -15
- package/src/wgsl/spmv-pull.wgsl.ts +110 -36
- package/dist/chunks/context-BR7fx3vR.js.map +0 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@graphty/webgpu-graph-algorithms",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.6.0",
|
|
4
4
|
"description": "WebGPU-accelerated graph algorithms and layouts over the @graphty/graph-format snapshot, for Node (Dawn) and browsers",
|
|
5
5
|
"author": "Adam Powers <apowers@ato.ms>",
|
|
6
6
|
"type": "module",
|
|
@@ -105,12 +105,13 @@
|
|
|
105
105
|
"typecheck:strict-consumer": "tsc -p tsconfig.strict-consumer.json",
|
|
106
106
|
"test": "vitest",
|
|
107
107
|
"test:ui": "vitest --ui",
|
|
108
|
-
"test:run": "vitest run --project=node",
|
|
109
|
-
"test:node": "vitest run --project=node",
|
|
108
|
+
"test:run": "vitest run --project=node --project=node-device-errors",
|
|
109
|
+
"test:node": "vitest run --project=node --project=node-device-errors",
|
|
110
|
+
"test:node:ci": "node scripts/run-node-shard.js",
|
|
110
111
|
"test:browser": "vitest run --project=browser",
|
|
111
112
|
"test:browser:ci": "node scripts/run-browser-project.js",
|
|
112
113
|
"test:limits": "vitest run --project=node-limits",
|
|
113
|
-
"coverage": "vitest run --project=node --coverage",
|
|
114
|
+
"coverage": "vitest run --project=node --project=node-device-errors --coverage",
|
|
114
115
|
"coverage:preview": "npx serve coverage -p 9058",
|
|
115
116
|
"bench": "tsx benchmarks/run.ts",
|
|
116
117
|
"bench:compare": "node scripts/bench-compare.js",
|
|
@@ -19,11 +19,13 @@ import { type GraphSnapshot, renumberPartition, type U32 } from "@graphty/graph-
|
|
|
19
19
|
|
|
20
20
|
import { U32_MAX } from "../constants.js";
|
|
21
21
|
import { type GpuContext } from "../context.js";
|
|
22
|
-
import {
|
|
22
|
+
import { WebGpuGraphError } from "../errors.js";
|
|
23
23
|
import { CommandBatch } from "../kernel/batch.js";
|
|
24
24
|
import { plan1d, planGridStride } from "../kernel/dispatch.js";
|
|
25
25
|
import { FILL_PARAMS, graphBindings, graphOverrides, kernelSpec, WCC_PARAMS } from "../kernels.js";
|
|
26
26
|
import { type CoreBinding } from "../memory/residency.js";
|
|
27
|
+
import { assertWholeCore } from "../primitives/core-shape.js";
|
|
28
|
+
import { assertDeviceComputes } from "../primitives/verify.js";
|
|
27
29
|
import { type ComponentsOptions, type GpuLabelResult } from "../types/algorithms.js";
|
|
28
30
|
import { type Binding } from "../types/memory.js";
|
|
29
31
|
import { type GpuRunOptions } from "../types/run.js";
|
|
@@ -66,25 +68,16 @@ function checkDest(dest: Float32Array | Uint32Array | undefined, n: number): U32
|
|
|
66
68
|
}
|
|
67
69
|
|
|
68
70
|
/**
|
|
69
|
-
* The resident core; a windowed plan
|
|
70
|
-
*
|
|
71
|
+
* The resident core; a windowed plan is refused with `E_TOO_LARGE { path: "windowed", algorithm }` (spec 3.8, 3.12;
|
|
72
|
+
* DEP-P4-B: only degree and segmentedReduce execute windows).
|
|
71
73
|
* @param ctx - the context
|
|
72
74
|
* @param s - the snapshot
|
|
73
75
|
* @returns the core binding
|
|
74
76
|
*/
|
|
75
77
|
function coreOf(ctx: GpuContext, s: GraphSnapshot): CoreBinding {
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
if (isWebGpuGraphError(error) && error.code === "E_TOO_LARGE" && error.details.path === "windowed") {
|
|
80
|
-
throw new WebGpuGraphError(
|
|
81
|
-
"E_TOO_LARGE",
|
|
82
|
-
`${ALGORITHM}: the arc arrays need a windowed upload, which P1-P3 plan but do not execute`,
|
|
83
|
-
{ ...error.details, algorithm: ALGORITHM },
|
|
84
|
-
);
|
|
85
|
-
}
|
|
86
|
-
throw error;
|
|
87
|
-
}
|
|
78
|
+
const core = ctx.residency.core(s);
|
|
79
|
+
assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM);
|
|
80
|
+
return core;
|
|
88
81
|
}
|
|
89
82
|
|
|
90
83
|
/**
|
|
@@ -197,6 +190,7 @@ export async function connectedComponents(
|
|
|
197
190
|
options?: ComponentsOptions & GpuRunOptions,
|
|
198
191
|
): Promise<GpuLabelResult> {
|
|
199
192
|
ctx.assertReady();
|
|
193
|
+
await assertDeviceComputes(ctx);
|
|
200
194
|
const n = s.nodeCount;
|
|
201
195
|
const renumber = options?.renumber !== false;
|
|
202
196
|
const dest = checkDest(options?.dest, n);
|
|
@@ -312,7 +306,9 @@ export async function connectedComponents(
|
|
|
312
306
|
rounds += ROUNDS_PER_BATCH;
|
|
313
307
|
ctx.assertReady();
|
|
314
308
|
if (options?.signal?.aborted) {
|
|
315
|
-
throw new WebGpuGraphError("E_ABORTED", `${ALGORITHM}: the signal was aborted`, {
|
|
309
|
+
throw new WebGpuGraphError("E_ABORTED", `${ALGORITHM}: the signal was aborted`, {
|
|
310
|
+
batchId: submitted.id,
|
|
311
|
+
});
|
|
316
312
|
}
|
|
317
313
|
if (new Uint32Array(back, flagRequest.offset, 1)[0] === 0) {
|
|
318
314
|
break;
|
package/src/algorithms/degree.ts
CHANGED
|
@@ -9,24 +9,24 @@
|
|
|
9
9
|
*
|
|
10
10
|
* Contract (3.12): `ctx.assertReady()` first; `nodeCount === 0` returns an empty `Uint32Array` (or `dest`) with no
|
|
11
11
|
* GPU work (spec 5.6: there is no work, which is not a fallback); an already-aborted `signal` is `E_ABORTED` before
|
|
12
|
-
* any work; `dest` must be a `Uint32Array` of length n over an `ArrayBuffer`;
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
* fill precedes it), a readback into the result, `onProgress(1, 1)`, and the scratch returned in a finally. Two runs
|
|
19
|
-
* are bitwise identical (spec 11.9 item 4).
|
|
12
|
+
* any work; `dest` must be a `Uint32Array` of length n over an `ArrayBuffer`; `arcCount === 0` skips the dispatch
|
|
13
|
+
* (only `rowPtr` is resident) and the result is zeros; otherwise ONE dispatch over rows [0, n) with `accumulate = 0`
|
|
14
|
+
* (every row is written, so no fill precedes it) -- or, on a windowed core (spec 4.2, PD-8), a `fill` of zeros and
|
|
15
|
+
* one dispatch per window over its rows with `arcBase = w.start`, `arcEnd = w.end`, `accumulate = 1`, so a row split
|
|
16
|
+
* across windows adds its partial counts -- a readback into the result, `onProgress(1, 1)`, and the scratch returned
|
|
17
|
+
* in a finally. Two runs are bitwise identical (spec 11.9 item 4).
|
|
20
18
|
*/
|
|
21
19
|
|
|
22
20
|
import { type GraphSnapshot, type U32 } from "@graphty/graph-format";
|
|
23
21
|
|
|
24
22
|
import { type GpuContext } from "../context.js";
|
|
25
23
|
import { BufferUsage } from "../device/webgpu-constants.js";
|
|
26
|
-
import {
|
|
24
|
+
import { WebGpuGraphError } from "../errors.js";
|
|
27
25
|
import { plan1d } from "../kernel/dispatch.js";
|
|
28
|
-
import {
|
|
29
|
-
import {
|
|
26
|
+
import { type UniformBlock, type UniformValues } from "../kernel/struct-block.js";
|
|
27
|
+
import { FILL_PARAMS, graphBindings, graphOverrides, kernelSpec, RANGE_PARAMS } from "../kernels.js";
|
|
28
|
+
import { windowBinding } from "../primitives/core-shape.js";
|
|
29
|
+
import { assertDeviceComputes } from "../primitives/verify.js";
|
|
30
30
|
import { type Binding } from "../types/memory.js";
|
|
31
31
|
import { type GpuRunOptions } from "../types/run.js";
|
|
32
32
|
|
|
@@ -55,26 +55,21 @@ function checkDest(dest: Float32Array | Uint32Array | undefined, n: number): U32
|
|
|
55
55
|
}
|
|
56
56
|
|
|
57
57
|
/**
|
|
58
|
-
*
|
|
59
|
-
*
|
|
60
|
-
*
|
|
61
|
-
* @param
|
|
62
|
-
* @param
|
|
63
|
-
* @
|
|
58
|
+
* One pool-acquired uniform buffer holding one params record (a params buffer per dispatch; the caller releases
|
|
59
|
+
* `pooled` in its finally).
|
|
60
|
+
* @param ctx - the context whose pool and queue are used
|
|
61
|
+
* @param pooled - the list the buffer is appended to for release
|
|
62
|
+
* @param block - the uniform block
|
|
63
|
+
* @param values - the record's values
|
|
64
|
+
* @returns the whole-buffer binding
|
|
64
65
|
*/
|
|
65
|
-
function
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
"degree: the arc arrays need a windowed upload, which P1-P3 plan but do not execute",
|
|
73
|
-
{ ...error.details, algorithm: "degree" },
|
|
74
|
-
);
|
|
75
|
-
}
|
|
76
|
-
throw error;
|
|
77
|
-
}
|
|
66
|
+
function paramsBinding(ctx: GpuContext, pooled: GPUBuffer[], block: UniformBlock, values: UniformValues): Binding {
|
|
67
|
+
const buffer = ctx.pool.acquire(block.byteLength, BufferUsage.UNIFORM | BufferUsage.COPY_DST, "degree/params");
|
|
68
|
+
pooled.push(buffer);
|
|
69
|
+
const bytes = new ArrayBuffer(block.byteLength);
|
|
70
|
+
block.write(new DataView(bytes), values);
|
|
71
|
+
ctx.device.queue.writeBuffer(buffer, 0, bytes);
|
|
72
|
+
return { buffer, offset: 0, size: block.byteLength, window: null };
|
|
78
73
|
}
|
|
79
74
|
|
|
80
75
|
/**
|
|
@@ -89,6 +84,7 @@ function coreOf(ctx: GpuContext, s: GraphSnapshot): CoreBinding {
|
|
|
89
84
|
*/
|
|
90
85
|
export async function degree(ctx: GpuContext, s: GraphSnapshot, options?: GpuRunOptions): Promise<U32> {
|
|
91
86
|
ctx.assertReady();
|
|
87
|
+
await assertDeviceComputes(ctx);
|
|
92
88
|
const n = s.nodeCount;
|
|
93
89
|
const dest = checkDest(options?.dest, n);
|
|
94
90
|
if (options?.signal?.aborted) {
|
|
@@ -98,7 +94,7 @@ export async function degree(ctx: GpuContext, s: GraphSnapshot, options?: GpuRun
|
|
|
98
94
|
options?.onProgress?.(1, 1);
|
|
99
95
|
return dest ?? new Uint32Array(0);
|
|
100
96
|
}
|
|
101
|
-
const core =
|
|
97
|
+
const core = ctx.residency.core(s);
|
|
102
98
|
if (s.arcCount === 0) {
|
|
103
99
|
const zeros = dest ?? new Uint32Array(n);
|
|
104
100
|
zeros.fill(0);
|
|
@@ -111,23 +107,40 @@ export async function degree(ctx: GpuContext, s: GraphSnapshot, options?: GpuRun
|
|
|
111
107
|
BufferUsage.STORAGE | BufferUsage.COPY_SRC | BufferUsage.COPY_DST,
|
|
112
108
|
"degree/out",
|
|
113
109
|
);
|
|
114
|
-
const
|
|
115
|
-
|
|
116
|
-
BufferUsage.UNIFORM | BufferUsage.COPY_DST,
|
|
117
|
-
"degree/params",
|
|
118
|
-
);
|
|
110
|
+
const pooled: GPUBuffer[] = [];
|
|
111
|
+
const params = (block: UniformBlock, values: UniformValues): Binding => paramsBinding(ctx, pooled, block, values);
|
|
119
112
|
try {
|
|
120
113
|
await ctx.allocator.check();
|
|
121
114
|
const kernel = await ctx.pipelines.kernel(kernelSpec("degree", graphOverrides(core, null)));
|
|
122
|
-
const bytes = new ArrayBuffer(RANGE_PARAMS.byteLength);
|
|
123
|
-
RANGE_PARAMS.write(new DataView(bytes), { start: 0, end: n, arcBase: 0, arcEnd: s.arcCount, accumulate: 0, n });
|
|
124
|
-
ctx.device.queue.writeBuffer(params, 0, bytes);
|
|
125
115
|
const outBinding: Binding = { buffer: out, offset: 0, size: byteLength, window: null };
|
|
126
|
-
const paramsBinding: Binding = { buffer: params, offset: 0, size: RANGE_PARAMS.byteLength, window: null };
|
|
127
|
-
const bound = kernel.bind({ ...graphBindings(core, null), out: outBinding, P: paramsBinding });
|
|
128
116
|
const encoder = ctx.device.createCommandEncoder({ label: "degree" });
|
|
129
117
|
const pass = encoder.beginComputePass({ label: "degree" });
|
|
130
|
-
|
|
118
|
+
if (core.windows === null) {
|
|
119
|
+
const P = params(RANGE_PARAMS, { start: 0, end: n, arcBase: 0, arcEnd: s.arcCount, accumulate: 0, n });
|
|
120
|
+
const bound = kernel.bind({ ...graphBindings(core, null), out: outBinding, P });
|
|
121
|
+
kernel.dispatch(pass, bound, plan1d(n, ctx.workgroupSize, ctx.caps), [0]);
|
|
122
|
+
} else {
|
|
123
|
+
const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
|
|
124
|
+
const zero = fill.bind({ dst: outBinding, P: params(FILL_PARAMS, { count: n, value: 0, mode: 0 }) });
|
|
125
|
+
fill.dispatch(pass, zero, plan1d(n, ctx.workgroupSize, ctx.caps), [0]);
|
|
126
|
+
for (const w of core.windows) {
|
|
127
|
+
const windowed = {
|
|
128
|
+
...core,
|
|
129
|
+
colIdx: windowBinding(core, "colIdx", w),
|
|
130
|
+
weights: core.weights === null ? null : windowBinding(core, "weights", w),
|
|
131
|
+
};
|
|
132
|
+
const P = params(RANGE_PARAMS, {
|
|
133
|
+
start: w.rowFirst,
|
|
134
|
+
end: w.rowLast + 1,
|
|
135
|
+
arcBase: w.start,
|
|
136
|
+
arcEnd: w.end,
|
|
137
|
+
accumulate: 1,
|
|
138
|
+
n,
|
|
139
|
+
});
|
|
140
|
+
const bound = kernel.bind({ ...graphBindings(windowed, null), out: outBinding, P });
|
|
141
|
+
kernel.dispatch(pass, bound, plan1d(w.rowLast - w.rowFirst + 1, ctx.workgroupSize, ctx.caps), [0]);
|
|
142
|
+
}
|
|
143
|
+
}
|
|
131
144
|
pass.end();
|
|
132
145
|
ctx.device.queue.submit([encoder.finish()]);
|
|
133
146
|
ctx.assertReady();
|
|
@@ -136,7 +149,9 @@ export async function degree(ctx: GpuContext, s: GraphSnapshot, options?: GpuRun
|
|
|
136
149
|
options?.onProgress?.(1, 1);
|
|
137
150
|
return dest ?? new Uint32Array(result);
|
|
138
151
|
} finally {
|
|
139
|
-
|
|
152
|
+
for (const buffer of pooled) {
|
|
153
|
+
ctx.pool.release(buffer);
|
|
154
|
+
}
|
|
140
155
|
ctx.pool.release(out);
|
|
141
156
|
}
|
|
142
157
|
}
|
|
@@ -18,14 +18,15 @@ import { type F32, type GraphSnapshot } from "@graphty/graph-format";
|
|
|
18
18
|
|
|
19
19
|
import { U32_MAX } from "../constants.js";
|
|
20
20
|
import { type GpuContext } from "../context.js";
|
|
21
|
-
import {
|
|
21
|
+
import { WebGpuGraphError } from "../errors.js";
|
|
22
22
|
import { CommandBatch } from "../kernel/batch.js";
|
|
23
23
|
import { groupsOf, plan1d } from "../kernel/dispatch.js";
|
|
24
24
|
import { kernelSpec, PR_PARAMS, PR_PARTIAL } from "../kernels.js";
|
|
25
25
|
import { type ArrayBinding, type CoreBinding } from "../memory/residency.js";
|
|
26
|
-
import { coreOfView } from "../primitives/core-shape.js";
|
|
26
|
+
import { assertWholeCore, coreOfView } from "../primitives/core-shape.js";
|
|
27
27
|
import { prepareSegmentedReduce } from "../primitives/segmented-reduce.js";
|
|
28
28
|
import { prepareSpmvPull } from "../primitives/spmv.js";
|
|
29
|
+
import { assertDeviceComputes } from "../primitives/verify.js";
|
|
29
30
|
import { type GpuPageRankResult, type PageRankOptions } from "../types/algorithms.js";
|
|
30
31
|
import { type Binding } from "../types/memory.js";
|
|
31
32
|
import { type GpuRunOptions } from "../types/run.js";
|
|
@@ -62,26 +63,17 @@ function checkDest(dest: Float32Array | Uint32Array | undefined, n: number, algo
|
|
|
62
63
|
}
|
|
63
64
|
|
|
64
65
|
/**
|
|
65
|
-
* The resident core; a windowed plan
|
|
66
|
-
*
|
|
66
|
+
* The resident core; a windowed plan is refused with `E_TOO_LARGE { path: "windowed", algorithm }` (spec 3.8, 3.12;
|
|
67
|
+
* DEP-P4-B: only degree and segmentedReduce execute windows).
|
|
67
68
|
* @param ctx - the context
|
|
68
69
|
* @param s - the snapshot
|
|
69
70
|
* @param algorithm - the caller's name
|
|
70
71
|
* @returns the core binding
|
|
71
72
|
*/
|
|
72
73
|
function coreOf(ctx: GpuContext, s: GraphSnapshot, algorithm: string): CoreBinding {
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
if (isWebGpuGraphError(error) && error.code === "E_TOO_LARGE" && error.details.path === "windowed") {
|
|
77
|
-
throw new WebGpuGraphError(
|
|
78
|
-
"E_TOO_LARGE",
|
|
79
|
-
`${algorithm}: the arc arrays need a windowed upload, which P1-P3 plan but do not execute`,
|
|
80
|
-
{ ...error.details, algorithm },
|
|
81
|
-
);
|
|
82
|
-
}
|
|
83
|
-
throw error;
|
|
84
|
-
}
|
|
74
|
+
const core = ctx.residency.core(s);
|
|
75
|
+
assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, algorithm);
|
|
76
|
+
return core;
|
|
85
77
|
}
|
|
86
78
|
|
|
87
79
|
/**
|
|
@@ -126,6 +118,7 @@ async function run(
|
|
|
126
118
|
algorithm: string,
|
|
127
119
|
): Promise<GpuPageRankResult> {
|
|
128
120
|
ctx.assertReady();
|
|
121
|
+
await assertDeviceComputes(ctx);
|
|
129
122
|
const n = s.nodeCount;
|
|
130
123
|
const alpha = options?.dampingFactor ?? 0.85;
|
|
131
124
|
const maxIterations = options?.maxIterations ?? 100;
|
|
@@ -144,7 +137,13 @@ async function run(
|
|
|
144
137
|
}
|
|
145
138
|
if (n === 0) {
|
|
146
139
|
options?.onProgress?.(maxIterations, maxIterations);
|
|
147
|
-
return {
|
|
140
|
+
return {
|
|
141
|
+
scores: dest ?? new Float32Array(0),
|
|
142
|
+
iterations: 0,
|
|
143
|
+
converged: true,
|
|
144
|
+
danglingMass: 0,
|
|
145
|
+
precision: "f32",
|
|
146
|
+
};
|
|
148
147
|
}
|
|
149
148
|
const core = coreOf(ctx, s, algorithm);
|
|
150
149
|
if (s.arcCount === 0) {
|
|
@@ -322,7 +321,10 @@ export async function personalizedPageRank(
|
|
|
322
321
|
expected,
|
|
323
322
|
});
|
|
324
323
|
if (!(personalization instanceof Float32Array) || personalization.length !== n) {
|
|
325
|
-
throw invalid(
|
|
324
|
+
throw invalid(
|
|
325
|
+
`${personalization.constructor.name}(${personalization.length})`,
|
|
326
|
+
`a Float32Array of length ${n}`,
|
|
327
|
+
);
|
|
326
328
|
}
|
|
327
329
|
let total = 0;
|
|
328
330
|
for (let v = 0; v < n; v++) {
|
|
@@ -20,13 +20,14 @@ import { type F32, type GraphSnapshot } from "@graphty/graph-format";
|
|
|
20
20
|
|
|
21
21
|
import { U32_MAX } from "../constants.js";
|
|
22
22
|
import { type GpuContext } from "../context.js";
|
|
23
|
-
import {
|
|
23
|
+
import { WebGpuGraphError } from "../errors.js";
|
|
24
24
|
import { CommandBatch } from "../kernel/batch.js";
|
|
25
25
|
import { groupsOf, plan1d } from "../kernel/dispatch.js";
|
|
26
26
|
import { kernelSpec, PR_PARAMS, PR_PARTIAL } from "../kernels.js";
|
|
27
27
|
import { type CoreBinding } from "../memory/residency.js";
|
|
28
|
-
import { coreOfView } from "../primitives/core-shape.js";
|
|
28
|
+
import { assertWholeCore, coreOfView } from "../primitives/core-shape.js";
|
|
29
29
|
import { prepareSpmvPull } from "../primitives/spmv.js";
|
|
30
|
+
import { assertDeviceComputes } from "../primitives/verify.js";
|
|
30
31
|
import { type Binding } from "../types/memory.js";
|
|
31
32
|
import { algorithmScope } from "./scope.js";
|
|
32
33
|
|
|
@@ -103,26 +104,17 @@ export function checkDest(dest: Float32Array | Uint32Array | undefined, n: numbe
|
|
|
103
104
|
}
|
|
104
105
|
|
|
105
106
|
/**
|
|
106
|
-
* The resident core; a windowed plan
|
|
107
|
-
*
|
|
107
|
+
* The resident core; a windowed plan is refused with `E_TOO_LARGE { path: "windowed", algorithm }` (spec 3.8, 3.12;
|
|
108
|
+
* DEP-P4-B: only degree and segmentedReduce execute windows).
|
|
108
109
|
* @param ctx - the context
|
|
109
110
|
* @param s - the snapshot
|
|
110
111
|
* @param algorithm - the caller's name
|
|
111
112
|
* @returns the core binding
|
|
112
113
|
*/
|
|
113
114
|
export function coreOf(ctx: GpuContext, s: GraphSnapshot, algorithm: string): CoreBinding {
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
if (isWebGpuGraphError(error) && error.code === "E_TOO_LARGE" && error.details.path === "windowed") {
|
|
118
|
-
throw new WebGpuGraphError(
|
|
119
|
-
"E_TOO_LARGE",
|
|
120
|
-
`${algorithm}: the arc arrays need a windowed upload, which P1-P3 plan but do not execute`,
|
|
121
|
-
{ ...error.details, algorithm },
|
|
122
|
-
);
|
|
123
|
-
}
|
|
124
|
-
throw error;
|
|
125
|
-
}
|
|
115
|
+
const core = ctx.residency.core(s);
|
|
116
|
+
assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, algorithm);
|
|
117
|
+
return core;
|
|
126
118
|
}
|
|
127
119
|
|
|
128
120
|
/**
|
|
@@ -179,6 +171,7 @@ export async function runPowerIteration(
|
|
|
179
171
|
n: number,
|
|
180
172
|
config: PowerIterationConfig,
|
|
181
173
|
): Promise<PowerIterationRun> {
|
|
174
|
+
await assertDeviceComputes(ctx);
|
|
182
175
|
const scope = algorithmScope(ctx, config.label, RING_SLOTS);
|
|
183
176
|
try {
|
|
184
177
|
const bytes = 4 * n;
|
|
@@ -231,7 +224,14 @@ export async function runPowerIteration(
|
|
|
231
224
|
const rankOut = ring[iteration % ring.length];
|
|
232
225
|
const { core, pull } = pulls[(iteration - 1) % pulls.length];
|
|
233
226
|
// outWeightSum takes rankIn as its dummy: storage-ro, read only under NORM_MODE 0, never compiled here
|
|
234
|
-
const scaleBindings = {
|
|
227
|
+
const scaleBindings = {
|
|
228
|
+
rankIn,
|
|
229
|
+
rankPrev: rankOut,
|
|
230
|
+
outWeightSum: rankIn,
|
|
231
|
+
xNorm,
|
|
232
|
+
partials,
|
|
233
|
+
P: params.binding,
|
|
234
|
+
};
|
|
235
235
|
scaleNorm.dispatch(pass, scaleNorm.bind(scaleBindings), scalePlan, [params.offset]);
|
|
236
236
|
finalize.dispatch(pass, finalize.bind({ partials, P: params.binding }), finalizePlan, [params.offset]);
|
|
237
237
|
if (scaleApply !== null) {
|
|
@@ -265,7 +265,8 @@ export async function runPowerIteration(
|
|
|
265
265
|
}
|
|
266
266
|
return {
|
|
267
267
|
scores: new Float32Array(back, scoresRequest.offset, n).slice(),
|
|
268
|
-
previous:
|
|
268
|
+
previous:
|
|
269
|
+
previousRequest === null ? null : new Float32Array(back, previousRequest.offset, n).slice(),
|
|
269
270
|
iterations: converged ? firstConverged : iterationsRun,
|
|
270
271
|
converged,
|
|
271
272
|
iterationsRun,
|
package/src/constants.ts
CHANGED
|
@@ -13,6 +13,8 @@ export const MAX_WORKGROUPS_PER_DIM = 65535;
|
|
|
13
13
|
export const MAX_1D_ITEMS = MAX_WORKGROUPS_PER_DIM * WORKGROUP_SIZE;
|
|
14
14
|
/** The largest u32 (the `min` identity of the u32 reduce); interpolated into the prelude as `U32_MAX` so no body types the literal (contract 4.1). */
|
|
15
15
|
export const U32_MAX = 0xffffffff;
|
|
16
|
+
/** The bins of one radix-sort pass, 2^8 (spec 6 row 6; P4 PD-5: 8 bits per pass); interpolated into the prelude as `RADIX_BINS` and `RADIX_DIGIT_MASK` (= bins - 1) so no body types the literal. */
|
|
17
|
+
export const RADIX_BINS = 256;
|
|
16
18
|
/** Arc-window boundaries are multiples of 64 arcs = 256 bytes (design 10.6). */
|
|
17
19
|
export const ARC_WINDOW_ALIGN = 64;
|
|
18
20
|
/** Storage-binding offset alignment the package always honours (spec 2.6): the graph-format arena is 256-aligned. */
|
|
@@ -20,14 +22,30 @@ export const STORAGE_ALIGN = 256;
|
|
|
20
22
|
/** Stride of one UniformRing slot: minUniformBufferOffsetAlignment is 256 on every runtime the package targets (spec 5.3). */
|
|
21
23
|
export const UNIFORM_SLOT_BYTES = 256;
|
|
22
24
|
/**
|
|
23
|
-
* Exact-tier crossover default, re-fixed at
|
|
24
|
-
*
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
*
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
*
|
|
25
|
+
* Exact-tier crossover default, re-fixed at G4 by the spec 7.8 rule in full (spec Q-6; docs/decisions/G3.md section 3
|
|
26
|
+
* for the budget clause, docs/decisions/G4.md section 7 for the re-check): the largest rung of the T-4 ladder (1k / 4k /
|
|
27
|
+
* 8k / 16k / 32k / 65k, E = 10n, 2D) with <= 4 ms per iteration AND not slower than the grid tier at the same n (the
|
|
28
|
+
* `layout-grid` 2D rows: the shared rungs 32k / 65k and the re-check rungs 1k / 4k / 8k / 16k the group runs so every
|
|
29
|
+
* exact rung has a grid row), rounded down to a power of two. Measured on the RTX 4070 SUPER under Dawn-node:
|
|
30
|
+
* benchmarks/results/nvidia-lovelace-driver580.json, session 2026-09-21T06:15:17.027Z, the "ms/iteration (profiler)"
|
|
31
|
+
* rows of layout-exact and the "grid ms/iteration (profiler) ... 2D" rows of layout-grid (the GPU time of the
|
|
32
|
+
* iteration's passes at the card's working clock), exact / grid ms per iteration: 1k 0.102 / 0.290, 4k 0.262 / 0.188,
|
|
33
|
+
* 8k 0.484 / 0.211, 16k 1.064 / 0.225, 32k 2.579 / 0.298, 65k 8.450 / 0.409. The grid tier's cost is a near-constant
|
|
34
|
+
* ~0.19-0.3 ms floor (its sort, scan and pyramid passes) up to 32k, so the exact tier wins only at 1k and the rule's
|
|
35
|
+
* answer is 1024 (32768 under the budget clause alone, the G3 value; 16384 was the P4-T14 draft's value, which only
|
|
36
|
+
* evaluated the grid clause at 32k / 65k and was not a measurement). calibrateLayout's finer probe on the same card
|
|
37
|
+
* (tmp/p4/t14-review/calibrate-defaults.log: exact 0.150 vs grid 0.190 ms at 2048) puts the true crossover between
|
|
38
|
+
* 2k and 4k; the ladder has no 2k rung, and the design's basis (cosmos switches at 4,096) is within 0.07 ms per
|
|
39
|
+
* iteration of the rule's value, so the choice among 1k / 2k / 4k is an accuracy preference (the exact tier is the
|
|
40
|
+
* oracle), not a speed one. OWNER DECISION G4-D1 (docs/decisions/G4.md section 7, rows G4-D1 / G4-F9): the constant
|
|
41
|
+
* KEEPS the G3 value 32768 while the accuracy work G4-F1 leaves is open -- G4-F1 (the grid tier's exact-vs-grid RMS
|
|
42
|
+
* on the clumpy and degenerate fixtures) closed on 2026-09-21 by the owner's asserted / printed split, G4-F2 (the
|
|
43
|
+
* unbiasedness item) the same day by its re-scope to the whole-field ratio, and the far field's accuracy on those
|
|
44
|
+
* fixtures is the follow-up -- because "auto" is the default every consumer sees and the exact tier is the accurate
|
|
45
|
+
* one; the rule's answer on the dev box (1024) is recorded, not shipped, until that work closes and the value is
|
|
46
|
+
* re-fixed. A
|
|
47
|
+
* consumer whose GPU differs (integrated, Apple, T4) passes its own value through
|
|
48
|
+
* createAccelerator(ctx, { layout: { exactMaxNodes } }); calibrateLayout(ctx) measures it.
|
|
31
49
|
*/
|
|
32
50
|
export const EXACT_MAX_NODES = 32768;
|
|
33
51
|
/** Default number of MAP_READ staging buffers in the Readback ring (spec 4.4). */
|
|
@@ -184,3 +202,15 @@ export const SE_DEFAULTS: Readonly<{
|
|
|
184
202
|
iterationsPerStep: 1,
|
|
185
203
|
maxInFlight: 2,
|
|
186
204
|
});
|
|
205
|
+
/** The smallest finest grid side `G` (spec 7.7 geometry table: `clamp(nextPow2(2 n^(1/dim)), 8, gridMax)`; P4 PD-9). */
|
|
206
|
+
export const GRID_MIN_SIDE = 8;
|
|
207
|
+
/** The coarsest pyramid level's side (spec 7.7: "levels (coarsest 4 per axis)", `levels = log2(G / 4) + 1`). */
|
|
208
|
+
export const GRID_COARSEST_SIDE = 4;
|
|
209
|
+
/** A cell with more than this many entries is summed by a workgroup (G4b) instead of the thread-per-cell loop of G4 (spec 7.7 G4: `count > 1024`). */
|
|
210
|
+
export const GRID_HUB_CELL = 1024;
|
|
211
|
+
/** The extent floor of spec 7.7: `cellSize = max(extent, 1e-6) / G`, so an all-coincident load never divides by zero. */
|
|
212
|
+
export const GRID_EXTENT_FLOOR = 1e-6;
|
|
213
|
+
/** The bbox margin of spec 7.7: `bboxExtent` is the largest axis of `state.min / max` "with a 1% margin". */
|
|
214
|
+
export const GRID_BBOX_MARGIN = 1.01;
|
|
215
|
+
/** The key width the grid always sorts with (P4 PD-5): three 8-bit passes cover the 19-bit 2D keys at G = 512 and the 22-bit 3D keys at G = 128, and an odd pass count leaves `sortedIdx` in the scratch pair. */
|
|
216
|
+
export const GRID_SORT_BITS = 24;
|
package/src/errors.ts
CHANGED
|
@@ -9,7 +9,8 @@
|
|
|
9
9
|
* `details` keys per code (contract 3.1, so tests can assert them): E_NO_WEBGPU { reason, hint };
|
|
10
10
|
* E_NO_ADAPTER { reason }; E_NO_DEVICE { reason, adapter, limit?, requested?, available? } (reason "consumed" |
|
|
11
11
|
* "requestDevice" | "limit" | "feature" | "maxComputeWorkgroupsPerDimension"); E_SOFTWARE_ONLY { adapter };
|
|
12
|
-
* E_DEVICE_LOST { reason, message };
|
|
12
|
+
* E_DEVICE_LOST { reason, message }; E_DEVICE_INCORRECT { check, where, expected, actual, poison, count, blocks,
|
|
13
|
+
* workgroupSize, adapter, hint? }; E_DISPOSED { label }; E_VALIDATION { label, message, batchId? };
|
|
13
14
|
* E_SHADER_COMPILE { id, stage: "compose" | "compile", slot?, messages? }; E_OUT_OF_MEMORY { requested, resident,
|
|
14
15
|
* label }; E_TOO_LARGE { needed, limit, path, algorithm }; E_UNSUPPORTED { feature? | option?, hint? } (exactly
|
|
15
16
|
* one of feature / option); E_INVALID_ARGUMENT { argument, value, expected? }; E_SNAPSHOT { reason, serial };
|
|
@@ -23,6 +24,7 @@ export type WebGpuGraphErrorCode =
|
|
|
23
24
|
| "E_NO_DEVICE"
|
|
24
25
|
| "E_SOFTWARE_ONLY"
|
|
25
26
|
| "E_DEVICE_LOST"
|
|
27
|
+
| "E_DEVICE_INCORRECT"
|
|
26
28
|
| "E_DISPOSED"
|
|
27
29
|
| "E_VALIDATION"
|
|
28
30
|
| "E_SHADER_COMPILE"
|
package/src/index.ts
CHANGED
|
@@ -8,8 +8,8 @@
|
|
|
8
8
|
* isSoftwareAdapter. P1 adds GpuContext and degree (values) and the context / run / profiler types. P2 adds nothing
|
|
9
9
|
* (Lease, CommandBatch, UniformRing are internal). P3 adds the layout factory, the accelerator, the two default
|
|
10
10
|
* tables, the seeder and the layout / accelerator types. P5 adds the two factories, the two default tables and the
|
|
11
|
-
* two stats records.
|
|
12
|
-
*
|
|
11
|
+
* two stats records. P4 adds calibrateLayout and its two records. test/index.test.ts pins the value list and
|
|
12
|
+
* test/types/public-api.test-d.ts the type list. This comment must never spell the internal
|
|
13
13
|
* JSDoc tag: it is the leading comment of the first export statement, and stripInternal would drop that statement
|
|
14
14
|
* from the emitted declarations.
|
|
15
15
|
*/
|
|
@@ -41,6 +41,11 @@ export { GpuContext } from "./context.js";
|
|
|
41
41
|
export { isSoftwareAdapter } from "./device/acquire.js";
|
|
42
42
|
export type { PassTiming, Profiler } from "./kernel/profiler.js";
|
|
43
43
|
|
|
44
|
+
// ==================== the device self-check: what this device computed when it was asked an answer we already
|
|
45
|
+
// know. Every compute entry point awaits it and refuses a device that got it wrong (E_DEVICE_INCORRECT); a
|
|
46
|
+
// caller may await it first to ask before committing.
|
|
47
|
+
export { verifyDevice } from "./primitives/verify.js";
|
|
48
|
+
|
|
44
49
|
// ==================== algorithms (P1: the walking-skeleton diagnostic, spec 3.3)
|
|
45
50
|
export { degree } from "./algorithms/degree.js";
|
|
46
51
|
|
|
@@ -49,8 +54,9 @@ export { connectedComponents } from "./algorithms/components.js";
|
|
|
49
54
|
export { pageRank, personalizedPageRank } from "./algorithms/pagerank.js";
|
|
50
55
|
export { eigenvectorCentrality, hits, katzCentrality } from "./algorithms/spectral.js";
|
|
51
56
|
|
|
52
|
-
// ==================== layouts and the accelerator (P3; the two P5 factories)
|
|
57
|
+
// ==================== layouts and the accelerator (P3; the two P5 factories; P4's calibrateLayout, spec 2.2)
|
|
53
58
|
export { createAccelerator } from "./accelerator.js";
|
|
59
|
+
export { calibrateLayout } from "./layouts/calibrate.js";
|
|
54
60
|
export { createForceAtlas2 } from "./layouts/forceatlas2.js";
|
|
55
61
|
export { createFruchtermanReingold } from "./layouts/fruchterman-reingold.js";
|
|
56
62
|
export { seedPositions } from "./layouts/seed.js";
|
|
@@ -97,6 +103,8 @@ export type {
|
|
|
97
103
|
export type {
|
|
98
104
|
AdapterInfoLike,
|
|
99
105
|
AdapterSummary,
|
|
106
|
+
DeviceCheck,
|
|
107
|
+
DeviceCheckMismatch,
|
|
100
108
|
GpuCaps,
|
|
101
109
|
GpuContextOptions,
|
|
102
110
|
LimitPolicy,
|
|
@@ -107,12 +115,14 @@ export type {
|
|
|
107
115
|
RaisableLimit,
|
|
108
116
|
} from "./types/context.js";
|
|
109
117
|
|
|
110
|
-
// ==================== types: layouts (P3; the P5 stats and trace records)
|
|
118
|
+
// ==================== types: layouts (P3; the P5 stats and trace records; the P4 calibration records)
|
|
111
119
|
export type {
|
|
120
|
+
CalibrateOptions,
|
|
112
121
|
ForceAtlas2Stats,
|
|
113
122
|
ForceAtlas2TraceRecord,
|
|
114
123
|
FruchtermanReingoldStats,
|
|
115
124
|
FruchtermanReingoldTraceRecord,
|
|
125
|
+
GpuCalibration,
|
|
116
126
|
GpuLayoutSimulation,
|
|
117
127
|
GpuLayoutTuning,
|
|
118
128
|
LayoutStatsBase,
|
package/src/kernel/dispatch.ts
CHANGED
|
@@ -147,16 +147,27 @@ export function planGridStride(items: number, wg: number, caps: PlanCaps, maxGro
|
|
|
147
147
|
}
|
|
148
148
|
|
|
149
149
|
/**
|
|
150
|
-
* P4 (spec 5.4): the (x, y, 1) args
|
|
151
|
-
*
|
|
152
|
-
*
|
|
150
|
+
* P4 (spec 5.4): the (x, y, 1) args the device-side finalize kernel writes for a count -- plan1d's rule applied to a
|
|
151
|
+
* u32 count through the same grid() as plan1d and plan2d, so the host twin and the kernel cannot drift; the kernel
|
|
152
|
+
* (src/wgsl/indirect-finalize.wgsl.ts) mirrors exactly this arithmetic. `items` of the plan is the count. A count
|
|
153
|
+
* outside [0, 2^32) is E_INVALID_ARGUMENT (the device holds it as a u32); for any u32 count y <= 1,025, so
|
|
154
|
+
* E_TOO_LARGE is unreachable here.
|
|
155
|
+
* @param count - the device-side count (a u32)
|
|
156
|
+
* @param wg - the workgroup size (a power of two)
|
|
153
157
|
* @param caps - the capability table
|
|
158
|
+
* @returns the plan
|
|
154
159
|
*/
|
|
155
160
|
export function planIndirect(count: number, wg: number, caps: PlanCaps): DispatchPlan {
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
161
|
+
assertCount("count", count);
|
|
162
|
+
if (count > 0xffffffff) {
|
|
163
|
+
throw new WebGpuGraphError("E_INVALID_ARGUMENT", `count must fit a u32, got ${count}`, {
|
|
164
|
+
argument: "count",
|
|
165
|
+
value: count,
|
|
166
|
+
expected: "an integer in [0, 2^32)",
|
|
167
|
+
});
|
|
168
|
+
}
|
|
169
|
+
assertWorkgroupSize(wg);
|
|
170
|
+
return grid(Math.ceil(count / wg), count, caps);
|
|
160
171
|
}
|
|
161
172
|
|
|
162
173
|
/**
|