@graphty/webgpu-graph-algorithms 0.6.4 → 0.6.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-Dvq-Cc6v.js → context-hzGggHeM.js} +28 -30
- package/dist/chunks/context-hzGggHeM.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/algorithms/bfs.d.ts +1 -1
- package/dist/src/algorithms/bfs.js +1 -1
- package/dist/src/algorithms/scope.d.ts +3 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +0 -2
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +4 -3
- package/dist/src/algorithms/sssp.d.ts.map +1 -1
- package/dist/src/algorithms/sssp.js +4 -3
- package/dist/src/algorithms/sssp.js.map +1 -1
- package/dist/src/constants.d.ts +0 -2
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +0 -2
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernels.d.ts +8 -7
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +12 -15
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +3 -2
- package/dist/src/primitives/advance.d.ts.map +1 -1
- package/dist/src/primitives/advance.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +33 -38
- package/dist/src/primitives/frontier.d.ts.map +1 -1
- package/dist/src/primitives/frontier.js +23 -32
- package/dist/src/primitives/frontier.js.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +24 -30
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.js +36 -82
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
- package/dist/webgpu-graph-algorithms.js +27 -86
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +1 -1
- package/src/algorithms/bfs.ts +1 -1
- package/src/algorithms/scope.ts +3 -10
- package/src/algorithms/sssp.ts +4 -3
- package/src/constants.ts +0 -2
- package/src/kernels.ts +12 -15
- package/src/primitives/advance.ts +5 -4
- package/src/primitives/frontier.ts +40 -56
- package/src/wgsl/frontier-finalize.wgsl.ts +36 -82
- package/dist/chunks/context-Dvq-Cc6v.js.map +0 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@graphty/webgpu-graph-algorithms",
|
|
3
|
-
"version": "0.6.
|
|
3
|
+
"version": "0.6.5",
|
|
4
4
|
"description": "WebGPU-accelerated graph algorithms and layouts over the @graphty/graph-format snapshot, for Node (Dawn) and browsers",
|
|
5
5
|
"author": "Adam Powers <apowers@ato.ms>",
|
|
6
6
|
"type": "module",
|
package/src/algorithms/bfs.ts
CHANGED
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
* subtraction inside the selector (whose JSDoc states the boundary rule). The host records `MAX_LEVELS_PER_SUBMIT`
|
|
21
21
|
* levels into ONE command buffer, submits, and reads four bytes, the `done` word (PD-7): a road network has
|
|
22
22
|
* thousands of levels and a per-level `mapAsync` would be slower than the CPU. A level recorded past the end is a
|
|
23
|
-
* no-op (its boundary finds `done` set,
|
|
23
|
+
* no-op (its boundary finds `done` set, writes path 0 and moves no counter), so the loop needs no diameter and
|
|
24
24
|
* ends on `done`; a traversal has at most `n` levels, so more submits than that is E_VALIDATION, never a hang. The
|
|
25
25
|
* thresholds are uniform fields (`FUSED_FRONTIER_MAX`; alpha derived as `max(1, floor(arcCount / n))`, PD-21;
|
|
26
26
|
* `BEAMER_BETA`; `mode 1` pinning top-down -- each unless the tuning says otherwise), so a test forces any path
|
package/src/algorithms/scope.ts
CHANGED
|
@@ -8,12 +8,11 @@
|
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
10
|
import { type GpuContext } from "../context.js";
|
|
11
|
-
import { BufferUsage } from "../device/webgpu-constants.js";
|
|
12
11
|
import { UniformRing } from "../kernel/uniform-ring.js";
|
|
13
|
-
import { type
|
|
12
|
+
import { type ReduceScope } from "../primitives/reduce.js";
|
|
14
13
|
|
|
15
|
-
/** A
|
|
16
|
-
export interface AlgorithmScope extends
|
|
14
|
+
/** A ReduceScope over a context plus the two lifecycle calls an algorithm makes: flush() before submit, dispose() in its finally. (The `indirect()` lease it once added for the frontier's args buffer went with that buffer, 2026-09-25.) */
|
|
15
|
+
export interface AlgorithmScope extends ReduceScope {
|
|
17
16
|
/** queue.writeBuffer of the params slots written since the last flush (called before the batch is submitted). */
|
|
18
17
|
flush(): void;
|
|
19
18
|
/** Destroys the ring and releases every scratch buffer of the lease; idempotent. */
|
|
@@ -39,12 +38,6 @@ export function algorithmScope(ctx: GpuContext, label: string, slots: number): A
|
|
|
39
38
|
pool: ctx.pool,
|
|
40
39
|
workgroupSize: ctx.workgroupSize,
|
|
41
40
|
scratch: (byteLength, scratchLabel) => lease.storage(byteLength, `${label}/${scratchLabel}`),
|
|
42
|
-
indirect: (byteLength, indirectLabel) =>
|
|
43
|
-
lease.acquire(
|
|
44
|
-
byteLength,
|
|
45
|
-
BufferUsage.STORAGE | BufferUsage.INDIRECT | BufferUsage.COPY_DST | BufferUsage.COPY_SRC,
|
|
46
|
-
`${label}/${indirectLabel}`,
|
|
47
|
-
),
|
|
48
41
|
params(block, values) {
|
|
49
42
|
const slot = ring.reserve(1);
|
|
50
43
|
ring.write(slot, block, values);
|
package/src/algorithms/sssp.ts
CHANGED
|
@@ -11,9 +11,10 @@
|
|
|
11
11
|
* delta and dedupes the far half into `farIn` for a pass-through; both empty is `done`), `dedupe-claim` and
|
|
12
12
|
* `dedupe-filter` over each half (direct grid-stride dispatches; role 2 writes the chosen half's raw count into that
|
|
13
13
|
* dedupe's count word -- `edgeCount` for the near half, `edgeCountUnclamped` for the far one, two words SSSP borrows
|
|
14
|
-
* -- and 0 into the other's, so the half not chosen is a no-op), role 3 (restarts the raw half),
|
|
15
|
-
* twice (role 0 over `nearIn`, role 1 over `farIn
|
|
16
|
-
* the other role's dispatch a no-op).
|
|
14
|
+
* -- and 0 into the other's, so the half not chosen is a no-op), role 3 (restarts the raw half the round consumed),
|
|
15
|
+
* and `sssp-relax` twice (role 0 over `nearIn` sized by `frontierCount`, role 1 over `farIn` sized by `farCount`;
|
|
16
|
+
* the block's `path` word, 5 a near round and 6 a far one, makes the other role's dispatch a no-op). Nothing in a
|
|
17
|
+
* round is an indirect dispatch (2026-09-25). The far pile is re-bucketed by the
|
|
17
18
|
* relax kernel's pass-through, not by `compact` (PD-20): a far entry whose settled distance fell below the previous
|
|
18
19
|
* threshold was relaxed in the near band already and is dropped, the rest go back to near or far against the raised
|
|
19
20
|
* threshold. The near pile is ONE pile (no sub-partitions). The host records `MAX_LEVELS_PER_SUBMIT` rounds per
|
package/src/constants.ts
CHANGED
|
@@ -218,8 +218,6 @@ export const GRID_SORT_BITS = 24;
|
|
|
218
218
|
export const MAX_LEVELS_PER_SUBMIT = 32;
|
|
219
219
|
/** Design 8.4 and 6 row 8 (P8): a frontier at most this long runs the fused expand-contract kernel (Merrill's "fleeting iterations"); the default of the `fusedMax` uniform, which a test may set to 0 or `U32_MAX`. */
|
|
220
220
|
export const FUSED_FRONTIER_MAX = 4096;
|
|
221
|
-
/** P8-T4: the indirect dispatch slots `frontier-finalize` writes per level (expand, contract, fused, fill-bits, bitset-build, bottom-up, fused-retry); the args buffer is `MAX_LEVELS_PER_SUBMIT x FRONTIER_CANDIDATES x 16` bytes. */
|
|
222
|
-
export const FRONTIER_CANDIDATES = 7;
|
|
223
221
|
/** Design 8.4 (P8 PD-21): Beamer's beta -- switch back to top-down when `frontierCount * BEAMER_BETA < unvisitedCount` and the frontier is shrinking; alpha is derived from the graph, so it has no constant. */
|
|
224
222
|
export const BEAMER_BETA = 24;
|
|
225
223
|
/** Design 8.4 (P8 PD-22): the near-far split `delta = SSSP_DELTA_FACTOR * avgWeight / avgDegree`, computed on the host from the weight vector the run uses. */
|
package/src/kernels.ts
CHANGED
|
@@ -423,17 +423,17 @@ export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
|
423
423
|
|
|
424
424
|
/**
|
|
425
425
|
* `FrontierParams` (uniform, 80 B; P8-T4): the params block every P8 kernel except the three compact / dedupe
|
|
426
|
-
* primitives and `bf-relax` binds -- `role` @0 (the finalize role), `
|
|
427
|
-
* `
|
|
428
|
-
* `
|
|
429
|
-
*
|
|
430
|
-
*
|
|
431
|
-
*
|
|
432
|
-
* `
|
|
426
|
+
* primitives and `bf-relax` binds -- `role` @0 (the finalize role), `wg` @4 (the consumers' workgroup size),
|
|
427
|
+
* `alpha` @8, `beta` @12 (Beamer's thresholds, P8-T8), `fusedMax` @16, `edgeCapacity` @20, `maxDepth` @24, `n` @28,
|
|
428
|
+
* `mode` @32 (BFS: 0 auto, 1 top-down only; `sssp-pred`: the PD-27 key rule), `cutoffBits` @36, `arcBase` @40,
|
|
429
|
+
* `arcEnd` @44 (the bound arc window), `predKind` @48 (0 arc, 1 node), `bitsBase` @52, `source` @56, `stride` @60
|
|
430
|
+
* (a grid-stride plan's stride), `firstOfSubmit` @64 (the boundary's index inside its submit, clamped to 2: the
|
|
431
|
+
* unvisited-count subtraction runs at >= 1, the degree-sum one at >= 2), `iteration` @68 (an `sssp-pred` hop pass,
|
|
432
|
+
* P8-T9), `pad1` @72, `pad2` @76. The `slotBase` field that once addressed the selector's indirect slots went with
|
|
433
|
+
* the slots (2026-09-25); `pad2` keeps the block an explicit 80 bytes, the way every block here is padded.
|
|
433
434
|
*/
|
|
434
435
|
export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams", [
|
|
435
436
|
["role", "u32"],
|
|
436
|
-
["slotBase", "u32"],
|
|
437
437
|
["wg", "u32"],
|
|
438
438
|
["alpha", "u32"],
|
|
439
439
|
["beta", "u32"],
|
|
@@ -452,6 +452,7 @@ export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams
|
|
|
452
452
|
["firstOfSubmit", "u32"],
|
|
453
453
|
["iteration", "u32"],
|
|
454
454
|
["pad1", "u32"],
|
|
455
|
+
["pad2", "u32"],
|
|
455
456
|
]);
|
|
456
457
|
|
|
457
458
|
/** `BfParams` (uniform, 16 B; P8-T10): `edgeCount` @0 (the logical edges of the `edgeList` view), `stride` @4 (the grid-stride plan's stride), `maxRetries` @8 (PD-12's compare-exchange bound), `cutoffBits` @12 (the f32 bit pattern of the CPU port's `cutoff`, `+Inf` when absent). */
|
|
@@ -1117,16 +1118,12 @@ const DEDUPE_FILTER: KernelEntry = {
|
|
|
1117
1118
|
phase: "P8",
|
|
1118
1119
|
};
|
|
1119
1120
|
|
|
1120
|
-
/** `frontier-finalize` (design 5.4, 8.10 "BFS finalizeArgs"; P8-T4, PD-3): the one-lane level-boundary selector that rotates the counters block and writes the level's
|
|
1121
|
+
/** `frontier-finalize` (design 5.4, 8.10 "BFS finalizeArgs"; P8-T4, PD-3): the one-lane level-boundary selector that rotates the counters block and writes the level's `path` word (role 0), then clamps the edge count or switches the path to the fused retry (role 1); 1 storage binding (the block as `array<atomic<u32>>`). Since 2026-09-25 it writes no indirect slots: every level kernel is a direct dispatch gated by the path word. */
|
|
1121
1122
|
const FRONTIER_FINALIZE: KernelEntry = {
|
|
1122
1123
|
id: "frontier-finalize",
|
|
1123
1124
|
body: frontierFinalizeWgsl,
|
|
1124
1125
|
entryPoint: "frontier_finalize",
|
|
1125
|
-
bindings: [
|
|
1126
|
-
decl(1, 0, "counters", "storage", "array<atomic<u32>>"),
|
|
1127
|
-
decl(1, 1, "args", "storage", "array<u32>"),
|
|
1128
|
-
decl(2, 0, "P", "uniform", "FrontierParams"),
|
|
1129
|
-
],
|
|
1126
|
+
bindings: [decl(1, 0, "counters", "storage", "array<atomic<u32>>"), decl(2, 0, "P", "uniform", "FrontierParams")],
|
|
1130
1127
|
overrideDecls: [],
|
|
1131
1128
|
uniforms: [FRONTIER_PARAMS],
|
|
1132
1129
|
needs: [],
|
|
@@ -1188,7 +1185,7 @@ const SSSP_PRED: KernelEntry = {
|
|
|
1188
1185
|
phase: "P8",
|
|
1189
1186
|
};
|
|
1190
1187
|
|
|
1191
|
-
/** `bfs-fused` (design 8.4 "the fused variant", 6 row 8 "the workgroup-per-row tier", 8.10 "BFS fused expand-contract"; P8-T7, PD-23): one level's expansion and contraction in one dispatch, one WORKGROUP per frontier entry, every lane stripping the entry's row with `bfs-contract`'s claim inline and no edge queue traffic;
|
|
1188
|
+
/** `bfs-fused` (design 8.4 "the fused variant", 6 row 8 "the workgroup-per-row tier", 8.10 "BFS fused expand-contract"; P8-T7, PD-23): one level's expansion and contraction in one dispatch, one WORKGROUP per frontier entry, every lane stripping the entry's row with `bfs-contract`'s claim inline and no edge queue traffic; a direct grid-stride dispatch that runs when the path word is 2 (a frontier below `P.fusedMax`) or 4 (the overflow retry, role 1's), sized from `frontierCount`; 8 storage bindings (the four graph slots, `frontierIn`, the counters block as `array<atomic<u32>>`, `depth` as `array<atomic<u32>>`, `frontierOut`) -- exactly at the budget, which is why no `parent` lives here (PD-24). */
|
|
1192
1189
|
const BFS_FUSED: KernelEntry = {
|
|
1193
1190
|
id: "bfs-fused",
|
|
1194
1191
|
body: bfsFusedWgsl,
|
|
@@ -34,7 +34,8 @@ import { type Kernel } from "../kernel/kernel.js";
|
|
|
34
34
|
import { FRONTIER_PARAMS, graphBindings, graphOverrides, kernelSpec } from "../kernels.js";
|
|
35
35
|
import { type CoreBinding } from "../memory/residency.js";
|
|
36
36
|
import { type CoreWindow, coreWindows, rowCountOf } from "./core-shape.js";
|
|
37
|
-
import { type Frontier
|
|
37
|
+
import { type Frontier } from "./frontier.js";
|
|
38
|
+
import { type ReduceScope } from "./reduce.js";
|
|
38
39
|
|
|
39
40
|
/** A prepared advance (design 6 row 8): records one expansion of a frontier per level. */
|
|
40
41
|
export interface AdvancePlanner {
|
|
@@ -64,7 +65,7 @@ export interface AdvancePlanner {
|
|
|
64
65
|
* @param core - the resident core arrays of the snapshot the frontier walks (windowed or not)
|
|
65
66
|
* @returns the planner
|
|
66
67
|
*/
|
|
67
|
-
export async function prepareAdvance(scope:
|
|
68
|
+
export async function prepareAdvance(scope: ReduceScope, core: CoreBinding): Promise<AdvancePlanner> {
|
|
68
69
|
const kernel = await scope.pipelines.kernel(kernelSpec("advance-expand", graphOverrides(core, null)));
|
|
69
70
|
return new AdvancePlannerImpl(scope, core, kernel);
|
|
70
71
|
}
|
|
@@ -73,7 +74,7 @@ export async function prepareAdvance(scope: FrontierScope, core: CoreBinding): P
|
|
|
73
74
|
class AdvancePlannerImpl implements AdvancePlanner {
|
|
74
75
|
readonly kernel: Kernel;
|
|
75
76
|
readonly windows: readonly CoreWindow[];
|
|
76
|
-
private readonly scope:
|
|
77
|
+
private readonly scope: ReduceScope;
|
|
77
78
|
private readonly n: number;
|
|
78
79
|
|
|
79
80
|
/**
|
|
@@ -82,7 +83,7 @@ class AdvancePlannerImpl implements AdvancePlanner {
|
|
|
82
83
|
* @param core - the core the kernel was compiled for
|
|
83
84
|
* @param kernel - the `advance-expand` kernel
|
|
84
85
|
*/
|
|
85
|
-
constructor(scope:
|
|
86
|
+
constructor(scope: ReduceScope, core: CoreBinding, kernel: Kernel) {
|
|
86
87
|
this.scope = scope;
|
|
87
88
|
this.kernel = kernel;
|
|
88
89
|
this.windows = coreWindows(core);
|
|
@@ -1,46 +1,48 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The `Frontier` of design 6 row 7 and the device-side dispatch selector of design 5.4 (P8-T4; the P8 plan's PD-1,
|
|
3
|
-
* PD-3, PD-8, PD-23, DEP-P8-A, DEP-P8-C). A traversal's per-level state is two n-slot vertex queues, ONE
|
|
3
|
+
* PD-3, PD-8, PD-23, DEP-P8-A, DEP-P8-C). A traversal's per-level state is two n-slot vertex queues, ONE 25-word
|
|
4
4
|
* counters block (every counter of the phase is a word of it: a four-byte word is never a legal storage-binding
|
|
5
|
-
* offset, and the selector must reach every count it acts on through one binding)
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
5
|
+
* offset, and the selector must reach every count it acts on through one binding) and the edge queue. The host
|
|
6
|
+
* never reads a counter inside a submit: `frontier-finalize`, one lane, runs at the START of every level (role 0:
|
|
7
|
+
* rotates `nextFrontierCount` into `frontierCount`, advances `level`, decides `done`, chooses the level's path) and
|
|
8
|
+
* again once the edge queue is filled (role 1: clamps `edgeCount`, or on an overflow -- `edgeCountUnclamped >
|
|
9
|
+
* edgeCapacity` -- switches the path to the fused retry, PD-23). The choice is the block's `path` word (`W.path`,
|
|
10
|
+
* word 24); every level kernel is a direct grid-stride dispatch that reads it first and does nothing unless the
|
|
11
|
+
* word names it (design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md: the seven indirect slots the
|
|
12
|
+
* selector once wrote per level cost about 0.4 ms of Dawn validation each and were 97 % of a traversal's wall
|
|
13
|
+
* time; they and their args buffer are gone). The host records up to `MAX_LEVELS_PER_SUBMIT` levels per submit and
|
|
14
|
+
* reads `done` (four bytes) once per submit; a boundary that finds `done` set writes path 0 and moves no counter,
|
|
15
|
+
* so the recorded levels past the end are no-ops and the counters freeze at the finishing boundary's values.
|
|
14
16
|
*
|
|
15
17
|
* The seed: `frontierCount` is NEVER seeded, because the first boundary rotates it out unread. A BFS driver calls
|
|
16
18
|
* `reset(queue, source, { nextFrontierCount: 1, level: U32_MAX })`: the first boundary rotates the 1 in, adds it into
|
|
17
19
|
* `visitedCount`, and wraps `level` to 0, so the level-0 expansion claims the source's neighbours at `level + 1 == 1`.
|
|
18
20
|
* `reset` is a queue write, ordered before the submit that follows, and it puts the source on side 0.
|
|
19
21
|
*
|
|
20
|
-
* Every buffer comes from the caller's ONE lease (`scope.scratch
|
|
22
|
+
* Every buffer comes from the caller's ONE lease (`scope.scratch`) so the algorithm's dispose()
|
|
21
23
|
* releases them together (design 4.4). The two vertex queues are two BUFFERS, never two ranges of one: `Kernel.bind`
|
|
22
24
|
* rejects one buffer bound read-only and read-write in one dispatch even for disjoint ranges. `src/primitives/**`
|
|
23
25
|
* never imports `src/context.ts`.
|
|
24
26
|
*/
|
|
25
27
|
|
|
26
|
-
import {
|
|
28
|
+
import { MAX_LEVELS_PER_SUBMIT, U32_MAX } from "../constants.js";
|
|
27
29
|
import { WebGpuGraphError } from "../errors.js";
|
|
28
30
|
import { plan1d } from "../kernel/dispatch.js";
|
|
29
|
-
import {
|
|
31
|
+
import { type Kernel } from "../kernel/kernel.js";
|
|
30
32
|
import { FRONTIER_COUNTERS, FRONTIER_PARAMS, kernelSpec } from "../kernels.js";
|
|
31
33
|
import { type Binding } from "../types/memory.js";
|
|
32
34
|
import { type ReduceScope } from "./reduce.js";
|
|
33
35
|
|
|
34
|
-
/** The
|
|
35
|
-
export const
|
|
36
|
-
|
|
37
|
-
|
|
36
|
+
/** The values of the `path` word (`W.path`): what a level's kernels run; every level kernel reads it first. */
|
|
37
|
+
export const PATH: Readonly<{
|
|
38
|
+
none: 0;
|
|
39
|
+
twoPhase: 1;
|
|
38
40
|
fused: 2;
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
}> = Object.freeze({
|
|
41
|
+
bottomUp: 3;
|
|
42
|
+
fusedRetry: 4;
|
|
43
|
+
near: 5;
|
|
44
|
+
far: 6;
|
|
45
|
+
}> = Object.freeze({ none: 0, twoPhase: 1, fused: 2, bottomUp: 3, fusedRetry: 4, near: 5, far: 6 });
|
|
44
46
|
|
|
45
47
|
/** The words of the counters block (`FRONTIER_COUNTERS`), by index: byte offset 4 x word; no driver types a number. */
|
|
46
48
|
export const W: Readonly<{
|
|
@@ -100,13 +102,7 @@ export const W: Readonly<{
|
|
|
100
102
|
/** The words a `reset` seeds (every other word is zeroed). */
|
|
101
103
|
export type FrontierSeed = Readonly<Partial<Record<keyof typeof W, number>>>;
|
|
102
104
|
|
|
103
|
-
/**
|
|
104
|
-
export interface FrontierScope extends ReduceScope {
|
|
105
|
-
/** A STORAGE | INDIRECT | COPY_DST | COPY_SRC buffer of the scope's lease. */
|
|
106
|
-
indirect(byteLength: number, label: string): GPUBuffer;
|
|
107
|
-
}
|
|
108
|
-
|
|
109
|
-
/** The `FrontierParams` fields a caller passes to `recordFinalize`; the planner fills `role`, `slotBase`, `wg`, `edgeCapacity` and `n` itself. A missing field is written as 0. */
|
|
105
|
+
/** The `FrontierParams` fields a caller passes to `recordFinalize`; the planner fills `role`, `wg`, `edgeCapacity` and `n` itself. A missing field is written as 0. */
|
|
110
106
|
export type FrontierFinalizeFields = Readonly<
|
|
111
107
|
Partial<
|
|
112
108
|
Record<
|
|
@@ -163,14 +159,12 @@ function assertCount(argument: string, value: number): void {
|
|
|
163
159
|
}
|
|
164
160
|
}
|
|
165
161
|
|
|
166
|
-
/** The frontier queue of design 6 row 7: two vertex queues, the counters block
|
|
162
|
+
/** The frontier queue of design 6 row 7: two vertex queues, the counters block and the edge queue, all leased by the caller's scope. */
|
|
167
163
|
export class Frontier {
|
|
168
164
|
/** The two n-slot u32 vertex queues (`vertices[side]` is the input of the current level). */
|
|
169
165
|
readonly vertices: readonly [Binding, Binding];
|
|
170
|
-
/** The `FrontierCounters` block,
|
|
166
|
+
/** The `FrontierCounters` block, 112 B, bound by every kernel as `array<atomic<u32>>`. */
|
|
171
167
|
readonly counters: Binding;
|
|
172
|
-
/** The indirect args, `MAX_LEVELS_PER_SUBMIT x FRONTIER_CANDIDATES` 16-byte slots (usage INDIRECT | STORAGE | COPY_DST | COPY_SRC). */
|
|
173
|
-
readonly args: Binding;
|
|
174
168
|
/** The edge queue: `edgeCapacity` entries of u32 (the target vertex of an arc). */
|
|
175
169
|
readonly edgeQueue: Binding;
|
|
176
170
|
/** How many entries the edge queue holds; role 1 clamps `edgeCount` to it and detects an overflow above it. */
|
|
@@ -183,7 +177,6 @@ export class Frontier {
|
|
|
183
177
|
* Wraps the leased buffers; use prepareFrontier().
|
|
184
178
|
* @param vertices - the two vertex queues
|
|
185
179
|
* @param counters - the counters block
|
|
186
|
-
* @param args - the args buffer
|
|
187
180
|
* @param edgeQueue - the edge queue
|
|
188
181
|
* @param edgeCapacity - the edge queue's entry count
|
|
189
182
|
* @param n - the vertex count
|
|
@@ -191,14 +184,12 @@ export class Frontier {
|
|
|
191
184
|
constructor(
|
|
192
185
|
vertices: readonly [Binding, Binding],
|
|
193
186
|
counters: Binding,
|
|
194
|
-
args: Binding,
|
|
195
187
|
edgeQueue: Binding,
|
|
196
188
|
edgeCapacity: number,
|
|
197
189
|
n: number,
|
|
198
190
|
) {
|
|
199
191
|
this.vertices = vertices;
|
|
200
192
|
this.counters = counters;
|
|
201
|
-
this.args = args;
|
|
202
193
|
this.edgeQueue = edgeQueue;
|
|
203
194
|
this.edgeCapacity = edgeCapacity;
|
|
204
195
|
this.n = n;
|
|
@@ -234,7 +225,7 @@ export class Frontier {
|
|
|
234
225
|
}
|
|
235
226
|
|
|
236
227
|
/**
|
|
237
|
-
* Seeds a traversal: one `queue.writeBuffer` of the whole
|
|
228
|
+
* Seeds a traversal: one `queue.writeBuffer` of the whole 112-byte block (zero except the caller's words) and one of
|
|
238
229
|
* `vertices[0][0] = source`, both ordered before the submit that follows; the source is on side 0 afterwards.
|
|
239
230
|
* `frontierCount` is not a word to seed: the first boundary rotates word 1 into it (the BFS seed is
|
|
240
231
|
* `{ nextFrontierCount: 1, level: U32_MAX }`). A source outside `[0, n)`, an unknown word or a value that is not a
|
|
@@ -261,13 +252,14 @@ export class Frontier {
|
|
|
261
252
|
|
|
262
253
|
/** A prepared frontier (design 6 row 7): the leased queue and the recorded selector dispatches. */
|
|
263
254
|
export interface FrontierPlanner {
|
|
264
|
-
/** The queue the planner
|
|
255
|
+
/** The queue the planner's selector rotates and chooses the path of. */
|
|
265
256
|
readonly frontier: Frontier;
|
|
266
257
|
/**
|
|
267
258
|
* Records one `frontier-finalize` dispatch (one workgroup) in `role` for `level` of the current submit: one
|
|
268
|
-
* `FrontierParams` record with `
|
|
269
|
-
*
|
|
270
|
-
* `[0, 3]` is E_INVALID_ARGUMENT before
|
|
259
|
+
* `FrontierParams` record with `wg`, `edgeCapacity` and `n` filled by the planner and every other field from
|
|
260
|
+
* `fields`. A level outside `[0, MAX_LEVELS_PER_SUBMIT)` (the selector addresses nothing by level any more, but
|
|
261
|
+
* the host's submit cadence still is the bound) or a role outside `[0, 3]` is E_INVALID_ARGUMENT before
|
|
262
|
+
* anything is recorded.
|
|
271
263
|
* @param pass - the compute pass
|
|
272
264
|
* @param role - 0 the level boundary, 1 the edge-queue role (2 and 3 are P8-T9's)
|
|
273
265
|
* @param level - the level inside the submit
|
|
@@ -281,14 +273,14 @@ export interface FrontierPlanner {
|
|
|
281
273
|
* edge capacity defaults to `max(1, min(arcCount, floor(maxStorageBufferBindingSize / 4)))` (never a zero-length
|
|
282
274
|
* buffer: the one-node graph has no arcs); a test passes a small one to force the overflow path. The planner lives
|
|
283
275
|
* exactly as long as the scope: never use it after the scope's dispose().
|
|
284
|
-
* @param scope - the caller's scope (device, caps, cache, scratch,
|
|
276
|
+
* @param scope - the caller's scope (device, caps, cache, scratch, params)
|
|
285
277
|
* @param n - the vertex count
|
|
286
278
|
* @param arcCount - the arc count (the natural edge-queue size)
|
|
287
279
|
* @param edgeCapacity - the edge queue's entry count, when the caller chooses it (an integer >= 1)
|
|
288
280
|
* @returns the planner
|
|
289
281
|
*/
|
|
290
282
|
export async function prepareFrontier(
|
|
291
|
-
scope:
|
|
283
|
+
scope: ReduceScope,
|
|
292
284
|
n: number,
|
|
293
285
|
arcCount: number,
|
|
294
286
|
edgeCapacity?: number,
|
|
@@ -306,7 +298,6 @@ export async function prepareFrontier(
|
|
|
306
298
|
}
|
|
307
299
|
const kernel = await scope.pipelines.kernel(kernelSpec("frontier-finalize"));
|
|
308
300
|
const queueBytes = 4 * Math.max(1, n);
|
|
309
|
-
const argsBytes = MAX_LEVELS_PER_SUBMIT * FRONTIER_CANDIDATES * INDIRECT_ARGS_STRIDE;
|
|
310
301
|
const vertices: readonly [Binding, Binding] = [
|
|
311
302
|
{ buffer: scope.scratch(queueBytes, "frontier/vertices-0"), offset: 0, size: queueBytes, window: null },
|
|
312
303
|
{ buffer: scope.scratch(queueBytes, "frontier/vertices-1"), offset: 0, size: queueBytes, window: null },
|
|
@@ -317,26 +308,20 @@ export async function prepareFrontier(
|
|
|
317
308
|
size: FRONTIER_COUNTERS.byteLength,
|
|
318
309
|
window: null,
|
|
319
310
|
};
|
|
320
|
-
const args: Binding = {
|
|
321
|
-
buffer: scope.indirect(argsBytes, "frontier/args"),
|
|
322
|
-
offset: 0,
|
|
323
|
-
size: argsBytes,
|
|
324
|
-
window: null,
|
|
325
|
-
};
|
|
326
311
|
const edgeQueue: Binding = {
|
|
327
312
|
buffer: scope.scratch(4 * capacity, "frontier/edge-queue"),
|
|
328
313
|
offset: 0,
|
|
329
314
|
size: 4 * capacity,
|
|
330
315
|
window: null,
|
|
331
316
|
};
|
|
332
|
-
const frontier = new Frontier(vertices, counters,
|
|
317
|
+
const frontier = new Frontier(vertices, counters, edgeQueue, capacity, n);
|
|
333
318
|
return new FrontierPlannerImpl(scope, kernel, frontier);
|
|
334
319
|
}
|
|
335
320
|
|
|
336
|
-
/** The planner: the selector kernel bound once to the frontier's block
|
|
321
|
+
/** The planner: the selector kernel bound once to the frontier's block. */
|
|
337
322
|
class FrontierPlannerImpl implements FrontierPlanner {
|
|
338
323
|
readonly frontier: Frontier;
|
|
339
|
-
private readonly scope:
|
|
324
|
+
private readonly scope: ReduceScope;
|
|
340
325
|
private readonly kernel: Kernel;
|
|
341
326
|
|
|
342
327
|
/**
|
|
@@ -345,7 +330,7 @@ class FrontierPlannerImpl implements FrontierPlanner {
|
|
|
345
330
|
* @param kernel - the `frontier-finalize` kernel
|
|
346
331
|
* @param frontier - the leased queue
|
|
347
332
|
*/
|
|
348
|
-
constructor(scope:
|
|
333
|
+
constructor(scope: ReduceScope, kernel: Kernel, frontier: Frontier) {
|
|
349
334
|
this.scope = scope;
|
|
350
335
|
this.kernel = kernel;
|
|
351
336
|
this.frontier = frontier;
|
|
@@ -377,12 +362,11 @@ class FrontierPlannerImpl implements FrontierPlanner {
|
|
|
377
362
|
const params = scope.params(FRONTIER_PARAMS, {
|
|
378
363
|
...definedWords(fields),
|
|
379
364
|
role,
|
|
380
|
-
slotBase: level * FRONTIER_CANDIDATES,
|
|
381
365
|
wg: scope.workgroupSize,
|
|
382
366
|
edgeCapacity: frontier.edgeCapacity,
|
|
383
367
|
n: frontier.n,
|
|
384
368
|
});
|
|
385
|
-
const bound = this.kernel.bind({ counters: frontier.counters,
|
|
369
|
+
const bound = this.kernel.bind({ counters: frontier.counters, P: params.binding });
|
|
386
370
|
this.kernel.dispatch(pass, bound, plan1d(1, scope.workgroupSize, scope.caps), [params.offset]);
|
|
387
371
|
}
|
|
388
372
|
}
|