@graphty/webgpu-graph-algorithms 0.6.4 → 0.6.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +38 -17
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-Dvq-Cc6v.js → context-DiSr6eiz.js} +45 -33
- package/dist/chunks/context-DiSr6eiz.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/algorithms/bfs.d.ts +11 -7
- package/dist/src/algorithms/bfs.d.ts.map +1 -1
- package/dist/src/algorithms/bfs.js +33 -10
- package/dist/src/algorithms/bfs.js.map +1 -1
- package/dist/src/algorithms/scope.d.ts +3 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +0 -2
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +4 -3
- package/dist/src/algorithms/sssp.d.ts.map +1 -1
- package/dist/src/algorithms/sssp.js +4 -3
- package/dist/src/algorithms/sssp.js.map +1 -1
- package/dist/src/constants.d.ts +33 -2
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +33 -2
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernel/dispatch.d.ts +2 -2
- package/dist/src/kernel/dispatch.d.ts.map +1 -1
- package/dist/src/kernel/kernel.d.ts +1 -1
- package/dist/src/kernel/kernel.js +2 -2
- package/dist/src/kernel/kernel.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +2 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernels.d.ts +15 -11
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +42 -21
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
- package/dist/src/layouts/forceatlas2.js +2 -1
- package/dist/src/layouts/forceatlas2.js.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.js +4 -2
- package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
- package/dist/src/layouts/repulsion-exact.d.ts +16 -0
- package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
- package/dist/src/layouts/repulsion-exact.js +21 -1
- package/dist/src/layouts/repulsion-exact.js.map +1 -1
- package/dist/src/layouts/repulsion-grid.d.ts +1 -1
- package/dist/src/layouts/repulsion-grid.js +1 -1
- package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
- package/dist/src/layouts/spring-electrical.js +6 -2
- package/dist/src/layouts/spring-electrical.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +3 -2
- package/dist/src/primitives/advance.d.ts.map +1 -1
- package/dist/src/primitives/advance.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +34 -38
- package/dist/src/primitives/frontier.d.ts.map +1 -1
- package/dist/src/primitives/frontier.js +24 -32
- package/dist/src/primitives/frontier.js.map +1 -1
- package/dist/src/primitives/grid-pyramid.d.ts +4 -4
- package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
- package/dist/src/primitives/grid-pyramid.js +4 -3
- package/dist/src/primitives/grid-pyramid.js.map +1 -1
- package/dist/src/primitives/grid.d.ts +13 -10
- package/dist/src/primitives/grid.d.ts.map +1 -1
- package/dist/src/primitives/grid.js +10 -7
- package/dist/src/primitives/grid.js.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +44 -49
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.js +62 -107
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
- package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
- package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
- package/dist/src/wgsl/histogram.wgsl.js +1 -1
- package/dist/webgpu-graph-algorithms.js +144 -119
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +3 -3
- package/src/algorithms/bfs.ts +34 -10
- package/src/algorithms/pagerank.ts +19 -5
- package/src/algorithms/power-iteration.ts +8 -2
- package/src/algorithms/scope.ts +3 -10
- package/src/algorithms/sssp.ts +4 -3
- package/src/constants.ts +35 -2
- package/src/kernel/dispatch.ts +2 -2
- package/src/kernel/kernel.ts +2 -2
- package/src/kernel/prelude.ts +2 -0
- package/src/kernels.ts +44 -21
- package/src/layouts/forceatlas2.ts +2 -0
- package/src/layouts/fruchterman-reingold.ts +4 -1
- package/src/layouts/repulsion-exact.ts +29 -1
- package/src/layouts/repulsion-grid.ts +1 -1
- package/src/layouts/spring-electrical.ts +8 -1
- package/src/memory/residency.ts +14 -4
- package/src/primitives/advance.ts +5 -4
- package/src/primitives/frontier.ts +42 -56
- package/src/primitives/grid-pyramid.ts +6 -5
- package/src/primitives/grid.ts +17 -12
- package/src/wgsl/advance-expand.wgsl.ts +4 -3
- package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
- package/src/wgsl/bfs-fused.wgsl.ts +6 -6
- package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
- package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
- package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
- package/src/wgsl/frontier-finalize.wgsl.ts +62 -107
- package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
- package/src/wgsl/grid-centroid.wgsl.ts +3 -2
- package/src/wgsl/grid-downsample.wgsl.ts +1 -1
- package/src/wgsl/grid-far-field.wgsl.ts +15 -8
- package/src/wgsl/grid-near-field.wgsl.ts +6 -2
- package/src/wgsl/histogram.wgsl.ts +1 -1
- package/dist/chunks/context-Dvq-Cc6v.js.map +0 -1
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
* override set is a distinct pipeline and the subgroup twin is selected by the device's features (spec 5.1, D16).
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
|
+
import { EXACT_TILES_PER_PASS } from "../constants.js";
|
|
10
11
|
import { WebGpuGraphError } from "../errors.js";
|
|
11
12
|
import { type DispatchPlan, plan1d } from "../kernel/dispatch.js";
|
|
12
13
|
import { type BoundKernel, type Kernel } from "../kernel/kernel.js";
|
|
@@ -35,6 +36,33 @@ export interface RepulsionExactOverrides {
|
|
|
35
36
|
readonly GRAVITY_CENTER: 0 | 1;
|
|
36
37
|
}
|
|
37
38
|
|
|
39
|
+
/**
|
|
40
|
+
* Records K3 over `n` nodes as ceil(tiles / EXACT_TILES_PER_PASS) dispatches (issue #87: llvmpipe's per-invocation
|
|
41
|
+
* loop budget), pass p with p + 1 z slices of which only the last works (the kernel reads the pass from
|
|
42
|
+
* `num_workgroups.z`). One pass up to 32,768 nodes at WG 256. Every model's exact tier records K3 through here.
|
|
43
|
+
* ponytail: pass p also launches p idle slices, sum p over P passes; negligible against the O(n^2) pass work (31
|
|
44
|
+
* passes at 1M nodes); a per-pass uniform removes them if they ever show in a profile.
|
|
45
|
+
* @param kernel - the compiled `fa2-repulsion-exact`
|
|
46
|
+
* @param pass - the open compute pass
|
|
47
|
+
* @param bound - the kernel's bound groups
|
|
48
|
+
* @param plan - plan1d(n) of the kernel
|
|
49
|
+
* @param n - the node count
|
|
50
|
+
* @param paramsOffset - the dynamic offset of the Fa2Params slot
|
|
51
|
+
*/
|
|
52
|
+
export function recordExactRepulsion(
|
|
53
|
+
kernel: Kernel,
|
|
54
|
+
pass: GPUComputePassEncoder,
|
|
55
|
+
bound: BoundKernel,
|
|
56
|
+
plan: DispatchPlan,
|
|
57
|
+
n: number,
|
|
58
|
+
paramsOffset: number,
|
|
59
|
+
): void {
|
|
60
|
+
const passes = Math.max(1, Math.ceil(Math.ceil(n / kernel.workgroupSize) / EXACT_TILES_PER_PASS));
|
|
61
|
+
for (let z = 1; z <= passes; z++) {
|
|
62
|
+
kernel.dispatch(pass, bound, { ...plan, z }, [paramsOffset]);
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
|
|
38
66
|
/** K3 (tiled all-pairs repulsion + gravity + the swing / traction epilogue) followed by K4 (the one-workgroup speed finalize) (spec 7.6, 7.10). */
|
|
39
67
|
export class RepulsionExact {
|
|
40
68
|
/** The overrides both kernels were compiled with (a frozen copy of the argument of create()). */
|
|
@@ -148,7 +176,7 @@ export class RepulsionExact {
|
|
|
148
176
|
recordRepulsion(pass: GPUComputePassEncoder, n: number, paramsOffset: number): void {
|
|
149
177
|
const bound = this.bound(this.boundRepulsion, "recordRepulsion");
|
|
150
178
|
const plan = plan1d(n, this.repulsion.workgroupSize, this.caps);
|
|
151
|
-
this.repulsion
|
|
179
|
+
recordExactRepulsion(this.repulsion, pass, bound, plan, n, paramsOffset);
|
|
152
180
|
}
|
|
153
181
|
|
|
154
182
|
/**
|
|
@@ -216,7 +216,7 @@ export class RepulsionGrid {
|
|
|
216
216
|
|
|
217
217
|
/**
|
|
218
218
|
* The model-owned buffers of the grid tier (spec 7.3; PD-11): `cellKey` / `cellVal` / `sortedKey` / `sortedIdx`
|
|
219
|
-
* 4n, `cellHist` / `cellStart` 4 (cells + 2) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
|
|
219
|
+
* 4n, `cellHist` / `cellStart` 4 histWords (cells + 2^dim + 1) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
|
|
220
220
|
* indirect slot, `pyramid` 16 B per pyramid cell zeroed. `hubCounters` (16 B, zeroed) is the MODEL's on every
|
|
221
221
|
* tier (PD-14: K1 binds it on the exact tier too). n = 0 reports one node's worth of bytes (spec 3.6).
|
|
222
222
|
* @param n - the node count
|
|
@@ -25,6 +25,8 @@ import {
|
|
|
25
25
|
MAX_ITERATIONS_PER_STEP,
|
|
26
26
|
SE_DEFAULTS,
|
|
27
27
|
SE_SCALE_REFERENCE_NODES,
|
|
28
|
+
SETTLE_FLOOR_FRACTION,
|
|
29
|
+
SETTLE_FLOOR_REFERENCE_NODES,
|
|
28
30
|
TRACE_RECORD_BYTES,
|
|
29
31
|
UNIFORM_SLOT_BYTES,
|
|
30
32
|
} from "../constants.js";
|
|
@@ -77,6 +79,7 @@ import {
|
|
|
77
79
|
subset,
|
|
78
80
|
vector,
|
|
79
81
|
} from "./model-common.js";
|
|
82
|
+
import { recordExactRepulsion } from "./repulsion-exact.js";
|
|
80
83
|
import { type GridStage, RepulsionGrid, type RepulsionGridOverrides } from "./repulsion-grid.js";
|
|
81
84
|
|
|
82
85
|
// ============================================================ constants
|
|
@@ -555,6 +558,10 @@ export class SpringElectricalModel implements ForceModel<SpringElectricalOptions
|
|
|
555
558
|
frK: 0,
|
|
556
559
|
temperature: 0,
|
|
557
560
|
springLength: resolved.springLength,
|
|
561
|
+
settleFloor:
|
|
562
|
+
SETTLE_FLOOR_FRACTION.springElectrical *
|
|
563
|
+
resolved.springLength *
|
|
564
|
+
(SETTLE_FLOOR_REFERENCE_NODES / Math.max(n, 1)) ** 0.25,
|
|
558
565
|
springCoefficient: resolved.springCoefficient ?? SE_DEFAULTS.springCoefficient * springSizeFactor(n),
|
|
559
566
|
coulomb: resolved.gravity ?? SE_DEFAULTS.gravity * springSizeFactor(n),
|
|
560
567
|
dragCoefficient: resolved.dragCoefficient,
|
|
@@ -609,7 +616,7 @@ export class SpringElectricalModel implements ForceModel<SpringElectricalOptions
|
|
|
609
616
|
if (stop < 2) {
|
|
610
617
|
return;
|
|
611
618
|
}
|
|
612
|
-
k3
|
|
619
|
+
recordExactRepulsion(k3, pass, k3Bound, bound.plan, bound.n, offset);
|
|
613
620
|
if (stop < STAGE_K5) {
|
|
614
621
|
return;
|
|
615
622
|
}
|
package/src/memory/residency.ts
CHANGED
|
@@ -579,7 +579,9 @@ export class GraphResidency {
|
|
|
579
579
|
}
|
|
580
580
|
|
|
581
581
|
/**
|
|
582
|
-
* One resident per array (spec 4.3: views upload in perArray mode, never into the arena).
|
|
582
|
+
* One resident per array (spec 4.3: views upload in perArray mode, never into the arena). An empty array (the
|
|
583
|
+
* colIdx of an edgeless directed reverse view, the src / dst of an edgeless edgeList) is skipped: spec 5.6 never
|
|
584
|
+
* uploads a zero-length array, and Kernel.bind rejects a zero-size binding, so it is absent as in core().
|
|
583
585
|
* @param record - the owning record
|
|
584
586
|
* @param arrays - the named arrays
|
|
585
587
|
* @param label - the buffer label prefix
|
|
@@ -592,6 +594,9 @@ export class GraphResidency {
|
|
|
592
594
|
): Readonly<Record<string, Binding>> {
|
|
593
595
|
const bindings: Record<string, Binding> = {};
|
|
594
596
|
for (const [name, array] of arrays) {
|
|
597
|
+
if (array.byteLength === 0) {
|
|
598
|
+
continue;
|
|
599
|
+
}
|
|
595
600
|
const resident = this.upload(record, array, array, `${label}:${name}`);
|
|
596
601
|
bindings[name] = { buffer: resident.buffer, offset: 0, size: resident.byteLength, window: null };
|
|
597
602
|
}
|
|
@@ -606,19 +611,24 @@ export class GraphResidency {
|
|
|
606
611
|
* lengths, so keying the packed buffer on `rev.rowPtr` would make the packed and the unpacked view of one
|
|
607
612
|
* snapshot collide -- whichever was built second would get the other's buffer. The record still owns the
|
|
608
613
|
* resident, so release(s) destroys it with the rest. Offsets are STORAGE_ALIGN-aligned because Kernel.bind
|
|
609
|
-
* rejects any other offset synchronously (E_INVALID_ARGUMENT { argument: "offset" }).
|
|
614
|
+
* rejects any other offset synchronously (E_INVALID_ARGUMENT { argument: "offset" }). Empty arrays are left
|
|
615
|
+
* out as in separateArrays; when nothing is left, nothing is uploaded.
|
|
610
616
|
* @param record - the owning record
|
|
611
|
-
* @param
|
|
617
|
+
* @param all - the named arrays, in buffer order
|
|
612
618
|
* @param key - the marker object the resident is keyed on
|
|
613
619
|
* @param label - the buffer label
|
|
614
620
|
* @returns the bindings by name, all into the one buffer
|
|
615
621
|
*/
|
|
616
622
|
private packArrays(
|
|
617
623
|
record: ResidencyRecord,
|
|
618
|
-
|
|
624
|
+
all: readonly (readonly [string, TypedArrayData])[],
|
|
619
625
|
key: object,
|
|
620
626
|
label: string,
|
|
621
627
|
): Readonly<Record<string, Binding>> {
|
|
628
|
+
const arrays = all.filter(([, array]) => array.byteLength > 0);
|
|
629
|
+
if (arrays.length === 0) {
|
|
630
|
+
return Object.freeze({});
|
|
631
|
+
}
|
|
622
632
|
const offsets: number[] = [];
|
|
623
633
|
let total = 0;
|
|
624
634
|
for (const [, array] of arrays) {
|
|
@@ -34,7 +34,8 @@ import { type Kernel } from "../kernel/kernel.js";
|
|
|
34
34
|
import { FRONTIER_PARAMS, graphBindings, graphOverrides, kernelSpec } from "../kernels.js";
|
|
35
35
|
import { type CoreBinding } from "../memory/residency.js";
|
|
36
36
|
import { type CoreWindow, coreWindows, rowCountOf } from "./core-shape.js";
|
|
37
|
-
import { type Frontier
|
|
37
|
+
import { type Frontier } from "./frontier.js";
|
|
38
|
+
import { type ReduceScope } from "./reduce.js";
|
|
38
39
|
|
|
39
40
|
/** A prepared advance (design 6 row 8): records one expansion of a frontier per level. */
|
|
40
41
|
export interface AdvancePlanner {
|
|
@@ -64,7 +65,7 @@ export interface AdvancePlanner {
|
|
|
64
65
|
* @param core - the resident core arrays of the snapshot the frontier walks (windowed or not)
|
|
65
66
|
* @returns the planner
|
|
66
67
|
*/
|
|
67
|
-
export async function prepareAdvance(scope:
|
|
68
|
+
export async function prepareAdvance(scope: ReduceScope, core: CoreBinding): Promise<AdvancePlanner> {
|
|
68
69
|
const kernel = await scope.pipelines.kernel(kernelSpec("advance-expand", graphOverrides(core, null)));
|
|
69
70
|
return new AdvancePlannerImpl(scope, core, kernel);
|
|
70
71
|
}
|
|
@@ -73,7 +74,7 @@ export async function prepareAdvance(scope: FrontierScope, core: CoreBinding): P
|
|
|
73
74
|
class AdvancePlannerImpl implements AdvancePlanner {
|
|
74
75
|
readonly kernel: Kernel;
|
|
75
76
|
readonly windows: readonly CoreWindow[];
|
|
76
|
-
private readonly scope:
|
|
77
|
+
private readonly scope: ReduceScope;
|
|
77
78
|
private readonly n: number;
|
|
78
79
|
|
|
79
80
|
/**
|
|
@@ -82,7 +83,7 @@ class AdvancePlannerImpl implements AdvancePlanner {
|
|
|
82
83
|
* @param core - the core the kernel was compiled for
|
|
83
84
|
* @param kernel - the `advance-expand` kernel
|
|
84
85
|
*/
|
|
85
|
-
constructor(scope:
|
|
86
|
+
constructor(scope: ReduceScope, core: CoreBinding, kernel: Kernel) {
|
|
86
87
|
this.scope = scope;
|
|
87
88
|
this.kernel = kernel;
|
|
88
89
|
this.windows = coreWindows(core);
|
|
@@ -1,46 +1,48 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The `Frontier` of design 6 row 7 and the device-side dispatch selector of design 5.4 (P8-T4; the P8 plan's PD-1,
|
|
3
|
-
* PD-3, PD-8, PD-23, DEP-P8-A, DEP-P8-C). A traversal's per-level state is two n-slot vertex queues, ONE
|
|
3
|
+
* PD-3, PD-8, PD-23, DEP-P8-A, DEP-P8-C). A traversal's per-level state is two n-slot vertex queues, ONE 25-word
|
|
4
4
|
* counters block (every counter of the phase is a word of it: a four-byte word is never a legal storage-binding
|
|
5
|
-
* offset, and the selector must reach every count it acts on through one binding)
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
5
|
+
* offset, and the selector must reach every count it acts on through one binding) and the edge queue. The host
|
|
6
|
+
* never reads a counter inside a submit: `frontier-finalize`, one lane, runs at the START of every level (role 0:
|
|
7
|
+
* rotates `nextFrontierCount` into `frontierCount`, advances `level`, decides `done`, chooses the level's path) and
|
|
8
|
+
* again once the edge queue is filled (role 1: clamps `edgeCount`, or on an overflow -- `edgeCountUnclamped >
|
|
9
|
+
* edgeCapacity` -- switches the path to the fused retry, PD-23). The choice is the block's `path` word (`W.path`,
|
|
10
|
+
* word 24); every level kernel is a direct grid-stride dispatch that reads it first and does nothing unless the
|
|
11
|
+
* word names it (design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md: the seven indirect slots the
|
|
12
|
+
* selector once wrote per level cost about 0.4 ms of Dawn validation each and were 97 % of a traversal's wall
|
|
13
|
+
* time; they and their args buffer are gone). The host records up to `MAX_LEVELS_PER_SUBMIT` levels per submit and
|
|
14
|
+
* reads `done` (four bytes) once per submit; a boundary that finds `done` set writes path 0 and moves no counter,
|
|
15
|
+
* so the recorded levels past the end are no-ops and the counters freeze at the finishing boundary's values.
|
|
14
16
|
*
|
|
15
17
|
* The seed: `frontierCount` is NEVER seeded, because the first boundary rotates it out unread. A BFS driver calls
|
|
16
18
|
* `reset(queue, source, { nextFrontierCount: 1, level: U32_MAX })`: the first boundary rotates the 1 in, adds it into
|
|
17
19
|
* `visitedCount`, and wraps `level` to 0, so the level-0 expansion claims the source's neighbours at `level + 1 == 1`.
|
|
18
20
|
* `reset` is a queue write, ordered before the submit that follows, and it puts the source on side 0.
|
|
19
21
|
*
|
|
20
|
-
* Every buffer comes from the caller's ONE lease (`scope.scratch
|
|
22
|
+
* Every buffer comes from the caller's ONE lease (`scope.scratch`) so the algorithm's dispose()
|
|
21
23
|
* releases them together (design 4.4). The two vertex queues are two BUFFERS, never two ranges of one: `Kernel.bind`
|
|
22
24
|
* rejects one buffer bound read-only and read-write in one dispatch even for disjoint ranges. `src/primitives/**`
|
|
23
25
|
* never imports `src/context.ts`.
|
|
24
26
|
*/
|
|
25
27
|
|
|
26
|
-
import {
|
|
28
|
+
import { MAX_LEVELS_PER_SUBMIT, U32_MAX } from "../constants.js";
|
|
27
29
|
import { WebGpuGraphError } from "../errors.js";
|
|
28
30
|
import { plan1d } from "../kernel/dispatch.js";
|
|
29
|
-
import {
|
|
31
|
+
import { type Kernel } from "../kernel/kernel.js";
|
|
30
32
|
import { FRONTIER_COUNTERS, FRONTIER_PARAMS, kernelSpec } from "../kernels.js";
|
|
31
33
|
import { type Binding } from "../types/memory.js";
|
|
32
34
|
import { type ReduceScope } from "./reduce.js";
|
|
33
35
|
|
|
34
|
-
/** The
|
|
35
|
-
export const
|
|
36
|
-
|
|
37
|
-
|
|
36
|
+
/** The values of the `path` word (`W.path`): what a level's kernels run; every level kernel reads it first. */
|
|
37
|
+
export const PATH: Readonly<{
|
|
38
|
+
none: 0;
|
|
39
|
+
twoPhase: 1;
|
|
38
40
|
fused: 2;
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
}> = Object.freeze({
|
|
41
|
+
bottomUp: 3;
|
|
42
|
+
fusedRetry: 4;
|
|
43
|
+
near: 5;
|
|
44
|
+
far: 6;
|
|
45
|
+
}> = Object.freeze({ none: 0, twoPhase: 1, fused: 2, bottomUp: 3, fusedRetry: 4, near: 5, far: 6 });
|
|
44
46
|
|
|
45
47
|
/** The words of the counters block (`FRONTIER_COUNTERS`), by index: byte offset 4 x word; no driver types a number. */
|
|
46
48
|
export const W: Readonly<{
|
|
@@ -69,6 +71,7 @@ export const W: Readonly<{
|
|
|
69
71
|
thresholdBits: 22;
|
|
70
72
|
deltaBits: 23;
|
|
71
73
|
path: 24;
|
|
74
|
+
nextDegreeSum: 25;
|
|
72
75
|
}> = Object.freeze({
|
|
73
76
|
frontierCount: 0,
|
|
74
77
|
nextFrontierCount: 1,
|
|
@@ -95,18 +98,13 @@ export const W: Readonly<{
|
|
|
95
98
|
thresholdBits: 22,
|
|
96
99
|
deltaBits: 23,
|
|
97
100
|
path: 24,
|
|
101
|
+
nextDegreeSum: 25,
|
|
98
102
|
});
|
|
99
103
|
|
|
100
104
|
/** The words a `reset` seeds (every other word is zeroed). */
|
|
101
105
|
export type FrontierSeed = Readonly<Partial<Record<keyof typeof W, number>>>;
|
|
102
106
|
|
|
103
|
-
/**
|
|
104
|
-
export interface FrontierScope extends ReduceScope {
|
|
105
|
-
/** A STORAGE | INDIRECT | COPY_DST | COPY_SRC buffer of the scope's lease. */
|
|
106
|
-
indirect(byteLength: number, label: string): GPUBuffer;
|
|
107
|
-
}
|
|
108
|
-
|
|
109
|
-
/** The `FrontierParams` fields a caller passes to `recordFinalize`; the planner fills `role`, `slotBase`, `wg`, `edgeCapacity` and `n` itself. A missing field is written as 0. */
|
|
107
|
+
/** The `FrontierParams` fields a caller passes to `recordFinalize`; the planner fills `role`, `wg`, `edgeCapacity` and `n` itself. A missing field is written as 0. */
|
|
110
108
|
export type FrontierFinalizeFields = Readonly<
|
|
111
109
|
Partial<
|
|
112
110
|
Record<
|
|
@@ -163,14 +161,12 @@ function assertCount(argument: string, value: number): void {
|
|
|
163
161
|
}
|
|
164
162
|
}
|
|
165
163
|
|
|
166
|
-
/** The frontier queue of design 6 row 7: two vertex queues, the counters block
|
|
164
|
+
/** The frontier queue of design 6 row 7: two vertex queues, the counters block and the edge queue, all leased by the caller's scope. */
|
|
167
165
|
export class Frontier {
|
|
168
166
|
/** The two n-slot u32 vertex queues (`vertices[side]` is the input of the current level). */
|
|
169
167
|
readonly vertices: readonly [Binding, Binding];
|
|
170
|
-
/** The `FrontierCounters` block,
|
|
168
|
+
/** The `FrontierCounters` block, 112 B, bound by every kernel as `array<atomic<u32>>`. */
|
|
171
169
|
readonly counters: Binding;
|
|
172
|
-
/** The indirect args, `MAX_LEVELS_PER_SUBMIT x FRONTIER_CANDIDATES` 16-byte slots (usage INDIRECT | STORAGE | COPY_DST | COPY_SRC). */
|
|
173
|
-
readonly args: Binding;
|
|
174
170
|
/** The edge queue: `edgeCapacity` entries of u32 (the target vertex of an arc). */
|
|
175
171
|
readonly edgeQueue: Binding;
|
|
176
172
|
/** How many entries the edge queue holds; role 1 clamps `edgeCount` to it and detects an overflow above it. */
|
|
@@ -183,7 +179,6 @@ export class Frontier {
|
|
|
183
179
|
* Wraps the leased buffers; use prepareFrontier().
|
|
184
180
|
* @param vertices - the two vertex queues
|
|
185
181
|
* @param counters - the counters block
|
|
186
|
-
* @param args - the args buffer
|
|
187
182
|
* @param edgeQueue - the edge queue
|
|
188
183
|
* @param edgeCapacity - the edge queue's entry count
|
|
189
184
|
* @param n - the vertex count
|
|
@@ -191,14 +186,12 @@ export class Frontier {
|
|
|
191
186
|
constructor(
|
|
192
187
|
vertices: readonly [Binding, Binding],
|
|
193
188
|
counters: Binding,
|
|
194
|
-
args: Binding,
|
|
195
189
|
edgeQueue: Binding,
|
|
196
190
|
edgeCapacity: number,
|
|
197
191
|
n: number,
|
|
198
192
|
) {
|
|
199
193
|
this.vertices = vertices;
|
|
200
194
|
this.counters = counters;
|
|
201
|
-
this.args = args;
|
|
202
195
|
this.edgeQueue = edgeQueue;
|
|
203
196
|
this.edgeCapacity = edgeCapacity;
|
|
204
197
|
this.n = n;
|
|
@@ -234,7 +227,7 @@ export class Frontier {
|
|
|
234
227
|
}
|
|
235
228
|
|
|
236
229
|
/**
|
|
237
|
-
* Seeds a traversal: one `queue.writeBuffer` of the whole
|
|
230
|
+
* Seeds a traversal: one `queue.writeBuffer` of the whole 112-byte block (zero except the caller's words) and one of
|
|
238
231
|
* `vertices[0][0] = source`, both ordered before the submit that follows; the source is on side 0 afterwards.
|
|
239
232
|
* `frontierCount` is not a word to seed: the first boundary rotates word 1 into it (the BFS seed is
|
|
240
233
|
* `{ nextFrontierCount: 1, level: U32_MAX }`). A source outside `[0, n)`, an unknown word or a value that is not a
|
|
@@ -261,13 +254,14 @@ export class Frontier {
|
|
|
261
254
|
|
|
262
255
|
/** A prepared frontier (design 6 row 7): the leased queue and the recorded selector dispatches. */
|
|
263
256
|
export interface FrontierPlanner {
|
|
264
|
-
/** The queue the planner
|
|
257
|
+
/** The queue the planner's selector rotates and chooses the path of. */
|
|
265
258
|
readonly frontier: Frontier;
|
|
266
259
|
/**
|
|
267
260
|
* Records one `frontier-finalize` dispatch (one workgroup) in `role` for `level` of the current submit: one
|
|
268
|
-
* `FrontierParams` record with `
|
|
269
|
-
*
|
|
270
|
-
* `[0, 3]` is E_INVALID_ARGUMENT before
|
|
261
|
+
* `FrontierParams` record with `wg`, `edgeCapacity` and `n` filled by the planner and every other field from
|
|
262
|
+
* `fields`. A level outside `[0, MAX_LEVELS_PER_SUBMIT)` (the selector addresses nothing by level any more, but
|
|
263
|
+
* the host's submit cadence still is the bound) or a role outside `[0, 3]` is E_INVALID_ARGUMENT before
|
|
264
|
+
* anything is recorded.
|
|
271
265
|
* @param pass - the compute pass
|
|
272
266
|
* @param role - 0 the level boundary, 1 the edge-queue role (2 and 3 are P8-T9's)
|
|
273
267
|
* @param level - the level inside the submit
|
|
@@ -281,14 +275,14 @@ export interface FrontierPlanner {
|
|
|
281
275
|
* edge capacity defaults to `max(1, min(arcCount, floor(maxStorageBufferBindingSize / 4)))` (never a zero-length
|
|
282
276
|
* buffer: the one-node graph has no arcs); a test passes a small one to force the overflow path. The planner lives
|
|
283
277
|
* exactly as long as the scope: never use it after the scope's dispose().
|
|
284
|
-
* @param scope - the caller's scope (device, caps, cache, scratch,
|
|
278
|
+
* @param scope - the caller's scope (device, caps, cache, scratch, params)
|
|
285
279
|
* @param n - the vertex count
|
|
286
280
|
* @param arcCount - the arc count (the natural edge-queue size)
|
|
287
281
|
* @param edgeCapacity - the edge queue's entry count, when the caller chooses it (an integer >= 1)
|
|
288
282
|
* @returns the planner
|
|
289
283
|
*/
|
|
290
284
|
export async function prepareFrontier(
|
|
291
|
-
scope:
|
|
285
|
+
scope: ReduceScope,
|
|
292
286
|
n: number,
|
|
293
287
|
arcCount: number,
|
|
294
288
|
edgeCapacity?: number,
|
|
@@ -306,7 +300,6 @@ export async function prepareFrontier(
|
|
|
306
300
|
}
|
|
307
301
|
const kernel = await scope.pipelines.kernel(kernelSpec("frontier-finalize"));
|
|
308
302
|
const queueBytes = 4 * Math.max(1, n);
|
|
309
|
-
const argsBytes = MAX_LEVELS_PER_SUBMIT * FRONTIER_CANDIDATES * INDIRECT_ARGS_STRIDE;
|
|
310
303
|
const vertices: readonly [Binding, Binding] = [
|
|
311
304
|
{ buffer: scope.scratch(queueBytes, "frontier/vertices-0"), offset: 0, size: queueBytes, window: null },
|
|
312
305
|
{ buffer: scope.scratch(queueBytes, "frontier/vertices-1"), offset: 0, size: queueBytes, window: null },
|
|
@@ -317,26 +310,20 @@ export async function prepareFrontier(
|
|
|
317
310
|
size: FRONTIER_COUNTERS.byteLength,
|
|
318
311
|
window: null,
|
|
319
312
|
};
|
|
320
|
-
const args: Binding = {
|
|
321
|
-
buffer: scope.indirect(argsBytes, "frontier/args"),
|
|
322
|
-
offset: 0,
|
|
323
|
-
size: argsBytes,
|
|
324
|
-
window: null,
|
|
325
|
-
};
|
|
326
313
|
const edgeQueue: Binding = {
|
|
327
314
|
buffer: scope.scratch(4 * capacity, "frontier/edge-queue"),
|
|
328
315
|
offset: 0,
|
|
329
316
|
size: 4 * capacity,
|
|
330
317
|
window: null,
|
|
331
318
|
};
|
|
332
|
-
const frontier = new Frontier(vertices, counters,
|
|
319
|
+
const frontier = new Frontier(vertices, counters, edgeQueue, capacity, n);
|
|
333
320
|
return new FrontierPlannerImpl(scope, kernel, frontier);
|
|
334
321
|
}
|
|
335
322
|
|
|
336
|
-
/** The planner: the selector kernel bound once to the frontier's block
|
|
323
|
+
/** The planner: the selector kernel bound once to the frontier's block. */
|
|
337
324
|
class FrontierPlannerImpl implements FrontierPlanner {
|
|
338
325
|
readonly frontier: Frontier;
|
|
339
|
-
private readonly scope:
|
|
326
|
+
private readonly scope: ReduceScope;
|
|
340
327
|
private readonly kernel: Kernel;
|
|
341
328
|
|
|
342
329
|
/**
|
|
@@ -345,7 +332,7 @@ class FrontierPlannerImpl implements FrontierPlanner {
|
|
|
345
332
|
* @param kernel - the `frontier-finalize` kernel
|
|
346
333
|
* @param frontier - the leased queue
|
|
347
334
|
*/
|
|
348
|
-
constructor(scope:
|
|
335
|
+
constructor(scope: ReduceScope, kernel: Kernel, frontier: Frontier) {
|
|
349
336
|
this.scope = scope;
|
|
350
337
|
this.kernel = kernel;
|
|
351
338
|
this.frontier = frontier;
|
|
@@ -377,12 +364,11 @@ class FrontierPlannerImpl implements FrontierPlanner {
|
|
|
377
364
|
const params = scope.params(FRONTIER_PARAMS, {
|
|
378
365
|
...definedWords(fields),
|
|
379
366
|
role,
|
|
380
|
-
slotBase: level * FRONTIER_CANDIDATES,
|
|
381
367
|
wg: scope.workgroupSize,
|
|
382
368
|
edgeCapacity: frontier.edgeCapacity,
|
|
383
369
|
n: frontier.n,
|
|
384
370
|
});
|
|
385
|
-
const bound = this.kernel.bind({ counters: frontier.counters,
|
|
371
|
+
const bound = this.kernel.bind({ counters: frontier.counters, P: params.binding });
|
|
386
372
|
this.kernel.dispatch(pass, bound, plan1d(1, scope.workgroupSize, scope.caps), [params.offset]);
|
|
387
373
|
}
|
|
388
374
|
}
|
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The grid pyramid (spec 6 row 12, 7.7 G4-G5; P4-T9): the planner that records, into the caller's pass, the finest
|
|
3
|
-
* centroids (G4, `grid-centroid`: thread per cell over `cells +
|
|
3
|
+
* centroids (G4, `grid-centroid`: thread per cell over `cells + outsideCells`, the pseudo-cells included), the hub-cell
|
|
4
4
|
* completion (G4a: the T1 `indirect-finalize` over `hubCounters[0]` into `hubArgs` with `wg = 1`, so the finalize's
|
|
5
5
|
* `ceil(count / wg)` is ONE workgroup per hub cell; G4b: `grid-centroid-hub`, one workgroup per hub cell, dispatched
|
|
6
6
|
* indirectly; PD-13, DEP-P4-I) and one `grid-downsample` dispatch per coarser
|
|
7
7
|
* level (G5). Level 0 holds `[sum m x, sum m y, sum m z, sum m]` per cell; every parent is the sum of its 2^dim
|
|
8
|
-
* children; the pseudo-
|
|
8
|
+
* children; the pseudo-cells (indices `cells ..` of level 0) are never children. No atomics touch the sums (design 6 row 12:
|
|
9
9
|
* bitwise reproducible); the only atomics are the hub append and the occupancy max.
|
|
10
10
|
*
|
|
11
11
|
* The named grid buffers (`pyramid`, `hubList`, `hubCounters`, `hubArgs`) are the caller's (the model's
|
|
@@ -34,9 +34,9 @@ export interface GridPyramidBindings {
|
|
|
34
34
|
readonly params: Binding;
|
|
35
35
|
/** `n` words: the sorted node indices (the T8 build). */
|
|
36
36
|
readonly sortedIdx: Binding;
|
|
37
|
-
/** `cells + 2` words: the exclusive scan of the cell histogram (the T8 build). */
|
|
37
|
+
/** `histWords` (`cells + 2^dim + 1`) words: the exclusive scan of the cell histogram (the T8 build). */
|
|
38
38
|
readonly cellStart: Binding;
|
|
39
|
-
/** `pyramidCells` vec4f: every level, level 0 first with the pseudo-
|
|
39
|
+
/** `pyramidCells` vec4f: every level, level 0 first with the 2^dim orthant pseudo-cells from index `cells`. */
|
|
40
40
|
readonly pyramid: Binding;
|
|
41
41
|
/** The hub cells' indices, appended by G4 (at least one word; at most `floor(n / (GRID_HUB_CELL + 1))` are ever written, so `ceil(n / GRID_HUB_CELL)` words always suffice). */
|
|
42
42
|
readonly hubList: Binding;
|
|
@@ -204,7 +204,8 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
|
|
|
204
204
|
}
|
|
205
205
|
const { centroid, finalize, hub, downsample } = this.kernels;
|
|
206
206
|
const one: DispatchPlan = { x: 1, y: 1, z: 1, items: 1, stride: null };
|
|
207
|
-
|
|
207
|
+
const level0 = spec.cells + spec.outsideCells;
|
|
208
|
+
centroid.dispatch(pass, bound.centroid, plan1d(level0, scope.workgroupSize, scope.caps), [paramsOffset]);
|
|
208
209
|
finalize.dispatch(pass, bound.finalize, one, [bound.finalizeOffset]);
|
|
209
210
|
hub.dispatchIndirect(pass, bound.hub, bound.hubArgs, 0, [paramsOffset]);
|
|
210
211
|
this.dispatches = 3;
|
package/src/primitives/grid.ts
CHANGED
|
@@ -3,8 +3,9 @@
|
|
|
3
3
|
* the caller's pass, the cell keys (G1, `grid-cell-key`), the stable sort by key (G2: `radixSort` at GRID_SORT_BITS,
|
|
4
4
|
* or `countingSortByKey` when the caller asks for the set-deterministic path) and the per-cell histogram with its
|
|
5
5
|
* exclusive scan (G3: the `histogram` kernel over `cellKey` and the `scan` of it; DEP-P4-I names no grid-specific
|
|
6
|
-
* id). `cellHist` and `cellStart` hold `cells + 2` words: every real cell, the outside
|
|
7
|
-
*
|
|
6
|
+
* id). `cellHist` and `cellStart` hold `histWords = cells + 2^dim + 1` words: every real cell, the 2^dim outside
|
|
7
|
+
* pseudo-cells (one per orthant about the grid centre, issue #90) from index `cells`, and one more so
|
|
8
|
+
* `cellStart[histWords - 1] === n` closes the last range. Every zeroing is a `fill` dispatch inside the
|
|
8
9
|
* pass (PD-12), never an encoder clear.
|
|
9
10
|
*
|
|
10
11
|
* The named grid buffers (`cellKey`, `cellVal`, `sortedKey`, `sortedIdx`, `cellHist`, `cellStart`) are the caller's
|
|
@@ -39,11 +40,13 @@ export interface GridSpec {
|
|
|
39
40
|
readonly g: number;
|
|
40
41
|
/** `log2(G / GRID_COARSEST_SIDE) + 1`. */
|
|
41
42
|
readonly levels: number;
|
|
42
|
-
/** `G^dim` finest cells; the outside pseudo-
|
|
43
|
+
/** `G^dim` finest cells; the outside pseudo-cells are indices `cells .. cells + outsideCells - 1`. */
|
|
43
44
|
readonly cells: number;
|
|
44
|
-
/** `
|
|
45
|
+
/** `2^dim`: one outside pseudo-cell per orthant about the grid centre (issue #90). */
|
|
46
|
+
readonly outsideCells: number;
|
|
47
|
+
/** `cells + outsideCells + 1`: the length of `cellHist` / `cellStart`. */
|
|
45
48
|
readonly histWords: number;
|
|
46
|
-
/** The first cell of every level inside the pyramid: `levelOffsets[0] = 0`, level 0 holds `cells +
|
|
49
|
+
/** The first cell of every level inside the pyramid: `levelOffsets[0] = 0`, level 0 holds `cells + outsideCells` (the pseudo-cells last), level L `(G / 2^L)^dim`. */
|
|
47
50
|
readonly levelOffsets: readonly number[];
|
|
48
51
|
/** Every level's cells together: `levelOffsets[levels - 1] + GRID_COARSEST_SIDE^dim`. */
|
|
49
52
|
readonly pyramidCells: number;
|
|
@@ -81,8 +84,8 @@ function floorPow2(x: number): number {
|
|
|
81
84
|
* The grid of `n` nodes in `dim` dimensions under the tuning (spec 7.7 geometry table; PD-9): `G = clamp(nextPow2(2 *
|
|
82
85
|
* ceil(n^(1 / dim))), GRID_MIN_SIDE, floorPow2(gridMax))` where `gridMax` is `gridMax2D` or `gridMax3D`, rounded DOWN
|
|
83
86
|
* to a power of two so every level's side is an integer (512 and 128 stay; 100 becomes 64); `levels = log2(G /
|
|
84
|
-
* GRID_COARSEST_SIDE) + 1`. At the caps: 349,
|
|
85
|
-
* pseudo-
|
|
87
|
+
* GRID_COARSEST_SIDE) + 1`. At the caps: 349,524 pyramid cells in 2D, 2,396,744 in 3D (the design's counts plus the
|
|
88
|
+
* 2^dim pseudo-cells).
|
|
86
89
|
* @param n - the node count (>= 0)
|
|
87
90
|
* @param dim - 2 or 3
|
|
88
91
|
* @param tuning - the resolved layout tuning (`gridMax2D`, `gridMax3D`, `deterministic`)
|
|
@@ -102,10 +105,11 @@ export function gridSpecFor(
|
|
|
102
105
|
levels++;
|
|
103
106
|
}
|
|
104
107
|
const cells = g ** dim;
|
|
108
|
+
const outsideCells = 2 ** dim;
|
|
105
109
|
const levelOffsets: number[] = [0];
|
|
106
110
|
let s = g;
|
|
107
111
|
for (let level = 0; level + 1 < levels; level++) {
|
|
108
|
-
levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ?
|
|
112
|
+
levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? outsideCells : 0));
|
|
109
113
|
s /= 2;
|
|
110
114
|
}
|
|
111
115
|
return {
|
|
@@ -113,7 +117,8 @@ export function gridSpecFor(
|
|
|
113
117
|
g,
|
|
114
118
|
levels,
|
|
115
119
|
cells,
|
|
116
|
-
|
|
120
|
+
outsideCells,
|
|
121
|
+
histWords: cells + outsideCells + 1,
|
|
117
122
|
levelOffsets: Object.freeze(levelOffsets),
|
|
118
123
|
pyramidCells: levelOffsets[levels - 1] + GRID_COARSEST_SIDE ** dim,
|
|
119
124
|
deterministic: tuning.deterministic,
|
|
@@ -121,7 +126,7 @@ export function gridSpecFor(
|
|
|
121
126
|
}
|
|
122
127
|
|
|
123
128
|
/**
|
|
124
|
-
* The bytes of the pyramid (spec 7.7: 16 B per cell, every level, the pseudo-
|
|
129
|
+
* The bytes of the pyramid (spec 7.7: 16 B per cell, every level, the pseudo-cells included): 38,347,904 at the 3D cap.
|
|
125
130
|
* @param spec - the grid
|
|
126
131
|
* @returns the byte length
|
|
127
132
|
*/
|
|
@@ -155,9 +160,9 @@ export interface GridBuildBindings {
|
|
|
155
160
|
readonly sortedKey: Binding;
|
|
156
161
|
/** `n` words: the sorted node indices. */
|
|
157
162
|
readonly sortedIdx: Binding;
|
|
158
|
-
/** `cells + 2` words: the per-cell counts. */
|
|
163
|
+
/** `histWords` (`cells + 2^dim + 1`) words: the per-cell counts. */
|
|
159
164
|
readonly cellHist: Binding;
|
|
160
|
-
/** `cells + 2` words: the exclusive scan of `cellHist`. */
|
|
165
|
+
/** `histWords` (`cells + 2^dim + 1`) words: the exclusive scan of `cellHist`. */
|
|
161
166
|
readonly cellStart: Binding;
|
|
162
167
|
}
|
|
163
168
|
|
|
@@ -7,8 +7,9 @@
|
|
|
7
7
|
* the twin's two compilations -- and then every invocation strips the range `[0, aggregate)` with a binary search
|
|
8
8
|
* (`upper_bound`) over the scanned degrees to find which entry its arc belongs to. One `atomicAdd` per WORKGROUP
|
|
9
9
|
* reserves the block's span in the queue (`edgeCount`), the same aggregate lands in `edgeCountUnclamped` (the overflow
|
|
10
|
-
* detector, never clamped) and in `frontierDegreeSum` (
|
|
11
|
-
* `
|
|
10
|
+
* detector, never clamped) and in `frontierDegreeSum` (the inspect seam's per-level expansion count, rotated into
|
|
11
|
+
* `prevDegreeSum` by the boundary; Beamer's m_f is `nextDegreeSum`, measured by `bfs-next-degree` -- issue #391);
|
|
12
|
+
* a lane whose queue position is at or past `P.edgeCapacity` writes nothing (the clamp). The queue holds the TARGET vertex of each arc only (PD-24: `parent`
|
|
12
13
|
* comes from the post-pass). Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan call and the
|
|
13
14
|
* `workgroupUniformLoad` sit unconditionally after the guard, and the strip loop is bounded by a uniform value.
|
|
14
15
|
* There is no `TIER` override: a hub row is balanced over all `WG` lanes inside its block, and the small-frontier
|
|
@@ -44,7 +45,7 @@ fn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
|
|
|
44
45
|
if (lid.x == 0u) {
|
|
45
46
|
base = atomicAdd(&counters[8], aggregate); // edgeCount: ONE reservation per workgroup, not one per arc
|
|
46
47
|
atomicAdd(&counters[9], aggregate); // edgeCountUnclamped: the overflow detector (PD-23)
|
|
47
|
-
atomicAdd(&counters[2], aggregate); // frontierDegreeSum:
|
|
48
|
+
atomicAdd(&counters[2], aggregate); // frontierDegreeSum: what this level expanded (the inspect seam)
|
|
48
49
|
}
|
|
49
50
|
workgroupBarrier();
|
|
50
51
|
for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
|
|
@@ -11,9 +11,10 @@
|
|
|
11
11
|
* The winners are packed into the output vertex queue by `bfs-contract`'s workgroup scan and one `atomicAdd` per
|
|
12
12
|
* workgroup on `nextFrontierCount`, and claim with a plain `atomicStore`: the list holds every vertex once and the
|
|
13
13
|
* sweep is vertex-parallel, so no two lanes claim one vertex. It adds nothing to `frontierDegreeSum` (a bottom-up
|
|
14
|
-
* level expands nothing), which is why
|
|
15
|
-
*
|
|
16
|
-
*
|
|
14
|
+
* level expands nothing), which is why that word is not Beamer's m_f: `bfs-next-degree` sums the degree of what
|
|
15
|
+
* this sweep CLAIMS into `nextDegreeSum`, so the boundary's test and its `unvisitedDegreeSum` subtraction are exact
|
|
16
|
+
* on a bottom-up level like any other (issue #391; the selector's JSDoc). Uniformity (spec 3.5 rule 1): the
|
|
17
|
+
* guarded walk writes locals, the scan and the reduction run unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
|
|
17
18
|
* test/helpers/sabotage.ts are textual edits of it.
|
|
18
19
|
*/
|
|
19
20
|
export const bfsBottomUpWgsl = /* wgsl */ `
|
|
@@ -4,15 +4,15 @@
|
|
|
4
4
|
* ONE dispatch, chosen by `frontier-finalize` for a frontier below `P.fusedMax` entries (`path` 2) and for the retry
|
|
5
5
|
* of a level whose edge queue overflowed (`path` 4, PD-23). One WORKGROUP per frontier entry, the workgroups striding
|
|
6
6
|
* the entries by the dispatch's group count (`P.stride`): lane 0 reads the entry's row clipped
|
|
7
|
-
* to the bound arc window, adds its degree to `frontierDegreeSum` (
|
|
8
|
-
* too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim
|
|
9
|
-
* `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
|
|
7
|
+
* to the bound arc window, adds its degree to `frontierDegreeSum` (so the inspect seam's per-level expansion count
|
|
8
|
+
* covers fused levels too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim
|
|
9
|
+
* inline -- `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
|
|
10
10
|
* packing the strip's winners into the output vertex queue by the same Hillis-Steele scan and one `atomicAdd` per
|
|
11
11
|
* strip on `nextFrontierCount`. No edge queue is written or read, which is the whole win for a tiny frontier
|
|
12
12
|
* (Merrill's fleeting iterations) and what makes the overflow retry exact: the partial edge queue is never consulted.
|
|
13
13
|
* On a retry level `advance-expand` has already added the frontier's degree to `frontierDegreeSum`, so that word
|
|
14
|
-
* holds
|
|
15
|
-
*
|
|
14
|
+
* holds twice the level's expanded degree; since issue #391 no decision reads it (Beamer's m_f is `nextDegreeSum`
|
|
15
|
+
* and the boundary subtracts that), so the doubling only reaches the inspect seam's `prevDegreeSum`. Nothing here
|
|
16
16
|
* writes a parent (PD-24: the post-pass does), which is what keeps the kernel at the eight-storage-buffer budget
|
|
17
17
|
* with the four graph slots. Uniformity (spec 3.5 rule 1): the strip loop's bound and the row start are
|
|
18
18
|
* `workgroupUniformLoad`s, so every barrier of the per-strip append is in uniform control flow; the guarded claim
|
|
@@ -44,7 +44,7 @@ fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
|
|
|
44
44
|
let d = select(0u, a1 - a0, a1 > a0);
|
|
45
45
|
wdeg = d;
|
|
46
46
|
wstart = a0;
|
|
47
|
-
atomicAdd(&counters[2], d); // frontierDegreeSum, so
|
|
47
|
+
atomicAdd(&counters[2], d); // frontierDegreeSum, so the inspect seam sees fused levels too
|
|
48
48
|
}
|
|
49
49
|
let deg = workgroupUniformLoad(&wdeg); // uniform: the loop below may hold barriers
|
|
50
50
|
let start = workgroupUniformLoad(&wstart);
|