@graphty/webgpu-graph-algorithms 0.6.5 → 0.6.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +38 -17
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-hzGggHeM.js → context-Cezi7qpi.js} +46 -22
- package/dist/chunks/context-Cezi7qpi.js.map +1 -0
- package/dist/node.js +19 -11
- package/dist/node.js.map +1 -1
- package/dist/src/algorithms/bfs.d.ts +10 -6
- package/dist/src/algorithms/bfs.d.ts.map +1 -1
- package/dist/src/algorithms/bfs.js +32 -9
- package/dist/src/algorithms/bfs.js.map +1 -1
- package/dist/src/algorithms/pagerank.d.ts.map +1 -1
- package/dist/src/algorithms/pagerank.js +19 -5
- package/dist/src/algorithms/pagerank.js.map +1 -1
- package/dist/src/algorithms/power-iteration.d.ts.map +1 -1
- package/dist/src/algorithms/power-iteration.js +8 -2
- package/dist/src/algorithms/power-iteration.js.map +1 -1
- package/dist/src/constants.d.ts +33 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +33 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/kernel/dispatch.d.ts +2 -2
- package/dist/src/kernel/dispatch.d.ts.map +1 -1
- package/dist/src/kernel/kernel.d.ts +1 -1
- package/dist/src/kernel/kernel.js +2 -2
- package/dist/src/kernel/kernel.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +2 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernels.d.ts +10 -7
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +33 -9
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
- package/dist/src/layouts/forceatlas2.js +2 -1
- package/dist/src/layouts/forceatlas2.js.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
- package/dist/src/layouts/fruchterman-reingold.js +4 -2
- package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
- package/dist/src/layouts/repulsion-exact.d.ts +16 -0
- package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
- package/dist/src/layouts/repulsion-exact.js +21 -1
- package/dist/src/layouts/repulsion-exact.js.map +1 -1
- package/dist/src/layouts/repulsion-grid.d.ts +1 -1
- package/dist/src/layouts/repulsion-grid.js +1 -1
- package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
- package/dist/src/layouts/spring-electrical.js +6 -2
- package/dist/src/layouts/spring-electrical.js.map +1 -1
- package/dist/src/memory/residency.js +14 -4
- package/dist/src/memory/residency.js.map +1 -1
- package/dist/src/node/index.d.ts +13 -8
- package/dist/src/node/index.d.ts.map +1 -1
- package/dist/src/node/index.js +36 -17
- package/dist/src/node/index.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +1 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -1
- package/dist/src/primitives/frontier.js +1 -0
- package/dist/src/primitives/frontier.js.map +1 -1
- package/dist/src/primitives/grid-pyramid.d.ts +4 -4
- package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
- package/dist/src/primitives/grid-pyramid.js +4 -3
- package/dist/src/primitives/grid-pyramid.js.map +1 -1
- package/dist/src/primitives/grid.d.ts +13 -10
- package/dist/src/primitives/grid.d.ts.map +1 -1
- package/dist/src/primitives/grid.js +10 -7
- package/dist/src/primitives/grid.js.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
- package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
- package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
- package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +21 -20
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/frontier-finalize.wgsl.js +26 -25
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
- package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
- package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
- package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
- package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
- package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
- package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
- package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
- package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
- package/dist/src/wgsl/histogram.wgsl.js +1 -1
- package/dist/webgpu-graph-algorithms.js +144 -45
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +3 -3
- package/src/algorithms/bfs.ts +33 -9
- package/src/algorithms/pagerank.ts +19 -5
- package/src/algorithms/power-iteration.ts +8 -2
- package/src/constants.ts +35 -0
- package/src/kernel/dispatch.ts +2 -2
- package/src/kernel/kernel.ts +2 -2
- package/src/kernel/prelude.ts +2 -0
- package/src/kernels.ts +35 -9
- package/src/layouts/forceatlas2.ts +2 -0
- package/src/layouts/fruchterman-reingold.ts +4 -1
- package/src/layouts/repulsion-exact.ts +29 -1
- package/src/layouts/repulsion-grid.ts +1 -1
- package/src/layouts/spring-electrical.ts +8 -1
- package/src/memory/residency.ts +14 -4
- package/src/node/index.ts +42 -18
- package/src/primitives/frontier.ts +2 -0
- package/src/primitives/grid-pyramid.ts +6 -5
- package/src/primitives/grid.ts +17 -12
- package/src/wgsl/advance-expand.wgsl.ts +4 -3
- package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
- package/src/wgsl/bfs-fused.wgsl.ts +6 -6
- package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
- package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
- package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
- package/src/wgsl/frontier-finalize.wgsl.ts +26 -25
- package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
- package/src/wgsl/grid-centroid.wgsl.ts +3 -2
- package/src/wgsl/grid-downsample.wgsl.ts +1 -1
- package/src/wgsl/grid-far-field.wgsl.ts +15 -8
- package/src/wgsl/grid-near-field.wgsl.ts +6 -2
- package/src/wgsl/histogram.wgsl.ts +1 -1
- package/dist/chunks/context-hzGggHeM.js.map +0 -1
package/src/memory/residency.ts
CHANGED
|
@@ -579,7 +579,9 @@ export class GraphResidency {
|
|
|
579
579
|
}
|
|
580
580
|
|
|
581
581
|
/**
|
|
582
|
-
* One resident per array (spec 4.3: views upload in perArray mode, never into the arena).
|
|
582
|
+
* One resident per array (spec 4.3: views upload in perArray mode, never into the arena). An empty array (the
|
|
583
|
+
* colIdx of an edgeless directed reverse view, the src / dst of an edgeless edgeList) is skipped: spec 5.6 never
|
|
584
|
+
* uploads a zero-length array, and Kernel.bind rejects a zero-size binding, so it is absent as in core().
|
|
583
585
|
* @param record - the owning record
|
|
584
586
|
* @param arrays - the named arrays
|
|
585
587
|
* @param label - the buffer label prefix
|
|
@@ -592,6 +594,9 @@ export class GraphResidency {
|
|
|
592
594
|
): Readonly<Record<string, Binding>> {
|
|
593
595
|
const bindings: Record<string, Binding> = {};
|
|
594
596
|
for (const [name, array] of arrays) {
|
|
597
|
+
if (array.byteLength === 0) {
|
|
598
|
+
continue;
|
|
599
|
+
}
|
|
595
600
|
const resident = this.upload(record, array, array, `${label}:${name}`);
|
|
596
601
|
bindings[name] = { buffer: resident.buffer, offset: 0, size: resident.byteLength, window: null };
|
|
597
602
|
}
|
|
@@ -606,19 +611,24 @@ export class GraphResidency {
|
|
|
606
611
|
* lengths, so keying the packed buffer on `rev.rowPtr` would make the packed and the unpacked view of one
|
|
607
612
|
* snapshot collide -- whichever was built second would get the other's buffer. The record still owns the
|
|
608
613
|
* resident, so release(s) destroys it with the rest. Offsets are STORAGE_ALIGN-aligned because Kernel.bind
|
|
609
|
-
* rejects any other offset synchronously (E_INVALID_ARGUMENT { argument: "offset" }).
|
|
614
|
+
* rejects any other offset synchronously (E_INVALID_ARGUMENT { argument: "offset" }). Empty arrays are left
|
|
615
|
+
* out as in separateArrays; when nothing is left, nothing is uploaded.
|
|
610
616
|
* @param record - the owning record
|
|
611
|
-
* @param
|
|
617
|
+
* @param all - the named arrays, in buffer order
|
|
612
618
|
* @param key - the marker object the resident is keyed on
|
|
613
619
|
* @param label - the buffer label
|
|
614
620
|
* @returns the bindings by name, all into the one buffer
|
|
615
621
|
*/
|
|
616
622
|
private packArrays(
|
|
617
623
|
record: ResidencyRecord,
|
|
618
|
-
|
|
624
|
+
all: readonly (readonly [string, TypedArrayData])[],
|
|
619
625
|
key: object,
|
|
620
626
|
label: string,
|
|
621
627
|
): Readonly<Record<string, Binding>> {
|
|
628
|
+
const arrays = all.filter(([, array]) => array.byteLength > 0);
|
|
629
|
+
if (arrays.length === 0) {
|
|
630
|
+
return Object.freeze({});
|
|
631
|
+
}
|
|
622
632
|
const offsets: number[] = [];
|
|
623
633
|
let total = 0;
|
|
624
634
|
for (const [, array] of arrays) {
|
package/src/node/index.ts
CHANGED
|
@@ -32,11 +32,15 @@ export interface NodeGpuOptions extends Omit<GpuContextOptions, "gpu" | "adapter
|
|
|
32
32
|
readonly loadModule?: (() => Promise<unknown>) | undefined;
|
|
33
33
|
}
|
|
34
34
|
|
|
35
|
-
/**
|
|
35
|
+
/**
|
|
36
|
+
* The Dawn GPU handle (spec 2.3). The Dawn instance behind `gpu` is shared by every handle created with the
|
|
37
|
+
* same flags and lives until the process exits (see `createNodeGpu`); it holds no event-loop handle, so it
|
|
38
|
+
* never keeps the process alive.
|
|
39
|
+
*/
|
|
36
40
|
export interface NodeGpuHandle {
|
|
37
41
|
/** The GPU of `dawn.create(flags)`; reading it after dispose() throws E_DISPOSED. */
|
|
38
42
|
readonly gpu: GPU;
|
|
39
|
-
/** Drops
|
|
43
|
+
/** Drops this handle's GPU reference (the shared instance stays alive); idempotent. */
|
|
40
44
|
dispose(): void;
|
|
41
45
|
}
|
|
42
46
|
|
|
@@ -49,6 +53,18 @@ interface DawnModule {
|
|
|
49
53
|
globals?: unknown;
|
|
50
54
|
}
|
|
51
55
|
|
|
56
|
+
/**
|
|
57
|
+
* Every GPU object `dawn.create()` returned, per module and flag list, kept for the life of the process.
|
|
58
|
+
* webgpu@0.4.0's adapters, devices and queues run their promises through an AsyncRunner that polls the Dawn
|
|
59
|
+
* instance by RAW pointer, and only the GPU object owns that instance: once the GPU object is collected, the
|
|
60
|
+
* next promise on any adapter or device it produced (a requestDevice on a probed adapter, a queue call or a
|
|
61
|
+
* late map / lost callback of a destroyed device) polls freed memory -- SIGSEGV in
|
|
62
|
+
* dawn::native::InstanceBase::ProcessEvents, on Metal and lavapipe alike (issue #30). dawn-node signals no
|
|
63
|
+
* point at which the instance has drained, so no GPU object is ever released; sharing one per flag list
|
|
64
|
+
* bounds what that keeps to one instance per configuration.
|
|
65
|
+
*/
|
|
66
|
+
const instances = new Map<DawnModule, Map<string, GPU>>();
|
|
67
|
+
|
|
52
68
|
/**
|
|
53
69
|
* Whether a loaded module is usable as Dawn.
|
|
54
70
|
* @param loaded - the module namespace
|
|
@@ -125,9 +141,11 @@ export function dawnFlags(options: NodeGpuOptions | undefined): string[] {
|
|
|
125
141
|
}
|
|
126
142
|
|
|
127
143
|
/**
|
|
128
|
-
* import("webgpu"), install dawn.globals unless installGlobals === false, dawn.create(flags) (spec 2.3).
|
|
144
|
+
* import("webgpu"), install dawn.globals unless installGlobals === false, dawn.create(flags) (spec 2.3). The GPU
|
|
145
|
+
* object is created once per flag list and reused by every later call with the same flags; it is never released,
|
|
146
|
+
* because Dawn keeps polling its instance for the adapters and devices it produced (issue #30).
|
|
129
147
|
* @param options - adapter / backend / dawnFeatures / software / installGlobals (and the test seam)
|
|
130
|
-
* @returns the handle; `dispose()` drops the
|
|
148
|
+
* @returns the handle; `dispose()` drops the handle's reference, never the shared instance
|
|
131
149
|
* @throws WebGpuGraphError E_NO_WEBGPU { reason, hint } when the module does not load (missing, or its glibc is too old), has no create(), or create(flags) throws
|
|
132
150
|
*/
|
|
133
151
|
export async function createNodeGpu(options?: NodeGpuOptions): Promise<NodeGpuHandle> {
|
|
@@ -150,12 +168,22 @@ export async function createNodeGpu(options?: NodeGpuOptions): Promise<NodeGpuHa
|
|
|
150
168
|
if (options?.installGlobals !== false && typeof loaded.globals === "object" && loaded.globals !== null) {
|
|
151
169
|
Object.assign(globalThis, loaded.globals);
|
|
152
170
|
}
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
171
|
+
const flags = dawnFlags(options);
|
|
172
|
+
const key = flags.join("\n");
|
|
173
|
+
let byFlags = instances.get(loaded);
|
|
174
|
+
if (byFlags === undefined) {
|
|
175
|
+
byFlags = new Map();
|
|
176
|
+
instances.set(loaded, byFlags);
|
|
177
|
+
}
|
|
178
|
+
let gpu = byFlags.get(key);
|
|
179
|
+
if (gpu === undefined) {
|
|
180
|
+
try {
|
|
181
|
+
gpu = loaded.create(flags);
|
|
182
|
+
} catch (err) {
|
|
183
|
+
const reason = `dawn.create() threw: ${messageOf(err)}`;
|
|
184
|
+
throw new WebGpuGraphError("E_NO_WEBGPU", `${reason}; ${INSTALL_HINT}`, { reason, hint: INSTALL_HINT });
|
|
185
|
+
}
|
|
186
|
+
byFlags.set(key, gpu);
|
|
159
187
|
}
|
|
160
188
|
return new DawnHandle(gpu);
|
|
161
189
|
}
|
|
@@ -181,10 +209,9 @@ function contextOptionsOf(options: NodeGpuOptions): Omit<GpuContextOptions, "gpu
|
|
|
181
209
|
|
|
182
210
|
/**
|
|
183
211
|
* createNodeGpu + GpuContext.create({ gpu, runtime: "node", ...options }); ctx.dispose() also disposes the
|
|
184
|
-
* handle, and a create() failure disposes it before rethrowing.
|
|
185
|
-
*
|
|
186
|
-
*
|
|
187
|
-
* P1-T1; measured with tmp/p1t1/gc-race2.mjs), and device.destroy() reports the loss right away.
|
|
212
|
+
* handle, and a create() failure disposes it before rethrowing. Disposing is safe at any moment: the Dawn
|
|
213
|
+
* instance itself stays alive for the process (createNodeGpu), so a callback of the destroyed device that
|
|
214
|
+
* arrives late, or a call on `ctx.device` after dispose(), never reaches a freed instance.
|
|
188
215
|
* @param options - the Node options
|
|
189
216
|
* @returns the context
|
|
190
217
|
*/
|
|
@@ -197,11 +224,8 @@ export async function createNodeGpuContext(options?: NodeGpuOptions): Promise<Gp
|
|
|
197
224
|
handle.dispose();
|
|
198
225
|
throw err;
|
|
199
226
|
}
|
|
200
|
-
const { lost } = ctx;
|
|
201
227
|
ctx.attachDisposer(() => {
|
|
202
|
-
|
|
203
|
-
handle.dispose();
|
|
204
|
-
});
|
|
228
|
+
handle.dispose();
|
|
205
229
|
});
|
|
206
230
|
return ctx;
|
|
207
231
|
}
|
|
@@ -71,6 +71,7 @@ export const W: Readonly<{
|
|
|
71
71
|
thresholdBits: 22;
|
|
72
72
|
deltaBits: 23;
|
|
73
73
|
path: 24;
|
|
74
|
+
nextDegreeSum: 25;
|
|
74
75
|
}> = Object.freeze({
|
|
75
76
|
frontierCount: 0,
|
|
76
77
|
nextFrontierCount: 1,
|
|
@@ -97,6 +98,7 @@ export const W: Readonly<{
|
|
|
97
98
|
thresholdBits: 22,
|
|
98
99
|
deltaBits: 23,
|
|
99
100
|
path: 24,
|
|
101
|
+
nextDegreeSum: 25,
|
|
100
102
|
});
|
|
101
103
|
|
|
102
104
|
/** The words a `reset` seeds (every other word is zeroed). */
|
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The grid pyramid (spec 6 row 12, 7.7 G4-G5; P4-T9): the planner that records, into the caller's pass, the finest
|
|
3
|
-
* centroids (G4, `grid-centroid`: thread per cell over `cells +
|
|
3
|
+
* centroids (G4, `grid-centroid`: thread per cell over `cells + outsideCells`, the pseudo-cells included), the hub-cell
|
|
4
4
|
* completion (G4a: the T1 `indirect-finalize` over `hubCounters[0]` into `hubArgs` with `wg = 1`, so the finalize's
|
|
5
5
|
* `ceil(count / wg)` is ONE workgroup per hub cell; G4b: `grid-centroid-hub`, one workgroup per hub cell, dispatched
|
|
6
6
|
* indirectly; PD-13, DEP-P4-I) and one `grid-downsample` dispatch per coarser
|
|
7
7
|
* level (G5). Level 0 holds `[sum m x, sum m y, sum m z, sum m]` per cell; every parent is the sum of its 2^dim
|
|
8
|
-
* children; the pseudo-
|
|
8
|
+
* children; the pseudo-cells (indices `cells ..` of level 0) are never children. No atomics touch the sums (design 6 row 12:
|
|
9
9
|
* bitwise reproducible); the only atomics are the hub append and the occupancy max.
|
|
10
10
|
*
|
|
11
11
|
* The named grid buffers (`pyramid`, `hubList`, `hubCounters`, `hubArgs`) are the caller's (the model's
|
|
@@ -34,9 +34,9 @@ export interface GridPyramidBindings {
|
|
|
34
34
|
readonly params: Binding;
|
|
35
35
|
/** `n` words: the sorted node indices (the T8 build). */
|
|
36
36
|
readonly sortedIdx: Binding;
|
|
37
|
-
/** `cells + 2` words: the exclusive scan of the cell histogram (the T8 build). */
|
|
37
|
+
/** `histWords` (`cells + 2^dim + 1`) words: the exclusive scan of the cell histogram (the T8 build). */
|
|
38
38
|
readonly cellStart: Binding;
|
|
39
|
-
/** `pyramidCells` vec4f: every level, level 0 first with the pseudo-
|
|
39
|
+
/** `pyramidCells` vec4f: every level, level 0 first with the 2^dim orthant pseudo-cells from index `cells`. */
|
|
40
40
|
readonly pyramid: Binding;
|
|
41
41
|
/** The hub cells' indices, appended by G4 (at least one word; at most `floor(n / (GRID_HUB_CELL + 1))` are ever written, so `ceil(n / GRID_HUB_CELL)` words always suffice). */
|
|
42
42
|
readonly hubList: Binding;
|
|
@@ -204,7 +204,8 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
|
|
|
204
204
|
}
|
|
205
205
|
const { centroid, finalize, hub, downsample } = this.kernels;
|
|
206
206
|
const one: DispatchPlan = { x: 1, y: 1, z: 1, items: 1, stride: null };
|
|
207
|
-
|
|
207
|
+
const level0 = spec.cells + spec.outsideCells;
|
|
208
|
+
centroid.dispatch(pass, bound.centroid, plan1d(level0, scope.workgroupSize, scope.caps), [paramsOffset]);
|
|
208
209
|
finalize.dispatch(pass, bound.finalize, one, [bound.finalizeOffset]);
|
|
209
210
|
hub.dispatchIndirect(pass, bound.hub, bound.hubArgs, 0, [paramsOffset]);
|
|
210
211
|
this.dispatches = 3;
|
package/src/primitives/grid.ts
CHANGED
|
@@ -3,8 +3,9 @@
|
|
|
3
3
|
* the caller's pass, the cell keys (G1, `grid-cell-key`), the stable sort by key (G2: `radixSort` at GRID_SORT_BITS,
|
|
4
4
|
* or `countingSortByKey` when the caller asks for the set-deterministic path) and the per-cell histogram with its
|
|
5
5
|
* exclusive scan (G3: the `histogram` kernel over `cellKey` and the `scan` of it; DEP-P4-I names no grid-specific
|
|
6
|
-
* id). `cellHist` and `cellStart` hold `cells + 2` words: every real cell, the outside
|
|
7
|
-
*
|
|
6
|
+
* id). `cellHist` and `cellStart` hold `histWords = cells + 2^dim + 1` words: every real cell, the 2^dim outside
|
|
7
|
+
* pseudo-cells (one per orthant about the grid centre, issue #90) from index `cells`, and one more so
|
|
8
|
+
* `cellStart[histWords - 1] === n` closes the last range. Every zeroing is a `fill` dispatch inside the
|
|
8
9
|
* pass (PD-12), never an encoder clear.
|
|
9
10
|
*
|
|
10
11
|
* The named grid buffers (`cellKey`, `cellVal`, `sortedKey`, `sortedIdx`, `cellHist`, `cellStart`) are the caller's
|
|
@@ -39,11 +40,13 @@ export interface GridSpec {
|
|
|
39
40
|
readonly g: number;
|
|
40
41
|
/** `log2(G / GRID_COARSEST_SIDE) + 1`. */
|
|
41
42
|
readonly levels: number;
|
|
42
|
-
/** `G^dim` finest cells; the outside pseudo-
|
|
43
|
+
/** `G^dim` finest cells; the outside pseudo-cells are indices `cells .. cells + outsideCells - 1`. */
|
|
43
44
|
readonly cells: number;
|
|
44
|
-
/** `
|
|
45
|
+
/** `2^dim`: one outside pseudo-cell per orthant about the grid centre (issue #90). */
|
|
46
|
+
readonly outsideCells: number;
|
|
47
|
+
/** `cells + outsideCells + 1`: the length of `cellHist` / `cellStart`. */
|
|
45
48
|
readonly histWords: number;
|
|
46
|
-
/** The first cell of every level inside the pyramid: `levelOffsets[0] = 0`, level 0 holds `cells +
|
|
49
|
+
/** The first cell of every level inside the pyramid: `levelOffsets[0] = 0`, level 0 holds `cells + outsideCells` (the pseudo-cells last), level L `(G / 2^L)^dim`. */
|
|
47
50
|
readonly levelOffsets: readonly number[];
|
|
48
51
|
/** Every level's cells together: `levelOffsets[levels - 1] + GRID_COARSEST_SIDE^dim`. */
|
|
49
52
|
readonly pyramidCells: number;
|
|
@@ -81,8 +84,8 @@ function floorPow2(x: number): number {
|
|
|
81
84
|
* The grid of `n` nodes in `dim` dimensions under the tuning (spec 7.7 geometry table; PD-9): `G = clamp(nextPow2(2 *
|
|
82
85
|
* ceil(n^(1 / dim))), GRID_MIN_SIDE, floorPow2(gridMax))` where `gridMax` is `gridMax2D` or `gridMax3D`, rounded DOWN
|
|
83
86
|
* to a power of two so every level's side is an integer (512 and 128 stay; 100 becomes 64); `levels = log2(G /
|
|
84
|
-
* GRID_COARSEST_SIDE) + 1`. At the caps: 349,
|
|
85
|
-
* pseudo-
|
|
87
|
+
* GRID_COARSEST_SIDE) + 1`. At the caps: 349,524 pyramid cells in 2D, 2,396,744 in 3D (the design's counts plus the
|
|
88
|
+
* 2^dim pseudo-cells).
|
|
86
89
|
* @param n - the node count (>= 0)
|
|
87
90
|
* @param dim - 2 or 3
|
|
88
91
|
* @param tuning - the resolved layout tuning (`gridMax2D`, `gridMax3D`, `deterministic`)
|
|
@@ -102,10 +105,11 @@ export function gridSpecFor(
|
|
|
102
105
|
levels++;
|
|
103
106
|
}
|
|
104
107
|
const cells = g ** dim;
|
|
108
|
+
const outsideCells = 2 ** dim;
|
|
105
109
|
const levelOffsets: number[] = [0];
|
|
106
110
|
let s = g;
|
|
107
111
|
for (let level = 0; level + 1 < levels; level++) {
|
|
108
|
-
levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ?
|
|
112
|
+
levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? outsideCells : 0));
|
|
109
113
|
s /= 2;
|
|
110
114
|
}
|
|
111
115
|
return {
|
|
@@ -113,7 +117,8 @@ export function gridSpecFor(
|
|
|
113
117
|
g,
|
|
114
118
|
levels,
|
|
115
119
|
cells,
|
|
116
|
-
|
|
120
|
+
outsideCells,
|
|
121
|
+
histWords: cells + outsideCells + 1,
|
|
117
122
|
levelOffsets: Object.freeze(levelOffsets),
|
|
118
123
|
pyramidCells: levelOffsets[levels - 1] + GRID_COARSEST_SIDE ** dim,
|
|
119
124
|
deterministic: tuning.deterministic,
|
|
@@ -121,7 +126,7 @@ export function gridSpecFor(
|
|
|
121
126
|
}
|
|
122
127
|
|
|
123
128
|
/**
|
|
124
|
-
* The bytes of the pyramid (spec 7.7: 16 B per cell, every level, the pseudo-
|
|
129
|
+
* The bytes of the pyramid (spec 7.7: 16 B per cell, every level, the pseudo-cells included): 38,347,904 at the 3D cap.
|
|
125
130
|
* @param spec - the grid
|
|
126
131
|
* @returns the byte length
|
|
127
132
|
*/
|
|
@@ -155,9 +160,9 @@ export interface GridBuildBindings {
|
|
|
155
160
|
readonly sortedKey: Binding;
|
|
156
161
|
/** `n` words: the sorted node indices. */
|
|
157
162
|
readonly sortedIdx: Binding;
|
|
158
|
-
/** `cells + 2` words: the per-cell counts. */
|
|
163
|
+
/** `histWords` (`cells + 2^dim + 1`) words: the per-cell counts. */
|
|
159
164
|
readonly cellHist: Binding;
|
|
160
|
-
/** `cells + 2` words: the exclusive scan of `cellHist`. */
|
|
165
|
+
/** `histWords` (`cells + 2^dim + 1`) words: the exclusive scan of `cellHist`. */
|
|
161
166
|
readonly cellStart: Binding;
|
|
162
167
|
}
|
|
163
168
|
|
|
@@ -7,8 +7,9 @@
|
|
|
7
7
|
* the twin's two compilations -- and then every invocation strips the range `[0, aggregate)` with a binary search
|
|
8
8
|
* (`upper_bound`) over the scanned degrees to find which entry its arc belongs to. One `atomicAdd` per WORKGROUP
|
|
9
9
|
* reserves the block's span in the queue (`edgeCount`), the same aggregate lands in `edgeCountUnclamped` (the overflow
|
|
10
|
-
* detector, never clamped) and in `frontierDegreeSum` (
|
|
11
|
-
* `
|
|
10
|
+
* detector, never clamped) and in `frontierDegreeSum` (the inspect seam's per-level expansion count, rotated into
|
|
11
|
+
* `prevDegreeSum` by the boundary; Beamer's m_f is `nextDegreeSum`, measured by `bfs-next-degree` -- issue #391);
|
|
12
|
+
* a lane whose queue position is at or past `P.edgeCapacity` writes nothing (the clamp). The queue holds the TARGET vertex of each arc only (PD-24: `parent`
|
|
12
13
|
* comes from the post-pass). Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan call and the
|
|
13
14
|
* `workgroupUniformLoad` sit unconditionally after the guard, and the strip loop is bounded by a uniform value.
|
|
14
15
|
* There is no `TIER` override: a hub row is balanced over all `WG` lanes inside its block, and the small-frontier
|
|
@@ -44,7 +45,7 @@ fn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
|
|
|
44
45
|
if (lid.x == 0u) {
|
|
45
46
|
base = atomicAdd(&counters[8], aggregate); // edgeCount: ONE reservation per workgroup, not one per arc
|
|
46
47
|
atomicAdd(&counters[9], aggregate); // edgeCountUnclamped: the overflow detector (PD-23)
|
|
47
|
-
atomicAdd(&counters[2], aggregate); // frontierDegreeSum:
|
|
48
|
+
atomicAdd(&counters[2], aggregate); // frontierDegreeSum: what this level expanded (the inspect seam)
|
|
48
49
|
}
|
|
49
50
|
workgroupBarrier();
|
|
50
51
|
for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
|
|
@@ -11,9 +11,10 @@
|
|
|
11
11
|
* The winners are packed into the output vertex queue by `bfs-contract`'s workgroup scan and one `atomicAdd` per
|
|
12
12
|
* workgroup on `nextFrontierCount`, and claim with a plain `atomicStore`: the list holds every vertex once and the
|
|
13
13
|
* sweep is vertex-parallel, so no two lanes claim one vertex. It adds nothing to `frontierDegreeSum` (a bottom-up
|
|
14
|
-
* level expands nothing), which is why
|
|
15
|
-
*
|
|
16
|
-
*
|
|
14
|
+
* level expands nothing), which is why that word is not Beamer's m_f: `bfs-next-degree` sums the degree of what
|
|
15
|
+
* this sweep CLAIMS into `nextDegreeSum`, so the boundary's test and its `unvisitedDegreeSum` subtraction are exact
|
|
16
|
+
* on a bottom-up level like any other (issue #391; the selector's JSDoc). Uniformity (spec 3.5 rule 1): the
|
|
17
|
+
* guarded walk writes locals, the scan and the reduction run unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
|
|
17
18
|
* test/helpers/sabotage.ts are textual edits of it.
|
|
18
19
|
*/
|
|
19
20
|
export const bfsBottomUpWgsl = /* wgsl */ `
|
|
@@ -4,15 +4,15 @@
|
|
|
4
4
|
* ONE dispatch, chosen by `frontier-finalize` for a frontier below `P.fusedMax` entries (`path` 2) and for the retry
|
|
5
5
|
* of a level whose edge queue overflowed (`path` 4, PD-23). One WORKGROUP per frontier entry, the workgroups striding
|
|
6
6
|
* the entries by the dispatch's group count (`P.stride`): lane 0 reads the entry's row clipped
|
|
7
|
-
* to the bound arc window, adds its degree to `frontierDegreeSum` (
|
|
8
|
-
* too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim
|
|
9
|
-
* `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
|
|
7
|
+
* to the bound arc window, adds its degree to `frontierDegreeSum` (so the inspect seam's per-level expansion count
|
|
8
|
+
* covers fused levels too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim
|
|
9
|
+
* inline -- `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
|
|
10
10
|
* packing the strip's winners into the output vertex queue by the same Hillis-Steele scan and one `atomicAdd` per
|
|
11
11
|
* strip on `nextFrontierCount`. No edge queue is written or read, which is the whole win for a tiny frontier
|
|
12
12
|
* (Merrill's fleeting iterations) and what makes the overflow retry exact: the partial edge queue is never consulted.
|
|
13
13
|
* On a retry level `advance-expand` has already added the frontier's degree to `frontierDegreeSum`, so that word
|
|
14
|
-
* holds
|
|
15
|
-
*
|
|
14
|
+
* holds twice the level's expanded degree; since issue #391 no decision reads it (Beamer's m_f is `nextDegreeSum`
|
|
15
|
+
* and the boundary subtracts that), so the doubling only reaches the inspect seam's `prevDegreeSum`. Nothing here
|
|
16
16
|
* writes a parent (PD-24: the post-pass does), which is what keeps the kernel at the eight-storage-buffer budget
|
|
17
17
|
* with the four graph slots. Uniformity (spec 3.5 rule 1): the strip loop's bound and the row start are
|
|
18
18
|
* `workgroupUniformLoad`s, so every barrier of the per-strip append is in uniform control flow; the guarded claim
|
|
@@ -44,7 +44,7 @@ fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
|
|
|
44
44
|
let d = select(0u, a1 - a0, a1 > a0);
|
|
45
45
|
wdeg = d;
|
|
46
46
|
wstart = a0;
|
|
47
|
-
atomicAdd(&counters[2], d); // frontierDegreeSum, so
|
|
47
|
+
atomicAdd(&counters[2], d); // frontierDegreeSum, so the inspect seam sees fused levels too
|
|
48
48
|
}
|
|
49
49
|
let deg = workgroupUniformLoad(&wdeg); // uniform: the loop below may hold barriers
|
|
50
50
|
let start = workgroupUniformLoad(&wstart);
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `bfs-next-degree` kernel body (design 8.4; issue #391, the amendment to P8-T8's PD-21): Beamer's m_f measured
|
|
3
|
+
* EXACTLY, at the end of every level, as the out-degree sum of the vertices the level just claimed -- the next
|
|
4
|
+
* frontier, which is what the next boundary decides the direction FOR. It grid-strides over the output vertex
|
|
5
|
+
* queue (`nextFrontierCount`, word 1, the claim kernels' append span; `P.stride` the plan's stride), reads each
|
|
6
|
+
* entry's out-degree from the `outDegree` view, reduces the lane sums with the prelude's `wg_reduce_u32` and lands
|
|
7
|
+
* ONE `atomicAdd` per workgroup in `nextDegreeSum` (word 25), which `frontier-finalize` role 0 reads for the
|
|
8
|
+
* switch-into-bottom-up test, subtracts from `unvisitedDegreeSum` and zeroes for the next level. It runs on every
|
|
9
|
+
* path that claims (the path word 24 non-zero: two-phase, fused, bottom-up, the retry) and does nothing on a level
|
|
10
|
+
* past the end.
|
|
11
|
+
*
|
|
12
|
+
* Why a kernel of its own: before it, the test used `frontierDegreeSum` (word 2), which the EXPANSION of the
|
|
13
|
+
* previous frontier accumulates, so the boundary compared the degree of the frontier it had just finished with the
|
|
14
|
+
* unvisited set, one level stale, and a bottom-up level (which expands nothing) left it at 0. On the 1M / 10M R-MAT
|
|
15
|
+
* that misses the switch at the level that matters: the frontier of 46,524 hubs at level 1 has 13.6M out-arcs, the
|
|
16
|
+
* unvisited set 7.3M, and the boundary saw the source's 86,405 instead -- top-down wrote 13.6M edge-queue entries
|
|
17
|
+
* where the bottom-up sweep reads 0.6M. Measuring the next frontier's degree at claim time is Beamer's own m_f, and
|
|
18
|
+
* the same word makes `unvisitedDegreeSum` exact after a bottom-up level too. Uniformity (spec 3.5 rule 1): the
|
|
19
|
+
* loop holds no barrier (its trip count is per lane), and the reduction runs unconditionally after it. Body only
|
|
20
|
+
* (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
|
|
21
|
+
*/
|
|
22
|
+
export const bfsNextDegreeWgsl = /* wgsl */ `
|
|
23
|
+
@compute @workgroup_size(WG)
|
|
24
|
+
fn bfs_next_degree(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
25
|
+
let count = select(0u, atomicLoad(&counters[1]), atomicLoad(&counters[24]) != 0u); // nextFrontierCount, on a level that claimed (the path word)
|
|
26
|
+
var sum = 0u;
|
|
27
|
+
for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // no barrier inside: the trip count is per lane
|
|
28
|
+
sum = sum + outDegree[frontier[i]];
|
|
29
|
+
}
|
|
30
|
+
let total = wg_reduce_u32(sum, lid.x, 0u); // the prelude's workgroup sum; uniform: after the loop
|
|
31
|
+
if (lid.x == 0u) { atomicAdd(&counters[25], total); } // nextDegreeSum: ONE atomic per workgroup
|
|
32
|
+
}
|
|
33
|
+
`;
|
|
@@ -9,6 +9,9 @@
|
|
|
9
9
|
* `partials[group].swingTraction`. `pos` is `array<vec4f>` with the mass in `.w`; `force` / `oldForce` are stride-3
|
|
10
10
|
* `array<f32>` read through the per-body helpers (4.4 rule 6). Normative text, copied verbatim: the P1-T5 sabotage
|
|
11
11
|
* rows (gravity sign, `k / d2`, the `jj != i` guard, the `.w` mass lane) are textual edits of this string.
|
|
12
|
+
* The j range is split into passes of at most EXACT_TILES_PER_PASS tiles (issue #87: llvmpipe's per-invocation loop
|
|
13
|
+
* budget); an earlier pass adds its partial sum into `force`, the last one runs gravity and the epilogue. A graph of
|
|
14
|
+
* at most 32,768 nodes is one pass, bitwise the single-pass kernel. Record it through recordExactRepulsion.
|
|
12
15
|
*/
|
|
13
16
|
export const fa2RepulsionExactWgsl = /* wgsl */ `
|
|
14
17
|
var<workgroup> tile: array<vec4f, WG>; // xyz + mass, 4 KiB at WG = 256
|
|
@@ -35,14 +38,23 @@ fn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: cent
|
|
|
35
38
|
}
|
|
36
39
|
|
|
37
40
|
@compute @workgroup_size(WG)
|
|
38
|
-
fn repulsion(
|
|
41
|
+
fn repulsion(
|
|
42
|
+
@builtin(workgroup_id) wid: vec3<u32>,
|
|
43
|
+
@builtin(local_invocation_id) lid: vec3<u32>,
|
|
44
|
+
@builtin(num_workgroups) nwg: vec3<u32>,
|
|
45
|
+
) {
|
|
46
|
+
// issue #87: pass p of the tile range is dispatched with p + 1 z slices; only the last slice works, so the pass
|
|
47
|
+
// index needs no uniform. Uniform: keyed on workgroup_id and num_workgroups only.
|
|
48
|
+
if (wid.z + 1u < nwg.z) { return; }
|
|
39
49
|
let i = linear_id(wid, lid.x);
|
|
40
50
|
let valid = i < P.n;
|
|
41
51
|
var pi = vec4f(0.0);
|
|
42
52
|
if (valid) { pi = pos[i]; }
|
|
43
53
|
var f = vec3f(0.0);
|
|
44
54
|
let tiles = (P.n + WG - 1u) / WG;
|
|
45
|
-
|
|
55
|
+
let tileBegin = (nwg.z - 1u) * EXACT_TILES_PER_PASS;
|
|
56
|
+
let tileEnd = min(tiles, tileBegin + EXACT_TILES_PER_PASS);
|
|
57
|
+
for (var t = tileBegin; t < tileEnd; t = t + 1u) {
|
|
46
58
|
let j = t * WG + lid.x;
|
|
47
59
|
if (j < P.n) { tile[lid.x] = pos[j]; } else { tile[lid.x] = vec4f(0.0); } // guarded fill; mass 0 marks the pad
|
|
48
60
|
workgroupBarrier(); // uniform: every invocation reaches it
|
|
@@ -65,6 +77,10 @@ fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
|
|
|
65
77
|
}
|
|
66
78
|
workgroupBarrier();
|
|
67
79
|
}
|
|
80
|
+
if (tileEnd < tiles) { // an earlier pass: its partial sum only (uniform: P.n, nwg)
|
|
81
|
+
if (valid) { store_force(i, load_force(i) + f); }
|
|
82
|
+
return;
|
|
83
|
+
}
|
|
68
84
|
// epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it
|
|
69
85
|
var sw = 0.0;
|
|
70
86
|
var tr = 0.0;
|
|
@@ -61,7 +61,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
61
61
|
S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)
|
|
62
62
|
let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)
|
|
63
63
|
S.meanDisplacement = meanDisp;
|
|
64
|
-
S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);
|
|
64
|
+
S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= min(P.settleThreshold * S.rmsRadius, P.settleFloor)); // relative AND absolute (issue #97)
|
|
65
65
|
}
|
|
66
66
|
S.iteration = S.iteration + 1u;
|
|
67
67
|
T[P.iterationIndex].meanDisplacement = S.meanDisplacement;
|
|
@@ -79,7 +79,9 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
79
79
|
S.invCellSize = 1.0 / cellSize;
|
|
80
80
|
S.eps = 0.25 * cellSize;
|
|
81
81
|
}
|
|
82
|
-
|
|
82
|
+
var outside = 0u; // the previous iteration's pseudo-cell counts, one per orthant (issue #90; 0 after load)
|
|
83
|
+
for (var o = 0u; o < select(4u, 8u, P.dim == 3u); o = o + 1u) { outside = outside + cellHist[cells + o]; }
|
|
84
|
+
S.outsideGrid = outside;
|
|
83
85
|
S.maxCellOccupancy = atomicLoad(&hubCounters[1]);
|
|
84
86
|
atomicStore(&hubCounters[0], 0u);
|
|
85
87
|
atomicStore(&hubCounters[1], 0u);
|
|
@@ -29,25 +29,26 @@
|
|
|
29
29
|
* raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
|
|
30
30
|
* to 0); the relax kernels size themselves from words 0 and 20.
|
|
31
31
|
*
|
|
32
|
-
* Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a
|
|
33
|
-
* the done boundary too, which the host model of the tests mirrors): top-down switches to
|
|
34
|
-
* `
|
|
35
|
-
* unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up
|
|
36
|
-
* `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no
|
|
37
|
-
* reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the
|
|
38
|
-
* change is counted in `switches`, the previous direction is word 14.
|
|
39
|
-
*
|
|
40
|
-
*
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
*
|
|
47
|
-
*
|
|
48
|
-
*
|
|
49
|
-
*
|
|
50
|
-
*
|
|
32
|
+
* Beamer's test (P8-T8, PD-21; amended for issue #391), evaluated at every boundary BEFORE the `done` branch (so a
|
|
33
|
+
* switch can be counted at the done boundary too, which the host model of the tests mirrors): top-down switches to
|
|
34
|
+
* bottom-up when `nextDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's
|
|
35
|
+
* `max(1, floor(arcCount / n))` unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up
|
|
36
|
+
* switches back when `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no
|
|
37
|
+
* admitted device reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the
|
|
38
|
+
* direction at 0. Every change is counted in `switches`, the previous direction is word 14. `nextDegreeSum` (word
|
|
39
|
+
* 25) is Beamer's m_f measured EXACTLY: `bfs-next-degree` sums the out-degrees of the vertices a level claims at
|
|
40
|
+
* the end of that level, so the boundary that rotates them in as `next` compares the degree of the frontier it is
|
|
41
|
+
* about to expand -- not, as before the amendment, `frontierDegreeSum` (word 2), the degree of the frontier the
|
|
42
|
+
* previous level EXPANDED, one level stale and 0 after a bottom-up level, which on the 1M / 10M R-MAT missed the
|
|
43
|
+
* switch at the level holding 13.6M of the 21M arcs. Word 2 is still accumulated by the expansion and rotated into
|
|
44
|
+
* word 4 for the inspect seam. The two unvisited words are rebuilt exactly once per submit by `bfs-unvisited-flags`
|
|
45
|
+
* (PD-18) and maintained here by subtraction from the SECOND boundary of a submit on, because a boundary may only
|
|
46
|
+
* subtract what the submit's rebuild counted: the rebuild counts the vertices unclaimed when it runs, the frontier
|
|
47
|
+
* rotated in at boundary 0 was claimed by the previous submit's last level, so it was never in the sums, and
|
|
48
|
+
* boundary b subtracts `next = |F_b|` and `nextDegreeSum = deg(F_b)`, both inside the sums iff b >= 1. Both words
|
|
49
|
+
* are therefore exact at every boundary, bottom-up levels included (the sweep's claims are summed like any other).
|
|
50
|
+
* Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits
|
|
51
|
+
* of it.
|
|
51
52
|
*/
|
|
52
53
|
export const frontierFinalizeWgsl = /* wgsl */ `
|
|
53
54
|
@compute @workgroup_size(WG)
|
|
@@ -61,19 +62,19 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
61
62
|
let finished = atomicLoad(&counters[0]);
|
|
62
63
|
let next = atomicLoad(&counters[1]);
|
|
63
64
|
let degSum = atomicLoad(&counters[2]);
|
|
65
|
+
let nextDeg = atomicLoad(&counters[25]); // deg(F_b), summed by bfs-next-degree when F_b was claimed (issue #391)
|
|
64
66
|
atomicStore(&counters[3], finished); // prevFrontierCount
|
|
65
67
|
atomicStore(&counters[4], degSum); // prevDegreeSum
|
|
66
68
|
atomicStore(&counters[0], next); // the rotation
|
|
67
69
|
atomicStore(&counters[1], 0u);
|
|
68
70
|
atomicStore(&counters[2], 0u);
|
|
71
|
+
atomicStore(&counters[25], 0u); // the next level's claims sum from 0
|
|
69
72
|
atomicStore(&counters[8], 0u); // edgeCount
|
|
70
73
|
atomicStore(&counters[9], 0u); // edgeCountUnclamped
|
|
71
74
|
atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount
|
|
72
|
-
if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped
|
|
73
|
-
atomicStore(&counters[5], atomicLoad(&counters[5]) - next);
|
|
74
|
-
|
|
75
|
-
if (P.firstOfSubmit >= 2u) {
|
|
76
|
-
atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2
|
|
75
|
+
if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped (P8-T8, PD-18): F_b was inside the submit's rebuilt sums iff b >= 1
|
|
76
|
+
atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount, exact
|
|
77
|
+
atomicStore(&counters[6], atomicLoad(&counters[6]) - nextDeg); // unvisitedDegreeSum, exact (issue #391: no longer one level stale)
|
|
77
78
|
}
|
|
78
79
|
let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0
|
|
79
80
|
atomicStore(&counters[11], level);
|
|
@@ -83,7 +84,7 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
|
|
|
83
84
|
if (P.mode == 1u) {
|
|
84
85
|
direction = 0u; // top-down only (the test seam)
|
|
85
86
|
} else if (direction == 0u) {
|
|
86
|
-
if (
|
|
87
|
+
if (nextDeg > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing, m_f the degree of the frontier about to be expanded
|
|
87
88
|
} else {
|
|
88
89
|
if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking
|
|
89
90
|
}
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* G1, the `grid-cell-key` kernel body (spec 7.7; P4-T8): the finest cell of every node from the state's robust extent,
|
|
3
3
|
* `floor((p - gridMin) * invCellSize)` (a multiply, correctly rounded everywhere: PD-10), linearised when every axis
|
|
4
|
-
* is in [0, G) and the outside pseudo-
|
|
4
|
+
* is in [0, G), and otherwise one of the 2^dim outside pseudo-cells `cells + orthant`, the orthant of the cell about the
|
|
5
|
+
* grid centre (bit a set when `c[a] >= G / 2`; issue #90); `cellVal[i] = i`. The clamp before the floor keeps a
|
|
5
6
|
* far-away or NaN coordinate out of an out-of-range float-to-int conversion. Body only; normative text.
|
|
6
7
|
*/
|
|
7
8
|
export const gridCellKeyWgsl = /* wgsl */ `
|
|
@@ -18,7 +19,8 @@ fn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocatio
|
|
|
18
19
|
let g = i32(P.gridMax);
|
|
19
20
|
var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;
|
|
20
21
|
if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }
|
|
21
|
-
var key = cells;
|
|
22
|
+
var key = cells + select(0u, 1u, c.x >= g / 2) + select(0u, 2u, c.y >= g / 2); // an outside pseudo-cell: its orthant (issue #90)
|
|
23
|
+
if (P.dim == 3u) { key = key + select(0u, 4u, c.z >= g / 2); }
|
|
22
24
|
if (inside) {
|
|
23
25
|
key = u32(c.x) + P.gridMax * u32(c.y);
|
|
24
26
|
if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the pseudo-
|
|
2
|
+
* G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the 2^dim outside pseudo-cells
|
|
3
|
+
* included (issue #90); the
|
|
3
4
|
* mass-weighted position sum of a cell's sorted range in index order (no atomics: deterministic), the largest
|
|
4
5
|
* occupancy into hubCounters[1], and cells above GRID_HUB_CELL entries appended to hubList for G4b (PD-13). Body
|
|
5
6
|
* only; normative text.
|
|
@@ -10,7 +11,7 @@ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.
|
|
|
10
11
|
@compute @workgroup_size(WG)
|
|
11
12
|
fn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
|
|
12
13
|
let c = linear_id(wid, lid.x);
|
|
13
|
-
if (c
|
|
14
|
+
if (c >= grid_cells() + select(4u, 8u, P.dim == 3u)) { return; } // cells [0, cells + 2^dim): the pseudo-cells follow the real ones; no barrier follows
|
|
14
15
|
let start = cellStart[c];
|
|
15
16
|
let count = cellStart[c + 1u] - start;
|
|
16
17
|
atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* G5, the `grid-downsample` kernel body (spec 7.7; P4-T9): one dispatch per coarser level; every parent cell is the
|
|
3
3
|
* sum of its 4 (2D) or 8 (3D) children at the level below, read at P.childBase and written at P.parentBase (the
|
|
4
|
-
* pseudo-
|
|
4
|
+
* pseudo-cells, indices cells .. of level 0, are never children). No atomics. Body only; normative text.
|
|
5
5
|
*/
|
|
6
6
|
export const gridDownsampleWgsl = /* wgsl */ `
|
|
7
7
|
@compute @workgroup_size(WG)
|