@graphty/webgpu-graph-algorithms 0.6.3 → 0.6.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +62 -32
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-BXqgCifx.js → context-Dvq-Cc6v.js} +71 -25
- package/dist/chunks/context-Dvq-Cc6v.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +8 -6
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +57 -6
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/bellman-ford.d.ts +60 -0
- package/dist/src/algorithms/bellman-ford.d.ts.map +1 -0
- package/dist/src/algorithms/bellman-ford.js +301 -0
- package/dist/src/algorithms/bellman-ford.js.map +1 -0
- package/dist/src/algorithms/bfs.d.ts +67 -0
- package/dist/src/algorithms/bfs.d.ts.map +1 -0
- package/dist/src/algorithms/bfs.js +534 -0
- package/dist/src/algorithms/bfs.js.map +1 -0
- package/dist/src/algorithms/closeness.d.ts +53 -0
- package/dist/src/algorithms/closeness.d.ts.map +1 -0
- package/dist/src/algorithms/closeness.js +323 -0
- package/dist/src/algorithms/closeness.js.map +1 -0
- package/dist/src/algorithms/scope.d.ts +5 -3
- package/dist/src/algorithms/scope.d.ts.map +1 -1
- package/dist/src/algorithms/scope.js +3 -0
- package/dist/src/algorithms/scope.js.map +1 -1
- package/dist/src/algorithms/sssp.d.ts +71 -0
- package/dist/src/algorithms/sssp.d.ts.map +1 -0
- package/dist/src/algorithms/sssp.js +585 -0
- package/dist/src/algorithms/sssp.js.map +1 -0
- package/dist/src/constants.d.ts +12 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +12 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/index.d.ts +8 -2
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +7 -1
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts +4 -4
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +39 -5
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernel/uniform-ring.d.ts +8 -0
- package/dist/src/kernel/uniform-ring.d.ts.map +1 -1
- package/dist/src/kernel/uniform-ring.js +13 -0
- package/dist/src/kernel/uniform-ring.js.map +1 -1
- package/dist/src/kernels.d.ts +44 -4
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +371 -3
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/primitives/advance.d.ts +62 -0
- package/dist/src/primitives/advance.d.ts.map +1 -0
- package/dist/src/primitives/advance.js +95 -0
- package/dist/src/primitives/advance.js.map +1 -0
- package/dist/src/primitives/compact.d.ts +89 -0
- package/dist/src/primitives/compact.d.ts.map +1 -0
- package/dist/src/primitives/compact.js +233 -0
- package/dist/src/primitives/compact.js.map +1 -0
- package/dist/src/primitives/core-shape.d.ts +22 -1
- package/dist/src/primitives/core-shape.d.ts.map +1 -1
- package/dist/src/primitives/core-shape.js +33 -3
- package/dist/src/primitives/core-shape.js.map +1 -1
- package/dist/src/primitives/frontier.d.ts +156 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -0
- package/dist/src/primitives/frontier.js +259 -0
- package/dist/src/primitives/frontier.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +16 -7
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/traversal.d.ts +53 -0
- package/dist/src/types/traversal.d.ts.map +1 -0
- package/dist/src/types/traversal.js +10 -0
- package/dist/src/types/traversal.js.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts +19 -0
- package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/advance-expand.wgsl.js +69 -0
- package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts +22 -0
- package/dist/src/wgsl/bf-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bf-relax.wgsl.js +58 -0
- package/dist/src/wgsl/bf-relax.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js +24 -0
- package/dist/src/wgsl/bfs-bitset-build.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js +67 -0
- package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts +20 -0
- package/dist/src/wgsl/bfs-contract.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js +55 -0
- package/dist/src/wgsl/bfs-contract.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts +25 -0
- package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js +78 -0
- package/dist/src/wgsl/bfs-fused.wgsl.js.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts +18 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js +42 -0
- package/dist/src/wgsl/bfs-unvisited-flags.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +17 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js +65 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +20 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js +96 -0
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts +9 -0
- package/dist/src/wgsl/compact-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js +17 -0
- package/dist/src/wgsl/compact-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts +10 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js +19 -0
- package/dist/src/wgsl/dedupe-claim.wgsl.js.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts +12 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js +46 -0
- package/dist/src/wgsl/dedupe-filter.wgsl.js.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +59 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js +210 -0
- package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts +28 -0
- package/dist/src/wgsl/sssp-pred.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js +80 -0
- package/dist/src/wgsl/sssp-pred.wgsl.js.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts +30 -0
- package/dist/src/wgsl/sssp-relax.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js +72 -0
- package/dist/src/wgsl/sssp-relax.wgsl.js.map +1 -0
- package/dist/webgpu-graph-algorithms.js +3207 -377
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +5 -4
- package/src/accelerator.ts +65 -7
- package/src/algorithms/bellman-ford.ts +387 -0
- package/src/algorithms/bfs.ts +626 -0
- package/src/algorithms/closeness.ts +395 -0
- package/src/algorithms/scope.ts +13 -3
- package/src/algorithms/sssp.ts +767 -0
- package/src/constants.ts +12 -0
- package/src/index.ts +14 -1
- package/src/kernel/prelude.ts +39 -4
- package/src/kernel/uniform-ring.ts +14 -0
- package/src/kernels.ts +450 -6
- package/src/primitives/advance.ts +130 -0
- package/src/primitives/compact.ts +323 -0
- package/src/primitives/core-shape.ts +41 -3
- package/src/primitives/frontier.ts +388 -0
- package/src/types/accelerator.ts +18 -5
- package/src/types/traversal.ts +56 -0
- package/src/wgsl/advance-expand.wgsl.ts +68 -0
- package/src/wgsl/bf-relax.wgsl.ts +57 -0
- package/src/wgsl/bfs-bitset-build.wgsl.ts +23 -0
- package/src/wgsl/bfs-bottom-up.wgsl.ts +66 -0
- package/src/wgsl/bfs-contract.wgsl.ts +54 -0
- package/src/wgsl/bfs-fused.wgsl.ts +77 -0
- package/src/wgsl/bfs-unvisited-flags.wgsl.ts +41 -0
- package/src/wgsl/closeness-reduce.wgsl.ts +64 -0
- package/src/wgsl/closeness-sweep.wgsl.ts +95 -0
- package/src/wgsl/compact-scatter.wgsl.ts +16 -0
- package/src/wgsl/dedupe-claim.wgsl.ts +18 -0
- package/src/wgsl/dedupe-filter.wgsl.ts +45 -0
- package/src/wgsl/frontier-finalize.wgsl.ts +209 -0
- package/src/wgsl/sssp-pred.wgsl.ts +79 -0
- package/src/wgsl/sssp-relax.wgsl.ts +71 -0
- package/dist/chunks/context-BXqgCifx.js.map +0 -1
|
@@ -0,0 +1,323 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `compact` and `dedupe` primitive driver (spec 6 row 4; P8-T3, the P8 plan's PD-1 / DEP-P8-A). `compact` is
|
|
3
|
+
* flag, scan, scatter: the flags are the caller's (one u32 per entry, written by the kernel that produced the queue),
|
|
4
|
+
* `exclusiveScan` turns them into offsets, and `compact-scatter` lands every flagged entry at its offset in queue
|
|
5
|
+
* order -- order-preserving, so bitwise reproducible -- and writes the total into one word of the caller's block.
|
|
6
|
+
* `dedupe` is Davidson's ownership trick in two dispatches: `dedupe-claim` stores every entry's index into
|
|
7
|
+
* `owner[vertex]`, `dedupe-filter` keeps the entries that still own their vertex. Two dispatches because a plain
|
|
8
|
+
* store read back in the same dispatch is a data race (WGSL 6.5.7); between dispatches the relaxed atomics make
|
|
9
|
+
* last-writer-wins well defined, so exactly one index per distinct vertex survives (set-deterministic: the SET is
|
|
10
|
+
* fixed, the order inside `out` follows the schedule).
|
|
11
|
+
*
|
|
12
|
+
* `owner` needs NO reset between calls, whatever it holds: a word is read only by an entry `i` whose vertex `v` a
|
|
13
|
+
* claim of the SAME call has just written, so a stale index or garbage at `owner[v]` is overwritten before any lane
|
|
14
|
+
* compares against it, and a word no current entry names is never read. The design's row implies a reset; the `fill`
|
|
15
|
+
* over `n` per round is the dispatch this saves.
|
|
16
|
+
*
|
|
17
|
+
* The count of `compact` is host-known. `dedupe`'s callers (the SSSP piles of P8-T9) only know their entry count on
|
|
18
|
+
* the device, so `countIndex` names a word of the counters block and the kernels read `min(counters[countIndex],
|
|
19
|
+
* count)` -- the device word clamped to the capacity `count` -- or `count` alone when `countIndex` is `U32_MAX`. The
|
|
20
|
+
* counters block is bound WHOLE and indexed, for the count word and for the output word alike, because
|
|
21
|
+
* `Kernel.bind` rejects a binding whose offset is not a multiple of 256 and a four-byte word never is.
|
|
22
|
+
*
|
|
23
|
+
* The driver owns no device objects: the caller supplies a ReduceScope (the same record `reduce` takes) and the
|
|
24
|
+
* compute pass to record into. `src/primitives/**` never imports `src/context.ts`.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
import { U32_MAX } from "../constants.js";
|
|
28
|
+
import { WebGpuGraphError } from "../errors.js";
|
|
29
|
+
import { plan1d, planGridStride } from "../kernel/dispatch.js";
|
|
30
|
+
import { type BoundKernel, type Kernel } from "../kernel/kernel.js";
|
|
31
|
+
import { COMPACT_PARAMS, kernelSpec } from "../kernels.js";
|
|
32
|
+
import { type Binding } from "../types/memory.js";
|
|
33
|
+
import { type ReduceScope } from "./reduce.js";
|
|
34
|
+
import { prepareScan, type ScanPlanner } from "./scan.js";
|
|
35
|
+
|
|
36
|
+
/** One compaction: `count` entries of `queue` whose `flags` word is non-zero go to `out` in order; the total goes to `outCount[outIndex]`. */
|
|
37
|
+
export interface CompactRecord {
|
|
38
|
+
/** The entries (at least 4 x count bytes). */
|
|
39
|
+
readonly queue: Binding;
|
|
40
|
+
/** One u32 per entry, 0 or 1 (at least 4 x count bytes). */
|
|
41
|
+
readonly flags: Binding;
|
|
42
|
+
/** The entry count (a non-negative integer below 2^32; host-known). */
|
|
43
|
+
readonly count: number;
|
|
44
|
+
/** The compacted entries (at least 4 x count bytes; the words past the total are not written). */
|
|
45
|
+
readonly out: Binding;
|
|
46
|
+
/** The block whose word `outIndex` receives the total (bound whole). */
|
|
47
|
+
readonly outCount: Binding;
|
|
48
|
+
/** The word of `outCount` to write. */
|
|
49
|
+
readonly outIndex: number;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** One dedupe: the entries of `queue` are filtered to one per distinct vertex into `out`; the surviving count is ADDED to `outCount[outIndex]`, which the caller zeroes before the pass. */
|
|
53
|
+
export interface DedupeRecord {
|
|
54
|
+
/** The entries, vertex indices below the owner's length (at least 4 x count bytes). */
|
|
55
|
+
readonly queue: Binding;
|
|
56
|
+
/** The entry count when `countIndex` is `U32_MAX`, else the capacity the device count is clamped to (a non-negative integer below 2^32). */
|
|
57
|
+
readonly count: number;
|
|
58
|
+
/** `U32_MAX` for a host-known count; else the word of `counters` holding the entry count on the device. */
|
|
59
|
+
readonly countIndex: number;
|
|
60
|
+
/** The block the count word is read from; with a device count it must be the SAME range as `outCount` (both kernels read the word through their own binding), and it is bound but never read when `countIndex` is `U32_MAX`. */
|
|
61
|
+
readonly counters: Binding;
|
|
62
|
+
/** The ownership words, one per vertex (the caller's promise: at least 4 x n bytes; never reset). */
|
|
63
|
+
readonly owner: Binding;
|
|
64
|
+
/** The surviving entries (at least 4 x count bytes; the words past the count are not written). */
|
|
65
|
+
readonly out: Binding;
|
|
66
|
+
/** The block whose word `outIndex` accumulates the surviving count (bound whole; zeroed by the caller before the pass). */
|
|
67
|
+
readonly outCount: Binding;
|
|
68
|
+
/** The word of `outCount` to add to; never equal to `countIndex`. */
|
|
69
|
+
readonly outIndex: number;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/** A prepared compact / dedupe (spec 6 row 4): records the dispatches of one call into a pass. */
|
|
73
|
+
export interface CompactPlanner {
|
|
74
|
+
/**
|
|
75
|
+
* Records the compaction (the flags' exclusive scan, then the scatter). For count 0 nothing is recorded and
|
|
76
|
+
* `outCount[outIndex]` keeps its prior value: a four-byte write at an unaligned offset is impossible, so every
|
|
77
|
+
* caller zeroes its block before the pass.
|
|
78
|
+
* @param pass - the compute pass
|
|
79
|
+
* @param record - the buffers and the count
|
|
80
|
+
*/
|
|
81
|
+
record(pass: GPUComputePassEncoder, record: CompactRecord): void;
|
|
82
|
+
/**
|
|
83
|
+
* Records the claim and the filter over `plan1d(count)`; nothing for count 0 (`outCount[outIndex]` then keeps
|
|
84
|
+
* its prior value).
|
|
85
|
+
* @param pass - the compute pass
|
|
86
|
+
* @param record - the buffers, the count and the count word
|
|
87
|
+
*/
|
|
88
|
+
recordDedupe(pass: GPUComputePassEncoder, record: DedupeRecord): void;
|
|
89
|
+
/** Dispatches the last record*() issued: compact 0 for count 0, else the scan's `2 x levels - 1` plus one (`2 x levels`); dedupe 0 for a direct count 0, else 2. */
|
|
90
|
+
readonly lastDispatches: number;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* Prepares the three pipelines and the scan of a scope (compiles once) so record() is synchronous. The planner
|
|
95
|
+
* lives exactly as long as the scope (the scan keeps a scratch word of it): never use it after the scope's dispose().
|
|
96
|
+
* @param scope - the caller's scope (device, caps, cache, scratch, params)
|
|
97
|
+
* @returns the planner
|
|
98
|
+
*/
|
|
99
|
+
export async function prepareCompact(scope: ReduceScope): Promise<CompactPlanner> {
|
|
100
|
+
const scatter = await scope.pipelines.kernel(kernelSpec("compact-scatter"));
|
|
101
|
+
const claim = await scope.pipelines.kernel(kernelSpec("dedupe-claim"));
|
|
102
|
+
const filter = await scope.pipelines.kernel(kernelSpec("dedupe-filter"));
|
|
103
|
+
const scan = await prepareScan(scope);
|
|
104
|
+
return new CompactPlannerImpl(scope, scatter, claim, filter, scan);
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* The E_INVALID_ARGUMENT of a bad count.
|
|
109
|
+
* @param what - the primitive's name for the message
|
|
110
|
+
* @param count - the count
|
|
111
|
+
*/
|
|
112
|
+
function checkCount(what: string, count: number): void {
|
|
113
|
+
if (!Number.isSafeInteger(count) || count < 0 || count > U32_MAX) {
|
|
114
|
+
throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${what}: count must be a non-negative integer below 2^32`, {
|
|
115
|
+
argument: "count",
|
|
116
|
+
value: count,
|
|
117
|
+
});
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* The E_INVALID_ARGUMENT of a binding shorter than `words` u32.
|
|
123
|
+
* @param what - the primitive's name for the message
|
|
124
|
+
* @param name - the argument name
|
|
125
|
+
* @param binding - the binding
|
|
126
|
+
* @param words - the words it must hold
|
|
127
|
+
*/
|
|
128
|
+
function checkWords(what: string, name: string, binding: Binding, words: number): void {
|
|
129
|
+
if (binding.size < 4 * words) {
|
|
130
|
+
throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${what}: ${name} is smaller than 4 x ${words} bytes`, {
|
|
131
|
+
argument: name,
|
|
132
|
+
value: binding.size,
|
|
133
|
+
expected: 4 * words,
|
|
134
|
+
});
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/**
|
|
139
|
+
* The E_INVALID_ARGUMENT of a word index outside a block.
|
|
140
|
+
* @param what - the primitive's name for the message
|
|
141
|
+
* @param name - the argument name
|
|
142
|
+
* @param index - the word index
|
|
143
|
+
* @param block - the block it indexes
|
|
144
|
+
*/
|
|
145
|
+
function checkWordIndex(what: string, name: string, index: number, block: Binding): void {
|
|
146
|
+
if (!Number.isSafeInteger(index) || index < 0 || 4 * (index + 1) > block.size) {
|
|
147
|
+
throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${what}: ${name} ${index} is outside the block`, {
|
|
148
|
+
argument: name,
|
|
149
|
+
value: index,
|
|
150
|
+
expected: `0 <= ${name} < ${Math.floor(block.size / 4)}`,
|
|
151
|
+
});
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/**
|
|
156
|
+
* The argument checks of record() (E_INVALID_ARGUMENT before anything is recorded).
|
|
157
|
+
* @param r - the record
|
|
158
|
+
*/
|
|
159
|
+
function checkCompactArguments(r: CompactRecord): void {
|
|
160
|
+
checkCount("compact", r.count);
|
|
161
|
+
checkWords("compact", "queue", r.queue, r.count);
|
|
162
|
+
checkWords("compact", "flags", r.flags, r.count);
|
|
163
|
+
checkWords("compact", "out", r.out, r.count);
|
|
164
|
+
checkWordIndex("compact", "outIndex", r.outIndex, r.outCount);
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* The argument checks of recordDedupe(): the count, the two entry buffers (`out` holds at
|
|
169
|
+
* most one entry per owner word, so it is checked against `min(count, owner words)`: the SSSP piles of P8-T9 dedupe
|
|
170
|
+
* a raw half of `arcCount` capacity into an `n`-word pile), the output word, and -- with a device count -- a count
|
|
171
|
+
* word that is inside `counters`, is not the output word (the filter's atomicAdd would corrupt the count other
|
|
172
|
+
* workgroups of the same dispatch still read) and lives in the block `outCount` names (the two kernels read it
|
|
173
|
+
* through their own binding).
|
|
174
|
+
* @param r - the record
|
|
175
|
+
*/
|
|
176
|
+
function checkDedupeArguments(r: DedupeRecord): void {
|
|
177
|
+
checkCount("dedupe", r.count);
|
|
178
|
+
checkWords("dedupe", "queue", r.queue, r.count);
|
|
179
|
+
checkWords("dedupe", "out", r.out, Math.min(r.count, Math.floor(r.owner.size / 4)));
|
|
180
|
+
checkWordIndex("dedupe", "outIndex", r.outIndex, r.outCount);
|
|
181
|
+
if (r.countIndex === U32_MAX) {
|
|
182
|
+
return;
|
|
183
|
+
}
|
|
184
|
+
checkWordIndex("dedupe", "countIndex", r.countIndex, r.counters);
|
|
185
|
+
if (r.countIndex === r.outIndex) {
|
|
186
|
+
throw new WebGpuGraphError("E_INVALID_ARGUMENT", "dedupe: countIndex must not be the output word outIndex", {
|
|
187
|
+
argument: "countIndex",
|
|
188
|
+
value: r.countIndex,
|
|
189
|
+
});
|
|
190
|
+
}
|
|
191
|
+
if (
|
|
192
|
+
r.counters.buffer !== r.outCount.buffer ||
|
|
193
|
+
r.counters.offset !== r.outCount.offset ||
|
|
194
|
+
r.counters.size !== r.outCount.size
|
|
195
|
+
) {
|
|
196
|
+
throw new WebGpuGraphError(
|
|
197
|
+
"E_INVALID_ARGUMENT",
|
|
198
|
+
"dedupe: counters must be the same range as outCount when countIndex names a device word",
|
|
199
|
+
{
|
|
200
|
+
argument: "counters",
|
|
201
|
+
value: r.counters.size,
|
|
202
|
+
expected: r.outCount.size,
|
|
203
|
+
},
|
|
204
|
+
);
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/** The two bound dedupe kernels and the dynamic offset of the one params record they share. */
|
|
209
|
+
interface BoundDedupe {
|
|
210
|
+
readonly claim: BoundKernel;
|
|
211
|
+
readonly filter: BoundKernel;
|
|
212
|
+
readonly offset: number;
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
/** The planner: the three resolved kernels and the scan over one scope. */
|
|
216
|
+
class CompactPlannerImpl implements CompactPlanner {
|
|
217
|
+
private readonly scope: ReduceScope;
|
|
218
|
+
private readonly scatter: Kernel;
|
|
219
|
+
private readonly claim: Kernel;
|
|
220
|
+
private readonly filter: Kernel;
|
|
221
|
+
private readonly scan: ScanPlanner;
|
|
222
|
+
private dispatches = 0;
|
|
223
|
+
|
|
224
|
+
/**
|
|
225
|
+
* Wraps the resolved kernels; use prepareCompact().
|
|
226
|
+
* @param scope - the caller's scope
|
|
227
|
+
* @param scatter - the `compact-scatter` kernel
|
|
228
|
+
* @param claim - the `dedupe-claim` kernel
|
|
229
|
+
* @param filter - the `dedupe-filter` kernel
|
|
230
|
+
* @param scan - the scan planner of the same scope
|
|
231
|
+
*/
|
|
232
|
+
constructor(scope: ReduceScope, scatter: Kernel, claim: Kernel, filter: Kernel, scan: ScanPlanner) {
|
|
233
|
+
this.scope = scope;
|
|
234
|
+
this.scatter = scatter;
|
|
235
|
+
this.claim = claim;
|
|
236
|
+
this.filter = filter;
|
|
237
|
+
this.scan = scan;
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
/**
|
|
241
|
+
* Dispatches the last record*() issued.
|
|
242
|
+
* @returns the count
|
|
243
|
+
*/
|
|
244
|
+
get lastDispatches(): number {
|
|
245
|
+
return this.dispatches;
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/**
|
|
249
|
+
* Records the scan and the scatter (see the interface).
|
|
250
|
+
* @param pass - the compute pass
|
|
251
|
+
* @param r - the record
|
|
252
|
+
*/
|
|
253
|
+
record(pass: GPUComputePassEncoder, r: CompactRecord): void {
|
|
254
|
+
checkCompactArguments(r);
|
|
255
|
+
if (r.count === 0) {
|
|
256
|
+
this.dispatches = 0;
|
|
257
|
+
return;
|
|
258
|
+
}
|
|
259
|
+
const size = 4 * r.count;
|
|
260
|
+
const offsets: Binding = { buffer: this.scope.scratch(size, "compact/offsets"), offset: 0, size, window: null };
|
|
261
|
+
this.scan.record(pass, r.flags, r.count, offsets);
|
|
262
|
+
const params = this.scope.params(COMPACT_PARAMS, {
|
|
263
|
+
count: r.count,
|
|
264
|
+
outIndex: r.outIndex,
|
|
265
|
+
countIndex: U32_MAX,
|
|
266
|
+
stride: 0,
|
|
267
|
+
});
|
|
268
|
+
const bound = this.scatter.bind({
|
|
269
|
+
queue: r.queue,
|
|
270
|
+
flags: r.flags,
|
|
271
|
+
offsets,
|
|
272
|
+
out: r.out,
|
|
273
|
+
outCount: r.outCount,
|
|
274
|
+
P: params.binding,
|
|
275
|
+
});
|
|
276
|
+
this.scatter.dispatch(pass, bound, plan1d(r.count, this.scope.workgroupSize, this.scope.caps), [params.offset]);
|
|
277
|
+
this.dispatches = this.scan.lastDispatches + 1;
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
/**
|
|
281
|
+
* Records the claim and the filter over plan1d(count) (see the interface).
|
|
282
|
+
* @param pass - the compute pass
|
|
283
|
+
* @param r - the record
|
|
284
|
+
*/
|
|
285
|
+
recordDedupe(pass: GPUComputePassEncoder, r: DedupeRecord): void {
|
|
286
|
+
checkDedupeArguments(r);
|
|
287
|
+
if (r.count === 0) {
|
|
288
|
+
this.dispatches = 0;
|
|
289
|
+
return;
|
|
290
|
+
}
|
|
291
|
+
const plan = planGridStride(r.count, this.scope.workgroupSize, this.scope.caps);
|
|
292
|
+
const bound = this.bindDedupe(r, plan.stride ?? this.scope.workgroupSize);
|
|
293
|
+
this.claim.dispatch(pass, bound.claim, plan, [bound.offset]);
|
|
294
|
+
this.filter.dispatch(pass, bound.filter, plan, [bound.offset]);
|
|
295
|
+
this.dispatches = 2;
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
/**
|
|
299
|
+
* One params record (both kernels read the same values) and the two bind groups.
|
|
300
|
+
* @param r - the record
|
|
301
|
+
* @param stride - the grid-stride plan's stride (entries per pass over the grid)
|
|
302
|
+
* @returns the bound kernels and the record's dynamic offset
|
|
303
|
+
*/
|
|
304
|
+
private bindDedupe(r: DedupeRecord, stride: number): BoundDedupe {
|
|
305
|
+
const params = this.scope.params(COMPACT_PARAMS, {
|
|
306
|
+
count: r.count,
|
|
307
|
+
outIndex: r.outIndex,
|
|
308
|
+
countIndex: r.countIndex,
|
|
309
|
+
stride,
|
|
310
|
+
});
|
|
311
|
+
return {
|
|
312
|
+
claim: this.claim.bind({ queue: r.queue, owner: r.owner, counters: r.counters, P: params.binding }),
|
|
313
|
+
filter: this.filter.bind({
|
|
314
|
+
queue: r.queue,
|
|
315
|
+
owner: r.owner,
|
|
316
|
+
out: r.out,
|
|
317
|
+
outCount: r.outCount,
|
|
318
|
+
P: params.binding,
|
|
319
|
+
}),
|
|
320
|
+
offset: params.offset,
|
|
321
|
+
};
|
|
322
|
+
}
|
|
323
|
+
}
|
|
@@ -107,6 +107,40 @@ export function windowBinding(core: CoreBinding, name: "colIdx" | "weights" | "a
|
|
|
107
107
|
return { buffer, offset: w.offset, size: 4 * (w.end - w.start), window: w };
|
|
108
108
|
}
|
|
109
109
|
|
|
110
|
+
/** One dispatch of a frontier-walking kernel over a core (P8-T12): the arc range it owns and the core with its arc-indexed arrays bound to that window (the core itself, over every arc, when it is not windowed). */
|
|
111
|
+
export interface CoreWindow {
|
|
112
|
+
readonly arcBase: number;
|
|
113
|
+
readonly arcEnd: number;
|
|
114
|
+
readonly core: CoreBinding;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* The per-window dispatches of a kernel that walks rows named by a FRONTIER rather than by a row range (P8-T12:
|
|
119
|
+
* `advance-expand`, `bfs-fused`, `bfs-bottom-up`, `sssp-pred`): every window sees every frontier entry and clips
|
|
120
|
+
* each row to `[arcBase, arcEnd)`, so the ranges must PARTITION `[0, arcCount)` or a row is expanded twice. P4's
|
|
121
|
+
* windows do not: `planArcWindows` opens a window at the previous window's end aligned DOWN to 64 arcs, so two
|
|
122
|
+
* consecutive windows overlap by up to 63 arcs at an unaligned row boundary (the overlap belongs to the earlier
|
|
123
|
+
* window's rows, which a row-range dispatch such as `degree` never revisits). Here a window owns `[start, next
|
|
124
|
+
* window's start)` -- the last one `[start, end)` -- which is inside its binding (`next.start <= end`) and
|
|
125
|
+
* partitions the arcs exactly; a window whose owned range is empty dispatches, harmlessly, over nothing.
|
|
126
|
+
* @param core - the core (windowed or not)
|
|
127
|
+
* @returns one entry per window, in arc order; exactly one entry over `[0, arcCount)` when the core is not windowed
|
|
128
|
+
*/
|
|
129
|
+
export function coreWindows(core: CoreBinding): readonly CoreWindow[] {
|
|
130
|
+
if (core.windows === null) {
|
|
131
|
+
return [{ arcBase: 0, arcEnd: arcCountOf(core), core }];
|
|
132
|
+
}
|
|
133
|
+
return core.windows.map((w, k, all) => ({
|
|
134
|
+
arcBase: w.start,
|
|
135
|
+
arcEnd: k + 1 < all.length ? all[k + 1].start : w.end,
|
|
136
|
+
core: {
|
|
137
|
+
...core,
|
|
138
|
+
colIdx: windowBinding(core, "colIdx", w),
|
|
139
|
+
weights: core.weights === null ? null : windowBinding(core, "weights", w),
|
|
140
|
+
},
|
|
141
|
+
}));
|
|
142
|
+
}
|
|
143
|
+
|
|
110
144
|
/**
|
|
111
145
|
* Rejects a windowed core: the pull cannot accumulate its affine epilogue across windows (DEP-P4-B).
|
|
112
146
|
* @param core - the core
|
|
@@ -146,7 +180,9 @@ export function assertWholeCore(core: CoreBinding, arcCount: number, limit: numb
|
|
|
146
180
|
* package. `plan` is "perArray" because views always upload per array (spec 4.3 lines 1182-1186) and `windows` is
|
|
147
181
|
* null because a view is never windowed, which is what makes assertNotWindowed pass for a view. `serial` is -1: a
|
|
148
182
|
* view is not a core and no caller of this function reads serial (nothing in src/primitives or src/algorithms
|
|
149
|
-
* reads CoreBinding.serial).
|
|
183
|
+
* reads CoreBinding.serial). A zero-size `colIdx` or `weights` binding (the reverse view of an arc-less directed
|
|
184
|
+
* snapshot uploads zero-byte arc arrays) becomes null, the core's own spelling of "no arcs", so `graphBindings` binds
|
|
185
|
+
* the dummy instead of a zero-length range (P8-T8: the one-node directed fixture of the BFS suite).
|
|
150
186
|
* @param v - the view binding, from residency.view(s, "reverse") or view(s, "edgeList")
|
|
151
187
|
* @param arcCount - the arc count of the view, from v.scalars.arcCount[0]; it must agree with the colIdx binding,
|
|
152
188
|
* which is what record() derives the arc window from (E_INVALID_ARGUMENT { argument: "arcCount" } otherwise)
|
|
@@ -161,8 +197,10 @@ export function coreOfView(v: ViewBinding, arcCount: number): CoreBinding {
|
|
|
161
197
|
expected: "a view with a rowPtr binding (reverse)",
|
|
162
198
|
});
|
|
163
199
|
}
|
|
164
|
-
|
|
165
|
-
|
|
200
|
+
// an arc-less DIRECTED snapshot's reverse view uploads zero-byte arc arrays; the core spells "no arcs" as null
|
|
201
|
+
// (graphBindings then binds the rowPtr dummy), and a zero-size binding is never bound (spec 3.6)
|
|
202
|
+
const colIdx = v.bindings.colIdx === undefined || v.bindings.colIdx.size === 0 ? null : v.bindings.colIdx;
|
|
203
|
+
const weights = v.bindings.weights === undefined || v.bindings.weights.size === 0 ? null : v.bindings.weights;
|
|
166
204
|
const bound = colIdx === null ? 0 : colIdx.size / 4;
|
|
167
205
|
if (bound !== arcCount) {
|
|
168
206
|
throw new WebGpuGraphError(
|