@graphty/webgpu-graph-algorithms 0.6.14 → 0.6.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +52 -52
- package/dist/browser.js +1 -1
- package/dist/chunks/{context-Bi6AhScG.js → context-VIvatQOo.js} +69 -34
- package/dist/chunks/context-VIvatQOo.js.map +1 -0
- package/dist/node.js +1 -1
- package/dist/src/accelerator.d.ts +5 -3
- package/dist/src/accelerator.d.ts.map +1 -1
- package/dist/src/accelerator.js +101 -5
- package/dist/src/accelerator.js.map +1 -1
- package/dist/src/algorithms/all-pairs.d.ts +41 -0
- package/dist/src/algorithms/all-pairs.d.ts.map +1 -0
- package/dist/src/algorithms/all-pairs.js +181 -0
- package/dist/src/algorithms/all-pairs.js.map +1 -0
- package/dist/src/algorithms/betweenness.d.ts +70 -0
- package/dist/src/algorithms/betweenness.d.ts.map +1 -0
- package/dist/src/algorithms/betweenness.js +538 -0
- package/dist/src/algorithms/betweenness.js.map +1 -0
- package/dist/src/algorithms/closeness.d.ts +15 -5
- package/dist/src/algorithms/closeness.d.ts.map +1 -1
- package/dist/src/algorithms/closeness.js +112 -26
- package/dist/src/algorithms/closeness.js.map +1 -1
- package/dist/src/algorithms/components.d.ts +9 -1
- package/dist/src/algorithms/components.d.ts.map +1 -1
- package/dist/src/algorithms/components.js +2 -2
- package/dist/src/algorithms/components.js.map +1 -1
- package/dist/src/algorithms/label-propagation.d.ts +31 -0
- package/dist/src/algorithms/label-propagation.d.ts.map +1 -0
- package/dist/src/algorithms/label-propagation.js +254 -0
- package/dist/src/algorithms/label-propagation.js.map +1 -0
- package/dist/src/algorithms/simple-symmetric.d.ts +88 -0
- package/dist/src/algorithms/simple-symmetric.d.ts.map +1 -0
- package/dist/src/algorithms/simple-symmetric.js +347 -0
- package/dist/src/algorithms/simple-symmetric.js.map +1 -0
- package/dist/src/algorithms/triangles.d.ts +34 -0
- package/dist/src/algorithms/triangles.d.ts.map +1 -0
- package/dist/src/algorithms/triangles.js +203 -0
- package/dist/src/algorithms/triangles.js.map +1 -0
- package/dist/src/constants.d.ts +53 -0
- package/dist/src/constants.d.ts.map +1 -1
- package/dist/src/constants.js +53 -0
- package/dist/src/constants.js.map +1 -1
- package/dist/src/index.d.ts +12 -3
- package/dist/src/index.d.ts.map +1 -1
- package/dist/src/index.js +8 -1
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/prelude.d.ts.map +1 -1
- package/dist/src/kernel/prelude.js +4 -1
- package/dist/src/kernel/prelude.js.map +1 -1
- package/dist/src/kernels.d.ts +24 -6
- package/dist/src/kernels.d.ts.map +1 -1
- package/dist/src/kernels.js +373 -7
- package/dist/src/kernels.js.map +1 -1
- package/dist/src/memory/residency.js +15 -4
- package/dist/src/memory/residency.js.map +1 -1
- package/dist/src/primitives/coo-to-csr.d.ts +73 -0
- package/dist/src/primitives/coo-to-csr.d.ts.map +1 -0
- package/dist/src/primitives/coo-to-csr.js +183 -0
- package/dist/src/primitives/coo-to-csr.js.map +1 -0
- package/dist/src/primitives/frontier.d.ts +2 -0
- package/dist/src/primitives/frontier.d.ts.map +1 -1
- package/dist/src/primitives/frontier.js +2 -0
- package/dist/src/primitives/frontier.js.map +1 -1
- package/dist/src/primitives/group-by-key.d.ts +82 -0
- package/dist/src/primitives/group-by-key.d.ts.map +1 -0
- package/dist/src/primitives/group-by-key.js +147 -0
- package/dist/src/primitives/group-by-key.js.map +1 -0
- package/dist/src/types/accelerator.d.ts +19 -7
- package/dist/src/types/accelerator.d.ts.map +1 -1
- package/dist/src/types/algorithms.d.ts +4 -0
- package/dist/src/types/algorithms.d.ts.map +1 -1
- package/dist/src/types/all-pairs.d.ts +35 -0
- package/dist/src/types/all-pairs.d.ts.map +1 -0
- package/dist/src/types/all-pairs.js +8 -0
- package/dist/src/types/all-pairs.js.map +1 -0
- package/dist/src/types/betweenness.d.ts +35 -0
- package/dist/src/types/betweenness.d.ts.map +1 -0
- package/dist/src/types/betweenness.js +7 -0
- package/dist/src/types/betweenness.js.map +1 -0
- package/dist/src/types/community.d.ts +18 -0
- package/dist/src/types/community.d.ts.map +1 -0
- package/dist/src/types/community.js +5 -0
- package/dist/src/types/community.js.map +1 -0
- package/dist/src/types/structure.d.ts +27 -0
- package/dist/src/types/structure.d.ts.map +1 -0
- package/dist/src/types/structure.js +8 -0
- package/dist/src/types/structure.js.map +1 -0
- package/dist/src/wgsl/apsp-fw.wgsl.d.ts +25 -0
- package/dist/src/wgsl/apsp-fw.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/apsp-fw.wgsl.js +113 -0
- package/dist/src/wgsl/apsp-fw.wgsl.js.map +1 -0
- package/dist/src/wgsl/apsp-init.wgsl.d.ts +12 -0
- package/dist/src/wgsl/apsp-init.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/apsp-init.wgsl.js +26 -0
- package/dist/src/wgsl/apsp-init.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-backward.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bc-backward.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-backward.wgsl.js +34 -0
- package/dist/src/wgsl/bc-backward.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts +12 -0
- package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-edge-gather.wgsl.js +36 -0
- package/dist/src/wgsl/bc-edge-gather.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-finalize.wgsl.d.ts +21 -0
- package/dist/src/wgsl/bc-finalize.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-finalize.wgsl.js +47 -0
- package/dist/src/wgsl/bc-finalize.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts +15 -0
- package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-forward-edge.wgsl.js +76 -0
- package/dist/src/wgsl/bc-forward-edge.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-forward.wgsl.d.ts +23 -0
- package/dist/src/wgsl/bc-forward.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-forward.wgsl.js +106 -0
- package/dist/src/wgsl/bc-forward.wgsl.js.map +1 -0
- package/dist/src/wgsl/bc-gather.wgsl.d.ts +9 -0
- package/dist/src/wgsl/bc-gather.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/bc-gather.wgsl.js +20 -0
- package/dist/src/wgsl/bc-gather.wgsl.js.map +1 -0
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +4 -1
- package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/closeness-reduce.wgsl.js +8 -4
- package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -1
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +4 -2
- package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -1
- package/dist/src/wgsl/closeness-sweep.wgsl.js +12 -2
- package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -1
- package/dist/src/wgsl/coo-emit.wgsl.d.ts +10 -0
- package/dist/src/wgsl/coo-emit.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/coo-emit.wgsl.js +33 -0
- package/dist/src/wgsl/coo-emit.wgsl.js.map +1 -0
- package/dist/src/wgsl/coo-scatter.wgsl.d.ts +15 -0
- package/dist/src/wgsl/coo-scatter.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/coo-scatter.wgsl.js +32 -0
- package/dist/src/wgsl/coo-scatter.wgsl.js.map +1 -0
- package/dist/src/wgsl/group-by-key-row.wgsl.d.ts +26 -0
- package/dist/src/wgsl/group-by-key-row.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/group-by-key-row.wgsl.js +146 -0
- package/dist/src/wgsl/group-by-key-row.wgsl.js.map +1 -0
- package/dist/src/wgsl/lpa-step.wgsl.d.ts +10 -0
- package/dist/src/wgsl/lpa-step.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/lpa-step.wgsl.js +35 -0
- package/dist/src/wgsl/lpa-step.wgsl.js.map +1 -0
- package/dist/src/wgsl/orient-flags.wgsl.d.ts +9 -0
- package/dist/src/wgsl/orient-flags.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/orient-flags.wgsl.js +21 -0
- package/dist/src/wgsl/orient-flags.wgsl.js.map +1 -0
- package/dist/src/wgsl/run-flags.wgsl.d.ts +8 -0
- package/dist/src/wgsl/run-flags.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/run-flags.wgsl.js +18 -0
- package/dist/src/wgsl/run-flags.wgsl.js.map +1 -0
- package/dist/src/wgsl/tri-intersect.wgsl.d.ts +11 -0
- package/dist/src/wgsl/tri-intersect.wgsl.d.ts.map +1 -0
- package/dist/src/wgsl/tri-intersect.wgsl.js +64 -0
- package/dist/src/wgsl/tri-intersect.wgsl.js.map +1 -0
- package/dist/webgpu-graph-algorithms.js +2828 -321
- package/dist/webgpu-graph-algorithms.js.map +1 -1
- package/package.json +5 -5
- package/src/accelerator.ts +130 -7
- package/src/algorithms/all-pairs.ts +228 -0
- package/src/algorithms/betweenness.ts +739 -0
- package/src/algorithms/closeness.ts +124 -32
- package/src/algorithms/components.ts +2 -2
- package/src/algorithms/label-propagation.ts +280 -0
- package/src/algorithms/simple-symmetric.ts +409 -0
- package/src/algorithms/triangles.ts +240 -0
- package/src/constants.ts +53 -0
- package/src/index.ts +20 -1
- package/src/kernel/prelude.ts +6 -0
- package/src/kernels.ts +411 -10
- package/src/memory/residency.ts +15 -4
- package/src/primitives/coo-to-csr.ts +251 -0
- package/src/primitives/frontier.ts +4 -0
- package/src/primitives/group-by-key.ts +209 -0
- package/src/types/accelerator.ts +26 -6
- package/src/types/algorithms.ts +5 -0
- package/src/types/all-pairs.ts +37 -0
- package/src/types/betweenness.ts +38 -0
- package/src/types/community.ts +18 -0
- package/src/types/structure.ts +28 -0
- package/src/wgsl/apsp-fw.wgsl.ts +112 -0
- package/src/wgsl/apsp-init.wgsl.ts +25 -0
- package/src/wgsl/bc-backward.wgsl.ts +33 -0
- package/src/wgsl/bc-edge-gather.wgsl.ts +35 -0
- package/src/wgsl/bc-finalize.wgsl.ts +46 -0
- package/src/wgsl/bc-forward-edge.wgsl.ts +75 -0
- package/src/wgsl/bc-forward.wgsl.ts +105 -0
- package/src/wgsl/bc-gather.wgsl.ts +19 -0
- package/src/wgsl/closeness-reduce.wgsl.ts +8 -4
- package/src/wgsl/closeness-sweep.wgsl.ts +12 -2
- package/src/wgsl/coo-emit.wgsl.ts +32 -0
- package/src/wgsl/coo-scatter.wgsl.ts +31 -0
- package/src/wgsl/group-by-key-row.wgsl.ts +145 -0
- package/src/wgsl/lpa-step.wgsl.ts +34 -0
- package/src/wgsl/orient-flags.wgsl.ts +20 -0
- package/src/wgsl/run-flags.wgsl.ts +17 -0
- package/src/wgsl/tri-intersect.wgsl.ts +63 -0
- package/dist/chunks/context-Bi6AhScG.js.map +0 -1
package/src/kernels.ts
CHANGED
|
@@ -12,7 +12,12 @@
|
|
|
12
12
|
* dedupe-claim and dedupe-filter; P8-T4 adds frontier-finalize with the FrontierCounters and FrontierParams blocks;
|
|
13
13
|
* P8-T5 adds advance-expand; P8-T6 adds bfs-contract and sssp-pred; P8-T7 adds bfs-fused; P8-T8 adds bfs-bottom-up,
|
|
14
14
|
* bfs-bitset-build and bfs-unvisited-flags; P8-T9 adds sssp-relax; P8-T10 adds bf-relax with the BfParams and BfFlags
|
|
15
|
-
* blocks; P8-T11 adds closeness-sweep and closeness-reduce
|
|
15
|
+
* blocks; P8-T11 adds closeness-sweep and closeness-reduce; P9 (betweenness) adds bc-finalize, bc-forward,
|
|
16
|
+
* bc-backward, bc-gather, bc-edge-gather and bc-forward-edge with the BcParams block; all-pairs shortest paths
|
|
17
|
+
* (design 8.7) adds apsp-init and apsp-fw with the ApspParams block. P11 (the structure and community phase, plan
|
|
18
|
+
* design/webgpu/plans/2026-09-23-webgpu-p11-structure-and-community.md) adds the graph build on the device (coo-emit,
|
|
19
|
+
* run-flags, coo-scatter), the per-row group-by-key (group-by-key-row), label propagation's step (lpa-step) and
|
|
20
|
+
* triangle counting (orient-flags, tri-intersect). This file is the only importer of src/wgsl/** (spec 3.2;
|
|
16
21
|
* test/layers.test.ts).
|
|
17
22
|
*/
|
|
18
23
|
|
|
@@ -23,6 +28,14 @@ import { type BindingDecl, type OverrideDecl, type WgslModuleSpec } from "./kern
|
|
|
23
28
|
import { type CoreBinding } from "./memory/residency.js";
|
|
24
29
|
import { type Binding } from "./types/memory.js";
|
|
25
30
|
import { advanceExpandWgsl } from "./wgsl/advance-expand.wgsl.js";
|
|
31
|
+
import { apspFwWgsl } from "./wgsl/apsp-fw.wgsl.js";
|
|
32
|
+
import { apspInitWgsl } from "./wgsl/apsp-init.wgsl.js";
|
|
33
|
+
import { bcBackwardWgsl } from "./wgsl/bc-backward.wgsl.js";
|
|
34
|
+
import { bcEdgeGatherWgsl } from "./wgsl/bc-edge-gather.wgsl.js";
|
|
35
|
+
import { bcFinalizeWgsl } from "./wgsl/bc-finalize.wgsl.js";
|
|
36
|
+
import { bcForwardWgsl } from "./wgsl/bc-forward.wgsl.js";
|
|
37
|
+
import { bcForwardEdgeWgsl } from "./wgsl/bc-forward-edge.wgsl.js";
|
|
38
|
+
import { bcGatherWgsl } from "./wgsl/bc-gather.wgsl.js";
|
|
26
39
|
import { bfRelaxWgsl } from "./wgsl/bf-relax.wgsl.js";
|
|
27
40
|
import { bfsBitsetBuildWgsl } from "./wgsl/bfs-bitset-build.wgsl.js";
|
|
28
41
|
import { bfsBottomUpWgsl } from "./wgsl/bfs-bottom-up.wgsl.js";
|
|
@@ -33,6 +46,8 @@ import { bfsUnvisitedFlagsWgsl } from "./wgsl/bfs-unvisited-flags.wgsl.js";
|
|
|
33
46
|
import { closenessReduceWgsl } from "./wgsl/closeness-reduce.wgsl.js";
|
|
34
47
|
import { closenessSweepWgsl } from "./wgsl/closeness-sweep.wgsl.js";
|
|
35
48
|
import { compactScatterWgsl } from "./wgsl/compact-scatter.wgsl.js";
|
|
49
|
+
import { cooEmitWgsl } from "./wgsl/coo-emit.wgsl.js";
|
|
50
|
+
import { cooScatterWgsl } from "./wgsl/coo-scatter.wgsl.js";
|
|
36
51
|
import { countingScatterWgsl } from "./wgsl/counting-scatter.wgsl.js";
|
|
37
52
|
import { dedupeClaimWgsl } from "./wgsl/dedupe-claim.wgsl.js";
|
|
38
53
|
import { dedupeFilterWgsl } from "./wgsl/dedupe-filter.wgsl.js";
|
|
@@ -51,25 +66,30 @@ import { gridCentroidHubWgsl } from "./wgsl/grid-centroid-hub.wgsl.js";
|
|
|
51
66
|
import { gridDownsampleWgsl } from "./wgsl/grid-downsample.wgsl.js";
|
|
52
67
|
import { gridFarFieldWgsl } from "./wgsl/grid-far-field.wgsl.js";
|
|
53
68
|
import { gridNearFieldWgsl } from "./wgsl/grid-near-field.wgsl.js";
|
|
69
|
+
import { groupByKeyRowWgsl } from "./wgsl/group-by-key-row.wgsl.js";
|
|
54
70
|
import { histogramWgsl } from "./wgsl/histogram.wgsl.js";
|
|
55
71
|
import { indirectFinalizeWgsl } from "./wgsl/indirect-finalize.wgsl.js";
|
|
72
|
+
import { lpaStepWgsl } from "./wgsl/lpa-step.wgsl.js";
|
|
73
|
+
import { orientFlagsWgsl } from "./wgsl/orient-flags.wgsl.js";
|
|
56
74
|
import { prFinalizeWgsl } from "./wgsl/pr-finalize.wgsl.js";
|
|
57
75
|
import { prScaleWgsl } from "./wgsl/pr-scale.wgsl.js";
|
|
58
76
|
import { radixHistWgsl } from "./wgsl/radix-hist.wgsl.js";
|
|
59
77
|
import { radixScatterWgsl } from "./wgsl/radix-scatter.wgsl.js";
|
|
60
78
|
import { reduceWgsl } from "./wgsl/reduce.wgsl.js";
|
|
79
|
+
import { runFlagsWgsl } from "./wgsl/run-flags.wgsl.js";
|
|
61
80
|
import { scanAddWgsl } from "./wgsl/scan-add.wgsl.js";
|
|
62
81
|
import { scanBlockWgsl } from "./wgsl/scan-block.wgsl.js";
|
|
63
82
|
import { segmentedReduceWgsl } from "./wgsl/segmented-reduce.wgsl.js";
|
|
64
83
|
import { spmvPullWgsl } from "./wgsl/spmv-pull.wgsl.js";
|
|
65
84
|
import { ssspPredWgsl } from "./wgsl/sssp-pred.wgsl.js";
|
|
66
85
|
import { ssspRelaxWgsl } from "./wgsl/sssp-relax.wgsl.js";
|
|
86
|
+
import { triIntersectWgsl } from "./wgsl/tri-intersect.wgsl.js";
|
|
67
87
|
import { wccCompressWgsl } from "./wgsl/wcc-compress.wgsl.js";
|
|
68
88
|
import { wccLinkEdgesWgsl } from "./wgsl/wcc-link-edges.wgsl.js";
|
|
69
89
|
import { wccLinkSampleWgsl } from "./wgsl/wcc-link-sample.wgsl.js";
|
|
70
90
|
import { wccSampleWgsl } from "./wgsl/wcc-sample.wgsl.js";
|
|
71
91
|
|
|
72
|
-
/** Every module id of P1-P4, P7 and
|
|
92
|
+
/** Every module id of P1-P4, P7, P8, P9 and P11 (later ids are appended, never renamed). */
|
|
73
93
|
export type KernelId =
|
|
74
94
|
| "degree"
|
|
75
95
|
| "reduce"
|
|
@@ -116,7 +136,22 @@ export type KernelId =
|
|
|
116
136
|
| "sssp-relax"
|
|
117
137
|
| "bf-relax"
|
|
118
138
|
| "closeness-sweep"
|
|
119
|
-
| "closeness-reduce"
|
|
139
|
+
| "closeness-reduce"
|
|
140
|
+
| "bc-finalize"
|
|
141
|
+
| "bc-forward"
|
|
142
|
+
| "bc-backward"
|
|
143
|
+
| "bc-gather"
|
|
144
|
+
| "bc-edge-gather"
|
|
145
|
+
| "bc-forward-edge"
|
|
146
|
+
| "apsp-init"
|
|
147
|
+
| "apsp-fw"
|
|
148
|
+
| "coo-emit"
|
|
149
|
+
| "run-flags"
|
|
150
|
+
| "coo-scatter"
|
|
151
|
+
| "orient-flags"
|
|
152
|
+
| "tri-intersect"
|
|
153
|
+
| "group-by-key-row"
|
|
154
|
+
| "lpa-step";
|
|
120
155
|
|
|
121
156
|
/** One registry entry: everything of a WgslModuleSpec except the per-variant overrides and snippets. */
|
|
122
157
|
export interface KernelEntry {
|
|
@@ -131,7 +166,7 @@ export interface KernelEntry {
|
|
|
131
166
|
/** The snippet marker names the body carries (segmented-reduce: ["VALUE"]). */
|
|
132
167
|
readonly snippetSlots: readonly string[];
|
|
133
168
|
/** The phase the entry landed in (documentation and the compile-matrix filter). */
|
|
134
|
-
readonly phase: "P1" | "P2" | "P3" | "P4" | "P7" | "P8";
|
|
169
|
+
readonly phase: "P1" | "P2" | "P3" | "P4" | "P7" | "P8" | "P9" | "P11";
|
|
135
170
|
}
|
|
136
171
|
|
|
137
172
|
// ---- the generated blocks (spec 5.3; contract 3.10.2): field order = byte order, offsets in the JSDoc
|
|
@@ -392,7 +427,9 @@ export const COMPACT_PARAMS: UniformBlock = UniformBlock.define("CompactParams",
|
|
|
392
427
|
* one; every level kernel is a direct dispatch that reads it first -- G8-F5), `nextDegreeSum` @100 (issue #391: the
|
|
393
428
|
* out-degree sum of the vertices the level claimed, accumulated by `bfs-next-degree` at the end of every level and
|
|
394
429
|
* read, subtracted and zeroed by the next boundary -- Beamer's m_f measured on the frontier the boundary decides
|
|
395
|
-
* for, not on the one it has just expanded)
|
|
430
|
+
* for, not on the one it has just expanded); `stackTop` @104 (betweenness: the append cursor of the claim log, which
|
|
431
|
+
* `bc-finalize` also writes into `ends` at every level boundary) and `sigmaOverflow` @108 (betweenness: 1 once a u32
|
|
432
|
+
* path count wrapped) are APPENDED so every earlier word keeps its byte offset. The words nothing writes before P8-T8 / P8-T9 are declared now because
|
|
396
433
|
* the byte layout is what the single result copy decodes.
|
|
397
434
|
*/
|
|
398
435
|
export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
@@ -424,6 +461,8 @@ export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
|
424
461
|
["deltaBits", "u32"],
|
|
425
462
|
["path", "u32"],
|
|
426
463
|
["nextDegreeSum", "u32"],
|
|
464
|
+
["stackTop", "u32"],
|
|
465
|
+
["sigmaOverflow", "u32"],
|
|
427
466
|
],
|
|
428
467
|
{ layout: "storage" },
|
|
429
468
|
);
|
|
@@ -436,7 +475,8 @@ export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
|
|
|
436
475
|
* `arcEnd` @44 (the bound arc window), `predKind` @48 (0 arc, 1 node), `bitsBase` @52, `source` @56, `stride` @60
|
|
437
476
|
* (a grid-stride plan's stride), `firstOfSubmit` @64 (the boundary's index inside its submit, clamped to 1: both
|
|
438
477
|
* the unvisited-count and the unvisited-degree-sum subtraction run at >= 1, issue #391), `iteration` @68 (an
|
|
439
|
-
* `sssp-pred` hop pass, P8-T9), `
|
|
478
|
+
* `sssp-pred` hop pass, P8-T9), `perNode` @72 (`closeness-sweep`: 1 also folds every claim into the per-node
|
|
479
|
+
* distance sums of a sampled run), `pad2` @76. The `slotBase` field that once addressed the selector's indirect slots went with
|
|
440
480
|
* the slots (2026-09-25); `pad2` keeps the block an explicit 80 bytes, the way every block here is padded.
|
|
441
481
|
*/
|
|
442
482
|
export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams", [
|
|
@@ -458,10 +498,22 @@ export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams
|
|
|
458
498
|
["stride", "u32"],
|
|
459
499
|
["firstOfSubmit", "u32"],
|
|
460
500
|
["iteration", "u32"],
|
|
461
|
-
["
|
|
501
|
+
["perNode", "u32"],
|
|
462
502
|
["pad2", "u32"],
|
|
463
503
|
]);
|
|
464
504
|
|
|
505
|
+
/** `BcParams` (uniform, 32 B; betweenness): `n` @0, `k` @4 (the batch's sources), `start` @8 (the first log index of a backward level), `count` @12 (a backward level's entries, or the edge count of `bc-forward-edge`), `stride` @16 (a grid-stride plan's stride), `role` @20 (`bc-finalize`: 0 the level boundary, 1 the seed), `pad0` @24, `pad1` @28. */
|
|
506
|
+
export const BC_PARAMS: UniformBlock = UniformBlock.define("BcParams", [
|
|
507
|
+
["n", "u32"],
|
|
508
|
+
["k", "u32"],
|
|
509
|
+
["start", "u32"],
|
|
510
|
+
["count", "u32"],
|
|
511
|
+
["stride", "u32"],
|
|
512
|
+
["role", "u32"],
|
|
513
|
+
["pad0", "u32"],
|
|
514
|
+
["pad1", "u32"],
|
|
515
|
+
]);
|
|
516
|
+
|
|
465
517
|
/** `BfParams` (uniform, 16 B; P8-T10): `edgeCount` @0 (the logical edges of the `edgeList` view), `stride` @4 (the grid-stride plan's stride), `maxRetries` @8 (PD-12's compare-exchange bound), `cutoffBits` @12 (the f32 bit pattern of the CPU port's `cutoff`, `+Inf` when absent). */
|
|
466
518
|
export const BF_PARAMS: UniformBlock = UniformBlock.define("BfParams", [
|
|
467
519
|
["edgeCount", "u32"],
|
|
@@ -482,6 +534,38 @@ export const BF_FLAGS: UniformBlock = UniformBlock.define(
|
|
|
482
534
|
{ layout: "storage" },
|
|
483
535
|
);
|
|
484
536
|
|
|
537
|
+
/** `ApspParams` (uniform, 16 B; design 8.7): `n` @0 (the node count, the matrix side), `round` @4 (the pivot block index of the blocked Floyd-Warshall round), `blocks` @8 (`ceil(n / APSP_TILE)`), `infBits` @12 (`F32_INF_BITS`: a kernel reads `+Infinity` from a uniform because Tint refuses it as a constant expression). */
|
|
538
|
+
export const APSP_PARAMS: UniformBlock = UniformBlock.define("ApspParams", [
|
|
539
|
+
["n", "u32"],
|
|
540
|
+
["round", "u32"],
|
|
541
|
+
["blocks", "u32"],
|
|
542
|
+
["infBits", "u32"],
|
|
543
|
+
]);
|
|
544
|
+
|
|
545
|
+
/** `CooParams` (uniform, 16 B; P11): `count` @0 (the arcs or positions of the dispatch), `pad0` @4, `pad1` @8, `pad2` @12. The one params block of `coo-emit`, `run-flags`, `coo-scatter`, `orient-flags` and `tri-intersect`. */
|
|
546
|
+
export const COO_PARAMS: UniformBlock = UniformBlock.define("CooParams", [
|
|
547
|
+
["count", "u32"],
|
|
548
|
+
["pad0", "u32"],
|
|
549
|
+
["pad1", "u32"],
|
|
550
|
+
["pad2", "u32"],
|
|
551
|
+
]);
|
|
552
|
+
|
|
553
|
+
/** `GroupParams` (uniform, 16 B; P11): `rowsBase` @0 (the word of `rows` where the dispatch's row list starts), `basesBase` @4 (the word where the workgroup tier's region offsets start), `count` @8 (the rows of the dispatch), `pad0` @12. */
|
|
554
|
+
export const GROUP_PARAMS: UniformBlock = UniformBlock.define("GroupParams", [
|
|
555
|
+
["rowsBase", "u32"],
|
|
556
|
+
["basesBase", "u32"],
|
|
557
|
+
["count", "u32"],
|
|
558
|
+
["pad0", "u32"],
|
|
559
|
+
]);
|
|
560
|
+
|
|
561
|
+
/** `LpaParams` (uniform, 16 B; P11): `n` @0, `direction` @4 (0: a pass that moves labels down only, 1: up only), `counterIndex` @8 (the word of `counters` that receives the pass's move count), `pad0` @12. */
|
|
562
|
+
export const LPA_PARAMS: UniformBlock = UniformBlock.define("LpaParams", [
|
|
563
|
+
["n", "u32"],
|
|
564
|
+
["direction", "u32"],
|
|
565
|
+
["counterIndex", "u32"],
|
|
566
|
+
["pad0", "u32"],
|
|
567
|
+
]);
|
|
568
|
+
|
|
485
569
|
// ---- the entries (contract 3.10.1; group 0 = graph, 1 = state, 2 = params, 3 = cold)
|
|
486
570
|
|
|
487
571
|
/**
|
|
@@ -1345,7 +1429,7 @@ const CLOSENESS_SWEEP: KernelEntry = {
|
|
|
1345
1429
|
phase: "P8",
|
|
1346
1430
|
};
|
|
1347
1431
|
|
|
1348
|
-
/** `closeness-reduce` (design 8.4, 9.7; P8-T11, PD-13): the one-lane bookkeeping of the sweep -- role 0 the level boundary (`done` from the previous level's compacted count, `newCount` folded into `reached` and the 64-bit `sum` at `level + 1` with the 16-bit split product and the carry, `level` advanced), role 1 the seed of a batch (the sources' bits into `visited` and the level-0 frontier region, their flags, `counters[0] = k`, `level = U32_MAX`); 3 storage bindings (`counters` and `perSource` as `array<atomic<u32>>`, `bits` plain: one lane writes the seed). */
|
|
1432
|
+
/** `closeness-reduce` (design 8.4, 9.7; P8-T11, PD-13): the one-lane bookkeeping of the sweep -- role 0 the level boundary (`done` from the previous level's compacted count, `newCount` folded into `reached` and the 64-bit `sum` at `level + 1` with the 16-bit split product and the carry, `level` advanced), role 1 the seed of a batch (the sources' bits into `visited` and the level-0 frontier region, their flags, `counters[0] = k`, `level = U32_MAX`), role 2 the same seed from a sampled run's source list; 3 storage bindings (`counters` and `perSource` as `array<atomic<u32>>`, `bits` plain: one lane writes the seed). */
|
|
1349
1433
|
const CLOSENESS_REDUCE: KernelEntry = {
|
|
1350
1434
|
id: "closeness-reduce",
|
|
1351
1435
|
body: closenessReduceWgsl,
|
|
@@ -1363,6 +1447,306 @@ const CLOSENESS_REDUCE: KernelEntry = {
|
|
|
1363
1447
|
phase: "P8",
|
|
1364
1448
|
};
|
|
1365
1449
|
|
|
1450
|
+
/** `bc-finalize` (design 8.4, 5.4): the one-lane bookkeeping of a betweenness batch -- role 1 seeds it (depth 0 and one path for the k seed entries of the claim log, `stackTop = k`, `level = U32_MAX`), role 0 is the level boundary (`ends[level + 1] = stackTop`, `frontierCount`, `done` on an empty level); 5 storage bindings (the counters block as `array<atomic<u32>>`, `ends`, `S` read-only, `depthK` and `sigmaK` plain: one lane writes the seed). The design's finalize row has 2; `ends` is the third (the level boundary), and the seed's `S`, `depthK` and `sigmaK` make it five. */
|
|
1451
|
+
const BC_FINALIZE: KernelEntry = {
|
|
1452
|
+
id: "bc-finalize",
|
|
1453
|
+
body: bcFinalizeWgsl,
|
|
1454
|
+
entryPoint: "bc_finalize",
|
|
1455
|
+
bindings: [
|
|
1456
|
+
decl(1, 0, "counters", "storage", "array<atomic<u32>>"),
|
|
1457
|
+
decl(1, 1, "ends", "storage", "array<u32>"),
|
|
1458
|
+
decl(1, 2, "S", "storage-ro", "array<u32>"),
|
|
1459
|
+
decl(1, 3, "depthK", "storage", "array<u32>"),
|
|
1460
|
+
decl(1, 4, "sigmaK", "storage", "array<u32>"),
|
|
1461
|
+
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1462
|
+
],
|
|
1463
|
+
overrideDecls: [],
|
|
1464
|
+
uniforms: [BC_PARAMS],
|
|
1465
|
+
needs: [],
|
|
1466
|
+
snippetSlots: [],
|
|
1467
|
+
phase: "P9",
|
|
1468
|
+
};
|
|
1469
|
+
|
|
1470
|
+
/** `bc-forward` (design 8.4, 8.10 "BC forward (tagged)", 16.1): one level of the tagged multi-source BFS -- the block-mapped strip over the level's range of the claim log with the claim, the path count and the overflow report inline, the winners appended to the log; 7 storage bindings (`rowPtr`, `colIdx`, `S` read-write -- the level being read and the appends are ranges of ONE binding --, `ends` read-only, the counters block, `depthK` and `sigmaK` as `array<atomic<u32>>`); the inlined Hillis-Steele scan, so `needs: []`. */
|
|
1471
|
+
const BC_FORWARD: KernelEntry = {
|
|
1472
|
+
id: "bc-forward",
|
|
1473
|
+
body: bcForwardWgsl,
|
|
1474
|
+
entryPoint: "bc_forward",
|
|
1475
|
+
bindings: [
|
|
1476
|
+
decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
|
|
1477
|
+
decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
|
|
1478
|
+
decl(1, 2, "S", "storage", "array<u32>"),
|
|
1479
|
+
decl(1, 3, "ends", "storage-ro", "array<u32>"),
|
|
1480
|
+
decl(1, 4, "counters", "storage", "array<atomic<u32>>"),
|
|
1481
|
+
decl(1, 5, "depthK", "storage", "array<atomic<u32>>"),
|
|
1482
|
+
decl(1, 6, "sigmaK", "storage", "array<atomic<u32>>"),
|
|
1483
|
+
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1484
|
+
],
|
|
1485
|
+
overrideDecls: [],
|
|
1486
|
+
uniforms: [BC_PARAMS],
|
|
1487
|
+
needs: [],
|
|
1488
|
+
snippetSlots: [],
|
|
1489
|
+
phase: "P9",
|
|
1490
|
+
};
|
|
1491
|
+
|
|
1492
|
+
/** `bc-backward` (design 8.4, 8.10 "BC backward (successor pull)"): one level of the dependency accumulation, one invocation per log entry of a host-planned range, each pulling over its successors and writing its delta once; 6 storage bindings (`rowPtr`, `colIdx`, `S`, `depthK`, `sigmaK` read-only, `deltaK`). */
|
|
1493
|
+
const BC_BACKWARD: KernelEntry = {
|
|
1494
|
+
id: "bc-backward",
|
|
1495
|
+
body: bcBackwardWgsl,
|
|
1496
|
+
entryPoint: "bc_backward",
|
|
1497
|
+
bindings: [
|
|
1498
|
+
decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
|
|
1499
|
+
decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
|
|
1500
|
+
decl(1, 2, "S", "storage-ro", "array<u32>"),
|
|
1501
|
+
decl(1, 3, "depthK", "storage-ro", "array<u32>"),
|
|
1502
|
+
decl(1, 4, "sigmaK", "storage-ro", "array<u32>"),
|
|
1503
|
+
decl(1, 5, "deltaK", "storage", "array<f32>"),
|
|
1504
|
+
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1505
|
+
],
|
|
1506
|
+
overrideDecls: [],
|
|
1507
|
+
uniforms: [BC_PARAMS],
|
|
1508
|
+
needs: [],
|
|
1509
|
+
snippetSlots: [],
|
|
1510
|
+
phase: "P9",
|
|
1511
|
+
};
|
|
1512
|
+
|
|
1513
|
+
/** `bc-gather` (design 8.4, 8.10 "BC gather"): `bc[w] += sum over s of delta[s][w]`, one invocation per vertex, no atomic; 2 storage bindings (`deltaK` read-only, `bc`). */
|
|
1514
|
+
const BC_GATHER: KernelEntry = {
|
|
1515
|
+
id: "bc-gather",
|
|
1516
|
+
body: bcGatherWgsl,
|
|
1517
|
+
entryPoint: "bc_gather",
|
|
1518
|
+
bindings: [
|
|
1519
|
+
decl(1, 0, "deltaK", "storage-ro", "array<f32>"),
|
|
1520
|
+
decl(1, 1, "bc", "storage", "array<f32>"),
|
|
1521
|
+
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1522
|
+
],
|
|
1523
|
+
overrideDecls: [],
|
|
1524
|
+
uniforms: [BC_PARAMS],
|
|
1525
|
+
needs: [],
|
|
1526
|
+
snippetSlots: [],
|
|
1527
|
+
phase: "P9",
|
|
1528
|
+
};
|
|
1529
|
+
|
|
1530
|
+
/** `bc-edge-gather` (design 8.4 "edge BC accumulates per arc from the same n x k deltas"): the per-arc twin of `bc-gather`, one invocation per arc (its row found by an upper-bound search over `rowPtr`) adding the arc's term over the batch's sources; 6 storage bindings (`rowPtr`, `colIdx`, `depthK`, `sigmaK`, `deltaK` read-only, `arcScores`). */
|
|
1531
|
+
const BC_EDGE_GATHER: KernelEntry = {
|
|
1532
|
+
id: "bc-edge-gather",
|
|
1533
|
+
body: bcEdgeGatherWgsl,
|
|
1534
|
+
entryPoint: "bc_edge_gather",
|
|
1535
|
+
bindings: [
|
|
1536
|
+
decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
|
|
1537
|
+
decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
|
|
1538
|
+
decl(1, 2, "depthK", "storage-ro", "array<u32>"),
|
|
1539
|
+
decl(1, 3, "sigmaK", "storage-ro", "array<u32>"),
|
|
1540
|
+
decl(1, 4, "deltaK", "storage-ro", "array<f32>"),
|
|
1541
|
+
decl(1, 5, "arcScores", "storage", "array<f32>"),
|
|
1542
|
+
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1543
|
+
],
|
|
1544
|
+
overrideDecls: [],
|
|
1545
|
+
uniforms: [BC_PARAMS],
|
|
1546
|
+
needs: [],
|
|
1547
|
+
snippetSlots: [],
|
|
1548
|
+
phase: "P9",
|
|
1549
|
+
};
|
|
1550
|
+
|
|
1551
|
+
/** `bc-forward-edge` (design 8.4 "the edge-parallel form", 8.8 row 7): one forward level edge-parallel over the `edgeList` view for every source of the batch, with `bc-forward`'s claim, count and overflow report, appending to the same claim log; `UNDIRECTED` relaxes both directions of every edge; 7 storage bindings (`edgeSrc`, `edgeDst`, `S` read-write, `ends` read-only, the counters block, `depthK` and `sigmaK` as `array<atomic<u32>>`). */
|
|
1552
|
+
const BC_FORWARD_EDGE: KernelEntry = {
|
|
1553
|
+
id: "bc-forward-edge",
|
|
1554
|
+
body: bcForwardEdgeWgsl,
|
|
1555
|
+
entryPoint: "bc_forward_edge",
|
|
1556
|
+
bindings: [
|
|
1557
|
+
decl(1, 0, "edgeSrc", "storage-ro", "array<u32>"),
|
|
1558
|
+
decl(1, 1, "edgeDst", "storage-ro", "array<u32>"),
|
|
1559
|
+
decl(1, 2, "S", "storage", "array<u32>"),
|
|
1560
|
+
decl(1, 3, "ends", "storage-ro", "array<u32>"),
|
|
1561
|
+
decl(1, 4, "counters", "storage", "array<atomic<u32>>"),
|
|
1562
|
+
decl(1, 5, "depthK", "storage", "array<atomic<u32>>"),
|
|
1563
|
+
decl(1, 6, "sigmaK", "storage", "array<atomic<u32>>"),
|
|
1564
|
+
decl(2, 0, "P", "uniform", "BcParams"),
|
|
1565
|
+
],
|
|
1566
|
+
overrideDecls: [{ name: "UNDIRECTED", type: "bool", default: false }],
|
|
1567
|
+
uniforms: [BC_PARAMS],
|
|
1568
|
+
needs: [],
|
|
1569
|
+
snippetSlots: [],
|
|
1570
|
+
phase: "P9",
|
|
1571
|
+
};
|
|
1572
|
+
|
|
1573
|
+
/** `apsp-init` (design 8.7): one lane per row writes that row's arcs into the `+Infinity`-filled `n x n` matrix, the cheapest of parallel arcs, then the diagonal zero; 5 storage bindings (the four graph slots -- `perm` bound to its dummy, the rows are never permuted -- and `dist`). */
|
|
1574
|
+
const APSP_INIT: KernelEntry = {
|
|
1575
|
+
id: "apsp-init",
|
|
1576
|
+
body: apspInitWgsl,
|
|
1577
|
+
entryPoint: "apsp_init",
|
|
1578
|
+
bindings: GRAPH_SLOTS.concat(decl(1, 0, "dist", "storage", "array<f32>"), decl(2, 0, "P", "uniform", "ApspParams")),
|
|
1579
|
+
overrideDecls: [],
|
|
1580
|
+
uniforms: [APSP_PARAMS],
|
|
1581
|
+
needs: [],
|
|
1582
|
+
snippetSlots: [],
|
|
1583
|
+
phase: "P9",
|
|
1584
|
+
};
|
|
1585
|
+
|
|
1586
|
+
/** `apsp-fw` (design 8.7): one phase of one blocked Floyd-Warshall round over 32 x 32 tiles in workgroup memory -- `PHASE` 0 the pivot block, 1 the pivot row and column, 2 every other block; 1 storage binding (`dist`, read-write). */
|
|
1587
|
+
const APSP_FW: KernelEntry = {
|
|
1588
|
+
id: "apsp-fw",
|
|
1589
|
+
body: apspFwWgsl,
|
|
1590
|
+
entryPoint: "apsp_fw",
|
|
1591
|
+
bindings: [decl(1, 0, "dist", "storage", "array<f32>"), decl(2, 0, "P", "uniform", "ApspParams")],
|
|
1592
|
+
overrideDecls: [{ name: "PHASE", type: "u32", default: 0 }],
|
|
1593
|
+
uniforms: [APSP_PARAMS],
|
|
1594
|
+
needs: [],
|
|
1595
|
+
snippetSlots: [],
|
|
1596
|
+
phase: "P9",
|
|
1597
|
+
};
|
|
1598
|
+
|
|
1599
|
+
/** `coo-emit` (design 6 row 10; P11): position i takes arc i, or `order[i]` under INDEXED, and writes its source, target and weight -- arc 2e is edge e as declared, 2e + 1 its reverse; a self-loop's two arcs become `INVALID_INDEX`; 7 storage bindings. */
|
|
1600
|
+
const COO_EMIT: KernelEntry = {
|
|
1601
|
+
id: "coo-emit",
|
|
1602
|
+
body: cooEmitWgsl,
|
|
1603
|
+
entryPoint: "coo_emit",
|
|
1604
|
+
bindings: [
|
|
1605
|
+
decl(1, 0, "edgeSrc", "storage-ro", "array<u32>"),
|
|
1606
|
+
decl(1, 1, "edgeDst", "storage-ro", "array<u32>"),
|
|
1607
|
+
decl(1, 2, "edgeWeight", "storage-ro", "array<f32>"),
|
|
1608
|
+
decl(1, 3, "order", "storage-ro", "array<u32>"),
|
|
1609
|
+
decl(1, 4, "outSrc", "storage", "array<u32>"),
|
|
1610
|
+
decl(1, 5, "outDst", "storage", "array<u32>"),
|
|
1611
|
+
decl(1, 6, "outWeight", "storage", "array<f32>"),
|
|
1612
|
+
decl(2, 0, "P", "uniform", "CooParams"),
|
|
1613
|
+
],
|
|
1614
|
+
overrideDecls: [
|
|
1615
|
+
{ name: "INDEXED", type: "bool", default: false },
|
|
1616
|
+
{ name: "WEIGHTED", type: "bool", default: false },
|
|
1617
|
+
],
|
|
1618
|
+
uniforms: [COO_PARAMS],
|
|
1619
|
+
needs: [],
|
|
1620
|
+
snippetSlots: [],
|
|
1621
|
+
phase: "P11",
|
|
1622
|
+
};
|
|
1623
|
+
|
|
1624
|
+
/** `run-flags` (design 6 row 10; P11): 1 where an arc opens a run of equal (keysA, keysB) pairs and is not dropped; 3 storage bindings. */
|
|
1625
|
+
const RUN_FLAGS: KernelEntry = {
|
|
1626
|
+
id: "run-flags",
|
|
1627
|
+
body: runFlagsWgsl,
|
|
1628
|
+
entryPoint: "run_flags",
|
|
1629
|
+
bindings: [
|
|
1630
|
+
decl(1, 0, "keysA", "storage-ro", "array<u32>"),
|
|
1631
|
+
decl(1, 1, "keysB", "storage-ro", "array<u32>"),
|
|
1632
|
+
decl(1, 2, "flags", "storage", "array<u32>"),
|
|
1633
|
+
decl(2, 0, "P", "uniform", "CooParams"),
|
|
1634
|
+
],
|
|
1635
|
+
overrideDecls: [],
|
|
1636
|
+
uniforms: [COO_PARAMS],
|
|
1637
|
+
needs: [],
|
|
1638
|
+
snippetSlots: [],
|
|
1639
|
+
phase: "P11",
|
|
1640
|
+
};
|
|
1641
|
+
|
|
1642
|
+
/** `coo-scatter` (design 6 row 10; P11): the scatter of `cooToCsr` -- the cursor mode, or under SORTED_INPUT the order-preserving mode whose precondition flag is `cursors[0]`; 7 storage bindings. */
|
|
1643
|
+
const COO_SCATTER: KernelEntry = {
|
|
1644
|
+
id: "coo-scatter",
|
|
1645
|
+
body: cooScatterWgsl,
|
|
1646
|
+
entryPoint: "coo_scatter",
|
|
1647
|
+
bindings: [
|
|
1648
|
+
decl(1, 0, "src", "storage-ro", "array<u32>"),
|
|
1649
|
+
decl(1, 1, "dst", "storage-ro", "array<u32>"),
|
|
1650
|
+
decl(1, 2, "weight", "storage-ro", "array<f32>"),
|
|
1651
|
+
decl(1, 3, "rowPtr", "storage-ro", "array<u32>"),
|
|
1652
|
+
decl(1, 4, "cursors", "storage", "array<atomic<u32>>"),
|
|
1653
|
+
decl(1, 5, "colIdx", "storage", "array<u32>"),
|
|
1654
|
+
decl(1, 6, "outWeight", "storage", "array<f32>"),
|
|
1655
|
+
decl(2, 0, "P", "uniform", "CooParams"),
|
|
1656
|
+
],
|
|
1657
|
+
overrideDecls: [
|
|
1658
|
+
{ name: "SORTED_INPUT", type: "bool", default: false },
|
|
1659
|
+
{ name: "WEIGHTED", type: "bool", default: false },
|
|
1660
|
+
],
|
|
1661
|
+
uniforms: [COO_PARAMS],
|
|
1662
|
+
needs: [],
|
|
1663
|
+
snippetSlots: [],
|
|
1664
|
+
phase: "P11",
|
|
1665
|
+
};
|
|
1666
|
+
|
|
1667
|
+
/** `orient-flags` (design 8.5; P11): 1 where arc (u, v) points up the (degree, id) order, one arc per undirected edge; 4 storage bindings. */
|
|
1668
|
+
const ORIENT_FLAGS: KernelEntry = {
|
|
1669
|
+
id: "orient-flags",
|
|
1670
|
+
body: orientFlagsWgsl,
|
|
1671
|
+
entryPoint: "orient_flags",
|
|
1672
|
+
bindings: [
|
|
1673
|
+
decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
|
|
1674
|
+
decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
|
|
1675
|
+
decl(1, 2, "src", "storage-ro", "array<u32>"),
|
|
1676
|
+
decl(1, 3, "flags", "storage", "array<u32>"),
|
|
1677
|
+
decl(2, 0, "P", "uniform", "CooParams"),
|
|
1678
|
+
],
|
|
1679
|
+
overrideDecls: [],
|
|
1680
|
+
uniforms: [COO_PARAMS],
|
|
1681
|
+
needs: [],
|
|
1682
|
+
snippetSlots: [],
|
|
1683
|
+
phase: "P11",
|
|
1684
|
+
};
|
|
1685
|
+
|
|
1686
|
+
/** `tri-intersect` (design 8.5, 8.10 "Triangle intersection"; P11): one oriented arc per invocation, merge or binary-search intersection (SEARCH 0 chooses, 1 merge, 2 search), u32 atomic per-node counts; 4 storage bindings. */
|
|
1687
|
+
const TRI_INTERSECT: KernelEntry = {
|
|
1688
|
+
id: "tri-intersect",
|
|
1689
|
+
body: triIntersectWgsl,
|
|
1690
|
+
entryPoint: "tri_intersect",
|
|
1691
|
+
bindings: [
|
|
1692
|
+
decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
|
|
1693
|
+
decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
|
|
1694
|
+
decl(1, 2, "src", "storage-ro", "array<u32>"),
|
|
1695
|
+
decl(1, 3, "counts", "storage", "array<atomic<u32>>"),
|
|
1696
|
+
decl(2, 0, "P", "uniform", "CooParams"),
|
|
1697
|
+
],
|
|
1698
|
+
overrideDecls: [{ name: "SEARCH", type: "u32", default: 0 }],
|
|
1699
|
+
uniforms: [COO_PARAMS],
|
|
1700
|
+
needs: [],
|
|
1701
|
+
snippetSlots: [],
|
|
1702
|
+
phase: "P11",
|
|
1703
|
+
};
|
|
1704
|
+
|
|
1705
|
+
/** `group-by-key-row` (design 8.6; P11): per listed row, the target key with the largest summed weight (lowest key on a tie), TIER 0 a thread per row, else a workgroup per row over a global hash region; 8 storage bindings. */
|
|
1706
|
+
const GROUP_BY_KEY_ROW: KernelEntry = {
|
|
1707
|
+
id: "group-by-key-row",
|
|
1708
|
+
body: groupByKeyRowWgsl,
|
|
1709
|
+
entryPoint: "group_by_key_row",
|
|
1710
|
+
bindings: [
|
|
1711
|
+
decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
|
|
1712
|
+
decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
|
|
1713
|
+
decl(1, 2, "weights", "storage-ro", "array<f32>"),
|
|
1714
|
+
decl(1, 3, "keyIn", "storage-ro", "array<u32>"),
|
|
1715
|
+
decl(1, 4, "rows", "storage-ro", "array<u32>"),
|
|
1716
|
+
decl(1, 5, "hashRegion", "storage", "array<atomic<u32>>"),
|
|
1717
|
+
decl(1, 6, "bestKey", "storage", "array<u32>"),
|
|
1718
|
+
decl(1, 7, "bestScore", "storage", "array<f32>"),
|
|
1719
|
+
decl(2, 0, "P", "uniform", "GroupParams"),
|
|
1720
|
+
],
|
|
1721
|
+
overrideDecls: [
|
|
1722
|
+
{ name: "TIER", type: "u32", default: 0 },
|
|
1723
|
+
{ name: "WEIGHTED", type: "bool", default: false },
|
|
1724
|
+
],
|
|
1725
|
+
uniforms: [GROUP_PARAMS],
|
|
1726
|
+
needs: [],
|
|
1727
|
+
snippetSlots: [],
|
|
1728
|
+
phase: "P11",
|
|
1729
|
+
};
|
|
1730
|
+
|
|
1731
|
+
/** `lpa-step` (design 8.6, 8.10 "Label propagation"; P11): a vertex adopts its best neighbour label when the move goes the pass's direction; one atomic per workgroup adds the moves to `counters[P.counterIndex]`; 4 storage bindings. */
|
|
1732
|
+
const LPA_STEP: KernelEntry = {
|
|
1733
|
+
id: "lpa-step",
|
|
1734
|
+
body: lpaStepWgsl,
|
|
1735
|
+
entryPoint: "lpa_step",
|
|
1736
|
+
bindings: [
|
|
1737
|
+
decl(1, 0, "labelsIn", "storage-ro", "array<u32>"),
|
|
1738
|
+
decl(1, 1, "bestKey", "storage-ro", "array<u32>"),
|
|
1739
|
+
decl(1, 2, "labelsOut", "storage", "array<u32>"),
|
|
1740
|
+
decl(1, 3, "counters", "storage", "array<atomic<u32>>"),
|
|
1741
|
+
decl(2, 0, "P", "uniform", "LpaParams"),
|
|
1742
|
+
],
|
|
1743
|
+
overrideDecls: [],
|
|
1744
|
+
uniforms: [LPA_PARAMS],
|
|
1745
|
+
needs: [],
|
|
1746
|
+
snippetSlots: [],
|
|
1747
|
+
phase: "P11",
|
|
1748
|
+
};
|
|
1749
|
+
|
|
1366
1750
|
/**
|
|
1367
1751
|
* The entries by id, in dispatch order. PLAN DECISION: `KernelId` is declared in full (contract 3.10) while the
|
|
1368
1752
|
* entries landed phase by phase, so the table is built as a Partial record and exported below through the
|
|
@@ -1372,8 +1756,10 @@ const CLOSENESS_REDUCE: KernelEntry = {
|
|
|
1372
1756
|
* and `"fa2-to-scene"`; M8b-T3 landed the seven P7 entries and P4 its thirteen; P8-T3 landed the three compact /
|
|
1373
1757
|
* dedupe entries, P8-T4 `"frontier-finalize"`, P8-T5 `"advance-expand"`, P8-T6 `"bfs-contract"` and `"sssp-pred"` and
|
|
1374
1758
|
* P8-T7 `"bfs-fused"`, P8-T8 `"bfs-bottom-up"`, `"bfs-bitset-build"` and `"bfs-unvisited-flags"`, P8-T9
|
|
1375
|
-
* `"sssp-relax"`, P8-T10 `"bf-relax"` and P8-T11 `"closeness-sweep"` and `"closeness-reduce"`,
|
|
1376
|
-
* `
|
|
1759
|
+
* `"sssp-relax"`, P8-T10 `"bf-relax"` and P8-T11 `"closeness-sweep"` and `"closeness-reduce"`, betweenness the
|
|
1760
|
+
* six `"bc-*"` entries, all-pairs shortest paths `"apsp-init"` and `"apsp-fw"`, and P11 its seven (the graph
|
|
1761
|
+
* build, the group-by-key, label propagation's step and triangle counting), so every member of `KernelId`
|
|
1762
|
+
* is present and the assertion is exact.
|
|
1377
1763
|
*/
|
|
1378
1764
|
const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze({
|
|
1379
1765
|
degree: DEGREE,
|
|
@@ -1422,6 +1808,21 @@ const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze
|
|
|
1422
1808
|
"bf-relax": BF_RELAX,
|
|
1423
1809
|
"closeness-sweep": CLOSENESS_SWEEP,
|
|
1424
1810
|
"closeness-reduce": CLOSENESS_REDUCE,
|
|
1811
|
+
"bc-finalize": BC_FINALIZE,
|
|
1812
|
+
"bc-forward": BC_FORWARD,
|
|
1813
|
+
"bc-backward": BC_BACKWARD,
|
|
1814
|
+
"bc-gather": BC_GATHER,
|
|
1815
|
+
"bc-edge-gather": BC_EDGE_GATHER,
|
|
1816
|
+
"bc-forward-edge": BC_FORWARD_EDGE,
|
|
1817
|
+
"apsp-init": APSP_INIT,
|
|
1818
|
+
"apsp-fw": APSP_FW,
|
|
1819
|
+
"coo-emit": COO_EMIT,
|
|
1820
|
+
"run-flags": RUN_FLAGS,
|
|
1821
|
+
"coo-scatter": COO_SCATTER,
|
|
1822
|
+
"orient-flags": ORIENT_FLAGS,
|
|
1823
|
+
"tri-intersect": TRI_INTERSECT,
|
|
1824
|
+
"group-by-key-row": GROUP_BY_KEY_ROW,
|
|
1825
|
+
"lpa-step": LPA_STEP,
|
|
1425
1826
|
});
|
|
1426
1827
|
|
|
1427
1828
|
/** THE registry (spec 3.5): every entry, keyed by id. */
|
package/src/memory/residency.ts
CHANGED
|
@@ -393,7 +393,8 @@ export class GraphResidency {
|
|
|
393
393
|
/**
|
|
394
394
|
* Uploads (or finds) a view: outDegree / inDegree / degreeOrder / reverseDegreeOrder upload one array each;
|
|
395
395
|
* reverse and edgeList (P7) upload their arrays perArray, never into the arena, and are memoised per record so
|
|
396
|
-
* a second call uploads nothing (spec 4.3). coo
|
|
396
|
+
* a second call uploads nothing (spec 4.3). coo uploads its per-arc `src` (P11; the rest aliases the core); mate ->
|
|
397
|
+
* E_UNSUPPORTED. packViews concatenates a
|
|
397
398
|
* reverse or edgeList view into ONE buffer at STORAGE_ALIGN offsets; on any other view `true` is E_UNSUPPORTED
|
|
398
399
|
* { option: "packViews" }.
|
|
399
400
|
* @param s - the snapshot
|
|
@@ -465,10 +466,20 @@ export class GraphResidency {
|
|
|
465
466
|
break;
|
|
466
467
|
}
|
|
467
468
|
case "coo":
|
|
469
|
+
// dst, arcToEdge and weights alias the core's colIdx / arcToEdge / weights (graph-format design 7.2):
|
|
470
|
+
// only the per-arc source is new, and a kernel binds the rest from core()
|
|
471
|
+
this.assertNotReleased(s);
|
|
472
|
+
this.assertNonEmpty(s);
|
|
473
|
+
if (s.arcCount === 0) {
|
|
474
|
+
return Object.freeze({ view: name, bindings: Object.freeze({}), scalars: Object.freeze({}) });
|
|
475
|
+
}
|
|
476
|
+
array = s.coo().src;
|
|
477
|
+
bindingName = "src";
|
|
478
|
+
break;
|
|
468
479
|
case "mate":
|
|
469
|
-
throw new WebGpuGraphError("E_UNSUPPORTED",
|
|
470
|
-
feature:
|
|
471
|
-
hint: "outDegree, inDegree, degreeOrder, reverseDegreeOrder, reverse and edgeList are uploaded",
|
|
480
|
+
throw new WebGpuGraphError("E_UNSUPPORTED", "the mate view is not uploaded: no kernel reads it", {
|
|
481
|
+
feature: "view:mate",
|
|
482
|
+
hint: "outDegree, inDegree, degreeOrder, reverseDegreeOrder, coo, reverse and edgeList are uploaded",
|
|
472
483
|
});
|
|
473
484
|
default:
|
|
474
485
|
throw invalid("name", name, "a view name");
|