@graphty/webgpu-graph-algorithms 0.6.15 → 0.6.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (134) hide show
  1. package/README.md +5 -5
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-oXphO3yj.js → context-VIvatQOo.js} +61 -34
  4. package/dist/chunks/context-VIvatQOo.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/accelerator.d.ts +2 -1
  7. package/dist/src/accelerator.d.ts.map +1 -1
  8. package/dist/src/accelerator.js +53 -1
  9. package/dist/src/accelerator.js.map +1 -1
  10. package/dist/src/algorithms/all-pairs.d.ts +41 -0
  11. package/dist/src/algorithms/all-pairs.d.ts.map +1 -0
  12. package/dist/src/algorithms/all-pairs.js +181 -0
  13. package/dist/src/algorithms/all-pairs.js.map +1 -0
  14. package/dist/src/algorithms/components.d.ts +9 -1
  15. package/dist/src/algorithms/components.d.ts.map +1 -1
  16. package/dist/src/algorithms/components.js +2 -2
  17. package/dist/src/algorithms/components.js.map +1 -1
  18. package/dist/src/algorithms/label-propagation.d.ts +31 -0
  19. package/dist/src/algorithms/label-propagation.d.ts.map +1 -0
  20. package/dist/src/algorithms/label-propagation.js +254 -0
  21. package/dist/src/algorithms/label-propagation.js.map +1 -0
  22. package/dist/src/algorithms/simple-symmetric.d.ts +88 -0
  23. package/dist/src/algorithms/simple-symmetric.d.ts.map +1 -0
  24. package/dist/src/algorithms/simple-symmetric.js +347 -0
  25. package/dist/src/algorithms/simple-symmetric.js.map +1 -0
  26. package/dist/src/algorithms/triangles.d.ts +34 -0
  27. package/dist/src/algorithms/triangles.d.ts.map +1 -0
  28. package/dist/src/algorithms/triangles.js +203 -0
  29. package/dist/src/algorithms/triangles.js.map +1 -0
  30. package/dist/src/constants.d.ts +45 -0
  31. package/dist/src/constants.d.ts.map +1 -1
  32. package/dist/src/constants.js +45 -0
  33. package/dist/src/constants.js.map +1 -1
  34. package/dist/src/index.d.ts +8 -1
  35. package/dist/src/index.d.ts.map +1 -1
  36. package/dist/src/index.js +7 -1
  37. package/dist/src/index.js.map +1 -1
  38. package/dist/src/kernel/prelude.d.ts.map +1 -1
  39. package/dist/src/kernel/prelude.js +4 -1
  40. package/dist/src/kernel/prelude.js.map +1 -1
  41. package/dist/src/kernels.d.ts +16 -4
  42. package/dist/src/kernels.d.ts.map +1 -1
  43. package/dist/src/kernels.js +223 -3
  44. package/dist/src/kernels.js.map +1 -1
  45. package/dist/src/memory/residency.js +15 -4
  46. package/dist/src/memory/residency.js.map +1 -1
  47. package/dist/src/primitives/coo-to-csr.d.ts +73 -0
  48. package/dist/src/primitives/coo-to-csr.d.ts.map +1 -0
  49. package/dist/src/primitives/coo-to-csr.js +183 -0
  50. package/dist/src/primitives/coo-to-csr.js.map +1 -0
  51. package/dist/src/primitives/group-by-key.d.ts +82 -0
  52. package/dist/src/primitives/group-by-key.d.ts.map +1 -0
  53. package/dist/src/primitives/group-by-key.js +147 -0
  54. package/dist/src/primitives/group-by-key.js.map +1 -0
  55. package/dist/src/types/accelerator.d.ts +10 -2
  56. package/dist/src/types/accelerator.d.ts.map +1 -1
  57. package/dist/src/types/all-pairs.d.ts +35 -0
  58. package/dist/src/types/all-pairs.d.ts.map +1 -0
  59. package/dist/src/types/all-pairs.js +8 -0
  60. package/dist/src/types/all-pairs.js.map +1 -0
  61. package/dist/src/types/community.d.ts +18 -0
  62. package/dist/src/types/community.d.ts.map +1 -0
  63. package/dist/src/types/community.js +5 -0
  64. package/dist/src/types/community.js.map +1 -0
  65. package/dist/src/types/structure.d.ts +27 -0
  66. package/dist/src/types/structure.d.ts.map +1 -0
  67. package/dist/src/types/structure.js +8 -0
  68. package/dist/src/types/structure.js.map +1 -0
  69. package/dist/src/wgsl/apsp-fw.wgsl.d.ts +25 -0
  70. package/dist/src/wgsl/apsp-fw.wgsl.d.ts.map +1 -0
  71. package/dist/src/wgsl/apsp-fw.wgsl.js +113 -0
  72. package/dist/src/wgsl/apsp-fw.wgsl.js.map +1 -0
  73. package/dist/src/wgsl/apsp-init.wgsl.d.ts +12 -0
  74. package/dist/src/wgsl/apsp-init.wgsl.d.ts.map +1 -0
  75. package/dist/src/wgsl/apsp-init.wgsl.js +26 -0
  76. package/dist/src/wgsl/apsp-init.wgsl.js.map +1 -0
  77. package/dist/src/wgsl/coo-emit.wgsl.d.ts +10 -0
  78. package/dist/src/wgsl/coo-emit.wgsl.d.ts.map +1 -0
  79. package/dist/src/wgsl/coo-emit.wgsl.js +33 -0
  80. package/dist/src/wgsl/coo-emit.wgsl.js.map +1 -0
  81. package/dist/src/wgsl/coo-scatter.wgsl.d.ts +15 -0
  82. package/dist/src/wgsl/coo-scatter.wgsl.d.ts.map +1 -0
  83. package/dist/src/wgsl/coo-scatter.wgsl.js +32 -0
  84. package/dist/src/wgsl/coo-scatter.wgsl.js.map +1 -0
  85. package/dist/src/wgsl/group-by-key-row.wgsl.d.ts +26 -0
  86. package/dist/src/wgsl/group-by-key-row.wgsl.d.ts.map +1 -0
  87. package/dist/src/wgsl/group-by-key-row.wgsl.js +146 -0
  88. package/dist/src/wgsl/group-by-key-row.wgsl.js.map +1 -0
  89. package/dist/src/wgsl/lpa-step.wgsl.d.ts +10 -0
  90. package/dist/src/wgsl/lpa-step.wgsl.d.ts.map +1 -0
  91. package/dist/src/wgsl/lpa-step.wgsl.js +35 -0
  92. package/dist/src/wgsl/lpa-step.wgsl.js.map +1 -0
  93. package/dist/src/wgsl/orient-flags.wgsl.d.ts +9 -0
  94. package/dist/src/wgsl/orient-flags.wgsl.d.ts.map +1 -0
  95. package/dist/src/wgsl/orient-flags.wgsl.js +21 -0
  96. package/dist/src/wgsl/orient-flags.wgsl.js.map +1 -0
  97. package/dist/src/wgsl/run-flags.wgsl.d.ts +8 -0
  98. package/dist/src/wgsl/run-flags.wgsl.d.ts.map +1 -0
  99. package/dist/src/wgsl/run-flags.wgsl.js +18 -0
  100. package/dist/src/wgsl/run-flags.wgsl.js.map +1 -0
  101. package/dist/src/wgsl/tri-intersect.wgsl.d.ts +11 -0
  102. package/dist/src/wgsl/tri-intersect.wgsl.d.ts.map +1 -0
  103. package/dist/src/wgsl/tri-intersect.wgsl.js +64 -0
  104. package/dist/src/wgsl/tri-intersect.wgsl.js.map +1 -0
  105. package/dist/webgpu-graph-algorithms.js +1819 -181
  106. package/dist/webgpu-graph-algorithms.js.map +1 -1
  107. package/package.json +4 -4
  108. package/src/accelerator.ts +56 -1
  109. package/src/algorithms/all-pairs.ts +228 -0
  110. package/src/algorithms/components.ts +2 -2
  111. package/src/algorithms/label-propagation.ts +280 -0
  112. package/src/algorithms/simple-symmetric.ts +409 -0
  113. package/src/algorithms/triangles.ts +240 -0
  114. package/src/constants.ts +45 -0
  115. package/src/index.ts +12 -1
  116. package/src/kernel/prelude.ts +6 -0
  117. package/src/kernels.ts +248 -6
  118. package/src/memory/residency.ts +15 -4
  119. package/src/primitives/coo-to-csr.ts +251 -0
  120. package/src/primitives/group-by-key.ts +209 -0
  121. package/src/types/accelerator.ts +10 -2
  122. package/src/types/all-pairs.ts +37 -0
  123. package/src/types/community.ts +18 -0
  124. package/src/types/structure.ts +28 -0
  125. package/src/wgsl/apsp-fw.wgsl.ts +112 -0
  126. package/src/wgsl/apsp-init.wgsl.ts +25 -0
  127. package/src/wgsl/coo-emit.wgsl.ts +32 -0
  128. package/src/wgsl/coo-scatter.wgsl.ts +31 -0
  129. package/src/wgsl/group-by-key-row.wgsl.ts +145 -0
  130. package/src/wgsl/lpa-step.wgsl.ts +34 -0
  131. package/src/wgsl/orient-flags.wgsl.ts +20 -0
  132. package/src/wgsl/run-flags.wgsl.ts +17 -0
  133. package/src/wgsl/tri-intersect.wgsl.ts +63 -0
  134. package/dist/chunks/context-oXphO3yj.js.map +0 -1
@@ -0,0 +1,251 @@
1
+ /**
2
+ * The `cooToCsr` primitive driver (design 6 row 10; the P11 plan's PD-1): compressed sparse rows built on the device
3
+ * from `count` arcs `(src[i], dst[i], weights[i])` over `n` nodes. It is the histogram of the sources (over `n + 1`
4
+ * bins, so the last word is 0), their exclusive scan into `rowPtr` (so `rowPtr[n]` is the arc count) and the
5
+ * `coo-scatter` of the targets and weights into `colIdx` and `outWeights`.
6
+ *
7
+ * The scatter has two modes, and they promise different things:
8
+ * - `sortedInput: false` is design 6 row 10's cursor scatter. Each arc reserves its slot with an atomic on its row's
9
+ * cursor, so slots go out in race order: `rowPtr` is deterministic, but the order inside a row is not, and a row is
10
+ * NOT sorted by target even when the input was.
11
+ * - `sortedInput: true` requires the arcs ordered by source and writes each at its own index minus its row's start:
12
+ * no cursor, no atomic, the input order survives into every row, and the output is a pure function of the input.
13
+ * Arcs sorted by (source, target) therefore give rows sorted by target -- which every intersection relies on. The
14
+ * precondition is checked on the device: an out-of-order source raises word 0 of `out.flag`, which the caller
15
+ * must read back and refuse, because unsorted input in this mode gives rows that are silently scrambled.
16
+ *
17
+ * A caller who passes unsorted arcs with `sortedInput: true`, or sorted arcs with `sortedInput: false`, and needs
18
+ * sorted rows gets a graph whose rows are not sorted and whose triangle intersection is then silently wrong.
19
+ *
20
+ * A planner over a ReduceScope with a synchronous `record` (the shape of every primitive here), not the design's free
21
+ * function: pipelines compile once in `prepareCooToCsr`, and `src/primitives/**` never imports `src/context.ts`.
22
+ */
23
+
24
+ import { U32_MAX } from "../constants.js";
25
+ import { WebGpuGraphError } from "../errors.js";
26
+ import { plan1d } from "../kernel/dispatch.js";
27
+ import { type Kernel } from "../kernel/kernel.js";
28
+ import { COO_PARAMS, FILL_PARAMS, kernelSpec } from "../kernels.js";
29
+ import { type Binding } from "../types/memory.js";
30
+ import { type HistogramPlanner, prepareHistogram } from "./histogram.js";
31
+ import { type ReduceScope } from "./reduce.js";
32
+ import { prepareScan, type ScanPlanner } from "./scan.js";
33
+
34
+ /**
35
+ * The graph `record` writes.
36
+ * @public
37
+ */
38
+ export interface CooToCsrOutput {
39
+ /** `n + 1` words. */
40
+ readonly rowPtr: Binding;
41
+ /** At least `count` words. */
42
+ readonly colIdx: Binding;
43
+ /** At least `count` f32, or null when the input has no weights. */
44
+ readonly weights: Binding | null;
45
+ /** Word 0 is zeroed, then raised to 1 by an out-of-order source under `sortedInput`; a scratch word otherwise. */
46
+ readonly flag: Binding;
47
+ }
48
+
49
+ /**
50
+ * One build: `count` arcs whose sources are below `n`.
51
+ * @public
52
+ */
53
+ export interface CooToCsrRecord {
54
+ readonly src: Binding;
55
+ readonly dst: Binding;
56
+ /** One f32 per arc, or null (the output then has no weights). */
57
+ readonly weights: Binding | null;
58
+ readonly count: number;
59
+ readonly n: number;
60
+ /** The arcs are ordered by source: take the order-preserving scatter (see the file header). */
61
+ readonly sortedInput: boolean;
62
+ readonly out: CooToCsrOutput;
63
+ }
64
+
65
+ /**
66
+ * A prepared `cooToCsr`.
67
+ * @public
68
+ */
69
+ export interface CooToCsrPlanner {
70
+ /**
71
+ * Records the histogram, the scan and (for count > 0) the scatter of one build into the pass.
72
+ * @param pass - the compute pass
73
+ * @param record - the arcs and the output
74
+ */
75
+ record(pass: GPUComputePassEncoder, record: CooToCsrRecord): void;
76
+ }
77
+
78
+ /**
79
+ * Compiles the scatter's four variants, the histogram's, the scan's and `fill` so record() is synchronous. The
80
+ * planner lives as long as the scope: its scratch comes from it.
81
+ * @param scope - the caller's scope
82
+ * @returns the planner
83
+ */
84
+ export async function prepareCooToCsr(scope: ReduceScope): Promise<CooToCsrPlanner> {
85
+ const histogram = await prepareHistogram(scope);
86
+ const scan = await prepareScan(scope);
87
+ const fill = await scope.pipelines.kernel(kernelSpec("fill"));
88
+ const scatter = new Map<string, Kernel>();
89
+ for (const sorted of [false, true]) {
90
+ for (const weighted of [false, true]) {
91
+ const spec = kernelSpec("coo-scatter", { SORTED_INPUT: sorted, WEIGHTED: weighted });
92
+ scatter.set(`${sorted}/${weighted}`, await scope.pipelines.kernel(spec));
93
+ }
94
+ }
95
+ return new CooToCsrPlannerImpl(scope, histogram, scan, fill, scatter);
96
+ }
97
+
98
+ /**
99
+ * The E_INVALID_ARGUMENT of a binding shorter than `words` u32.
100
+ * @param name - the argument name
101
+ * @param binding - the binding
102
+ * @param words - the words it must hold
103
+ */
104
+ function checkWords(name: string, binding: Binding, words: number): void {
105
+ if (binding.size < 4 * words) {
106
+ throw new WebGpuGraphError("E_INVALID_ARGUMENT", `cooToCsr: ${name} is smaller than 4 x ${words} bytes`, {
107
+ argument: name,
108
+ value: binding.size,
109
+ expected: 4 * words,
110
+ });
111
+ }
112
+ }
113
+
114
+ /**
115
+ * The argument checks of record() (E_INVALID_ARGUMENT before anything is recorded).
116
+ * @param r - the record
117
+ */
118
+ function checkRecord(r: CooToCsrRecord): void {
119
+ for (const [name, value] of [
120
+ ["count", r.count],
121
+ ["n", r.n],
122
+ ] as const) {
123
+ if (!Number.isSafeInteger(value) || value < 0 || value >= U32_MAX) {
124
+ throw new WebGpuGraphError("E_INVALID_ARGUMENT", `cooToCsr: ${name} must be an integer in [0, 2^32 - 1)`, {
125
+ argument: name,
126
+ value,
127
+ });
128
+ }
129
+ }
130
+ if ((r.weights === null) !== (r.out.weights === null)) {
131
+ throw new WebGpuGraphError(
132
+ "E_INVALID_ARGUMENT",
133
+ "cooToCsr: weights and out.weights must be both present or both null",
134
+ {
135
+ argument: "weights",
136
+ value: r.weights === null ? "null" : "present",
137
+ expected: r.out.weights === null ? "null" : "present",
138
+ },
139
+ );
140
+ }
141
+ checkWords("src", r.src, r.count);
142
+ checkWords("dst", r.dst, r.count);
143
+ checkWords("out.rowPtr", r.out.rowPtr, r.n + 1);
144
+ checkWords("out.colIdx", r.out.colIdx, r.count);
145
+ checkWords("out.flag", r.out.flag, 1);
146
+ if (r.weights !== null && r.out.weights !== null) {
147
+ checkWords("weights", r.weights, r.count);
148
+ checkWords("out.weights", r.out.weights, r.count);
149
+ }
150
+ }
151
+
152
+ /** The planner: the histogram and scan planners, `fill` and the four scatter variants over one scope. */
153
+ class CooToCsrPlannerImpl implements CooToCsrPlanner {
154
+ private readonly scope: ReduceScope;
155
+ private readonly histogram: HistogramPlanner;
156
+ private readonly scan: ScanPlanner;
157
+ private readonly fill: Kernel;
158
+ private readonly scatter: ReadonlyMap<string, Kernel>;
159
+
160
+ /**
161
+ * Wraps the resolved planners and kernels; use prepareCooToCsr().
162
+ * @param scope - the caller's scope
163
+ * @param histogram - the histogram planner
164
+ * @param scan - the scan planner
165
+ * @param fill - the `fill` kernel
166
+ * @param scatter - the `coo-scatter` variants by `sorted/weighted`
167
+ */
168
+ constructor(
169
+ scope: ReduceScope,
170
+ histogram: HistogramPlanner,
171
+ scan: ScanPlanner,
172
+ fill: Kernel,
173
+ scatter: ReadonlyMap<string, Kernel>,
174
+ ) {
175
+ this.scope = scope;
176
+ this.histogram = histogram;
177
+ this.scan = scan;
178
+ this.fill = fill;
179
+ this.scatter = scatter;
180
+ }
181
+
182
+ /**
183
+ * Records one build (see the interface).
184
+ * @param pass - the compute pass
185
+ * @param r - the arcs and the output
186
+ */
187
+ record(pass: GPUComputePassEncoder, r: CooToCsrRecord): void {
188
+ checkRecord(r);
189
+ const { scope } = this;
190
+ const wg = scope.workgroupSize;
191
+ const degreeBytes = 4 * (r.n + 1);
192
+ const degrees: Binding = {
193
+ buffer: scope.scratch(degreeBytes, "cooToCsr/degrees"),
194
+ offset: 0,
195
+ size: degreeBytes,
196
+ window: null,
197
+ };
198
+ this.histogram.record(pass, r.src, r.count, r.n + 1, degrees);
199
+ this.scan.record(pass, degrees, r.n + 1, r.out.rowPtr);
200
+ let cursors = r.out.flag;
201
+ if (r.sortedInput) {
202
+ this.zero(pass, r.out.flag, 1);
203
+ } else if (r.n > 0) {
204
+ const bytes = 4 * r.n;
205
+ cursors = { buffer: scope.scratch(bytes, "cooToCsr/cursors"), offset: 0, size: bytes, window: null };
206
+ this.zero(pass, cursors, r.n);
207
+ }
208
+ if (r.count === 0) {
209
+ return;
210
+ }
211
+ const weighted = r.weights !== null;
212
+ const kernel = this.scatter.get(`${r.sortedInput}/${weighted}`);
213
+ if (kernel === undefined) {
214
+ throw new WebGpuGraphError("E_INVALID_ARGUMENT", "cooToCsr: no scatter variant was prepared", {
215
+ argument: "sortedInput",
216
+ value: r.sortedInput,
217
+ });
218
+ }
219
+ // the unweighted variant never touches the weight slots: a read-only dummy and a writable scratch word
220
+ const outWeight = r.out.weights ?? {
221
+ buffer: scope.scratch(4, "cooToCsr/weight-dummy"),
222
+ offset: 0,
223
+ size: 4,
224
+ window: null,
225
+ };
226
+ const params = scope.params(COO_PARAMS, { count: r.count, pad0: 0, pad1: 0, pad2: 0 });
227
+ const bound = kernel.bind({
228
+ src: r.src,
229
+ dst: r.dst,
230
+ weight: r.weights ?? r.dst,
231
+ rowPtr: r.out.rowPtr,
232
+ cursors,
233
+ colIdx: r.out.colIdx,
234
+ outWeight,
235
+ P: params.binding,
236
+ });
237
+ kernel.dispatch(pass, bound, plan1d(r.count, wg, scope.caps), [params.offset]);
238
+ }
239
+
240
+ /**
241
+ * Records a `fill` of `words` zeros.
242
+ * @param pass - the compute pass
243
+ * @param dst - the words
244
+ * @param words - how many (>= 1)
245
+ */
246
+ private zero(pass: GPUComputePassEncoder, dst: Binding, words: number): void {
247
+ const params = this.scope.params(FILL_PARAMS, { count: words, value: 0, mode: 0, pad0: 0 });
248
+ const bound = this.fill.bind({ dst, P: params.binding });
249
+ this.fill.dispatch(pass, bound, plan1d(words, this.scope.workgroupSize, this.scope.caps), [params.offset]);
250
+ }
251
+ }
@@ -0,0 +1,209 @@
1
+ /**
2
+ * The per-row group-by-key primitive (design 8.6; the P11 plan's PD-8): for every row `v` of a CSR graph, group the
3
+ * row's arcs by the key of their target (`keyIn[colIdx[a]]`), sum the weights per key, and write the key with the
4
+ * largest sum to `bestKey[v]` -- the LOWEST such key on a tie, which makes the answer independent of the visiting
5
+ * order -- and its summed weight to `bestScore[v]`. An empty row gets `INVALID_INDEX` and 0. Label propagation's step
6
+ * is its first caller: the best key of a row is the weighted mode of the neighbours' labels.
7
+ *
8
+ * Sums are u32 fixed point at a per-row power-of-two scale (the kernel header says how), so the two tiers and any
9
+ * reference that follows the same arithmetic agree bitwise. A key must be below `INVALID_INDEX`.
10
+ *
11
+ * Two tiers, chosen per row on the host from an upper bound of each row's length (`planGroupRows`):
12
+ * - TIER 0, a thread per row, for rows of at most `threadMax` arcs (GROUP_ROW_THREAD_MAX by default): a pairwise scan
13
+ * in registers, no workgroup memory, no barrier.
14
+ * - TIER 1, a workgroup per row, for the rest: the row's keys are hashed into its own region of
15
+ * `GROUP_HASH_LOAD_FACTOR x length` slots of `hashRegion`, then the workgroup reduces the slots to the best one.
16
+ *
17
+ * PLAN DECISION (the P11 plan, PD-8): the plan's small tier sorted a row of up to 256 arcs in workgroup memory with a
18
+ * bitonic sort, one workgroup per row; this tier is a thread per row instead, because a workgroup of 256 lanes spent
19
+ * on a row of ten arcs -- the typical row -- idles 246 of them, and the pairwise scan needs no barrier at all. Rows
20
+ * above GROUP_ROW_THREAD_MAX take the workgroup hash tier, which is the plan's large tier unchanged.
21
+ */
22
+
23
+ import { GROUP_HASH_LOAD_FACTOR, GROUP_ROW_THREAD_LIMIT, GROUP_ROW_THREAD_MAX } from "../constants.js";
24
+ import { WebGpuGraphError } from "../errors.js";
25
+ import { plan1d, plan2d } from "../kernel/dispatch.js";
26
+ import { type Kernel } from "../kernel/kernel.js";
27
+ import { GROUP_PARAMS, kernelSpec } from "../kernels.js";
28
+ import { type Binding } from "../types/memory.js";
29
+ import { type ReduceScope } from "./reduce.js";
30
+
31
+ /**
32
+ * The rows of each tier, as the host planned them: upload `words` into the `rows` binding and size `hashRegion` by `regionWords`.
33
+ * @public
34
+ */
35
+ export interface GroupRows {
36
+ /** The thread tier's rows, then the workgroup tier's rows, then each workgroup-tier row's region offset (in words of `hashRegion`). */
37
+ readonly words: Uint32Array<ArrayBuffer>;
38
+ readonly threadCount: number;
39
+ readonly hashCount: number;
40
+ /** The words of `hashRegion`: word 0 is the exhausted flag, then two words (key, sum) per slot of every workgroup-tier row. */
41
+ readonly regionWords: number;
42
+ }
43
+
44
+ /**
45
+ * Splits the rows into the two tiers from an upper bound of every row's length (the exact length works too; a bound
46
+ * only costs region space): rows whose bound is at most `threadMax` go to the thread tier, the rest to the workgroup
47
+ * tier with `GROUP_HASH_LOAD_FACTOR x bound` slots each.
48
+ * @param lengthBound - an upper bound of every row's arc count, one per row
49
+ * @param threadMax - the longest row the thread tier takes (0 sends every row to the workgroup tier; at most GROUP_ROW_THREAD_LIMIT)
50
+ * @returns the plan
51
+ */
52
+ export function planGroupRows(lengthBound: ArrayLike<number>, threadMax: number = GROUP_ROW_THREAD_MAX): GroupRows {
53
+ if (!Number.isInteger(threadMax) || threadMax < 0 || threadMax > GROUP_ROW_THREAD_LIMIT) {
54
+ throw new WebGpuGraphError(
55
+ "E_INVALID_ARGUMENT",
56
+ `planGroupRows: threadMax must be an integer in [0, ${GROUP_ROW_THREAD_LIMIT}]`,
57
+ { argument: "threadMax", value: threadMax, expected: `[0, ${GROUP_ROW_THREAD_LIMIT}]` },
58
+ );
59
+ }
60
+ const n = lengthBound.length;
61
+ const thread: number[] = [];
62
+ const hash: number[] = [];
63
+ for (let v = 0; v < n; v++) {
64
+ (lengthBound[v] <= threadMax ? thread : hash).push(v);
65
+ }
66
+ const words = new Uint32Array(thread.length + 2 * hash.length);
67
+ words.set(thread, 0);
68
+ words.set(hash, thread.length);
69
+ let next = 1;
70
+ for (let g = 0; g < hash.length; g++) {
71
+ words[thread.length + hash.length + g] = next;
72
+ next += 2 * GROUP_HASH_LOAD_FACTOR * lengthBound[hash[g]];
73
+ }
74
+ return { words, threadCount: thread.length, hashCount: hash.length, regionWords: next };
75
+ }
76
+
77
+ /**
78
+ * One grouping over a graph.
79
+ * @public
80
+ */
81
+ export interface GroupByKeyRecord {
82
+ readonly rowPtr: Binding;
83
+ readonly colIdx: Binding;
84
+ /** One f32 per arc, or null (every arc weighs 1). */
85
+ readonly weights: Binding | null;
86
+ /** One key per node, each below `INVALID_INDEX`. */
87
+ readonly keyIn: Binding;
88
+ /** The plan's counts; its `words` must be what `rows` holds. */
89
+ readonly plan: GroupRows;
90
+ readonly rows: Binding;
91
+ /** At least `plan.regionWords` words; word 0 must be zeroed by the caller once, and is 1 afterwards iff a probe loop exhausted its bound. */
92
+ readonly hashRegion: Binding;
93
+ readonly bestKey: Binding;
94
+ readonly bestScore: Binding;
95
+ }
96
+
97
+ /**
98
+ * A prepared group-by-key.
99
+ * @public
100
+ */
101
+ export interface GroupByKeyPlanner {
102
+ /**
103
+ * Records the (up to two) tier dispatches of one grouping into the pass.
104
+ * @param pass - the compute pass
105
+ * @param record - the graph, the keys, the tier plan and the outputs
106
+ */
107
+ record(pass: GPUComputePassEncoder, record: GroupByKeyRecord): void;
108
+ }
109
+
110
+ /**
111
+ * Compiles the four variants (two tiers, weighted or not) so record() is synchronous.
112
+ * @param scope - the caller's scope
113
+ * @returns the planner
114
+ */
115
+ export async function prepareGroupByKeyRow(scope: ReduceScope): Promise<GroupByKeyPlanner> {
116
+ const kernels = new Map<string, Kernel>();
117
+ for (const tier of [0, 1]) {
118
+ for (const weighted of [false, true]) {
119
+ const spec = kernelSpec("group-by-key-row", { TIER: tier, WEIGHTED: weighted });
120
+ kernels.set(`${tier}/${weighted}`, await scope.pipelines.kernel(spec));
121
+ }
122
+ }
123
+ return new GroupByKeyPlannerImpl(scope, kernels);
124
+ }
125
+
126
+ /**
127
+ * The E_INVALID_ARGUMENT of a binding shorter than `words` u32.
128
+ * @param name - the argument name
129
+ * @param binding - the binding
130
+ * @param words - the words it must hold
131
+ */
132
+ function checkWords(name: string, binding: Binding, words: number): void {
133
+ if (binding.size < 4 * words) {
134
+ throw new WebGpuGraphError("E_INVALID_ARGUMENT", `groupByKeyRow: ${name} is smaller than 4 x ${words} bytes`, {
135
+ argument: name,
136
+ value: binding.size,
137
+ expected: 4 * words,
138
+ });
139
+ }
140
+ }
141
+
142
+ /** The planner: the four variants over one scope. */
143
+ class GroupByKeyPlannerImpl implements GroupByKeyPlanner {
144
+ private readonly scope: ReduceScope;
145
+ private readonly kernels: ReadonlyMap<string, Kernel>;
146
+
147
+ /**
148
+ * Wraps the compiled variants; use prepareGroupByKeyRow().
149
+ * @param scope - the caller's scope
150
+ * @param kernels - the variants by `tier/weighted`
151
+ */
152
+ constructor(scope: ReduceScope, kernels: ReadonlyMap<string, Kernel>) {
153
+ this.scope = scope;
154
+ this.kernels = kernels;
155
+ }
156
+
157
+ /**
158
+ * Records the tier dispatches (see the interface).
159
+ * @param pass - the compute pass
160
+ * @param r - the record
161
+ */
162
+ record(pass: GPUComputePassEncoder, r: GroupByKeyRecord): void {
163
+ const { threadCount, hashCount } = r.plan;
164
+ const rows = threadCount + hashCount;
165
+ checkWords("rows", r.rows, rows + hashCount);
166
+ checkWords("hashRegion", r.hashRegion, r.plan.regionWords);
167
+ checkWords("rowPtr", r.rowPtr, rows + 1);
168
+ if (rows === 0) {
169
+ return;
170
+ }
171
+ const weighted = r.weights !== null;
172
+ const tiers: readonly (readonly [0 | 1, number, number])[] = [
173
+ [0, 0, threadCount],
174
+ [1, threadCount, hashCount],
175
+ ];
176
+ for (const [tier, rowsBase, count] of tiers) {
177
+ if (count === 0) {
178
+ continue;
179
+ }
180
+ const kernel = this.kernels.get(`${tier}/${weighted}`);
181
+ if (kernel === undefined) {
182
+ throw new WebGpuGraphError("E_INVALID_ARGUMENT", "groupByKeyRow: no variant was prepared", {
183
+ argument: "tier",
184
+ value: tier,
185
+ });
186
+ }
187
+ const params = this.scope.params(GROUP_PARAMS, {
188
+ rowsBase,
189
+ basesBase: threadCount + hashCount,
190
+ count,
191
+ pad0: 0,
192
+ });
193
+ const bound = kernel.bind({
194
+ rowPtr: r.rowPtr,
195
+ colIdx: r.colIdx,
196
+ weights: r.weights ?? r.colIdx,
197
+ keyIn: r.keyIn,
198
+ rows: r.rows,
199
+ hashRegion: r.hashRegion,
200
+ bestKey: r.bestKey,
201
+ bestScore: r.bestScore,
202
+ P: params.binding,
203
+ });
204
+ const plan =
205
+ tier === 0 ? plan1d(count, this.scope.workgroupSize, this.scope.caps) : plan2d(count, this.scope.caps);
206
+ kernel.dispatch(pass, bound, plan, [params.offset]);
207
+ }
208
+ }
209
+ }
@@ -42,6 +42,7 @@ import type {
42
42
  KatzOptions,
43
43
  PageRankOptions,
44
44
  } from "./algorithms.js";
45
+ import type { GpuApspResult } from "./all-pairs.js";
45
46
  import type { GpuBetweennessResult, GpuEdgeScoresResult } from "./betweenness.js";
46
47
  import type {
47
48
  ForceAtlas2Stats,
@@ -51,6 +52,7 @@ import type {
51
52
  SpringElectricalStats,
52
53
  } from "./layout.js";
53
54
  import type { ForceAtlas2Options, FruchtermanReingoldOptions, SpringElectricalOptions } from "./options.js";
55
+ import type { GpuTriangleResult } from "./structure.js";
54
56
  import type { GpuBellmanFordResult, GpuBfsResult, GpuSsspResult } from "./traversal.js";
55
57
 
56
58
  // ---- the real @graphty/layout interfaces (spec 9.3, D27): imported at W1b, re-exported so the package's public
@@ -115,8 +117,11 @@ export interface AcceleratorOptions {
115
117
  * take the seam's OWN option types (PD-19: `BfsOptions`, `SsspOptions` for both `sssp` and `bellmanFord`,
116
118
  * `ClosenessAcceleratorOptions` for `closenessCentrality`), so a key the CPU dispatcher forwards is exactly a key the GPU reads;
117
119
  * `test/types/conformance.test-d.ts` holds each parameter EQUAL to the seam's, not merely assignable. The two
118
- * betweenness members take the seam's `BetweennessAcceleratorOptions`. Later phases add one member per shipped
119
- * algorithm.
120
+ * betweenness members take the seam's `BetweennessAcceleratorOptions`. `allPairsShortestPath` (design 8.7) takes the
121
+ * seam's `SsspOptions` too and refuses both of its keys. P11 adds `triangleCount` (its result carries `coefficient`
122
+ * and `transitivity` beyond the seam's `{ perNode, total }`, which a wider object satisfies) and `labelPropagation`
123
+ * (the seam's `HitsOptionsLike`: `maxIterations` and `weighted` honoured, `tolerance` refused). Later phases add one
124
+ * member per shipped algorithm.
120
125
  * Exported: implemented by src/accelerator.ts (P3-T3); re-exported from src/index.ts at P3-T3.
121
126
  * @public
122
127
  */
@@ -148,6 +153,9 @@ export interface GpuAccelerator extends AlgorithmAccelerator, LayoutAccelerator
148
153
  closenessCentrality(s: GraphSnapshot, options?: ClosenessAcceleratorOptions): Promise<GpuClosenessResult>;
149
154
  betweennessCentrality(s: GraphSnapshot, options?: BetweennessAcceleratorOptions): Promise<GpuBetweennessResult>;
150
155
  edgeBetweennessCentrality(s: GraphSnapshot, options?: BetweennessAcceleratorOptions): Promise<GpuEdgeScoresResult>;
156
+ allPairsShortestPath(s: GraphSnapshot, options?: SsspOptions): Promise<GpuApspResult>;
157
+ triangleCount(s: GraphSnapshot): Promise<GpuTriangleResult>;
158
+ labelPropagation(s: GraphSnapshot, options?: HitsOptionsLike): Promise<GpuLabelResult>;
151
159
  release(s: GraphSnapshot): void;
152
160
  dispose(): void;
153
161
  }
@@ -0,0 +1,37 @@
1
+ /**
2
+ * The all-pairs shortest-path result and option records (design 3.3 lines 813 and 835, 8.7, 9.7). The result is the
3
+ * design's shape verbatim; every sentinel is spelled on its field, because a consumer who reads a `0` off the diagonal
4
+ * or an `Infinity` off the matrix and guesses what it means is the failure this file prevents. Types only: this file
5
+ * imports nothing at runtime.
6
+ */
7
+
8
+ import type { F32 } from "@graphty/graph-format";
9
+
10
+ /**
11
+ * Design 3.3 line 835: what `allPairsShortestPath` returns. Satisfies the seam's `ApspResultLike` (`dist:
12
+ * NumericVector` admits `F32`).
13
+ * @public
14
+ */
15
+ export interface GpuApspResult {
16
+ /**
17
+ * The `n * n` distances, row-major: `dist[i * n + j]` is the shortest distance FROM `i` TO `j` (the i-to-j
18
+ * direction on a directed snapshot). `+Infinity` when `j` is unreachable from `i`. The diagonal is `0` even when
19
+ * a self-loop carries a weight. f32 throughout; hop counts are exact integers.
20
+ */
21
+ readonly dist: F32;
22
+ /** The node count; `dist.length === n * n`. */
23
+ readonly n: number;
24
+ }
25
+
26
+ /**
27
+ * The options of `allPairsShortestPath` beyond `GpuRunOptions`. Nothing else: a cutoff would change the meaning of
28
+ * `+Infinity`, and the design asks for none.
29
+ * @public
30
+ */
31
+ export interface ApspOptions {
32
+ /**
33
+ * Use the snapshot's weight column. Default: true when the snapshot has weights. `false` on a weighted snapshot
34
+ * computes HOP COUNTS (every arc costs 1), not distances.
35
+ */
36
+ readonly weighted?: boolean | undefined;
37
+ }
@@ -0,0 +1,18 @@
1
+ /**
2
+ * The option record of label propagation (design 3.3 line 807, 8.6). Types only.
3
+ */
4
+
5
+ /**
6
+ * Label propagation's options. Ties between neighbour labels are broken by the LOWEST label, never at random, so
7
+ * the result is bitwise reproducible on one device; that is also why there is no `randomSeed`.
8
+ * @public
9
+ */
10
+ export interface LabelPropagationOptions {
11
+ /** The largest number of passes (default 100, as in `@graphty/algorithms`); a non-negative integer. */
12
+ readonly maxIterations?: number | undefined;
13
+ /**
14
+ * Sum the weights of the edges to each neighbour label (default true; an unweighted snapshot's edges weigh 1
15
+ * each, so a parallel edge counts once per copy); false counts every distinct neighbour once.
16
+ */
17
+ readonly weighted?: boolean | undefined;
18
+ }
@@ -0,0 +1,28 @@
1
+ /**
2
+ * The result record of triangle counting (design 3.3 line 828, 8.5; design 17 line 5067). Design 3.3 declares
3
+ * `{ perNode, total }`; this package also returns the clustering coefficient and the transitivity, because they are
4
+ * an epilogue over the counts and degrees the call already holds and nothing else in the monorepo computes them.
5
+ * Types only.
6
+ */
7
+
8
+ import type { F32, U32 } from "@graphty/graph-format";
9
+
10
+ /**
11
+ * Triangle counting over the simple undirected graph underlying the snapshot: arc directions are ignored, parallel
12
+ * edges count once and self-loops not at all, so a directed snapshot and its undirected twin give the same answer.
13
+ * @public
14
+ */
15
+ export interface GpuTriangleResult {
16
+ /** The triangles each node lies in. */
17
+ readonly perNode: U32;
18
+ /** The triangles of the graph (each counted once). */
19
+ readonly total: number;
20
+ /**
21
+ * The local clustering coefficient of every node, `2 T(v) / (d(v) (d(v) - 1))` with `d` the node's number of
22
+ * distinct neighbours. It is 0 -- a defined value, not a missing one -- for a node with fewer than two
23
+ * neighbours, so a star graph returns all zeros.
24
+ */
25
+ readonly coefficient: F32;
26
+ /** The graph's transitivity, `3 x triangles / connected triples`; 0 when the graph has no connected triple. */
27
+ readonly transitivity: number;
28
+ }