@graphty/webgpu-graph-algorithms 0.6.15 → 0.6.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (134) hide show
  1. package/README.md +5 -5
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-oXphO3yj.js → context-VIvatQOo.js} +61 -34
  4. package/dist/chunks/context-VIvatQOo.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/accelerator.d.ts +2 -1
  7. package/dist/src/accelerator.d.ts.map +1 -1
  8. package/dist/src/accelerator.js +53 -1
  9. package/dist/src/accelerator.js.map +1 -1
  10. package/dist/src/algorithms/all-pairs.d.ts +41 -0
  11. package/dist/src/algorithms/all-pairs.d.ts.map +1 -0
  12. package/dist/src/algorithms/all-pairs.js +181 -0
  13. package/dist/src/algorithms/all-pairs.js.map +1 -0
  14. package/dist/src/algorithms/components.d.ts +9 -1
  15. package/dist/src/algorithms/components.d.ts.map +1 -1
  16. package/dist/src/algorithms/components.js +2 -2
  17. package/dist/src/algorithms/components.js.map +1 -1
  18. package/dist/src/algorithms/label-propagation.d.ts +31 -0
  19. package/dist/src/algorithms/label-propagation.d.ts.map +1 -0
  20. package/dist/src/algorithms/label-propagation.js +254 -0
  21. package/dist/src/algorithms/label-propagation.js.map +1 -0
  22. package/dist/src/algorithms/simple-symmetric.d.ts +88 -0
  23. package/dist/src/algorithms/simple-symmetric.d.ts.map +1 -0
  24. package/dist/src/algorithms/simple-symmetric.js +347 -0
  25. package/dist/src/algorithms/simple-symmetric.js.map +1 -0
  26. package/dist/src/algorithms/triangles.d.ts +34 -0
  27. package/dist/src/algorithms/triangles.d.ts.map +1 -0
  28. package/dist/src/algorithms/triangles.js +203 -0
  29. package/dist/src/algorithms/triangles.js.map +1 -0
  30. package/dist/src/constants.d.ts +45 -0
  31. package/dist/src/constants.d.ts.map +1 -1
  32. package/dist/src/constants.js +45 -0
  33. package/dist/src/constants.js.map +1 -1
  34. package/dist/src/index.d.ts +8 -1
  35. package/dist/src/index.d.ts.map +1 -1
  36. package/dist/src/index.js +7 -1
  37. package/dist/src/index.js.map +1 -1
  38. package/dist/src/kernel/prelude.d.ts.map +1 -1
  39. package/dist/src/kernel/prelude.js +4 -1
  40. package/dist/src/kernel/prelude.js.map +1 -1
  41. package/dist/src/kernels.d.ts +16 -4
  42. package/dist/src/kernels.d.ts.map +1 -1
  43. package/dist/src/kernels.js +223 -3
  44. package/dist/src/kernels.js.map +1 -1
  45. package/dist/src/memory/residency.js +15 -4
  46. package/dist/src/memory/residency.js.map +1 -1
  47. package/dist/src/primitives/coo-to-csr.d.ts +73 -0
  48. package/dist/src/primitives/coo-to-csr.d.ts.map +1 -0
  49. package/dist/src/primitives/coo-to-csr.js +183 -0
  50. package/dist/src/primitives/coo-to-csr.js.map +1 -0
  51. package/dist/src/primitives/group-by-key.d.ts +82 -0
  52. package/dist/src/primitives/group-by-key.d.ts.map +1 -0
  53. package/dist/src/primitives/group-by-key.js +147 -0
  54. package/dist/src/primitives/group-by-key.js.map +1 -0
  55. package/dist/src/types/accelerator.d.ts +10 -2
  56. package/dist/src/types/accelerator.d.ts.map +1 -1
  57. package/dist/src/types/all-pairs.d.ts +35 -0
  58. package/dist/src/types/all-pairs.d.ts.map +1 -0
  59. package/dist/src/types/all-pairs.js +8 -0
  60. package/dist/src/types/all-pairs.js.map +1 -0
  61. package/dist/src/types/community.d.ts +18 -0
  62. package/dist/src/types/community.d.ts.map +1 -0
  63. package/dist/src/types/community.js +5 -0
  64. package/dist/src/types/community.js.map +1 -0
  65. package/dist/src/types/structure.d.ts +27 -0
  66. package/dist/src/types/structure.d.ts.map +1 -0
  67. package/dist/src/types/structure.js +8 -0
  68. package/dist/src/types/structure.js.map +1 -0
  69. package/dist/src/wgsl/apsp-fw.wgsl.d.ts +25 -0
  70. package/dist/src/wgsl/apsp-fw.wgsl.d.ts.map +1 -0
  71. package/dist/src/wgsl/apsp-fw.wgsl.js +113 -0
  72. package/dist/src/wgsl/apsp-fw.wgsl.js.map +1 -0
  73. package/dist/src/wgsl/apsp-init.wgsl.d.ts +12 -0
  74. package/dist/src/wgsl/apsp-init.wgsl.d.ts.map +1 -0
  75. package/dist/src/wgsl/apsp-init.wgsl.js +26 -0
  76. package/dist/src/wgsl/apsp-init.wgsl.js.map +1 -0
  77. package/dist/src/wgsl/coo-emit.wgsl.d.ts +10 -0
  78. package/dist/src/wgsl/coo-emit.wgsl.d.ts.map +1 -0
  79. package/dist/src/wgsl/coo-emit.wgsl.js +33 -0
  80. package/dist/src/wgsl/coo-emit.wgsl.js.map +1 -0
  81. package/dist/src/wgsl/coo-scatter.wgsl.d.ts +15 -0
  82. package/dist/src/wgsl/coo-scatter.wgsl.d.ts.map +1 -0
  83. package/dist/src/wgsl/coo-scatter.wgsl.js +32 -0
  84. package/dist/src/wgsl/coo-scatter.wgsl.js.map +1 -0
  85. package/dist/src/wgsl/group-by-key-row.wgsl.d.ts +26 -0
  86. package/dist/src/wgsl/group-by-key-row.wgsl.d.ts.map +1 -0
  87. package/dist/src/wgsl/group-by-key-row.wgsl.js +146 -0
  88. package/dist/src/wgsl/group-by-key-row.wgsl.js.map +1 -0
  89. package/dist/src/wgsl/lpa-step.wgsl.d.ts +10 -0
  90. package/dist/src/wgsl/lpa-step.wgsl.d.ts.map +1 -0
  91. package/dist/src/wgsl/lpa-step.wgsl.js +35 -0
  92. package/dist/src/wgsl/lpa-step.wgsl.js.map +1 -0
  93. package/dist/src/wgsl/orient-flags.wgsl.d.ts +9 -0
  94. package/dist/src/wgsl/orient-flags.wgsl.d.ts.map +1 -0
  95. package/dist/src/wgsl/orient-flags.wgsl.js +21 -0
  96. package/dist/src/wgsl/orient-flags.wgsl.js.map +1 -0
  97. package/dist/src/wgsl/run-flags.wgsl.d.ts +8 -0
  98. package/dist/src/wgsl/run-flags.wgsl.d.ts.map +1 -0
  99. package/dist/src/wgsl/run-flags.wgsl.js +18 -0
  100. package/dist/src/wgsl/run-flags.wgsl.js.map +1 -0
  101. package/dist/src/wgsl/tri-intersect.wgsl.d.ts +11 -0
  102. package/dist/src/wgsl/tri-intersect.wgsl.d.ts.map +1 -0
  103. package/dist/src/wgsl/tri-intersect.wgsl.js +64 -0
  104. package/dist/src/wgsl/tri-intersect.wgsl.js.map +1 -0
  105. package/dist/webgpu-graph-algorithms.js +1819 -181
  106. package/dist/webgpu-graph-algorithms.js.map +1 -1
  107. package/package.json +4 -4
  108. package/src/accelerator.ts +56 -1
  109. package/src/algorithms/all-pairs.ts +228 -0
  110. package/src/algorithms/components.ts +2 -2
  111. package/src/algorithms/label-propagation.ts +280 -0
  112. package/src/algorithms/simple-symmetric.ts +409 -0
  113. package/src/algorithms/triangles.ts +240 -0
  114. package/src/constants.ts +45 -0
  115. package/src/index.ts +12 -1
  116. package/src/kernel/prelude.ts +6 -0
  117. package/src/kernels.ts +248 -6
  118. package/src/memory/residency.ts +15 -4
  119. package/src/primitives/coo-to-csr.ts +251 -0
  120. package/src/primitives/group-by-key.ts +209 -0
  121. package/src/types/accelerator.ts +10 -2
  122. package/src/types/all-pairs.ts +37 -0
  123. package/src/types/community.ts +18 -0
  124. package/src/types/structure.ts +28 -0
  125. package/src/wgsl/apsp-fw.wgsl.ts +112 -0
  126. package/src/wgsl/apsp-init.wgsl.ts +25 -0
  127. package/src/wgsl/coo-emit.wgsl.ts +32 -0
  128. package/src/wgsl/coo-scatter.wgsl.ts +31 -0
  129. package/src/wgsl/group-by-key-row.wgsl.ts +145 -0
  130. package/src/wgsl/lpa-step.wgsl.ts +34 -0
  131. package/src/wgsl/orient-flags.wgsl.ts +20 -0
  132. package/src/wgsl/run-flags.wgsl.ts +17 -0
  133. package/src/wgsl/tri-intersect.wgsl.ts +63 -0
  134. package/dist/chunks/context-oXphO3yj.js.map +0 -1
package/src/constants.ts CHANGED
@@ -267,3 +267,48 @@ export const BC_MAX_BATCH = 64;
267
267
  export const BC_EDGE_PARALLEL_GAMMA = 2;
268
268
  /** Backward-pass levels recorded per submit: each level is one dispatch with its own parameter record, so this bounds the uniform ring. */
269
269
  export const BC_BACKWARD_LEVELS_PER_SUBMIT = 64;
270
+ /**
271
+ * Design 8.7: all-pairs shortest paths is a blocked Floyd-Warshall over `APSP_TILE x APSP_TILE` tiles. One tile of
272
+ * f32 is 4 KiB of workgroup memory and a workgroup stages at most two (8 KiB), inside the 16 KiB
273
+ * `maxComputeWorkgroupStorageSize` every WebGPU device reports. Interpolated into the prelude as `APSP_TILE`.
274
+ */
275
+ export const APSP_TILE = 32;
276
+ /**
277
+ * The blocked sweep records `3 x ceil(n / APSP_TILE)` dispatches (543 at the 5,792-node ceiling of a 128 MiB binding,
278
+ * 2,175 at a 2 GiB binding's 23,170, 3,072 at a 4 GiB binding's 32,767); above this many the driver splits the sweep
279
+ * into further submits. No binding offered today reaches it, so every sweep is one submit; the cap only stops a
280
+ * device with a binding above 4 GiB (`maxStorageBufferBindingSize` is a GPUSize64) from building one unbounded
281
+ * command buffer.
282
+ */
283
+ export const APSP_MAX_DISPATCHES_PER_SUBMIT = 4096;
284
+ /**
285
+ * Label propagation (design 8.6): passes recorded per submit, with ONE readback of the per-pass changed counts at the
286
+ * end of the submit. A readback costs about 2 ms in Chromium whatever it carries, so at one readback per pass a
287
+ * 10,000-node call spends more on synchronisation than the CPU spends on the whole algorithm
288
+ * (design/decisions/2026-09-26-which-algorithms-earn-the-gpu.md: 101 passes x 2 ms, 0.79x); eight passes per
289
+ * submit is the floor that decision sets. Even, so every submit holds as many descending as ascending passes of the
290
+ * alternating direction rule.
291
+ */
292
+ export const LABEL_PROP_PASSES_PER_SUBMIT = 8;
293
+ /**
294
+ * Boruvka's minimum spanning tree (design 8.5): rounds recorded per submit, with one readback of the counters block
295
+ * per submit. Each readback is a device-to-host synchronisation that costs about 2 ms in Chromium, and at one per
296
+ * round the syncs are 61 % of the 100,000-node call; four rounds per submit amortise them over O(log n) rounds, moving
297
+ * the Chromium crossover from 6,000 to 4,600 nodes (design/decisions/2026-09-26-which-algorithms-earn-the-gpu.md).
298
+ * Declared ahead of the minimum-spanning-tree driver, which reads it when it lands.
299
+ */
300
+ export const BORUVKA_ROUNDS_PER_SUBMIT = 4;
301
+ /** The per-row group-by-key (design 8.6): a row of at most this many arcs is grouped by one thread in registers; a longer row by a workgroup over a global open-addressing region. */
302
+ export const GROUP_ROW_THREAD_MAX = 32;
303
+ /** The largest row the thread tier accepts when a caller forces the tier: its pairwise scan is about d^2 / 2 loop steps, and llvmpipe stops every loop of an invocation after 65,535 steps in total. */
304
+ export const GROUP_ROW_THREAD_LIMIT = 128;
305
+ /**
306
+ * The most parallel arcs the simple symmetric graph build merges into one weighted arc. The merge sums each run of
307
+ * parallel arcs in one invocation, and llvmpipe stops every loop of an invocation after 65,535 steps in total and
308
+ * then quietly returns a short sum; a weighted build whose pair repeats more often is refused on every adapter.
309
+ */
310
+ export const PARALLEL_MERGE_LIMIT = 65_000;
311
+ /** The per-row group-by-key (design 8.6): the global open-addressing region of a workgroup-tier row holds this many slots per arc. */
312
+ export const GROUP_HASH_LOAD_FACTOR = 2;
313
+ /** Triangle counting (design 8.5): intersect two oriented rows by merge, but binary-search each element of the shorter row into the longer when their lengths differ by more than this factor. */
314
+ export const TRIANGLE_BINARY_SEARCH_RATIO = 32;
package/src/index.ts CHANGED
@@ -9,7 +9,8 @@
9
9
  * (Lease, CommandBatch, UniformRing are internal). P3 adds the layout factory, the accelerator, the two default
10
10
  * tables, the seeder and the layout / accelerator types. P5 adds the two factories, the two default tables and the
11
11
  * two stats records. P4 adds calibrateLayout and its two records. P8 adds the four traversals and their three result
12
- * records. test/index.test.ts pins the value list and
12
+ * records. All-pairs shortest paths (design 8.7) adds allPairsShortestPath and its result and option records.
13
+ * test/index.test.ts pins the value list and
13
14
  * test/types/public-api.test-d.ts the type list. This comment must never spell the internal
14
15
  * JSDoc tag: it is the leading comment of the first export statement, and stripInternal would drop that statement
15
16
  * from the emitted declarations.
@@ -62,6 +63,12 @@ export { breadthFirstSearch } from "./algorithms/bfs.js";
62
63
  export { closenessCentrality } from "./algorithms/closeness.js";
63
64
  export { sssp } from "./algorithms/sssp.js";
64
65
 
66
+ // ==================== algorithms (all-pairs shortest paths, design 3.3 line 813, 8.7)
67
+ export { allPairsShortestPath } from "./algorithms/all-pairs.js";
68
+ // ==================== algorithms (P11: structure and community, design 3.3 lines 806-807, 8.5, 8.6)
69
+ export { labelPropagation } from "./algorithms/label-propagation.js";
70
+ export { triangleCount } from "./algorithms/triangles.js";
71
+
65
72
  // ==================== layouts and the accelerator (P3; the two P5 factories; P4's calibrateLayout, spec 2.2)
66
73
  export { createAccelerator } from "./accelerator.js";
67
74
  export { calibrateLayout } from "./layouts/calibrate.js";
@@ -100,11 +107,15 @@ export type {
100
107
 
101
108
  // ==================== types: the P8 traversal results (spec 3.3 lines 830-832, 9.7); the option types are the seam's
102
109
  // BfsOptions / SsspOptions / HitsOptionsLike above (P8 PD-19)
110
+ export type { LabelPropagationOptions } from "./types/community.js";
111
+ export type { GpuTriangleResult } from "./types/structure.js";
103
112
  export type { GpuBellmanFordResult, GpuBfsResult, GpuSsspResult } from "./types/traversal.js";
104
113
 
105
114
  // ==================== types: the betweenness results (spec 3.3 lines 833-834); the option type is the seam's
106
115
  // BetweennessAcceleratorOptions above
107
116
  export type { GpuBetweennessResult, GpuEdgeScoresResult } from "./types/betweenness.js";
117
+ // ==================== types: all-pairs shortest paths (design 3.3 line 835)
118
+ export type { ApspOptions, GpuApspResult } from "./types/all-pairs.js";
108
119
 
109
120
  // ==================== types: the P7 algorithm results and option records (spec 3.3 lines 815-828, 9.7)
110
121
  export type {
@@ -12,6 +12,7 @@
12
12
  import { INVALID_INDEX } from "@graphty/graph-format";
13
13
 
14
14
  import {
15
+ APSP_TILE,
15
16
  EXACT_TILES_PER_PASS,
16
17
  F32_INF_BITS,
17
18
  FA2_COINCIDENT_SQ,
@@ -24,8 +25,10 @@ import {
24
25
  GRID_BBOX_MARGIN,
25
26
  GRID_EXTENT_FLOOR,
26
27
  GRID_HUB_CELL,
28
+ GROUP_HASH_LOAD_FACTOR,
27
29
  MAX_WORKGROUPS_PER_DIM,
28
30
  RADIX_BINS,
31
+ TRIANGLE_BINARY_SEARCH_RATIO,
29
32
  U32_MAX,
30
33
  WORKGROUP_SIZE,
31
34
  } from "../constants.js";
@@ -68,6 +71,9 @@ const GRID_EXTENT_FLOOR: f32 = ${wgslF32Literal(GRID_EXTENT_FLOOR)};
68
71
  const GRID_BBOX_MARGIN: f32 = ${wgslF32Literal(GRID_BBOX_MARGIN)};
69
72
  const RADIX_BINS: u32 = ${RADIX_BINS}u;
70
73
  const RADIX_DIGIT_MASK: u32 = ${RADIX_BINS - 1}u;
74
+ const APSP_TILE: u32 = ${APSP_TILE}u;
75
+ const GROUP_HASH_LOAD_FACTOR: u32 = ${GROUP_HASH_LOAD_FACTOR}u;
76
+ const TRIANGLE_BINARY_SEARCH_RATIO: u32 = ${TRIANGLE_BINARY_SEARCH_RATIO}u;
71
77
  const F32_MAX: f32 = 0x1.fffffep+127;
72
78
  override WG: u32 = ${WORKGROUP_SIZE}u;
73
79
  override USE_PERM: bool = false;
package/src/kernels.ts CHANGED
@@ -13,7 +13,11 @@
13
13
  * P8-T5 adds advance-expand; P8-T6 adds bfs-contract and sssp-pred; P8-T7 adds bfs-fused; P8-T8 adds bfs-bottom-up,
14
14
  * bfs-bitset-build and bfs-unvisited-flags; P8-T9 adds sssp-relax; P8-T10 adds bf-relax with the BfParams and BfFlags
15
15
  * blocks; P8-T11 adds closeness-sweep and closeness-reduce; P9 (betweenness) adds bc-finalize, bc-forward,
16
- * bc-backward, bc-gather, bc-edge-gather and bc-forward-edge with the BcParams block. This file is the only importer of src/wgsl/** (spec 3.2;
16
+ * bc-backward, bc-gather, bc-edge-gather and bc-forward-edge with the BcParams block; all-pairs shortest paths
17
+ * (design 8.7) adds apsp-init and apsp-fw with the ApspParams block. P11 (the structure and community phase, plan
18
+ * design/webgpu/plans/2026-09-23-webgpu-p11-structure-and-community.md) adds the graph build on the device (coo-emit,
19
+ * run-flags, coo-scatter), the per-row group-by-key (group-by-key-row), label propagation's step (lpa-step) and
20
+ * triangle counting (orient-flags, tri-intersect). This file is the only importer of src/wgsl/** (spec 3.2;
17
21
  * test/layers.test.ts).
18
22
  */
19
23
 
@@ -24,6 +28,8 @@ import { type BindingDecl, type OverrideDecl, type WgslModuleSpec } from "./kern
24
28
  import { type CoreBinding } from "./memory/residency.js";
25
29
  import { type Binding } from "./types/memory.js";
26
30
  import { advanceExpandWgsl } from "./wgsl/advance-expand.wgsl.js";
31
+ import { apspFwWgsl } from "./wgsl/apsp-fw.wgsl.js";
32
+ import { apspInitWgsl } from "./wgsl/apsp-init.wgsl.js";
27
33
  import { bcBackwardWgsl } from "./wgsl/bc-backward.wgsl.js";
28
34
  import { bcEdgeGatherWgsl } from "./wgsl/bc-edge-gather.wgsl.js";
29
35
  import { bcFinalizeWgsl } from "./wgsl/bc-finalize.wgsl.js";
@@ -40,6 +46,8 @@ import { bfsUnvisitedFlagsWgsl } from "./wgsl/bfs-unvisited-flags.wgsl.js";
40
46
  import { closenessReduceWgsl } from "./wgsl/closeness-reduce.wgsl.js";
41
47
  import { closenessSweepWgsl } from "./wgsl/closeness-sweep.wgsl.js";
42
48
  import { compactScatterWgsl } from "./wgsl/compact-scatter.wgsl.js";
49
+ import { cooEmitWgsl } from "./wgsl/coo-emit.wgsl.js";
50
+ import { cooScatterWgsl } from "./wgsl/coo-scatter.wgsl.js";
43
51
  import { countingScatterWgsl } from "./wgsl/counting-scatter.wgsl.js";
44
52
  import { dedupeClaimWgsl } from "./wgsl/dedupe-claim.wgsl.js";
45
53
  import { dedupeFilterWgsl } from "./wgsl/dedupe-filter.wgsl.js";
@@ -58,25 +66,30 @@ import { gridCentroidHubWgsl } from "./wgsl/grid-centroid-hub.wgsl.js";
58
66
  import { gridDownsampleWgsl } from "./wgsl/grid-downsample.wgsl.js";
59
67
  import { gridFarFieldWgsl } from "./wgsl/grid-far-field.wgsl.js";
60
68
  import { gridNearFieldWgsl } from "./wgsl/grid-near-field.wgsl.js";
69
+ import { groupByKeyRowWgsl } from "./wgsl/group-by-key-row.wgsl.js";
61
70
  import { histogramWgsl } from "./wgsl/histogram.wgsl.js";
62
71
  import { indirectFinalizeWgsl } from "./wgsl/indirect-finalize.wgsl.js";
72
+ import { lpaStepWgsl } from "./wgsl/lpa-step.wgsl.js";
73
+ import { orientFlagsWgsl } from "./wgsl/orient-flags.wgsl.js";
63
74
  import { prFinalizeWgsl } from "./wgsl/pr-finalize.wgsl.js";
64
75
  import { prScaleWgsl } from "./wgsl/pr-scale.wgsl.js";
65
76
  import { radixHistWgsl } from "./wgsl/radix-hist.wgsl.js";
66
77
  import { radixScatterWgsl } from "./wgsl/radix-scatter.wgsl.js";
67
78
  import { reduceWgsl } from "./wgsl/reduce.wgsl.js";
79
+ import { runFlagsWgsl } from "./wgsl/run-flags.wgsl.js";
68
80
  import { scanAddWgsl } from "./wgsl/scan-add.wgsl.js";
69
81
  import { scanBlockWgsl } from "./wgsl/scan-block.wgsl.js";
70
82
  import { segmentedReduceWgsl } from "./wgsl/segmented-reduce.wgsl.js";
71
83
  import { spmvPullWgsl } from "./wgsl/spmv-pull.wgsl.js";
72
84
  import { ssspPredWgsl } from "./wgsl/sssp-pred.wgsl.js";
73
85
  import { ssspRelaxWgsl } from "./wgsl/sssp-relax.wgsl.js";
86
+ import { triIntersectWgsl } from "./wgsl/tri-intersect.wgsl.js";
74
87
  import { wccCompressWgsl } from "./wgsl/wcc-compress.wgsl.js";
75
88
  import { wccLinkEdgesWgsl } from "./wgsl/wcc-link-edges.wgsl.js";
76
89
  import { wccLinkSampleWgsl } from "./wgsl/wcc-link-sample.wgsl.js";
77
90
  import { wccSampleWgsl } from "./wgsl/wcc-sample.wgsl.js";
78
91
 
79
- /** Every module id of P1-P4, P7, P8 and P9 (later ids are appended, never renamed). */
92
+ /** Every module id of P1-P4, P7, P8, P9 and P11 (later ids are appended, never renamed). */
80
93
  export type KernelId =
81
94
  | "degree"
82
95
  | "reduce"
@@ -129,7 +142,16 @@ export type KernelId =
129
142
  | "bc-backward"
130
143
  | "bc-gather"
131
144
  | "bc-edge-gather"
132
- | "bc-forward-edge";
145
+ | "bc-forward-edge"
146
+ | "apsp-init"
147
+ | "apsp-fw"
148
+ | "coo-emit"
149
+ | "run-flags"
150
+ | "coo-scatter"
151
+ | "orient-flags"
152
+ | "tri-intersect"
153
+ | "group-by-key-row"
154
+ | "lpa-step";
133
155
 
134
156
  /** One registry entry: everything of a WgslModuleSpec except the per-variant overrides and snippets. */
135
157
  export interface KernelEntry {
@@ -144,7 +166,7 @@ export interface KernelEntry {
144
166
  /** The snippet marker names the body carries (segmented-reduce: ["VALUE"]). */
145
167
  readonly snippetSlots: readonly string[];
146
168
  /** The phase the entry landed in (documentation and the compile-matrix filter). */
147
- readonly phase: "P1" | "P2" | "P3" | "P4" | "P7" | "P8" | "P9";
169
+ readonly phase: "P1" | "P2" | "P3" | "P4" | "P7" | "P8" | "P9" | "P11";
148
170
  }
149
171
 
150
172
  // ---- the generated blocks (spec 5.3; contract 3.10.2): field order = byte order, offsets in the JSDoc
@@ -512,6 +534,38 @@ export const BF_FLAGS: UniformBlock = UniformBlock.define(
512
534
  { layout: "storage" },
513
535
  );
514
536
 
537
+ /** `ApspParams` (uniform, 16 B; design 8.7): `n` @0 (the node count, the matrix side), `round` @4 (the pivot block index of the blocked Floyd-Warshall round), `blocks` @8 (`ceil(n / APSP_TILE)`), `infBits` @12 (`F32_INF_BITS`: a kernel reads `+Infinity` from a uniform because Tint refuses it as a constant expression). */
538
+ export const APSP_PARAMS: UniformBlock = UniformBlock.define("ApspParams", [
539
+ ["n", "u32"],
540
+ ["round", "u32"],
541
+ ["blocks", "u32"],
542
+ ["infBits", "u32"],
543
+ ]);
544
+
545
+ /** `CooParams` (uniform, 16 B; P11): `count` @0 (the arcs or positions of the dispatch), `pad0` @4, `pad1` @8, `pad2` @12. The one params block of `coo-emit`, `run-flags`, `coo-scatter`, `orient-flags` and `tri-intersect`. */
546
+ export const COO_PARAMS: UniformBlock = UniformBlock.define("CooParams", [
547
+ ["count", "u32"],
548
+ ["pad0", "u32"],
549
+ ["pad1", "u32"],
550
+ ["pad2", "u32"],
551
+ ]);
552
+
553
+ /** `GroupParams` (uniform, 16 B; P11): `rowsBase` @0 (the word of `rows` where the dispatch's row list starts), `basesBase` @4 (the word where the workgroup tier's region offsets start), `count` @8 (the rows of the dispatch), `pad0` @12. */
554
+ export const GROUP_PARAMS: UniformBlock = UniformBlock.define("GroupParams", [
555
+ ["rowsBase", "u32"],
556
+ ["basesBase", "u32"],
557
+ ["count", "u32"],
558
+ ["pad0", "u32"],
559
+ ]);
560
+
561
+ /** `LpaParams` (uniform, 16 B; P11): `n` @0, `direction` @4 (0: a pass that moves labels down only, 1: up only), `counterIndex` @8 (the word of `counters` that receives the pass's move count), `pad0` @12. */
562
+ export const LPA_PARAMS: UniformBlock = UniformBlock.define("LpaParams", [
563
+ ["n", "u32"],
564
+ ["direction", "u32"],
565
+ ["counterIndex", "u32"],
566
+ ["pad0", "u32"],
567
+ ]);
568
+
515
569
  // ---- the entries (contract 3.10.1; group 0 = graph, 1 = state, 2 = params, 3 = cold)
516
570
 
517
571
  /**
@@ -1516,6 +1570,183 @@ const BC_FORWARD_EDGE: KernelEntry = {
1516
1570
  phase: "P9",
1517
1571
  };
1518
1572
 
1573
+ /** `apsp-init` (design 8.7): one lane per row writes that row's arcs into the `+Infinity`-filled `n x n` matrix, the cheapest of parallel arcs, then the diagonal zero; 5 storage bindings (the four graph slots -- `perm` bound to its dummy, the rows are never permuted -- and `dist`). */
1574
+ const APSP_INIT: KernelEntry = {
1575
+ id: "apsp-init",
1576
+ body: apspInitWgsl,
1577
+ entryPoint: "apsp_init",
1578
+ bindings: GRAPH_SLOTS.concat(decl(1, 0, "dist", "storage", "array<f32>"), decl(2, 0, "P", "uniform", "ApspParams")),
1579
+ overrideDecls: [],
1580
+ uniforms: [APSP_PARAMS],
1581
+ needs: [],
1582
+ snippetSlots: [],
1583
+ phase: "P9",
1584
+ };
1585
+
1586
+ /** `apsp-fw` (design 8.7): one phase of one blocked Floyd-Warshall round over 32 x 32 tiles in workgroup memory -- `PHASE` 0 the pivot block, 1 the pivot row and column, 2 every other block; 1 storage binding (`dist`, read-write). */
1587
+ const APSP_FW: KernelEntry = {
1588
+ id: "apsp-fw",
1589
+ body: apspFwWgsl,
1590
+ entryPoint: "apsp_fw",
1591
+ bindings: [decl(1, 0, "dist", "storage", "array<f32>"), decl(2, 0, "P", "uniform", "ApspParams")],
1592
+ overrideDecls: [{ name: "PHASE", type: "u32", default: 0 }],
1593
+ uniforms: [APSP_PARAMS],
1594
+ needs: [],
1595
+ snippetSlots: [],
1596
+ phase: "P9",
1597
+ };
1598
+
1599
+ /** `coo-emit` (design 6 row 10; P11): position i takes arc i, or `order[i]` under INDEXED, and writes its source, target and weight -- arc 2e is edge e as declared, 2e + 1 its reverse; a self-loop's two arcs become `INVALID_INDEX`; 7 storage bindings. */
1600
+ const COO_EMIT: KernelEntry = {
1601
+ id: "coo-emit",
1602
+ body: cooEmitWgsl,
1603
+ entryPoint: "coo_emit",
1604
+ bindings: [
1605
+ decl(1, 0, "edgeSrc", "storage-ro", "array<u32>"),
1606
+ decl(1, 1, "edgeDst", "storage-ro", "array<u32>"),
1607
+ decl(1, 2, "edgeWeight", "storage-ro", "array<f32>"),
1608
+ decl(1, 3, "order", "storage-ro", "array<u32>"),
1609
+ decl(1, 4, "outSrc", "storage", "array<u32>"),
1610
+ decl(1, 5, "outDst", "storage", "array<u32>"),
1611
+ decl(1, 6, "outWeight", "storage", "array<f32>"),
1612
+ decl(2, 0, "P", "uniform", "CooParams"),
1613
+ ],
1614
+ overrideDecls: [
1615
+ { name: "INDEXED", type: "bool", default: false },
1616
+ { name: "WEIGHTED", type: "bool", default: false },
1617
+ ],
1618
+ uniforms: [COO_PARAMS],
1619
+ needs: [],
1620
+ snippetSlots: [],
1621
+ phase: "P11",
1622
+ };
1623
+
1624
+ /** `run-flags` (design 6 row 10; P11): 1 where an arc opens a run of equal (keysA, keysB) pairs and is not dropped; 3 storage bindings. */
1625
+ const RUN_FLAGS: KernelEntry = {
1626
+ id: "run-flags",
1627
+ body: runFlagsWgsl,
1628
+ entryPoint: "run_flags",
1629
+ bindings: [
1630
+ decl(1, 0, "keysA", "storage-ro", "array<u32>"),
1631
+ decl(1, 1, "keysB", "storage-ro", "array<u32>"),
1632
+ decl(1, 2, "flags", "storage", "array<u32>"),
1633
+ decl(2, 0, "P", "uniform", "CooParams"),
1634
+ ],
1635
+ overrideDecls: [],
1636
+ uniforms: [COO_PARAMS],
1637
+ needs: [],
1638
+ snippetSlots: [],
1639
+ phase: "P11",
1640
+ };
1641
+
1642
+ /** `coo-scatter` (design 6 row 10; P11): the scatter of `cooToCsr` -- the cursor mode, or under SORTED_INPUT the order-preserving mode whose precondition flag is `cursors[0]`; 7 storage bindings. */
1643
+ const COO_SCATTER: KernelEntry = {
1644
+ id: "coo-scatter",
1645
+ body: cooScatterWgsl,
1646
+ entryPoint: "coo_scatter",
1647
+ bindings: [
1648
+ decl(1, 0, "src", "storage-ro", "array<u32>"),
1649
+ decl(1, 1, "dst", "storage-ro", "array<u32>"),
1650
+ decl(1, 2, "weight", "storage-ro", "array<f32>"),
1651
+ decl(1, 3, "rowPtr", "storage-ro", "array<u32>"),
1652
+ decl(1, 4, "cursors", "storage", "array<atomic<u32>>"),
1653
+ decl(1, 5, "colIdx", "storage", "array<u32>"),
1654
+ decl(1, 6, "outWeight", "storage", "array<f32>"),
1655
+ decl(2, 0, "P", "uniform", "CooParams"),
1656
+ ],
1657
+ overrideDecls: [
1658
+ { name: "SORTED_INPUT", type: "bool", default: false },
1659
+ { name: "WEIGHTED", type: "bool", default: false },
1660
+ ],
1661
+ uniforms: [COO_PARAMS],
1662
+ needs: [],
1663
+ snippetSlots: [],
1664
+ phase: "P11",
1665
+ };
1666
+
1667
+ /** `orient-flags` (design 8.5; P11): 1 where arc (u, v) points up the (degree, id) order, one arc per undirected edge; 4 storage bindings. */
1668
+ const ORIENT_FLAGS: KernelEntry = {
1669
+ id: "orient-flags",
1670
+ body: orientFlagsWgsl,
1671
+ entryPoint: "orient_flags",
1672
+ bindings: [
1673
+ decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
1674
+ decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
1675
+ decl(1, 2, "src", "storage-ro", "array<u32>"),
1676
+ decl(1, 3, "flags", "storage", "array<u32>"),
1677
+ decl(2, 0, "P", "uniform", "CooParams"),
1678
+ ],
1679
+ overrideDecls: [],
1680
+ uniforms: [COO_PARAMS],
1681
+ needs: [],
1682
+ snippetSlots: [],
1683
+ phase: "P11",
1684
+ };
1685
+
1686
+ /** `tri-intersect` (design 8.5, 8.10 "Triangle intersection"; P11): one oriented arc per invocation, merge or binary-search intersection (SEARCH 0 chooses, 1 merge, 2 search), u32 atomic per-node counts; 4 storage bindings. */
1687
+ const TRI_INTERSECT: KernelEntry = {
1688
+ id: "tri-intersect",
1689
+ body: triIntersectWgsl,
1690
+ entryPoint: "tri_intersect",
1691
+ bindings: [
1692
+ decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
1693
+ decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
1694
+ decl(1, 2, "src", "storage-ro", "array<u32>"),
1695
+ decl(1, 3, "counts", "storage", "array<atomic<u32>>"),
1696
+ decl(2, 0, "P", "uniform", "CooParams"),
1697
+ ],
1698
+ overrideDecls: [{ name: "SEARCH", type: "u32", default: 0 }],
1699
+ uniforms: [COO_PARAMS],
1700
+ needs: [],
1701
+ snippetSlots: [],
1702
+ phase: "P11",
1703
+ };
1704
+
1705
+ /** `group-by-key-row` (design 8.6; P11): per listed row, the target key with the largest summed weight (lowest key on a tie), TIER 0 a thread per row, else a workgroup per row over a global hash region; 8 storage bindings. */
1706
+ const GROUP_BY_KEY_ROW: KernelEntry = {
1707
+ id: "group-by-key-row",
1708
+ body: groupByKeyRowWgsl,
1709
+ entryPoint: "group_by_key_row",
1710
+ bindings: [
1711
+ decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
1712
+ decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
1713
+ decl(1, 2, "weights", "storage-ro", "array<f32>"),
1714
+ decl(1, 3, "keyIn", "storage-ro", "array<u32>"),
1715
+ decl(1, 4, "rows", "storage-ro", "array<u32>"),
1716
+ decl(1, 5, "hashRegion", "storage", "array<atomic<u32>>"),
1717
+ decl(1, 6, "bestKey", "storage", "array<u32>"),
1718
+ decl(1, 7, "bestScore", "storage", "array<f32>"),
1719
+ decl(2, 0, "P", "uniform", "GroupParams"),
1720
+ ],
1721
+ overrideDecls: [
1722
+ { name: "TIER", type: "u32", default: 0 },
1723
+ { name: "WEIGHTED", type: "bool", default: false },
1724
+ ],
1725
+ uniforms: [GROUP_PARAMS],
1726
+ needs: [],
1727
+ snippetSlots: [],
1728
+ phase: "P11",
1729
+ };
1730
+
1731
+ /** `lpa-step` (design 8.6, 8.10 "Label propagation"; P11): a vertex adopts its best neighbour label when the move goes the pass's direction; one atomic per workgroup adds the moves to `counters[P.counterIndex]`; 4 storage bindings. */
1732
+ const LPA_STEP: KernelEntry = {
1733
+ id: "lpa-step",
1734
+ body: lpaStepWgsl,
1735
+ entryPoint: "lpa_step",
1736
+ bindings: [
1737
+ decl(1, 0, "labelsIn", "storage-ro", "array<u32>"),
1738
+ decl(1, 1, "bestKey", "storage-ro", "array<u32>"),
1739
+ decl(1, 2, "labelsOut", "storage", "array<u32>"),
1740
+ decl(1, 3, "counters", "storage", "array<atomic<u32>>"),
1741
+ decl(2, 0, "P", "uniform", "LpaParams"),
1742
+ ],
1743
+ overrideDecls: [],
1744
+ uniforms: [LPA_PARAMS],
1745
+ needs: [],
1746
+ snippetSlots: [],
1747
+ phase: "P11",
1748
+ };
1749
+
1519
1750
  /**
1520
1751
  * The entries by id, in dispatch order. PLAN DECISION: `KernelId` is declared in full (contract 3.10) while the
1521
1752
  * entries landed phase by phase, so the table is built as a Partial record and exported below through the
@@ -1525,8 +1756,10 @@ const BC_FORWARD_EDGE: KernelEntry = {
1525
1756
  * and `"fa2-to-scene"`; M8b-T3 landed the seven P7 entries and P4 its thirteen; P8-T3 landed the three compact /
1526
1757
  * dedupe entries, P8-T4 `"frontier-finalize"`, P8-T5 `"advance-expand"`, P8-T6 `"bfs-contract"` and `"sssp-pred"` and
1527
1758
  * P8-T7 `"bfs-fused"`, P8-T8 `"bfs-bottom-up"`, `"bfs-bitset-build"` and `"bfs-unvisited-flags"`, P8-T9
1528
- * `"sssp-relax"`, P8-T10 `"bf-relax"` and P8-T11 `"closeness-sweep"` and `"closeness-reduce"`, and betweenness the
1529
- * six `"bc-*"` entries, so every member of `KernelId` is present and the assertion is exact.
1759
+ * `"sssp-relax"`, P8-T10 `"bf-relax"` and P8-T11 `"closeness-sweep"` and `"closeness-reduce"`, betweenness the
1760
+ * six `"bc-*"` entries, all-pairs shortest paths `"apsp-init"` and `"apsp-fw"`, and P11 its seven (the graph
1761
+ * build, the group-by-key, label propagation's step and triangle counting), so every member of `KernelId`
1762
+ * is present and the assertion is exact.
1530
1763
  */
1531
1764
  const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze({
1532
1765
  degree: DEGREE,
@@ -1581,6 +1814,15 @@ const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze
1581
1814
  "bc-gather": BC_GATHER,
1582
1815
  "bc-edge-gather": BC_EDGE_GATHER,
1583
1816
  "bc-forward-edge": BC_FORWARD_EDGE,
1817
+ "apsp-init": APSP_INIT,
1818
+ "apsp-fw": APSP_FW,
1819
+ "coo-emit": COO_EMIT,
1820
+ "run-flags": RUN_FLAGS,
1821
+ "coo-scatter": COO_SCATTER,
1822
+ "orient-flags": ORIENT_FLAGS,
1823
+ "tri-intersect": TRI_INTERSECT,
1824
+ "group-by-key-row": GROUP_BY_KEY_ROW,
1825
+ "lpa-step": LPA_STEP,
1584
1826
  });
1585
1827
 
1586
1828
  /** THE registry (spec 3.5): every entry, keyed by id. */
@@ -393,7 +393,8 @@ export class GraphResidency {
393
393
  /**
394
394
  * Uploads (or finds) a view: outDegree / inDegree / degreeOrder / reverseDegreeOrder upload one array each;
395
395
  * reverse and edgeList (P7) upload their arrays perArray, never into the arena, and are memoised per record so
396
- * a second call uploads nothing (spec 4.3). coo and mate -> E_UNSUPPORTED until P11. packViews concatenates a
396
+ * a second call uploads nothing (spec 4.3). coo uploads its per-arc `src` (P11; the rest aliases the core); mate ->
397
+ * E_UNSUPPORTED. packViews concatenates a
397
398
  * reverse or edgeList view into ONE buffer at STORAGE_ALIGN offsets; on any other view `true` is E_UNSUPPORTED
398
399
  * { option: "packViews" }.
399
400
  * @param s - the snapshot
@@ -465,10 +466,20 @@ export class GraphResidency {
465
466
  break;
466
467
  }
467
468
  case "coo":
469
+ // dst, arcToEdge and weights alias the core's colIdx / arcToEdge / weights (graph-format design 7.2):
470
+ // only the per-arc source is new, and a kernel binds the rest from core()
471
+ this.assertNotReleased(s);
472
+ this.assertNonEmpty(s);
473
+ if (s.arcCount === 0) {
474
+ return Object.freeze({ view: name, bindings: Object.freeze({}), scalars: Object.freeze({}) });
475
+ }
476
+ array = s.coo().src;
477
+ bindingName = "src";
478
+ break;
468
479
  case "mate":
469
- throw new WebGpuGraphError("E_UNSUPPORTED", `the ${name} view is not uploaded before P11`, {
470
- feature: `view:${name}`,
471
- hint: "outDegree, inDegree, degreeOrder, reverseDegreeOrder, reverse and edgeList are uploaded",
480
+ throw new WebGpuGraphError("E_UNSUPPORTED", "the mate view is not uploaded: no kernel reads it", {
481
+ feature: "view:mate",
482
+ hint: "outDegree, inDegree, degreeOrder, reverseDegreeOrder, coo, reverse and edgeList are uploaded",
472
483
  });
473
484
  default:
474
485
  throw invalid("name", name, "a view name");