@graphty/webgpu-graph-algorithms 0.6.13 → 0.6.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. package/README.md +52 -52
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-Bi6AhScG.js → context-oXphO3yj.js} +36 -28
  4. package/dist/chunks/context-oXphO3yj.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/accelerator.d.ts +3 -2
  7. package/dist/src/accelerator.d.ts.map +1 -1
  8. package/dist/src/accelerator.js +48 -4
  9. package/dist/src/accelerator.js.map +1 -1
  10. package/dist/src/algorithms/betweenness.d.ts +70 -0
  11. package/dist/src/algorithms/betweenness.d.ts.map +1 -0
  12. package/dist/src/algorithms/betweenness.js +538 -0
  13. package/dist/src/algorithms/betweenness.js.map +1 -0
  14. package/dist/src/algorithms/closeness.d.ts +15 -5
  15. package/dist/src/algorithms/closeness.d.ts.map +1 -1
  16. package/dist/src/algorithms/closeness.js +112 -26
  17. package/dist/src/algorithms/closeness.js.map +1 -1
  18. package/dist/src/constants.d.ts +8 -0
  19. package/dist/src/constants.d.ts.map +1 -1
  20. package/dist/src/constants.js +8 -0
  21. package/dist/src/constants.js.map +1 -1
  22. package/dist/src/index.d.ts +4 -2
  23. package/dist/src/index.d.ts.map +1 -1
  24. package/dist/src/index.js +1 -0
  25. package/dist/src/index.js.map +1 -1
  26. package/dist/src/kernels.d.ts +12 -6
  27. package/dist/src/kernels.d.ts.map +1 -1
  28. package/dist/src/kernels.js +153 -7
  29. package/dist/src/kernels.js.map +1 -1
  30. package/dist/src/primitives/frontier.d.ts +2 -0
  31. package/dist/src/primitives/frontier.d.ts.map +1 -1
  32. package/dist/src/primitives/frontier.js +2 -0
  33. package/dist/src/primitives/frontier.js.map +1 -1
  34. package/dist/src/types/accelerator.d.ts +11 -7
  35. package/dist/src/types/accelerator.d.ts.map +1 -1
  36. package/dist/src/types/algorithms.d.ts +4 -0
  37. package/dist/src/types/algorithms.d.ts.map +1 -1
  38. package/dist/src/types/betweenness.d.ts +35 -0
  39. package/dist/src/types/betweenness.d.ts.map +1 -0
  40. package/dist/src/types/betweenness.js +7 -0
  41. package/dist/src/types/betweenness.js.map +1 -0
  42. package/dist/src/wgsl/bc-backward.wgsl.d.ts +15 -0
  43. package/dist/src/wgsl/bc-backward.wgsl.d.ts.map +1 -0
  44. package/dist/src/wgsl/bc-backward.wgsl.js +34 -0
  45. package/dist/src/wgsl/bc-backward.wgsl.js.map +1 -0
  46. package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts +12 -0
  47. package/dist/src/wgsl/bc-edge-gather.wgsl.d.ts.map +1 -0
  48. package/dist/src/wgsl/bc-edge-gather.wgsl.js +36 -0
  49. package/dist/src/wgsl/bc-edge-gather.wgsl.js.map +1 -0
  50. package/dist/src/wgsl/bc-finalize.wgsl.d.ts +21 -0
  51. package/dist/src/wgsl/bc-finalize.wgsl.d.ts.map +1 -0
  52. package/dist/src/wgsl/bc-finalize.wgsl.js +47 -0
  53. package/dist/src/wgsl/bc-finalize.wgsl.js.map +1 -0
  54. package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts +15 -0
  55. package/dist/src/wgsl/bc-forward-edge.wgsl.d.ts.map +1 -0
  56. package/dist/src/wgsl/bc-forward-edge.wgsl.js +76 -0
  57. package/dist/src/wgsl/bc-forward-edge.wgsl.js.map +1 -0
  58. package/dist/src/wgsl/bc-forward.wgsl.d.ts +23 -0
  59. package/dist/src/wgsl/bc-forward.wgsl.d.ts.map +1 -0
  60. package/dist/src/wgsl/bc-forward.wgsl.js +106 -0
  61. package/dist/src/wgsl/bc-forward.wgsl.js.map +1 -0
  62. package/dist/src/wgsl/bc-gather.wgsl.d.ts +9 -0
  63. package/dist/src/wgsl/bc-gather.wgsl.d.ts.map +1 -0
  64. package/dist/src/wgsl/bc-gather.wgsl.js +20 -0
  65. package/dist/src/wgsl/bc-gather.wgsl.js.map +1 -0
  66. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +4 -1
  67. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +1 -1
  68. package/dist/src/wgsl/closeness-reduce.wgsl.js +8 -4
  69. package/dist/src/wgsl/closeness-reduce.wgsl.js.map +1 -1
  70. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +4 -2
  71. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +1 -1
  72. package/dist/src/wgsl/closeness-sweep.wgsl.js +12 -2
  73. package/dist/src/wgsl/closeness-sweep.wgsl.js.map +1 -1
  74. package/dist/webgpu-graph-algorithms.js +1037 -168
  75. package/dist/webgpu-graph-algorithms.js.map +1 -1
  76. package/package.json +5 -5
  77. package/src/accelerator.ts +75 -7
  78. package/src/algorithms/betweenness.ts +739 -0
  79. package/src/algorithms/closeness.ts +124 -32
  80. package/src/constants.ts +8 -0
  81. package/src/index.ts +8 -0
  82. package/src/kernels.ts +169 -10
  83. package/src/primitives/frontier.ts +4 -0
  84. package/src/types/accelerator.ts +18 -6
  85. package/src/types/algorithms.ts +5 -0
  86. package/src/types/betweenness.ts +38 -0
  87. package/src/wgsl/bc-backward.wgsl.ts +33 -0
  88. package/src/wgsl/bc-edge-gather.wgsl.ts +35 -0
  89. package/src/wgsl/bc-finalize.wgsl.ts +46 -0
  90. package/src/wgsl/bc-forward-edge.wgsl.ts +75 -0
  91. package/src/wgsl/bc-forward.wgsl.ts +105 -0
  92. package/src/wgsl/bc-gather.wgsl.ts +19 -0
  93. package/src/wgsl/closeness-reduce.wgsl.ts +8 -4
  94. package/src/wgsl/closeness-sweep.wgsl.ts +12 -2
  95. package/dist/chunks/context-Bi6AhScG.js.map +0 -1
@@ -1,6 +1,6 @@
1
- import { W as WebGpuGraphError, U as UNIFORM_SLOT_BYTES, B as BufferUsage, M as MAX_WORKGROUPS_PER_DIM, a as WGSL_RESERVED_WORDS, S as STATE_HEADER_BYTES, d as deviceLostError, i as isWebGpuGraphError, b as U32_MAX$2, c as MAX_LEVELS_PER_SUBMIT, R as RADIX_BINS, F as FUSED_FRONTIER_MAX, e as BEAMER_BETA, f as SSSP_DELTA_FACTOR, g as F32_INF_BITS, h as GRID_COARSEST_SIDE, j as GRID_MIN_SIDE, k as GRID_SORT_BITS, l as FA2_DEFAULTS, m as MAX_ITERATIONS_PER_STEP, n as MAX_1D_ITEMS, o as hasErrorCode, p as FA2_FLAG_FIRST, P as PARTIAL_BYTES, E as EXACT_TILES_PER_PASS, I as INDIRECT_ARGS_STRIDE, q as GRID_HUB_CELL, L as LAYOUT_TUNING_DEFAULTS, r as EXACT_MAX_NODES, s as SETTLE_FLOOR_UNBOUNDED, T as TRACE_RECORD_BYTES, t as GRID_BBOX_MARGIN, u as GRID_EXTENT_FLOOR, v as FR_ADAPTIVE_MAX_ITERATIONS, w as FR_START_TEMPERATURE, x as FA2_FLAG_ADAPTIVE, y as SETTLE_FLOOR_FRACTION, z as FR_REHEAT_FRACTION, A as FR_DEFAULTS, C as SE_DEFAULTS, D as SETTLE_FLOOR_REFERENCE_NODES, H as SE_SCALE_REFERENCE_NODES } from "./chunks/context-Bi6AhScG.js";
2
- import { J, G, K, N, O, Q } from "./chunks/context-Bi6AhScG.js";
3
- import { renumberPartition, INVALID_INDEX, makeMask, maskTest, expandEdges, fromEdgeArrays } from "@graphty/graph-format";
1
+ import { W as WebGpuGraphError, U as UNIFORM_SLOT_BYTES, B as BufferUsage, M as MAX_WORKGROUPS_PER_DIM, a as WGSL_RESERVED_WORDS, S as STATE_HEADER_BYTES, d as deviceLostError, i as isWebGpuGraphError, b as U32_MAX$2, c as MAX_LEVELS_PER_SUBMIT, R as RADIX_BINS, F as FUSED_FRONTIER_MAX, e as BEAMER_BETA, f as SSSP_DELTA_FACTOR, g as F32_INF_BITS, h as BC_EDGE_PARALLEL_GAMMA, j as BC_BATCH_BUDGET_FRACTION, k as BC_MAX_BATCH, l as BC_BACKWARD_LEVELS_PER_SUBMIT, m as GRID_COARSEST_SIDE, n as GRID_MIN_SIDE, o as GRID_SORT_BITS, p as FA2_DEFAULTS, q as MAX_ITERATIONS_PER_STEP, r as MAX_1D_ITEMS, s as hasErrorCode, t as FA2_FLAG_FIRST, P as PARTIAL_BYTES, E as EXACT_TILES_PER_PASS, I as INDIRECT_ARGS_STRIDE, u as GRID_HUB_CELL, L as LAYOUT_TUNING_DEFAULTS, v as EXACT_MAX_NODES, w as SETTLE_FLOOR_UNBOUNDED, T as TRACE_RECORD_BYTES, x as GRID_BBOX_MARGIN, y as GRID_EXTENT_FLOOR, z as FR_ADAPTIVE_MAX_ITERATIONS, A as FR_START_TEMPERATURE, C as FA2_FLAG_ADAPTIVE, D as SETTLE_FLOOR_FRACTION, H as FR_REHEAT_FRACTION, J as FR_DEFAULTS, K as SE_DEFAULTS, N as SETTLE_FLOOR_REFERENCE_NODES, O as SE_SCALE_REFERENCE_NODES } from "./chunks/context-oXphO3yj.js";
2
+ import { Q, G, V, X, Y, Z } from "./chunks/context-oXphO3yj.js";
3
+ import { renumberPartition, INVALID_INDEX, foldArcs, makeMask, maskTest, expandEdges, fromEdgeArrays } from "@graphty/graph-format";
4
4
  class UniformRing {
5
5
  /**
6
6
  * Creates the ring buffer (`slots x UNIFORM_SLOT_BYTES` bytes, UNIFORM | COPY_DST) through the allocator.
@@ -605,6 +605,254 @@ fn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
605
605
  }
606
606
  `
607
607
  );
608
+ const bcBackwardWgsl = (
609
+ /* wgsl */
610
+ `
611
+ @compute @workgroup_size(WG)
612
+ fn bc_backward(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
613
+ for (var i = linear_id(wid, lid.x); i < P.count; i = i + P.stride) {
614
+ let t = S[P.start + i]; // s * n + w
615
+ let w = t % P.n;
616
+ let base = t - w; // s * n
617
+ let succ = depthK[t] + 1u;
618
+ let sw = f32(sigmaK[t]);
619
+ var acc = 0.0;
620
+ for (var a = rowPtr[w]; a < rowPtr[w + 1u]; a = a + 1u) {
621
+ let v = base + colIdx[a];
622
+ if (depthK[v] == succ) { // v is a successor of w for source s
623
+ acc = acc + (sw / f32(sigmaK[v])) * (1.0 + deltaK[v]);
624
+ }
625
+ }
626
+ deltaK[t] = acc; // written once per (w, s)
627
+ }
628
+ }
629
+ `
630
+ );
631
+ const bcEdgeGatherWgsl = (
632
+ /* wgsl */
633
+ `
634
+ @compute @workgroup_size(WG)
635
+ fn bc_edge_gather(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
636
+ for (var a = linear_id(wid, lid.x); a < P.count; a = a + P.stride) {
637
+ var lo = 0u; // the row w with rowPtr[w] <= a < rowPtr[w + 1]
638
+ var hi = P.n;
639
+ loop {
640
+ if (lo >= hi) { break; }
641
+ let mid = (lo + hi) / 2u;
642
+ if (rowPtr[mid + 1u] <= a) { lo = mid + 1u; } else { hi = mid; }
643
+ }
644
+ let w = lo;
645
+ let nbr = colIdx[a];
646
+ var acc = arcScores[a];
647
+ for (var s = 0u; s < P.k; s = s + 1u) {
648
+ let base = s * P.n;
649
+ let dw = depthK[base + w];
650
+ if (dw != INVALID_INDEX && depthK[base + nbr] == dw + 1u) { // (w, nbr) is on a shortest path from s
651
+ acc = acc + (f32(sigmaK[base + w]) / f32(sigmaK[base + nbr])) * (1.0 + deltaK[base + nbr]);
652
+ }
653
+ }
654
+ arcScores[a] = acc;
655
+ }
656
+ }
657
+ `
658
+ );
659
+ const bcFinalizeWgsl = (
660
+ /* wgsl */
661
+ `
662
+ @compute @workgroup_size(WG)
663
+ fn bc_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
664
+ if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
665
+ if (P.role == 1u) { // the seed of a batch
666
+ for (var i = 0u; i < P.k; i = i + 1u) {
667
+ let t = S[i];
668
+ depthK[t] = 0u; // the source is at depth 0
669
+ sigmaK[t] = 1u; // with one shortest path, itself
670
+ }
671
+ ends[0] = 0u;
672
+ atomicStore(&counters[26], P.k); // stackTop: the seeds are the log's first k entries
673
+ atomicStore(&counters[27], 0u); // sigmaOverflow
674
+ atomicStore(&counters[11], U32_MAX); // level: the first boundary brings it to 0
675
+ atomicStore(&counters[15], 0u); // done
676
+ return;
677
+ }
678
+ if (atomicLoad(&counters[15]) != 0u) { return; } // done: a no-op level the host recorded past the end
679
+ let level = atomicLoad(&counters[11]) + 1u;
680
+ let top = atomicLoad(&counters[26]);
681
+ ends[level + 1u] = top; // the level's entries end where the log ends now
682
+ let count = top - ends[level];
683
+ atomicStore(&counters[0], count); // frontierCount (the inspect seam reads it)
684
+ atomicStore(&counters[11], level);
685
+ atomicStore(&counters[15], select(0u, 1u, count == 0u)); // an empty level ends the batch
686
+ }
687
+ `
688
+ );
689
+ const bcForwardWgsl = (
690
+ /* wgsl */
691
+ `
692
+ var<workgroup> sh: array<u32, WG>; // the block's degrees, then their inclusive scan
693
+ var<workgroup> rowStart: array<u32, WG>; // the first arc of each entry's row
694
+ var<workgroup> entryOf: array<u32, WG>; // each entry, s * n + u
695
+ var<workgroup> wstart: u32; // the level's first log index
696
+ var<workgroup> wcount: u32; // the level's entry count
697
+ var<workgroup> wwon: atomic<u32>; // the strip's winners
698
+ var<workgroup> wbase: u32; // where the strip's winners go in the log
699
+
700
+ @compute @workgroup_size(WG)
701
+ fn bc_forward(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
702
+ let level = atomicLoad(&counters[11]);
703
+ if (lid.x == 0u) {
704
+ let lo = ends[level];
705
+ wstart = lo;
706
+ wcount = ends[level + 1u] - lo;
707
+ }
708
+ let start = workgroupUniformLoad(&wstart);
709
+ let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
710
+ let next = level + 1u;
711
+ for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
712
+ let i = b0 + lid.x;
713
+ var deg = 0u;
714
+ var first = 0u;
715
+ var entry = 0u;
716
+ if (i < count) { // guarded loads into locals (3.5 rule 1)
717
+ entry = S[start + i];
718
+ let u = entry % P.n;
719
+ first = rowPtr[u];
720
+ deg = rowPtr[u + 1u] - first;
721
+ }
722
+ sh[lid.x] = deg;
723
+ rowStart[lid.x] = first;
724
+ entryOf[lid.x] = entry;
725
+ workgroupBarrier();
726
+ for (var s = 1u; s < WG; s = s * 2u) { // Hillis-Steele inclusive scan of the degrees
727
+ var t = 0u;
728
+ if (lid.x >= s) { t = sh[lid.x - s]; }
729
+ workgroupBarrier();
730
+ sh[lid.x] = sh[lid.x] + t;
731
+ workgroupBarrier();
732
+ }
733
+ let aggregate = workgroupUniformLoad(&sh[WG - 1u]); // uniform; includes a barrier
734
+ for (var p0 = 0u; p0 < aggregate; p0 = p0 + WG) { // strip [0, aggregate) WG arcs at a time
735
+ let p = p0 + lid.x;
736
+ var won = false;
737
+ var claimed = 0u;
738
+ if (p < aggregate) {
739
+ var lo = 0u; // upper_bound: the first k with sh[k] > p owns arc p
740
+ var hi = WG;
741
+ loop {
742
+ if (lo >= hi) { break; }
743
+ let mid = (lo + hi) / 2u;
744
+ if (sh[mid] > p) { hi = mid; } else { lo = mid + 1u; }
745
+ }
746
+ let k = lo;
747
+ var exclusive = 0u;
748
+ if (k > 0u) { exclusive = sh[k - 1u]; }
749
+ let origin = entryOf[k]; // s * n + u
750
+ let x = (origin - (origin % P.n)) + colIdx[rowStart[k] + (p - exclusive)]; // s * n + x
751
+ if (atomicLoad(&depthK[x]) == INVALID_INDEX) { // the pre-check of design 16.1
752
+ won = atomicMin(&depthK[x], next) == INVALID_INDEX; // the claim: the one winner appends
753
+ }
754
+ if (atomicLoad(&depthK[x]) == next) { // the count: EVERY arc on a shortest path adds
755
+ let add = atomicLoad(&sigmaK[origin]);
756
+ let old = atomicAdd(&sigmaK[x], add);
757
+ if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported
758
+ }
759
+ claimed = x;
760
+ }
761
+ var slot = 0u;
762
+ if (won) { slot = atomicAdd(&wwon, 1u); }
763
+ workgroupBarrier();
764
+ if (lid.x == 0u) {
765
+ wbase = atomicAdd(&counters[26], atomicLoad(&wwon)); // stackTop: one global atomic per strip
766
+ atomicStore(&wwon, 0u);
767
+ }
768
+ workgroupBarrier();
769
+ if (won) { S[wbase + slot] = claimed; }
770
+ }
771
+ workgroupBarrier(); // sh, rowStart and entryOf are reused by the next block
772
+ }
773
+ }
774
+ `
775
+ );
776
+ const bcForwardEdgeWgsl = (
777
+ /* wgsl */
778
+ `
779
+ var<workgroup> wlive: u32; // 1 when the level has entries
780
+ var<workgroup> wwon: atomic<u32>; // the strip's winners
781
+ var<workgroup> wbase: u32; // where the strip's winners go in the log
782
+
783
+ fn claim(x: u32, next: u32) -> bool {
784
+ if (atomicLoad(&depthK[x]) != INVALID_INDEX) { return false; } // the pre-check of design 16.1
785
+ return atomicMin(&depthK[x], next) == INVALID_INDEX;
786
+ }
787
+
788
+ fn count_paths(origin: u32, x: u32, next: u32) {
789
+ if (atomicLoad(&depthK[x]) == next) { // every arc on a shortest path adds
790
+ let add = atomicLoad(&sigmaK[origin]);
791
+ let old = atomicAdd(&sigmaK[x], add);
792
+ if (old + add < old) { atomicOr(&counters[27], 1u); } // the u32 wrap, reported
793
+ }
794
+ }
795
+
796
+ @compute @workgroup_size(WG)
797
+ fn bc_forward_edge(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
798
+ let level = atomicLoad(&counters[11]);
799
+ if (lid.x == 0u) { wlive = select(0u, 1u, ends[level + 1u] > ends[level]); }
800
+ if (workgroupUniformLoad(&wlive) == 0u) { return; } // uniform: nothing below runs on an empty level
801
+ let next = level + 1u;
802
+ for (var s = 0u; s < P.k; s = s + 1u) {
803
+ let base = s * P.n;
804
+ for (var e0 = group_id(wid) * WG; e0 < P.count; e0 = e0 + P.stride) { // grid-stride over the edges
805
+ let e = e0 + lid.x;
806
+ var a = INVALID_INDEX; // the claims this lane won
807
+ var b = INVALID_INDEX;
808
+ if (e < P.count) {
809
+ let u = base + edgeSrc[e];
810
+ let x = base + edgeDst[e];
811
+ if (atomicLoad(&depthK[u]) == level) {
812
+ if (claim(x, next)) { a = x; }
813
+ count_paths(u, x, next);
814
+ }
815
+ if (UNDIRECTED) { // the other direction of an undirected edge
816
+ if (atomicLoad(&depthK[x]) == level) {
817
+ if (claim(u, next)) { b = u; }
818
+ count_paths(x, u, next);
819
+ }
820
+ }
821
+ }
822
+ let mine = select(0u, 1u, a != INVALID_INDEX) + select(0u, 1u, b != INVALID_INDEX);
823
+ var slot = 0u;
824
+ if (mine != 0u) { slot = atomicAdd(&wwon, mine); }
825
+ workgroupBarrier();
826
+ if (lid.x == 0u) {
827
+ wbase = atomicAdd(&counters[26], atomicLoad(&wwon)); // stackTop: one global atomic per strip
828
+ atomicStore(&wwon, 0u);
829
+ }
830
+ workgroupBarrier();
831
+ if (a != INVALID_INDEX) {
832
+ S[wbase + slot] = a;
833
+ slot = slot + 1u;
834
+ }
835
+ if (b != INVALID_INDEX) { S[wbase + slot] = b; }
836
+ }
837
+ }
838
+ }
839
+ `
840
+ );
841
+ const bcGatherWgsl = (
842
+ /* wgsl */
843
+ `
844
+ @compute @workgroup_size(WG)
845
+ fn bc_gather(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
846
+ for (var w = linear_id(wid, lid.x); w < P.n; w = w + P.stride) {
847
+ var acc = bc[w];
848
+ for (var s = 0u; s < P.k; s = s + 1u) {
849
+ acc = acc + deltaK[s * P.n + w];
850
+ }
851
+ bc[w] = acc;
852
+ }
853
+ }
854
+ `
855
+ );
608
856
  const bfRelaxWgsl = (
609
857
  /* wgsl */
610
858
  `
@@ -854,13 +1102,14 @@ const closenessReduceWgsl = (
854
1102
  @compute @workgroup_size(WG)
855
1103
  fn closeness_reduce(@builtin(local_invocation_id) lid: vec3<u32>) {
856
1104
  if (lid.x != 0u) { return; } // one lane; no barrier follows (3.5 rule 1)
857
- if (P.role == 1u) { // the seed of a batch: P.source is its first source
1105
+ if (P.role != 0u) { // the seed of a batch: P.source is its first source
858
1106
  let k = min(32u, P.n - P.source);
859
1107
  for (var s = 0u; s < k; s = s + 1u) {
860
- let v = P.source + s;
1108
+ var v = P.source + s;
1109
+ if (P.role == 2u) { v = atomicLoad(&perSource[128u + P.bitsBase + P.source + s]); } // a sampled run's list
861
1110
  let bit = 1u << s;
862
- bits[v] = bit; // visited
863
- bits[P.bitsBase + v] = bit; // the frontier level 0 reads (region 1: level 0's parity is 0)
1111
+ bits[v] = bits[v] | bit; // visited
1112
+ bits[P.bitsBase + v] = bits[P.bitsBase + v] | bit; // the frontier level 0 reads (region 1: level 0's parity is 0)
864
1113
  bits[3u * P.bitsBase + v] = 1u; // flags: level 0's compact turns them into the list
865
1114
  }
866
1115
  atomicStore(&counters[0], k); // not done
@@ -908,11 +1157,16 @@ var<workgroup> rowStart: array<u32, WG>; // the first bound arc of each
908
1157
  var<workgroup> rowOf: array<u32, WG>; // the frontier vertex of each entry (the source end of its arcs)
909
1158
  var<workgroup> local: array<atomic<u32>, 32>; // this workgroup's fresh claims per source
910
1159
  var<workgroup> wcount: u32; // the frontier list's length
1160
+ var<workgroup> wdist: u32; // the distance of this level's claims
911
1161
 
912
1162
  @compute @workgroup_size(WG)
913
1163
  fn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
914
- if (lid.x == 0u) { wcount = atomicLoad(&counters[0]); } // the frontier list's length (compact's total)
1164
+ if (lid.x == 0u) {
1165
+ wcount = atomicLoad(&counters[0]); // the frontier list's length (compact's total)
1166
+ wdist = atomicLoad(&counters[11]) + 1u; // the level word: this level claims at level + 1
1167
+ }
915
1168
  let count = workgroupUniformLoad(&wcount); // uniform: the block loop below holds barriers
1169
+ let dist = workgroupUniformLoad(&wdist);
916
1170
  let nextBase = select(2u * P.bitsBase, P.bitsBase, P.mode == 1u); // the region that is next this level
917
1171
  let frontierBase = 3u * P.bitsBase - nextBase; // the other one: the region that is the frontier
918
1172
  for (var b0 = group_id(wid) * WG; b0 < count; b0 = b0 + P.stride) { // grid-stride over blocks of WG entries
@@ -960,6 +1214,9 @@ fn closeness_sweep(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocat
960
1214
  if (fresh != 0u) {
961
1215
  atomicOr(&bits[nextBase + x], fresh);
962
1216
  atomicStore(&bits[3u * P.bitsBase + x], 1u); // flags: x is in the next frontier list (compact reads it)
1217
+ if (P.perNode == 1u) { // a sampled run: x's distance to each source won
1218
+ atomicAdd(&perSource[128u + x], countOneBits(fresh) * dist);
1219
+ }
963
1220
  var b = fresh;
964
1221
  loop { // one tally per set bit of fresh
965
1222
  if (b == 0u) { break; }
@@ -2805,7 +3062,9 @@ const FRONTIER_COUNTERS = UniformBlock.define(
2805
3062
  ["thresholdBits", "u32"],
2806
3063
  ["deltaBits", "u32"],
2807
3064
  ["path", "u32"],
2808
- ["nextDegreeSum", "u32"]
3065
+ ["nextDegreeSum", "u32"],
3066
+ ["stackTop", "u32"],
3067
+ ["sigmaOverflow", "u32"]
2809
3068
  ],
2810
3069
  { layout: "storage" }
2811
3070
  );
@@ -2828,9 +3087,19 @@ const FRONTIER_PARAMS = UniformBlock.define("FrontierParams", [
2828
3087
  ["stride", "u32"],
2829
3088
  ["firstOfSubmit", "u32"],
2830
3089
  ["iteration", "u32"],
2831
- ["pad1", "u32"],
3090
+ ["perNode", "u32"],
2832
3091
  ["pad2", "u32"]
2833
3092
  ]);
3093
+ const BC_PARAMS = UniformBlock.define("BcParams", [
3094
+ ["n", "u32"],
3095
+ ["k", "u32"],
3096
+ ["start", "u32"],
3097
+ ["count", "u32"],
3098
+ ["stride", "u32"],
3099
+ ["role", "u32"],
3100
+ ["pad0", "u32"],
3101
+ ["pad1", "u32"]
3102
+ ]);
2834
3103
  const BF_PARAMS = UniformBlock.define("BfParams", [
2835
3104
  ["edgeCount", "u32"],
2836
3105
  ["stride", "u32"],
@@ -3616,6 +3885,117 @@ const CLOSENESS_REDUCE = {
3616
3885
  snippetSlots: [],
3617
3886
  phase: "P8"
3618
3887
  };
3888
+ const BC_FINALIZE = {
3889
+ id: "bc-finalize",
3890
+ body: bcFinalizeWgsl,
3891
+ entryPoint: "bc_finalize",
3892
+ bindings: [
3893
+ decl(1, 0, "counters", "storage", "array<atomic<u32>>"),
3894
+ decl(1, 1, "ends", "storage", "array<u32>"),
3895
+ decl(1, 2, "S", "storage-ro", "array<u32>"),
3896
+ decl(1, 3, "depthK", "storage", "array<u32>"),
3897
+ decl(1, 4, "sigmaK", "storage", "array<u32>"),
3898
+ decl(2, 0, "P", "uniform", "BcParams")
3899
+ ],
3900
+ overrideDecls: [],
3901
+ uniforms: [BC_PARAMS],
3902
+ needs: [],
3903
+ snippetSlots: [],
3904
+ phase: "P9"
3905
+ };
3906
+ const BC_FORWARD = {
3907
+ id: "bc-forward",
3908
+ body: bcForwardWgsl,
3909
+ entryPoint: "bc_forward",
3910
+ bindings: [
3911
+ decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
3912
+ decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
3913
+ decl(1, 2, "S", "storage", "array<u32>"),
3914
+ decl(1, 3, "ends", "storage-ro", "array<u32>"),
3915
+ decl(1, 4, "counters", "storage", "array<atomic<u32>>"),
3916
+ decl(1, 5, "depthK", "storage", "array<atomic<u32>>"),
3917
+ decl(1, 6, "sigmaK", "storage", "array<atomic<u32>>"),
3918
+ decl(2, 0, "P", "uniform", "BcParams")
3919
+ ],
3920
+ overrideDecls: [],
3921
+ uniforms: [BC_PARAMS],
3922
+ needs: [],
3923
+ snippetSlots: [],
3924
+ phase: "P9"
3925
+ };
3926
+ const BC_BACKWARD = {
3927
+ id: "bc-backward",
3928
+ body: bcBackwardWgsl,
3929
+ entryPoint: "bc_backward",
3930
+ bindings: [
3931
+ decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
3932
+ decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
3933
+ decl(1, 2, "S", "storage-ro", "array<u32>"),
3934
+ decl(1, 3, "depthK", "storage-ro", "array<u32>"),
3935
+ decl(1, 4, "sigmaK", "storage-ro", "array<u32>"),
3936
+ decl(1, 5, "deltaK", "storage", "array<f32>"),
3937
+ decl(2, 0, "P", "uniform", "BcParams")
3938
+ ],
3939
+ overrideDecls: [],
3940
+ uniforms: [BC_PARAMS],
3941
+ needs: [],
3942
+ snippetSlots: [],
3943
+ phase: "P9"
3944
+ };
3945
+ const BC_GATHER = {
3946
+ id: "bc-gather",
3947
+ body: bcGatherWgsl,
3948
+ entryPoint: "bc_gather",
3949
+ bindings: [
3950
+ decl(1, 0, "deltaK", "storage-ro", "array<f32>"),
3951
+ decl(1, 1, "bc", "storage", "array<f32>"),
3952
+ decl(2, 0, "P", "uniform", "BcParams")
3953
+ ],
3954
+ overrideDecls: [],
3955
+ uniforms: [BC_PARAMS],
3956
+ needs: [],
3957
+ snippetSlots: [],
3958
+ phase: "P9"
3959
+ };
3960
+ const BC_EDGE_GATHER = {
3961
+ id: "bc-edge-gather",
3962
+ body: bcEdgeGatherWgsl,
3963
+ entryPoint: "bc_edge_gather",
3964
+ bindings: [
3965
+ decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
3966
+ decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
3967
+ decl(1, 2, "depthK", "storage-ro", "array<u32>"),
3968
+ decl(1, 3, "sigmaK", "storage-ro", "array<u32>"),
3969
+ decl(1, 4, "deltaK", "storage-ro", "array<f32>"),
3970
+ decl(1, 5, "arcScores", "storage", "array<f32>"),
3971
+ decl(2, 0, "P", "uniform", "BcParams")
3972
+ ],
3973
+ overrideDecls: [],
3974
+ uniforms: [BC_PARAMS],
3975
+ needs: [],
3976
+ snippetSlots: [],
3977
+ phase: "P9"
3978
+ };
3979
+ const BC_FORWARD_EDGE = {
3980
+ id: "bc-forward-edge",
3981
+ body: bcForwardEdgeWgsl,
3982
+ entryPoint: "bc_forward_edge",
3983
+ bindings: [
3984
+ decl(1, 0, "edgeSrc", "storage-ro", "array<u32>"),
3985
+ decl(1, 1, "edgeDst", "storage-ro", "array<u32>"),
3986
+ decl(1, 2, "S", "storage", "array<u32>"),
3987
+ decl(1, 3, "ends", "storage-ro", "array<u32>"),
3988
+ decl(1, 4, "counters", "storage", "array<atomic<u32>>"),
3989
+ decl(1, 5, "depthK", "storage", "array<atomic<u32>>"),
3990
+ decl(1, 6, "sigmaK", "storage", "array<atomic<u32>>"),
3991
+ decl(2, 0, "P", "uniform", "BcParams")
3992
+ ],
3993
+ overrideDecls: [{ name: "UNDIRECTED", type: "bool", default: false }],
3994
+ uniforms: [BC_PARAMS],
3995
+ needs: [],
3996
+ snippetSlots: [],
3997
+ phase: "P9"
3998
+ };
3619
3999
  const REGISTRY = Object.freeze({
3620
4000
  degree: DEGREE,
3621
4001
  reduce: REDUCE,
@@ -3662,7 +4042,13 @@ const REGISTRY = Object.freeze({
3662
4042
  "sssp-relax": SSSP_RELAX,
3663
4043
  "bf-relax": BF_RELAX,
3664
4044
  "closeness-sweep": CLOSENESS_SWEEP,
3665
- "closeness-reduce": CLOSENESS_REDUCE
4045
+ "closeness-reduce": CLOSENESS_REDUCE,
4046
+ "bc-finalize": BC_FINALIZE,
4047
+ "bc-forward": BC_FORWARD,
4048
+ "bc-backward": BC_BACKWARD,
4049
+ "bc-gather": BC_GATHER,
4050
+ "bc-edge-gather": BC_EDGE_GATHER,
4051
+ "bc-forward-edge": BC_FORWARD_EDGE
3666
4052
  });
3667
4053
  const bodyOverrides = /* @__PURE__ */ new Map();
3668
4054
  function entryOf(id) {
@@ -3821,7 +4207,7 @@ class ScanPlannerImpl {
3821
4207
  }
3822
4208
  const POISON = 3735928559;
3823
4209
  const CHECK_BLOCKS = 32;
3824
- const RING_SLOTS$6 = 8;
4210
+ const RING_SLOTS$7 = 8;
3825
4211
  const checked = /* @__PURE__ */ new WeakMap();
3826
4212
  function inputAt(i) {
3827
4213
  return i + 1;
@@ -3849,7 +4235,7 @@ async function runCheck(ctx) {
3849
4235
  const count = CHECK_BLOCKS * wg + 1;
3850
4236
  const bytes = 4 * count;
3851
4237
  const lease = ctx.pool.lease();
3852
- const ring = new UniformRing(ctx.device, ctx.allocator, RING_SLOTS$6, "device-check/ring");
4238
+ const ring = new UniformRing(ctx.device, ctx.allocator, RING_SLOTS$7, "device-check/ring");
3853
4239
  try {
3854
4240
  const scope = {
3855
4241
  device: ctx.device,
@@ -4516,12 +4902,12 @@ function algorithmScope(ctx, label, slots) {
4516
4902
  ringOverruns: () => ring.overruns
4517
4903
  };
4518
4904
  }
4519
- const ALGORITHM$4 = "connectedComponents";
4905
+ const ALGORITHM$5 = "connectedComponents";
4520
4906
  const ROUNDS_PER_BATCH$1 = 4;
4521
4907
  const MAX_WCC_ROUNDS = 64;
4522
4908
  const SAMPLE_SIZE = 1024;
4523
4909
  const MAX_STEPS = 1024;
4524
- const RING_SLOTS$5 = 2 * ROUNDS_PER_BATCH$1;
4910
+ const RING_SLOTS$6 = 2 * ROUNDS_PER_BATCH$1;
4525
4911
  function checkDest$4(dest, n) {
4526
4912
  if (dest === void 0) {
4527
4913
  return null;
@@ -4531,7 +4917,7 @@ function checkDest$4(dest, n) {
4531
4917
  }
4532
4918
  throw new WebGpuGraphError(
4533
4919
  "E_INVALID_ARGUMENT",
4534
- `${ALGORITHM$4}: dest must be a Uint32Array of length ${n} over an ArrayBuffer`,
4920
+ `${ALGORITHM$5}: dest must be a Uint32Array of length ${n} over an ArrayBuffer`,
4535
4921
  {
4536
4922
  argument: "dest",
4537
4923
  value: `${dest.constructor.name}(${dest.length})`,
@@ -4541,7 +4927,7 @@ function checkDest$4(dest, n) {
4541
4927
  }
4542
4928
  function coreOf$2(ctx, s) {
4543
4929
  const core = ctx.residency.core(s);
4544
- assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM$4);
4930
+ assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM$5);
4545
4931
  return core;
4546
4932
  }
4547
4933
  function bindingOf$3(buffer, size) {
@@ -4600,8 +4986,8 @@ function checkLabels(raw) {
4600
4986
  for (let v = 0; v < n; v++) {
4601
4987
  const label = raw[v];
4602
4988
  if (label >= n) {
4603
- throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM$4}: labels[${v}] = ${label} is not a node index`, {
4604
- label: `${ALGORITHM$4}/labels`,
4989
+ throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM$5}: labels[${v}] = ${label} is not a node index`, {
4990
+ label: `${ALGORITHM$5}/labels`,
4605
4991
  message: `the device produced a label outside [0, ${n})`
4606
4992
  });
4607
4993
  }
@@ -4619,7 +5005,7 @@ async function connectedComponents(ctx, s, options) {
4619
5005
  const renumber = options?.renumber !== false;
4620
5006
  const dest = checkDest$4(options?.dest, n);
4621
5007
  if (options?.signal?.aborted) {
4622
- throw new WebGpuGraphError("E_ABORTED", `${ALGORITHM$4}: the signal was aborted before any work started`, {});
5008
+ throw new WebGpuGraphError("E_ABORTED", `${ALGORITHM$5}: the signal was aborted before any work started`, {});
4623
5009
  }
4624
5010
  if (n === 0) {
4625
5011
  options?.onProgress?.(1, 1);
@@ -4636,7 +5022,7 @@ async function connectedComponents(ctx, s, options) {
4636
5022
  }
4637
5023
  const edges = ctx.residency.view(s, "edgeList");
4638
5024
  const edgeCount = edges.scalars.edgeCount[0];
4639
- const scope = algorithmScope(ctx, ALGORITHM$4, RING_SLOTS$5);
5025
+ const scope = algorithmScope(ctx, ALGORITHM$5, RING_SLOTS$6);
4640
5026
  try {
4641
5027
  const compBytes = 4 * (n + 1);
4642
5028
  const comp = scope.scratch(compBytes, "comp");
@@ -4660,12 +5046,12 @@ async function connectedComponents(ctx, s, options) {
4660
5046
  const params = wccParams({ items: n, stride: rowPlan.stride ?? n, r: 0, giant: U32_MAX$2 });
4661
5047
  compress.dispatch(pass2, compress.bind({ comp: compBinding, P: params.binding }), rowPlan, [params.offset]);
4662
5048
  };
4663
- const submit = (batch) => {
5049
+ const submit2 = (batch) => {
4664
5050
  scope.flush();
4665
5051
  return batch.submit();
4666
5052
  };
4667
5053
  queue.writeBuffer(comp, 4 * flagIndex, zero);
4668
- const setup = new CommandBatch(ctx, `${ALGORITHM$4}/setup`);
5054
+ const setup = new CommandBatch(ctx, `${ALGORITHM$5}/setup`);
4669
5055
  let pass = setup.pass("sample-rounds");
4670
5056
  const fillParams = scope.params(FILL_PARAMS, { count: n, value: 0, mode: 1, pad0: 0 });
4671
5057
  fill.dispatch(
@@ -4682,9 +5068,9 @@ async function connectedComponents(ctx, s, options) {
4682
5068
  }
4683
5069
  recordCompress(pass);
4684
5070
  setup.endPass();
4685
- await submit(setup).readback;
5071
+ await submit2(setup).readback;
4686
5072
  ctx.assertReady();
4687
- const sampler = new CommandBatch(ctx, `${ALGORITHM$4}/sample`);
5073
+ const sampler = new CommandBatch(ctx, `${ALGORITHM$5}/sample`);
4688
5074
  pass = sampler.pass("sample");
4689
5075
  const sampleParams = wccParams({ items, stride: 0, r: 0, giant: U32_MAX$2 });
4690
5076
  sample.dispatch(
@@ -4695,14 +5081,14 @@ async function connectedComponents(ctx, s, options) {
4695
5081
  );
4696
5082
  sampler.endPass();
4697
5083
  const histRequest = sampler.readback(hist, 0, 4 * items);
4698
- const histBytes = await submit(sampler).readback;
5084
+ const histBytes = await submit2(sampler).readback;
4699
5085
  ctx.assertReady();
4700
5086
  const giant = modeOf(new Uint32Array(histBytes, histRequest.offset, items));
4701
5087
  const edgeBindings = { edgeSrc: edges.bindings.src, edgeDst: edges.bindings.dst, comp: compBinding };
4702
5088
  let rounds = 0;
4703
5089
  for (; ; ) {
4704
5090
  queue.writeBuffer(comp, 4 * flagIndex, zero);
4705
- const batch = new CommandBatch(ctx, `${ALGORITHM$4}/rounds`);
5091
+ const batch = new CommandBatch(ctx, `${ALGORITHM$5}/rounds`);
4706
5092
  pass = batch.pass("edge-rounds");
4707
5093
  for (let i = 0; i < ROUNDS_PER_BATCH$1; i++) {
4708
5094
  const params = wccParams({ items: edgeCount, stride: edgePlan.stride ?? edgeCount, r: 0, giant });
@@ -4712,12 +5098,12 @@ async function connectedComponents(ctx, s, options) {
4712
5098
  }
4713
5099
  batch.endPass();
4714
5100
  const flagRequest = batch.readback(comp, 4 * flagIndex, 4);
4715
- const submitted = submit(batch);
5101
+ const submitted = submit2(batch);
4716
5102
  const back = await submitted.readback;
4717
5103
  rounds += ROUNDS_PER_BATCH$1;
4718
5104
  ctx.assertReady();
4719
5105
  if (options?.signal?.aborted) {
4720
- throw new WebGpuGraphError("E_ABORTED", `${ALGORITHM$4}: the signal was aborted`, {
5106
+ throw new WebGpuGraphError("E_ABORTED", `${ALGORITHM$5}: the signal was aborted`, {
4721
5107
  batchId: submitted.id
4722
5108
  });
4723
5109
  }
@@ -4727,15 +5113,15 @@ async function connectedComponents(ctx, s, options) {
4727
5113
  if (rounds >= MAX_WCC_ROUNDS) {
4728
5114
  throw new WebGpuGraphError(
4729
5115
  "E_VALIDATION",
4730
- `${ALGORITHM$4}: the changed flag never settled in ${MAX_WCC_ROUNDS} rounds`,
4731
- { label: ALGORITHM$4, message: `the changed flag never settled in ${MAX_WCC_ROUNDS} rounds` }
5116
+ `${ALGORITHM$5}: the changed flag never settled in ${MAX_WCC_ROUNDS} rounds`,
5117
+ { label: ALGORITHM$5, message: `the changed flag never settled in ${MAX_WCC_ROUNDS} rounds` }
4732
5118
  );
4733
5119
  }
4734
5120
  }
4735
- const final = new CommandBatch(ctx, `${ALGORITHM$4}/final`);
5121
+ const final = new CommandBatch(ctx, `${ALGORITHM$5}/final`);
4736
5122
  recordCompress(final.pass("compress"));
4737
5123
  final.endPass();
4738
- await submit(final).readback;
5124
+ await submit2(final).readback;
4739
5125
  ctx.assertReady();
4740
5126
  const raw = !renumber && dest !== null ? dest : new Uint32Array(n);
4741
5127
  await ctx.readback.read(comp, 4 * n, raw);
@@ -5098,7 +5484,7 @@ async function prepareSpmvPull(scope, rev, options) {
5098
5484
  return new SpmvPullPlannerImpl(scope, compiled, perm, options.weights);
5099
5485
  }
5100
5486
  const PR_BATCH = 8;
5101
- const RING_SLOTS$4 = 2 * PR_BATCH + 2;
5487
+ const RING_SLOTS$5 = 2 * PR_BATCH + 2;
5102
5488
  function checkDest$3(dest, n, algorithm) {
5103
5489
  if (dest === void 0) {
5104
5490
  return null;
@@ -5176,7 +5562,7 @@ async function run(ctx, s, personalization, options, algorithm) {
5176
5562
  const weights = useWeights ? void 0 : null;
5177
5563
  const weightedCore = useWeights ? core : { ...core, weights: null, hasWeights: false };
5178
5564
  const weightedRev = useWeights ? rev : { ...rev, weights: null, hasWeights: false };
5179
- const scope = algorithmScope(ctx, algorithm, RING_SLOTS$4);
5565
+ const scope = algorithmScope(ctx, algorithm, RING_SLOTS$5);
5180
5566
  let uploaded = null;
5181
5567
  try {
5182
5568
  const bytes = 4 * n;
@@ -5193,7 +5579,7 @@ async function run(ctx, s, personalization, options, algorithm) {
5193
5579
  uploaded = ctx.residency.array(personalization, `${algorithm}/personalization`);
5194
5580
  }
5195
5581
  await ctx.allocator.check();
5196
- const normaliser = await prepareSegmentedReduce(scope, weightedCore, {
5582
+ const normaliser2 = await prepareSegmentedReduce(scope, weightedCore, {
5197
5583
  op: "sum",
5198
5584
  valueSnippet: "v = weight;",
5199
5585
  tiers: null
@@ -5226,7 +5612,7 @@ async function run(ctx, s, personalization, options, algorithm) {
5226
5612
  const batch = new CommandBatch(ctx, algorithm);
5227
5613
  let pass = batch.pass("iterations");
5228
5614
  if (iterationsRun === 0) {
5229
- normaliser.record(pass, weightedCore, outWeightSumBinding);
5615
+ normaliser2.record(pass, weightedCore, outWeightSumBinding);
5230
5616
  }
5231
5617
  const last = iterationsRun + k === maxIterations;
5232
5618
  for (let i = 0; i < k + (last ? 1 : 0); i++) {
@@ -5339,7 +5725,7 @@ async function personalizedPageRank(ctx, s, personalization, options) {
5339
5725
  return run(ctx, s, normalised2, options, "personalizedPageRank");
5340
5726
  }
5341
5727
  const BATCH = 8;
5342
- const RING_SLOTS$3 = 4 * BATCH + 8;
5728
+ const RING_SLOTS$4 = 4 * BATCH + 8;
5343
5729
  function checkDest$2(dest, n, algorithm) {
5344
5730
  if (dest === void 0) {
5345
5731
  return null;
@@ -5378,7 +5764,7 @@ function whole(buffer, size) {
5378
5764
  }
5379
5765
  async function runPowerIteration(ctx, n, config) {
5380
5766
  await assertDeviceComputes(ctx);
5381
- const scope = algorithmScope(ctx, config.label, RING_SLOTS$3);
5767
+ const scope = algorithmScope(ctx, config.label, RING_SLOTS$4);
5382
5768
  try {
5383
5769
  const bytes = 4 * n;
5384
5770
  const ring = (config.alternate === null ? ["rankA", "rankB"] : ["rankA", "rankB", "rankC"]).map(
@@ -5822,7 +6208,9 @@ const W = Object.freeze({
5822
6208
  thresholdBits: 22,
5823
6209
  deltaBits: 23,
5824
6210
  path: 24,
5825
- nextDegreeSum: 25
6211
+ nextDegreeSum: 25,
6212
+ stackTop: 26,
6213
+ sigmaOverflow: 27
5826
6214
  });
5827
6215
  function definedWords(words) {
5828
6216
  const out = {};
@@ -6156,7 +6544,7 @@ class RadixSortPlannerImpl {
6156
6544
  return src;
6157
6545
  }
6158
6546
  }
6159
- const ALGORITHM$3 = "breadthFirstSearch";
6547
+ const ALGORITHM$4 = "breadthFirstSearch";
6160
6548
  const NEXT_DEGREE_MAX_GROUPS = 128;
6161
6549
  function bfsRingSlots(windows, levelsPerSubmit) {
6162
6550
  return Math.max((5 + 4 * windows) * levelsPerSubmit + 16, RESULT_BATCH_SLOTS + windows);
@@ -6177,7 +6565,7 @@ function checkDest$1(dest, n) {
6177
6565
  }
6178
6566
  throw new WebGpuGraphError(
6179
6567
  "E_INVALID_ARGUMENT",
6180
- `${ALGORITHM$3}: dest must be a Uint32Array of length ${n} over an ArrayBuffer`,
6568
+ `${ALGORITHM$4}: dest must be a Uint32Array of length ${n} over an ArrayBuffer`,
6181
6569
  {
6182
6570
  argument: "dest",
6183
6571
  value: `${dest.constructor.name}(${dest.length})`,
@@ -6192,8 +6580,8 @@ function degreeView(ctx, s, name) {
6192
6580
  const { bindings } = ctx.residency.view(s, name);
6193
6581
  const { [name]: binding } = bindings;
6194
6582
  if (binding === void 0) {
6195
- throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM$3}: the ${name} view has no ${name} binding`, {
6196
- label: `${ALGORITHM$3}/${name}`,
6583
+ throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM$4}: the ${name} view has no ${name} binding`, {
6584
+ label: `${ALGORITHM$4}/${name}`,
6197
6585
  message: `the ${name} view has no ${name} binding`
6198
6586
  });
6199
6587
  }
@@ -6202,15 +6590,15 @@ function degreeView(ctx, s, name) {
6202
6590
  function aborted$1(batchId) {
6203
6591
  return new WebGpuGraphError(
6204
6592
  "E_ABORTED",
6205
- `${ALGORITHM$3}: the signal was aborted`,
6593
+ `${ALGORITHM$4}: the signal was aborted`,
6206
6594
  batchId === void 0 ? {} : { batchId }
6207
6595
  );
6208
6596
  }
6209
6597
  function wordOf$1(block, name) {
6210
6598
  const value = block[name];
6211
6599
  if (typeof value !== "number") {
6212
- throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM$3}: counters.${name} did not decode to a number`, {
6213
- label: `${ALGORITHM$3}/counters`,
6600
+ throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM$4}: counters.${name} did not decode to a number`, {
6601
+ label: `${ALGORITHM$4}/counters`,
6214
6602
  message: `the field ${name} did not decode to a number`
6215
6603
  });
6216
6604
  }
@@ -6221,7 +6609,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6221
6609
  await assertDeviceComputes(ctx);
6222
6610
  const n = s.nodeCount;
6223
6611
  if (!Number.isInteger(source) || source < 0 || source >= n) {
6224
- throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM$3}: source ${source} is outside [0, ${n})`, {
6612
+ throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM$4}: source ${source} is outside [0, ${n})`, {
6225
6613
  argument: "source",
6226
6614
  value: source,
6227
6615
  expected: `an integer in [0, ${n})`
@@ -6231,7 +6619,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6231
6619
  if (!Number.isInteger(levelsPerSubmit) || levelsPerSubmit < 1 || levelsPerSubmit > MAX_LEVELS_PER_SUBMIT) {
6232
6620
  throw new WebGpuGraphError(
6233
6621
  "E_INVALID_ARGUMENT",
6234
- `${ALGORITHM$3}: levelsPerSubmit must be an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
6622
+ `${ALGORITHM$4}: levelsPerSubmit must be an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
6235
6623
  {
6236
6624
  argument: "levelsPerSubmit",
6237
6625
  value: levelsPerSubmit,
@@ -6250,12 +6638,12 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6250
6638
  if (s.directed && 4 * s.arcCount > ctx.caps.limits.maxStorageBufferBindingSize) {
6251
6639
  throw new WebGpuGraphError(
6252
6640
  "E_TOO_LARGE",
6253
- `${ALGORITHM$3}: the reverse adjacency of a directed snapshot (${4 * s.arcCount} bytes) needs arc windows, which no view executes (spec 4.3); the bottom-up sweep binds it whole`,
6641
+ `${ALGORITHM$4}: the reverse adjacency of a directed snapshot (${4 * s.arcCount} bytes) needs arc windows, which no view executes (spec 4.3); the bottom-up sweep binds it whole`,
6254
6642
  {
6255
6643
  needed: 4 * s.arcCount,
6256
6644
  limit: ctx.caps.limits.maxStorageBufferBindingSize,
6257
6645
  path: "windowed",
6258
- algorithm: ALGORITHM$3
6646
+ algorithm: ALGORITHM$4
6259
6647
  }
6260
6648
  );
6261
6649
  }
@@ -6263,7 +6651,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6263
6651
  const backward = coreWindows(reverse);
6264
6652
  const outDegree = degreeView(ctx, s, "outDegree");
6265
6653
  const inDegree = degreeView(ctx, s, "inDegree");
6266
- const scope = algorithmScope(ctx, ALGORITHM$3, bfsRingSlots(forward.length, levelsPerSubmit));
6654
+ const scope = algorithmScope(ctx, ALGORITHM$4, bfsRingSlots(forward.length, levelsPerSubmit));
6267
6655
  tuning.onScope?.(scope);
6268
6656
  try {
6269
6657
  const bytes = 4 * n;
@@ -6305,7 +6693,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6305
6693
  const degreePlan = planGridStride(n, wg, ctx.caps, NEXT_DEGREE_MAX_GROUPS);
6306
6694
  const fusedPlan = planGridStride(n * wg, wg, ctx.caps);
6307
6695
  const bitsPlan = plan1d(bitsWords, wg, ctx.caps);
6308
- const recordFill = (pass2, dst, value, mode) => {
6696
+ const recordFill2 = (pass2, dst, value, mode) => {
6309
6697
  const params = scope.params(FILL_PARAMS, { count: n, value, mode, pad0: 0 });
6310
6698
  fill.dispatch(pass2, fill.bind({ dst, P: params.binding }), fillPlan, [params.offset]);
6311
6699
  };
@@ -6323,16 +6711,16 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6323
6711
  outIndex: 0
6324
6712
  });
6325
6713
  };
6326
- const submit = (batch) => {
6714
+ const submit2 = (batch) => {
6327
6715
  scope.flush();
6328
6716
  return batch.submit();
6329
6717
  };
6330
- const setup = new CommandBatch(ctx, `${ALGORITHM$3}/setup`);
6718
+ const setup = new CommandBatch(ctx, `${ALGORITHM$4}/setup`);
6331
6719
  const setupPass = setup.pass("fill");
6332
- recordFill(setupPass, depth, INVALID_INDEX, 0);
6333
- recordFill(setupPass, iota, 0, 1);
6720
+ recordFill2(setupPass, depth, INVALID_INDEX, 0);
6721
+ recordFill2(setupPass, iota, 0, 1);
6334
6722
  setup.endPass();
6335
- await submit(setup).readback;
6723
+ await submit2(setup).readback;
6336
6724
  ctx.assertReady();
6337
6725
  queue.writeBuffer(depth.buffer, depth.offset + 4 * source, Uint32Array.of(0));
6338
6726
  frontier.reset(queue, source, { nextFrontierCount: 1, level: U32_MAX$2 });
@@ -6347,7 +6735,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6347
6735
  let submits = 0;
6348
6736
  for (; ; ) {
6349
6737
  queue.writeBuffer(counters.buffer, counters.offset + 4 * W.unvisitedCount, new Uint32Array(3));
6350
- const batch = new CommandBatch(ctx, `${ALGORITHM$3}/levels`);
6738
+ const batch = new CommandBatch(ctx, `${ALGORITHM$4}/levels`);
6351
6739
  const pass2 = batch.pass("bfs");
6352
6740
  recordRebuild(pass2);
6353
6741
  const bitsParams = scope.params(FILL_PARAMS, { count: bitsWords, value: 0, mode: 0, pad0: 0 });
@@ -6436,7 +6824,7 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6436
6824
  frontier: batch.readback(frontier.input.buffer, frontier.input.offset, frontier.input.size),
6437
6825
  count: batch.readback(compactCount.buffer, compactCount.offset, 4)
6438
6826
  };
6439
- const submitted = submit(batch);
6827
+ const submitted = submit2(batch);
6440
6828
  const back2 = await submitted.readback;
6441
6829
  levelsRecorded += levelsPerSubmit;
6442
6830
  submits += 1;
@@ -6457,8 +6845,8 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6457
6845
  if (submits > n + 1) {
6458
6846
  throw new WebGpuGraphError(
6459
6847
  "E_VALIDATION",
6460
- `${ALGORITHM$3}: the done flag never rose in ${submits} submits (a traversal has at most ${n} levels)`,
6461
- { label: ALGORITHM$3, message: `the done flag never rose in ${submits} submits` }
6848
+ `${ALGORITHM$4}: the done flag never rose in ${submits} submits (a traversal has at most ${n} levels)`,
6849
+ { label: ALGORITHM$4, message: `the done flag never rose in ${submits} submits` }
6462
6850
  );
6463
6851
  }
6464
6852
  }
@@ -6472,12 +6860,12 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6472
6860
  offsets: bindingOf$1(scope.scratch(histBytes, "order/offsets"), histBytes)
6473
6861
  };
6474
6862
  const parent = bindingOf$1(scope.scratch(bytes, "parent"), bytes);
6475
- const result = new CommandBatch(ctx, `${ALGORITHM$3}/result`);
6863
+ const result = new CommandBatch(ctx, `${ALGORITHM$4}/result`);
6476
6864
  result.copy(depth, keys, bytes);
6477
6865
  const pass = result.pass("result");
6478
- recordFill(pass, vals, 0, 1);
6866
+ recordFill2(pass, vals, 0, 1);
6479
6867
  const sorted = sort.record(pass, keys, vals, n, 32, scratch);
6480
- recordFill(pass, parent, INVALID_INDEX, 0);
6868
+ recordFill2(pass, parent, INVALID_INDEX, 0);
6481
6869
  const predPlan = planGridStride(n, wg, ctx.caps);
6482
6870
  for (const w of forward) {
6483
6871
  const predParams = scope.params(FRONTIER_PARAMS, {
@@ -6502,16 +6890,16 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6502
6890
  const parentRequest = result.readback(parent.buffer, parent.offset, bytes);
6503
6891
  const orderRequest = result.readback(sorted.vals.buffer, sorted.vals.offset, bytes);
6504
6892
  const blockRequest = result.readback(counters.buffer, counters.offset, FRONTIER_COUNTERS.byteLength);
6505
- const back = await submit(result).readback;
6893
+ const back = await submit2(result).readback;
6506
6894
  ctx.assertReady();
6507
6895
  const block = FRONTIER_COUNTERS.read(new DataView(back), blockRequest.offset);
6508
6896
  const visitedCount = wordOf$1(block, "visitedCount");
6509
6897
  if (visitedCount > n) {
6510
6898
  throw new WebGpuGraphError(
6511
6899
  "E_VALIDATION",
6512
- `${ALGORITHM$3}: visitedCount ${visitedCount} exceeds the ${n} vertices (a duplicate claim)`,
6900
+ `${ALGORITHM$4}: visitedCount ${visitedCount} exceeds the ${n} vertices (a duplicate claim)`,
6513
6901
  {
6514
- label: `${ALGORITHM$3}/visitedCount`,
6902
+ label: `${ALGORITHM$4}/visitedCount`,
6515
6903
  message: `the device counted ${visitedCount} visits of ${n} vertices`
6516
6904
  }
6517
6905
  );
@@ -6535,9 +6923,9 @@ async function bfsWithTuning(ctx, s, source, options, tuning) {
6535
6923
  function breadthFirstSearch(ctx, s, source, options) {
6536
6924
  return bfsWithTuning(ctx, s, source, options, {});
6537
6925
  }
6538
- const ALGORITHM$2 = "sssp";
6926
+ const ALGORITHM$3 = "sssp";
6539
6927
  const HALF_ALIGN = 64;
6540
- const RING_SLOTS$2 = 4 * MAX_LEVELS_PER_SUBMIT + 40;
6928
+ const RING_SLOTS$3 = 4 * MAX_LEVELS_PER_SUBMIT + 40;
6541
6929
  function bitsOf(value) {
6542
6930
  return new Uint32Array(Float32Array.of(value).buffer)[0];
6543
6931
  }
@@ -6563,8 +6951,8 @@ function assertSource(algorithm, source, n) {
6563
6951
  function wordOf(block, name) {
6564
6952
  const value = block[name];
6565
6953
  if (typeof value !== "number") {
6566
- throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM$2}: counters.${name} did not decode to a number`, {
6567
- label: `${ALGORITHM$2}/counters`,
6954
+ throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM$3}: counters.${name} did not decode to a number`, {
6955
+ label: `${ALGORITHM$3}/counters`,
6568
6956
  message: `the field ${name} did not decode to a number`
6569
6957
  });
6570
6958
  }
@@ -6664,7 +7052,7 @@ function predBufferWords(n) {
6664
7052
  return 2 * Math.ceil(n / 64) * 64 + 64;
6665
7053
  }
6666
7054
  async function predecessorPass(input) {
6667
- const { algorithm, ctx, scope, predKernel, recordFill, graph, dist, pred, n, arcCount, source, mode } = input;
7055
+ const { algorithm, ctx, scope, predKernel, recordFill: recordFill2, graph, dist, pred, n, arcCount, source, mode } = input;
6668
7056
  const wg = ctx.workgroupSize;
6669
7057
  const { queue } = ctx.device;
6670
7058
  const bytes = 4 * n;
@@ -6699,7 +7087,7 @@ async function predecessorPass(input) {
6699
7087
  for (let iteration = 0; iteration < MAX_LEVELS_PER_SUBMIT; iteration++) {
6700
7088
  recordRole(1, iteration);
6701
7089
  }
6702
- recordFill(pass, predArcs, n, INVALID_INDEX);
7090
+ recordFill2(pass, predArcs, n, INVALID_INDEX);
6703
7091
  recordRole(2, 0);
6704
7092
  batch.endPass();
6705
7093
  const distRequest = batch.readback(dist.buffer, dist.offset, bytes);
@@ -6731,12 +7119,12 @@ async function ssspWithTuning(ctx, s, source, options, tuning) {
6731
7119
  ctx.assertReady();
6732
7120
  await assertDeviceComputes(ctx);
6733
7121
  const n = s.nodeCount;
6734
- assertSource(ALGORITHM$2, source, n);
7122
+ assertSource(ALGORITHM$3, source, n);
6735
7123
  const roundsPerSubmit = tuning.roundsPerSubmit ?? MAX_LEVELS_PER_SUBMIT;
6736
7124
  if (!Number.isInteger(roundsPerSubmit) || roundsPerSubmit < 1 || roundsPerSubmit > MAX_LEVELS_PER_SUBMIT) {
6737
7125
  throw new WebGpuGraphError(
6738
7126
  "E_INVALID_ARGUMENT",
6739
- `${ALGORITHM$2}: roundsPerSubmit must be an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
7127
+ `${ALGORITHM$3}: roundsPerSubmit must be an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
6740
7128
  {
6741
7129
  argument: "roundsPerSubmit",
6742
7130
  value: roundsPerSubmit,
@@ -6745,42 +7133,42 @@ async function ssspWithTuning(ctx, s, source, options, tuning) {
6745
7133
  );
6746
7134
  }
6747
7135
  if (tuning.delta !== void 0 && !(Number.isFinite(tuning.delta) && tuning.delta > 0)) {
6748
- throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM$2}: delta must be a finite positive number`, {
7136
+ throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM$3}: delta must be a finite positive number`, {
6749
7137
  argument: "delta",
6750
7138
  value: tuning.delta,
6751
7139
  expected: "a finite positive number"
6752
7140
  });
6753
7141
  }
6754
- const dest = checkDest(ALGORITHM$2, options?.dest, n);
6755
- const vector2 = resolveWeights$1(ALGORITHM$2, s, options?.weights);
6756
- const cutoff = normaliseCutoff(ALGORITHM$2, options?.cutoff);
7142
+ const dest = checkDest(ALGORITHM$3, options?.dest, n);
7143
+ const vector2 = resolveWeights$1(ALGORITHM$3, s, options?.weights);
7144
+ const cutoff = normaliseCutoff(ALGORITHM$3, options?.cutoff);
6757
7145
  if (options?.signal?.aborted) {
6758
- throw aborted(ALGORITHM$2);
7146
+ throw aborted(ALGORITHM$3);
6759
7147
  }
6760
7148
  if (vector2 === null || vector2.allOne) {
6761
7149
  return unitWeightRoute(ctx, s, source, cutoff, dest, options);
6762
7150
  }
6763
7151
  if (!vector2.nonNegative) {
6764
- throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM$2}: a negative weight has no shortest path here`, {
7152
+ throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM$3}: a negative weight has no shortest path here`, {
6765
7153
  feature: "sssp.negativeWeights",
6766
7154
  hint: "use bellmanFord"
6767
7155
  });
6768
7156
  }
6769
7157
  if (!vector2.finite) {
6770
- throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM$2}: a NaN or infinite weight has no bit-pattern order`, {
7158
+ throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM$3}: a NaN or infinite weight has no bit-pattern order`, {
6771
7159
  feature: "sssp.nonFiniteWeights"
6772
7160
  });
6773
7161
  }
6774
7162
  const { arcCount } = s;
6775
7163
  const core = ctx.residency.core(s);
6776
7164
  const limit = ctx.caps.limits.maxStorageBufferBindingSize;
6777
- assertWholeCore(core, arcCount, limit, ALGORITHM$2);
7165
+ assertWholeCore(core, arcCount, limit, ALGORITHM$3);
6778
7166
  const cap = Math.ceil(Math.max(1, arcCount) / HALF_ALIGN) * HALF_ALIGN;
6779
7167
  if (8 * cap > limit) {
6780
7168
  throw new WebGpuGraphError(
6781
7169
  "E_TOO_LARGE",
6782
- `${ALGORITHM$2}: the near-far queue of ${cap} entries per half needs ${8 * cap} bytes, above the ${limit}-byte binding limit (the relax is never windowed)`,
6783
- { needed: 8 * cap, limit, path: "sssp.queue", algorithm: ALGORITHM$2 }
7170
+ `${ALGORITHM$3}: the near-far queue of ${cap} entries per half needs ${8 * cap} bytes, above the ${limit}-byte binding limit (the relax is never windowed)`,
7171
+ { needed: 8 * cap, limit, path: "sssp.queue", algorithm: ALGORITHM$3 }
6784
7172
  );
6785
7173
  }
6786
7174
  const delta = Math.fround(
@@ -6789,7 +7177,7 @@ async function ssspWithTuning(ctx, s, source, options, tuning) {
6789
7177
  const deltaBits = bitsOf(delta);
6790
7178
  const maxRounds = n + Math.ceil(vector2.sum / delta) + 1;
6791
7179
  const maxSubmits = Math.ceil((maxRounds + 1) / roundsPerSubmit) + 1;
6792
- const scope = algorithmScope(ctx, ALGORITHM$2, RING_SLOTS$2);
7180
+ const scope = algorithmScope(ctx, ALGORITHM$3, RING_SLOTS$3);
6793
7181
  try {
6794
7182
  const wg = ctx.workgroupSize;
6795
7183
  const bytes = 4 * n;
@@ -6819,20 +7207,20 @@ async function ssspWithTuning(ctx, s, source, options, tuning) {
6819
7207
  const { counters } = frontier;
6820
7208
  const nearIn = frontier.vertices[0];
6821
7209
  const farIn = frontier.vertices[1];
6822
- const recordFill = (pass, dst, count, value) => {
7210
+ const recordFill2 = (pass, dst, count, value) => {
6823
7211
  const params = scope.params(FILL_PARAMS, { count, value, mode: 0, pad0: 0 });
6824
7212
  fill.dispatch(pass, fill.bind({ dst, P: params.binding }), plan1d(count, wg, ctx.caps), [params.offset]);
6825
7213
  };
6826
- const submit = (batch) => {
7214
+ const submit2 = (batch) => {
6827
7215
  scope.flush();
6828
7216
  return batch.submit();
6829
7217
  };
6830
- const setup = new CommandBatch(ctx, `${ALGORITHM$2}/setup`);
7218
+ const setup = new CommandBatch(ctx, `${ALGORITHM$3}/setup`);
6831
7219
  const setupPass = setup.pass("fill");
6832
- recordFill(setupPass, dist, n, F32_INF_BITS);
6833
- recordFill(setupPass, pred, predWords, INVALID_INDEX);
7220
+ recordFill2(setupPass, dist, n, F32_INF_BITS);
7221
+ recordFill2(setupPass, pred, predWords, INVALID_INDEX);
6834
7222
  setup.endPass();
6835
- await submit(setup).readback;
7223
+ await submit2(setup).readback;
6836
7224
  ctx.assertReady();
6837
7225
  queue.writeBuffer(dist.buffer, dist.offset + 4 * source, Uint32Array.of(0));
6838
7226
  frontier.reset(queue, source, { nextFrontierCount: 1, thresholdBits: deltaBits, deltaBits });
@@ -6873,7 +7261,7 @@ async function ssspWithTuning(ctx, s, source, options, tuning) {
6873
7261
  let roundsRecorded = 0;
6874
7262
  let submits = 0;
6875
7263
  for (; ; ) {
6876
- const batch = new CommandBatch(ctx, `${ALGORITHM$2}/rounds`);
7264
+ const batch = new CommandBatch(ctx, `${ALGORITHM$3}/rounds`);
6877
7265
  const pass = batch.pass("sssp");
6878
7266
  const near = scope.params(FRONTIER_PARAMS, { ...relaxFields, role: 0 });
6879
7267
  const far = scope.params(FRONTIER_PARAMS, { ...relaxFields, role: 1 });
@@ -6890,13 +7278,13 @@ async function ssspWithTuning(ctx, s, source, options, tuning) {
6890
7278
  batch.endPass();
6891
7279
  const doneRequest = batch.readback(counters.buffer, counters.offset + 4 * W.done, 4);
6892
7280
  const inspect = tuning.onRound === void 0 ? null : batch.readback(counters.buffer, counters.offset, FRONTIER_COUNTERS.byteLength);
6893
- const submitted = submit(batch);
7281
+ const submitted = submit2(batch);
6894
7282
  const back = await submitted.readback;
6895
7283
  roundsRecorded += roundsPerSubmit;
6896
7284
  submits += 1;
6897
7285
  ctx.assertReady();
6898
7286
  if (options?.signal?.aborted) {
6899
- throw aborted(ALGORITHM$2, submitted.id);
7287
+ throw aborted(ALGORITHM$3, submitted.id);
6900
7288
  }
6901
7289
  options?.onProgress?.(Math.min(roundsRecorded, maxRounds), maxRounds);
6902
7290
  if (inspect !== null && tuning.onRound !== void 0) {
@@ -6921,30 +7309,30 @@ async function ssspWithTuning(ctx, s, source, options, tuning) {
6921
7309
  const needed = Math.max(wordOf(block, "nextFrontierCount"), wordOf(block, "nextFarCount"));
6922
7310
  throw new WebGpuGraphError(
6923
7311
  "E_TOO_LARGE",
6924
- `${ALGORITHM$2}: a raw pile of ${needed} entries overflowed its ${cap}-entry half`,
6925
- { needed, limit: cap, path: "sssp.pile", algorithm: ALGORITHM$2 }
7312
+ `${ALGORITHM$3}: a raw pile of ${needed} entries overflowed its ${cap}-entry half`,
7313
+ { needed, limit: cap, path: "sssp.pile", algorithm: ALGORITHM$3 }
6926
7314
  );
6927
7315
  }
6928
7316
  throw new WebGpuGraphError(
6929
7317
  "E_UNSUPPORTED",
6930
- `${ALGORITHM$2}: the f32 threshold ${wordOf(block, "thresholdBits")} absorbed the delta ${deltaBits} (as bit patterns); the far pile can no longer be bucketed`,
7318
+ `${ALGORITHM$3}: the f32 threshold ${wordOf(block, "thresholdBits")} absorbed the delta ${deltaBits} (as bit patterns); the far pile can no longer be bucketed`,
6931
7319
  { feature: "sssp.thresholdAbsorbed", hint: "the distances outgrew the delta's f32 precision" }
6932
7320
  );
6933
7321
  }
6934
7322
  if (submits > maxSubmits) {
6935
7323
  throw new WebGpuGraphError(
6936
7324
  "E_VALIDATION",
6937
- `${ALGORITHM$2}: the done flag never rose in ${submits} submits (at most ${maxRounds} rounds)`,
6938
- { label: `${ALGORITHM$2}/rounds`, message: `the done flag never rose in ${submits} submits` }
7325
+ `${ALGORITHM$3}: the done flag never rose in ${submits} submits (at most ${maxRounds} rounds)`,
7326
+ { label: `${ALGORITHM$3}/rounds`, message: `the done flag never rose in ${submits} submits` }
6939
7327
  );
6940
7328
  }
6941
7329
  }
6942
7330
  const passed = await predecessorPass({
6943
- algorithm: ALGORITHM$2,
7331
+ algorithm: ALGORITHM$3,
6944
7332
  ctx,
6945
7333
  scope,
6946
7334
  predKernel,
6947
- recordFill,
7335
+ recordFill: recordFill2,
6948
7336
  graph,
6949
7337
  dist,
6950
7338
  pred,
@@ -6956,8 +7344,8 @@ async function ssspWithTuning(ctx, s, source, options, tuning) {
6956
7344
  if (passed.orphans !== 0) {
6957
7345
  throw new WebGpuGraphError(
6958
7346
  "E_VALIDATION",
6959
- `${ALGORITHM$2}: ${passed.orphans} reached node(s) the predecessor key never reached (a kernel bug)`,
6960
- { label: `${ALGORITHM$2}/pred`, message: `${passed.orphans} orphan(s) in the predecessor pass` }
7347
+ `${ALGORITHM$3}: ${passed.orphans} reached node(s) the predecessor key never reached (a kernel bug)`,
7348
+ { label: `${ALGORITHM$3}/pred`, message: `${passed.orphans} orphan(s) in the predecessor pass` }
6961
7349
  );
6962
7350
  }
6963
7351
  const distOut = dest ?? new Float32Array(n);
@@ -6976,10 +7364,10 @@ async function ssspWithTuning(ctx, s, source, options, tuning) {
6976
7364
  function sssp(ctx, s, source, options) {
6977
7365
  return ssspWithTuning(ctx, s, source, options, {});
6978
7366
  }
6979
- const ALGORITHM$1 = "bellmanFord";
7367
+ const ALGORITHM$2 = "bellmanFord";
6980
7368
  const ROUNDS_PER_BATCH = 8;
6981
7369
  const MAX_RETRIES = 16;
6982
- const RING_SLOTS$1 = MAX_LEVELS_PER_SUBMIT + 16;
7370
+ const RING_SLOTS$2 = MAX_LEVELS_PER_SUBMIT + 16;
6983
7371
  function assertSymmetric(s, vector2) {
6984
7372
  const { arcToEdge, edgeToArc } = s;
6985
7373
  for (let a = 0; a < s.arcCount; a++) {
@@ -6987,7 +7375,7 @@ function assertSymmetric(s, vector2) {
6987
7375
  if (vector2[a] !== vector2[forward]) {
6988
7376
  throw new WebGpuGraphError(
6989
7377
  "E_UNSUPPORTED",
6990
- `${ALGORITHM$1}: weights[${a}] = ${vector2[a]} differs from weights[${forward}] = ${vector2[forward]}, the forward arc of the same undirected edge; the kernel reads one weight per edge`,
7378
+ `${ALGORITHM$2}: weights[${a}] = ${vector2[a]} differs from weights[${forward}] = ${vector2[forward]}, the forward arc of the same undirected edge; the kernel reads one weight per edge`,
6991
7379
  { feature: "bellmanFord.asymmetricUndirectedWeights", hint: "use a directed snapshot" }
6992
7380
  );
6993
7381
  }
@@ -6997,10 +7385,10 @@ async function bellmanFordWithTuning(ctx, s, source, options, tuning) {
6997
7385
  ctx.assertReady();
6998
7386
  await assertDeviceComputes(ctx);
6999
7387
  const n = s.nodeCount;
7000
- assertSource(ALGORITHM$1, source, n);
7388
+ assertSource(ALGORITHM$2, source, n);
7001
7389
  const maxRetries = tuning.maxRetries ?? MAX_RETRIES;
7002
7390
  if (!Number.isInteger(maxRetries) || maxRetries < 1) {
7003
- throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM$1}: maxRetries must be an integer >= 1`, {
7391
+ throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM$2}: maxRetries must be an integer >= 1`, {
7004
7392
  argument: "maxRetries",
7005
7393
  value: maxRetries,
7006
7394
  expected: "an integer >= 1"
@@ -7010,7 +7398,7 @@ async function bellmanFordWithTuning(ctx, s, source, options, tuning) {
7010
7398
  if (!Number.isInteger(roundsPerBatch) || roundsPerBatch < 1 || roundsPerBatch > MAX_LEVELS_PER_SUBMIT) {
7011
7399
  throw new WebGpuGraphError(
7012
7400
  "E_INVALID_ARGUMENT",
7013
- `${ALGORITHM$1}: roundsPerBatch must be an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
7401
+ `${ALGORITHM$2}: roundsPerBatch must be an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`,
7014
7402
  {
7015
7403
  argument: "roundsPerBatch",
7016
7404
  value: roundsPerBatch,
@@ -7018,18 +7406,18 @@ async function bellmanFordWithTuning(ctx, s, source, options, tuning) {
7018
7406
  }
7019
7407
  );
7020
7408
  }
7021
- const dest = checkDest(ALGORITHM$1, options?.dest, n);
7022
- const vector2 = resolveWeights$1(ALGORITHM$1, s, options?.weights);
7023
- const cutoff = normaliseCutoff(ALGORITHM$1, options?.cutoff);
7409
+ const dest = checkDest(ALGORITHM$2, options?.dest, n);
7410
+ const vector2 = resolveWeights$1(ALGORITHM$2, s, options?.weights);
7411
+ const cutoff = normaliseCutoff(ALGORITHM$2, options?.cutoff);
7024
7412
  if (options?.signal?.aborted) {
7025
- throw aborted(ALGORITHM$1);
7413
+ throw aborted(ALGORITHM$2);
7026
7414
  }
7027
7415
  if (vector2 === null || vector2.allOne) {
7028
7416
  const unit = await unitWeightRoute(ctx, s, source, cutoff, dest, options);
7029
7417
  return { result: { ...unit, hasNegativeCycle: false }, rounds: 0, retryExhaustedRounds: 0 };
7030
7418
  }
7031
7419
  if (!vector2.finite) {
7032
- throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM$1}: a NaN or infinite weight has no shortest path`, {
7420
+ throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM$2}: a NaN or infinite weight has no shortest path`, {
7033
7421
  feature: "bellmanFord.nonFiniteWeights"
7034
7422
  });
7035
7423
  }
@@ -7038,10 +7426,10 @@ async function bellmanFordWithTuning(ctx, s, source, options, tuning) {
7038
7426
  }
7039
7427
  const { arcCount } = s;
7040
7428
  const core = ctx.residency.core(s, ["rowPtr", "colIdx", "weights", "edgeToArc"]);
7041
- assertWholeCore(core, arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM$1);
7429
+ assertWholeCore(core, arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM$2);
7042
7430
  const edges = ctx.residency.view(s, "edgeList");
7043
7431
  const edgeCount = edges.scalars.edgeCount[0];
7044
- const scope = algorithmScope(ctx, ALGORITHM$1, RING_SLOTS$1);
7432
+ const scope = algorithmScope(ctx, ALGORITHM$2, RING_SLOTS$2);
7045
7433
  try {
7046
7434
  const wg = ctx.workgroupSize;
7047
7435
  const bytes = 4 * n;
@@ -7068,23 +7456,23 @@ async function bellmanFordWithTuning(ctx, s, source, options, tuning) {
7068
7456
  const predKernel = await ctx.pipelines.kernel(kernelSpec("sssp-pred", { ...overrides, MODE: 0 }));
7069
7457
  const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
7070
7458
  const graph = graphBindings(core, null, weightsBinding);
7071
- const recordFill = (pass, dst, count, value, mode = 0) => {
7459
+ const recordFill2 = (pass, dst, count, value, mode = 0) => {
7072
7460
  const params = scope.params(FILL_PARAMS, { count, value, mode, pad0: 0 });
7073
7461
  fill.dispatch(pass, fill.bind({ dst, P: params.binding }), plan1d(count, wg, ctx.caps), [params.offset]);
7074
7462
  };
7075
- const submit = (batch) => {
7463
+ const submit2 = (batch) => {
7076
7464
  scope.flush();
7077
7465
  return batch.submit();
7078
7466
  };
7079
- const setup = new CommandBatch(ctx, `${ALGORITHM$1}/setup`);
7467
+ const setup = new CommandBatch(ctx, `${ALGORITHM$2}/setup`);
7080
7468
  const setupPass = setup.pass("fill");
7081
- recordFill(setupPass, dist, n, F32_INF_BITS);
7082
- recordFill(setupPass, pred, predWords, INVALID_INDEX);
7469
+ recordFill2(setupPass, dist, n, F32_INF_BITS);
7470
+ recordFill2(setupPass, pred, predWords, INVALID_INDEX);
7083
7471
  if (iota !== null) {
7084
- recordFill(setupPass, iota, edgeCount, 0, 1);
7472
+ recordFill2(setupPass, iota, edgeCount, 0, 1);
7085
7473
  }
7086
7474
  setup.endPass();
7087
- await submit(setup).readback;
7475
+ await submit2(setup).readback;
7088
7476
  ctx.assertReady();
7089
7477
  queue.writeBuffer(dist.buffer, dist.offset + 4 * source, Uint32Array.of(0));
7090
7478
  const edgePlan = planGridStride(edgeCount, wg, ctx.caps);
@@ -7105,7 +7493,7 @@ async function bellmanFordWithTuning(ctx, s, source, options, tuning) {
7105
7493
  const zero = new Uint32Array(BF_FLAGS.byteLength / 4);
7106
7494
  const runRounds = async (count, label) => {
7107
7495
  queue.writeBuffer(flags.buffer, flags.offset, zero);
7108
- const batch = new CommandBatch(ctx, `${ALGORITHM$1}/${label}`);
7496
+ const batch = new CommandBatch(ctx, `${ALGORITHM$2}/${label}`);
7109
7497
  const pass = batch.pass("relax");
7110
7498
  const params = scope.params(BF_PARAMS, relaxFields);
7111
7499
  const bound = relax.bind({ ...relaxBindings, P: params.binding });
@@ -7114,7 +7502,7 @@ async function bellmanFordWithTuning(ctx, s, source, options, tuning) {
7114
7502
  }
7115
7503
  batch.endPass();
7116
7504
  const request = batch.readback(flags.buffer, flags.offset, BF_FLAGS.byteLength);
7117
- const submitted = submit(batch);
7505
+ const submitted = submit2(batch);
7118
7506
  const back = await submitted.readback;
7119
7507
  ctx.assertReady();
7120
7508
  const block = BF_FLAGS.read(new DataView(back), request.offset);
@@ -7130,9 +7518,9 @@ async function bellmanFordWithTuning(ctx, s, source, options, tuning) {
7130
7518
  if (decision.retryExhausted !== 0) {
7131
7519
  throw new WebGpuGraphError(
7132
7520
  "E_VALIDATION",
7133
- `${ALGORITHM$1}: a lane exhausted the ${maxRetries}-retry compare-exchange bound in the decision round, so its change is not a verdict`,
7521
+ `${ALGORITHM$2}: a lane exhausted the ${maxRetries}-retry compare-exchange bound in the decision round, so its change is not a verdict`,
7134
7522
  {
7135
- label: `${ALGORITHM$1}/retry`,
7523
+ label: `${ALGORITHM$2}/retry`,
7136
7524
  message: "retryExhausted in the decision round",
7137
7525
  batchId: decision.id
7138
7526
  }
@@ -7148,7 +7536,7 @@ async function bellmanFordWithTuning(ctx, s, source, options, tuning) {
7148
7536
  retryExhaustedRounds += 1;
7149
7537
  }
7150
7538
  if (options?.signal?.aborted) {
7151
- throw aborted(ALGORITHM$1, batch.id);
7539
+ throw aborted(ALGORITHM$2, batch.id);
7152
7540
  }
7153
7541
  options?.onProgress?.(rounds, n);
7154
7542
  if (batch.changed === 0 && batch.retryExhausted === 0) {
@@ -7156,11 +7544,11 @@ async function bellmanFordWithTuning(ctx, s, source, options, tuning) {
7156
7544
  }
7157
7545
  }
7158
7546
  const passed = await predecessorPass({
7159
- algorithm: ALGORITHM$1,
7547
+ algorithm: ALGORITHM$2,
7160
7548
  ctx,
7161
7549
  scope,
7162
7550
  predKernel,
7163
- recordFill,
7551
+ recordFill: recordFill2,
7164
7552
  graph,
7165
7553
  dist,
7166
7554
  pred,
@@ -7172,7 +7560,7 @@ async function bellmanFordWithTuning(ctx, s, source, options, tuning) {
7172
7560
  if (passed.orphans !== 0 && !hasNegativeCycle) {
7173
7561
  throw new WebGpuGraphError(
7174
7562
  "E_UNSUPPORTED",
7175
- `${ALGORITHM$1}: ${passed.orphans} reached node(s) the tight subgraph never reaches (a cycle of weights below one f32 ulp relaxed once)`,
7563
+ `${ALGORITHM$2}: ${passed.orphans} reached node(s) the tight subgraph never reaches (a cycle of weights below one f32 ulp relaxed once)`,
7176
7564
  {
7177
7565
  feature: "bellmanFord.roundedCycle",
7178
7566
  hint: "a cycle of weights below one f32 ulp relaxed once at a distance above 2^24; scale the weights or shorten the distances"
@@ -7199,6 +7587,381 @@ async function bellmanFordWithTuning(ctx, s, source, options, tuning) {
7199
7587
  async function bellmanFord(ctx, s, source, options) {
7200
7588
  return (await bellmanFordWithTuning(ctx, s, source, options, {})).result;
7201
7589
  }
7590
+ const ALGORITHM$1 = "betweennessCentrality";
7591
+ const BYTES_PER_NODE_SOURCE = 16;
7592
+ const SAMPLE_SEED = 2654435769;
7593
+ const RING_SLOTS$1 = BC_BACKWARD_LEVELS_PER_SUBMIT + 16;
7594
+ function planBatchSize(n, remaining, limits) {
7595
+ const needed = 4 * (n + 2);
7596
+ if (needed > limits.maxStorageBufferBindingSize) {
7597
+ throw new WebGpuGraphError(
7598
+ "E_TOO_LARGE",
7599
+ `${ALGORITHM$1}: one source needs ${needed} bytes in its largest binding at n = ${n}, above maxStorageBufferBindingSize = ${limits.maxStorageBufferBindingSize}; a limit of at least ${needed} admits one source per batch`,
7600
+ { needed, limit: limits.maxStorageBufferBindingSize, path: "binding", algorithm: ALGORITHM$1 }
7601
+ );
7602
+ }
7603
+ const kByBinding = Math.floor(limits.maxStorageBufferBindingSize / (4 * n));
7604
+ const kByBudget = Math.floor(BC_BATCH_BUDGET_FRACTION * limits.maxBufferSize / (BYTES_PER_NODE_SOURCE * n));
7605
+ return Math.max(1, Math.min(kByBinding, kByBudget, BC_MAX_BATCH, remaining));
7606
+ }
7607
+ function drawSources(n, k) {
7608
+ const pool = Array.from({ length: n }, (_, i) => i);
7609
+ let state = SAMPLE_SEED;
7610
+ for (let i = 0; i < k; i++) {
7611
+ state = state + 1831565813 >>> 0;
7612
+ let t = Math.imul(state ^ state >>> 15, state | 1);
7613
+ t = t + Math.imul(t ^ t >>> 7, t | 61) ^ t;
7614
+ const unit = ((t ^ t >>> 14) >>> 0) / 2 ** 32;
7615
+ const j = i + Math.floor(unit * (n - i));
7616
+ [pool[i], pool[j]] = [pool[j], pool[i]];
7617
+ }
7618
+ return pool.slice(0, k);
7619
+ }
7620
+ function badArgument(argument, value, expected) {
7621
+ return new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM$1}: ${argument} must be ${expected}`, {
7622
+ argument,
7623
+ value,
7624
+ expected
7625
+ });
7626
+ }
7627
+ function resolveSources(options, n) {
7628
+ const k = options?.k;
7629
+ if (k !== void 0 && (!Number.isInteger(k) || k < 0 || k > n)) {
7630
+ throw badArgument("k", k, `an integer in [0, ${n}]`);
7631
+ }
7632
+ const given = options?.sources;
7633
+ if (given !== void 0) {
7634
+ for (const v of given) {
7635
+ if (!Number.isInteger(v) || v < 0 || v >= n) {
7636
+ throw badArgument("sources", v, `node indices in [0, ${n})`);
7637
+ }
7638
+ }
7639
+ if (k !== void 0 && k !== given.length) {
7640
+ throw badArgument("k", k, `absent or equal to sources.length (${given.length})`);
7641
+ }
7642
+ return [...given];
7643
+ }
7644
+ if (k !== void 0) {
7645
+ return drawSources(n, k);
7646
+ }
7647
+ return Array.from({ length: n }, (_, i) => i);
7648
+ }
7649
+ function recordFill(state, pass, dst, count, value) {
7650
+ const { scope, fill, ctx } = state;
7651
+ const params = scope.params(FILL_PARAMS, { count, value, mode: 0, pad0: 0 });
7652
+ fill.dispatch(pass, fill.bind({ dst, P: params.binding }), plan1d(count, ctx.workgroupSize, ctx.caps), [
7653
+ params.offset
7654
+ ]);
7655
+ }
7656
+ function recordBc(state, pass, kernel, resources, fields, items) {
7657
+ const { scope, ctx } = state;
7658
+ const plan = planGridStride(items, ctx.workgroupSize, ctx.caps);
7659
+ const params = scope.params(BC_PARAMS, { ...fields, stride: plan.stride ?? 0 });
7660
+ let bound = state.bound.get(kernel);
7661
+ if (bound === void 0) {
7662
+ bound = kernel.bind({ ...resources, P: params.binding });
7663
+ state.bound.set(kernel, bound);
7664
+ }
7665
+ kernel.dispatch(pass, bound, items === 0 ? plan1d(1, ctx.workgroupSize, ctx.caps) : plan, [params.offset]);
7666
+ }
7667
+ async function submit(state, batch, signal) {
7668
+ state.scope.flush();
7669
+ const submitted = batch.submit();
7670
+ const back = await submitted.readback;
7671
+ state.ctx.assertReady();
7672
+ if (signal?.aborted) {
7673
+ throw aborted(ALGORITHM$1, submitted.id);
7674
+ }
7675
+ return back;
7676
+ }
7677
+ async function runBatch(state, sources, form, levelsPerSubmit, tuning, signal) {
7678
+ const { ctx, n, S, ends, depthK, sigmaK, deltaK, counters } = state;
7679
+ const k = sources.length;
7680
+ const words = n * k;
7681
+ const seeds = Uint32Array.from(sources, (v, s) => s * n + v);
7682
+ ctx.device.queue.writeBuffer(S.buffer, S.offset, seeds);
7683
+ const forwardFields = { n, k, count: form === "edge" ? state.edgeCount : 0 };
7684
+ const forwardItems = form === "edge" ? state.edgeCount : words;
7685
+ let recorded = 0;
7686
+ let levels = 0;
7687
+ let overflow = false;
7688
+ let endsWords = null;
7689
+ for (let first = true; endsWords === null; first = false) {
7690
+ const batch = new CommandBatch(ctx, `${ALGORITHM$1}/forward`);
7691
+ const pass = batch.pass("forward");
7692
+ if (first) {
7693
+ recordFill(state, pass, depthK, words, 4294967295);
7694
+ recordFill(state, pass, sigmaK, words, 0);
7695
+ recordFill(state, pass, deltaK, words, 0);
7696
+ recordBc(state, pass, state.finalize, { counters, ends, S, depthK, sigmaK }, { n, k, role: 1 }, 1);
7697
+ }
7698
+ for (let level = 0; level < levelsPerSubmit; level++) {
7699
+ recordBc(state, pass, state.finalize, { counters, ends, S, depthK, sigmaK }, { n, k, role: 0 }, 1);
7700
+ if (form === "edge" && state.forwardEdge !== null && state.edgeSrc !== null && state.edgeDst !== null) {
7701
+ recordBc(
7702
+ state,
7703
+ pass,
7704
+ state.forwardEdge,
7705
+ { edgeSrc: state.edgeSrc, edgeDst: state.edgeDst, S, ends, counters, depthK, sigmaK },
7706
+ forwardFields,
7707
+ forwardItems
7708
+ );
7709
+ } else {
7710
+ recordBc(
7711
+ state,
7712
+ pass,
7713
+ state.forward,
7714
+ { rowPtr: state.rowPtr, colIdx: state.colIdx, S, ends, counters, depthK, sigmaK },
7715
+ forwardFields,
7716
+ forwardItems
7717
+ );
7718
+ }
7719
+ }
7720
+ recorded += levelsPerSubmit;
7721
+ batch.endPass();
7722
+ const endsCount = Math.min(n + 2, recorded + 1);
7723
+ const countersRequest = batch.readback(counters.buffer, counters.offset, FRONTIER_COUNTERS.byteLength);
7724
+ const endsRequest = batch.readback(ends.buffer, ends.offset, 4 * endsCount);
7725
+ const back = await submit(state, batch, signal);
7726
+ const words32 = new Uint32Array(back, countersRequest.offset, FRONTIER_COUNTERS.byteLength / 4);
7727
+ if (words32[W.done] !== 0) {
7728
+ levels = words32[W.level];
7729
+ overflow = words32[W.sigmaOverflow] !== 0;
7730
+ endsWords = new Uint32Array(back, endsRequest.offset, endsCount).slice(0, levels + 1);
7731
+ } else if (recorded > n + 2) {
7732
+ throw new WebGpuGraphError("E_VALIDATION", `${ALGORITHM$1}: the done flag never rose in ${recorded} levels`, {
7733
+ label: ALGORITHM$1,
7734
+ message: `the done flag never rose in ${recorded} levels`
7735
+ });
7736
+ }
7737
+ }
7738
+ const backwardLevels = [];
7739
+ for (let level = levels - 1; level >= 1; level--) {
7740
+ backwardLevels.push(level);
7741
+ }
7742
+ let arrays = null;
7743
+ for (let i = 0; ; i += BC_BACKWARD_LEVELS_PER_SUBMIT) {
7744
+ const chunk = backwardLevels.slice(i, i + BC_BACKWARD_LEVELS_PER_SUBMIT);
7745
+ const last = i + BC_BACKWARD_LEVELS_PER_SUBMIT >= backwardLevels.length;
7746
+ const batch = new CommandBatch(ctx, `${ALGORITHM$1}/backward`);
7747
+ const pass = batch.pass("backward");
7748
+ for (const level of chunk) {
7749
+ const start = endsWords[level];
7750
+ const count = endsWords[level + 1] - start;
7751
+ recordBc(
7752
+ state,
7753
+ pass,
7754
+ state.backward,
7755
+ { rowPtr: state.rowPtr, colIdx: state.colIdx, S, depthK, sigmaK, deltaK },
7756
+ { n, k, start, count },
7757
+ count
7758
+ );
7759
+ }
7760
+ if (last) {
7761
+ recordBc(state, pass, state.gather, { deltaK, bc: state.bc }, { n, k }, n);
7762
+ if (state.edgeGather !== null && state.arcScores !== null) {
7763
+ recordBc(
7764
+ state,
7765
+ pass,
7766
+ state.edgeGather,
7767
+ { rowPtr: state.rowPtr, colIdx: state.colIdx, depthK, sigmaK, deltaK, arcScores: state.arcScores },
7768
+ { n, k, count: state.arcCount },
7769
+ state.arcCount
7770
+ );
7771
+ }
7772
+ }
7773
+ batch.endPass();
7774
+ const wantArrays = last && tuning.readArrays === true;
7775
+ const requests = wantArrays ? [depthK, sigmaK, deltaK].map((b) => batch.readback(b.buffer, b.offset, 4 * words)) : [];
7776
+ const back = await submit(state, batch, signal);
7777
+ if (wantArrays) {
7778
+ arrays = {
7779
+ depthK: new Uint32Array(back, requests[0].offset, words).slice(),
7780
+ sigmaK: new Uint32Array(back, requests[1].offset, words).slice(),
7781
+ deltaK: new Float32Array(back, requests[2].offset, words).slice()
7782
+ };
7783
+ }
7784
+ if (last) {
7785
+ break;
7786
+ }
7787
+ }
7788
+ tuning.onBatch?.({
7789
+ sources,
7790
+ forward: form,
7791
+ levels,
7792
+ ends: endsWords.slice(),
7793
+ sigmaOverflow: overflow,
7794
+ depthK: arrays?.depthK ?? null,
7795
+ sigmaK: arrays?.sigmaK ?? null,
7796
+ deltaK: arrays?.deltaK ?? null
7797
+ });
7798
+ return { levels, overflow };
7799
+ }
7800
+ async function runRaw(ctx, s, sources, withEdges, tuning, options) {
7801
+ const n = s.nodeCount;
7802
+ const pinned = tuning.forward ?? "auto";
7803
+ const levelsPerSubmit = tuning.levelsPerSubmit ?? MAX_LEVELS_PER_SUBMIT;
7804
+ if (!Number.isInteger(levelsPerSubmit) || levelsPerSubmit < 1 || levelsPerSubmit > MAX_LEVELS_PER_SUBMIT) {
7805
+ throw badArgument("levelsPerSubmit", levelsPerSubmit, `an integer in [1, ${MAX_LEVELS_PER_SUBMIT}]`);
7806
+ }
7807
+ const limits = tuning.limits ?? ctx.caps.limits;
7808
+ const kMax = planBatchSize(n, sources.length, limits);
7809
+ const core = ctx.residency.core(s);
7810
+ assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM$1);
7811
+ const { edgeCount } = s;
7812
+ const mayRunEdge = pinned === "edge" || pinned === "auto" && sources.length > kMax;
7813
+ const edgeView = mayRunEdge && edgeCount > 0 ? ctx.residency.view(s, "edgeList") : null;
7814
+ const scope = algorithmScope(ctx, ALGORITHM$1, RING_SLOTS$1);
7815
+ try {
7816
+ const arrayBytes = 4 * n * kMax;
7817
+ const lease = (bytes, label) => bindingOf(scope.scratch(bytes, label), bytes);
7818
+ const arcBytes = 4 * Math.max(1, s.arcCount);
7819
+ const [fill, finalize, forward, backward, gather] = await Promise.all(
7820
+ ["fill", "bc-finalize", "bc-forward", "bc-backward", "bc-gather"].map(
7821
+ (id) => ctx.pipelines.kernel(kernelSpec(id))
7822
+ )
7823
+ );
7824
+ const forwardEdge = edgeView === null ? null : await ctx.pipelines.kernel(kernelSpec("bc-forward-edge", { UNDIRECTED: !s.directed }));
7825
+ const edgeGather = withEdges ? await ctx.pipelines.kernel(kernelSpec("bc-edge-gather")) : null;
7826
+ const state = {
7827
+ ctx,
7828
+ scope,
7829
+ n,
7830
+ S: lease(arrayBytes, "S"),
7831
+ ends: lease(4 * (n + 2), "ends"),
7832
+ depthK: lease(arrayBytes, "depthK"),
7833
+ sigmaK: lease(arrayBytes, "sigmaK"),
7834
+ deltaK: lease(arrayBytes, "deltaK"),
7835
+ counters: lease(FRONTIER_COUNTERS.byteLength, "counters"),
7836
+ bc: lease(4 * n, "bc"),
7837
+ arcScores: withEdges ? lease(arcBytes, "arc-scores") : null,
7838
+ rowPtr: core.rowPtr,
7839
+ colIdx: core.colIdx ?? core.rowPtr,
7840
+ edgeSrc: edgeView?.bindings.src ?? null,
7841
+ edgeDst: edgeView?.bindings.dst ?? null,
7842
+ edgeCount,
7843
+ arcCount: s.arcCount,
7844
+ fill,
7845
+ finalize,
7846
+ forward,
7847
+ forwardEdge,
7848
+ backward,
7849
+ gather,
7850
+ edgeGather,
7851
+ bound: /* @__PURE__ */ new Map()
7852
+ };
7853
+ await ctx.allocator.check();
7854
+ const setup = new CommandBatch(ctx, `${ALGORITHM$1}/setup`);
7855
+ const setupPass = setup.pass("setup");
7856
+ recordFill(state, setupPass, state.bc, n, 0);
7857
+ if (state.arcScores !== null) {
7858
+ recordFill(state, setupPass, state.arcScores, arcBytes / 4, 0);
7859
+ }
7860
+ setup.endPass();
7861
+ await submit(state, setup, options?.signal);
7862
+ let overflow = false;
7863
+ let batches = 0;
7864
+ let previousLevels = -1;
7865
+ for (let start = 0; start < sources.length; ) {
7866
+ const k = planBatchSize(n, sources.length - start, limits);
7867
+ let form = pinned === "edge" ? "edge" : "frontier";
7868
+ if (pinned === "auto" && previousLevels >= 0) {
7869
+ form = previousLevels < BC_EDGE_PARALLEL_GAMMA * Math.log2(n) ? "edge" : "frontier";
7870
+ }
7871
+ if (state.forwardEdge === null) {
7872
+ form = "frontier";
7873
+ }
7874
+ const batch = sources.slice(start, start + k);
7875
+ const outcome = await runBatch(state, batch, form, levelsPerSubmit, tuning, options?.signal);
7876
+ overflow = overflow || outcome.overflow;
7877
+ previousLevels = outcome.levels;
7878
+ start += k;
7879
+ batches += 1;
7880
+ options?.onProgress?.(start, sources.length);
7881
+ }
7882
+ const result = new CommandBatch(ctx, `${ALGORITHM$1}/result`);
7883
+ const vertexRequest = result.readback(state.bc.buffer, state.bc.offset, 4 * n);
7884
+ const arcRequest = state.arcScores === null ? null : result.readback(state.arcScores.buffer, state.arcScores.offset, 4 * s.arcCount);
7885
+ const back = await submit(state, result, options?.signal);
7886
+ return {
7887
+ vertex: new Float32Array(back, vertexRequest.offset, n).slice(),
7888
+ perArc: arcRequest === null ? null : new Float32Array(back, arcRequest.offset, s.arcCount).slice(),
7889
+ sourcesUsed: sources.length,
7890
+ sigmaOverflow: overflow,
7891
+ batches
7892
+ };
7893
+ } finally {
7894
+ scope.dispose();
7895
+ }
7896
+ }
7897
+ function normaliser(s, normalized) {
7898
+ const n = s.nodeCount;
7899
+ const factor = s.directed ? (n - 1) * (n - 2) : (n - 1) * (n - 2) / 2;
7900
+ return normalized === true && factor > 0 ? factor : 1;
7901
+ }
7902
+ async function precheck(ctx, options) {
7903
+ ctx.assertReady();
7904
+ await assertDeviceComputes(ctx);
7905
+ if (options?.endpoints === true) {
7906
+ throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM$1}: endpoints: true is not supported`, {
7907
+ feature: "betweenness.endpoints",
7908
+ hint: "the CPU endpoints branch in algorithms/src/algorithms/centrality/betweenness.ts (predecessors.length === 0 && w !== source) never fires, so there is no convention to match"
7909
+ });
7910
+ }
7911
+ }
7912
+ async function betweennessWithTuning(ctx, s, options, tuning) {
7913
+ await precheck(ctx, options);
7914
+ const n = s.nodeCount;
7915
+ const scores = checkDest(ALGORITHM$1, options?.dest, n) ?? new Float32Array(n);
7916
+ const sources = resolveSources(options, n);
7917
+ if (options?.signal?.aborted) {
7918
+ throw aborted(ALGORITHM$1);
7919
+ }
7920
+ if (n === 0 || sources.length === 0) {
7921
+ scores.fill(0);
7922
+ return { scores, iterations: 0, converged: true, precision: "f32", sourcesUsed: 0, sigmaOverflow: false };
7923
+ }
7924
+ const raw = await runRaw(ctx, s, sources, false, tuning, options);
7925
+ const divisor = (s.directed ? 1 : 2) * normaliser(s, options?.normalized);
7926
+ for (let v = 0; v < n; v++) {
7927
+ scores[v] = raw.vertex[v] / divisor;
7928
+ }
7929
+ return {
7930
+ scores,
7931
+ iterations: raw.batches,
7932
+ converged: true,
7933
+ precision: "f32",
7934
+ sourcesUsed: raw.sourcesUsed,
7935
+ sigmaOverflow: raw.sigmaOverflow
7936
+ };
7937
+ }
7938
+ async function edgeBetweennessWithTuning(ctx, s, options, tuning, onArcs) {
7939
+ await precheck(ctx, options);
7940
+ const n = s.nodeCount;
7941
+ const scores = checkDest("edgeBetweennessCentrality", options?.dest, s.edgeCount) ?? new Float32Array(s.edgeCount);
7942
+ const sources = resolveSources(options, n);
7943
+ if (options?.signal?.aborted) {
7944
+ throw aborted(ALGORITHM$1);
7945
+ }
7946
+ if (n === 0 || sources.length === 0 || s.arcCount === 0) {
7947
+ scores.fill(0);
7948
+ return { scores, precision: "f32", sourcesUsed: sources.length, sigmaOverflow: false };
7949
+ }
7950
+ const raw = await runRaw(ctx, s, sources, true, tuning, options);
7951
+ const perArc = raw.perArc ?? new Float32Array(s.arcCount);
7952
+ const folded = foldArcs(s, perArc, "sum");
7953
+ const divisor = (s.directed ? 1 : 2) * normaliser(s, options?.normalized);
7954
+ for (let e = 0; e < s.edgeCount; e++) {
7955
+ scores[e] = folded[e] / divisor;
7956
+ }
7957
+ return { scores, precision: "f32", sourcesUsed: raw.sourcesUsed, sigmaOverflow: raw.sigmaOverflow };
7958
+ }
7959
+ function betweennessCentrality(ctx, s, options) {
7960
+ return betweennessWithTuning(ctx, s, options, {});
7961
+ }
7962
+ function edgeBetweennessCentrality(ctx, s, options) {
7963
+ return edgeBetweennessWithTuning(ctx, s, options, {});
7964
+ }
7202
7965
  const ALGORITHM = "closenessCentrality";
7203
7966
  const SOURCES_PER_BATCH = 32;
7204
7967
  const PER_SOURCE_WORDS = 4 * SOURCES_PER_BATCH;
@@ -7218,29 +7981,44 @@ function reusingScratch(scope) {
7218
7981
  }
7219
7982
  };
7220
7983
  }
7221
- async function weightedRoute(ctx, s, scores, options) {
7984
+ async function weightedRoute(ctx, s, scores, sources, options) {
7222
7985
  const n = s.nodeCount;
7223
- for (let source = 0; source < n; source++) {
7986
+ const count = sources?.length ?? n;
7987
+ const totals = sources === null ? null : new Float64Array(n);
7988
+ for (let i = 0; i < count; i++) {
7224
7989
  if (options?.signal?.aborted) {
7225
7990
  throw aborted(ALGORITHM);
7226
7991
  }
7992
+ const source = sources === null ? i : sources[i];
7227
7993
  const { dist } = await sssp(ctx, s, source, { signal: options?.signal });
7228
7994
  let sum = 0;
7229
7995
  for (let v = 0; v < n; v++) {
7230
7996
  const d = dist[v];
7231
7997
  if (v !== source && d !== Infinity) {
7232
- sum += d;
7998
+ if (totals === null) {
7999
+ sum += d;
8000
+ } else {
8001
+ totals[v] += d;
8002
+ }
7233
8003
  }
7234
8004
  }
7235
- scores[source] = sum === 0 ? 0 : 1 / sum;
7236
- options?.onProgress?.(source + 1, n);
8005
+ if (totals === null) {
8006
+ scores[source] = sum === 0 ? 0 : 1 / sum;
8007
+ }
8008
+ options?.onProgress?.(i + 1, count);
7237
8009
  }
7238
- return { scores, iterations: n, converged: true, precision: "f32" };
8010
+ if (totals !== null) {
8011
+ totals.forEach((sum, v) => {
8012
+ scores[v] = sum === 0 ? 0 : 1 / sum;
8013
+ });
8014
+ }
8015
+ return { scores, iterations: count, converged: true, precision: "f32", sourcesUsed: count };
7239
8016
  }
7240
- async function sweepRoute(ctx, s, scores, levelsPerSubmit, options, tuning) {
8017
+ async function sweepRoute(ctx, s, scores, sources, levelsPerSubmit, options, tuning) {
7241
8018
  const n = s.nodeCount;
7242
- if (n === 0) {
7243
- return { scores, iterations: 0, converged: true, precision: "f32" };
8019
+ const seedCount = sources?.length ?? n;
8020
+ if (seedCount === 0) {
8021
+ return { scores, iterations: 0, converged: true, precision: "f32", sourcesUsed: 0 };
7244
8022
  }
7245
8023
  const core = ctx.residency.core(s);
7246
8024
  assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM);
@@ -7265,7 +8043,17 @@ async function sweepRoute(ctx, s, scores, levelsPerSubmit, options, tuning) {
7265
8043
  FRONTIER_COUNTERS.byteLength
7266
8044
  );
7267
8045
  const perSourceBytes = 4 * PER_SOURCE_WORDS;
7268
- const perSource = bindingOf(scope.scratch(perSourceBytes, "per-source"), perSourceBytes);
8046
+ const zeroedWords = PER_SOURCE_WORDS + (sources === null ? 0 : bitsBase);
8047
+ const perSourceAll = 4 * (zeroedWords + (sources === null ? 0 : sources.length));
8048
+ const perSource = bindingOf(scope.scratch(perSourceAll, "per-source"), perSourceAll);
8049
+ if (sources !== null) {
8050
+ ctx.device.queue.writeBuffer(
8051
+ perSource.buffer,
8052
+ perSource.offset + 4 * zeroedWords,
8053
+ Uint32Array.from(sources)
8054
+ );
8055
+ }
8056
+ const totals = sources === null ? null : new Float64Array(n);
7269
8057
  await ctx.allocator.check();
7270
8058
  const compact = await prepareCompact(reusingScratch(scope));
7271
8059
  const sweep = await ctx.pipelines.kernel(kernelSpec("closeness-sweep", graphOverrides(core, null)));
@@ -7275,29 +8063,34 @@ async function sweepRoute(ctx, s, scores, levelsPerSubmit, options, tuning) {
7275
8063
  const onePlan = plan1d(1, wg, ctx.caps);
7276
8064
  const regionPlan = plan1d(bitsBase, wg, ctx.caps);
7277
8065
  const sweepPlan = planGridStride(n, wg, ctx.caps);
7278
- const recordFill = (pass, dst, count, mode) => {
8066
+ const recordFill2 = (pass, dst, count, mode) => {
7279
8067
  const params = scope.params(FILL_PARAMS, { count, value: 0, mode, pad0: 0 });
7280
8068
  fill.dispatch(pass, fill.bind({ dst, P: params.binding }), plan1d(count, wg, ctx.caps), [params.offset]);
7281
8069
  };
7282
- const submit = (batch) => {
8070
+ const submit2 = (batch) => {
7283
8071
  scope.flush();
7284
8072
  return batch.submit();
7285
8073
  };
7286
8074
  const setup = new CommandBatch(ctx, `${ALGORITHM}/setup`);
7287
- recordFill(setup.pass("fill"), iota, n, 1);
8075
+ recordFill2(setup.pass("fill"), iota, n, 1);
7288
8076
  setup.endPass();
7289
- await submit(setup).readback;
8077
+ await submit2(setup).readback;
7290
8078
  ctx.assertReady();
7291
8079
  let batches = 0;
7292
- for (let batchStart = 0; batchStart < n; batchStart += SOURCES_PER_BATCH) {
8080
+ for (let batchStart = 0; batchStart < seedCount; batchStart += SOURCES_PER_BATCH) {
7293
8081
  let level = 0;
7294
8082
  for (let first = true; ; first = false) {
7295
8083
  const batch = new CommandBatch(ctx, `${ALGORITHM}/levels`);
7296
8084
  const pass = batch.pass("closeness");
7297
8085
  if (first) {
7298
- recordFill(pass, bits, 4 * bitsBase, 0);
7299
- recordFill(pass, perSource, PER_SOURCE_WORDS, 0);
7300
- const seed = scope.params(FRONTIER_PARAMS, { role: 1, n, bitsBase, source: batchStart });
8086
+ recordFill2(pass, bits, 4 * bitsBase, 0);
8087
+ recordFill2(pass, perSource, zeroedWords, 0);
8088
+ const seed = scope.params(FRONTIER_PARAMS, {
8089
+ role: sources === null ? 1 : 2,
8090
+ n: seedCount,
8091
+ bitsBase,
8092
+ source: batchStart
8093
+ });
7301
8094
  reduce.dispatch(pass, reduce.bind({ counters, perSource, bits, P: seed.binding }), onePlan, [
7302
8095
  seed.offset
7303
8096
  ]);
@@ -7315,7 +8108,8 @@ async function sweepRoute(ctx, s, scores, levelsPerSubmit, options, tuning) {
7315
8108
  arcBase: 0,
7316
8109
  arcEnd: s.arcCount,
7317
8110
  mode,
7318
- stride: sweepPlan.stride ?? wg
8111
+ stride: sweepPlan.stride ?? wg,
8112
+ perNode: sources === null ? 0 : 1
7319
8113
  });
7320
8114
  return {
7321
8115
  bound: sweep.bind({ ...graph, frontierList, counters, bits, perSource, P: params.binding }),
@@ -7340,7 +8134,8 @@ async function sweepRoute(ctx, s, scores, levelsPerSubmit, options, tuning) {
7340
8134
  batch.endPass();
7341
8135
  const doneRequest = batch.readback(counters.buffer, counters.offset + 4 * W.done, 4);
7342
8136
  const blockRequest = batch.readback(perSource.buffer, perSource.offset, perSourceBytes);
7343
- const submitted = submit(batch);
8137
+ const nodeRequest = totals === null ? null : batch.readback(perSource.buffer, perSource.offset + perSourceBytes, 4 * n);
8138
+ const submitted = submit2(batch);
7344
8139
  const back = await submitted.readback;
7345
8140
  ctx.assertReady();
7346
8141
  if (options?.signal?.aborted) {
@@ -7348,10 +8143,17 @@ async function sweepRoute(ctx, s, scores, levelsPerSubmit, options, tuning) {
7348
8143
  }
7349
8144
  if (new Uint32Array(back, doneRequest.offset, 1)[0] !== 0) {
7350
8145
  const block = new Uint32Array(back, blockRequest.offset, PER_SOURCE_WORDS);
7351
- const count = Math.min(SOURCES_PER_BATCH, n - batchStart);
7352
- for (let i = 0; i < count; i++) {
7353
- const sum = block[3 * SOURCES_PER_BATCH + i] * 2 ** 32 + block[2 * SOURCES_PER_BATCH + i];
7354
- scores[batchStart + i] = sum === 0 ? 0 : 1 / sum;
8146
+ if (totals === null || nodeRequest === null) {
8147
+ const count = Math.min(SOURCES_PER_BATCH, n - batchStart);
8148
+ for (let i = 0; i < count; i++) {
8149
+ const sum = block[3 * SOURCES_PER_BATCH + i] * 2 ** 32 + block[2 * SOURCES_PER_BATCH + i];
8150
+ scores[batchStart + i] = sum === 0 ? 0 : 1 / sum;
8151
+ }
8152
+ } else {
8153
+ const sums = new Uint32Array(back, nodeRequest.offset, n);
8154
+ for (let v = 0; v < n; v++) {
8155
+ totals[v] += sums[v];
8156
+ }
7355
8157
  }
7356
8158
  tuning.onBatch?.(batchStart, block.slice());
7357
8159
  break;
@@ -7365,13 +8167,37 @@ async function sweepRoute(ctx, s, scores, levelsPerSubmit, options, tuning) {
7365
8167
  }
7366
8168
  }
7367
8169
  batches += 1;
7368
- options?.onProgress?.(Math.min(batchStart + SOURCES_PER_BATCH, n), n);
8170
+ options?.onProgress?.(Math.min(batchStart + SOURCES_PER_BATCH, seedCount), seedCount);
7369
8171
  }
7370
- return { scores, iterations: batches, converged: true, precision: "f32" };
8172
+ totals?.forEach((sum, v) => {
8173
+ scores[v] = sum === 0 ? 0 : 1 / sum;
8174
+ });
8175
+ return { scores, iterations: batches, converged: true, precision: "f32", sourcesUsed: seedCount };
7371
8176
  } finally {
7372
8177
  scope.dispose();
7373
8178
  }
7374
8179
  }
8180
+ function checkSources(s, sources) {
8181
+ if (sources === void 0) {
8182
+ return null;
8183
+ }
8184
+ if (s.directed) {
8185
+ throw new WebGpuGraphError("E_UNSUPPORTED", `${ALGORITHM}: sampled sources need an undirected snapshot`, {
8186
+ feature: "closenessCentrality.directedSources",
8187
+ hint: "run the CPU port, which searches the in-arcs"
8188
+ });
8189
+ }
8190
+ for (const v of sources) {
8191
+ if (!Number.isInteger(v) || v < 0 || v >= s.nodeCount) {
8192
+ throw new WebGpuGraphError("E_INVALID_ARGUMENT", `${ALGORITHM}: a source is not a node index`, {
8193
+ argument: "sources",
8194
+ value: v,
8195
+ expected: `an integer in [0, ${s.nodeCount})`
8196
+ });
8197
+ }
8198
+ }
8199
+ return sources;
8200
+ }
7375
8201
  async function closenessWithTuning(ctx, s, options, tuning) {
7376
8202
  ctx.assertReady();
7377
8203
  await assertDeviceComputes(ctx);
@@ -7384,6 +8210,7 @@ async function closenessWithTuning(ctx, s, options, tuning) {
7384
8210
  }
7385
8211
  }
7386
8212
  const n = s.nodeCount;
8213
+ const sources = checkSources(s, options?.sources);
7387
8214
  const levelsPerSubmit = tuning.levelsPerSubmit ?? MAX_LEVELS_PER_SUBMIT;
7388
8215
  if (!Number.isInteger(levelsPerSubmit) || levelsPerSubmit < 1 || levelsPerSubmit > MAX_LEVELS_PER_SUBMIT) {
7389
8216
  throw new WebGpuGraphError(
@@ -7417,9 +8244,9 @@ async function closenessWithTuning(ctx, s, options, tuning) {
7417
8244
  feature: "closenessCentrality.nonFiniteWeights"
7418
8245
  });
7419
8246
  }
7420
- return weightedRoute(ctx, s, scores, options);
8247
+ return weightedRoute(ctx, s, scores, sources, options);
7421
8248
  }
7422
- return sweepRoute(ctx, s, scores, levelsPerSubmit, options, tuning);
8249
+ return sweepRoute(ctx, s, scores, sources, levelsPerSubmit, options, tuning);
7423
8250
  }
7424
8251
  function closenessCentrality(ctx, s, options) {
7425
8252
  return closenessWithTuning(ctx, s, options, {});
@@ -12412,6 +13239,15 @@ function copyBetweenness(defaults) {
12412
13239
  }
12413
13240
  return Object.freeze(copy);
12414
13241
  }
13242
+ function withBetweennessDefaults(defaults, options, nodeCount) {
13243
+ if (defaults === void 0 || options?.sources !== void 0 || options?.k !== void 0) {
13244
+ return options;
13245
+ }
13246
+ if (defaults.sources !== void 0) {
13247
+ return { ...options, sources: defaults.sources.filter((v) => v < nodeCount) };
13248
+ }
13249
+ return { ...options, k: defaults.k !== void 0 && defaults.k < nodeCount ? defaults.k : void 0 };
13250
+ }
12415
13251
  function copyAlgorithms(algorithms) {
12416
13252
  const copy = { ...algorithms };
12417
13253
  if (algorithms.betweenness !== void 0) {
@@ -12573,12 +13409,43 @@ function createAccelerator(ctx, options) {
12573
13409
  ctx.assertReady();
12574
13410
  return await bellmanFord(ctx, gs, source, o);
12575
13411
  },
13412
+ /**
13413
+ * Betweenness centrality on the device (spec 8.4): exact, or sampled through `sources` / `k` (the call's own,
13414
+ * else the accelerator's `algorithms.betweenness` defaults), the unscaled sum over the sources run.
13415
+ * `endpoints: true` is refused.
13416
+ * @param gs - the snapshot
13417
+ * @param o - the seam's `BetweennessAcceleratorOptions`
13418
+ * @returns the f32 scores with `sourcesUsed` and `sigmaOverflow`
13419
+ */
13420
+ async betweennessCentrality(gs, o) {
13421
+ ctx.assertReady();
13422
+ return await betweennessCentrality(
13423
+ ctx,
13424
+ gs,
13425
+ withBetweennessDefaults(frozen.algorithms?.betweenness, o, gs.nodeCount)
13426
+ );
13427
+ },
13428
+ /**
13429
+ * Edge betweenness on the device (spec 8.4): one score per edge, arcs summed and halved when undirected; sampling
13430
+ * and defaults as `betweennessCentrality`.
13431
+ * @param gs - the snapshot
13432
+ * @param o - the seam's `BetweennessAcceleratorOptions`
13433
+ * @returns the f32 per-edge scores with `sourcesUsed` and `sigmaOverflow`
13434
+ */
13435
+ async edgeBetweennessCentrality(gs, o) {
13436
+ ctx.assertReady();
13437
+ return await edgeBetweennessCentrality(
13438
+ ctx,
13439
+ gs,
13440
+ withBetweennessDefaults(frozen.algorithms?.betweenness, o, gs.nodeCount)
13441
+ );
13442
+ },
12576
13443
  /**
12577
13444
  * Closeness centrality on the device (spec 8.4; P8-T13): the bit-parallel multi-source sweep, or one `sssp`
12578
13445
  * per source when `weighted`. `maxIterations` / `tolerance` are refused when defined (P8 PD-25).
12579
13446
  * @param gs - the snapshot
12580
- * @param o - the seam's placeholder `HitsOptionsLike` (`weighted`)
12581
- * @returns the f32 scores with `precision: "f32"` (spec 9.7)
13447
+ * @param o - `weighted`, and a sampled run's `sources` (undirected snapshots only)
13448
+ * @returns the f32 scores with `precision: "f32"` (spec 9.7) and `sourcesUsed`
12582
13449
  */
12583
13450
  async closenessCentrality(gs, o) {
12584
13451
  ctx.assertReady();
@@ -12693,7 +13560,7 @@ async function calibrateLayout(ctx, options) {
12693
13560
  };
12694
13561
  }
12695
13562
  export {
12696
- J as ARC_WINDOW_ALIGN,
13563
+ Q as ARC_WINDOW_ALIGN,
12697
13564
  EXACT_MAX_NODES,
12698
13565
  FA2_DEFAULTS,
12699
13566
  FR_DEFAULTS,
@@ -12701,12 +13568,13 @@ export {
12701
13568
  LAYOUT_TUNING_DEFAULTS,
12702
13569
  MAX_1D_ITEMS,
12703
13570
  MAX_WORKGROUPS_PER_DIM,
12704
- K as PASSTHROUGH_FORMAT_CODES,
13571
+ V as PASSTHROUGH_FORMAT_CODES,
12705
13572
  SE_DEFAULTS,
12706
- N as STORAGE_ALIGN,
12707
- O as WORKGROUP_SIZE,
13573
+ X as STORAGE_ALIGN,
13574
+ Y as WORKGROUP_SIZE,
12708
13575
  WebGpuGraphError,
12709
13576
  bellmanFord,
13577
+ betweennessCentrality,
12710
13578
  breadthFirstSearch,
12711
13579
  calibrateLayout,
12712
13580
  closenessCentrality,
@@ -12716,10 +13584,11 @@ export {
12716
13584
  createFruchtermanReingold,
12717
13585
  createSpringElectrical,
12718
13586
  degree,
13587
+ edgeBetweennessCentrality,
12719
13588
  eigenvectorCentrality,
12720
13589
  hasErrorCode,
12721
13590
  hits,
12722
- Q as isSoftwareAdapter,
13591
+ Z as isSoftwareAdapter,
12723
13592
  isWebGpuGraphError,
12724
13593
  katzCentrality,
12725
13594
  pageRank,