@graphty/webgpu-graph-algorithms 0.6.27 → 0.6.28

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/README.md +56 -6
  2. package/dist/acquire.d.ts +2 -0
  3. package/dist/browser.js +18 -1
  4. package/dist/browser.js.map +1 -1
  5. package/dist/chunks/accelerator-B-FjQwaA.js +19173 -0
  6. package/dist/chunks/accelerator-B-FjQwaA.js.map +1 -0
  7. package/dist/chunks/managed-D_GdQtnu.js +98 -0
  8. package/dist/chunks/managed-D_GdQtnu.js.map +1 -0
  9. package/dist/node.js +18 -1
  10. package/dist/node.js.map +1 -1
  11. package/dist/src/accelerator.d.ts.map +1 -1
  12. package/dist/src/accelerator.js +5 -3
  13. package/dist/src/accelerator.js.map +1 -1
  14. package/dist/src/algorithms/all-pairs.d.ts.map +1 -1
  15. package/dist/src/algorithms/all-pairs.js +72 -47
  16. package/dist/src/algorithms/all-pairs.js.map +1 -1
  17. package/dist/src/algorithms/betweenness.d.ts +1 -1
  18. package/dist/src/algorithms/betweenness.js +2 -2
  19. package/dist/src/algorithms/closeness.d.ts +45 -42
  20. package/dist/src/algorithms/closeness.d.ts.map +1 -1
  21. package/dist/src/algorithms/closeness.js +295 -226
  22. package/dist/src/algorithms/closeness.js.map +1 -1
  23. package/dist/src/browser/index.d.ts +10 -0
  24. package/dist/src/browser/index.d.ts.map +1 -1
  25. package/dist/src/browser/index.js +22 -0
  26. package/dist/src/browser/index.js.map +1 -1
  27. package/dist/src/constants.d.ts +13 -3
  28. package/dist/src/constants.d.ts.map +1 -1
  29. package/dist/src/constants.js +13 -3
  30. package/dist/src/constants.js.map +1 -1
  31. package/dist/src/kernels.d.ts +14 -4
  32. package/dist/src/kernels.d.ts.map +1 -1
  33. package/dist/src/kernels.js +62 -28
  34. package/dist/src/kernels.js.map +1 -1
  35. package/dist/src/layouts/force-simulation.d.ts.map +1 -1
  36. package/dist/src/layouts/force-simulation.js +0 -1
  37. package/dist/src/layouts/force-simulation.js.map +1 -1
  38. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  39. package/dist/src/layouts/forceatlas2.js +0 -1
  40. package/dist/src/layouts/forceatlas2.js.map +1 -1
  41. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  42. package/dist/src/layouts/fruchterman-reingold.js +0 -1
  43. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  44. package/dist/src/layouts/repulsion-grid.d.ts +1 -3
  45. package/dist/src/layouts/repulsion-grid.d.ts.map +1 -1
  46. package/dist/src/layouts/repulsion-grid.js +1 -6
  47. package/dist/src/layouts/repulsion-grid.js.map +1 -1
  48. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  49. package/dist/src/layouts/spring-electrical.js +0 -1
  50. package/dist/src/layouts/spring-electrical.js.map +1 -1
  51. package/dist/src/managed.d.ts +11 -0
  52. package/dist/src/managed.d.ts.map +1 -0
  53. package/dist/src/managed.js +129 -0
  54. package/dist/src/managed.js.map +1 -0
  55. package/dist/src/node/index.d.ts +11 -0
  56. package/dist/src/node/index.d.ts.map +1 -1
  57. package/dist/src/node/index.js +21 -0
  58. package/dist/src/node/index.js.map +1 -1
  59. package/dist/src/primitives/grid-pyramid.d.ts +16 -15
  60. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  61. package/dist/src/primitives/grid-pyramid.js +20 -28
  62. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  63. package/dist/src/types/accelerator.d.ts +2 -0
  64. package/dist/src/types/accelerator.d.ts.map +1 -1
  65. package/dist/src/types/managed.d.ts +81 -0
  66. package/dist/src/types/managed.d.ts.map +1 -0
  67. package/dist/src/types/managed.js +7 -0
  68. package/dist/src/types/managed.js.map +1 -0
  69. package/dist/src/wgsl/bc-forward.wgsl.d.ts +1 -1
  70. package/dist/src/wgsl/bc-forward.wgsl.js +1 -1
  71. package/dist/src/wgsl/closeness-level.wgsl.d.ts +37 -0
  72. package/dist/src/wgsl/closeness-level.wgsl.d.ts.map +1 -0
  73. package/dist/src/wgsl/closeness-level.wgsl.js +204 -0
  74. package/dist/src/wgsl/closeness-level.wgsl.js.map +1 -0
  75. package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts +11 -0
  76. package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts.map +1 -0
  77. package/dist/src/wgsl/closeness-rowsum.wgsl.js +42 -0
  78. package/dist/src/wgsl/closeness-rowsum.wgsl.js.map +1 -0
  79. package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +2 -2
  80. package/dist/src/wgsl/grid-centroid-hub.wgsl.js +2 -2
  81. package/dist/webgpu-graph-algorithms.js +142 -15586
  82. package/dist/webgpu-graph-algorithms.js.map +1 -1
  83. package/package.json +10 -4
  84. package/src/accelerator.ts +5 -3
  85. package/src/algorithms/all-pairs.ts +86 -56
  86. package/src/algorithms/betweenness.ts +2 -2
  87. package/src/algorithms/closeness.ts +353 -256
  88. package/src/browser/index.ts +37 -0
  89. package/src/constants.ts +13 -3
  90. package/src/kernels.ts +65 -36
  91. package/src/layouts/force-simulation.ts +0 -1
  92. package/src/layouts/forceatlas2.ts +0 -1
  93. package/src/layouts/fruchterman-reingold.ts +0 -1
  94. package/src/layouts/repulsion-grid.ts +2 -7
  95. package/src/layouts/spring-electrical.ts +0 -1
  96. package/src/managed.ts +172 -0
  97. package/src/node/index.ts +36 -0
  98. package/src/primitives/grid-pyramid.ts +29 -41
  99. package/src/types/accelerator.ts +2 -0
  100. package/src/types/managed.ts +86 -0
  101. package/src/wgsl/bc-forward.wgsl.ts +1 -1
  102. package/src/wgsl/closeness-level.wgsl.ts +203 -0
  103. package/src/wgsl/closeness-rowsum.wgsl.ts +41 -0
  104. package/src/wgsl/grid-centroid-hub.wgsl.ts +2 -2
  105. package/dist/chunks/context-BZY6SMsM.js +0 -3615
  106. package/dist/chunks/context-BZY6SMsM.js.map +0 -1
  107. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +0 -20
  108. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +0 -1
  109. package/dist/src/wgsl/closeness-reduce.wgsl.js +0 -69
  110. package/dist/src/wgsl/closeness-reduce.wgsl.js.map +0 -1
  111. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +0 -22
  112. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +0 -1
  113. package/dist/src/wgsl/closeness-sweep.wgsl.js +0 -106
  114. package/dist/src/wgsl/closeness-sweep.wgsl.js.map +0 -1
  115. package/src/wgsl/closeness-reduce.wgsl.ts +0 -68
  116. package/src/wgsl/closeness-sweep.wgsl.ts +0 -105
@@ -6,7 +6,17 @@
6
6
 
7
7
  import { GpuContext } from "../context.js";
8
8
  import { WebGpuGraphError } from "../errors.js";
9
+ import { manageAccelerator } from "../managed.js";
9
10
  import type { GpuContextOptions, ProbeResult } from "../types/context.js";
11
+ import type { AcquireAcceleratorOptions, ManagedAccelerator } from "../types/managed.js";
12
+
13
+ export type {
14
+ AcceleratorDeclined,
15
+ AcceleratorReady,
16
+ AcquireAcceleratorOptions,
17
+ AcquireResult,
18
+ ManagedAccelerator,
19
+ } from "../types/managed.js";
10
20
 
11
21
  /** Options of the browser helpers (spec 3.4); a type alias, not an empty `extends` interface, which strictTypeChecked's no-empty-object-type (allowInterfaces "never") reports. */
12
22
  export type BrowserGpuOptions = Omit<GpuContextOptions, "gpu" | "device" | "runtime">;
@@ -55,3 +65,30 @@ export function requestGpuContext(options?: BrowserGpuOptions): Promise<GpuConte
55
65
  }
56
66
  return GpuContext.create({ powerPreference: "high-performance", ...options, gpu, runtime: "browser" });
57
67
  }
68
+
69
+ /**
70
+ * The managed accelerator on navigator.gpu: probe, context, device self-check and accelerator on the first
71
+ * `current()`, a new device after a loss, disposal by `dispose()`. A software adapter is declined unless
72
+ * `acceptSoftware` is set. The same function as the `./acquire` subpath resolves to in a browser.
73
+ * @param options - see AcquireAcceleratorOptions; `adapter` is ignored here
74
+ * @returns the handle; nothing is probed until `current()` is called
75
+ */
76
+ export function acquireAccelerator(options?: AcquireAcceleratorOptions): ManagedAccelerator {
77
+ return manageAccelerator(
78
+ {
79
+ probe: (o, rejectSoftware) => probeBrowserWebGpu({ powerPreference: o.powerPreference, rejectSoftware }),
80
+ open: (o, probe, rejectSoftware) =>
81
+ requestGpuContext({
82
+ adapter: probe.adapter ?? undefined,
83
+ powerPreference: o.powerPreference,
84
+ rejectSoftware,
85
+ warnUnreleasedSnapshots: o.warnUnreleasedSnapshots,
86
+ }),
87
+ noWebGpuFix: () =>
88
+ (globalThis as { isSecureContext?: boolean }).isSecureContext === false
89
+ ? "serve the page over https or from localhost: WebGPU needs a secure context"
90
+ : null,
91
+ },
92
+ options,
93
+ );
94
+ }
package/src/constants.ts CHANGED
@@ -43,7 +43,9 @@ export const UNIFORM_SLOT_BYTES = 256;
43
43
  * unbiasedness item) the same day by its re-scope to the whole-field ratio, and the far field's accuracy on those
44
44
  * fixtures is the follow-up -- because "auto" is the default every consumer sees and the exact tier is the accurate
45
45
  * one; the rule's answer on the dev box (1024) is recorded, not shipped, until that work closes and the value is
46
- * re-fixed. A
46
+ * re-fixed. Re-measured on 2026-10-02 after the grid tier's hub-cell centroid became a direct dispatch (issue #732):
47
+ * the rule still answers 1024 (exact 0.26 against grid 0.18 ms per iteration at 4,096 nodes), because the indirect
48
+ * dispatch's cost was Dawn's validation pass, which the profiler's per-pass times never included. A
47
49
  * consumer whose GPU differs (integrated, Apple, T4) passes its own value through
48
50
  * createAccelerator(ctx, { layout: { exactMaxNodes } }); calibrateLayout(ctx) measures it.
49
51
  */
@@ -261,8 +263,16 @@ export const SSSP_DELTA_FACTOR = 32;
261
263
  export const F32_INF_BITS = 0x7f800000;
262
264
  /** Design 8.4 "k planned from maxBufferSize and a 25% budget": the share of `maxBufferSize` one betweenness source batch may hold. WebGPU exposes no device memory size, so this is a fraction of the largest buffer, not a memory measurement. */
263
265
  export const BC_BATCH_BUDGET_FRACTION = 0.25;
264
- /** Design 10.1's betweenness column: the most sources one betweenness batch runs together. */
265
- export const BC_MAX_BATCH = 64;
266
+ /**
267
+ * The most sources one betweenness batch runs together. 256 rather than design 10.1's 64 (issue #733): a batch's
268
+ * cost is dominated by its per-level submits and readbacks, not by its width, so a quarter of the batches runs an
269
+ * exact call on random graphs of 1k / 2k / 4k nodes (4 edges per node, RTX 4070 SUPER) 2.8x / 2.5x / 1.8x faster
270
+ * with the frontier forward form and 1.9x / 1.6x / 1.5x with the per-batch choice, with bit-identical vertex and
271
+ * edge scores (the gathers add every source's dependency in source order whatever the batching). The memory budget
272
+ * still decides k above about 16k nodes at default limits, so large graphs are unchanged. Forward levels per submit
273
+ * stay at `MAX_LEVELS_PER_SUBMIT`: the parameter ring is sized for it.
274
+ */
275
+ export const BC_MAX_BATCH = 256;
266
276
  /** Design 8.4 (McLaughlin-Bader): a betweenness batch runs the edge-parallel forward pass when the previous batch's level count is below `BC_EDGE_PARALLEL_GAMMA * log2(n)`. The design names the rule and no value; 2 is unmeasured and a benchmark run re-fixes it. */
267
277
  export const BC_EDGE_PARALLEL_GAMMA = 2;
268
278
  /**
package/src/kernels.ts CHANGED
@@ -12,7 +12,7 @@
12
12
  * dedupe-claim and dedupe-filter; P8-T4 adds frontier-finalize with the FrontierCounters and FrontierParams blocks;
13
13
  * P8-T5 adds advance-expand; P8-T6 adds bfs-contract and sssp-pred; P8-T7 adds bfs-fused; P8-T8 adds bfs-bottom-up,
14
14
  * bfs-bitset-build and bfs-unvisited-flags; P8-T9 adds sssp-relax; P8-T10 adds bf-relax with the BfParams and BfFlags
15
- * blocks; P8-T11 adds closeness-sweep and closeness-reduce; P9 (betweenness) adds bc-finalize, bc-forward,
15
+ * blocks; closeness adds closeness-level and closeness-rowsum with the ClosenessParams block; P9 (betweenness) adds bc-finalize, bc-forward,
16
16
  * bc-backward, bc-gather, bc-edge-gather, bc-forward-edge and bc-count (issue #719) with the BcParams block; all-pairs shortest paths
17
17
  * (design 8.7) adds apsp-init and apsp-fw with the ApspParams block. P11 (the structure and community phase, plan
18
18
  * design/webgpu/plans/2026-09-23-webgpu-p11-structure-and-community.md) adds the graph build on the device (coo-emit,
@@ -45,8 +45,8 @@ import { bfsContractWgsl } from "./wgsl/bfs-contract.wgsl.js";
45
45
  import { bfsFusedWgsl } from "./wgsl/bfs-fused.wgsl.js";
46
46
  import { bfsNextDegreeWgsl } from "./wgsl/bfs-next-degree.wgsl.js";
47
47
  import { bfsUnvisitedFlagsWgsl } from "./wgsl/bfs-unvisited-flags.wgsl.js";
48
- import { closenessReduceWgsl } from "./wgsl/closeness-reduce.wgsl.js";
49
- import { closenessSweepWgsl } from "./wgsl/closeness-sweep.wgsl.js";
48
+ import { closenessLevelWgsl } from "./wgsl/closeness-level.wgsl.js";
49
+ import { closenessRowsumWgsl } from "./wgsl/closeness-rowsum.wgsl.js";
50
50
  import { compactScatterWgsl } from "./wgsl/compact-scatter.wgsl.js";
51
51
  import { cooEmitWgsl } from "./wgsl/coo-emit.wgsl.js";
52
52
  import { cooScatterWgsl } from "./wgsl/coo-scatter.wgsl.js";
@@ -139,8 +139,8 @@ export type KernelId =
139
139
  | "bfs-next-degree"
140
140
  | "sssp-relax"
141
141
  | "bf-relax"
142
- | "closeness-sweep"
143
- | "closeness-reduce"
142
+ | "closeness-level"
143
+ | "closeness-rowsum"
144
144
  | "bc-finalize"
145
145
  | "bc-forward"
146
146
  | "bc-backward"
@@ -482,8 +482,7 @@ export const FRONTIER_COUNTERS: UniformBlock = UniformBlock.define(
482
482
  * `arcEnd` @44 (the bound arc window), `predKind` @48 (0 arc, 1 node), `bitsBase` @52, `source` @56, `stride` @60
483
483
  * (a grid-stride plan's stride), `firstOfSubmit` @64 (the boundary's index inside its submit, clamped to 1: both
484
484
  * the unvisited-count and the unvisited-degree-sum subtraction run at >= 1, issue #391), `iteration` @68 (an
485
- * `sssp-pred` hop pass, P8-T9), `perNode` @72 (`closeness-sweep`: 1 also folds every claim into the per-node
486
- * distance sums of a sampled run), `pad2` @76. The `slotBase` field that once addressed the selector's indirect slots went with
485
+ * `sssp-pred` hop pass, P8-T9), `pad3` @72, `pad2` @76. The `slotBase` field that once addressed the selector's indirect slots went with
487
486
  * the slots (2026-09-25); `pad2` keeps the block an explicit 80 bytes, the way every block here is padded.
488
487
  */
489
488
  export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams", [
@@ -505,7 +504,7 @@ export const FRONTIER_PARAMS: UniformBlock = UniformBlock.define("FrontierParams
505
504
  ["stride", "u32"],
506
505
  ["firstOfSubmit", "u32"],
507
506
  ["iteration", "u32"],
508
- ["perNode", "u32"],
507
+ ["pad3", "u32"],
509
508
  ["pad2", "u32"],
510
509
  ]);
511
510
 
@@ -521,6 +520,35 @@ export const BC_PARAMS: UniformBlock = UniformBlock.define("BcParams", [
521
520
  ["pad1", "u32"],
522
521
  ]);
523
522
 
523
+ /**
524
+ * `ClosenessParams` (uniform, 64 B): the params of `closeness-level` and `closeness-rowsum` -- `role` @0 (level: 0 a
525
+ * level, 1 the seed's clear, 2 the seed's sources; rowsum: 0 integer, 1 f32, 2 harmonic), `n` @4, `words` @8 (32-bit
526
+ * words per node, so `32 x words` sources per batch), `base` @12 (the words of one `bits` region), `total` @16 (the
527
+ * words the clear zeroes), `level` @20, `row` @24 (the level's row of the submit's count table), `ctrl` @28 (the word
528
+ * of the control ring in `table`), `count` @32 (the batch's sources), `source` @36 (its first source, or its first
529
+ * word of the source list), `sourcesAt` @40 (the word of a sampled run's source list in `table`, 0 for the exact
530
+ * run), `pullAt` @44 (the frontier arcs above which a level pulls), `perNode` @48 (1: a sampled run's per-node
531
+ * distance sums), `arcCount` @52 (the arcs a push level covers), `pullOk` @56 (1: a level may pull), `pad0` @60.
532
+ */
533
+ export const CLOSENESS_PARAMS: UniformBlock = UniformBlock.define("ClosenessParams", [
534
+ ["role", "u32"],
535
+ ["n", "u32"],
536
+ ["words", "u32"],
537
+ ["base", "u32"],
538
+ ["total", "u32"],
539
+ ["level", "u32"],
540
+ ["row", "u32"],
541
+ ["ctrl", "u32"],
542
+ ["count", "u32"],
543
+ ["source", "u32"],
544
+ ["sourcesAt", "u32"],
545
+ ["pullAt", "u32"],
546
+ ["perNode", "u32"],
547
+ ["arcCount", "u32"],
548
+ ["pullOk", "u32"],
549
+ ["pad0", "u32"],
550
+ ]);
551
+
524
552
  /** `BfParams` (uniform, 16 B; P8-T10): `edgeCount` @0 (the logical edges of the `edgeList` view), `stride` @4 (the grid-stride plan's stride), `maxRetries` @8 (PD-12's compare-exchange bound), `cutoffBits` @12 (the f32 bit pattern of the CPU port's `cutoff`, `+Inf` when absent). */
525
553
  export const BF_PARAMS: UniformBlock = UniformBlock.define("BfParams", [
526
554
  ["edgeCount", "u32"],
@@ -1085,7 +1113,7 @@ const GRID_CENTROID: KernelEntry = {
1085
1113
  phase: "P4",
1086
1114
  };
1087
1115
 
1088
- /** `grid-centroid-hub` (G4b, spec 7.7; P4-T9, PD-13, DEP-P4-L): one workgroup per hub cell, dispatched indirectly, a WG-strided sum through `wg_reduce_vec4` guarded by `h < hubCount[0]`; 6 storage bindings (`hubCount` is a read-only view of `hubCounters`). */
1116
+ /** `grid-centroid-hub` (G4b, spec 7.7; P4-T9, PD-13, DEP-P4-L): one workgroup per word of `hubList`, dispatched directly, a WG-strided sum through `wg_reduce_vec4` guarded by `h < hubCount[0]`; 6 storage bindings (`hubCount` is a read-only view of `hubCounters`). */
1089
1117
  const GRID_CENTROID_HUB: KernelEntry = {
1090
1118
  id: "grid-centroid-hub",
1091
1119
  body: gridCentroidHubWgsl,
@@ -1425,41 +1453,42 @@ const BF_RELAX: KernelEntry = {
1425
1453
  phase: "P8",
1426
1454
  };
1427
1455
 
1428
- /** `closeness-sweep` (design 8.4 "32 sources per u32 word"; P8-T11, PD-13 / DEP-P8-E): one level of the bit-parallel multi-source BFS -- `advance-expand`'s block-mapped strip over the compacted frontier list with the claim inline (`atomicOr` on the visited word of the four-region `bits` buffer, the won bits into the level's next region and the flags region, one workgroup-memory tally per source flushed by one `atomicAdd` per source per workgroup into `perSource`); 8 storage bindings (the four graph slots, `frontierList` read-only, `counters`, `bits` and `perSource` as `array<atomic<u32>>`) -- exactly at the budget; the inlined Hillis-Steele scan, so `needs: []`. */
1429
- const CLOSENESS_SWEEP: KernelEntry = {
1430
- id: "closeness-sweep",
1431
- body: closenessSweepWgsl,
1432
- entryPoint: "closeness_sweep",
1433
- bindings: GRAPH_SLOTS.concat(
1434
- decl(1, 0, "frontierList", "storage-ro", "array<u32>"),
1435
- decl(1, 1, "counters", "storage", "array<atomic<u32>>"),
1436
- decl(1, 2, "bits", "storage", "array<atomic<u32>>"),
1437
- decl(1, 3, "perSource", "storage", "array<atomic<u32>>"),
1438
- decl(2, 0, "P", "uniform", "FrontierParams"),
1439
- ),
1456
+ /** `closeness-level` (design 8.4 "32 sources per u32 word"): one level of closeness's bit-parallel multi-source BFS in one dispatch, `32 x P.words` sources per batch -- role 0 the level (push over the out-arcs or pull over the in-arcs, push one invocation per arc, pull one per node, chosen per level on the device from the frontier's arcs against `P.pullAt`, the pull with an early exit once every source has reached the node), role 1 and role 2 the batch's seed; 6 storage bindings (the forward and the reverse CSR as group-1 state, `bits` and `table` as `array<atomic<u32>>`). */
1457
+ const CLOSENESS_LEVEL: KernelEntry = {
1458
+ id: "closeness-level",
1459
+ body: closenessLevelWgsl,
1460
+ entryPoint: "closeness_level",
1461
+ bindings: [
1462
+ decl(1, 0, "rowPtr", "storage-ro", "array<u32>"),
1463
+ decl(1, 1, "colIdx", "storage-ro", "array<u32>"),
1464
+ decl(1, 2, "inRowPtr", "storage-ro", "array<u32>"),
1465
+ decl(1, 3, "inColIdx", "storage-ro", "array<u32>"),
1466
+ decl(1, 4, "bits", "storage", "array<atomic<u32>>"),
1467
+ decl(1, 5, "table", "storage", "array<atomic<u32>>"),
1468
+ decl(2, 0, "P", "uniform", "ClosenessParams"),
1469
+ ],
1440
1470
  overrideDecls: [],
1441
- uniforms: [FRONTIER_PARAMS],
1471
+ uniforms: [CLOSENESS_PARAMS],
1442
1472
  needs: [],
1443
1473
  snippetSlots: [],
1444
1474
  phase: "P8",
1445
1475
  };
1446
1476
 
1447
- /** `closeness-reduce` (design 8.4, 9.7; P8-T11, PD-13): the one-lane bookkeeping of the sweep -- role 0 the level boundary (`done` from the previous level's compacted count, `newCount` folded into `reached` and the 64-bit `sum` at `level + 1` with the 16-bit split product and the carry, `level` advanced), role 1 the seed of a batch (the sources' bits into `visited` and the level-0 frontier region, their flags, `counters[0] = k`, `level = U32_MAX`), role 2 the same seed from a sampled run's source list; 3 storage bindings (`counters` and `perSource` as `array<atomic<u32>>`, `bits` plain: one lane writes the seed). */
1448
- const CLOSENESS_REDUCE: KernelEntry = {
1449
- id: "closeness-reduce",
1450
- body: closenessReduceWgsl,
1451
- entryPoint: "closeness_reduce",
1477
+ /** `closeness-rowsum` (design 8.7): closeness from a finished all-pairs matrix, one workgroup per row -- the off-diagonal finite entries summed as integers (`P.role` 0), as f32 (1) or as reciprocals (2, harmonic), by a tree reduction; 2 storage bindings (`dist` read-only, `out`). */
1478
+ const CLOSENESS_ROWSUM: KernelEntry = {
1479
+ id: "closeness-rowsum",
1480
+ body: closenessRowsumWgsl,
1481
+ entryPoint: "closeness_rowsum",
1452
1482
  bindings: [
1453
- decl(1, 0, "counters", "storage", "array<atomic<u32>>"),
1454
- decl(1, 1, "perSource", "storage", "array<atomic<u32>>"),
1455
- decl(1, 2, "bits", "storage", "array<u32>"),
1456
- decl(2, 0, "P", "uniform", "FrontierParams"),
1483
+ decl(1, 0, "dist", "storage-ro", "array<f32>"),
1484
+ decl(1, 1, "out", "storage", "array<u32>"),
1485
+ decl(2, 0, "P", "uniform", "ClosenessParams"),
1457
1486
  ],
1458
1487
  overrideDecls: [],
1459
- uniforms: [FRONTIER_PARAMS],
1488
+ uniforms: [CLOSENESS_PARAMS],
1460
1489
  needs: [],
1461
1490
  snippetSlots: [],
1462
- phase: "P8",
1491
+ phase: "P9",
1463
1492
  };
1464
1493
 
1465
1494
  /** `bc-finalize` (design 8.4, 5.4): the one-lane bookkeeping of a betweenness batch -- role 1 seeds it (depth 0 and one path for the k seed entries of the claim log, `stackTop = k`, `level = U32_MAX`), role 0 is the level boundary (`ends[level + 1] = stackTop`, `frontierCount`, `done` on an empty level); 6 storage bindings (the counters block as `array<atomic<u32>>`, `ends`, `S` read-only, `depthK`, `sigmaK` and `levelMax` plain: one lane writes the seed; `SCALED` seeds f32 bits). The design's finalize row has 2; `ends` is the third (the level boundary), and the seed's `S`, `depthK`, `sigmaK` and `levelMax` make it six. */
@@ -1845,7 +1874,7 @@ const MST_LINK: KernelEntry = {
1845
1874
  * and `"fa2-to-scene"`; M8b-T3 landed the seven P7 entries and P4 its thirteen; P8-T3 landed the three compact /
1846
1875
  * dedupe entries, P8-T4 `"frontier-finalize"`, P8-T5 `"advance-expand"`, P8-T6 `"bfs-contract"` and `"sssp-pred"` and
1847
1876
  * P8-T7 `"bfs-fused"`, P8-T8 `"bfs-bottom-up"`, `"bfs-bitset-build"` and `"bfs-unvisited-flags"`, P8-T9
1848
- * `"sssp-relax"`, P8-T10 `"bf-relax"` and P8-T11 `"closeness-sweep"` and `"closeness-reduce"`, betweenness the
1877
+ * `"sssp-relax"`, P8-T10 `"bf-relax"`, closeness `"closeness-level"` and `"closeness-rowsum"`, betweenness the
1849
1878
  * six `"bc-*"` entries, all-pairs shortest paths `"apsp-init"` and `"apsp-fw"`, and P11 its seven (the graph
1850
1879
  * build, the group-by-key, label propagation's step and triangle counting) plus Boruvka's `"mst-best"` and
1851
1880
  * `"mst-link"`, so every member of `KernelId`
@@ -1896,8 +1925,8 @@ const REGISTRY: Readonly<Partial<Record<KernelId, KernelEntry>>> = Object.freeze
1896
1925
  "bfs-next-degree": BFS_NEXT_DEGREE,
1897
1926
  "sssp-relax": SSSP_RELAX,
1898
1927
  "bf-relax": BF_RELAX,
1899
- "closeness-sweep": CLOSENESS_SWEEP,
1900
- "closeness-reduce": CLOSENESS_REDUCE,
1928
+ "closeness-level": CLOSENESS_LEVEL,
1929
+ "closeness-rowsum": CLOSENESS_ROWSUM,
1901
1930
  "bc-finalize": BC_FINALIZE,
1902
1931
  "bc-forward": BC_FORWARD,
1903
1932
  "bc-backward": BC_BACKWARD,
@@ -183,7 +183,6 @@ const U32_BUFFER_NAMES: ReadonlySet<string> = new Set([
183
183
  "cellStart",
184
184
  "hubList",
185
185
  "hubCounters",
186
- "hubArgs",
187
186
  ]);
188
187
 
189
188
  /** The per-batch epilogue stage: iterations 0..k-2 stop after the stage that precedes it (PLAN DECISION 2). */
@@ -520,7 +520,6 @@ export class ForceAtlas2Model implements ForceModel<ForceAtlas2Options, ForceAtl
520
520
  cellStart: resources.buffer("cellStart"),
521
521
  hubList: resources.buffer("hubList"),
522
522
  hubCounters,
523
- hubArgs: resources.buffer("hubArgs"),
524
523
  pyramid: resources.buffer("pyramid"),
525
524
  });
526
525
  const wg = k1.workgroupSize;
@@ -513,7 +513,6 @@ export class FruchtermanReingoldModel implements ForceModel<FruchtermanReingoldO
513
513
  cellStart: resources.buffer("cellStart"),
514
514
  hubList: resources.buffer("hubList"),
515
515
  hubCounters,
516
- hubArgs: resources.buffer("hubArgs"),
517
516
  pyramid: resources.buffer("pyramid"),
518
517
  });
519
518
  const wg = k1.workgroupSize;
@@ -15,7 +15,7 @@ import { GRID_HUB_CELL } from "../constants.js";
15
15
  import { BufferUsage } from "../device/webgpu-constants.js";
16
16
  import { WebGpuGraphError } from "../errors.js";
17
17
  import { plan1d } from "../kernel/dispatch.js";
18
- import { type BoundKernel, INDIRECT_ARGS_STRIDE, type Kernel } from "../kernel/kernel.js";
18
+ import { type BoundKernel, type Kernel } from "../kernel/kernel.js";
19
19
  import { type PipelineCache } from "../kernel/pipeline-cache.js";
20
20
  import { type UniformBlock, type UniformValues } from "../kernel/struct-block.js";
21
21
  import { type WgslModuleSpec } from "../kernel/wgsl.js";
@@ -56,7 +56,6 @@ export interface RepulsionGridResources extends RepulsionExactResources {
56
56
  readonly cellStart: Binding;
57
57
  readonly hubList: Binding;
58
58
  readonly hubCounters: Binding;
59
- readonly hubArgs: Binding;
60
59
  readonly pyramid: Binding;
61
60
  }
62
61
 
@@ -216,8 +215,7 @@ export class RepulsionGrid {
216
215
 
217
216
  /**
218
217
  * The model-owned buffers of the grid tier (spec 7.3; PD-11): `cellKey` / `cellVal` / `sortedKey` / `sortedIdx`
219
- * 4n, `cellHist` / `cellStart` 4 histWords (cells + 2^dim + 1) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
220
- * indirect slot, `pyramid` 16 B per pyramid cell zeroed. `hubCounters` (16 B, zeroed) is the MODEL's on every
218
+ * 4n, `cellHist` / `cellStart` 4 histWords (cells + 2^dim + 1) zeroed, `hubList` one word per possible hub cell, `pyramid` 16 B per pyramid cell zeroed. `hubCounters` (16 B, zeroed) is the MODEL's on every
221
219
  * tier (PD-14: K1 binds it on the exact tier too). n = 0 reports one node's worth of bytes (spec 3.6).
222
220
  * @param n - the node count
223
221
  * @param spec - the grid
@@ -238,7 +236,6 @@ export class RepulsionGrid {
238
236
  usage: STORAGE_RW,
239
237
  zero: false,
240
238
  },
241
- { name: "hubArgs", byteLength: INDIRECT_ARGS_STRIDE, usage: STORAGE_RW | BufferUsage.INDIRECT, zero: true },
242
239
  { name: "pyramid", byteLength: gridPyramidBytes(spec), usage: STORAGE_RW, zero: true },
243
240
  ];
244
241
  }
@@ -261,7 +258,6 @@ export class RepulsionGrid {
261
258
  kernelSpec("histogram"),
262
259
  kernelSpec("fill"),
263
260
  kernelSpec("grid-centroid"),
264
- kernelSpec("indirect-finalize"),
265
261
  kernelSpec("grid-centroid-hub"),
266
262
  kernelSpec("grid-downsample"),
267
263
  farFieldSpec(overrides),
@@ -348,7 +344,6 @@ export class RepulsionGrid {
348
344
  pyramid: r.pyramid,
349
345
  hubList: r.hubList,
350
346
  hubCounters: r.hubCounters,
351
- hubArgs: r.hubArgs,
352
347
  });
353
348
  this.bound = {
354
349
  far: this.far.bind({
@@ -483,7 +483,6 @@ export class SpringElectricalModel implements ForceModel<SpringElectricalOptions
483
483
  cellStart: resources.buffer("cellStart"),
484
484
  hubList: resources.buffer("hubList"),
485
485
  hubCounters,
486
- hubArgs: resources.buffer("hubArgs"),
487
486
  pyramid: resources.buffer("pyramid"),
488
487
  });
489
488
  const wg = k1.workgroupSize;
package/src/managed.ts ADDED
@@ -0,0 +1,172 @@
1
+ /**
2
+ * The managed accelerator behind `acquireAccelerator` (the `./acquire` subpath, and the same function on `./browser`
3
+ * and `./node`): probe, context, device self-check, accelerator, re-acquisition after device loss, disposal. The
4
+ * runtime-specific half (how to probe and how to open a context) comes from the entry that calls `manageAccelerator`;
5
+ * everything else is here once, so no consumer writes it.
6
+ *
7
+ * Every decline is decided before any work runs: a caller with a CPU implementation runs it and reports the reason.
8
+ * A failure of work that already started on the device is never caught here; it reaches the caller of that work.
9
+ */
10
+
11
+ import { createAccelerator } from "./accelerator.js";
12
+ import type { GpuContext } from "./context.js";
13
+ import { isWebGpuGraphError, WebGpuGraphError } from "./errors.js";
14
+ import { verifyDevice } from "./primitives/verify.js";
15
+ import type { GpuAccelerator } from "./types/accelerator.js";
16
+ import type { DeviceCheck, ProbeResult } from "./types/context.js";
17
+ import type {
18
+ AcceleratorDeclined,
19
+ AcquireAcceleratorOptions,
20
+ AcquireResult,
21
+ ManagedAccelerator,
22
+ } from "./types/managed.js";
23
+
24
+ /**
25
+ * How one runtime finds a device; supplied by the browser and Node entries.
26
+ * @internal
27
+ */
28
+ export interface AcceleratorPlatform {
29
+ /** Probes without creating a device; never throws. */
30
+ probe(options: AcquireAcceleratorOptions, rejectSoftware: boolean): Promise<ProbeResult>;
31
+ /** Opens a context on the probed adapter. */
32
+ open(options: AcquireAcceleratorOptions, probe: ProbeResult, rejectSoftware: boolean): Promise<GpuContext>;
33
+ /** What the user can do about E_NO_WEBGPU on this runtime, or null. */
34
+ noWebGpuFix(): string | null;
35
+ /** Test seam: replaces `verifyDevice`. */
36
+ readonly verify?: ((ctx: GpuContext) => Promise<DeviceCheck>) | undefined;
37
+ }
38
+
39
+ /** The fix of a declined software adapter, on every runtime. */
40
+ const SOFTWARE_FIX = "pass acceptSoftware: true to use the software adapter (usually slower than the CPU)";
41
+
42
+ const DECLINE_CODES: ReadonlySet<string> = new Set(["E_NO_WEBGPU", "E_NO_ADAPTER", "E_SOFTWARE_ONLY"]);
43
+
44
+ /**
45
+ * A decline record.
46
+ * @param platform - the runtime, for the E_NO_WEBGPU fix
47
+ * @param code - the decline code
48
+ * @param reason - the reason in words
49
+ * @param rest - the adapter and the check, when known
50
+ * @returns the record
51
+ */
52
+ function declined(
53
+ platform: AcceleratorPlatform,
54
+ code: AcceleratorDeclined["code"],
55
+ reason: string,
56
+ rest: Partial<Pick<AcceleratorDeclined, "adapter" | "check">> = {},
57
+ ): AcceleratorDeclined {
58
+ let fix: string | null = null;
59
+ if (code === "E_SOFTWARE_ONLY") {
60
+ fix = SOFTWARE_FIX;
61
+ } else if (code === "E_NO_WEBGPU") {
62
+ fix = platform.noWebGpuFix();
63
+ }
64
+ return { ok: false, code, reason, fix, adapter: rest.adapter ?? null, check: rest.check ?? null };
65
+ }
66
+
67
+ /**
68
+ * One acquisition: probe, open, self-check, accelerator. The context is disposed on every path that does not hand
69
+ * it over.
70
+ * @param platform - the runtime
71
+ * @param options - the caller's options
72
+ * @returns the accelerator, or why there is none
73
+ */
74
+ async function acquireOnce(platform: AcceleratorPlatform, options: AcquireAcceleratorOptions): Promise<AcquireResult> {
75
+ const rejectSoftware = options.acceptSoftware !== true;
76
+ const probe = await platform.probe(options, rejectSoftware);
77
+ if (!probe.ok || probe.code !== "OK") {
78
+ const code = probe.code === "OK" ? "E_NO_ADAPTER" : probe.code;
79
+ return declined(platform, code, probe.reason ?? code, { adapter: probe.summary });
80
+ }
81
+ let ctx: GpuContext;
82
+ try {
83
+ ctx = await platform.open(options, probe, rejectSoftware);
84
+ } catch (err) {
85
+ // the adapter can change between the probe and the device request (a GPU process restart)
86
+ if (isWebGpuGraphError(err) && DECLINE_CODES.has(err.code)) {
87
+ return declined(platform, err.code as AcceleratorDeclined["code"], err.message, { adapter: probe.summary });
88
+ }
89
+ throw err;
90
+ }
91
+ let accelerator: GpuAccelerator;
92
+ try {
93
+ const check = await (platform.verify ?? verifyDevice)(ctx);
94
+ if (check.mismatch !== null) {
95
+ ctx.dispose();
96
+ const where = check.mismatch.poison
97
+ ? `${check.mismatch.where} was never written`
98
+ : `${check.mismatch.where} came back as ${String(check.mismatch.actual)} where ${String(check.mismatch.expected)} was required`;
99
+ return declined(
100
+ platform,
101
+ "E_DEVICE_INCORRECT",
102
+ `the ${check.vendor} device computed a known prefix sum incorrectly: ${where}`,
103
+ { adapter: probe.summary, check },
104
+ );
105
+ }
106
+ accelerator = createAccelerator(ctx, options.accelerator);
107
+ } catch (err) {
108
+ ctx.dispose();
109
+ throw err;
110
+ }
111
+ return { ok: true, code: "OK", accelerator };
112
+ }
113
+
114
+ /**
115
+ * The managed accelerator over one runtime.
116
+ * @param platform - how this runtime probes and opens a context
117
+ * @param options - the caller's options
118
+ * @returns the handle
119
+ * @internal
120
+ */
121
+ export function manageAccelerator(
122
+ platform: AcceleratorPlatform,
123
+ options: AcquireAcceleratorOptions = {},
124
+ ): ManagedAccelerator {
125
+ let pending: Promise<AcquireResult> | null = null;
126
+ let ready: GpuAccelerator | null = null;
127
+ let disposed = false;
128
+ const disposedError = (): WebGpuGraphError =>
129
+ new WebGpuGraphError("E_DISPOSED", "the managed accelerator was disposed", { label: "acquireAccelerator" });
130
+ return {
131
+ current(): Promise<AcquireResult> {
132
+ if (disposed) {
133
+ return Promise.reject(disposedError());
134
+ }
135
+ if (ready !== null && ready.ctx.state !== "ready") {
136
+ // lost, or disposed by its holder: the next answer is a new device
137
+ ready = null;
138
+ pending = null;
139
+ }
140
+ if (pending === null) {
141
+ const attempt = acquireOnce(platform, options).then(
142
+ (result) => {
143
+ if (!result.ok) {
144
+ return result;
145
+ }
146
+ if (disposed) {
147
+ result.accelerator.dispose();
148
+ throw disposedError();
149
+ }
150
+ ready = result.accelerator;
151
+ return result;
152
+ },
153
+ (err: unknown) => {
154
+ // not a decline: the next call tries again rather than remembering a transient failure
155
+ if (pending === attempt) {
156
+ pending = null;
157
+ }
158
+ throw err;
159
+ },
160
+ );
161
+ pending = attempt;
162
+ }
163
+ return pending;
164
+ },
165
+ dispose(): void {
166
+ disposed = true;
167
+ ready?.dispose();
168
+ ready = null;
169
+ pending = null;
170
+ },
171
+ };
172
+ }
package/src/node/index.ts CHANGED
@@ -11,7 +11,17 @@
11
11
 
12
12
  import { GpuContext } from "../context.js";
13
13
  import { WebGpuGraphError } from "../errors.js";
14
+ import { manageAccelerator } from "../managed.js";
14
15
  import type { GpuContextOptions, ProbeResult } from "../types/context.js";
16
+ import type { AcquireAcceleratorOptions, ManagedAccelerator } from "../types/managed.js";
17
+
18
+ export type {
19
+ AcceleratorDeclined,
20
+ AcceleratorReady,
21
+ AcquireAcceleratorOptions,
22
+ AcquireResult,
23
+ ManagedAccelerator,
24
+ } from "../types/managed.js";
15
25
 
16
26
  /** Options of the Node helpers (spec 3.4). */
17
27
  export interface NodeGpuOptions extends Omit<GpuContextOptions, "gpu" | "adapter" | "device" | "runtime"> {
@@ -252,3 +262,29 @@ export async function probeNodeWebGpu(options?: NodeGpuOptions): Promise<ProbeRe
252
262
  handle.dispose();
253
263
  }
254
264
  }
265
+
266
+ /**
267
+ * The managed accelerator on Dawn: probe, context, device self-check and accelerator on the first `current()`, a
268
+ * new device after a loss, disposal by `dispose()`. Without the optional `webgpu` package `current()` declines with
269
+ * E_NO_WEBGPU and the install command as its fix; a software adapter is declined unless `acceptSoftware` is set.
270
+ * The same function as the `./acquire` subpath resolves to under Node.
271
+ * @param options - see AcquireAcceleratorOptions; `adapter` picks the Dawn adapter by name
272
+ * @returns the handle; nothing is loaded or probed until `current()` is called
273
+ */
274
+ export function acquireAccelerator(options?: AcquireAcceleratorOptions): ManagedAccelerator {
275
+ return manageAccelerator(
276
+ {
277
+ probe: (o, rejectSoftware) =>
278
+ probeNodeWebGpu({ adapter: o.adapter, powerPreference: o.powerPreference, rejectSoftware }),
279
+ open: (o, _probe, rejectSoftware) =>
280
+ createNodeGpuContext({
281
+ adapter: o.adapter,
282
+ powerPreference: o.powerPreference,
283
+ rejectSoftware,
284
+ warnUnreleasedSnapshots: o.warnUnreleasedSnapshots,
285
+ }),
286
+ noWebGpuFix: () => INSTALL_HINT,
287
+ },
288
+ options,
289
+ );
290
+ }