@graphty/webgpu-graph-algorithms 0.6.26 → 0.6.28

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/README.md +56 -6
  2. package/dist/acquire.d.ts +2 -0
  3. package/dist/browser.js +18 -1
  4. package/dist/browser.js.map +1 -1
  5. package/dist/chunks/accelerator-B-FjQwaA.js +19173 -0
  6. package/dist/chunks/accelerator-B-FjQwaA.js.map +1 -0
  7. package/dist/chunks/managed-D_GdQtnu.js +98 -0
  8. package/dist/chunks/managed-D_GdQtnu.js.map +1 -0
  9. package/dist/node.js +18 -1
  10. package/dist/node.js.map +1 -1
  11. package/dist/src/accelerator.d.ts.map +1 -1
  12. package/dist/src/accelerator.js +5 -3
  13. package/dist/src/accelerator.js.map +1 -1
  14. package/dist/src/algorithms/all-pairs.d.ts.map +1 -1
  15. package/dist/src/algorithms/all-pairs.js +72 -47
  16. package/dist/src/algorithms/all-pairs.js.map +1 -1
  17. package/dist/src/algorithms/betweenness.d.ts +1 -1
  18. package/dist/src/algorithms/betweenness.js +2 -2
  19. package/dist/src/algorithms/closeness.d.ts +45 -42
  20. package/dist/src/algorithms/closeness.d.ts.map +1 -1
  21. package/dist/src/algorithms/closeness.js +295 -226
  22. package/dist/src/algorithms/closeness.js.map +1 -1
  23. package/dist/src/browser/index.d.ts +10 -0
  24. package/dist/src/browser/index.d.ts.map +1 -1
  25. package/dist/src/browser/index.js +22 -0
  26. package/dist/src/browser/index.js.map +1 -1
  27. package/dist/src/constants.d.ts +13 -3
  28. package/dist/src/constants.d.ts.map +1 -1
  29. package/dist/src/constants.js +13 -3
  30. package/dist/src/constants.js.map +1 -1
  31. package/dist/src/kernels.d.ts +14 -4
  32. package/dist/src/kernels.d.ts.map +1 -1
  33. package/dist/src/kernels.js +62 -28
  34. package/dist/src/kernels.js.map +1 -1
  35. package/dist/src/layouts/force-simulation.d.ts.map +1 -1
  36. package/dist/src/layouts/force-simulation.js +0 -1
  37. package/dist/src/layouts/force-simulation.js.map +1 -1
  38. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  39. package/dist/src/layouts/forceatlas2.js +0 -1
  40. package/dist/src/layouts/forceatlas2.js.map +1 -1
  41. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  42. package/dist/src/layouts/fruchterman-reingold.js +0 -1
  43. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  44. package/dist/src/layouts/repulsion-grid.d.ts +1 -3
  45. package/dist/src/layouts/repulsion-grid.d.ts.map +1 -1
  46. package/dist/src/layouts/repulsion-grid.js +1 -6
  47. package/dist/src/layouts/repulsion-grid.js.map +1 -1
  48. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  49. package/dist/src/layouts/spring-electrical.js +0 -1
  50. package/dist/src/layouts/spring-electrical.js.map +1 -1
  51. package/dist/src/managed.d.ts +11 -0
  52. package/dist/src/managed.d.ts.map +1 -0
  53. package/dist/src/managed.js +129 -0
  54. package/dist/src/managed.js.map +1 -0
  55. package/dist/src/node/index.d.ts +11 -0
  56. package/dist/src/node/index.d.ts.map +1 -1
  57. package/dist/src/node/index.js +21 -0
  58. package/dist/src/node/index.js.map +1 -1
  59. package/dist/src/primitives/grid-pyramid.d.ts +16 -15
  60. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  61. package/dist/src/primitives/grid-pyramid.js +20 -28
  62. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  63. package/dist/src/types/accelerator.d.ts +2 -0
  64. package/dist/src/types/accelerator.d.ts.map +1 -1
  65. package/dist/src/types/managed.d.ts +81 -0
  66. package/dist/src/types/managed.d.ts.map +1 -0
  67. package/dist/src/types/managed.js +7 -0
  68. package/dist/src/types/managed.js.map +1 -0
  69. package/dist/src/wgsl/bc-forward.wgsl.d.ts +1 -1
  70. package/dist/src/wgsl/bc-forward.wgsl.js +1 -1
  71. package/dist/src/wgsl/closeness-level.wgsl.d.ts +37 -0
  72. package/dist/src/wgsl/closeness-level.wgsl.d.ts.map +1 -0
  73. package/dist/src/wgsl/closeness-level.wgsl.js +204 -0
  74. package/dist/src/wgsl/closeness-level.wgsl.js.map +1 -0
  75. package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts +11 -0
  76. package/dist/src/wgsl/closeness-rowsum.wgsl.d.ts.map +1 -0
  77. package/dist/src/wgsl/closeness-rowsum.wgsl.js +42 -0
  78. package/dist/src/wgsl/closeness-rowsum.wgsl.js.map +1 -0
  79. package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +2 -2
  80. package/dist/src/wgsl/grid-centroid-hub.wgsl.js +2 -2
  81. package/dist/webgpu-graph-algorithms.js +142 -15586
  82. package/dist/webgpu-graph-algorithms.js.map +1 -1
  83. package/package.json +10 -4
  84. package/src/accelerator.ts +5 -3
  85. package/src/algorithms/all-pairs.ts +86 -56
  86. package/src/algorithms/betweenness.ts +2 -2
  87. package/src/algorithms/closeness.ts +353 -256
  88. package/src/browser/index.ts +37 -0
  89. package/src/constants.ts +13 -3
  90. package/src/kernels.ts +65 -36
  91. package/src/layouts/force-simulation.ts +0 -1
  92. package/src/layouts/forceatlas2.ts +0 -1
  93. package/src/layouts/fruchterman-reingold.ts +0 -1
  94. package/src/layouts/repulsion-grid.ts +2 -7
  95. package/src/layouts/spring-electrical.ts +0 -1
  96. package/src/managed.ts +172 -0
  97. package/src/node/index.ts +36 -0
  98. package/src/primitives/grid-pyramid.ts +29 -41
  99. package/src/types/accelerator.ts +2 -0
  100. package/src/types/managed.ts +86 -0
  101. package/src/wgsl/bc-forward.wgsl.ts +1 -1
  102. package/src/wgsl/closeness-level.wgsl.ts +203 -0
  103. package/src/wgsl/closeness-rowsum.wgsl.ts +41 -0
  104. package/src/wgsl/grid-centroid-hub.wgsl.ts +2 -2
  105. package/dist/chunks/context-BZY6SMsM.js +0 -3615
  106. package/dist/chunks/context-BZY6SMsM.js.map +0 -1
  107. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts +0 -20
  108. package/dist/src/wgsl/closeness-reduce.wgsl.d.ts.map +0 -1
  109. package/dist/src/wgsl/closeness-reduce.wgsl.js +0 -69
  110. package/dist/src/wgsl/closeness-reduce.wgsl.js.map +0 -1
  111. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts +0 -22
  112. package/dist/src/wgsl/closeness-sweep.wgsl.d.ts.map +0 -1
  113. package/dist/src/wgsl/closeness-sweep.wgsl.js +0 -106
  114. package/dist/src/wgsl/closeness-sweep.wgsl.js.map +0 -1
  115. package/src/wgsl/closeness-reduce.wgsl.ts +0 -68
  116. package/src/wgsl/closeness-sweep.wgsl.ts +0 -105
@@ -1,23 +1,26 @@
1
1
  /**
2
2
  * The grid pyramid (spec 6 row 12, 7.7 G4-G5; P4-T9): the planner that records, into the caller's pass, the finest
3
3
  * centroids (G4, `grid-centroid`: thread per cell over `cells + outsideCells`, the pseudo-cells included), the hub-cell
4
- * completion (G4a: the T1 `indirect-finalize` over `hubCounters[0]` into `hubArgs` with `wg = 1`, so the finalize's
5
- * `ceil(count / wg)` is ONE workgroup per hub cell; G4b: `grid-centroid-hub`, one workgroup per hub cell, dispatched
6
- * indirectly; PD-13, DEP-P4-I) and one `grid-downsample` dispatch per coarser
7
- * level (G5). Level 0 holds `[sum m x, sum m y, sum m z, sum m]` per cell; every parent is the sum of its 2^dim
4
+ * completion (G4b: `grid-centroid-hub`, one workgroup per hub cell; PD-13) and one `grid-downsample` dispatch per
5
+ * coarser level (G5). G4b is a DIRECT dispatch of one workgroup per word of `hubList` -- the most hub cells the graph
6
+ * can hold -- and the kernel's `h < hubCount[0]` guard idles the workgroups past the count. It was an indirect
7
+ * dispatch from a finalize-written args slot until issue #732: Dawn validates every indirect dispatch with a hidden
8
+ * pass that cost 0.3 to 0.9 ms of device time per iteration, several times the whole grid tier's own work at 10k
9
+ * nodes (the frontier kernels' lesson, docs/decisions/G8.md G8-F5). The idle workgroups cost far less: `hubList` holds
10
+ * `ceil(n / GRID_HUB_CELL)` words, so 977 workgroups at 1M nodes, each reading one word. Level 0 holds `[sum m x, sum m y, sum m z, sum m]` per cell; every parent is the sum of its 2^dim
8
11
  * children; the pseudo-cells (indices `cells ..` of level 0) are never children. No atomics touch the sums (design 6 row 12:
9
12
  * bitwise reproducible); the only atomics are the hub append and the occupancy max.
10
13
  *
11
- * The named grid buffers (`pyramid`, `hubList`, `hubCounters`, `hubArgs`) are the caller's (the model's
12
- * `BufferSpec`s, so `inspect(name)` reaches them); the static params of the finalize and of every level are written
14
+ * The named grid buffers (`pyramid`, `hubList`, `hubCounters`) are the caller's (the model's
15
+ * `BufferSpec`s, so `inspect(name)` reaches them); the static params of every level are written
13
16
  * ONCE at bind() through the scope's params writer (PD-11), so record() writes no uniform. `src/primitives/**` never
14
17
  * imports `src/context.ts`.
15
18
  */
16
19
 
17
20
  import { WebGpuGraphError } from "../errors.js";
18
- import { type DispatchPlan, plan1d } from "../kernel/dispatch.js";
21
+ import { plan1d, plan2d } from "../kernel/dispatch.js";
19
22
  import { type BoundKernel, type Kernel } from "../kernel/kernel.js";
20
- import { GRID_LEVEL_PARAMS, INDIRECT_PARAMS, kernelSpec } from "../kernels.js";
23
+ import { GRID_LEVEL_PARAMS, kernelSpec } from "../kernels.js";
21
24
  import { type Binding } from "../types/memory.js";
22
25
  import { type GridSpec } from "./grid.js";
23
26
  import { type ReduceScope } from "./reduce.js";
@@ -38,39 +41,37 @@ export interface GridPyramidBindings {
38
41
  readonly cellStart: Binding;
39
42
  /** `pyramidCells` vec4f: every level, level 0 first with the 2^dim orthant pseudo-cells from index `cells`. */
40
43
  readonly pyramid: Binding;
41
- /** The hub cells' indices, appended by G4 (at least one word; at most `floor(n / (GRID_HUB_CELL + 1))` are ever written, so `ceil(n / GRID_HUB_CELL)` words always suffice). */
44
+ /** The hub cells' indices, appended by G4 (at least one word; at most `floor(n / (GRID_HUB_CELL + 1))` are ever written, so `ceil(n / GRID_HUB_CELL)` words always suffice). Its word count is G4b's workgroup count. */
42
45
  readonly hubList: Binding;
43
46
  /** Two u32 (bound whole, at least 8 bytes): `[0]` the hub count, `[1]` the largest cell occupancy; the caller zeroes both before every build (K1, T10). */
44
47
  readonly hubCounters: Binding;
45
- /** One 16-byte indirect args slot with the INDIRECT usage: G4a writes it, G4b dispatches from it. */
46
- readonly hubArgs: Binding;
47
48
  }
48
49
 
49
- /** Where `record()` stops: after level 0 is whole (G4, G4a and G4b) or after every coarser level (G5, the default). */
50
+ /** Where `record()` stops: after level 0 is whole (G4 and G4b) or after every coarser level (G5, the default). */
50
51
  export type GridPyramidStage = "G4" | "G5";
51
52
 
52
53
  /** A prepared pyramid build (spec 7.7 G4-G5): binds once per load, records the stages of one iteration into a pass. */
53
54
  export interface GridPyramidPlanner {
54
55
  /**
55
- * Binds the named buffers and writes the static params of the finalize and of every level (PD-11); called once
56
+ * Binds the named buffers and writes the static params of every level (PD-11); called once
56
57
  * per load (a second call rebinds and writes fresh params, so it belongs to a reload, never to an iteration).
57
58
  * @param bindings - the buffers
58
59
  */
59
60
  bind(bindings: GridPyramidBindings): void;
60
61
  /**
61
- * Records G4, G4a, G4b and then G5 for every coarser level at the `Fa2Params` slot `paramsOffset`; `upTo: "G4"`
62
- * stops after G4b (level 0 is whole: G4a and G4b are G4's completion).
62
+ * Records G4, G4b and then G5 for every coarser level at the `Fa2Params` slot `paramsOffset`; `upTo: "G4"`
63
+ * stops after G4b (level 0 is whole: G4b is G4's completion).
63
64
  * @param pass - the compute pass
64
65
  * @param paramsOffset - the dynamic offset of this iteration's `Fa2Params`
65
66
  * @param upTo - the last stage to record (default "G5")
66
67
  */
67
68
  record(pass: GPUComputePassEncoder, paramsOffset: number, upTo?: GridPyramidStage): void;
68
- /** Dispatches the last record() issued: 3 after `upTo: "G4"`, `3 + (levels - 1)` for a full record. */
69
+ /** Dispatches the last record() issued: 2 after `upTo: "G4"`, `2 + (levels - 1)` for a full record. */
69
70
  readonly lastDispatches: number;
70
71
  }
71
72
 
72
73
  /**
73
- * Prepares the pyramid's pipelines over a scope (G4, the finalize, G4b and G5; compiles once) so bind() and record()
74
+ * Prepares the pyramid's pipelines over a scope (G4, G4b and G5; compiles once) so bind() and record()
74
75
  * are synchronous. The planner lives exactly as long as the scope.
75
76
  * @param scope - the caller's scope (device, caps, cache, scratch, params)
76
77
  * @param spec - the grid
@@ -78,31 +79,28 @@ export interface GridPyramidPlanner {
78
79
  */
79
80
  export async function preparePyramid(scope: ReduceScope, spec: GridSpec): Promise<GridPyramidPlanner> {
80
81
  const centroid = await scope.pipelines.kernel(kernelSpec("grid-centroid"));
81
- const finalize = await scope.pipelines.kernel(kernelSpec("indirect-finalize"));
82
82
  const hub = await scope.pipelines.kernel(kernelSpec("grid-centroid-hub"));
83
83
  const downsample = await scope.pipelines.kernel(kernelSpec("grid-downsample"));
84
- return new GridPyramidPlannerImpl(scope, spec, { centroid, finalize, hub, downsample });
84
+ return new GridPyramidPlannerImpl(scope, spec, { centroid, hub, downsample });
85
85
  }
86
86
 
87
- /** The four kernels of the build. */
87
+ /** The three kernels of the build. */
88
88
  interface Kernels {
89
89
  readonly centroid: Kernel;
90
- readonly finalize: Kernel;
91
90
  readonly hub: Kernel;
92
91
  readonly downsample: Kernel;
93
92
  }
94
93
 
95
94
  /** What bind() prepared: the bound groups and, per coarser level, its bound group with its params offset. */
96
95
  interface Bound {
97
- readonly hubArgs: Binding;
98
96
  readonly centroid: BoundKernel;
99
- readonly finalize: BoundKernel;
100
- readonly finalizeOffset: number;
101
97
  readonly hub: BoundKernel;
98
+ /** G4b's workgroups: one per word of `hubList`. */
99
+ readonly hubSlots: number;
102
100
  readonly levels: readonly { readonly bound: BoundKernel; readonly offset: number; readonly parentCells: number }[];
103
101
  }
104
102
 
105
- /** The planner: G4, G4a, G4b and G5 over one scope. */
103
+ /** The planner: G4, G4b and G5 over one scope. */
106
104
  class GridPyramidPlannerImpl implements GridPyramidPlanner {
107
105
  private readonly scope: ReduceScope;
108
106
  private readonly spec: GridSpec;
@@ -114,7 +112,7 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
114
112
  * Wraps the resolved kernels; use preparePyramid().
115
113
  * @param scope - the caller's scope
116
114
  * @param spec - the grid
117
- * @param kernels - the four kernels
115
+ * @param kernels - the three kernels
118
116
  */
119
117
  constructor(scope: ReduceScope, spec: GridSpec, kernels: Kernels) {
120
118
  this.scope = scope;
@@ -135,15 +133,9 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
135
133
  * @param bindings - the buffers
136
134
  */
137
135
  bind(bindings: GridPyramidBindings): void {
138
- const { centroid, finalize, hub, downsample } = this.kernels;
136
+ const { centroid, hub, downsample } = this.kernels;
139
137
  const { spec, scope } = this;
140
138
  const b = bindings;
141
- const finalizeParams = scope.params(INDIRECT_PARAMS, {
142
- countIndex: 0,
143
- wg: 1, // the finalize plans ceil(count / wg) workgroups over ITEMS; G4b's item is a hub cell, one workgroup each
144
- slot: 0,
145
- pad0: 0,
146
- });
147
139
  const levels: { readonly bound: BoundKernel; readonly offset: number; readonly parentCells: number }[] = [];
148
140
  let parentSide = spec.g;
149
141
  for (let level = 0; level + 1 < spec.levels; level++) {
@@ -166,7 +158,6 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
166
158
  });
167
159
  }
168
160
  this.bound = {
169
- hubArgs: b.hubArgs,
170
161
  centroid: centroid.bind({
171
162
  sortedIdx: b.sortedIdx,
172
163
  cellStart: b.cellStart,
@@ -176,8 +167,6 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
176
167
  hubCounters: b.hubCounters,
177
168
  P: b.params,
178
169
  }),
179
- finalize: finalize.bind({ counters: b.hubCounters, args: b.hubArgs, P: finalizeParams.binding }),
180
- finalizeOffset: finalizeParams.offset,
181
170
  hub: hub.bind({
182
171
  sortedIdx: b.sortedIdx,
183
172
  cellStart: b.cellStart,
@@ -187,6 +176,7 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
187
176
  hubCount: b.hubCounters,
188
177
  P: b.params,
189
178
  }),
179
+ hubSlots: Math.floor(b.hubList.size / 4),
190
180
  levels,
191
181
  };
192
182
  }
@@ -202,13 +192,11 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
202
192
  if (bound === null) {
203
193
  throw new WebGpuGraphError("E_NOT_LOADED", "gridPyramid: record() before bind()", { argument: "bind" });
204
194
  }
205
- const { centroid, finalize, hub, downsample } = this.kernels;
206
- const one: DispatchPlan = { x: 1, y: 1, z: 1, items: 1, stride: null };
195
+ const { centroid, hub, downsample } = this.kernels;
207
196
  const level0 = spec.cells + spec.outsideCells;
208
197
  centroid.dispatch(pass, bound.centroid, plan1d(level0, scope.workgroupSize, scope.caps), [paramsOffset]);
209
- finalize.dispatch(pass, bound.finalize, one, [bound.finalizeOffset]);
210
- hub.dispatchIndirect(pass, bound.hub, bound.hubArgs, 0, [paramsOffset]);
211
- this.dispatches = 3;
198
+ hub.dispatch(pass, bound.hub, plan2d(bound.hubSlots, scope.caps), [paramsOffset]);
199
+ this.dispatches = 2;
212
200
  if ((upTo ?? "G5") === "G4") {
213
201
  return;
214
202
  }
@@ -130,6 +130,8 @@ export interface AcceleratorOptions {
130
130
  */
131
131
  export interface GpuAccelerator extends AlgorithmAccelerator, LayoutAccelerator {
132
132
  readonly kind: "webgpu";
133
+ /** `closenessCentrality` honours `harmonic` on an exact run, so the CPU dispatcher sends harmonic closeness here. */
134
+ readonly harmonicCloseness: true;
133
135
  readonly ctx: GpuContext;
134
136
  readonly options: Readonly<AcceleratorOptions>;
135
137
  forceAtlas2(options?: ForceAtlas2Options): GpuLayoutSimulation<ForceAtlas2Options, ForceAtlas2Stats>;
@@ -0,0 +1,86 @@
1
+ /**
2
+ * The types of `acquireAccelerator` (the `./acquire` subpath, and the same function on `./browser` and `./node`):
3
+ * one call that finds a device, checks that it computes correctly and hands back an accelerator, or says why it
4
+ * did not. Types only.
5
+ */
6
+
7
+ import type { AcceleratorOptions, GpuAccelerator } from "./accelerator.js";
8
+ import type { AdapterSummary, DeviceCheck } from "./context.js";
9
+
10
+ /**
11
+ * Options of `acquireAccelerator`.
12
+ * @public
13
+ */
14
+ export interface AcquireAcceleratorOptions {
15
+ /**
16
+ * Accept a software adapter (llvmpipe, SwiftShader, WARP). Default false: a software adapter is usually slower
17
+ * than a CPU implementation, so it is declined with `E_SOFTWARE_ONLY`.
18
+ */
19
+ readonly acceptSoftware?: boolean | undefined;
20
+ /** The adapter power preference; default "high-performance". */
21
+ readonly powerPreference?: GPUPowerPreference | undefined;
22
+ /** Passed to `createAccelerator` (layout tuning, betweenness defaults). */
23
+ readonly accelerator?: AcceleratorOptions | undefined;
24
+ /** Warn once when more than this many snapshots stay resident on the device (the context's default when absent). */
25
+ readonly warnUnreleasedSnapshots?: number | undefined;
26
+ /** Node only: a substring of the Dawn adapter name to pick (`"llvmpipe"`, `"4070"`). Ignored in a browser. */
27
+ readonly adapter?: string | undefined;
28
+ }
29
+
30
+ /**
31
+ * A verified accelerator.
32
+ * @public
33
+ */
34
+ export interface AcceleratorReady {
35
+ readonly ok: true;
36
+ readonly code: "OK";
37
+ /** The accelerator. Its `ctx.lost` resolves when the device is lost; the handle then acquires a new one. */
38
+ readonly accelerator: GpuAccelerator;
39
+ }
40
+
41
+ /**
42
+ * Why no accelerator was handed over. Every code here is decided before any work runs, so a caller that has a CPU
43
+ * implementation runs it and reports the reason; that is detection, not a fallback.
44
+ * @public
45
+ */
46
+ export interface AcceleratorDeclined {
47
+ readonly ok: false;
48
+ /**
49
+ * E_NO_WEBGPU: this runtime has no WebGPU (in Node: the optional `webgpu` package is missing);
50
+ * E_NO_ADAPTER: WebGPU exists but no adapter answered; E_SOFTWARE_ONLY: the adapter is a software renderer and
51
+ * `acceptSoftware` was not set; E_DEVICE_INCORRECT: the device got a known answer wrong in the self-check.
52
+ */
53
+ readonly code: "E_NO_WEBGPU" | "E_NO_ADAPTER" | "E_SOFTWARE_ONLY" | "E_DEVICE_INCORRECT";
54
+ /** The reason in words. */
55
+ readonly reason: string;
56
+ /** What the user can change to get the GPU, or null when nothing they can do would help. */
57
+ readonly fix: string | null;
58
+ /** The adapter that was found and declined, when one was. */
59
+ readonly adapter: AdapterSummary | null;
60
+ /** The self-check record of an E_DEVICE_INCORRECT decline (what disagreed); null otherwise. */
61
+ readonly check: DeviceCheck | null;
62
+ }
63
+
64
+ /**
65
+ * What `ManagedAccelerator.current()` resolves to.
66
+ * @public
67
+ */
68
+ export type AcquireResult = AcceleratorReady | AcceleratorDeclined;
69
+
70
+ /**
71
+ * The handle `acquireAccelerator` returns. It owns the device: probing, the device self-check, re-acquiring after
72
+ * device loss, and disposal.
73
+ * @public
74
+ */
75
+ export interface ManagedAccelerator {
76
+ /**
77
+ * The accelerator, or why there is none. The first call acquires (probe, context, self-check); later calls
78
+ * return the same answer until the device is lost or the accelerator is disposed, after which the next call
79
+ * acquires a new device. Concurrent calls share one acquisition. A decline is remembered for the life of the
80
+ * handle. Rejects only for a failure that is not a decline (a device request that failed, a check that could not
81
+ * run), and with E_DISPOSED after `dispose()`.
82
+ */
83
+ current(): Promise<AcquireResult>;
84
+ /** Disposes the current accelerator and its device; idempotent. */
85
+ dispose(): void;
86
+ }
@@ -2,7 +2,7 @@
2
2
  * The `bc-forward` kernel body (design 8.4 "forward pass = BFS with sigma as array<atomic<u32>>", 8.10 "BC forward
3
3
  * (tagged)", 16.1): one level of the tagged multi-source breadth-first search of a betweenness batch. The level's
4
4
  * frontier is the range `S[ends[level] .. ends[level + 1])` of the claim log, every entry a packed `s * n + u`; the
5
- * expansion is `closeness-sweep`'s block-mapped strip (each workgroup loads up to `WG` entries, scans their degrees
5
+ * expansion is `advance-expand`'s block-mapped strip (each workgroup loads up to `WG` entries, scans their degrees
6
6
  * in workgroup memory, and every lane strips the aggregate by an upper-bound binary search), fused with the claim, so
7
7
  * no edge queue exists.
8
8
  *
@@ -0,0 +1,203 @@
1
+ /**
2
+ * The `closeness-level` kernel body: one level of the bit-parallel multi-source breadth-first search of closeness, in
3
+ * ONE dispatch, plus the two seed roles of a batch. A batch runs `32 x P.words` sources at once: bit `b` of word `j`
4
+ * of node `v` is source lane `32 j + b`. The `bits` buffer holds five regions of `P.base` words, `P.words` words per
5
+ * node: `visited` at 0, three frontier regions at `P.base`, `2 P.base` and `3 P.base` that rotate by the level
6
+ * (level `L` reads region `1 + L % 3`, writes region `1 + (L + 1) % 3` and zeroes region `1 + (L + 2) % 3`, the
7
+ * frontier of level `L - 1` that nothing reads any more, so no fill runs between levels; a stale bit could never
8
+ * claim anything, since its node's neighbours were claimed the level after, so the clear only keeps a later frontier
9
+ * from re-walking old nodes), then a sampled run's
10
+ * per-node distance sums at `4 P.base`. The `table` buffer holds the per-level claim counts of the submit (row `P.row`
11
+ * at `P.row x 32 P.words`, one word per source lane), then a ring of three control slots of four words at `P.ctrl`
12
+ * (`any` @0: the level claimed something; `arcs` @1: the out-degree of every (node, word) that joined the next
13
+ * frontier; `pull` @2: `arcs` passed `P.pullAt`), then a sampled run's source list at `P.sourcesAt`.
14
+ *
15
+ * Role 0 (one invocation per node and per arc, `max(n, arcs)` in all): a level whose predecessor claimed nothing
16
+ * returns at once (one uniform load per workgroup), so the host can record more levels than the batch needs.
17
+ * Otherwise the level chooses its step from the slot its predecessor filled, the same way for every invocation:
18
+ * - PUSH (the frontier is cheap to expand): one invocation per out-arc `(u, x)` -- `u` found by a binary search of
19
+ * `rowPtr` -- claims the sources at `u` that `x` has not seen with `atomicOr` on `x`'s visited word; the bits it
20
+ * won (`fresh`) join the next frontier;
21
+ * - PULL (the frontier's arcs pass `P.pullAt`, and `P.pullOk`): one invocation per node not yet reached by every
22
+ * source walks its IN-arcs, ORs the neighbours' frontier words, and keeps the bits it had not seen; it stops at
23
+ * the first in-arc after which every source has reached it (the early exit). Only the owner writes a node's
24
+ * words. The host allows the pull only when no node has more than a few thousand in-arcs, so no
25
+ * invocation of either step loops more than a few tens of thousands of times (llvmpipe silently ends every loop
26
+ * of an invocation past 65,535 iterations).
27
+ * Every claimed bit is tallied per source lane in workgroup memory and flushed with one global `atomicAdd` per lane
28
+ * per workgroup into the level's row; the host turns the counts into exact sums (`count x (L + 1)`) and harmonic
29
+ * sums (`count / (L + 1)`). Role 1 (one invocation per word): zeroes the regions, sets every dead lane of a partial
30
+ * batch as already visited (so the early exit and the "reached by every source" test see a full word), and zeroes the
31
+ * control ring. Role 2 (one invocation per lane): seeds lane `b`'s source -- `P.source + b`, or word `P.source + b` of
32
+ * the source list -- into `visited` and level 0's frontier with `atomicOr` (a node listed twice carries both bits),
33
+ * and fills the control slot level 0 reads. Body only; the text is normative: the sabotage rows of
34
+ * test/helpers/sabotage.ts are textual edits of it.
35
+ */
36
+ export const closenessLevelWgsl = /* wgsl */ `
37
+ const max_words: u32 = 8u;
38
+ var<workgroup> tally: array<atomic<u32>, 32u * max_words>; // 32 x max_words: this workgroup's claims per source lane
39
+ var<workgroup> wlive: u32;
40
+ var<workgroup> wpull: u32;
41
+ var<workgroup> wany: atomic<u32>;
42
+ var<workgroup> warcs: atomic<u32>;
43
+ var<workgroup> wover: atomic<u32>;
44
+
45
+ fn dead_lanes(j: u32) -> u32 { // the lanes of word j at or past the batch's source count
46
+ let first = 32u * j;
47
+ if (P.count >= first + 32u) { return 0u; }
48
+ if (P.count <= first) { return U32_MAX; }
49
+ return ~((1u << (P.count - first)) - 1u);
50
+ }
51
+
52
+ fn add_arcs(slot: u32, value: u32) { // the arcs total of a control slot; pull once it passes P.pullAt
53
+ if (value == 0u) { return; }
54
+ let before = atomicAdd(&table[slot + 1u], value);
55
+ if (value > P.pullAt || before + value > P.pullAt || before + value < before) {
56
+ atomicStore(&table[slot + 2u], 1u);
57
+ }
58
+ }
59
+
60
+ fn record(j: u32, node: u32, fresh: u32, dist: u32) { // the claims of one word: tallied per lane, a sampled run's sums
61
+ if (P.perNode == 1u) { atomicAdd(&bits[4u * P.base + node], countOneBits(fresh) * dist); }
62
+ var b = fresh;
63
+ loop {
64
+ if (b == 0u) { break; }
65
+ atomicAdd(&tally[32u * j + firstTrailingBit(b)], 1u);
66
+ b = b & (b - 1u);
67
+ }
68
+ }
69
+
70
+ @compute @workgroup_size(WG)
71
+ fn closeness_level(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
72
+ let i = linear_id(wid, lid.x);
73
+ let W = P.words;
74
+ if (P.role == 1u) { // seed, part 1: one invocation per word
75
+ if (i == 0u) {
76
+ for (var k = 0u; k < 12u; k = k + 1u) { atomicStore(&table[P.ctrl + k], 0u); }
77
+ }
78
+ if (i < P.total) { atomicStore(&bits[i], select(0u, dead_lanes(i % W), i < P.base)); }
79
+ return; // uniform: P.role is
80
+ }
81
+ if (P.role == 2u) { // seed, part 2: one invocation per source lane
82
+ if (i < P.count) {
83
+ var v = P.source + i;
84
+ if (P.sourcesAt != 0u) { v = atomicLoad(&table[P.sourcesAt + P.source + i]); }
85
+ let w = v * W + i / 32u;
86
+ let bit = 1u << (i % 32u);
87
+ atomicOr(&bits[w], bit); // visited
88
+ let before = atomicOr(&bits[P.base + w], bit); // level 0's frontier (region 1)
89
+ let slot = P.ctrl + 8u; // the slot of "level -1", which level 0 reads
90
+ atomicStore(&table[slot], 1u);
91
+ if (before == 0u) { add_arcs(slot, rowPtr[v + 1u] - rowPtr[v]); }
92
+ }
93
+ return;
94
+ }
95
+
96
+ // role 0: one level
97
+ let L = P.level;
98
+ let prev = P.ctrl + 4u * ((L + 2u) % 3u);
99
+ let cur = P.ctrl + 4u * (L % 3u);
100
+ if (lid.x == 0u) {
101
+ wlive = atomicLoad(&table[prev]);
102
+ wpull = select(0u, atomicLoad(&table[prev + 2u]), P.pullOk == 1u);
103
+ atomicStore(&wany, 0u);
104
+ atomicStore(&warcs, 0u);
105
+ atomicStore(&wover, 0u);
106
+ }
107
+ for (var k = lid.x; k < 32u * W; k = k + WG) { atomicStore(&tally[k], 0u); }
108
+ let live = workgroupUniformLoad(&wlive); // uniform; includes a barrier
109
+ if (live == 0u) { return; } // the previous level claimed nothing
110
+ let pull = workgroupUniformLoad(&wpull) == 1u;
111
+ if (i == 0u) { // the slot of level L + 1 held level L - 2's
112
+ let nxt = P.ctrl + 4u * ((L + 1u) % 3u);
113
+ atomicStore(&table[nxt], 0u);
114
+ atomicStore(&table[nxt + 1u], 0u);
115
+ atomicStore(&table[nxt + 2u], 0u);
116
+ }
117
+ let frontierBase = P.base * (1u + L % 3u);
118
+ let nextBase = P.base * (1u + (L + 1u) % 3u);
119
+ let staleBase = P.base * (1u + (L + 2u) % 3u);
120
+ let dist = L + 1u;
121
+ var arcs = 0u;
122
+ if (i < P.n) {
123
+ for (var j = 0u; j < W; j = j + 1u) { atomicStore(&bits[staleBase + i * W + j], 0u); }
124
+ }
125
+ if (pull) {
126
+ if (i < P.n) { // one invocation per node: its in-arcs
127
+ var vis: array<u32, max_words>;
128
+ var unseen = 0u;
129
+ for (var j = 0u; j < W; j = j + 1u) {
130
+ vis[j] = atomicLoad(&bits[i * W + j]);
131
+ unseen = unseen | ~vis[j];
132
+ }
133
+ if (unseen != 0u) { // some source has not reached this node yet
134
+ var acc: array<u32, max_words>;
135
+ let end = inRowPtr[i + 1u];
136
+ for (var a = inRowPtr[i]; a < end; a = a + 1u) {
137
+ let u = inColIdx[a];
138
+ var missing = 0u;
139
+ for (var j = 0u; j < W; j = j + 1u) {
140
+ acc[j] = acc[j] | atomicLoad(&bits[frontierBase + u * W + j]);
141
+ missing = missing | ~(acc[j] | vis[j]);
142
+ }
143
+ if (missing == 0u) { break; } // every source has reached it: the early exit
144
+ }
145
+ let degree = rowPtr[i + 1u] - rowPtr[i];
146
+ for (var j = 0u; j < W; j = j + 1u) {
147
+ let fresh = acc[j] & ~vis[j];
148
+ if (fresh != 0u) {
149
+ atomicStore(&bits[i * W + j], vis[j] | fresh);
150
+ atomicStore(&bits[nextBase + i * W + j], fresh);
151
+ arcs = arcs + degree;
152
+ record(j, i, fresh, dist);
153
+ }
154
+ }
155
+ }
156
+ }
157
+ } else if (i < P.arcCount) { // one invocation per arc: no row is walked whole
158
+ var lo = 0u; // the arc's source: the last row starting at or before it
159
+ var hi = P.n - 1u;
160
+ loop {
161
+ if (lo >= hi) { break; }
162
+ let mid = (lo + hi + 1u) / 2u;
163
+ if (rowPtr[mid] <= i) { lo = mid; } else { hi = mid - 1u; }
164
+ }
165
+ let u = lo;
166
+ let x = colIdx[i];
167
+ for (var j = 0u; j < W; j = j + 1u) {
168
+ let mask = atomicLoad(&bits[frontierBase + u * W + j]) & ~atomicLoad(&bits[x * W + j]); // at u, not yet at x
169
+ if (mask == 0u) { continue; }
170
+ let fresh = mask & ~atomicOr(&bits[x * W + j], mask); // the claims this invocation won
171
+ if (fresh == 0u) { continue; }
172
+ if (atomicOr(&bits[nextBase + x * W + j], fresh) == 0u) {
173
+ arcs = arcs + (rowPtr[x + 1u] - rowPtr[x]);
174
+ }
175
+ record(j, x, fresh, dist);
176
+ }
177
+ }
178
+ if (arcs > P.pullAt) {
179
+ atomicStore(&wover, 1u);
180
+ } else if (arcs != 0u) {
181
+ let before = atomicAdd(&warcs, arcs);
182
+ if (before + arcs > P.pullAt || before + arcs < before) { atomicStore(&wover, 1u); }
183
+ }
184
+ workgroupBarrier();
185
+ let row = P.row * 32u * W;
186
+ for (var k = lid.x; k < 32u * W; k = k + WG) { // ONE global atomic per source lane per workgroup
187
+ let c = atomicLoad(&tally[k]);
188
+ if (c != 0u) {
189
+ atomicAdd(&table[row + k], c);
190
+ atomicStore(&wany, 1u);
191
+ }
192
+ }
193
+ workgroupBarrier();
194
+ if (lid.x == 0u) {
195
+ if (atomicLoad(&wany) != 0u) { atomicStore(&table[cur], 1u); }
196
+ if (atomicLoad(&wover) != 0u) {
197
+ atomicStore(&table[cur + 2u], 1u);
198
+ } else {
199
+ add_arcs(cur, atomicLoad(&warcs));
200
+ }
201
+ }
202
+ }
203
+ `;
@@ -0,0 +1,41 @@
1
+ /**
2
+ * The `closeness-rowsum` kernel body: closeness from a finished all-pairs distance matrix, one workgroup per row.
3
+ * Every lane walks its strided columns of row `r` (the distances FROM `r`), skips the diagonal and every unreachable
4
+ * entry (`+Infinity`, compared by its bit pattern because WGSL lets a compiler assume no infinities), and adds, by
5
+ * `P.role`: 0 the hop count as an integer (exact: a row of at most 23,170 hops below 23,170 sums below 2^32), 1 the
6
+ * f32 distance, 2 its reciprocal (harmonic closeness; a zero distance adds nothing, as in the CPU port). A tree reduction in workgroup memory folds the lanes and lane
7
+ * 0 writes `out[r]` -- the integer, or the f32 bit pattern. The host turns the row into the score. Body only; the text
8
+ * is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
9
+ */
10
+ export const closenessRowsumWgsl = /* wgsl */ `
11
+ var<workgroup> partial: array<u32, WG>;
12
+
13
+ @compute @workgroup_size(WG)
14
+ fn closeness_rowsum(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
15
+ let row = group_id(wid); // uniform: one workgroup per row
16
+ if (row >= P.n) { return; }
17
+ var whole = 0u;
18
+ var real = 0.0;
19
+ for (var c = lid.x; c < P.n; c = c + WG) {
20
+ let d = dist[row * P.n + c];
21
+ if (c == row || bitcast<u32>(d) == F32_INF_BITS) { continue; }
22
+ if (P.role == 0u) {
23
+ whole = whole + u32(d);
24
+ } else if (P.role == 1u) {
25
+ real = real + d;
26
+ } else if (d > 0.0) {
27
+ real = real + 1.0 / d; // a zero distance adds nothing, as on the CPU
28
+ }
29
+ }
30
+ partial[lid.x] = select(whole, bitcast<u32>(real), P.role != 0u);
31
+ for (var s = WG / 2u; s > 0u; s = s / 2u) {
32
+ workgroupBarrier();
33
+ if (lid.x < s) {
34
+ let a = partial[lid.x];
35
+ let b = partial[lid.x + s];
36
+ partial[lid.x] = select(a + b, bitcast<u32>(bitcast<f32>(a) + bitcast<f32>(b)), P.role != 0u);
37
+ }
38
+ }
39
+ if (lid.x == 0u) { out[row] = partial[0]; }
40
+ }
41
+ `;
@@ -1,6 +1,6 @@
1
1
  /**
2
- * G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per hub cell of hubList, dispatched
3
- * indirectly from hubArgs (the T1 finalize over hubCounters[0]); a WG-strided mass-weighted sum reduced by the
2
+ * G4b, the `grid-centroid-hub` kernel body (spec 7.7; P4-T9): one workgroup per word of hubList, dispatched
3
+ * directly (issue #732), so the workgroups past hubCounters[0] idle; a WG-strided mass-weighted sum reduced by the
4
4
  * prelude's tree. The work is guarded by `valid`, never an early return, so the reduction is uniform (PD-13). Body
5
5
  * only; normative text.
6
6
  */