@graphty/webgpu-graph-algorithms 0.6.5 → 0.6.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/README.md +38 -17
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-hzGggHeM.js → context-Cezi7qpi.js} +46 -22
  4. package/dist/chunks/context-Cezi7qpi.js.map +1 -0
  5. package/dist/node.js +19 -11
  6. package/dist/node.js.map +1 -1
  7. package/dist/src/algorithms/bfs.d.ts +10 -6
  8. package/dist/src/algorithms/bfs.d.ts.map +1 -1
  9. package/dist/src/algorithms/bfs.js +32 -9
  10. package/dist/src/algorithms/bfs.js.map +1 -1
  11. package/dist/src/algorithms/pagerank.d.ts.map +1 -1
  12. package/dist/src/algorithms/pagerank.js +19 -5
  13. package/dist/src/algorithms/pagerank.js.map +1 -1
  14. package/dist/src/algorithms/power-iteration.d.ts.map +1 -1
  15. package/dist/src/algorithms/power-iteration.js +8 -2
  16. package/dist/src/algorithms/power-iteration.js.map +1 -1
  17. package/dist/src/constants.d.ts +33 -0
  18. package/dist/src/constants.d.ts.map +1 -1
  19. package/dist/src/constants.js +33 -0
  20. package/dist/src/constants.js.map +1 -1
  21. package/dist/src/kernel/dispatch.d.ts +2 -2
  22. package/dist/src/kernel/dispatch.d.ts.map +1 -1
  23. package/dist/src/kernel/kernel.d.ts +1 -1
  24. package/dist/src/kernel/kernel.js +2 -2
  25. package/dist/src/kernel/kernel.js.map +1 -1
  26. package/dist/src/kernel/prelude.d.ts.map +1 -1
  27. package/dist/src/kernel/prelude.js +2 -1
  28. package/dist/src/kernel/prelude.js.map +1 -1
  29. package/dist/src/kernels.d.ts +10 -7
  30. package/dist/src/kernels.d.ts.map +1 -1
  31. package/dist/src/kernels.js +33 -9
  32. package/dist/src/kernels.js.map +1 -1
  33. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  34. package/dist/src/layouts/forceatlas2.js +2 -1
  35. package/dist/src/layouts/forceatlas2.js.map +1 -1
  36. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  37. package/dist/src/layouts/fruchterman-reingold.js +4 -2
  38. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  39. package/dist/src/layouts/repulsion-exact.d.ts +16 -0
  40. package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
  41. package/dist/src/layouts/repulsion-exact.js +21 -1
  42. package/dist/src/layouts/repulsion-exact.js.map +1 -1
  43. package/dist/src/layouts/repulsion-grid.d.ts +1 -1
  44. package/dist/src/layouts/repulsion-grid.js +1 -1
  45. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  46. package/dist/src/layouts/spring-electrical.js +6 -2
  47. package/dist/src/layouts/spring-electrical.js.map +1 -1
  48. package/dist/src/memory/residency.js +14 -4
  49. package/dist/src/memory/residency.js.map +1 -1
  50. package/dist/src/node/index.d.ts +13 -8
  51. package/dist/src/node/index.d.ts.map +1 -1
  52. package/dist/src/node/index.js +36 -17
  53. package/dist/src/node/index.js.map +1 -1
  54. package/dist/src/primitives/frontier.d.ts +1 -0
  55. package/dist/src/primitives/frontier.d.ts.map +1 -1
  56. package/dist/src/primitives/frontier.js +1 -0
  57. package/dist/src/primitives/frontier.js.map +1 -1
  58. package/dist/src/primitives/grid-pyramid.d.ts +4 -4
  59. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  60. package/dist/src/primitives/grid-pyramid.js +4 -3
  61. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  62. package/dist/src/primitives/grid.d.ts +13 -10
  63. package/dist/src/primitives/grid.d.ts.map +1 -1
  64. package/dist/src/primitives/grid.js +10 -7
  65. package/dist/src/primitives/grid.js.map +1 -1
  66. package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
  67. package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
  68. package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
  69. package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
  70. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
  71. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
  72. package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
  73. package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
  74. package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
  75. package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
  76. package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
  77. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
  78. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
  79. package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
  80. package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
  81. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
  82. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
  83. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
  84. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
  85. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
  86. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
  87. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
  88. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
  89. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +21 -20
  90. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
  91. package/dist/src/wgsl/frontier-finalize.wgsl.js +26 -25
  92. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
  93. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
  94. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
  95. package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
  96. package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
  97. package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
  98. package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
  99. package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
  100. package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
  101. package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
  102. package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
  103. package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
  104. package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
  105. package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
  106. package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
  107. package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
  108. package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
  109. package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
  110. package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
  111. package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
  112. package/dist/src/wgsl/histogram.wgsl.js +1 -1
  113. package/dist/webgpu-graph-algorithms.js +144 -45
  114. package/dist/webgpu-graph-algorithms.js.map +1 -1
  115. package/package.json +3 -3
  116. package/src/algorithms/bfs.ts +33 -9
  117. package/src/algorithms/pagerank.ts +19 -5
  118. package/src/algorithms/power-iteration.ts +8 -2
  119. package/src/constants.ts +35 -0
  120. package/src/kernel/dispatch.ts +2 -2
  121. package/src/kernel/kernel.ts +2 -2
  122. package/src/kernel/prelude.ts +2 -0
  123. package/src/kernels.ts +35 -9
  124. package/src/layouts/forceatlas2.ts +2 -0
  125. package/src/layouts/fruchterman-reingold.ts +4 -1
  126. package/src/layouts/repulsion-exact.ts +29 -1
  127. package/src/layouts/repulsion-grid.ts +1 -1
  128. package/src/layouts/spring-electrical.ts +8 -1
  129. package/src/memory/residency.ts +14 -4
  130. package/src/node/index.ts +42 -18
  131. package/src/primitives/frontier.ts +2 -0
  132. package/src/primitives/grid-pyramid.ts +6 -5
  133. package/src/primitives/grid.ts +17 -12
  134. package/src/wgsl/advance-expand.wgsl.ts +4 -3
  135. package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
  136. package/src/wgsl/bfs-fused.wgsl.ts +6 -6
  137. package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
  138. package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
  139. package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
  140. package/src/wgsl/frontier-finalize.wgsl.ts +26 -25
  141. package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
  142. package/src/wgsl/grid-centroid.wgsl.ts +3 -2
  143. package/src/wgsl/grid-downsample.wgsl.ts +1 -1
  144. package/src/wgsl/grid-far-field.wgsl.ts +15 -8
  145. package/src/wgsl/grid-near-field.wgsl.ts +6 -2
  146. package/src/wgsl/histogram.wgsl.ts +1 -1
  147. package/dist/chunks/context-hzGggHeM.js.map +0 -1
@@ -579,7 +579,9 @@ export class GraphResidency {
579
579
  }
580
580
 
581
581
  /**
582
- * One resident per array (spec 4.3: views upload in perArray mode, never into the arena).
582
+ * One resident per array (spec 4.3: views upload in perArray mode, never into the arena). An empty array (the
583
+ * colIdx of an edgeless directed reverse view, the src / dst of an edgeless edgeList) is skipped: spec 5.6 never
584
+ * uploads a zero-length array, and Kernel.bind rejects a zero-size binding, so it is absent as in core().
583
585
  * @param record - the owning record
584
586
  * @param arrays - the named arrays
585
587
  * @param label - the buffer label prefix
@@ -592,6 +594,9 @@ export class GraphResidency {
592
594
  ): Readonly<Record<string, Binding>> {
593
595
  const bindings: Record<string, Binding> = {};
594
596
  for (const [name, array] of arrays) {
597
+ if (array.byteLength === 0) {
598
+ continue;
599
+ }
595
600
  const resident = this.upload(record, array, array, `${label}:${name}`);
596
601
  bindings[name] = { buffer: resident.buffer, offset: 0, size: resident.byteLength, window: null };
597
602
  }
@@ -606,19 +611,24 @@ export class GraphResidency {
606
611
  * lengths, so keying the packed buffer on `rev.rowPtr` would make the packed and the unpacked view of one
607
612
  * snapshot collide -- whichever was built second would get the other's buffer. The record still owns the
608
613
  * resident, so release(s) destroys it with the rest. Offsets are STORAGE_ALIGN-aligned because Kernel.bind
609
- * rejects any other offset synchronously (E_INVALID_ARGUMENT { argument: "offset" }).
614
+ * rejects any other offset synchronously (E_INVALID_ARGUMENT { argument: "offset" }). Empty arrays are left
615
+ * out as in separateArrays; when nothing is left, nothing is uploaded.
610
616
  * @param record - the owning record
611
- * @param arrays - the named arrays, in buffer order
617
+ * @param all - the named arrays, in buffer order
612
618
  * @param key - the marker object the resident is keyed on
613
619
  * @param label - the buffer label
614
620
  * @returns the bindings by name, all into the one buffer
615
621
  */
616
622
  private packArrays(
617
623
  record: ResidencyRecord,
618
- arrays: readonly (readonly [string, TypedArrayData])[],
624
+ all: readonly (readonly [string, TypedArrayData])[],
619
625
  key: object,
620
626
  label: string,
621
627
  ): Readonly<Record<string, Binding>> {
628
+ const arrays = all.filter(([, array]) => array.byteLength > 0);
629
+ if (arrays.length === 0) {
630
+ return Object.freeze({});
631
+ }
622
632
  const offsets: number[] = [];
623
633
  let total = 0;
624
634
  for (const [, array] of arrays) {
package/src/node/index.ts CHANGED
@@ -32,11 +32,15 @@ export interface NodeGpuOptions extends Omit<GpuContextOptions, "gpu" | "adapter
32
32
  readonly loadModule?: (() => Promise<unknown>) | undefined;
33
33
  }
34
34
 
35
- /** The Dawn GPU handle (spec 2.3): dispose() drops the reference so the process can exit. */
35
+ /**
36
+ * The Dawn GPU handle (spec 2.3). The Dawn instance behind `gpu` is shared by every handle created with the
37
+ * same flags and lives until the process exits (see `createNodeGpu`); it holds no event-loop handle, so it
38
+ * never keeps the process alive.
39
+ */
36
40
  export interface NodeGpuHandle {
37
41
  /** The GPU of `dawn.create(flags)`; reading it after dispose() throws E_DISPOSED. */
38
42
  readonly gpu: GPU;
39
- /** Drops the GPU reference; idempotent. */
43
+ /** Drops this handle's GPU reference (the shared instance stays alive); idempotent. */
40
44
  dispose(): void;
41
45
  }
42
46
 
@@ -49,6 +53,18 @@ interface DawnModule {
49
53
  globals?: unknown;
50
54
  }
51
55
 
56
+ /**
57
+ * Every GPU object `dawn.create()` returned, per module and flag list, kept for the life of the process.
58
+ * webgpu@0.4.0's adapters, devices and queues run their promises through an AsyncRunner that polls the Dawn
59
+ * instance by RAW pointer, and only the GPU object owns that instance: once the GPU object is collected, the
60
+ * next promise on any adapter or device it produced (a requestDevice on a probed adapter, a queue call or a
61
+ * late map / lost callback of a destroyed device) polls freed memory -- SIGSEGV in
62
+ * dawn::native::InstanceBase::ProcessEvents, on Metal and lavapipe alike (issue #30). dawn-node signals no
63
+ * point at which the instance has drained, so no GPU object is ever released; sharing one per flag list
64
+ * bounds what that keeps to one instance per configuration.
65
+ */
66
+ const instances = new Map<DawnModule, Map<string, GPU>>();
67
+
52
68
  /**
53
69
  * Whether a loaded module is usable as Dawn.
54
70
  * @param loaded - the module namespace
@@ -125,9 +141,11 @@ export function dawnFlags(options: NodeGpuOptions | undefined): string[] {
125
141
  }
126
142
 
127
143
  /**
128
- * import("webgpu"), install dawn.globals unless installGlobals === false, dawn.create(flags) (spec 2.3).
144
+ * import("webgpu"), install dawn.globals unless installGlobals === false, dawn.create(flags) (spec 2.3). The GPU
145
+ * object is created once per flag list and reused by every later call with the same flags; it is never released,
146
+ * because Dawn keeps polling its instance for the adapters and devices it produced (issue #30).
129
147
  * @param options - adapter / backend / dawnFeatures / software / installGlobals (and the test seam)
130
- * @returns the handle; `dispose()` drops the GPU reference so the process can exit
148
+ * @returns the handle; `dispose()` drops the handle's reference, never the shared instance
131
149
  * @throws WebGpuGraphError E_NO_WEBGPU { reason, hint } when the module does not load (missing, or its glibc is too old), has no create(), or create(flags) throws
132
150
  */
133
151
  export async function createNodeGpu(options?: NodeGpuOptions): Promise<NodeGpuHandle> {
@@ -150,12 +168,22 @@ export async function createNodeGpu(options?: NodeGpuOptions): Promise<NodeGpuHa
150
168
  if (options?.installGlobals !== false && typeof loaded.globals === "object" && loaded.globals !== null) {
151
169
  Object.assign(globalThis, loaded.globals);
152
170
  }
153
- let gpu: GPU;
154
- try {
155
- gpu = loaded.create(dawnFlags(options));
156
- } catch (err) {
157
- const reason = `dawn.create() threw: ${messageOf(err)}`;
158
- throw new WebGpuGraphError("E_NO_WEBGPU", `${reason}; ${INSTALL_HINT}`, { reason, hint: INSTALL_HINT });
171
+ const flags = dawnFlags(options);
172
+ const key = flags.join("\n");
173
+ let byFlags = instances.get(loaded);
174
+ if (byFlags === undefined) {
175
+ byFlags = new Map();
176
+ instances.set(loaded, byFlags);
177
+ }
178
+ let gpu = byFlags.get(key);
179
+ if (gpu === undefined) {
180
+ try {
181
+ gpu = loaded.create(flags);
182
+ } catch (err) {
183
+ const reason = `dawn.create() threw: ${messageOf(err)}`;
184
+ throw new WebGpuGraphError("E_NO_WEBGPU", `${reason}; ${INSTALL_HINT}`, { reason, hint: INSTALL_HINT });
185
+ }
186
+ byFlags.set(key, gpu);
159
187
  }
160
188
  return new DawnHandle(gpu);
161
189
  }
@@ -181,10 +209,9 @@ function contextOptionsOf(options: NodeGpuOptions): Omit<GpuContextOptions, "gpu
181
209
 
182
210
  /**
183
211
  * createNodeGpu + GpuContext.create({ gpu, runtime: "node", ...options }); ctx.dispose() also disposes the
184
- * handle, and a create() failure disposes it before rethrowing. The handle is dropped once `ctx.lost` has
185
- * settled, never before: under webgpu@0.4.0 a GPU object collected while the device it created is still
186
- * tearing down (its lost / work-done callbacks in flight) crashes or deadlocks the process (PLAN DECISION,
187
- * P1-T1; measured with tmp/p1t1/gc-race2.mjs), and device.destroy() reports the loss right away.
212
+ * handle, and a create() failure disposes it before rethrowing. Disposing is safe at any moment: the Dawn
213
+ * instance itself stays alive for the process (createNodeGpu), so a callback of the destroyed device that
214
+ * arrives late, or a call on `ctx.device` after dispose(), never reaches a freed instance.
188
215
  * @param options - the Node options
189
216
  * @returns the context
190
217
  */
@@ -197,11 +224,8 @@ export async function createNodeGpuContext(options?: NodeGpuOptions): Promise<Gp
197
224
  handle.dispose();
198
225
  throw err;
199
226
  }
200
- const { lost } = ctx;
201
227
  ctx.attachDisposer(() => {
202
- void lost.then(() => {
203
- handle.dispose();
204
- });
228
+ handle.dispose();
205
229
  });
206
230
  return ctx;
207
231
  }
@@ -71,6 +71,7 @@ export const W: Readonly<{
71
71
  thresholdBits: 22;
72
72
  deltaBits: 23;
73
73
  path: 24;
74
+ nextDegreeSum: 25;
74
75
  }> = Object.freeze({
75
76
  frontierCount: 0,
76
77
  nextFrontierCount: 1,
@@ -97,6 +98,7 @@ export const W: Readonly<{
97
98
  thresholdBits: 22,
98
99
  deltaBits: 23,
99
100
  path: 24,
101
+ nextDegreeSum: 25,
100
102
  });
101
103
 
102
104
  /** The words a `reset` seeds (every other word is zeroed). */
@@ -1,11 +1,11 @@
1
1
  /**
2
2
  * The grid pyramid (spec 6 row 12, 7.7 G4-G5; P4-T9): the planner that records, into the caller's pass, the finest
3
- * centroids (G4, `grid-centroid`: thread per cell over `cells + 1`, the pseudo-cell included), the hub-cell
3
+ * centroids (G4, `grid-centroid`: thread per cell over `cells + outsideCells`, the pseudo-cells included), the hub-cell
4
4
  * completion (G4a: the T1 `indirect-finalize` over `hubCounters[0]` into `hubArgs` with `wg = 1`, so the finalize's
5
5
  * `ceil(count / wg)` is ONE workgroup per hub cell; G4b: `grid-centroid-hub`, one workgroup per hub cell, dispatched
6
6
  * indirectly; PD-13, DEP-P4-I) and one `grid-downsample` dispatch per coarser
7
7
  * level (G5). Level 0 holds `[sum m x, sum m y, sum m z, sum m]` per cell; every parent is the sum of its 2^dim
8
- * children; the pseudo-cell (index `cells` of level 0) is never a child. No atomics touch the sums (design 6 row 12:
8
+ * children; the pseudo-cells (indices `cells ..` of level 0) are never children. No atomics touch the sums (design 6 row 12:
9
9
  * bitwise reproducible); the only atomics are the hub append and the occupancy max.
10
10
  *
11
11
  * The named grid buffers (`pyramid`, `hubList`, `hubCounters`, `hubArgs`) are the caller's (the model's
@@ -34,9 +34,9 @@ export interface GridPyramidBindings {
34
34
  readonly params: Binding;
35
35
  /** `n` words: the sorted node indices (the T8 build). */
36
36
  readonly sortedIdx: Binding;
37
- /** `cells + 2` words: the exclusive scan of the cell histogram (the T8 build). */
37
+ /** `histWords` (`cells + 2^dim + 1`) words: the exclusive scan of the cell histogram (the T8 build). */
38
38
  readonly cellStart: Binding;
39
- /** `pyramidCells` vec4f: every level, level 0 first with the pseudo-cell at index `cells`. */
39
+ /** `pyramidCells` vec4f: every level, level 0 first with the 2^dim orthant pseudo-cells from index `cells`. */
40
40
  readonly pyramid: Binding;
41
41
  /** The hub cells' indices, appended by G4 (at least one word; at most `floor(n / (GRID_HUB_CELL + 1))` are ever written, so `ceil(n / GRID_HUB_CELL)` words always suffice). */
42
42
  readonly hubList: Binding;
@@ -204,7 +204,8 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
204
204
  }
205
205
  const { centroid, finalize, hub, downsample } = this.kernels;
206
206
  const one: DispatchPlan = { x: 1, y: 1, z: 1, items: 1, stride: null };
207
- centroid.dispatch(pass, bound.centroid, plan1d(spec.cells + 1, scope.workgroupSize, scope.caps), [paramsOffset]);
207
+ const level0 = spec.cells + spec.outsideCells;
208
+ centroid.dispatch(pass, bound.centroid, plan1d(level0, scope.workgroupSize, scope.caps), [paramsOffset]);
208
209
  finalize.dispatch(pass, bound.finalize, one, [bound.finalizeOffset]);
209
210
  hub.dispatchIndirect(pass, bound.hub, bound.hubArgs, 0, [paramsOffset]);
210
211
  this.dispatches = 3;
@@ -3,8 +3,9 @@
3
3
  * the caller's pass, the cell keys (G1, `grid-cell-key`), the stable sort by key (G2: `radixSort` at GRID_SORT_BITS,
4
4
  * or `countingSortByKey` when the caller asks for the set-deterministic path) and the per-cell histogram with its
5
5
  * exclusive scan (G3: the `histogram` kernel over `cellKey` and the `scan` of it; DEP-P4-I names no grid-specific
6
- * id). `cellHist` and `cellStart` hold `cells + 2` words: every real cell, the outside pseudo-cell at index `cells`
7
- * and one more so `cellStart[cells + 1] === n` closes the last range. Every zeroing is a `fill` dispatch inside the
6
+ * id). `cellHist` and `cellStart` hold `histWords = cells + 2^dim + 1` words: every real cell, the 2^dim outside
7
+ * pseudo-cells (one per orthant about the grid centre, issue #90) from index `cells`, and one more so
8
+ * `cellStart[histWords - 1] === n` closes the last range. Every zeroing is a `fill` dispatch inside the
8
9
  * pass (PD-12), never an encoder clear.
9
10
  *
10
11
  * The named grid buffers (`cellKey`, `cellVal`, `sortedKey`, `sortedIdx`, `cellHist`, `cellStart`) are the caller's
@@ -39,11 +40,13 @@ export interface GridSpec {
39
40
  readonly g: number;
40
41
  /** `log2(G / GRID_COARSEST_SIDE) + 1`. */
41
42
  readonly levels: number;
42
- /** `G^dim` finest cells; the outside pseudo-cell is index `cells`. */
43
+ /** `G^dim` finest cells; the outside pseudo-cells are indices `cells .. cells + outsideCells - 1`. */
43
44
  readonly cells: number;
44
- /** `cells + 2`: the length of `cellHist` / `cellStart`. */
45
+ /** `2^dim`: one outside pseudo-cell per orthant about the grid centre (issue #90). */
46
+ readonly outsideCells: number;
47
+ /** `cells + outsideCells + 1`: the length of `cellHist` / `cellStart`. */
45
48
  readonly histWords: number;
46
- /** The first cell of every level inside the pyramid: `levelOffsets[0] = 0`, level 0 holds `cells + 1` (the pseudo-cell), level L `(G / 2^L)^dim`. */
49
+ /** The first cell of every level inside the pyramid: `levelOffsets[0] = 0`, level 0 holds `cells + outsideCells` (the pseudo-cells last), level L `(G / 2^L)^dim`. */
47
50
  readonly levelOffsets: readonly number[];
48
51
  /** Every level's cells together: `levelOffsets[levels - 1] + GRID_COARSEST_SIDE^dim`. */
49
52
  readonly pyramidCells: number;
@@ -81,8 +84,8 @@ function floorPow2(x: number): number {
81
84
  * The grid of `n` nodes in `dim` dimensions under the tuning (spec 7.7 geometry table; PD-9): `G = clamp(nextPow2(2 *
82
85
  * ceil(n^(1 / dim))), GRID_MIN_SIDE, floorPow2(gridMax))` where `gridMax` is `gridMax2D` or `gridMax3D`, rounded DOWN
83
86
  * to a power of two so every level's side is an integer (512 and 128 stay; 100 becomes 64); `levels = log2(G /
84
- * GRID_COARSEST_SIDE) + 1`. At the caps: 349,521 pyramid cells in 2D, 2,396,737 in 3D (the design's counts plus the
85
- * pseudo-cell).
87
+ * GRID_COARSEST_SIDE) + 1`. At the caps: 349,524 pyramid cells in 2D, 2,396,744 in 3D (the design's counts plus the
88
+ * 2^dim pseudo-cells).
86
89
  * @param n - the node count (>= 0)
87
90
  * @param dim - 2 or 3
88
91
  * @param tuning - the resolved layout tuning (`gridMax2D`, `gridMax3D`, `deterministic`)
@@ -102,10 +105,11 @@ export function gridSpecFor(
102
105
  levels++;
103
106
  }
104
107
  const cells = g ** dim;
108
+ const outsideCells = 2 ** dim;
105
109
  const levelOffsets: number[] = [0];
106
110
  let s = g;
107
111
  for (let level = 0; level + 1 < levels; level++) {
108
- levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? 1 : 0));
112
+ levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? outsideCells : 0));
109
113
  s /= 2;
110
114
  }
111
115
  return {
@@ -113,7 +117,8 @@ export function gridSpecFor(
113
117
  g,
114
118
  levels,
115
119
  cells,
116
- histWords: cells + 2,
120
+ outsideCells,
121
+ histWords: cells + outsideCells + 1,
117
122
  levelOffsets: Object.freeze(levelOffsets),
118
123
  pyramidCells: levelOffsets[levels - 1] + GRID_COARSEST_SIDE ** dim,
119
124
  deterministic: tuning.deterministic,
@@ -121,7 +126,7 @@ export function gridSpecFor(
121
126
  }
122
127
 
123
128
  /**
124
- * The bytes of the pyramid (spec 7.7: 16 B per cell, every level, the pseudo-cell included): 38,347,792 at the 3D cap.
129
+ * The bytes of the pyramid (spec 7.7: 16 B per cell, every level, the pseudo-cells included): 38,347,904 at the 3D cap.
125
130
  * @param spec - the grid
126
131
  * @returns the byte length
127
132
  */
@@ -155,9 +160,9 @@ export interface GridBuildBindings {
155
160
  readonly sortedKey: Binding;
156
161
  /** `n` words: the sorted node indices. */
157
162
  readonly sortedIdx: Binding;
158
- /** `cells + 2` words: the per-cell counts. */
163
+ /** `histWords` (`cells + 2^dim + 1`) words: the per-cell counts. */
159
164
  readonly cellHist: Binding;
160
- /** `cells + 2` words: the exclusive scan of `cellHist`. */
165
+ /** `histWords` (`cells + 2^dim + 1`) words: the exclusive scan of `cellHist`. */
161
166
  readonly cellStart: Binding;
162
167
  }
163
168
 
@@ -7,8 +7,9 @@
7
7
  * the twin's two compilations -- and then every invocation strips the range `[0, aggregate)` with a binary search
8
8
  * (`upper_bound`) over the scanned degrees to find which entry its arc belongs to. One `atomicAdd` per WORKGROUP
9
9
  * reserves the block's span in the queue (`edgeCount`), the same aggregate lands in `edgeCountUnclamped` (the overflow
10
- * detector, never clamped) and in `frontierDegreeSum` (Beamer's m_f); a lane whose queue position is at or past
11
- * `P.edgeCapacity` writes nothing (the clamp). The queue holds the TARGET vertex of each arc only (PD-24: `parent`
10
+ * detector, never clamped) and in `frontierDegreeSum` (the inspect seam's per-level expansion count, rotated into
11
+ * `prevDegreeSum` by the boundary; Beamer's m_f is `nextDegreeSum`, measured by `bfs-next-degree` -- issue #391);
12
+ * a lane whose queue position is at or past `P.edgeCapacity` writes nothing (the clamp). The queue holds the TARGET vertex of each arc only (PD-24: `parent`
12
13
  * comes from the post-pass). Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan call and the
13
14
  * `workgroupUniformLoad` sit unconditionally after the guard, and the strip loop is bounded by a uniform value.
14
15
  * There is no `TIER` override: a hub row is balanced over all `WG` lanes inside its block, and the small-frontier
@@ -44,7 +45,7 @@ fn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
44
45
  if (lid.x == 0u) {
45
46
  base = atomicAdd(&counters[8], aggregate); // edgeCount: ONE reservation per workgroup, not one per arc
46
47
  atomicAdd(&counters[9], aggregate); // edgeCountUnclamped: the overflow detector (PD-23)
47
- atomicAdd(&counters[2], aggregate); // frontierDegreeSum: Beamer's m_f (P8-T8)
48
+ atomicAdd(&counters[2], aggregate); // frontierDegreeSum: what this level expanded (the inspect seam)
48
49
  }
49
50
  workgroupBarrier();
50
51
  for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
@@ -11,9 +11,10 @@
11
11
  * The winners are packed into the output vertex queue by `bfs-contract`'s workgroup scan and one `atomicAdd` per
12
12
  * workgroup on `nextFrontierCount`, and claim with a plain `atomicStore`: the list holds every vertex once and the
13
13
  * sweep is vertex-parallel, so no two lanes claim one vertex. It adds nothing to `frontierDegreeSum` (a bottom-up
14
- * level expands nothing), which is why `unvisitedDegreeSum` stops falling while bottom-up runs (the selector's
15
- * JSDoc). Uniformity (spec 3.5 rule 1): the guarded walk writes locals, the scan and the reduction run
16
- * unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
14
+ * level expands nothing), which is why that word is not Beamer's m_f: `bfs-next-degree` sums the degree of what
15
+ * this sweep CLAIMS into `nextDegreeSum`, so the boundary's test and its `unvisitedDegreeSum` subtraction are exact
16
+ * on a bottom-up level like any other (issue #391; the selector's JSDoc). Uniformity (spec 3.5 rule 1): the
17
+ * guarded walk writes locals, the scan and the reduction run unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
17
18
  * test/helpers/sabotage.ts are textual edits of it.
18
19
  */
19
20
  export const bfsBottomUpWgsl = /* wgsl */ `
@@ -4,15 +4,15 @@
4
4
  * ONE dispatch, chosen by `frontier-finalize` for a frontier below `P.fusedMax` entries (`path` 2) and for the retry
5
5
  * of a level whose edge queue overflowed (`path` 4, PD-23). One WORKGROUP per frontier entry, the workgroups striding
6
6
  * the entries by the dispatch's group count (`P.stride`): lane 0 reads the entry's row clipped
7
- * to the bound arc window, adds its degree to `frontierDegreeSum` (Beamer's m_f, so P8-T8's test sees fused levels
8
- * too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim inline --
9
- * `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
7
+ * to the bound arc window, adds its degree to `frontierDegreeSum` (so the inspect seam's per-level expansion count
8
+ * covers fused levels too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim
9
+ * inline -- `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
10
10
  * packing the strip's winners into the output vertex queue by the same Hillis-Steele scan and one `atomicAdd` per
11
11
  * strip on `nextFrontierCount`. No edge queue is written or read, which is the whole win for a tiny frontier
12
12
  * (Merrill's fleeting iterations) and what makes the overflow retry exact: the partial edge queue is never consulted.
13
13
  * On a retry level `advance-expand` has already added the frontier's degree to `frontierDegreeSum`, so that word
14
- * holds 2 x m_f for the level and the next boundary's Beamer test and degree-sum subtraction see the doubled value;
15
- * reachable only with a faked capacity or an absurd graph, accepted and said here rather than guarded. Nothing here
14
+ * holds twice the level's expanded degree; since issue #391 no decision reads it (Beamer's m_f is `nextDegreeSum`
15
+ * and the boundary subtracts that), so the doubling only reaches the inspect seam's `prevDegreeSum`. Nothing here
16
16
  * writes a parent (PD-24: the post-pass does), which is what keeps the kernel at the eight-storage-buffer budget
17
17
  * with the four graph slots. Uniformity (spec 3.5 rule 1): the strip loop's bound and the row start are
18
18
  * `workgroupUniformLoad`s, so every barrier of the per-strip append is in uniform control flow; the guarded claim
@@ -44,7 +44,7 @@ fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
44
44
  let d = select(0u, a1 - a0, a1 > a0);
45
45
  wdeg = d;
46
46
  wstart = a0;
47
- atomicAdd(&counters[2], d); // frontierDegreeSum, so Beamer's test (P8-T8) sees fused levels too
47
+ atomicAdd(&counters[2], d); // frontierDegreeSum, so the inspect seam sees fused levels too
48
48
  }
49
49
  let deg = workgroupUniformLoad(&wdeg); // uniform: the loop below may hold barriers
50
50
  let start = workgroupUniformLoad(&wstart);
@@ -0,0 +1,33 @@
1
+ /**
2
+ * The `bfs-next-degree` kernel body (design 8.4; issue #391, the amendment to P8-T8's PD-21): Beamer's m_f measured
3
+ * EXACTLY, at the end of every level, as the out-degree sum of the vertices the level just claimed -- the next
4
+ * frontier, which is what the next boundary decides the direction FOR. It grid-strides over the output vertex
5
+ * queue (`nextFrontierCount`, word 1, the claim kernels' append span; `P.stride` the plan's stride), reads each
6
+ * entry's out-degree from the `outDegree` view, reduces the lane sums with the prelude's `wg_reduce_u32` and lands
7
+ * ONE `atomicAdd` per workgroup in `nextDegreeSum` (word 25), which `frontier-finalize` role 0 reads for the
8
+ * switch-into-bottom-up test, subtracts from `unvisitedDegreeSum` and zeroes for the next level. It runs on every
9
+ * path that claims (the path word 24 non-zero: two-phase, fused, bottom-up, the retry) and does nothing on a level
10
+ * past the end.
11
+ *
12
+ * Why a kernel of its own: before it, the test used `frontierDegreeSum` (word 2), which the EXPANSION of the
13
+ * previous frontier accumulates, so the boundary compared the degree of the frontier it had just finished with the
14
+ * unvisited set, one level stale, and a bottom-up level (which expands nothing) left it at 0. On the 1M / 10M R-MAT
15
+ * that misses the switch at the level that matters: the frontier of 46,524 hubs at level 1 has 13.6M out-arcs, the
16
+ * unvisited set 7.3M, and the boundary saw the source's 86,405 instead -- top-down wrote 13.6M edge-queue entries
17
+ * where the bottom-up sweep reads 0.6M. Measuring the next frontier's degree at claim time is Beamer's own m_f, and
18
+ * the same word makes `unvisitedDegreeSum` exact after a bottom-up level too. Uniformity (spec 3.5 rule 1): the
19
+ * loop holds no barrier (its trip count is per lane), and the reduction runs unconditionally after it. Body only
20
+ * (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits of it.
21
+ */
22
+ export const bfsNextDegreeWgsl = /* wgsl */ `
23
+ @compute @workgroup_size(WG)
24
+ fn bfs_next_degree(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
25
+ let count = select(0u, atomicLoad(&counters[1]), atomicLoad(&counters[24]) != 0u); // nextFrontierCount, on a level that claimed (the path word)
26
+ var sum = 0u;
27
+ for (var i = linear_id(wid, lid.x); i < count; i = i + P.stride) { // no barrier inside: the trip count is per lane
28
+ sum = sum + outDegree[frontier[i]];
29
+ }
30
+ let total = wg_reduce_u32(sum, lid.x, 0u); // the prelude's workgroup sum; uniform: after the loop
31
+ if (lid.x == 0u) { atomicAdd(&counters[25], total); } // nextDegreeSum: ONE atomic per workgroup
32
+ }
33
+ `;
@@ -9,6 +9,9 @@
9
9
  * `partials[group].swingTraction`. `pos` is `array<vec4f>` with the mass in `.w`; `force` / `oldForce` are stride-3
10
10
  * `array<f32>` read through the per-body helpers (4.4 rule 6). Normative text, copied verbatim: the P1-T5 sabotage
11
11
  * rows (gravity sign, `k / d2`, the `jj != i` guard, the `.w` mass lane) are textual edits of this string.
12
+ * The j range is split into passes of at most EXACT_TILES_PER_PASS tiles (issue #87: llvmpipe's per-invocation loop
13
+ * budget); an earlier pass adds its partial sum into `force`, the last one runs gravity and the epilogue. A graph of
14
+ * at most 32,768 nodes is one pass, bitwise the single-pass kernel. Record it through recordExactRepulsion.
12
15
  */
13
16
  export const fa2RepulsionExactWgsl = /* wgsl */ `
14
17
  var<workgroup> tile: array<vec4f, WG>; // xyz + mass, 4 KiB at WG = 256
@@ -35,14 +38,23 @@ fn gravity_force(pi: vec4f) -> vec3f { // spec 7.9: cent
35
38
  }
36
39
 
37
40
  @compute @workgroup_size(WG)
38
- fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
41
+ fn repulsion(
42
+ @builtin(workgroup_id) wid: vec3<u32>,
43
+ @builtin(local_invocation_id) lid: vec3<u32>,
44
+ @builtin(num_workgroups) nwg: vec3<u32>,
45
+ ) {
46
+ // issue #87: pass p of the tile range is dispatched with p + 1 z slices; only the last slice works, so the pass
47
+ // index needs no uniform. Uniform: keyed on workgroup_id and num_workgroups only.
48
+ if (wid.z + 1u < nwg.z) { return; }
39
49
  let i = linear_id(wid, lid.x);
40
50
  let valid = i < P.n;
41
51
  var pi = vec4f(0.0);
42
52
  if (valid) { pi = pos[i]; }
43
53
  var f = vec3f(0.0);
44
54
  let tiles = (P.n + WG - 1u) / WG;
45
- for (var t = 0u; t < tiles; t = t + 1u) {
55
+ let tileBegin = (nwg.z - 1u) * EXACT_TILES_PER_PASS;
56
+ let tileEnd = min(tiles, tileBegin + EXACT_TILES_PER_PASS);
57
+ for (var t = tileBegin; t < tileEnd; t = t + 1u) {
46
58
  let j = t * WG + lid.x;
47
59
  if (j < P.n) { tile[lid.x] = pos[j]; } else { tile[lid.x] = vec4f(0.0); } // guarded fill; mass 0 marks the pad
48
60
  workgroupBarrier(); // uniform: every invocation reaches it
@@ -65,6 +77,10 @@ fn repulsion(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
65
77
  }
66
78
  workgroupBarrier();
67
79
  }
80
+ if (tileEnd < tiles) { // an earlier pass: its partial sum only (uniform: P.n, nwg)
81
+ if (valid) { store_force(i, load_force(i) + f); }
82
+ return;
83
+ }
68
84
  // epilogue (7.9, 7.10): gravity and force += under the guard, the swing / traction reduction outside it
69
85
  var sw = 0.0;
70
86
  var tr = 0.0;
@@ -61,7 +61,7 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
61
61
  S.radius = sqrt(max(tHi.w, 0.0)); // max |p - centroid| about the same previous centroid as rmsRadius (K5 puts |q|^2 in max.w)
62
62
  let meanDisp = select(tDisp / f32(tFree), 0.0, tFree == 0u); // all-fixed: 0, never NaN (7.4)
63
63
  S.meanDisplacement = meanDisp;
64
- S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= P.settleThreshold * S.rmsRadius);
64
+ S.settledCount = select(0u, S.settledCount + 1u, meanDisp <= min(P.settleThreshold * S.rmsRadius, P.settleFloor)); // relative AND absolute (issue #97)
65
65
  }
66
66
  S.iteration = S.iteration + 1u;
67
67
  T[P.iterationIndex].meanDisplacement = S.meanDisplacement;
@@ -79,7 +79,9 @@ fn stats_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
79
79
  S.invCellSize = 1.0 / cellSize;
80
80
  S.eps = 0.25 * cellSize;
81
81
  }
82
- S.outsideGrid = cellHist[cells]; // the previous iteration's pseudo-cell count (0 after load)
82
+ var outside = 0u; // the previous iteration's pseudo-cell counts, one per orthant (issue #90; 0 after load)
83
+ for (var o = 0u; o < select(4u, 8u, P.dim == 3u); o = o + 1u) { outside = outside + cellHist[cells + o]; }
84
+ S.outsideGrid = outside;
83
85
  S.maxCellOccupancy = atomicLoad(&hubCounters[1]);
84
86
  atomicStore(&hubCounters[0], 0u);
85
87
  atomicStore(&hubCounters[1], 0u);
@@ -29,25 +29,26 @@
29
29
  * raw half the round consumed: mode 0 restarts the raw near half (word 1 to 0), mode 1 the raw far half (word 21
30
30
  * to 0); the relax kernels size themselves from words 0 and 20.
31
31
  *
32
- * Beamer's test (P8-T8, PD-21), evaluated at every boundary BEFORE the `done` branch (so a switch can be counted at
33
- * the done boundary too, which the host model of the tests mirrors): top-down switches to bottom-up when
34
- * `frontierDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's `max(1, floor(arcCount / n))`
35
- * unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up switches back when
36
- * `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no admitted device
37
- * reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the direction at 0. Every
38
- * change is counted in `switches`, the previous direction is word 14. The two unvisited words the test reads are
39
- * rebuilt exactly once per submit by `bfs-unvisited-flags` (PD-18) and maintained here by subtraction: the count is
40
- * subtracted from the SECOND boundary of a submit on and the degree sum from the THIRD on, because a boundary may
41
- * only subtract what the submit's rebuild counted, and the frontier whose degree sum the second boundary holds was
42
- * claimed before the rebuild ran (the rebuild counts the vertices unclaimed when it runs; the frontier rotated in at
43
- * boundary 0 was claimed by the previous submit's last contract, so it was never in the sum; boundary b subtracts
44
- * `next = |F_b|`, inside the sum iff b >= 1, and `degSum = deg(F_{b-1})`, inside it iff b >= 2). The degree sum is
45
- * the "unvisited degree estimate" of the design rather than an exact count for two reasons: it is one level stale
46
- * (a frontier's degree sum is only known once it has been expanded), and a bottom-up level expands nothing, so the
47
- * word stops falling while bottom-up runs and overstates the set afterwards. The bias is one-directional -- an
48
- * overstated m_u makes the switch INTO bottom-up harder, never easier -- and the next submit's rebuild makes it
49
- * exact again. Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are
50
- * textual edits of it.
32
+ * Beamer's test (P8-T8, PD-21; amended for issue #391), evaluated at every boundary BEFORE the `done` branch (so a
33
+ * switch can be counted at the done boundary too, which the host model of the tests mirrors): top-down switches to
34
+ * bottom-up when `nextDegreeSum > unvisitedDegreeSum / alpha` (u32 division; alpha the host's
35
+ * `max(1, floor(arcCount / n))` unless tuned) and the frontier is growing (`next > frontierCount`); bottom-up
36
+ * switches back when `next * beta < unvisitedCount` (a u32 product, wrapping only above 178M vertices, which no
37
+ * admitted device reaches) and the frontier is shrinking; `P.mode == 1` (the driver's `"top-down"`) pins the
38
+ * direction at 0. Every change is counted in `switches`, the previous direction is word 14. `nextDegreeSum` (word
39
+ * 25) is Beamer's m_f measured EXACTLY: `bfs-next-degree` sums the out-degrees of the vertices a level claims at
40
+ * the end of that level, so the boundary that rotates them in as `next` compares the degree of the frontier it is
41
+ * about to expand -- not, as before the amendment, `frontierDegreeSum` (word 2), the degree of the frontier the
42
+ * previous level EXPANDED, one level stale and 0 after a bottom-up level, which on the 1M / 10M R-MAT missed the
43
+ * switch at the level holding 13.6M of the 21M arcs. Word 2 is still accumulated by the expansion and rotated into
44
+ * word 4 for the inspect seam. The two unvisited words are rebuilt exactly once per submit by `bfs-unvisited-flags`
45
+ * (PD-18) and maintained here by subtraction from the SECOND boundary of a submit on, because a boundary may only
46
+ * subtract what the submit's rebuild counted: the rebuild counts the vertices unclaimed when it runs, the frontier
47
+ * rotated in at boundary 0 was claimed by the previous submit's last level, so it was never in the sums, and
48
+ * boundary b subtracts `next = |F_b|` and `nextDegreeSum = deg(F_b)`, both inside the sums iff b >= 1. Both words
49
+ * are therefore exact at every boundary, bottom-up levels included (the sweep's claims are summed like any other).
50
+ * Body only (spec 3.5, D9); the text is normative: the sabotage rows of test/helpers/sabotage.ts are textual edits
51
+ * of it.
51
52
  */
52
53
  export const frontierFinalizeWgsl = /* wgsl */ `
53
54
  @compute @workgroup_size(WG)
@@ -61,19 +62,19 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
61
62
  let finished = atomicLoad(&counters[0]);
62
63
  let next = atomicLoad(&counters[1]);
63
64
  let degSum = atomicLoad(&counters[2]);
65
+ let nextDeg = atomicLoad(&counters[25]); // deg(F_b), summed by bfs-next-degree when F_b was claimed (issue #391)
64
66
  atomicStore(&counters[3], finished); // prevFrontierCount
65
67
  atomicStore(&counters[4], degSum); // prevDegreeSum
66
68
  atomicStore(&counters[0], next); // the rotation
67
69
  atomicStore(&counters[1], 0u);
68
70
  atomicStore(&counters[2], 0u);
71
+ atomicStore(&counters[25], 0u); // the next level's claims sum from 0
69
72
  atomicStore(&counters[8], 0u); // edgeCount
70
73
  atomicStore(&counters[9], 0u); // edgeCountUnclamped
71
74
  atomicStore(&counters[12], atomicLoad(&counters[12]) + next); // visitedCount
72
- if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped to 2 (P8-T8, PD-18)
73
- atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount (exact): F_b was inside the submit's rebuilt sum iff b >= 1
74
- }
75
- if (P.firstOfSubmit >= 2u) {
76
- atomicStore(&counters[6], atomicLoad(&counters[6]) - degSum); // unvisitedDegreeSum (one level stale): F_{b-1} was inside it iff b >= 2
75
+ if (P.firstOfSubmit >= 1u) { // b = the boundary's index inside its submit, clamped (P8-T8, PD-18): F_b was inside the submit's rebuilt sums iff b >= 1
76
+ atomicStore(&counters[5], atomicLoad(&counters[5]) - next); // unvisitedCount, exact
77
+ atomicStore(&counters[6], atomicLoad(&counters[6]) - nextDeg); // unvisitedDegreeSum, exact (issue #391: no longer one level stale)
77
78
  }
78
79
  let level = atomicLoad(&counters[11]) + 1u; // the seed is U32_MAX, so the first boundary lands on 0
79
80
  atomicStore(&counters[11], level);
@@ -83,7 +84,7 @@ fn frontier_finalize(@builtin(local_invocation_id) lid: vec3<u32>) {
83
84
  if (P.mode == 1u) {
84
85
  direction = 0u; // top-down only (the test seam)
85
86
  } else if (direction == 0u) {
86
- if (degSum > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing
87
+ if (nextDeg > atomicLoad(&counters[6]) / P.alpha && next > finished) { direction = 1u; } // m_f > m_u / alpha and growing, m_f the degree of the frontier about to be expanded
87
88
  } else {
88
89
  if (next * P.beta < atomicLoad(&counters[5]) && next < finished) { direction = 0u; } // next * beta < unvisited and shrinking
89
90
  }
@@ -1,7 +1,8 @@
1
1
  /**
2
2
  * G1, the `grid-cell-key` kernel body (spec 7.7; P4-T8): the finest cell of every node from the state's robust extent,
3
3
  * `floor((p - gridMin) * invCellSize)` (a multiply, correctly rounded everywhere: PD-10), linearised when every axis
4
- * is in [0, G) and the outside pseudo-cell `cells` otherwise; `cellVal[i] = i`. The clamp before the floor keeps a
4
+ * is in [0, G), and otherwise one of the 2^dim outside pseudo-cells `cells + orthant`, the orthant of the cell about the
5
+ * grid centre (bit a set when `c[a] >= G / 2`; issue #90); `cellVal[i] = i`. The clamp before the floor keeps a
5
6
  * far-away or NaN coordinate out of an out-of-range float-to-int conversion. Body only; normative text.
6
7
  */
7
8
  export const gridCellKeyWgsl = /* wgsl */ `
@@ -18,7 +19,8 @@ fn grid_cell_key(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocatio
18
19
  let g = i32(P.gridMax);
19
20
  var inside = c.x >= 0 && c.x < g && c.y >= 0 && c.y < g;
20
21
  if (P.dim == 3u) { inside = inside && c.z >= 0 && c.z < g; }
21
- var key = cells; // the outside pseudo-cell (7.7)
22
+ var key = cells + select(0u, 1u, c.x >= g / 2) + select(0u, 2u, c.y >= g / 2); // an outside pseudo-cell: its orthant (issue #90)
23
+ if (P.dim == 3u) { key = key + select(0u, 4u, c.z >= g / 2); }
22
24
  if (inside) {
23
25
  key = u32(c.x) + P.gridMax * u32(c.y);
24
26
  if (P.dim == 3u) { key = key + P.gridMax * P.gridMax * u32(c.z); }
@@ -1,5 +1,6 @@
1
1
  /**
2
- * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the pseudo-cell included; the
2
+ * G4, the `grid-centroid` kernel body (spec 7.7; P4-T9): thread per finest cell, the 2^dim outside pseudo-cells
3
+ * included (issue #90); the
3
4
  * mass-weighted position sum of a cell's sorted range in index order (no atomics: deterministic), the largest
4
5
  * occupancy into hubCounters[1], and cells above GRID_HUB_CELL entries appended to hubList for G4b (PD-13). Body
5
6
  * only; normative text.
@@ -10,7 +11,7 @@ fn grid_cells() -> u32 { return P.gridMax * P.gridMax * select(1u, P.gridMax, P.
10
11
  @compute @workgroup_size(WG)
11
12
  fn grid_centroid(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id) lid: vec3<u32>) {
12
13
  let c = linear_id(wid, lid.x);
13
- if (c > grid_cells()) { return; } // cells [0, cells]: the pseudo-cell is index cells; no barrier follows
14
+ if (c >= grid_cells() + select(4u, 8u, P.dim == 3u)) { return; } // cells [0, cells + 2^dim): the pseudo-cells follow the real ones; no barrier follows
14
15
  let start = cellStart[c];
15
16
  let count = cellStart[c + 1u] - start;
16
17
  atomicMax(&hubCounters[1], count); // maxCellOccupancy, read by K1 next iteration
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * G5, the `grid-downsample` kernel body (spec 7.7; P4-T9): one dispatch per coarser level; every parent cell is the
3
3
  * sum of its 4 (2D) or 8 (3D) children at the level below, read at P.childBase and written at P.parentBase (the
4
- * pseudo-cell, index cells of level 0, is never a child). No atomics. Body only; normative text.
4
+ * pseudo-cells, indices cells .. of level 0, are never children). No atomics. Body only; normative text.
5
5
  */
6
6
  export const gridDownsampleWgsl = /* wgsl */ `
7
7
  @compute @workgroup_size(WG)