@graphty/webgpu-graph-algorithms 0.6.4 → 0.6.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/README.md +38 -17
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-Dvq-Cc6v.js → context-DiSr6eiz.js} +45 -33
  4. package/dist/chunks/context-DiSr6eiz.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/algorithms/bfs.d.ts +11 -7
  7. package/dist/src/algorithms/bfs.d.ts.map +1 -1
  8. package/dist/src/algorithms/bfs.js +33 -10
  9. package/dist/src/algorithms/bfs.js.map +1 -1
  10. package/dist/src/algorithms/scope.d.ts +3 -3
  11. package/dist/src/algorithms/scope.d.ts.map +1 -1
  12. package/dist/src/algorithms/scope.js +0 -2
  13. package/dist/src/algorithms/scope.js.map +1 -1
  14. package/dist/src/algorithms/sssp.d.ts +4 -3
  15. package/dist/src/algorithms/sssp.d.ts.map +1 -1
  16. package/dist/src/algorithms/sssp.js +4 -3
  17. package/dist/src/algorithms/sssp.js.map +1 -1
  18. package/dist/src/constants.d.ts +33 -2
  19. package/dist/src/constants.d.ts.map +1 -1
  20. package/dist/src/constants.js +33 -2
  21. package/dist/src/constants.js.map +1 -1
  22. package/dist/src/kernel/dispatch.d.ts +2 -2
  23. package/dist/src/kernel/dispatch.d.ts.map +1 -1
  24. package/dist/src/kernel/kernel.d.ts +1 -1
  25. package/dist/src/kernel/kernel.js +2 -2
  26. package/dist/src/kernel/kernel.js.map +1 -1
  27. package/dist/src/kernel/prelude.d.ts.map +1 -1
  28. package/dist/src/kernel/prelude.js +2 -1
  29. package/dist/src/kernel/prelude.js.map +1 -1
  30. package/dist/src/kernels.d.ts +15 -11
  31. package/dist/src/kernels.d.ts.map +1 -1
  32. package/dist/src/kernels.js +42 -21
  33. package/dist/src/kernels.js.map +1 -1
  34. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  35. package/dist/src/layouts/forceatlas2.js +2 -1
  36. package/dist/src/layouts/forceatlas2.js.map +1 -1
  37. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  38. package/dist/src/layouts/fruchterman-reingold.js +4 -2
  39. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  40. package/dist/src/layouts/repulsion-exact.d.ts +16 -0
  41. package/dist/src/layouts/repulsion-exact.d.ts.map +1 -1
  42. package/dist/src/layouts/repulsion-exact.js +21 -1
  43. package/dist/src/layouts/repulsion-exact.js.map +1 -1
  44. package/dist/src/layouts/repulsion-grid.d.ts +1 -1
  45. package/dist/src/layouts/repulsion-grid.js +1 -1
  46. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  47. package/dist/src/layouts/spring-electrical.js +6 -2
  48. package/dist/src/layouts/spring-electrical.js.map +1 -1
  49. package/dist/src/primitives/advance.d.ts +3 -2
  50. package/dist/src/primitives/advance.d.ts.map +1 -1
  51. package/dist/src/primitives/advance.js.map +1 -1
  52. package/dist/src/primitives/frontier.d.ts +34 -38
  53. package/dist/src/primitives/frontier.d.ts.map +1 -1
  54. package/dist/src/primitives/frontier.js +24 -32
  55. package/dist/src/primitives/frontier.js.map +1 -1
  56. package/dist/src/primitives/grid-pyramid.d.ts +4 -4
  57. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -1
  58. package/dist/src/primitives/grid-pyramid.js +4 -3
  59. package/dist/src/primitives/grid-pyramid.js.map +1 -1
  60. package/dist/src/primitives/grid.d.ts +13 -10
  61. package/dist/src/primitives/grid.d.ts.map +1 -1
  62. package/dist/src/primitives/grid.js +10 -7
  63. package/dist/src/primitives/grid.js.map +1 -1
  64. package/dist/src/wgsl/advance-expand.wgsl.d.ts +4 -3
  65. package/dist/src/wgsl/advance-expand.wgsl.d.ts.map +1 -1
  66. package/dist/src/wgsl/advance-expand.wgsl.js +4 -3
  67. package/dist/src/wgsl/advance-expand.wgsl.js.map +1 -1
  68. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts +4 -3
  69. package/dist/src/wgsl/bfs-bottom-up.wgsl.d.ts.map +1 -1
  70. package/dist/src/wgsl/bfs-bottom-up.wgsl.js +4 -3
  71. package/dist/src/wgsl/bfs-bottom-up.wgsl.js.map +1 -1
  72. package/dist/src/wgsl/bfs-fused.wgsl.d.ts +6 -6
  73. package/dist/src/wgsl/bfs-fused.wgsl.d.ts.map +1 -1
  74. package/dist/src/wgsl/bfs-fused.wgsl.js +6 -6
  75. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts +23 -0
  76. package/dist/src/wgsl/bfs-next-degree.wgsl.d.ts.map +1 -0
  77. package/dist/src/wgsl/bfs-next-degree.wgsl.js +34 -0
  78. package/dist/src/wgsl/bfs-next-degree.wgsl.js.map +1 -0
  79. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts +4 -1
  80. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.d.ts.map +1 -1
  81. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js +18 -2
  82. package/dist/src/wgsl/fa2-repulsion-exact.wgsl.js.map +1 -1
  83. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +1 -1
  84. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
  85. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +4 -2
  86. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
  87. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts +44 -49
  88. package/dist/src/wgsl/frontier-finalize.wgsl.d.ts.map +1 -1
  89. package/dist/src/wgsl/frontier-finalize.wgsl.js +62 -107
  90. package/dist/src/wgsl/frontier-finalize.wgsl.js.map +1 -1
  91. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +3 -2
  92. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -1
  93. package/dist/src/wgsl/grid-cell-key.wgsl.js +4 -2
  94. package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -1
  95. package/dist/src/wgsl/grid-centroid.wgsl.d.ts +3 -2
  96. package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -1
  97. package/dist/src/wgsl/grid-centroid.wgsl.js +3 -2
  98. package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -1
  99. package/dist/src/wgsl/grid-downsample.wgsl.d.ts +1 -1
  100. package/dist/src/wgsl/grid-downsample.wgsl.js +1 -1
  101. package/dist/src/wgsl/grid-far-field.wgsl.d.ts +6 -4
  102. package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -1
  103. package/dist/src/wgsl/grid-far-field.wgsl.js +15 -8
  104. package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -1
  105. package/dist/src/wgsl/grid-near-field.wgsl.d.ts +2 -2
  106. package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -1
  107. package/dist/src/wgsl/grid-near-field.wgsl.js +6 -2
  108. package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -1
  109. package/dist/src/wgsl/histogram.wgsl.d.ts +1 -1
  110. package/dist/src/wgsl/histogram.wgsl.js +1 -1
  111. package/dist/webgpu-graph-algorithms.js +144 -119
  112. package/dist/webgpu-graph-algorithms.js.map +1 -1
  113. package/package.json +3 -3
  114. package/src/algorithms/bfs.ts +34 -10
  115. package/src/algorithms/pagerank.ts +19 -5
  116. package/src/algorithms/power-iteration.ts +8 -2
  117. package/src/algorithms/scope.ts +3 -10
  118. package/src/algorithms/sssp.ts +4 -3
  119. package/src/constants.ts +35 -2
  120. package/src/kernel/dispatch.ts +2 -2
  121. package/src/kernel/kernel.ts +2 -2
  122. package/src/kernel/prelude.ts +2 -0
  123. package/src/kernels.ts +44 -21
  124. package/src/layouts/forceatlas2.ts +2 -0
  125. package/src/layouts/fruchterman-reingold.ts +4 -1
  126. package/src/layouts/repulsion-exact.ts +29 -1
  127. package/src/layouts/repulsion-grid.ts +1 -1
  128. package/src/layouts/spring-electrical.ts +8 -1
  129. package/src/memory/residency.ts +14 -4
  130. package/src/primitives/advance.ts +5 -4
  131. package/src/primitives/frontier.ts +42 -56
  132. package/src/primitives/grid-pyramid.ts +6 -5
  133. package/src/primitives/grid.ts +17 -12
  134. package/src/wgsl/advance-expand.wgsl.ts +4 -3
  135. package/src/wgsl/bfs-bottom-up.wgsl.ts +4 -3
  136. package/src/wgsl/bfs-fused.wgsl.ts +6 -6
  137. package/src/wgsl/bfs-next-degree.wgsl.ts +33 -0
  138. package/src/wgsl/fa2-repulsion-exact.wgsl.ts +18 -2
  139. package/src/wgsl/fa2-stats-finalize.wgsl.ts +4 -2
  140. package/src/wgsl/frontier-finalize.wgsl.ts +62 -107
  141. package/src/wgsl/grid-cell-key.wgsl.ts +4 -2
  142. package/src/wgsl/grid-centroid.wgsl.ts +3 -2
  143. package/src/wgsl/grid-downsample.wgsl.ts +1 -1
  144. package/src/wgsl/grid-far-field.wgsl.ts +15 -8
  145. package/src/wgsl/grid-near-field.wgsl.ts +6 -2
  146. package/src/wgsl/histogram.wgsl.ts +1 -1
  147. package/dist/chunks/context-Dvq-Cc6v.js.map +0 -1
@@ -7,6 +7,7 @@
7
7
  * override set is a distinct pipeline and the subgroup twin is selected by the device's features (spec 5.1, D16).
8
8
  */
9
9
 
10
+ import { EXACT_TILES_PER_PASS } from "../constants.js";
10
11
  import { WebGpuGraphError } from "../errors.js";
11
12
  import { type DispatchPlan, plan1d } from "../kernel/dispatch.js";
12
13
  import { type BoundKernel, type Kernel } from "../kernel/kernel.js";
@@ -35,6 +36,33 @@ export interface RepulsionExactOverrides {
35
36
  readonly GRAVITY_CENTER: 0 | 1;
36
37
  }
37
38
 
39
+ /**
40
+ * Records K3 over `n` nodes as ceil(tiles / EXACT_TILES_PER_PASS) dispatches (issue #87: llvmpipe's per-invocation
41
+ * loop budget), pass p with p + 1 z slices of which only the last works (the kernel reads the pass from
42
+ * `num_workgroups.z`). One pass up to 32,768 nodes at WG 256. Every model's exact tier records K3 through here.
43
+ * ponytail: pass p also launches p idle slices, sum p over P passes; negligible against the O(n^2) pass work (31
44
+ * passes at 1M nodes); a per-pass uniform removes them if they ever show in a profile.
45
+ * @param kernel - the compiled `fa2-repulsion-exact`
46
+ * @param pass - the open compute pass
47
+ * @param bound - the kernel's bound groups
48
+ * @param plan - plan1d(n) of the kernel
49
+ * @param n - the node count
50
+ * @param paramsOffset - the dynamic offset of the Fa2Params slot
51
+ */
52
+ export function recordExactRepulsion(
53
+ kernel: Kernel,
54
+ pass: GPUComputePassEncoder,
55
+ bound: BoundKernel,
56
+ plan: DispatchPlan,
57
+ n: number,
58
+ paramsOffset: number,
59
+ ): void {
60
+ const passes = Math.max(1, Math.ceil(Math.ceil(n / kernel.workgroupSize) / EXACT_TILES_PER_PASS));
61
+ for (let z = 1; z <= passes; z++) {
62
+ kernel.dispatch(pass, bound, { ...plan, z }, [paramsOffset]);
63
+ }
64
+ }
65
+
38
66
  /** K3 (tiled all-pairs repulsion + gravity + the swing / traction epilogue) followed by K4 (the one-workgroup speed finalize) (spec 7.6, 7.10). */
39
67
  export class RepulsionExact {
40
68
  /** The overrides both kernels were compiled with (a frozen copy of the argument of create()). */
@@ -148,7 +176,7 @@ export class RepulsionExact {
148
176
  recordRepulsion(pass: GPUComputePassEncoder, n: number, paramsOffset: number): void {
149
177
  const bound = this.bound(this.boundRepulsion, "recordRepulsion");
150
178
  const plan = plan1d(n, this.repulsion.workgroupSize, this.caps);
151
- this.repulsion.dispatch(pass, bound, plan, [paramsOffset]);
179
+ recordExactRepulsion(this.repulsion, pass, bound, plan, n, paramsOffset);
152
180
  }
153
181
 
154
182
  /**
@@ -216,7 +216,7 @@ export class RepulsionGrid {
216
216
 
217
217
  /**
218
218
  * The model-owned buffers of the grid tier (spec 7.3; PD-11): `cellKey` / `cellVal` / `sortedKey` / `sortedIdx`
219
- * 4n, `cellHist` / `cellStart` 4 (cells + 2) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
219
+ * 4n, `cellHist` / `cellStart` 4 histWords (cells + 2^dim + 1) zeroed, `hubList` one word per possible hub cell, `hubArgs` one
220
220
  * indirect slot, `pyramid` 16 B per pyramid cell zeroed. `hubCounters` (16 B, zeroed) is the MODEL's on every
221
221
  * tier (PD-14: K1 binds it on the exact tier too). n = 0 reports one node's worth of bytes (spec 3.6).
222
222
  * @param n - the node count
@@ -25,6 +25,8 @@ import {
25
25
  MAX_ITERATIONS_PER_STEP,
26
26
  SE_DEFAULTS,
27
27
  SE_SCALE_REFERENCE_NODES,
28
+ SETTLE_FLOOR_FRACTION,
29
+ SETTLE_FLOOR_REFERENCE_NODES,
28
30
  TRACE_RECORD_BYTES,
29
31
  UNIFORM_SLOT_BYTES,
30
32
  } from "../constants.js";
@@ -77,6 +79,7 @@ import {
77
79
  subset,
78
80
  vector,
79
81
  } from "./model-common.js";
82
+ import { recordExactRepulsion } from "./repulsion-exact.js";
80
83
  import { type GridStage, RepulsionGrid, type RepulsionGridOverrides } from "./repulsion-grid.js";
81
84
 
82
85
  // ============================================================ constants
@@ -555,6 +558,10 @@ export class SpringElectricalModel implements ForceModel<SpringElectricalOptions
555
558
  frK: 0,
556
559
  temperature: 0,
557
560
  springLength: resolved.springLength,
561
+ settleFloor:
562
+ SETTLE_FLOOR_FRACTION.springElectrical *
563
+ resolved.springLength *
564
+ (SETTLE_FLOOR_REFERENCE_NODES / Math.max(n, 1)) ** 0.25,
558
565
  springCoefficient: resolved.springCoefficient ?? SE_DEFAULTS.springCoefficient * springSizeFactor(n),
559
566
  coulomb: resolved.gravity ?? SE_DEFAULTS.gravity * springSizeFactor(n),
560
567
  dragCoefficient: resolved.dragCoefficient,
@@ -609,7 +616,7 @@ export class SpringElectricalModel implements ForceModel<SpringElectricalOptions
609
616
  if (stop < 2) {
610
617
  return;
611
618
  }
612
- k3.dispatch(pass, k3Bound, bound.plan, [offset]);
619
+ recordExactRepulsion(k3, pass, k3Bound, bound.plan, bound.n, offset);
613
620
  if (stop < STAGE_K5) {
614
621
  return;
615
622
  }
@@ -579,7 +579,9 @@ export class GraphResidency {
579
579
  }
580
580
 
581
581
  /**
582
- * One resident per array (spec 4.3: views upload in perArray mode, never into the arena).
582
+ * One resident per array (spec 4.3: views upload in perArray mode, never into the arena). An empty array (the
583
+ * colIdx of an edgeless directed reverse view, the src / dst of an edgeless edgeList) is skipped: spec 5.6 never
584
+ * uploads a zero-length array, and Kernel.bind rejects a zero-size binding, so it is absent as in core().
583
585
  * @param record - the owning record
584
586
  * @param arrays - the named arrays
585
587
  * @param label - the buffer label prefix
@@ -592,6 +594,9 @@ export class GraphResidency {
592
594
  ): Readonly<Record<string, Binding>> {
593
595
  const bindings: Record<string, Binding> = {};
594
596
  for (const [name, array] of arrays) {
597
+ if (array.byteLength === 0) {
598
+ continue;
599
+ }
595
600
  const resident = this.upload(record, array, array, `${label}:${name}`);
596
601
  bindings[name] = { buffer: resident.buffer, offset: 0, size: resident.byteLength, window: null };
597
602
  }
@@ -606,19 +611,24 @@ export class GraphResidency {
606
611
  * lengths, so keying the packed buffer on `rev.rowPtr` would make the packed and the unpacked view of one
607
612
  * snapshot collide -- whichever was built second would get the other's buffer. The record still owns the
608
613
  * resident, so release(s) destroys it with the rest. Offsets are STORAGE_ALIGN-aligned because Kernel.bind
609
- * rejects any other offset synchronously (E_INVALID_ARGUMENT { argument: "offset" }).
614
+ * rejects any other offset synchronously (E_INVALID_ARGUMENT { argument: "offset" }). Empty arrays are left
615
+ * out as in separateArrays; when nothing is left, nothing is uploaded.
610
616
  * @param record - the owning record
611
- * @param arrays - the named arrays, in buffer order
617
+ * @param all - the named arrays, in buffer order
612
618
  * @param key - the marker object the resident is keyed on
613
619
  * @param label - the buffer label
614
620
  * @returns the bindings by name, all into the one buffer
615
621
  */
616
622
  private packArrays(
617
623
  record: ResidencyRecord,
618
- arrays: readonly (readonly [string, TypedArrayData])[],
624
+ all: readonly (readonly [string, TypedArrayData])[],
619
625
  key: object,
620
626
  label: string,
621
627
  ): Readonly<Record<string, Binding>> {
628
+ const arrays = all.filter(([, array]) => array.byteLength > 0);
629
+ if (arrays.length === 0) {
630
+ return Object.freeze({});
631
+ }
622
632
  const offsets: number[] = [];
623
633
  let total = 0;
624
634
  for (const [, array] of arrays) {
@@ -34,7 +34,8 @@ import { type Kernel } from "../kernel/kernel.js";
34
34
  import { FRONTIER_PARAMS, graphBindings, graphOverrides, kernelSpec } from "../kernels.js";
35
35
  import { type CoreBinding } from "../memory/residency.js";
36
36
  import { type CoreWindow, coreWindows, rowCountOf } from "./core-shape.js";
37
- import { type Frontier, type FrontierScope } from "./frontier.js";
37
+ import { type Frontier } from "./frontier.js";
38
+ import { type ReduceScope } from "./reduce.js";
38
39
 
39
40
  /** A prepared advance (design 6 row 8): records one expansion of a frontier per level. */
40
41
  export interface AdvancePlanner {
@@ -64,7 +65,7 @@ export interface AdvancePlanner {
64
65
  * @param core - the resident core arrays of the snapshot the frontier walks (windowed or not)
65
66
  * @returns the planner
66
67
  */
67
- export async function prepareAdvance(scope: FrontierScope, core: CoreBinding): Promise<AdvancePlanner> {
68
+ export async function prepareAdvance(scope: ReduceScope, core: CoreBinding): Promise<AdvancePlanner> {
68
69
  const kernel = await scope.pipelines.kernel(kernelSpec("advance-expand", graphOverrides(core, null)));
69
70
  return new AdvancePlannerImpl(scope, core, kernel);
70
71
  }
@@ -73,7 +74,7 @@ export async function prepareAdvance(scope: FrontierScope, core: CoreBinding): P
73
74
  class AdvancePlannerImpl implements AdvancePlanner {
74
75
  readonly kernel: Kernel;
75
76
  readonly windows: readonly CoreWindow[];
76
- private readonly scope: FrontierScope;
77
+ private readonly scope: ReduceScope;
77
78
  private readonly n: number;
78
79
 
79
80
  /**
@@ -82,7 +83,7 @@ class AdvancePlannerImpl implements AdvancePlanner {
82
83
  * @param core - the core the kernel was compiled for
83
84
  * @param kernel - the `advance-expand` kernel
84
85
  */
85
- constructor(scope: FrontierScope, core: CoreBinding, kernel: Kernel) {
86
+ constructor(scope: ReduceScope, core: CoreBinding, kernel: Kernel) {
86
87
  this.scope = scope;
87
88
  this.kernel = kernel;
88
89
  this.windows = coreWindows(core);
@@ -1,46 +1,48 @@
1
1
  /**
2
2
  * The `Frontier` of design 6 row 7 and the device-side dispatch selector of design 5.4 (P8-T4; the P8 plan's PD-1,
3
- * PD-3, PD-8, PD-23, DEP-P8-A, DEP-P8-C). A traversal's per-level state is two n-slot vertex queues, ONE 24-word
3
+ * PD-3, PD-8, PD-23, DEP-P8-A, DEP-P8-C). A traversal's per-level state is two n-slot vertex queues, ONE 25-word
4
4
  * counters block (every counter of the phase is a word of it: a four-byte word is never a legal storage-binding
5
- * offset, and the selector must reach every count it acts on through one binding), the indirect args buffer of
6
- * `MAX_LEVELS_PER_SUBMIT x FRONTIER_CANDIDATES` 16-byte slots, and the edge queue. The host never reads a counter
7
- * inside a submit: `frontier-finalize`, one lane, runs at the START of every level (role 0: rotates
8
- * `nextFrontierCount` into `frontierCount`, advances `level`, decides `done`, writes this level's slots) and again once
9
- * the edge queue is filled (role 1: clamps `edgeCount`, sizes the contract slot, or on an overflow --
10
- * `edgeCountUnclamped > edgeCapacity` -- sizes the fused-retry slot instead, PD-23), and the kernels that follow
11
- * dispatch indirectly from those slots. The host records up to `MAX_LEVELS_PER_SUBMIT` levels per submit and reads
12
- * `done` (four bytes) once per submit; a boundary that finds `done` set zeroes its slots and moves no counter, so the
13
- * recorded levels past the end are no-ops and the counters freeze at the finishing boundary's values.
5
+ * offset, and the selector must reach every count it acts on through one binding) and the edge queue. The host
6
+ * never reads a counter inside a submit: `frontier-finalize`, one lane, runs at the START of every level (role 0:
7
+ * rotates `nextFrontierCount` into `frontierCount`, advances `level`, decides `done`, chooses the level's path) and
8
+ * again once the edge queue is filled (role 1: clamps `edgeCount`, or on an overflow -- `edgeCountUnclamped >
9
+ * edgeCapacity` -- switches the path to the fused retry, PD-23). The choice is the block's `path` word (`W.path`,
10
+ * word 24); every level kernel is a direct grid-stride dispatch that reads it first and does nothing unless the
11
+ * word names it (design/decisions/2026-09-25-frontier-kernels-dispatch-directly.md: the seven indirect slots the
12
+ * selector once wrote per level cost about 0.4 ms of Dawn validation each and were 97 % of a traversal's wall
13
+ * time; they and their args buffer are gone). The host records up to `MAX_LEVELS_PER_SUBMIT` levels per submit and
14
+ * reads `done` (four bytes) once per submit; a boundary that finds `done` set writes path 0 and moves no counter,
15
+ * so the recorded levels past the end are no-ops and the counters freeze at the finishing boundary's values.
14
16
  *
15
17
  * The seed: `frontierCount` is NEVER seeded, because the first boundary rotates it out unread. A BFS driver calls
16
18
  * `reset(queue, source, { nextFrontierCount: 1, level: U32_MAX })`: the first boundary rotates the 1 in, adds it into
17
19
  * `visitedCount`, and wraps `level` to 0, so the level-0 expansion claims the source's neighbours at `level + 1 == 1`.
18
20
  * `reset` is a queue write, ordered before the submit that follows, and it puts the source on side 0.
19
21
  *
20
- * Every buffer comes from the caller's ONE lease (`scope.scratch`, `scope.indirect`) so the algorithm's dispose()
22
+ * Every buffer comes from the caller's ONE lease (`scope.scratch`) so the algorithm's dispose()
21
23
  * releases them together (design 4.4). The two vertex queues are two BUFFERS, never two ranges of one: `Kernel.bind`
22
24
  * rejects one buffer bound read-only and read-write in one dispatch even for disjoint ranges. `src/primitives/**`
23
25
  * never imports `src/context.ts`.
24
26
  */
25
27
 
26
- import { FRONTIER_CANDIDATES, MAX_LEVELS_PER_SUBMIT, U32_MAX } from "../constants.js";
28
+ import { MAX_LEVELS_PER_SUBMIT, U32_MAX } from "../constants.js";
27
29
  import { WebGpuGraphError } from "../errors.js";
28
30
  import { plan1d } from "../kernel/dispatch.js";
29
- import { INDIRECT_ARGS_STRIDE, type Kernel } from "../kernel/kernel.js";
31
+ import { type Kernel } from "../kernel/kernel.js";
30
32
  import { FRONTIER_COUNTERS, FRONTIER_PARAMS, kernelSpec } from "../kernels.js";
31
33
  import { type Binding } from "../types/memory.js";
32
34
  import { type ReduceScope } from "./reduce.js";
33
35
 
34
- /** The seven indirect slots of one level (`FRONTIER_CANDIDATES`), by the kernel that dispatches from each. */
35
- export const SLOT: Readonly<{
36
- expand: 0;
37
- contract: 1;
36
+ /** The values of the `path` word (`W.path`): what a level's kernels run; every level kernel reads it first. */
37
+ export const PATH: Readonly<{
38
+ none: 0;
39
+ twoPhase: 1;
38
40
  fused: 2;
39
- fillBits: 3;
40
- bitset: 4;
41
- bottomUp: 5;
42
- fusedRetry: 6;
43
- }> = Object.freeze({ expand: 0, contract: 1, fused: 2, fillBits: 3, bitset: 4, bottomUp: 5, fusedRetry: 6 });
41
+ bottomUp: 3;
42
+ fusedRetry: 4;
43
+ near: 5;
44
+ far: 6;
45
+ }> = Object.freeze({ none: 0, twoPhase: 1, fused: 2, bottomUp: 3, fusedRetry: 4, near: 5, far: 6 });
44
46
 
45
47
  /** The words of the counters block (`FRONTIER_COUNTERS`), by index: byte offset 4 x word; no driver types a number. */
46
48
  export const W: Readonly<{
@@ -69,6 +71,7 @@ export const W: Readonly<{
69
71
  thresholdBits: 22;
70
72
  deltaBits: 23;
71
73
  path: 24;
74
+ nextDegreeSum: 25;
72
75
  }> = Object.freeze({
73
76
  frontierCount: 0,
74
77
  nextFrontierCount: 1,
@@ -95,18 +98,13 @@ export const W: Readonly<{
95
98
  thresholdBits: 22,
96
99
  deltaBits: 23,
97
100
  path: 24,
101
+ nextDegreeSum: 25,
98
102
  });
99
103
 
100
104
  /** The words a `reset` seeds (every other word is zeroed). */
101
105
  export type FrontierSeed = Readonly<Partial<Record<keyof typeof W, number>>>;
102
106
 
103
- /** A ReduceScope that can also lease an INDIRECT buffer (the args of the frontier); `algorithmScope` is one. */
104
- export interface FrontierScope extends ReduceScope {
105
- /** A STORAGE | INDIRECT | COPY_DST | COPY_SRC buffer of the scope's lease. */
106
- indirect(byteLength: number, label: string): GPUBuffer;
107
- }
108
-
109
- /** The `FrontierParams` fields a caller passes to `recordFinalize`; the planner fills `role`, `slotBase`, `wg`, `edgeCapacity` and `n` itself. A missing field is written as 0. */
107
+ /** The `FrontierParams` fields a caller passes to `recordFinalize`; the planner fills `role`, `wg`, `edgeCapacity` and `n` itself. A missing field is written as 0. */
110
108
  export type FrontierFinalizeFields = Readonly<
111
109
  Partial<
112
110
  Record<
@@ -163,14 +161,12 @@ function assertCount(argument: string, value: number): void {
163
161
  }
164
162
  }
165
163
 
166
- /** The frontier queue of design 6 row 7: two vertex queues, the counters block, the args buffer and the edge queue, all leased by the caller's scope. */
164
+ /** The frontier queue of design 6 row 7: two vertex queues, the counters block and the edge queue, all leased by the caller's scope. */
167
165
  export class Frontier {
168
166
  /** The two n-slot u32 vertex queues (`vertices[side]` is the input of the current level). */
169
167
  readonly vertices: readonly [Binding, Binding];
170
- /** The `FrontierCounters` block, 96 B, bound by every kernel as `array<atomic<u32>>`. */
168
+ /** The `FrontierCounters` block, 112 B, bound by every kernel as `array<atomic<u32>>`. */
171
169
  readonly counters: Binding;
172
- /** The indirect args, `MAX_LEVELS_PER_SUBMIT x FRONTIER_CANDIDATES` 16-byte slots (usage INDIRECT | STORAGE | COPY_DST | COPY_SRC). */
173
- readonly args: Binding;
174
170
  /** The edge queue: `edgeCapacity` entries of u32 (the target vertex of an arc). */
175
171
  readonly edgeQueue: Binding;
176
172
  /** How many entries the edge queue holds; role 1 clamps `edgeCount` to it and detects an overflow above it. */
@@ -183,7 +179,6 @@ export class Frontier {
183
179
  * Wraps the leased buffers; use prepareFrontier().
184
180
  * @param vertices - the two vertex queues
185
181
  * @param counters - the counters block
186
- * @param args - the args buffer
187
182
  * @param edgeQueue - the edge queue
188
183
  * @param edgeCapacity - the edge queue's entry count
189
184
  * @param n - the vertex count
@@ -191,14 +186,12 @@ export class Frontier {
191
186
  constructor(
192
187
  vertices: readonly [Binding, Binding],
193
188
  counters: Binding,
194
- args: Binding,
195
189
  edgeQueue: Binding,
196
190
  edgeCapacity: number,
197
191
  n: number,
198
192
  ) {
199
193
  this.vertices = vertices;
200
194
  this.counters = counters;
201
- this.args = args;
202
195
  this.edgeQueue = edgeQueue;
203
196
  this.edgeCapacity = edgeCapacity;
204
197
  this.n = n;
@@ -234,7 +227,7 @@ export class Frontier {
234
227
  }
235
228
 
236
229
  /**
237
- * Seeds a traversal: one `queue.writeBuffer` of the whole 96-byte block (zero except the caller's words) and one of
230
+ * Seeds a traversal: one `queue.writeBuffer` of the whole 112-byte block (zero except the caller's words) and one of
238
231
  * `vertices[0][0] = source`, both ordered before the submit that follows; the source is on side 0 afterwards.
239
232
  * `frontierCount` is not a word to seed: the first boundary rotates word 1 into it (the BFS seed is
240
233
  * `{ nextFrontierCount: 1, level: U32_MAX }`). A source outside `[0, n)`, an unknown word or a value that is not a
@@ -261,13 +254,14 @@ export class Frontier {
261
254
 
262
255
  /** A prepared frontier (design 6 row 7): the leased queue and the recorded selector dispatches. */
263
256
  export interface FrontierPlanner {
264
- /** The queue the planner sizes the slots of. */
257
+ /** The queue the planner's selector rotates and chooses the path of. */
265
258
  readonly frontier: Frontier;
266
259
  /**
267
260
  * Records one `frontier-finalize` dispatch (one workgroup) in `role` for `level` of the current submit: one
268
- * `FrontierParams` record with `slotBase = level x FRONTIER_CANDIDATES`, `wg`, `edgeCapacity` and `n` filled by the
269
- * planner and every other field from `fields`. A level outside `[0, MAX_LEVELS_PER_SUBMIT)` or a role outside
270
- * `[0, 3]` is E_INVALID_ARGUMENT before anything is recorded.
261
+ * `FrontierParams` record with `wg`, `edgeCapacity` and `n` filled by the planner and every other field from
262
+ * `fields`. A level outside `[0, MAX_LEVELS_PER_SUBMIT)` (the selector addresses nothing by level any more, but
263
+ * the host's submit cadence still is the bound) or a role outside `[0, 3]` is E_INVALID_ARGUMENT before
264
+ * anything is recorded.
271
265
  * @param pass - the compute pass
272
266
  * @param role - 0 the level boundary, 1 the edge-queue role (2 and 3 are P8-T9's)
273
267
  * @param level - the level inside the submit
@@ -281,14 +275,14 @@ export interface FrontierPlanner {
281
275
  * edge capacity defaults to `max(1, min(arcCount, floor(maxStorageBufferBindingSize / 4)))` (never a zero-length
282
276
  * buffer: the one-node graph has no arcs); a test passes a small one to force the overflow path. The planner lives
283
277
  * exactly as long as the scope: never use it after the scope's dispose().
284
- * @param scope - the caller's scope (device, caps, cache, scratch, indirect, params)
278
+ * @param scope - the caller's scope (device, caps, cache, scratch, params)
285
279
  * @param n - the vertex count
286
280
  * @param arcCount - the arc count (the natural edge-queue size)
287
281
  * @param edgeCapacity - the edge queue's entry count, when the caller chooses it (an integer >= 1)
288
282
  * @returns the planner
289
283
  */
290
284
  export async function prepareFrontier(
291
- scope: FrontierScope,
285
+ scope: ReduceScope,
292
286
  n: number,
293
287
  arcCount: number,
294
288
  edgeCapacity?: number,
@@ -306,7 +300,6 @@ export async function prepareFrontier(
306
300
  }
307
301
  const kernel = await scope.pipelines.kernel(kernelSpec("frontier-finalize"));
308
302
  const queueBytes = 4 * Math.max(1, n);
309
- const argsBytes = MAX_LEVELS_PER_SUBMIT * FRONTIER_CANDIDATES * INDIRECT_ARGS_STRIDE;
310
303
  const vertices: readonly [Binding, Binding] = [
311
304
  { buffer: scope.scratch(queueBytes, "frontier/vertices-0"), offset: 0, size: queueBytes, window: null },
312
305
  { buffer: scope.scratch(queueBytes, "frontier/vertices-1"), offset: 0, size: queueBytes, window: null },
@@ -317,26 +310,20 @@ export async function prepareFrontier(
317
310
  size: FRONTIER_COUNTERS.byteLength,
318
311
  window: null,
319
312
  };
320
- const args: Binding = {
321
- buffer: scope.indirect(argsBytes, "frontier/args"),
322
- offset: 0,
323
- size: argsBytes,
324
- window: null,
325
- };
326
313
  const edgeQueue: Binding = {
327
314
  buffer: scope.scratch(4 * capacity, "frontier/edge-queue"),
328
315
  offset: 0,
329
316
  size: 4 * capacity,
330
317
  window: null,
331
318
  };
332
- const frontier = new Frontier(vertices, counters, args, edgeQueue, capacity, n);
319
+ const frontier = new Frontier(vertices, counters, edgeQueue, capacity, n);
333
320
  return new FrontierPlannerImpl(scope, kernel, frontier);
334
321
  }
335
322
 
336
- /** The planner: the selector kernel bound once to the frontier's block and args. */
323
+ /** The planner: the selector kernel bound once to the frontier's block. */
337
324
  class FrontierPlannerImpl implements FrontierPlanner {
338
325
  readonly frontier: Frontier;
339
- private readonly scope: FrontierScope;
326
+ private readonly scope: ReduceScope;
340
327
  private readonly kernel: Kernel;
341
328
 
342
329
  /**
@@ -345,7 +332,7 @@ class FrontierPlannerImpl implements FrontierPlanner {
345
332
  * @param kernel - the `frontier-finalize` kernel
346
333
  * @param frontier - the leased queue
347
334
  */
348
- constructor(scope: FrontierScope, kernel: Kernel, frontier: Frontier) {
335
+ constructor(scope: ReduceScope, kernel: Kernel, frontier: Frontier) {
349
336
  this.scope = scope;
350
337
  this.kernel = kernel;
351
338
  this.frontier = frontier;
@@ -377,12 +364,11 @@ class FrontierPlannerImpl implements FrontierPlanner {
377
364
  const params = scope.params(FRONTIER_PARAMS, {
378
365
  ...definedWords(fields),
379
366
  role,
380
- slotBase: level * FRONTIER_CANDIDATES,
381
367
  wg: scope.workgroupSize,
382
368
  edgeCapacity: frontier.edgeCapacity,
383
369
  n: frontier.n,
384
370
  });
385
- const bound = this.kernel.bind({ counters: frontier.counters, args: frontier.args, P: params.binding });
371
+ const bound = this.kernel.bind({ counters: frontier.counters, P: params.binding });
386
372
  this.kernel.dispatch(pass, bound, plan1d(1, scope.workgroupSize, scope.caps), [params.offset]);
387
373
  }
388
374
  }
@@ -1,11 +1,11 @@
1
1
  /**
2
2
  * The grid pyramid (spec 6 row 12, 7.7 G4-G5; P4-T9): the planner that records, into the caller's pass, the finest
3
- * centroids (G4, `grid-centroid`: thread per cell over `cells + 1`, the pseudo-cell included), the hub-cell
3
+ * centroids (G4, `grid-centroid`: thread per cell over `cells + outsideCells`, the pseudo-cells included), the hub-cell
4
4
  * completion (G4a: the T1 `indirect-finalize` over `hubCounters[0]` into `hubArgs` with `wg = 1`, so the finalize's
5
5
  * `ceil(count / wg)` is ONE workgroup per hub cell; G4b: `grid-centroid-hub`, one workgroup per hub cell, dispatched
6
6
  * indirectly; PD-13, DEP-P4-I) and one `grid-downsample` dispatch per coarser
7
7
  * level (G5). Level 0 holds `[sum m x, sum m y, sum m z, sum m]` per cell; every parent is the sum of its 2^dim
8
- * children; the pseudo-cell (index `cells` of level 0) is never a child. No atomics touch the sums (design 6 row 12:
8
+ * children; the pseudo-cells (indices `cells ..` of level 0) are never children. No atomics touch the sums (design 6 row 12:
9
9
  * bitwise reproducible); the only atomics are the hub append and the occupancy max.
10
10
  *
11
11
  * The named grid buffers (`pyramid`, `hubList`, `hubCounters`, `hubArgs`) are the caller's (the model's
@@ -34,9 +34,9 @@ export interface GridPyramidBindings {
34
34
  readonly params: Binding;
35
35
  /** `n` words: the sorted node indices (the T8 build). */
36
36
  readonly sortedIdx: Binding;
37
- /** `cells + 2` words: the exclusive scan of the cell histogram (the T8 build). */
37
+ /** `histWords` (`cells + 2^dim + 1`) words: the exclusive scan of the cell histogram (the T8 build). */
38
38
  readonly cellStart: Binding;
39
- /** `pyramidCells` vec4f: every level, level 0 first with the pseudo-cell at index `cells`. */
39
+ /** `pyramidCells` vec4f: every level, level 0 first with the 2^dim orthant pseudo-cells from index `cells`. */
40
40
  readonly pyramid: Binding;
41
41
  /** The hub cells' indices, appended by G4 (at least one word; at most `floor(n / (GRID_HUB_CELL + 1))` are ever written, so `ceil(n / GRID_HUB_CELL)` words always suffice). */
42
42
  readonly hubList: Binding;
@@ -204,7 +204,8 @@ class GridPyramidPlannerImpl implements GridPyramidPlanner {
204
204
  }
205
205
  const { centroid, finalize, hub, downsample } = this.kernels;
206
206
  const one: DispatchPlan = { x: 1, y: 1, z: 1, items: 1, stride: null };
207
- centroid.dispatch(pass, bound.centroid, plan1d(spec.cells + 1, scope.workgroupSize, scope.caps), [paramsOffset]);
207
+ const level0 = spec.cells + spec.outsideCells;
208
+ centroid.dispatch(pass, bound.centroid, plan1d(level0, scope.workgroupSize, scope.caps), [paramsOffset]);
208
209
  finalize.dispatch(pass, bound.finalize, one, [bound.finalizeOffset]);
209
210
  hub.dispatchIndirect(pass, bound.hub, bound.hubArgs, 0, [paramsOffset]);
210
211
  this.dispatches = 3;
@@ -3,8 +3,9 @@
3
3
  * the caller's pass, the cell keys (G1, `grid-cell-key`), the stable sort by key (G2: `radixSort` at GRID_SORT_BITS,
4
4
  * or `countingSortByKey` when the caller asks for the set-deterministic path) and the per-cell histogram with its
5
5
  * exclusive scan (G3: the `histogram` kernel over `cellKey` and the `scan` of it; DEP-P4-I names no grid-specific
6
- * id). `cellHist` and `cellStart` hold `cells + 2` words: every real cell, the outside pseudo-cell at index `cells`
7
- * and one more so `cellStart[cells + 1] === n` closes the last range. Every zeroing is a `fill` dispatch inside the
6
+ * id). `cellHist` and `cellStart` hold `histWords = cells + 2^dim + 1` words: every real cell, the 2^dim outside
7
+ * pseudo-cells (one per orthant about the grid centre, issue #90) from index `cells`, and one more so
8
+ * `cellStart[histWords - 1] === n` closes the last range. Every zeroing is a `fill` dispatch inside the
8
9
  * pass (PD-12), never an encoder clear.
9
10
  *
10
11
  * The named grid buffers (`cellKey`, `cellVal`, `sortedKey`, `sortedIdx`, `cellHist`, `cellStart`) are the caller's
@@ -39,11 +40,13 @@ export interface GridSpec {
39
40
  readonly g: number;
40
41
  /** `log2(G / GRID_COARSEST_SIDE) + 1`. */
41
42
  readonly levels: number;
42
- /** `G^dim` finest cells; the outside pseudo-cell is index `cells`. */
43
+ /** `G^dim` finest cells; the outside pseudo-cells are indices `cells .. cells + outsideCells - 1`. */
43
44
  readonly cells: number;
44
- /** `cells + 2`: the length of `cellHist` / `cellStart`. */
45
+ /** `2^dim`: one outside pseudo-cell per orthant about the grid centre (issue #90). */
46
+ readonly outsideCells: number;
47
+ /** `cells + outsideCells + 1`: the length of `cellHist` / `cellStart`. */
45
48
  readonly histWords: number;
46
- /** The first cell of every level inside the pyramid: `levelOffsets[0] = 0`, level 0 holds `cells + 1` (the pseudo-cell), level L `(G / 2^L)^dim`. */
49
+ /** The first cell of every level inside the pyramid: `levelOffsets[0] = 0`, level 0 holds `cells + outsideCells` (the pseudo-cells last), level L `(G / 2^L)^dim`. */
47
50
  readonly levelOffsets: readonly number[];
48
51
  /** Every level's cells together: `levelOffsets[levels - 1] + GRID_COARSEST_SIDE^dim`. */
49
52
  readonly pyramidCells: number;
@@ -81,8 +84,8 @@ function floorPow2(x: number): number {
81
84
  * The grid of `n` nodes in `dim` dimensions under the tuning (spec 7.7 geometry table; PD-9): `G = clamp(nextPow2(2 *
82
85
  * ceil(n^(1 / dim))), GRID_MIN_SIDE, floorPow2(gridMax))` where `gridMax` is `gridMax2D` or `gridMax3D`, rounded DOWN
83
86
  * to a power of two so every level's side is an integer (512 and 128 stay; 100 becomes 64); `levels = log2(G /
84
- * GRID_COARSEST_SIDE) + 1`. At the caps: 349,521 pyramid cells in 2D, 2,396,737 in 3D (the design's counts plus the
85
- * pseudo-cell).
87
+ * GRID_COARSEST_SIDE) + 1`. At the caps: 349,524 pyramid cells in 2D, 2,396,744 in 3D (the design's counts plus the
88
+ * 2^dim pseudo-cells).
86
89
  * @param n - the node count (>= 0)
87
90
  * @param dim - 2 or 3
88
91
  * @param tuning - the resolved layout tuning (`gridMax2D`, `gridMax3D`, `deterministic`)
@@ -102,10 +105,11 @@ export function gridSpecFor(
102
105
  levels++;
103
106
  }
104
107
  const cells = g ** dim;
108
+ const outsideCells = 2 ** dim;
105
109
  const levelOffsets: number[] = [0];
106
110
  let s = g;
107
111
  for (let level = 0; level + 1 < levels; level++) {
108
- levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? 1 : 0));
112
+ levelOffsets.push(levelOffsets[level] + s ** dim + (level === 0 ? outsideCells : 0));
109
113
  s /= 2;
110
114
  }
111
115
  return {
@@ -113,7 +117,8 @@ export function gridSpecFor(
113
117
  g,
114
118
  levels,
115
119
  cells,
116
- histWords: cells + 2,
120
+ outsideCells,
121
+ histWords: cells + outsideCells + 1,
117
122
  levelOffsets: Object.freeze(levelOffsets),
118
123
  pyramidCells: levelOffsets[levels - 1] + GRID_COARSEST_SIDE ** dim,
119
124
  deterministic: tuning.deterministic,
@@ -121,7 +126,7 @@ export function gridSpecFor(
121
126
  }
122
127
 
123
128
  /**
124
- * The bytes of the pyramid (spec 7.7: 16 B per cell, every level, the pseudo-cell included): 38,347,792 at the 3D cap.
129
+ * The bytes of the pyramid (spec 7.7: 16 B per cell, every level, the pseudo-cells included): 38,347,904 at the 3D cap.
125
130
  * @param spec - the grid
126
131
  * @returns the byte length
127
132
  */
@@ -155,9 +160,9 @@ export interface GridBuildBindings {
155
160
  readonly sortedKey: Binding;
156
161
  /** `n` words: the sorted node indices. */
157
162
  readonly sortedIdx: Binding;
158
- /** `cells + 2` words: the per-cell counts. */
163
+ /** `histWords` (`cells + 2^dim + 1`) words: the per-cell counts. */
159
164
  readonly cellHist: Binding;
160
- /** `cells + 2` words: the exclusive scan of `cellHist`. */
165
+ /** `histWords` (`cells + 2^dim + 1`) words: the exclusive scan of `cellHist`. */
161
166
  readonly cellStart: Binding;
162
167
  }
163
168
 
@@ -7,8 +7,9 @@
7
7
  * the twin's two compilations -- and then every invocation strips the range `[0, aggregate)` with a binary search
8
8
  * (`upper_bound`) over the scanned degrees to find which entry its arc belongs to. One `atomicAdd` per WORKGROUP
9
9
  * reserves the block's span in the queue (`edgeCount`), the same aggregate lands in `edgeCountUnclamped` (the overflow
10
- * detector, never clamped) and in `frontierDegreeSum` (Beamer's m_f); a lane whose queue position is at or past
11
- * `P.edgeCapacity` writes nothing (the clamp). The queue holds the TARGET vertex of each arc only (PD-24: `parent`
10
+ * detector, never clamped) and in `frontierDegreeSum` (the inspect seam's per-level expansion count, rotated into
11
+ * `prevDegreeSum` by the boundary; Beamer's m_f is `nextDegreeSum`, measured by `bfs-next-degree` -- issue #391);
12
+ * a lane whose queue position is at or past `P.edgeCapacity` writes nothing (the clamp). The queue holds the TARGET vertex of each arc only (PD-24: `parent`
12
13
  * comes from the post-pass). Uniformity (spec 3.5 rule 1): the guarded loads write locals, the scan call and the
13
14
  * `workgroupUniformLoad` sit unconditionally after the guard, and the strip loop is bounded by a uniform value.
14
15
  * There is no `TIER` override: a hub row is balanced over all `WG` lanes inside its block, and the small-frontier
@@ -44,7 +45,7 @@ fn advance_expand(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocati
44
45
  if (lid.x == 0u) {
45
46
  base = atomicAdd(&counters[8], aggregate); // edgeCount: ONE reservation per workgroup, not one per arc
46
47
  atomicAdd(&counters[9], aggregate); // edgeCountUnclamped: the overflow detector (PD-23)
47
- atomicAdd(&counters[2], aggregate); // frontierDegreeSum: Beamer's m_f (P8-T8)
48
+ atomicAdd(&counters[2], aggregate); // frontierDegreeSum: what this level expanded (the inspect seam)
48
49
  }
49
50
  workgroupBarrier();
50
51
  for (var p = lid.x; p < aggregate; p = p + WG) { // strip [0, aggregate): lane j takes j, j + WG, ...
@@ -11,9 +11,10 @@
11
11
  * The winners are packed into the output vertex queue by `bfs-contract`'s workgroup scan and one `atomicAdd` per
12
12
  * workgroup on `nextFrontierCount`, and claim with a plain `atomicStore`: the list holds every vertex once and the
13
13
  * sweep is vertex-parallel, so no two lanes claim one vertex. It adds nothing to `frontierDegreeSum` (a bottom-up
14
- * level expands nothing), which is why `unvisitedDegreeSum` stops falling while bottom-up runs (the selector's
15
- * JSDoc). Uniformity (spec 3.5 rule 1): the guarded walk writes locals, the scan and the reduction run
16
- * unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
14
+ * level expands nothing), which is why that word is not Beamer's m_f: `bfs-next-degree` sums the degree of what
15
+ * this sweep CLAIMS into `nextDegreeSum`, so the boundary's test and its `unvisitedDegreeSum` subtraction are exact
16
+ * on a bottom-up level like any other (issue #391; the selector's JSDoc). Uniformity (spec 3.5 rule 1): the
17
+ * guarded walk writes locals, the scan and the reduction run unconditionally after it. Body only (spec 3.5, D9); the text is normative: the sabotage rows of
17
18
  * test/helpers/sabotage.ts are textual edits of it.
18
19
  */
19
20
  export const bfsBottomUpWgsl = /* wgsl */ `
@@ -4,15 +4,15 @@
4
4
  * ONE dispatch, chosen by `frontier-finalize` for a frontier below `P.fusedMax` entries (`path` 2) and for the retry
5
5
  * of a level whose edge queue overflowed (`path` 4, PD-23). One WORKGROUP per frontier entry, the workgroups striding
6
6
  * the entries by the dispatch's group count (`P.stride`): lane 0 reads the entry's row clipped
7
- * to the bound arc window, adds its degree to `frontierDegreeSum` (Beamer's m_f, so P8-T8's test sees fused levels
8
- * too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim inline --
9
- * `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
7
+ * to the bound arc window, adds its degree to `frontierDegreeSum` (so the inspect seam's per-level expansion count
8
+ * covers fused levels too), and every lane strips the row `WG` arcs at a time, applying `bfs-contract`'s claim
9
+ * inline -- `atomicMin(&depth[v], level + 1)`, the invocation that observes `INVALID_INDEX` the unique winner (PD-6) -- and
10
10
  * packing the strip's winners into the output vertex queue by the same Hillis-Steele scan and one `atomicAdd` per
11
11
  * strip on `nextFrontierCount`. No edge queue is written or read, which is the whole win for a tiny frontier
12
12
  * (Merrill's fleeting iterations) and what makes the overflow retry exact: the partial edge queue is never consulted.
13
13
  * On a retry level `advance-expand` has already added the frontier's degree to `frontierDegreeSum`, so that word
14
- * holds 2 x m_f for the level and the next boundary's Beamer test and degree-sum subtraction see the doubled value;
15
- * reachable only with a faked capacity or an absurd graph, accepted and said here rather than guarded. Nothing here
14
+ * holds twice the level's expanded degree; since issue #391 no decision reads it (Beamer's m_f is `nextDegreeSum`
15
+ * and the boundary subtracts that), so the doubling only reaches the inspect seam's `prevDegreeSum`. Nothing here
16
16
  * writes a parent (PD-24: the post-pass does), which is what keeps the kernel at the eight-storage-buffer budget
17
17
  * with the four graph slots. Uniformity (spec 3.5 rule 1): the strip loop's bound and the row start are
18
18
  * `workgroupUniformLoad`s, so every barrier of the per-strip append is in uniform control flow; the guarded claim
@@ -44,7 +44,7 @@ fn bfs_fused(@builtin(workgroup_id) wid: vec3<u32>, @builtin(local_invocation_id
44
44
  let d = select(0u, a1 - a0, a1 > a0);
45
45
  wdeg = d;
46
46
  wstart = a0;
47
- atomicAdd(&counters[2], d); // frontierDegreeSum, so Beamer's test (P8-T8) sees fused levels too
47
+ atomicAdd(&counters[2], d); // frontierDegreeSum, so the inspect seam sees fused levels too
48
48
  }
49
49
  let deg = workgroupUniformLoad(&wdeg); // uniform: the loop below may hold barriers
50
50
  let start = workgroupUniformLoad(&wstart);