@graphty/webgpu-graph-algorithms 0.5.1 → 0.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (243) hide show
  1. package/README.md +459 -58
  2. package/dist/browser.js +1 -1
  3. package/dist/chunks/{context-BR7fx3vR.js → context-BXqgCifx.js} +190 -40
  4. package/dist/chunks/context-BXqgCifx.js.map +1 -0
  5. package/dist/node.js +1 -1
  6. package/dist/src/algorithms/components.d.ts.map +1 -1
  7. package/dist/src/algorithms/components.js +12 -13
  8. package/dist/src/algorithms/components.js.map +1 -1
  9. package/dist/src/algorithms/degree.d.ts +6 -8
  10. package/dist/src/algorithms/degree.d.ts.map +1 -1
  11. package/dist/src/algorithms/degree.js +58 -35
  12. package/dist/src/algorithms/degree.js.map +1 -1
  13. package/dist/src/algorithms/pagerank.d.ts.map +1 -1
  14. package/dist/src/algorithms/pagerank.js +16 -14
  15. package/dist/src/algorithms/pagerank.js.map +1 -1
  16. package/dist/src/algorithms/power-iteration.d.ts +2 -2
  17. package/dist/src/algorithms/power-iteration.d.ts.map +1 -1
  18. package/dist/src/algorithms/power-iteration.js +17 -14
  19. package/dist/src/algorithms/power-iteration.js.map +1 -1
  20. package/dist/src/constants.d.ts +38 -8
  21. package/dist/src/constants.d.ts.map +1 -1
  22. package/dist/src/constants.js +38 -8
  23. package/dist/src/constants.js.map +1 -1
  24. package/dist/src/errors.d.ts +3 -2
  25. package/dist/src/errors.d.ts.map +1 -1
  26. package/dist/src/errors.js +2 -1
  27. package/dist/src/errors.js.map +1 -1
  28. package/dist/src/index.d.ts +6 -4
  29. package/dist/src/index.d.ts.map +1 -1
  30. package/dist/src/index.js +8 -3
  31. package/dist/src/index.js.map +1 -1
  32. package/dist/src/kernel/dispatch.d.ts +8 -3
  33. package/dist/src/kernel/dispatch.d.ts.map +1 -1
  34. package/dist/src/kernel/dispatch.js +18 -7
  35. package/dist/src/kernel/dispatch.js.map +1 -1
  36. package/dist/src/kernel/kernel.d.ts +30 -1
  37. package/dist/src/kernel/kernel.d.ts.map +1 -1
  38. package/dist/src/kernel/kernel.js +49 -5
  39. package/dist/src/kernel/kernel.js.map +1 -1
  40. package/dist/src/kernel/prelude.d.ts.map +1 -1
  41. package/dist/src/kernel/prelude.js +6 -1
  42. package/dist/src/kernel/prelude.js.map +1 -1
  43. package/dist/src/kernel/profiler.d.ts +15 -3
  44. package/dist/src/kernel/profiler.d.ts.map +1 -1
  45. package/dist/src/kernel/profiler.js +27 -4
  46. package/dist/src/kernel/profiler.js.map +1 -1
  47. package/dist/src/kernels.d.ts +17 -7
  48. package/dist/src/kernels.d.ts.map +1 -1
  49. package/dist/src/kernels.js +323 -16
  50. package/dist/src/kernels.js.map +1 -1
  51. package/dist/src/layouts/calibrate.d.ts +51 -0
  52. package/dist/src/layouts/calibrate.d.ts.map +1 -0
  53. package/dist/src/layouts/calibrate.js +172 -0
  54. package/dist/src/layouts/calibrate.js.map +1 -0
  55. package/dist/src/layouts/force-simulation.d.ts +39 -4
  56. package/dist/src/layouts/force-simulation.d.ts.map +1 -1
  57. package/dist/src/layouts/force-simulation.js +71 -19
  58. package/dist/src/layouts/force-simulation.js.map +1 -1
  59. package/dist/src/layouts/forceatlas2.d.ts +107 -36
  60. package/dist/src/layouts/forceatlas2.d.ts.map +1 -1
  61. package/dist/src/layouts/forceatlas2.js +296 -100
  62. package/dist/src/layouts/forceatlas2.js.map +1 -1
  63. package/dist/src/layouts/fruchterman-reingold.d.ts +73 -27
  64. package/dist/src/layouts/fruchterman-reingold.d.ts.map +1 -1
  65. package/dist/src/layouts/fruchterman-reingold.js +230 -70
  66. package/dist/src/layouts/fruchterman-reingold.js.map +1 -1
  67. package/dist/src/layouts/model-common.d.ts +41 -3
  68. package/dist/src/layouts/model-common.d.ts.map +1 -1
  69. package/dist/src/layouts/model-common.js +74 -3
  70. package/dist/src/layouts/model-common.js.map +1 -1
  71. package/dist/src/layouts/repulsion-grid.d.ts +152 -0
  72. package/dist/src/layouts/repulsion-grid.d.ts.map +1 -0
  73. package/dist/src/layouts/repulsion-grid.js +318 -0
  74. package/dist/src/layouts/repulsion-grid.js.map +1 -0
  75. package/dist/src/layouts/spring-electrical.d.ts +75 -30
  76. package/dist/src/layouts/spring-electrical.d.ts.map +1 -1
  77. package/dist/src/layouts/spring-electrical.js +231 -74
  78. package/dist/src/layouts/spring-electrical.js.map +1 -1
  79. package/dist/src/memory/residency.d.ts +6 -2
  80. package/dist/src/memory/residency.d.ts.map +1 -1
  81. package/dist/src/memory/residency.js +84 -14
  82. package/dist/src/memory/residency.js.map +1 -1
  83. package/dist/src/primitives/core-shape.d.ts +38 -2
  84. package/dist/src/primitives/core-shape.d.ts.map +1 -1
  85. package/dist/src/primitives/core-shape.js +71 -3
  86. package/dist/src/primitives/core-shape.js.map +1 -1
  87. package/dist/src/primitives/grid-pyramid.d.ts +71 -0
  88. package/dist/src/primitives/grid-pyramid.d.ts.map +1 -0
  89. package/dist/src/primitives/grid-pyramid.js +143 -0
  90. package/dist/src/primitives/grid-pyramid.js.map +1 -0
  91. package/dist/src/primitives/grid.d.ts +118 -0
  92. package/dist/src/primitives/grid.d.ts.map +1 -0
  93. package/dist/src/primitives/grid.js +225 -0
  94. package/dist/src/primitives/grid.js.map +1 -0
  95. package/dist/src/primitives/histogram.d.ts +67 -0
  96. package/dist/src/primitives/histogram.d.ts.map +1 -0
  97. package/dist/src/primitives/histogram.js +190 -0
  98. package/dist/src/primitives/histogram.js.map +1 -0
  99. package/dist/src/primitives/radix-sort.d.ts +75 -0
  100. package/dist/src/primitives/radix-sort.d.ts.map +1 -0
  101. package/dist/src/primitives/radix-sort.js +168 -0
  102. package/dist/src/primitives/radix-sort.js.map +1 -0
  103. package/dist/src/primitives/scan.d.ts +44 -0
  104. package/dist/src/primitives/scan.d.ts.map +1 -0
  105. package/dist/src/primitives/scan.js +151 -0
  106. package/dist/src/primitives/scan.js.map +1 -0
  107. package/dist/src/primitives/segmented-reduce.d.ts +25 -17
  108. package/dist/src/primitives/segmented-reduce.d.ts.map +1 -1
  109. package/dist/src/primitives/segmented-reduce.js +166 -47
  110. package/dist/src/primitives/segmented-reduce.js.map +1 -1
  111. package/dist/src/primitives/spmv.d.ts +18 -14
  112. package/dist/src/primitives/spmv.d.ts.map +1 -1
  113. package/dist/src/primitives/spmv.js +94 -58
  114. package/dist/src/primitives/spmv.js.map +1 -1
  115. package/dist/src/primitives/verify.d.ts +49 -0
  116. package/dist/src/primitives/verify.d.ts.map +1 -0
  117. package/dist/src/primitives/verify.js +229 -0
  118. package/dist/src/primitives/verify.js.map +1 -0
  119. package/dist/src/types/context.d.ts +53 -0
  120. package/dist/src/types/context.d.ts.map +1 -1
  121. package/dist/src/types/layout.d.ts +20 -0
  122. package/dist/src/types/layout.d.ts.map +1 -1
  123. package/dist/src/wgsl/counting-scatter.wgsl.d.ts +8 -0
  124. package/dist/src/wgsl/counting-scatter.wgsl.d.ts.map +1 -0
  125. package/dist/src/wgsl/counting-scatter.wgsl.js +17 -0
  126. package/dist/src/wgsl/counting-scatter.wgsl.js.map +1 -0
  127. package/dist/src/wgsl/fa2-attraction.wgsl.d.ts +23 -11
  128. package/dist/src/wgsl/fa2-attraction.wgsl.d.ts.map +1 -1
  129. package/dist/src/wgsl/fa2-attraction.wgsl.js +98 -20
  130. package/dist/src/wgsl/fa2-attraction.wgsl.js.map +1 -1
  131. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts +6 -2
  132. package/dist/src/wgsl/fa2-stats-finalize.wgsl.d.ts.map +1 -1
  133. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js +22 -1
  134. package/dist/src/wgsl/fa2-stats-finalize.wgsl.js.map +1 -1
  135. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts +8 -0
  136. package/dist/src/wgsl/grid-cell-key.wgsl.d.ts.map +1 -0
  137. package/dist/src/wgsl/grid-cell-key.wgsl.js +30 -0
  138. package/dist/src/wgsl/grid-cell-key.wgsl.js.map +1 -0
  139. package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts +8 -0
  140. package/dist/src/wgsl/grid-centroid-hub.wgsl.d.ts.map +1 -0
  141. package/dist/src/wgsl/grid-centroid-hub.wgsl.js +29 -0
  142. package/dist/src/wgsl/grid-centroid-hub.wgsl.js.map +1 -0
  143. package/dist/src/wgsl/grid-centroid.wgsl.d.ts +8 -0
  144. package/dist/src/wgsl/grid-centroid.wgsl.d.ts.map +1 -0
  145. package/dist/src/wgsl/grid-centroid.wgsl.js +29 -0
  146. package/dist/src/wgsl/grid-centroid.wgsl.js.map +1 -0
  147. package/dist/src/wgsl/grid-downsample.wgsl.d.ts +7 -0
  148. package/dist/src/wgsl/grid-downsample.wgsl.d.ts.map +1 -0
  149. package/dist/src/wgsl/grid-downsample.wgsl.js +28 -0
  150. package/dist/src/wgsl/grid-downsample.wgsl.js.map +1 -0
  151. package/dist/src/wgsl/grid-far-field.wgsl.d.ts +13 -0
  152. package/dist/src/wgsl/grid-far-field.wgsl.d.ts.map +1 -0
  153. package/dist/src/wgsl/grid-far-field.wgsl.js +98 -0
  154. package/dist/src/wgsl/grid-far-field.wgsl.js.map +1 -0
  155. package/dist/src/wgsl/grid-near-field.wgsl.d.ts +19 -0
  156. package/dist/src/wgsl/grid-near-field.wgsl.d.ts.map +1 -0
  157. package/dist/src/wgsl/grid-near-field.wgsl.js +129 -0
  158. package/dist/src/wgsl/grid-near-field.wgsl.js.map +1 -0
  159. package/dist/src/wgsl/histogram.wgsl.d.ts +7 -0
  160. package/dist/src/wgsl/histogram.wgsl.d.ts.map +1 -0
  161. package/dist/src/wgsl/histogram.wgsl.js +15 -0
  162. package/dist/src/wgsl/histogram.wgsl.js.map +1 -0
  163. package/dist/src/wgsl/indirect-finalize.wgsl.d.ts +8 -0
  164. package/dist/src/wgsl/indirect-finalize.wgsl.d.ts.map +1 -0
  165. package/dist/src/wgsl/indirect-finalize.wgsl.js +26 -0
  166. package/dist/src/wgsl/indirect-finalize.wgsl.js.map +1 -0
  167. package/dist/src/wgsl/radix-hist.wgsl.d.ts +9 -0
  168. package/dist/src/wgsl/radix-hist.wgsl.d.ts.map +1 -0
  169. package/dist/src/wgsl/radix-hist.wgsl.js +31 -0
  170. package/dist/src/wgsl/radix-hist.wgsl.js.map +1 -0
  171. package/dist/src/wgsl/radix-scatter.wgsl.d.ts +9 -0
  172. package/dist/src/wgsl/radix-scatter.wgsl.d.ts.map +1 -0
  173. package/dist/src/wgsl/radix-scatter.wgsl.js +40 -0
  174. package/dist/src/wgsl/radix-scatter.wgsl.js.map +1 -0
  175. package/dist/src/wgsl/scan-add.wgsl.d.ts +6 -0
  176. package/dist/src/wgsl/scan-add.wgsl.d.ts.map +1 -0
  177. package/dist/src/wgsl/scan-add.wgsl.js +14 -0
  178. package/dist/src/wgsl/scan-add.wgsl.js.map +1 -0
  179. package/dist/src/wgsl/scan-block.wgsl.d.ts +8 -0
  180. package/dist/src/wgsl/scan-block.wgsl.d.ts.map +1 -0
  181. package/dist/src/wgsl/scan-block.wgsl.js +30 -0
  182. package/dist/src/wgsl/scan-block.wgsl.js.map +1 -0
  183. package/dist/src/wgsl/segmented-reduce.wgsl.d.ts +22 -8
  184. package/dist/src/wgsl/segmented-reduce.wgsl.d.ts.map +1 -1
  185. package/dist/src/wgsl/segmented-reduce.wgsl.js +84 -15
  186. package/dist/src/wgsl/segmented-reduce.wgsl.js.map +1 -1
  187. package/dist/src/wgsl/spmv-pull.wgsl.d.ts +22 -11
  188. package/dist/src/wgsl/spmv-pull.wgsl.d.ts.map +1 -1
  189. package/dist/src/wgsl/spmv-pull.wgsl.js +110 -36
  190. package/dist/src/wgsl/spmv-pull.wgsl.js.map +1 -1
  191. package/dist/tsconfig.build.tsbuildinfo +1 -1
  192. package/dist/webgpu-graph-algorithms.js +3815 -1003
  193. package/dist/webgpu-graph-algorithms.js.map +1 -1
  194. package/package.json +9 -8
  195. package/src/algorithms/components.ts +12 -16
  196. package/src/algorithms/degree.ts +58 -43
  197. package/src/algorithms/pagerank.ts +20 -18
  198. package/src/algorithms/power-iteration.ts +19 -18
  199. package/src/constants.ts +38 -8
  200. package/src/errors.ts +3 -1
  201. package/src/index.ts +14 -4
  202. package/src/kernel/dispatch.ts +18 -7
  203. package/src/kernel/kernel.ts +59 -5
  204. package/src/kernel/prelude.ts +9 -0
  205. package/src/kernel/profiler.ts +28 -4
  206. package/src/kernels.ts +356 -18
  207. package/src/layouts/calibrate.ts +187 -0
  208. package/src/layouts/force-simulation.ts +91 -23
  209. package/src/layouts/forceatlas2.ts +331 -106
  210. package/src/layouts/fruchterman-reingold.ts +255 -74
  211. package/src/layouts/model-common.ts +98 -3
  212. package/src/layouts/repulsion-grid.ts +451 -0
  213. package/src/layouts/spring-electrical.ts +257 -78
  214. package/src/memory/residency.ts +126 -20
  215. package/src/primitives/core-shape.ts +91 -4
  216. package/src/primitives/grid-pyramid.ts +221 -0
  217. package/src/primitives/grid.ts +349 -0
  218. package/src/primitives/histogram.ts +273 -0
  219. package/src/primitives/radix-sort.ts +246 -0
  220. package/src/primitives/scan.ts +197 -0
  221. package/src/primitives/segmented-reduce.ts +214 -56
  222. package/src/primitives/spmv.ts +125 -65
  223. package/src/primitives/verify.ts +249 -0
  224. package/src/types/context.ts +56 -0
  225. package/src/types/layout.ts +22 -0
  226. package/src/wgsl/counting-scatter.wgsl.ts +16 -0
  227. package/src/wgsl/fa2-attraction.wgsl.ts +98 -20
  228. package/src/wgsl/fa2-stats-finalize.wgsl.ts +22 -1
  229. package/src/wgsl/grid-cell-key.wgsl.ts +29 -0
  230. package/src/wgsl/grid-centroid-hub.wgsl.ts +28 -0
  231. package/src/wgsl/grid-centroid.wgsl.ts +28 -0
  232. package/src/wgsl/grid-downsample.wgsl.ts +27 -0
  233. package/src/wgsl/grid-far-field.wgsl.ts +97 -0
  234. package/src/wgsl/grid-near-field.wgsl.ts +128 -0
  235. package/src/wgsl/histogram.wgsl.ts +14 -0
  236. package/src/wgsl/indirect-finalize.wgsl.ts +25 -0
  237. package/src/wgsl/radix-hist.wgsl.ts +30 -0
  238. package/src/wgsl/radix-scatter.wgsl.ts +39 -0
  239. package/src/wgsl/scan-add.wgsl.ts +13 -0
  240. package/src/wgsl/scan-block.wgsl.ts +29 -0
  241. package/src/wgsl/segmented-reduce.wgsl.ts +84 -15
  242. package/src/wgsl/spmv-pull.wgsl.ts +110 -36
  243. package/dist/chunks/context-BR7fx3vR.js.map +0 -1
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@graphty/webgpu-graph-algorithms",
3
- "version": "0.5.1",
3
+ "version": "0.6.1",
4
4
  "description": "WebGPU-accelerated graph algorithms and layouts over the @graphty/graph-format snapshot, for Node (Dawn) and browsers",
5
5
  "author": "Adam Powers <apowers@ato.ms>",
6
6
  "type": "module",
@@ -63,7 +63,7 @@
63
63
  "@graphty/graph-format": "^1.0.2"
64
64
  },
65
65
  "peerDependencies": {
66
- "@graphty/algorithms": "^1.0.0",
66
+ "@graphty/algorithms": "^1.0.0 || ^2.0.0",
67
67
  "@graphty/graph-format": "^1.0.0",
68
68
  "@graphty/layout": "^1.7.0",
69
69
  "webgpu": ">=0.4.0 <1.0.0"
@@ -92,8 +92,8 @@
92
92
  "vite": "^7.0.5",
93
93
  "vitest": "^3.2.4",
94
94
  "webgpu": "0.4.0",
95
- "@graphty/layout": "^1.8.0",
96
- "@graphty/algorithms": "^1.8.1"
95
+ "@graphty/algorithms": "^2.0.0",
96
+ "@graphty/layout": "^1.8.1"
97
97
  },
98
98
  "scripts": {
99
99
  "build": "tsc -p tsconfig.build.json",
@@ -105,13 +105,14 @@
105
105
  "typecheck:strict-consumer": "tsc -p tsconfig.strict-consumer.json",
106
106
  "test": "vitest",
107
107
  "test:ui": "vitest --ui",
108
- "test:run": "vitest run --project=node",
109
- "test:node": "vitest run --project=node",
108
+ "test:run": "vitest run --project=node --project=node-device-errors",
109
+ "test:node": "vitest run --project=node --project=node-device-errors",
110
+ "test:node:ci": "node scripts/run-node-shard.js",
110
111
  "test:browser": "vitest run --project=browser",
111
112
  "test:browser:ci": "node scripts/run-browser-project.js",
112
113
  "test:limits": "vitest run --project=node-limits",
113
- "coverage": "vitest run --project=node --coverage",
114
- "coverage:preview": "npx serve coverage -p 9058",
114
+ "coverage": "vitest run --project=node --project=node-device-errors --coverage",
115
+ "coverage:preview": "npx serve coverage -p ${PORT:?start it through servherd, which sets PORT}",
115
116
  "bench": "tsx benchmarks/run.ts",
116
117
  "bench:compare": "node scripts/bench-compare.js",
117
118
  "benchmark": "tsx benchmarks/run.ts",
@@ -19,11 +19,13 @@ import { type GraphSnapshot, renumberPartition, type U32 } from "@graphty/graph-
19
19
 
20
20
  import { U32_MAX } from "../constants.js";
21
21
  import { type GpuContext } from "../context.js";
22
- import { isWebGpuGraphError, WebGpuGraphError } from "../errors.js";
22
+ import { WebGpuGraphError } from "../errors.js";
23
23
  import { CommandBatch } from "../kernel/batch.js";
24
24
  import { plan1d, planGridStride } from "../kernel/dispatch.js";
25
25
  import { FILL_PARAMS, graphBindings, graphOverrides, kernelSpec, WCC_PARAMS } from "../kernels.js";
26
26
  import { type CoreBinding } from "../memory/residency.js";
27
+ import { assertWholeCore } from "../primitives/core-shape.js";
28
+ import { assertDeviceComputes } from "../primitives/verify.js";
27
29
  import { type ComponentsOptions, type GpuLabelResult } from "../types/algorithms.js";
28
30
  import { type Binding } from "../types/memory.js";
29
31
  import { type GpuRunOptions } from "../types/run.js";
@@ -66,25 +68,16 @@ function checkDest(dest: Float32Array | Uint32Array | undefined, n: number): U32
66
68
  }
67
69
 
68
70
  /**
69
- * The resident core; a windowed plan (`E_TOO_LARGE { path: "windowed", algorithm: null }`, spec 3.8) is re-thrown
70
- * with the algorithm name (3.12). Transcribed from degree.ts, whose coreOf is module-private.
71
+ * The resident core; a windowed plan is refused with `E_TOO_LARGE { path: "windowed", algorithm }` (spec 3.8, 3.12;
72
+ * DEP-P4-B: only degree and segmentedReduce execute windows).
71
73
  * @param ctx - the context
72
74
  * @param s - the snapshot
73
75
  * @returns the core binding
74
76
  */
75
77
  function coreOf(ctx: GpuContext, s: GraphSnapshot): CoreBinding {
76
- try {
77
- return ctx.residency.core(s);
78
- } catch (error: unknown) {
79
- if (isWebGpuGraphError(error) && error.code === "E_TOO_LARGE" && error.details.path === "windowed") {
80
- throw new WebGpuGraphError(
81
- "E_TOO_LARGE",
82
- `${ALGORITHM}: the arc arrays need a windowed upload, which P1-P3 plan but do not execute`,
83
- { ...error.details, algorithm: ALGORITHM },
84
- );
85
- }
86
- throw error;
87
- }
78
+ const core = ctx.residency.core(s);
79
+ assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, ALGORITHM);
80
+ return core;
88
81
  }
89
82
 
90
83
  /**
@@ -197,6 +190,7 @@ export async function connectedComponents(
197
190
  options?: ComponentsOptions & GpuRunOptions,
198
191
  ): Promise<GpuLabelResult> {
199
192
  ctx.assertReady();
193
+ await assertDeviceComputes(ctx);
200
194
  const n = s.nodeCount;
201
195
  const renumber = options?.renumber !== false;
202
196
  const dest = checkDest(options?.dest, n);
@@ -312,7 +306,9 @@ export async function connectedComponents(
312
306
  rounds += ROUNDS_PER_BATCH;
313
307
  ctx.assertReady();
314
308
  if (options?.signal?.aborted) {
315
- throw new WebGpuGraphError("E_ABORTED", `${ALGORITHM}: the signal was aborted`, { batchId: submitted.id });
309
+ throw new WebGpuGraphError("E_ABORTED", `${ALGORITHM}: the signal was aborted`, {
310
+ batchId: submitted.id,
311
+ });
316
312
  }
317
313
  if (new Uint32Array(back, flagRequest.offset, 1)[0] === 0) {
318
314
  break;
@@ -9,24 +9,24 @@
9
9
  *
10
10
  * Contract (3.12): `ctx.assertReady()` first; `nodeCount === 0` returns an empty `Uint32Array` (or `dest`) with no
11
11
  * GPU work (spec 5.6: there is no work, which is not a fallback); an already-aborted `signal` is `E_ABORTED` before
12
- * any work; `dest` must be a `Uint32Array` of length n over an `ArrayBuffer`; a windowed core is `E_TOO_LARGE
13
- * { path: "windowed", algorithm: "degree" }` until P4 executes windows -- `residency.core()` (3.8) throws that error
14
- * itself with `algorithm: null` before it would ever return a windowed `CoreBinding`, so `degree` catches it and
15
- * re-throws the same error with `algorithm: "degree"` (the 3.12 detail) instead of testing `core.plan`, which never
16
- * observes "windowed" in P1-P3; `arcCount === 0` skips the dispatch (only `rowPtr` is resident)
17
- * and the result is zeros; otherwise ONE dispatch over rows [0, n) with `accumulate = 0` (every row is written, so no
18
- * fill precedes it), a readback into the result, `onProgress(1, 1)`, and the scratch returned in a finally. Two runs
19
- * are bitwise identical (spec 11.9 item 4).
12
+ * any work; `dest` must be a `Uint32Array` of length n over an `ArrayBuffer`; `arcCount === 0` skips the dispatch
13
+ * (only `rowPtr` is resident) and the result is zeros; otherwise ONE dispatch over rows [0, n) with `accumulate = 0`
14
+ * (every row is written, so no fill precedes it) -- or, on a windowed core (spec 4.2, PD-8), a `fill` of zeros and
15
+ * one dispatch per window over its rows with `arcBase = w.start`, `arcEnd = w.end`, `accumulate = 1`, so a row split
16
+ * across windows adds its partial counts -- a readback into the result, `onProgress(1, 1)`, and the scratch returned
17
+ * in a finally. Two runs are bitwise identical (spec 11.9 item 4).
20
18
  */
21
19
 
22
20
  import { type GraphSnapshot, type U32 } from "@graphty/graph-format";
23
21
 
24
22
  import { type GpuContext } from "../context.js";
25
23
  import { BufferUsage } from "../device/webgpu-constants.js";
26
- import { isWebGpuGraphError, WebGpuGraphError } from "../errors.js";
24
+ import { WebGpuGraphError } from "../errors.js";
27
25
  import { plan1d } from "../kernel/dispatch.js";
28
- import { graphBindings, graphOverrides, kernelSpec, RANGE_PARAMS } from "../kernels.js";
29
- import { type CoreBinding } from "../memory/residency.js";
26
+ import { type UniformBlock, type UniformValues } from "../kernel/struct-block.js";
27
+ import { FILL_PARAMS, graphBindings, graphOverrides, kernelSpec, RANGE_PARAMS } from "../kernels.js";
28
+ import { windowBinding } from "../primitives/core-shape.js";
29
+ import { assertDeviceComputes } from "../primitives/verify.js";
30
30
  import { type Binding } from "../types/memory.js";
31
31
  import { type GpuRunOptions } from "../types/run.js";
32
32
 
@@ -55,26 +55,21 @@ function checkDest(dest: Float32Array | Uint32Array | undefined, n: number): U32
55
55
  }
56
56
 
57
57
  /**
58
- * Uploads (or finds) the core through the residency; a windowed plan, which the residency reports as `E_TOO_LARGE
59
- * { path: "windowed", algorithm: null }` (3.8), is re-thrown with `algorithm: "degree"` (3.12) and its other details
60
- * (`needed`, `limit`, `path`) kept. Every other error passes through untouched.
61
- * @param ctx - the context whose residency holds the core
62
- * @param s - the snapshot
63
- * @returns the resident core binding (arena or perArray)
58
+ * One pool-acquired uniform buffer holding one params record (a params buffer per dispatch; the caller releases
59
+ * `pooled` in its finally).
60
+ * @param ctx - the context whose pool and queue are used
61
+ * @param pooled - the list the buffer is appended to for release
62
+ * @param block - the uniform block
63
+ * @param values - the record's values
64
+ * @returns the whole-buffer binding
64
65
  */
65
- function coreOf(ctx: GpuContext, s: GraphSnapshot): CoreBinding {
66
- try {
67
- return ctx.residency.core(s);
68
- } catch (error: unknown) {
69
- if (isWebGpuGraphError(error) && error.code === "E_TOO_LARGE" && error.details.path === "windowed") {
70
- throw new WebGpuGraphError(
71
- "E_TOO_LARGE",
72
- "degree: the arc arrays need a windowed upload, which P1-P3 plan but do not execute",
73
- { ...error.details, algorithm: "degree" },
74
- );
75
- }
76
- throw error;
77
- }
66
+ function paramsBinding(ctx: GpuContext, pooled: GPUBuffer[], block: UniformBlock, values: UniformValues): Binding {
67
+ const buffer = ctx.pool.acquire(block.byteLength, BufferUsage.UNIFORM | BufferUsage.COPY_DST, "degree/params");
68
+ pooled.push(buffer);
69
+ const bytes = new ArrayBuffer(block.byteLength);
70
+ block.write(new DataView(bytes), values);
71
+ ctx.device.queue.writeBuffer(buffer, 0, bytes);
72
+ return { buffer, offset: 0, size: block.byteLength, window: null };
78
73
  }
79
74
 
80
75
  /**
@@ -89,6 +84,7 @@ function coreOf(ctx: GpuContext, s: GraphSnapshot): CoreBinding {
89
84
  */
90
85
  export async function degree(ctx: GpuContext, s: GraphSnapshot, options?: GpuRunOptions): Promise<U32> {
91
86
  ctx.assertReady();
87
+ await assertDeviceComputes(ctx);
92
88
  const n = s.nodeCount;
93
89
  const dest = checkDest(options?.dest, n);
94
90
  if (options?.signal?.aborted) {
@@ -98,7 +94,7 @@ export async function degree(ctx: GpuContext, s: GraphSnapshot, options?: GpuRun
98
94
  options?.onProgress?.(1, 1);
99
95
  return dest ?? new Uint32Array(0);
100
96
  }
101
- const core = coreOf(ctx, s);
97
+ const core = ctx.residency.core(s);
102
98
  if (s.arcCount === 0) {
103
99
  const zeros = dest ?? new Uint32Array(n);
104
100
  zeros.fill(0);
@@ -111,23 +107,40 @@ export async function degree(ctx: GpuContext, s: GraphSnapshot, options?: GpuRun
111
107
  BufferUsage.STORAGE | BufferUsage.COPY_SRC | BufferUsage.COPY_DST,
112
108
  "degree/out",
113
109
  );
114
- const params = ctx.pool.acquire(
115
- RANGE_PARAMS.byteLength,
116
- BufferUsage.UNIFORM | BufferUsage.COPY_DST,
117
- "degree/params",
118
- );
110
+ const pooled: GPUBuffer[] = [];
111
+ const params = (block: UniformBlock, values: UniformValues): Binding => paramsBinding(ctx, pooled, block, values);
119
112
  try {
120
113
  await ctx.allocator.check();
121
114
  const kernel = await ctx.pipelines.kernel(kernelSpec("degree", graphOverrides(core, null)));
122
- const bytes = new ArrayBuffer(RANGE_PARAMS.byteLength);
123
- RANGE_PARAMS.write(new DataView(bytes), { start: 0, end: n, arcBase: 0, arcEnd: s.arcCount, accumulate: 0, n });
124
- ctx.device.queue.writeBuffer(params, 0, bytes);
125
115
  const outBinding: Binding = { buffer: out, offset: 0, size: byteLength, window: null };
126
- const paramsBinding: Binding = { buffer: params, offset: 0, size: RANGE_PARAMS.byteLength, window: null };
127
- const bound = kernel.bind({ ...graphBindings(core, null), out: outBinding, P: paramsBinding });
128
116
  const encoder = ctx.device.createCommandEncoder({ label: "degree" });
129
117
  const pass = encoder.beginComputePass({ label: "degree" });
130
- kernel.dispatch(pass, bound, plan1d(n, ctx.workgroupSize, ctx.caps), [0]);
118
+ if (core.windows === null) {
119
+ const P = params(RANGE_PARAMS, { start: 0, end: n, arcBase: 0, arcEnd: s.arcCount, accumulate: 0, n });
120
+ const bound = kernel.bind({ ...graphBindings(core, null), out: outBinding, P });
121
+ kernel.dispatch(pass, bound, plan1d(n, ctx.workgroupSize, ctx.caps), [0]);
122
+ } else {
123
+ const fill = await ctx.pipelines.kernel(kernelSpec("fill"));
124
+ const zero = fill.bind({ dst: outBinding, P: params(FILL_PARAMS, { count: n, value: 0, mode: 0 }) });
125
+ fill.dispatch(pass, zero, plan1d(n, ctx.workgroupSize, ctx.caps), [0]);
126
+ for (const w of core.windows) {
127
+ const windowed = {
128
+ ...core,
129
+ colIdx: windowBinding(core, "colIdx", w),
130
+ weights: core.weights === null ? null : windowBinding(core, "weights", w),
131
+ };
132
+ const P = params(RANGE_PARAMS, {
133
+ start: w.rowFirst,
134
+ end: w.rowLast + 1,
135
+ arcBase: w.start,
136
+ arcEnd: w.end,
137
+ accumulate: 1,
138
+ n,
139
+ });
140
+ const bound = kernel.bind({ ...graphBindings(windowed, null), out: outBinding, P });
141
+ kernel.dispatch(pass, bound, plan1d(w.rowLast - w.rowFirst + 1, ctx.workgroupSize, ctx.caps), [0]);
142
+ }
143
+ }
131
144
  pass.end();
132
145
  ctx.device.queue.submit([encoder.finish()]);
133
146
  ctx.assertReady();
@@ -136,7 +149,9 @@ export async function degree(ctx: GpuContext, s: GraphSnapshot, options?: GpuRun
136
149
  options?.onProgress?.(1, 1);
137
150
  return dest ?? new Uint32Array(result);
138
151
  } finally {
139
- ctx.pool.release(params);
152
+ for (const buffer of pooled) {
153
+ ctx.pool.release(buffer);
154
+ }
140
155
  ctx.pool.release(out);
141
156
  }
142
157
  }
@@ -18,14 +18,15 @@ import { type F32, type GraphSnapshot } from "@graphty/graph-format";
18
18
 
19
19
  import { U32_MAX } from "../constants.js";
20
20
  import { type GpuContext } from "../context.js";
21
- import { isWebGpuGraphError, WebGpuGraphError } from "../errors.js";
21
+ import { WebGpuGraphError } from "../errors.js";
22
22
  import { CommandBatch } from "../kernel/batch.js";
23
23
  import { groupsOf, plan1d } from "../kernel/dispatch.js";
24
24
  import { kernelSpec, PR_PARAMS, PR_PARTIAL } from "../kernels.js";
25
25
  import { type ArrayBinding, type CoreBinding } from "../memory/residency.js";
26
- import { coreOfView } from "../primitives/core-shape.js";
26
+ import { assertWholeCore, coreOfView } from "../primitives/core-shape.js";
27
27
  import { prepareSegmentedReduce } from "../primitives/segmented-reduce.js";
28
28
  import { prepareSpmvPull } from "../primitives/spmv.js";
29
+ import { assertDeviceComputes } from "../primitives/verify.js";
29
30
  import { type GpuPageRankResult, type PageRankOptions } from "../types/algorithms.js";
30
31
  import { type Binding } from "../types/memory.js";
31
32
  import { type GpuRunOptions } from "../types/run.js";
@@ -62,26 +63,17 @@ function checkDest(dest: Float32Array | Uint32Array | undefined, n: number, algo
62
63
  }
63
64
 
64
65
  /**
65
- * The resident core; a windowed plan (`E_TOO_LARGE { path: "windowed", algorithm: null }`, spec 3.8) is re-thrown
66
- * with the algorithm name (3.12). Transcribed from degree.ts, whose coreOf is module-private.
66
+ * The resident core; a windowed plan is refused with `E_TOO_LARGE { path: "windowed", algorithm }` (spec 3.8, 3.12;
67
+ * DEP-P4-B: only degree and segmentedReduce execute windows).
67
68
  * @param ctx - the context
68
69
  * @param s - the snapshot
69
70
  * @param algorithm - the caller's name
70
71
  * @returns the core binding
71
72
  */
72
73
  function coreOf(ctx: GpuContext, s: GraphSnapshot, algorithm: string): CoreBinding {
73
- try {
74
- return ctx.residency.core(s);
75
- } catch (error: unknown) {
76
- if (isWebGpuGraphError(error) && error.code === "E_TOO_LARGE" && error.details.path === "windowed") {
77
- throw new WebGpuGraphError(
78
- "E_TOO_LARGE",
79
- `${algorithm}: the arc arrays need a windowed upload, which P1-P3 plan but do not execute`,
80
- { ...error.details, algorithm },
81
- );
82
- }
83
- throw error;
84
- }
74
+ const core = ctx.residency.core(s);
75
+ assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, algorithm);
76
+ return core;
85
77
  }
86
78
 
87
79
  /**
@@ -126,6 +118,7 @@ async function run(
126
118
  algorithm: string,
127
119
  ): Promise<GpuPageRankResult> {
128
120
  ctx.assertReady();
121
+ await assertDeviceComputes(ctx);
129
122
  const n = s.nodeCount;
130
123
  const alpha = options?.dampingFactor ?? 0.85;
131
124
  const maxIterations = options?.maxIterations ?? 100;
@@ -144,7 +137,13 @@ async function run(
144
137
  }
145
138
  if (n === 0) {
146
139
  options?.onProgress?.(maxIterations, maxIterations);
147
- return { scores: dest ?? new Float32Array(0), iterations: 0, converged: true, danglingMass: 0, precision: "f32" };
140
+ return {
141
+ scores: dest ?? new Float32Array(0),
142
+ iterations: 0,
143
+ converged: true,
144
+ danglingMass: 0,
145
+ precision: "f32",
146
+ };
148
147
  }
149
148
  const core = coreOf(ctx, s, algorithm);
150
149
  if (s.arcCount === 0) {
@@ -322,7 +321,10 @@ export async function personalizedPageRank(
322
321
  expected,
323
322
  });
324
323
  if (!(personalization instanceof Float32Array) || personalization.length !== n) {
325
- throw invalid(`${personalization.constructor.name}(${personalization.length})`, `a Float32Array of length ${n}`);
324
+ throw invalid(
325
+ `${personalization.constructor.name}(${personalization.length})`,
326
+ `a Float32Array of length ${n}`,
327
+ );
326
328
  }
327
329
  let total = 0;
328
330
  for (let v = 0; v < n; v++) {
@@ -20,13 +20,14 @@ import { type F32, type GraphSnapshot } from "@graphty/graph-format";
20
20
 
21
21
  import { U32_MAX } from "../constants.js";
22
22
  import { type GpuContext } from "../context.js";
23
- import { isWebGpuGraphError, WebGpuGraphError } from "../errors.js";
23
+ import { WebGpuGraphError } from "../errors.js";
24
24
  import { CommandBatch } from "../kernel/batch.js";
25
25
  import { groupsOf, plan1d } from "../kernel/dispatch.js";
26
26
  import { kernelSpec, PR_PARAMS, PR_PARTIAL } from "../kernels.js";
27
27
  import { type CoreBinding } from "../memory/residency.js";
28
- import { coreOfView } from "../primitives/core-shape.js";
28
+ import { assertWholeCore, coreOfView } from "../primitives/core-shape.js";
29
29
  import { prepareSpmvPull } from "../primitives/spmv.js";
30
+ import { assertDeviceComputes } from "../primitives/verify.js";
30
31
  import { type Binding } from "../types/memory.js";
31
32
  import { algorithmScope } from "./scope.js";
32
33
 
@@ -103,26 +104,17 @@ export function checkDest(dest: Float32Array | Uint32Array | undefined, n: numbe
103
104
  }
104
105
 
105
106
  /**
106
- * The resident core; a windowed plan (`E_TOO_LARGE { path: "windowed", algorithm: null }`, spec 3.8) is re-thrown
107
- * with the algorithm name (3.12). Transcribed from degree.ts, whose coreOf is module-private.
107
+ * The resident core; a windowed plan is refused with `E_TOO_LARGE { path: "windowed", algorithm }` (spec 3.8, 3.12;
108
+ * DEP-P4-B: only degree and segmentedReduce execute windows).
108
109
  * @param ctx - the context
109
110
  * @param s - the snapshot
110
111
  * @param algorithm - the caller's name
111
112
  * @returns the core binding
112
113
  */
113
114
  export function coreOf(ctx: GpuContext, s: GraphSnapshot, algorithm: string): CoreBinding {
114
- try {
115
- return ctx.residency.core(s);
116
- } catch (error: unknown) {
117
- if (isWebGpuGraphError(error) && error.code === "E_TOO_LARGE" && error.details.path === "windowed") {
118
- throw new WebGpuGraphError(
119
- "E_TOO_LARGE",
120
- `${algorithm}: the arc arrays need a windowed upload, which P1-P3 plan but do not execute`,
121
- { ...error.details, algorithm },
122
- );
123
- }
124
- throw error;
125
- }
115
+ const core = ctx.residency.core(s);
116
+ assertWholeCore(core, s.arcCount, ctx.caps.limits.maxStorageBufferBindingSize, algorithm);
117
+ return core;
126
118
  }
127
119
 
128
120
  /**
@@ -179,6 +171,7 @@ export async function runPowerIteration(
179
171
  n: number,
180
172
  config: PowerIterationConfig,
181
173
  ): Promise<PowerIterationRun> {
174
+ await assertDeviceComputes(ctx);
182
175
  const scope = algorithmScope(ctx, config.label, RING_SLOTS);
183
176
  try {
184
177
  const bytes = 4 * n;
@@ -231,7 +224,14 @@ export async function runPowerIteration(
231
224
  const rankOut = ring[iteration % ring.length];
232
225
  const { core, pull } = pulls[(iteration - 1) % pulls.length];
233
226
  // outWeightSum takes rankIn as its dummy: storage-ro, read only under NORM_MODE 0, never compiled here
234
- const scaleBindings = { rankIn, rankPrev: rankOut, outWeightSum: rankIn, xNorm, partials, P: params.binding };
227
+ const scaleBindings = {
228
+ rankIn,
229
+ rankPrev: rankOut,
230
+ outWeightSum: rankIn,
231
+ xNorm,
232
+ partials,
233
+ P: params.binding,
234
+ };
235
235
  scaleNorm.dispatch(pass, scaleNorm.bind(scaleBindings), scalePlan, [params.offset]);
236
236
  finalize.dispatch(pass, finalize.bind({ partials, P: params.binding }), finalizePlan, [params.offset]);
237
237
  if (scaleApply !== null) {
@@ -265,7 +265,8 @@ export async function runPowerIteration(
265
265
  }
266
266
  return {
267
267
  scores: new Float32Array(back, scoresRequest.offset, n).slice(),
268
- previous: previousRequest === null ? null : new Float32Array(back, previousRequest.offset, n).slice(),
268
+ previous:
269
+ previousRequest === null ? null : new Float32Array(back, previousRequest.offset, n).slice(),
269
270
  iterations: converged ? firstConverged : iterationsRun,
270
271
  converged,
271
272
  iterationsRun,
package/src/constants.ts CHANGED
@@ -13,6 +13,8 @@ export const MAX_WORKGROUPS_PER_DIM = 65535;
13
13
  export const MAX_1D_ITEMS = MAX_WORKGROUPS_PER_DIM * WORKGROUP_SIZE;
14
14
  /** The largest u32 (the `min` identity of the u32 reduce); interpolated into the prelude as `U32_MAX` so no body types the literal (contract 4.1). */
15
15
  export const U32_MAX = 0xffffffff;
16
+ /** The bins of one radix-sort pass, 2^8 (spec 6 row 6; P4 PD-5: 8 bits per pass); interpolated into the prelude as `RADIX_BINS` and `RADIX_DIGIT_MASK` (= bins - 1) so no body types the literal. */
17
+ export const RADIX_BINS = 256;
16
18
  /** Arc-window boundaries are multiples of 64 arcs = 256 bytes (design 10.6). */
17
19
  export const ARC_WINDOW_ALIGN = 64;
18
20
  /** Storage-binding offset alignment the package always honours (spec 2.6): the graph-format arena is 256-aligned. */
@@ -20,14 +22,30 @@ export const STORAGE_ALIGN = 256;
20
22
  /** Stride of one UniformRing slot: minUniformBufferOffsetAlignment is 256 on every runtime the package targets (spec 5.3). */
21
23
  export const UNIFORM_SLOT_BYTES = 256;
22
24
  /**
23
- * Exact-tier crossover default, re-fixed at G3 by the spec 7.8 rule (spec Q-6; docs/decisions/G3.md section 3): the
24
- * largest rung of the T-4 ladder (1k / 4k / 8k / 16k / 32k / 65k, E = 10n, 2D) with <= 4 ms per iteration, rounded down
25
- * to a power of two. Measured on the RTX 4070 SUPER under Dawn-node: benchmarks/results/nvidia-lovelace-driver580.json,
26
- * session 2026-09-16T02:07:45.933Z, the "ms/iteration (profiler)" rows (the GPU time of the iteration's passes at the
27
- * card's working clock; the step(1) wall rows beside them include the 12n readback): 2.560 ms at 32k, 8.405 ms at 65k.
28
- * The rule's second clause -- not slower than the grid tier at the same n -- has no grid tier to compare with in P3 and
29
- * is re-checked at G4 (spec 7.8). A consumer whose GPU differs (integrated, Apple, T4) passes its own value through
30
- * createAccelerator(ctx, { layout: { exactMaxNodes } }); P4's calibrateLayout(ctx) measures it.
25
+ * Exact-tier crossover default, re-fixed at G4 by the spec 7.8 rule in full (spec Q-6; docs/decisions/G3.md section 3
26
+ * for the budget clause, docs/decisions/G4.md section 7 for the re-check): the largest rung of the T-4 ladder (1k / 4k /
27
+ * 8k / 16k / 32k / 65k, E = 10n, 2D) with <= 4 ms per iteration AND not slower than the grid tier at the same n (the
28
+ * `layout-grid` 2D rows: the shared rungs 32k / 65k and the re-check rungs 1k / 4k / 8k / 16k the group runs so every
29
+ * exact rung has a grid row), rounded down to a power of two. Measured on the RTX 4070 SUPER under Dawn-node:
30
+ * benchmarks/results/nvidia-lovelace-driver580.json, session 2026-09-21T06:15:17.027Z, the "ms/iteration (profiler)"
31
+ * rows of layout-exact and the "grid ms/iteration (profiler) ... 2D" rows of layout-grid (the GPU time of the
32
+ * iteration's passes at the card's working clock), exact / grid ms per iteration: 1k 0.102 / 0.290, 4k 0.262 / 0.188,
33
+ * 8k 0.484 / 0.211, 16k 1.064 / 0.225, 32k 2.579 / 0.298, 65k 8.450 / 0.409. The grid tier's cost is a near-constant
34
+ * ~0.19-0.3 ms floor (its sort, scan and pyramid passes) up to 32k, so the exact tier wins only at 1k and the rule's
35
+ * answer is 1024 (32768 under the budget clause alone, the G3 value; 16384 was the P4-T14 draft's value, which only
36
+ * evaluated the grid clause at 32k / 65k and was not a measurement). calibrateLayout's finer probe on the same card
37
+ * (tmp/p4/t14-review/calibrate-defaults.log: exact 0.150 vs grid 0.190 ms at 2048) puts the true crossover between
38
+ * 2k and 4k; the ladder has no 2k rung, and the design's basis (cosmos switches at 4,096) is within 0.07 ms per
39
+ * iteration of the rule's value, so the choice among 1k / 2k / 4k is an accuracy preference (the exact tier is the
40
+ * oracle), not a speed one. OWNER DECISION G4-D1 (docs/decisions/G4.md section 7, rows G4-D1 / G4-F9): the constant
41
+ * KEEPS the G3 value 32768 while the accuracy work G4-F1 leaves is open -- G4-F1 (the grid tier's exact-vs-grid RMS
42
+ * on the clumpy and degenerate fixtures) closed on 2026-09-21 by the owner's asserted / printed split, G4-F2 (the
43
+ * unbiasedness item) the same day by its re-scope to the whole-field ratio, and the far field's accuracy on those
44
+ * fixtures is the follow-up -- because "auto" is the default every consumer sees and the exact tier is the accurate
45
+ * one; the rule's answer on the dev box (1024) is recorded, not shipped, until that work closes and the value is
46
+ * re-fixed. A
47
+ * consumer whose GPU differs (integrated, Apple, T4) passes its own value through
48
+ * createAccelerator(ctx, { layout: { exactMaxNodes } }); calibrateLayout(ctx) measures it.
31
49
  */
32
50
  export const EXACT_MAX_NODES = 32768;
33
51
  /** Default number of MAP_READ staging buffers in the Readback ring (spec 4.4). */
@@ -184,3 +202,15 @@ export const SE_DEFAULTS: Readonly<{
184
202
  iterationsPerStep: 1,
185
203
  maxInFlight: 2,
186
204
  });
205
+ /** The smallest finest grid side `G` (spec 7.7 geometry table: `clamp(nextPow2(2 n^(1/dim)), 8, gridMax)`; P4 PD-9). */
206
+ export const GRID_MIN_SIDE = 8;
207
+ /** The coarsest pyramid level's side (spec 7.7: "levels (coarsest 4 per axis)", `levels = log2(G / 4) + 1`). */
208
+ export const GRID_COARSEST_SIDE = 4;
209
+ /** A cell with more than this many entries is summed by a workgroup (G4b) instead of the thread-per-cell loop of G4 (spec 7.7 G4: `count > 1024`). */
210
+ export const GRID_HUB_CELL = 1024;
211
+ /** The extent floor of spec 7.7: `cellSize = max(extent, 1e-6) / G`, so an all-coincident load never divides by zero. */
212
+ export const GRID_EXTENT_FLOOR = 1e-6;
213
+ /** The bbox margin of spec 7.7: `bboxExtent` is the largest axis of `state.min / max` "with a 1% margin". */
214
+ export const GRID_BBOX_MARGIN = 1.01;
215
+ /** The key width the grid always sorts with (P4 PD-5): three 8-bit passes cover the 19-bit 2D keys at G = 512 and the 22-bit 3D keys at G = 128, and an odd pass count leaves `sortedIdx` in the scratch pair. */
216
+ export const GRID_SORT_BITS = 24;
package/src/errors.ts CHANGED
@@ -9,7 +9,8 @@
9
9
  * `details` keys per code (contract 3.1, so tests can assert them): E_NO_WEBGPU { reason, hint };
10
10
  * E_NO_ADAPTER { reason }; E_NO_DEVICE { reason, adapter, limit?, requested?, available? } (reason "consumed" |
11
11
  * "requestDevice" | "limit" | "feature" | "maxComputeWorkgroupsPerDimension"); E_SOFTWARE_ONLY { adapter };
12
- * E_DEVICE_LOST { reason, message }; E_DISPOSED { label }; E_VALIDATION { label, message, batchId? };
12
+ * E_DEVICE_LOST { reason, message }; E_DEVICE_INCORRECT { check, where, expected, actual, poison, count, blocks,
13
+ * workgroupSize, adapter, hint? }; E_DISPOSED { label }; E_VALIDATION { label, message, batchId? };
13
14
  * E_SHADER_COMPILE { id, stage: "compose" | "compile", slot?, messages? }; E_OUT_OF_MEMORY { requested, resident,
14
15
  * label }; E_TOO_LARGE { needed, limit, path, algorithm }; E_UNSUPPORTED { feature? | option?, hint? } (exactly
15
16
  * one of feature / option); E_INVALID_ARGUMENT { argument, value, expected? }; E_SNAPSHOT { reason, serial };
@@ -23,6 +24,7 @@ export type WebGpuGraphErrorCode =
23
24
  | "E_NO_DEVICE"
24
25
  | "E_SOFTWARE_ONLY"
25
26
  | "E_DEVICE_LOST"
27
+ | "E_DEVICE_INCORRECT"
26
28
  | "E_DISPOSED"
27
29
  | "E_VALIDATION"
28
30
  | "E_SHADER_COMPILE"
package/src/index.ts CHANGED
@@ -8,8 +8,8 @@
8
8
  * isSoftwareAdapter. P1 adds GpuContext and degree (values) and the context / run / profiler types. P2 adds nothing
9
9
  * (Lease, CommandBatch, UniformRing are internal). P3 adds the layout factory, the accelerator, the two default
10
10
  * tables, the seeder and the layout / accelerator types. P5 adds the two factories, the two default tables and the
11
- * two stats records. test/index.test.ts pins the value list and test/types/public-api.test-d.ts the type list; P4+
12
- * extends both. This comment must never spell the internal
11
+ * two stats records. P4 adds calibrateLayout and its two records. test/index.test.ts pins the value list and
12
+ * test/types/public-api.test-d.ts the type list. This comment must never spell the internal
13
13
  * JSDoc tag: it is the leading comment of the first export statement, and stripInternal would drop that statement
14
14
  * from the emitted declarations.
15
15
  */
@@ -41,6 +41,11 @@ export { GpuContext } from "./context.js";
41
41
  export { isSoftwareAdapter } from "./device/acquire.js";
42
42
  export type { PassTiming, Profiler } from "./kernel/profiler.js";
43
43
 
44
+ // ==================== the device self-check: what this device computed when it was asked an answer we already
45
+ // know. Every compute entry point awaits it and refuses a device that got it wrong (E_DEVICE_INCORRECT); a
46
+ // caller may await it first to ask before committing.
47
+ export { verifyDevice } from "./primitives/verify.js";
48
+
44
49
  // ==================== algorithms (P1: the walking-skeleton diagnostic, spec 3.3)
45
50
  export { degree } from "./algorithms/degree.js";
46
51
 
@@ -49,8 +54,9 @@ export { connectedComponents } from "./algorithms/components.js";
49
54
  export { pageRank, personalizedPageRank } from "./algorithms/pagerank.js";
50
55
  export { eigenvectorCentrality, hits, katzCentrality } from "./algorithms/spectral.js";
51
56
 
52
- // ==================== layouts and the accelerator (P3; the two P5 factories)
57
+ // ==================== layouts and the accelerator (P3; the two P5 factories; P4's calibrateLayout, spec 2.2)
53
58
  export { createAccelerator } from "./accelerator.js";
59
+ export { calibrateLayout } from "./layouts/calibrate.js";
54
60
  export { createForceAtlas2 } from "./layouts/forceatlas2.js";
55
61
  export { createFruchtermanReingold } from "./layouts/fruchterman-reingold.js";
56
62
  export { seedPositions } from "./layouts/seed.js";
@@ -97,6 +103,8 @@ export type {
97
103
  export type {
98
104
  AdapterInfoLike,
99
105
  AdapterSummary,
106
+ DeviceCheck,
107
+ DeviceCheckMismatch,
100
108
  GpuCaps,
101
109
  GpuContextOptions,
102
110
  LimitPolicy,
@@ -107,12 +115,14 @@ export type {
107
115
  RaisableLimit,
108
116
  } from "./types/context.js";
109
117
 
110
- // ==================== types: layouts (P3; the P5 stats and trace records)
118
+ // ==================== types: layouts (P3; the P5 stats and trace records; the P4 calibration records)
111
119
  export type {
120
+ CalibrateOptions,
112
121
  ForceAtlas2Stats,
113
122
  ForceAtlas2TraceRecord,
114
123
  FruchtermanReingoldStats,
115
124
  FruchtermanReingoldTraceRecord,
125
+ GpuCalibration,
116
126
  GpuLayoutSimulation,
117
127
  GpuLayoutTuning,
118
128
  LayoutStatsBase,
@@ -147,16 +147,27 @@ export function planGridStride(items: number, wg: number, caps: PlanCaps, maxGro
147
147
  }
148
148
 
149
149
  /**
150
- * P4 (spec 5.4): the (x, y, 1) args a device-side finalize kernel writes for a count. P1-P3: throws E_UNSUPPORTED { feature: "planIndirect" } (lead f).
151
- * @param count - the device-side count
152
- * @param wg - the workgroup size
150
+ * P4 (spec 5.4): the (x, y, 1) args the device-side finalize kernel writes for a count -- plan1d's rule applied to a
151
+ * u32 count through the same grid() as plan1d and plan2d, so the host twin and the kernel cannot drift; the kernel
152
+ * (src/wgsl/indirect-finalize.wgsl.ts) mirrors exactly this arithmetic. `items` of the plan is the count. A count
153
+ * outside [0, 2^32) is E_INVALID_ARGUMENT (the device holds it as a u32); for any u32 count y <= 1,025, so
154
+ * E_TOO_LARGE is unreachable here.
155
+ * @param count - the device-side count (a u32)
156
+ * @param wg - the workgroup size (a power of two)
153
157
  * @param caps - the capability table
158
+ * @returns the plan
154
159
  */
155
160
  export function planIndirect(count: number, wg: number, caps: PlanCaps): DispatchPlan {
156
- throw new WebGpuGraphError("E_UNSUPPORTED", "planIndirect lands with the indirect dispatch of P4 (spec 5.4)", {
157
- feature: "planIndirect",
158
- hint: `count ${count}, wg ${wg}, limit ${perDimension(caps)}`,
159
- });
161
+ assertCount("count", count);
162
+ if (count > 0xffffffff) {
163
+ throw new WebGpuGraphError("E_INVALID_ARGUMENT", `count must fit a u32, got ${count}`, {
164
+ argument: "count",
165
+ value: count,
166
+ expected: "an integer in [0, 2^32)",
167
+ });
168
+ }
169
+ assertWorkgroupSize(wg);
170
+ return grid(Math.ceil(count / wg), count, caps);
160
171
  }
161
172
 
162
173
  /**