makewfs 2.0.0__tar.gz → 2.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. {makewfs-2.0.0 → makewfs-2.1.1}/AGENTS.md +34 -2
  2. {makewfs-2.0.0 → makewfs-2.1.1}/CHANGELOG.md +77 -0
  3. {makewfs-2.0.0 → makewfs-2.1.1}/CITATION.cff +1 -1
  4. {makewfs-2.0.0 → makewfs-2.1.1}/PKG-INFO +4 -4
  5. {makewfs-2.0.0 → makewfs-2.1.1}/README.md +3 -3
  6. makewfs-2.1.1/benchmarks/device-results-neoverse-n1-rtx4060.json +357 -0
  7. makewfs-2.1.1/benchmarks/device-results-neoverse-n1-rtxa400.json +195 -0
  8. makewfs-2.1.1/benchmarks/device-results-neoverse-n1.md +40 -0
  9. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/run.py +21 -9
  10. {makewfs-2.0.0 → makewfs-2.1.1}/docs/configuration.md +1 -1
  11. {makewfs-2.0.0 → makewfs-2.1.1}/docs/performance.md +57 -5
  12. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/__about__.py +1 -1
  13. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/api.py +1 -0
  14. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/backend.py +194 -8
  15. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/provenance.py +5 -0
  16. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/sampling.py +81 -22
  17. makewfs-2.1.1/src/makewfs/sensors/_cuda_graph.py +84 -0
  18. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/sensors/_shack_hartmann_cuda.py +19 -2
  19. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/sensors/pyramid.py +73 -21
  20. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/sensors/shack_hartmann.py +6 -1
  21. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_gpu_backend.py +166 -0
  22. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_numerics.py +144 -0
  23. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_pyramid.py +64 -0
  24. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_shack_hartmann_sampling.py +3 -3
  25. {makewfs-2.0.0 → makewfs-2.1.1}/.gitignore +0 -0
  26. {makewfs-2.0.0 → makewfs-2.1.1}/CONTRIBUTING.md +0 -0
  27. {makewfs-2.0.0 → makewfs-2.1.1}/LICENSE +0 -0
  28. {makewfs-2.0.0 → makewfs-2.1.1}/ROADMAP.md +0 -0
  29. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/__init__.py +0 -0
  30. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/benchmark_compiled_sh_executor.py +0 -0
  31. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/benchmark_sh_state_batching.py +0 -0
  32. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/check_regression.py +0 -0
  33. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/configs/pyramid_40_float32.toml +0 -0
  34. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/configs/pyramid_60_mod8_float32.toml +0 -0
  35. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/configs/pyramid_80_mod32_float64.toml +0 -0
  36. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/configs/shack_hartmann_20x20_float32.toml +0 -0
  37. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/configs/shack_hartmann_60x60_float64.toml +0 -0
  38. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/configs/shack_hartmann_quadrature_9sample.toml +0 -0
  39. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/device-results.json +0 -0
  40. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/device-results.md +0 -0
  41. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/haka-compiled-sh-executor-quadro-p620.json +0 -0
  42. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/haka-sh-state-batching-quadro-p620.json +0 -0
  43. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/profile_warm.py +0 -0
  44. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/reference-results.json +0 -0
  45. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/reference-table.md +0 -0
  46. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/render_device_table.py +0 -0
  47. {makewfs-2.0.0 → makewfs-2.1.1}/benchmarks/render_table.py +0 -0
  48. {makewfs-2.0.0 → makewfs-2.1.1}/docs/adr/0001-units-coordinates.md +0 -0
  49. {makewfs-2.0.0 → makewfs-2.1.1}/docs/adr/0002-flux-normalization.md +0 -0
  50. {makewfs-2.0.0 → makewfs-2.1.1}/docs/adr/0003-public-api.md +0 -0
  51. {makewfs-2.0.0 → makewfs-2.1.1}/docs/adr/0004-backend-boundary.md +0 -0
  52. {makewfs-2.0.0 → makewfs-2.1.1}/docs/adr/index.md +0 -0
  53. {makewfs-2.0.0 → makewfs-2.1.1}/docs/api.md +0 -0
  54. {makewfs-2.0.0 → makewfs-2.1.1}/docs/concepts.md +0 -0
  55. {makewfs-2.0.0 → makewfs-2.1.1}/docs/contributing.md +0 -0
  56. {makewfs-2.0.0 → makewfs-2.1.1}/docs/detectors.md +0 -0
  57. {makewfs-2.0.0 → makewfs-2.1.1}/docs/examples.md +0 -0
  58. {makewfs-2.0.0 → makewfs-2.1.1}/docs/gallery/makewfs-gallery.json +0 -0
  59. {makewfs-2.0.0 → makewfs-2.1.1}/docs/gallery/makewfs-gallery.svg +0 -0
  60. {makewfs-2.0.0 → makewfs-2.1.1}/docs/gallery.md +0 -0
  61. {makewfs-2.0.0 → makewfs-2.1.1}/docs/guide-stars.md +0 -0
  62. {makewfs-2.0.0 → makewfs-2.1.1}/docs/index.md +0 -0
  63. {makewfs-2.0.0 → makewfs-2.1.1}/docs/interop.md +0 -0
  64. {makewfs-2.0.0 → makewfs-2.1.1}/docs/pyramid.md +0 -0
  65. {makewfs-2.0.0 → makewfs-2.1.1}/docs/quickstart.md +0 -0
  66. {makewfs-2.0.0 → makewfs-2.1.1}/docs/release.md +0 -0
  67. {makewfs-2.0.0 → makewfs-2.1.1}/docs/shack-hartmann.md +0 -0
  68. {makewfs-2.0.0 → makewfs-2.1.1}/docs/stability.md +0 -0
  69. {makewfs-2.0.0 → makewfs-2.1.1}/docs/troubleshooting.md +0 -0
  70. {makewfs-2.0.0 → makewfs-2.1.1}/docs/units-and-coordinates.md +0 -0
  71. {makewfs-2.0.0 → makewfs-2.1.1}/docs/validation.md +0 -0
  72. {makewfs-2.0.0 → makewfs-2.1.1}/examples/README.md +0 -0
  73. {makewfs-2.0.0 → makewfs-2.1.1}/examples/cds_readout.py +0 -0
  74. {makewfs-2.0.0 → makewfs-2.1.1}/examples/closed_loop_injection.py +0 -0
  75. {makewfs-2.0.0 → makewfs-2.1.1}/examples/compare_sensors.py +0 -0
  76. {makewfs-2.0.0 → makewfs-2.1.1}/examples/configs/angular_kernel.txt +0 -0
  77. {makewfs-2.0.0 → makewfs-2.1.1}/examples/configs/lgs_thin_beacon.toml +0 -0
  78. {makewfs-2.0.0 → makewfs-2.1.1}/examples/configs/precision_throughput.toml +0 -0
  79. {makewfs-2.0.0 → makewfs-2.1.1}/examples/configs/pyramid_minimal.toml +0 -0
  80. {makewfs-2.0.0 → makewfs-2.1.1}/examples/configs/qe_curve.txt +0 -0
  81. {makewfs-2.0.0 → makewfs-2.1.1}/examples/configs/shack_hartmann_extended_source.toml +0 -0
  82. {makewfs-2.0.0 → makewfs-2.1.1}/examples/configs/shack_hartmann_minimal.toml +0 -0
  83. {makewfs-2.0.0 → makewfs-2.1.1}/examples/configs/shack_hartmann_spectral_qe.toml +0 -0
  84. {makewfs-2.0.0 → makewfs-2.1.1}/examples/detector_choices.py +0 -0
  85. {makewfs-2.0.0 → makewfs-2.1.1}/examples/gallery.py +0 -0
  86. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/README.md +0 -0
  87. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/analyze_lut.py +0 -0
  88. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/benchmark.py +0 -0
  89. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/camera_modes.csv +0 -0
  90. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/camera_modes_empirical_floor_continuous.csv +0 -0
  91. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/compare_real.py +0 -0
  92. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/fit_secondary.py +0 -0
  93. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/haka_cpu_gpu_benchmark.json +0 -0
  94. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/haka_lut_snr.json +0 -0
  95. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/keck_haka.json +0 -0
  96. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/keck_haka.toml +0 -0
  97. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/mauna_kea_extinction.csv +0 -0
  98. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/mauna_kea_extinction_nir.csv +0 -0
  99. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/ocam_20260720/extract_ocam_images.py +0 -0
  100. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/ocam_20260720/make_ocam_video.py +0 -0
  101. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/real_vs_simulation.json +0 -0
  102. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/secondary_fit.json +0 -0
  103. {makewfs-2.0.0 → makewfs-2.1.1}/examples/keck_haka/simulate.py +0 -0
  104. {makewfs-2.0.0 → makewfs-2.1.1}/examples/lgs_elongation.py +0 -0
  105. {makewfs-2.0.0 → makewfs-2.1.1}/examples/lgs_thin_beacon.py +0 -0
  106. {makewfs-2.0.0 → makewfs-2.1.1}/examples/magnitude_series.py +0 -0
  107. {makewfs-2.0.0 → makewfs-2.1.1}/examples/makewfs_showcase.webp +0 -0
  108. {makewfs-2.0.0 → makewfs-2.1.1}/examples/moving_atmosphere.py +0 -0
  109. {makewfs-2.0.0 → makewfs-2.1.1}/examples/precision_throughput.py +0 -0
  110. {makewfs-2.0.0 → makewfs-2.1.1}/examples/pyramid_modulation.py +0 -0
  111. {makewfs-2.0.0 → makewfs-2.1.1}/examples/quickstart.py +0 -0
  112. {makewfs-2.0.0 → makewfs-2.1.1}/examples/realistic_broadband.py +0 -0
  113. {makewfs-2.0.0 → makewfs-2.1.1}/examples/sh_design_trade.py +0 -0
  114. {makewfs-2.0.0 → makewfs-2.1.1}/examples/showcase.py +0 -0
  115. {makewfs-2.0.0 → makewfs-2.1.1}/examples/spectral_qe.py +0 -0
  116. {makewfs-2.0.0 → makewfs-2.1.1}/pyproject.toml +0 -0
  117. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/__init__.py +0 -0
  118. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/cli.py +0 -0
  119. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/config.py +0 -0
  120. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/detector.py +0 -0
  121. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/pupil.py +0 -0
  122. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/py.typed +0 -0
  123. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/radiometry.py +0 -0
  124. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/sensors/__init__.py +0 -0
  125. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/sensors/base.py +0 -0
  126. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/source.py +0 -0
  127. {makewfs-2.0.0 → makewfs-2.1.1}/src/makewfs/wavefront.py +0 -0
  128. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_backend_audit.py +0 -0
  129. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_benchmarks.py +0 -0
  130. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_cli.py +0 -0
  131. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_config.py +0 -0
  132. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_conformance.py +0 -0
  133. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_hcipy_validation.py +0 -0
  134. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_input_rms.py +0 -0
  135. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_interop.py +0 -0
  136. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_keck_haka_example.py +0 -0
  137. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_oopao_validation.py +0 -0
  138. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_optics_validation.py +0 -0
  139. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_provenance.py +0 -0
  140. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_public_api.py +0 -0
  141. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_radiometry.py +0 -0
  142. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_shack_hartmann.py +0 -0
  143. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_source.py +0 -0
  144. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_validation_report.py +0 -0
  145. {makewfs-2.0.0 → makewfs-2.1.1}/tests/test_wavefront.py +0 -0
  146. {makewfs-2.0.0 → makewfs-2.1.1}/validation/__init__.py +0 -0
  147. {makewfs-2.0.0 → makewfs-2.1.1}/validation/run.py +0 -0
@@ -172,8 +172,10 @@ Follow the target layout in `ROADMAP.md`:
172
172
  only) does not cover.
173
173
  - `sensors/` contains deterministic ideal optical engines and no camera noise.
174
174
  `_shack_hartmann_cuda.py` is a private first-use-JIT execution plan for exact
175
- compatible CUDA geometries; `shack_hartmann.py` remains the readable physics
176
- reference and must stay as the automatic feature-complete fallback.
175
+ compatible CUDA geometries (sampled-DFT and integer-FFT spot grids alike);
176
+ `shack_hartmann.py` remains the readable physics reference and must stay as
177
+ the automatic feature-complete fallback. `_cuda_graph.py` replays the
178
+ pyramid's fixed-shape GPU propagation from a captured CUDA graph.
177
179
  - `radiometry.py` produces source photon budgets using public `getframes` tools.
178
180
  - `detector.py` is a narrow adapter to `getframes.Camera.expose` and the
179
181
  optional public `expose_spectral` cube API, plus the
@@ -236,6 +238,36 @@ python benchmarks/check_regression.py /tmp/makewfs-benchmark.json
236
238
  MPLBACKEND=Agg python examples/gallery.py
237
239
  ```
238
240
 
241
+ ## Performance gotchas
242
+
243
+ - `ArrayBackend.pruned_fft2` skips all-zero input lines and cropped output
244
+ lines of a 2-D FFT. Keep its `axes` equal to the pass order of the transform
245
+ it replaces: SciPy's `fft2` runs the listed axes in order when
246
+ `overwrite_x=True` but the last axis first when it allocates its output, and
247
+ the two orders round differently. SciPy also transforms lines in SIMD groups
248
+ whose short remainder group rounds differently, so a pruned transform can
249
+ differ from the full one by an ulp where the line counts differ. Check
250
+ candidate changes with a saved before/after render matrix, not only tests.
251
+ - The pyramid applies its mask on the unshifted FFT grid (the stored mask is
252
+ `ifftshift`-ed) and turns the outer shifts into wrapped start indices. Do not
253
+ reintroduce `fftshift`/`ifftshift` copies there; they are pure permutations.
254
+ - `PyramidEngine._propagate` is captured as a CUDA graph on a GPU. It must keep
255
+ fixed shapes and never synchronize with the host (no `scalar`, `.item()`,
256
+ `to_host`, or data-dependent shapes). A failed capture silently falls back to
257
+ eager execution and records why in `engine._graph.failure`; check it after
258
+ changing that method. Anything the graph uses must outlive it: its FFTs run
259
+ on the engine's own cuFFT plans (`plans=` in `fft_axis`), never CuPy's global
260
+ plan cache, which evicts and frees plans the graph still references.
261
+ - The compiled SH kernel reads the OPD as float64; `_CompiledShackHartmannExecutor.render`
262
+ widens a float32 lenslet-grid resample (exact). Its static thread and
263
+ shared-memory checks cannot see register pressure, so construction also
264
+ checks the compiled kernel's `max_threads_per_block` and falls back.
265
+ - Grid sizes set the physics, so they must never depend on the backend.
266
+ `backend.next_fast_length` is one 7-smooth rule for CPU and GPU (it equals
267
+ CuPy's `next_fast_len`); do not call `scipy.fft.next_fast_len` or
268
+ `cupyx.scipy.fft.next_fast_len` for a size. SciPy's admits 11, which made the
269
+ pyramid pad differently on the two devices before 2.1.1.
270
+
239
271
  ## Documentation and examples
240
272
 
241
273
  - Every public feature lands with its API docstring and the relevant user guide.
@@ -4,6 +4,83 @@ All notable changes to `makewfs` are documented here.
4
4
 
5
5
  ## [Unreleased]
6
6
 
7
+ ## [2.1.1] - 2026-10-07
8
+
9
+ ### Fixed
10
+
11
+ - **CPU and GPU pyramids used different propagation grids for some
12
+ geometries** ([#15](https://github.com/jacotay7/makewfs/issues/15)). The FFT
13
+ size came from `scipy.fft.next_fast_len` on the CPU and
14
+ `cupyx.scipy.fft.next_fast_len` on the GPU, and SciPy also admits a factor
15
+ of 11. Whenever it picked one (requests such as 33, 66 or 99 samples), the
16
+ CPU padded to 33/66/99 while the GPU padded to 35/70/100, and the photon
17
+ rates differed by up to 46% (24-pixel pupil, 9-pixel separation, 12-point
18
+ modulation at 3 lambda/D). Both backends now use one rule: the smallest
19
+ length with no prime factor above 7, which equals CuPy's choice, so **GPU
20
+ results are unchanged** and CPU and GPU now agree to rounding.
21
+ - **Behaviour change on the CPU** for every pyramid whose old SciPy size had a
22
+ factor of 11: the grid grows by at most 9.1% and the rate changes. At the
23
+ default `fft_oversampling = 2` this affects detector widths
24
+ (`pixels_across_pupil + pupil_separation_pixels + 2 * detector_margin_pixels`)
25
+ of 11, 22, 33, 38, 43, 44, 55, 65, 66, 76, 77, 82, 88, 99, 109, 110,
26
+ 113-115, 121, 129-132, 136, 137, 148, 151-154, 163-165, 176, 181, 197, 198,
27
+ 217-220, 226-231, 241, 242, 246, 247, 263-269, 271-275 and 295-297 pixels
28
+ (up to 300). Overall, 1,342 of the requests 1-4096 change size.
29
+ - **Unaffected:** the shipped example and benchmark pyramids (grids of 288,
30
+ 108, 160 and 216 samples) and every Shack-Hartmann configuration, whose FFT
31
+ sizes follow from the configured sampling.
32
+ - Frames now record the grid as `wfs_pyramid_fft_size_px`.
33
+ - **The 2.1.0 GPU pyramid's CUDA graph could replay freed cuFFT memory.** It
34
+ used plans from CuPy's global plan cache, which evicts them once other
35
+ transforms fill it (for example many sensors or user FFTs in one process),
36
+ and replay then read freed memory (`cudaErrorIllegalAddress`). The pyramid
37
+ now owns its plans, built exactly as CuPy builds them, so results are
38
+ unchanged.
39
+
40
+ ## [2.1.0] - 2026-10-07
41
+
42
+ ### Performance
43
+
44
+ - **Pruned FFTs on CPU and GPU.** Shack-Hartmann spots and the pyramid
45
+ transform only the FFT lines that hold data and keep only the cropped
46
+ output lines; the pyramid applies its mask on the unshifted grid instead of
47
+ shifting four full grids per state. CPU results are unchanged (bit-identical
48
+ in 451 of 468 arrays of a before/after matrix, otherwise within 2.5e-7
49
+ relative in float32).
50
+ - **CUDA-graph replay of the pyramid.** On a GPU the pyramid's fixed-shape
51
+ propagation is captured once and replayed with one launch.
52
+ - **Compiled CUDA executor for integer-FFT Shack-Hartmann grids**, which
53
+ previously ran the array path. Results agree to float rounding.
54
+ - End-to-end frames/s on an Ampere Neoverse-N1 (12 cores) and an RTX 4060,
55
+ 2.0.0 -> now: SH 20x20 float32 CPU 38.5 -> 126.1 (3.3x), GPU 496 -> 813
56
+ (1.6x); SH 60x60 float64 CPU 10.4 -> 29.8 (2.9x), GPU 199 -> 640 (3.2x);
57
+ nine-sample SH CPU 52.0 -> 96.3 (1.9x), GPU 189 -> 764 (4.1x); pyramid 40
58
+ CPU 923 -> 1,101 (1.2x), GPU 397 -> 759 (1.9x); pyramid 60 mod-8 CPU 173 ->
59
+ 272 (1.6x), GPU 408 -> 762 (1.9x); pyramid 80 mod-32 float64 CPU 11.4 ->
60
+ 20.9 (1.8x), GPU 204 -> 277 (1.4x). See `docs/performance.md`.
61
+
62
+ ### Fixed
63
+
64
+ - **Compiled CUDA Shack-Hartmann executor misread float32 rotated or offset
65
+ lenslet grids.** Such grids resample the OPD in float32, but the kernel read
66
+ it as float64, so the GPU spot pattern was wrong (82-96% error). The OPD is
67
+ now widened (exactly) before the launch.
68
+ - **Compiled CUDA Shack-Hartmann executor failed to launch for large
69
+ lenslet sampling** (e.g. 32x32 samples, 1024 threads) when register use left
70
+ fewer threads per block. Such geometries now fall back to the array path.
71
+
72
+ - **Benchmark artifacts named the wrong GPU on multi-GPU hosts.**
73
+ `benchmarks/run.py` recorded the first line of `nvidia-smi`, which ignores
74
+ `CUDA_VISIBLE_DEVICES`. It now asks CuPy for the device the run used. On
75
+ Arm hosts, whose `/proc/cpuinfo` has no model name, it reads the CPU model
76
+ from `lscpu` (e.g. `Neoverse-N1`) instead of reporting `aarch64`.
77
+
78
+ ### Added
79
+
80
+ - **Arm benchmark data point** (`benchmarks/device-results-neoverse-n1.*`): the
81
+ device table on an Ampere Neoverse-N1 host (16 pinned cores) with an RTX
82
+ 4060 and an RTX A400.
83
+
7
84
  ## [2.0.0] - 2026-10-07
8
85
 
9
86
  ### Breaking
@@ -5,7 +5,7 @@ type: software
5
5
  authors:
6
6
  - family-names: Taylor
7
7
  given-names: Jacob
8
- version: 2.0.0
8
+ version: 2.1.1
9
9
  date-released: 2026-10-07
10
10
  license: MIT
11
11
  repository-code: "https://github.com/jacotay7/makewfs"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: makewfs
3
- Version: 2.0.0
3
+ Version: 2.1.1
4
4
  Summary: Configuration-driven adaptive-optics wavefront sensor image simulation.
5
5
  Project-URL: Homepage, https://github.com/jacotay7/makewfs
6
6
  Project-URL: Documentation, https://jacotay7.github.io/makewfs/
@@ -137,9 +137,9 @@ LGS is monochromatic at the 589 nm D2 line, broadened by only ~0.003 nm, so its
137
137
  spectral axis exists there purely to exercise the polychromatic path. The
138
138
  showcase clip above instead runs a physically correct narrowband beacon sampled
139
139
  across the sodium layer's depth, which is where spot elongation actually comes
140
- from. Compatible sampled-DFT Shack–Hartmann geometries are compiled
141
- on first use, so warm each fixed sensor before measuring or entering a real-time
142
- loop. Reproduce the table with
140
+ from. On a GPU, compatible Shack–Hartmann geometries are compiled and
141
+ pyramid frames are captured as a CUDA graph on first use, so warm each fixed
142
+ sensor before measuring or entering a real-time loop. Reproduce the table with
143
143
 
144
144
  ```bash
145
145
  python benchmarks/run.py --device both --frames 100 \
@@ -88,9 +88,9 @@ LGS is monochromatic at the 589 nm D2 line, broadened by only ~0.003 nm, so its
88
88
  spectral axis exists there purely to exercise the polychromatic path. The
89
89
  showcase clip above instead runs a physically correct narrowband beacon sampled
90
90
  across the sodium layer's depth, which is where spot elongation actually comes
91
- from. Compatible sampled-DFT Shack–Hartmann geometries are compiled
92
- on first use, so warm each fixed sensor before measuring or entering a real-time
93
- loop. Reproduce the table with
91
+ from. On a GPU, compatible Shack–Hartmann geometries are compiled and
92
+ pyramid frames are captured as a CUDA graph on first use, so warm each fixed
93
+ sensor before measuring or entering a real-time loop. Reproduce the table with
94
94
 
95
95
  ```bash
96
96
  python benchmarks/run.py --device both --frames 100 \
@@ -0,0 +1,357 @@
1
+ {
2
+ "schema_version": 2,
3
+ "generated_at_utc": "2026-10-07T07:11:47.632479+00:00",
4
+ "command": "/home/jtaylor/miniforge3/envs/aosim-bench/bin/python benchmarks/run.py --device both --frames 100 --config benchmarks/configs/pyramid_40_float32.toml --config benchmarks/configs/pyramid_60_mod8_float32.toml --config benchmarks/configs/pyramid_80_mod32_float64.toml --config benchmarks/configs/shack_hartmann_20x20_float32.toml --config benchmarks/configs/shack_hartmann_60x60_float64.toml --config benchmarks/configs/shack_hartmann_quadrature_9sample.toml --output /home/jtaylor/aosim-bench-logs/artifacts/makewfs-device-table-rtx4060.json",
5
+ "revision": "74b53697569dc8b2b5ce93f0b66e39a1de3b303d",
6
+ "source_dirty": true,
7
+ "python": "3.13.15 | packaged by conda-forge | (main, Sep 2 2026, 22:02:03) [GCC 15.3.0]",
8
+ "platform": "Linux-6.17.9-76061709-generic-aarch64-with-glibc2.39",
9
+ "processor": "aarch64",
10
+ "cpu": "Neoverse-N1",
11
+ "gpu": "NVIDIA GeForce RTX 4060",
12
+ "dependencies": {
13
+ "makewfs": "2.0.0",
14
+ "numpy": "2.5.3",
15
+ "scipy": "1.18.1",
16
+ "cupy": "14.2.0",
17
+ "getframes": "2.4.0",
18
+ "pyturb": "2.2.0"
19
+ },
20
+ "methodology": {
21
+ "frames_per_cell": 100,
22
+ "persistent_sensor": true,
23
+ "warmup_frames": 1,
24
+ "device_resident_opd_rate_and_output": true,
25
+ "include_detector_truth": true,
26
+ "rng": "a distinct deterministic per-frame seed",
27
+ "cuda_synchronization": "before and after each timed region",
28
+ "construction_included_in_throughput": false,
29
+ "host_transfers_included": false
30
+ },
31
+ "results": [
32
+ {
33
+ "config": "benchmarks/configs/pyramid_40_float32.toml",
34
+ "sensor": "pyramid",
35
+ "device": "cpu",
36
+ "shape": [
37
+ 54,
38
+ 54
39
+ ],
40
+ "source_states": 1,
41
+ "wavelength_samples": 1,
42
+ "range_samples": 1,
43
+ "modulation_samples": 1,
44
+ "modulation_radius_lambda_over_d": 0.0,
45
+ "dtype": "float32",
46
+ "fft_workers": 1,
47
+ "frames": 100,
48
+ "construction_s": 0.7629785779863596,
49
+ "warm_optical_total_s": 0.06306239901459776,
50
+ "warm_optical_frame_s": 0.0006306239901459775,
51
+ "warm_detector_total_s": 0.10588375400402583,
52
+ "warm_detector_frame_s": 0.0010588375400402584,
53
+ "warm_optical_frames_per_s": 1585.7309833210102,
54
+ "warm_detector_frames_per_s": 944.4319474751327,
55
+ "warm_detector_only_frame_s": 0.00022277187992585822,
56
+ "warm_detector_only_frames_per_s": 4488.896894584786,
57
+ "python_peak_memory_mib": null
58
+ },
59
+ {
60
+ "config": "benchmarks/configs/pyramid_40_float32.toml",
61
+ "sensor": "pyramid",
62
+ "device": "gpu",
63
+ "shape": [
64
+ 54,
65
+ 54
66
+ ],
67
+ "source_states": 1,
68
+ "wavelength_samples": 1,
69
+ "range_samples": 1,
70
+ "modulation_samples": 1,
71
+ "modulation_radius_lambda_over_d": 0.0,
72
+ "dtype": "float32",
73
+ "fft_workers": 1,
74
+ "frames": 100,
75
+ "construction_s": 0.2768877709750086,
76
+ "warm_optical_total_s": 0.13808018100098707,
77
+ "warm_optical_frame_s": 0.0013808018100098707,
78
+ "warm_detector_total_s": 0.24444962001871318,
79
+ "warm_detector_frame_s": 0.0024444962001871316,
80
+ "warm_optical_frames_per_s": 724.2168953941706,
81
+ "warm_detector_frames_per_s": 409.08224767273015,
82
+ "warm_detector_only_frame_s": 0.0006185778399230912,
83
+ "warm_detector_only_frames_per_s": 1616.6114197112713,
84
+ "python_peak_memory_mib": null
85
+ },
86
+ {
87
+ "config": "benchmarks/configs/pyramid_60_mod8_float32.toml",
88
+ "sensor": "pyramid",
89
+ "device": "cpu",
90
+ "shape": [
91
+ 80,
92
+ 80
93
+ ],
94
+ "source_states": 1,
95
+ "wavelength_samples": 1,
96
+ "range_samples": 1,
97
+ "modulation_samples": 8,
98
+ "modulation_radius_lambda_over_d": 2.0,
99
+ "dtype": "float32",
100
+ "fft_workers": 1,
101
+ "frames": 100,
102
+ "construction_s": 0.07376596701215021,
103
+ "warm_optical_total_s": 0.5042883460118901,
104
+ "warm_optical_frame_s": 0.005042883460118901,
105
+ "warm_detector_total_s": 0.5606069839850534,
106
+ "warm_detector_frame_s": 0.005606069839850534,
107
+ "warm_optical_frames_per_s": 198.29924841777367,
108
+ "warm_detector_frames_per_s": 178.37808457032378,
109
+ "warm_detector_only_frame_s": 0.00035893032007152216,
110
+ "warm_detector_only_frames_per_s": 2786.056078518904,
111
+ "python_peak_memory_mib": null
112
+ },
113
+ {
114
+ "config": "benchmarks/configs/pyramid_60_mod8_float32.toml",
115
+ "sensor": "pyramid",
116
+ "device": "gpu",
117
+ "shape": [
118
+ 80,
119
+ 80
120
+ ],
121
+ "source_states": 1,
122
+ "wavelength_samples": 1,
123
+ "range_samples": 1,
124
+ "modulation_samples": 8,
125
+ "modulation_radius_lambda_over_d": 2.0,
126
+ "dtype": "float32",
127
+ "fft_workers": 1,
128
+ "frames": 100,
129
+ "construction_s": 0.012662593013374135,
130
+ "warm_optical_total_s": 0.1342540149926208,
131
+ "warm_optical_frame_s": 0.0013425401499262079,
132
+ "warm_detector_total_s": 0.2391750770038925,
133
+ "warm_detector_frame_s": 0.002391750770038925,
134
+ "warm_optical_frames_per_s": 744.8566808634844,
135
+ "warm_detector_frames_per_s": 418.10376420772525,
136
+ "warm_detector_only_frame_s": 0.0006130585700157099,
137
+ "warm_detector_only_frames_per_s": 1631.1655181239448,
138
+ "python_peak_memory_mib": null
139
+ },
140
+ {
141
+ "config": "benchmarks/configs/pyramid_80_mod32_float64.toml",
142
+ "sensor": "pyramid",
143
+ "device": "cpu",
144
+ "shape": [
145
+ 108,
146
+ 108
147
+ ],
148
+ "source_states": 1,
149
+ "wavelength_samples": 1,
150
+ "range_samples": 1,
151
+ "modulation_samples": 32,
152
+ "modulation_radius_lambda_over_d": 3.0,
153
+ "dtype": "float64",
154
+ "fft_workers": 1,
155
+ "frames": 100,
156
+ "construction_s": 0.08048216797760688,
157
+ "warm_optical_total_s": 8.533555491012521,
158
+ "warm_optical_frame_s": 0.0853355549101252,
159
+ "warm_detector_total_s": 8.6366516520211,
160
+ "warm_detector_frame_s": 0.086366516520211,
161
+ "warm_optical_frames_per_s": 11.718444920797582,
162
+ "warm_detector_frames_per_s": 11.578561232882254,
163
+ "warm_detector_only_frame_s": 0.0006370968700502999,
164
+ "warm_detector_only_frames_per_s": 1569.620017001572,
165
+ "python_peak_memory_mib": null
166
+ },
167
+ {
168
+ "config": "benchmarks/configs/pyramid_80_mod32_float64.toml",
169
+ "sensor": "pyramid",
170
+ "device": "gpu",
171
+ "shape": [
172
+ 108,
173
+ 108
174
+ ],
175
+ "source_states": 1,
176
+ "wavelength_samples": 1,
177
+ "range_samples": 1,
178
+ "modulation_samples": 32,
179
+ "modulation_radius_lambda_over_d": 3.0,
180
+ "dtype": "float64",
181
+ "fft_workers": 1,
182
+ "frames": 100,
183
+ "construction_s": 0.025755188980838284,
184
+ "warm_optical_total_s": 0.4056679999921471,
185
+ "warm_optical_frame_s": 0.004056679999921471,
186
+ "warm_detector_total_s": 0.4848081910167821,
187
+ "warm_detector_frame_s": 0.004848081910167821,
188
+ "warm_optical_frames_per_s": 246.50699587331462,
189
+ "warm_detector_frames_per_s": 206.26714204285878,
190
+ "warm_detector_only_frame_s": 0.0005712536702048964,
191
+ "warm_detector_only_frames_per_s": 1750.5357989933289,
192
+ "python_peak_memory_mib": null
193
+ },
194
+ {
195
+ "config": "benchmarks/configs/shack_hartmann_20x20_float32.toml",
196
+ "sensor": "shack_hartmann",
197
+ "device": "cpu",
198
+ "shape": [
199
+ 160,
200
+ 160
201
+ ],
202
+ "source_states": 1,
203
+ "wavelength_samples": 1,
204
+ "range_samples": 1,
205
+ "modulation_samples": 1,
206
+ "modulation_radius_lambda_over_d": 0.0,
207
+ "dtype": "float32",
208
+ "fft_workers": 1,
209
+ "frames": 100,
210
+ "construction_s": 0.08199134599999525,
211
+ "warm_optical_total_s": 2.28855264998856,
212
+ "warm_optical_frame_s": 0.0228855264998856,
213
+ "warm_detector_total_s": 2.4490860599908046,
214
+ "warm_detector_frame_s": 0.024490860599908047,
215
+ "warm_optical_frames_per_s": 43.695739313884644,
216
+ "warm_detector_frames_per_s": 40.831558201909594,
217
+ "warm_detector_only_frame_s": 0.0010764133499469608,
218
+ "warm_detector_only_frames_per_s": 929.0111461821558,
219
+ "python_peak_memory_mib": null
220
+ },
221
+ {
222
+ "config": "benchmarks/configs/shack_hartmann_20x20_float32.toml",
223
+ "sensor": "shack_hartmann",
224
+ "device": "gpu",
225
+ "shape": [
226
+ 160,
227
+ 160
228
+ ],
229
+ "source_states": 1,
230
+ "wavelength_samples": 1,
231
+ "range_samples": 1,
232
+ "modulation_samples": 1,
233
+ "modulation_radius_lambda_over_d": 0.0,
234
+ "dtype": "float32",
235
+ "fft_workers": 1,
236
+ "frames": 100,
237
+ "construction_s": 0.01180534201557748,
238
+ "warm_optical_total_s": 0.08716972899856046,
239
+ "warm_optical_frame_s": 0.0008716972899856046,
240
+ "warm_detector_total_s": 0.19479694298934191,
241
+ "warm_detector_frame_s": 0.0019479694298934192,
242
+ "warm_optical_frames_per_s": 1147.187230576929,
243
+ "warm_detector_frames_per_s": 513.3550787060933,
244
+ "warm_detector_only_frame_s": 0.0006126053698244505,
245
+ "warm_detector_only_frames_per_s": 1632.372240365053,
246
+ "python_peak_memory_mib": null
247
+ },
248
+ {
249
+ "config": "benchmarks/configs/shack_hartmann_60x60_float64.toml",
250
+ "sensor": "shack_hartmann",
251
+ "device": "cpu",
252
+ "shape": [
253
+ 360,
254
+ 360
255
+ ],
256
+ "source_states": 1,
257
+ "wavelength_samples": 1,
258
+ "range_samples": 1,
259
+ "modulation_samples": 1,
260
+ "modulation_radius_lambda_over_d": 0.0,
261
+ "dtype": "float64",
262
+ "fft_workers": 1,
263
+ "frames": 100,
264
+ "construction_s": 0.08440429499023594,
265
+ "warm_optical_total_s": 8.684691948990803,
266
+ "warm_optical_frame_s": 0.08684691948990803,
267
+ "warm_detector_total_s": 9.488772381999297,
268
+ "warm_detector_frame_s": 0.09488772381999297,
269
+ "warm_optical_frames_per_s": 11.514513190260066,
270
+ "warm_detector_frames_per_s": 10.538771083781636,
271
+ "warm_detector_only_frame_s": 0.005910381099965889,
272
+ "warm_detector_only_frames_per_s": 169.19382745146018,
273
+ "python_peak_memory_mib": null
274
+ },
275
+ {
276
+ "config": "benchmarks/configs/shack_hartmann_60x60_float64.toml",
277
+ "sensor": "shack_hartmann",
278
+ "device": "gpu",
279
+ "shape": [
280
+ 360,
281
+ 360
282
+ ],
283
+ "source_states": 1,
284
+ "wavelength_samples": 1,
285
+ "range_samples": 1,
286
+ "modulation_samples": 1,
287
+ "modulation_radius_lambda_over_d": 0.0,
288
+ "dtype": "float64",
289
+ "fft_workers": 1,
290
+ "frames": 100,
291
+ "construction_s": 0.026143675000639632,
292
+ "warm_optical_total_s": 0.4227883660059888,
293
+ "warm_optical_frame_s": 0.004227883660059888,
294
+ "warm_detector_total_s": 0.4985939179896377,
295
+ "warm_detector_frame_s": 0.004985939179896377,
296
+ "warm_optical_frames_per_s": 236.52495678791573,
297
+ "warm_detector_frames_per_s": 200.5640189178527,
298
+ "warm_detector_only_frame_s": 0.0005747833198984153,
299
+ "warm_detector_only_frames_per_s": 1739.7860469867767,
300
+ "python_peak_memory_mib": null
301
+ },
302
+ {
303
+ "config": "benchmarks/configs/shack_hartmann_quadrature_9sample.toml",
304
+ "sensor": "shack_hartmann",
305
+ "device": "cpu",
306
+ "shape": [
307
+ 64,
308
+ 64
309
+ ],
310
+ "source_states": 9,
311
+ "wavelength_samples": 3,
312
+ "range_samples": 3,
313
+ "modulation_samples": 1,
314
+ "modulation_radius_lambda_over_d": 0.0,
315
+ "dtype": "float32",
316
+ "fft_workers": 1,
317
+ "frames": 100,
318
+ "construction_s": 0.06993808099650778,
319
+ "warm_optical_total_s": 1.8203168570180424,
320
+ "warm_optical_frame_s": 0.018203168570180422,
321
+ "warm_detector_total_s": 1.8593171660031658,
322
+ "warm_detector_frame_s": 0.018593171660031656,
323
+ "warm_optical_frames_per_s": 54.93549082647914,
324
+ "warm_detector_frames_per_s": 53.78318547715153,
325
+ "warm_detector_only_frame_s": 0.00024450053984764964,
326
+ "warm_detector_only_frames_per_s": 4089.9705195870265,
327
+ "python_peak_memory_mib": null
328
+ },
329
+ {
330
+ "config": "benchmarks/configs/shack_hartmann_quadrature_9sample.toml",
331
+ "sensor": "shack_hartmann",
332
+ "device": "gpu",
333
+ "shape": [
334
+ 64,
335
+ 64
336
+ ],
337
+ "source_states": 9,
338
+ "wavelength_samples": 3,
339
+ "range_samples": 3,
340
+ "modulation_samples": 1,
341
+ "modulation_radius_lambda_over_d": 0.0,
342
+ "dtype": "float32",
343
+ "fft_workers": 1,
344
+ "frames": 100,
345
+ "construction_s": 0.014552935026586056,
346
+ "warm_optical_total_s": 0.4028777669882402,
347
+ "warm_optical_frame_s": 0.004028777669882402,
348
+ "warm_detector_total_s": 0.5134580160083715,
349
+ "warm_detector_frame_s": 0.005134580160083715,
350
+ "warm_optical_frames_per_s": 248.2142431128967,
351
+ "warm_detector_frames_per_s": 194.75789038683465,
352
+ "warm_detector_only_frame_s": 0.0006140789898927323,
353
+ "warm_detector_only_frames_per_s": 1628.4549975804912,
354
+ "python_peak_memory_mib": null
355
+ }
356
+ ]
357
+ }