makewfs 1.2.0__tar.gz → 2.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (148) hide show
  1. {makewfs-1.2.0 → makewfs-2.1.0}/AGENTS.md +48 -4
  2. {makewfs-1.2.0 → makewfs-2.1.0}/CHANGELOG.md +103 -0
  3. {makewfs-1.2.0 → makewfs-2.1.0}/CITATION.cff +1 -1
  4. {makewfs-1.2.0 → makewfs-2.1.0}/PKG-INFO +5 -5
  5. {makewfs-1.2.0 → makewfs-2.1.0}/README.md +3 -3
  6. makewfs-2.1.0/benchmarks/device-results-neoverse-n1-rtx4060.json +357 -0
  7. makewfs-2.1.0/benchmarks/device-results-neoverse-n1-rtxa400.json +195 -0
  8. makewfs-2.1.0/benchmarks/device-results-neoverse-n1.md +40 -0
  9. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/run.py +21 -9
  10. makewfs-2.1.0/docs/concepts.md +67 -0
  11. {makewfs-1.2.0 → makewfs-2.1.0}/docs/performance.md +57 -5
  12. {makewfs-1.2.0 → makewfs-2.1.0}/docs/stability.md +17 -0
  13. {makewfs-1.2.0 → makewfs-2.1.0}/examples/closed_loop_injection.py +8 -5
  14. {makewfs-1.2.0 → makewfs-2.1.0}/examples/showcase.py +3 -2
  15. {makewfs-1.2.0 → makewfs-2.1.0}/pyproject.toml +1 -1
  16. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/__about__.py +1 -1
  17. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/api.py +59 -11
  18. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/backend.py +133 -0
  19. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/provenance.py +10 -16
  20. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/sampling.py +146 -42
  21. makewfs-2.1.0/src/makewfs/sensors/_cuda_graph.py +84 -0
  22. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/sensors/_shack_hartmann_cuda.py +19 -2
  23. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/sensors/base.py +6 -0
  24. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/sensors/pyramid.py +67 -20
  25. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/sensors/shack_hartmann.py +8 -3
  26. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/wavefront.py +85 -6
  27. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_gpu_backend.py +151 -0
  28. makewfs-2.1.0/tests/test_input_rms.py +213 -0
  29. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_numerics.py +114 -0
  30. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_provenance.py +0 -30
  31. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_pyramid.py +46 -0
  32. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_shack_hartmann_sampling.py +3 -3
  33. makewfs-1.2.0/docs/concepts.md +0 -38
  34. {makewfs-1.2.0 → makewfs-2.1.0}/.gitignore +0 -0
  35. {makewfs-1.2.0 → makewfs-2.1.0}/CONTRIBUTING.md +0 -0
  36. {makewfs-1.2.0 → makewfs-2.1.0}/LICENSE +0 -0
  37. {makewfs-1.2.0 → makewfs-2.1.0}/ROADMAP.md +0 -0
  38. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/__init__.py +0 -0
  39. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/benchmark_compiled_sh_executor.py +0 -0
  40. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/benchmark_sh_state_batching.py +0 -0
  41. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/check_regression.py +0 -0
  42. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/configs/pyramid_40_float32.toml +0 -0
  43. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/configs/pyramid_60_mod8_float32.toml +0 -0
  44. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/configs/pyramid_80_mod32_float64.toml +0 -0
  45. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/configs/shack_hartmann_20x20_float32.toml +0 -0
  46. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/configs/shack_hartmann_60x60_float64.toml +0 -0
  47. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/configs/shack_hartmann_quadrature_9sample.toml +0 -0
  48. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/device-results.json +0 -0
  49. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/device-results.md +0 -0
  50. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/haka-compiled-sh-executor-quadro-p620.json +0 -0
  51. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/haka-sh-state-batching-quadro-p620.json +0 -0
  52. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/profile_warm.py +0 -0
  53. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/reference-results.json +0 -0
  54. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/reference-table.md +0 -0
  55. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/render_device_table.py +0 -0
  56. {makewfs-1.2.0 → makewfs-2.1.0}/benchmarks/render_table.py +0 -0
  57. {makewfs-1.2.0 → makewfs-2.1.0}/docs/adr/0001-units-coordinates.md +0 -0
  58. {makewfs-1.2.0 → makewfs-2.1.0}/docs/adr/0002-flux-normalization.md +0 -0
  59. {makewfs-1.2.0 → makewfs-2.1.0}/docs/adr/0003-public-api.md +0 -0
  60. {makewfs-1.2.0 → makewfs-2.1.0}/docs/adr/0004-backend-boundary.md +0 -0
  61. {makewfs-1.2.0 → makewfs-2.1.0}/docs/adr/index.md +0 -0
  62. {makewfs-1.2.0 → makewfs-2.1.0}/docs/api.md +0 -0
  63. {makewfs-1.2.0 → makewfs-2.1.0}/docs/configuration.md +0 -0
  64. {makewfs-1.2.0 → makewfs-2.1.0}/docs/contributing.md +0 -0
  65. {makewfs-1.2.0 → makewfs-2.1.0}/docs/detectors.md +0 -0
  66. {makewfs-1.2.0 → makewfs-2.1.0}/docs/examples.md +0 -0
  67. {makewfs-1.2.0 → makewfs-2.1.0}/docs/gallery/makewfs-gallery.json +0 -0
  68. {makewfs-1.2.0 → makewfs-2.1.0}/docs/gallery/makewfs-gallery.svg +0 -0
  69. {makewfs-1.2.0 → makewfs-2.1.0}/docs/gallery.md +0 -0
  70. {makewfs-1.2.0 → makewfs-2.1.0}/docs/guide-stars.md +0 -0
  71. {makewfs-1.2.0 → makewfs-2.1.0}/docs/index.md +0 -0
  72. {makewfs-1.2.0 → makewfs-2.1.0}/docs/interop.md +0 -0
  73. {makewfs-1.2.0 → makewfs-2.1.0}/docs/pyramid.md +0 -0
  74. {makewfs-1.2.0 → makewfs-2.1.0}/docs/quickstart.md +0 -0
  75. {makewfs-1.2.0 → makewfs-2.1.0}/docs/release.md +0 -0
  76. {makewfs-1.2.0 → makewfs-2.1.0}/docs/shack-hartmann.md +0 -0
  77. {makewfs-1.2.0 → makewfs-2.1.0}/docs/troubleshooting.md +0 -0
  78. {makewfs-1.2.0 → makewfs-2.1.0}/docs/units-and-coordinates.md +0 -0
  79. {makewfs-1.2.0 → makewfs-2.1.0}/docs/validation.md +0 -0
  80. {makewfs-1.2.0 → makewfs-2.1.0}/examples/README.md +0 -0
  81. {makewfs-1.2.0 → makewfs-2.1.0}/examples/cds_readout.py +0 -0
  82. {makewfs-1.2.0 → makewfs-2.1.0}/examples/compare_sensors.py +0 -0
  83. {makewfs-1.2.0 → makewfs-2.1.0}/examples/configs/angular_kernel.txt +0 -0
  84. {makewfs-1.2.0 → makewfs-2.1.0}/examples/configs/lgs_thin_beacon.toml +0 -0
  85. {makewfs-1.2.0 → makewfs-2.1.0}/examples/configs/precision_throughput.toml +0 -0
  86. {makewfs-1.2.0 → makewfs-2.1.0}/examples/configs/pyramid_minimal.toml +0 -0
  87. {makewfs-1.2.0 → makewfs-2.1.0}/examples/configs/qe_curve.txt +0 -0
  88. {makewfs-1.2.0 → makewfs-2.1.0}/examples/configs/shack_hartmann_extended_source.toml +0 -0
  89. {makewfs-1.2.0 → makewfs-2.1.0}/examples/configs/shack_hartmann_minimal.toml +0 -0
  90. {makewfs-1.2.0 → makewfs-2.1.0}/examples/configs/shack_hartmann_spectral_qe.toml +0 -0
  91. {makewfs-1.2.0 → makewfs-2.1.0}/examples/detector_choices.py +0 -0
  92. {makewfs-1.2.0 → makewfs-2.1.0}/examples/gallery.py +0 -0
  93. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/README.md +0 -0
  94. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/analyze_lut.py +0 -0
  95. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/benchmark.py +0 -0
  96. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/camera_modes.csv +0 -0
  97. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/camera_modes_empirical_floor_continuous.csv +0 -0
  98. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/compare_real.py +0 -0
  99. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/fit_secondary.py +0 -0
  100. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/haka_cpu_gpu_benchmark.json +0 -0
  101. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/haka_lut_snr.json +0 -0
  102. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/keck_haka.json +0 -0
  103. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/keck_haka.toml +0 -0
  104. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/mauna_kea_extinction.csv +0 -0
  105. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/mauna_kea_extinction_nir.csv +0 -0
  106. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/ocam_20260720/extract_ocam_images.py +0 -0
  107. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/ocam_20260720/make_ocam_video.py +0 -0
  108. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/real_vs_simulation.json +0 -0
  109. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/secondary_fit.json +0 -0
  110. {makewfs-1.2.0 → makewfs-2.1.0}/examples/keck_haka/simulate.py +0 -0
  111. {makewfs-1.2.0 → makewfs-2.1.0}/examples/lgs_elongation.py +0 -0
  112. {makewfs-1.2.0 → makewfs-2.1.0}/examples/lgs_thin_beacon.py +0 -0
  113. {makewfs-1.2.0 → makewfs-2.1.0}/examples/magnitude_series.py +0 -0
  114. {makewfs-1.2.0 → makewfs-2.1.0}/examples/makewfs_showcase.webp +0 -0
  115. {makewfs-1.2.0 → makewfs-2.1.0}/examples/moving_atmosphere.py +0 -0
  116. {makewfs-1.2.0 → makewfs-2.1.0}/examples/precision_throughput.py +0 -0
  117. {makewfs-1.2.0 → makewfs-2.1.0}/examples/pyramid_modulation.py +0 -0
  118. {makewfs-1.2.0 → makewfs-2.1.0}/examples/quickstart.py +0 -0
  119. {makewfs-1.2.0 → makewfs-2.1.0}/examples/realistic_broadband.py +0 -0
  120. {makewfs-1.2.0 → makewfs-2.1.0}/examples/sh_design_trade.py +0 -0
  121. {makewfs-1.2.0 → makewfs-2.1.0}/examples/spectral_qe.py +0 -0
  122. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/__init__.py +0 -0
  123. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/cli.py +0 -0
  124. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/config.py +0 -0
  125. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/detector.py +0 -0
  126. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/pupil.py +0 -0
  127. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/py.typed +0 -0
  128. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/radiometry.py +0 -0
  129. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/sensors/__init__.py +0 -0
  130. {makewfs-1.2.0 → makewfs-2.1.0}/src/makewfs/source.py +0 -0
  131. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_backend_audit.py +0 -0
  132. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_benchmarks.py +0 -0
  133. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_cli.py +0 -0
  134. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_config.py +0 -0
  135. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_conformance.py +0 -0
  136. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_hcipy_validation.py +0 -0
  137. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_interop.py +0 -0
  138. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_keck_haka_example.py +0 -0
  139. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_oopao_validation.py +0 -0
  140. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_optics_validation.py +0 -0
  141. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_public_api.py +0 -0
  142. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_radiometry.py +0 -0
  143. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_shack_hartmann.py +0 -0
  144. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_source.py +0 -0
  145. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_validation_report.py +0 -0
  146. {makewfs-1.2.0 → makewfs-2.1.0}/tests/test_wavefront.py +0 -0
  147. {makewfs-1.2.0 → makewfs-2.1.0}/validation/__init__.py +0 -0
  148. {makewfs-1.2.0 → makewfs-2.1.0}/validation/run.py +0 -0
@@ -87,6 +87,16 @@ write a failing integration test/design note, use the conditional gates in
87
87
  - Intensities, not fields, are summed over incoherent wavelengths, modulation
88
88
  points, finite-source samples, and sodium slices.
89
89
  - Cropping reports lost flux; it does not renormalize it away.
90
+ - Wavefront metrics follow aocore CONVENTIONS 4.1. Frame metadata
91
+ `wfs_input_opd_rms_m` is the pupil-intensity-weighted, piston-removed RMS of
92
+ the input OPD on the input grid (weights from `WavefrontSensor._rms_weights`:
93
+ the analytic pupil evaluated on `input.shape`, or a custom mask area-averaged
94
+ from the engine's `configured_pupil`); `wfs_input_opd_rms_unweighted_m` is the
95
+ whole-grid quadratic mean with piston kept. Both are reduced on the device and
96
+ cross in the one `backend.scalars` batch, so do not swap in `aocore.rms` or
97
+ `aocore.rms_unweighted` there: they return host floats and would add a
98
+ synchronization each. Any other RMS-like key must say its variant in its
99
+ name (`_unweighted`, `_tiptilt_removed`).
90
100
  - The intended top-level API is `load_config`, `WavefrontSensor`, and `simulate`.
91
101
  Keep other implementation objects out of `makewfs.__init__` unless an API review
92
102
  explicitly accepts them.
@@ -153,12 +163,19 @@ Follow the target layout in `ROADMAP.md`:
153
163
  rotation and rectangular grids on the selected backend, none of which
154
164
  `aocore.Pupil` models. `backend.ArrayBackend` takes a dtype per array, explicit
155
165
  FFT workers and `ndimage` helpers, which `aocore.Backend` does not, and its
156
- centred FFTs keep the `fftshift` convention. `sampling.block_sum` wraps
157
- `aocore.block_sum` but keeps its own factor-two fast path.
166
+ centred FFTs keep the `fftshift` convention; `ArrayBackend.centered_coordinates`
167
+ builds aocore's coordinates directly on the device. `sampling.block_sum` is a
168
+ thin wrapper over `aocore.block_sum` (no local fast path since aocore 0.1.3,
169
+ which measured at least as fast at the SH call sites).
170
+ `sampling.area_rebin` is exact-overlap area averaging between grids of the
171
+ same extent and any shape ratio, which `aocore.block_sum` (integer factors
172
+ only) does not cover.
158
173
  - `sensors/` contains deterministic ideal optical engines and no camera noise.
159
174
  `_shack_hartmann_cuda.py` is a private first-use-JIT execution plan for exact
160
- compatible CUDA geometries; `shack_hartmann.py` remains the readable physics
161
- reference and must stay as the automatic feature-complete fallback.
175
+ compatible CUDA geometries (sampled-DFT and integer-FFT spot grids alike);
176
+ `shack_hartmann.py` remains the readable physics reference and must stay as
177
+ the automatic feature-complete fallback. `_cuda_graph.py` replays the
178
+ pyramid's fixed-shape GPU propagation from a captured CUDA graph.
162
179
  - `radiometry.py` produces source photon budgets using public `getframes` tools.
163
180
  - `detector.py` is a narrow adapter to `getframes.Camera.expose` and the
164
181
  optional public `expose_spectral` cube API, plus the
@@ -221,6 +238,33 @@ python benchmarks/check_regression.py /tmp/makewfs-benchmark.json
221
238
  MPLBACKEND=Agg python examples/gallery.py
222
239
  ```
223
240
 
241
+ ## Performance gotchas
242
+
243
+ - `ArrayBackend.pruned_fft2` skips all-zero input lines and cropped output
244
+ lines of a 2-D FFT. Keep its `axes` equal to the pass order of the transform
245
+ it replaces: SciPy's `fft2` runs the listed axes in order when
246
+ `overwrite_x=True` but the last axis first when it allocates its output, and
247
+ the two orders round differently. SciPy also transforms lines in SIMD groups
248
+ whose short remainder group rounds differently, so a pruned transform can
249
+ differ from the full one by an ulp where the line counts differ. Check
250
+ candidate changes with a saved before/after render matrix, not only tests.
251
+ - The pyramid applies its mask on the unshifted FFT grid (the stored mask is
252
+ `ifftshift`-ed) and turns the outer shifts into wrapped start indices. Do not
253
+ reintroduce `fftshift`/`ifftshift` copies there; they are pure permutations.
254
+ - `PyramidEngine._propagate` is captured as a CUDA graph on a GPU. It must keep
255
+ fixed shapes and never synchronize with the host (no `scalar`, `.item()`,
256
+ `to_host`, or data-dependent shapes). A failed capture silently falls back to
257
+ eager execution and records why in `engine._graph.failure`; check it after
258
+ changing that method.
259
+ - The compiled SH kernel reads the OPD as float64; `_CompiledShackHartmannExecutor.render`
260
+ widens a float32 lenslet-grid resample (exact). Its static thread and
261
+ shared-memory checks cannot see register pressure, so construction also
262
+ checks the compiled kernel's `max_threads_per_block` and falls back.
263
+ - `ArrayBackend.next_fast_length` uses SciPy on the CPU and CuPy on the GPU,
264
+ which disagree when SciPy picks a factor of 11 (for example 33, 66, 99). The
265
+ pyramid's FFT size, and therefore its result, can then differ between
266
+ devices; parity tests use geometries where they agree.
267
+
224
268
  ## Documentation and examples
225
269
 
226
270
  - Every public feature lands with its API docstring and the relevant user guide.
@@ -4,6 +4,109 @@ All notable changes to `makewfs` are documented here.
4
4
 
5
5
  ## [Unreleased]
6
6
 
7
+ ## [2.1.0] - 2026-10-07
8
+
9
+ ### Performance
10
+
11
+ - **Pruned FFTs on CPU and GPU.** Shack-Hartmann spots and the pyramid
12
+ transform only the FFT lines that hold data and keep only the cropped
13
+ output lines; the pyramid applies its mask on the unshifted grid instead of
14
+ shifting four full grids per state. CPU results are unchanged (bit-identical
15
+ in 451 of 468 arrays of a before/after matrix, otherwise within 2.5e-7
16
+ relative in float32).
17
+ - **CUDA-graph replay of the pyramid.** On a GPU the pyramid's fixed-shape
18
+ propagation is captured once and replayed with one launch.
19
+ - **Compiled CUDA executor for integer-FFT Shack-Hartmann grids**, which
20
+ previously ran the array path. Results agree to float rounding.
21
+ - End-to-end frames/s on an Ampere Neoverse-N1 (12 cores) and an RTX 4060,
22
+ 2.0.0 -> now: SH 20x20 float32 CPU 38.5 -> 126.1 (3.3x), GPU 496 -> 813
23
+ (1.6x); SH 60x60 float64 CPU 10.4 -> 29.8 (2.9x), GPU 199 -> 640 (3.2x);
24
+ nine-sample SH CPU 52.0 -> 96.3 (1.9x), GPU 189 -> 764 (4.1x); pyramid 40
25
+ CPU 923 -> 1,101 (1.2x), GPU 397 -> 759 (1.9x); pyramid 60 mod-8 CPU 173 ->
26
+ 272 (1.6x), GPU 408 -> 762 (1.9x); pyramid 80 mod-32 float64 CPU 11.4 ->
27
+ 20.9 (1.8x), GPU 204 -> 277 (1.4x). See `docs/performance.md`.
28
+
29
+ ### Fixed
30
+
31
+ - **Compiled CUDA Shack-Hartmann executor misread float32 rotated or offset
32
+ lenslet grids.** Such grids resample the OPD in float32, but the kernel read
33
+ it as float64, so the GPU spot pattern was wrong (82-96% error). The OPD is
34
+ now widened (exactly) before the launch.
35
+ - **Compiled CUDA Shack-Hartmann executor failed to launch for large
36
+ lenslet sampling** (e.g. 32x32 samples, 1024 threads) when register use left
37
+ fewer threads per block. Such geometries now fall back to the array path.
38
+
39
+ - **Benchmark artifacts named the wrong GPU on multi-GPU hosts.**
40
+ `benchmarks/run.py` recorded the first line of `nvidia-smi`, which ignores
41
+ `CUDA_VISIBLE_DEVICES`. It now asks CuPy for the device the run used. On
42
+ Arm hosts, whose `/proc/cpuinfo` has no model name, it reads the CPU model
43
+ from `lscpu` (e.g. `Neoverse-N1`) instead of reporting `aarch64`.
44
+
45
+ ### Added
46
+
47
+ - **Arm benchmark data point** (`benchmarks/device-results-neoverse-n1.*`): the
48
+ device table on an Ampere Neoverse-N1 host (16 pinned cores) with an RTX
49
+ 4060 and an RTX A400.
50
+
51
+ ## [2.0.0] - 2026-10-07
52
+
53
+ ### Breaking
54
+
55
+ - **`wfs_input_opd_rms_m` is now the pupil-weighted, piston-removed RMS.**
56
+ It follows aocore CONVENTIONS.md 4.1: the RMS of the input OPD in metres,
57
+ weighted by the intensity of the pupil the optics use and with the
58
+ intensity-weighted mean (piston) removed. Before 2.0 it was the quadratic
59
+ mean over the whole input grid, with every pixel counted equally, pixels
60
+ outside the pupil included and piston kept. For the same wavefront the new
61
+ value is usually smaller; a pure piston now reports 0. The weights are the
62
+ configured pupil's intensity (the amplitude the engines use, squared) on the
63
+ input grid: an analytic pupil is evaluated on `input.shape` with the
64
+ configured `numerics.pupil_supersampling`, the same map
65
+ `WavefrontSensor.pupil_illumination()` returns; a custom mask, which exists
66
+ only on the engine's pupil grid, is area-averaged onto the input grid.
67
+ Phase input is converted to OPD first, and `expose_integrated` reports the
68
+ RMS of the mean OPD, as before. The value is still reduced on the device and
69
+ crosses to the host in the same single batched transfer as the captured
70
+ photon rate. A pupil with no transmission on the input grid is now rejected
71
+ when the sensor is built.
72
+ - **`makewfs.provenance.metadata`**, an internal helper, takes the two
73
+ reduced RMS values as required `opd_rms_m` and `opd_rms_unweighted_m`
74
+ arguments and no longer accepts `opd_m`.
75
+
76
+ See "Migrating to 2.0" in the stability guide.
77
+
78
+ ### Added
79
+
80
+ - **`wfs_input_opd_rms_unweighted_m`** frame metadata keeps the 1.x quantity,
81
+ the unweighted RMS over the whole input grid with piston included, so no
82
+ information is lost. Read it wherever the old number is still wanted.
83
+ - Both sensor engines expose `configured_pupil`, the pupil amplitude on their
84
+ own pupil grid, and `makewfs.sampling.area_rebin` area-averages a map onto
85
+ another grid of the same extent, for any shape ratio.
86
+
87
+ ### Changed
88
+
89
+ - The `closed_loop_injection.py` and `showcase.py` examples take the residual
90
+ RMS they plot from each frame's `wfs_input_opd_rms_m` instead of a
91
+ whole-grid `np.std`, so the numbers they print are pupil-weighted. The
92
+ checked-in `makewfs_showcase.webp` was not regenerated.
93
+ - **Requires `aocore>=0.1.3,<0.2`.** `makewfs.sampling.block_sum` drops its
94
+ own factor-two shortcut and delegates every factor to `aocore.block_sum`,
95
+ whose strided CPU adds and single CuPy kernel measured at least as fast on
96
+ the spot stacks the Shack-Hartmann actually bins (for example
97
+ `(400, 16, 16)` float32: CPU 112 vs 173 us, Quadro P620 65 vs 94 us;
98
+ `(3600, 12, 12)` float64: CPU 0.90 vs 1.27 ms, GPU 98 vs 173 us). The
99
+ additions happen in a different order, so Shack-Hartmann float32 images
100
+ differ from 1.2.0 by float32 rounding (at most about 1e-7 relative) and
101
+ float64 ones by about 2e-16; pyramid images are bit-for-bit unchanged.
102
+ Pixel-centre coordinates for the input, pupil and DFT detector grids are
103
+ now built on the selected device by `aocore.centered_coordinates`
104
+ (`ArrayBackend.centered_coordinates`), with identical values.
105
+ - The input-RMS tests check the unweighted key against
106
+ `aocore.rms_unweighted`. The per-frame path keeps its own device reduction,
107
+ because the aocore functions return host floats and would each add a
108
+ synchronization.
109
+
7
110
  ## [1.2.0] - 2026-10-07
8
111
 
9
112
  - **Fixed: phase input was reported as metres.** With
@@ -5,7 +5,7 @@ type: software
5
5
  authors:
6
6
  - family-names: Taylor
7
7
  given-names: Jacob
8
- version: 1.2.0
8
+ version: 2.1.0
9
9
  date-released: 2026-10-07
10
10
  license: MIT
11
11
  repository-code: "https://github.com/jacotay7/makewfs"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: makewfs
3
- Version: 1.2.0
3
+ Version: 2.1.0
4
4
  Summary: Configuration-driven adaptive-optics wavefront sensor image simulation.
5
5
  Project-URL: Homepage, https://github.com/jacotay7/makewfs
6
6
  Project-URL: Documentation, https://jacotay7.github.io/makewfs/
@@ -22,7 +22,7 @@ Classifier: Programming Language :: Python :: 3.13
22
22
  Classifier: Topic :: Scientific/Engineering :: Astronomy
23
23
  Classifier: Typing :: Typed
24
24
  Requires-Python: >=3.10
25
- Requires-Dist: aocore<0.2,>=0.1.2
25
+ Requires-Dist: aocore<0.2,>=0.1.3
26
26
  Requires-Dist: getframes>=2.2.0
27
27
  Requires-Dist: numpy>=1.23
28
28
  Requires-Dist: scipy>=1.10
@@ -137,9 +137,9 @@ LGS is monochromatic at the 589 nm D2 line, broadened by only ~0.003 nm, so its
137
137
  spectral axis exists there purely to exercise the polychromatic path. The
138
138
  showcase clip above instead runs a physically correct narrowband beacon sampled
139
139
  across the sodium layer's depth, which is where spot elongation actually comes
140
- from. Compatible sampled-DFT Shack–Hartmann geometries are compiled
141
- on first use, so warm each fixed sensor before measuring or entering a real-time
142
- loop. Reproduce the table with
140
+ from. On a GPU, compatible Shack–Hartmann geometries are compiled and
141
+ pyramid frames are captured as a CUDA graph on first use, so warm each fixed
142
+ sensor before measuring or entering a real-time loop. Reproduce the table with
143
143
 
144
144
  ```bash
145
145
  python benchmarks/run.py --device both --frames 100 \
@@ -88,9 +88,9 @@ LGS is monochromatic at the 589 nm D2 line, broadened by only ~0.003 nm, so its
88
88
  spectral axis exists there purely to exercise the polychromatic path. The
89
89
  showcase clip above instead runs a physically correct narrowband beacon sampled
90
90
  across the sodium layer's depth, which is where spot elongation actually comes
91
- from. Compatible sampled-DFT Shack–Hartmann geometries are compiled
92
- on first use, so warm each fixed sensor before measuring or entering a real-time
93
- loop. Reproduce the table with
91
+ from. On a GPU, compatible Shack–Hartmann geometries are compiled and
92
+ pyramid frames are captured as a CUDA graph on first use, so warm each fixed
93
+ sensor before measuring or entering a real-time loop. Reproduce the table with
94
94
 
95
95
  ```bash
96
96
  python benchmarks/run.py --device both --frames 100 \
@@ -0,0 +1,357 @@
1
+ {
2
+ "schema_version": 2,
3
+ "generated_at_utc": "2026-10-07T07:11:47.632479+00:00",
4
+ "command": "/home/jtaylor/miniforge3/envs/aosim-bench/bin/python benchmarks/run.py --device both --frames 100 --config benchmarks/configs/pyramid_40_float32.toml --config benchmarks/configs/pyramid_60_mod8_float32.toml --config benchmarks/configs/pyramid_80_mod32_float64.toml --config benchmarks/configs/shack_hartmann_20x20_float32.toml --config benchmarks/configs/shack_hartmann_60x60_float64.toml --config benchmarks/configs/shack_hartmann_quadrature_9sample.toml --output /home/jtaylor/aosim-bench-logs/artifacts/makewfs-device-table-rtx4060.json",
5
+ "revision": "74b53697569dc8b2b5ce93f0b66e39a1de3b303d",
6
+ "source_dirty": true,
7
+ "python": "3.13.15 | packaged by conda-forge | (main, Sep 2 2026, 22:02:03) [GCC 15.3.0]",
8
+ "platform": "Linux-6.17.9-76061709-generic-aarch64-with-glibc2.39",
9
+ "processor": "aarch64",
10
+ "cpu": "Neoverse-N1",
11
+ "gpu": "NVIDIA GeForce RTX 4060",
12
+ "dependencies": {
13
+ "makewfs": "2.0.0",
14
+ "numpy": "2.5.3",
15
+ "scipy": "1.18.1",
16
+ "cupy": "14.2.0",
17
+ "getframes": "2.4.0",
18
+ "pyturb": "2.2.0"
19
+ },
20
+ "methodology": {
21
+ "frames_per_cell": 100,
22
+ "persistent_sensor": true,
23
+ "warmup_frames": 1,
24
+ "device_resident_opd_rate_and_output": true,
25
+ "include_detector_truth": true,
26
+ "rng": "a distinct deterministic per-frame seed",
27
+ "cuda_synchronization": "before and after each timed region",
28
+ "construction_included_in_throughput": false,
29
+ "host_transfers_included": false
30
+ },
31
+ "results": [
32
+ {
33
+ "config": "benchmarks/configs/pyramid_40_float32.toml",
34
+ "sensor": "pyramid",
35
+ "device": "cpu",
36
+ "shape": [
37
+ 54,
38
+ 54
39
+ ],
40
+ "source_states": 1,
41
+ "wavelength_samples": 1,
42
+ "range_samples": 1,
43
+ "modulation_samples": 1,
44
+ "modulation_radius_lambda_over_d": 0.0,
45
+ "dtype": "float32",
46
+ "fft_workers": 1,
47
+ "frames": 100,
48
+ "construction_s": 0.7629785779863596,
49
+ "warm_optical_total_s": 0.06306239901459776,
50
+ "warm_optical_frame_s": 0.0006306239901459775,
51
+ "warm_detector_total_s": 0.10588375400402583,
52
+ "warm_detector_frame_s": 0.0010588375400402584,
53
+ "warm_optical_frames_per_s": 1585.7309833210102,
54
+ "warm_detector_frames_per_s": 944.4319474751327,
55
+ "warm_detector_only_frame_s": 0.00022277187992585822,
56
+ "warm_detector_only_frames_per_s": 4488.896894584786,
57
+ "python_peak_memory_mib": null
58
+ },
59
+ {
60
+ "config": "benchmarks/configs/pyramid_40_float32.toml",
61
+ "sensor": "pyramid",
62
+ "device": "gpu",
63
+ "shape": [
64
+ 54,
65
+ 54
66
+ ],
67
+ "source_states": 1,
68
+ "wavelength_samples": 1,
69
+ "range_samples": 1,
70
+ "modulation_samples": 1,
71
+ "modulation_radius_lambda_over_d": 0.0,
72
+ "dtype": "float32",
73
+ "fft_workers": 1,
74
+ "frames": 100,
75
+ "construction_s": 0.2768877709750086,
76
+ "warm_optical_total_s": 0.13808018100098707,
77
+ "warm_optical_frame_s": 0.0013808018100098707,
78
+ "warm_detector_total_s": 0.24444962001871318,
79
+ "warm_detector_frame_s": 0.0024444962001871316,
80
+ "warm_optical_frames_per_s": 724.2168953941706,
81
+ "warm_detector_frames_per_s": 409.08224767273015,
82
+ "warm_detector_only_frame_s": 0.0006185778399230912,
83
+ "warm_detector_only_frames_per_s": 1616.6114197112713,
84
+ "python_peak_memory_mib": null
85
+ },
86
+ {
87
+ "config": "benchmarks/configs/pyramid_60_mod8_float32.toml",
88
+ "sensor": "pyramid",
89
+ "device": "cpu",
90
+ "shape": [
91
+ 80,
92
+ 80
93
+ ],
94
+ "source_states": 1,
95
+ "wavelength_samples": 1,
96
+ "range_samples": 1,
97
+ "modulation_samples": 8,
98
+ "modulation_radius_lambda_over_d": 2.0,
99
+ "dtype": "float32",
100
+ "fft_workers": 1,
101
+ "frames": 100,
102
+ "construction_s": 0.07376596701215021,
103
+ "warm_optical_total_s": 0.5042883460118901,
104
+ "warm_optical_frame_s": 0.005042883460118901,
105
+ "warm_detector_total_s": 0.5606069839850534,
106
+ "warm_detector_frame_s": 0.005606069839850534,
107
+ "warm_optical_frames_per_s": 198.29924841777367,
108
+ "warm_detector_frames_per_s": 178.37808457032378,
109
+ "warm_detector_only_frame_s": 0.00035893032007152216,
110
+ "warm_detector_only_frames_per_s": 2786.056078518904,
111
+ "python_peak_memory_mib": null
112
+ },
113
+ {
114
+ "config": "benchmarks/configs/pyramid_60_mod8_float32.toml",
115
+ "sensor": "pyramid",
116
+ "device": "gpu",
117
+ "shape": [
118
+ 80,
119
+ 80
120
+ ],
121
+ "source_states": 1,
122
+ "wavelength_samples": 1,
123
+ "range_samples": 1,
124
+ "modulation_samples": 8,
125
+ "modulation_radius_lambda_over_d": 2.0,
126
+ "dtype": "float32",
127
+ "fft_workers": 1,
128
+ "frames": 100,
129
+ "construction_s": 0.012662593013374135,
130
+ "warm_optical_total_s": 0.1342540149926208,
131
+ "warm_optical_frame_s": 0.0013425401499262079,
132
+ "warm_detector_total_s": 0.2391750770038925,
133
+ "warm_detector_frame_s": 0.002391750770038925,
134
+ "warm_optical_frames_per_s": 744.8566808634844,
135
+ "warm_detector_frames_per_s": 418.10376420772525,
136
+ "warm_detector_only_frame_s": 0.0006130585700157099,
137
+ "warm_detector_only_frames_per_s": 1631.1655181239448,
138
+ "python_peak_memory_mib": null
139
+ },
140
+ {
141
+ "config": "benchmarks/configs/pyramid_80_mod32_float64.toml",
142
+ "sensor": "pyramid",
143
+ "device": "cpu",
144
+ "shape": [
145
+ 108,
146
+ 108
147
+ ],
148
+ "source_states": 1,
149
+ "wavelength_samples": 1,
150
+ "range_samples": 1,
151
+ "modulation_samples": 32,
152
+ "modulation_radius_lambda_over_d": 3.0,
153
+ "dtype": "float64",
154
+ "fft_workers": 1,
155
+ "frames": 100,
156
+ "construction_s": 0.08048216797760688,
157
+ "warm_optical_total_s": 8.533555491012521,
158
+ "warm_optical_frame_s": 0.0853355549101252,
159
+ "warm_detector_total_s": 8.6366516520211,
160
+ "warm_detector_frame_s": 0.086366516520211,
161
+ "warm_optical_frames_per_s": 11.718444920797582,
162
+ "warm_detector_frames_per_s": 11.578561232882254,
163
+ "warm_detector_only_frame_s": 0.0006370968700502999,
164
+ "warm_detector_only_frames_per_s": 1569.620017001572,
165
+ "python_peak_memory_mib": null
166
+ },
167
+ {
168
+ "config": "benchmarks/configs/pyramid_80_mod32_float64.toml",
169
+ "sensor": "pyramid",
170
+ "device": "gpu",
171
+ "shape": [
172
+ 108,
173
+ 108
174
+ ],
175
+ "source_states": 1,
176
+ "wavelength_samples": 1,
177
+ "range_samples": 1,
178
+ "modulation_samples": 32,
179
+ "modulation_radius_lambda_over_d": 3.0,
180
+ "dtype": "float64",
181
+ "fft_workers": 1,
182
+ "frames": 100,
183
+ "construction_s": 0.025755188980838284,
184
+ "warm_optical_total_s": 0.4056679999921471,
185
+ "warm_optical_frame_s": 0.004056679999921471,
186
+ "warm_detector_total_s": 0.4848081910167821,
187
+ "warm_detector_frame_s": 0.004848081910167821,
188
+ "warm_optical_frames_per_s": 246.50699587331462,
189
+ "warm_detector_frames_per_s": 206.26714204285878,
190
+ "warm_detector_only_frame_s": 0.0005712536702048964,
191
+ "warm_detector_only_frames_per_s": 1750.5357989933289,
192
+ "python_peak_memory_mib": null
193
+ },
194
+ {
195
+ "config": "benchmarks/configs/shack_hartmann_20x20_float32.toml",
196
+ "sensor": "shack_hartmann",
197
+ "device": "cpu",
198
+ "shape": [
199
+ 160,
200
+ 160
201
+ ],
202
+ "source_states": 1,
203
+ "wavelength_samples": 1,
204
+ "range_samples": 1,
205
+ "modulation_samples": 1,
206
+ "modulation_radius_lambda_over_d": 0.0,
207
+ "dtype": "float32",
208
+ "fft_workers": 1,
209
+ "frames": 100,
210
+ "construction_s": 0.08199134599999525,
211
+ "warm_optical_total_s": 2.28855264998856,
212
+ "warm_optical_frame_s": 0.0228855264998856,
213
+ "warm_detector_total_s": 2.4490860599908046,
214
+ "warm_detector_frame_s": 0.024490860599908047,
215
+ "warm_optical_frames_per_s": 43.695739313884644,
216
+ "warm_detector_frames_per_s": 40.831558201909594,
217
+ "warm_detector_only_frame_s": 0.0010764133499469608,
218
+ "warm_detector_only_frames_per_s": 929.0111461821558,
219
+ "python_peak_memory_mib": null
220
+ },
221
+ {
222
+ "config": "benchmarks/configs/shack_hartmann_20x20_float32.toml",
223
+ "sensor": "shack_hartmann",
224
+ "device": "gpu",
225
+ "shape": [
226
+ 160,
227
+ 160
228
+ ],
229
+ "source_states": 1,
230
+ "wavelength_samples": 1,
231
+ "range_samples": 1,
232
+ "modulation_samples": 1,
233
+ "modulation_radius_lambda_over_d": 0.0,
234
+ "dtype": "float32",
235
+ "fft_workers": 1,
236
+ "frames": 100,
237
+ "construction_s": 0.01180534201557748,
238
+ "warm_optical_total_s": 0.08716972899856046,
239
+ "warm_optical_frame_s": 0.0008716972899856046,
240
+ "warm_detector_total_s": 0.19479694298934191,
241
+ "warm_detector_frame_s": 0.0019479694298934192,
242
+ "warm_optical_frames_per_s": 1147.187230576929,
243
+ "warm_detector_frames_per_s": 513.3550787060933,
244
+ "warm_detector_only_frame_s": 0.0006126053698244505,
245
+ "warm_detector_only_frames_per_s": 1632.372240365053,
246
+ "python_peak_memory_mib": null
247
+ },
248
+ {
249
+ "config": "benchmarks/configs/shack_hartmann_60x60_float64.toml",
250
+ "sensor": "shack_hartmann",
251
+ "device": "cpu",
252
+ "shape": [
253
+ 360,
254
+ 360
255
+ ],
256
+ "source_states": 1,
257
+ "wavelength_samples": 1,
258
+ "range_samples": 1,
259
+ "modulation_samples": 1,
260
+ "modulation_radius_lambda_over_d": 0.0,
261
+ "dtype": "float64",
262
+ "fft_workers": 1,
263
+ "frames": 100,
264
+ "construction_s": 0.08440429499023594,
265
+ "warm_optical_total_s": 8.684691948990803,
266
+ "warm_optical_frame_s": 0.08684691948990803,
267
+ "warm_detector_total_s": 9.488772381999297,
268
+ "warm_detector_frame_s": 0.09488772381999297,
269
+ "warm_optical_frames_per_s": 11.514513190260066,
270
+ "warm_detector_frames_per_s": 10.538771083781636,
271
+ "warm_detector_only_frame_s": 0.005910381099965889,
272
+ "warm_detector_only_frames_per_s": 169.19382745146018,
273
+ "python_peak_memory_mib": null
274
+ },
275
+ {
276
+ "config": "benchmarks/configs/shack_hartmann_60x60_float64.toml",
277
+ "sensor": "shack_hartmann",
278
+ "device": "gpu",
279
+ "shape": [
280
+ 360,
281
+ 360
282
+ ],
283
+ "source_states": 1,
284
+ "wavelength_samples": 1,
285
+ "range_samples": 1,
286
+ "modulation_samples": 1,
287
+ "modulation_radius_lambda_over_d": 0.0,
288
+ "dtype": "float64",
289
+ "fft_workers": 1,
290
+ "frames": 100,
291
+ "construction_s": 0.026143675000639632,
292
+ "warm_optical_total_s": 0.4227883660059888,
293
+ "warm_optical_frame_s": 0.004227883660059888,
294
+ "warm_detector_total_s": 0.4985939179896377,
295
+ "warm_detector_frame_s": 0.004985939179896377,
296
+ "warm_optical_frames_per_s": 236.52495678791573,
297
+ "warm_detector_frames_per_s": 200.5640189178527,
298
+ "warm_detector_only_frame_s": 0.0005747833198984153,
299
+ "warm_detector_only_frames_per_s": 1739.7860469867767,
300
+ "python_peak_memory_mib": null
301
+ },
302
+ {
303
+ "config": "benchmarks/configs/shack_hartmann_quadrature_9sample.toml",
304
+ "sensor": "shack_hartmann",
305
+ "device": "cpu",
306
+ "shape": [
307
+ 64,
308
+ 64
309
+ ],
310
+ "source_states": 9,
311
+ "wavelength_samples": 3,
312
+ "range_samples": 3,
313
+ "modulation_samples": 1,
314
+ "modulation_radius_lambda_over_d": 0.0,
315
+ "dtype": "float32",
316
+ "fft_workers": 1,
317
+ "frames": 100,
318
+ "construction_s": 0.06993808099650778,
319
+ "warm_optical_total_s": 1.8203168570180424,
320
+ "warm_optical_frame_s": 0.018203168570180422,
321
+ "warm_detector_total_s": 1.8593171660031658,
322
+ "warm_detector_frame_s": 0.018593171660031656,
323
+ "warm_optical_frames_per_s": 54.93549082647914,
324
+ "warm_detector_frames_per_s": 53.78318547715153,
325
+ "warm_detector_only_frame_s": 0.00024450053984764964,
326
+ "warm_detector_only_frames_per_s": 4089.9705195870265,
327
+ "python_peak_memory_mib": null
328
+ },
329
+ {
330
+ "config": "benchmarks/configs/shack_hartmann_quadrature_9sample.toml",
331
+ "sensor": "shack_hartmann",
332
+ "device": "gpu",
333
+ "shape": [
334
+ 64,
335
+ 64
336
+ ],
337
+ "source_states": 9,
338
+ "wavelength_samples": 3,
339
+ "range_samples": 3,
340
+ "modulation_samples": 1,
341
+ "modulation_radius_lambda_over_d": 0.0,
342
+ "dtype": "float32",
343
+ "fft_workers": 1,
344
+ "frames": 100,
345
+ "construction_s": 0.014552935026586056,
346
+ "warm_optical_total_s": 0.4028777669882402,
347
+ "warm_optical_frame_s": 0.004028777669882402,
348
+ "warm_detector_total_s": 0.5134580160083715,
349
+ "warm_detector_frame_s": 0.005134580160083715,
350
+ "warm_optical_frames_per_s": 248.2142431128967,
351
+ "warm_detector_frames_per_s": 194.75789038683465,
352
+ "warm_detector_only_frame_s": 0.0006140789898927323,
353
+ "warm_detector_only_frames_per_s": 1628.4549975804912,
354
+ "python_peak_memory_mib": null
355
+ }
356
+ ]
357
+ }