solvephase 0.1.0__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. {solvephase-0.1.0 → solvephase-0.2.0}/AGENTS.md +30 -1
  2. {solvephase-0.1.0 → solvephase-0.2.0}/CHANGELOG.md +62 -0
  3. {solvephase-0.1.0 → solvephase-0.2.0}/PKG-INFO +1 -1
  4. solvephase-0.2.0/benchmarks/artifacts/v0.1.0-compare-neoverse-n1-rtx4060.json +97 -0
  5. solvephase-0.2.0/benchmarks/artifacts/v0.1.0-methods-neoverse-n1-rtx4060.json +288 -0
  6. solvephase-0.2.0/benchmarks/artifacts/v0.1.0-neoverse-n1-rtx4060.json +803 -0
  7. solvephase-0.2.0/benchmarks/artifacts/v0.1.0-neoverse-n1-rtxa400.json +412 -0
  8. solvephase-0.2.0/docs/benchmarks.md +370 -0
  9. {solvephase-0.1.0 → solvephase-0.2.0}/docs/guide/performance.md +11 -3
  10. solvephase-0.2.0/docs/validation.md +128 -0
  11. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/__about__.py +1 -1
  12. solvephase-0.2.0/src/solvephase/_kernels.py +281 -0
  13. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/algorithms/fast_furious.py +207 -13
  14. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/algorithms/gerchberg_saxton.py +31 -12
  15. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/algorithms/phase_diversity.py +32 -8
  16. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/algorithms/tie.py +8 -7
  17. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/algorithms/wirtinger.py +7 -6
  18. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/focal.py +71 -14
  19. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/losses.py +88 -1
  20. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/optimize.py +22 -9
  21. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/retrieval.py +62 -23
  22. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_cdi.py +25 -12
  23. solvephase-0.2.0/tests/test_kernels.py +221 -0
  24. solvephase-0.2.0/validation/artifacts/nirc2.json +233 -0
  25. solvephase-0.2.0/validation/artifacts/nirc2_fit.png +0 -0
  26. solvephase-0.2.0/validation/artifacts/nirc2_injections.png +0 -0
  27. solvephase-0.2.0/validation/nirc2.py +548 -0
  28. solvephase-0.1.0/docs/benchmarks.md +0 -165
  29. solvephase-0.1.0/docs/validation.md +0 -54
  30. {solvephase-0.1.0 → solvephase-0.2.0}/.gitignore +0 -0
  31. {solvephase-0.1.0 → solvephase-0.2.0}/CONTRIBUTING.md +0 -0
  32. {solvephase-0.1.0 → solvephase-0.2.0}/LICENSE +0 -0
  33. {solvephase-0.1.0 → solvephase-0.2.0}/README.md +0 -0
  34. {solvephase-0.1.0 → solvephase-0.2.0}/ROADMAP.md +0 -0
  35. {solvephase-0.1.0 → solvephase-0.2.0}/benchmarks/artifacts/v0.1.0-compare-i7-10700-p620.json +0 -0
  36. {solvephase-0.1.0 → solvephase-0.2.0}/benchmarks/artifacts/v0.1.0-i7-10700-p620.json +0 -0
  37. {solvephase-0.1.0 → solvephase-0.2.0}/benchmarks/artifacts/v0.1.0-methods-i7-10700-p620.json +0 -0
  38. {solvephase-0.1.0 → solvephase-0.2.0}/benchmarks/bench_algorithms.py +0 -0
  39. {solvephase-0.1.0 → solvephase-0.2.0}/benchmarks/compare.py +0 -0
  40. {solvephase-0.1.0 → solvephase-0.2.0}/benchmarks/methods.py +0 -0
  41. {solvephase-0.1.0 → solvephase-0.2.0}/benchmarks/render_table.py +0 -0
  42. {solvephase-0.1.0 → solvephase-0.2.0}/benchmarks/run.py +0 -0
  43. {solvephase-0.1.0 → solvephase-0.2.0}/docs/api.md +0 -0
  44. {solvephase-0.1.0 → solvephase-0.2.0}/docs/changelog.md +0 -0
  45. {solvephase-0.1.0 → solvephase-0.2.0}/docs/choosing.md +0 -0
  46. {solvephase-0.1.0 → solvephase-0.2.0}/docs/concepts.md +0 -0
  47. {solvephase-0.1.0 → solvephase-0.2.0}/docs/contributing.md +0 -0
  48. {solvephase-0.1.0 → solvephase-0.2.0}/docs/conventions.md +0 -0
  49. {solvephase-0.1.0 → solvephase-0.2.0}/docs/examples.md +0 -0
  50. {solvephase-0.1.0 → solvephase-0.2.0}/docs/guide/ao-wavefront-sensing.md +0 -0
  51. {solvephase-0.1.0 → solvephase-0.2.0}/docs/guide/cdi.md +0 -0
  52. {solvephase-0.1.0 → solvephase-0.2.0}/docs/guide/focal-plane.md +0 -0
  53. {solvephase-0.1.0 → solvephase-0.2.0}/docs/guide/generic.md +0 -0
  54. {solvephase-0.1.0 → solvephase-0.2.0}/docs/guide/interop.md +0 -0
  55. {solvephase-0.1.0 → solvephase-0.2.0}/docs/guide/phase-diversity.md +0 -0
  56. {solvephase-0.1.0 → solvephase-0.2.0}/docs/guide/pupils-and-bases.md +0 -0
  57. {solvephase-0.1.0 → solvephase-0.2.0}/docs/guide/tie.md +0 -0
  58. {solvephase-0.1.0 → solvephase-0.2.0}/docs/index.md +0 -0
  59. {solvephase-0.1.0 → solvephase-0.2.0}/docs/installation.md +0 -0
  60. {solvephase-0.1.0 → solvephase-0.2.0}/docs/javascripts/mathjax.js +0 -0
  61. {solvephase-0.1.0 → solvephase-0.2.0}/docs/quickstart.md +0 -0
  62. {solvephase-0.1.0 → solvephase-0.2.0}/docs/roadmap.md +0 -0
  63. {solvephase-0.1.0 → solvephase-0.2.0}/examples/01_ncpa_calibration.py +0 -0
  64. {solvephase-0.1.0 → solvephase-0.2.0}/examples/02_broadband_undersampled.py +0 -0
  65. {solvephase-0.1.0 → solvephase-0.2.0}/examples/03_segment_phasing.py +0 -0
  66. {solvephase-0.1.0 → solvephase-0.2.0}/examples/04_extended_scene_diversity.py +0 -0
  67. {solvephase-0.1.0 → solvephase-0.2.0}/examples/05_focal_plane_wfs.py +0 -0
  68. {solvephase-0.1.0 → solvephase-0.2.0}/examples/06_cdi_multistart.py +0 -0
  69. {solvephase-0.1.0 → solvephase-0.2.0}/examples/07_generic_coded_diffraction.py +0 -0
  70. {solvephase-0.1.0 → solvephase-0.2.0}/examples/08_tie_microscopy.py +0 -0
  71. {solvephase-0.1.0 → solvephase-0.2.0}/examples/_common.py +0 -0
  72. {solvephase-0.1.0 → solvephase-0.2.0}/examples/showcase.py +0 -0
  73. {solvephase-0.1.0 → solvephase-0.2.0}/examples/solvephase_showcase.webp +0 -0
  74. {solvephase-0.1.0 → solvephase-0.2.0}/mkdocs.yml +0 -0
  75. {solvephase-0.1.0 → solvephase-0.2.0}/pyproject.toml +0 -0
  76. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/__init__.py +0 -0
  77. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/algorithms/__init__.py +0 -0
  78. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/algorithms/cdi.py +0 -0
  79. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/algorithms/lift.py +0 -0
  80. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/api.py +0 -0
  81. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/backend.py +0 -0
  82. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/basis.py +0 -0
  83. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/cli.py +0 -0
  84. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/interop.py +0 -0
  85. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/metrics.py +0 -0
  86. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/operators.py +0 -0
  87. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/propagation.py +0 -0
  88. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/pupil.py +0 -0
  89. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/py.typed +0 -0
  90. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/result.py +0 -0
  91. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/simulate.py +0 -0
  92. {solvephase-0.1.0 → solvephase-0.2.0}/src/solvephase/unwrap.py +0 -0
  93. {solvephase-0.1.0 → solvephase-0.2.0}/tests/conftest.py +0 -0
  94. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_api.py +0 -0
  95. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_cli.py +0 -0
  96. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_conformance.py +0 -0
  97. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_docs.py +0 -0
  98. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_fast_furious.py +0 -0
  99. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_focal.py +0 -0
  100. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_gs_unwrap_metrics.py +0 -0
  101. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_interop.py +0 -0
  102. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_lift.py +0 -0
  103. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_losses_optimize.py +0 -0
  104. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_operators.py +0 -0
  105. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_phase_diversity.py +0 -0
  106. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_pupil_basis.py +0 -0
  107. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_retrieval.py +0 -0
  108. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_tie.py +0 -0
  109. {solvephase-0.1.0 → solvephase-0.2.0}/tests/test_wirtinger.py +0 -0
  110. {solvephase-0.1.0 → solvephase-0.2.0}/validation/artifacts/validation.json +0 -0
  111. {solvephase-0.1.0 → solvephase-0.2.0}/validation/artifacts/validation.png +0 -0
  112. {solvephase-0.1.0 → solvephase-0.2.0}/validation/extensions.py +0 -0
  113. {solvephase-0.1.0 → solvephase-0.2.0}/validation/validate.py +0 -0
@@ -50,12 +50,14 @@ src/solvephase/
50
50
  unwrap.py re-exports aocore.unwrap (unwrap_phase, wrap)
51
51
  metrics.py re-exports aocore.metrics (rms, wavefront_error, strehl_from_rms)
52
52
  result.py Result returned by every focal-plane solver
53
+ _kernels.py fused CuPy kernels for the GPU hot paths (same arithmetic as the xp code)
53
54
  algorithms/ one module per algorithm family (gerchberg_saxton, cdi, ...)
54
55
  api.py retrieve(): the one-call high-level entry point
55
56
  tests/ pytest; mirrors the module names
56
57
  benchmarks/ speed suite (run.py), head-to-head baselines (compare.py), method
57
58
  comparison feeding docs/choosing.md (methods.py), JSON artifacts
58
- validation/ physics/statistics evidence (Cramer-Rao, independent references)
59
+ validation/ physics/statistics evidence (Cramer-Rao, independent references);
60
+ nirc2.py validates on real Keck/NIRC2 data (data not in the repo)
59
61
  docs/ mkdocs-material site
60
62
  examples/ headless, deterministic scripts
61
63
  ```
@@ -214,3 +216,30 @@ note the GPU and CuPy version when you report GPU results.
214
216
  is not a package).
215
217
  - GPU timings on a shared device (or CPU timings under load) vary 2-5x;
216
218
  record the load and hardware with any performance claim.
219
+ - Small GPU problems are bound by the host: each CuPy operation costs
220
+ 15-30 us of Python and launch overhead on the Arm bench, whatever the
221
+ array size. Fuse elementwise chains in `_kernels.py`, keep the exact
222
+ expression order, and compile with `--fmad=false` so results match the
223
+ unfused CuPy chain to the bit. Products of two complex arrays are the
224
+ exception: write them as thrust complex products compiled with default
225
+ flags (as CuPy's own multiply kernel is). Inlining a conjugate into a
226
+ product (`a * conj(b)`) or summing with a custom `ReductionKernel` changes
227
+ the rounding; keep `cupy.sum` for sums. `cupy.vdot` of real vectors
228
+ rounds exactly like `(a * b).sum()`, which has less overhead.
229
+ - CuPy refuses cuBLAS calls (any `matmul`) during CUDA-graph stream capture,
230
+ so only matmul-free bodies (the Fast & Furious step) are replayed from
231
+ graphs; capture under a private memory pool and warm FFT plans first.
232
+ - CPU micro-optimizations of elementwise code on MB-sized temporaries are
233
+ dominated by allocation and page-fault patterns; a rewrite 5x faster in
234
+ isolation made `jvp` slower in context. Measure inside the solver.
235
+ - CI has no GPU and gates coverage at 85%: mark GPU-only blocks
236
+ `# pragma: no cover - GPU only` and test them with `--run-gpu`
237
+ (`tests/test_kernels.py` compares them with the plain expressions).
238
+ - `AOCORE_FFT_WORKERS` fixes the FFT thread count for every transform,
239
+ including tiny ones the default heuristic would run on one thread.
240
+ - Real data (validation/nirc2.py): the Keck daytime bench pupil is a full
241
+ circle; a segmented Keck pupil predicts six-fold spikes the images don't
242
+ have. Crop each defocused frame around its own centroid (the image walks
243
+ with focus) and leave tip, tilt and focus out of run-to-run comparisons.
244
+ Beyond ~36 Zernikes the likelihood on these data has flat directions:
245
+ cold- and warm-started fits differ by ~300 nm RMS at nearly equal chi2.
@@ -5,6 +5,68 @@ All notable changes to `solvephase` are documented here. The project follows
5
5
 
6
6
  ## [Unreleased]
7
7
 
8
+ ## [0.2.0] - 2026-10-07
9
+
10
+ ### Performance
11
+
12
+ Small GPU problems were bound by host-side launch overhead (15-30 us per CuPy
13
+ operation on an Arm host), not by the GPU. Results are unchanged: bitwise
14
+ identical to 0.1.0 on CPU and GPU on the reference machine (Ampere
15
+ Neoverse-N1, 12 pinned cores, NumPy 2.5.3/SciPy 1.18.1; RTX 4060 with CuPy
16
+ 14.2.0). Speed-ups below are medians of 0.1.0 and 0.2.0 run alternately,
17
+ three times each (the Benchmarks page, `docs/benchmarks.md`, has the tables).
18
+
19
+ - **Fast & Furious on the GPU replays each step from a CUDA graph**: one copy
20
+ in, one graph launch and one copy out per frame, running the same kernels.
21
+ A step takes 0.11 ms at 64² (was 1.41 ms, 12x) and 0.27 ms at 256² (5.8x),
22
+ so the GPU now beats the CPU on every size.
23
+ - **Fused GPU kernels** (`solvephase._kernels`) for the focal-plane model,
24
+ its reverse- and forward-mode derivatives, the Poisson, Gaussian and
25
+ amplitude losses, the Gerchberg-Saxton projection, the Fast & Furious step
26
+ and the phase-diversity metric. On the RTX 4060: objective + gradient
27
+ 1.5-1.8x, Levenberg-Marquardt 1.4-1.7x, LIFT 1.8x, phase diversity
28
+ 1.4-1.5x, `retrieve()` 1.2x, Gerchberg-Saxton 1.1-1.2x.
29
+ - **L-BFGS** reuses the scalar products it already has, fuses its vector
30
+ updates and evaluates real scalar products on the GPU with less overhead:
31
+ coded-diffraction L-BFGS 1.2-1.3x on the GPU.
32
+ - **`FocalPlaneProblem`** keeps device copies of its flux, background and
33
+ scaling vectors (no host-to-device copy per evaluation), caches the
34
+ Gauss-Newton direction maps and assembles gradients and Jacobian blocks
35
+ without per-channel loops.
36
+ - **CPU**: unit phasors are built from `cos`/`sin` instead of a complex
37
+ `exp` (same values, about 1.4x faster), Gerchberg-Saxton computes its
38
+ diversity phasors once instead of every iteration, and the Poisson and
39
+ amplitude losses skip their Taylor terms when no pixel is below the floor.
40
+ Gerchberg-Saxton 1.2x, focal-plane objective and LM up to 1.1x on 12
41
+ Neoverse-N1 cores.
42
+
43
+ ### Fixed
44
+
45
+ - **The CDI CPU/GPU parity test failed on aarch64.** After 50 iterations it
46
+ compared the objects to `atol=1e-9`, but RAAR and DM are chaotic and amplify
47
+ the FFT libraries' last-bit differences (about 1e5x between iterations 10
48
+ and 40 on an Arm Neoverse-N1 with an RTX 4060). The test now compares the
49
+ objects after a short run (`atol=1e-10`) and the error histories after the
50
+ long one (`rtol=1e-6`). The implementations already agreed.
51
+
52
+ ### Added
53
+
54
+ - **Validation on real Keck/NIRC2 data** (`validation/nirc2.py`). Daytime
55
+ focus-diversity calibrations of the Keck AO bench, with known coma and
56
+ trefoil patterns injected on the Xinetics DM, are retrieved with
57
+ `FocalPlaneProblem`.
58
+ - The injected patterns are recovered with correlations of 0.945–0.965.
59
+ One DM gain fits all six runs (3 % scatter), and strength ratios come
60
+ out within 1 % of the commanded values.
61
+ - The IDL sharpening correction is predicted with no free parameter
62
+ (correlation 0.90, amplitude within 1 %).
63
+ - The fitted DM gain (652 nm OPD/V) and pupil size (9.75 actuator pitches)
64
+ agree with the Keck AO software's Xinetics parameters (600 nm/V; a
65
+ 10.0-pitch control aperture).
66
+ - The data are not distributed; the script reads them from `--data` or
67
+ `SOLVEPHASE_NIRC2_DATA`. The report and figures are in
68
+ `validation/artifacts`.
69
+
8
70
  ## [0.1.0] - 2026-10-07
9
71
 
10
72
  First release.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: solvephase
3
- Version: 0.1.0
3
+ Version: 0.2.0
4
4
  Summary: Fast, GPU-optional phase retrieval for adaptive optics, optical metrology and coherent imaging.
5
5
  Project-URL: Homepage, https://github.com/jacotay7/solvephase
6
6
  Project-URL: Documentation, https://jacotay7.github.io/solvephase/
@@ -0,0 +1,97 @@
1
+ {
2
+ "environment": {
3
+ "solvephase": "0.1.0",
4
+ "python": "3.13.15",
5
+ "platform": "Linux-6.17.9-76061709-generic-aarch64-with-glibc2.39",
6
+ "processor": "aarch64",
7
+ "cpu_count": 80,
8
+ "numpy": "2.5.3",
9
+ "timestamp": "2026-10-07T06:59:07+00:00",
10
+ "scipy": "1.18.1",
11
+ "gpu": "NVIDIA GeForce RTX 4060",
12
+ "cupy": "14.2.0",
13
+ "cuda_runtime": 12090,
14
+ "host": "cfl-test-bench: 80-core Ampere Neoverse-N1 (aarch64), shared",
15
+ "cpu_pinning": "taskset -c 16-31 (16 cores); OMP/OpenBLAS/FFT threads = 16",
16
+ "gpus": "NVIDIA GeForce RTX 4060 (8 GB), NVIDIA RTX A400 (4 GB); driver 580.173.02"
17
+ },
18
+ "rows": [
19
+ {
20
+ "seconds": 0.6354510160163045,
21
+ "error_nm": 0.12939041062796763,
22
+ "iterations": 6,
23
+ "problem": "focal 128x128, 36 modes, 2 images",
24
+ "method": "solvephase LM (cpu)"
25
+ },
26
+ {
27
+ "seconds": 0.09766029001912102,
28
+ "error_nm": 0.12939044804752545,
29
+ "iterations": 9,
30
+ "problem": "focal 128x128, 36 modes, 2 images",
31
+ "method": "solvephase LM (gpu)"
32
+ },
33
+ {
34
+ "seconds": 2.9365519550046884,
35
+ "error_nm": 0.1911658513284471,
36
+ "evaluations": 8,
37
+ "problem": "focal 128x128, 36 modes, 2 images",
38
+ "method": "HCIPy + SciPy least_squares"
39
+ },
40
+ {
41
+ "seconds": 3.0905606810119934,
42
+ "error_nm": 0.12076580233938293,
43
+ "iterations": 12,
44
+ "evaluations": 518,
45
+ "problem": "focal 128x128, 36 modes, 2 images",
46
+ "method": "HCIPy + SciPy L-BFGS-B"
47
+ },
48
+ {
49
+ "seconds": 1.0338798309967387,
50
+ "iterations_per_s": 96.7230397594587,
51
+ "problem": "Misell/GS 128x128, 2 images (iterations/s)",
52
+ "method": "NumPy textbook loop"
53
+ },
54
+ {
55
+ "seconds": 0.5860432240006048,
56
+ "iterations_per_s": 170.63587787493435,
57
+ "problem": "Misell/GS 128x128, 2 images (iterations/s)",
58
+ "method": "solvephase (cpu)"
59
+ },
60
+ {
61
+ "seconds": 0.08268631101236679,
62
+ "iterations_per_s": 1209.3900281153399,
63
+ "problem": "Misell/GS 128x128, 2 images (iterations/s)",
64
+ "method": "solvephase (gpu)"
65
+ },
66
+ {
67
+ "seconds": 0.6534776719927322,
68
+ "iterations_per_s": 153.02741667524361,
69
+ "problem": "CDI HIO 256x256 (iterations/s)",
70
+ "method": "NumPy textbook loop"
71
+ },
72
+ {
73
+ "seconds": 0.24491209498955868,
74
+ "iterations_per_s": 408.3097652007072,
75
+ "problem": "CDI HIO 256x256 (iterations/s)",
76
+ "method": "solvephase (cpu)"
77
+ },
78
+ {
79
+ "seconds": 1.5308567469764967,
80
+ "iterations_per_s": 522.5831885184764,
81
+ "problem": "CDI HIO 256x256 (iterations/s)",
82
+ "method": "solvephase (cpu, 16 starts, per start)"
83
+ },
84
+ {
85
+ "seconds": 0.03796361497370526,
86
+ "iterations_per_s": 2634.1011009953345,
87
+ "problem": "CDI HIO 256x256 (iterations/s)",
88
+ "method": "solvephase (gpu)"
89
+ },
90
+ {
91
+ "seconds": 0.029988599999342114,
92
+ "iterations_per_s": 26676.803852715708,
93
+ "problem": "CDI HIO 256x256 (iterations/s)",
94
+ "method": "solvephase (gpu, 16 starts, per start)"
95
+ }
96
+ ]
97
+ }
@@ -0,0 +1,288 @@
1
+ {
2
+ "environment": {
3
+ "solvephase": "0.1.0",
4
+ "python": "3.13.15",
5
+ "platform": "Linux-6.17.9-76061709-generic-aarch64-with-glibc2.39",
6
+ "processor": "aarch64",
7
+ "cpu_count": 80,
8
+ "numpy": "2.5.3",
9
+ "timestamp": "2026-10-07T07:08:36+00:00",
10
+ "scipy": "1.18.1",
11
+ "gpu": "NVIDIA GeForce RTX 4060",
12
+ "cupy": "14.2.0",
13
+ "cuda_runtime": 12090,
14
+ "host": "cfl-test-bench: 80-core Ampere Neoverse-N1 (aarch64), shared",
15
+ "cpu_pinning": "taskset -c 16-31 (16 cores); OMP/OpenBLAS/FFT threads = 16",
16
+ "gpus": "NVIDIA GeForce RTX 4060 (8 GB), NVIDIA RTX A400 (4 GB); driver 580.173.02"
17
+ },
18
+ "capture_levels": [
19
+ 0.05,
20
+ 0.1,
21
+ 0.2,
22
+ 0.3,
23
+ 0.4,
24
+ 0.5,
25
+ 0.7
26
+ ],
27
+ "methods": [
28
+ {
29
+ "key": "misell",
30
+ "label": "Gerchberg-Saxton / Misell",
31
+ "family": "focal-plane",
32
+ "data": "2+ images with known diversity",
33
+ "requires": "known pupil",
34
+ "unknowns": "zonal phase, unwrapped and fitted to a basis",
35
+ "seconds": 0.6054978969914373,
36
+ "seconds_gpu": 0.32186949701281264,
37
+ "error": 0.010525870689563108,
38
+ "error_abs": 1.6839799555150254,
39
+ "error_unit": "nm",
40
+ "iterations": 300,
41
+ "capture": {
42
+ "0.05": 1.0,
43
+ "0.1": 1.0,
44
+ "0.2": 1.0,
45
+ "0.3": 0.16666666666666666,
46
+ "0.4": 0.3333333333333333,
47
+ "0.5": 0.16666666666666666,
48
+ "0.7": 0.0
49
+ },
50
+ "trace": {
51
+ "times": [],
52
+ "errors": []
53
+ },
54
+ "extra": {}
55
+ },
56
+ {
57
+ "key": "lm",
58
+ "label": "Nonlinear ML, modal (LM)",
59
+ "family": "focal-plane",
60
+ "data": "1+ images; 2+ with diversity to fix the sign",
61
+ "requires": "known pupil",
62
+ "unknowns": "modal coefficients + flux, background, registration",
63
+ "seconds": 0.18697032099589705,
64
+ "seconds_gpu": 0.09825141800683923,
65
+ "error": 0.005106228168692953,
66
+ "error_abs": 0.8169192020277456,
67
+ "error_unit": "nm",
68
+ "iterations": 6,
69
+ "capture": {
70
+ "0.05": 1.0,
71
+ "0.1": 1.0,
72
+ "0.2": 1.0,
73
+ "0.3": 0.3333333333333333,
74
+ "0.4": 0.16666666666666666,
75
+ "0.5": 0.0,
76
+ "0.7": 0.0
77
+ },
78
+ "trace": {
79
+ "times": [],
80
+ "errors": []
81
+ },
82
+ "extra": {}
83
+ },
84
+ {
85
+ "key": "lbfgs",
86
+ "label": "Nonlinear ML, zonal (L-BFGS)",
87
+ "family": "focal-plane",
88
+ "data": "2+ images with known diversity",
89
+ "requires": "known pupil",
90
+ "unknowns": "phase per pixel (+ amplitude)",
91
+ "seconds": 0.6653045339917298,
92
+ "seconds_gpu": 1.8744644250255078,
93
+ "error": 0.1589518492913621,
94
+ "error_abs": 25.429889459321217,
95
+ "error_unit": "nm",
96
+ "iterations": 300,
97
+ "capture": {
98
+ "0.05": 0.0,
99
+ "0.1": 1.0,
100
+ "0.2": 0.3333333333333333,
101
+ "0.3": 0.0,
102
+ "0.4": 0.0,
103
+ "0.5": 0.0,
104
+ "0.7": 0.0
105
+ },
106
+ "trace": {
107
+ "times": [],
108
+ "errors": []
109
+ },
110
+ "extra": {}
111
+ },
112
+ {
113
+ "key": "retrieve",
114
+ "label": "retrieve() (robust default)",
115
+ "family": "focal-plane",
116
+ "data": "2+ images with known diversity",
117
+ "requires": "known pupil",
118
+ "unknowns": "modal (+ optional zonal refinement)",
119
+ "seconds": 0.8604175919899717,
120
+ "seconds_gpu": 0.454890371998772,
121
+ "error": 0.005106226280309669,
122
+ "error_abs": 0.8169188999150089,
123
+ "error_unit": "nm",
124
+ "iterations": 3,
125
+ "capture": {
126
+ "0.05": 1.0,
127
+ "0.1": 1.0,
128
+ "0.2": 1.0,
129
+ "0.3": 1.0,
130
+ "0.4": 0.8333333333333334,
131
+ "0.5": 0.5,
132
+ "0.7": 0.5
133
+ },
134
+ "trace": {
135
+ "times": [],
136
+ "errors": []
137
+ },
138
+ "extra": {}
139
+ },
140
+ {
141
+ "key": "pd",
142
+ "label": "Phase diversity, extended object",
143
+ "family": "focal-plane",
144
+ "data": "2+ images of the same scene with known diversity",
145
+ "requires": "known pupil; compact scene",
146
+ "unknowns": "modal coefficients + the object",
147
+ "seconds": 0.29611618898343295,
148
+ "seconds_gpu": 0.6530910670117009,
149
+ "error": 0.06562706370545356,
150
+ "error_abs": 10.499336641943728,
151
+ "error_unit": "nm",
152
+ "iterations": 119,
153
+ "capture": {
154
+ "0.05": 1.0,
155
+ "0.1": 0.8333333333333334,
156
+ "0.2": 1.0,
157
+ "0.3": 0.6666666666666666,
158
+ "0.4": 0.3333333333333333,
159
+ "0.5": 0.3333333333333333,
160
+ "0.7": 0.0
161
+ },
162
+ "trace": {
163
+ "times": [],
164
+ "errors": []
165
+ },
166
+ "extra": {}
167
+ },
168
+ {
169
+ "key": "lift",
170
+ "label": "LIFT",
171
+ "family": "focal-plane",
172
+ "data": "1 image with a known astigmatism bias",
173
+ "requires": "known pupil",
174
+ "unknowns": "~10 low-order modes",
175
+ "seconds": 0.0662676340143662,
176
+ "seconds_gpu": 0.10143513602088206,
177
+ "error": 0.2686901758718495,
178
+ "error_abs": 42.986360345529,
179
+ "error_unit": "nm",
180
+ "iterations": 12,
181
+ "capture": {
182
+ "0.05": 1.0,
183
+ "0.1": 1.0,
184
+ "0.2": 1.0,
185
+ "0.3": 0.6666666666666666,
186
+ "0.4": 0.16666666666666666,
187
+ "0.5": 0.0,
188
+ "0.7": 0.0
189
+ },
190
+ "trace": {
191
+ "times": [],
192
+ "errors": []
193
+ },
194
+ "extra": {
195
+ "in_model_error_nm": 6.756301510929042,
196
+ "n_modes": 10
197
+ }
198
+ },
199
+ {
200
+ "key": "ff",
201
+ "label": "Fast & Furious",
202
+ "family": "focal-plane",
203
+ "data": "1 image per step, in closed loop",
204
+ "requires": "a DM; small residuals; symmetric pupil",
205
+ "unknowns": "zonal phase (sequential)",
206
+ "seconds": 0.022502428153529763,
207
+ "seconds_gpu": 0.035019537986954674,
208
+ "error": 0.11888414940395173,
209
+ "error_abs": 19.01966407617134,
210
+ "error_unit": "nm",
211
+ "iterations": 30,
212
+ "capture": {
213
+ "0.05": 1.0,
214
+ "0.1": 1.0,
215
+ "0.2": 0.0,
216
+ "0.3": 0.0,
217
+ "0.4": 0.0,
218
+ "0.5": 0.0,
219
+ "0.7": 0.0
220
+ },
221
+ "trace": {
222
+ "times": [],
223
+ "errors": []
224
+ },
225
+ "extra": {}
226
+ },
227
+ {
228
+ "key": "cdi",
229
+ "label": "CDI: HIO\u2192ER + shrinkwrap, 8 starts",
230
+ "family": "imaging",
231
+ "data": "1 oversampled diffraction pattern",
232
+ "requires": "isolated object (oversampling \u2265 2)",
233
+ "unknowns": "complex object",
234
+ "seconds": 1.581951920001302,
235
+ "seconds_gpu": 0.29780680901603773,
236
+ "error": 0.0009197853412670216,
237
+ "error_abs": 0.0009197853412670216,
238
+ "error_unit": "relative",
239
+ "iterations": 700,
240
+ "capture": {},
241
+ "trace": {
242
+ "times": [],
243
+ "errors": []
244
+ },
245
+ "extra": {}
246
+ },
247
+ {
248
+ "key": "generic",
249
+ "label": "Generic: L-BFGS (coded diffraction, 6 masks)",
250
+ "family": "generic",
251
+ "data": "6 coded diffraction patterns",
252
+ "requires": "known random masks (m/n \u2273 4)",
253
+ "unknowns": "complex vector",
254
+ "seconds": 0.22192049998557195,
255
+ "seconds_gpu": 0.2058028259780258,
256
+ "error": 1.876543333502858e-08,
257
+ "error_abs": 1.876543333502858e-08,
258
+ "error_unit": "relative",
259
+ "iterations": 40,
260
+ "capture": {},
261
+ "trace": {
262
+ "times": [],
263
+ "errors": []
264
+ },
265
+ "extra": {}
266
+ },
267
+ {
268
+ "key": "tie",
269
+ "label": "TIE (non-uniform intensity, DCT)",
270
+ "family": "near-field",
271
+ "data": "3 intensities at \u00b1dz",
272
+ "requires": "weak defocus; sample inside the field",
273
+ "unknowns": "phase map",
274
+ "seconds": 0.004973458999302238,
275
+ "seconds_gpu": 0.004885978007223457,
276
+ "error": 0.044832407811043105,
277
+ "error_abs": 0.044832407811043105,
278
+ "error_unit": "relative",
279
+ "iterations": 1,
280
+ "capture": {},
281
+ "trace": {
282
+ "times": [],
283
+ "errors": []
284
+ },
285
+ "extra": {}
286
+ }
287
+ ]
288
+ }