getframes 2.3.0__tar.gz → 2.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {getframes-2.3.0 → getframes-2.5.0}/CHANGELOG.md +101 -1
- {getframes-2.3.0 → getframes-2.5.0}/PKG-INFO +5 -4
- {getframes-2.3.0 → getframes-2.5.0}/README.md +3 -3
- {getframes-2.3.0 → getframes-2.5.0}/benchmarks/bench_devices.py +9 -0
- getframes-2.5.0/benchmarks/device-results-neoverse-n1-rtx4060.json +221 -0
- getframes-2.5.0/benchmarks/device-results-neoverse-n1-rtxa400.json +126 -0
- getframes-2.5.0/benchmarks/device-results-neoverse-n1.md +39 -0
- {getframes-2.3.0 → getframes-2.5.0}/pyproject.toml +2 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/__about__.py +1 -1
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/__init__.py +2 -1
- getframes-2.5.0/src/getframes/_cuda.py +224 -0
- getframes-2.5.0/src/getframes/backend.py +435 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/camera.py +55 -28
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/cli.py +12 -7
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/dataset.py +3 -2
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/frame.py +8 -7
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/noise.py +222 -91
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/scene/scene.py +19 -5
- getframes-2.5.0/tests/test_backend.py +242 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_cli.py +29 -0
- getframes-2.5.0/tests/test_conformance.py +149 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_gpu.py +140 -1
- getframes-2.3.0/src/getframes/backend.py +0 -150
- getframes-2.3.0/tests/test_conformance.py +0 -83
- {getframes-2.3.0 → getframes-2.5.0}/.gitignore +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/LICENSE +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/benchmarks/__init__.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/benchmarks/bench_detector_workspace.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/benchmarks/bench_fixed_map_dtype.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/benchmarks/detector-workspace-results.json +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/benchmarks/device-results.json +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/benchmarks/device-results.md +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/benchmarks/fixed-map-dtype-results.json +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/benchmarks/render_device_table.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/benchmarks/run.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/01_basic_dark_frame.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/02_custom_camera.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/03_master_dark.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/04_browse_presets.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/05_visualise.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/06_photon_transfer_curve.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/07_star_field_exposure.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/08_ao_limiting_magnitude.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/09_transit_photometry.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/10_detector_realism.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/11_radiometry_and_ir.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/12_ml_dataset.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/13_crowded_field.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/14_keck_lgs_ttf_trade_study.ipynb +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/15_detector_characterization.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/16_detector_showcase.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/README.md +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/_common.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/examples/detector_showcase.webp +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/analysis/__init__.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/analysis/apertures.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/analysis/characterize.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/analysis/nondestructive.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/analysis/ptc.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/calibrate.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/config.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/observation.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/__init__.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/__init__.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/andor_cb1_0_5mp.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/andor_ikon_m934.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/andor_ixon_ultra_888.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/andor_marana_4_2b_11.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/andor_ocam2k.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/first_light_imaging_cred_one.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/generic_ccd.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/generic_cmos.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/generic_eapd.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/generic_emccd.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/generic_scmos.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/hamamatsu_orca_fusion.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/hamamatsu_orca_quest_2.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/leonardo_saphira.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/nuvu_hnu_128_omega.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/nuvu_hnu_240.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/photometrics_prime_95b.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/princeton_instruments_kuro_1200b.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/qhy530_pro_ii.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/scimeasure_little_joe_ccd39.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/tucsen_aries_6504_pro.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/presets/data/zwo_asi2600mm.toml +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/py.typed +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/scene/__init__.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/scene/optics.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/scene/photometry.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/scene/psf.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/scene/sources.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/scene/thermal.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/scene/wcs.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/src/getframes/spectral.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_analysis.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_benchmarks.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_calibrate.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_camera.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_characterize.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_config.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_dataset.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_detector.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_frame.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_gain.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_noise.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_nondestructive_analysis.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_observation.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_presets.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_radiometry.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_realism.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_scale.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_scene.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_scene_enrich.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_signal.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_spectral.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_validation.py +0 -0
- {getframes-2.3.0 → getframes-2.5.0}/tests/test_workspace.py +0 -0
|
@@ -6,6 +6,104 @@ to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [2.5.0] - 2026-10-07
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- **Arm benchmark data point** (`benchmarks/device-results-neoverse-n1.*`): the
|
|
14
|
+
device table on an Ampere Neoverse-N1 host (16 pinned cores) with an RTX
|
|
15
|
+
4060 and an RTX A400.
|
|
16
|
+
|
|
17
|
+
### Performance
|
|
18
|
+
|
|
19
|
+
- **Small GPU frames run up to 2.6x faster, with bit-identical seeded output.**
|
|
20
|
+
WFS-sized frames were launch-bound: the host spent longer issuing about twenty
|
|
21
|
+
small CuPy kernels per frame than the GPU spent running them. The GPU path now
|
|
22
|
+
computes the photo and total expectations in one fused kernel and the whole
|
|
23
|
+
readout (full-well clip, defects, reset/avalanche/read noise, gain, bias
|
|
24
|
+
pedestal and structure, common mode, rounding, ADC saturation, `uint32`
|
|
25
|
+
conversion) in another; keeps scalar inputs (`background`, `extra_electrons`,
|
|
26
|
+
the EM-gain scale, the CIC rate) on the host instead of uploading them every
|
|
27
|
+
frame; samples Poisson counts directly in the working precision; and reuses
|
|
28
|
+
the chain's `photo + dark + extra` sum as `FrameTruth.mean_electrons`. The
|
|
29
|
+
noise is drawn by the same calls in the same order and the fused kernels
|
|
30
|
+
repeat the same floating-point operations (FMA contraction off), so every
|
|
31
|
+
seeded GPU frame and truth array is unchanged; a new GPU test compares them
|
|
32
|
+
bit for bit against the separate operations. `bench_devices.py` on an Ampere
|
|
33
|
+
Neoverse-N1 host (12 pinned cores), frames/s, 2.4.0 → now, median of three
|
|
34
|
+
interleaved runs: RTX 4060 — Pyramid 80x80 2,422 → 6,418 (2.65x),
|
|
35
|
+
Shack-Hartmann 160x160 2,424 → 6,340 (2.62x), OCAM2K 240x240 1,650 → 2,733
|
|
36
|
+
(1.66x), SAPHIRA 256x320 1,809 → 2,072 (1.15x), 1024x1024 240 → 247 (1.03x);
|
|
37
|
+
RTX A400 — Pyramid 2,403 → 6,315 (2.63x), the larger frames 1.02–1.06x.
|
|
38
|
+
Larger frames are bound by CuPy's double-precision Poisson and Gamma
|
|
39
|
+
samplers. See the [GPU guide](docs/guides/gpu.md#small-frames-fused-kernels).
|
|
40
|
+
- The NumPy path skips whole-frame identity arithmetic (adding a zero offset,
|
|
41
|
+
dividing by a unit gain, multiplying by a unit avalanche-gain map) and reuses
|
|
42
|
+
the expectation sum as the truth: 1–5% faster (1024x1024: 13.4 → 14.1 frames/s
|
|
43
|
+
on the same host), with identical seeded output. It remains bound by NumPy's
|
|
44
|
+
Poisson sampler (about 80% of a 1024x1024 frame), whose single sequential
|
|
45
|
+
stream cannot be threaded without changing seeded frames.
|
|
46
|
+
|
|
47
|
+
### Fixed
|
|
48
|
+
|
|
49
|
+
- `benchmarks/bench_devices.py` reported the CPU of Arm hosts as `aarch64`;
|
|
50
|
+
it now reads the model from `lscpu` (e.g. `Neoverse-N1`) when
|
|
51
|
+
`/proc/cpuinfo` has no model name.
|
|
52
|
+
|
|
53
|
+
## [2.4.0] - 2026-10-07
|
|
54
|
+
|
|
55
|
+
### Added
|
|
56
|
+
|
|
57
|
+
- **`device="gpu:N"` and `device="auto"`.** Every `device` argument (`Camera`,
|
|
58
|
+
`get_backend`, and the CLI's `[camera]` table) now speaks the AO stack's
|
|
59
|
+
vocabulary (aocore CONVENTIONS 8.1): `"cpu"`, `"gpu"`, `"gpu:N"` for CUDA
|
|
60
|
+
device `N`, and `"auto"`, which picks the GPU when CuPy is installed and sees
|
|
61
|
+
a device and the CPU otherwise. A `"gpu:N"` beyond the devices CuPy sees
|
|
62
|
+
raises a `ValueError` naming the count. The old spellings (`"numpy"`,
|
|
63
|
+
`"cuda"`, `"cupy"`) still work, now case-insensitively. A GPU camera is
|
|
64
|
+
pinned to its card: the fixed-pattern maps and the cuRAND streams are created
|
|
65
|
+
on it and every camera method runs with it current, so a `"gpu:1"` camera
|
|
66
|
+
works whichever device is current at the call, and `with_config` keeps it.
|
|
67
|
+
New: `Camera.device_id`, `ArrayBackend.device_id`, `ArrayBackend.spec`
|
|
68
|
+
(`"cpu"` or `"gpu:N"`) and `ArrayBackend.activate()` (the device context,
|
|
69
|
+
for calling the low-level `noise` functions on another card).
|
|
70
|
+
- **`precision="single"` / `"double"`.** The working precision takes the
|
|
71
|
+
shared names (aocore CONVENTIONS 8.2), with `"float32"`/`"float64"` kept as
|
|
72
|
+
aliases: on `Camera`, as a new `precision` keyword on
|
|
73
|
+
`Scene.photon_rate_map`/`photoelectron_rate_map` (beside `dtype`), and on the
|
|
74
|
+
`noise` functions that take a `float_dtype` (`simulate_frame`,
|
|
75
|
+
`fixed_pattern_maps`, `dark_signal_map`, `photo_signal_map`). A `dtype` and a
|
|
76
|
+
`precision` that disagree raise `ValueError`. `getframes.resolve_precision`
|
|
77
|
+
maps any of these names to the NumPy dtype. `Camera.precision` still reports
|
|
78
|
+
the dtype name (`"float32"`/`"float64"`) whichever spelling was passed.
|
|
79
|
+
`dataset.pairs(dtype=...)` is unchanged: it is the host *storage* type of the
|
|
80
|
+
finished arrays, not a working precision.
|
|
81
|
+
- The CLI's `[camera]` table takes a `device` key.
|
|
82
|
+
- **Conformance tests** for the device and precision vocabulary, including that
|
|
83
|
+
each precision name selects the same dtype as in aocore.
|
|
84
|
+
- **Edge-flux conformance for every PSF model.** `tests/test_conformance.py`
|
|
85
|
+
now uses aocore 0.1.3's image-builder checks (`check_point_source_centring`,
|
|
86
|
+
`check_point_source_flux`) instead of feeding analytic PSFs through the
|
|
87
|
+
OPD-driven checks with a dummy OPD, and adds `check_edge_flux_loss` for
|
|
88
|
+
Gaussian, Moffat, elliptical Gaussian, Airy and array PSFs, guarding the
|
|
89
|
+
2.3.0 fix. The `dev` extra pins `aocore>=0.1.3,<0.2`; the runtime
|
|
90
|
+
requirement is unchanged.
|
|
91
|
+
|
|
92
|
+
### Changed
|
|
93
|
+
|
|
94
|
+
- An unknown `device` string now raises `ValueError` listing the accepted words
|
|
95
|
+
(`'cpu', 'gpu', 'gpu:N' ... or 'auto'`), and a non-string `device` a
|
|
96
|
+
`TypeError`. `device="gpu"` with CuPy installed but no CUDA device raises
|
|
97
|
+
`RuntimeError` at construction rather than failing at the first frame.
|
|
98
|
+
- `Camera.__repr__` shows the GPU number (`device='gpu:0'`).
|
|
99
|
+
- The noise functions' `float_dtype` default is now `None` (still float64).
|
|
100
|
+
|
|
101
|
+
### Fixed
|
|
102
|
+
|
|
103
|
+
- `dataset.pairs` with a GPU camera failed on an implicit CuPy-to-NumPy
|
|
104
|
+
conversion; it now copies each frame to the host through `to_numpy`, as do
|
|
105
|
+
the CLI's `.npy`/`.npz` writers now that the CLI can select a GPU.
|
|
106
|
+
|
|
9
107
|
## [2.3.0] - 2026-10-07
|
|
10
108
|
|
|
11
109
|
### Fixed
|
|
@@ -592,7 +690,9 @@ together in 1.0.
|
|
|
592
690
|
- Documentation, runnable examples, and CI (lint, type-check, test matrix, PyPI
|
|
593
691
|
release via Trusted Publishing).
|
|
594
692
|
|
|
595
|
-
[Unreleased]: https://github.com/jacotay7/getframes/compare/2.
|
|
693
|
+
[Unreleased]: https://github.com/jacotay7/getframes/compare/2.5.0...HEAD
|
|
694
|
+
[2.5.0]: https://github.com/jacotay7/getframes/compare/2.4.0...2.5.0
|
|
695
|
+
[2.4.0]: https://github.com/jacotay7/getframes/compare/2.3.0...2.4.0
|
|
596
696
|
[2.3.0]: https://github.com/jacotay7/getframes/compare/2.2.0...2.3.0
|
|
597
697
|
[2.2.0]: https://github.com/jacotay7/getframes/compare/2.1.1...2.2.0
|
|
598
698
|
[2.1.1]: https://github.com/jacotay7/getframes/compare/2.1.0...2.1.1
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: getframes
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.5.0
|
|
4
4
|
Summary: Generate physically realistic synthetic camera frames (CCD/CMOS/EMCCD/eAPD/sCMOS) — dark, bias, flat, and rendered star fields — with auditable noise physics for scientific imaging pipelines.
|
|
5
5
|
Project-URL: Homepage, https://github.com/jacotay7/getframes
|
|
6
6
|
Project-URL: Documentation, https://jacotay7.github.io/getframes/
|
|
@@ -29,6 +29,7 @@ Requires-Dist: numpy>=1.23
|
|
|
29
29
|
Requires-Dist: scipy>=1.10
|
|
30
30
|
Requires-Dist: tomli>=2.0; python_version < '3.11'
|
|
31
31
|
Provides-Extra: dev
|
|
32
|
+
Requires-Dist: aocore<0.2,>=0.1.3; extra == 'dev'
|
|
32
33
|
Requires-Dist: build>=1.0; extra == 'dev'
|
|
33
34
|
Requires-Dist: mypy>=1.8; extra == 'dev'
|
|
34
35
|
Requires-Dist: pytest-cov>=4.0; extra == 'dev'
|
|
@@ -106,7 +107,7 @@ frame = cam.with_config(resolution=(256, 256)).observe(scene, exposure=300.0, se
|
|
|
106
107
|
|
|
107
108
|
import cupy as cp # and the same path on a GPU
|
|
108
109
|
|
|
109
|
-
cam = gf.Camera.from_preset("andor_ocam2k", device="gpu", precision="
|
|
110
|
+
cam = gf.Camera.from_preset("andor_ocam2k", device="gpu", precision="single")
|
|
110
111
|
rate = cp.full(cam.resolution, 2.0e6, dtype=cp.float32) # photons/s/pixel
|
|
111
112
|
frame = cam.expose(rate, exposure=1.0e-3, seed=0) # CuPy ADU, no host copy
|
|
112
113
|
```
|
|
@@ -186,8 +187,8 @@ for the methodology.
|
|
|
186
187
|
- **Scale & datasets** — a float32 fast path, vectorised multi-source rendering,
|
|
187
188
|
a streaming raw+truth `dataset` generator and a `getframes` CLI; see
|
|
188
189
|
**[Scale & datasets](https://jacotay7.github.io/getframes/guides/datasets/)**.
|
|
189
|
-
- **GPU-optional** — every camera takes `device="gpu"`
|
|
190
|
-
detector path and truth arrays device-resident. CPU and GPU have independent
|
|
190
|
+
- **GPU-optional** — every camera takes `device="gpu"`, `"gpu:N"` or `"auto"`
|
|
191
|
+
(CuPy) and keeps the detector path and truth arrays device-resident. CPU and GPU have independent
|
|
191
192
|
RNG streams, so a `seed` repeats exactly on a fixed backend while parity across
|
|
192
193
|
backends means matching statistics, not identical pixels.
|
|
193
194
|
- **Reproducible and typed** — all randomness flows through a camera-owned seeded
|
|
@@ -60,7 +60,7 @@ frame = cam.with_config(resolution=(256, 256)).observe(scene, exposure=300.0, se
|
|
|
60
60
|
|
|
61
61
|
import cupy as cp # and the same path on a GPU
|
|
62
62
|
|
|
63
|
-
cam = gf.Camera.from_preset("andor_ocam2k", device="gpu", precision="
|
|
63
|
+
cam = gf.Camera.from_preset("andor_ocam2k", device="gpu", precision="single")
|
|
64
64
|
rate = cp.full(cam.resolution, 2.0e6, dtype=cp.float32) # photons/s/pixel
|
|
65
65
|
frame = cam.expose(rate, exposure=1.0e-3, seed=0) # CuPy ADU, no host copy
|
|
66
66
|
```
|
|
@@ -140,8 +140,8 @@ for the methodology.
|
|
|
140
140
|
- **Scale & datasets** — a float32 fast path, vectorised multi-source rendering,
|
|
141
141
|
a streaming raw+truth `dataset` generator and a `getframes` CLI; see
|
|
142
142
|
**[Scale & datasets](https://jacotay7.github.io/getframes/guides/datasets/)**.
|
|
143
|
-
- **GPU-optional** — every camera takes `device="gpu"`
|
|
144
|
-
detector path and truth arrays device-resident. CPU and GPU have independent
|
|
143
|
+
- **GPU-optional** — every camera takes `device="gpu"`, `"gpu:N"` or `"auto"`
|
|
144
|
+
(CuPy) and keeps the detector path and truth arrays device-resident. CPU and GPU have independent
|
|
145
145
|
RNG streams, so a `seed` repeats exactly on a fixed backend while parity across
|
|
146
146
|
backends means matching statistics, not identical pixels.
|
|
147
147
|
- **Reproducible and typed** — all randomness flows through a camera-owned seeded
|
|
@@ -112,6 +112,15 @@ def _cpu_model() -> str:
|
|
|
112
112
|
return line.split(":", 1)[1].strip()
|
|
113
113
|
except OSError:
|
|
114
114
|
pass
|
|
115
|
+
# Arm /proc/cpuinfo has no "model name"; lscpu decodes the part number
|
|
116
|
+
# (e.g. "Neoverse-N1").
|
|
117
|
+
try:
|
|
118
|
+
output = subprocess.run(["lscpu"], check=True, capture_output=True, text=True).stdout
|
|
119
|
+
except (OSError, subprocess.CalledProcessError):
|
|
120
|
+
output = ""
|
|
121
|
+
for line in output.splitlines():
|
|
122
|
+
if line.startswith("Model name:"):
|
|
123
|
+
return line.split(":", 1)[1].strip()
|
|
115
124
|
return platform.processor() or "unknown CPU"
|
|
116
125
|
|
|
117
126
|
|
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schema_version": 1,
|
|
3
|
+
"generated_at_utc": "2026-10-07T07:14:14.461693+00:00",
|
|
4
|
+
"command": "/home/jtaylor/miniforge3/envs/aosim-bench/bin/python benchmarks/bench_devices.py --seconds 2 --warmup 10 --device both --output /home/jtaylor/aosim-bench-logs/artifacts/getframes-device-rtx4060.json",
|
|
5
|
+
"revision": "7b01daf765d432ab141176548f2b11cf0e7fc11c",
|
|
6
|
+
"source_dirty": true,
|
|
7
|
+
"python": "3.13.15 | packaged by conda-forge | (main, Sep 2 2026, 22:02:03) [GCC 15.3.0]",
|
|
8
|
+
"platform": "Linux-6.17.9-76061709-generic-aarch64-with-glibc2.39",
|
|
9
|
+
"processor": "aarch64",
|
|
10
|
+
"cpu": "Neoverse-N1",
|
|
11
|
+
"gpu": "NVIDIA GeForce RTX 4060",
|
|
12
|
+
"dependencies": {
|
|
13
|
+
"getframes": "2.4.0",
|
|
14
|
+
"numpy": "2.5.3",
|
|
15
|
+
"scipy": "1.18.1",
|
|
16
|
+
"cupy": "14.2.0"
|
|
17
|
+
},
|
|
18
|
+
"methodology": {
|
|
19
|
+
"seconds_per_cell": 2.0,
|
|
20
|
+
"warmup_frames": 10,
|
|
21
|
+
"persistent_camera": true,
|
|
22
|
+
"device_resident_rate_and_output": true,
|
|
23
|
+
"include_truth": true,
|
|
24
|
+
"rng": "one generator seeded at camera construction and advanced per frame",
|
|
25
|
+
"cuda_synchronization": "before and after each timed region",
|
|
26
|
+
"construction_included": false,
|
|
27
|
+
"host_transfers_included": false
|
|
28
|
+
},
|
|
29
|
+
"results": [
|
|
30
|
+
{
|
|
31
|
+
"workflow": "pyramid_cmos_80",
|
|
32
|
+
"label": "Pyramid WFS CMOS",
|
|
33
|
+
"preset": "generic_cmos",
|
|
34
|
+
"sensor": "CMOS",
|
|
35
|
+
"shape": [
|
|
36
|
+
80,
|
|
37
|
+
80
|
|
38
|
+
],
|
|
39
|
+
"precision": "float32",
|
|
40
|
+
"exposure_s": 0.001,
|
|
41
|
+
"photon_rate_per_s": 2000000.0,
|
|
42
|
+
"device": "cpu",
|
|
43
|
+
"frames": 4295,
|
|
44
|
+
"elapsed_s": 2.0004646239976864,
|
|
45
|
+
"frame_s": 0.000465765919440672,
|
|
46
|
+
"frames_per_s": 2147.001225853703,
|
|
47
|
+
"megapixels_per_s": 13.740807845463701
|
|
48
|
+
},
|
|
49
|
+
{
|
|
50
|
+
"workflow": "pyramid_cmos_80",
|
|
51
|
+
"label": "Pyramid WFS CMOS",
|
|
52
|
+
"preset": "generic_cmos",
|
|
53
|
+
"sensor": "CMOS",
|
|
54
|
+
"shape": [
|
|
55
|
+
80,
|
|
56
|
+
80
|
|
57
|
+
],
|
|
58
|
+
"precision": "float32",
|
|
59
|
+
"exposure_s": 0.001,
|
|
60
|
+
"photon_rate_per_s": 2000000.0,
|
|
61
|
+
"device": "gpu",
|
|
62
|
+
"frames": 5011,
|
|
63
|
+
"elapsed_s": 2.0001363799965475,
|
|
64
|
+
"frame_s": 0.0003991491478739867,
|
|
65
|
+
"frames_per_s": 2505.329161608795,
|
|
66
|
+
"megapixels_per_s": 16.034106634296286
|
|
67
|
+
},
|
|
68
|
+
{
|
|
69
|
+
"workflow": "shack_hartmann_cmos_160",
|
|
70
|
+
"label": "Shack-Hartmann WFS CMOS",
|
|
71
|
+
"preset": "generic_cmos",
|
|
72
|
+
"sensor": "CMOS",
|
|
73
|
+
"shape": [
|
|
74
|
+
160,
|
|
75
|
+
160
|
|
76
|
+
],
|
|
77
|
+
"precision": "float32",
|
|
78
|
+
"exposure_s": 0.001,
|
|
79
|
+
"photon_rate_per_s": 2000000.0,
|
|
80
|
+
"device": "cpu",
|
|
81
|
+
"frames": 1149,
|
|
82
|
+
"elapsed_s": 2.001656759006437,
|
|
83
|
+
"frame_s": 0.001742085952137891,
|
|
84
|
+
"frames_per_s": 574.0244898782395,
|
|
85
|
+
"megapixels_per_s": 14.69502694088293
|
|
86
|
+
},
|
|
87
|
+
{
|
|
88
|
+
"workflow": "shack_hartmann_cmos_160",
|
|
89
|
+
"label": "Shack-Hartmann WFS CMOS",
|
|
90
|
+
"preset": "generic_cmos",
|
|
91
|
+
"sensor": "CMOS",
|
|
92
|
+
"shape": [
|
|
93
|
+
160,
|
|
94
|
+
160
|
|
95
|
+
],
|
|
96
|
+
"precision": "float32",
|
|
97
|
+
"exposure_s": 0.001,
|
|
98
|
+
"photon_rate_per_s": 2000000.0,
|
|
99
|
+
"device": "gpu",
|
|
100
|
+
"frames": 5026,
|
|
101
|
+
"elapsed_s": 2.000239181012148,
|
|
102
|
+
"frame_s": 0.0003979783487887282,
|
|
103
|
+
"frames_per_s": 2512.699504994586,
|
|
104
|
+
"megapixels_per_s": 64.3251073278614
|
|
105
|
+
},
|
|
106
|
+
{
|
|
107
|
+
"workflow": "ocam2k_emccd_240",
|
|
108
|
+
"label": "OCAM2K EMCCD",
|
|
109
|
+
"preset": "andor_ocam2k",
|
|
110
|
+
"sensor": "EMCCD",
|
|
111
|
+
"shape": [
|
|
112
|
+
240,
|
|
113
|
+
240
|
|
114
|
+
],
|
|
115
|
+
"precision": "float32",
|
|
116
|
+
"exposure_s": 0.001,
|
|
117
|
+
"photon_rate_per_s": 2000000.0,
|
|
118
|
+
"device": "cpu",
|
|
119
|
+
"frames": 288,
|
|
120
|
+
"elapsed_s": 2.003861945006065,
|
|
121
|
+
"frame_s": 0.006957853975715504,
|
|
122
|
+
"frames_per_s": 143.72247585106385,
|
|
123
|
+
"megapixels_per_s": 8.278414609021278
|
|
124
|
+
},
|
|
125
|
+
{
|
|
126
|
+
"workflow": "ocam2k_emccd_240",
|
|
127
|
+
"label": "OCAM2K EMCCD",
|
|
128
|
+
"preset": "andor_ocam2k",
|
|
129
|
+
"sensor": "EMCCD",
|
|
130
|
+
"shape": [
|
|
131
|
+
240,
|
|
132
|
+
240
|
|
133
|
+
],
|
|
134
|
+
"precision": "float32",
|
|
135
|
+
"exposure_s": 0.001,
|
|
136
|
+
"photon_rate_per_s": 2000000.0,
|
|
137
|
+
"device": "gpu",
|
|
138
|
+
"frames": 3420,
|
|
139
|
+
"elapsed_s": 2.0004429029941093,
|
|
140
|
+
"frame_s": 0.000584924825436874,
|
|
141
|
+
"frames_per_s": 1709.621401781179,
|
|
142
|
+
"megapixels_per_s": 98.47419274259589
|
|
143
|
+
},
|
|
144
|
+
{
|
|
145
|
+
"workflow": "saphira_eapd_256x320",
|
|
146
|
+
"label": "SAPHIRA eAPD",
|
|
147
|
+
"preset": "leonardo_saphira",
|
|
148
|
+
"sensor": "EAPD",
|
|
149
|
+
"shape": [
|
|
150
|
+
256,
|
|
151
|
+
320
|
|
152
|
+
],
|
|
153
|
+
"precision": "float32",
|
|
154
|
+
"exposure_s": 0.001,
|
|
155
|
+
"photon_rate_per_s": 2000000.0,
|
|
156
|
+
"device": "cpu",
|
|
157
|
+
"frames": 228,
|
|
158
|
+
"elapsed_s": 2.0011549520131666,
|
|
159
|
+
"frame_s": 0.00877699540356652,
|
|
160
|
+
"frames_per_s": 113.93420572986187,
|
|
161
|
+
"megapixels_per_s": 9.333490133390285
|
|
162
|
+
},
|
|
163
|
+
{
|
|
164
|
+
"workflow": "saphira_eapd_256x320",
|
|
165
|
+
"label": "SAPHIRA eAPD",
|
|
166
|
+
"preset": "leonardo_saphira",
|
|
167
|
+
"sensor": "EAPD",
|
|
168
|
+
"shape": [
|
|
169
|
+
256,
|
|
170
|
+
320
|
|
171
|
+
],
|
|
172
|
+
"precision": "float32",
|
|
173
|
+
"exposure_s": 0.001,
|
|
174
|
+
"photon_rate_per_s": 2000000.0,
|
|
175
|
+
"device": "gpu",
|
|
176
|
+
"frames": 3786,
|
|
177
|
+
"elapsed_s": 2.000311462994432,
|
|
178
|
+
"frame_s": 0.000528344284995888,
|
|
179
|
+
"frames_per_s": 1892.7052461782241,
|
|
180
|
+
"megapixels_per_s": 155.05041376692012
|
|
181
|
+
},
|
|
182
|
+
{
|
|
183
|
+
"workflow": "science_cmos_1024",
|
|
184
|
+
"label": "Large science CMOS",
|
|
185
|
+
"preset": "generic_cmos",
|
|
186
|
+
"sensor": "CMOS",
|
|
187
|
+
"shape": [
|
|
188
|
+
1024,
|
|
189
|
+
1024
|
|
190
|
+
],
|
|
191
|
+
"precision": "float32",
|
|
192
|
+
"exposure_s": 5.0,
|
|
193
|
+
"photon_rate_per_s": 200.0,
|
|
194
|
+
"device": "cpu",
|
|
195
|
+
"frames": 27,
|
|
196
|
+
"elapsed_s": 2.000652825983707,
|
|
197
|
+
"frame_s": 0.07409825281421137,
|
|
198
|
+
"frames_per_s": 13.495594862504088,
|
|
199
|
+
"megapixels_per_s": 14.151156878545088
|
|
200
|
+
},
|
|
201
|
+
{
|
|
202
|
+
"workflow": "science_cmos_1024",
|
|
203
|
+
"label": "Large science CMOS",
|
|
204
|
+
"preset": "generic_cmos",
|
|
205
|
+
"sensor": "CMOS",
|
|
206
|
+
"shape": [
|
|
207
|
+
1024,
|
|
208
|
+
1024
|
|
209
|
+
],
|
|
210
|
+
"precision": "float32",
|
|
211
|
+
"exposure_s": 5.0,
|
|
212
|
+
"photon_rate_per_s": 200.0,
|
|
213
|
+
"device": "gpu",
|
|
214
|
+
"frames": 525,
|
|
215
|
+
"elapsed_s": 2.2030622210004367,
|
|
216
|
+
"frame_s": 0.0041963089923817845,
|
|
217
|
+
"frames_per_s": 238.30466293484497,
|
|
218
|
+
"megapixels_per_s": 249.880550241568
|
|
219
|
+
}
|
|
220
|
+
]
|
|
221
|
+
}
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schema_version": 1,
|
|
3
|
+
"generated_at_utc": "2026-10-07T07:14:37.452446+00:00",
|
|
4
|
+
"command": "/home/jtaylor/miniforge3/envs/aosim-bench/bin/python benchmarks/bench_devices.py --seconds 2 --warmup 10 --device gpu --output /home/jtaylor/aosim-bench-logs/artifacts/getframes-device-rtxa400.json",
|
|
5
|
+
"revision": "7b01daf765d432ab141176548f2b11cf0e7fc11c",
|
|
6
|
+
"source_dirty": true,
|
|
7
|
+
"python": "3.13.15 | packaged by conda-forge | (main, Sep 2 2026, 22:02:03) [GCC 15.3.0]",
|
|
8
|
+
"platform": "Linux-6.17.9-76061709-generic-aarch64-with-glibc2.39",
|
|
9
|
+
"processor": "aarch64",
|
|
10
|
+
"cpu": "Neoverse-N1",
|
|
11
|
+
"gpu": "NVIDIA RTX A400",
|
|
12
|
+
"dependencies": {
|
|
13
|
+
"getframes": "2.4.0",
|
|
14
|
+
"numpy": "2.5.3",
|
|
15
|
+
"scipy": "1.18.1",
|
|
16
|
+
"cupy": "14.2.0"
|
|
17
|
+
},
|
|
18
|
+
"methodology": {
|
|
19
|
+
"seconds_per_cell": 2.0,
|
|
20
|
+
"warmup_frames": 10,
|
|
21
|
+
"persistent_camera": true,
|
|
22
|
+
"device_resident_rate_and_output": true,
|
|
23
|
+
"include_truth": true,
|
|
24
|
+
"rng": "one generator seeded at camera construction and advanced per frame",
|
|
25
|
+
"cuda_synchronization": "before and after each timed region",
|
|
26
|
+
"construction_included": false,
|
|
27
|
+
"host_transfers_included": false
|
|
28
|
+
},
|
|
29
|
+
"results": [
|
|
30
|
+
{
|
|
31
|
+
"workflow": "pyramid_cmos_80",
|
|
32
|
+
"label": "Pyramid WFS CMOS",
|
|
33
|
+
"preset": "generic_cmos",
|
|
34
|
+
"sensor": "CMOS",
|
|
35
|
+
"shape": [
|
|
36
|
+
80,
|
|
37
|
+
80
|
|
38
|
+
],
|
|
39
|
+
"precision": "float32",
|
|
40
|
+
"exposure_s": 0.001,
|
|
41
|
+
"photon_rate_per_s": 2000000.0,
|
|
42
|
+
"device": "gpu",
|
|
43
|
+
"frames": 4672,
|
|
44
|
+
"elapsed_s": 2.0003948630183004,
|
|
45
|
+
"frame_s": 0.0004281667086939855,
|
|
46
|
+
"frames_per_s": 2335.5388910320644,
|
|
47
|
+
"megapixels_per_s": 14.947448902605213
|
|
48
|
+
},
|
|
49
|
+
{
|
|
50
|
+
"workflow": "shack_hartmann_cmos_160",
|
|
51
|
+
"label": "Shack-Hartmann WFS CMOS",
|
|
52
|
+
"preset": "generic_cmos",
|
|
53
|
+
"sensor": "CMOS",
|
|
54
|
+
"shape": [
|
|
55
|
+
160,
|
|
56
|
+
160
|
|
57
|
+
],
|
|
58
|
+
"precision": "float32",
|
|
59
|
+
"exposure_s": 0.001,
|
|
60
|
+
"photon_rate_per_s": 2000000.0,
|
|
61
|
+
"device": "gpu",
|
|
62
|
+
"frames": 3458,
|
|
63
|
+
"elapsed_s": 2.0286170829785988,
|
|
64
|
+
"frame_s": 0.00058664461624598,
|
|
65
|
+
"frames_per_s": 1704.6095239041624,
|
|
66
|
+
"megapixels_per_s": 43.638003811946554
|
|
67
|
+
},
|
|
68
|
+
{
|
|
69
|
+
"workflow": "ocam2k_emccd_240",
|
|
70
|
+
"label": "OCAM2K EMCCD",
|
|
71
|
+
"preset": "andor_ocam2k",
|
|
72
|
+
"sensor": "EMCCD",
|
|
73
|
+
"shape": [
|
|
74
|
+
240,
|
|
75
|
+
240
|
|
76
|
+
],
|
|
77
|
+
"precision": "float32",
|
|
78
|
+
"exposure_s": 0.001,
|
|
79
|
+
"photon_rate_per_s": 2000000.0,
|
|
80
|
+
"device": "gpu",
|
|
81
|
+
"frames": 1073,
|
|
82
|
+
"elapsed_s": 2.0662656150234398,
|
|
83
|
+
"frame_s": 0.0019256902283536252,
|
|
84
|
+
"frames_per_s": 519.2943212133101,
|
|
85
|
+
"megapixels_per_s": 29.911352901886666
|
|
86
|
+
},
|
|
87
|
+
{
|
|
88
|
+
"workflow": "saphira_eapd_256x320",
|
|
89
|
+
"label": "SAPHIRA eAPD",
|
|
90
|
+
"preset": "leonardo_saphira",
|
|
91
|
+
"sensor": "EAPD",
|
|
92
|
+
"shape": [
|
|
93
|
+
256,
|
|
94
|
+
320
|
|
95
|
+
],
|
|
96
|
+
"precision": "float32",
|
|
97
|
+
"exposure_s": 0.001,
|
|
98
|
+
"photon_rate_per_s": 2000000.0,
|
|
99
|
+
"device": "gpu",
|
|
100
|
+
"frames": 807,
|
|
101
|
+
"elapsed_s": 2.099866060016211,
|
|
102
|
+
"frame_s": 0.0026020645105529257,
|
|
103
|
+
"frames_per_s": 384.3102259549687,
|
|
104
|
+
"megapixels_per_s": 31.482693710231036
|
|
105
|
+
},
|
|
106
|
+
{
|
|
107
|
+
"workflow": "science_cmos_1024",
|
|
108
|
+
"label": "Large science CMOS",
|
|
109
|
+
"preset": "generic_cmos",
|
|
110
|
+
"sensor": "CMOS",
|
|
111
|
+
"shape": [
|
|
112
|
+
1024,
|
|
113
|
+
1024
|
|
114
|
+
],
|
|
115
|
+
"precision": "float32",
|
|
116
|
+
"exposure_s": 5.0,
|
|
117
|
+
"photon_rate_per_s": 200.0,
|
|
118
|
+
"device": "gpu",
|
|
119
|
+
"frames": 133,
|
|
120
|
+
"elapsed_s": 3.1430114879913162,
|
|
121
|
+
"frame_s": 0.02363166532324298,
|
|
122
|
+
"frames_per_s": 42.31610368214074,
|
|
123
|
+
"megapixels_per_s": 44.37165073460441
|
|
124
|
+
}
|
|
125
|
+
]
|
|
126
|
+
}
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# CPU/GPU detector throughput on an Arm host
|
|
2
|
+
|
|
3
|
+
Frames/s, higher is better. These come from
|
|
4
|
+
`benchmarks/device-results-neoverse-n1-rtx4060.json` and
|
|
5
|
+
`benchmarks/device-results-neoverse-n1-rtxa400.json`. They use the same
|
|
6
|
+
method and workflows as [device-results.md](device-results.md), whose
|
|
7
|
+
x86 + RTX 5090 numbers (getframes 2.1.1) are repeated here for reference.
|
|
8
|
+
|
|
9
|
+
- Host: cfl-test-bench, an 80-core Ampere Neoverse-N1 (aarch64), shared.
|
|
10
|
+
Every run was pinned to 16 cores (`taskset -c 16-31`) with 16 BLAS
|
|
11
|
+
threads.
|
|
12
|
+
- GPUs: NVIDIA GeForce RTX 4060 (8 GB) and NVIDIA RTX A400 (4 GB), driver 580.
|
|
13
|
+
- Dependencies: getframes 2.4.0, NumPy 2.5.3, SciPy 1.18.1, CuPy 14.2.0.
|
|
14
|
+
- Method: persistent float32 camera, warm device-resident rate and output,
|
|
15
|
+
truth enabled, construction and host transfers excluded, CUDA synchronized.
|
|
16
|
+
|
|
17
|
+
| Workflow | Detector | Native shape | Neoverse-N1 CPU (16 cores) | RTX A400 | RTX 4060 | Ryzen 9 9950X3D CPU | RTX 5090 |
|
|
18
|
+
| --- | --- | ---: | ---: | ---: | ---: | ---: | ---: |
|
|
19
|
+
| Pyramid WFS CMOS | CMOS | 80x80 | 2,147.0 | 2,335.5 | 2,505.3 | 5,240.3 | 11,513.6 |
|
|
20
|
+
| Shack-Hartmann WFS CMOS | CMOS | 160x160 | 574.0 | 1,704.6 | 2,512.7 | 1,386.3 | 11,470.8 |
|
|
21
|
+
| OCAM2K EMCCD | EMCCD | 240x240 | 143.7 | 519.3 | 1,709.6 | 357.0 | 8,045.0 |
|
|
22
|
+
| SAPHIRA eAPD | EAPD | 256x320 | 113.9 | 384.3 | 1,892.7 | 280.4 | 7,496.5 |
|
|
23
|
+
| Large science CMOS | CMOS | 1024x1024 | 13.5 | 42.3 | 238.3 | 30.8 | 1,453.2 |
|
|
24
|
+
|
|
25
|
+
Reading this:
|
|
26
|
+
|
|
27
|
+
- **The GPU advantage grows with the detector.** At 80x80 the GPU barely
|
|
28
|
+
beats 16 N1 cores (1.2x on the RTX 4060), because each frame is a handful
|
|
29
|
+
of kernel launches. At 1024x1024 it is 18x.
|
|
30
|
+
- **The two cards only separate on large detectors.** The RTX 4060 runs
|
|
31
|
+
1.1x the RTX A400 at 80x80, 3.3x on the OCAM2K and 5.6x at 1024x1024.
|
|
32
|
+
- **16 Neoverse-N1 cores reach a steady 0.40–0.44x a 16-core Ryzen 9
|
|
33
|
+
9950X3D** across every workflow.
|
|
34
|
+
|
|
35
|
+
These are getframes 2.4.0 numbers. From 2.5.0 the GPU fuses the per-frame
|
|
36
|
+
chain into far fewer kernels, which lifts the launch-bound Pyramid 80x80
|
|
37
|
+
frame on this host to about 6,300–6,400 frames/s on either card, and
|
|
38
|
+
Shack-Hartmann 160x160 to about 6,300 on the RTX 4060 (2.6x); see
|
|
39
|
+
[Small frames: fused kernels](../docs/guides/gpu.md#small-frames-fused-kernels).
|
|
@@ -21,7 +21,7 @@ from .analysis import (
|
|
|
21
21
|
nondestructive_stack_statistics,
|
|
22
22
|
ramp_photon_transfer,
|
|
23
23
|
)
|
|
24
|
-
from .backend import ArrayBackend, get_array_module, get_backend, to_numpy
|
|
24
|
+
from .backend import ArrayBackend, get_array_module, get_backend, resolve_precision, to_numpy
|
|
25
25
|
from .calibrate import calibrate, combine
|
|
26
26
|
from .camera import Camera
|
|
27
27
|
from .config import CameraConfig, SensorType
|
|
@@ -108,5 +108,6 @@ __all__ = [
|
|
|
108
108
|
"load_preset",
|
|
109
109
|
"nondestructive_stack_statistics",
|
|
110
110
|
"ramp_photon_transfer",
|
|
111
|
+
"resolve_precision",
|
|
111
112
|
"to_numpy",
|
|
112
113
|
]
|