dkist-processing-trend 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. changelog/.gitempty +0 -0
  2. dkist_processing_trend/__init__.py +10 -0
  3. dkist_processing_trend/config.py +11 -0
  4. dkist_processing_trend/models/__init__.py +1 -0
  5. dkist_processing_trend/models/constants.py +143 -0
  6. dkist_processing_trend/models/fit_options.py +15 -0
  7. dkist_processing_trend/models/fits_access.py +65 -0
  8. dkist_processing_trend/models/instrument.py +35 -0
  9. dkist_processing_trend/models/instrument_options.py +35 -0
  10. dkist_processing_trend/models/parameters.py +134 -0
  11. dkist_processing_trend/models/tags.py +117 -0
  12. dkist_processing_trend/models/task_name.py +20 -0
  13. dkist_processing_trend/parsers/__init__.py +1 -0
  14. dkist_processing_trend/parsers/arm_id.py +103 -0
  15. dkist_processing_trend/parsers/instrument_unique_bud.py +40 -0
  16. dkist_processing_trend/parsers/time.py +27 -0
  17. dkist_processing_trend/parsers/trend_l0_fits_access.py +123 -0
  18. dkist_processing_trend/tasks/__init__.py +27 -0
  19. dkist_processing_trend/tasks/arm_task_factory.py +57 -0
  20. dkist_processing_trend/tasks/dark.py +56 -0
  21. dkist_processing_trend/tasks/gain.py +74 -0
  22. dkist_processing_trend/tasks/initialize_arm_tasks.py +50 -0
  23. dkist_processing_trend/tasks/parse.py +169 -0
  24. dkist_processing_trend/tasks/prepare_fit_data_base.py +257 -0
  25. dkist_processing_trend/tasks/run_pac_fitter.py +392 -0
  26. dkist_processing_trend/tasks/trend_base.py +97 -0
  27. dkist_processing_trend/tasks/trend_output_data.py +167 -0
  28. dkist_processing_trend/tasks/visp/__init__.py +6 -0
  29. dkist_processing_trend/tasks/visp/visp_dmpd.py +410 -0
  30. dkist_processing_trend/tasks/visp/visp_extract_beam.py +14 -0
  31. dkist_processing_trend/tasks/visp/visp_geometric.py +260 -0
  32. dkist_processing_trend/tasks/visp/visp_prep_fit_data.py +162 -0
  33. dkist_processing_trend/tasks/visp/visp_process_demod.py +236 -0
  34. dkist_processing_trend/tasks/write_trend.py +663 -0
  35. dkist_processing_trend/tests/__init__.py +1 -0
  36. dkist_processing_trend/tests/conftest.py +718 -0
  37. dkist_processing_trend/tests/local_trial_workflows/__init__.py +0 -0
  38. dkist_processing_trend/tests/local_trial_workflows/l0_to_trend_visp_polcal.py +294 -0
  39. dkist_processing_trend/tests/local_trial_workflows/local_trial_helpers.py +488 -0
  40. dkist_processing_trend/tests/test_arm_task_factory.py +82 -0
  41. dkist_processing_trend/tests/test_base_tasks.py +86 -0
  42. dkist_processing_trend/tests/test_constants.py +120 -0
  43. dkist_processing_trend/tests/test_dark.py +97 -0
  44. dkist_processing_trend/tests/test_gain.py +135 -0
  45. dkist_processing_trend/tests/test_parameters.py +149 -0
  46. dkist_processing_trend/tests/test_parse.py +276 -0
  47. dkist_processing_trend/tests/test_prep_fit_data_base.py +233 -0
  48. dkist_processing_trend/tests/test_publish_catalog_messages.py +45 -0
  49. dkist_processing_trend/tests/test_run_pac_fitter.py +371 -0
  50. dkist_processing_trend/tests/test_stems.py +75 -0
  51. dkist_processing_trend/tests/test_transfer_output_data.py +76 -0
  52. dkist_processing_trend/tests/test_trend_fits_access.py +173 -0
  53. dkist_processing_trend/tests/test_visp.py +874 -0
  54. dkist_processing_trend/tests/test_workflows.py +10 -0
  55. dkist_processing_trend/tests/test_write_trend.py +460 -0
  56. dkist_processing_trend/workflows/__init__.py +3 -0
  57. dkist_processing_trend/workflows/visp.py +58 -0
  58. dkist_processing_trend-0.1.0.dist-info/METADATA +549 -0
  59. dkist_processing_trend-0.1.0.dist-info/RECORD +66 -0
  60. dkist_processing_trend-0.1.0.dist-info/WHEEL +5 -0
  61. dkist_processing_trend-0.1.0.dist-info/top_level.txt +3 -0
  62. docs/conf.py +57 -0
  63. docs/index.rst +10 -0
  64. docs/l0_to_trend_visp_polcal.rst +4 -0
  65. docs/landing_page.rst +11 -0
  66. docs/requirements_table.rst +8 -0
@@ -0,0 +1,162 @@
1
+ """Task for preparing ViSP data for fitting by `dkist_processing_pac`."""
2
+
3
+ import numpy as np
4
+ import scipy.ndimage as spnd
5
+
6
+ from dkist_processing_trend.models.instrument_options import VispInstrumentOptions
7
+ from dkist_processing_trend.tasks.prepare_fit_data_base import PrepareFitDataBase
8
+ from dkist_processing_trend.tasks.visp.visp_extract_beam import extract_visp_beam
9
+
10
+ __all__ = ["VispPrepareFitData"]
11
+
12
+
13
+ class VispPrepareFitData(PrepareFitDataBase):
14
+ """
15
+ Subclass of `~dkist_processing_trend.tasks.prepare_fit_data_base.PrepareFitDataBase` that adds functionality for working with ViSP data.
16
+
17
+ Parameters
18
+ ----------
19
+ arm_id
20
+ id of the instrument arm to operate on
21
+
22
+ recipe_run_id
23
+ id of the recipe run used to identify the workflow run this task is part of
24
+
25
+ workflow_name
26
+ name of the workflow to which this instance of the task belongs
27
+
28
+ workflow_version
29
+ version of the workflow to which this instance of the task belongs
30
+ """
31
+
32
+ def extract_beam(self, array: np.ndarray, beam: int) -> np.ndarray:
33
+ """Extract a given beam array from a full ViSP frame."""
34
+ beam_border = self.parameters.visp_beam_border
35
+ return extract_visp_beam(array=array, beam=beam, beam_border=beam_border)
36
+
37
+ def apply_global_instrument_options(
38
+ self, array: np.ndarray, instrument_options: VispInstrumentOptions
39
+ ) -> np.ndarray:
40
+ """
41
+ Prepare global ViSP data for PAC fitting.
42
+
43
+ We first mask the slit hairlines and then simply compute a global median of the array.
44
+ """
45
+ filtered_array = self.mask_hairlines(array)
46
+
47
+ return np.nanmedian(filtered_array)[None, None]
48
+
49
+ def apply_local_instrument_options(
50
+ self, array: np.ndarray, instrument_options: VispInstrumentOptions
51
+ ) -> np.ndarray:
52
+ """
53
+ Prepare local ViSP data for PAC fitting.
54
+
55
+ Algorithm:
56
+
57
+ #. Mask the slit hairlines
58
+ #. Collapse the array by computing the median along the spectral dimension
59
+ #. Smooth the result in the spatial dimension
60
+ #. Bin the array in the spatial dimension using the local median to compute the value of each bin
61
+ """
62
+ filtered_array = self.mask_hairlines(array)
63
+
64
+ # Add back in a dummy spectral dimension so things stay 2D, which helps thinking about it later on.
65
+ spectral_binned_array = np.nanmedian(filtered_array, axis=0)[None, :]
66
+
67
+ spatially_smoothed_array = spnd.median_filter(
68
+ spectral_binned_array,
69
+ # The 1 below means we don't smooth in the spectral dimension
70
+ size=(1, self.parameters.visp_polcal_spatial_median_filter_width_px),
71
+ )
72
+
73
+ local_array = self.downsample_spatial_dimension_local_median(
74
+ spatially_smoothed_array, instrument_options.num_spatial_px
75
+ )
76
+
77
+ return local_array
78
+
79
+ def mask_hairlines(self, array: np.ndarray) -> np.ndarray:
80
+ # This method is copied directly from the `CorrectionsMixin` in `dkist-processing-visp`
81
+ """
82
+ Mask hairlines from an array.
83
+
84
+ The hairlines will be replaced with data from a median-filtered version of the input array.
85
+
86
+ Hairlines are identified by first subtracting a spatially smoothed copy of the array and then looking for pixels
87
+ that have large differences. This works because the hairlines are the only features that are sharp in the
88
+ spatial dimension. The identified hairlines are then slightly smoothed spatially to ensure that their
89
+ higher-flux wings are correctly masked.
90
+ """
91
+ filtered_array = self.median_filter_array_for_hairline_identification(array)
92
+ hairline_locations = self.find_hairline_pixels(
93
+ input_array=array, filtered_array=filtered_array
94
+ )
95
+
96
+ # Replace hairline pixels with data from the spatially-filtered array
97
+ array[hairline_locations] = filtered_array[hairline_locations]
98
+
99
+ return array
100
+
101
+ def median_filter_array_for_hairline_identification(self, array: np.ndarray) -> np.ndarray:
102
+ # This method is copied directly from the `CorrectionsMixin` in `dkist-processing-visp`
103
+ """
104
+ Small helper to separate out the median filter step of hairline identification.
105
+
106
+ This step has been factored out so that functions that need the filtered array for further processing can avoid
107
+ repeating this expensive computation.
108
+ """
109
+ # The size=(1, X) means we only smooth in the spatial dimension (1st axis)
110
+ filtered_array = spnd.median_filter(
111
+ array, size=(1, self.parameters.visp_hairline_median_spatial_smoothing_width_px)
112
+ )
113
+ return filtered_array
114
+
115
+ def find_hairline_pixels(
116
+ self, input_array: np.ndarray, filtered_array: np.ndarray
117
+ ) -> np.ndarray:
118
+ # This method is copied directly from the `CorrectionsMixin` in `dkist-processing-visp`
119
+ """
120
+ Find pixels that likely correspond to hairlines.
121
+
122
+ This also slightly smooths the identified pixels so that high-flux wings of the hairlines are included.
123
+ """
124
+ diff = (input_array - filtered_array) / filtered_array
125
+ hairline_locations = np.abs(diff) > self.parameters.visp_hairline_fraction
126
+
127
+ # Now smooth the hairline mask in the spatial dimension to capture the higher-flux wings
128
+ mask_array = np.zeros_like(input_array)
129
+ mask_array[hairline_locations] = 1.0
130
+ mask_array = spnd.gaussian_filter1d(
131
+ mask_array, self.parameters.visp_hairline_mask_spatial_smoothing_width_px, axis=1
132
+ )
133
+
134
+ hairline_locations = np.where(
135
+ mask_array
136
+ > mask_array.max() * self.parameters.visp_hairline_mask_gaussian_peak_cutoff_fraction
137
+ )
138
+
139
+ return hairline_locations
140
+
141
+ @staticmethod
142
+ def downsample_spatial_dimension_local_median(
143
+ data: np.ndarray, num_spatial_bins: int
144
+ ) -> np.ndarray:
145
+ # Copied directly from the `DownsampleMixin` of `dkist-processing-visp`
146
+ """Resample a stack of spectra along the spatial dimension.
147
+
148
+ This is a separate function because calling `skimage.measure.block_reduce` on the entire (large) array all at once
149
+ creates a huge memory strain that is not needed since we're only reducing over one of the dimensions. Instead,
150
+ this function does some very cool tricks with reshaping and base numpy to keep the memory footprint lower.
151
+ """
152
+ # Taken from the amazing answer in
153
+ # https://stackoverflow.com/questions/44527579/whats-the-best-way-to-downsample-a-numpy-array
154
+ num_wave, num_spat_pos = data.shape
155
+
156
+ if (num_spat_pos / num_spatial_bins) % 1 != 0:
157
+ raise ValueError(
158
+ f"The number of spatial bins must evenly divide the spatial dimension. {num_spatial_bins} bins do not evenly divide {num_spat_pos}."
159
+ )
160
+
161
+ reshaped = data.reshape((num_wave, num_spatial_bins, num_spat_pos // num_spatial_bins))
162
+ return np.median(reshaped, axis=2)
@@ -0,0 +1,236 @@
1
+ """Task for producing full-frame ViSP demodulation matrices from the output of PAC fitting."""
2
+
3
+ from typing import Literal
4
+
5
+ import numpy as np
6
+ from dkist_processing_common.codecs.fits import fits_array_decoder
7
+ from dkist_processing_common.codecs.fits import fits_array_encoder
8
+ from dkist_processing_math.transform.binning import resize_arrays
9
+ from dkist_service_configuration.logging import logger
10
+ from sklearn.linear_model import RANSACRegressor
11
+ from sklearn.pipeline import Pipeline
12
+ from sklearn.pipeline import make_pipeline
13
+ from sklearn.preprocessing import PolynomialFeatures
14
+ from sklearn.preprocessing import RobustScaler
15
+
16
+ from dkist_processing_trend.models.tags import TrendTag
17
+ from dkist_processing_trend.tasks.trend_base import TrendArmTaskBase
18
+ from dkist_processing_trend.tasks.visp.visp_extract_beam import extract_visp_beam
19
+
20
+ __all__ = ["VispProcessDemodulationMatrices"]
21
+
22
+
23
+ class VispProcessDemodulationMatrices(TrendArmTaskBase):
24
+ """
25
+ Task class for producing full-frame demodulation ViSP matrices from the output of PAC fits.
26
+
27
+ This task mimics the functionality (and code) of the `InstrumentPolarizationCalibration <https://docs.dkist.nso.edu/projects/visp/en/stable/autoapi/dkist_processing_visp/tasks/instrument_polarization/index.html>`_
28
+ in the L1 ViSP pipeline; it first smooths the fit modulation matrices in the spatial dimension before upsampling to
29
+ the full-frame shape.
30
+
31
+ Parameters
32
+ ----------
33
+ arm_id
34
+ id of the instrument arm to operate on
35
+
36
+ recipe_run_id
37
+ id of the recipe run used to identify the workflow run this task is part of
38
+
39
+ workflow_name
40
+ name of the workflow to which this instance of the task belongs
41
+
42
+ workflow_version
43
+ version of the workflow to which this instance of the task belongs
44
+ """
45
+
46
+ record_provenance = True
47
+
48
+ def run(self):
49
+ """
50
+ Smooth and upsample raw best-fit demodulation matrices to match the full ViSP beam size.
51
+
52
+ First, the MODulation matrices are smoothed element-by-element along the spatial dimension by fitting with a polynomial.
53
+ This also has the effect of upsampling the spatial dimension to the full-beam size.
54
+ Finally, the wavelength dimension is upsampled to the full-beam size via interpolation.
55
+ """
56
+ for beam in [1, 2]:
57
+ full_beam_shape = self.get_full_beam_shape(beam)
58
+ logger.info(f"{full_beam_shape = }")
59
+ for fit_options in self.parameters.fit_options_list:
60
+ for instrument_options in self.parameters.instrument_processing_options:
61
+ log_str = f"fit options '{fit_options.name}', inst options '{instrument_options.name}', and {beam = }"
62
+ base_tags = [
63
+ TrendTag.intermediate(),
64
+ TrendTag.pac_fit_options(fit_options.name),
65
+ TrendTag.instrument_processing_options(instrument_options.name),
66
+ TrendTag.beam(beam),
67
+ TrendTag.arm_id(self.arm_id),
68
+ ]
69
+ with self.telemetry_span(f"Resampling demodulation matrices for {log_str}"):
70
+ raw_demod_matrices = next(
71
+ self.read(
72
+ tags=base_tags + [TrendTag.task_best_fit_demodulation_matrices()],
73
+ decoder=fits_array_decoder,
74
+ )
75
+ )
76
+
77
+ logger.info(f"Smoothing demodulation matrices for {log_str}")
78
+ smoothed_demod = self.smooth_demod_matrices(
79
+ demod_matrices=raw_demod_matrices,
80
+ fit_order=instrument_options.spatial_smoothing_fit_order,
81
+ num_full_slit_pos=full_beam_shape[1],
82
+ )
83
+
84
+ # Reshaping the demodulation matrix to get rid of unit length dimensions
85
+ logger.info(f"Resampling demodulation matrices for {log_str}")
86
+ final_demod_matrices = self.reshape_demod_matrices(
87
+ smoothed_demod, full_beam_shape
88
+ )
89
+ logger.info(
90
+ f"Shape of resampled demodulation matrices: {final_demod_matrices.shape}"
91
+ )
92
+
93
+ with self.telemetry_span(f"Writing demodulation matrices for {log_str}"):
94
+ self.write(
95
+ data=final_demod_matrices,
96
+ tags=base_tags + [TrendTag.task_processed_demodulation_matrices()],
97
+ encoder=fits_array_encoder,
98
+ )
99
+
100
+ def get_full_beam_shape(self, beam: Literal[1, 2]) -> tuple[int, int]:
101
+ """Read an INTERMEDIATE gain frame to measure the shape of a single beam array."""
102
+ gain_array = next(
103
+ self.read(
104
+ tags=[
105
+ TrendTag.intermediate(),
106
+ TrendTag.frame(),
107
+ TrendTag.arm_id(self.arm_id),
108
+ TrendTag.task_gain(),
109
+ ],
110
+ decoder=fits_array_decoder,
111
+ )
112
+ )
113
+ beam_array = extract_visp_beam(
114
+ gain_array, beam=beam, beam_border=self.parameters.visp_beam_border
115
+ )
116
+ return beam_array.shape
117
+
118
+ def smooth_demod_matrices(
119
+ self, demod_matrices: np.ndarray, fit_order: int, num_full_slit_pos: int
120
+ ) -> np.ndarray:
121
+ """
122
+ Smooth demodulation matrices in the spatial dimension.
123
+
124
+ The output will fully sample the spatial dimension so, as a side effect, this function also up-samples any data
125
+ that were binned spatially.
126
+
127
+ Smoothing is done using a RANSAC regression estimator to perform a polynomial fit with high resilience to outliers.
128
+
129
+ This method is taken directly from the ViSP L1 pipeline.
130
+ """
131
+ # We need to smooth the *modulation* (not DEmodulation) matrices to preserve the normalization of the Stokes-I
132
+ # values. linalg.pinv makes this easy
133
+ modulation_matrices = np.linalg.pinv(demod_matrices)
134
+
135
+ num_wave, num_binned_slit_pos, num_mod, num_stokes = modulation_matrices.shape
136
+
137
+ smoothed_mod = np.zeros((num_wave, num_full_slit_pos, num_mod, num_stokes))
138
+
139
+ # The binned abscissa is the "x" locations of the bins along the full spatial range
140
+ # The full abscissa is the "x" locations of all spatial pixels
141
+ # Add dummy dimensions because sklearn requires it
142
+ binned_abscissa = np.linspace(
143
+ start=0, stop=num_full_slit_pos, num=num_binned_slit_pos, endpoint=False
144
+ )[:, None]
145
+ full_abscissa = np.arange(num_full_slit_pos)[:, None]
146
+
147
+ model = self.build_RANSAC_model(fit_order=fit_order)
148
+ for w in range(num_wave):
149
+ for m in range(num_mod):
150
+ for s in range(num_stokes):
151
+ curve = modulation_matrices[w, :, m, s]
152
+
153
+ # Clean weirdo pixels
154
+ fill_value = np.nanmedian(curve)
155
+ curve[~np.isfinite(curve)] = fill_value
156
+
157
+ model.fit(binned_abscissa, curve)
158
+ fit_curve = model.predict(full_abscissa)
159
+ smoothed_mod[w, :, m, s] = fit_curve
160
+
161
+ # Now compute the inverse again to return the DEmodulation matrices
162
+ smoothed_demod = np.linalg.pinv(smoothed_mod)
163
+
164
+ return smoothed_demod
165
+
166
+ def build_RANSAC_model(self, fit_order: int) -> Pipeline:
167
+ """
168
+ Build a scikit-learn pipeline from a set of estimators.
169
+
170
+ This method is taken directly from the ViSP L1 pipeline.
171
+ """
172
+ # PolynomialFeatures casts the pipeline as a polynomial fit
173
+ # see https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.PolynomialFeatures.html#sklearn.preprocessing.PolynomialFeatures
174
+ poly_feature = PolynomialFeatures(degree=fit_order)
175
+
176
+ # RobustScaler is a scale factor that is robust to outliers
177
+ # see https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.RobustScaler.html#sklearn.preprocessing.RobustScaler
178
+ scaler = RobustScaler()
179
+
180
+ # The RANSAC regressor iteratively sub-samples the input data and constructs a model with this sub-sample.
181
+ # The method used allows it to be robust to outliers.
182
+ # see https://scikit-learn.org/stable/modules/linear_model.html#ransac-regression
183
+ RANSAC = RANSACRegressor(
184
+ min_samples=self.parameters.visp_polcal_demod_spatial_smooth_min_samples
185
+ )
186
+
187
+ return make_pipeline(poly_feature, scaler, RANSAC)
188
+
189
+ def reshape_demod_matrices(
190
+ self, demod_matrices: np.ndarray, full_beam_shape: tuple[int, int]
191
+ ) -> np.ndarray:
192
+ """Upsample demodulation matrices to match the full beam size.
193
+
194
+ Given an input set of demodulation matrices with shape ``(X', Y', 4, M)`` resample the output to shape
195
+ ``(X, Y, 4, M)``, where ``X'`` and ``Y'`` are the binned size of the beam FOV, ``X`` and ``Y`` are the full beam shape, and ``M`` is the
196
+ number of modulator states.
197
+
198
+ If only a single demodulation matrix was made then it is returned as a single array with shape ``(4, M)``.
199
+
200
+
201
+ This method is taken directly from the ViSP L1 pipeline.
202
+
203
+ Parameters
204
+ ----------
205
+ demod_matrices
206
+ A set of demodulation matrices with shape ``(X', Y', 4, M)``
207
+
208
+ Returns
209
+ -------
210
+ If ``X'`` or ``Y'`` > 1 then upsampled matrices that are the full beam size ``(X, Y, 4, M)``. If ``X' == Y' == 1`` then a single matric for the whole FOV with shape ``(4, M)``
211
+
212
+ """
213
+ if len(demod_matrices.shape) != 4:
214
+ raise ValueError(
215
+ f"Expected demodulation matrices to have 4 dimensions. Got shape {demod_matrices.shape}"
216
+ )
217
+
218
+ # The non-demodulation matrix part of the larger array
219
+ data_shape = demod_matrices.shape[:2]
220
+ # The shape of a single demodulation matrix
221
+ demod_shape = demod_matrices.shape[-2:]
222
+ logger.info(f"Demodulation FOV sampling shape: {data_shape}")
223
+ logger.info(f"Demodulation matrix shape: {demod_shape}")
224
+ if data_shape == (1, 1):
225
+ # A single modulation matrix can be used directly, so just return it after removing extraneous dimensions
226
+ logger.info(f"Single demodulation matrix detected")
227
+ return demod_matrices[0, 0, :, :]
228
+
229
+ target_shape = full_beam_shape + demod_shape
230
+ logger.info(f"Target full-frame demodulation shape: {target_shape}")
231
+ full_frame_matrices = next(
232
+ resize_arrays(
233
+ demod_matrices, target_shape, order=self.parameters.visp_polcal_demod_upsample_order
234
+ )
235
+ )
236
+ return full_frame_matrices