corgidrp 0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- corgidrp-0.1/LICENSE +27 -0
- corgidrp-0.1/MANIFEST.in +7 -0
- corgidrp-0.1/PKG-INFO +17 -0
- corgidrp-0.1/README.md +158 -0
- corgidrp-0.1/corgidrp/__init__.py +63 -0
- corgidrp-0.1/corgidrp/caldb.py +304 -0
- corgidrp-0.1/corgidrp/data.py +1013 -0
- corgidrp-0.1/corgidrp/detector.py +425 -0
- corgidrp-0.1/corgidrp/l1_to_l2a.py +289 -0
- corgidrp-0.1/corgidrp/l2a_to_l2b.py +317 -0
- corgidrp-0.1/corgidrp/l2b_to_l3.py +31 -0
- corgidrp-0.1/corgidrp/l3_to_l4.py +45 -0
- corgidrp-0.1/corgidrp/mocks.py +358 -0
- corgidrp-0.1/corgidrp/recipe_templates/l1_to_l2a_basic.json +35 -0
- corgidrp-0.1/corgidrp/recipe_templates/l1_to_l2a_eng.json +35 -0
- corgidrp-0.1/corgidrp/walker.py +168 -0
- corgidrp-0.1/corgidrp.egg-info/PKG-INFO +17 -0
- corgidrp-0.1/corgidrp.egg-info/SOURCES.txt +44 -0
- corgidrp-0.1/corgidrp.egg-info/dependency_links.txt +1 -0
- corgidrp-0.1/corgidrp.egg-info/requires.txt +5 -0
- corgidrp-0.1/corgidrp.egg-info/top_level.txt +1 -0
- corgidrp-0.1/requirements.txt +5 -0
- corgidrp-0.1/setup.cfg +4 -0
- corgidrp-0.1/setup.py +32 -0
- corgidrp-0.1/tests/test_add_phot_noise.py +35 -0
- corgidrp-0.1/tests/test_badpixel.py +105 -0
- corgidrp-0.1/tests/test_caldb.py +152 -0
- corgidrp-0.1/tests/test_cr_detection.py +647 -0
- corgidrp-0.1/tests/test_dark_sub.py +80 -0
- corgidrp-0.1/tests/test_data/example_L1_input.fits +10999 -0
- corgidrp-0.1/tests/test_data/metadata.yaml +92 -0
- corgidrp-0.1/tests/test_data/metadata_eng.yaml +61 -0
- corgidrp-0.1/tests/test_data/nonlin_sample.csv +41 -0
- corgidrp-0.1/tests/test_data/nonlin_sample.fits +0 -0
- corgidrp-0.1/tests/test_data_levels.py +34 -0
- corgidrp-0.1/tests/test_dataset.py +127 -0
- corgidrp-0.1/tests/test_desmear.py +63 -0
- corgidrp-0.1/tests/test_detector_params.py +33 -0
- corgidrp-0.1/tests/test_emgain_div.py +42 -0
- corgidrp-0.1/tests/test_err_dq.py +245 -0
- corgidrp-0.1/tests/test_flat_div.py +85 -0
- corgidrp-0.1/tests/test_frame_selection.py +121 -0
- corgidrp-0.1/tests/test_kgain.py +60 -0
- corgidrp-0.1/tests/test_non_linearity_correction.py +162 -0
- corgidrp-0.1/tests/test_prescan_sub.py +524 -0
- corgidrp-0.1/tests/test_walker.py +79 -0
corgidrp-0.1/LICENSE
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
Copyright (c) 2024, corgidrp Developers
|
|
2
|
+
All rights reserved.
|
|
3
|
+
|
|
4
|
+
Redistribution and use in source and binary forms, with or without
|
|
5
|
+
modification, are permitted provided that the following conditions are met:
|
|
6
|
+
|
|
7
|
+
1. Redistributions of source code must retain the above copyright notice, this
|
|
8
|
+
list of conditions and the following disclaimer.
|
|
9
|
+
|
|
10
|
+
2. Redistributions in binary form must reproduce the above copyright notice,
|
|
11
|
+
this list of conditions and the following disclaimer in the documentation
|
|
12
|
+
and/or other materials provided with the distribution.
|
|
13
|
+
|
|
14
|
+
3. Neither the name of the copyright holder nor the names of its contributors
|
|
15
|
+
may be used to endorse or promote products derived from this software without
|
|
16
|
+
specific prior written permission.
|
|
17
|
+
|
|
18
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
|
|
19
|
+
ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
|
|
20
|
+
WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
21
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
22
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
23
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
24
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
25
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
26
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
27
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
corgidrp-0.1/MANIFEST.in
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
include requirements.txt
|
|
2
|
+
include corgidrp/recipe_templates/*.json
|
|
3
|
+
include tests/test_data/example_L1_input.fits
|
|
4
|
+
include tests/test_data/metadata.yaml
|
|
5
|
+
include tests/test_data/metadata_eng.yaml
|
|
6
|
+
include tests/test_data/nonlin_sample.csv
|
|
7
|
+
include tests/test_data/nonlin_sample.fits
|
corgidrp-0.1/PKG-INFO
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: corgidrp
|
|
3
|
+
Version: 0.1
|
|
4
|
+
Summary: (Roman Space Telescope) CORonaGraph Instrument Data Reduction Pipeline
|
|
5
|
+
Home-page: https://github.com/roman-corgi/corgidrp
|
|
6
|
+
Author: Roman Coronagraph Instrument CPP
|
|
7
|
+
License: BSD
|
|
8
|
+
Keywords: Roman Space Telescope Exoplanets Astronomy
|
|
9
|
+
Classifier: Intended Audience :: Science/Research
|
|
10
|
+
Classifier: Topic :: Scientific/Engineering :: Astronomy
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Requires-Dist: numpy
|
|
14
|
+
Requires-Dist: astropy
|
|
15
|
+
Requires-Dist: pandas
|
|
16
|
+
Requires-Dist: pytest
|
|
17
|
+
Requires-Dist: scipy
|
corgidrp-0.1/README.md
ADDED
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
# corgidrp: CORonaGraph Instrument Data Reduction Pipeline
|
|
2
|
+
This is the data reduction pipeline for the Nancy Grace Roman Space Telescope Coronagraph Instrument
|
|
3
|
+
|
|
4
|
+

|
|
5
|
+
|
|
6
|
+
## Install
|
|
7
|
+
As the code is very much still in development, clone this repository, enter the top-level folder, and run the following command:
|
|
8
|
+
```
|
|
9
|
+
pip install -e .
|
|
10
|
+
```
|
|
11
|
+
Then you can import `corgidrp` like any other python package!
|
|
12
|
+
|
|
13
|
+
The installation will create a configuration folder in your home directory called `.corgidrp`.
|
|
14
|
+
That configuration directory will be used to locate things on your computer such as the location of the calibration database and the pipeline configuration file. The configuration files stores setting such as whether to track each individual error term added to the noise.
|
|
15
|
+
|
|
16
|
+
## How to Contribute
|
|
17
|
+
|
|
18
|
+
We encourage you to chat with Jason, Max, and Marie (e.g., on Slack) to discuss what to do before you get started. Brainstorming
|
|
19
|
+
about how to implement something is a very good use of time and makes sure you aren't going down the wrong path.
|
|
20
|
+
Contact Jason is you have any questions on how to get started on programming details (e.g., git).
|
|
21
|
+
|
|
22
|
+
Below is a quick tutorial that outlines the general contribution process.
|
|
23
|
+
|
|
24
|
+
### The basics of getting setup
|
|
25
|
+
#### Find a task to work on
|
|
26
|
+
Check out the [Github issues page](https://github.com/roman-corgi/corgidrp/issues) for tasks that need attention. Alternatively, contact Jason (@semaphoreP). _Make sure to tag yourself on the issue and mention in the comments if you start working on it._
|
|
27
|
+
|
|
28
|
+
#### Clone the git repository and install
|
|
29
|
+
|
|
30
|
+
See install instructions above. Contact Jason (@semaphoreP) if you need write access to push changes as you make edits. If you do not have write access to the repository, you can still contribute by creating a fork of the repository under your own GitHub user. See here for details: https://docs.github.com/en/get-started/quickstart/fork-a-repo. You can then use the same commands given below, but just replace `roman-corgi` with your own GitHub username. If you fork the repository, you will need to make sure that your fork is up to date with the main repository (https://docs.github.com/en/pull-requests/collaborating-with-pull-requests/working-with-forks/syncing-a-fork).
|
|
31
|
+
|
|
32
|
+
To quickly get up an running with the repository, execute the following commands in a terminal (or command prompt - you will need to have the git executable installed on your system):
|
|
33
|
+
```
|
|
34
|
+
> git clone https://github.com/roman-corgi/corgidrp.git
|
|
35
|
+
> cd corgidrp
|
|
36
|
+
> pip install -e .
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
##### Make a new git branch to work on
|
|
40
|
+
|
|
41
|
+
You will create a "feature branch" so you can develop your feature without impacting other people's code. Let's say I'm working on dark subtraction. I could create a feature branch and switch to it like this
|
|
42
|
+
|
|
43
|
+
```
|
|
44
|
+
> git branch dark-sub
|
|
45
|
+
> git checkout dark-sub
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
### Write your pipeline step
|
|
49
|
+
|
|
50
|
+
In `corgidrp`, each pipeline step is a function, that is contained in one of the lX_to_lY.py files (where X and Y are various data levels).
|
|
51
|
+
Think about how your feature can be implemented as a function that takes in some data and outputs processed data. Please see below for some
|
|
52
|
+
`corgidrp` design principles.
|
|
53
|
+
|
|
54
|
+
All functions should follow this example:
|
|
55
|
+
|
|
56
|
+
```
|
|
57
|
+
def example_step(dataset, calib_data, tuneable_arg=1, another_arg="test"):
|
|
58
|
+
"""
|
|
59
|
+
Function docstrings are required and should follow Google style docstrings.
|
|
60
|
+
We will not demonstrate it here for brevity.
|
|
61
|
+
"""
|
|
62
|
+
# unless you don't alter the input dataset at all, plan to make a copy of the data
|
|
63
|
+
# this is to ensure functions are reproducible
|
|
64
|
+
processed_dataset = input_dataset.copy()
|
|
65
|
+
|
|
66
|
+
### Your code here that does the real work
|
|
67
|
+
# here is a convience field to grab all the data in a dataset
|
|
68
|
+
all_data = processed_dataset.all_data
|
|
69
|
+
### End of your code that does the real work
|
|
70
|
+
|
|
71
|
+
# update the header of the new dataset with your processing step
|
|
72
|
+
history_msg = "I did an example step"
|
|
73
|
+
# update the output dataset with the new data and update the history
|
|
74
|
+
processed_dataset.update_after_processing_step(history_msg, new_all_data=all_data)
|
|
75
|
+
|
|
76
|
+
# return the processed data
|
|
77
|
+
return processed_dataset
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
Inside the function can be nearly anything you want, but the function signature and start/end of the function should follow a few rules.
|
|
81
|
+
|
|
82
|
+
* Each function should include a docstring that descibes what the function is doing, what the inputs (including units if appropriate) are and what the outputs (also with units). The dosctrings should be [goggle style docstrings](https://sphinxcontrib-napoleon.readthedocs.io/en/latest/example_google.html).
|
|
83
|
+
* The input dataset should always be the first input
|
|
84
|
+
* Additional arguments and keywords exist only if you need them--many relevant parameters might already by in Dataset headers. A pipeline step can only have a single argument (the input dataset) if needed.
|
|
85
|
+
* All additional function arguments/keywords should only consist of the following types: int, float, str, or a class defined in corgidrp.Data.
|
|
86
|
+
* (Long explaination for the curious: The reason for this is that pipeline steps can be written out as text files. Int/float/str are easily represented succinctly by textfiles. All classes in corgidrp.Data can be created simply by passing in a filepath. Therefore, all pipeline steps have easily recordable arguments for easy reproducibility.)
|
|
87
|
+
* The first line of the function generally should be creating a copy of the dataset (which will be the output dataset). This way, the output dataset is not the same instance as the input dataset. This will make it easier to ensure reproducibility.
|
|
88
|
+
* The function should always end with updating the header and (typically) the data of the output dataset. The history of running this pipeline step should be recorded in the header.
|
|
89
|
+
|
|
90
|
+
You can check out `corgidrp.l2a_to_l2b.dark_subtraction` function as an example of a basic pipeline step.
|
|
91
|
+
|
|
92
|
+
### Write a unit test to debug your pipeline step
|
|
93
|
+
|
|
94
|
+
We are required to write tests to verify the functionality of the code. Instead of seeing this as an extra chore, I encourage you to write unit tests to be your debug script to get your code working (this is called "test-driven development").
|
|
95
|
+
|
|
96
|
+
All tests are stored in the `tests` folder and each test is a function that starts with `test_`. See `tests/test_dark_sub.py` as an example. Within each test, you will likely need to simulate some mock data, run it through your function you wrote, and verify it ran correctly using assert statements. Your tests should cover the primary use cases of your code, and check that the function outputs what you expect. You do not need high fidelity data for your test: focus on making sure the data is in the correct format as real data, and less on making sure the data values are simulated to high fidelity (see the examples in the `mocks.py` module).
|
|
97
|
+
|
|
98
|
+
Importantly, these tests will allow code reviewers to test and understand your code. We will also run these tests in an automated test suite (continuous integration) with the pipeline to verify the functions continue to work (e.g., as dependencies change).
|
|
99
|
+
|
|
100
|
+
#### How to run your tests locally
|
|
101
|
+
|
|
102
|
+
You can either run tests individually yourself (to debug individual tests) or run the entire test suite to make sure you didn't break anything.
|
|
103
|
+
|
|
104
|
+
To run an individual test, call the test function you want to test at the bottom of its `test_*.py` script. Then, you just need to run the `test_*.py` script. See `tests/test_dark_sub.py` for an example.
|
|
105
|
+
|
|
106
|
+
To run all the tests in the test suite, go to the base corgidrp folder in a terminal and run the `pytest` command.
|
|
107
|
+
|
|
108
|
+
### Linting
|
|
109
|
+
|
|
110
|
+
In addition to unit tests, your code will need to pass a static analysis before being merged. `corgidrp` currently runs a subset of flake8 tests, which you can replicate on your local system by running:
|
|
111
|
+
|
|
112
|
+
```
|
|
113
|
+
flake8 . --count --select=E9,F63,F7,F82,DCO020,DCO021,DCO022,DCO023,DCO024,DCO030,DCO031,DCO032,DCO060,DCO061,DCO062,DCO063,DCO064,DCO065 --show-source --statistics
|
|
114
|
+
```
|
|
115
|
+
from the top-level directory of the repository. In order to run these tests you will need to have `flake8` and `flake8-docstrings-complete` installed (both are pip-installable). Note that the test subset may be updated in the future. To see the current set of tests being applied, look in the continuous integration GitHub action, located in the repository in file `.github/workflows/python-app.yml`.
|
|
116
|
+
|
|
117
|
+
### Create a pull request to merge your changes
|
|
118
|
+
Before creating a pull request, review the design Principles below. Use the Github pull request feature to request that your changes get merged into the `main` branch. Assign Jason/Max to be your reviewers. Your changes will be reviewed, and possibly some edits will be requested. You can simply make additional pushes to your branch to update the pull request with those changes (you don't need to delete the PR and make a new one). When the branch is satisfactory, we will pull your changes in. When preparing your pull request, you may find it helpful to follow this checklist:
|
|
119
|
+
|
|
120
|
+
- [ ] If working with a fork, sync your fork to the upstream repository
|
|
121
|
+
- [ ] Ensure that your pull can be merged automatically by merging main onto the branch you wish to pull from
|
|
122
|
+
- [ ] Ensure that all of your new additions have properly formatted docstrings (this will happen automatically if you run the `flake8` command given above
|
|
123
|
+
- [ ] Ensure that all of the commits going in to your pull have informative messages
|
|
124
|
+
- [ ] Ensure that all unit tests pass locally, and that you have provided new unit tests for all new functionality in your pull
|
|
125
|
+
- [ ] Create a new pull request, fully describing all changes/additions
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
## Overarching Design Principles
|
|
129
|
+
* Minimize the use of external packages, unless it saves us a lot of time. If you need to use something external, default to well-established and maintained packages, such as `numpy`, `scipy` or `Astropy`. If you think you need to use something else, please check with Jason and Max.
|
|
130
|
+
* Minimize the use of new classes, with the exception of new classes that extend the existing data framework.
|
|
131
|
+
* The python module files (i.e. the *.py files) should typically hold on the order of 5-10 different functions. You can create new ones if need be, but the new files should be general enough in topic to encapsulate other future functions as well.
|
|
132
|
+
* All the image data in the Dataset and Image objects should be saved as standard numpy arrays. Other types of arrays (such as masked arrays) can be used as intermediate products within a function.
|
|
133
|
+
* Keep things simple
|
|
134
|
+
* Use _descriptive_ variable names **always**.
|
|
135
|
+
* Comments should be used to describe a section of the code where it's not immediately obvious what the code is doing. Using descriptive variable names will minimize the amount of comments required.
|
|
136
|
+
|
|
137
|
+
## FAQ
|
|
138
|
+
|
|
139
|
+
* Does my pipeline function need to save files?
|
|
140
|
+
* Files will be saved by a higher level pipeline code. As long as you output an object that's an instance of a `corgidrp.Data` class, it will have a `save()` function that will be used.
|
|
141
|
+
* Can I create new data classes?
|
|
142
|
+
* Yes, you can feel free to make new data classes. Generally, they should be a subclass of the `Image` class, and you can look at the `Dark` class as an example. Each calibration type should have its own `Image` subclass defined. Talk with Jason and Max to discuss how your class should be implemented!
|
|
143
|
+
|
|
144
|
+
* What python version should I develop in?
|
|
145
|
+
* Python 3.12
|
|
146
|
+
|
|
147
|
+
* How should I treat hard-coded variables
|
|
148
|
+
* If a variable value is extremely unlikely to change AND is only required by one module, it can be hard coded inside that module.
|
|
149
|
+
* If it is unlikely to change but will need to be referenced by multiple modules, it should be added to the central constants file (name TBD).
|
|
150
|
+
* If it is a setting about pipeline behavior that may change, it should be implemented by a config file (examples are location of the calibration database and whether to save individual error terms in the output FITS files).
|
|
151
|
+
|
|
152
|
+
* Where should I store computed variables so they can be referenced later in the pipeline?
|
|
153
|
+
* If possible, in the header of the dataset being processed or in a hew HDU extension
|
|
154
|
+
* If not possible, then let's chat!
|
|
155
|
+
|
|
156
|
+
* Where do I save FITS files or other data files I need to use for my tests?
|
|
157
|
+
* Auxiliary data to run tests should be stored in the tests/test_data folder
|
|
158
|
+
* If they are larger than 1 MB, they should be stored using `git lfs`. Ask Jason about setting up git lfs (as of writing, we have not set up git lfs yet).
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
import configparser
|
|
2
|
+
import os
|
|
3
|
+
import pathlib
|
|
4
|
+
import configparser
|
|
5
|
+
|
|
6
|
+
__version__ = "0.1"
|
|
7
|
+
version = __version__ # temporary backwards compatability
|
|
8
|
+
|
|
9
|
+
#### Create a configuration file for the corgidrp if it doesn't exist.
|
|
10
|
+
def create_config_dir():
|
|
11
|
+
"""
|
|
12
|
+
Checks if the default .corgidrp directory exists, and if not, it sets it up
|
|
13
|
+
"""
|
|
14
|
+
homedir = pathlib.Path.home()
|
|
15
|
+
config_folder = os.path.join(homedir, ".corgidrp")
|
|
16
|
+
# replace legacy file with folder if needed
|
|
17
|
+
if os.path.isfile(config_folder):
|
|
18
|
+
oldconfig = configparser.ConfigParser()
|
|
19
|
+
oldconfig.read(config_folder)
|
|
20
|
+
os.remove(config_folder)
|
|
21
|
+
else:
|
|
22
|
+
oldconfig = None
|
|
23
|
+
|
|
24
|
+
# make folder if doesn't exist
|
|
25
|
+
if not os.path.isdir(config_folder):
|
|
26
|
+
os.mkdir(config_folder)
|
|
27
|
+
|
|
28
|
+
# make default calibrations folder
|
|
29
|
+
default_cal_dir = os.path.join(config_folder, "default_calibs")
|
|
30
|
+
if not os.path.exists(default_cal_dir):
|
|
31
|
+
os.mkdir(default_cal_dir)
|
|
32
|
+
|
|
33
|
+
# write config
|
|
34
|
+
config_filepath = os.path.join(config_folder, "corgidrp.cfg")
|
|
35
|
+
if not os.path.exists(config_filepath):
|
|
36
|
+
config = configparser.ConfigParser()
|
|
37
|
+
config["PATH"] = {}
|
|
38
|
+
config["PATH"]["caldb"] = os.path.join(config_folder, "corgidrp_caldb.csv") # location to store caldb
|
|
39
|
+
config["PATH"]["default_calibs"] = default_cal_dir
|
|
40
|
+
config["DATA"] = {}
|
|
41
|
+
config["DATA"]["track_individual_errors"] = "False"
|
|
42
|
+
# overwrite with old settings if needed
|
|
43
|
+
if oldconfig is not None:
|
|
44
|
+
config["PATH"]["caldb"] = oldconfig["PATH"]["caldb"]
|
|
45
|
+
|
|
46
|
+
with open(config_filepath, 'w') as f:
|
|
47
|
+
config.write(f)
|
|
48
|
+
|
|
49
|
+
print("corgidrp: Configuration file written to {0}. Please edit if you want things stored in different locations.".format(config_filepath))
|
|
50
|
+
create_config_dir()
|
|
51
|
+
|
|
52
|
+
_bool_map = {"true" : True, "false" : False}
|
|
53
|
+
|
|
54
|
+
# borrowed from the kpicdrp caldb
|
|
55
|
+
# load in default caldbs based on configuration file
|
|
56
|
+
config_filepath = os.path.join(pathlib.Path.home(), ".corgidrp", "corgidrp.cfg")
|
|
57
|
+
config = configparser.ConfigParser()
|
|
58
|
+
config.read(config_filepath)
|
|
59
|
+
|
|
60
|
+
## pipeline settings
|
|
61
|
+
caldb_filepath = config.get("PATH", "caldb", fallback=None)
|
|
62
|
+
default_cal_dir = config.get("PATH", "default_calibs", fallback=None)
|
|
63
|
+
track_individual_errors = _bool_map[config.get("DATA", "track_individual_errors").lower()]
|
|
@@ -0,0 +1,304 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Calibration tracking system. Modified from kpicdrp caldb implmentation (Copyright (c) 2024, KPIC Team)
|
|
3
|
+
"""
|
|
4
|
+
import os
|
|
5
|
+
import numpy as np
|
|
6
|
+
import pandas as pd
|
|
7
|
+
import corgidrp
|
|
8
|
+
import corgidrp.data as data
|
|
9
|
+
import astropy.time as time
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
column_names = [
|
|
13
|
+
"Filepath",
|
|
14
|
+
"Type",
|
|
15
|
+
"MJD",
|
|
16
|
+
"EXPTIME",
|
|
17
|
+
"Files Used",
|
|
18
|
+
"Date Created",
|
|
19
|
+
"Hash",
|
|
20
|
+
"DRPVERSN",
|
|
21
|
+
"OBSID",
|
|
22
|
+
"NAXIS1",
|
|
23
|
+
"NAXIS2",
|
|
24
|
+
"OPMODE",
|
|
25
|
+
"CMDGAIN",
|
|
26
|
+
"EXCAMT",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
labels = {data.Dark: "Dark",
|
|
30
|
+
data.NonLinearityCalibration: "NonLinearityCalibration",
|
|
31
|
+
data.BadPixelMap: "BadPixelMap",
|
|
32
|
+
data.KGain : "KGain",
|
|
33
|
+
data.DetectorParams : "DetectorParams"}
|
|
34
|
+
|
|
35
|
+
class CalDB:
|
|
36
|
+
"""
|
|
37
|
+
Database for tracking calibration files saved to disk. Modified from the kpicdrp version
|
|
38
|
+
|
|
39
|
+
Note that database is not parallelism-safe, but should be ok in most cases.
|
|
40
|
+
(Jason: look at using posix_ipc to guarantee thread safety if we really need it)
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
filepath (str): [optional] filepath to a CSV file with an existing database
|
|
44
|
+
|
|
45
|
+
Fields:
|
|
46
|
+
columns (list): column names of dataframe
|
|
47
|
+
filepath(str): full filepath to data
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
def __init__(self, filepath=""):
|
|
51
|
+
"""
|
|
52
|
+
Args:
|
|
53
|
+
filepath (str): [optional] filepath to a CSV file with an existing database
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
# If filepath is not passed in, use the default (majority of case)
|
|
57
|
+
if len(filepath) == 0:
|
|
58
|
+
self.filepath = corgidrp.caldb_filepath
|
|
59
|
+
else:
|
|
60
|
+
# possibly edge case where we want to use specialized caldb
|
|
61
|
+
self.filepath = filepath
|
|
62
|
+
|
|
63
|
+
# if database does't exist, create a blank one
|
|
64
|
+
if not os.path.exists(self.filepath):
|
|
65
|
+
# new database
|
|
66
|
+
self.columns = column_names
|
|
67
|
+
self._db = pd.DataFrame(columns=self.columns)
|
|
68
|
+
self.save()
|
|
69
|
+
else:
|
|
70
|
+
# already a database exists
|
|
71
|
+
self.load()
|
|
72
|
+
self.columns = list(self._db.columns.values)
|
|
73
|
+
|
|
74
|
+
def load(self):
|
|
75
|
+
"""
|
|
76
|
+
Load/update db from filepath
|
|
77
|
+
"""
|
|
78
|
+
self._db = pd.read_csv(self.filepath)
|
|
79
|
+
|
|
80
|
+
def save(self):
|
|
81
|
+
"""
|
|
82
|
+
Save file without numbered index to disk with user specified filepath as a CSV file
|
|
83
|
+
"""
|
|
84
|
+
self._db.to_csv(self.filepath, index=False)
|
|
85
|
+
|
|
86
|
+
def _get_values_from_entry(self, entry, is_calib=True):
|
|
87
|
+
"""
|
|
88
|
+
Extract the properties from this data entry to ingest them into the database
|
|
89
|
+
|
|
90
|
+
Args:
|
|
91
|
+
entry (corgidrp.data.Image subclass): calibration frame to add to the database
|
|
92
|
+
is_calib (bool): is a calibration frame. if Not, it won't look up filetype.
|
|
93
|
+
Only used in get_calib() to grab metadata for science frames
|
|
94
|
+
|
|
95
|
+
Returns:
|
|
96
|
+
tuple:
|
|
97
|
+
row (list):
|
|
98
|
+
List of data entry properties
|
|
99
|
+
row_dict (dict):
|
|
100
|
+
Dictionary of data entry properties keyed by column names
|
|
101
|
+
|
|
102
|
+
"""
|
|
103
|
+
filepath = os.path.abspath(entry.filepath)
|
|
104
|
+
if is_calib:
|
|
105
|
+
datatype = labels[entry.__class__] # get the database str representation
|
|
106
|
+
else:
|
|
107
|
+
datatype = "Sci"
|
|
108
|
+
mjd = time.Time(entry.ext_hdr["SCTSRT"]).mjd
|
|
109
|
+
exptime = entry.ext_hdr["EXPTIME"]
|
|
110
|
+
|
|
111
|
+
# check if this exists. will be a keyword written by corgidrp
|
|
112
|
+
if "DRPNFILE" in entry.ext_hdr:
|
|
113
|
+
files_used = entry.ext_hdr["DRPNFILE"]
|
|
114
|
+
else:
|
|
115
|
+
files_used = 0
|
|
116
|
+
|
|
117
|
+
if "DRPCTIME" in entry.ext_hdr:
|
|
118
|
+
date_created = time.Time(entry.ext_hdr["DRPCTIME"]).mjd
|
|
119
|
+
else:
|
|
120
|
+
date_created = -1
|
|
121
|
+
|
|
122
|
+
if "DRPVERSN" in entry.ext_hdr:
|
|
123
|
+
drp_version = entry.ext_hdr["DRPVERSN"]
|
|
124
|
+
else:
|
|
125
|
+
drp_version = ""
|
|
126
|
+
|
|
127
|
+
obsid = entry.pri_hdr["OBSID"]
|
|
128
|
+
|
|
129
|
+
hash_val = entry.get_hash()
|
|
130
|
+
|
|
131
|
+
# this only works for 2D images. may need to adapt for non-2D calibration frames
|
|
132
|
+
naxis1 = entry.data.shape[-1]
|
|
133
|
+
naxis2 = entry.data.shape[-2]
|
|
134
|
+
|
|
135
|
+
row = [
|
|
136
|
+
filepath,
|
|
137
|
+
datatype,
|
|
138
|
+
mjd,
|
|
139
|
+
exptime,
|
|
140
|
+
files_used,
|
|
141
|
+
date_created,
|
|
142
|
+
hash_val,
|
|
143
|
+
drp_version,
|
|
144
|
+
obsid,
|
|
145
|
+
naxis1,
|
|
146
|
+
naxis2,
|
|
147
|
+
]
|
|
148
|
+
|
|
149
|
+
# rest are ext_hdr keys we can copy
|
|
150
|
+
start_index = len(row)
|
|
151
|
+
for i in range(start_index, len(self.columns)):
|
|
152
|
+
row.append(entry.ext_hdr[self.columns[i]]) # add value staright from header
|
|
153
|
+
|
|
154
|
+
row_dict = {}
|
|
155
|
+
for key, val in zip(self.columns, row):
|
|
156
|
+
row_dict[key] = val
|
|
157
|
+
|
|
158
|
+
return row, row_dict
|
|
159
|
+
|
|
160
|
+
def create_entry(self, entry, to_disk=True):
|
|
161
|
+
"""
|
|
162
|
+
Add a new entry to or update an existing one in the database. Note that function by default will load and save db to disk
|
|
163
|
+
|
|
164
|
+
Args:
|
|
165
|
+
entry (corgidrp.data.Image subclass): calibration frame to add to the database
|
|
166
|
+
to_disk (bool): True by default, will update DB from disk before adding entry and saving it back to disk
|
|
167
|
+
"""
|
|
168
|
+
new_row, row_dict = self._get_values_from_entry(entry)
|
|
169
|
+
|
|
170
|
+
# update database from disk in case anything changed
|
|
171
|
+
if to_disk:
|
|
172
|
+
self.load()
|
|
173
|
+
|
|
174
|
+
# use filepath as key to see if it's already in database
|
|
175
|
+
if row_dict["Filepath"] in self._db.values:
|
|
176
|
+
row_index = self._db[
|
|
177
|
+
self._db["Filepath"] == row_dict["Filepath"]
|
|
178
|
+
].index.values
|
|
179
|
+
self._db.loc[row_index, self.columns] = new_row
|
|
180
|
+
# otherwise create new entry
|
|
181
|
+
else:
|
|
182
|
+
new_entry = pd.DataFrame([new_row], columns=self.columns)
|
|
183
|
+
self._db = pd.concat([self._db, new_entry], ignore_index=True)
|
|
184
|
+
|
|
185
|
+
# save to disk to update changes
|
|
186
|
+
if to_disk:
|
|
187
|
+
self.save()
|
|
188
|
+
|
|
189
|
+
def remove_entry(self, entry, to_disk=True):
|
|
190
|
+
"""
|
|
191
|
+
Remove an entry from the database. Removes the entire row
|
|
192
|
+
|
|
193
|
+
Args:
|
|
194
|
+
entry (corgidrp.data.Image subclass): calibration frame to add to the database
|
|
195
|
+
to_disk (bool): True by default, will update DB from disk before adding entry and saving it back to disk
|
|
196
|
+
"""
|
|
197
|
+
new_row, row_dict = self._get_values_from_entry(entry)
|
|
198
|
+
|
|
199
|
+
# update database from disk in case anything changed
|
|
200
|
+
if to_disk:
|
|
201
|
+
self.load()
|
|
202
|
+
|
|
203
|
+
if row_dict["Filepath"] in self._db.values:
|
|
204
|
+
entry_index = self._db[
|
|
205
|
+
self._db["Filepath"] == row_dict["Filepath"]
|
|
206
|
+
].index.values
|
|
207
|
+
self._db = self._db.drop(self._db.index[entry_index])
|
|
208
|
+
self._db = self._db.reset_index(drop=True)
|
|
209
|
+
else:
|
|
210
|
+
raise ValueError("No filepath found so could not remove.")
|
|
211
|
+
|
|
212
|
+
# save to disk to update changes
|
|
213
|
+
if to_disk:
|
|
214
|
+
self.save()
|
|
215
|
+
|
|
216
|
+
def get_calib(self, frame, dtype, to_disk=True):
|
|
217
|
+
"""
|
|
218
|
+
Outputs the best calibration file of the given type for the input sciene frame.
|
|
219
|
+
|
|
220
|
+
Args:
|
|
221
|
+
frame (corgidrp.data.Image): an image frame to request a calibratio for
|
|
222
|
+
dtype (corgidrp.data Class): for example: corgidrp.data.Dark (TODO: document the entire list of options)
|
|
223
|
+
to_disk (bool): True by default, will update DB from disk before matching
|
|
224
|
+
|
|
225
|
+
Returns:
|
|
226
|
+
corgidrp.data.*: an instance of the appropriate calibration type (Exact type depends on calibration type)
|
|
227
|
+
"""
|
|
228
|
+
if dtype not in labels:
|
|
229
|
+
raise ValueError(
|
|
230
|
+
"Requested calibration dtype of {0} not a valid option".format(dtype)
|
|
231
|
+
)
|
|
232
|
+
dtype_label = labels[dtype]
|
|
233
|
+
|
|
234
|
+
# get values for this science frame
|
|
235
|
+
_, frame_dict = self._get_values_from_entry(frame, is_calib=False)
|
|
236
|
+
|
|
237
|
+
# update database from disk in case anything changed
|
|
238
|
+
if to_disk:
|
|
239
|
+
self.load()
|
|
240
|
+
|
|
241
|
+
# downselect to only calibs of this type
|
|
242
|
+
calibdf = self._db[self._db["Type"] == dtype_label]
|
|
243
|
+
|
|
244
|
+
if dtype_label in ["Dark"]:
|
|
245
|
+
# general selection criteria for 2D image frames. Can use different selection criteria for different dtypes
|
|
246
|
+
options = calibdf.loc[
|
|
247
|
+
(
|
|
248
|
+
(calibdf["EXPTIME"] == frame_dict["EXPTIME"])
|
|
249
|
+
& (calibdf["NAXIS1"] == frame_dict["NAXIS1"])
|
|
250
|
+
& (calibdf["NAXIS2"] == frame_dict["NAXIS2"])
|
|
251
|
+
)
|
|
252
|
+
]
|
|
253
|
+
else:
|
|
254
|
+
options = calibdf
|
|
255
|
+
|
|
256
|
+
# select the one closest in time
|
|
257
|
+
result_index = np.abs(options["MJD"] - frame_dict["MJD"]).argmin()
|
|
258
|
+
calib_filepath = options.iloc[result_index, 0]
|
|
259
|
+
|
|
260
|
+
# load the object from disk and return it
|
|
261
|
+
return dtype(calib_filepath)
|
|
262
|
+
|
|
263
|
+
def scan_dir_for_new_entries(self, filedir, look_in_subfolders=True, to_disk=True):
|
|
264
|
+
"""
|
|
265
|
+
Scan a folder and subfolder for calibration files and add them all to the caldb
|
|
266
|
+
|
|
267
|
+
Args:
|
|
268
|
+
filedir (str): path to folder to scan (includes all subfolders by default)
|
|
269
|
+
look_in_subfolders (bool): whether to look in subfolders for files. True by default
|
|
270
|
+
to_disk (bool): True by default, will update DB from disk before adding entry and saving it back to disk
|
|
271
|
+
"""
|
|
272
|
+
calib_frames = []
|
|
273
|
+
# walk the directory to find all the calibration files
|
|
274
|
+
for dirpath, subfolders, filenames in os.walk(filedir):
|
|
275
|
+
for filename in filenames:
|
|
276
|
+
# hard coded check only for files that end in .fits
|
|
277
|
+
if filename[-5:] != ".fits":
|
|
278
|
+
continue
|
|
279
|
+
|
|
280
|
+
filepath = os.path.join(dirpath, filename)
|
|
281
|
+
frame = data.autoload(filepath)
|
|
282
|
+
|
|
283
|
+
# check what class it has been loaded as. only save frames that fall into calibration classes
|
|
284
|
+
if frame.__class__ in labels:
|
|
285
|
+
calib_frames.append(frame)
|
|
286
|
+
|
|
287
|
+
# the first iteration looks in the basedir
|
|
288
|
+
# if we don't wnat to look in subdirs now, we should break
|
|
289
|
+
if not look_in_subfolders:
|
|
290
|
+
break
|
|
291
|
+
|
|
292
|
+
# load all these files into the caldb
|
|
293
|
+
for calib_frame in calib_frames:
|
|
294
|
+
self.create_entry(calib_frame, to_disk=to_disk)
|
|
295
|
+
|
|
296
|
+
### Create set of default calibrations
|
|
297
|
+
# Add default detector_params calibration file if it doesn't exist
|
|
298
|
+
if not os.path.exists(os.path.join(corgidrp.default_cal_dir, "DetectorParams_2023-11-01T00:00:00.000.fits")):
|
|
299
|
+
default_detparams = data.DetectorParams({}, date_valid=time.Time("2023-11-01 00:00:00", scale='utc'))
|
|
300
|
+
default_detparams.save(filedir=corgidrp.default_cal_dir)
|
|
301
|
+
|
|
302
|
+
# add default caldb entries
|
|
303
|
+
default_caldb = CalDB()
|
|
304
|
+
default_caldb.scan_dir_for_new_entries(corgidrp.default_cal_dir)
|