corgidrp 0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. corgidrp-0.1/LICENSE +27 -0
  2. corgidrp-0.1/MANIFEST.in +7 -0
  3. corgidrp-0.1/PKG-INFO +17 -0
  4. corgidrp-0.1/README.md +158 -0
  5. corgidrp-0.1/corgidrp/__init__.py +63 -0
  6. corgidrp-0.1/corgidrp/caldb.py +304 -0
  7. corgidrp-0.1/corgidrp/data.py +1013 -0
  8. corgidrp-0.1/corgidrp/detector.py +425 -0
  9. corgidrp-0.1/corgidrp/l1_to_l2a.py +289 -0
  10. corgidrp-0.1/corgidrp/l2a_to_l2b.py +317 -0
  11. corgidrp-0.1/corgidrp/l2b_to_l3.py +31 -0
  12. corgidrp-0.1/corgidrp/l3_to_l4.py +45 -0
  13. corgidrp-0.1/corgidrp/mocks.py +358 -0
  14. corgidrp-0.1/corgidrp/recipe_templates/l1_to_l2a_basic.json +35 -0
  15. corgidrp-0.1/corgidrp/recipe_templates/l1_to_l2a_eng.json +35 -0
  16. corgidrp-0.1/corgidrp/walker.py +168 -0
  17. corgidrp-0.1/corgidrp.egg-info/PKG-INFO +17 -0
  18. corgidrp-0.1/corgidrp.egg-info/SOURCES.txt +44 -0
  19. corgidrp-0.1/corgidrp.egg-info/dependency_links.txt +1 -0
  20. corgidrp-0.1/corgidrp.egg-info/requires.txt +5 -0
  21. corgidrp-0.1/corgidrp.egg-info/top_level.txt +1 -0
  22. corgidrp-0.1/requirements.txt +5 -0
  23. corgidrp-0.1/setup.cfg +4 -0
  24. corgidrp-0.1/setup.py +32 -0
  25. corgidrp-0.1/tests/test_add_phot_noise.py +35 -0
  26. corgidrp-0.1/tests/test_badpixel.py +105 -0
  27. corgidrp-0.1/tests/test_caldb.py +152 -0
  28. corgidrp-0.1/tests/test_cr_detection.py +647 -0
  29. corgidrp-0.1/tests/test_dark_sub.py +80 -0
  30. corgidrp-0.1/tests/test_data/example_L1_input.fits +10999 -0
  31. corgidrp-0.1/tests/test_data/metadata.yaml +92 -0
  32. corgidrp-0.1/tests/test_data/metadata_eng.yaml +61 -0
  33. corgidrp-0.1/tests/test_data/nonlin_sample.csv +41 -0
  34. corgidrp-0.1/tests/test_data/nonlin_sample.fits +0 -0
  35. corgidrp-0.1/tests/test_data_levels.py +34 -0
  36. corgidrp-0.1/tests/test_dataset.py +127 -0
  37. corgidrp-0.1/tests/test_desmear.py +63 -0
  38. corgidrp-0.1/tests/test_detector_params.py +33 -0
  39. corgidrp-0.1/tests/test_emgain_div.py +42 -0
  40. corgidrp-0.1/tests/test_err_dq.py +245 -0
  41. corgidrp-0.1/tests/test_flat_div.py +85 -0
  42. corgidrp-0.1/tests/test_frame_selection.py +121 -0
  43. corgidrp-0.1/tests/test_kgain.py +60 -0
  44. corgidrp-0.1/tests/test_non_linearity_correction.py +162 -0
  45. corgidrp-0.1/tests/test_prescan_sub.py +524 -0
  46. corgidrp-0.1/tests/test_walker.py +79 -0
corgidrp-0.1/LICENSE ADDED
@@ -0,0 +1,27 @@
1
+ Copyright (c) 2024, corgidrp Developers
2
+ All rights reserved.
3
+
4
+ Redistribution and use in source and binary forms, with or without
5
+ modification, are permitted provided that the following conditions are met:
6
+
7
+ 1. Redistributions of source code must retain the above copyright notice, this
8
+ list of conditions and the following disclaimer.
9
+
10
+ 2. Redistributions in binary form must reproduce the above copyright notice,
11
+ this list of conditions and the following disclaimer in the documentation
12
+ and/or other materials provided with the distribution.
13
+
14
+ 3. Neither the name of the copyright holder nor the names of its contributors
15
+ may be used to endorse or promote products derived from this software without
16
+ specific prior written permission.
17
+
18
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
19
+ ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
20
+ WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
21
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
22
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
23
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
24
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
25
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
26
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
27
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,7 @@
1
+ include requirements.txt
2
+ include corgidrp/recipe_templates/*.json
3
+ include tests/test_data/example_L1_input.fits
4
+ include tests/test_data/metadata.yaml
5
+ include tests/test_data/metadata_eng.yaml
6
+ include tests/test_data/nonlin_sample.csv
7
+ include tests/test_data/nonlin_sample.fits
corgidrp-0.1/PKG-INFO ADDED
@@ -0,0 +1,17 @@
1
+ Metadata-Version: 2.1
2
+ Name: corgidrp
3
+ Version: 0.1
4
+ Summary: (Roman Space Telescope) CORonaGraph Instrument Data Reduction Pipeline
5
+ Home-page: https://github.com/roman-corgi/corgidrp
6
+ Author: Roman Coronagraph Instrument CPP
7
+ License: BSD
8
+ Keywords: Roman Space Telescope Exoplanets Astronomy
9
+ Classifier: Intended Audience :: Science/Research
10
+ Classifier: Topic :: Scientific/Engineering :: Astronomy
11
+ Classifier: Programming Language :: Python :: 3
12
+ License-File: LICENSE
13
+ Requires-Dist: numpy
14
+ Requires-Dist: astropy
15
+ Requires-Dist: pandas
16
+ Requires-Dist: pytest
17
+ Requires-Dist: scipy
corgidrp-0.1/README.md ADDED
@@ -0,0 +1,158 @@
1
+ # corgidrp: CORonaGraph Instrument Data Reduction Pipeline
2
+ This is the data reduction pipeline for the Nancy Grace Roman Space Telescope Coronagraph Instrument
3
+
4
+ ![Testing Badge](https://github.com/roman-corgi/corgidrp/actions/workflows/python-app.yml/badge.svg)
5
+
6
+ ## Install
7
+ As the code is very much still in development, clone this repository, enter the top-level folder, and run the following command:
8
+ ```
9
+ pip install -e .
10
+ ```
11
+ Then you can import `corgidrp` like any other python package!
12
+
13
+ The installation will create a configuration folder in your home directory called `.corgidrp`.
14
+ That configuration directory will be used to locate things on your computer such as the location of the calibration database and the pipeline configuration file. The configuration files stores setting such as whether to track each individual error term added to the noise.
15
+
16
+ ## How to Contribute
17
+
18
+ We encourage you to chat with Jason, Max, and Marie (e.g., on Slack) to discuss what to do before you get started. Brainstorming
19
+ about how to implement something is a very good use of time and makes sure you aren't going down the wrong path.
20
+ Contact Jason is you have any questions on how to get started on programming details (e.g., git).
21
+
22
+ Below is a quick tutorial that outlines the general contribution process.
23
+
24
+ ### The basics of getting setup
25
+ #### Find a task to work on
26
+ Check out the [Github issues page](https://github.com/roman-corgi/corgidrp/issues) for tasks that need attention. Alternatively, contact Jason (@semaphoreP). _Make sure to tag yourself on the issue and mention in the comments if you start working on it._
27
+
28
+ #### Clone the git repository and install
29
+
30
+ See install instructions above. Contact Jason (@semaphoreP) if you need write access to push changes as you make edits. If you do not have write access to the repository, you can still contribute by creating a fork of the repository under your own GitHub user. See here for details: https://docs.github.com/en/get-started/quickstart/fork-a-repo. You can then use the same commands given below, but just replace `roman-corgi` with your own GitHub username. If you fork the repository, you will need to make sure that your fork is up to date with the main repository (https://docs.github.com/en/pull-requests/collaborating-with-pull-requests/working-with-forks/syncing-a-fork).
31
+
32
+ To quickly get up an running with the repository, execute the following commands in a terminal (or command prompt - you will need to have the git executable installed on your system):
33
+ ```
34
+ > git clone https://github.com/roman-corgi/corgidrp.git
35
+ > cd corgidrp
36
+ > pip install -e .
37
+ ```
38
+
39
+ ##### Make a new git branch to work on
40
+
41
+ You will create a "feature branch" so you can develop your feature without impacting other people's code. Let's say I'm working on dark subtraction. I could create a feature branch and switch to it like this
42
+
43
+ ```
44
+ > git branch dark-sub
45
+ > git checkout dark-sub
46
+ ```
47
+
48
+ ### Write your pipeline step
49
+
50
+ In `corgidrp`, each pipeline step is a function, that is contained in one of the lX_to_lY.py files (where X and Y are various data levels).
51
+ Think about how your feature can be implemented as a function that takes in some data and outputs processed data. Please see below for some
52
+ `corgidrp` design principles.
53
+
54
+ All functions should follow this example:
55
+
56
+ ```
57
+ def example_step(dataset, calib_data, tuneable_arg=1, another_arg="test"):
58
+ """
59
+ Function docstrings are required and should follow Google style docstrings.
60
+ We will not demonstrate it here for brevity.
61
+ """
62
+ # unless you don't alter the input dataset at all, plan to make a copy of the data
63
+ # this is to ensure functions are reproducible
64
+ processed_dataset = input_dataset.copy()
65
+
66
+ ### Your code here that does the real work
67
+ # here is a convience field to grab all the data in a dataset
68
+ all_data = processed_dataset.all_data
69
+ ### End of your code that does the real work
70
+
71
+ # update the header of the new dataset with your processing step
72
+ history_msg = "I did an example step"
73
+ # update the output dataset with the new data and update the history
74
+ processed_dataset.update_after_processing_step(history_msg, new_all_data=all_data)
75
+
76
+ # return the processed data
77
+ return processed_dataset
78
+ ```
79
+
80
+ Inside the function can be nearly anything you want, but the function signature and start/end of the function should follow a few rules.
81
+
82
+ * Each function should include a docstring that descibes what the function is doing, what the inputs (including units if appropriate) are and what the outputs (also with units). The dosctrings should be [goggle style docstrings](https://sphinxcontrib-napoleon.readthedocs.io/en/latest/example_google.html).
83
+ * The input dataset should always be the first input
84
+ * Additional arguments and keywords exist only if you need them--many relevant parameters might already by in Dataset headers. A pipeline step can only have a single argument (the input dataset) if needed.
85
+ * All additional function arguments/keywords should only consist of the following types: int, float, str, or a class defined in corgidrp.Data.
86
+ * (Long explaination for the curious: The reason for this is that pipeline steps can be written out as text files. Int/float/str are easily represented succinctly by textfiles. All classes in corgidrp.Data can be created simply by passing in a filepath. Therefore, all pipeline steps have easily recordable arguments for easy reproducibility.)
87
+ * The first line of the function generally should be creating a copy of the dataset (which will be the output dataset). This way, the output dataset is not the same instance as the input dataset. This will make it easier to ensure reproducibility.
88
+ * The function should always end with updating the header and (typically) the data of the output dataset. The history of running this pipeline step should be recorded in the header.
89
+
90
+ You can check out `corgidrp.l2a_to_l2b.dark_subtraction` function as an example of a basic pipeline step.
91
+
92
+ ### Write a unit test to debug your pipeline step
93
+
94
+ We are required to write tests to verify the functionality of the code. Instead of seeing this as an extra chore, I encourage you to write unit tests to be your debug script to get your code working (this is called "test-driven development").
95
+
96
+ All tests are stored in the `tests` folder and each test is a function that starts with `test_`. See `tests/test_dark_sub.py` as an example. Within each test, you will likely need to simulate some mock data, run it through your function you wrote, and verify it ran correctly using assert statements. Your tests should cover the primary use cases of your code, and check that the function outputs what you expect. You do not need high fidelity data for your test: focus on making sure the data is in the correct format as real data, and less on making sure the data values are simulated to high fidelity (see the examples in the `mocks.py` module).
97
+
98
+ Importantly, these tests will allow code reviewers to test and understand your code. We will also run these tests in an automated test suite (continuous integration) with the pipeline to verify the functions continue to work (e.g., as dependencies change).
99
+
100
+ #### How to run your tests locally
101
+
102
+ You can either run tests individually yourself (to debug individual tests) or run the entire test suite to make sure you didn't break anything.
103
+
104
+ To run an individual test, call the test function you want to test at the bottom of its `test_*.py` script. Then, you just need to run the `test_*.py` script. See `tests/test_dark_sub.py` for an example.
105
+
106
+ To run all the tests in the test suite, go to the base corgidrp folder in a terminal and run the `pytest` command.
107
+
108
+ ### Linting
109
+
110
+ In addition to unit tests, your code will need to pass a static analysis before being merged. `corgidrp` currently runs a subset of flake8 tests, which you can replicate on your local system by running:
111
+
112
+ ```
113
+ flake8 . --count --select=E9,F63,F7,F82,DCO020,DCO021,DCO022,DCO023,DCO024,DCO030,DCO031,DCO032,DCO060,DCO061,DCO062,DCO063,DCO064,DCO065 --show-source --statistics
114
+ ```
115
+ from the top-level directory of the repository. In order to run these tests you will need to have `flake8` and `flake8-docstrings-complete` installed (both are pip-installable). Note that the test subset may be updated in the future. To see the current set of tests being applied, look in the continuous integration GitHub action, located in the repository in file `.github/workflows/python-app.yml`.
116
+
117
+ ### Create a pull request to merge your changes
118
+ Before creating a pull request, review the design Principles below. Use the Github pull request feature to request that your changes get merged into the `main` branch. Assign Jason/Max to be your reviewers. Your changes will be reviewed, and possibly some edits will be requested. You can simply make additional pushes to your branch to update the pull request with those changes (you don't need to delete the PR and make a new one). When the branch is satisfactory, we will pull your changes in. When preparing your pull request, you may find it helpful to follow this checklist:
119
+
120
+ - [ ] If working with a fork, sync your fork to the upstream repository
121
+ - [ ] Ensure that your pull can be merged automatically by merging main onto the branch you wish to pull from
122
+ - [ ] Ensure that all of your new additions have properly formatted docstrings (this will happen automatically if you run the `flake8` command given above
123
+ - [ ] Ensure that all of the commits going in to your pull have informative messages
124
+ - [ ] Ensure that all unit tests pass locally, and that you have provided new unit tests for all new functionality in your pull
125
+ - [ ] Create a new pull request, fully describing all changes/additions
126
+
127
+
128
+ ## Overarching Design Principles
129
+ * Minimize the use of external packages, unless it saves us a lot of time. If you need to use something external, default to well-established and maintained packages, such as `numpy`, `scipy` or `Astropy`. If you think you need to use something else, please check with Jason and Max.
130
+ * Minimize the use of new classes, with the exception of new classes that extend the existing data framework.
131
+ * The python module files (i.e. the *.py files) should typically hold on the order of 5-10 different functions. You can create new ones if need be, but the new files should be general enough in topic to encapsulate other future functions as well.
132
+ * All the image data in the Dataset and Image objects should be saved as standard numpy arrays. Other types of arrays (such as masked arrays) can be used as intermediate products within a function.
133
+ * Keep things simple
134
+ * Use _descriptive_ variable names **always**.
135
+ * Comments should be used to describe a section of the code where it's not immediately obvious what the code is doing. Using descriptive variable names will minimize the amount of comments required.
136
+
137
+ ## FAQ
138
+
139
+ * Does my pipeline function need to save files?
140
+ * Files will be saved by a higher level pipeline code. As long as you output an object that's an instance of a `corgidrp.Data` class, it will have a `save()` function that will be used.
141
+ * Can I create new data classes?
142
+ * Yes, you can feel free to make new data classes. Generally, they should be a subclass of the `Image` class, and you can look at the `Dark` class as an example. Each calibration type should have its own `Image` subclass defined. Talk with Jason and Max to discuss how your class should be implemented!
143
+
144
+ * What python version should I develop in?
145
+ * Python 3.12
146
+
147
+ * How should I treat hard-coded variables
148
+ * If a variable value is extremely unlikely to change AND is only required by one module, it can be hard coded inside that module.
149
+ * If it is unlikely to change but will need to be referenced by multiple modules, it should be added to the central constants file (name TBD).
150
+ * If it is a setting about pipeline behavior that may change, it should be implemented by a config file (examples are location of the calibration database and whether to save individual error terms in the output FITS files).
151
+
152
+ * Where should I store computed variables so they can be referenced later in the pipeline?
153
+ * If possible, in the header of the dataset being processed or in a hew HDU extension
154
+ * If not possible, then let's chat!
155
+
156
+ * Where do I save FITS files or other data files I need to use for my tests?
157
+ * Auxiliary data to run tests should be stored in the tests/test_data folder
158
+ * If they are larger than 1 MB, they should be stored using `git lfs`. Ask Jason about setting up git lfs (as of writing, we have not set up git lfs yet).
@@ -0,0 +1,63 @@
1
+ import configparser
2
+ import os
3
+ import pathlib
4
+ import configparser
5
+
6
+ __version__ = "0.1"
7
+ version = __version__ # temporary backwards compatability
8
+
9
+ #### Create a configuration file for the corgidrp if it doesn't exist.
10
+ def create_config_dir():
11
+ """
12
+ Checks if the default .corgidrp directory exists, and if not, it sets it up
13
+ """
14
+ homedir = pathlib.Path.home()
15
+ config_folder = os.path.join(homedir, ".corgidrp")
16
+ # replace legacy file with folder if needed
17
+ if os.path.isfile(config_folder):
18
+ oldconfig = configparser.ConfigParser()
19
+ oldconfig.read(config_folder)
20
+ os.remove(config_folder)
21
+ else:
22
+ oldconfig = None
23
+
24
+ # make folder if doesn't exist
25
+ if not os.path.isdir(config_folder):
26
+ os.mkdir(config_folder)
27
+
28
+ # make default calibrations folder
29
+ default_cal_dir = os.path.join(config_folder, "default_calibs")
30
+ if not os.path.exists(default_cal_dir):
31
+ os.mkdir(default_cal_dir)
32
+
33
+ # write config
34
+ config_filepath = os.path.join(config_folder, "corgidrp.cfg")
35
+ if not os.path.exists(config_filepath):
36
+ config = configparser.ConfigParser()
37
+ config["PATH"] = {}
38
+ config["PATH"]["caldb"] = os.path.join(config_folder, "corgidrp_caldb.csv") # location to store caldb
39
+ config["PATH"]["default_calibs"] = default_cal_dir
40
+ config["DATA"] = {}
41
+ config["DATA"]["track_individual_errors"] = "False"
42
+ # overwrite with old settings if needed
43
+ if oldconfig is not None:
44
+ config["PATH"]["caldb"] = oldconfig["PATH"]["caldb"]
45
+
46
+ with open(config_filepath, 'w') as f:
47
+ config.write(f)
48
+
49
+ print("corgidrp: Configuration file written to {0}. Please edit if you want things stored in different locations.".format(config_filepath))
50
+ create_config_dir()
51
+
52
+ _bool_map = {"true" : True, "false" : False}
53
+
54
+ # borrowed from the kpicdrp caldb
55
+ # load in default caldbs based on configuration file
56
+ config_filepath = os.path.join(pathlib.Path.home(), ".corgidrp", "corgidrp.cfg")
57
+ config = configparser.ConfigParser()
58
+ config.read(config_filepath)
59
+
60
+ ## pipeline settings
61
+ caldb_filepath = config.get("PATH", "caldb", fallback=None)
62
+ default_cal_dir = config.get("PATH", "default_calibs", fallback=None)
63
+ track_individual_errors = _bool_map[config.get("DATA", "track_individual_errors").lower()]
@@ -0,0 +1,304 @@
1
+ """
2
+ Calibration tracking system. Modified from kpicdrp caldb implmentation (Copyright (c) 2024, KPIC Team)
3
+ """
4
+ import os
5
+ import numpy as np
6
+ import pandas as pd
7
+ import corgidrp
8
+ import corgidrp.data as data
9
+ import astropy.time as time
10
+
11
+
12
+ column_names = [
13
+ "Filepath",
14
+ "Type",
15
+ "MJD",
16
+ "EXPTIME",
17
+ "Files Used",
18
+ "Date Created",
19
+ "Hash",
20
+ "DRPVERSN",
21
+ "OBSID",
22
+ "NAXIS1",
23
+ "NAXIS2",
24
+ "OPMODE",
25
+ "CMDGAIN",
26
+ "EXCAMT",
27
+ ]
28
+
29
+ labels = {data.Dark: "Dark",
30
+ data.NonLinearityCalibration: "NonLinearityCalibration",
31
+ data.BadPixelMap: "BadPixelMap",
32
+ data.KGain : "KGain",
33
+ data.DetectorParams : "DetectorParams"}
34
+
35
+ class CalDB:
36
+ """
37
+ Database for tracking calibration files saved to disk. Modified from the kpicdrp version
38
+
39
+ Note that database is not parallelism-safe, but should be ok in most cases.
40
+ (Jason: look at using posix_ipc to guarantee thread safety if we really need it)
41
+
42
+ Args:
43
+ filepath (str): [optional] filepath to a CSV file with an existing database
44
+
45
+ Fields:
46
+ columns (list): column names of dataframe
47
+ filepath(str): full filepath to data
48
+ """
49
+
50
+ def __init__(self, filepath=""):
51
+ """
52
+ Args:
53
+ filepath (str): [optional] filepath to a CSV file with an existing database
54
+ """
55
+
56
+ # If filepath is not passed in, use the default (majority of case)
57
+ if len(filepath) == 0:
58
+ self.filepath = corgidrp.caldb_filepath
59
+ else:
60
+ # possibly edge case where we want to use specialized caldb
61
+ self.filepath = filepath
62
+
63
+ # if database does't exist, create a blank one
64
+ if not os.path.exists(self.filepath):
65
+ # new database
66
+ self.columns = column_names
67
+ self._db = pd.DataFrame(columns=self.columns)
68
+ self.save()
69
+ else:
70
+ # already a database exists
71
+ self.load()
72
+ self.columns = list(self._db.columns.values)
73
+
74
+ def load(self):
75
+ """
76
+ Load/update db from filepath
77
+ """
78
+ self._db = pd.read_csv(self.filepath)
79
+
80
+ def save(self):
81
+ """
82
+ Save file without numbered index to disk with user specified filepath as a CSV file
83
+ """
84
+ self._db.to_csv(self.filepath, index=False)
85
+
86
+ def _get_values_from_entry(self, entry, is_calib=True):
87
+ """
88
+ Extract the properties from this data entry to ingest them into the database
89
+
90
+ Args:
91
+ entry (corgidrp.data.Image subclass): calibration frame to add to the database
92
+ is_calib (bool): is a calibration frame. if Not, it won't look up filetype.
93
+ Only used in get_calib() to grab metadata for science frames
94
+
95
+ Returns:
96
+ tuple:
97
+ row (list):
98
+ List of data entry properties
99
+ row_dict (dict):
100
+ Dictionary of data entry properties keyed by column names
101
+
102
+ """
103
+ filepath = os.path.abspath(entry.filepath)
104
+ if is_calib:
105
+ datatype = labels[entry.__class__] # get the database str representation
106
+ else:
107
+ datatype = "Sci"
108
+ mjd = time.Time(entry.ext_hdr["SCTSRT"]).mjd
109
+ exptime = entry.ext_hdr["EXPTIME"]
110
+
111
+ # check if this exists. will be a keyword written by corgidrp
112
+ if "DRPNFILE" in entry.ext_hdr:
113
+ files_used = entry.ext_hdr["DRPNFILE"]
114
+ else:
115
+ files_used = 0
116
+
117
+ if "DRPCTIME" in entry.ext_hdr:
118
+ date_created = time.Time(entry.ext_hdr["DRPCTIME"]).mjd
119
+ else:
120
+ date_created = -1
121
+
122
+ if "DRPVERSN" in entry.ext_hdr:
123
+ drp_version = entry.ext_hdr["DRPVERSN"]
124
+ else:
125
+ drp_version = ""
126
+
127
+ obsid = entry.pri_hdr["OBSID"]
128
+
129
+ hash_val = entry.get_hash()
130
+
131
+ # this only works for 2D images. may need to adapt for non-2D calibration frames
132
+ naxis1 = entry.data.shape[-1]
133
+ naxis2 = entry.data.shape[-2]
134
+
135
+ row = [
136
+ filepath,
137
+ datatype,
138
+ mjd,
139
+ exptime,
140
+ files_used,
141
+ date_created,
142
+ hash_val,
143
+ drp_version,
144
+ obsid,
145
+ naxis1,
146
+ naxis2,
147
+ ]
148
+
149
+ # rest are ext_hdr keys we can copy
150
+ start_index = len(row)
151
+ for i in range(start_index, len(self.columns)):
152
+ row.append(entry.ext_hdr[self.columns[i]]) # add value staright from header
153
+
154
+ row_dict = {}
155
+ for key, val in zip(self.columns, row):
156
+ row_dict[key] = val
157
+
158
+ return row, row_dict
159
+
160
+ def create_entry(self, entry, to_disk=True):
161
+ """
162
+ Add a new entry to or update an existing one in the database. Note that function by default will load and save db to disk
163
+
164
+ Args:
165
+ entry (corgidrp.data.Image subclass): calibration frame to add to the database
166
+ to_disk (bool): True by default, will update DB from disk before adding entry and saving it back to disk
167
+ """
168
+ new_row, row_dict = self._get_values_from_entry(entry)
169
+
170
+ # update database from disk in case anything changed
171
+ if to_disk:
172
+ self.load()
173
+
174
+ # use filepath as key to see if it's already in database
175
+ if row_dict["Filepath"] in self._db.values:
176
+ row_index = self._db[
177
+ self._db["Filepath"] == row_dict["Filepath"]
178
+ ].index.values
179
+ self._db.loc[row_index, self.columns] = new_row
180
+ # otherwise create new entry
181
+ else:
182
+ new_entry = pd.DataFrame([new_row], columns=self.columns)
183
+ self._db = pd.concat([self._db, new_entry], ignore_index=True)
184
+
185
+ # save to disk to update changes
186
+ if to_disk:
187
+ self.save()
188
+
189
+ def remove_entry(self, entry, to_disk=True):
190
+ """
191
+ Remove an entry from the database. Removes the entire row
192
+
193
+ Args:
194
+ entry (corgidrp.data.Image subclass): calibration frame to add to the database
195
+ to_disk (bool): True by default, will update DB from disk before adding entry and saving it back to disk
196
+ """
197
+ new_row, row_dict = self._get_values_from_entry(entry)
198
+
199
+ # update database from disk in case anything changed
200
+ if to_disk:
201
+ self.load()
202
+
203
+ if row_dict["Filepath"] in self._db.values:
204
+ entry_index = self._db[
205
+ self._db["Filepath"] == row_dict["Filepath"]
206
+ ].index.values
207
+ self._db = self._db.drop(self._db.index[entry_index])
208
+ self._db = self._db.reset_index(drop=True)
209
+ else:
210
+ raise ValueError("No filepath found so could not remove.")
211
+
212
+ # save to disk to update changes
213
+ if to_disk:
214
+ self.save()
215
+
216
+ def get_calib(self, frame, dtype, to_disk=True):
217
+ """
218
+ Outputs the best calibration file of the given type for the input sciene frame.
219
+
220
+ Args:
221
+ frame (corgidrp.data.Image): an image frame to request a calibratio for
222
+ dtype (corgidrp.data Class): for example: corgidrp.data.Dark (TODO: document the entire list of options)
223
+ to_disk (bool): True by default, will update DB from disk before matching
224
+
225
+ Returns:
226
+ corgidrp.data.*: an instance of the appropriate calibration type (Exact type depends on calibration type)
227
+ """
228
+ if dtype not in labels:
229
+ raise ValueError(
230
+ "Requested calibration dtype of {0} not a valid option".format(dtype)
231
+ )
232
+ dtype_label = labels[dtype]
233
+
234
+ # get values for this science frame
235
+ _, frame_dict = self._get_values_from_entry(frame, is_calib=False)
236
+
237
+ # update database from disk in case anything changed
238
+ if to_disk:
239
+ self.load()
240
+
241
+ # downselect to only calibs of this type
242
+ calibdf = self._db[self._db["Type"] == dtype_label]
243
+
244
+ if dtype_label in ["Dark"]:
245
+ # general selection criteria for 2D image frames. Can use different selection criteria for different dtypes
246
+ options = calibdf.loc[
247
+ (
248
+ (calibdf["EXPTIME"] == frame_dict["EXPTIME"])
249
+ & (calibdf["NAXIS1"] == frame_dict["NAXIS1"])
250
+ & (calibdf["NAXIS2"] == frame_dict["NAXIS2"])
251
+ )
252
+ ]
253
+ else:
254
+ options = calibdf
255
+
256
+ # select the one closest in time
257
+ result_index = np.abs(options["MJD"] - frame_dict["MJD"]).argmin()
258
+ calib_filepath = options.iloc[result_index, 0]
259
+
260
+ # load the object from disk and return it
261
+ return dtype(calib_filepath)
262
+
263
+ def scan_dir_for_new_entries(self, filedir, look_in_subfolders=True, to_disk=True):
264
+ """
265
+ Scan a folder and subfolder for calibration files and add them all to the caldb
266
+
267
+ Args:
268
+ filedir (str): path to folder to scan (includes all subfolders by default)
269
+ look_in_subfolders (bool): whether to look in subfolders for files. True by default
270
+ to_disk (bool): True by default, will update DB from disk before adding entry and saving it back to disk
271
+ """
272
+ calib_frames = []
273
+ # walk the directory to find all the calibration files
274
+ for dirpath, subfolders, filenames in os.walk(filedir):
275
+ for filename in filenames:
276
+ # hard coded check only for files that end in .fits
277
+ if filename[-5:] != ".fits":
278
+ continue
279
+
280
+ filepath = os.path.join(dirpath, filename)
281
+ frame = data.autoload(filepath)
282
+
283
+ # check what class it has been loaded as. only save frames that fall into calibration classes
284
+ if frame.__class__ in labels:
285
+ calib_frames.append(frame)
286
+
287
+ # the first iteration looks in the basedir
288
+ # if we don't wnat to look in subdirs now, we should break
289
+ if not look_in_subfolders:
290
+ break
291
+
292
+ # load all these files into the caldb
293
+ for calib_frame in calib_frames:
294
+ self.create_entry(calib_frame, to_disk=to_disk)
295
+
296
+ ### Create set of default calibrations
297
+ # Add default detector_params calibration file if it doesn't exist
298
+ if not os.path.exists(os.path.join(corgidrp.default_cal_dir, "DetectorParams_2023-11-01T00:00:00.000.fits")):
299
+ default_detparams = data.DetectorParams({}, date_valid=time.Time("2023-11-01 00:00:00", scale='utc'))
300
+ default_detparams.save(filedir=corgidrp.default_cal_dir)
301
+
302
+ # add default caldb entries
303
+ default_caldb = CalDB()
304
+ default_caldb.scan_dir_for_new_entries(corgidrp.default_cal_dir)