cloudcatalog 0.4__tar.gz → 0.6.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. cloudcatalog-0.6.1/.github/bug_report.md +38 -0
  2. cloudcatalog-0.6.1/.github/feature_request.md +20 -0
  3. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/.gitignore +10 -0
  4. cloudcatalog-0.6.1/.gitlab-ci.yml +90 -0
  5. {cloudcatalog-0.4/src/cloudcatalog.egg-info → cloudcatalog-0.6.1}/PKG-INFO +7 -8
  6. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/README.md +3 -3
  7. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/docs/Notes.md +1 -1
  8. cloudcatalog-0.6.1/docs/cloudcatalog-spec-06.md +390 -0
  9. cloudcatalog-0.6.1/docs/cloudcatalog-spec.md +390 -0
  10. cloudcatalog-0.4/docs/cloudcatalog_demo.py → cloudcatalog-0.6.1/docs/demo.py +6 -10
  11. cloudcatalog-0.6.1/docs/earlier/cloudcatalog-spec-05.md +398 -0
  12. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/pyproject.toml +8 -4
  13. cloudcatalog-0.6.1/requirements-dev.txt +11 -0
  14. cloudcatalog-0.6.1/requirements.txt +4 -0
  15. cloudcatalog-0.4/src/cloudcatalog.py → cloudcatalog-0.6.1/src/cc.py +157 -20
  16. {cloudcatalog-0.4 → cloudcatalog-0.6.1/src/cloudcatalog.egg-info}/PKG-INFO +7 -8
  17. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/src/cloudcatalog.egg-info/SOURCES.txt +14 -5
  18. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/src/cloudcatalog.egg-info/top_level.txt +1 -0
  19. cloudcatalog-0.6.1/src/cloudcatalog.py +799 -0
  20. cloudcatalog-0.6.1/tests/test_hdrl.py +42 -0
  21. cloudcatalog-0.4/LICENSE +0 -21
  22. /cloudcatalog-0.4/LICENSE.txt → /cloudcatalog-0.6.1/LICENSE.MD +0 -0
  23. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/TODO.md +0 -0
  24. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/docs/Makefile +0 -0
  25. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/docs/conf.py +0 -0
  26. {cloudcatalog-0.4/docs → cloudcatalog-0.6.1/docs/earlier}/cloudcatalog-spec-04.md +0 -0
  27. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/docs/index.rst +0 -0
  28. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/docs/make.bat +0 -0
  29. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/docs/scr_logo.png +0 -0
  30. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/setup.cfg +0 -0
  31. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/src/__init__.py +0 -0
  32. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/src/cloudcatalog.egg-info/dependency_links.txt +0 -0
  33. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/src/cloudcatalog.egg-info/requires.txt +0 -0
  34. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/src/data/README.rst +0 -0
  35. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/tests/__init__.py +0 -0
  36. /cloudcatalog-0.4/tests/test_file_registry.py → /cloudcatalog-0.6.1/tests/older_test_file_registry.py +0 -0
  37. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/tests/test_catalog_registry.py +0 -0
  38. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/tests/test_entire_catalog_search.py +0 -0
  39. {cloudcatalog-0.4 → cloudcatalog-0.6.1}/tests/validation.py +0 -0
@@ -0,0 +1,38 @@
1
+ ---
2
+ name: Bug report
3
+ about: Create a report to help us improve
4
+ title: ''
5
+ labels: ''
6
+ assignees: ''
7
+
8
+ ---
9
+
10
+ **Describe the bug**
11
+ A clear and concise description of what the bug is.
12
+
13
+ **To Reproduce**
14
+ Steps to reproduce the behavior:
15
+ 1. Go to '...'
16
+ 2. Click on '....'
17
+ 3. Scroll down to '....'
18
+ 4. See error
19
+
20
+ **Expected behavior**
21
+ A clear and concise description of what you expected to happen.
22
+
23
+ **Screenshots**
24
+ If applicable, add screenshots to help explain your problem.
25
+
26
+ **Error Log**
27
+ If applicable, add the log dump produced by the bug.
28
+
29
+ **Desktop (please complete the following information):**
30
+ - OS: [e.g. iOS]
31
+ - Browser [e.g. chrome, safari] (if applicable)
32
+ - Platform version [e.g. 1.0.0]
33
+ - Python version
34
+ - AWS CLI version
35
+
36
+
37
+ **Additional context**
38
+ Add any other context about the problem here.
@@ -0,0 +1,20 @@
1
+ ---
2
+ name: Feature request
3
+ about: Suggest an idea for this project
4
+ title: ''
5
+ labels: ''
6
+ assignees: ''
7
+
8
+ ---
9
+
10
+ **Is your feature request related to a problem? Please describe.**
11
+ A clear and concise description of what the problem is. Ex. I'm always frustrated when [...]
12
+
13
+ **Describe the solution you'd like**
14
+ A clear and concise description of what you want to happen.
15
+
16
+ **Describe alternatives you've considered**
17
+ A clear and concise description of any alternative solutions or features you've considered.
18
+
19
+ **Additional context**
20
+ Add any other context or screenshots about the feature request here.
@@ -3,6 +3,16 @@ __pycache__/
3
3
  *.py[cod]
4
4
  *$py.class
5
5
 
6
+ # cloud catalog common caches
7
+ *_cache/
8
+ src/*_cache/
9
+
10
+ # emacs save files
11
+ *~
12
+
13
+ # my junk folder
14
+ junk/
15
+
6
16
  # C extensions
7
17
  *.so
8
18
 
@@ -0,0 +1,90 @@
1
+ default:
2
+ image: python:3.9
3
+
4
+ stages:
5
+ - test
6
+
7
+ # At the momemnt, there are a subset of unit tests that require nodejs
8
+ # to be installed.
9
+ unit-test:
10
+ stage: test
11
+ before_script:
12
+ - mkdir -p public/badges
13
+ script:
14
+ - export PYTHONPATH=.
15
+ - python -m pip install -r requirements.txt
16
+ - python -m pip install -r requirements-dev.txt
17
+ - pytest -c pytest-unit.ini --junit-xml=TEST-HelioCloud-cloudcatalog.xml --junit-prefix=HelioCloud-cloudcatalog
18
+ artifacts:
19
+ when: always
20
+ paths:
21
+ - TEST-HelioCloud-cloudcatalog.xml
22
+ reports:
23
+ junit: TEST-HelioCloud-cloudcatalog.xml
24
+
25
+ # See
26
+ # * https://docs.gitlab.com/ee/ci/testing/test_coverage_visualization.html
27
+ coverage:
28
+ stage: test
29
+ script:
30
+ - export PYTHONPATH=.
31
+ - python -m pip install -r requirements.txt
32
+ - python -m pip install -r requirements-dev.txt
33
+ - coverage run -m pytest -c pytest-unit.ini
34
+ - coverage report
35
+ - coverage xml
36
+ - coverage html
37
+ coverage: '/(?i)TOTAL.*? (100(?:\.0+)?\%|[1-9]?\d(?:\.\d+)?\%)$/'
38
+ artifacts:
39
+ when: always
40
+ paths:
41
+ - coverage.xml
42
+ - htmlcov
43
+ reports:
44
+ coverage_report:
45
+ coverage_format: cobertura
46
+ path: coverage.xml
47
+
48
+ # For static-analysis, we're going to run the analysis once for badge generation
49
+ # and once for generating the codeclimate report, which is required integration
50
+ # with the gitlab pull request.
51
+ #
52
+ # This configuration only runs pylint on the main source files of the project,
53
+ # it excludes the unit and integration test folders as well as the cdk output
54
+ # build folder.
55
+ #
56
+ # See:
57
+ # * https://pypi.org/project/pylint-gitlab/
58
+ static-analysis:
59
+ stage: test
60
+ variables:
61
+ PYLINT_TEXT_OUTPUT_FILE: 'pylint.txt'
62
+ PYLINT_SCORE_OUTPUT_FILE: public/pylint.score
63
+ PYLINT_BADGE_OUTPUT_FILE: 'public/pylint.svg'
64
+ CODE_CLIMATE_OUTPUT_FILE: 'codeclimate.json'
65
+ before_script:
66
+ - mkdir -p public
67
+ script:
68
+ - export PYTHONPATH=.
69
+ - python -m pip install -r requirements.txt
70
+ - python -m pip install -r requirements-dev.txt
71
+ - pylint --exit-zero --output-format=text $(find -type f -name "*.py" ! -path "**/.venv/**" | grep -v '/test/' | grep -v '/cdk.out/') | tee ${PYLINT_TEXT_OUTPUT_FILE}
72
+ - sed -n 's/^Your code has been rated at \([-0-9.]*\)\/.*/\1/p' ${PYLINT_TEXT_OUTPUT_FILE} > ${PYLINT_SCORE_OUTPUT_FILE}
73
+ - anybadge --value=$(cat ${PYLINT_SCORE_OUTPUT_FILE}) --file=${PYLINT_BADGE_OUTPUT_FILE} pylint
74
+ - pylint --exit-zero --load-plugins=pylint_gitlab --output-format=gitlab-codeclimate $(find -type f -name "*.py" ! -path "**/.venv/**" | grep -v '/test/' | grep -v '/cdk.out/') > ${CODE_CLIMATE_OUTPUT_FILE}
75
+ artifacts:
76
+ when: always
77
+ paths:
78
+ - ${CODE_CLIMATE_OUTPUT_FILE}
79
+ - ${PYLINT_TEXT_OUTPUT_FILE}
80
+ - ${PYLINT_SCORE_OUTPUT_FILE}
81
+ - ${PYLINT_BADGE_OUTPUT_FILE}
82
+ reports:
83
+ codequality: ${CODE_CLIMATE_OUTPUT_FILE}
84
+
85
+ black:
86
+ stage: test
87
+ script:
88
+ - export PYTHONPATH=.
89
+ - python -m pip install -r requirements-dev.txt
90
+ - black --check .
@@ -1,20 +1,19 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: cloudcatalog
3
- Version: 0.4
3
+ Version: 0.6.1
4
4
  Summary: API for accessing the generalized Cloud Catalog (cloudcatalog) specification for sharing data in and across clouds
5
5
  Author-email: Johns Hopkins University Applied Physics Laboratory LLC <sandy.antunes@jhuapl.edu>
6
6
  License: MIT License
7
7
  Project-URL: Homepage, https://heliocloud.org
8
- Project-URL: Documentation, https://heliocloud.org
9
- Project-URL: Repository, https://gitlab.smce.nasa.gov
8
+ Project-URL: Documentation, https://github.com/heliocloud-data/cloudcatalog
9
+ Project-URL: Repository, https://github.com/heliocloud-data/cloudcatalog
10
10
  Keywords: cloud,index,catalog,AWS
11
11
  Classifier: Development Status :: 4 - Beta
12
12
  Classifier: Topic :: Utilities
13
13
  Classifier: License :: OSI Approved :: MIT License
14
14
  Requires-Python: >=3.8
15
15
  Description-Content-Type: text/markdown
16
- License-File: LICENSE
17
- License-File: LICENSE.txt
16
+ License-File: LICENSE.MD
18
17
  Requires-Dist: boto3
19
18
  Requires-Dist: pandas
20
19
 
@@ -66,17 +65,17 @@ print(fr.get_entries())
66
65
  # also save the downloaded file index
67
66
  fr_id = 'a_dataset_id_from_the_catalog'
68
67
  start_date = '2007-02-01T00:00:00Z' # A ISO 8601 standard time and a valid time witin the mission/file-index
69
- end_date = None # A ISO 8601 standard time or None if want all the file indices after start_date
68
+ stop_date = None # A ISO 8601 standard time or None if want all the file indices after start_date
70
69
  myfiles = fr.request_cloud_catalog(fr_id, start_date=start_date, end_date=end_date, overwrite=False)
71
70
  ```
72
71
 
73
72
  ### Streaming Data from the File Catalog
74
- You now have a pandas DataFrame with startdate, key, and filesize for all the files of the mission within your specified start and end dates. From here, you can use the key to stream some of the data through EC2, a Lambda, or other processing methods.
73
+ You now have a pandas DataFrame with startdate, stopdate, key, and filesize for all the files of the mission within your specified start and end dates. From here, you can use the key to stream some of the data through EC2, a Lambda, or other processing methods.
75
74
 
76
75
  This tool also offers a simple function for streaming the data once the file catalog is obtained:
77
76
 
78
77
  ```python
79
- cloudcatalog.CloudCatalog.stream(cloud_catalog, lambda bfile, startdate, filesize: print(len(bo.read()), filesize))
78
+ cloudcatalog.CloudCatalog.stream(cloud_catalog, lambda bfile, startdate, stopdate, filesize: print(len(bo.read()), filesize))
80
79
  ```
81
80
 
82
81
  ### Searching the Entire Catalog
@@ -46,17 +46,17 @@ print(fr.get_entries())
46
46
  # also save the downloaded file index
47
47
  fr_id = 'a_dataset_id_from_the_catalog'
48
48
  start_date = '2007-02-01T00:00:00Z' # A ISO 8601 standard time and a valid time witin the mission/file-index
49
- end_date = None # A ISO 8601 standard time or None if want all the file indices after start_date
49
+ stop_date = None # A ISO 8601 standard time or None if want all the file indices after start_date
50
50
  myfiles = fr.request_cloud_catalog(fr_id, start_date=start_date, end_date=end_date, overwrite=False)
51
51
  ```
52
52
 
53
53
  ### Streaming Data from the File Catalog
54
- You now have a pandas DataFrame with startdate, key, and filesize for all the files of the mission within your specified start and end dates. From here, you can use the key to stream some of the data through EC2, a Lambda, or other processing methods.
54
+ You now have a pandas DataFrame with startdate, stopdate, key, and filesize for all the files of the mission within your specified start and end dates. From here, you can use the key to stream some of the data through EC2, a Lambda, or other processing methods.
55
55
 
56
56
  This tool also offers a simple function for streaming the data once the file catalog is obtained:
57
57
 
58
58
  ```python
59
- cloudcatalog.CloudCatalog.stream(cloud_catalog, lambda bfile, startdate, filesize: print(len(bo.read()), filesize))
59
+ cloudcatalog.CloudCatalog.stream(cloud_catalog, lambda bfile, startdate, stopdate, filesize: print(len(bo.read()), filesize))
60
60
  ```
61
61
 
62
62
  ### Searching the Entire Catalog
@@ -9,7 +9,7 @@ Test the wheel
9
9
 
10
10
  Do a test package upload then install with:
11
11
 
12
- * twine upload -repository testpypi dist/*
12
+ * twine upload -r testpypi dist/*
13
13
  * pip install -i https://test.pypi.org/simple/cloudcatalog==0.4
14
14
 
15
15
  Commit with
@@ -0,0 +1,390 @@
1
+ <!-- TOC -->
2
+ [1 Introduction](#1-intro)<br/>
3
+ [2 Global data registry](#2-dataregistry)<br/>
4
+ [3 Catalog](#3-catalog)<br/>
5
+ [4 File Indices](#4-fileindex)<br/>
6
+ [5 Info Metadata](#5-info)<br/>
7
+ <!-- \TOC -->
8
+
9
+ Version 0.6.0 \| HelioCloud \|
10
+
11
+ # 1 The generalized Cloud Catalog specification for HelioCloud
12
+
13
+ The shared Cloud Catalog specification can be used for sharing datasets across cloud frameworks as well as exposing cloud archives outside of the cloud.
14
+
15
+ For HelioCloud, this specification creates a global data registry of publicly-accessible disks ('HelioDataRegistry'), maintained at the HDRL HelioCloud.org website. Individual dataset owners then define their dataset file catalogs ('cloudCatalog') for each dataset, that resides in the S3 (or equivalent) bucket alongside the dataset.
16
+
17
+ That global 'HelioDataRegistry.json' is a minimal JSON file that only lists buckets (disks) that contain one or more dataset. It consists of **name** and **endpoint** and lists buckets as endpoints, not datasets. Tools for fetching individual dataset indices visit each **endpoint** to get the cloudCatalog-format **catalog.json** listing datasets available in that bucket. These 'cloudCatalog' consist of the required index to the actual files, and an optional dataset summary file. The cloudCatalog itself is a set of ***<id>_YYYY.csv*** (or csp-zip or parquet) index files, one per year. The optional summary file is named **<id>.json** file and provides additional potentially searchable metadata.
18
+
19
+ ## 1.1 Flow
20
+
21
+ Findability is through the 'HelioDataRegistry' JSON index of S3 buckets, which crawls the **catalog.json** for each S3 bucket that indicates available datasets, which points to the individual dataset file catalog index files **<id>_YYYY.csv** and its optional associated **<id>.info** metadata auxillary file.
22
+
23
+ (Diagram here)
24
+ http://heliocloud.org/catalog/HelioDataRegistry.json -> s3://aplcloud.com/mybucket/catalog.json -> individual dataset file catalogs (.csv)
25
+
26
+ ![Data Registry Schema](dataRegistry_diagram.png)
27
+
28
+ For scientists, datasets are findable by going to 'HelioDataRegistry.json' to get endpoints, then visiting each S3 endpoint to get the catalog listing of datasets available. There is also a search function in the Python client.
29
+
30
+ Accessing the actual data involves accessing each dataset's catalog index files. The cloudCatalog consists of CSV files, one per year. Therefore, a user has the choice to:
31
+ * (a) directly download the CSV index file(s)
32
+ * (b) use our provided Python API to fetch a subset of the CSV file, by date range
33
+ * (c) use AWS Athena to run queries on the CSV file
34
+
35
+ ## 1.2 Uploading Data
36
+
37
+ Data providers must provide both the Data, the Dataset Description (metadata in .json as per the catalog.json spec) and a Manifest (in the .csv format we specify). Sample Python tools that fit this API are part of this repository.
38
+
39
+ Note that users have full control over their datasets and indices; HelioCloud.org only lists buckets (disks) that users wish to make available to the HelioCloud network. To make a bucket (disk) available within the HelioCloud network, currently email the cloud location to heliocloud@groups.io to start the process.
40
+
41
+ We provide example Python tools that fit this API for copying or indexing datasets (based primarily on CDAWeb).
42
+
43
+ # 2. Global data registry 'HelioDataRegistry.json'
44
+
45
+ The global data registry is an unsorted json list of names and bucket endpoints. This is for buckets, not datasets (a bucket can hold multiple datasets). Queries to each endpoint can fetch the list off datasets and more information from the **endpoint**/catalog.json file.
46
+
47
+ The definitive listing of available dataset is kept in the **name for GSFC HelioCloud registry**. Providers who wish to make data visibile to the public do a one-time registration of their S3 bucket (or equivalent) to the main registry by emailing heliocloud@groups.io. The global data registry can be accessed directly or mirrored by anyone.
48
+
49
+ Once an S3 bucket (or equivalent) is registered in the HelioDataRegistry.json file, it does not have to be updated when new datasets are added, only when new buckets are made public.
50
+
51
+ For this global registry, only S3 buckets, not subbuckets, are allowed. For example, 's3://helio-public/' is allowed, but 's3://helio-public/MMS/' is not.
52
+
53
+ ## 2.1 Global Data Registry specification
54
+
55
+ The registry is a JSON file containing the version of this specification, the date it was last modified, and the registry list of datasets. Each dataset has two items: **endpoint** and **name**. **endpoint**s must be unique; names do not enforce uniqueness.
56
+
57
+ * **endpoint** - An accessble S3 (or equivalent) bucket link
58
+ * **name** - A descriptive name for the dataset.
59
+ * **provider** - default 'aws', indicates which cloud provider. Included for future federation.
60
+ * **region** - The AWS or equivalent region the bucket resides in
61
+
62
+ The **name** should be brief and one-line; length is not enforced by the specification but may be truncated when registration to keep things readable.
63
+
64
+ ## 2.2 Sample global data registry
65
+
66
+ Here is a sample 'HelioDataRegistry.json' for three buckets at two sites.
67
+
68
+ ```javascript
69
+ {
70
+ "version": "0.3",
71
+ "modificationDate": "2022-01-01T00:00.00Z",
72
+ "registry": [
73
+ {
74
+ "endpoint": "s3://gov-nasa-hdrl-data1/",
75
+ "name": "GSFC HelioCloud Set 1",
76
+ "provider": "aws",
77
+ "region": "us-east-1"
78
+ },
79
+ {
80
+ "endpoint": "s3://gov-nasa-hdrl-data2/",
81
+ "name": "GSFC HelioCloud Set 2",
82
+ "provider": "aws",
83
+ "region": "us-east-1"
84
+ },
85
+ {
86
+ "endpoint": "s3://edu-apl-helio-public/",
87
+ "name": "APL HelioCLoud",
88
+ "provider": "aws",
89
+ "region": "us-west-1"
90
+ }
91
+ ]
92
+ }
93
+ ```
94
+
95
+ # 3. Catalog of File Registries (per bucket catalog.json)
96
+
97
+ The catalog.json file has an entry for each dataset stored within the given S3 bucket. **endpoint** and **name** are identical to the item in the global data registry.
98
+
99
+ Globally the catalog.json describes the endpoint with the following items. Note that **endpoint** and **name** are included for clarity sake (being duplicates of the main registry) as JSON does not allow for comments, and should be the same as was provided to the global registry.
100
+
101
+ * **endpoint** - same as was provided to GlobalDataRegistry.json, an accessble S3 (or equivalent) bucket link
102
+ * **name** - same as was provided to GlobalDataRegistry.json, a descriptive name for the dataset.
103
+ * **provider** - same as was provided to GlobalDataRegistry.json, default 'aws', indicates which cloud provider. Included for future federation.
104
+ * **region** - same as was provided to GlobalDataRegistry.json, which AWS region
105
+ * **egress** - one of 'no-egress', 'user-pays', 'egress-allowed', or 'none'
106
+ * **status** - A return code, typically "1200/OK". Site owners can temporarily set this to other values
107
+ * **contact** - Who to contact for issues with this bucket, e.g. "Dr. Contact, dr_contact@example.com"
108
+
109
+ * **description** - Optional description of this collection
110
+ * **citation** - Optional how to cite, preferably a DOI for the server
111
+ * **comment** - A catch-all comment field for data provider and developer use. It should not contain information required to parse the data items.
112
+
113
+ For each dataset, the catalog entry requires:
114
+
115
+ * **id** a unique ID for the dataset that follows the ID naming requirements
116
+ * **index** a fully qualified pointer to the object directory containing both the dataset and the required cloudCatalog. It MUST start with s3:// or "https://" (or equivalent) end in a terminating '/'.
117
+ * **start**: string, Restricted ISO 8601 date/time of first record of data in the entire dataset OR the word 'static' for items such as model shapes that lack a time field.
118
+ * **stop**: string, Restricted ISO 8601 date/time of end of the last
119
+ record of data in the entire dataset OR the word 'static' for items such as model shapes that lack a time field.
120
+ * **modification**: string, Restricted ISO 8601 date/time of last time this dataset was updated
121
+ * **title** a short descriptive title sufficient to identify the dataset and its utility to users
122
+ * **indextype** Defines what format the actual cloudCatalog is, one of 'csv', 'csv-zip' or 'parquet'
123
+ * **filetype** the file format of the actual data. Must be from the prescribed list of files.
124
+
125
+ * **description** optional description for dataset".
126
+ * **resource** optional identifier e.g. SPASE ID, dataset description URL, DOI, json link of model parameters, or similar ancillary information".
127
+ * **creation** optional ISO 8601 date/time of the dataset creation".
128
+ * **expiration** optional ISO 8601 date/time after which the dataset will be expired, migrated, or not maintained".
129
+ * **verified** optional ISO 8601 date/time for when the dataset was last tested by verifier programs".
130
+ * **citation** optional how to cite this dataset, DOI or similar".
131
+ * **contact** optional contact name".
132
+ * **about** optional website URL for info, team, etc.
133
+ * **multiyear** optional True/False field (default: False) for use when dataitems span multiple years (see 3.2 below)
134
+
135
+ For file formats, there is a prescibed list. As new file formats are introduced, we will update this specification to give a single unique identifier for it. The reason for the prescribed list is to avoid ambiguity or the need for users to parse (for example, avoiding figuring out '.fts', 'fits', '.FTS', etc) Repositories with multiple files types can specify them as a comma-separated list with no spaces, e.g. 'fits,csv' for a dataset that contains both images and event lists. Currently defined types are 'fits,csv,cdf,netcdf3,netcdf4,hdf5,datamap,txt,binary,other'.
136
+
137
+ The catalog.json file has to be updated when new data is added to a dataset, by updating the **stop** item. Also, the catalog.json file is updated when a new dataset is put into that S3 bucket.
138
+
139
+ Note, currently this specification is defined around "s3://" architecture with some support for "https://" endpoints; future versions may support other protocols.
140
+
141
+ ## 3.1 ID naming requirements
142
+
143
+ The **id** field can only contain alphanumeric characters, dashes, or underscores. No spaces or other characters are allowed. **id** should be unique across the HelioCloud network (to avoid namespace clashes, e.g. multiple datasets called '0094A' are to be avoided.)
144
+
145
+ The **id** field will match the cloudCatalog files but does not have to match the sub-bucket names. The cloudCatalog includes the **id**_YYYY.csv file indices and the optional **id**.json metadata file.
146
+
147
+ ## 3.2 Data items spanning multiple years
148
+
149
+ Some data or model outputs span multiple years. An output product that covers 3 years, for example (2011-2013) would be index in "id"_2011.csv (based on the start date of the data). A time-based search that is looking for 2012 coverage would therefore not be able to find it, as no "id"_2012.csv file exists or is needed. To accommodate these edge cases, the **multiyear** field should be set to True, which will allow subsequent search layers to be aware that long-duration files exist in this dataset.
150
+
151
+ This field should not be set for the typical case where a datafile extends slightly into the next year, but only when a datafile exists to provide data for a calendar year and that datafile would not be findable based purely on its given start date.
152
+
153
+
154
+ ## 3.2 Status codes
155
+
156
+ The default status code for a catalog.json item is "1200/OK" indicating the data is available. Status codes are informative and do not enforce any limits. They exist to communicate to users and client programs if a dataset is temporarily down or has other constraints.
157
+ Other status codes defined so far include:
158
+ * "code": "1200", "message": "OK": system is up and running
159
+ * "code": "1400", "message": "temporarily unavailable": providers can set this if they temporarily are doing maintenance or need to stop access due to costs
160
+
161
+ ## 3.3 Example
162
+
163
+ Here is an example catalog, for which only the first item has decided to fill out the optional 'ownership' block. If this was the GSFC catalog.json, it would reside at s3://gov-nasa-helio-public/catalog.json.
164
+
165
+ ```javascript
166
+ {
167
+ "version": "0.3",
168
+ "endpoint": "s3://gov-nasa-helio-public/",
169
+ "name": "GSFC HelioCloud",
170
+ "provider": "aws",
171
+ "region": "us-east-1",
172
+ "egress": "no-egress",
173
+ "contact": "Dr. Contact, dr_contact@example.com",
174
+ "description": "Optional description of this collection",
175
+ "citation": "Optional how to cite, preferably a DOI for the server",
176
+ "catalog":[
177
+ {
178
+ "id": "euvml",
179
+ "index": "s3://gov-nasa-helio-public/euvml/",
180
+ "title": "EUV-ML dataset",
181
+ "start": "1995-01-01T00:00.00Z",
182
+ "stop": "2022-01-01T00:00.00Z",
183
+ "modification": "2022-01-01T00:00.00Z",
184
+ "indextype": "csv",
185
+ "filetype": "fits",
186
+ "description": "Optional description for dataset",
187
+ "resource": "optional SPASE ID, DOI, URL, or json modelset",
188
+ "creation": "optional ISO 8601 date/time of the dataset creation",
189
+ "citation": "optional how to cite this dataset, DOI or similar",
190
+ "contact": "optional contact name",
191
+ "about": "optional website URL for info, team, etc"
192
+ },
193
+ {
194
+ "id": "mms_hmi",
195
+ "index": "s3://gov-nasa-helio-public/mms/hmi/",
196
+ "title": "MMS HMI data"
197
+ "start": "2015-01-01T00:00.00Z",
198
+ "stop": "2022-01-01T00:00.00Z",
199
+ "modification": "2022-01-01T00:00.00Z",
200
+ "indextype": "csv-zip",
201
+ "filetype": "cdf"
202
+ },
203
+ {
204
+ "id": "mms_feeps",
205
+ "index": "s3://gov-nasa-helio-public/mms/feeps/",
206
+ "title": "MMS FEEPS data"
207
+ "start": "2015-01-01T00:00.00Z",
208
+ "stop": "2022-01-01T00:00.00Z",
209
+ "modification": "2022-01-01T00:00.00Z",
210
+ "indextype": "csv-zip",
211
+ "filetype": "cdf"
212
+ },
213
+ {
214
+ "id": "fluxrope",
215
+ "index": "s3://heliotest/models/",
216
+ "title": "Instatiation of 3D fluxropes"
217
+ "start": "static",
218
+ "stop": "static",
219
+ "modification": "2022-01-01T00:00.00Z",
220
+ "indextype": "csv-zip",
221
+ "filetype": "cdf"
222
+ }
223
+ ],
224
+ "status": {
225
+ "code": 1200,
226
+ "message": "OK request successful"
227
+ }
228
+ }
229
+ ```
230
+
231
+ ## 3.4 Indexes should reside in same bucket as data
232
+
233
+ Index CSV files must be in the same bucket as the data, but do not have to be in the same sub-bucket or directory as their data. The catalog points to the index files, and the index files use absolute paths to point to the data items.
234
+
235
+ This also enables design a collection of datasets, wherein the data in the actual file catalog <ID>_<YYYY>.csv files can span sub-buckets. The use of absolute file paths is mandated.
236
+
237
+ ```
238
+ Example data itself is in:
239
+ s3://example/mms1/feeps/
240
+ s3://example/mms2/feeps/
241
+ s3://example/mms3/feeps/
242
+ s3://example/mms4/feeps/
243
+ ```
244
+
245
+ ```
246
+ Case 1: Matching
247
+ file catalog locations:
248
+ s3://example/mms1/feeps/mms1_feeps.CSV
249
+ s3://example/mms2/feeps/mms2_feeps.CSV
250
+ s3://example/mms3/feeps/mms3_feeps.CSV
251
+ s3://example/mms4/feeps/mms4_feeps.CSV
252
+ ```
253
+
254
+ ```
255
+ Case 2: Not Matching
256
+ file catalog locations:
257
+ s3://example/mms_all/feeps/mms_feeps.CSV (contents point to 4 subbuckets)
258
+ ```
259
+
260
+ The first case matches the idea of a 'dataset' and MUST be provided for any provided dataset. The second matches the idea of a 'collection' and is supported but not required (i.e. additional extra cloudCatalog endpoints do not have to match the underlying data structure).
261
+
262
+ ## 3.5 Concerns
263
+
264
+ Concerns were voiced about clashes if multiple people attempt to edit the catalog.json for a bucket at the same time. Our default HelioCloud will maintain the catalog.json contents in a serverless DynamoDB (which will handle transaction collisions and provide rollback), which then outputs the 'catalog.json' file that users and data requests use.
265
+
266
+ # 4 File Catalog
267
+
268
+ The file catalog consists of one index for each year of the dataset in either csv, zipped csv, or parquet format. The file must be in time sequence and the first four items must be the **start**, **stop**, **datakey**, and **filesize** fields in that order.
269
+
270
+ The index cloudCatalog is a set of CSV or Parquet files named "index"/"id"_YYYY.csv, "index"/"id"_YYYY.csv.zip or "index"/"id"_YYYY.parquet. For the case of static non-time sequence outputs, the index cloudCatalog are named "index"/"id"_static.csv (or .csv.zip or .parquet).
271
+
272
+ ## 4.1 Required Items
273
+
274
+ * **start**: string, Restricted ISO 8601 date/time of start for that data OR the word 'static' for items such as model shapes that lack a time field
275
+ * **stop**: string, Restricted ISO 8601 date/time of stop for that data OR the word 'static' for items such as model shapes that lack a time field
276
+ * **datakey**: string, full filename or S3 object identifier sufficient to actually obtain the file
277
+ * **filesize**: integer, file size in bytes
278
+
279
+ ## 4.2 Optional Items
280
+
281
+ * **checksum**: checksum for that file. If given, **checksum_algorithm** must also be listed
282
+ * **checksum_algorithm**: checksum algorithm used if checksums are generated. Examples include SHA, others.
283
+
284
+ ## 4.3 Optional Parameters
285
+
286
+ The default expected search capability is _filetime_ within requested time range.
287
+
288
+ Anything past **filesize** is fully optional; the minimal API expects only a start time, datakey aka filehandle, and file size IN THAT ORDER in the actual file index. It is up to individual client programs to do anything past that.
289
+
290
+ Any metadata included in this per-file index must be defined in the <id>.info json file to allow parseability.
291
+
292
+ The csv or zipped csv files may or may not include a one-line header that is prefaced by "#". It is the responsibility of client programs to determine if there is a skippable header or not.
293
+
294
+ Any optional parameters from the above list or the info JSON must be in the same order specified in the JSON.
295
+
296
+ For 'static' items, since the 'start' field will be the same constant value 'static' for all items, optional parameters to distinguish the items are suggested.
297
+
298
+ ## 4.4 Accessing
299
+
300
+ Users have 3 options with the cloudCatalog CSV file:
301
+ * download it directly and parse yourself,
302
+ * use our Python API (provided) to extract a subset of filehandles from CSV,
303
+ * use AWS Athena for queries
304
+
305
+ ## 4.5 Justification for Yearly CSV/Parquet Files
306
+
307
+ Unlike a database, yearly index files are both fetchable and parseable. They have a lower cost profile than the equivalent database, reasonably fast access, and allow for client and search programs independent of a specific database implementation.
308
+
309
+ Using yearly files rather than a single file or a database is to maintain an inexpensive, stateless, easily updated file index. In AWS, the "S3 Inventory" command can generate a list of filenames and datakeys, or filenames and datakeys since the last time inventory was run. Being able to add to the catalog index files incrementally is easier served if they are chunked into yearly files.
310
+
311
+ In addition, many use cases for long time baseline datasets will not need to access the entire multi-decadal span of the data, so parsing into years reduces the downloads needed to obtain the indices.
312
+
313
+ The use of CSV or Parquet also enables AWS Athena searches within the index with little overhead, so long as optional metadata is provided.
314
+
315
+ ## 4.6 Example File Catalog
316
+
317
+ Here is a short minimal CSV example index file.
318
+
319
+ ```
320
+ # start, stop, datakey, filesize
321
+ '2010-05-08T12:05:30.000Z','2010-05-08T12:06:14.000Z','s3://edu-apl-helio-public/euvml/stereo/a/195/20100508_120530_n4euA.fts','246000'
322
+ '2010-05-08T12:06:15.000Z','2010-05-08T12:10:29.00Z','s3://edu-apl-helio-public/euvml/stereo/a/195/20100508_120615_n4euA.fts','246000'
323
+ '2010-05-08T12:10:30.000Z','2010-05-08T12:14:29.000Z','s3://edu-apl-helio-public/euvml/stereo/a/195/20100508_121030_n4euA.fts','246000'
324
+ ```
325
+
326
+ Here is an example with additional metadata and a CSV header as well.
327
+
328
+ ```
329
+ # start, stop, datakey, filesize, wavelength, carr_lon, carr_lat
330
+ '2010-05-08T12:05:30.000Z','2010-05-08T12:06:14.000Z','s3://edu-apl-helio-public/euvml/stereo/a/195/20100508_120530_n4euA.fts','246000','195','20.4','30.0'
331
+ '2010-05-08T12:06:15.000Z','2010-05-08T12:10:29.000Z','s3://edu-apl-helio-public/euvml/stereo/a/195/20100508_120615_n4euA.fts','246000','195','21.8','30.0'
332
+ '2010-05-08T12:10:30.000Z','2010-05-08T12:14:29.000Z','s3://edu-apl-helio-public/euvml/stereo/a/195/20100508_121030_n4euA.fts','246000','195','22.4','30.0'
333
+ ```
334
+ Here is an example with additional metadata and a CSV header as the EUV-ML project would like. Items in CAPS are directly from FITS keywords.
335
+ ```
336
+ # start, stop, datakey, filesize, spacecraft, instrument, WAVELNTH, CRLT_OBS, CRLN_OBS, CRPIX1, CRPIX2, RSUN, quality, generation_flag
337
+ '2010-05-08T12:05:30.000Z','2010-05-08T12:06:14.000Z','s3://edu-apl-helio-public/euvml/stereo/a/195/20100508_120530_n4euA.fts','246000','A','euvi','195,45.0, 23.1, 512, 510, 26.5, 1, 1
338
+ '2010-05-08T12:06:15.000Z','2010-05-08T12:10:29.000Z','s3://edu-apl-helio-public/euvml/stereo/a/195/20100508_120615_n4euA.fts','246000','A','euvi','195',45.0, 23.1, 512, 510, 26.5, 1, 1
339
+ '2010-05-08T12:10:30.000Z','2010-05-08T12:14:29.000Z','s3://edu-apl-helio-public/euvml/stereo/a/195/20100508_121030_n4euA.fts','246000','A','euvi','195',45.0, 23.1, 512, 510, 26.5, 1, 1
340
+ ```
341
+
342
+ # 5 Time Specification and ISO 8601
343
+
344
+ Time values are always strings, and the SCR Time format (taken from the HAPI Time format) is a subset of the ISO 8601 standard. The restriction on the ISO 8601 standard is that time must be represented as
345
+
346
+ ```
347
+ yyyy-mm-ddThh:mm:ss.sssZ
348
+ ```
349
+
350
+ and the trailing Z is required. Strings with less precision are allowed as per ISO 8601. Any date or time elements missing from the string are assumed to take on their smallest possible value. For example, the string 2017-01-15T23:00:00.000Z could be given in truncated form as 2017-01-15T23:00Z. A dataset must use only one format and length within that given dataset. The times values must not have any local time zone offset, and they must indicate this by including the trailing Z.
351
+
352
+
353
+ # 6 Info Metadata
354
+
355
+ Each dataset may also include an optional info json file that gives the time range, date last modified, ownership information, and optional additional metadata for that dataset. The files are by default searchable and selectable by time window. Additional search capability is not within scope of the file catalog per se, but data providers can indicate metadata for adding a search layer.
356
+
357
+ If an <id>.info file exists and lists additional parameters, the resulting catalog index files must contain those parameters in the same order as expressed in this json file.
358
+
359
+ ## 6.1 Optional Items
360
+
361
+ * **parameters**: optional list of searchable parameters available in the actual catalog index files
362
+ file.
363
+ (* = available from S3 Inventory)
364
+
365
+ ## 6.2 Example
366
+
367
+ Here is an example for a sample optional Info json file. This is used to indicate additional searchable metadata that exists within the catalog index files, and enables higher searchability in datasets.
368
+
369
+ ```javascript
370
+ {
371
+ "version": "0.3",
372
+ "parameters": [
373
+ {"name": "spacecraft", "type": "string"},
374
+ {"name": "wavelength", "type": "int", "units": "Angstroms"},
375
+ {"name": "crlt", "type": "double", "units": "degrees", "desc": "Carrington latitude"},
376
+ {"name": "crln", "type": "double", "units": "degrees", "desc": "Carrington longitude"},
377
+ {"name": "rsun", "type": "double", "units": "pixels", "desc": "Size of sun in pizels"},
378
+ {"name": "crpix1", "type": "integer", "units": "pixels", "desc": "x coord of sun center"},
379
+ {"name": "crpix2", "type": "integer", "units": "pixels", "desc": "x coord of sun center"},
380
+ {"name": "quality", "type": "integer", "desc": "data quality and level of interpolation"}
381
+ ]
382
+ }
383
+ ```
384
+
385
+ ## 7.0 Changes from 0.4 to 0.5
386
+
387
+ Changed
388
+ ```
389
+ Added 'stop' as a mandatory field
390
+ ```