caltechdata-api 2.1.0__tar.gz → 2.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/PKG-INFO +1 -1
  2. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/__init__.py +4 -0
  3. caltechdata_api-2.2.0/caltechdata_api/download_from_record.py +236 -0
  4. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api.egg-info/PKG-INFO +1 -1
  5. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api.egg-info/SOURCES.txt +1 -1
  6. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/pyproject.toml +1 -1
  7. caltechdata_api-2.1.0/caltechdata_api/get_files.py +0 -49
  8. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/LICENSE +0 -0
  9. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/README.md +0 -0
  10. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/caltechdata_edit.py +0 -0
  11. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/caltechdata_write.py +0 -0
  12. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/cli.py +0 -0
  13. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/customize_schema.py +0 -0
  14. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/download_file.py +0 -0
  15. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/get_metadata.py +0 -0
  16. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/md_to_json.py +0 -0
  17. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/utils.py +0 -0
  18. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/vocabularies/date_types.yaml +0 -0
  19. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/vocabularies/description_types.yaml +0 -0
  20. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/vocabularies/identifier_types.yaml +0 -0
  21. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/vocabularies/licenses.csv +0 -0
  22. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/vocabularies/relation_types.yaml +0 -0
  23. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/vocabularies/resource_types.yaml +0 -0
  24. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/vocabularies/roles.yaml +0 -0
  25. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/vocabularies/title_types.yaml +0 -0
  26. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api/vocabularies.yaml +0 -0
  27. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api.egg-info/dependency_links.txt +0 -0
  28. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api.egg-info/entry_points.txt +0 -0
  29. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api.egg-info/requires.txt +0 -0
  30. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/caltechdata_api.egg-info/top_level.txt +0 -0
  31. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/setup.cfg +0 -0
  32. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/tests/test_download.py +0 -0
  33. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/tests/test_rdm.py +0 -0
  34. {caltechdata_api-2.1.0 → caltechdata_api-2.2.0}/tests/test_unit.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: caltechdata_api
3
- Version: 2.1.0
3
+ Version: 2.2.0
4
4
  Summary: Python wrapper for CaltechDATA API.
5
5
  Author: Rohan Bhattarai, Elizabeth Won
6
6
  Author-email: Thomas E Morrell <tmorrell@caltech.edu>, Alexander A Abakah <aabakah@caltech.edu>, Kshemaahna Nagi <knagi@caltech.edu>
@@ -13,5 +13,9 @@ from .caltechdata_edit import (
13
13
  from .customize_schema import customize_schema, validate_metadata
14
14
  from .get_metadata import get_metadata
15
15
  from .download_file import download_file, download_url
16
+ from .download_from_record import (
17
+ get_files_from_record,
18
+ download_files_from_record,
19
+ )
16
20
  from .utils import humanbytes
17
21
  from .md_to_json import parse_readme_to_json
@@ -0,0 +1,236 @@
1
+ """Functions to download files directly from a CaltechData record.
2
+
3
+ A CaltechData record ID is the ten characters, separated into two groups
4
+ of five by a dash, found at the end of a CaltechData URL or default DOI.
5
+ For instance, in the URL ``https://data.caltech.edu/records/4rgh7-zss31``
6
+ or DOI ``10.22002/4rgh7-zss31``, the record ID is ``4rgh7-zss31``.
7
+
8
+ The same functions can download from CaltechAUTHORS by passing
9
+ ``authors=True``, and from the test (development) instances of either
10
+ repository by passing ``production=False``.
11
+ """
12
+
13
+ import argparse
14
+ import os
15
+ import requests
16
+ from tqdm.auto import tqdm
17
+
18
+ from typing import Any, Dict, Optional, Sequence
19
+
20
+
21
+ def get_base_url(production: bool = True, authors: bool = False) -> str:
22
+ """Get the base URL of the repository to download from.
23
+
24
+ Parameters
25
+ ----------
26
+ production
27
+ Whether to use the production repository. If ``False``, the
28
+ test (development) instance is used instead.
29
+
30
+ authors
31
+ Whether to use CaltechAUTHORS instead of CaltechDATA.
32
+
33
+ Returns
34
+ -------
35
+ str
36
+ The base URL, without a trailing slash.
37
+ """
38
+ if production:
39
+ if authors:
40
+ return "https://authors.library.caltech.edu"
41
+ return "https://data.caltech.edu"
42
+ else:
43
+ if authors:
44
+ return "https://authors.caltechlibrary.dev"
45
+ return "https://data.caltechlibrary.dev"
46
+
47
+
48
+ def get_files_from_record(
49
+ record_id: str, production: bool = True, authors: bool = False
50
+ ) -> Dict[str, Any]:
51
+ """Get a dictionary of files associated with a record.
52
+
53
+ Parameters
54
+ ----------
55
+ record_id
56
+ The record ID for which to obtain the list of files. See
57
+ the module documentation for how to find the record ID.
58
+
59
+ production
60
+ Whether to query the production repository. If ``False``, the
61
+ test (development) instance is used instead.
62
+
63
+ authors
64
+ Whether to query CaltechAUTHORS instead of CaltechDATA.
65
+
66
+ Returns
67
+ -------
68
+ dict
69
+ A dictionary with the file names as keys and the metadata
70
+ associated with those files as values.
71
+ """
72
+ base_url = get_base_url(production=production, authors=authors)
73
+ with requests.get(f"{base_url}/api/records/{record_id}/files") as r:
74
+ r.raise_for_status()
75
+ files = dict()
76
+ for entry in r.json().get("entries", []):
77
+ key = entry.pop("key")
78
+ files[key] = entry
79
+
80
+ return files
81
+
82
+
83
+ def download_files_from_record(
84
+ record_id: str,
85
+ output_path: os.PathLike,
86
+ filenames: Optional[Sequence[str]] = None,
87
+ max_redirects: int = 5,
88
+ production: bool = True,
89
+ authors: bool = False,
90
+ ):
91
+ """Download one or more files from a record.
92
+
93
+ Parameters
94
+ ----------
95
+ record_id
96
+ The record ID from which to download the file. See
97
+ the module documentation for how to find the record ID.
98
+
99
+ output_path
100
+ Where to download the files. Must be an existing directory.
101
+ If you want control over exactly how each file is named,
102
+ use :func:`get_files_from_record` to get the URLs for each
103
+ file, then manually call :func:`download_content` for each
104
+ file you wish to download.
105
+
106
+ filenames
107
+ The name of the files in the record. Must match exactly.
108
+ If you are not sure of the filenames in the record,
109
+ they are the keys of the dictionary returned by
110
+ :func:`get_files_from_record`. If omitted, all files
111
+ are downloaded.
112
+
113
+ max_redirects
114
+ If the record's link to the file content returns a
115
+ redirection to another location, this function will follow
116
+ that redirection. This parameter sets the maximum number of
117
+ hops that the function can take. Usually, only a single
118
+ redirection should be necessary (from the record to the file
119
+ provider), but the default allows for a few extra hops.
120
+
121
+ production
122
+ Whether to download from the production repository. If ``False``,
123
+ the test (development) instance is used instead.
124
+
125
+ authors
126
+ Whether to download from CaltechAUTHORS instead of CaltechDATA.
127
+ """
128
+ if not os.path.isdir(output_path):
129
+ raise IOError(f"{output_path} is not an extant directory")
130
+
131
+ files = get_files_from_record(record_id, production=production, authors=authors)
132
+ if filenames is None:
133
+ filenames = sorted(files.keys())
134
+
135
+ for filename in filenames:
136
+ output_file = os.path.join(output_path, filename)
137
+ if filename not in files:
138
+ raise KeyError(f'File "{filename}" not found in record {record_id}')
139
+
140
+ entry = files[filename]
141
+ if "links" not in entry or "content" not in entry["links"]:
142
+ raise NotImplementedError(
143
+ f'Metadata for file "{filename}" in record {record_id} is '
144
+ "missing the content link"
145
+ )
146
+
147
+ content_url = entry["links"]["content"]
148
+ download_content(content_url, output_file, max_redirects=max_redirects)
149
+
150
+
151
+ def download_content(content_url: str, fname: os.PathLike, max_redirects=5):
152
+ """Download the contents of a file.
153
+
154
+ Parameters
155
+ ----------
156
+ content_url
157
+ The URL pointing to the contents of the file. In the dictionary
158
+ returned by :func:`get_files_from_record` (call it ``D``), this
159
+ is usually the value at ``D[FILENAME]["links"]["content"]``.
160
+
161
+ fname
162
+ The path to which to save the file contents. Will be overwritten!
163
+
164
+
165
+ max_redirects
166
+ If the ``content_url`` returns a redirection to another location,
167
+ this function will follow that redirection. This parameter sets
168
+ the maximum number of hops that the function can take. Usually,
169
+ only a single redirection should be necessary (from the record to
170
+ the file provider), but the default allows for a few extra hops.
171
+ """
172
+ url = content_url
173
+ # If max_redirects == 0, then this loop will never run. Since the
174
+ # likely expected behavior for ``max_redirects = 0`` is to only
175
+ # look at the link directly given back by the record, we add 1
176
+ # for the loop to ensure it runs once in that case.
177
+ for _ in range(max_redirects + 1):
178
+ with requests.get(url, stream=True) as r:
179
+ r.raise_for_status()
180
+ if "Location" in r.headers:
181
+ # Redirection - follow to the next URL
182
+ url = r.headers["Location"]
183
+ else:
184
+ with open(fname, "wb") as f:
185
+ content_length = r.headers.get("content-length", None)
186
+ if content_length is not None:
187
+ content_length = int(content_length)
188
+ pbar = tqdm(total=content_length // 1024, unit="kB")
189
+ for chunk in r.iter_content(chunk_size=1024):
190
+ if chunk:
191
+ pbar.update()
192
+ f.write(chunk)
193
+ print(f"Download to {fname} complete.")
194
+ return
195
+ raise RuntimeError(f"Exceeded maximum number of redirects ({max_redirects})")
196
+
197
+
198
+ if __name__ == "__main__":
199
+ parser = argparse.ArgumentParser(
200
+ description="download_from_record downloads the files attached to a "
201
+ "CaltechDATA or CaltechAUTHORS record"
202
+ )
203
+ parser.add_argument(
204
+ "record_id", help="The record ID for the record to download files from"
205
+ )
206
+ parser.add_argument(
207
+ "filenames",
208
+ nargs="*",
209
+ default=None,
210
+ help="The names of the files to download. If omitted, all files in "
211
+ "the record are downloaded.",
212
+ )
213
+ parser.add_argument(
214
+ "-output_path",
215
+ default=".",
216
+ help="Existing directory to download the files into",
217
+ )
218
+ parser.add_argument(
219
+ "-max_redirects",
220
+ type=int,
221
+ default=5,
222
+ help="Maximum number of redirects to follow per file",
223
+ )
224
+ parser.add_argument("-test", dest="production", action="store_false")
225
+ parser.add_argument("-authors", dest="authors", action="store_true")
226
+
227
+ args = parser.parse_args()
228
+
229
+ download_files_from_record(
230
+ args.record_id,
231
+ args.output_path,
232
+ filenames=args.filenames or None,
233
+ max_redirects=args.max_redirects,
234
+ production=args.production,
235
+ authors=args.authors,
236
+ )
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: caltechdata_api
3
- Version: 2.1.0
3
+ Version: 2.2.0
4
4
  Summary: Python wrapper for CaltechDATA API.
5
5
  Author: Rohan Bhattarai, Elizabeth Won
6
6
  Author-email: Thomas E Morrell <tmorrell@caltech.edu>, Alexander A Abakah <aabakah@caltech.edu>, Kshemaahna Nagi <knagi@caltech.edu>
@@ -7,7 +7,7 @@ caltechdata_api/caltechdata_write.py
7
7
  caltechdata_api/cli.py
8
8
  caltechdata_api/customize_schema.py
9
9
  caltechdata_api/download_file.py
10
- caltechdata_api/get_files.py
10
+ caltechdata_api/download_from_record.py
11
11
  caltechdata_api/get_metadata.py
12
12
  caltechdata_api/md_to_json.py
13
13
  caltechdata_api/utils.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "caltechdata_api"
7
- version = "2.1.0"
7
+ version = "2.2.0"
8
8
  description = "Python wrapper for CaltechDATA API."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.9"
@@ -1,49 +0,0 @@
1
- import argparse
2
- import requests
3
-
4
-
5
- def get_files(idv, production=True):
6
- # Returns file block
7
-
8
- if production == True:
9
- api_url = "https://data.caltech.edu/api/records/"
10
- else:
11
- api_url = "https://data.caltechlibrary.dev/api/records/"
12
-
13
- r = requests.get(api_url + str(idv) + "/files")
14
- r_data = r.json()
15
- if "message" in r_data:
16
- raise AssertionError(
17
- "id "
18
- + str(idv)
19
- + " expected http status 200, got "
20
- + str(r.status_code)
21
- + " "
22
- + r_data["message"]
23
- )
24
- if not "entries" in r_data:
25
- raise AssertionError("expected as entries property in response, got " + r_data)
26
- return r_data["entries"]
27
-
28
-
29
- if __name__ == "__main__":
30
- parser = argparse.ArgumentParser(
31
- description="get_files queries the caltechDATA (Invenio 3) API\
32
- and returns file information"
33
- )
34
- parser.add_argument(
35
- "ids",
36
- metavar="ID",
37
- type=str,
38
- nargs="+",
39
- help="The CaltechDATA ID for each record of interest",
40
- )
41
- parser.add_argument("-test", dest="production", action="store_false")
42
-
43
- args = parser.parse_args()
44
-
45
- production = args.production
46
-
47
- for idv in args.ids:
48
- metadata = get_files(idv, production)
49
- print(metadata)
File without changes