datalake 1.1__tar.gz → 2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. datalake-2.2/LICENSE +201 -0
  2. datalake-2.2/PKG-INFO +33 -0
  3. {datalake-1.1 → datalake-2.2}/datalake/_version.py +3 -3
  4. {datalake-1.1 → datalake-2.2}/datalake/archive.py +100 -79
  5. {datalake-1.1 → datalake-2.2}/datalake/common/record.py +58 -51
  6. {datalake-1.1 → datalake-2.2}/datalake/logging_helpers.py +1 -1
  7. {datalake-1.1 → datalake-2.2}/datalake/tests/conftest.py +25 -11
  8. datalake-2.2/datalake.egg-info/PKG-INFO +33 -0
  9. {datalake-1.1 → datalake-2.2}/datalake.egg-info/SOURCES.txt +1 -0
  10. datalake-2.2/datalake.egg-info/entry_points.txt +2 -0
  11. {datalake-1.1 → datalake-2.2}/datalake.egg-info/requires.txt +2 -2
  12. {datalake-1.1 → datalake-2.2}/setup.cfg +1 -1
  13. {datalake-1.1 → datalake-2.2}/setup.py +2 -2
  14. {datalake-1.1 → datalake-2.2}/test/test_archive.py +13 -9
  15. {datalake-1.1 → datalake-2.2}/test/test_cli.py +8 -7
  16. {datalake-1.1 → datalake-2.2}/test/test_fetch.py +24 -0
  17. {datalake-1.1 → datalake-2.2}/test/test_queue.py +24 -9
  18. datalake-1.1/PKG-INFO +0 -17
  19. datalake-1.1/datalake.egg-info/PKG-INFO +0 -17
  20. datalake-1.1/datalake.egg-info/entry_points.txt +0 -4
  21. {datalake-1.1 → datalake-2.2}/MANIFEST.in +0 -0
  22. {datalake-1.1 → datalake-2.2}/README.md +0 -0
  23. {datalake-1.1 → datalake-2.2}/datalake/__init__.py +0 -0
  24. {datalake-1.1 → datalake-2.2}/datalake/common/__init__.py +0 -0
  25. {datalake-1.1 → datalake-2.2}/datalake/common/conf.py +0 -0
  26. {datalake-1.1 → datalake-2.2}/datalake/common/errors.py +0 -0
  27. {datalake-1.1 → datalake-2.2}/datalake/common/metadata.py +0 -0
  28. {datalake-1.1 → datalake-2.2}/datalake/config_helpers.py +0 -0
  29. {datalake-1.1 → datalake-2.2}/datalake/crtime.py +0 -0
  30. {datalake-1.1 → datalake-2.2}/datalake/dlfile.py +0 -0
  31. {datalake-1.1 → datalake-2.2}/datalake/queue.py +0 -0
  32. {datalake-1.1 → datalake-2.2}/datalake/scripts/__init__.py +0 -0
  33. {datalake-1.1 → datalake-2.2}/datalake/scripts/cli.py +0 -0
  34. {datalake-1.1 → datalake-2.2}/datalake/tests/__init__.py +0 -0
  35. {datalake-1.1 → datalake-2.2}/datalake/tests/test_conf.py +0 -0
  36. {datalake-1.1 → datalake-2.2}/datalake/tests/test_metadata.py +0 -0
  37. {datalake-1.1 → datalake-2.2}/datalake/tests/test_record.py +0 -0
  38. {datalake-1.1 → datalake-2.2}/datalake/translator.py +0 -0
  39. {datalake-1.1 → datalake-2.2}/datalake.egg-info/dependency_links.txt +0 -0
  40. {datalake-1.1 → datalake-2.2}/datalake.egg-info/top_level.txt +0 -0
  41. {datalake-1.1 → datalake-2.2}/test/test_config.py +0 -0
  42. {datalake-1.1 → datalake-2.2}/test/test_crtime.py +0 -0
  43. {datalake-1.1 → datalake-2.2}/test/test_file.py +0 -0
  44. {datalake-1.1 → datalake-2.2}/test/test_latest.py +0 -0
  45. {datalake-1.1 → datalake-2.2}/test/test_list.py +0 -0
  46. {datalake-1.1 → datalake-2.2}/test/test_logging.py +0 -0
  47. {datalake-1.1 → datalake-2.2}/test/test_translator.py +0 -0
  48. {datalake-1.1 → datalake-2.2}/versioneer.py +0 -0
datalake-2.2/LICENSE ADDED
@@ -0,0 +1,201 @@
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction,
10
+ and distribution as defined by Sections 1 through 9 of this document.
11
+
12
+ "Licensor" shall mean the copyright owner or entity authorized by
13
+ the copyright owner that is granting the License.
14
+
15
+ "Legal Entity" shall mean the union of the acting entity and all
16
+ other entities that control, are controlled by, or are under common
17
+ control with that entity. For the purposes of this definition,
18
+ "control" means (i) the power, direct or indirect, to cause the
19
+ direction or management of such entity, whether by contract or
20
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21
+ outstanding shares, or (iii) beneficial ownership of such entity.
22
+
23
+ "You" (or "Your") shall mean an individual or Legal Entity
24
+ exercising permissions granted by this License.
25
+
26
+ "Source" form shall mean the preferred form for making modifications,
27
+ including but not limited to software source code, documentation
28
+ source, and configuration files.
29
+
30
+ "Object" form shall mean any form resulting from mechanical
31
+ transformation or translation of a Source form, including but
32
+ not limited to compiled object code, generated documentation,
33
+ and conversions to other media types.
34
+
35
+ "Work" shall mean the work of authorship, whether in Source or
36
+ Object form, made available under the License, as indicated by a
37
+ copyright notice that is included in or attached to the work
38
+ (an example is provided in the Appendix below).
39
+
40
+ "Derivative Works" shall mean any work, whether in Source or Object
41
+ form, that is based on (or derived from) the Work and for which the
42
+ editorial revisions, annotations, elaborations, or other modifications
43
+ represent, as a whole, an original work of authorship. For the purposes
44
+ of this License, Derivative Works shall not include works that remain
45
+ separable from, or merely link (or bind by name) to the interfaces of,
46
+ the Work and Derivative Works thereof.
47
+
48
+ "Contribution" shall mean any work of authorship, including
49
+ the original version of the Work and any modifications or additions
50
+ to that Work or Derivative Works thereof, that is intentionally
51
+ submitted to Licensor for inclusion in the Work by the copyright owner
52
+ or by an individual or Legal Entity authorized to submit on behalf of
53
+ the copyright owner. For the purposes of this definition, "submitted"
54
+ means any form of electronic, verbal, or written communication sent
55
+ to the Licensor or its representatives, including but not limited to
56
+ communication on electronic mailing lists, source code control systems,
57
+ and issue tracking systems that are managed by, or on behalf of, the
58
+ Licensor for the purpose of discussing and improving the Work, but
59
+ excluding communication that is conspicuously marked or otherwise
60
+ designated in writing by the copyright owner as "Not a Contribution."
61
+
62
+ "Contributor" shall mean Licensor and any individual or Legal Entity
63
+ on behalf of whom a Contribution has been received by Licensor and
64
+ subsequently incorporated within the Work.
65
+
66
+ 2. Grant of Copyright License. Subject to the terms and conditions of
67
+ this License, each Contributor hereby grants to You a perpetual,
68
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69
+ copyright license to reproduce, prepare Derivative Works of,
70
+ publicly display, publicly perform, sublicense, and distribute the
71
+ Work and such Derivative Works in Source or Object form.
72
+
73
+ 3. Grant of Patent License. Subject to the terms and conditions of
74
+ this License, each Contributor hereby grants to You a perpetual,
75
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76
+ (except as stated in this section) patent license to make, have made,
77
+ use, offer to sell, sell, import, and otherwise transfer the Work,
78
+ where such license applies only to those patent claims licensable
79
+ by such Contributor that are necessarily infringed by their
80
+ Contribution(s) alone or by combination of their Contribution(s)
81
+ with the Work to which such Contribution(s) was submitted. If You
82
+ institute patent litigation against any entity (including a
83
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84
+ or a Contribution incorporated within the Work constitutes direct
85
+ or contributory patent infringement, then any patent licenses
86
+ granted to You under this License for that Work shall terminate
87
+ as of the date such litigation is filed.
88
+
89
+ 4. Redistribution. You may reproduce and distribute copies of the
90
+ Work or Derivative Works thereof in any medium, with or without
91
+ modifications, and in Source or Object form, provided that You
92
+ meet the following conditions:
93
+
94
+ (a) You must give any other recipients of the Work or
95
+ Derivative Works a copy of this License; and
96
+
97
+ (b) You must cause any modified files to carry prominent notices
98
+ stating that You changed the files; and
99
+
100
+ (c) You must retain, in the Source form of any Derivative Works
101
+ that You distribute, all copyright, patent, trademark, and
102
+ attribution notices from the Source form of the Work,
103
+ excluding those notices that do not pertain to any part of
104
+ the Derivative Works; and
105
+
106
+ (d) If the Work includes a "NOTICE" text file as part of its
107
+ distribution, then any Derivative Works that You distribute must
108
+ include a readable copy of the attribution notices contained
109
+ within such NOTICE file, excluding those notices that do not
110
+ pertain to any part of the Derivative Works, in at least one
111
+ of the following places: within a NOTICE text file distributed
112
+ as part of the Derivative Works; within the Source form or
113
+ documentation, if provided along with the Derivative Works; or,
114
+ within a display generated by the Derivative Works, if and
115
+ wherever such third-party notices normally appear. The contents
116
+ of the NOTICE file are for informational purposes only and
117
+ do not modify the License. You may add Your own attribution
118
+ notices within Derivative Works that You distribute, alongside
119
+ or as an addendum to the NOTICE text from the Work, provided
120
+ that such additional attribution notices cannot be construed
121
+ as modifying the License.
122
+
123
+ You may add Your own copyright statement to Your modifications and
124
+ may provide additional or different license terms and conditions
125
+ for use, reproduction, or distribution of Your modifications, or
126
+ for any such Derivative Works as a whole, provided Your use,
127
+ reproduction, and distribution of the Work otherwise complies with
128
+ the conditions stated in this License.
129
+
130
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131
+ any Contribution intentionally submitted for inclusion in the Work
132
+ by You to the Licensor shall be under the terms and conditions of
133
+ this License, without any additional terms or conditions.
134
+ Notwithstanding the above, nothing herein shall supersede or modify
135
+ the terms of any separate license agreement you may have executed
136
+ with Licensor regarding such Contributions.
137
+
138
+ 6. Trademarks. This License does not grant permission to use the trade
139
+ names, trademarks, service marks, or product names of the Licensor,
140
+ except as required for reasonable and customary use in describing the
141
+ origin of the Work and reproducing the content of the NOTICE file.
142
+
143
+ 7. Disclaimer of Warranty. Unless required by applicable law or
144
+ agreed to in writing, Licensor provides the Work (and each
145
+ Contributor provides its Contributions) on an "AS IS" BASIS,
146
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147
+ implied, including, without limitation, any warranties or conditions
148
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149
+ PARTICULAR PURPOSE. You are solely responsible for determining the
150
+ appropriateness of using or redistributing the Work and assume any
151
+ risks associated with Your exercise of permissions under this License.
152
+
153
+ 8. Limitation of Liability. In no event and under no legal theory,
154
+ whether in tort (including negligence), contract, or otherwise,
155
+ unless required by applicable law (such as deliberate and grossly
156
+ negligent acts) or agreed to in writing, shall any Contributor be
157
+ liable to You for damages, including any direct, indirect, special,
158
+ incidental, or consequential damages of any character arising as a
159
+ result of this License or out of the use or inability to use the
160
+ Work (including but not limited to damages for loss of goodwill,
161
+ work stoppage, computer failure or malfunction, or any and all
162
+ other commercial damages or losses), even if such Contributor
163
+ has been advised of the possibility of such damages.
164
+
165
+ 9. Accepting Warranty or Additional Liability. While redistributing
166
+ the Work or Derivative Works thereof, You may choose to offer,
167
+ and charge a fee for, acceptance of support, warranty, indemnity,
168
+ or other liability obligations and/or rights consistent with this
169
+ License. However, in accepting such obligations, You may act only
170
+ on Your own behalf and on Your sole responsibility, not on behalf
171
+ of any other Contributor, and only if You agree to indemnify,
172
+ defend, and hold each Contributor harmless for any liability
173
+ incurred by, or claims asserted against, such Contributor by reason
174
+ of your accepting any such warranty or additional liability.
175
+
176
+ END OF TERMS AND CONDITIONS
177
+
178
+ APPENDIX: How to apply the Apache License to your work.
179
+
180
+ To apply the Apache License to your work, attach the following
181
+ boilerplate notice, with the fields enclosed by brackets "{}"
182
+ replaced with your own identifying information. (Don't include
183
+ the brackets!) The text should be enclosed in the appropriate
184
+ comment syntax for the file format. We also recommend that a
185
+ file or class name and description of purpose be included on the
186
+ same "printed page" as the copyright notice for easier
187
+ identification within third-party archives.
188
+
189
+ Copyright 2016 Planet Labs Inc.
190
+
191
+ Licensed under the Apache License, Version 2.0 (the "License");
192
+ you may not use this file except in compliance with the License.
193
+ You may obtain a copy of the License at
194
+
195
+ http://www.apache.org/licenses/LICENSE-2.0
196
+
197
+ Unless required by applicable law or agreed to in writing, software
198
+ distributed under the License is distributed on an "AS IS" BASIS,
199
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
200
+ See the License for the specific language governing permissions and
201
+ limitations under the License.
datalake-2.2/PKG-INFO ADDED
@@ -0,0 +1,33 @@
1
+ Metadata-Version: 2.1
2
+ Name: datalake
3
+ Version: 2.2
4
+ Summary: datalake: a metadata-aware archive
5
+ Home-page: https://github.com/planetlabs/datalake
6
+ Author: Brian Cavagnolo
7
+ Author-email: brian@planet.com
8
+ Classifier: Programming Language :: Python :: 3.6
9
+ Classifier: Programming Language :: Python :: 3.7
10
+ Classifier: Programming Language :: Python :: 3.8
11
+ Classifier: Programming Language :: Python :: 3.9
12
+ License-File: LICENSE
13
+ Requires-Dist: boto3>=1.9.68
14
+ Requires-Dist: memoized_property>=1.0.1
15
+ Requires-Dist: pyblake2>=0.9.3; python_version < "3.6"
16
+ Requires-Dist: click>=4.1
17
+ Requires-Dist: python-dotenv>=0.1.3
18
+ Requires-Dist: requests>=2.5
19
+ Requires-Dist: six>=1.10.0
20
+ Requires-Dist: python-dateutil>=2.4.2
21
+ Requires-Dist: pytz>=2015.4
22
+ Provides-Extra: test
23
+ Requires-Dist: pytest<8.0.0; extra == "test"
24
+ Requires-Dist: moto[s3]<5,>4; extra == "test"
25
+ Requires-Dist: twine<4.0.0; extra == "test"
26
+ Requires-Dist: pip<22.0.0,>=20.0.0; extra == "test"
27
+ Requires-Dist: wheel<0.38.0; extra == "test"
28
+ Requires-Dist: flake8<4.1,>=2.5.0; extra == "test"
29
+ Requires-Dist: responses<0.22.0; extra == "test"
30
+ Provides-Extra: queuable
31
+ Requires-Dist: pyinotify>=0.9.4; extra == "queuable"
32
+ Provides-Extra: sentry
33
+ Requires-Dist: raven>=5.0.0; extra == "sentry"
@@ -8,11 +8,11 @@ import json
8
8
 
9
9
  version_json = '''
10
10
  {
11
- "date": "2022-05-06T09:52:24-0700",
11
+ "date": "2024-01-10T16:28:12-0800",
12
12
  "dirty": false,
13
13
  "error": null,
14
- "full-revisionid": "4215d7f1982009e5597f6f8e1423a8e728eac1bf",
15
- "version": "1.1"
14
+ "full-revisionid": "b1ffd3c6211b7d759ceb70f8f07bfee4ac21e93a",
15
+ "version": "2.2"
16
16
  }
17
17
  ''' # END VERSION_JSON
18
18
 
@@ -28,10 +28,10 @@ from .common import Metadata
28
28
  import requests
29
29
  from io import BytesIO
30
30
  import errno
31
+ from copy import deepcopy
32
+ from datetime import datetime
31
33
 
32
- from boto.s3.connection import S3Connection
33
- from boto.s3.key import Key
34
- from boto.s3.connection import NoHostProvided
34
+ import boto3
35
35
  import math
36
36
  from logging import getLogger
37
37
  log = getLogger('datalake-archive')
@@ -210,7 +210,26 @@ class Archive(object):
210
210
  return self.url_from_file(f)
211
211
 
212
212
  def _upload_file(self, f):
213
- key = self._s3_key_from_metadata(f)
213
+
214
+ # Implementation inspired by https://stackoverflow.com/a/60892027
215
+ obj = self._s3_object_from_metadata(f)
216
+
217
+ # NB: we have an opportunitiy to turn on threading here, which may
218
+ # improve performance. However, in some cases (i.e., queue-based
219
+ # uploader) we already use threads. So let's add it later as a
220
+ # configuration if/when we want to experiment.
221
+ config = boto3.s3.transfer.TransferConfig(
222
+ # All sizes are bytes
223
+ multipart_threshold=CHUNK_SIZE(),
224
+ use_threads=False,
225
+ multipart_chunksize=CHUNK_SIZE(),
226
+ )
227
+
228
+ extra = {
229
+ 'Metadata': {
230
+ METADATA_NAME: json.dumps(f.metadata)
231
+ }
232
+ }
214
233
 
215
234
  spos = f.tell()
216
235
  f.seek(0, os.SEEK_END)
@@ -220,38 +239,22 @@ class Archive(object):
220
239
 
221
240
  num_chunks = int(math.ceil(f_size / float(CHUNK_SIZE())))
222
241
  log.info("Uploading {} ({} B / {} chunks)".format(
223
- key.name, f_size, num_chunks))
224
- if num_chunks <= 1:
225
- key.set_metadata(METADATA_NAME, json.dumps(f.metadata))
226
- completed_size = key.set_contents_from_file(f)
227
- log.info("Upload of {} complete (1 part / {} B).".format(
228
- key.name, completed_size))
229
- return
230
- completed_size = 0
242
+ obj.key, f_size, num_chunks))
243
+
231
244
  chunk = 0
232
- mp = key.bucket.initiate_multipart_upload(
233
- key.name, metadata={
234
- METADATA_NAME: json.dumps(f.metadata)
235
- })
236
- try:
237
- for chunk in range(1, num_chunks + 1):
238
- part = mp.upload_part_from_file(
239
- f, chunk, size=CHUNK_SIZE())
240
- completed_size += part.size
241
- log.debug("Uploaded chunk {}/{} ({}B)".format(
242
- chunk, num_chunks, part.size))
243
- except: # NOQA
244
- # Any exception we want to attempt to cancel_upload, otherwise
245
- # AWS will bill us every month indefnitely for storing the
246
- # partial-uploaded chunks.
247
- log.exception("Upload of {} failed on chunk {}".format(
248
- key.name, chunk))
249
- mp.cancel_upload()
250
- raise
251
- else:
252
- completed = mp.complete_upload()
253
- log.info("Upload of {} complete ({} parts / {} B).".format(
254
- completed.key_name, chunk, completed_size))
245
+
246
+ def _progress(number_of_bytes):
247
+ nonlocal chunk
248
+ log.info("Uploaded chunk {}/{} ({}B)".format(
249
+ chunk, num_chunks, CHUNK_SIZE()))
250
+ chunk += 1
251
+
252
+ # NB: deep under the hood, upload_fileobj creates a
253
+ # CreateMultipartUploadTask. And that object cleans up after itself:
254
+ # https://github.com/boto/s3transfer/blob/develop/s3transfer/tasks.py#L353-L360 # noqa
255
+ obj.upload_fileobj(f, ExtraArgs=extra, Config=config,
256
+ Callback=_progress)
257
+ obj.wait_until_exists()
255
258
 
256
259
  def url_from_file(self, f):
257
260
  return self._get_s3_url(f)
@@ -279,12 +282,11 @@ class Archive(object):
279
282
  return url.startswith('http') and url.endswith('/data')
280
283
 
281
284
  def _fetch_s3_url(self, url, stream=False):
282
- k = self._get_key_from_url(url)
283
- m = self._get_metadata_from_key(k)
285
+ obj, m = self._get_object_from_url(url)
284
286
  if stream:
285
- return StreamingFile(k, **m)
287
+ return StreamingFile(obj._datalake_details['Body'], **m)
286
288
  fd = BytesIO()
287
- k.get_contents_to_file(fd)
289
+ self._s3_bucket.download_fileobj(obj.key, fd)
288
290
  fd.seek(0)
289
291
  return File(fd, **m)
290
292
 
@@ -323,16 +325,19 @@ class Archive(object):
323
325
  example, to store a file based on `what` it is, you could pass
324
326
  something like {what}.log. Or if you gathering many `what`'s that
325
327
  originate from many `where`'s you might want to use something like
326
- {where}/{what}-{start}.log. If filename_template is None (the default),
327
- files are stored in the current directory and the filenames are the ids
328
- from the metadata.
328
+ {where}/{what}-{start}.log. Note that the template variables
329
+ {start_iso} and {end_iso} are also supported and expand to the ISO
330
+ timestamps with millisecond precision (e.g., 2023-12-19T00:10:22.123).
331
+
332
+ If filename_template is None (the default), files are stored in the
333
+ current directory and the filenames are the ids from the metadata.
329
334
 
330
335
  Returns the filename written.
336
+
331
337
  '''
332
338
  k = None
333
339
  if url.startswith('s3://'):
334
- k = self._get_key_from_url(url)
335
- m = self._get_metadata_from_key(k)
340
+ obj, m = self._get_object_from_url(url)
336
341
  else:
337
342
  m = self._get_metadata_from_http_url(url)
338
343
  fname = self._get_filename_from_template(filename_template, m)
@@ -357,38 +362,58 @@ class Archive(object):
357
362
  else:
358
363
  raise
359
364
 
360
- def _get_key_from_url(self, url):
365
+ def _get_object_from_url(self, url):
361
366
  self._validate_fetch_url(url)
362
367
  key_name = self._get_key_name_from_url(url)
363
- k = self._s3_bucket.get_key(key_name)
364
- if k is None:
368
+ obj = self._s3.Object(self._s3_bucket_name, key_name)
369
+ try:
370
+ # cache the results of the get on the obj to avoid superfluous
371
+ # network calls.
372
+ obj._datalake_details = obj.get()
373
+ m = obj._datalake_details['Metadata'].get(METADATA_NAME)
374
+ except self._s3.meta.client.exceptions.NoSuchKey:
365
375
  msg = 'Failed to find {} in the datalake.'.format(url)
366
376
  raise InvalidDatalakePath(msg)
367
- return k
368
-
369
- def _get_metadata_from_key(self, key):
370
- m = key.get_metadata(METADATA_NAME)
371
- return Metadata.from_json(m)
377
+ return obj, Metadata.from_json(m)
372
378
 
373
379
  def _get_filename_from_template(self, template, metadata):
380
+ template_vars = deepcopy(metadata)
381
+ template_vars.update(
382
+ start_iso=self._ms_to_iso(metadata.get('start')),
383
+ end_iso=self._ms_to_iso(metadata.get('end')),
384
+ )
374
385
  if template is None:
375
386
  template = '{id}'
376
387
  try:
377
- return template.format(**metadata)
388
+ return template.format(**template_vars)
378
389
  except KeyError as e:
379
- m = '"{}" does not appear in the datalake metadata'
390
+ m = '"{}" does not appear to be a supported template variable.'
380
391
  m = m.format(str(e))
381
392
  raise InvalidDatalakePath(m)
382
393
  except ValueError as e:
383
394
  raise InvalidDatalakePath(str(e))
384
395
 
396
+ _ISO_FORMAT_MS = '%Y-%m-%dT%H:%M:%S.%f'
397
+
398
+ def _ms_to_iso(self, ts):
399
+ if ts is None:
400
+ return None
401
+ d = datetime.utcfromtimestamp(ts/1000.0)
402
+ # drop to ms precision
403
+ return d.strftime(self._ISO_FORMAT_MS)[:-3]
404
+
385
405
  def _get_key_name_from_url(self, url):
386
406
  parts = urlparse(url)
387
407
  if not parts.path:
388
408
  msg = '{} is not a valid datalake url'.format(url)
389
409
  raise InvalidDatalakePath(msg)
390
410
 
391
- return parts.path
411
+ # NB: under boto 2 we didn't used to have to have the lstrip. It seems
412
+ # that boto2 explicitly stripped these leading slashes for us:
413
+ # https://groups.google.com/g/boto-users/c/mv--NMPUXoU ...but boto3
414
+ # does not. So we must take care to strip it whenever we parse a URL to
415
+ # get a key.
416
+ return parts.path.lstrip('/')
392
417
 
393
418
  def _validate_fetch_url(self, url):
394
419
  valid_base_urls = (self.storage_url, self.http_url)
@@ -398,9 +423,9 @@ class Archive(object):
398
423
  raise InvalidDatalakePath(msg)
399
424
 
400
425
  def _get_s3_url(self, f):
401
- key = self._s3_key_from_metadata(f)
426
+ obj = self._s3_object_from_metadata(f)
402
427
  return self._URL_FORMAT.format(bucket=self._s3_bucket_name,
403
- key=key.name)
428
+ key=obj.key)
404
429
 
405
430
  @property
406
431
  def _s3_bucket_name(self):
@@ -408,41 +433,37 @@ class Archive(object):
408
433
 
409
434
  @memoized_property
410
435
  def _s3_bucket(self):
411
- # Note: we pass validate=False because we may just have push
412
- # permissions. If validate is not False, boto tries to list the
413
- # bucket. And this will 403.
414
- return self._s3_conn.get_bucket(self._s3_bucket_name, validate=False)
436
+ return self._s3.Bucket(self._s3_bucket_name)
415
437
 
416
438
  _KEY_FORMAT = '{id}/data'
417
439
 
418
- def _s3_key_from_metadata(self, f):
419
- # For performance reasons, s3 keys should start with a short random
420
- # sequence:
421
- # https://aws.amazon.com/blogs/aws/amazon-s3-performance-tips-tricks-seattle-hiring-event/
422
- # http://docs.aws.amazon.com/AmazonS3/latest/dev/request-rate-perf-considerations.html
440
+ def _s3_object_from_metadata(self, f):
423
441
  key_name = self._KEY_FORMAT.format(**f.metadata)
424
- return Key(self._s3_bucket, name=key_name)
442
+ return self._s3_bucket.Object(key_name)
425
443
 
426
444
  @property
427
445
  def _s3_host(self):
428
446
  h = environ.get('AWS_S3_HOST')
429
447
  if h is not None:
430
- return h
448
+ return 'https://' + h
431
449
  r = environ.get('AWS_REGION') or environ.get('AWS_DEFAULT_REGION')
432
450
  if r is not None:
433
- return 's3-' + r + '.amazonaws.com'
451
+ return 'https://s3-' + r + '.amazonaws.com'
434
452
  else:
435
- return NoHostProvided
453
+ return None
436
454
 
437
- @property
438
- def _s3_conn(self):
439
- if not hasattr(self, '_conn'):
440
- k = environ.get('AWS_ACCESS_KEY_ID')
441
- s = environ.get('AWS_SECRET_ACCESS_KEY')
442
- self._conn = S3Connection(aws_access_key_id=k,
443
- aws_secret_access_key=s,
444
- host=self._s3_host)
445
- return self._conn
455
+ @memoized_property
456
+ def _s3(self):
457
+ # boto3 uses AWS_ACCESS_KEY_ID and AWS_SECRET_ACCESS_KEY
458
+ # boto3 will use AWS_DEFAULT_REGION if AWS_REGION is not set
459
+ return boto3.resource('s3',
460
+ region_name=environ.get('AWS_REGION'),
461
+ endpoint_url=self._s3_host)
462
+
463
+ @memoized_property
464
+ def _s3_client(self):
465
+ boto_session = boto3.Session()
466
+ return boto_session.client('s3')
446
467
 
447
468
  def _requests_get(self, url, **kwargs):
448
469
  return self._session.get(url, timeout=TIMEOUT(), **kwargs)
@@ -12,7 +12,7 @@
12
12
  # License for the specific language governing permissions and limitations under
13
13
  # the License.
14
14
 
15
- from . import Metadata, InvalidDatalakeMetadata
15
+ from . import Metadata
16
16
  from six.moves.urllib.parse import urlparse
17
17
  import json
18
18
  import os
@@ -28,8 +28,7 @@ they are unavailable, the affected functions will raise
28
28
  InsufficientConfiguration.'''
29
29
  has_s3 = True
30
30
  try:
31
- import boto.s3
32
- from boto.exception import S3ResponseError
31
+ import boto3
33
32
  except ImportError:
34
33
  has_s3 = False
35
34
 
@@ -44,6 +43,11 @@ def requires_s3(f):
44
43
  return wrapped
45
44
 
46
45
 
46
+ # NB: Some time ago we migrated this class from datalake-common to datalake in
47
+ # order to reduce the number of packages that comprise the datalake. As a side
48
+ # effect of this migration this class contains some code duplicated with the
49
+ # datalake.Archive. There is an opportunity to clean this up, but we can take
50
+ # that on after getting off of boto 2.
47
51
  class DatalakeRecord(dict):
48
52
 
49
53
  def __init__(self, url, metadata, time_bucket, create_time, size):
@@ -64,66 +68,58 @@ class DatalakeRecord(dict):
64
68
  @requires_s3
65
69
  def list_from_url(cls, url):
66
70
  '''return a list of DatalakeRecords for the specified url'''
67
- key = cls._get_key(url)
68
- metadata = cls._get_metadata_from_key(key)
69
- ct = cls._get_create_time(key)
71
+ obj, metadata = cls._get_object(url)
72
+ ct = cls._get_create_time(obj)
70
73
  time_buckets = cls.get_time_buckets_from_metadata(metadata)
71
- return [cls(url, metadata, t, ct, key.size) for t in time_buckets]
74
+
75
+ return [
76
+ cls(url, metadata, t, ct, obj.content_length) for t in time_buckets
77
+ ]
72
78
 
73
79
  @classmethod
74
80
  @requires_s3
75
81
  def list_from_metadata(cls, url, metadata):
76
82
  '''return a list of DatalakeRecords for the url and metadata'''
77
- key = cls._get_key(url)
83
+ obj, _ = cls._get_object(url)
78
84
  metadata = Metadata(**metadata)
79
- ct = cls._get_create_time(key)
85
+ ct = cls._get_create_time(obj)
80
86
  time_buckets = cls.get_time_buckets_from_metadata(metadata)
81
- return [cls(url, metadata, t, ct, key.size) for t in time_buckets]
87
+ return [
88
+ cls(url, metadata, t, ct, obj.content_length) for t in time_buckets
89
+ ]
82
90
 
83
91
  @classmethod
84
- def _get_create_time(cls, key):
85
- return Metadata.normalize_date(key.last_modified)
92
+ def _get_create_time(cls, obj):
93
+ return Metadata.normalize_date(obj.last_modified)
86
94
 
87
95
  @classmethod
88
- def _get_key(cls, url):
96
+ def _get_object(cls, url):
89
97
  parsed_url = urlparse(url)
90
- bucket = cls._get_bucket(parsed_url.netloc)
91
- key = bucket.get_key(parsed_url.path)
92
- if key is None:
98
+ bucket = parsed_url.netloc
99
+
100
+ # NB: under boto 2 we didn't used to have to have the lstrip. It seems
101
+ # that boto2 explicitly stripped these leading slashes for us:
102
+ # https://groups.google.com/g/boto-users/c/mv--NMPUXoU ...but boto3
103
+ # does not. So we must take care to strip it whenever we parse a URL to
104
+ # get a key.
105
+ key_name = parsed_url.path.lstrip('/')
106
+ obj = cls._connection().Object(parsed_url.netloc, key_name)
107
+
108
+ try:
109
+ # cache the results of the get on the obj to avoid superfluous
110
+ # network calls.
111
+ obj._datalake_details = obj.get()
112
+ m = obj._datalake_details['Metadata'].get('datalake')
113
+ except cls._connection().meta.client.exceptions.NoSuchKey:
93
114
  msg = '{} does not appear to be in the datalake'
94
115
  msg = msg.format(url)
95
116
  raise NoSuchDatalakeFile(msg)
96
- return key
97
-
98
- @classmethod
99
- def _get_metadata_from_key(cls, key):
100
- metadata = key.get_metadata('datalake')
101
- if not metadata:
102
- msg = 'No datalake metadata for s3://{}{}'
103
- msg = msg.format(key.bucket.name, key.name)
104
- raise InvalidDatalakeMetadata(msg)
105
- return Metadata.from_json(metadata)
106
-
107
- _BUCKETS = {}
108
-
109
- @classmethod
110
- def _get_bucket(cls, bucket_name):
111
- if bucket_name not in cls._BUCKETS:
112
- bucket = cls._get_bucket_from_s3(bucket_name)
113
- DatalakeRecord._BUCKETS[bucket_name] = bucket
114
- return cls._BUCKETS[bucket_name]
117
+ except cls._connection().meta.client.exceptions.NoSuchBucket:
118
+ msg = 'Cannot find datalake file (s3 bucket {} does not exist)'
119
+ msg = msg.format(bucket)
120
+ raise NoSuchDatalakeFile(msg)
115
121
 
116
- @classmethod
117
- def _get_bucket_from_s3(cls, bucket_name):
118
- try:
119
- return cls._connection().get_bucket(bucket_name)
120
- except S3ResponseError as e:
121
- if e.error_code == 'NoSuchBucket':
122
- msg = 'Cannot find datalake file (s3 bucket {} does not exist)'
123
- msg = msg.format(bucket_name)
124
- raise NoSuchDatalakeFile(msg)
125
- else:
126
- raise
122
+ return obj, Metadata.from_json(m)
127
123
 
128
124
  _CONNECTION = None
129
125
 
@@ -133,13 +129,24 @@ class DatalakeRecord(dict):
133
129
  cls._CONNECTION = cls._prepare_connection()
134
130
  return cls._CONNECTION
135
131
 
132
+ @classmethod
133
+ def _s3_host(cls):
134
+ h = os.environ.get('AWS_S3_HOST')
135
+ if h is not None:
136
+ return 'https://' + h
137
+ r = (os.environ.get('AWS_REGION') or
138
+ os.environ.get('AWS_DEFAULT_REGION'))
139
+ if r is not None:
140
+ return 'https://s3-' + r + '.amazonaws.com'
141
+ else:
142
+ return None
143
+
136
144
  @classmethod
137
145
  def _prepare_connection(cls):
138
- kwargs = {}
139
- s3_host = os.environ.get('AWS_S3_HOST')
140
- if s3_host:
141
- kwargs['host'] = s3_host
142
- return boto.connect_s3(**kwargs)
146
+
147
+ return boto3.resource('s3',
148
+ region_name=os.environ.get('AWS_REGION'),
149
+ endpoint_url=cls._s3_host())
143
150
 
144
151
  _ONE_DAY_IN_MS = 24*60*60*1000
145
152
 
@@ -5,7 +5,7 @@ command line client may choose to configure logging in some cases. Users with
5
5
  sentry accounts may wish to configure it by installing the sentry extras.
6
6
  '''
7
7
  import os
8
- import logging
8
+ import logging.config
9
9
  from .common.errors import InsufficientConfiguration
10
10
 
11
11
 
@@ -20,9 +20,8 @@ import six
20
20
 
21
21
 
22
22
  try:
23
- from moto import mock_s3_deprecated
24
- import boto.s3
25
- from boto.s3.key import Key
23
+ from moto import mock_s3
24
+ import boto3
26
25
  from six.moves.urllib.parse import urlparse
27
26
  import json
28
27
  except ImportError:
@@ -113,6 +112,18 @@ def tmpfile(tmpdir):
113
112
  return get_tmpfile
114
113
 
115
114
 
115
+ @pytest.fixture
116
+ def tmpfile_maker(tmpdir):
117
+
118
+ def get_tmpfile(content):
119
+ name = random_word(10)
120
+ f = tmpdir.join(name)
121
+ f.write(content)
122
+ return str(f)
123
+
124
+ return get_tmpfile
125
+
126
+
116
127
  @pytest.fixture
117
128
  def aws_connector(request):
118
129
 
@@ -131,14 +142,17 @@ def aws_connector(request):
131
142
 
132
143
  @pytest.fixture
133
144
  def s3_connection(aws_connector):
134
- return aws_connector(mock_s3_deprecated, boto.connect_s3)
145
+ with mock_s3():
146
+ yield boto3.resource('s3')
135
147
 
136
148
 
137
149
  @pytest.fixture
138
150
  def s3_bucket_maker(s3_connection):
139
151
 
140
152
  def maker(bucket_name):
141
- return s3_connection.create_bucket(bucket_name)
153
+ b = s3_connection.Bucket(bucket_name)
154
+ b.create()
155
+ return b
142
156
 
143
157
  return maker
144
158
 
@@ -148,11 +162,10 @@ def s3_file_maker(s3_bucket_maker):
148
162
 
149
163
  def maker(bucket, key, content, metadata):
150
164
  b = s3_bucket_maker(bucket)
151
- k = Key(b)
152
- k.key = key
153
- if metadata:
154
- k.set_metadata('datalake', json.dumps(metadata))
155
- k.set_contents_from_string(content)
165
+ b.Object(key).put(
166
+ Body=content,
167
+ Metadata={'datalake': json.dumps(metadata)} if metadata else {}
168
+ )
156
169
 
157
170
  return maker
158
171
 
@@ -163,6 +176,7 @@ def s3_file_from_metadata(s3_file_maker):
163
176
  def maker(url, metadata):
164
177
  url = urlparse(url)
165
178
  assert url.scheme == 's3'
166
- s3_file_maker(url.netloc, url.path, '', metadata)
179
+ # NB: clean up leading slash.
180
+ s3_file_maker(url.netloc, url.path.lstrip('/'), '', metadata)
167
181
 
168
182
  return maker
@@ -0,0 +1,33 @@
1
+ Metadata-Version: 2.1
2
+ Name: datalake
3
+ Version: 2.2
4
+ Summary: datalake: a metadata-aware archive
5
+ Home-page: https://github.com/planetlabs/datalake
6
+ Author: Brian Cavagnolo
7
+ Author-email: brian@planet.com
8
+ Classifier: Programming Language :: Python :: 3.6
9
+ Classifier: Programming Language :: Python :: 3.7
10
+ Classifier: Programming Language :: Python :: 3.8
11
+ Classifier: Programming Language :: Python :: 3.9
12
+ License-File: LICENSE
13
+ Requires-Dist: boto3>=1.9.68
14
+ Requires-Dist: memoized_property>=1.0.1
15
+ Requires-Dist: pyblake2>=0.9.3; python_version < "3.6"
16
+ Requires-Dist: click>=4.1
17
+ Requires-Dist: python-dotenv>=0.1.3
18
+ Requires-Dist: requests>=2.5
19
+ Requires-Dist: six>=1.10.0
20
+ Requires-Dist: python-dateutil>=2.4.2
21
+ Requires-Dist: pytz>=2015.4
22
+ Provides-Extra: test
23
+ Requires-Dist: pytest<8.0.0; extra == "test"
24
+ Requires-Dist: moto[s3]<5,>4; extra == "test"
25
+ Requires-Dist: twine<4.0.0; extra == "test"
26
+ Requires-Dist: pip<22.0.0,>=20.0.0; extra == "test"
27
+ Requires-Dist: wheel<0.38.0; extra == "test"
28
+ Requires-Dist: flake8<4.1,>=2.5.0; extra == "test"
29
+ Requires-Dist: responses<0.22.0; extra == "test"
30
+ Provides-Extra: queuable
31
+ Requires-Dist: pyinotify>=0.9.4; extra == "queuable"
32
+ Provides-Extra: sentry
33
+ Requires-Dist: raven>=5.0.0; extra == "sentry"
@@ -1,3 +1,4 @@
1
+ LICENSE
1
2
  MANIFEST.in
2
3
  README.md
3
4
  setup.cfg
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ datalake = datalake.scripts.cli:cli
@@ -1,4 +1,4 @@
1
- boto>=2.38.0
1
+ boto3>=1.9.68
2
2
  memoized_property>=1.0.1
3
3
  click>=4.1
4
4
  python-dotenv>=0.1.3
@@ -18,7 +18,7 @@ raven>=5.0.0
18
18
 
19
19
  [test]
20
20
  pytest<8.0.0
21
- moto<3.0.0
21
+ moto[s3]<5,>4
22
22
  twine<4.0.0
23
23
  pip<22.0.0,>=20.0.0
24
24
  wheel<0.38.0
@@ -1,5 +1,5 @@
1
1
  [versioneer]
2
- vcs = git
2
+ VCS = git
3
3
  style = pep440
4
4
  versionfile_source = datalake/_version.py
5
5
  versionfile_build = datalake/_version.py
@@ -21,7 +21,7 @@ setup(name='datalake',
21
21
  author_email='brian@planet.com',
22
22
  packages=find_packages(exclude=['test']),
23
23
  install_requires=[
24
- 'boto>=2.38.0',
24
+ 'boto3>=1.9.68',
25
25
  'memoized_property>=1.0.1',
26
26
  'pyblake2>=0.9.3; python_version<"3.6"',
27
27
  'click>=4.1',
@@ -34,7 +34,7 @@ setup(name='datalake',
34
34
  extras_require={
35
35
  'test': [
36
36
  'pytest<8.0.0',
37
- 'moto<3.0.0',
37
+ 'moto[s3]>4,<5',
38
38
  'twine<4.0.0',
39
39
  'pip>=20.0.0,<22.0.0',
40
40
  'wheel<0.38.0',
@@ -16,13 +16,17 @@ import json
16
16
  import re
17
17
 
18
18
 
19
- def test_push_file(archive, random_metadata, tmpfile, s3_key):
19
+ def _get_contents_as_string(obj):
20
+ return obj.get()['Body'].read()
21
+
22
+
23
+ def test_push_file(archive, random_metadata, tmpfile, s3_object):
20
24
  expected_content = 'mwahaha'.encode('utf-8')
21
25
  f = tmpfile(expected_content)
22
26
  url = archive.prepare_metadata_and_push(f, **random_metadata)
23
- from_s3 = s3_key(url)
24
- assert from_s3.get_contents_as_string() == expected_content
25
- metadata = from_s3.get_metadata('datalake')
27
+ from_s3 = s3_object(url)
28
+ assert _get_contents_as_string(from_s3) == expected_content
29
+ metadata = from_s3.get()['Metadata'].get('datalake')
26
30
  assert metadata is not None
27
31
  metadata = json.loads(metadata)
28
32
  common_keys = set(metadata.keys()).intersection(random_metadata.keys())
@@ -30,14 +34,14 @@ def test_push_file(archive, random_metadata, tmpfile, s3_key):
30
34
 
31
35
 
32
36
  def test_push_large_file(
33
- monkeypatch, archive, random_metadata, tmpfile, s3_key):
34
- monkeypatch.setenv('DATALAKE_CHUNK_SIZE_MB', 5)
37
+ monkeypatch, archive, random_metadata, tmpfile, s3_object):
38
+ monkeypatch.setenv('DATALAKE_CHUNK_SIZE_MB', '5')
35
39
  expected_content = ('big data' * 1024 * 1024).encode('utf-8')
36
40
  f = tmpfile(expected_content)
37
41
  url = archive.prepare_metadata_and_push(f, **random_metadata)
38
- from_s3 = s3_key(url)
39
- assert from_s3.get_contents_as_string() == expected_content
40
- metadata = from_s3.get_metadata('datalake')
42
+ from_s3 = s3_object(url)
43
+ assert _get_contents_as_string(from_s3) == expected_content
44
+ metadata = from_s3.get()['Metadata'].get('datalake')
41
45
  assert metadata is not None
42
46
  metadata = json.loads(metadata)
43
47
  common_keys = set(metadata.keys()).intersection(random_metadata.keys())
@@ -14,9 +14,9 @@
14
14
 
15
15
  import pytest
16
16
  from test_crtime import crtime_setuid
17
- from click.testing import CliRunner
18
17
  import random
19
18
 
19
+
20
20
  def test_cli_without_command_fails(cli_tester):
21
21
  cli_tester('', expected_exit=2)
22
22
 
@@ -62,17 +62,18 @@ def test_translate_with_good_args_succceeds(cli_tester):
62
62
  cli_tester(cmd)
63
63
 
64
64
 
65
- @pytest.mark.parametrize("content", [['text'], ['multiple', 'things']])
66
- def test_cat(cli_tester, datalake_url_maker, random_metadata, content):
65
+ @pytest.mark.parametrize("content", [['text'], ['multiple', 'things']])
66
+ def test_cat(cli_tester, datalake_url_maker, random_metadata, content,
67
+ s3_connection):
67
68
  metadata1 = random_metadata
68
69
  metadata2 = random_metadata.copy()
69
70
  metadata2['id'] = ('%0' + str(40) + 'x') % random.randrange(16**40)
70
71
  url = ""+datalake_url_maker(metadata=metadata1, content=content[0])
71
72
  if content[0] == 'multiple':
72
- for i in range (1, len(content)):
73
- url = url + " " + datalake_url_maker(metadata=metadata2, content=content[i])
74
- cmd = ("cat "+url)
75
- print(cmd)
73
+ for i in range(1, len(content)):
74
+ url = (url + " " +
75
+ datalake_url_maker(metadata=metadata2, content=content[i]))
76
+ cmd = "cat " + url
76
77
  output = cli_tester(cmd)
77
78
  output = output.split('\n')
78
79
  for i in range(len(content)):
@@ -227,3 +227,27 @@ def test_metadata_from_http_url(archive, random_metadata):
227
227
  f = archive.fetch(url + '/data')
228
228
  assert f.read() == content
229
229
  assert f.metadata == random_metadata
230
+
231
+
232
+ def test_fetch_template_with_iso(archive, datalake_url_maker, random_metadata,
233
+ tmpdir):
234
+ random_metadata['start']=1702944622123
235
+ random_metadata['end']=1702944634456
236
+ url = datalake_url_maker(metadata=random_metadata)
237
+ t = os.path.join(str(tmpdir), '{start_iso}-{end_iso}-foobar.log')
238
+ fname = '2023-12-19T00:10:22.123-2023-12-19T00:10:34.456-foobar.log'
239
+ expected_path = os.path.join(str(tmpdir), fname)
240
+ archive.fetch_to_filename(url, filename_template=t)
241
+ assert os.path.exists(expected_path)
242
+
243
+
244
+ def test_fetch_template_with_none(archive, datalake_url_maker, random_metadata,
245
+ tmpdir):
246
+ random_metadata['start']=1702944622123
247
+ random_metadata['end']=None
248
+ url = datalake_url_maker(metadata=random_metadata)
249
+ t = os.path.join(str(tmpdir), '{start_iso}-{end_iso}-foobar.log')
250
+ fname = '2023-12-19T00:10:22.123-None-foobar.log'
251
+ expected_path = os.path.join(str(tmpdir), fname)
252
+ archive.fetch_to_filename(url, filename_template=t)
253
+ assert os.path.exists(expected_path)
@@ -16,7 +16,7 @@ import pytest
16
16
  import json
17
17
  from threading import Timer
18
18
  import os
19
- from datalake.tests import random_word
19
+ from datalake.tests import random_word, generate_random_metadata
20
20
  from datalake.common.errors import InsufficientConfiguration
21
21
  from datalake import Enqueuer, Uploader, InvalidDatalakeBundle
22
22
  from datalake.queue import has_queue
@@ -53,18 +53,23 @@ def faulty_uploader(archive, queue_dir):
53
53
 
54
54
 
55
55
  @pytest.fixture
56
- def uploaded_content_validator(s3_key):
56
+ def uploaded_content_validator(s3_object):
57
57
 
58
58
  def validator(expected_content, expected_metadata=None, compressed=False):
59
59
 
60
- from_s3 = s3_key()
60
+ if expected_metadata:
61
+ from_s3 = s3_object(
62
+ f's3://datalake-test/{expected_metadata["id"]}/data'
63
+ )
64
+ else:
65
+ from_s3 = s3_object()
61
66
  assert from_s3 is not None
62
- content = from_s3.get_contents_as_string()
67
+ content = from_s3.get()['Body'].read()
63
68
  if compressed:
64
69
  content = zlib.decompress(content, 16 + zlib.MAX_WBITS)
65
70
  assert content == expected_content
66
71
  if expected_metadata is not None:
67
- metadata = json.loads(from_s3.get_metadata('datalake'))
72
+ metadata = json.loads(from_s3.get()['Metadata'].get('datalake'))
68
73
  assert metadata == expected_metadata
69
74
 
70
75
  return validator
@@ -84,7 +89,7 @@ def uploaded_file_validator(archive, uploaded_content_validator):
84
89
  def assert_s3_bucket_empty(s3_bucket):
85
90
 
86
91
  def asserter():
87
- assert len([k for k in s3_bucket.list()]) == 0
92
+ assert len(list(s3_bucket.objects.all())) == 0
88
93
 
89
94
  return asserter
90
95
 
@@ -95,6 +100,16 @@ def random_file(tmpfile, random_metadata):
95
100
  return tmpfile(expected_content)
96
101
 
97
102
 
103
+ @pytest.fixture
104
+ def random_file_maker(tmpfile_maker):
105
+
106
+ def maker():
107
+ expected_content = random_word(100)
108
+ return tmpfile_maker(expected_content)
109
+
110
+ return maker
111
+
112
+
98
113
  @pytest.mark.skipif(not has_queue, reason='requires queuable features')
99
114
  def test_upload_existing(enqueuer, uploader, random_file, random_metadata,
100
115
  uploaded_file_validator):
@@ -220,7 +235,7 @@ def test_enqueue_compress_cli(cli_tester, uploader, random_file,
220
235
 
221
236
 
222
237
  @pytest.mark.skipif(not has_queue, reason='requires queuable features')
223
- def test_threaded_upload(enqueuer, uploader, random_file, random_metadata,
238
+ def test_threaded_upload(enqueuer, uploader, random_file_maker,
224
239
  uploaded_file_validator):
225
240
 
226
241
  # This test does not actually validate that multiple threads are running.
@@ -229,7 +244,8 @@ def test_threaded_upload(enqueuer, uploader, random_file, random_metadata,
229
244
  enqueued_files = []
230
245
 
231
246
  def enqueue():
232
- f = enqueuer.enqueue(random_file, **random_metadata)
247
+ m = generate_random_metadata()
248
+ f = enqueuer.enqueue(random_file_maker(), **m)
233
249
  enqueued_files.append(f)
234
250
  if len(enqueued_files) < 3:
235
251
  t = Timer(0.1, enqueue)
@@ -238,7 +254,6 @@ def test_threaded_upload(enqueuer, uploader, random_file, random_metadata,
238
254
  t = Timer(0.5, enqueue)
239
255
  t.start()
240
256
  uploader.listen(timeout=1.0, workers=3)
241
-
242
257
  assert len(enqueued_files) == 3
243
258
  for f in enqueued_files:
244
259
  uploaded_file_validator(f)
datalake-1.1/PKG-INFO DELETED
@@ -1,17 +0,0 @@
1
- Metadata-Version: 2.1
2
- Name: datalake
3
- Version: 1.1
4
- Summary: datalake: a metadata-aware archive
5
- Home-page: https://github.com/planetlabs/datalake
6
- Author: Brian Cavagnolo
7
- Author-email: brian@planet.com
8
- License: UNKNOWN
9
- Description: UNKNOWN
10
- Platform: UNKNOWN
11
- Classifier: Programming Language :: Python :: 3.6
12
- Classifier: Programming Language :: Python :: 3.7
13
- Classifier: Programming Language :: Python :: 3.8
14
- Classifier: Programming Language :: Python :: 3.9
15
- Provides-Extra: test
16
- Provides-Extra: queuable
17
- Provides-Extra: sentry
@@ -1,17 +0,0 @@
1
- Metadata-Version: 2.1
2
- Name: datalake
3
- Version: 1.1
4
- Summary: datalake: a metadata-aware archive
5
- Home-page: https://github.com/planetlabs/datalake
6
- Author: Brian Cavagnolo
7
- Author-email: brian@planet.com
8
- License: UNKNOWN
9
- Description: UNKNOWN
10
- Platform: UNKNOWN
11
- Classifier: Programming Language :: Python :: 3.6
12
- Classifier: Programming Language :: Python :: 3.7
13
- Classifier: Programming Language :: Python :: 3.8
14
- Classifier: Programming Language :: Python :: 3.9
15
- Provides-Extra: test
16
- Provides-Extra: queuable
17
- Provides-Extra: sentry
@@ -1,4 +0,0 @@
1
-
2
- [console_scripts]
3
- datalake=datalake.scripts.cli:cli
4
-
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes