datalake 2.4.1__tar.gz → 2.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. datalake-2.5/PKG-INFO +334 -0
  2. datalake-2.5/datalake/_version.py +520 -0
  3. {datalake-2.4.1 → datalake-2.5}/datalake/queue.py +19 -35
  4. datalake-2.5/datalake.egg-info/PKG-INFO +334 -0
  5. {datalake-2.4.1 → datalake-2.5}/datalake.egg-info/SOURCES.txt +1 -3
  6. {datalake-2.4.1 → datalake-2.5}/datalake.egg-info/requires.txt +2 -1
  7. datalake-2.5/datalake.egg-info/top_level.txt +3 -0
  8. datalake-2.5/pyproject.toml +79 -0
  9. datalake-2.5/setup.cfg +4 -0
  10. {datalake-2.4.1 → datalake-2.5}/test/test_queue.py +1 -1
  11. datalake-2.4.1/PKG-INFO +0 -33
  12. datalake-2.4.1/datalake/_version.py +0 -21
  13. datalake-2.4.1/datalake.egg-info/PKG-INFO +0 -33
  14. datalake-2.4.1/datalake.egg-info/top_level.txt +0 -1
  15. datalake-2.4.1/setup.cfg +0 -11
  16. datalake-2.4.1/setup.py +0 -62
  17. datalake-2.4.1/versioneer.py +0 -1822
  18. {datalake-2.4.1 → datalake-2.5}/LICENSE +0 -0
  19. {datalake-2.4.1 → datalake-2.5}/MANIFEST.in +0 -0
  20. {datalake-2.4.1 → datalake-2.5}/README.md +0 -0
  21. {datalake-2.4.1 → datalake-2.5}/datalake/__init__.py +0 -0
  22. {datalake-2.4.1 → datalake-2.5}/datalake/archive.py +0 -0
  23. {datalake-2.4.1 → datalake-2.5}/datalake/common/__init__.py +0 -0
  24. {datalake-2.4.1 → datalake-2.5}/datalake/common/conf.py +0 -0
  25. {datalake-2.4.1 → datalake-2.5}/datalake/common/errors.py +0 -0
  26. {datalake-2.4.1 → datalake-2.5}/datalake/common/metadata.py +0 -0
  27. {datalake-2.4.1 → datalake-2.5}/datalake/common/record.py +0 -0
  28. {datalake-2.4.1 → datalake-2.5}/datalake/config_helpers.py +0 -0
  29. {datalake-2.4.1 → datalake-2.5}/datalake/crtime.py +0 -0
  30. {datalake-2.4.1 → datalake-2.5}/datalake/dlfile.py +0 -0
  31. {datalake-2.4.1 → datalake-2.5}/datalake/logging_helpers.py +0 -0
  32. {datalake-2.4.1 → datalake-2.5}/datalake/scripts/__init__.py +0 -0
  33. {datalake-2.4.1 → datalake-2.5}/datalake/scripts/cli.py +0 -0
  34. {datalake-2.4.1 → datalake-2.5}/datalake/tests/__init__.py +0 -0
  35. {datalake-2.4.1 → datalake-2.5}/datalake/tests/conftest.py +0 -0
  36. {datalake-2.4.1 → datalake-2.5}/datalake/tests/test_conf.py +0 -0
  37. {datalake-2.4.1 → datalake-2.5}/datalake/tests/test_metadata.py +0 -0
  38. {datalake-2.4.1 → datalake-2.5}/datalake/tests/test_record.py +0 -0
  39. {datalake-2.4.1 → datalake-2.5}/datalake/translator.py +0 -0
  40. {datalake-2.4.1 → datalake-2.5}/datalake.egg-info/dependency_links.txt +0 -0
  41. {datalake-2.4.1 → datalake-2.5}/datalake.egg-info/entry_points.txt +0 -0
  42. {datalake-2.4.1 → datalake-2.5}/test/test_archive.py +0 -0
  43. {datalake-2.4.1 → datalake-2.5}/test/test_cli.py +0 -0
  44. {datalake-2.4.1 → datalake-2.5}/test/test_config.py +0 -0
  45. {datalake-2.4.1 → datalake-2.5}/test/test_crtime.py +0 -0
  46. {datalake-2.4.1 → datalake-2.5}/test/test_fetch.py +0 -0
  47. {datalake-2.4.1 → datalake-2.5}/test/test_file.py +0 -0
  48. {datalake-2.4.1 → datalake-2.5}/test/test_latest.py +0 -0
  49. {datalake-2.4.1 → datalake-2.5}/test/test_list.py +0 -0
  50. {datalake-2.4.1 → datalake-2.5}/test/test_logging.py +0 -0
  51. {datalake-2.4.1 → datalake-2.5}/test/test_translator.py +0 -0
datalake-2.5/PKG-INFO ADDED
@@ -0,0 +1,334 @@
1
+ Metadata-Version: 2.1
2
+ Name: datalake
3
+ Version: 2.5
4
+ Summary: datalake: a metadata-aware archive
5
+ Author-email: Brian Cavagnolo <brian@planet.com>
6
+ Classifier: Development Status :: 5 - Production/Stable
7
+ Classifier: Environment :: Console
8
+ Classifier: Operating System :: OS Independent
9
+ Classifier: Programming Language :: Python
10
+ Classifier: Programming Language :: Python :: 3
11
+ Requires-Python: >=3.8
12
+ Description-Content-Type: text/markdown
13
+ License-File: LICENSE
14
+ Requires-Dist: boto3>=1.9.68
15
+ Requires-Dist: memoized_property>=1.0.1
16
+ Requires-Dist: pyblake2>=0.9.3; python_version < "3.6"
17
+ Requires-Dist: click>=4.1
18
+ Requires-Dist: python-dotenv>=0.1.3
19
+ Requires-Dist: requests>=2.5
20
+ Requires-Dist: six>=1.10.0
21
+ Requires-Dist: python-dateutil>=2.4.2
22
+ Requires-Dist: pytz>=2015.4
23
+ Provides-Extra: test
24
+ Requires-Dist: pytest<8.0.0; extra == "test"
25
+ Requires-Dist: pytest-cov<4,>=2.5.1; extra == "test"
26
+ Requires-Dist: moto[s3]<5,>4; extra == "test"
27
+ Requires-Dist: twine<4.0.0; extra == "test"
28
+ Requires-Dist: pip<22.0.0,>=20.0.0; extra == "test"
29
+ Requires-Dist: wheel<0.38.0; extra == "test"
30
+ Requires-Dist: flake8<4.1,>=2.5.0; extra == "test"
31
+ Requires-Dist: responses<0.22.0; extra == "test"
32
+ Provides-Extra: queuable
33
+ Requires-Dist: inotify_simple>=1.3.5; extra == "queuable"
34
+ Provides-Extra: sentry
35
+ Requires-Dist: raven>=5.0.0; extra == "sentry"
36
+
37
+ Introduction
38
+ ============
39
+
40
+ [![Build Status](https://travis-ci.org/planetlabs/datalake.svg)](https://travis-ci.org/planetlabs/datalake)
41
+
42
+ A datalake is an archive that contains files and metadata records about those
43
+ files. The datalake project consists of a number of pieces:
44
+
45
+ - The ingester that listens for new files pushed to the datalake and ingests
46
+ their metadata so it can be searched.
47
+
48
+ - The api to query over the files in the datalake.
49
+
50
+ - The client, which is a python and command-line interface to the datalake. You
51
+ can use it to push files to the datalake, list the files available in the
52
+ datalake, and retrieve files from the datalake.
53
+
54
+ To use this client, you (or somebody on your behalf) must be operating an
55
+ instance of the datalake-ingester and the datalake-api. You will need some
56
+ configuration information from them.
57
+
58
+ Why would I use this? Because you just want to get all of the files into one
59
+ place with nice uniform metadata so you can know what is what. Then you can
60
+ pull the files onto your hardrive for your grepping and awking pleasure. Or
61
+ perhaps you can feed them to a compute cluster of some sort for mapping and
62
+ reducing. Or maybe you don't want to set up and maintain a bunch of log
63
+ ingestion infrastructure, or you don't trust that log ingestion infrastructure
64
+ to be your source of truth. Or maybe you just get that warm fuzzy feeling when
65
+ things are archived somewhere.
66
+
67
+ Client Usage
68
+ ============
69
+
70
+ Install
71
+ -------
72
+
73
+ pip install datalake
74
+
75
+ If you plan to use the queuing feature, you must install some extra
76
+ dependencies:
77
+
78
+ apt-get install libffi-dev # or equivalent
79
+ pip install datalake[queuable]
80
+
81
+ Configure
82
+ ---------
83
+
84
+ datalake needs a bit of configuration. Every configuration variable can either
85
+ be set in /etc/datalake.conf, set as an environment variable, or passed in as
86
+ an argument. For documentation on the configuration variables, invoke `datalake
87
+ --help`.
88
+
89
+ Usage
90
+ -----
91
+
92
+ datalake has a python API and a command-line client. What you can do with one,
93
+ you can do with the other. Here's how it works:
94
+
95
+ Push a log file:
96
+
97
+ datalake push --start 2015-03-20T00:05:32.345Z
98
+ --end 2015-03-20T23:59.114Z \
99
+ --where webserver01 --what nginx /path/to/nginx.log
100
+
101
+ Push a log file with a specific work-id:
102
+
103
+ datalake push --start 2015-03-20T00:00:05:32.345Z \
104
+ --end 2015-03-20T00:00:34.114Z \
105
+ --what blappo-etl --where backend01 \
106
+ --work-id blappo-14321359
107
+
108
+ The work-id is convenient for tracking processing jobs or other entities that
109
+ may pass through many log-generating machines as they proceed through life. It
110
+ must be unique within the datalake. So usually some kind of domain-specific
111
+ prefix is recommended here.
112
+
113
+ List the syslog and foobar files available from webserver01 since the specified
114
+ start date.
115
+
116
+ datalake list --where webserver01 --start 2015-03-20 --end `date -u` \
117
+ --what syslog,foobar
118
+
119
+ Fetch the blappo gather, etl, and cleanup log files with work id
120
+ blappo-14321359:
121
+
122
+ datalake fetch --what gather,etl,cleanup --work-id blappo-14321359
123
+
124
+ Developer Setup
125
+ ===============
126
+
127
+ make docker test
128
+
129
+ Datalake Metadata
130
+ =================
131
+
132
+ Files that are shipped to the datalake are accompanied by a JSON metadata
133
+ document. Here it is:
134
+
135
+ {
136
+ "version": 0,
137
+ "start": 1426809920345,
138
+ "end": 1426895999114,
139
+ "path": "/var/log/syslog.1"
140
+ "work_id": null,
141
+ "where": "webserver02",
142
+ "what": "syslog",
143
+ "id": "6309e115c2914d0f8622973422626954",
144
+ "hash": "a3e75ee4f45f676422e038f2c116d000"
145
+ }
146
+
147
+ version: This is the metadata version. It should be 0.
148
+
149
+ start: This is the time of the first event in the file in milliseconds since
150
+ the epoch. Alternatively, if the file is associated with an instant, this is
151
+ the only relevant time. It is required.
152
+
153
+ end: This is the time of the last event in the file in milliseconds since the
154
+ epoch. If the key is not present or if the value is `None`, the file represents a
155
+ snapshot of something like a weekly report where only one date (`start`) is
156
+ relevant.
157
+
158
+ path: The absolute path to the file in the originating filesystem.
159
+
160
+ where: This is the location or server that generated the file. It is required
161
+ and must only contain lowercase alpha-numeric characters, - and _. It should be
162
+ concise. 'localhost' and 'vagrant' are bad names. Something like
163
+ 'whirlyweb02-prod' is good.
164
+
165
+ what: This is the process or program that generated the file. It is required
166
+ and must only contain lowercase alpha-numeric characters, - and _. It must not
167
+ have trailing file extension (e.g., .log). The name should be concise to limit
168
+ the chances that it conflicts with other whats in the datalake. So names like
169
+ 'job' or 'task' are bad. Names like 'balyhoo-source-audit' or
170
+ 'rawfood-ingester' are good.
171
+
172
+ id: An ID for the file assigned by the datalake. It is required.
173
+
174
+ hash: A 16-byte blake2 hash of the file content. This is calcluated and
175
+ assigned by the datalake. It is required.
176
+
177
+ work_id: This is an application-specific id that can be used later to retrieve
178
+ the file. It is required but may be null. In fact the datalake utilities will
179
+ generally default it to null if it is not set. It must not be the string
180
+ "null". It should be prepended with a domain-specific prefix to prevent
181
+ conflicts with other work id spaces. It must only contain lowercase
182
+ alpha-numeric characters, -, and _.
183
+
184
+ Index Design
185
+ ============
186
+
187
+ In practice, metadata is stored in DynamoDB, which has strict but simple rules
188
+ about defining and querying indexes. We wish to support a few simple queries
189
+ over our metadata:
190
+
191
+ 1. give me all of the WHATs for a given WHERE from t=START to t=END
192
+ 2. give me all of the WHATs from t=START to t=END
193
+ 3. give me all of the WHATs for a given WHERE with a given WORK_ID
194
+ 4. give me all of the WHATs with a given WORK_ID
195
+
196
+ To achieve this using DynamoDB, we adopt the notion of "time buckets," each of
197
+ which is one day long. So a file whose data spans the period of today from
198
+ 1:00-2:00 would have a single record in today's time bucket. A file whose data
199
+ spans the period from yesterday at noon to today at noon has two records: one
200
+ in yesterday's bucket and one in today's bucket. And so when a user queries
201
+ over a time period, we simply calculate the buckets that the time period spans,
202
+ then look in each bucket for relevant files.
203
+
204
+ But doesn't that mean we have to sometimes write multiple records per file?
205
+ Yes. What if a file spans 100 days? Do we really want to put a record in each
206
+ of 100 buckets? Well, this would be a pretty uncommon case for the uses that we
207
+ are envisioning. In practice, such files should be broken up into smaller files
208
+ and uploaded more frequently. What if a user queries for 100 days worth of
209
+ data? Well, we examine a bunch of buckets and it takes a while. Users that are
210
+ not prepared to wait this long should make smaller requests.
211
+
212
+ To enable these queries, we have two hash-and-range indexes. They have the
213
+ following HASHKEY RANGEKEY format:
214
+
215
+ TIME_BUCKET:WHAT WHERE:ID
216
+
217
+ WORK_ID:WHAT WHERE:ID
218
+
219
+ The first index is to support query types 1 and 2. By using TIME_BUCKET:WHAT as
220
+ the hash key we prevent "hot" hash keys by distributing writes and queries
221
+ across WHATs. So while all the records for a day will be written to the same
222
+ TIME_BUCKET, and while users are much more likely to query recent things from
223
+ the last few TIME_BUCKETs, we spread the load across a diversity of WHATs. The
224
+ WHERE:ID range key can be used to retrieve a subset of WHEREs if
225
+ necessary. Finally, we append the file ID to ensure that the key is unique as
226
+ required by DynamoDB.
227
+
228
+ The second index supports query types 3 and 4 and follows a pattern similar to
229
+ the first. However, it should be noted that the WORK_ID is optional metadata,
230
+ but required for indexing purposes. To work around this without introducing a
231
+ hot hash key in the second index, the ingester generates a random WORK_ID with
232
+ the reserved prefix "null".
233
+
234
+ Datalake Record Format
235
+ ======================
236
+
237
+ The datalake client specifies metadata that is recorded when a file is pushed
238
+ to the datalake. We need to store some administrative fields to get our queries
239
+ to work with the dynamodb indexes. These records have the following format:
240
+
241
+ {
242
+ "version": 0,
243
+ "url": "s3://datalake/d-nebraska/nginx/1437375600000/91dd2525a5924c6c972e3d67fee8cda9-nginx-523.txt",
244
+ "time_index_key": "16636:nginx",
245
+ "work_id_index_key": "nullc177bfc032c548ba9e056c8e8672dba8:nginx",
246
+ "range_key": "nebraska:91dd2525a5924c6c972e3d67fee8cda9",
247
+ "create_time": 1426896791333,
248
+ "size": 7892341,
249
+ "metadata": { ... },
250
+ }
251
+
252
+ version: the version of the datalake record format. What we describe here is
253
+ version 0.
254
+
255
+ url: the url of the resource to which the datalake record pertains.
256
+
257
+ time_index_key: the hash key for the index used for time-based queries. It is
258
+ formed by joining the "time bucket" number and the "what" from the metadata.
259
+
260
+ work_id_index_key: the hash key for the index used for work_id-based
261
+ queries. It is formed by joining the work_id and the "what" from the
262
+ metadata. Note that if the work_id is null, a random work_id will be generated
263
+ to prevent ingestion failures and hot hash keys. Of course in this case
264
+ retrieving by work_id is not meaninful or possible.
265
+
266
+ range_key: the range key used by the time-based and work_id-based indexes. It
267
+ is formed by joining the "where" and the "id" from the metadata.
268
+
269
+ create_time: the creation time of the file in the datalake
270
+
271
+ size: the size of the file in bytes
272
+
273
+ Ingester
274
+ ========
275
+
276
+ The datalake-ingester ingests datalake metadata records into a database so that
277
+ they may be queried by other datalake components.
278
+
279
+ The ingester looks something like this:
280
+
281
+ +----------+ +---------+
282
+ +-------+ +-----------------+ | |---->| storage |
283
+ -->| queue |--->| s3_notification |--->| ingester | +---------+
284
+ +-------+ +-----------------+ | |--+
285
+ +----------+ | +----------+
286
+ +->| reporter |
287
+ +----------+
288
+
289
+
290
+ A queue receives notice that an event has occured in the datalake's s3
291
+ bucket. An s3_notification object translates the event from the queue's format
292
+ to the datalake record format (see above). Next the ingester updates the
293
+ storage (i.e., dynamodb) and reports the ingestion status to the reporter
294
+ (i.e., SNS).
295
+
296
+ Datalake Ingester Report Format
297
+ ===============================
298
+
299
+ The datalake ingester emits a Datalake Ingester Report for each file that it
300
+ ingests. The report has the following format:
301
+
302
+ {
303
+ "version": 0,
304
+ "status": "success",
305
+ "start": 1437375854967,
306
+ "duration": 0.738383,
307
+ "records": [
308
+ {
309
+ "url": "s3://datalake/d-nebraska/nginx/1437375600000/91dd2525a5924c6c972e3d67fee8cda9-nginx-523.txt",
310
+ "metadata": { ... }
311
+ }
312
+ ]
313
+ }
314
+
315
+ version: the version of the datalake ingester report format. What we describe
316
+ here is version 0.
317
+
318
+ status: Either "success", "warning", or "error" depending on how successful
319
+ ingestion was. If status is not "success" expect "message" to be set with a
320
+ human-readable explanation.
321
+
322
+ start: ms since the epoch when the ingestion started.
323
+
324
+ duration: time in seconds that it took to ingest the record.
325
+
326
+ records: a list of records that were ingested. Note that this is typically a
327
+ list with one element. However, some underlying protocols (e.g., s3
328
+ notifications) may carry information about multiple records. Under these
329
+ circumstances multiple records may appear.
330
+
331
+ API
332
+ ===
333
+
334
+ The datalake-api offers and HTTP interface to the datalake.