dc-python-sdk 1.5.49__tar.gz → 1.5.51__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dc_python_sdk-1.5.49/src/dc_python_sdk.egg-info → dc_python_sdk-1.5.51}/PKG-INFO +1 -1
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/pyproject.toml +1 -1
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/setup.cfg +1 -1
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51/src/dc_python_sdk.egg-info}/PKG-INFO +1 -1
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_python_sdk.egg-info/SOURCES.txt +2 -0
- dc_python_sdk-1.5.51/src/dc_sdk/__init__.py +3 -0
- dc_python_sdk-1.5.51/src/dc_sdk/data_stream.py +33 -0
- dc_python_sdk-1.5.51/src/dc_sdk/src/destination_object_template.py +74 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/mapping.py +15 -5
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/pipeline.py +140 -5
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/services/aws.py +52 -1
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/types.py +11 -1
- dc_python_sdk-1.5.49/src/dc_sdk/src/services/__init__.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/LICENSE +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/README.md +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_python_sdk.egg-info/dependency_links.txt +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_python_sdk.egg-info/entry_points.txt +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_python_sdk.egg-info/requires.txt +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_python_sdk.egg-info/top_level.txt +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/app.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/cli.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/errors.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/handler.py +0 -0
- {dc_python_sdk-1.5.49/src/dc_sdk → dc_python_sdk-1.5.51/src/dc_sdk/src}/__init__.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/ai.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/ai_http.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/connection_status.py +0 -0
- {dc_python_sdk-1.5.49/src/dc_sdk/src → dc_python_sdk-1.5.51/src/dc_sdk/src/models}/__init__.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/models/enums.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/models/errors.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/models/log_templates.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/models/pipeline_details.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/server.py +0 -0
- {dc_python_sdk-1.5.49/src/dc_sdk/src/models → dc_python_sdk-1.5.51/src/dc_sdk/src/services}/__init__.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/services/api.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/services/environment.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/services/loader.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/services/logger.py +0 -0
- {dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_sdk/src/services/session.py +0 -0
|
@@ -11,6 +11,7 @@ src/dc_python_sdk.egg-info/top_level.txt
|
|
|
11
11
|
src/dc_sdk/__init__.py
|
|
12
12
|
src/dc_sdk/app.py
|
|
13
13
|
src/dc_sdk/cli.py
|
|
14
|
+
src/dc_sdk/data_stream.py
|
|
14
15
|
src/dc_sdk/errors.py
|
|
15
16
|
src/dc_sdk/handler.py
|
|
16
17
|
src/dc_sdk/types.py
|
|
@@ -18,6 +19,7 @@ src/dc_sdk/src/__init__.py
|
|
|
18
19
|
src/dc_sdk/src/ai.py
|
|
19
20
|
src/dc_sdk/src/ai_http.py
|
|
20
21
|
src/dc_sdk/src/connection_status.py
|
|
22
|
+
src/dc_sdk/src/destination_object_template.py
|
|
21
23
|
src/dc_sdk/src/mapping.py
|
|
22
24
|
src/dc_sdk/src/pipeline.py
|
|
23
25
|
src/dc_sdk/src/server.py
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
from dataclasses import dataclass, field
|
|
2
|
+
from typing import Any, Dict, Iterable, Optional
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
@dataclass
|
|
6
|
+
class DataStream:
|
|
7
|
+
"""Byte stream returned by connector get_data_stream for transfer/upload."""
|
|
8
|
+
|
|
9
|
+
stream: Iterable[bytes]
|
|
10
|
+
|
|
11
|
+
# What are the bytes?
|
|
12
|
+
media_type: str
|
|
13
|
+
|
|
14
|
+
# Optional processing hints
|
|
15
|
+
container: Optional[str] = None # zip, gzip, tar
|
|
16
|
+
format: Optional[str] = None # csv, ndjson, parquet, xml, xlsx
|
|
17
|
+
|
|
18
|
+
# Optional metadata
|
|
19
|
+
filename: Optional[str] = None
|
|
20
|
+
encoding: Optional[str] = None
|
|
21
|
+
|
|
22
|
+
metadata: Dict[str, Any] = field(default_factory=dict)
|
|
23
|
+
|
|
24
|
+
def hints(self) -> Dict[str, Any]:
|
|
25
|
+
"""Serializable extract/normalize hints (excludes the live stream)."""
|
|
26
|
+
return {
|
|
27
|
+
"media_type": self.media_type,
|
|
28
|
+
"container": self.container,
|
|
29
|
+
"format": self.format,
|
|
30
|
+
"filename": self.filename,
|
|
31
|
+
"encoding": self.encoding,
|
|
32
|
+
"metadata": self.metadata,
|
|
33
|
+
}
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Resolve dynamic tokens in destination object paths / file names at run time."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from datetime import datetime, timezone
|
|
7
|
+
from typing import Optional
|
|
8
|
+
|
|
9
|
+
_TOKEN_RE = re.compile(r"\{([A-Za-z0-9_-]+)\}")
|
|
10
|
+
|
|
11
|
+
_DATE_FORMATS = {
|
|
12
|
+
"YYYYMMDD": "%Y%m%d",
|
|
13
|
+
"YYYY-MM-DD": "%Y-%m-%d",
|
|
14
|
+
"YYMMDD": "%y%m%d",
|
|
15
|
+
"HHMMSS": "%H%M%S",
|
|
16
|
+
"YYYYMMDDHHMMSS": "%Y%m%d%H%M%S",
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _slug(value: Optional[str]) -> str:
|
|
21
|
+
if value is None:
|
|
22
|
+
return ""
|
|
23
|
+
text = str(value).strip()
|
|
24
|
+
if not text:
|
|
25
|
+
return ""
|
|
26
|
+
text = re.sub(r"[^A-Za-z0-9]+", "_", text)
|
|
27
|
+
return text.strip("_")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def resolve_destination_object_template(
|
|
31
|
+
template: Optional[str],
|
|
32
|
+
*,
|
|
33
|
+
source_object_id: Optional[str] = None,
|
|
34
|
+
source_connector_nm: Optional[str] = None,
|
|
35
|
+
when: Optional[datetime] = None,
|
|
36
|
+
) -> str:
|
|
37
|
+
"""
|
|
38
|
+
Expand `{YYYYMMDD}`, `{object}`, `{connector}`, etc. in a destination path.
|
|
39
|
+
|
|
40
|
+
Unknown tokens are left unchanged so mis-typed braces remain visible in logs.
|
|
41
|
+
"""
|
|
42
|
+
if template is None:
|
|
43
|
+
return ""
|
|
44
|
+
value = str(template)
|
|
45
|
+
if "{" not in value:
|
|
46
|
+
return value
|
|
47
|
+
|
|
48
|
+
now = when or datetime.now(timezone.utc)
|
|
49
|
+
if now.tzinfo is None:
|
|
50
|
+
now = now.replace(tzinfo=timezone.utc)
|
|
51
|
+
|
|
52
|
+
replacements = {
|
|
53
|
+
"object": _slug(source_object_id),
|
|
54
|
+
"connector": _slug(source_connector_nm),
|
|
55
|
+
}
|
|
56
|
+
for token, fmt in _DATE_FORMATS.items():
|
|
57
|
+
replacements[token] = now.strftime(fmt)
|
|
58
|
+
|
|
59
|
+
def _replace(match: re.Match) -> str:
|
|
60
|
+
key = match.group(1)
|
|
61
|
+
if key in replacements:
|
|
62
|
+
return replacements[key]
|
|
63
|
+
# Autocomplete may insert uppercase tokens; users often type lowercase
|
|
64
|
+
# (`{hhmmss}`). Resolve date formats case-insensitively, and
|
|
65
|
+
# `{object}` / `{connector}` case-insensitively.
|
|
66
|
+
upper = key.upper()
|
|
67
|
+
if upper in _DATE_FORMATS:
|
|
68
|
+
return replacements[upper]
|
|
69
|
+
lower = key.lower()
|
|
70
|
+
if lower in ("object", "connector"):
|
|
71
|
+
return replacements[lower]
|
|
72
|
+
return match.group(0)
|
|
73
|
+
|
|
74
|
+
return _TOKEN_RE.sub(_replace, value)
|
|
@@ -71,12 +71,21 @@ class Mapping():
|
|
|
71
71
|
|
|
72
72
|
self._ensure_session(force_authenticate=False)
|
|
73
73
|
|
|
74
|
-
def
|
|
74
|
+
def _get_metadata(self, required=False):
|
|
75
|
+
"""
|
|
76
|
+
Fetch connector metadata.
|
|
77
|
+
|
|
78
|
+
On connect/authenticate paths, failures must surface (required=True) so
|
|
79
|
+
report-catalog connectors cannot look Connected with empty accounts.
|
|
80
|
+
Soft-fail (required=False) only for optional metadata enrichment.
|
|
81
|
+
"""
|
|
75
82
|
try:
|
|
76
83
|
return self.connector.get_metadata()
|
|
77
84
|
except errors.NotImplementedError:
|
|
78
85
|
return None
|
|
79
86
|
except Exception:
|
|
87
|
+
if required:
|
|
88
|
+
raise
|
|
80
89
|
logger.exception("Error getting metadata")
|
|
81
90
|
return None
|
|
82
91
|
|
|
@@ -86,7 +95,7 @@ class Mapping():
|
|
|
86
95
|
|
|
87
96
|
self._ensure_session(force_authenticate=force_authenticate)
|
|
88
97
|
|
|
89
|
-
metadata = self.
|
|
98
|
+
metadata = self._get_metadata(required=True)
|
|
90
99
|
results = {
|
|
91
100
|
"metadata": metadata
|
|
92
101
|
}
|
|
@@ -106,7 +115,7 @@ class Mapping():
|
|
|
106
115
|
)
|
|
107
116
|
|
|
108
117
|
objects = self.connector.get_objects()
|
|
109
|
-
metadata = self.
|
|
118
|
+
metadata = self._get_metadata(required=True)
|
|
110
119
|
|
|
111
120
|
results = {
|
|
112
121
|
"metadata": metadata,
|
|
@@ -122,7 +131,7 @@ class Mapping():
|
|
|
122
131
|
force_authenticate=force_authenticate,
|
|
123
132
|
)
|
|
124
133
|
|
|
125
|
-
metadata = self.
|
|
134
|
+
metadata = self._get_metadata(required=True)
|
|
126
135
|
return [{"metadata": metadata}, "Retrieved metadata"]
|
|
127
136
|
|
|
128
137
|
def get_objects_only(self, skip_authenticate=False, force_authenticate=False, include_metadata=True):
|
|
@@ -135,7 +144,8 @@ class Mapping():
|
|
|
135
144
|
results = {"objects": objects}
|
|
136
145
|
|
|
137
146
|
if include_metadata:
|
|
138
|
-
|
|
147
|
+
# Optional enrichment on object-only fetches; connect path uses required=True.
|
|
148
|
+
results["metadata"] = self._get_metadata(required=False)
|
|
139
149
|
|
|
140
150
|
return [results, "Retrieved objects"]
|
|
141
151
|
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import json, io, math, re, inspect
|
|
2
|
+
import os
|
|
2
3
|
import time
|
|
3
4
|
from .services.environment import PipelineEnvironment
|
|
4
5
|
from .services.api import DataConnectorAPI
|
|
@@ -6,10 +7,26 @@ from .models.log_templates import LogTemplates
|
|
|
6
7
|
from .services.aws import AwsService
|
|
7
8
|
from .models import errors
|
|
8
9
|
from .services.loader import load_connector
|
|
10
|
+
from .destination_object_template import resolve_destination_object_template
|
|
11
|
+
from dc_sdk.data_stream import DataStream
|
|
9
12
|
|
|
10
13
|
TEMP_UPLOADS = 'temporary-files'
|
|
11
14
|
SUB_FOLDER = 'etlJobHistory'
|
|
12
15
|
|
|
16
|
+
_FORMAT_EXTENSIONS = {
|
|
17
|
+
"csv": ".csv",
|
|
18
|
+
"ndjson": ".ndjson",
|
|
19
|
+
"json": ".json",
|
|
20
|
+
"parquet": ".parquet",
|
|
21
|
+
"xml": ".xml",
|
|
22
|
+
"xlsx": ".xlsx",
|
|
23
|
+
}
|
|
24
|
+
_CONTAINER_EXTENSIONS = {
|
|
25
|
+
"zip": ".zip",
|
|
26
|
+
"gzip": ".gz",
|
|
27
|
+
"tar": ".tar",
|
|
28
|
+
}
|
|
29
|
+
|
|
13
30
|
|
|
14
31
|
class PipelineConductor:
|
|
15
32
|
def __init__(self,
|
|
@@ -159,9 +176,49 @@ class PipelineConductor:
|
|
|
159
176
|
|
|
160
177
|
self.log(self.log_templates.GET_DATA_FINISH.format(self.row_count, self.pipeline_details.source_object_id))
|
|
161
178
|
|
|
179
|
+
def get_data_stream(self):
|
|
180
|
+
"""
|
|
181
|
+
POC: pull a DataStream from the connector and upload raw bytes to S3.
|
|
182
|
+
|
|
183
|
+
Not wired into the default SOURCE path yet — call explicitly when testing.
|
|
184
|
+
Retains extract/normalize hints via a sibling .meta.json object.
|
|
185
|
+
"""
|
|
186
|
+
if not hasattr(self.connector, "get_data_stream"):
|
|
187
|
+
raise errors.DataError(
|
|
188
|
+
"Connector does not implement get_data_stream; cannot use stream transfer."
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
self.log(self.log_templates.GET_DATA_START.format(
|
|
192
|
+
self.pipeline_details.source_object_id,
|
|
193
|
+
", ".join(str(fid) for fid in self._get_field_ids())
|
|
194
|
+
))
|
|
195
|
+
|
|
196
|
+
data_stream = self.connector.get_data_stream(
|
|
197
|
+
self.pipeline_details.source_object_id,
|
|
198
|
+
self._get_field_ids(),
|
|
199
|
+
filters=self._get_filters(),
|
|
200
|
+
options=self.pipeline_details.options,
|
|
201
|
+
)
|
|
202
|
+
if not isinstance(data_stream, DataStream):
|
|
203
|
+
raise errors.DataError(
|
|
204
|
+
"get_data_stream must return a dc_sdk.data_stream.DataStream instance."
|
|
205
|
+
)
|
|
206
|
+
|
|
207
|
+
key_name = self._process_data_stream(data_stream)
|
|
208
|
+
|
|
209
|
+
self.log(self.log_templates.GET_DATA_FINISH.format(
|
|
210
|
+
self.row_count, self.pipeline_details.source_object_id
|
|
211
|
+
))
|
|
212
|
+
return key_name
|
|
213
|
+
|
|
162
214
|
def load_data(self, batch_start):
|
|
163
215
|
self._ensure_mapping_has_default_data_types()
|
|
164
|
-
|
|
216
|
+
destination_object_id = resolve_destination_object_template(
|
|
217
|
+
self.pipeline_details.destination_object_id,
|
|
218
|
+
source_object_id=getattr(self.pipeline_details, "source_object_id", None),
|
|
219
|
+
source_connector_nm=getattr(self.pipeline_details, "source_connector_nm", None),
|
|
220
|
+
)
|
|
221
|
+
self.log(self.log_templates.LOAD_DATA_START.format(destination_object_id, len(self._get_field_ids())))
|
|
165
222
|
|
|
166
223
|
keys = []
|
|
167
224
|
if PipelineEnvironment.platform == "aws":
|
|
@@ -171,7 +228,7 @@ class PipelineConductor:
|
|
|
171
228
|
|
|
172
229
|
if not keys:
|
|
173
230
|
# Call load_data with empty data when there are no keys
|
|
174
|
-
loaded = self._call_connector_load_data([],
|
|
231
|
+
loaded = self._call_connector_load_data([], destination_object_id, self._get_mapping(), self.pipeline_details.update_method_cd, 0, 1)
|
|
175
232
|
if not loaded:
|
|
176
233
|
raise errors.LoadDataError("Loading data failed.")
|
|
177
234
|
else:
|
|
@@ -184,8 +241,8 @@ class PipelineConductor:
|
|
|
184
241
|
|
|
185
242
|
|
|
186
243
|
data = json.load(file_object)
|
|
187
|
-
self.log(self.log_templates.LOAD_DATA_LOADED.format(len(data),
|
|
188
|
-
loaded = self._call_connector_load_data(data,
|
|
244
|
+
self.log(self.log_templates.LOAD_DATA_LOADED.format(len(data), destination_object_id))
|
|
245
|
+
loaded = self._call_connector_load_data(data, destination_object_id, self._get_mapping(), self.pipeline_details.update_method_cd, index, len(keys))
|
|
189
246
|
if loaded:
|
|
190
247
|
self.row_count += len(data)
|
|
191
248
|
if self.mode == "prod":
|
|
@@ -196,7 +253,7 @@ class PipelineConductor:
|
|
|
196
253
|
self.aws.delete_object(key)
|
|
197
254
|
else:
|
|
198
255
|
raise errors.LoadDataError("Loading data failed.")
|
|
199
|
-
self.log(self.log_templates.LOAD_DATA_FINISHED.format(self.row_count,
|
|
256
|
+
self.log(self.log_templates.LOAD_DATA_FINISHED.format(self.row_count, destination_object_id))
|
|
200
257
|
self._persist_sync_cursor_after_load()
|
|
201
258
|
|
|
202
259
|
def _create_connector(self, connector_cls, credentials):
|
|
@@ -422,6 +479,84 @@ class PipelineConductor:
|
|
|
422
479
|
|
|
423
480
|
return limit_reached
|
|
424
481
|
|
|
482
|
+
def _process_data_stream(self, data_stream: DataStream):
|
|
483
|
+
"""Upload a DataStream's bytes to object storage and persist normalize hints."""
|
|
484
|
+
self.batch_index += 1
|
|
485
|
+
extension = self._data_stream_extension(data_stream)
|
|
486
|
+
|
|
487
|
+
if self.mode == "prod":
|
|
488
|
+
key_name = f"e{self.pipeline_run_history_id}-b{self.batch_index}{extension}"
|
|
489
|
+
else:
|
|
490
|
+
key_name = f"devTransfers/e{self.prefix}-b{self.batch_index}{extension}"
|
|
491
|
+
|
|
492
|
+
if PipelineEnvironment.platform == "aws":
|
|
493
|
+
key_name = f"transfers/{key_name}"
|
|
494
|
+
self.aws.upload_stream(
|
|
495
|
+
key_name,
|
|
496
|
+
data_stream.stream,
|
|
497
|
+
content_type=data_stream.media_type,
|
|
498
|
+
)
|
|
499
|
+
meta_key = f"{key_name}.meta.json"
|
|
500
|
+
meta_buffer = io.StringIO(json.dumps(data_stream.hints()))
|
|
501
|
+
self.aws.upload_object(meta_key, json_buffer=meta_buffer)
|
|
502
|
+
self.bytes_transferred += self.aws.get_object_size(key_name)
|
|
503
|
+
elif PipelineEnvironment.platform == "azure":
|
|
504
|
+
key_name = f"{PipelineEnvironment.app_env}/{key_name}"
|
|
505
|
+
if not hasattr(self.azure, "upload_stream"):
|
|
506
|
+
raise errors.DataError(
|
|
507
|
+
"Azure upload_stream is not implemented for DataStream POC."
|
|
508
|
+
)
|
|
509
|
+
self.azure.upload_stream(
|
|
510
|
+
key_name,
|
|
511
|
+
data_stream.stream,
|
|
512
|
+
content_type=data_stream.media_type,
|
|
513
|
+
)
|
|
514
|
+
meta_key = f"{key_name}.meta.json"
|
|
515
|
+
meta_buffer = io.StringIO(json.dumps(data_stream.hints()))
|
|
516
|
+
self.azure.upload_object(meta_key, json_buffer=meta_buffer)
|
|
517
|
+
self.bytes_transferred += self.azure.get_object_size(key_name)
|
|
518
|
+
else:
|
|
519
|
+
raise errors.DataError(
|
|
520
|
+
f"Unsupported platform for DataStream upload: {PipelineEnvironment.platform}"
|
|
521
|
+
)
|
|
522
|
+
|
|
523
|
+
self.successful_keys.append(key_name)
|
|
524
|
+
self.internal_log(self.log_templates.INTERNAL_S3_UPLOAD_LOG.format(key_name))
|
|
525
|
+
|
|
526
|
+
row_count = data_stream.metadata.get("row_count")
|
|
527
|
+
if row_count is not None:
|
|
528
|
+
try:
|
|
529
|
+
self.row_count += int(row_count)
|
|
530
|
+
except (TypeError, ValueError):
|
|
531
|
+
pass
|
|
532
|
+
|
|
533
|
+
if self.mode == "prod":
|
|
534
|
+
self.update_history({
|
|
535
|
+
"updateAction": "ROWS_RETRIEVED",
|
|
536
|
+
"RowsRetrievedNBR": self.row_count,
|
|
537
|
+
})
|
|
538
|
+
|
|
539
|
+
return key_name
|
|
540
|
+
|
|
541
|
+
@staticmethod
|
|
542
|
+
def _data_stream_extension(data_stream: DataStream) -> str:
|
|
543
|
+
if data_stream.filename:
|
|
544
|
+
_, ext = os.path.splitext(data_stream.filename)
|
|
545
|
+
if ext:
|
|
546
|
+
return ext.lower()
|
|
547
|
+
|
|
548
|
+
if data_stream.container:
|
|
549
|
+
container_ext = _CONTAINER_EXTENSIONS.get(data_stream.container.lower())
|
|
550
|
+
if container_ext:
|
|
551
|
+
return container_ext
|
|
552
|
+
|
|
553
|
+
if data_stream.format:
|
|
554
|
+
format_ext = _FORMAT_EXTENSIONS.get(data_stream.format.lower())
|
|
555
|
+
if format_ext:
|
|
556
|
+
return format_ext
|
|
557
|
+
|
|
558
|
+
return ".bin"
|
|
559
|
+
|
|
425
560
|
def _get_credentials(self):
|
|
426
561
|
is_source = self.task == "SOURCE"
|
|
427
562
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import os
|
|
2
2
|
import boto3
|
|
3
3
|
import json
|
|
4
|
-
from typing import Dict
|
|
4
|
+
from typing import Dict, Iterable, Optional
|
|
5
5
|
from base64 import b64decode
|
|
6
6
|
from Crypto.Cipher import AES
|
|
7
7
|
|
|
@@ -14,6 +14,38 @@ s3_resource = boto3.resource('s3')
|
|
|
14
14
|
ecs_resource = boto3.client('ecs')
|
|
15
15
|
kms_client = boto3.client('kms')
|
|
16
16
|
|
|
17
|
+
|
|
18
|
+
class _ChunkIteratorReader:
|
|
19
|
+
"""File-like wrapper so boto3 can upload an Iterable[bytes] via upload_fileobj."""
|
|
20
|
+
|
|
21
|
+
def __init__(self, chunks: Iterable[bytes]):
|
|
22
|
+
self._iterator = iter(chunks)
|
|
23
|
+
self._buffer = b""
|
|
24
|
+
self._exhausted = False
|
|
25
|
+
|
|
26
|
+
def read(self, size: int = -1) -> bytes:
|
|
27
|
+
if size == 0:
|
|
28
|
+
return b""
|
|
29
|
+
|
|
30
|
+
if size < 0:
|
|
31
|
+
parts = [self._buffer]
|
|
32
|
+
self._buffer = b""
|
|
33
|
+
if not self._exhausted:
|
|
34
|
+
parts.extend(self._iterator)
|
|
35
|
+
self._exhausted = True
|
|
36
|
+
return b"".join(parts)
|
|
37
|
+
|
|
38
|
+
while len(self._buffer) < size and not self._exhausted:
|
|
39
|
+
try:
|
|
40
|
+
self._buffer += next(self._iterator)
|
|
41
|
+
except StopIteration:
|
|
42
|
+
self._exhausted = True
|
|
43
|
+
|
|
44
|
+
out = self._buffer[:size]
|
|
45
|
+
self._buffer = self._buffer[size:]
|
|
46
|
+
return out
|
|
47
|
+
|
|
48
|
+
|
|
17
49
|
class AwsService:
|
|
18
50
|
def __init__(self, s3_bucket) -> None:
|
|
19
51
|
self.s3_bucket = s3_bucket
|
|
@@ -48,6 +80,25 @@ class AwsService:
|
|
|
48
80
|
else:
|
|
49
81
|
raise ValueError("Either json_buffer or file_path must be provided")
|
|
50
82
|
|
|
83
|
+
def upload_stream(
|
|
84
|
+
self,
|
|
85
|
+
key_name: str,
|
|
86
|
+
stream: Iterable[bytes],
|
|
87
|
+
content_type: Optional[str] = None,
|
|
88
|
+
):
|
|
89
|
+
"""Upload an iterable of byte chunks to S3 without requiring a local file."""
|
|
90
|
+
extra_args = {}
|
|
91
|
+
if content_type:
|
|
92
|
+
extra_args["ContentType"] = content_type
|
|
93
|
+
|
|
94
|
+
reader = _ChunkIteratorReader(stream)
|
|
95
|
+
if extra_args:
|
|
96
|
+
s3_client.upload_fileobj(
|
|
97
|
+
reader, self.s3_bucket, key_name, ExtraArgs=extra_args
|
|
98
|
+
)
|
|
99
|
+
else:
|
|
100
|
+
s3_client.upload_fileobj(reader, self.s3_bucket, key_name)
|
|
101
|
+
|
|
51
102
|
def get_object_size(self, key_name):
|
|
52
103
|
response = s3_client.head_object(
|
|
53
104
|
Bucket=self.s3_bucket, Key=key_name)
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
from typing import Protocol, Any, Dict, List, Optional
|
|
2
2
|
|
|
3
|
+
from dc_sdk.data_stream import DataStream
|
|
4
|
+
|
|
3
5
|
|
|
4
6
|
class ConnectorProtocol(Protocol):
|
|
5
7
|
|
|
@@ -25,6 +27,14 @@ class ConnectorProtocol(Protocol):
|
|
|
25
27
|
options: Dict[str, Any] = {}
|
|
26
28
|
) -> Any: ...
|
|
27
29
|
|
|
30
|
+
def get_data_stream(
|
|
31
|
+
self,
|
|
32
|
+
object_id: str,
|
|
33
|
+
field_ids: List[str],
|
|
34
|
+
filters: Optional[Dict[str, Any]] = None,
|
|
35
|
+
options: Dict[str, Any] = {}
|
|
36
|
+
) -> DataStream: ...
|
|
37
|
+
|
|
28
38
|
def load_data(
|
|
29
39
|
self,
|
|
30
40
|
data: List[Dict[str, Any]],
|
|
@@ -35,4 +45,4 @@ class ConnectorProtocol(Protocol):
|
|
|
35
45
|
total_batches: int
|
|
36
46
|
) -> Any: ...
|
|
37
47
|
|
|
38
|
-
def close(self) -> None: ...
|
|
48
|
+
def close(self) -> None: ...
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{dc_python_sdk-1.5.49 → dc_python_sdk-1.5.51}/src/dc_python_sdk.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{dc_python_sdk-1.5.49/src/dc_sdk/src → dc_python_sdk-1.5.51/src/dc_sdk/src/models}/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|