dc-python-sdk 1.5.54__tar.gz → 1.6.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dc_python_sdk-1.5.54/src/dc_python_sdk.egg-info → dc_python_sdk-1.6.1}/PKG-INFO +37 -2
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/README.md +35 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/pyproject.toml +2 -2
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/setup.cfg +2 -2
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1/src/dc_python_sdk.egg-info}/PKG-INFO +37 -2
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_python_sdk.egg-info/SOURCES.txt +1 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_python_sdk.egg-info/requires.txt +1 -1
- dc_python_sdk-1.6.1/src/dc_sdk/handler.py +122 -0
- dc_python_sdk-1.6.1/src/dc_sdk/row_filters.py +246 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/mapping.py +32 -12
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/pipeline.py +63 -31
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/services/api.py +16 -1
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/services/aws.py +8 -11
- dc_python_sdk-1.5.54/src/dc_sdk/handler.py +0 -242
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/LICENSE +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_python_sdk.egg-info/dependency_links.txt +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_python_sdk.egg-info/entry_points.txt +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_python_sdk.egg-info/top_level.txt +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/__init__.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/app.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/cli.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/data_stream.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/errors.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/file_utils.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/json_safe.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/__init__.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/ai.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/ai_http.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/connection_status.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/destination_object_template.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/models/__init__.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/models/enums.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/models/errors.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/models/log_templates.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/models/pipeline_details.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/server.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/services/__init__.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/services/environment.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/services/loader.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/services/logger.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/services/session.py +0 -0
- {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/types.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dc-python-sdk
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.6.1
|
|
4
4
|
Summary: Data Connector Python SDK
|
|
5
5
|
Home-page: https://github.com/data-connector/dc-python-sdk
|
|
6
6
|
Author: DataConnector
|
|
@@ -14,7 +14,7 @@ Description-Content-Type: text/markdown
|
|
|
14
14
|
License-File: LICENSE
|
|
15
15
|
Requires-Dist: fastapi
|
|
16
16
|
Requires-Dist: uvicorn
|
|
17
|
-
Requires-Dist: awslambdaric
|
|
17
|
+
Requires-Dist: awslambdaric==4.0.4
|
|
18
18
|
Requires-Dist: requests
|
|
19
19
|
Requires-Dist: boto3>=1.40.0
|
|
20
20
|
Requires-Dist: openai
|
|
@@ -94,6 +94,41 @@ def get_available_objects():
|
|
|
94
94
|
return objects
|
|
95
95
|
```
|
|
96
96
|
|
|
97
|
+
### Selective discovery in the Lambda handler
|
|
98
|
+
|
|
99
|
+
Authenticate with action `3`, then request only the metadata needed by the current
|
|
100
|
+
screen with action `5`:
|
|
101
|
+
|
|
102
|
+
```json
|
|
103
|
+
{"action": 5, "sections": ["accounts"], "credentials": {}}
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
Connectors opt in by defining `get_metadata(self, sections=None)`. Fabric supports
|
|
107
|
+
`accounts`, `schemas`, and `folders`; callers can request more than one section.
|
|
108
|
+
Requested sections retain their usual keys in the metadata dictionary, and static
|
|
109
|
+
capability fields may also be returned. Unrequested dynamic keys are omitted,
|
|
110
|
+
not returned as empty lists. Callers should merge partial metadata into existing
|
|
111
|
+
metadata for the same account, and clear cached metadata when switching accounts.
|
|
112
|
+
|
|
113
|
+
Action `6` accepts the same optional parameter for `get_objects(self, sections=None)`;
|
|
114
|
+
Fabric supports `tables` and `files`. Use `include_metadata: false` to avoid an
|
|
115
|
+
additional full metadata request when fetching only objects. Object results remain
|
|
116
|
+
a list in the existing response envelope.
|
|
117
|
+
|
|
118
|
+
Omitting `sections` (or using null) retains full discovery. An empty list requests
|
|
119
|
+
no dynamic sections on an opted-in connector. The SDK validates a list of non-empty
|
|
120
|
+
strings and passes it only to methods with an explicit `sections` keyword; legacy
|
|
121
|
+
connectors still perform full discovery with no arguments. Each opted-in connector
|
|
122
|
+
validates its supported section names. Internal connector TypeErrors are propagated,
|
|
123
|
+
never retried as a legacy call. The SDK does not filter a legacy response or claim
|
|
124
|
+
that a legacy connector avoided fetching unrequested data.
|
|
125
|
+
|
|
126
|
+
The SDK returns `error_phase: "metadata"` for discovery errors and keeps successful
|
|
127
|
+
authentication status. `AuthenticationError` still invalidates authentication.
|
|
128
|
+
API and UI consumers must use this distinction to show a discovery retry instead
|
|
129
|
+
of requiring reconnection. This contract applies to the Lambda handler; the local
|
|
130
|
+
HTTP server below uses a separate method-based request format.
|
|
131
|
+
|
|
97
132
|
### Running the local HTTP server
|
|
98
133
|
|
|
99
134
|
Install the SDK and start the HTTP server that wraps your connector:
|
|
@@ -61,6 +61,41 @@ def get_available_objects():
|
|
|
61
61
|
return objects
|
|
62
62
|
```
|
|
63
63
|
|
|
64
|
+
### Selective discovery in the Lambda handler
|
|
65
|
+
|
|
66
|
+
Authenticate with action `3`, then request only the metadata needed by the current
|
|
67
|
+
screen with action `5`:
|
|
68
|
+
|
|
69
|
+
```json
|
|
70
|
+
{"action": 5, "sections": ["accounts"], "credentials": {}}
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
Connectors opt in by defining `get_metadata(self, sections=None)`. Fabric supports
|
|
74
|
+
`accounts`, `schemas`, and `folders`; callers can request more than one section.
|
|
75
|
+
Requested sections retain their usual keys in the metadata dictionary, and static
|
|
76
|
+
capability fields may also be returned. Unrequested dynamic keys are omitted,
|
|
77
|
+
not returned as empty lists. Callers should merge partial metadata into existing
|
|
78
|
+
metadata for the same account, and clear cached metadata when switching accounts.
|
|
79
|
+
|
|
80
|
+
Action `6` accepts the same optional parameter for `get_objects(self, sections=None)`;
|
|
81
|
+
Fabric supports `tables` and `files`. Use `include_metadata: false` to avoid an
|
|
82
|
+
additional full metadata request when fetching only objects. Object results remain
|
|
83
|
+
a list in the existing response envelope.
|
|
84
|
+
|
|
85
|
+
Omitting `sections` (or using null) retains full discovery. An empty list requests
|
|
86
|
+
no dynamic sections on an opted-in connector. The SDK validates a list of non-empty
|
|
87
|
+
strings and passes it only to methods with an explicit `sections` keyword; legacy
|
|
88
|
+
connectors still perform full discovery with no arguments. Each opted-in connector
|
|
89
|
+
validates its supported section names. Internal connector TypeErrors are propagated,
|
|
90
|
+
never retried as a legacy call. The SDK does not filter a legacy response or claim
|
|
91
|
+
that a legacy connector avoided fetching unrequested data.
|
|
92
|
+
|
|
93
|
+
The SDK returns `error_phase: "metadata"` for discovery errors and keeps successful
|
|
94
|
+
authentication status. `AuthenticationError` still invalidates authentication.
|
|
95
|
+
API and UI consumers must use this distinction to show a discovery retry instead
|
|
96
|
+
of requiring reconnection. This contract applies to the Lambda handler; the local
|
|
97
|
+
HTTP server below uses a separate method-based request format.
|
|
98
|
+
|
|
64
99
|
### Running the local HTTP server
|
|
65
100
|
|
|
66
101
|
Install the SDK and start the HTTP server that wraps your connector:
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "dc-python-sdk"
|
|
7
|
-
version = "1.
|
|
7
|
+
version = "1.6.1"
|
|
8
8
|
description = "Data Connector Python SDK"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.6"
|
|
@@ -19,7 +19,7 @@ classifiers = [
|
|
|
19
19
|
dependencies = [
|
|
20
20
|
"fastapi",
|
|
21
21
|
"uvicorn",
|
|
22
|
-
"awslambdaric",
|
|
22
|
+
"awslambdaric==4.0.4",
|
|
23
23
|
"requests",
|
|
24
24
|
"boto3>=1.40.0",
|
|
25
25
|
"openai",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[metadata]
|
|
2
2
|
name = dc-python-sdk
|
|
3
|
-
version = 1.
|
|
3
|
+
version = 1.6.1
|
|
4
4
|
author = DataConnector
|
|
5
5
|
author_email = josh@dataconnector.com
|
|
6
6
|
description = A small example package
|
|
@@ -22,7 +22,7 @@ python_requires = >=3.6
|
|
|
22
22
|
install_requires =
|
|
23
23
|
fastapi
|
|
24
24
|
uvicorn
|
|
25
|
-
awslambdaric
|
|
25
|
+
awslambdaric==4.0.4
|
|
26
26
|
pycryptodome
|
|
27
27
|
requests
|
|
28
28
|
boto3>=1.40.0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dc-python-sdk
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.6.1
|
|
4
4
|
Summary: Data Connector Python SDK
|
|
5
5
|
Home-page: https://github.com/data-connector/dc-python-sdk
|
|
6
6
|
Author: DataConnector
|
|
@@ -14,7 +14,7 @@ Description-Content-Type: text/markdown
|
|
|
14
14
|
License-File: LICENSE
|
|
15
15
|
Requires-Dist: fastapi
|
|
16
16
|
Requires-Dist: uvicorn
|
|
17
|
-
Requires-Dist: awslambdaric
|
|
17
|
+
Requires-Dist: awslambdaric==4.0.4
|
|
18
18
|
Requires-Dist: requests
|
|
19
19
|
Requires-Dist: boto3>=1.40.0
|
|
20
20
|
Requires-Dist: openai
|
|
@@ -94,6 +94,41 @@ def get_available_objects():
|
|
|
94
94
|
return objects
|
|
95
95
|
```
|
|
96
96
|
|
|
97
|
+
### Selective discovery in the Lambda handler
|
|
98
|
+
|
|
99
|
+
Authenticate with action `3`, then request only the metadata needed by the current
|
|
100
|
+
screen with action `5`:
|
|
101
|
+
|
|
102
|
+
```json
|
|
103
|
+
{"action": 5, "sections": ["accounts"], "credentials": {}}
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
Connectors opt in by defining `get_metadata(self, sections=None)`. Fabric supports
|
|
107
|
+
`accounts`, `schemas`, and `folders`; callers can request more than one section.
|
|
108
|
+
Requested sections retain their usual keys in the metadata dictionary, and static
|
|
109
|
+
capability fields may also be returned. Unrequested dynamic keys are omitted,
|
|
110
|
+
not returned as empty lists. Callers should merge partial metadata into existing
|
|
111
|
+
metadata for the same account, and clear cached metadata when switching accounts.
|
|
112
|
+
|
|
113
|
+
Action `6` accepts the same optional parameter for `get_objects(self, sections=None)`;
|
|
114
|
+
Fabric supports `tables` and `files`. Use `include_metadata: false` to avoid an
|
|
115
|
+
additional full metadata request when fetching only objects. Object results remain
|
|
116
|
+
a list in the existing response envelope.
|
|
117
|
+
|
|
118
|
+
Omitting `sections` (or using null) retains full discovery. An empty list requests
|
|
119
|
+
no dynamic sections on an opted-in connector. The SDK validates a list of non-empty
|
|
120
|
+
strings and passes it only to methods with an explicit `sections` keyword; legacy
|
|
121
|
+
connectors still perform full discovery with no arguments. Each opted-in connector
|
|
122
|
+
validates its supported section names. Internal connector TypeErrors are propagated,
|
|
123
|
+
never retried as a legacy call. The SDK does not filter a legacy response or claim
|
|
124
|
+
that a legacy connector avoided fetching unrequested data.
|
|
125
|
+
|
|
126
|
+
The SDK returns `error_phase: "metadata"` for discovery errors and keeps successful
|
|
127
|
+
authentication status. `AuthenticationError` still invalidates authentication.
|
|
128
|
+
API and UI consumers must use this distinction to show a discovery retry instead
|
|
129
|
+
of requiring reconnection. This contract applies to the Lambda handler; the local
|
|
130
|
+
HTTP server below uses a separate method-based request format.
|
|
131
|
+
|
|
97
132
|
### Running the local HTTP server
|
|
98
133
|
|
|
99
134
|
Install the SDK and start the HTTP server that wraps your connector:
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
from dc_sdk.row_filters import apply_data_filters, compile_sync_filters
|
|
2
|
+
from dc_sdk.errors import Error, AuthenticationError
|
|
3
|
+
from dc_sdk.json_safe import to_json_safe
|
|
4
|
+
from dc_sdk.src.mapping import Mapping
|
|
5
|
+
from dc_sdk.src.connection_status import STATUS_ERROR
|
|
6
|
+
import traceback
|
|
7
|
+
from importlib.metadata import version
|
|
8
|
+
|
|
9
|
+
def get_action_name(action: int) -> str:
|
|
10
|
+
return {
|
|
11
|
+
0: "authenticating",
|
|
12
|
+
1: "retrieving fields",
|
|
13
|
+
2: "retrieving 5 row preview",
|
|
14
|
+
3: "testing connection",
|
|
15
|
+
4: "starting the connector",
|
|
16
|
+
5: "retrieving metadata",
|
|
17
|
+
6: "retrieving objects",
|
|
18
|
+
}[action]
|
|
19
|
+
|
|
20
|
+
def handler(event, context):
|
|
21
|
+
print("version: ", version("dc-python-sdk"))
|
|
22
|
+
"""Lambda Handler"""
|
|
23
|
+
action_name = None
|
|
24
|
+
internal_error = False
|
|
25
|
+
message = None
|
|
26
|
+
results = None
|
|
27
|
+
error = None
|
|
28
|
+
error_phase = None
|
|
29
|
+
mapping = None
|
|
30
|
+
action = int(event['action']) if 'action' in event else 4
|
|
31
|
+
get_objects = event.get('get_objects', True)
|
|
32
|
+
skip_authenticate = event.get('skip_authenticate', False)
|
|
33
|
+
force_authenticate = event.get('force_authenticate', False)
|
|
34
|
+
include_metadata = event.get('include_metadata', True)
|
|
35
|
+
credentials_dict = event['credentials'] if 'credentials' in event else None
|
|
36
|
+
object_id = event['object_id'] if 'object_id' in event else None
|
|
37
|
+
field_ids = event['mapping'] if 'mapping' in event else None
|
|
38
|
+
options = event['options'] if 'options' in event else dict()
|
|
39
|
+
next_page = event['next_page'] if 'next_page' in event else None
|
|
40
|
+
n_rows = event['n_rows'] if 'n_rows' in event else None
|
|
41
|
+
filters = event['filters'] if 'filters' in event else None
|
|
42
|
+
data_filters = event['data_filters'] if 'data_filters' in event else None
|
|
43
|
+
|
|
44
|
+
try:
|
|
45
|
+
action_name = get_action_name(action)
|
|
46
|
+
mapping = Mapping(credentials_dict)
|
|
47
|
+
|
|
48
|
+
if action == 0:
|
|
49
|
+
if get_objects:
|
|
50
|
+
results, message = mapping.connect_get_objects(
|
|
51
|
+
skip_authenticate=skip_authenticate,
|
|
52
|
+
force_authenticate=force_authenticate,
|
|
53
|
+
include_metadata=include_metadata,
|
|
54
|
+
)
|
|
55
|
+
else:
|
|
56
|
+
results, message = mapping.authenticate(
|
|
57
|
+
force_authenticate=force_authenticate or not skip_authenticate,
|
|
58
|
+
include_metadata=include_metadata,
|
|
59
|
+
)
|
|
60
|
+
elif action == 1:
|
|
61
|
+
results, message = mapping.get_fields(object_id, options)
|
|
62
|
+
elif action == 2:
|
|
63
|
+
results, message = mapping.get_five_row_preview(
|
|
64
|
+
object_id, field_ids, options, n_rows, filters, next_page)
|
|
65
|
+
# During rollout callers send both formats. Apply the canonical one
|
|
66
|
+
# once, so legacy date comparisons cannot remove valid whole-day matches.
|
|
67
|
+
preview_filters = (
|
|
68
|
+
compile_sync_filters(options['sync_filters'])
|
|
69
|
+
if isinstance(options, dict) and 'sync_filters' in options
|
|
70
|
+
else data_filters
|
|
71
|
+
)
|
|
72
|
+
results = apply_data_filters(results, preview_filters)
|
|
73
|
+
elif action == 3:
|
|
74
|
+
results, message = mapping.test_connection()
|
|
75
|
+
elif action == 5:
|
|
76
|
+
results, message = mapping.get_metadata_only(
|
|
77
|
+
skip_authenticate=skip_authenticate,
|
|
78
|
+
force_authenticate=force_authenticate,
|
|
79
|
+
sections=event.get('sections'),
|
|
80
|
+
)
|
|
81
|
+
elif action == 6:
|
|
82
|
+
results, message = mapping.get_objects_only(
|
|
83
|
+
skip_authenticate=skip_authenticate,
|
|
84
|
+
force_authenticate=force_authenticate,
|
|
85
|
+
include_metadata=include_metadata,
|
|
86
|
+
sections=event.get('sections'),
|
|
87
|
+
)
|
|
88
|
+
else:
|
|
89
|
+
raise Error("Invalid action", "InvalidActionError")
|
|
90
|
+
except Error as e:
|
|
91
|
+
error_phase = "authentication" if isinstance(e, AuthenticationError) else getattr(mapping, "request_phase", None)
|
|
92
|
+
message = e.message or str(e)
|
|
93
|
+
internal_error = e.internal
|
|
94
|
+
error_trace = traceback.format_exc()
|
|
95
|
+
error = error_trace
|
|
96
|
+
if mapping and (isinstance(e, AuthenticationError) or getattr(mapping, "request_phase", None) != "metadata"):
|
|
97
|
+
mapping._set_connection_status(STATUS_ERROR)
|
|
98
|
+
|
|
99
|
+
except Exception as e:
|
|
100
|
+
error_phase = getattr(mapping, "request_phase", None)
|
|
101
|
+
error_trace = traceback.format_exc()
|
|
102
|
+
error = error_trace
|
|
103
|
+
internal_error = True
|
|
104
|
+
if mapping and getattr(mapping, "request_phase", None) != "metadata":
|
|
105
|
+
mapping._set_connection_status(STATUS_ERROR)
|
|
106
|
+
|
|
107
|
+
if error:
|
|
108
|
+
print(error)
|
|
109
|
+
|
|
110
|
+
last_status_dsc = None
|
|
111
|
+
if mapping and mapping.connector.credentials:
|
|
112
|
+
last_status_dsc = mapping.connector.credentials.get("last_status_dsc")
|
|
113
|
+
|
|
114
|
+
return {
|
|
115
|
+
'message': message if not internal_error else f"Something went wrong with {action_name}",
|
|
116
|
+
'results': to_json_safe(results),
|
|
117
|
+
'credentials': mapping.connector.credentials if mapping else None,
|
|
118
|
+
'last_status_dsc': last_status_dsc,
|
|
119
|
+
'error': error,
|
|
120
|
+
'internal_error': internal_error,
|
|
121
|
+
'error_phase': error_phase,
|
|
122
|
+
}
|
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
from datetime import date, datetime, timedelta, timezone
|
|
2
|
+
from zoneinfo import ZoneInfo
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def _calendar_date(value, zone):
|
|
6
|
+
if isinstance(value, datetime):
|
|
7
|
+
return value.astimezone(zone).date() if value.tzinfo else value.date()
|
|
8
|
+
if isinstance(value, date):
|
|
9
|
+
return value
|
|
10
|
+
text = str(value).strip()
|
|
11
|
+
# Date-only values represent calendar dates, not UTC midnight.
|
|
12
|
+
if len(text) == 10:
|
|
13
|
+
return date.fromisoformat(text)
|
|
14
|
+
parsed = datetime.fromisoformat(text.replace('Z', '+00:00'))
|
|
15
|
+
return parsed.astimezone(zone).date() if parsed.tzinfo else parsed.date()
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def compile_sync_filters(config, now=None):
|
|
19
|
+
"""Resolve relative dates once per run; date boundaries include entire days."""
|
|
20
|
+
if not config:
|
|
21
|
+
return []
|
|
22
|
+
zone_name = config.get('timeZone') or 'UTC'
|
|
23
|
+
zone = ZoneInfo(zone_name)
|
|
24
|
+
today = (now or datetime.now(timezone.utc)).astimezone(zone).date()
|
|
25
|
+
|
|
26
|
+
def resolve(value):
|
|
27
|
+
offsets = {'__dq:today': 0, '__dq:yesterday': -1, '__dq:tomorrow': 1}
|
|
28
|
+
if value in offsets:
|
|
29
|
+
return today + timedelta(days=offsets[value])
|
|
30
|
+
return _calendar_date(value, zone)
|
|
31
|
+
|
|
32
|
+
supported = {
|
|
33
|
+
'is_equal', 'is_not_equal', 'is_empty', 'is_not_empty',
|
|
34
|
+
'text_contains', 'text_not_contains', 'text_starts_with', 'text_ends_with',
|
|
35
|
+
'gt', 'gte', 'lt', 'lte', 'between', 'not_between',
|
|
36
|
+
'date_is', 'date_before', 'date_after',
|
|
37
|
+
}
|
|
38
|
+
result = []
|
|
39
|
+
for column, entry in (config.get('values') or {}).items():
|
|
40
|
+
operator = entry.get('operator', 'none')
|
|
41
|
+
if operator == 'none':
|
|
42
|
+
continue
|
|
43
|
+
if operator not in supported:
|
|
44
|
+
raise ValueError('Unsupported sync filter operator.')
|
|
45
|
+
is_date = entry.get('kind') == 'date' or operator.startswith('date_')
|
|
46
|
+
value, value2 = entry.get('value'), entry.get('value2')
|
|
47
|
+
if operator not in ('is_empty', 'is_not_empty'):
|
|
48
|
+
if value is None or str(value).strip() == '':
|
|
49
|
+
raise ValueError('A sync filter value is required.')
|
|
50
|
+
if is_date:
|
|
51
|
+
value = resolve(value).isoformat()
|
|
52
|
+
if operator in ('between', 'not_between'):
|
|
53
|
+
if value2 is None or str(value2).strip() == '':
|
|
54
|
+
raise ValueError('Both sync filter range values are required.')
|
|
55
|
+
if is_date:
|
|
56
|
+
value2 = resolve(value2).isoformat()
|
|
57
|
+
if value > value2:
|
|
58
|
+
raise ValueError('The filter start date must be on or before its end date.')
|
|
59
|
+
result.append({
|
|
60
|
+
'column_name': str(column), 'operator_cd': operator,
|
|
61
|
+
'value_1_txt': value, 'value_2_txt': value2,
|
|
62
|
+
'kind': 'date' if is_date else entry.get('kind'), 'time_zone': zone_name,
|
|
63
|
+
})
|
|
64
|
+
|
|
65
|
+
snapshot = config.get('date') or {}
|
|
66
|
+
bounds = {}
|
|
67
|
+
for side in ('start', 'end'):
|
|
68
|
+
selection = int(snapshot.get(side + 'SelectionDR') or 0)
|
|
69
|
+
if not selection:
|
|
70
|
+
continue
|
|
71
|
+
if not snapshot.get('filterObject'):
|
|
72
|
+
raise ValueError('Choose a column for the date range.')
|
|
73
|
+
if selection == 1:
|
|
74
|
+
value = today
|
|
75
|
+
elif selection == 2:
|
|
76
|
+
value = today - timedelta(days=1)
|
|
77
|
+
elif selection == 3:
|
|
78
|
+
days = int(snapshot[side + 'DaysFromToday'])
|
|
79
|
+
if days < 0:
|
|
80
|
+
raise ValueError('Days before today cannot be negative.')
|
|
81
|
+
value = today - timedelta(days=days)
|
|
82
|
+
elif selection == 4:
|
|
83
|
+
value = resolve(snapshot[side + 'CustomDate'])
|
|
84
|
+
else:
|
|
85
|
+
raise ValueError('Unsupported date range selection.')
|
|
86
|
+
bounds[side] = value
|
|
87
|
+
result.append({
|
|
88
|
+
'column_name': str(snapshot['filterObject']),
|
|
89
|
+
'operator_cd': 'gte' if side == 'start' else 'lte',
|
|
90
|
+
'value_1_txt': value.isoformat(), 'kind': 'date', 'time_zone': zone_name,
|
|
91
|
+
})
|
|
92
|
+
if 'start' in bounds and 'end' in bounds and bounds['start'] > bounds['end']:
|
|
93
|
+
raise ValueError('The filter start date must be on or before its end date.')
|
|
94
|
+
return result
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def apply_data_filters(results, data_filters):
|
|
98
|
+
if not data_filters:
|
|
99
|
+
return results
|
|
100
|
+
|
|
101
|
+
# Connectors may return either:
|
|
102
|
+
# 1) a bare list of rows, or
|
|
103
|
+
# 2) an envelope: {"data": [...], "next_page": ...}
|
|
104
|
+
is_envelope = isinstance(results, dict) and isinstance(results.get("data"), list)
|
|
105
|
+
rows = results.get("data") if is_envelope else results
|
|
106
|
+
if not isinstance(rows, list):
|
|
107
|
+
return results
|
|
108
|
+
|
|
109
|
+
# 🔹 UI → backend operator mapping
|
|
110
|
+
OPERATOR_MAP = {
|
|
111
|
+
# text
|
|
112
|
+
"text_contains": "CONTAINS",
|
|
113
|
+
"text_not_contains": "NOT_CONTAINS",
|
|
114
|
+
"text_starts_with": "STARTS_WITH",
|
|
115
|
+
"text_ends_with": "ENDS_WITH",
|
|
116
|
+
|
|
117
|
+
# equality
|
|
118
|
+
"is_equal": "EQ",
|
|
119
|
+
"is_not_equal": "NEQ",
|
|
120
|
+
"date_is": "EQ",
|
|
121
|
+
"date_before": "LT",
|
|
122
|
+
"date_after": "GT",
|
|
123
|
+
|
|
124
|
+
# empty
|
|
125
|
+
"is_empty": "IS_EMPTY",
|
|
126
|
+
"is_not_empty": "IS_NOT_EMPTY",
|
|
127
|
+
|
|
128
|
+
# numeric
|
|
129
|
+
"gt": "GT",
|
|
130
|
+
"gte": "GTE",
|
|
131
|
+
"lt": "LT",
|
|
132
|
+
"lte": "LTE",
|
|
133
|
+
|
|
134
|
+
# range
|
|
135
|
+
"between": "BETWEEN",
|
|
136
|
+
"not_between": "NOT_BETWEEN",
|
|
137
|
+
|
|
138
|
+
# ignore
|
|
139
|
+
"none": None,
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
def try_parse_number(val):
|
|
143
|
+
try:
|
|
144
|
+
return float(val)
|
|
145
|
+
except:
|
|
146
|
+
return val
|
|
147
|
+
|
|
148
|
+
def normalize(row_val, v1, v2=None):
|
|
149
|
+
row_num = try_parse_number(row_val)
|
|
150
|
+
v1_num = try_parse_number(v1)
|
|
151
|
+
v2_num = try_parse_number(v2) if v2 is not None else None
|
|
152
|
+
|
|
153
|
+
# If both are numbers → compare as numbers
|
|
154
|
+
if isinstance(row_num, float) and isinstance(v1_num, float):
|
|
155
|
+
return row_num, v1_num, v2_num
|
|
156
|
+
|
|
157
|
+
# Otherwise → compare as strings
|
|
158
|
+
return str(row_val), str(v1), str(v2) if v2 is not None else None
|
|
159
|
+
|
|
160
|
+
def match(row, f):
|
|
161
|
+
ui_operator = f.get("operator_cd")
|
|
162
|
+
operator = OPERATOR_MAP.get(ui_operator)
|
|
163
|
+
|
|
164
|
+
if operator is None:
|
|
165
|
+
if ui_operator == 'none':
|
|
166
|
+
return True
|
|
167
|
+
raise ValueError('Unsupported sync filter operator.')
|
|
168
|
+
|
|
169
|
+
col = f.get("column_name")
|
|
170
|
+
val1 = f.get("value_1_txt")
|
|
171
|
+
val2 = f.get("value_2_txt")
|
|
172
|
+
|
|
173
|
+
# Some connectors can emit non-dict rows; keep filtering resilient.
|
|
174
|
+
if not isinstance(row, dict):
|
|
175
|
+
return False
|
|
176
|
+
row_val = row.get(col)
|
|
177
|
+
|
|
178
|
+
# ---- EMPTY HANDLING ----
|
|
179
|
+
if operator == "IS_EMPTY":
|
|
180
|
+
return row_val is None or str(row_val).strip() == ""
|
|
181
|
+
|
|
182
|
+
if operator == "IS_NOT_EMPTY":
|
|
183
|
+
return row_val is not None and str(row_val).strip() != ""
|
|
184
|
+
|
|
185
|
+
if row_val is None:
|
|
186
|
+
return False
|
|
187
|
+
|
|
188
|
+
# Normalize values
|
|
189
|
+
if f.get('kind') == 'date' or ui_operator.startswith('date_'):
|
|
190
|
+
zone = ZoneInfo(f.get('time_zone') or 'UTC')
|
|
191
|
+
v1 = _calendar_date(val1, zone)
|
|
192
|
+
v2 = _calendar_date(val2, zone) if val2 not in (None, '') else None
|
|
193
|
+
try:
|
|
194
|
+
row_val = _calendar_date(row_val, zone)
|
|
195
|
+
except (ValueError, TypeError):
|
|
196
|
+
return False
|
|
197
|
+
else:
|
|
198
|
+
row_val, v1, v2 = normalize(row_val, val1, val2)
|
|
199
|
+
|
|
200
|
+
# ---- OPERATORS ----
|
|
201
|
+
|
|
202
|
+
if operator == "EQ":
|
|
203
|
+
return row_val == v1
|
|
204
|
+
|
|
205
|
+
elif operator == "NEQ":
|
|
206
|
+
return row_val != v1
|
|
207
|
+
|
|
208
|
+
elif operator == "GT":
|
|
209
|
+
return row_val > v1
|
|
210
|
+
|
|
211
|
+
elif operator == "GTE":
|
|
212
|
+
return row_val >= v1
|
|
213
|
+
|
|
214
|
+
elif operator == "LT":
|
|
215
|
+
return row_val < v1
|
|
216
|
+
|
|
217
|
+
elif operator == "LTE":
|
|
218
|
+
return row_val <= v1
|
|
219
|
+
|
|
220
|
+
elif operator == "BETWEEN":
|
|
221
|
+
return v1 <= row_val <= v2
|
|
222
|
+
|
|
223
|
+
elif operator == "NOT_BETWEEN":
|
|
224
|
+
return not (v1 <= row_val <= v2)
|
|
225
|
+
|
|
226
|
+
elif operator == "CONTAINS":
|
|
227
|
+
return str(v1).lower() in str(row_val).lower()
|
|
228
|
+
|
|
229
|
+
elif operator == "NOT_CONTAINS":
|
|
230
|
+
return str(v1).lower() not in str(row_val).lower()
|
|
231
|
+
|
|
232
|
+
elif operator == "STARTS_WITH":
|
|
233
|
+
return str(row_val).lower().startswith(str(v1).lower())
|
|
234
|
+
|
|
235
|
+
elif operator == "ENDS_WITH":
|
|
236
|
+
return str(row_val).lower().endswith(str(v1).lower())
|
|
237
|
+
|
|
238
|
+
return True
|
|
239
|
+
|
|
240
|
+
# 🔹 Apply filters (AND logic)
|
|
241
|
+
filtered_rows = [row for row in rows if all(match(row, f) for f in data_filters)]
|
|
242
|
+
if is_envelope:
|
|
243
|
+
return {**results, "data": filtered_rows}
|
|
244
|
+
return filtered_rows
|
|
245
|
+
|
|
246
|
+
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
from dc_sdk import errors
|
|
2
2
|
import logging
|
|
3
|
+
import inspect
|
|
3
4
|
from dc_sdk.src.services.loader import load_connector
|
|
4
5
|
from dc_sdk.src.connection_status import (
|
|
5
6
|
STATUS_CONNECTED,
|
|
@@ -14,6 +15,7 @@ class Mapping():
|
|
|
14
15
|
def __init__(self, credentials):
|
|
15
16
|
Connector = load_connector()
|
|
16
17
|
self.connector = Connector(credentials)
|
|
18
|
+
self.request_phase = "authentication"
|
|
17
19
|
|
|
18
20
|
def _set_connection_status(self, status):
|
|
19
21
|
if self.connector.credentials is None:
|
|
@@ -35,6 +37,7 @@ class Mapping():
|
|
|
35
37
|
raise errors.AuthenticationError(message="Failed to initialize connection session")
|
|
36
38
|
|
|
37
39
|
def _ensure_session(self, force_authenticate=False):
|
|
40
|
+
self.request_phase = "authentication"
|
|
38
41
|
credentials = self.connector.credentials or {}
|
|
39
42
|
status = credentials.get("last_status_dsc")
|
|
40
43
|
|
|
@@ -71,16 +74,31 @@ class Mapping():
|
|
|
71
74
|
|
|
72
75
|
self._ensure_session(force_authenticate=False)
|
|
73
76
|
|
|
74
|
-
def
|
|
77
|
+
def _discover(self, method_name, sections=None):
|
|
78
|
+
"""Pass sections only to connectors that explicitly support the keyword."""
|
|
79
|
+
self.request_phase = "metadata"
|
|
80
|
+
method = getattr(self.connector, method_name)
|
|
81
|
+
if sections is None:
|
|
82
|
+
return method()
|
|
83
|
+
if not isinstance(sections, list) or any(not isinstance(section, str) or not section.strip() for section in sections):
|
|
84
|
+
raise ValueError("sections must be a list of non-empty strings")
|
|
85
|
+
parameter = inspect.signature(method).parameters.get("sections")
|
|
86
|
+
if parameter and parameter.kind in (inspect.Parameter.POSITIONAL_OR_KEYWORD, inspect.Parameter.KEYWORD_ONLY):
|
|
87
|
+
return method(sections=list(dict.fromkeys(sections)))
|
|
88
|
+
# Legacy connectors retain full discovery and their original response shape.
|
|
89
|
+
return method()
|
|
90
|
+
|
|
91
|
+
def _get_metadata(self, required=False, sections=None):
|
|
75
92
|
"""
|
|
76
93
|
Fetch connector metadata.
|
|
77
94
|
|
|
78
|
-
|
|
79
|
-
|
|
95
|
+
Required discovery failures still surface to the caller, but do not
|
|
96
|
+
invalidate successful authentication unless credentials were rejected.
|
|
80
97
|
Soft-fail (required=False) only for optional metadata enrichment.
|
|
81
98
|
"""
|
|
99
|
+
self.request_phase = "metadata"
|
|
82
100
|
try:
|
|
83
|
-
return self.
|
|
101
|
+
return self._discover("get_metadata", sections)
|
|
84
102
|
except errors.NotImplementedError:
|
|
85
103
|
return None
|
|
86
104
|
except Exception:
|
|
@@ -89,13 +107,13 @@ class Mapping():
|
|
|
89
107
|
logger.exception("Error getting metadata")
|
|
90
108
|
return None
|
|
91
109
|
|
|
92
|
-
def authenticate(self, force_authenticate=True):
|
|
110
|
+
def authenticate(self, force_authenticate=True, include_metadata=True):
|
|
93
111
|
results = None
|
|
94
112
|
message = None
|
|
95
113
|
|
|
96
114
|
self._ensure_session(force_authenticate=force_authenticate)
|
|
97
115
|
|
|
98
|
-
metadata = self._get_metadata(required=True)
|
|
116
|
+
metadata = self._get_metadata(required=True) if include_metadata else None
|
|
99
117
|
results = {
|
|
100
118
|
"metadata": metadata
|
|
101
119
|
}
|
|
@@ -103,7 +121,7 @@ class Mapping():
|
|
|
103
121
|
|
|
104
122
|
return [results, message]
|
|
105
123
|
|
|
106
|
-
def connect_get_objects(self, skip_authenticate=False, force_authenticate=False):
|
|
124
|
+
def connect_get_objects(self, skip_authenticate=False, force_authenticate=False, include_metadata=True):
|
|
107
125
|
results = None
|
|
108
126
|
metadata = None
|
|
109
127
|
objects = None
|
|
@@ -114,8 +132,9 @@ class Mapping():
|
|
|
114
132
|
force_authenticate=force_authenticate,
|
|
115
133
|
)
|
|
116
134
|
|
|
135
|
+
self.request_phase = "metadata"
|
|
117
136
|
objects = self.connector.get_objects()
|
|
118
|
-
metadata = self._get_metadata(required=True)
|
|
137
|
+
metadata = self._get_metadata(required=True) if include_metadata else None
|
|
119
138
|
|
|
120
139
|
results = {
|
|
121
140
|
"metadata": metadata,
|
|
@@ -125,22 +144,23 @@ class Mapping():
|
|
|
125
144
|
|
|
126
145
|
return [results, message]
|
|
127
146
|
|
|
128
|
-
def get_metadata_only(self, skip_authenticate=False, force_authenticate=False):
|
|
147
|
+
def get_metadata_only(self, skip_authenticate=False, force_authenticate=False, sections=None):
|
|
129
148
|
self._resolve_session(
|
|
130
149
|
skip_authenticate=skip_authenticate,
|
|
131
150
|
force_authenticate=force_authenticate,
|
|
132
151
|
)
|
|
133
152
|
|
|
134
|
-
metadata = self._get_metadata(required=True)
|
|
153
|
+
metadata = self._get_metadata(required=True, sections=sections)
|
|
135
154
|
return [{"metadata": metadata}, "Retrieved metadata"]
|
|
136
155
|
|
|
137
|
-
def get_objects_only(self, skip_authenticate=False, force_authenticate=False, include_metadata=True):
|
|
156
|
+
def get_objects_only(self, skip_authenticate=False, force_authenticate=False, include_metadata=True, sections=None):
|
|
138
157
|
self._resolve_session(
|
|
139
158
|
skip_authenticate=skip_authenticate,
|
|
140
159
|
force_authenticate=force_authenticate,
|
|
141
160
|
)
|
|
142
161
|
|
|
143
|
-
|
|
162
|
+
self.request_phase = "metadata"
|
|
163
|
+
objects = self._discover("get_objects", sections)
|
|
144
164
|
results = {"objects": objects}
|
|
145
165
|
|
|
146
166
|
if include_metadata:
|
|
@@ -10,6 +10,7 @@ from .services.loader import load_connector
|
|
|
10
10
|
from .destination_object_template import resolve_destination_object_template
|
|
11
11
|
from dc_sdk.data_stream import DataStream
|
|
12
12
|
from dc_sdk.json_safe import to_json_safe
|
|
13
|
+
from dc_sdk.row_filters import compile_sync_filters, apply_data_filters
|
|
13
14
|
|
|
14
15
|
TEMP_UPLOADS = 'temporary-files'
|
|
15
16
|
SUB_FOLDER = 'etlJobHistory'
|
|
@@ -82,6 +83,8 @@ class PipelineConductor:
|
|
|
82
83
|
while self.authentication_tries <= 3 and not authenticated:
|
|
83
84
|
try:
|
|
84
85
|
authenticated = self.connector.authenticate()
|
|
86
|
+
if not authenticated:
|
|
87
|
+
raise errors.AuthenticationError("Connector authentication failed.")
|
|
85
88
|
except Exception as e:
|
|
86
89
|
if self.authentication_tries == 3 or "firewall" in str(e): # todo: fix add retry flg in raised errors
|
|
87
90
|
raise e
|
|
@@ -97,9 +100,9 @@ class PipelineConductor:
|
|
|
97
100
|
response = self.api.get_credential_updates(True, self.pipeline_id)
|
|
98
101
|
|
|
99
102
|
creds = self.aws.decrypt_customer_data_object(response['credential'], response['organization'],
|
|
100
|
-
encrypted_data_key_txt=
|
|
101
|
-
encryption_iv_txt=
|
|
102
|
-
encryption_auth_tag_txt=
|
|
103
|
+
encrypted_data_key_txt=response.get('encrypted_data_key_txt'),
|
|
104
|
+
encryption_iv_txt=response.get('encryption_iv_txt'),
|
|
105
|
+
encryption_auth_tag_txt=response.get('encryption_auth_tag_txt')
|
|
103
106
|
)
|
|
104
107
|
|
|
105
108
|
self.connector.credentials = creds
|
|
@@ -108,9 +111,9 @@ class PipelineConductor:
|
|
|
108
111
|
response = self.api.get_credential_updates(False, self.pipeline_id)
|
|
109
112
|
|
|
110
113
|
creds = self.aws.decrypt_customer_data_object(response['credential'], response['organization'],
|
|
111
|
-
encrypted_data_key_txt=
|
|
112
|
-
encryption_iv_txt=
|
|
113
|
-
encryption_auth_tag_txt=
|
|
114
|
+
encrypted_data_key_txt=response.get('encrypted_data_key_txt'),
|
|
115
|
+
encryption_iv_txt=response.get('encryption_iv_txt'),
|
|
116
|
+
encryption_auth_tag_txt=response.get('encryption_auth_tag_txt')
|
|
114
117
|
)
|
|
115
118
|
|
|
116
119
|
self.connector.credentials = creds
|
|
@@ -124,6 +127,8 @@ class PipelineConductor:
|
|
|
124
127
|
while self.authentication_tries <= 3 and not authenticated:
|
|
125
128
|
try:
|
|
126
129
|
authenticated = self.connector.authenticate()
|
|
130
|
+
if not authenticated:
|
|
131
|
+
raise errors.AuthenticationError("Connector authentication failed.")
|
|
127
132
|
except Exception as e:
|
|
128
133
|
if self.authentication_tries == 3 or "firewall" in str(e): # todo: fix add retry flg in raised errors
|
|
129
134
|
raise e
|
|
@@ -137,6 +142,7 @@ class PipelineConductor:
|
|
|
137
142
|
|
|
138
143
|
def get_data(self):
|
|
139
144
|
# Determine batch size
|
|
145
|
+
self._get_data_filters() # Validate and resolve relative dates before fetching any rows.
|
|
140
146
|
|
|
141
147
|
self.log(self.log_templates.GET_DATA_START.format(
|
|
142
148
|
self.pipeline_details.source_object_id,
|
|
@@ -146,31 +152,22 @@ class PipelineConductor:
|
|
|
146
152
|
nrows = self._get_batch_row_count()
|
|
147
153
|
max_allowed = self.pipeline_details.max_allowed_retrieval
|
|
148
154
|
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
self.
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
# Check if we've reached the limit before fetching next page
|
|
155
|
+
next_page = None
|
|
156
|
+
while True:
|
|
157
|
+
results = self.connector.get_data(
|
|
158
|
+
self.pipeline_details.source_object_id, self._get_source_field_ids(),
|
|
159
|
+
n_rows=nrows, filters=self._get_filters(),
|
|
160
|
+
options=self.pipeline_details.options, next_page=next_page,
|
|
161
|
+
)
|
|
162
|
+
self._update_extracted_sync_cursor(results.get("metadata"))
|
|
163
|
+
rows = results.get("data") or []
|
|
164
|
+
if rows and self._process_rows(rows, max_allowed):
|
|
165
|
+
break
|
|
161
166
|
if max_allowed is not None and self.row_count >= max_allowed:
|
|
162
167
|
break
|
|
163
|
-
|
|
164
|
-
if
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
if "data" in results and results["data"] != None and results["data"] != []:
|
|
168
|
-
self._update_extracted_sync_cursor(results.get("metadata"))
|
|
169
|
-
self._process_rows(results["data"], max_allowed)
|
|
170
|
-
elif results["data"] != []:
|
|
171
|
-
self.internal_log(self.log_templates.INTERNAL_GET_DATA_FETCHED.format(0))
|
|
172
|
-
else:
|
|
173
|
-
self._update_extracted_sync_cursor(results.get("metadata"))
|
|
168
|
+
next_page = results.get("next_page")
|
|
169
|
+
if next_page is None:
|
|
170
|
+
break
|
|
174
171
|
|
|
175
172
|
if self._should_track_sync_cursor():
|
|
176
173
|
self._save_pending_sync_cursor()
|
|
@@ -184,6 +181,8 @@ class PipelineConductor:
|
|
|
184
181
|
Not wired into the default SOURCE path yet — call explicitly when testing.
|
|
185
182
|
Retains extract/normalize hints via a sibling .meta.json object.
|
|
186
183
|
"""
|
|
184
|
+
if self._get_data_filters():
|
|
185
|
+
raise errors.DataError('Row filters require the standard row transfer path.')
|
|
187
186
|
if not hasattr(self.connector, "get_data_stream"):
|
|
188
187
|
raise errors.DataError(
|
|
189
188
|
"Connector does not implement get_data_stream; cannot use stream transfer."
|
|
@@ -233,6 +232,7 @@ class PipelineConductor:
|
|
|
233
232
|
if not loaded:
|
|
234
233
|
raise errors.LoadDataError("Loading data failed.")
|
|
235
234
|
else:
|
|
235
|
+
keys.sort(key=self._transfer_batch_sort_key)
|
|
236
236
|
for index, key in enumerate(keys):
|
|
237
237
|
file_object = None
|
|
238
238
|
if PipelineEnvironment.platform == "aws":
|
|
@@ -257,6 +257,11 @@ class PipelineConductor:
|
|
|
257
257
|
self.log(self.log_templates.LOAD_DATA_FINISHED.format(self.row_count, destination_object_id))
|
|
258
258
|
self._persist_sync_cursor_after_load()
|
|
259
259
|
|
|
260
|
+
@staticmethod
|
|
261
|
+
def _transfer_batch_sort_key(key):
|
|
262
|
+
match = re.search(r"-b(\d+)\.json$", key)
|
|
263
|
+
return (0, int(match.group(1))) if match else (1, key)
|
|
264
|
+
|
|
260
265
|
def _create_connector(self, connector_cls, credentials):
|
|
261
266
|
init_params = inspect.signature(connector_cls.__init__).parameters
|
|
262
267
|
if "pipeline_context" in init_params:
|
|
@@ -428,6 +433,14 @@ class PipelineConductor:
|
|
|
428
433
|
self.api.create_pipeline_mapping(self.pipeline_id, self.pipeline_details.pipeline_mapping_json)
|
|
429
434
|
|
|
430
435
|
def _process_rows(self, rows, max_allowed=None):
|
|
436
|
+
rows = apply_data_filters(rows, self._get_data_filters())
|
|
437
|
+
if not rows:
|
|
438
|
+
return False
|
|
439
|
+
selected_ids = {str(field) for field in self._get_field_ids()}
|
|
440
|
+
if selected_ids:
|
|
441
|
+
filter_only = {f['column_name'] for f in self._get_data_filters()} - selected_ids
|
|
442
|
+
if filter_only:
|
|
443
|
+
rows = [{key: value for key, value in row.items() if str(key) not in filter_only} for row in rows]
|
|
431
444
|
# Check if we have a limit and need to truncate rows
|
|
432
445
|
limit_reached = False
|
|
433
446
|
if max_allowed is not None:
|
|
@@ -590,7 +603,7 @@ class PipelineConductor:
|
|
|
590
603
|
def _get_batch_row_count(self):
|
|
591
604
|
results = self.connector.get_data(
|
|
592
605
|
self.pipeline_details.source_object_id,
|
|
593
|
-
self.
|
|
606
|
+
self._get_source_field_ids(),
|
|
594
607
|
n_rows=32,
|
|
595
608
|
filters=self._get_filters(),
|
|
596
609
|
options=self.pipeline_details.options
|
|
@@ -639,6 +652,9 @@ class PipelineConductor:
|
|
|
639
652
|
return batch_size
|
|
640
653
|
|
|
641
654
|
def _get_filters(self):
|
|
655
|
+
# New filters are enforced centrally; do not also apply a stale legacy range.
|
|
656
|
+
if 'sync_filters' in (self.pipeline_details.options or {}):
|
|
657
|
+
return None
|
|
642
658
|
if self.mode == "prod":
|
|
643
659
|
return {
|
|
644
660
|
"filtered_column_nm": self.pipeline_details.filtered_column_nm,
|
|
@@ -651,6 +667,22 @@ class PipelineConductor:
|
|
|
651
667
|
else:
|
|
652
668
|
return self.filters
|
|
653
669
|
|
|
670
|
+
def _get_data_filters(self):
|
|
671
|
+
if not hasattr(self, '_resolved_data_filters'):
|
|
672
|
+
self._resolved_data_filters = compile_sync_filters(
|
|
673
|
+
(self.pipeline_details.options or {}).get('sync_filters')
|
|
674
|
+
)
|
|
675
|
+
return self._resolved_data_filters
|
|
676
|
+
|
|
677
|
+
def _get_source_field_ids(self):
|
|
678
|
+
fields = self._get_field_ids()
|
|
679
|
+
if not fields: # An empty mapping requests every source column.
|
|
680
|
+
return fields
|
|
681
|
+
ids = {str(field) for field in fields}
|
|
682
|
+
return fields + list(dict.fromkeys(
|
|
683
|
+
f['column_name'] for f in self._get_data_filters() if f['column_name'] not in ids
|
|
684
|
+
))
|
|
685
|
+
|
|
654
686
|
def _get_field_ids(self):
|
|
655
687
|
if not self.pipeline_details.pipeline_mapping_json:
|
|
656
688
|
return []
|
|
@@ -745,4 +777,4 @@ class PipelineConductor:
|
|
|
745
777
|
i += 1
|
|
746
778
|
seen[name] = True
|
|
747
779
|
mapping.append({"column": name, "mapped": fid})
|
|
748
|
-
return mapping
|
|
780
|
+
return mapping
|
|
@@ -103,6 +103,21 @@ class DataConnectorAPI:
|
|
|
103
103
|
f"credentials/{pipeline_id}?source={'true' if source_flg else 'false'}"
|
|
104
104
|
)
|
|
105
105
|
|
|
106
|
+
envelope_fields = (
|
|
107
|
+
'encrypted_data_key_txt', 'encryption_iv_txt', 'encryption_auth_tag_txt'
|
|
108
|
+
)
|
|
109
|
+
if not all(field in response for field in envelope_fields):
|
|
110
|
+
# Older APIs omit the envelope. Fetch ciphertext and metadata together
|
|
111
|
+
# from a fresh pipeline snapshot; never mix separate credential reads.
|
|
112
|
+
details = self.get(str(pipeline_id))
|
|
113
|
+
side = 'source' if source_flg else 'destination'
|
|
114
|
+
information = details.get(f'{side}_credential_information') or {}
|
|
115
|
+
response = {
|
|
116
|
+
'credential': details[f'{side}_encryption_credential_txt'],
|
|
117
|
+
'organization': details['customer_metadata_uuid'],
|
|
118
|
+
**{field: information.get(field) for field in envelope_fields},
|
|
119
|
+
}
|
|
120
|
+
|
|
106
121
|
return response
|
|
107
122
|
|
|
108
123
|
def create_pipeline_mapping(self, pipeline_id, pipeline_mapping_json):
|
|
@@ -220,4 +235,4 @@ class DataConnectorAPI:
|
|
|
220
235
|
def _get_retry_delay(self, attempt):
|
|
221
236
|
base_delay = min(2 ** attempt, 10)
|
|
222
237
|
jitter = random.uniform(0, 0.5)
|
|
223
|
-
return base_delay + jitter
|
|
238
|
+
return base_delay + jitter
|
|
@@ -52,19 +52,16 @@ class AwsService:
|
|
|
52
52
|
self._ecs_cluster_name = APP_ENV if APP_ENV != "local" else "development"
|
|
53
53
|
|
|
54
54
|
def get_keys(self, pipeline_run_history_id):
|
|
55
|
-
|
|
56
|
-
response = s3_client.list_objects(Bucket=self.s3_bucket, Prefix=f"transfers/e{pipeline_run_history_id}")
|
|
57
|
-
if 'Contents' in response:
|
|
58
|
-
for key in response['Contents']:
|
|
59
|
-
keys.append(key['Key'])
|
|
60
|
-
return keys
|
|
55
|
+
return self._get_keys(f"transfers/e{pipeline_run_history_id}-b")
|
|
61
56
|
|
|
62
57
|
def get_dev_keys(self, prefix):
|
|
58
|
+
return self._get_keys(f"devTransfers/e{prefix}-b")
|
|
59
|
+
|
|
60
|
+
def _get_keys(self, prefix):
|
|
63
61
|
keys = []
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
for
|
|
67
|
-
keys.append(key['Key'])
|
|
62
|
+
paginator = s3_client.get_paginator('list_objects_v2')
|
|
63
|
+
for page in paginator.paginate(Bucket=self.s3_bucket, Prefix=prefix):
|
|
64
|
+
keys.extend(obj['Key'] for obj in page.get('Contents', []))
|
|
68
65
|
return keys
|
|
69
66
|
|
|
70
67
|
def download_object(self, key_name):
|
|
@@ -328,4 +325,4 @@ class EncryptionService:
|
|
|
328
325
|
encryption_auth_tag_txt=encryption_auth_tag_txt
|
|
329
326
|
)
|
|
330
327
|
|
|
331
|
-
return json.loads(data)
|
|
328
|
+
return json.loads(data)
|
|
@@ -1,242 +0,0 @@
|
|
|
1
|
-
from dc_sdk.errors import Error
|
|
2
|
-
from dc_sdk.json_safe import to_json_safe
|
|
3
|
-
from dc_sdk.src.mapping import Mapping
|
|
4
|
-
from dc_sdk.src.connection_status import STATUS_ERROR
|
|
5
|
-
import traceback
|
|
6
|
-
from importlib.metadata import version
|
|
7
|
-
|
|
8
|
-
def get_action_name(action: int) -> str:
|
|
9
|
-
return {
|
|
10
|
-
0: "authenticating",
|
|
11
|
-
1: "retrieving fields",
|
|
12
|
-
2: "retrieving 5 row preview",
|
|
13
|
-
3: "testing connection",
|
|
14
|
-
4: "starting the connector",
|
|
15
|
-
5: "retrieving metadata",
|
|
16
|
-
6: "retrieving objects",
|
|
17
|
-
}[action]
|
|
18
|
-
|
|
19
|
-
def apply_data_filters(results, data_filters):
|
|
20
|
-
if not data_filters:
|
|
21
|
-
return results
|
|
22
|
-
|
|
23
|
-
# Connectors may return either:
|
|
24
|
-
# 1) a bare list of rows, or
|
|
25
|
-
# 2) an envelope: {"data": [...], "next_page": ...}
|
|
26
|
-
is_envelope = isinstance(results, dict) and isinstance(results.get("data"), list)
|
|
27
|
-
rows = results.get("data") if is_envelope else results
|
|
28
|
-
if not isinstance(rows, list):
|
|
29
|
-
return results
|
|
30
|
-
|
|
31
|
-
# 🔹 UI → backend operator mapping
|
|
32
|
-
OPERATOR_MAP = {
|
|
33
|
-
# text
|
|
34
|
-
"text_contains": "CONTAINS",
|
|
35
|
-
"text_not_contains": "NOT_CONTAINS",
|
|
36
|
-
"text_starts_with": "STARTS_WITH",
|
|
37
|
-
"text_ends_with": "ENDS_WITH",
|
|
38
|
-
|
|
39
|
-
# equality
|
|
40
|
-
"is_equal": "EQ",
|
|
41
|
-
"is_not_equal": "NEQ",
|
|
42
|
-
|
|
43
|
-
# empty
|
|
44
|
-
"is_empty": "IS_EMPTY",
|
|
45
|
-
"is_not_empty": "IS_NOT_EMPTY",
|
|
46
|
-
|
|
47
|
-
# numeric
|
|
48
|
-
"gt": "GT",
|
|
49
|
-
"gte": "GTE",
|
|
50
|
-
"lt": "LT",
|
|
51
|
-
"lte": "LTE",
|
|
52
|
-
|
|
53
|
-
# range
|
|
54
|
-
"between": "BETWEEN",
|
|
55
|
-
"not_between": "NOT_BETWEEN",
|
|
56
|
-
|
|
57
|
-
# ignore
|
|
58
|
-
"none": None,
|
|
59
|
-
}
|
|
60
|
-
|
|
61
|
-
def try_parse_number(val):
|
|
62
|
-
try:
|
|
63
|
-
return float(val)
|
|
64
|
-
except:
|
|
65
|
-
return val
|
|
66
|
-
|
|
67
|
-
def normalize(row_val, v1, v2=None):
|
|
68
|
-
row_num = try_parse_number(row_val)
|
|
69
|
-
v1_num = try_parse_number(v1)
|
|
70
|
-
v2_num = try_parse_number(v2) if v2 is not None else None
|
|
71
|
-
|
|
72
|
-
# If both are numbers → compare as numbers
|
|
73
|
-
if isinstance(row_num, float) and isinstance(v1_num, float):
|
|
74
|
-
return row_num, v1_num, v2_num
|
|
75
|
-
|
|
76
|
-
# Otherwise → compare as strings
|
|
77
|
-
return str(row_val), str(v1), str(v2) if v2 is not None else None
|
|
78
|
-
|
|
79
|
-
def match(row, f):
|
|
80
|
-
ui_operator = f.get("operator_cd")
|
|
81
|
-
operator = OPERATOR_MAP.get(ui_operator)
|
|
82
|
-
|
|
83
|
-
if operator is None:
|
|
84
|
-
return True # skip "none"
|
|
85
|
-
|
|
86
|
-
col = f.get("column_name")
|
|
87
|
-
val1 = f.get("value_1_txt")
|
|
88
|
-
val2 = f.get("value_2_txt")
|
|
89
|
-
|
|
90
|
-
# Some connectors can emit non-dict rows; keep filtering resilient.
|
|
91
|
-
if not isinstance(row, dict):
|
|
92
|
-
return False
|
|
93
|
-
row_val = row.get(col)
|
|
94
|
-
|
|
95
|
-
# ---- EMPTY HANDLING ----
|
|
96
|
-
if operator == "IS_EMPTY":
|
|
97
|
-
return row_val is None or str(row_val).strip() == ""
|
|
98
|
-
|
|
99
|
-
if operator == "IS_NOT_EMPTY":
|
|
100
|
-
return row_val is not None and str(row_val).strip() != ""
|
|
101
|
-
|
|
102
|
-
if row_val is None:
|
|
103
|
-
return False
|
|
104
|
-
|
|
105
|
-
# Normalize values
|
|
106
|
-
row_val, v1, v2 = normalize(row_val, val1, val2)
|
|
107
|
-
|
|
108
|
-
# ---- OPERATORS ----
|
|
109
|
-
|
|
110
|
-
if operator == "EQ":
|
|
111
|
-
return row_val == v1
|
|
112
|
-
|
|
113
|
-
elif operator == "NEQ":
|
|
114
|
-
return row_val != v1
|
|
115
|
-
|
|
116
|
-
elif operator == "GT":
|
|
117
|
-
return row_val > v1
|
|
118
|
-
|
|
119
|
-
elif operator == "GTE":
|
|
120
|
-
return row_val >= v1
|
|
121
|
-
|
|
122
|
-
elif operator == "LT":
|
|
123
|
-
return row_val < v1
|
|
124
|
-
|
|
125
|
-
elif operator == "LTE":
|
|
126
|
-
return row_val <= v1
|
|
127
|
-
|
|
128
|
-
elif operator == "BETWEEN":
|
|
129
|
-
return v1 <= row_val <= v2
|
|
130
|
-
|
|
131
|
-
elif operator == "NOT_BETWEEN":
|
|
132
|
-
return not (v1 <= row_val <= v2)
|
|
133
|
-
|
|
134
|
-
elif operator == "CONTAINS":
|
|
135
|
-
return str(v1).lower() in str(row_val).lower()
|
|
136
|
-
|
|
137
|
-
elif operator == "NOT_CONTAINS":
|
|
138
|
-
return str(v1).lower() not in str(row_val).lower()
|
|
139
|
-
|
|
140
|
-
elif operator == "STARTS_WITH":
|
|
141
|
-
return str(row_val).lower().startswith(str(v1).lower())
|
|
142
|
-
|
|
143
|
-
elif operator == "ENDS_WITH":
|
|
144
|
-
return str(row_val).lower().endswith(str(v1).lower())
|
|
145
|
-
|
|
146
|
-
return True
|
|
147
|
-
|
|
148
|
-
# 🔹 Apply filters (AND logic)
|
|
149
|
-
filtered_rows = [row for row in rows if all(match(row, f) for f in data_filters)]
|
|
150
|
-
if is_envelope:
|
|
151
|
-
return {**results, "data": filtered_rows}
|
|
152
|
-
return filtered_rows
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
def handler(event, context):
|
|
156
|
-
print("version: ", version("dc-python-sdk"))
|
|
157
|
-
"""Lambda Handler"""
|
|
158
|
-
action_name = None
|
|
159
|
-
internal_error = False
|
|
160
|
-
message = None
|
|
161
|
-
results = None
|
|
162
|
-
error = None
|
|
163
|
-
mapping = None
|
|
164
|
-
action = int(event['action']) if 'action' in event else 4
|
|
165
|
-
get_objects = event.get('get_objects', True)
|
|
166
|
-
skip_authenticate = event.get('skip_authenticate', False)
|
|
167
|
-
force_authenticate = event.get('force_authenticate', False)
|
|
168
|
-
include_metadata = event.get('include_metadata', True)
|
|
169
|
-
credentials_dict = event['credentials'] if 'credentials' in event else None
|
|
170
|
-
object_id = event['object_id'] if 'object_id' in event else None
|
|
171
|
-
field_ids = event['mapping'] if 'mapping' in event else None
|
|
172
|
-
options = event['options'] if 'options' in event else dict()
|
|
173
|
-
next_page = event['next_page'] if 'next_page' in event else None
|
|
174
|
-
n_rows = event['n_rows'] if 'n_rows' in event else None
|
|
175
|
-
filters = event['filters'] if 'filters' in event else None
|
|
176
|
-
data_filters = event['data_filters'] if 'data_filters' in event else None
|
|
177
|
-
|
|
178
|
-
try:
|
|
179
|
-
action_name = get_action_name(action)
|
|
180
|
-
mapping = Mapping(credentials_dict)
|
|
181
|
-
|
|
182
|
-
if action == 0:
|
|
183
|
-
if get_objects:
|
|
184
|
-
results, message = mapping.connect_get_objects(
|
|
185
|
-
skip_authenticate=skip_authenticate,
|
|
186
|
-
force_authenticate=force_authenticate,
|
|
187
|
-
)
|
|
188
|
-
else:
|
|
189
|
-
results, message = mapping.authenticate(
|
|
190
|
-
force_authenticate=force_authenticate or not skip_authenticate,
|
|
191
|
-
)
|
|
192
|
-
elif action == 1:
|
|
193
|
-
results, message = mapping.get_fields(object_id, options)
|
|
194
|
-
elif action == 2:
|
|
195
|
-
results, message = mapping.get_five_row_preview(
|
|
196
|
-
object_id, field_ids, options, n_rows, filters, next_page)
|
|
197
|
-
results = apply_data_filters(results, data_filters)
|
|
198
|
-
elif action == 3:
|
|
199
|
-
results, message = mapping.test_connection()
|
|
200
|
-
elif action == 5:
|
|
201
|
-
results, message = mapping.get_metadata_only(
|
|
202
|
-
skip_authenticate=skip_authenticate,
|
|
203
|
-
force_authenticate=force_authenticate,
|
|
204
|
-
)
|
|
205
|
-
elif action == 6:
|
|
206
|
-
results, message = mapping.get_objects_only(
|
|
207
|
-
skip_authenticate=skip_authenticate,
|
|
208
|
-
force_authenticate=force_authenticate,
|
|
209
|
-
include_metadata=include_metadata,
|
|
210
|
-
)
|
|
211
|
-
else:
|
|
212
|
-
raise Error("Invalid action", "InvalidActionError")
|
|
213
|
-
except Error as e:
|
|
214
|
-
message = e.message or str(e)
|
|
215
|
-
internal_error = e.internal
|
|
216
|
-
error_trace = traceback.format_exc()
|
|
217
|
-
error = error_trace
|
|
218
|
-
if mapping:
|
|
219
|
-
mapping._set_connection_status(STATUS_ERROR)
|
|
220
|
-
|
|
221
|
-
except Exception as e:
|
|
222
|
-
error_trace = traceback.format_exc()
|
|
223
|
-
error = error_trace
|
|
224
|
-
internal_error = True
|
|
225
|
-
if mapping:
|
|
226
|
-
mapping._set_connection_status(STATUS_ERROR)
|
|
227
|
-
|
|
228
|
-
if error:
|
|
229
|
-
print(error)
|
|
230
|
-
|
|
231
|
-
last_status_dsc = None
|
|
232
|
-
if mapping and mapping.connector.credentials:
|
|
233
|
-
last_status_dsc = mapping.connector.credentials.get("last_status_dsc")
|
|
234
|
-
|
|
235
|
-
return {
|
|
236
|
-
'message': message if not internal_error else f"Something went wrong with {action_name}",
|
|
237
|
-
'results': to_json_safe(results),
|
|
238
|
-
'credentials': mapping.connector.credentials if mapping else None,
|
|
239
|
-
'last_status_dsc': last_status_dsc,
|
|
240
|
-
'error': error,
|
|
241
|
-
'internal_error': internal_error
|
|
242
|
-
}
|
|
File without changes
|
{dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_python_sdk.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|