dc-python-sdk 1.5.54__tar.gz → 1.6.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. {dc_python_sdk-1.5.54/src/dc_python_sdk.egg-info → dc_python_sdk-1.6.1}/PKG-INFO +37 -2
  2. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/README.md +35 -0
  3. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/pyproject.toml +2 -2
  4. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/setup.cfg +2 -2
  5. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1/src/dc_python_sdk.egg-info}/PKG-INFO +37 -2
  6. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_python_sdk.egg-info/SOURCES.txt +1 -0
  7. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_python_sdk.egg-info/requires.txt +1 -1
  8. dc_python_sdk-1.6.1/src/dc_sdk/handler.py +122 -0
  9. dc_python_sdk-1.6.1/src/dc_sdk/row_filters.py +246 -0
  10. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/mapping.py +32 -12
  11. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/pipeline.py +63 -31
  12. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/services/api.py +16 -1
  13. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/services/aws.py +8 -11
  14. dc_python_sdk-1.5.54/src/dc_sdk/handler.py +0 -242
  15. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/LICENSE +0 -0
  16. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_python_sdk.egg-info/dependency_links.txt +0 -0
  17. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_python_sdk.egg-info/entry_points.txt +0 -0
  18. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_python_sdk.egg-info/top_level.txt +0 -0
  19. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/__init__.py +0 -0
  20. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/app.py +0 -0
  21. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/cli.py +0 -0
  22. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/data_stream.py +0 -0
  23. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/errors.py +0 -0
  24. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/file_utils.py +0 -0
  25. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/json_safe.py +0 -0
  26. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/__init__.py +0 -0
  27. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/ai.py +0 -0
  28. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/ai_http.py +0 -0
  29. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/connection_status.py +0 -0
  30. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/destination_object_template.py +0 -0
  31. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/models/__init__.py +0 -0
  32. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/models/enums.py +0 -0
  33. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/models/errors.py +0 -0
  34. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/models/log_templates.py +0 -0
  35. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/models/pipeline_details.py +0 -0
  36. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/server.py +0 -0
  37. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/services/__init__.py +0 -0
  38. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/services/environment.py +0 -0
  39. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/services/loader.py +0 -0
  40. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/services/logger.py +0 -0
  41. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/src/services/session.py +0 -0
  42. {dc_python_sdk-1.5.54 → dc_python_sdk-1.6.1}/src/dc_sdk/types.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dc-python-sdk
3
- Version: 1.5.54
3
+ Version: 1.6.1
4
4
  Summary: Data Connector Python SDK
5
5
  Home-page: https://github.com/data-connector/dc-python-sdk
6
6
  Author: DataConnector
@@ -14,7 +14,7 @@ Description-Content-Type: text/markdown
14
14
  License-File: LICENSE
15
15
  Requires-Dist: fastapi
16
16
  Requires-Dist: uvicorn
17
- Requires-Dist: awslambdaric
17
+ Requires-Dist: awslambdaric==4.0.4
18
18
  Requires-Dist: requests
19
19
  Requires-Dist: boto3>=1.40.0
20
20
  Requires-Dist: openai
@@ -94,6 +94,41 @@ def get_available_objects():
94
94
  return objects
95
95
  ```
96
96
 
97
+ ### Selective discovery in the Lambda handler
98
+
99
+ Authenticate with action `3`, then request only the metadata needed by the current
100
+ screen with action `5`:
101
+
102
+ ```json
103
+ {"action": 5, "sections": ["accounts"], "credentials": {}}
104
+ ```
105
+
106
+ Connectors opt in by defining `get_metadata(self, sections=None)`. Fabric supports
107
+ `accounts`, `schemas`, and `folders`; callers can request more than one section.
108
+ Requested sections retain their usual keys in the metadata dictionary, and static
109
+ capability fields may also be returned. Unrequested dynamic keys are omitted,
110
+ not returned as empty lists. Callers should merge partial metadata into existing
111
+ metadata for the same account, and clear cached metadata when switching accounts.
112
+
113
+ Action `6` accepts the same optional parameter for `get_objects(self, sections=None)`;
114
+ Fabric supports `tables` and `files`. Use `include_metadata: false` to avoid an
115
+ additional full metadata request when fetching only objects. Object results remain
116
+ a list in the existing response envelope.
117
+
118
+ Omitting `sections` (or using null) retains full discovery. An empty list requests
119
+ no dynamic sections on an opted-in connector. The SDK validates a list of non-empty
120
+ strings and passes it only to methods with an explicit `sections` keyword; legacy
121
+ connectors still perform full discovery with no arguments. Each opted-in connector
122
+ validates its supported section names. Internal connector TypeErrors are propagated,
123
+ never retried as a legacy call. The SDK does not filter a legacy response or claim
124
+ that a legacy connector avoided fetching unrequested data.
125
+
126
+ The SDK returns `error_phase: "metadata"` for discovery errors and keeps successful
127
+ authentication status. `AuthenticationError` still invalidates authentication.
128
+ API and UI consumers must use this distinction to show a discovery retry instead
129
+ of requiring reconnection. This contract applies to the Lambda handler; the local
130
+ HTTP server below uses a separate method-based request format.
131
+
97
132
  ### Running the local HTTP server
98
133
 
99
134
  Install the SDK and start the HTTP server that wraps your connector:
@@ -61,6 +61,41 @@ def get_available_objects():
61
61
  return objects
62
62
  ```
63
63
 
64
+ ### Selective discovery in the Lambda handler
65
+
66
+ Authenticate with action `3`, then request only the metadata needed by the current
67
+ screen with action `5`:
68
+
69
+ ```json
70
+ {"action": 5, "sections": ["accounts"], "credentials": {}}
71
+ ```
72
+
73
+ Connectors opt in by defining `get_metadata(self, sections=None)`. Fabric supports
74
+ `accounts`, `schemas`, and `folders`; callers can request more than one section.
75
+ Requested sections retain their usual keys in the metadata dictionary, and static
76
+ capability fields may also be returned. Unrequested dynamic keys are omitted,
77
+ not returned as empty lists. Callers should merge partial metadata into existing
78
+ metadata for the same account, and clear cached metadata when switching accounts.
79
+
80
+ Action `6` accepts the same optional parameter for `get_objects(self, sections=None)`;
81
+ Fabric supports `tables` and `files`. Use `include_metadata: false` to avoid an
82
+ additional full metadata request when fetching only objects. Object results remain
83
+ a list in the existing response envelope.
84
+
85
+ Omitting `sections` (or using null) retains full discovery. An empty list requests
86
+ no dynamic sections on an opted-in connector. The SDK validates a list of non-empty
87
+ strings and passes it only to methods with an explicit `sections` keyword; legacy
88
+ connectors still perform full discovery with no arguments. Each opted-in connector
89
+ validates its supported section names. Internal connector TypeErrors are propagated,
90
+ never retried as a legacy call. The SDK does not filter a legacy response or claim
91
+ that a legacy connector avoided fetching unrequested data.
92
+
93
+ The SDK returns `error_phase: "metadata"` for discovery errors and keeps successful
94
+ authentication status. `AuthenticationError` still invalidates authentication.
95
+ API and UI consumers must use this distinction to show a discovery retry instead
96
+ of requiring reconnection. This contract applies to the Lambda handler; the local
97
+ HTTP server below uses a separate method-based request format.
98
+
64
99
  ### Running the local HTTP server
65
100
 
66
101
  Install the SDK and start the HTTP server that wraps your connector:
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "dc-python-sdk"
7
- version = "1.5.54"
7
+ version = "1.6.1"
8
8
  description = "Data Connector Python SDK"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.6"
@@ -19,7 +19,7 @@ classifiers = [
19
19
  dependencies = [
20
20
  "fastapi",
21
21
  "uvicorn",
22
- "awslambdaric",
22
+ "awslambdaric==4.0.4",
23
23
  "requests",
24
24
  "boto3>=1.40.0",
25
25
  "openai",
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = dc-python-sdk
3
- version = 1.5.54
3
+ version = 1.6.1
4
4
  author = DataConnector
5
5
  author_email = josh@dataconnector.com
6
6
  description = A small example package
@@ -22,7 +22,7 @@ python_requires = >=3.6
22
22
  install_requires =
23
23
  fastapi
24
24
  uvicorn
25
- awslambdaric
25
+ awslambdaric==4.0.4
26
26
  pycryptodome
27
27
  requests
28
28
  boto3>=1.40.0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dc-python-sdk
3
- Version: 1.5.54
3
+ Version: 1.6.1
4
4
  Summary: Data Connector Python SDK
5
5
  Home-page: https://github.com/data-connector/dc-python-sdk
6
6
  Author: DataConnector
@@ -14,7 +14,7 @@ Description-Content-Type: text/markdown
14
14
  License-File: LICENSE
15
15
  Requires-Dist: fastapi
16
16
  Requires-Dist: uvicorn
17
- Requires-Dist: awslambdaric
17
+ Requires-Dist: awslambdaric==4.0.4
18
18
  Requires-Dist: requests
19
19
  Requires-Dist: boto3>=1.40.0
20
20
  Requires-Dist: openai
@@ -94,6 +94,41 @@ def get_available_objects():
94
94
  return objects
95
95
  ```
96
96
 
97
+ ### Selective discovery in the Lambda handler
98
+
99
+ Authenticate with action `3`, then request only the metadata needed by the current
100
+ screen with action `5`:
101
+
102
+ ```json
103
+ {"action": 5, "sections": ["accounts"], "credentials": {}}
104
+ ```
105
+
106
+ Connectors opt in by defining `get_metadata(self, sections=None)`. Fabric supports
107
+ `accounts`, `schemas`, and `folders`; callers can request more than one section.
108
+ Requested sections retain their usual keys in the metadata dictionary, and static
109
+ capability fields may also be returned. Unrequested dynamic keys are omitted,
110
+ not returned as empty lists. Callers should merge partial metadata into existing
111
+ metadata for the same account, and clear cached metadata when switching accounts.
112
+
113
+ Action `6` accepts the same optional parameter for `get_objects(self, sections=None)`;
114
+ Fabric supports `tables` and `files`. Use `include_metadata: false` to avoid an
115
+ additional full metadata request when fetching only objects. Object results remain
116
+ a list in the existing response envelope.
117
+
118
+ Omitting `sections` (or using null) retains full discovery. An empty list requests
119
+ no dynamic sections on an opted-in connector. The SDK validates a list of non-empty
120
+ strings and passes it only to methods with an explicit `sections` keyword; legacy
121
+ connectors still perform full discovery with no arguments. Each opted-in connector
122
+ validates its supported section names. Internal connector TypeErrors are propagated,
123
+ never retried as a legacy call. The SDK does not filter a legacy response or claim
124
+ that a legacy connector avoided fetching unrequested data.
125
+
126
+ The SDK returns `error_phase: "metadata"` for discovery errors and keeps successful
127
+ authentication status. `AuthenticationError` still invalidates authentication.
128
+ API and UI consumers must use this distinction to show a discovery retry instead
129
+ of requiring reconnection. This contract applies to the Lambda handler; the local
130
+ HTTP server below uses a separate method-based request format.
131
+
97
132
  ### Running the local HTTP server
98
133
 
99
134
  Install the SDK and start the HTTP server that wraps your connector:
@@ -16,6 +16,7 @@ src/dc_sdk/errors.py
16
16
  src/dc_sdk/file_utils.py
17
17
  src/dc_sdk/handler.py
18
18
  src/dc_sdk/json_safe.py
19
+ src/dc_sdk/row_filters.py
19
20
  src/dc_sdk/types.py
20
21
  src/dc_sdk/src/__init__.py
21
22
  src/dc_sdk/src/ai.py
@@ -1,6 +1,6 @@
1
1
  fastapi
2
2
  uvicorn
3
- awslambdaric
3
+ awslambdaric==4.0.4
4
4
  requests
5
5
  boto3>=1.40.0
6
6
  openai
@@ -0,0 +1,122 @@
1
+ from dc_sdk.row_filters import apply_data_filters, compile_sync_filters
2
+ from dc_sdk.errors import Error, AuthenticationError
3
+ from dc_sdk.json_safe import to_json_safe
4
+ from dc_sdk.src.mapping import Mapping
5
+ from dc_sdk.src.connection_status import STATUS_ERROR
6
+ import traceback
7
+ from importlib.metadata import version
8
+
9
+ def get_action_name(action: int) -> str:
10
+ return {
11
+ 0: "authenticating",
12
+ 1: "retrieving fields",
13
+ 2: "retrieving 5 row preview",
14
+ 3: "testing connection",
15
+ 4: "starting the connector",
16
+ 5: "retrieving metadata",
17
+ 6: "retrieving objects",
18
+ }[action]
19
+
20
+ def handler(event, context):
21
+ print("version: ", version("dc-python-sdk"))
22
+ """Lambda Handler"""
23
+ action_name = None
24
+ internal_error = False
25
+ message = None
26
+ results = None
27
+ error = None
28
+ error_phase = None
29
+ mapping = None
30
+ action = int(event['action']) if 'action' in event else 4
31
+ get_objects = event.get('get_objects', True)
32
+ skip_authenticate = event.get('skip_authenticate', False)
33
+ force_authenticate = event.get('force_authenticate', False)
34
+ include_metadata = event.get('include_metadata', True)
35
+ credentials_dict = event['credentials'] if 'credentials' in event else None
36
+ object_id = event['object_id'] if 'object_id' in event else None
37
+ field_ids = event['mapping'] if 'mapping' in event else None
38
+ options = event['options'] if 'options' in event else dict()
39
+ next_page = event['next_page'] if 'next_page' in event else None
40
+ n_rows = event['n_rows'] if 'n_rows' in event else None
41
+ filters = event['filters'] if 'filters' in event else None
42
+ data_filters = event['data_filters'] if 'data_filters' in event else None
43
+
44
+ try:
45
+ action_name = get_action_name(action)
46
+ mapping = Mapping(credentials_dict)
47
+
48
+ if action == 0:
49
+ if get_objects:
50
+ results, message = mapping.connect_get_objects(
51
+ skip_authenticate=skip_authenticate,
52
+ force_authenticate=force_authenticate,
53
+ include_metadata=include_metadata,
54
+ )
55
+ else:
56
+ results, message = mapping.authenticate(
57
+ force_authenticate=force_authenticate or not skip_authenticate,
58
+ include_metadata=include_metadata,
59
+ )
60
+ elif action == 1:
61
+ results, message = mapping.get_fields(object_id, options)
62
+ elif action == 2:
63
+ results, message = mapping.get_five_row_preview(
64
+ object_id, field_ids, options, n_rows, filters, next_page)
65
+ # During rollout callers send both formats. Apply the canonical one
66
+ # once, so legacy date comparisons cannot remove valid whole-day matches.
67
+ preview_filters = (
68
+ compile_sync_filters(options['sync_filters'])
69
+ if isinstance(options, dict) and 'sync_filters' in options
70
+ else data_filters
71
+ )
72
+ results = apply_data_filters(results, preview_filters)
73
+ elif action == 3:
74
+ results, message = mapping.test_connection()
75
+ elif action == 5:
76
+ results, message = mapping.get_metadata_only(
77
+ skip_authenticate=skip_authenticate,
78
+ force_authenticate=force_authenticate,
79
+ sections=event.get('sections'),
80
+ )
81
+ elif action == 6:
82
+ results, message = mapping.get_objects_only(
83
+ skip_authenticate=skip_authenticate,
84
+ force_authenticate=force_authenticate,
85
+ include_metadata=include_metadata,
86
+ sections=event.get('sections'),
87
+ )
88
+ else:
89
+ raise Error("Invalid action", "InvalidActionError")
90
+ except Error as e:
91
+ error_phase = "authentication" if isinstance(e, AuthenticationError) else getattr(mapping, "request_phase", None)
92
+ message = e.message or str(e)
93
+ internal_error = e.internal
94
+ error_trace = traceback.format_exc()
95
+ error = error_trace
96
+ if mapping and (isinstance(e, AuthenticationError) or getattr(mapping, "request_phase", None) != "metadata"):
97
+ mapping._set_connection_status(STATUS_ERROR)
98
+
99
+ except Exception as e:
100
+ error_phase = getattr(mapping, "request_phase", None)
101
+ error_trace = traceback.format_exc()
102
+ error = error_trace
103
+ internal_error = True
104
+ if mapping and getattr(mapping, "request_phase", None) != "metadata":
105
+ mapping._set_connection_status(STATUS_ERROR)
106
+
107
+ if error:
108
+ print(error)
109
+
110
+ last_status_dsc = None
111
+ if mapping and mapping.connector.credentials:
112
+ last_status_dsc = mapping.connector.credentials.get("last_status_dsc")
113
+
114
+ return {
115
+ 'message': message if not internal_error else f"Something went wrong with {action_name}",
116
+ 'results': to_json_safe(results),
117
+ 'credentials': mapping.connector.credentials if mapping else None,
118
+ 'last_status_dsc': last_status_dsc,
119
+ 'error': error,
120
+ 'internal_error': internal_error,
121
+ 'error_phase': error_phase,
122
+ }
@@ -0,0 +1,246 @@
1
+ from datetime import date, datetime, timedelta, timezone
2
+ from zoneinfo import ZoneInfo
3
+
4
+
5
+ def _calendar_date(value, zone):
6
+ if isinstance(value, datetime):
7
+ return value.astimezone(zone).date() if value.tzinfo else value.date()
8
+ if isinstance(value, date):
9
+ return value
10
+ text = str(value).strip()
11
+ # Date-only values represent calendar dates, not UTC midnight.
12
+ if len(text) == 10:
13
+ return date.fromisoformat(text)
14
+ parsed = datetime.fromisoformat(text.replace('Z', '+00:00'))
15
+ return parsed.astimezone(zone).date() if parsed.tzinfo else parsed.date()
16
+
17
+
18
+ def compile_sync_filters(config, now=None):
19
+ """Resolve relative dates once per run; date boundaries include entire days."""
20
+ if not config:
21
+ return []
22
+ zone_name = config.get('timeZone') or 'UTC'
23
+ zone = ZoneInfo(zone_name)
24
+ today = (now or datetime.now(timezone.utc)).astimezone(zone).date()
25
+
26
+ def resolve(value):
27
+ offsets = {'__dq:today': 0, '__dq:yesterday': -1, '__dq:tomorrow': 1}
28
+ if value in offsets:
29
+ return today + timedelta(days=offsets[value])
30
+ return _calendar_date(value, zone)
31
+
32
+ supported = {
33
+ 'is_equal', 'is_not_equal', 'is_empty', 'is_not_empty',
34
+ 'text_contains', 'text_not_contains', 'text_starts_with', 'text_ends_with',
35
+ 'gt', 'gte', 'lt', 'lte', 'between', 'not_between',
36
+ 'date_is', 'date_before', 'date_after',
37
+ }
38
+ result = []
39
+ for column, entry in (config.get('values') or {}).items():
40
+ operator = entry.get('operator', 'none')
41
+ if operator == 'none':
42
+ continue
43
+ if operator not in supported:
44
+ raise ValueError('Unsupported sync filter operator.')
45
+ is_date = entry.get('kind') == 'date' or operator.startswith('date_')
46
+ value, value2 = entry.get('value'), entry.get('value2')
47
+ if operator not in ('is_empty', 'is_not_empty'):
48
+ if value is None or str(value).strip() == '':
49
+ raise ValueError('A sync filter value is required.')
50
+ if is_date:
51
+ value = resolve(value).isoformat()
52
+ if operator in ('between', 'not_between'):
53
+ if value2 is None or str(value2).strip() == '':
54
+ raise ValueError('Both sync filter range values are required.')
55
+ if is_date:
56
+ value2 = resolve(value2).isoformat()
57
+ if value > value2:
58
+ raise ValueError('The filter start date must be on or before its end date.')
59
+ result.append({
60
+ 'column_name': str(column), 'operator_cd': operator,
61
+ 'value_1_txt': value, 'value_2_txt': value2,
62
+ 'kind': 'date' if is_date else entry.get('kind'), 'time_zone': zone_name,
63
+ })
64
+
65
+ snapshot = config.get('date') or {}
66
+ bounds = {}
67
+ for side in ('start', 'end'):
68
+ selection = int(snapshot.get(side + 'SelectionDR') or 0)
69
+ if not selection:
70
+ continue
71
+ if not snapshot.get('filterObject'):
72
+ raise ValueError('Choose a column for the date range.')
73
+ if selection == 1:
74
+ value = today
75
+ elif selection == 2:
76
+ value = today - timedelta(days=1)
77
+ elif selection == 3:
78
+ days = int(snapshot[side + 'DaysFromToday'])
79
+ if days < 0:
80
+ raise ValueError('Days before today cannot be negative.')
81
+ value = today - timedelta(days=days)
82
+ elif selection == 4:
83
+ value = resolve(snapshot[side + 'CustomDate'])
84
+ else:
85
+ raise ValueError('Unsupported date range selection.')
86
+ bounds[side] = value
87
+ result.append({
88
+ 'column_name': str(snapshot['filterObject']),
89
+ 'operator_cd': 'gte' if side == 'start' else 'lte',
90
+ 'value_1_txt': value.isoformat(), 'kind': 'date', 'time_zone': zone_name,
91
+ })
92
+ if 'start' in bounds and 'end' in bounds and bounds['start'] > bounds['end']:
93
+ raise ValueError('The filter start date must be on or before its end date.')
94
+ return result
95
+
96
+
97
+ def apply_data_filters(results, data_filters):
98
+ if not data_filters:
99
+ return results
100
+
101
+ # Connectors may return either:
102
+ # 1) a bare list of rows, or
103
+ # 2) an envelope: {"data": [...], "next_page": ...}
104
+ is_envelope = isinstance(results, dict) and isinstance(results.get("data"), list)
105
+ rows = results.get("data") if is_envelope else results
106
+ if not isinstance(rows, list):
107
+ return results
108
+
109
+ # 🔹 UI → backend operator mapping
110
+ OPERATOR_MAP = {
111
+ # text
112
+ "text_contains": "CONTAINS",
113
+ "text_not_contains": "NOT_CONTAINS",
114
+ "text_starts_with": "STARTS_WITH",
115
+ "text_ends_with": "ENDS_WITH",
116
+
117
+ # equality
118
+ "is_equal": "EQ",
119
+ "is_not_equal": "NEQ",
120
+ "date_is": "EQ",
121
+ "date_before": "LT",
122
+ "date_after": "GT",
123
+
124
+ # empty
125
+ "is_empty": "IS_EMPTY",
126
+ "is_not_empty": "IS_NOT_EMPTY",
127
+
128
+ # numeric
129
+ "gt": "GT",
130
+ "gte": "GTE",
131
+ "lt": "LT",
132
+ "lte": "LTE",
133
+
134
+ # range
135
+ "between": "BETWEEN",
136
+ "not_between": "NOT_BETWEEN",
137
+
138
+ # ignore
139
+ "none": None,
140
+ }
141
+
142
+ def try_parse_number(val):
143
+ try:
144
+ return float(val)
145
+ except:
146
+ return val
147
+
148
+ def normalize(row_val, v1, v2=None):
149
+ row_num = try_parse_number(row_val)
150
+ v1_num = try_parse_number(v1)
151
+ v2_num = try_parse_number(v2) if v2 is not None else None
152
+
153
+ # If both are numbers → compare as numbers
154
+ if isinstance(row_num, float) and isinstance(v1_num, float):
155
+ return row_num, v1_num, v2_num
156
+
157
+ # Otherwise → compare as strings
158
+ return str(row_val), str(v1), str(v2) if v2 is not None else None
159
+
160
+ def match(row, f):
161
+ ui_operator = f.get("operator_cd")
162
+ operator = OPERATOR_MAP.get(ui_operator)
163
+
164
+ if operator is None:
165
+ if ui_operator == 'none':
166
+ return True
167
+ raise ValueError('Unsupported sync filter operator.')
168
+
169
+ col = f.get("column_name")
170
+ val1 = f.get("value_1_txt")
171
+ val2 = f.get("value_2_txt")
172
+
173
+ # Some connectors can emit non-dict rows; keep filtering resilient.
174
+ if not isinstance(row, dict):
175
+ return False
176
+ row_val = row.get(col)
177
+
178
+ # ---- EMPTY HANDLING ----
179
+ if operator == "IS_EMPTY":
180
+ return row_val is None or str(row_val).strip() == ""
181
+
182
+ if operator == "IS_NOT_EMPTY":
183
+ return row_val is not None and str(row_val).strip() != ""
184
+
185
+ if row_val is None:
186
+ return False
187
+
188
+ # Normalize values
189
+ if f.get('kind') == 'date' or ui_operator.startswith('date_'):
190
+ zone = ZoneInfo(f.get('time_zone') or 'UTC')
191
+ v1 = _calendar_date(val1, zone)
192
+ v2 = _calendar_date(val2, zone) if val2 not in (None, '') else None
193
+ try:
194
+ row_val = _calendar_date(row_val, zone)
195
+ except (ValueError, TypeError):
196
+ return False
197
+ else:
198
+ row_val, v1, v2 = normalize(row_val, val1, val2)
199
+
200
+ # ---- OPERATORS ----
201
+
202
+ if operator == "EQ":
203
+ return row_val == v1
204
+
205
+ elif operator == "NEQ":
206
+ return row_val != v1
207
+
208
+ elif operator == "GT":
209
+ return row_val > v1
210
+
211
+ elif operator == "GTE":
212
+ return row_val >= v1
213
+
214
+ elif operator == "LT":
215
+ return row_val < v1
216
+
217
+ elif operator == "LTE":
218
+ return row_val <= v1
219
+
220
+ elif operator == "BETWEEN":
221
+ return v1 <= row_val <= v2
222
+
223
+ elif operator == "NOT_BETWEEN":
224
+ return not (v1 <= row_val <= v2)
225
+
226
+ elif operator == "CONTAINS":
227
+ return str(v1).lower() in str(row_val).lower()
228
+
229
+ elif operator == "NOT_CONTAINS":
230
+ return str(v1).lower() not in str(row_val).lower()
231
+
232
+ elif operator == "STARTS_WITH":
233
+ return str(row_val).lower().startswith(str(v1).lower())
234
+
235
+ elif operator == "ENDS_WITH":
236
+ return str(row_val).lower().endswith(str(v1).lower())
237
+
238
+ return True
239
+
240
+ # 🔹 Apply filters (AND logic)
241
+ filtered_rows = [row for row in rows if all(match(row, f) for f in data_filters)]
242
+ if is_envelope:
243
+ return {**results, "data": filtered_rows}
244
+ return filtered_rows
245
+
246
+
@@ -1,5 +1,6 @@
1
1
  from dc_sdk import errors
2
2
  import logging
3
+ import inspect
3
4
  from dc_sdk.src.services.loader import load_connector
4
5
  from dc_sdk.src.connection_status import (
5
6
  STATUS_CONNECTED,
@@ -14,6 +15,7 @@ class Mapping():
14
15
  def __init__(self, credentials):
15
16
  Connector = load_connector()
16
17
  self.connector = Connector(credentials)
18
+ self.request_phase = "authentication"
17
19
 
18
20
  def _set_connection_status(self, status):
19
21
  if self.connector.credentials is None:
@@ -35,6 +37,7 @@ class Mapping():
35
37
  raise errors.AuthenticationError(message="Failed to initialize connection session")
36
38
 
37
39
  def _ensure_session(self, force_authenticate=False):
40
+ self.request_phase = "authentication"
38
41
  credentials = self.connector.credentials or {}
39
42
  status = credentials.get("last_status_dsc")
40
43
 
@@ -71,16 +74,31 @@ class Mapping():
71
74
 
72
75
  self._ensure_session(force_authenticate=False)
73
76
 
74
- def _get_metadata(self, required=False):
77
+ def _discover(self, method_name, sections=None):
78
+ """Pass sections only to connectors that explicitly support the keyword."""
79
+ self.request_phase = "metadata"
80
+ method = getattr(self.connector, method_name)
81
+ if sections is None:
82
+ return method()
83
+ if not isinstance(sections, list) or any(not isinstance(section, str) or not section.strip() for section in sections):
84
+ raise ValueError("sections must be a list of non-empty strings")
85
+ parameter = inspect.signature(method).parameters.get("sections")
86
+ if parameter and parameter.kind in (inspect.Parameter.POSITIONAL_OR_KEYWORD, inspect.Parameter.KEYWORD_ONLY):
87
+ return method(sections=list(dict.fromkeys(sections)))
88
+ # Legacy connectors retain full discovery and their original response shape.
89
+ return method()
90
+
91
+ def _get_metadata(self, required=False, sections=None):
75
92
  """
76
93
  Fetch connector metadata.
77
94
 
78
- On connect/authenticate paths, failures must surface (required=True) so
79
- report-catalog connectors cannot look Connected with empty accounts.
95
+ Required discovery failures still surface to the caller, but do not
96
+ invalidate successful authentication unless credentials were rejected.
80
97
  Soft-fail (required=False) only for optional metadata enrichment.
81
98
  """
99
+ self.request_phase = "metadata"
82
100
  try:
83
- return self.connector.get_metadata()
101
+ return self._discover("get_metadata", sections)
84
102
  except errors.NotImplementedError:
85
103
  return None
86
104
  except Exception:
@@ -89,13 +107,13 @@ class Mapping():
89
107
  logger.exception("Error getting metadata")
90
108
  return None
91
109
 
92
- def authenticate(self, force_authenticate=True):
110
+ def authenticate(self, force_authenticate=True, include_metadata=True):
93
111
  results = None
94
112
  message = None
95
113
 
96
114
  self._ensure_session(force_authenticate=force_authenticate)
97
115
 
98
- metadata = self._get_metadata(required=True)
116
+ metadata = self._get_metadata(required=True) if include_metadata else None
99
117
  results = {
100
118
  "metadata": metadata
101
119
  }
@@ -103,7 +121,7 @@ class Mapping():
103
121
 
104
122
  return [results, message]
105
123
 
106
- def connect_get_objects(self, skip_authenticate=False, force_authenticate=False):
124
+ def connect_get_objects(self, skip_authenticate=False, force_authenticate=False, include_metadata=True):
107
125
  results = None
108
126
  metadata = None
109
127
  objects = None
@@ -114,8 +132,9 @@ class Mapping():
114
132
  force_authenticate=force_authenticate,
115
133
  )
116
134
 
135
+ self.request_phase = "metadata"
117
136
  objects = self.connector.get_objects()
118
- metadata = self._get_metadata(required=True)
137
+ metadata = self._get_metadata(required=True) if include_metadata else None
119
138
 
120
139
  results = {
121
140
  "metadata": metadata,
@@ -125,22 +144,23 @@ class Mapping():
125
144
 
126
145
  return [results, message]
127
146
 
128
- def get_metadata_only(self, skip_authenticate=False, force_authenticate=False):
147
+ def get_metadata_only(self, skip_authenticate=False, force_authenticate=False, sections=None):
129
148
  self._resolve_session(
130
149
  skip_authenticate=skip_authenticate,
131
150
  force_authenticate=force_authenticate,
132
151
  )
133
152
 
134
- metadata = self._get_metadata(required=True)
153
+ metadata = self._get_metadata(required=True, sections=sections)
135
154
  return [{"metadata": metadata}, "Retrieved metadata"]
136
155
 
137
- def get_objects_only(self, skip_authenticate=False, force_authenticate=False, include_metadata=True):
156
+ def get_objects_only(self, skip_authenticate=False, force_authenticate=False, include_metadata=True, sections=None):
138
157
  self._resolve_session(
139
158
  skip_authenticate=skip_authenticate,
140
159
  force_authenticate=force_authenticate,
141
160
  )
142
161
 
143
- objects = self.connector.get_objects()
162
+ self.request_phase = "metadata"
163
+ objects = self._discover("get_objects", sections)
144
164
  results = {"objects": objects}
145
165
 
146
166
  if include_metadata:
@@ -10,6 +10,7 @@ from .services.loader import load_connector
10
10
  from .destination_object_template import resolve_destination_object_template
11
11
  from dc_sdk.data_stream import DataStream
12
12
  from dc_sdk.json_safe import to_json_safe
13
+ from dc_sdk.row_filters import compile_sync_filters, apply_data_filters
13
14
 
14
15
  TEMP_UPLOADS = 'temporary-files'
15
16
  SUB_FOLDER = 'etlJobHistory'
@@ -82,6 +83,8 @@ class PipelineConductor:
82
83
  while self.authentication_tries <= 3 and not authenticated:
83
84
  try:
84
85
  authenticated = self.connector.authenticate()
86
+ if not authenticated:
87
+ raise errors.AuthenticationError("Connector authentication failed.")
85
88
  except Exception as e:
86
89
  if self.authentication_tries == 3 or "firewall" in str(e): # todo: fix add retry flg in raised errors
87
90
  raise e
@@ -97,9 +100,9 @@ class PipelineConductor:
97
100
  response = self.api.get_credential_updates(True, self.pipeline_id)
98
101
 
99
102
  creds = self.aws.decrypt_customer_data_object(response['credential'], response['organization'],
100
- encrypted_data_key_txt=self.pipeline_details.source_credential_information.get('encrypted_data_key_txt'),
101
- encryption_iv_txt=self.pipeline_details.source_credential_information.get('encryption_iv_txt'),
102
- encryption_auth_tag_txt=self.pipeline_details.source_credential_information.get('encryption_auth_tag_txt')
103
+ encrypted_data_key_txt=response.get('encrypted_data_key_txt'),
104
+ encryption_iv_txt=response.get('encryption_iv_txt'),
105
+ encryption_auth_tag_txt=response.get('encryption_auth_tag_txt')
103
106
  )
104
107
 
105
108
  self.connector.credentials = creds
@@ -108,9 +111,9 @@ class PipelineConductor:
108
111
  response = self.api.get_credential_updates(False, self.pipeline_id)
109
112
 
110
113
  creds = self.aws.decrypt_customer_data_object(response['credential'], response['organization'],
111
- encrypted_data_key_txt=self.pipeline_details.destination_credential_information.get('encrypted_data_key_txt'),
112
- encryption_iv_txt=self.pipeline_details.destination_credential_information.get('encryption_iv_txt'),
113
- encryption_auth_tag_txt=self.pipeline_details.destination_credential_information.get('encryption_auth_tag_txt')
114
+ encrypted_data_key_txt=response.get('encrypted_data_key_txt'),
115
+ encryption_iv_txt=response.get('encryption_iv_txt'),
116
+ encryption_auth_tag_txt=response.get('encryption_auth_tag_txt')
114
117
  )
115
118
 
116
119
  self.connector.credentials = creds
@@ -124,6 +127,8 @@ class PipelineConductor:
124
127
  while self.authentication_tries <= 3 and not authenticated:
125
128
  try:
126
129
  authenticated = self.connector.authenticate()
130
+ if not authenticated:
131
+ raise errors.AuthenticationError("Connector authentication failed.")
127
132
  except Exception as e:
128
133
  if self.authentication_tries == 3 or "firewall" in str(e): # todo: fix add retry flg in raised errors
129
134
  raise e
@@ -137,6 +142,7 @@ class PipelineConductor:
137
142
 
138
143
  def get_data(self):
139
144
  # Determine batch size
145
+ self._get_data_filters() # Validate and resolve relative dates before fetching any rows.
140
146
 
141
147
  self.log(self.log_templates.GET_DATA_START.format(
142
148
  self.pipeline_details.source_object_id,
@@ -146,31 +152,22 @@ class PipelineConductor:
146
152
  nrows = self._get_batch_row_count()
147
153
  max_allowed = self.pipeline_details.max_allowed_retrieval
148
154
 
149
- results = self.connector.get_data(self.pipeline_details.source_object_id, self._get_field_ids(), n_rows=nrows, filters=self._get_filters(), options=self.pipeline_details.options)
150
-
151
- while "next_page" in results and results["next_page"] != None:
152
- if "data" in results and results["data"] != None and results["data"] != []:
153
- self._update_extracted_sync_cursor(results.get("metadata"))
154
- limit_reached = self._process_rows(results["data"], max_allowed)
155
- if limit_reached:
156
- break
157
- elif results["data"] != []:
158
- self.internal_log(self.log_templates.INTERNAL_GET_DATA_FETCHED.format(0))
159
-
160
- # Check if we've reached the limit before fetching next page
155
+ next_page = None
156
+ while True:
157
+ results = self.connector.get_data(
158
+ self.pipeline_details.source_object_id, self._get_source_field_ids(),
159
+ n_rows=nrows, filters=self._get_filters(),
160
+ options=self.pipeline_details.options, next_page=next_page,
161
+ )
162
+ self._update_extracted_sync_cursor(results.get("metadata"))
163
+ rows = results.get("data") or []
164
+ if rows and self._process_rows(rows, max_allowed):
165
+ break
161
166
  if max_allowed is not None and self.row_count >= max_allowed:
162
167
  break
163
-
164
- if "next_page" in results and results["next_page"] != None:
165
- results = self.connector.get_data(self.pipeline_details.source_object_id, self._get_field_ids(), n_rows=nrows, filters=self._get_filters(), options=self.pipeline_details.options, next_page=results["next_page"])
166
-
167
- if "data" in results and results["data"] != None and results["data"] != []:
168
- self._update_extracted_sync_cursor(results.get("metadata"))
169
- self._process_rows(results["data"], max_allowed)
170
- elif results["data"] != []:
171
- self.internal_log(self.log_templates.INTERNAL_GET_DATA_FETCHED.format(0))
172
- else:
173
- self._update_extracted_sync_cursor(results.get("metadata"))
168
+ next_page = results.get("next_page")
169
+ if next_page is None:
170
+ break
174
171
 
175
172
  if self._should_track_sync_cursor():
176
173
  self._save_pending_sync_cursor()
@@ -184,6 +181,8 @@ class PipelineConductor:
184
181
  Not wired into the default SOURCE path yet — call explicitly when testing.
185
182
  Retains extract/normalize hints via a sibling .meta.json object.
186
183
  """
184
+ if self._get_data_filters():
185
+ raise errors.DataError('Row filters require the standard row transfer path.')
187
186
  if not hasattr(self.connector, "get_data_stream"):
188
187
  raise errors.DataError(
189
188
  "Connector does not implement get_data_stream; cannot use stream transfer."
@@ -233,6 +232,7 @@ class PipelineConductor:
233
232
  if not loaded:
234
233
  raise errors.LoadDataError("Loading data failed.")
235
234
  else:
235
+ keys.sort(key=self._transfer_batch_sort_key)
236
236
  for index, key in enumerate(keys):
237
237
  file_object = None
238
238
  if PipelineEnvironment.platform == "aws":
@@ -257,6 +257,11 @@ class PipelineConductor:
257
257
  self.log(self.log_templates.LOAD_DATA_FINISHED.format(self.row_count, destination_object_id))
258
258
  self._persist_sync_cursor_after_load()
259
259
 
260
+ @staticmethod
261
+ def _transfer_batch_sort_key(key):
262
+ match = re.search(r"-b(\d+)\.json$", key)
263
+ return (0, int(match.group(1))) if match else (1, key)
264
+
260
265
  def _create_connector(self, connector_cls, credentials):
261
266
  init_params = inspect.signature(connector_cls.__init__).parameters
262
267
  if "pipeline_context" in init_params:
@@ -428,6 +433,14 @@ class PipelineConductor:
428
433
  self.api.create_pipeline_mapping(self.pipeline_id, self.pipeline_details.pipeline_mapping_json)
429
434
 
430
435
  def _process_rows(self, rows, max_allowed=None):
436
+ rows = apply_data_filters(rows, self._get_data_filters())
437
+ if not rows:
438
+ return False
439
+ selected_ids = {str(field) for field in self._get_field_ids()}
440
+ if selected_ids:
441
+ filter_only = {f['column_name'] for f in self._get_data_filters()} - selected_ids
442
+ if filter_only:
443
+ rows = [{key: value for key, value in row.items() if str(key) not in filter_only} for row in rows]
431
444
  # Check if we have a limit and need to truncate rows
432
445
  limit_reached = False
433
446
  if max_allowed is not None:
@@ -590,7 +603,7 @@ class PipelineConductor:
590
603
  def _get_batch_row_count(self):
591
604
  results = self.connector.get_data(
592
605
  self.pipeline_details.source_object_id,
593
- self._get_field_ids(),
606
+ self._get_source_field_ids(),
594
607
  n_rows=32,
595
608
  filters=self._get_filters(),
596
609
  options=self.pipeline_details.options
@@ -639,6 +652,9 @@ class PipelineConductor:
639
652
  return batch_size
640
653
 
641
654
  def _get_filters(self):
655
+ # New filters are enforced centrally; do not also apply a stale legacy range.
656
+ if 'sync_filters' in (self.pipeline_details.options or {}):
657
+ return None
642
658
  if self.mode == "prod":
643
659
  return {
644
660
  "filtered_column_nm": self.pipeline_details.filtered_column_nm,
@@ -651,6 +667,22 @@ class PipelineConductor:
651
667
  else:
652
668
  return self.filters
653
669
 
670
+ def _get_data_filters(self):
671
+ if not hasattr(self, '_resolved_data_filters'):
672
+ self._resolved_data_filters = compile_sync_filters(
673
+ (self.pipeline_details.options or {}).get('sync_filters')
674
+ )
675
+ return self._resolved_data_filters
676
+
677
+ def _get_source_field_ids(self):
678
+ fields = self._get_field_ids()
679
+ if not fields: # An empty mapping requests every source column.
680
+ return fields
681
+ ids = {str(field) for field in fields}
682
+ return fields + list(dict.fromkeys(
683
+ f['column_name'] for f in self._get_data_filters() if f['column_name'] not in ids
684
+ ))
685
+
654
686
  def _get_field_ids(self):
655
687
  if not self.pipeline_details.pipeline_mapping_json:
656
688
  return []
@@ -745,4 +777,4 @@ class PipelineConductor:
745
777
  i += 1
746
778
  seen[name] = True
747
779
  mapping.append({"column": name, "mapped": fid})
748
- return mapping
780
+ return mapping
@@ -103,6 +103,21 @@ class DataConnectorAPI:
103
103
  f"credentials/{pipeline_id}?source={'true' if source_flg else 'false'}"
104
104
  )
105
105
 
106
+ envelope_fields = (
107
+ 'encrypted_data_key_txt', 'encryption_iv_txt', 'encryption_auth_tag_txt'
108
+ )
109
+ if not all(field in response for field in envelope_fields):
110
+ # Older APIs omit the envelope. Fetch ciphertext and metadata together
111
+ # from a fresh pipeline snapshot; never mix separate credential reads.
112
+ details = self.get(str(pipeline_id))
113
+ side = 'source' if source_flg else 'destination'
114
+ information = details.get(f'{side}_credential_information') or {}
115
+ response = {
116
+ 'credential': details[f'{side}_encryption_credential_txt'],
117
+ 'organization': details['customer_metadata_uuid'],
118
+ **{field: information.get(field) for field in envelope_fields},
119
+ }
120
+
106
121
  return response
107
122
 
108
123
  def create_pipeline_mapping(self, pipeline_id, pipeline_mapping_json):
@@ -220,4 +235,4 @@ class DataConnectorAPI:
220
235
  def _get_retry_delay(self, attempt):
221
236
  base_delay = min(2 ** attempt, 10)
222
237
  jitter = random.uniform(0, 0.5)
223
- return base_delay + jitter
238
+ return base_delay + jitter
@@ -52,19 +52,16 @@ class AwsService:
52
52
  self._ecs_cluster_name = APP_ENV if APP_ENV != "local" else "development"
53
53
 
54
54
  def get_keys(self, pipeline_run_history_id):
55
- keys = []
56
- response = s3_client.list_objects(Bucket=self.s3_bucket, Prefix=f"transfers/e{pipeline_run_history_id}")
57
- if 'Contents' in response:
58
- for key in response['Contents']:
59
- keys.append(key['Key'])
60
- return keys
55
+ return self._get_keys(f"transfers/e{pipeline_run_history_id}-b")
61
56
 
62
57
  def get_dev_keys(self, prefix):
58
+ return self._get_keys(f"devTransfers/e{prefix}-b")
59
+
60
+ def _get_keys(self, prefix):
63
61
  keys = []
64
- response = s3_client.list_objects(Bucket=self.s3_bucket, Prefix=f"devTransfers/e{prefix}")
65
- if 'Contents' in response:
66
- for key in response['Contents']:
67
- keys.append(key['Key'])
62
+ paginator = s3_client.get_paginator('list_objects_v2')
63
+ for page in paginator.paginate(Bucket=self.s3_bucket, Prefix=prefix):
64
+ keys.extend(obj['Key'] for obj in page.get('Contents', []))
68
65
  return keys
69
66
 
70
67
  def download_object(self, key_name):
@@ -328,4 +325,4 @@ class EncryptionService:
328
325
  encryption_auth_tag_txt=encryption_auth_tag_txt
329
326
  )
330
327
 
331
- return json.loads(data)
328
+ return json.loads(data)
@@ -1,242 +0,0 @@
1
- from dc_sdk.errors import Error
2
- from dc_sdk.json_safe import to_json_safe
3
- from dc_sdk.src.mapping import Mapping
4
- from dc_sdk.src.connection_status import STATUS_ERROR
5
- import traceback
6
- from importlib.metadata import version
7
-
8
- def get_action_name(action: int) -> str:
9
- return {
10
- 0: "authenticating",
11
- 1: "retrieving fields",
12
- 2: "retrieving 5 row preview",
13
- 3: "testing connection",
14
- 4: "starting the connector",
15
- 5: "retrieving metadata",
16
- 6: "retrieving objects",
17
- }[action]
18
-
19
- def apply_data_filters(results, data_filters):
20
- if not data_filters:
21
- return results
22
-
23
- # Connectors may return either:
24
- # 1) a bare list of rows, or
25
- # 2) an envelope: {"data": [...], "next_page": ...}
26
- is_envelope = isinstance(results, dict) and isinstance(results.get("data"), list)
27
- rows = results.get("data") if is_envelope else results
28
- if not isinstance(rows, list):
29
- return results
30
-
31
- # 🔹 UI → backend operator mapping
32
- OPERATOR_MAP = {
33
- # text
34
- "text_contains": "CONTAINS",
35
- "text_not_contains": "NOT_CONTAINS",
36
- "text_starts_with": "STARTS_WITH",
37
- "text_ends_with": "ENDS_WITH",
38
-
39
- # equality
40
- "is_equal": "EQ",
41
- "is_not_equal": "NEQ",
42
-
43
- # empty
44
- "is_empty": "IS_EMPTY",
45
- "is_not_empty": "IS_NOT_EMPTY",
46
-
47
- # numeric
48
- "gt": "GT",
49
- "gte": "GTE",
50
- "lt": "LT",
51
- "lte": "LTE",
52
-
53
- # range
54
- "between": "BETWEEN",
55
- "not_between": "NOT_BETWEEN",
56
-
57
- # ignore
58
- "none": None,
59
- }
60
-
61
- def try_parse_number(val):
62
- try:
63
- return float(val)
64
- except:
65
- return val
66
-
67
- def normalize(row_val, v1, v2=None):
68
- row_num = try_parse_number(row_val)
69
- v1_num = try_parse_number(v1)
70
- v2_num = try_parse_number(v2) if v2 is not None else None
71
-
72
- # If both are numbers → compare as numbers
73
- if isinstance(row_num, float) and isinstance(v1_num, float):
74
- return row_num, v1_num, v2_num
75
-
76
- # Otherwise → compare as strings
77
- return str(row_val), str(v1), str(v2) if v2 is not None else None
78
-
79
- def match(row, f):
80
- ui_operator = f.get("operator_cd")
81
- operator = OPERATOR_MAP.get(ui_operator)
82
-
83
- if operator is None:
84
- return True # skip "none"
85
-
86
- col = f.get("column_name")
87
- val1 = f.get("value_1_txt")
88
- val2 = f.get("value_2_txt")
89
-
90
- # Some connectors can emit non-dict rows; keep filtering resilient.
91
- if not isinstance(row, dict):
92
- return False
93
- row_val = row.get(col)
94
-
95
- # ---- EMPTY HANDLING ----
96
- if operator == "IS_EMPTY":
97
- return row_val is None or str(row_val).strip() == ""
98
-
99
- if operator == "IS_NOT_EMPTY":
100
- return row_val is not None and str(row_val).strip() != ""
101
-
102
- if row_val is None:
103
- return False
104
-
105
- # Normalize values
106
- row_val, v1, v2 = normalize(row_val, val1, val2)
107
-
108
- # ---- OPERATORS ----
109
-
110
- if operator == "EQ":
111
- return row_val == v1
112
-
113
- elif operator == "NEQ":
114
- return row_val != v1
115
-
116
- elif operator == "GT":
117
- return row_val > v1
118
-
119
- elif operator == "GTE":
120
- return row_val >= v1
121
-
122
- elif operator == "LT":
123
- return row_val < v1
124
-
125
- elif operator == "LTE":
126
- return row_val <= v1
127
-
128
- elif operator == "BETWEEN":
129
- return v1 <= row_val <= v2
130
-
131
- elif operator == "NOT_BETWEEN":
132
- return not (v1 <= row_val <= v2)
133
-
134
- elif operator == "CONTAINS":
135
- return str(v1).lower() in str(row_val).lower()
136
-
137
- elif operator == "NOT_CONTAINS":
138
- return str(v1).lower() not in str(row_val).lower()
139
-
140
- elif operator == "STARTS_WITH":
141
- return str(row_val).lower().startswith(str(v1).lower())
142
-
143
- elif operator == "ENDS_WITH":
144
- return str(row_val).lower().endswith(str(v1).lower())
145
-
146
- return True
147
-
148
- # 🔹 Apply filters (AND logic)
149
- filtered_rows = [row for row in rows if all(match(row, f) for f in data_filters)]
150
- if is_envelope:
151
- return {**results, "data": filtered_rows}
152
- return filtered_rows
153
-
154
-
155
- def handler(event, context):
156
- print("version: ", version("dc-python-sdk"))
157
- """Lambda Handler"""
158
- action_name = None
159
- internal_error = False
160
- message = None
161
- results = None
162
- error = None
163
- mapping = None
164
- action = int(event['action']) if 'action' in event else 4
165
- get_objects = event.get('get_objects', True)
166
- skip_authenticate = event.get('skip_authenticate', False)
167
- force_authenticate = event.get('force_authenticate', False)
168
- include_metadata = event.get('include_metadata', True)
169
- credentials_dict = event['credentials'] if 'credentials' in event else None
170
- object_id = event['object_id'] if 'object_id' in event else None
171
- field_ids = event['mapping'] if 'mapping' in event else None
172
- options = event['options'] if 'options' in event else dict()
173
- next_page = event['next_page'] if 'next_page' in event else None
174
- n_rows = event['n_rows'] if 'n_rows' in event else None
175
- filters = event['filters'] if 'filters' in event else None
176
- data_filters = event['data_filters'] if 'data_filters' in event else None
177
-
178
- try:
179
- action_name = get_action_name(action)
180
- mapping = Mapping(credentials_dict)
181
-
182
- if action == 0:
183
- if get_objects:
184
- results, message = mapping.connect_get_objects(
185
- skip_authenticate=skip_authenticate,
186
- force_authenticate=force_authenticate,
187
- )
188
- else:
189
- results, message = mapping.authenticate(
190
- force_authenticate=force_authenticate or not skip_authenticate,
191
- )
192
- elif action == 1:
193
- results, message = mapping.get_fields(object_id, options)
194
- elif action == 2:
195
- results, message = mapping.get_five_row_preview(
196
- object_id, field_ids, options, n_rows, filters, next_page)
197
- results = apply_data_filters(results, data_filters)
198
- elif action == 3:
199
- results, message = mapping.test_connection()
200
- elif action == 5:
201
- results, message = mapping.get_metadata_only(
202
- skip_authenticate=skip_authenticate,
203
- force_authenticate=force_authenticate,
204
- )
205
- elif action == 6:
206
- results, message = mapping.get_objects_only(
207
- skip_authenticate=skip_authenticate,
208
- force_authenticate=force_authenticate,
209
- include_metadata=include_metadata,
210
- )
211
- else:
212
- raise Error("Invalid action", "InvalidActionError")
213
- except Error as e:
214
- message = e.message or str(e)
215
- internal_error = e.internal
216
- error_trace = traceback.format_exc()
217
- error = error_trace
218
- if mapping:
219
- mapping._set_connection_status(STATUS_ERROR)
220
-
221
- except Exception as e:
222
- error_trace = traceback.format_exc()
223
- error = error_trace
224
- internal_error = True
225
- if mapping:
226
- mapping._set_connection_status(STATUS_ERROR)
227
-
228
- if error:
229
- print(error)
230
-
231
- last_status_dsc = None
232
- if mapping and mapping.connector.credentials:
233
- last_status_dsc = mapping.connector.credentials.get("last_status_dsc")
234
-
235
- return {
236
- 'message': message if not internal_error else f"Something went wrong with {action_name}",
237
- 'results': to_json_safe(results),
238
- 'credentials': mapping.connector.credentials if mapping else None,
239
- 'last_status_dsc': last_status_dsc,
240
- 'error': error,
241
- 'internal_error': internal_error
242
- }
File without changes