easy-data-loader 0.1.8__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {easy_data_loader-0.1.8/src/easy_data_loader.egg-info → easy_data_loader-0.2.1}/PKG-INFO +1 -1
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/pyproject.toml +1 -1
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/cli.py +3 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/custom_exceptions.py +24 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/data_inferrence.py +13 -14
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/database_connector.py +3 -3
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/database_operations.py +9 -7
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/file_operations.py +4 -2
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/log.py +16 -12
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/models.py +2 -2
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/orchestrator.py +35 -23
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/pipeline.py +60 -37
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/pipeline_base.py +75 -6
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/procedure_pipeline.py +31 -17
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1/src/easy_data_loader.egg-info}/PKG-INFO +1 -1
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_cli.py +10 -5
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_database_connector.py +1 -1
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_orchestrator.py +4 -4
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_pipeline_transform_order.py +2 -2
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_procedure_execution_mode.py +3 -3
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_resource_access.py +5 -5
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_source_file_delete.py +12 -7
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_validation.py +12 -10
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/LICENSE +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/README.md +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/setup.cfg +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/__init__.py +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/config_loader.py +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/driver_detector.py +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/resource_access.py +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/utils.py +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/SOURCES.txt +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/dependency_links.txt +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/entry_points.txt +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/requires.txt +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/top_level.txt +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_config_loader.py +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_data_inference.py +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_file_operations.py +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_imports.py +0 -0
- {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_models.py +0 -0
|
@@ -412,6 +412,9 @@ def validate_pipelines():
|
|
|
412
412
|
else:
|
|
413
413
|
raise ValueError(f"Unknown pipeline type: {type(definition)}")
|
|
414
414
|
|
|
415
|
+
if instance is not None and instance._init_error is not None:
|
|
416
|
+
raise ValueError(instance._init_error.message)
|
|
417
|
+
|
|
415
418
|
results[name] = "OK"
|
|
416
419
|
except Exception as e:
|
|
417
420
|
results[name] = f"FAILED: {str(e)}"
|
|
@@ -26,3 +26,27 @@ class PipelineValidationError(Exception):
|
|
|
26
26
|
def __init__(self, message: str):
|
|
27
27
|
self.message = message
|
|
28
28
|
super().__init__(self.message)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class EmptyDataError(Exception):
|
|
32
|
+
def __init__(self, message: str):
|
|
33
|
+
self.message = message
|
|
34
|
+
super().__init__(self.message)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class NoValidDestination(Exception):
|
|
38
|
+
def __init__(self, message: str = "The pipeline destination is not valid"):
|
|
39
|
+
self.message = message
|
|
40
|
+
super().__init__(self.message)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class NestedPipelinesError(Exception):
|
|
44
|
+
def __init__(self, message: str = "Nested pipelines are not supported"):
|
|
45
|
+
self.message = message
|
|
46
|
+
super().__init__(self.message)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class UnknownPipelineError(Exception):
|
|
50
|
+
def __init__(self, message: str = "Unknown pipeline type"):
|
|
51
|
+
self.message = message
|
|
52
|
+
super().__init__(self.message)
|
|
@@ -88,7 +88,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
88
88
|
self.logger.error(
|
|
89
89
|
f"Failed to infer dtype for column '{col}': {str(e)}. "
|
|
90
90
|
f"Column will use default inference.",
|
|
91
|
-
exc_info=
|
|
91
|
+
exc_info=True,
|
|
92
92
|
)
|
|
93
93
|
# Skip this column - it will use default inference
|
|
94
94
|
continue
|
|
@@ -102,9 +102,8 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
102
102
|
|
|
103
103
|
self.logger.info(f"Dtype inference completed for {len(dtype_dict)} columns")
|
|
104
104
|
|
|
105
|
-
#
|
|
106
|
-
|
|
107
|
-
self._log_inference_summary(dtype_dict)
|
|
105
|
+
# Only visible in the file log, since it's logged at debug level
|
|
106
|
+
self._log_inference_summary(dtype_dict)
|
|
108
107
|
|
|
109
108
|
return dtype_dict
|
|
110
109
|
|
|
@@ -593,7 +592,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
593
592
|
self.logger.error(
|
|
594
593
|
f"Failed to infer dtype for column '{col_name}' from Parquet metadata: {str(e)}. "
|
|
595
594
|
f"Column will use default inference.",
|
|
596
|
-
exc_info=
|
|
595
|
+
exc_info=True,
|
|
597
596
|
)
|
|
598
597
|
continue
|
|
599
598
|
|
|
@@ -616,7 +615,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
616
615
|
self.logger.error(
|
|
617
616
|
f"Failed to analyze string length for column '{col}': {str(e)}. "
|
|
618
617
|
f"Column will use default inference.",
|
|
619
|
-
exc_info=
|
|
618
|
+
exc_info=True,
|
|
620
619
|
)
|
|
621
620
|
# Remove from dict so it uses default
|
|
622
621
|
dtype_dict.pop(col, None)
|
|
@@ -626,7 +625,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
626
625
|
self.logger.error(
|
|
627
626
|
f"Failed to read string columns from Parquet: {str(e)}. "
|
|
628
627
|
f"String columns will use default inference.",
|
|
629
|
-
exc_info=
|
|
628
|
+
exc_info=True,
|
|
630
629
|
)
|
|
631
630
|
# Remove string columns from dict
|
|
632
631
|
for col in string_columns:
|
|
@@ -643,8 +642,8 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
643
642
|
f"Parquet dtype inference completed for {len(dtype_dict)} columns"
|
|
644
643
|
)
|
|
645
644
|
|
|
646
|
-
|
|
647
|
-
|
|
645
|
+
# Only visible in the file log, since it's logged at debug level
|
|
646
|
+
self._log_inference_summary(dtype_dict)
|
|
648
647
|
|
|
649
648
|
return dtype_dict
|
|
650
649
|
|
|
@@ -703,7 +702,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
703
702
|
self.logger.error(
|
|
704
703
|
f"Failed to infer dtype for column '{col_name}' from ORC metadata: {str(e)}. "
|
|
705
704
|
f"Column will use default inference.",
|
|
706
|
-
exc_info=
|
|
705
|
+
exc_info=True,
|
|
707
706
|
)
|
|
708
707
|
continue
|
|
709
708
|
|
|
@@ -726,7 +725,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
726
725
|
self.logger.error(
|
|
727
726
|
f"Failed to analyze string length for column '{col}': {str(e)}. "
|
|
728
727
|
f"Column will use default inference.",
|
|
729
|
-
exc_info=
|
|
728
|
+
exc_info=True,
|
|
730
729
|
)
|
|
731
730
|
dtype_dict.pop(col, None)
|
|
732
731
|
continue
|
|
@@ -735,7 +734,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
735
734
|
self.logger.error(
|
|
736
735
|
f"Failed to read string columns from ORC: {str(e)}. "
|
|
737
736
|
f"String columns will use default inference.",
|
|
738
|
-
exc_info=
|
|
737
|
+
exc_info=True,
|
|
739
738
|
)
|
|
740
739
|
for col in string_columns:
|
|
741
740
|
dtype_dict.pop(col, None)
|
|
@@ -751,8 +750,8 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
751
750
|
f"ORC dtype inference completed for {len(dtype_dict)} columns"
|
|
752
751
|
)
|
|
753
752
|
|
|
754
|
-
|
|
755
|
-
|
|
753
|
+
# Only visible in the file log, since it's logged at debug level
|
|
754
|
+
self._log_inference_summary(dtype_dict)
|
|
756
755
|
|
|
757
756
|
return dtype_dict
|
|
758
757
|
|
{easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/database_connector.py
RENAMED
|
@@ -107,7 +107,7 @@ class SqlServerDatabaseConnector(LoggedComponent, DatabaseConnector):
|
|
|
107
107
|
max_overflow=10,
|
|
108
108
|
pool_timeout=30,
|
|
109
109
|
pool_recycle=3600,
|
|
110
|
-
echo=
|
|
110
|
+
echo=False,
|
|
111
111
|
)
|
|
112
112
|
return engine
|
|
113
113
|
except Exception as e:
|
|
@@ -183,7 +183,7 @@ class SQLiteDatabaseConnector(LoggedComponent, DatabaseConnector):
|
|
|
183
183
|
# SQLite specific engine creation
|
|
184
184
|
engine = create_engine(
|
|
185
185
|
connection_string,
|
|
186
|
-
echo=
|
|
186
|
+
echo=False,
|
|
187
187
|
)
|
|
188
188
|
return engine
|
|
189
189
|
except Exception as e:
|
|
@@ -280,7 +280,7 @@ class PostgresDatabaseConnector(LoggedComponent, DatabaseConnector):
|
|
|
280
280
|
max_overflow=10,
|
|
281
281
|
pool_timeout=30,
|
|
282
282
|
pool_recycle=3600,
|
|
283
|
-
echo=
|
|
283
|
+
echo=False,
|
|
284
284
|
)
|
|
285
285
|
return engine
|
|
286
286
|
except Exception as e:
|
{easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/database_operations.py
RENAMED
|
@@ -19,18 +19,20 @@ class DatabaseOperations(LoggedComponent):
|
|
|
19
19
|
self.engine = engine
|
|
20
20
|
self._inspector = inspect(self.engine)
|
|
21
21
|
|
|
22
|
-
def write_to_table(
|
|
23
|
-
self, table_name: Optional[str], df: DataFrame, **kwargs
|
|
24
|
-
) -> bool:
|
|
22
|
+
def write_to_table(self, table_name: Optional[str], df: DataFrame, **kwargs):
|
|
25
23
|
"""Write a dataframe to a specified table in the database"""
|
|
26
24
|
|
|
27
|
-
self.logger.
|
|
25
|
+
self.logger.debug(f"Writing {len(df)} rows to table: {table_name}")
|
|
28
26
|
try:
|
|
29
27
|
if table_name:
|
|
30
28
|
df.to_sql(table_name, con=self.engine, **kwargs)
|
|
31
|
-
|
|
29
|
+
self.logger.debug("Finalized write to table: {table_name}")
|
|
30
|
+
return
|
|
32
31
|
except Exception as e:
|
|
33
|
-
self.
|
|
32
|
+
self.logger.debug(
|
|
33
|
+
f"Failed to write to table {table_name} caused by the following exception: {e}",
|
|
34
|
+
exc_info=True,
|
|
35
|
+
)
|
|
34
36
|
raise
|
|
35
37
|
|
|
36
38
|
def read_data(self, sql: str, **kwargs) -> DataFrame:
|
|
@@ -151,7 +153,7 @@ class DatabaseOperations(LoggedComponent):
|
|
|
151
153
|
Column("file_last_modified", DateTime),
|
|
152
154
|
Column("sp_name", String(255)),
|
|
153
155
|
Column("sp_parameters", String),
|
|
154
|
-
Column("
|
|
156
|
+
Column("insert_timestamp", DateTime),
|
|
155
157
|
Column("error_details", String),
|
|
156
158
|
]
|
|
157
159
|
|
|
@@ -48,7 +48,9 @@ class FileOperations(LoggedComponent):
|
|
|
48
48
|
Identifies the file based on pattern (latest) or explicit name.
|
|
49
49
|
"""
|
|
50
50
|
if self.settings.file_pattern:
|
|
51
|
-
files = list(
|
|
51
|
+
files = list(
|
|
52
|
+
self.settings.folder_path.glob("*" + self.settings.file_pattern + "*")
|
|
53
|
+
)
|
|
52
54
|
if not files:
|
|
53
55
|
self.log_and_raise(
|
|
54
56
|
ValueError,
|
|
@@ -132,7 +134,7 @@ class FileOperations(LoggedComponent):
|
|
|
132
134
|
"""
|
|
133
135
|
|
|
134
136
|
current_path = self._find_file()
|
|
135
|
-
self.logger.
|
|
137
|
+
self.logger.debug(f"Current file path is: {current_path}")
|
|
136
138
|
|
|
137
139
|
if preprocessor_func is None:
|
|
138
140
|
self.file_path = current_path
|
|
@@ -21,25 +21,37 @@ class AppLogger:
|
|
|
21
21
|
def _setup_logging(self):
|
|
22
22
|
Path("logs").mkdir(exist_ok=True)
|
|
23
23
|
|
|
24
|
-
|
|
24
|
+
file_formatter = logging.Formatter(
|
|
25
25
|
"%(asctime)s - %(name)s - %(levelname)s - %(message)s",
|
|
26
26
|
datefmt="%Y-%m-%d %H:%M:%S",
|
|
27
27
|
)
|
|
28
28
|
|
|
29
|
+
console_formatter = logging.Formatter(
|
|
30
|
+
"[%(asctime)s]-[%(levelname)s] %(message)s"
|
|
31
|
+
)
|
|
32
|
+
|
|
29
33
|
root_logger = logging.getLogger()
|
|
30
|
-
root_logger.setLevel(
|
|
34
|
+
root_logger.setLevel(
|
|
35
|
+
logging.DEBUG
|
|
36
|
+
) # allow all messages and split them in the handlers
|
|
31
37
|
root_logger.handlers.clear()
|
|
32
38
|
|
|
33
39
|
# Console
|
|
34
40
|
console_handler = logging.StreamHandler()
|
|
35
|
-
console_handler.
|
|
41
|
+
console_handler.setLevel(
|
|
42
|
+
logging.INFO
|
|
43
|
+
) # the console log will have minimal friendly information
|
|
44
|
+
console_handler.setFormatter(console_formatter)
|
|
36
45
|
root_logger.addHandler(console_handler)
|
|
37
46
|
|
|
38
47
|
# File
|
|
39
48
|
file_handler = logging.handlers.RotatingFileHandler(
|
|
40
49
|
"logs/application.log", maxBytes=10 * 1024 * 1024, backupCount=5
|
|
41
50
|
)
|
|
42
|
-
file_handler.setFormatter(
|
|
51
|
+
file_handler.setFormatter(file_formatter)
|
|
52
|
+
file_handler.setLevel(
|
|
53
|
+
logging.DEBUG
|
|
54
|
+
) # the file log will have full detailed information
|
|
43
55
|
root_logger.addHandler(file_handler)
|
|
44
56
|
|
|
45
57
|
def get_logger(self, name: str) -> logging.Logger:
|
|
@@ -55,10 +67,6 @@ class AppLogger:
|
|
|
55
67
|
for handler in logging.getLogger().handlers:
|
|
56
68
|
handler.setLevel(log_level)
|
|
57
69
|
|
|
58
|
-
@property
|
|
59
|
-
def is_debug(self) -> bool:
|
|
60
|
-
return logging.getLogger().isEnabledFor(logging.DEBUG)
|
|
61
|
-
|
|
62
70
|
|
|
63
71
|
class LoggedComponent:
|
|
64
72
|
"""Base class providing logging functionality to all components"""
|
|
@@ -85,7 +93,3 @@ class LoggedComponent:
|
|
|
85
93
|
log_msg += f" | Context: {context_str}"
|
|
86
94
|
|
|
87
95
|
self.logger.error(log_msg, exc_info=True)
|
|
88
|
-
|
|
89
|
-
@property
|
|
90
|
-
def is_debug_enabled(self) -> bool:
|
|
91
|
-
return self.log.is_debug
|
|
@@ -191,7 +191,7 @@ class BasePipelineDefinition(BaseModel):
|
|
|
191
191
|
@property
|
|
192
192
|
def destination_table_invalid(self) -> Optional[str]:
|
|
193
193
|
if self.destination_table:
|
|
194
|
-
return
|
|
194
|
+
return "edl_validation_error_log"
|
|
195
195
|
return None
|
|
196
196
|
|
|
197
197
|
def file_pre_process(self, file_path: Path) -> Path:
|
|
@@ -282,7 +282,7 @@ class AuditEntry(BaseModel):
|
|
|
282
282
|
error_details: Optional[str] = None
|
|
283
283
|
sp_name: Optional[str] = None
|
|
284
284
|
sp_parameters: Optional[Dict[str, Any]] = None
|
|
285
|
-
|
|
285
|
+
insert_timestamp: pd.Timestamp = Field(default_factory=pd.Timestamp.now)
|
|
286
286
|
|
|
287
287
|
model_config = ConfigDict(arbitrary_types_allowed=True)
|
|
288
288
|
|
|
@@ -11,35 +11,55 @@ from .models import (
|
|
|
11
11
|
)
|
|
12
12
|
from .pipeline import LoadPipeline
|
|
13
13
|
from .procedure_pipeline import ProcedurePipeline
|
|
14
|
+
from .custom_exceptions import NestedPipelinesError, UnknownPipelineError
|
|
15
|
+
from .pipeline_base import PipelineResult
|
|
14
16
|
|
|
15
17
|
|
|
16
18
|
class OrchestratorPipeline(LoggedComponent):
|
|
17
|
-
"""Executes a chain of pipelines defined by
|
|
19
|
+
"""Executes a chain of pipelines defined by a OrchestratorDefinition"""
|
|
18
20
|
|
|
19
|
-
def __init__(self,
|
|
21
|
+
def __init__(self, file_name: str, orchestrator_id: Optional[str] = None):
|
|
20
22
|
super().__init__()
|
|
21
23
|
self.orchestrator_id = orchestrator_id if orchestrator_id else str(uuid4())
|
|
22
24
|
self.config = Configuration()
|
|
23
|
-
definition = self.config.get_pipeline(
|
|
25
|
+
definition = self.config.get_pipeline(file_name)
|
|
24
26
|
|
|
25
27
|
if not isinstance(definition, OrchestratorDefinition):
|
|
26
|
-
|
|
28
|
+
raise ValueError(
|
|
29
|
+
f"'{file_name}' does not contain a OrchestratorPipeline definition."
|
|
30
|
+
)
|
|
27
31
|
|
|
28
32
|
self.definition: OrchestratorDefinition = definition
|
|
29
33
|
|
|
30
|
-
def run(self) ->
|
|
34
|
+
def run(self) -> PipelineResult:
|
|
35
|
+
"""Entry point mirroring BasePipeline.run(), without audit/connector lifecycle."""
|
|
36
|
+
try:
|
|
37
|
+
success = self._run()
|
|
38
|
+
return PipelineResult(
|
|
39
|
+
success=success,
|
|
40
|
+
message="Orchestrator executed successfully."
|
|
41
|
+
if success
|
|
42
|
+
else "One or more pipelines failed.",
|
|
43
|
+
)
|
|
44
|
+
except Exception as e:
|
|
45
|
+
self.logger.warning(f"Orchestrator has [FAILED] {e}")
|
|
46
|
+
return PipelineResult(
|
|
47
|
+
success=False,
|
|
48
|
+
message="Orchestrator has failed",
|
|
49
|
+
error_code=e.__class__.__name__,
|
|
50
|
+
exception=e,
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
def _run(self) -> bool:
|
|
31
54
|
self.logger.info(
|
|
32
55
|
f"=== Starting Orchestrator: {self.definition.pipeline_name} (Orchestrator: {self.orchestrator_id}) ==="
|
|
33
56
|
)
|
|
34
57
|
|
|
35
58
|
success = True
|
|
36
59
|
for pipeline_name in self.definition.pipelines:
|
|
37
|
-
self.logger.info(
|
|
38
|
-
f"[{self.definition.pipeline_name}] -> Triggering pipeline: {pipeline_name}"
|
|
39
|
-
)
|
|
60
|
+
self.logger.info(f"Triggering pipeline: {pipeline_name}")
|
|
40
61
|
|
|
41
62
|
p_def: PipelineType = self.config.get_pipeline(pipeline_name)
|
|
42
|
-
p_success = False
|
|
43
63
|
|
|
44
64
|
# Instantiate and run
|
|
45
65
|
if isinstance(p_def, BasePipelineDefinition):
|
|
@@ -51,25 +71,17 @@ class OrchestratorPipeline(LoggedComponent):
|
|
|
51
71
|
pipeline_name, orchestrator_id=self.orchestrator_id
|
|
52
72
|
).run()
|
|
53
73
|
elif isinstance(p_def, OrchestratorDefinition):
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
)
|
|
57
|
-
p_success = False
|
|
74
|
+
# caught by run() and turned into a failed PipelineResult
|
|
75
|
+
raise NestedPipelinesError
|
|
58
76
|
else:
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
)
|
|
62
|
-
p_success = False
|
|
77
|
+
# caught by run() and turned into a failed PipelineResult
|
|
78
|
+
raise UnknownPipelineError
|
|
63
79
|
|
|
64
80
|
if not p_success:
|
|
65
81
|
success = False
|
|
66
|
-
self.logger.error(
|
|
67
|
-
f"[{self.definition.pipeline_name}] -> Pipeline failed: {pipeline_name}"
|
|
68
|
-
)
|
|
82
|
+
self.logger.error(f"Pipeline failed: {pipeline_name}")
|
|
69
83
|
if self.definition.fail_fast:
|
|
70
|
-
self.logger.error(
|
|
71
|
-
f"[{self.definition.pipeline_name}] -> Fail fast enabled. Stopping orchestrator."
|
|
72
|
-
)
|
|
84
|
+
self.logger.error("Fail fast enabled. Stopping orchestrator.")
|
|
73
85
|
break
|
|
74
86
|
else:
|
|
75
87
|
self.logger.info(
|
|
@@ -14,8 +14,12 @@ from .models import (
|
|
|
14
14
|
FileBasedConnectionSettings,
|
|
15
15
|
FileType,
|
|
16
16
|
)
|
|
17
|
-
from .custom_exceptions import
|
|
18
|
-
|
|
17
|
+
from .custom_exceptions import (
|
|
18
|
+
PipelineValidationError,
|
|
19
|
+
EmptyDataError,
|
|
20
|
+
NoValidDestination,
|
|
21
|
+
)
|
|
22
|
+
from .pipeline_base import BasePipeline, PipelineResult
|
|
19
23
|
|
|
20
24
|
|
|
21
25
|
class LoadPipeline(BasePipeline):
|
|
@@ -25,13 +29,13 @@ class LoadPipeline(BasePipeline):
|
|
|
25
29
|
Inherits shared logic from BasePipeline.
|
|
26
30
|
"""
|
|
27
31
|
|
|
28
|
-
def __init__(self,
|
|
32
|
+
def __init__(self, file_name: str, orchestrator_id: Optional[str] = None):
|
|
29
33
|
|
|
30
34
|
# Load definition from config
|
|
31
|
-
definition = Configuration().get_pipeline(
|
|
35
|
+
definition = Configuration().get_pipeline(file_name)
|
|
32
36
|
if not isinstance(definition, BasePipelineDefinition):
|
|
33
37
|
raise ValueError(
|
|
34
|
-
f"Pipeline '{
|
|
38
|
+
f"Pipeline '{file_name}' does not contain a LoadPipeline definition."
|
|
35
39
|
)
|
|
36
40
|
|
|
37
41
|
super().__init__(definition, orchestrator_id=orchestrator_id)
|
|
@@ -44,7 +48,18 @@ class LoadPipeline(BasePipeline):
|
|
|
44
48
|
self.dst_file_ops: Optional[FileOperations] = None
|
|
45
49
|
self.source_file_path: Optional[Path] = None # For auditing
|
|
46
50
|
|
|
47
|
-
self.
|
|
51
|
+
if self._init_error is None:
|
|
52
|
+
try:
|
|
53
|
+
self._initialize_components()
|
|
54
|
+
except Exception as e:
|
|
55
|
+
# deferred so run() can report it as a PipelineResult instead of
|
|
56
|
+
# raising out of the constructor
|
|
57
|
+
self._init_error = PipelineResult(
|
|
58
|
+
success=False,
|
|
59
|
+
message=f"Pipeline setup failed: {e}",
|
|
60
|
+
error_code=e.__class__.__name__,
|
|
61
|
+
exception=e,
|
|
62
|
+
)
|
|
48
63
|
|
|
49
64
|
def _initialize_components(self):
|
|
50
65
|
"""Dynamically initialize the components based on their type and definition"""
|
|
@@ -86,40 +101,37 @@ class LoadPipeline(BasePipeline):
|
|
|
86
101
|
):
|
|
87
102
|
self.dst_db_ops = self._setup_db_operations(destination_resource)
|
|
88
103
|
self.destination_name = destination_resource.conn_database
|
|
104
|
+
else:
|
|
105
|
+
raise ValueError(
|
|
106
|
+
f"Unsupported destination resource type: {type(source_resource)}"
|
|
107
|
+
)
|
|
89
108
|
|
|
90
|
-
def
|
|
109
|
+
def _run(self):
|
|
91
110
|
"""Executes the entire ETL flow"""
|
|
92
|
-
self.logger.info(
|
|
93
|
-
f">>> Starting Load Pipeline: {self.definition.pipeline_name} <<<"
|
|
94
|
-
)
|
|
95
111
|
|
|
96
112
|
try:
|
|
97
|
-
# 1.
|
|
113
|
+
# 1. READ
|
|
98
114
|
df, inferred_dtypes = self._extract_step()
|
|
99
115
|
if df.empty:
|
|
100
|
-
self.
|
|
101
|
-
|
|
116
|
+
self.error_details = "No data found when trying to read from source!"
|
|
117
|
+
self._log_audit("FAILED")
|
|
118
|
+
raise EmptyDataError(self.error_details)
|
|
102
119
|
|
|
103
120
|
self.input_rows = len(df)
|
|
121
|
+
self.logger.info(f"Rows read from source: {self.input_rows}")
|
|
104
122
|
|
|
105
123
|
# 2. TRANSFORM & VALIDATE
|
|
106
124
|
df, invalid_df = self._transform_step(df)
|
|
107
125
|
self.output_rows = len(df)
|
|
108
126
|
if df.empty:
|
|
109
|
-
self.
|
|
110
|
-
"
|
|
127
|
+
self.error_details = (
|
|
128
|
+
"After validation no data remaining - all data is invalid."
|
|
111
129
|
)
|
|
112
|
-
self._log_audit("
|
|
113
|
-
|
|
130
|
+
self._log_audit("FAILED")
|
|
131
|
+
raise EmptyDataError(self.error_details)
|
|
114
132
|
|
|
115
133
|
# 3. LOAD
|
|
116
|
-
|
|
117
|
-
if not load_success:
|
|
118
|
-
self.logger.error(
|
|
119
|
-
f">>> Pipeline {self.definition.pipeline_name} failed to load during the LOAD step."
|
|
120
|
-
)
|
|
121
|
-
self._log_audit("FAILED")
|
|
122
|
-
return False
|
|
134
|
+
self._load_step(df, invalid_df, inferred_dtypes)
|
|
123
135
|
|
|
124
136
|
self._log_audit("SUCCESS")
|
|
125
137
|
self.logger.info(
|
|
@@ -133,14 +145,14 @@ class LoadPipeline(BasePipeline):
|
|
|
133
145
|
f"Critical pipeline error - {self.definition.pipeline_name}: {str(e)}"
|
|
134
146
|
)
|
|
135
147
|
self._log_audit("FAILED")
|
|
136
|
-
|
|
148
|
+
raise
|
|
137
149
|
except Exception as e:
|
|
138
150
|
self.error_details = str(e)
|
|
139
151
|
self.log_exception(
|
|
140
152
|
e, f"Critical pipeline error - {self.definition.pipeline_name}"
|
|
141
153
|
)
|
|
142
154
|
self._log_audit("FAILED")
|
|
143
|
-
|
|
155
|
+
raise
|
|
144
156
|
finally:
|
|
145
157
|
self._cleanup()
|
|
146
158
|
|
|
@@ -156,15 +168,23 @@ class LoadPipeline(BasePipeline):
|
|
|
156
168
|
try:
|
|
157
169
|
if self.src_db_ops: # DB source
|
|
158
170
|
if self.definition.source_sql:
|
|
171
|
+
self.logger.debug("<< reading data from database >>")
|
|
159
172
|
df = self.src_db_ops.read_data(
|
|
160
173
|
self.definition.source_sql, **self.definition.read_parameters
|
|
161
174
|
)
|
|
162
|
-
|
|
175
|
+
dtype_map = self.definition.get_dtype_map()
|
|
176
|
+
self.logger.debug("<< identifying columns definition >>")
|
|
177
|
+
if not dtype_map:
|
|
178
|
+
self.logger.debug("<< no columns definition identified >>")
|
|
179
|
+
dtype_map = {}
|
|
180
|
+
return df, dtype_map
|
|
163
181
|
|
|
164
182
|
if self.src_file_ops: # File source
|
|
183
|
+
self.logger.debug("<< source is a file, atempting pre-processing >>")
|
|
165
184
|
self.src_file_ops._apply_file_preprocessor(
|
|
166
185
|
self.definition.file_pre_process
|
|
167
186
|
)
|
|
187
|
+
self.logger.debug("<< reading from file source >>")
|
|
168
188
|
df = self.src_file_ops.read_file(**self.definition.read_parameters)
|
|
169
189
|
self.source_file_path = self.src_file_ops.file_path
|
|
170
190
|
|
|
@@ -212,7 +232,8 @@ class LoadPipeline(BasePipeline):
|
|
|
212
232
|
rename_map = self.definition.get_rename_map()
|
|
213
233
|
if rename_map:
|
|
214
234
|
df.rename(columns=rename_map, inplace=True)
|
|
215
|
-
self.logger.info(
|
|
235
|
+
self.logger.info("Columns renamed")
|
|
236
|
+
self.logger.debug(f"Columns renamed: {rename_map}")
|
|
216
237
|
|
|
217
238
|
# 2. Pipeline hook transformation
|
|
218
239
|
df = self.definition.transform(df)
|
|
@@ -307,10 +328,13 @@ class LoadPipeline(BasePipeline):
|
|
|
307
328
|
df: pd.DataFrame,
|
|
308
329
|
invalid: pd.DataFrame,
|
|
309
330
|
dtype_map: Dict[str, types.TypeEngine],
|
|
310
|
-
) ->
|
|
331
|
+
) -> None:
|
|
311
332
|
"""Handles loading logic based on destination type."""
|
|
312
333
|
try:
|
|
313
334
|
if self.dst_db_ops and self.definition.destination_table: # DB destination
|
|
335
|
+
self.logger.info(
|
|
336
|
+
f"Writting to table {self.definition.destination_table}"
|
|
337
|
+
)
|
|
314
338
|
self.dst_db_ops.write_to_table(
|
|
315
339
|
table_name=self.definition.destination_table,
|
|
316
340
|
df=df,
|
|
@@ -318,11 +342,12 @@ class LoadPipeline(BasePipeline):
|
|
|
318
342
|
**self.definition.write_parameters,
|
|
319
343
|
)
|
|
320
344
|
if not invalid.empty:
|
|
345
|
+
self.logger.info("Invalid data found - writting to error log")
|
|
321
346
|
self.dst_db_ops.write_to_table(
|
|
322
347
|
table_name=self.definition.destination_table_invalid,
|
|
323
348
|
df=invalid,
|
|
324
349
|
index=False,
|
|
325
|
-
if_exists="
|
|
350
|
+
if_exists="append",
|
|
326
351
|
)
|
|
327
352
|
elif self.dst_file_ops: # File destination
|
|
328
353
|
valid_path, invalid_path = self.dst_file_ops._construct_output_path()
|
|
@@ -340,18 +365,16 @@ class LoadPipeline(BasePipeline):
|
|
|
340
365
|
invalid, invalid_path, **self.definition.write_parameters
|
|
341
366
|
)
|
|
342
367
|
else:
|
|
343
|
-
|
|
368
|
+
raise NoValidDestination
|
|
344
369
|
|
|
345
370
|
# file_post_process always runs against the source file, not the destination
|
|
346
371
|
if self.src_file_ops and self.source_file_path:
|
|
372
|
+
self.logger.debug("Applying file post-processing.")
|
|
347
373
|
self.source_file_path = self.definition.file_post_process(
|
|
348
374
|
self.source_file_path
|
|
349
375
|
)
|
|
350
376
|
if self.definition.source_file_delete_after_load == "yes":
|
|
351
377
|
self.source_file_path.unlink()
|
|
352
|
-
self.logger.
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
except Exception as e:
|
|
356
|
-
self.log_exception(e, "Error writing to destination")
|
|
357
|
-
return False
|
|
378
|
+
self.logger.debug(f"Deleted source file: {self.source_file_path}")
|
|
379
|
+
except Exception:
|
|
380
|
+
raise
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
from abc import ABC, abstractmethod
|
|
2
2
|
from typing import List, Optional, Union
|
|
3
3
|
from uuid import uuid4
|
|
4
|
+
from dataclasses import dataclass
|
|
4
5
|
|
|
5
6
|
import pandas as pd
|
|
6
7
|
|
|
@@ -18,6 +19,20 @@ from .models import (
|
|
|
18
19
|
)
|
|
19
20
|
|
|
20
21
|
|
|
22
|
+
@dataclass
|
|
23
|
+
class PipelineResult:
|
|
24
|
+
"""A definition of a pipeline execution outcome with some details"""
|
|
25
|
+
|
|
26
|
+
success: bool
|
|
27
|
+
message: str
|
|
28
|
+
error_code: Optional[str] = None
|
|
29
|
+
exception: Optional[Exception] = None
|
|
30
|
+
|
|
31
|
+
def __bool__(self):
|
|
32
|
+
"""Facilitate checking the result directly in if statements: if result: ..."""
|
|
33
|
+
return self.success
|
|
34
|
+
|
|
35
|
+
|
|
21
36
|
class BasePipeline(LoggedComponent, ABC):
|
|
22
37
|
"""
|
|
23
38
|
Abstract base class for all pipeline types.
|
|
@@ -40,6 +55,7 @@ class BasePipeline(LoggedComponent, ABC):
|
|
|
40
55
|
|
|
41
56
|
self._active_connectors: List[DatabaseConnector] = []
|
|
42
57
|
self.audit_db_ops: Optional[DatabaseOperations] = None
|
|
58
|
+
self._init_error: Optional[PipelineResult] = None
|
|
43
59
|
|
|
44
60
|
# Metadata for auditing
|
|
45
61
|
self.input_rows = 0
|
|
@@ -49,7 +65,17 @@ class BasePipeline(LoggedComponent, ABC):
|
|
|
49
65
|
self.file_size_bytes: Optional[int] = None
|
|
50
66
|
self.file_last_modified: Optional[pd.Timestamp] = None
|
|
51
67
|
|
|
52
|
-
|
|
68
|
+
try:
|
|
69
|
+
self._initialize_audit_resource()
|
|
70
|
+
except Exception as e:
|
|
71
|
+
# deferred so run() can report it as a PipelineResult instead of
|
|
72
|
+
# raising out of the constructor
|
|
73
|
+
self._init_error = PipelineResult(
|
|
74
|
+
success=False,
|
|
75
|
+
message=f"Pipeline setup failed: {e}",
|
|
76
|
+
error_code=e.__class__.__name__,
|
|
77
|
+
exception=e,
|
|
78
|
+
)
|
|
53
79
|
|
|
54
80
|
def _initialize_audit_resource(self):
|
|
55
81
|
"""Initialize the database operations for auditing."""
|
|
@@ -92,9 +118,14 @@ class BasePipeline(LoggedComponent, ABC):
|
|
|
92
118
|
return DatabaseOperations(connector.get_engine())
|
|
93
119
|
|
|
94
120
|
def _cleanup(self):
|
|
95
|
-
"""Dispose of all active database connectors.
|
|
121
|
+
"""Dispose of all active database connectors.
|
|
122
|
+
Never raises, so a disposal failure can't mask the pipeline's real outcome.
|
|
123
|
+
"""
|
|
96
124
|
for connector in self._active_connectors:
|
|
97
|
-
|
|
125
|
+
try:
|
|
126
|
+
connector._dispose_engine()
|
|
127
|
+
except Exception as e:
|
|
128
|
+
self.logger.warning(f"Failed to dispose connector cleanly: {e}")
|
|
98
129
|
self.logger.debug("Pipeline cleanup completed.")
|
|
99
130
|
|
|
100
131
|
def _log_audit(self, status: str):
|
|
@@ -118,7 +149,7 @@ class BasePipeline(LoggedComponent, ABC):
|
|
|
118
149
|
destination_name=getattr(self, "destination_name", None),
|
|
119
150
|
sp_name=getattr(self, "sp_name", None),
|
|
120
151
|
sp_parameters=getattr(self, "sp_parameters", None),
|
|
121
|
-
|
|
152
|
+
insert_timestamp=pd.Timestamp.now(),
|
|
122
153
|
error_details=self.error_details,
|
|
123
154
|
file_name=source_file_path.name if source_file_path else None,
|
|
124
155
|
file_path=str(source_file_path) if source_file_path else None,
|
|
@@ -127,11 +158,49 @@ class BasePipeline(LoggedComponent, ABC):
|
|
|
127
158
|
)
|
|
128
159
|
|
|
129
160
|
try:
|
|
130
|
-
self.audit_db_ops.write_audit("
|
|
161
|
+
self.audit_db_ops.write_audit("edl_execution_audit", entry)
|
|
131
162
|
except Exception as e:
|
|
132
163
|
self.logger.error(f"Failed to write audit log: {str(e)}")
|
|
133
164
|
|
|
165
|
+
def run(self) -> PipelineResult:
|
|
166
|
+
"""
|
|
167
|
+
Public entry point for pipeline execution.
|
|
168
|
+
Wraps internal execution and returns a PipelineResult object.
|
|
169
|
+
"""
|
|
170
|
+
if self._init_error is not None:
|
|
171
|
+
self.logger.warning(f"Pipeline has [FAILED] {self._init_error.message}")
|
|
172
|
+
return self._init_error
|
|
173
|
+
|
|
174
|
+
self.logger.info("Starting pipeline.")
|
|
175
|
+
|
|
176
|
+
try:
|
|
177
|
+
"""Call the method implemented by child classes"""
|
|
178
|
+
self._run()
|
|
179
|
+
self.logger.info("Pipeline has finalized with [SUCCESS].")
|
|
180
|
+
|
|
181
|
+
return PipelineResult(
|
|
182
|
+
success=True,
|
|
183
|
+
message="Pipeline executed succesfully.",
|
|
184
|
+
)
|
|
185
|
+
except Exception as e:
|
|
186
|
+
# file log
|
|
187
|
+
self.logger.debug(
|
|
188
|
+
f"Pipeline has encountered an exception: {e}",
|
|
189
|
+
exc_info=True,
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
user_msg = getattr(e, "user_message", str(e))
|
|
193
|
+
# console log
|
|
194
|
+
self.logger.warning(f"Pipeline has [FAILED] {user_msg}")
|
|
195
|
+
|
|
196
|
+
return PipelineResult(
|
|
197
|
+
success=False,
|
|
198
|
+
message="Pipeline has failed",
|
|
199
|
+
error_code=e.__class__.__name__,
|
|
200
|
+
exception=e,
|
|
201
|
+
)
|
|
202
|
+
|
|
134
203
|
@abstractmethod
|
|
135
|
-
def
|
|
204
|
+
def _run(self) -> bool:
|
|
136
205
|
"""Main execution logic to be implemented by child classes."""
|
|
137
206
|
pass
|
{easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/procedure_pipeline.py
RENAMED
|
@@ -6,7 +6,7 @@ from .models import (
|
|
|
6
6
|
ServerBasedConnectionSettings,
|
|
7
7
|
FileBasedConnectionSettings,
|
|
8
8
|
)
|
|
9
|
-
from .pipeline_base import BasePipeline
|
|
9
|
+
from .pipeline_base import BasePipeline, PipelineResult
|
|
10
10
|
|
|
11
11
|
|
|
12
12
|
class ProcedurePipeline(BasePipeline):
|
|
@@ -14,30 +14,44 @@ class ProcedurePipeline(BasePipeline):
|
|
|
14
14
|
Pipeline specialized in executing database stored procedures.
|
|
15
15
|
"""
|
|
16
16
|
|
|
17
|
-
def __init__(self,
|
|
17
|
+
def __init__(self, file_name: str, orchestrator_id: Optional[str] = None):
|
|
18
18
|
# Load definition from config
|
|
19
|
-
definition = Configuration().get_pipeline(
|
|
19
|
+
definition = Configuration().get_pipeline(file_name)
|
|
20
20
|
if not isinstance(definition, ProcedureDefinition):
|
|
21
21
|
raise ValueError(
|
|
22
|
-
f"Pipeline '{
|
|
22
|
+
f"Pipeline '{file_name}' does not contain a ProcedureDefinition."
|
|
23
23
|
)
|
|
24
24
|
|
|
25
25
|
super().__init__(definition, orchestrator_id=orchestrator_id)
|
|
26
26
|
self.definition: ProcedureDefinition = definition
|
|
27
27
|
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
28
|
+
if self._init_error is None:
|
|
29
|
+
try:
|
|
30
|
+
# Set up primary database connection
|
|
31
|
+
resource = self.config.get_resource(self.definition.resource)
|
|
32
|
+
if not isinstance(
|
|
33
|
+
resource,
|
|
34
|
+
(ServerBasedConnectionSettings, FileBasedConnectionSettings),
|
|
35
|
+
):
|
|
36
|
+
raise ValueError(
|
|
37
|
+
f"Resource '{self.definition.resource}' must be a database connection."
|
|
38
|
+
)
|
|
37
39
|
|
|
38
|
-
|
|
40
|
+
self.db_ops = self._setup_db_operations(resource)
|
|
41
|
+
except Exception as e:
|
|
42
|
+
self.error_details = str(e)
|
|
43
|
+
self._log_audit("FAILED")
|
|
44
|
+
self.logger.error(self.error_details)
|
|
45
|
+
# deferred so run() can report it as a PipelineResult instead of
|
|
46
|
+
# raising out of the constructor
|
|
47
|
+
self._init_error = PipelineResult(
|
|
48
|
+
success=False,
|
|
49
|
+
message=f"Pipeline setup failed: {e}",
|
|
50
|
+
error_code=e.__class__.__name__,
|
|
51
|
+
exception=e,
|
|
52
|
+
)
|
|
39
53
|
|
|
40
|
-
def
|
|
54
|
+
def _run(self):
|
|
41
55
|
"""Execute all procedures in the definition."""
|
|
42
56
|
self.logger.info(
|
|
43
57
|
f">>> Starting Procedure Pipeline: {self.definition.pipeline_name} <<<"
|
|
@@ -65,7 +79,7 @@ class ProcedurePipeline(BasePipeline):
|
|
|
65
79
|
self.logger.info(
|
|
66
80
|
f">>> Procedure Pipeline {self.definition.pipeline_name} finished successfully <<<"
|
|
67
81
|
)
|
|
68
|
-
return
|
|
82
|
+
return
|
|
69
83
|
|
|
70
84
|
except Exception as e:
|
|
71
85
|
self.error_details = str(e)
|
|
@@ -78,6 +92,6 @@ class ProcedurePipeline(BasePipeline):
|
|
|
78
92
|
self.sp_parameters = {p[0]: p[1] for p in self.definition.procedures}
|
|
79
93
|
|
|
80
94
|
self._log_audit("FAILED")
|
|
81
|
-
|
|
95
|
+
raise
|
|
82
96
|
finally:
|
|
83
97
|
self._cleanup()
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import os
|
|
2
|
+
import re
|
|
2
3
|
import sqlite3
|
|
3
4
|
|
|
4
5
|
import pytest
|
|
@@ -138,7 +139,9 @@ def test_validate_resources_reports_ok_and_failed(runner, tmp_path):
|
|
|
138
139
|
result = runner.invoke(main, ["validate-resources"])
|
|
139
140
|
assert result.exit_code == 0
|
|
140
141
|
assert "Resource: good_folder ... OK (Path Exists)" in result.output
|
|
141
|
-
|
|
142
|
+
# SQLAlchemy's echo logging can interleave stderr noise between the label
|
|
143
|
+
# and status, so match across lines instead of requiring adjacency.
|
|
144
|
+
assert re.search(r"Resource: good_db \.\.\.[\s\S]*?OK \(Connected\)", result.output)
|
|
142
145
|
assert "Resource: bad_db ... FAILED" in result.output
|
|
143
146
|
|
|
144
147
|
|
|
@@ -179,9 +182,11 @@ def test_validate_pipelines_reports_ok_for_load_procedure_and_orchestrator(
|
|
|
179
182
|
|
|
180
183
|
result = runner.invoke(main, ["validate-pipelines"])
|
|
181
184
|
assert result.exit_code == 0
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
assert "Pipeline:
|
|
185
|
+
# SQLAlchemy's echo logging can interleave stderr noise between the label
|
|
186
|
+
# and status, so match across lines instead of requiring adjacency.
|
|
187
|
+
assert re.search(r"Pipeline: load_pipeline \.\.\.[\s\S]*?OK\b", result.output)
|
|
188
|
+
assert re.search(r"Pipeline: procedure_pipeline \.\.\.[\s\S]*?OK\b", result.output)
|
|
189
|
+
assert re.search(r"Pipeline: my_orchestrator \.\.\.[\s\S]*?OK\b", result.output)
|
|
185
190
|
|
|
186
191
|
|
|
187
192
|
def test_validate_pipelines_reports_failed_for_unresolvable_resource(runner, tmp_path):
|
|
@@ -193,4 +198,4 @@ def test_validate_pipelines_reports_failed_for_unresolvable_resource(runner, tmp
|
|
|
193
198
|
|
|
194
199
|
result = runner.invoke(main, ["validate-pipelines"])
|
|
195
200
|
assert result.exit_code == 0
|
|
196
|
-
assert "Pipeline: broken_pipeline
|
|
201
|
+
assert re.search(r"Pipeline: broken_pipeline \.\.\.[\s\S]*?FAILED", result.output)
|
|
@@ -24,7 +24,7 @@ def test_connector_factory_maps_expected_connector_classes():
|
|
|
24
24
|
|
|
25
25
|
|
|
26
26
|
def test_sqlite_connector_builds_expected_connection_string():
|
|
27
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
27
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
28
28
|
db_path = f"{tmpdir}/test.db"
|
|
29
29
|
settings = FileBasedConnectionSettings(
|
|
30
30
|
conn_server_type=ServerType.SQLITE, file_path=db_path
|
|
@@ -25,7 +25,7 @@ def reset_configuration_singleton():
|
|
|
25
25
|
|
|
26
26
|
def test_orchestrator_passes_orchestrator_id_and_writes_audit():
|
|
27
27
|
# Create temporary file paths for our sqlite databases
|
|
28
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
28
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
29
29
|
src_db_path = os.path.join(tmpdir, "src.db")
|
|
30
30
|
dst_db_path = os.path.join(tmpdir, "dst.db")
|
|
31
31
|
audit_db_path = os.path.join(tmpdir, "audit.db")
|
|
@@ -93,7 +93,7 @@ def test_orchestrator_passes_orchestrator_id_and_writes_audit():
|
|
|
93
93
|
assert orchestrator.orchestrator_id is not None
|
|
94
94
|
|
|
95
95
|
success = orchestrator.run()
|
|
96
|
-
assert success is True
|
|
96
|
+
assert success.success is True
|
|
97
97
|
|
|
98
98
|
# Verify target tables were created and populated in destination
|
|
99
99
|
conn_dst = sqlite3.connect(dst_db_path)
|
|
@@ -111,9 +111,9 @@ def test_orchestrator_passes_orchestrator_id_and_writes_audit():
|
|
|
111
111
|
conn_audit = sqlite3.connect(audit_db_path)
|
|
112
112
|
cursor_audit = conn_audit.cursor()
|
|
113
113
|
|
|
114
|
-
# Read the
|
|
114
|
+
# Read the edl_execution_audit table columns and rows
|
|
115
115
|
cursor_audit.execute(
|
|
116
|
-
"SELECT pipeline_id, orchestrator_id, pipeline_name, status FROM
|
|
116
|
+
"SELECT pipeline_id, orchestrator_id, pipeline_name, status FROM edl_execution_audit"
|
|
117
117
|
)
|
|
118
118
|
rows = cursor_audit.fetchall()
|
|
119
119
|
assert len(rows) == 2
|
|
@@ -31,7 +31,7 @@ def _transform_requires_renamed_column(df):
|
|
|
31
31
|
|
|
32
32
|
|
|
33
33
|
def test_transform_hook_sees_renamed_columns():
|
|
34
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
34
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
35
35
|
src_db_path = os.path.join(tmpdir, "src.db")
|
|
36
36
|
dst_db_path = os.path.join(tmpdir, "dst.db")
|
|
37
37
|
audit_db_path = os.path.join(tmpdir, "audit.db")
|
|
@@ -72,7 +72,7 @@ def test_transform_hook_sees_renamed_columns():
|
|
|
72
72
|
pipeline = LoadPipeline("test_pipeline")
|
|
73
73
|
success = pipeline.run()
|
|
74
74
|
|
|
75
|
-
assert success is True
|
|
75
|
+
assert success.success is True
|
|
76
76
|
|
|
77
77
|
conn_dst = sqlite3.connect(dst_db_path)
|
|
78
78
|
cursor_dst = conn_dst.cursor()
|
|
@@ -60,7 +60,7 @@ def test_execution_mode_defaults_to_sequential():
|
|
|
60
60
|
|
|
61
61
|
|
|
62
62
|
def test_sequential_mode_runs_all_procedures_when_each_succeeds():
|
|
63
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
63
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
64
64
|
pipeline = _build_pipeline(tmpdir, execution_mode="sequential")
|
|
65
65
|
|
|
66
66
|
recorded_calls = []
|
|
@@ -72,12 +72,12 @@ def test_sequential_mode_runs_all_procedures_when_each_succeeds():
|
|
|
72
72
|
pipeline.db_ops.execute_stored_procedure = _record_call # type: ignore[method-assign]
|
|
73
73
|
|
|
74
74
|
success = pipeline.run()
|
|
75
|
-
assert success is True
|
|
75
|
+
assert success.success is True
|
|
76
76
|
assert [c[0] for c in recorded_calls] == ["proc_a", "proc_b"]
|
|
77
77
|
|
|
78
78
|
|
|
79
79
|
def test_loop_stops_after_first_failed_procedure():
|
|
80
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
80
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
81
81
|
pipeline = _build_pipeline(tmpdir)
|
|
82
82
|
|
|
83
83
|
recorded_calls = []
|
|
@@ -29,7 +29,7 @@ def reset_configuration_singleton():
|
|
|
29
29
|
|
|
30
30
|
|
|
31
31
|
def test_read_data_returns_dataframe_for_multiple_rows():
|
|
32
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
32
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
33
33
|
db_path = os.path.join(tmpdir, "db.sqlite")
|
|
34
34
|
conn = sqlite3.connect(db_path)
|
|
35
35
|
conn.execute("CREATE TABLE t (id INTEGER)")
|
|
@@ -48,7 +48,7 @@ def test_read_data_returns_dataframe_for_multiple_rows():
|
|
|
48
48
|
|
|
49
49
|
|
|
50
50
|
def test_read_data_returns_dict_for_single_row():
|
|
51
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
51
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
52
52
|
db_path = os.path.join(tmpdir, "db.sqlite")
|
|
53
53
|
conn = sqlite3.connect(db_path)
|
|
54
54
|
conn.execute("CREATE TABLE t (id INTEGER, name TEXT)")
|
|
@@ -66,7 +66,7 @@ def test_read_data_returns_dict_for_single_row():
|
|
|
66
66
|
|
|
67
67
|
|
|
68
68
|
def test_read_data_returns_empty_dataframe_for_no_rows():
|
|
69
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
69
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
70
70
|
db_path = os.path.join(tmpdir, "db.sqlite")
|
|
71
71
|
conn = sqlite3.connect(db_path)
|
|
72
72
|
conn.execute("CREATE TABLE t (id INTEGER)")
|
|
@@ -138,7 +138,7 @@ def test_procedure_definition_default_build_parameters_is_empty():
|
|
|
138
138
|
|
|
139
139
|
|
|
140
140
|
def test_procedure_pipeline_merges_dynamic_parameters():
|
|
141
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
141
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
142
142
|
db_path = os.path.join(tmpdir, "db.sqlite")
|
|
143
143
|
|
|
144
144
|
config = Configuration()
|
|
@@ -173,7 +173,7 @@ def test_procedure_pipeline_merges_dynamic_parameters():
|
|
|
173
173
|
pipeline.db_ops.execute_stored_procedure = _record_call # type: ignore[method-assign]
|
|
174
174
|
|
|
175
175
|
success = pipeline.run()
|
|
176
|
-
assert success is True
|
|
176
|
+
assert success.success is True
|
|
177
177
|
assert recorded_calls == [
|
|
178
178
|
("insert_call", {"name": "static_name", "value": "dynamic_value"})
|
|
179
179
|
]
|
|
@@ -27,7 +27,7 @@ def reset_configuration_singleton():
|
|
|
27
27
|
|
|
28
28
|
|
|
29
29
|
def test_source_file_deleted_after_successful_load_to_db():
|
|
30
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
30
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
31
31
|
(Path(tmpdir) / "data.csv").write_text("a\n1\n2\n")
|
|
32
32
|
dst_db_path = os.path.join(tmpdir, "dst.db")
|
|
33
33
|
|
|
@@ -52,12 +52,12 @@ def test_source_file_deleted_after_successful_load_to_db():
|
|
|
52
52
|
pipeline = LoadPipeline("test_pipeline")
|
|
53
53
|
success = pipeline.run()
|
|
54
54
|
|
|
55
|
-
assert success is True
|
|
55
|
+
assert success.success is True
|
|
56
56
|
assert not (Path(tmpdir) / "data.csv").exists()
|
|
57
57
|
|
|
58
58
|
|
|
59
59
|
def test_file_post_process_runs_on_source_when_destination_is_db():
|
|
60
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
60
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
61
61
|
(Path(tmpdir) / "data.csv").write_text("a\n1\n2\n")
|
|
62
62
|
dst_db_path = os.path.join(tmpdir, "dst.db")
|
|
63
63
|
|
|
@@ -86,12 +86,12 @@ def test_file_post_process_runs_on_source_when_destination_is_db():
|
|
|
86
86
|
pipeline = LoadPipeline("test_pipeline")
|
|
87
87
|
success = pipeline.run()
|
|
88
88
|
|
|
89
|
-
assert success is True
|
|
89
|
+
assert success.success is True
|
|
90
90
|
assert post_processed_paths == [Path(tmpdir) / "data.csv"]
|
|
91
91
|
|
|
92
92
|
|
|
93
93
|
def test_source_file_delete_after_load_raises_when_source_is_not_a_file():
|
|
94
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
94
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
95
95
|
src_db_path = os.path.join(tmpdir, "src.db")
|
|
96
96
|
conn = sqlite3.connect(src_db_path)
|
|
97
97
|
conn.execute("CREATE TABLE source_table (a INTEGER)")
|
|
@@ -118,5 +118,10 @@ def test_source_file_delete_after_load_raises_when_source_is_not_a_file():
|
|
|
118
118
|
)
|
|
119
119
|
config.pipelines["test_pipeline"] = pipeline_def
|
|
120
120
|
|
|
121
|
-
|
|
122
|
-
|
|
121
|
+
# Setup failures are deferred to run() rather than raised from the
|
|
122
|
+
# constructor, so they can be reported as a PipelineResult.
|
|
123
|
+
pipeline = LoadPipeline("test_pipeline")
|
|
124
|
+
result = pipeline.run()
|
|
125
|
+
|
|
126
|
+
assert result.success is False
|
|
127
|
+
assert "source_file_delete_after_load" in result.message
|
|
@@ -34,7 +34,7 @@ def reset_configuration_singleton():
|
|
|
34
34
|
|
|
35
35
|
|
|
36
36
|
def test_validation_fail_false_keeps_valid_rows():
|
|
37
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
37
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
38
38
|
src_db_path = os.path.join(tmpdir, "src.db")
|
|
39
39
|
dst_db_path = os.path.join(tmpdir, "dst.db")
|
|
40
40
|
audit_db_path = os.path.join(tmpdir, "audit.db")
|
|
@@ -87,7 +87,7 @@ def test_validation_fail_false_keeps_valid_rows():
|
|
|
87
87
|
success = pipeline.run()
|
|
88
88
|
|
|
89
89
|
# Should be successful because validation_fail=False
|
|
90
|
-
assert success is True
|
|
90
|
+
assert success.success is True
|
|
91
91
|
|
|
92
92
|
# Verify destination table only has valid rows (1 and 3)
|
|
93
93
|
conn_dst = sqlite3.connect(dst_db_path)
|
|
@@ -101,7 +101,7 @@ def test_validation_fail_false_keeps_valid_rows():
|
|
|
101
101
|
assert rows[1] == (3, "Charlie")
|
|
102
102
|
|
|
103
103
|
cursor_dst = sqlite3.connect(dst_db_path).cursor()
|
|
104
|
-
cursor_dst.execute("SELECT error FROM
|
|
104
|
+
cursor_dst.execute("SELECT error FROM edl_validation_error_log")
|
|
105
105
|
invalid_rows = cursor_dst.fetchall()
|
|
106
106
|
cursor_dst.connection.close()
|
|
107
107
|
|
|
@@ -111,7 +111,7 @@ def test_validation_fail_false_keeps_valid_rows():
|
|
|
111
111
|
|
|
112
112
|
|
|
113
113
|
def test_validation_writes_valid_and_invalid_file_outputs():
|
|
114
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
114
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
115
115
|
src_db_path = os.path.join(tmpdir, "src.db")
|
|
116
116
|
audit_db_path = os.path.join(tmpdir, "audit.db")
|
|
117
117
|
|
|
@@ -146,7 +146,7 @@ def test_validation_writes_valid_and_invalid_file_outputs():
|
|
|
146
146
|
|
|
147
147
|
success = LoadPipeline("test_pipeline").run()
|
|
148
148
|
|
|
149
|
-
assert success is True
|
|
149
|
+
assert success.success is True
|
|
150
150
|
|
|
151
151
|
valid_df = pd.read_csv(Path(tmpdir) / "output.csv")
|
|
152
152
|
invalid_df = pd.read_csv(Path(tmpdir) / "output_invalid.csv")
|
|
@@ -157,7 +157,7 @@ def test_validation_writes_valid_and_invalid_file_outputs():
|
|
|
157
157
|
|
|
158
158
|
|
|
159
159
|
def test_validation_fail_true_fails_pipeline():
|
|
160
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
160
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
161
161
|
src_db_path = os.path.join(tmpdir, "src.db")
|
|
162
162
|
dst_db_path = os.path.join(tmpdir, "dst.db")
|
|
163
163
|
audit_db_path = os.path.join(tmpdir, "audit.db")
|
|
@@ -206,14 +206,14 @@ def test_validation_fail_true_fails_pipeline():
|
|
|
206
206
|
success = pipeline.run()
|
|
207
207
|
|
|
208
208
|
# Should fail because validation_fail=True
|
|
209
|
-
assert success is False
|
|
209
|
+
assert success.success is False
|
|
210
210
|
assert pipeline.error_details is not None
|
|
211
211
|
assert "Validation failed" in pipeline.error_details
|
|
212
212
|
assert "Row 1:" in pipeline.error_details # Row 2 (0-indexed row 1) failed
|
|
213
213
|
|
|
214
214
|
|
|
215
215
|
def test_validation_fail_false_graceful_stop_when_all_rows_invalid():
|
|
216
|
-
with tempfile.TemporaryDirectory() as tmpdir:
|
|
216
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
|
217
217
|
src_db_path = os.path.join(tmpdir, "src.db")
|
|
218
218
|
dst_db_path = os.path.join(tmpdir, "dst.db")
|
|
219
219
|
audit_db_path = os.path.join(tmpdir, "audit.db")
|
|
@@ -261,8 +261,10 @@ def test_validation_fail_false_graceful_stop_when_all_rows_invalid():
|
|
|
261
261
|
pipeline = LoadPipeline("test_pipeline")
|
|
262
262
|
success = pipeline.run()
|
|
263
263
|
|
|
264
|
-
#
|
|
265
|
-
|
|
264
|
+
# validation_fail=False keeps invalid rows out of the load, but if that
|
|
265
|
+
# leaves no data at all the pipeline still fails with an EmptyDataError.
|
|
266
|
+
assert success.success is False
|
|
267
|
+
assert success.error_code == "EmptyDataError"
|
|
266
268
|
assert pipeline.output_rows == 0
|
|
267
269
|
|
|
268
270
|
# Destination table should NOT have been loaded or have any data
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
{easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/entry_points.txt
RENAMED
|
File without changes
|
{easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/requires.txt
RENAMED
|
File without changes
|
{easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/top_level.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|