easy-data-loader 0.1.9__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {easy_data_loader-0.1.9/src/easy_data_loader.egg-info → easy_data_loader-0.2.1}/PKG-INFO +1 -1
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/pyproject.toml +1 -1
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/cli.py +3 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/custom_exceptions.py +24 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/data_inferrence.py +13 -14
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/database_connector.py +3 -3
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/database_operations.py +8 -6
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/file_operations.py +1 -1
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/log.py +16 -12
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/models.py +1 -1
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/orchestrator.py +31 -21
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/pipeline.py +53 -33
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/pipeline_base.py +73 -4
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/procedure_pipeline.py +28 -14
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1/src/easy_data_loader.egg-info}/PKG-INFO +1 -1
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/tests/test_cli.py +10 -5
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/tests/test_orchestrator.py +1 -1
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/tests/test_pipeline_transform_order.py +1 -1
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/tests/test_procedure_execution_mode.py +1 -1
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/tests/test_resource_access.py +1 -1
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/tests/test_source_file_delete.py +9 -4
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/tests/test_validation.py +8 -6
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/LICENSE +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/README.md +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/setup.cfg +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/__init__.py +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/config_loader.py +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/driver_detector.py +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/resource_access.py +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/utils.py +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/SOURCES.txt +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/dependency_links.txt +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/entry_points.txt +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/requires.txt +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/top_level.txt +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/tests/test_config_loader.py +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/tests/test_data_inference.py +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/tests/test_database_connector.py +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/tests/test_file_operations.py +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/tests/test_imports.py +0 -0
- {easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/tests/test_models.py +0 -0
|
@@ -412,6 +412,9 @@ def validate_pipelines():
|
|
|
412
412
|
else:
|
|
413
413
|
raise ValueError(f"Unknown pipeline type: {type(definition)}")
|
|
414
414
|
|
|
415
|
+
if instance is not None and instance._init_error is not None:
|
|
416
|
+
raise ValueError(instance._init_error.message)
|
|
417
|
+
|
|
415
418
|
results[name] = "OK"
|
|
416
419
|
except Exception as e:
|
|
417
420
|
results[name] = f"FAILED: {str(e)}"
|
|
@@ -26,3 +26,27 @@ class PipelineValidationError(Exception):
|
|
|
26
26
|
def __init__(self, message: str):
|
|
27
27
|
self.message = message
|
|
28
28
|
super().__init__(self.message)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class EmptyDataError(Exception):
|
|
32
|
+
def __init__(self, message: str):
|
|
33
|
+
self.message = message
|
|
34
|
+
super().__init__(self.message)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class NoValidDestination(Exception):
|
|
38
|
+
def __init__(self, message: str = "The pipeline destination is not valid"):
|
|
39
|
+
self.message = message
|
|
40
|
+
super().__init__(self.message)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class NestedPipelinesError(Exception):
|
|
44
|
+
def __init__(self, message: str = "Nested pipelines are not supported"):
|
|
45
|
+
self.message = message
|
|
46
|
+
super().__init__(self.message)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class UnknownPipelineError(Exception):
|
|
50
|
+
def __init__(self, message: str = "Unknown pipeline type"):
|
|
51
|
+
self.message = message
|
|
52
|
+
super().__init__(self.message)
|
|
@@ -88,7 +88,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
88
88
|
self.logger.error(
|
|
89
89
|
f"Failed to infer dtype for column '{col}': {str(e)}. "
|
|
90
90
|
f"Column will use default inference.",
|
|
91
|
-
exc_info=
|
|
91
|
+
exc_info=True,
|
|
92
92
|
)
|
|
93
93
|
# Skip this column - it will use default inference
|
|
94
94
|
continue
|
|
@@ -102,9 +102,8 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
102
102
|
|
|
103
103
|
self.logger.info(f"Dtype inference completed for {len(dtype_dict)} columns")
|
|
104
104
|
|
|
105
|
-
#
|
|
106
|
-
|
|
107
|
-
self._log_inference_summary(dtype_dict)
|
|
105
|
+
# Only visible in the file log, since it's logged at debug level
|
|
106
|
+
self._log_inference_summary(dtype_dict)
|
|
108
107
|
|
|
109
108
|
return dtype_dict
|
|
110
109
|
|
|
@@ -593,7 +592,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
593
592
|
self.logger.error(
|
|
594
593
|
f"Failed to infer dtype for column '{col_name}' from Parquet metadata: {str(e)}. "
|
|
595
594
|
f"Column will use default inference.",
|
|
596
|
-
exc_info=
|
|
595
|
+
exc_info=True,
|
|
597
596
|
)
|
|
598
597
|
continue
|
|
599
598
|
|
|
@@ -616,7 +615,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
616
615
|
self.logger.error(
|
|
617
616
|
f"Failed to analyze string length for column '{col}': {str(e)}. "
|
|
618
617
|
f"Column will use default inference.",
|
|
619
|
-
exc_info=
|
|
618
|
+
exc_info=True,
|
|
620
619
|
)
|
|
621
620
|
# Remove from dict so it uses default
|
|
622
621
|
dtype_dict.pop(col, None)
|
|
@@ -626,7 +625,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
626
625
|
self.logger.error(
|
|
627
626
|
f"Failed to read string columns from Parquet: {str(e)}. "
|
|
628
627
|
f"String columns will use default inference.",
|
|
629
|
-
exc_info=
|
|
628
|
+
exc_info=True,
|
|
630
629
|
)
|
|
631
630
|
# Remove string columns from dict
|
|
632
631
|
for col in string_columns:
|
|
@@ -643,8 +642,8 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
643
642
|
f"Parquet dtype inference completed for {len(dtype_dict)} columns"
|
|
644
643
|
)
|
|
645
644
|
|
|
646
|
-
|
|
647
|
-
|
|
645
|
+
# Only visible in the file log, since it's logged at debug level
|
|
646
|
+
self._log_inference_summary(dtype_dict)
|
|
648
647
|
|
|
649
648
|
return dtype_dict
|
|
650
649
|
|
|
@@ -703,7 +702,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
703
702
|
self.logger.error(
|
|
704
703
|
f"Failed to infer dtype for column '{col_name}' from ORC metadata: {str(e)}. "
|
|
705
704
|
f"Column will use default inference.",
|
|
706
|
-
exc_info=
|
|
705
|
+
exc_info=True,
|
|
707
706
|
)
|
|
708
707
|
continue
|
|
709
708
|
|
|
@@ -726,7 +725,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
726
725
|
self.logger.error(
|
|
727
726
|
f"Failed to analyze string length for column '{col}': {str(e)}. "
|
|
728
727
|
f"Column will use default inference.",
|
|
729
|
-
exc_info=
|
|
728
|
+
exc_info=True,
|
|
730
729
|
)
|
|
731
730
|
dtype_dict.pop(col, None)
|
|
732
731
|
continue
|
|
@@ -735,7 +734,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
735
734
|
self.logger.error(
|
|
736
735
|
f"Failed to read string columns from ORC: {str(e)}. "
|
|
737
736
|
f"String columns will use default inference.",
|
|
738
|
-
exc_info=
|
|
737
|
+
exc_info=True,
|
|
739
738
|
)
|
|
740
739
|
for col in string_columns:
|
|
741
740
|
dtype_dict.pop(col, None)
|
|
@@ -751,8 +750,8 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
|
|
|
751
750
|
f"ORC dtype inference completed for {len(dtype_dict)} columns"
|
|
752
751
|
)
|
|
753
752
|
|
|
754
|
-
|
|
755
|
-
|
|
753
|
+
# Only visible in the file log, since it's logged at debug level
|
|
754
|
+
self._log_inference_summary(dtype_dict)
|
|
756
755
|
|
|
757
756
|
return dtype_dict
|
|
758
757
|
|
{easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/database_connector.py
RENAMED
|
@@ -107,7 +107,7 @@ class SqlServerDatabaseConnector(LoggedComponent, DatabaseConnector):
|
|
|
107
107
|
max_overflow=10,
|
|
108
108
|
pool_timeout=30,
|
|
109
109
|
pool_recycle=3600,
|
|
110
|
-
echo=
|
|
110
|
+
echo=False,
|
|
111
111
|
)
|
|
112
112
|
return engine
|
|
113
113
|
except Exception as e:
|
|
@@ -183,7 +183,7 @@ class SQLiteDatabaseConnector(LoggedComponent, DatabaseConnector):
|
|
|
183
183
|
# SQLite specific engine creation
|
|
184
184
|
engine = create_engine(
|
|
185
185
|
connection_string,
|
|
186
|
-
echo=
|
|
186
|
+
echo=False,
|
|
187
187
|
)
|
|
188
188
|
return engine
|
|
189
189
|
except Exception as e:
|
|
@@ -280,7 +280,7 @@ class PostgresDatabaseConnector(LoggedComponent, DatabaseConnector):
|
|
|
280
280
|
max_overflow=10,
|
|
281
281
|
pool_timeout=30,
|
|
282
282
|
pool_recycle=3600,
|
|
283
|
-
echo=
|
|
283
|
+
echo=False,
|
|
284
284
|
)
|
|
285
285
|
return engine
|
|
286
286
|
except Exception as e:
|
{easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/database_operations.py
RENAMED
|
@@ -19,18 +19,20 @@ class DatabaseOperations(LoggedComponent):
|
|
|
19
19
|
self.engine = engine
|
|
20
20
|
self._inspector = inspect(self.engine)
|
|
21
21
|
|
|
22
|
-
def write_to_table(
|
|
23
|
-
self, table_name: Optional[str], df: DataFrame, **kwargs
|
|
24
|
-
) -> bool:
|
|
22
|
+
def write_to_table(self, table_name: Optional[str], df: DataFrame, **kwargs):
|
|
25
23
|
"""Write a dataframe to a specified table in the database"""
|
|
26
24
|
|
|
27
|
-
self.logger.
|
|
25
|
+
self.logger.debug(f"Writing {len(df)} rows to table: {table_name}")
|
|
28
26
|
try:
|
|
29
27
|
if table_name:
|
|
30
28
|
df.to_sql(table_name, con=self.engine, **kwargs)
|
|
31
|
-
|
|
29
|
+
self.logger.debug("Finalized write to table: {table_name}")
|
|
30
|
+
return
|
|
32
31
|
except Exception as e:
|
|
33
|
-
self.
|
|
32
|
+
self.logger.debug(
|
|
33
|
+
f"Failed to write to table {table_name} caused by the following exception: {e}",
|
|
34
|
+
exc_info=True,
|
|
35
|
+
)
|
|
34
36
|
raise
|
|
35
37
|
|
|
36
38
|
def read_data(self, sql: str, **kwargs) -> DataFrame:
|
|
@@ -134,7 +134,7 @@ class FileOperations(LoggedComponent):
|
|
|
134
134
|
"""
|
|
135
135
|
|
|
136
136
|
current_path = self._find_file()
|
|
137
|
-
self.logger.
|
|
137
|
+
self.logger.debug(f"Current file path is: {current_path}")
|
|
138
138
|
|
|
139
139
|
if preprocessor_func is None:
|
|
140
140
|
self.file_path = current_path
|
|
@@ -21,25 +21,37 @@ class AppLogger:
|
|
|
21
21
|
def _setup_logging(self):
|
|
22
22
|
Path("logs").mkdir(exist_ok=True)
|
|
23
23
|
|
|
24
|
-
|
|
24
|
+
file_formatter = logging.Formatter(
|
|
25
25
|
"%(asctime)s - %(name)s - %(levelname)s - %(message)s",
|
|
26
26
|
datefmt="%Y-%m-%d %H:%M:%S",
|
|
27
27
|
)
|
|
28
28
|
|
|
29
|
+
console_formatter = logging.Formatter(
|
|
30
|
+
"[%(asctime)s]-[%(levelname)s] %(message)s"
|
|
31
|
+
)
|
|
32
|
+
|
|
29
33
|
root_logger = logging.getLogger()
|
|
30
|
-
root_logger.setLevel(
|
|
34
|
+
root_logger.setLevel(
|
|
35
|
+
logging.DEBUG
|
|
36
|
+
) # allow all messages and split them in the handlers
|
|
31
37
|
root_logger.handlers.clear()
|
|
32
38
|
|
|
33
39
|
# Console
|
|
34
40
|
console_handler = logging.StreamHandler()
|
|
35
|
-
console_handler.
|
|
41
|
+
console_handler.setLevel(
|
|
42
|
+
logging.INFO
|
|
43
|
+
) # the console log will have minimal friendly information
|
|
44
|
+
console_handler.setFormatter(console_formatter)
|
|
36
45
|
root_logger.addHandler(console_handler)
|
|
37
46
|
|
|
38
47
|
# File
|
|
39
48
|
file_handler = logging.handlers.RotatingFileHandler(
|
|
40
49
|
"logs/application.log", maxBytes=10 * 1024 * 1024, backupCount=5
|
|
41
50
|
)
|
|
42
|
-
file_handler.setFormatter(
|
|
51
|
+
file_handler.setFormatter(file_formatter)
|
|
52
|
+
file_handler.setLevel(
|
|
53
|
+
logging.DEBUG
|
|
54
|
+
) # the file log will have full detailed information
|
|
43
55
|
root_logger.addHandler(file_handler)
|
|
44
56
|
|
|
45
57
|
def get_logger(self, name: str) -> logging.Logger:
|
|
@@ -55,10 +67,6 @@ class AppLogger:
|
|
|
55
67
|
for handler in logging.getLogger().handlers:
|
|
56
68
|
handler.setLevel(log_level)
|
|
57
69
|
|
|
58
|
-
@property
|
|
59
|
-
def is_debug(self) -> bool:
|
|
60
|
-
return logging.getLogger().isEnabledFor(logging.DEBUG)
|
|
61
|
-
|
|
62
70
|
|
|
63
71
|
class LoggedComponent:
|
|
64
72
|
"""Base class providing logging functionality to all components"""
|
|
@@ -85,7 +93,3 @@ class LoggedComponent:
|
|
|
85
93
|
log_msg += f" | Context: {context_str}"
|
|
86
94
|
|
|
87
95
|
self.logger.error(log_msg, exc_info=True)
|
|
88
|
-
|
|
89
|
-
@property
|
|
90
|
-
def is_debug_enabled(self) -> bool:
|
|
91
|
-
return self.log.is_debug
|
|
@@ -191,7 +191,7 @@ class BasePipelineDefinition(BaseModel):
|
|
|
191
191
|
@property
|
|
192
192
|
def destination_table_invalid(self) -> Optional[str]:
|
|
193
193
|
if self.destination_table:
|
|
194
|
-
return
|
|
194
|
+
return "edl_validation_error_log"
|
|
195
195
|
return None
|
|
196
196
|
|
|
197
197
|
def file_pre_process(self, file_path: Path) -> Path:
|
|
@@ -11,6 +11,8 @@ from .models import (
|
|
|
11
11
|
)
|
|
12
12
|
from .pipeline import LoadPipeline
|
|
13
13
|
from .procedure_pipeline import ProcedurePipeline
|
|
14
|
+
from .custom_exceptions import NestedPipelinesError, UnknownPipelineError
|
|
15
|
+
from .pipeline_base import PipelineResult
|
|
14
16
|
|
|
15
17
|
|
|
16
18
|
class OrchestratorPipeline(LoggedComponent):
|
|
@@ -23,25 +25,41 @@ class OrchestratorPipeline(LoggedComponent):
|
|
|
23
25
|
definition = self.config.get_pipeline(file_name)
|
|
24
26
|
|
|
25
27
|
if not isinstance(definition, OrchestratorDefinition):
|
|
26
|
-
|
|
27
|
-
|
|
28
|
+
raise ValueError(
|
|
29
|
+
f"'{file_name}' does not contain a OrchestratorPipeline definition."
|
|
28
30
|
)
|
|
29
31
|
|
|
30
32
|
self.definition: OrchestratorDefinition = definition
|
|
31
33
|
|
|
32
|
-
def run(self) ->
|
|
34
|
+
def run(self) -> PipelineResult:
|
|
35
|
+
"""Entry point mirroring BasePipeline.run(), without audit/connector lifecycle."""
|
|
36
|
+
try:
|
|
37
|
+
success = self._run()
|
|
38
|
+
return PipelineResult(
|
|
39
|
+
success=success,
|
|
40
|
+
message="Orchestrator executed successfully."
|
|
41
|
+
if success
|
|
42
|
+
else "One or more pipelines failed.",
|
|
43
|
+
)
|
|
44
|
+
except Exception as e:
|
|
45
|
+
self.logger.warning(f"Orchestrator has [FAILED] {e}")
|
|
46
|
+
return PipelineResult(
|
|
47
|
+
success=False,
|
|
48
|
+
message="Orchestrator has failed",
|
|
49
|
+
error_code=e.__class__.__name__,
|
|
50
|
+
exception=e,
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
def _run(self) -> bool:
|
|
33
54
|
self.logger.info(
|
|
34
55
|
f"=== Starting Orchestrator: {self.definition.pipeline_name} (Orchestrator: {self.orchestrator_id}) ==="
|
|
35
56
|
)
|
|
36
57
|
|
|
37
58
|
success = True
|
|
38
59
|
for pipeline_name in self.definition.pipelines:
|
|
39
|
-
self.logger.info(
|
|
40
|
-
f"[{self.definition.pipeline_name}] -> Triggering pipeline: {pipeline_name}"
|
|
41
|
-
)
|
|
60
|
+
self.logger.info(f"Triggering pipeline: {pipeline_name}")
|
|
42
61
|
|
|
43
62
|
p_def: PipelineType = self.config.get_pipeline(pipeline_name)
|
|
44
|
-
p_success = False
|
|
45
63
|
|
|
46
64
|
# Instantiate and run
|
|
47
65
|
if isinstance(p_def, BasePipelineDefinition):
|
|
@@ -53,25 +71,17 @@ class OrchestratorPipeline(LoggedComponent):
|
|
|
53
71
|
pipeline_name, orchestrator_id=self.orchestrator_id
|
|
54
72
|
).run()
|
|
55
73
|
elif isinstance(p_def, OrchestratorDefinition):
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
)
|
|
59
|
-
p_success = False
|
|
74
|
+
# caught by run() and turned into a failed PipelineResult
|
|
75
|
+
raise NestedPipelinesError
|
|
60
76
|
else:
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
)
|
|
64
|
-
p_success = False
|
|
77
|
+
# caught by run() and turned into a failed PipelineResult
|
|
78
|
+
raise UnknownPipelineError
|
|
65
79
|
|
|
66
80
|
if not p_success:
|
|
67
81
|
success = False
|
|
68
|
-
self.logger.error(
|
|
69
|
-
f"[{self.definition.pipeline_name}] -> Pipeline failed: {pipeline_name}"
|
|
70
|
-
)
|
|
82
|
+
self.logger.error(f"Pipeline failed: {pipeline_name}")
|
|
71
83
|
if self.definition.fail_fast:
|
|
72
|
-
self.logger.error(
|
|
73
|
-
f"[{self.definition.pipeline_name}] -> Fail fast enabled. Stopping orchestrator."
|
|
74
|
-
)
|
|
84
|
+
self.logger.error("Fail fast enabled. Stopping orchestrator.")
|
|
75
85
|
break
|
|
76
86
|
else:
|
|
77
87
|
self.logger.info(
|
|
@@ -14,8 +14,12 @@ from .models import (
|
|
|
14
14
|
FileBasedConnectionSettings,
|
|
15
15
|
FileType,
|
|
16
16
|
)
|
|
17
|
-
from .custom_exceptions import
|
|
18
|
-
|
|
17
|
+
from .custom_exceptions import (
|
|
18
|
+
PipelineValidationError,
|
|
19
|
+
EmptyDataError,
|
|
20
|
+
NoValidDestination,
|
|
21
|
+
)
|
|
22
|
+
from .pipeline_base import BasePipeline, PipelineResult
|
|
19
23
|
|
|
20
24
|
|
|
21
25
|
class LoadPipeline(BasePipeline):
|
|
@@ -44,7 +48,18 @@ class LoadPipeline(BasePipeline):
|
|
|
44
48
|
self.dst_file_ops: Optional[FileOperations] = None
|
|
45
49
|
self.source_file_path: Optional[Path] = None # For auditing
|
|
46
50
|
|
|
47
|
-
self.
|
|
51
|
+
if self._init_error is None:
|
|
52
|
+
try:
|
|
53
|
+
self._initialize_components()
|
|
54
|
+
except Exception as e:
|
|
55
|
+
# deferred so run() can report it as a PipelineResult instead of
|
|
56
|
+
# raising out of the constructor
|
|
57
|
+
self._init_error = PipelineResult(
|
|
58
|
+
success=False,
|
|
59
|
+
message=f"Pipeline setup failed: {e}",
|
|
60
|
+
error_code=e.__class__.__name__,
|
|
61
|
+
exception=e,
|
|
62
|
+
)
|
|
48
63
|
|
|
49
64
|
def _initialize_components(self):
|
|
50
65
|
"""Dynamically initialize the components based on their type and definition"""
|
|
@@ -86,40 +101,37 @@ class LoadPipeline(BasePipeline):
|
|
|
86
101
|
):
|
|
87
102
|
self.dst_db_ops = self._setup_db_operations(destination_resource)
|
|
88
103
|
self.destination_name = destination_resource.conn_database
|
|
104
|
+
else:
|
|
105
|
+
raise ValueError(
|
|
106
|
+
f"Unsupported destination resource type: {type(source_resource)}"
|
|
107
|
+
)
|
|
89
108
|
|
|
90
|
-
def
|
|
109
|
+
def _run(self):
|
|
91
110
|
"""Executes the entire ETL flow"""
|
|
92
|
-
self.logger.info(
|
|
93
|
-
f">>> Starting Load Pipeline: {self.definition.pipeline_name} <<<"
|
|
94
|
-
)
|
|
95
111
|
|
|
96
112
|
try:
|
|
97
|
-
# 1.
|
|
113
|
+
# 1. READ
|
|
98
114
|
df, inferred_dtypes = self._extract_step()
|
|
99
115
|
if df.empty:
|
|
100
|
-
self.
|
|
101
|
-
|
|
116
|
+
self.error_details = "No data found when trying to read from source!"
|
|
117
|
+
self._log_audit("FAILED")
|
|
118
|
+
raise EmptyDataError(self.error_details)
|
|
102
119
|
|
|
103
120
|
self.input_rows = len(df)
|
|
121
|
+
self.logger.info(f"Rows read from source: {self.input_rows}")
|
|
104
122
|
|
|
105
123
|
# 2. TRANSFORM & VALIDATE
|
|
106
124
|
df, invalid_df = self._transform_step(df)
|
|
107
125
|
self.output_rows = len(df)
|
|
108
126
|
if df.empty:
|
|
109
|
-
self.
|
|
110
|
-
"
|
|
127
|
+
self.error_details = (
|
|
128
|
+
"After validation no data remaining - all data is invalid."
|
|
111
129
|
)
|
|
112
|
-
self._log_audit("
|
|
113
|
-
|
|
130
|
+
self._log_audit("FAILED")
|
|
131
|
+
raise EmptyDataError(self.error_details)
|
|
114
132
|
|
|
115
133
|
# 3. LOAD
|
|
116
|
-
|
|
117
|
-
if not load_success:
|
|
118
|
-
self.logger.error(
|
|
119
|
-
f">>> Pipeline {self.definition.pipeline_name} failed to load during the LOAD step."
|
|
120
|
-
)
|
|
121
|
-
self._log_audit("FAILED")
|
|
122
|
-
return False
|
|
134
|
+
self._load_step(df, invalid_df, inferred_dtypes)
|
|
123
135
|
|
|
124
136
|
self._log_audit("SUCCESS")
|
|
125
137
|
self.logger.info(
|
|
@@ -133,14 +145,14 @@ class LoadPipeline(BasePipeline):
|
|
|
133
145
|
f"Critical pipeline error - {self.definition.pipeline_name}: {str(e)}"
|
|
134
146
|
)
|
|
135
147
|
self._log_audit("FAILED")
|
|
136
|
-
|
|
148
|
+
raise
|
|
137
149
|
except Exception as e:
|
|
138
150
|
self.error_details = str(e)
|
|
139
151
|
self.log_exception(
|
|
140
152
|
e, f"Critical pipeline error - {self.definition.pipeline_name}"
|
|
141
153
|
)
|
|
142
154
|
self._log_audit("FAILED")
|
|
143
|
-
|
|
155
|
+
raise
|
|
144
156
|
finally:
|
|
145
157
|
self._cleanup()
|
|
146
158
|
|
|
@@ -156,18 +168,23 @@ class LoadPipeline(BasePipeline):
|
|
|
156
168
|
try:
|
|
157
169
|
if self.src_db_ops: # DB source
|
|
158
170
|
if self.definition.source_sql:
|
|
171
|
+
self.logger.debug("<< reading data from database >>")
|
|
159
172
|
df = self.src_db_ops.read_data(
|
|
160
173
|
self.definition.source_sql, **self.definition.read_parameters
|
|
161
174
|
)
|
|
162
175
|
dtype_map = self.definition.get_dtype_map()
|
|
176
|
+
self.logger.debug("<< identifying columns definition >>")
|
|
163
177
|
if not dtype_map:
|
|
178
|
+
self.logger.debug("<< no columns definition identified >>")
|
|
164
179
|
dtype_map = {}
|
|
165
180
|
return df, dtype_map
|
|
166
181
|
|
|
167
182
|
if self.src_file_ops: # File source
|
|
183
|
+
self.logger.debug("<< source is a file, atempting pre-processing >>")
|
|
168
184
|
self.src_file_ops._apply_file_preprocessor(
|
|
169
185
|
self.definition.file_pre_process
|
|
170
186
|
)
|
|
187
|
+
self.logger.debug("<< reading from file source >>")
|
|
171
188
|
df = self.src_file_ops.read_file(**self.definition.read_parameters)
|
|
172
189
|
self.source_file_path = self.src_file_ops.file_path
|
|
173
190
|
|
|
@@ -215,7 +232,8 @@ class LoadPipeline(BasePipeline):
|
|
|
215
232
|
rename_map = self.definition.get_rename_map()
|
|
216
233
|
if rename_map:
|
|
217
234
|
df.rename(columns=rename_map, inplace=True)
|
|
218
|
-
self.logger.info(
|
|
235
|
+
self.logger.info("Columns renamed")
|
|
236
|
+
self.logger.debug(f"Columns renamed: {rename_map}")
|
|
219
237
|
|
|
220
238
|
# 2. Pipeline hook transformation
|
|
221
239
|
df = self.definition.transform(df)
|
|
@@ -310,10 +328,13 @@ class LoadPipeline(BasePipeline):
|
|
|
310
328
|
df: pd.DataFrame,
|
|
311
329
|
invalid: pd.DataFrame,
|
|
312
330
|
dtype_map: Dict[str, types.TypeEngine],
|
|
313
|
-
) ->
|
|
331
|
+
) -> None:
|
|
314
332
|
"""Handles loading logic based on destination type."""
|
|
315
333
|
try:
|
|
316
334
|
if self.dst_db_ops and self.definition.destination_table: # DB destination
|
|
335
|
+
self.logger.info(
|
|
336
|
+
f"Writting to table {self.definition.destination_table}"
|
|
337
|
+
)
|
|
317
338
|
self.dst_db_ops.write_to_table(
|
|
318
339
|
table_name=self.definition.destination_table,
|
|
319
340
|
df=df,
|
|
@@ -321,11 +342,12 @@ class LoadPipeline(BasePipeline):
|
|
|
321
342
|
**self.definition.write_parameters,
|
|
322
343
|
)
|
|
323
344
|
if not invalid.empty:
|
|
345
|
+
self.logger.info("Invalid data found - writting to error log")
|
|
324
346
|
self.dst_db_ops.write_to_table(
|
|
325
347
|
table_name=self.definition.destination_table_invalid,
|
|
326
348
|
df=invalid,
|
|
327
349
|
index=False,
|
|
328
|
-
if_exists="
|
|
350
|
+
if_exists="append",
|
|
329
351
|
)
|
|
330
352
|
elif self.dst_file_ops: # File destination
|
|
331
353
|
valid_path, invalid_path = self.dst_file_ops._construct_output_path()
|
|
@@ -343,18 +365,16 @@ class LoadPipeline(BasePipeline):
|
|
|
343
365
|
invalid, invalid_path, **self.definition.write_parameters
|
|
344
366
|
)
|
|
345
367
|
else:
|
|
346
|
-
|
|
368
|
+
raise NoValidDestination
|
|
347
369
|
|
|
348
370
|
# file_post_process always runs against the source file, not the destination
|
|
349
371
|
if self.src_file_ops and self.source_file_path:
|
|
372
|
+
self.logger.debug("Applying file post-processing.")
|
|
350
373
|
self.source_file_path = self.definition.file_post_process(
|
|
351
374
|
self.source_file_path
|
|
352
375
|
)
|
|
353
376
|
if self.definition.source_file_delete_after_load == "yes":
|
|
354
377
|
self.source_file_path.unlink()
|
|
355
|
-
self.logger.
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
except Exception as e:
|
|
359
|
-
self.log_exception(e, "Error writing to destination")
|
|
360
|
-
return False
|
|
378
|
+
self.logger.debug(f"Deleted source file: {self.source_file_path}")
|
|
379
|
+
except Exception:
|
|
380
|
+
raise
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
from abc import ABC, abstractmethod
|
|
2
2
|
from typing import List, Optional, Union
|
|
3
3
|
from uuid import uuid4
|
|
4
|
+
from dataclasses import dataclass
|
|
4
5
|
|
|
5
6
|
import pandas as pd
|
|
6
7
|
|
|
@@ -18,6 +19,20 @@ from .models import (
|
|
|
18
19
|
)
|
|
19
20
|
|
|
20
21
|
|
|
22
|
+
@dataclass
|
|
23
|
+
class PipelineResult:
|
|
24
|
+
"""A definition of a pipeline execution outcome with some details"""
|
|
25
|
+
|
|
26
|
+
success: bool
|
|
27
|
+
message: str
|
|
28
|
+
error_code: Optional[str] = None
|
|
29
|
+
exception: Optional[Exception] = None
|
|
30
|
+
|
|
31
|
+
def __bool__(self):
|
|
32
|
+
"""Facilitate checking the result directly in if statements: if result: ..."""
|
|
33
|
+
return self.success
|
|
34
|
+
|
|
35
|
+
|
|
21
36
|
class BasePipeline(LoggedComponent, ABC):
|
|
22
37
|
"""
|
|
23
38
|
Abstract base class for all pipeline types.
|
|
@@ -40,6 +55,7 @@ class BasePipeline(LoggedComponent, ABC):
|
|
|
40
55
|
|
|
41
56
|
self._active_connectors: List[DatabaseConnector] = []
|
|
42
57
|
self.audit_db_ops: Optional[DatabaseOperations] = None
|
|
58
|
+
self._init_error: Optional[PipelineResult] = None
|
|
43
59
|
|
|
44
60
|
# Metadata for auditing
|
|
45
61
|
self.input_rows = 0
|
|
@@ -49,7 +65,17 @@ class BasePipeline(LoggedComponent, ABC):
|
|
|
49
65
|
self.file_size_bytes: Optional[int] = None
|
|
50
66
|
self.file_last_modified: Optional[pd.Timestamp] = None
|
|
51
67
|
|
|
52
|
-
|
|
68
|
+
try:
|
|
69
|
+
self._initialize_audit_resource()
|
|
70
|
+
except Exception as e:
|
|
71
|
+
# deferred so run() can report it as a PipelineResult instead of
|
|
72
|
+
# raising out of the constructor
|
|
73
|
+
self._init_error = PipelineResult(
|
|
74
|
+
success=False,
|
|
75
|
+
message=f"Pipeline setup failed: {e}",
|
|
76
|
+
error_code=e.__class__.__name__,
|
|
77
|
+
exception=e,
|
|
78
|
+
)
|
|
53
79
|
|
|
54
80
|
def _initialize_audit_resource(self):
|
|
55
81
|
"""Initialize the database operations for auditing."""
|
|
@@ -92,9 +118,14 @@ class BasePipeline(LoggedComponent, ABC):
|
|
|
92
118
|
return DatabaseOperations(connector.get_engine())
|
|
93
119
|
|
|
94
120
|
def _cleanup(self):
|
|
95
|
-
"""Dispose of all active database connectors.
|
|
121
|
+
"""Dispose of all active database connectors.
|
|
122
|
+
Never raises, so a disposal failure can't mask the pipeline's real outcome.
|
|
123
|
+
"""
|
|
96
124
|
for connector in self._active_connectors:
|
|
97
|
-
|
|
125
|
+
try:
|
|
126
|
+
connector._dispose_engine()
|
|
127
|
+
except Exception as e:
|
|
128
|
+
self.logger.warning(f"Failed to dispose connector cleanly: {e}")
|
|
98
129
|
self.logger.debug("Pipeline cleanup completed.")
|
|
99
130
|
|
|
100
131
|
def _log_audit(self, status: str):
|
|
@@ -131,7 +162,45 @@ class BasePipeline(LoggedComponent, ABC):
|
|
|
131
162
|
except Exception as e:
|
|
132
163
|
self.logger.error(f"Failed to write audit log: {str(e)}")
|
|
133
164
|
|
|
165
|
+
def run(self) -> PipelineResult:
|
|
166
|
+
"""
|
|
167
|
+
Public entry point for pipeline execution.
|
|
168
|
+
Wraps internal execution and returns a PipelineResult object.
|
|
169
|
+
"""
|
|
170
|
+
if self._init_error is not None:
|
|
171
|
+
self.logger.warning(f"Pipeline has [FAILED] {self._init_error.message}")
|
|
172
|
+
return self._init_error
|
|
173
|
+
|
|
174
|
+
self.logger.info("Starting pipeline.")
|
|
175
|
+
|
|
176
|
+
try:
|
|
177
|
+
"""Call the method implemented by child classes"""
|
|
178
|
+
self._run()
|
|
179
|
+
self.logger.info("Pipeline has finalized with [SUCCESS].")
|
|
180
|
+
|
|
181
|
+
return PipelineResult(
|
|
182
|
+
success=True,
|
|
183
|
+
message="Pipeline executed succesfully.",
|
|
184
|
+
)
|
|
185
|
+
except Exception as e:
|
|
186
|
+
# file log
|
|
187
|
+
self.logger.debug(
|
|
188
|
+
f"Pipeline has encountered an exception: {e}",
|
|
189
|
+
exc_info=True,
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
user_msg = getattr(e, "user_message", str(e))
|
|
193
|
+
# console log
|
|
194
|
+
self.logger.warning(f"Pipeline has [FAILED] {user_msg}")
|
|
195
|
+
|
|
196
|
+
return PipelineResult(
|
|
197
|
+
success=False,
|
|
198
|
+
message="Pipeline has failed",
|
|
199
|
+
error_code=e.__class__.__name__,
|
|
200
|
+
exception=e,
|
|
201
|
+
)
|
|
202
|
+
|
|
134
203
|
@abstractmethod
|
|
135
|
-
def
|
|
204
|
+
def _run(self) -> bool:
|
|
136
205
|
"""Main execution logic to be implemented by child classes."""
|
|
137
206
|
pass
|
{easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader/procedure_pipeline.py
RENAMED
|
@@ -6,7 +6,7 @@ from .models import (
|
|
|
6
6
|
ServerBasedConnectionSettings,
|
|
7
7
|
FileBasedConnectionSettings,
|
|
8
8
|
)
|
|
9
|
-
from .pipeline_base import BasePipeline
|
|
9
|
+
from .pipeline_base import BasePipeline, PipelineResult
|
|
10
10
|
|
|
11
11
|
|
|
12
12
|
class ProcedurePipeline(BasePipeline):
|
|
@@ -25,19 +25,33 @@ class ProcedurePipeline(BasePipeline):
|
|
|
25
25
|
super().__init__(definition, orchestrator_id=orchestrator_id)
|
|
26
26
|
self.definition: ProcedureDefinition = definition
|
|
27
27
|
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
28
|
+
if self._init_error is None:
|
|
29
|
+
try:
|
|
30
|
+
# Set up primary database connection
|
|
31
|
+
resource = self.config.get_resource(self.definition.resource)
|
|
32
|
+
if not isinstance(
|
|
33
|
+
resource,
|
|
34
|
+
(ServerBasedConnectionSettings, FileBasedConnectionSettings),
|
|
35
|
+
):
|
|
36
|
+
raise ValueError(
|
|
37
|
+
f"Resource '{self.definition.resource}' must be a database connection."
|
|
38
|
+
)
|
|
37
39
|
|
|
38
|
-
|
|
40
|
+
self.db_ops = self._setup_db_operations(resource)
|
|
41
|
+
except Exception as e:
|
|
42
|
+
self.error_details = str(e)
|
|
43
|
+
self._log_audit("FAILED")
|
|
44
|
+
self.logger.error(self.error_details)
|
|
45
|
+
# deferred so run() can report it as a PipelineResult instead of
|
|
46
|
+
# raising out of the constructor
|
|
47
|
+
self._init_error = PipelineResult(
|
|
48
|
+
success=False,
|
|
49
|
+
message=f"Pipeline setup failed: {e}",
|
|
50
|
+
error_code=e.__class__.__name__,
|
|
51
|
+
exception=e,
|
|
52
|
+
)
|
|
39
53
|
|
|
40
|
-
def
|
|
54
|
+
def _run(self):
|
|
41
55
|
"""Execute all procedures in the definition."""
|
|
42
56
|
self.logger.info(
|
|
43
57
|
f">>> Starting Procedure Pipeline: {self.definition.pipeline_name} <<<"
|
|
@@ -65,7 +79,7 @@ class ProcedurePipeline(BasePipeline):
|
|
|
65
79
|
self.logger.info(
|
|
66
80
|
f">>> Procedure Pipeline {self.definition.pipeline_name} finished successfully <<<"
|
|
67
81
|
)
|
|
68
|
-
return
|
|
82
|
+
return
|
|
69
83
|
|
|
70
84
|
except Exception as e:
|
|
71
85
|
self.error_details = str(e)
|
|
@@ -78,6 +92,6 @@ class ProcedurePipeline(BasePipeline):
|
|
|
78
92
|
self.sp_parameters = {p[0]: p[1] for p in self.definition.procedures}
|
|
79
93
|
|
|
80
94
|
self._log_audit("FAILED")
|
|
81
|
-
|
|
95
|
+
raise
|
|
82
96
|
finally:
|
|
83
97
|
self._cleanup()
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import os
|
|
2
|
+
import re
|
|
2
3
|
import sqlite3
|
|
3
4
|
|
|
4
5
|
import pytest
|
|
@@ -138,7 +139,9 @@ def test_validate_resources_reports_ok_and_failed(runner, tmp_path):
|
|
|
138
139
|
result = runner.invoke(main, ["validate-resources"])
|
|
139
140
|
assert result.exit_code == 0
|
|
140
141
|
assert "Resource: good_folder ... OK (Path Exists)" in result.output
|
|
141
|
-
|
|
142
|
+
# SQLAlchemy's echo logging can interleave stderr noise between the label
|
|
143
|
+
# and status, so match across lines instead of requiring adjacency.
|
|
144
|
+
assert re.search(r"Resource: good_db \.\.\.[\s\S]*?OK \(Connected\)", result.output)
|
|
142
145
|
assert "Resource: bad_db ... FAILED" in result.output
|
|
143
146
|
|
|
144
147
|
|
|
@@ -179,9 +182,11 @@ def test_validate_pipelines_reports_ok_for_load_procedure_and_orchestrator(
|
|
|
179
182
|
|
|
180
183
|
result = runner.invoke(main, ["validate-pipelines"])
|
|
181
184
|
assert result.exit_code == 0
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
assert "Pipeline:
|
|
185
|
+
# SQLAlchemy's echo logging can interleave stderr noise between the label
|
|
186
|
+
# and status, so match across lines instead of requiring adjacency.
|
|
187
|
+
assert re.search(r"Pipeline: load_pipeline \.\.\.[\s\S]*?OK\b", result.output)
|
|
188
|
+
assert re.search(r"Pipeline: procedure_pipeline \.\.\.[\s\S]*?OK\b", result.output)
|
|
189
|
+
assert re.search(r"Pipeline: my_orchestrator \.\.\.[\s\S]*?OK\b", result.output)
|
|
185
190
|
|
|
186
191
|
|
|
187
192
|
def test_validate_pipelines_reports_failed_for_unresolvable_resource(runner, tmp_path):
|
|
@@ -193,4 +198,4 @@ def test_validate_pipelines_reports_failed_for_unresolvable_resource(runner, tmp
|
|
|
193
198
|
|
|
194
199
|
result = runner.invoke(main, ["validate-pipelines"])
|
|
195
200
|
assert result.exit_code == 0
|
|
196
|
-
assert "Pipeline: broken_pipeline
|
|
201
|
+
assert re.search(r"Pipeline: broken_pipeline \.\.\.[\s\S]*?FAILED", result.output)
|
|
@@ -93,7 +93,7 @@ def test_orchestrator_passes_orchestrator_id_and_writes_audit():
|
|
|
93
93
|
assert orchestrator.orchestrator_id is not None
|
|
94
94
|
|
|
95
95
|
success = orchestrator.run()
|
|
96
|
-
assert success is True
|
|
96
|
+
assert success.success is True
|
|
97
97
|
|
|
98
98
|
# Verify target tables were created and populated in destination
|
|
99
99
|
conn_dst = sqlite3.connect(dst_db_path)
|
|
@@ -72,7 +72,7 @@ def test_transform_hook_sees_renamed_columns():
|
|
|
72
72
|
pipeline = LoadPipeline("test_pipeline")
|
|
73
73
|
success = pipeline.run()
|
|
74
74
|
|
|
75
|
-
assert success is True
|
|
75
|
+
assert success.success is True
|
|
76
76
|
|
|
77
77
|
conn_dst = sqlite3.connect(dst_db_path)
|
|
78
78
|
cursor_dst = conn_dst.cursor()
|
|
@@ -72,7 +72,7 @@ def test_sequential_mode_runs_all_procedures_when_each_succeeds():
|
|
|
72
72
|
pipeline.db_ops.execute_stored_procedure = _record_call # type: ignore[method-assign]
|
|
73
73
|
|
|
74
74
|
success = pipeline.run()
|
|
75
|
-
assert success is True
|
|
75
|
+
assert success.success is True
|
|
76
76
|
assert [c[0] for c in recorded_calls] == ["proc_a", "proc_b"]
|
|
77
77
|
|
|
78
78
|
|
|
@@ -173,7 +173,7 @@ def test_procedure_pipeline_merges_dynamic_parameters():
|
|
|
173
173
|
pipeline.db_ops.execute_stored_procedure = _record_call # type: ignore[method-assign]
|
|
174
174
|
|
|
175
175
|
success = pipeline.run()
|
|
176
|
-
assert success is True
|
|
176
|
+
assert success.success is True
|
|
177
177
|
assert recorded_calls == [
|
|
178
178
|
("insert_call", {"name": "static_name", "value": "dynamic_value"})
|
|
179
179
|
]
|
|
@@ -52,7 +52,7 @@ def test_source_file_deleted_after_successful_load_to_db():
|
|
|
52
52
|
pipeline = LoadPipeline("test_pipeline")
|
|
53
53
|
success = pipeline.run()
|
|
54
54
|
|
|
55
|
-
assert success is True
|
|
55
|
+
assert success.success is True
|
|
56
56
|
assert not (Path(tmpdir) / "data.csv").exists()
|
|
57
57
|
|
|
58
58
|
|
|
@@ -86,7 +86,7 @@ def test_file_post_process_runs_on_source_when_destination_is_db():
|
|
|
86
86
|
pipeline = LoadPipeline("test_pipeline")
|
|
87
87
|
success = pipeline.run()
|
|
88
88
|
|
|
89
|
-
assert success is True
|
|
89
|
+
assert success.success is True
|
|
90
90
|
assert post_processed_paths == [Path(tmpdir) / "data.csv"]
|
|
91
91
|
|
|
92
92
|
|
|
@@ -118,5 +118,10 @@ def test_source_file_delete_after_load_raises_when_source_is_not_a_file():
|
|
|
118
118
|
)
|
|
119
119
|
config.pipelines["test_pipeline"] = pipeline_def
|
|
120
120
|
|
|
121
|
-
|
|
122
|
-
|
|
121
|
+
# Setup failures are deferred to run() rather than raised from the
|
|
122
|
+
# constructor, so they can be reported as a PipelineResult.
|
|
123
|
+
pipeline = LoadPipeline("test_pipeline")
|
|
124
|
+
result = pipeline.run()
|
|
125
|
+
|
|
126
|
+
assert result.success is False
|
|
127
|
+
assert "source_file_delete_after_load" in result.message
|
|
@@ -87,7 +87,7 @@ def test_validation_fail_false_keeps_valid_rows():
|
|
|
87
87
|
success = pipeline.run()
|
|
88
88
|
|
|
89
89
|
# Should be successful because validation_fail=False
|
|
90
|
-
assert success is True
|
|
90
|
+
assert success.success is True
|
|
91
91
|
|
|
92
92
|
# Verify destination table only has valid rows (1 and 3)
|
|
93
93
|
conn_dst = sqlite3.connect(dst_db_path)
|
|
@@ -101,7 +101,7 @@ def test_validation_fail_false_keeps_valid_rows():
|
|
|
101
101
|
assert rows[1] == (3, "Charlie")
|
|
102
102
|
|
|
103
103
|
cursor_dst = sqlite3.connect(dst_db_path).cursor()
|
|
104
|
-
cursor_dst.execute("SELECT error FROM
|
|
104
|
+
cursor_dst.execute("SELECT error FROM edl_validation_error_log")
|
|
105
105
|
invalid_rows = cursor_dst.fetchall()
|
|
106
106
|
cursor_dst.connection.close()
|
|
107
107
|
|
|
@@ -146,7 +146,7 @@ def test_validation_writes_valid_and_invalid_file_outputs():
|
|
|
146
146
|
|
|
147
147
|
success = LoadPipeline("test_pipeline").run()
|
|
148
148
|
|
|
149
|
-
assert success is True
|
|
149
|
+
assert success.success is True
|
|
150
150
|
|
|
151
151
|
valid_df = pd.read_csv(Path(tmpdir) / "output.csv")
|
|
152
152
|
invalid_df = pd.read_csv(Path(tmpdir) / "output_invalid.csv")
|
|
@@ -206,7 +206,7 @@ def test_validation_fail_true_fails_pipeline():
|
|
|
206
206
|
success = pipeline.run()
|
|
207
207
|
|
|
208
208
|
# Should fail because validation_fail=True
|
|
209
|
-
assert success is False
|
|
209
|
+
assert success.success is False
|
|
210
210
|
assert pipeline.error_details is not None
|
|
211
211
|
assert "Validation failed" in pipeline.error_details
|
|
212
212
|
assert "Row 1:" in pipeline.error_details # Row 2 (0-indexed row 1) failed
|
|
@@ -261,8 +261,10 @@ def test_validation_fail_false_graceful_stop_when_all_rows_invalid():
|
|
|
261
261
|
pipeline = LoadPipeline("test_pipeline")
|
|
262
262
|
success = pipeline.run()
|
|
263
263
|
|
|
264
|
-
#
|
|
265
|
-
|
|
264
|
+
# validation_fail=False keeps invalid rows out of the load, but if that
|
|
265
|
+
# leaves no data at all the pipeline still fails with an EmptyDataError.
|
|
266
|
+
assert success.success is False
|
|
267
|
+
assert success.error_code == "EmptyDataError"
|
|
266
268
|
assert pipeline.output_rows == 0
|
|
267
269
|
|
|
268
270
|
# Destination table should NOT have been loaded or have any data
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
{easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/entry_points.txt
RENAMED
|
File without changes
|
{easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/requires.txt
RENAMED
|
File without changes
|
{easy_data_loader-0.1.9 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/top_level.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|