easy-data-loader 0.1.8__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. {easy_data_loader-0.1.8/src/easy_data_loader.egg-info → easy_data_loader-0.2.1}/PKG-INFO +1 -1
  2. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/pyproject.toml +1 -1
  3. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/cli.py +3 -0
  4. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/custom_exceptions.py +24 -0
  5. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/data_inferrence.py +13 -14
  6. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/database_connector.py +3 -3
  7. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/database_operations.py +9 -7
  8. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/file_operations.py +4 -2
  9. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/log.py +16 -12
  10. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/models.py +2 -2
  11. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/orchestrator.py +35 -23
  12. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/pipeline.py +60 -37
  13. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/pipeline_base.py +75 -6
  14. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/procedure_pipeline.py +31 -17
  15. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1/src/easy_data_loader.egg-info}/PKG-INFO +1 -1
  16. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_cli.py +10 -5
  17. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_database_connector.py +1 -1
  18. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_orchestrator.py +4 -4
  19. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_pipeline_transform_order.py +2 -2
  20. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_procedure_execution_mode.py +3 -3
  21. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_resource_access.py +5 -5
  22. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_source_file_delete.py +12 -7
  23. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_validation.py +12 -10
  24. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/LICENSE +0 -0
  25. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/README.md +0 -0
  26. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/setup.cfg +0 -0
  27. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/__init__.py +0 -0
  28. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/config_loader.py +0 -0
  29. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/driver_detector.py +0 -0
  30. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/resource_access.py +0 -0
  31. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader/utils.py +0 -0
  32. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/SOURCES.txt +0 -0
  33. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/dependency_links.txt +0 -0
  34. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/entry_points.txt +0 -0
  35. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/requires.txt +0 -0
  36. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/src/easy_data_loader.egg-info/top_level.txt +0 -0
  37. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_config_loader.py +0 -0
  38. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_data_inference.py +0 -0
  39. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_file_operations.py +0 -0
  40. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_imports.py +0 -0
  41. {easy_data_loader-0.1.8 → easy_data_loader-0.2.1}/tests/test_models.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: easy_data_loader
3
- Version: 0.1.8
3
+ Version: 0.2.1
4
4
  Summary: Data transfer utilities between files and databases
5
5
  Author-email: Bojoi Gabriel <bojoigabriel@gmail.com>
6
6
  Classifier: Development Status :: 3 - Alpha
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "easy_data_loader"
3
- version = "0.1.8"
3
+ version = "0.2.1"
4
4
  description = "Data transfer utilities between files and databases"
5
5
  authors = [{ name = "Bojoi Gabriel", email = "bojoigabriel@gmail.com" }]
6
6
  readme = "README.md"
@@ -412,6 +412,9 @@ def validate_pipelines():
412
412
  else:
413
413
  raise ValueError(f"Unknown pipeline type: {type(definition)}")
414
414
 
415
+ if instance is not None and instance._init_error is not None:
416
+ raise ValueError(instance._init_error.message)
417
+
415
418
  results[name] = "OK"
416
419
  except Exception as e:
417
420
  results[name] = f"FAILED: {str(e)}"
@@ -26,3 +26,27 @@ class PipelineValidationError(Exception):
26
26
  def __init__(self, message: str):
27
27
  self.message = message
28
28
  super().__init__(self.message)
29
+
30
+
31
+ class EmptyDataError(Exception):
32
+ def __init__(self, message: str):
33
+ self.message = message
34
+ super().__init__(self.message)
35
+
36
+
37
+ class NoValidDestination(Exception):
38
+ def __init__(self, message: str = "The pipeline destination is not valid"):
39
+ self.message = message
40
+ super().__init__(self.message)
41
+
42
+
43
+ class NestedPipelinesError(Exception):
44
+ def __init__(self, message: str = "Nested pipelines are not supported"):
45
+ self.message = message
46
+ super().__init__(self.message)
47
+
48
+
49
+ class UnknownPipelineError(Exception):
50
+ def __init__(self, message: str = "Unknown pipeline type"):
51
+ self.message = message
52
+ super().__init__(self.message)
@@ -88,7 +88,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
88
88
  self.logger.error(
89
89
  f"Failed to infer dtype for column '{col}': {str(e)}. "
90
90
  f"Column will use default inference.",
91
- exc_info=self.is_debug_enabled,
91
+ exc_info=True,
92
92
  )
93
93
  # Skip this column - it will use default inference
94
94
  continue
@@ -102,9 +102,8 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
102
102
 
103
103
  self.logger.info(f"Dtype inference completed for {len(dtype_dict)} columns")
104
104
 
105
- # Log summary if debug enabled
106
- if self.is_debug_enabled:
107
- self._log_inference_summary(dtype_dict)
105
+ # Only visible in the file log, since it's logged at debug level
106
+ self._log_inference_summary(dtype_dict)
108
107
 
109
108
  return dtype_dict
110
109
 
@@ -593,7 +592,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
593
592
  self.logger.error(
594
593
  f"Failed to infer dtype for column '{col_name}' from Parquet metadata: {str(e)}. "
595
594
  f"Column will use default inference.",
596
- exc_info=self.is_debug_enabled,
595
+ exc_info=True,
597
596
  )
598
597
  continue
599
598
 
@@ -616,7 +615,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
616
615
  self.logger.error(
617
616
  f"Failed to analyze string length for column '{col}': {str(e)}. "
618
617
  f"Column will use default inference.",
619
- exc_info=self.is_debug_enabled,
618
+ exc_info=True,
620
619
  )
621
620
  # Remove from dict so it uses default
622
621
  dtype_dict.pop(col, None)
@@ -626,7 +625,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
626
625
  self.logger.error(
627
626
  f"Failed to read string columns from Parquet: {str(e)}. "
628
627
  f"String columns will use default inference.",
629
- exc_info=self.is_debug_enabled,
628
+ exc_info=True,
630
629
  )
631
630
  # Remove string columns from dict
632
631
  for col in string_columns:
@@ -643,8 +642,8 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
643
642
  f"Parquet dtype inference completed for {len(dtype_dict)} columns"
644
643
  )
645
644
 
646
- if self.is_debug_enabled:
647
- self._log_inference_summary(dtype_dict)
645
+ # Only visible in the file log, since it's logged at debug level
646
+ self._log_inference_summary(dtype_dict)
648
647
 
649
648
  return dtype_dict
650
649
 
@@ -703,7 +702,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
703
702
  self.logger.error(
704
703
  f"Failed to infer dtype for column '{col_name}' from ORC metadata: {str(e)}. "
705
704
  f"Column will use default inference.",
706
- exc_info=self.is_debug_enabled,
705
+ exc_info=True,
707
706
  )
708
707
  continue
709
708
 
@@ -726,7 +725,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
726
725
  self.logger.error(
727
726
  f"Failed to analyze string length for column '{col}': {str(e)}. "
728
727
  f"Column will use default inference.",
729
- exc_info=self.is_debug_enabled,
728
+ exc_info=True,
730
729
  )
731
730
  dtype_dict.pop(col, None)
732
731
  continue
@@ -735,7 +734,7 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
735
734
  self.logger.error(
736
735
  f"Failed to read string columns from ORC: {str(e)}. "
737
736
  f"String columns will use default inference.",
738
- exc_info=self.is_debug_enabled,
737
+ exc_info=True,
739
738
  )
740
739
  for col in string_columns:
741
740
  dtype_dict.pop(col, None)
@@ -751,8 +750,8 @@ class SQLAlchemyDTypeInferrer(LoggedComponent):
751
750
  f"ORC dtype inference completed for {len(dtype_dict)} columns"
752
751
  )
753
752
 
754
- if self.is_debug_enabled:
755
- self._log_inference_summary(dtype_dict)
753
+ # Only visible in the file log, since it's logged at debug level
754
+ self._log_inference_summary(dtype_dict)
756
755
 
757
756
  return dtype_dict
758
757
 
@@ -107,7 +107,7 @@ class SqlServerDatabaseConnector(LoggedComponent, DatabaseConnector):
107
107
  max_overflow=10,
108
108
  pool_timeout=30,
109
109
  pool_recycle=3600,
110
- echo=self.is_debug_enabled,
110
+ echo=False,
111
111
  )
112
112
  return engine
113
113
  except Exception as e:
@@ -183,7 +183,7 @@ class SQLiteDatabaseConnector(LoggedComponent, DatabaseConnector):
183
183
  # SQLite specific engine creation
184
184
  engine = create_engine(
185
185
  connection_string,
186
- echo=self.is_debug_enabled,
186
+ echo=False,
187
187
  )
188
188
  return engine
189
189
  except Exception as e:
@@ -280,7 +280,7 @@ class PostgresDatabaseConnector(LoggedComponent, DatabaseConnector):
280
280
  max_overflow=10,
281
281
  pool_timeout=30,
282
282
  pool_recycle=3600,
283
- echo=self.is_debug_enabled,
283
+ echo=False,
284
284
  )
285
285
  return engine
286
286
  except Exception as e:
@@ -19,18 +19,20 @@ class DatabaseOperations(LoggedComponent):
19
19
  self.engine = engine
20
20
  self._inspector = inspect(self.engine)
21
21
 
22
- def write_to_table(
23
- self, table_name: Optional[str], df: DataFrame, **kwargs
24
- ) -> bool:
22
+ def write_to_table(self, table_name: Optional[str], df: DataFrame, **kwargs):
25
23
  """Write a dataframe to a specified table in the database"""
26
24
 
27
- self.logger.info(f"Writing {len(df)} rows to table: {table_name}")
25
+ self.logger.debug(f"Writing {len(df)} rows to table: {table_name}")
28
26
  try:
29
27
  if table_name:
30
28
  df.to_sql(table_name, con=self.engine, **kwargs)
31
- return True
29
+ self.logger.debug("Finalized write to table: {table_name}")
30
+ return
32
31
  except Exception as e:
33
- self.log_exception(e, f"Failed to write to table {table_name}")
32
+ self.logger.debug(
33
+ f"Failed to write to table {table_name} caused by the following exception: {e}",
34
+ exc_info=True,
35
+ )
34
36
  raise
35
37
 
36
38
  def read_data(self, sql: str, **kwargs) -> DataFrame:
@@ -151,7 +153,7 @@ class DatabaseOperations(LoggedComponent):
151
153
  Column("file_last_modified", DateTime),
152
154
  Column("sp_name", String(255)),
153
155
  Column("sp_parameters", String),
154
- Column("timestamp", DateTime),
156
+ Column("insert_timestamp", DateTime),
155
157
  Column("error_details", String),
156
158
  ]
157
159
 
@@ -48,7 +48,9 @@ class FileOperations(LoggedComponent):
48
48
  Identifies the file based on pattern (latest) or explicit name.
49
49
  """
50
50
  if self.settings.file_pattern:
51
- files = list(self.settings.folder_path.glob('*' + self.settings.file_pattern + '*'))
51
+ files = list(
52
+ self.settings.folder_path.glob("*" + self.settings.file_pattern + "*")
53
+ )
52
54
  if not files:
53
55
  self.log_and_raise(
54
56
  ValueError,
@@ -132,7 +134,7 @@ class FileOperations(LoggedComponent):
132
134
  """
133
135
 
134
136
  current_path = self._find_file()
135
- self.logger.info(f"Current file path is: {current_path}")
137
+ self.logger.debug(f"Current file path is: {current_path}")
136
138
 
137
139
  if preprocessor_func is None:
138
140
  self.file_path = current_path
@@ -21,25 +21,37 @@ class AppLogger:
21
21
  def _setup_logging(self):
22
22
  Path("logs").mkdir(exist_ok=True)
23
23
 
24
- formatter = logging.Formatter(
24
+ file_formatter = logging.Formatter(
25
25
  "%(asctime)s - %(name)s - %(levelname)s - %(message)s",
26
26
  datefmt="%Y-%m-%d %H:%M:%S",
27
27
  )
28
28
 
29
+ console_formatter = logging.Formatter(
30
+ "[%(asctime)s]-[%(levelname)s] %(message)s"
31
+ )
32
+
29
33
  root_logger = logging.getLogger()
30
- root_logger.setLevel(logging.INFO)
34
+ root_logger.setLevel(
35
+ logging.DEBUG
36
+ ) # allow all messages and split them in the handlers
31
37
  root_logger.handlers.clear()
32
38
 
33
39
  # Console
34
40
  console_handler = logging.StreamHandler()
35
- console_handler.setFormatter(formatter)
41
+ console_handler.setLevel(
42
+ logging.INFO
43
+ ) # the console log will have minimal friendly information
44
+ console_handler.setFormatter(console_formatter)
36
45
  root_logger.addHandler(console_handler)
37
46
 
38
47
  # File
39
48
  file_handler = logging.handlers.RotatingFileHandler(
40
49
  "logs/application.log", maxBytes=10 * 1024 * 1024, backupCount=5
41
50
  )
42
- file_handler.setFormatter(formatter)
51
+ file_handler.setFormatter(file_formatter)
52
+ file_handler.setLevel(
53
+ logging.DEBUG
54
+ ) # the file log will have full detailed information
43
55
  root_logger.addHandler(file_handler)
44
56
 
45
57
  def get_logger(self, name: str) -> logging.Logger:
@@ -55,10 +67,6 @@ class AppLogger:
55
67
  for handler in logging.getLogger().handlers:
56
68
  handler.setLevel(log_level)
57
69
 
58
- @property
59
- def is_debug(self) -> bool:
60
- return logging.getLogger().isEnabledFor(logging.DEBUG)
61
-
62
70
 
63
71
  class LoggedComponent:
64
72
  """Base class providing logging functionality to all components"""
@@ -85,7 +93,3 @@ class LoggedComponent:
85
93
  log_msg += f" | Context: {context_str}"
86
94
 
87
95
  self.logger.error(log_msg, exc_info=True)
88
-
89
- @property
90
- def is_debug_enabled(self) -> bool:
91
- return self.log.is_debug
@@ -191,7 +191,7 @@ class BasePipelineDefinition(BaseModel):
191
191
  @property
192
192
  def destination_table_invalid(self) -> Optional[str]:
193
193
  if self.destination_table:
194
- return f"{self.destination_table}_invalid"
194
+ return "edl_validation_error_log"
195
195
  return None
196
196
 
197
197
  def file_pre_process(self, file_path: Path) -> Path:
@@ -282,7 +282,7 @@ class AuditEntry(BaseModel):
282
282
  error_details: Optional[str] = None
283
283
  sp_name: Optional[str] = None
284
284
  sp_parameters: Optional[Dict[str, Any]] = None
285
- timestamp: pd.Timestamp = Field(default_factory=pd.Timestamp.now)
285
+ insert_timestamp: pd.Timestamp = Field(default_factory=pd.Timestamp.now)
286
286
 
287
287
  model_config = ConfigDict(arbitrary_types_allowed=True)
288
288
 
@@ -11,35 +11,55 @@ from .models import (
11
11
  )
12
12
  from .pipeline import LoadPipeline
13
13
  from .procedure_pipeline import ProcedurePipeline
14
+ from .custom_exceptions import NestedPipelinesError, UnknownPipelineError
15
+ from .pipeline_base import PipelineResult
14
16
 
15
17
 
16
18
  class OrchestratorPipeline(LoggedComponent):
17
- """Executes a chain of pipelines defined by an OrchestratorDefinition"""
19
+ """Executes a chain of pipelines defined by a OrchestratorDefinition"""
18
20
 
19
- def __init__(self, pipeline_name: str, orchestrator_id: Optional[str] = None):
21
+ def __init__(self, file_name: str, orchestrator_id: Optional[str] = None):
20
22
  super().__init__()
21
23
  self.orchestrator_id = orchestrator_id if orchestrator_id else str(uuid4())
22
24
  self.config = Configuration()
23
- definition = self.config.get_pipeline(pipeline_name)
25
+ definition = self.config.get_pipeline(file_name)
24
26
 
25
27
  if not isinstance(definition, OrchestratorDefinition):
26
- self.log_and_raise(ValueError, f"'{pipeline_name}' is not an orchestrator.")
28
+ raise ValueError(
29
+ f"'{file_name}' does not contain a OrchestratorPipeline definition."
30
+ )
27
31
 
28
32
  self.definition: OrchestratorDefinition = definition
29
33
 
30
- def run(self) -> bool:
34
+ def run(self) -> PipelineResult:
35
+ """Entry point mirroring BasePipeline.run(), without audit/connector lifecycle."""
36
+ try:
37
+ success = self._run()
38
+ return PipelineResult(
39
+ success=success,
40
+ message="Orchestrator executed successfully."
41
+ if success
42
+ else "One or more pipelines failed.",
43
+ )
44
+ except Exception as e:
45
+ self.logger.warning(f"Orchestrator has [FAILED] {e}")
46
+ return PipelineResult(
47
+ success=False,
48
+ message="Orchestrator has failed",
49
+ error_code=e.__class__.__name__,
50
+ exception=e,
51
+ )
52
+
53
+ def _run(self) -> bool:
31
54
  self.logger.info(
32
55
  f"=== Starting Orchestrator: {self.definition.pipeline_name} (Orchestrator: {self.orchestrator_id}) ==="
33
56
  )
34
57
 
35
58
  success = True
36
59
  for pipeline_name in self.definition.pipelines:
37
- self.logger.info(
38
- f"[{self.definition.pipeline_name}] -> Triggering pipeline: {pipeline_name}"
39
- )
60
+ self.logger.info(f"Triggering pipeline: {pipeline_name}")
40
61
 
41
62
  p_def: PipelineType = self.config.get_pipeline(pipeline_name)
42
- p_success = False
43
63
 
44
64
  # Instantiate and run
45
65
  if isinstance(p_def, BasePipelineDefinition):
@@ -51,25 +71,17 @@ class OrchestratorPipeline(LoggedComponent):
51
71
  pipeline_name, orchestrator_id=self.orchestrator_id
52
72
  ).run()
53
73
  elif isinstance(p_def, OrchestratorDefinition):
54
- self.logger.error(
55
- f"[{self.definition.pipeline_name}] -> Nested orchestrators are not supported."
56
- )
57
- p_success = False
74
+ # caught by run() and turned into a failed PipelineResult
75
+ raise NestedPipelinesError
58
76
  else:
59
- self.logger.error(
60
- f"[{self.definition.pipeline_name}] -> Unknown pipeline type for '{pipeline_name}'"
61
- )
62
- p_success = False
77
+ # caught by run() and turned into a failed PipelineResult
78
+ raise UnknownPipelineError
63
79
 
64
80
  if not p_success:
65
81
  success = False
66
- self.logger.error(
67
- f"[{self.definition.pipeline_name}] -> Pipeline failed: {pipeline_name}"
68
- )
82
+ self.logger.error(f"Pipeline failed: {pipeline_name}")
69
83
  if self.definition.fail_fast:
70
- self.logger.error(
71
- f"[{self.definition.pipeline_name}] -> Fail fast enabled. Stopping orchestrator."
72
- )
84
+ self.logger.error("Fail fast enabled. Stopping orchestrator.")
73
85
  break
74
86
  else:
75
87
  self.logger.info(
@@ -14,8 +14,12 @@ from .models import (
14
14
  FileBasedConnectionSettings,
15
15
  FileType,
16
16
  )
17
- from .custom_exceptions import PipelineValidationError
18
- from .pipeline_base import BasePipeline
17
+ from .custom_exceptions import (
18
+ PipelineValidationError,
19
+ EmptyDataError,
20
+ NoValidDestination,
21
+ )
22
+ from .pipeline_base import BasePipeline, PipelineResult
19
23
 
20
24
 
21
25
  class LoadPipeline(BasePipeline):
@@ -25,13 +29,13 @@ class LoadPipeline(BasePipeline):
25
29
  Inherits shared logic from BasePipeline.
26
30
  """
27
31
 
28
- def __init__(self, pipeline_name: str, orchestrator_id: Optional[str] = None):
32
+ def __init__(self, file_name: str, orchestrator_id: Optional[str] = None):
29
33
 
30
34
  # Load definition from config
31
- definition = Configuration().get_pipeline(pipeline_name)
35
+ definition = Configuration().get_pipeline(file_name)
32
36
  if not isinstance(definition, BasePipelineDefinition):
33
37
  raise ValueError(
34
- f"Pipeline '{pipeline_name}' is not a LoadPipeline definition."
38
+ f"Pipeline '{file_name}' does not contain a LoadPipeline definition."
35
39
  )
36
40
 
37
41
  super().__init__(definition, orchestrator_id=orchestrator_id)
@@ -44,7 +48,18 @@ class LoadPipeline(BasePipeline):
44
48
  self.dst_file_ops: Optional[FileOperations] = None
45
49
  self.source_file_path: Optional[Path] = None # For auditing
46
50
 
47
- self._initialize_components()
51
+ if self._init_error is None:
52
+ try:
53
+ self._initialize_components()
54
+ except Exception as e:
55
+ # deferred so run() can report it as a PipelineResult instead of
56
+ # raising out of the constructor
57
+ self._init_error = PipelineResult(
58
+ success=False,
59
+ message=f"Pipeline setup failed: {e}",
60
+ error_code=e.__class__.__name__,
61
+ exception=e,
62
+ )
48
63
 
49
64
  def _initialize_components(self):
50
65
  """Dynamically initialize the components based on their type and definition"""
@@ -86,40 +101,37 @@ class LoadPipeline(BasePipeline):
86
101
  ):
87
102
  self.dst_db_ops = self._setup_db_operations(destination_resource)
88
103
  self.destination_name = destination_resource.conn_database
104
+ else:
105
+ raise ValueError(
106
+ f"Unsupported destination resource type: {type(source_resource)}"
107
+ )
89
108
 
90
- def run(self) -> bool:
109
+ def _run(self):
91
110
  """Executes the entire ETL flow"""
92
- self.logger.info(
93
- f">>> Starting Load Pipeline: {self.definition.pipeline_name} <<<"
94
- )
95
111
 
96
112
  try:
97
- # 1. EXTRACT
113
+ # 1. READ
98
114
  df, inferred_dtypes = self._extract_step()
99
115
  if df.empty:
100
- self.logger.warning("No data found! Pipeline will stop!")
101
- return False
116
+ self.error_details = "No data found when trying to read from source!"
117
+ self._log_audit("FAILED")
118
+ raise EmptyDataError(self.error_details)
102
119
 
103
120
  self.input_rows = len(df)
121
+ self.logger.info(f"Rows read from source: {self.input_rows}")
104
122
 
105
123
  # 2. TRANSFORM & VALIDATE
106
124
  df, invalid_df = self._transform_step(df)
107
125
  self.output_rows = len(df)
108
126
  if df.empty:
109
- self.logger.warning(
110
- "No valid data remaining after validation! Pipeline will stop gracefully."
127
+ self.error_details = (
128
+ "After validation no data remaining - all data is invalid."
111
129
  )
112
- self._log_audit("SUCCESS")
113
- return True
130
+ self._log_audit("FAILED")
131
+ raise EmptyDataError(self.error_details)
114
132
 
115
133
  # 3. LOAD
116
- load_success = self._load_step(df, invalid_df, inferred_dtypes)
117
- if not load_success:
118
- self.logger.error(
119
- f">>> Pipeline {self.definition.pipeline_name} failed to load during the LOAD step."
120
- )
121
- self._log_audit("FAILED")
122
- return False
134
+ self._load_step(df, invalid_df, inferred_dtypes)
123
135
 
124
136
  self._log_audit("SUCCESS")
125
137
  self.logger.info(
@@ -133,14 +145,14 @@ class LoadPipeline(BasePipeline):
133
145
  f"Critical pipeline error - {self.definition.pipeline_name}: {str(e)}"
134
146
  )
135
147
  self._log_audit("FAILED")
136
- return False
148
+ raise
137
149
  except Exception as e:
138
150
  self.error_details = str(e)
139
151
  self.log_exception(
140
152
  e, f"Critical pipeline error - {self.definition.pipeline_name}"
141
153
  )
142
154
  self._log_audit("FAILED")
143
- return False
155
+ raise
144
156
  finally:
145
157
  self._cleanup()
146
158
 
@@ -156,15 +168,23 @@ class LoadPipeline(BasePipeline):
156
168
  try:
157
169
  if self.src_db_ops: # DB source
158
170
  if self.definition.source_sql:
171
+ self.logger.debug("<< reading data from database >>")
159
172
  df = self.src_db_ops.read_data(
160
173
  self.definition.source_sql, **self.definition.read_parameters
161
174
  )
162
- return df, {} # No dtype inference for DB sources
175
+ dtype_map = self.definition.get_dtype_map()
176
+ self.logger.debug("<< identifying columns definition >>")
177
+ if not dtype_map:
178
+ self.logger.debug("<< no columns definition identified >>")
179
+ dtype_map = {}
180
+ return df, dtype_map
163
181
 
164
182
  if self.src_file_ops: # File source
183
+ self.logger.debug("<< source is a file, atempting pre-processing >>")
165
184
  self.src_file_ops._apply_file_preprocessor(
166
185
  self.definition.file_pre_process
167
186
  )
187
+ self.logger.debug("<< reading from file source >>")
168
188
  df = self.src_file_ops.read_file(**self.definition.read_parameters)
169
189
  self.source_file_path = self.src_file_ops.file_path
170
190
 
@@ -212,7 +232,8 @@ class LoadPipeline(BasePipeline):
212
232
  rename_map = self.definition.get_rename_map()
213
233
  if rename_map:
214
234
  df.rename(columns=rename_map, inplace=True)
215
- self.logger.info(f"Columns renamed: {list(rename_map.values())}")
235
+ self.logger.info("Columns renamed")
236
+ self.logger.debug(f"Columns renamed: {rename_map}")
216
237
 
217
238
  # 2. Pipeline hook transformation
218
239
  df = self.definition.transform(df)
@@ -307,10 +328,13 @@ class LoadPipeline(BasePipeline):
307
328
  df: pd.DataFrame,
308
329
  invalid: pd.DataFrame,
309
330
  dtype_map: Dict[str, types.TypeEngine],
310
- ) -> bool:
331
+ ) -> None:
311
332
  """Handles loading logic based on destination type."""
312
333
  try:
313
334
  if self.dst_db_ops and self.definition.destination_table: # DB destination
335
+ self.logger.info(
336
+ f"Writting to table {self.definition.destination_table}"
337
+ )
314
338
  self.dst_db_ops.write_to_table(
315
339
  table_name=self.definition.destination_table,
316
340
  df=df,
@@ -318,11 +342,12 @@ class LoadPipeline(BasePipeline):
318
342
  **self.definition.write_parameters,
319
343
  )
320
344
  if not invalid.empty:
345
+ self.logger.info("Invalid data found - writting to error log")
321
346
  self.dst_db_ops.write_to_table(
322
347
  table_name=self.definition.destination_table_invalid,
323
348
  df=invalid,
324
349
  index=False,
325
- if_exists="replace",
350
+ if_exists="append",
326
351
  )
327
352
  elif self.dst_file_ops: # File destination
328
353
  valid_path, invalid_path = self.dst_file_ops._construct_output_path()
@@ -340,18 +365,16 @@ class LoadPipeline(BasePipeline):
340
365
  invalid, invalid_path, **self.definition.write_parameters
341
366
  )
342
367
  else:
343
- return False
368
+ raise NoValidDestination
344
369
 
345
370
  # file_post_process always runs against the source file, not the destination
346
371
  if self.src_file_ops and self.source_file_path:
372
+ self.logger.debug("Applying file post-processing.")
347
373
  self.source_file_path = self.definition.file_post_process(
348
374
  self.source_file_path
349
375
  )
350
376
  if self.definition.source_file_delete_after_load == "yes":
351
377
  self.source_file_path.unlink()
352
- self.logger.info(f"Deleted source file: {self.source_file_path}")
353
-
354
- return True
355
- except Exception as e:
356
- self.log_exception(e, "Error writing to destination")
357
- return False
378
+ self.logger.debug(f"Deleted source file: {self.source_file_path}")
379
+ except Exception:
380
+ raise
@@ -1,6 +1,7 @@
1
1
  from abc import ABC, abstractmethod
2
2
  from typing import List, Optional, Union
3
3
  from uuid import uuid4
4
+ from dataclasses import dataclass
4
5
 
5
6
  import pandas as pd
6
7
 
@@ -18,6 +19,20 @@ from .models import (
18
19
  )
19
20
 
20
21
 
22
+ @dataclass
23
+ class PipelineResult:
24
+ """A definition of a pipeline execution outcome with some details"""
25
+
26
+ success: bool
27
+ message: str
28
+ error_code: Optional[str] = None
29
+ exception: Optional[Exception] = None
30
+
31
+ def __bool__(self):
32
+ """Facilitate checking the result directly in if statements: if result: ..."""
33
+ return self.success
34
+
35
+
21
36
  class BasePipeline(LoggedComponent, ABC):
22
37
  """
23
38
  Abstract base class for all pipeline types.
@@ -40,6 +55,7 @@ class BasePipeline(LoggedComponent, ABC):
40
55
 
41
56
  self._active_connectors: List[DatabaseConnector] = []
42
57
  self.audit_db_ops: Optional[DatabaseOperations] = None
58
+ self._init_error: Optional[PipelineResult] = None
43
59
 
44
60
  # Metadata for auditing
45
61
  self.input_rows = 0
@@ -49,7 +65,17 @@ class BasePipeline(LoggedComponent, ABC):
49
65
  self.file_size_bytes: Optional[int] = None
50
66
  self.file_last_modified: Optional[pd.Timestamp] = None
51
67
 
52
- self._initialize_audit_resource()
68
+ try:
69
+ self._initialize_audit_resource()
70
+ except Exception as e:
71
+ # deferred so run() can report it as a PipelineResult instead of
72
+ # raising out of the constructor
73
+ self._init_error = PipelineResult(
74
+ success=False,
75
+ message=f"Pipeline setup failed: {e}",
76
+ error_code=e.__class__.__name__,
77
+ exception=e,
78
+ )
53
79
 
54
80
  def _initialize_audit_resource(self):
55
81
  """Initialize the database operations for auditing."""
@@ -92,9 +118,14 @@ class BasePipeline(LoggedComponent, ABC):
92
118
  return DatabaseOperations(connector.get_engine())
93
119
 
94
120
  def _cleanup(self):
95
- """Dispose of all active database connectors."""
121
+ """Dispose of all active database connectors.
122
+ Never raises, so a disposal failure can't mask the pipeline's real outcome.
123
+ """
96
124
  for connector in self._active_connectors:
97
- connector._dispose_engine()
125
+ try:
126
+ connector._dispose_engine()
127
+ except Exception as e:
128
+ self.logger.warning(f"Failed to dispose connector cleanly: {e}")
98
129
  self.logger.debug("Pipeline cleanup completed.")
99
130
 
100
131
  def _log_audit(self, status: str):
@@ -118,7 +149,7 @@ class BasePipeline(LoggedComponent, ABC):
118
149
  destination_name=getattr(self, "destination_name", None),
119
150
  sp_name=getattr(self, "sp_name", None),
120
151
  sp_parameters=getattr(self, "sp_parameters", None),
121
- timestamp=pd.Timestamp.now(),
152
+ insert_timestamp=pd.Timestamp.now(),
122
153
  error_details=self.error_details,
123
154
  file_name=source_file_path.name if source_file_path else None,
124
155
  file_path=str(source_file_path) if source_file_path else None,
@@ -127,11 +158,49 @@ class BasePipeline(LoggedComponent, ABC):
127
158
  )
128
159
 
129
160
  try:
130
- self.audit_db_ops.write_audit("execution_audit", entry)
161
+ self.audit_db_ops.write_audit("edl_execution_audit", entry)
131
162
  except Exception as e:
132
163
  self.logger.error(f"Failed to write audit log: {str(e)}")
133
164
 
165
+ def run(self) -> PipelineResult:
166
+ """
167
+ Public entry point for pipeline execution.
168
+ Wraps internal execution and returns a PipelineResult object.
169
+ """
170
+ if self._init_error is not None:
171
+ self.logger.warning(f"Pipeline has [FAILED] {self._init_error.message}")
172
+ return self._init_error
173
+
174
+ self.logger.info("Starting pipeline.")
175
+
176
+ try:
177
+ """Call the method implemented by child classes"""
178
+ self._run()
179
+ self.logger.info("Pipeline has finalized with [SUCCESS].")
180
+
181
+ return PipelineResult(
182
+ success=True,
183
+ message="Pipeline executed succesfully.",
184
+ )
185
+ except Exception as e:
186
+ # file log
187
+ self.logger.debug(
188
+ f"Pipeline has encountered an exception: {e}",
189
+ exc_info=True,
190
+ )
191
+
192
+ user_msg = getattr(e, "user_message", str(e))
193
+ # console log
194
+ self.logger.warning(f"Pipeline has [FAILED] {user_msg}")
195
+
196
+ return PipelineResult(
197
+ success=False,
198
+ message="Pipeline has failed",
199
+ error_code=e.__class__.__name__,
200
+ exception=e,
201
+ )
202
+
134
203
  @abstractmethod
135
- def run(self) -> bool:
204
+ def _run(self) -> bool:
136
205
  """Main execution logic to be implemented by child classes."""
137
206
  pass
@@ -6,7 +6,7 @@ from .models import (
6
6
  ServerBasedConnectionSettings,
7
7
  FileBasedConnectionSettings,
8
8
  )
9
- from .pipeline_base import BasePipeline
9
+ from .pipeline_base import BasePipeline, PipelineResult
10
10
 
11
11
 
12
12
  class ProcedurePipeline(BasePipeline):
@@ -14,30 +14,44 @@ class ProcedurePipeline(BasePipeline):
14
14
  Pipeline specialized in executing database stored procedures.
15
15
  """
16
16
 
17
- def __init__(self, pipeline_name: str, orchestrator_id: Optional[str] = None):
17
+ def __init__(self, file_name: str, orchestrator_id: Optional[str] = None):
18
18
  # Load definition from config
19
- definition = Configuration().get_pipeline(pipeline_name)
19
+ definition = Configuration().get_pipeline(file_name)
20
20
  if not isinstance(definition, ProcedureDefinition):
21
21
  raise ValueError(
22
- f"Pipeline '{pipeline_name}' is not a ProcedureDefinition."
22
+ f"Pipeline '{file_name}' does not contain a ProcedureDefinition."
23
23
  )
24
24
 
25
25
  super().__init__(definition, orchestrator_id=orchestrator_id)
26
26
  self.definition: ProcedureDefinition = definition
27
27
 
28
- # Set up primary database connection
29
- resource = self.config.get_resource(self.definition.resource)
30
- if not isinstance(
31
- resource, (ServerBasedConnectionSettings, FileBasedConnectionSettings)
32
- ):
33
- self.log_and_raise(
34
- ValueError,
35
- f"Resource '{self.definition.resource}' must be a database connection.",
36
- )
28
+ if self._init_error is None:
29
+ try:
30
+ # Set up primary database connection
31
+ resource = self.config.get_resource(self.definition.resource)
32
+ if not isinstance(
33
+ resource,
34
+ (ServerBasedConnectionSettings, FileBasedConnectionSettings),
35
+ ):
36
+ raise ValueError(
37
+ f"Resource '{self.definition.resource}' must be a database connection."
38
+ )
37
39
 
38
- self.db_ops = self._setup_db_operations(resource)
40
+ self.db_ops = self._setup_db_operations(resource)
41
+ except Exception as e:
42
+ self.error_details = str(e)
43
+ self._log_audit("FAILED")
44
+ self.logger.error(self.error_details)
45
+ # deferred so run() can report it as a PipelineResult instead of
46
+ # raising out of the constructor
47
+ self._init_error = PipelineResult(
48
+ success=False,
49
+ message=f"Pipeline setup failed: {e}",
50
+ error_code=e.__class__.__name__,
51
+ exception=e,
52
+ )
39
53
 
40
- def run(self) -> bool:
54
+ def _run(self):
41
55
  """Execute all procedures in the definition."""
42
56
  self.logger.info(
43
57
  f">>> Starting Procedure Pipeline: {self.definition.pipeline_name} <<<"
@@ -65,7 +79,7 @@ class ProcedurePipeline(BasePipeline):
65
79
  self.logger.info(
66
80
  f">>> Procedure Pipeline {self.definition.pipeline_name} finished successfully <<<"
67
81
  )
68
- return True
82
+ return
69
83
 
70
84
  except Exception as e:
71
85
  self.error_details = str(e)
@@ -78,6 +92,6 @@ class ProcedurePipeline(BasePipeline):
78
92
  self.sp_parameters = {p[0]: p[1] for p in self.definition.procedures}
79
93
 
80
94
  self._log_audit("FAILED")
81
- return False
95
+ raise
82
96
  finally:
83
97
  self._cleanup()
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: easy_data_loader
3
- Version: 0.1.8
3
+ Version: 0.2.1
4
4
  Summary: Data transfer utilities between files and databases
5
5
  Author-email: Bojoi Gabriel <bojoigabriel@gmail.com>
6
6
  Classifier: Development Status :: 3 - Alpha
@@ -1,4 +1,5 @@
1
1
  import os
2
+ import re
2
3
  import sqlite3
3
4
 
4
5
  import pytest
@@ -138,7 +139,9 @@ def test_validate_resources_reports_ok_and_failed(runner, tmp_path):
138
139
  result = runner.invoke(main, ["validate-resources"])
139
140
  assert result.exit_code == 0
140
141
  assert "Resource: good_folder ... OK (Path Exists)" in result.output
141
- assert "Resource: good_db ... OK (Connected)" in result.output
142
+ # SQLAlchemy's echo logging can interleave stderr noise between the label
143
+ # and status, so match across lines instead of requiring adjacency.
144
+ assert re.search(r"Resource: good_db \.\.\.[\s\S]*?OK \(Connected\)", result.output)
142
145
  assert "Resource: bad_db ... FAILED" in result.output
143
146
 
144
147
 
@@ -179,9 +182,11 @@ def test_validate_pipelines_reports_ok_for_load_procedure_and_orchestrator(
179
182
 
180
183
  result = runner.invoke(main, ["validate-pipelines"])
181
184
  assert result.exit_code == 0
182
- assert "Pipeline: load_pipeline ... OK" in result.output
183
- assert "Pipeline: procedure_pipeline ... OK" in result.output
184
- assert "Pipeline: my_orchestrator ... OK" in result.output
185
+ # SQLAlchemy's echo logging can interleave stderr noise between the label
186
+ # and status, so match across lines instead of requiring adjacency.
187
+ assert re.search(r"Pipeline: load_pipeline \.\.\.[\s\S]*?OK\b", result.output)
188
+ assert re.search(r"Pipeline: procedure_pipeline \.\.\.[\s\S]*?OK\b", result.output)
189
+ assert re.search(r"Pipeline: my_orchestrator \.\.\.[\s\S]*?OK\b", result.output)
185
190
 
186
191
 
187
192
  def test_validate_pipelines_reports_failed_for_unresolvable_resource(runner, tmp_path):
@@ -193,4 +198,4 @@ def test_validate_pipelines_reports_failed_for_unresolvable_resource(runner, tmp
193
198
 
194
199
  result = runner.invoke(main, ["validate-pipelines"])
195
200
  assert result.exit_code == 0
196
- assert "Pipeline: broken_pipeline ... FAILED" in result.output
201
+ assert re.search(r"Pipeline: broken_pipeline \.\.\.[\s\S]*?FAILED", result.output)
@@ -24,7 +24,7 @@ def test_connector_factory_maps_expected_connector_classes():
24
24
 
25
25
 
26
26
  def test_sqlite_connector_builds_expected_connection_string():
27
- with tempfile.TemporaryDirectory() as tmpdir:
27
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
28
28
  db_path = f"{tmpdir}/test.db"
29
29
  settings = FileBasedConnectionSettings(
30
30
  conn_server_type=ServerType.SQLITE, file_path=db_path
@@ -25,7 +25,7 @@ def reset_configuration_singleton():
25
25
 
26
26
  def test_orchestrator_passes_orchestrator_id_and_writes_audit():
27
27
  # Create temporary file paths for our sqlite databases
28
- with tempfile.TemporaryDirectory() as tmpdir:
28
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
29
29
  src_db_path = os.path.join(tmpdir, "src.db")
30
30
  dst_db_path = os.path.join(tmpdir, "dst.db")
31
31
  audit_db_path = os.path.join(tmpdir, "audit.db")
@@ -93,7 +93,7 @@ def test_orchestrator_passes_orchestrator_id_and_writes_audit():
93
93
  assert orchestrator.orchestrator_id is not None
94
94
 
95
95
  success = orchestrator.run()
96
- assert success is True
96
+ assert success.success is True
97
97
 
98
98
  # Verify target tables were created and populated in destination
99
99
  conn_dst = sqlite3.connect(dst_db_path)
@@ -111,9 +111,9 @@ def test_orchestrator_passes_orchestrator_id_and_writes_audit():
111
111
  conn_audit = sqlite3.connect(audit_db_path)
112
112
  cursor_audit = conn_audit.cursor()
113
113
 
114
- # Read the execution_audit table columns and rows
114
+ # Read the edl_execution_audit table columns and rows
115
115
  cursor_audit.execute(
116
- "SELECT pipeline_id, orchestrator_id, pipeline_name, status FROM execution_audit"
116
+ "SELECT pipeline_id, orchestrator_id, pipeline_name, status FROM edl_execution_audit"
117
117
  )
118
118
  rows = cursor_audit.fetchall()
119
119
  assert len(rows) == 2
@@ -31,7 +31,7 @@ def _transform_requires_renamed_column(df):
31
31
 
32
32
 
33
33
  def test_transform_hook_sees_renamed_columns():
34
- with tempfile.TemporaryDirectory() as tmpdir:
34
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
35
35
  src_db_path = os.path.join(tmpdir, "src.db")
36
36
  dst_db_path = os.path.join(tmpdir, "dst.db")
37
37
  audit_db_path = os.path.join(tmpdir, "audit.db")
@@ -72,7 +72,7 @@ def test_transform_hook_sees_renamed_columns():
72
72
  pipeline = LoadPipeline("test_pipeline")
73
73
  success = pipeline.run()
74
74
 
75
- assert success is True
75
+ assert success.success is True
76
76
 
77
77
  conn_dst = sqlite3.connect(dst_db_path)
78
78
  cursor_dst = conn_dst.cursor()
@@ -60,7 +60,7 @@ def test_execution_mode_defaults_to_sequential():
60
60
 
61
61
 
62
62
  def test_sequential_mode_runs_all_procedures_when_each_succeeds():
63
- with tempfile.TemporaryDirectory() as tmpdir:
63
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
64
64
  pipeline = _build_pipeline(tmpdir, execution_mode="sequential")
65
65
 
66
66
  recorded_calls = []
@@ -72,12 +72,12 @@ def test_sequential_mode_runs_all_procedures_when_each_succeeds():
72
72
  pipeline.db_ops.execute_stored_procedure = _record_call # type: ignore[method-assign]
73
73
 
74
74
  success = pipeline.run()
75
- assert success is True
75
+ assert success.success is True
76
76
  assert [c[0] for c in recorded_calls] == ["proc_a", "proc_b"]
77
77
 
78
78
 
79
79
  def test_loop_stops_after_first_failed_procedure():
80
- with tempfile.TemporaryDirectory() as tmpdir:
80
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
81
81
  pipeline = _build_pipeline(tmpdir)
82
82
 
83
83
  recorded_calls = []
@@ -29,7 +29,7 @@ def reset_configuration_singleton():
29
29
 
30
30
 
31
31
  def test_read_data_returns_dataframe_for_multiple_rows():
32
- with tempfile.TemporaryDirectory() as tmpdir:
32
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
33
33
  db_path = os.path.join(tmpdir, "db.sqlite")
34
34
  conn = sqlite3.connect(db_path)
35
35
  conn.execute("CREATE TABLE t (id INTEGER)")
@@ -48,7 +48,7 @@ def test_read_data_returns_dataframe_for_multiple_rows():
48
48
 
49
49
 
50
50
  def test_read_data_returns_dict_for_single_row():
51
- with tempfile.TemporaryDirectory() as tmpdir:
51
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
52
52
  db_path = os.path.join(tmpdir, "db.sqlite")
53
53
  conn = sqlite3.connect(db_path)
54
54
  conn.execute("CREATE TABLE t (id INTEGER, name TEXT)")
@@ -66,7 +66,7 @@ def test_read_data_returns_dict_for_single_row():
66
66
 
67
67
 
68
68
  def test_read_data_returns_empty_dataframe_for_no_rows():
69
- with tempfile.TemporaryDirectory() as tmpdir:
69
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
70
70
  db_path = os.path.join(tmpdir, "db.sqlite")
71
71
  conn = sqlite3.connect(db_path)
72
72
  conn.execute("CREATE TABLE t (id INTEGER)")
@@ -138,7 +138,7 @@ def test_procedure_definition_default_build_parameters_is_empty():
138
138
 
139
139
 
140
140
  def test_procedure_pipeline_merges_dynamic_parameters():
141
- with tempfile.TemporaryDirectory() as tmpdir:
141
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
142
142
  db_path = os.path.join(tmpdir, "db.sqlite")
143
143
 
144
144
  config = Configuration()
@@ -173,7 +173,7 @@ def test_procedure_pipeline_merges_dynamic_parameters():
173
173
  pipeline.db_ops.execute_stored_procedure = _record_call # type: ignore[method-assign]
174
174
 
175
175
  success = pipeline.run()
176
- assert success is True
176
+ assert success.success is True
177
177
  assert recorded_calls == [
178
178
  ("insert_call", {"name": "static_name", "value": "dynamic_value"})
179
179
  ]
@@ -27,7 +27,7 @@ def reset_configuration_singleton():
27
27
 
28
28
 
29
29
  def test_source_file_deleted_after_successful_load_to_db():
30
- with tempfile.TemporaryDirectory() as tmpdir:
30
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
31
31
  (Path(tmpdir) / "data.csv").write_text("a\n1\n2\n")
32
32
  dst_db_path = os.path.join(tmpdir, "dst.db")
33
33
 
@@ -52,12 +52,12 @@ def test_source_file_deleted_after_successful_load_to_db():
52
52
  pipeline = LoadPipeline("test_pipeline")
53
53
  success = pipeline.run()
54
54
 
55
- assert success is True
55
+ assert success.success is True
56
56
  assert not (Path(tmpdir) / "data.csv").exists()
57
57
 
58
58
 
59
59
  def test_file_post_process_runs_on_source_when_destination_is_db():
60
- with tempfile.TemporaryDirectory() as tmpdir:
60
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
61
61
  (Path(tmpdir) / "data.csv").write_text("a\n1\n2\n")
62
62
  dst_db_path = os.path.join(tmpdir, "dst.db")
63
63
 
@@ -86,12 +86,12 @@ def test_file_post_process_runs_on_source_when_destination_is_db():
86
86
  pipeline = LoadPipeline("test_pipeline")
87
87
  success = pipeline.run()
88
88
 
89
- assert success is True
89
+ assert success.success is True
90
90
  assert post_processed_paths == [Path(tmpdir) / "data.csv"]
91
91
 
92
92
 
93
93
  def test_source_file_delete_after_load_raises_when_source_is_not_a_file():
94
- with tempfile.TemporaryDirectory() as tmpdir:
94
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
95
95
  src_db_path = os.path.join(tmpdir, "src.db")
96
96
  conn = sqlite3.connect(src_db_path)
97
97
  conn.execute("CREATE TABLE source_table (a INTEGER)")
@@ -118,5 +118,10 @@ def test_source_file_delete_after_load_raises_when_source_is_not_a_file():
118
118
  )
119
119
  config.pipelines["test_pipeline"] = pipeline_def
120
120
 
121
- with pytest.raises(ValueError, match="source_file_delete_after_load"):
122
- LoadPipeline("test_pipeline")
121
+ # Setup failures are deferred to run() rather than raised from the
122
+ # constructor, so they can be reported as a PipelineResult.
123
+ pipeline = LoadPipeline("test_pipeline")
124
+ result = pipeline.run()
125
+
126
+ assert result.success is False
127
+ assert "source_file_delete_after_load" in result.message
@@ -34,7 +34,7 @@ def reset_configuration_singleton():
34
34
 
35
35
 
36
36
  def test_validation_fail_false_keeps_valid_rows():
37
- with tempfile.TemporaryDirectory() as tmpdir:
37
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
38
38
  src_db_path = os.path.join(tmpdir, "src.db")
39
39
  dst_db_path = os.path.join(tmpdir, "dst.db")
40
40
  audit_db_path = os.path.join(tmpdir, "audit.db")
@@ -87,7 +87,7 @@ def test_validation_fail_false_keeps_valid_rows():
87
87
  success = pipeline.run()
88
88
 
89
89
  # Should be successful because validation_fail=False
90
- assert success is True
90
+ assert success.success is True
91
91
 
92
92
  # Verify destination table only has valid rows (1 and 3)
93
93
  conn_dst = sqlite3.connect(dst_db_path)
@@ -101,7 +101,7 @@ def test_validation_fail_false_keeps_valid_rows():
101
101
  assert rows[1] == (3, "Charlie")
102
102
 
103
103
  cursor_dst = sqlite3.connect(dst_db_path).cursor()
104
- cursor_dst.execute("SELECT error FROM target_table_invalid")
104
+ cursor_dst.execute("SELECT error FROM edl_validation_error_log")
105
105
  invalid_rows = cursor_dst.fetchall()
106
106
  cursor_dst.connection.close()
107
107
 
@@ -111,7 +111,7 @@ def test_validation_fail_false_keeps_valid_rows():
111
111
 
112
112
 
113
113
  def test_validation_writes_valid_and_invalid_file_outputs():
114
- with tempfile.TemporaryDirectory() as tmpdir:
114
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
115
115
  src_db_path = os.path.join(tmpdir, "src.db")
116
116
  audit_db_path = os.path.join(tmpdir, "audit.db")
117
117
 
@@ -146,7 +146,7 @@ def test_validation_writes_valid_and_invalid_file_outputs():
146
146
 
147
147
  success = LoadPipeline("test_pipeline").run()
148
148
 
149
- assert success is True
149
+ assert success.success is True
150
150
 
151
151
  valid_df = pd.read_csv(Path(tmpdir) / "output.csv")
152
152
  invalid_df = pd.read_csv(Path(tmpdir) / "output_invalid.csv")
@@ -157,7 +157,7 @@ def test_validation_writes_valid_and_invalid_file_outputs():
157
157
 
158
158
 
159
159
  def test_validation_fail_true_fails_pipeline():
160
- with tempfile.TemporaryDirectory() as tmpdir:
160
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
161
161
  src_db_path = os.path.join(tmpdir, "src.db")
162
162
  dst_db_path = os.path.join(tmpdir, "dst.db")
163
163
  audit_db_path = os.path.join(tmpdir, "audit.db")
@@ -206,14 +206,14 @@ def test_validation_fail_true_fails_pipeline():
206
206
  success = pipeline.run()
207
207
 
208
208
  # Should fail because validation_fail=True
209
- assert success is False
209
+ assert success.success is False
210
210
  assert pipeline.error_details is not None
211
211
  assert "Validation failed" in pipeline.error_details
212
212
  assert "Row 1:" in pipeline.error_details # Row 2 (0-indexed row 1) failed
213
213
 
214
214
 
215
215
  def test_validation_fail_false_graceful_stop_when_all_rows_invalid():
216
- with tempfile.TemporaryDirectory() as tmpdir:
216
+ with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
217
217
  src_db_path = os.path.join(tmpdir, "src.db")
218
218
  dst_db_path = os.path.join(tmpdir, "dst.db")
219
219
  audit_db_path = os.path.join(tmpdir, "audit.db")
@@ -261,8 +261,10 @@ def test_validation_fail_false_graceful_stop_when_all_rows_invalid():
261
261
  pipeline = LoadPipeline("test_pipeline")
262
262
  success = pipeline.run()
263
263
 
264
- # Should stop gracefully and return True because validation_fail=False
265
- assert success is True
264
+ # validation_fail=False keeps invalid rows out of the load, but if that
265
+ # leaves no data at all the pipeline still fails with an EmptyDataError.
266
+ assert success.success is False
267
+ assert success.error_code == "EmptyDataError"
266
268
  assert pipeline.output_rows == 0
267
269
 
268
270
  # Destination table should NOT have been loaded or have any data