DataExcept 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dataexcept/__init__.py +81 -0
- dataexcept/__main__.py +74 -0
- dataexcept/database_exceptions.py +71 -0
- dataexcept/dataengineering_exceptions.py +125 -0
- dataexcept/datascience_exceptions/__init__.py +85 -0
- dataexcept/datascience_exceptions/base.py +17 -0
- dataexcept/datascience_exceptions/ingestion.py +312 -0
- dataexcept/datascience_exceptions/operations.py +130 -0
- dataexcept/datascience_exceptions/training.py +508 -0
- dataexcept/exceptions/__init__.py +36 -0
- dataexcept/exceptions/authentication.py +21 -0
- dataexcept/exceptions/base.py +4 -0
- dataexcept/exceptions/configuration.py +10 -0
- dataexcept/exceptions/external.py +42 -0
- dataexcept/exceptions/lifecycle.py +13 -0
- dataexcept/exceptions/notification.py +49 -0
- dataexcept/exceptions/parsing.py +31 -0
- dataexcept/exceptions/scheduling.py +21 -0
- dataexcept/exceptions/validation.py +12 -0
- dataexcept/io_exceptions.py +66 -0
- dataexcept/job_exceptions.py +59 -0
- dataexcept/logging_helpers.py +76 -0
- dataexcept/network_exceptions.py +102 -0
- dataexcept/pandas_exceptions.py +143 -0
- dataexcept/pipeline_exceptions.py +205 -0
- dataexcept/py.typed +1 -0
- dataexcept/security_exceptions.py +68 -0
- dataexcept-0.1.0.dist-info/METADATA +397 -0
- dataexcept-0.1.0.dist-info/RECORD +32 -0
- dataexcept-0.1.0.dist-info/WHEEL +4 -0
- dataexcept-0.1.0.dist-info/entry_points.txt +3 -0
- dataexcept-0.1.0.dist-info/licenses/LICENSE +22 -0
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
"""Additional exception classes for data pipeline workflows."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any, Optional
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class PipelineError(Exception):
|
|
9
|
+
"""Base exception for pipeline errors."""
|
|
10
|
+
|
|
11
|
+
pass
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class PreprocessingError(PipelineError):
|
|
15
|
+
"""Raised when a preprocessing step fails."""
|
|
16
|
+
|
|
17
|
+
def __init__(self, step_name: str, details: Optional[str] = None) -> None:
|
|
18
|
+
default = f"Preprocessing failed at step: '{step_name}'."
|
|
19
|
+
message = f"{default} Details: {details}" if details else default
|
|
20
|
+
self.step_name = step_name
|
|
21
|
+
self.details = details
|
|
22
|
+
super().__init__(message)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class FeatureEngineeringError(PreprocessingError):
|
|
26
|
+
"""Raised when feature engineering fails."""
|
|
27
|
+
|
|
28
|
+
def __init__(self, feature: str, reason: Optional[str] = None) -> None:
|
|
29
|
+
super().__init__(step_name=f"feature_{feature}", details=reason)
|
|
30
|
+
self.feature = feature
|
|
31
|
+
self.reason = reason
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class StorageError(PipelineError):
|
|
35
|
+
"""Raised when reading from or writing to storage fails."""
|
|
36
|
+
|
|
37
|
+
def __init__(
|
|
38
|
+
self,
|
|
39
|
+
location: str,
|
|
40
|
+
operation: str,
|
|
41
|
+
message: Optional[str] = None,
|
|
42
|
+
) -> None:
|
|
43
|
+
default = f"Storage {operation} failed at location: '{location}'."
|
|
44
|
+
self.location = location
|
|
45
|
+
self.operation = operation
|
|
46
|
+
super().__init__(message or default)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class PipelineNotificationError(PipelineError):
|
|
50
|
+
"""Raised when sending a notification fails."""
|
|
51
|
+
|
|
52
|
+
def __init__(
|
|
53
|
+
self,
|
|
54
|
+
channel: str,
|
|
55
|
+
payload: Any,
|
|
56
|
+
message: Optional[str] = None,
|
|
57
|
+
) -> None:
|
|
58
|
+
default = f"Notification via '{channel}' failed."
|
|
59
|
+
self.channel = channel
|
|
60
|
+
self.payload = payload
|
|
61
|
+
super().__init__(message or default)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class RetryLimitExceededError(PipelineError):
|
|
65
|
+
"""Raised when an operation is retried too many times."""
|
|
66
|
+
|
|
67
|
+
def __init__(
|
|
68
|
+
self,
|
|
69
|
+
operation: str,
|
|
70
|
+
retries: int,
|
|
71
|
+
message: Optional[str] = None,
|
|
72
|
+
) -> None:
|
|
73
|
+
default = (
|
|
74
|
+
"Retry limit exceeded for operation "
|
|
75
|
+
f"'{operation}' after {retries} attempts."
|
|
76
|
+
)
|
|
77
|
+
self.operation = operation
|
|
78
|
+
self.retries = retries
|
|
79
|
+
super().__init__(message or default)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class ExternalServiceError(PipelineError):
|
|
83
|
+
"""General failure when calling an external service."""
|
|
84
|
+
|
|
85
|
+
def __init__(
|
|
86
|
+
self,
|
|
87
|
+
service_name: str,
|
|
88
|
+
status_code: Optional[int] = None,
|
|
89
|
+
response: Optional[Any] = None,
|
|
90
|
+
message: Optional[str] = None,
|
|
91
|
+
) -> None:
|
|
92
|
+
default = f"Call to external service '{service_name}' failed."
|
|
93
|
+
self.service_name = service_name
|
|
94
|
+
self.status_code = status_code
|
|
95
|
+
self.response = response
|
|
96
|
+
super().__init__(message or default)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
class ServiceAuthenticationError(ExternalServiceError):
|
|
100
|
+
"""Authentication to an external service failed."""
|
|
101
|
+
|
|
102
|
+
def __init__(
|
|
103
|
+
self,
|
|
104
|
+
service_name: str,
|
|
105
|
+
message: Optional[str] = None,
|
|
106
|
+
) -> None:
|
|
107
|
+
default = f"Authentication failed for service '{service_name}'."
|
|
108
|
+
super().__init__(service_name=service_name, message=message or default)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class ServiceAuthorizationError(ExternalServiceError):
|
|
112
|
+
"""Authorization was denied by an external service."""
|
|
113
|
+
|
|
114
|
+
def __init__(
|
|
115
|
+
self,
|
|
116
|
+
service_name: str,
|
|
117
|
+
message: Optional[str] = None,
|
|
118
|
+
) -> None:
|
|
119
|
+
default = f"Authorization denied for service '{service_name}'."
|
|
120
|
+
super().__init__(service_name=service_name, message=message or default)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
class ServiceTimeoutError(ExternalServiceError):
|
|
124
|
+
"""A call to an external service exceeded the allotted time."""
|
|
125
|
+
|
|
126
|
+
def __init__(
|
|
127
|
+
self,
|
|
128
|
+
service_name: str,
|
|
129
|
+
timeout_seconds: Optional[float] = None,
|
|
130
|
+
) -> None:
|
|
131
|
+
default = (
|
|
132
|
+
"Operation timed out after "
|
|
133
|
+
f"{timeout_seconds}s on service '{service_name}'."
|
|
134
|
+
)
|
|
135
|
+
self.timeout_seconds = timeout_seconds
|
|
136
|
+
super().__init__(service_name=service_name, message=default)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class ApiError(PipelineError):
|
|
140
|
+
"""Failure calling a REST API endpoint."""
|
|
141
|
+
|
|
142
|
+
def __init__(
|
|
143
|
+
self,
|
|
144
|
+
endpoint: str,
|
|
145
|
+
status_code: Optional[int] = None,
|
|
146
|
+
message: Optional[str] = None,
|
|
147
|
+
) -> None:
|
|
148
|
+
default = f"API call failed: {endpoint}"
|
|
149
|
+
if status_code is not None:
|
|
150
|
+
default += f" (status {status_code})"
|
|
151
|
+
self.endpoint = endpoint
|
|
152
|
+
self.status_code = status_code
|
|
153
|
+
super().__init__(message or default)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
class TimeDeltaTooLargeError(PipelineError):
|
|
157
|
+
"""The time span between records exceeded a threshold."""
|
|
158
|
+
|
|
159
|
+
def __init__(
|
|
160
|
+
self,
|
|
161
|
+
user: str,
|
|
162
|
+
delta_minutes: float,
|
|
163
|
+
message: Optional[str] = None,
|
|
164
|
+
) -> None:
|
|
165
|
+
default = f"Time delta {delta_minutes}m too large for user {user}"
|
|
166
|
+
self.user = user
|
|
167
|
+
self.delta_minutes = delta_minutes
|
|
168
|
+
super().__init__(message or default)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
class TypeCheckError(PipelineError):
|
|
172
|
+
"""Invalid type detected during recursive type inspection."""
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
class DataFetchError(PipelineError):
|
|
176
|
+
"""Failed to fetch data from a storage backend."""
|
|
177
|
+
|
|
178
|
+
def __init__(
|
|
179
|
+
self,
|
|
180
|
+
source: str,
|
|
181
|
+
cid: str,
|
|
182
|
+
message: Optional[str] = None,
|
|
183
|
+
) -> None:
|
|
184
|
+
default = f"Failed to fetch '{source}' data for cid={cid}"
|
|
185
|
+
self.source = source
|
|
186
|
+
self.cid = cid
|
|
187
|
+
super().__init__(message or default)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
__all__ = [
|
|
191
|
+
"PipelineError",
|
|
192
|
+
"PreprocessingError",
|
|
193
|
+
"FeatureEngineeringError",
|
|
194
|
+
"StorageError",
|
|
195
|
+
"PipelineNotificationError",
|
|
196
|
+
"RetryLimitExceededError",
|
|
197
|
+
"ExternalServiceError",
|
|
198
|
+
"ServiceAuthenticationError",
|
|
199
|
+
"ServiceAuthorizationError",
|
|
200
|
+
"ServiceTimeoutError",
|
|
201
|
+
"ApiError",
|
|
202
|
+
"TimeDeltaTooLargeError",
|
|
203
|
+
"TypeCheckError",
|
|
204
|
+
"DataFetchError",
|
|
205
|
+
]
|
dataexcept/py.typed
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Custom exceptions for security-related operations."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class SecurityError(Exception):
|
|
7
|
+
"""Base exception for security errors."""
|
|
8
|
+
|
|
9
|
+
pass
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class EncryptionError(SecurityError):
|
|
13
|
+
"""Raised when data encryption fails."""
|
|
14
|
+
|
|
15
|
+
def __init__(self, algorithm: str, message: str | None = None) -> None:
|
|
16
|
+
"""Initialize EncryptionError.
|
|
17
|
+
|
|
18
|
+
Args:
|
|
19
|
+
algorithm: Name of the encryption algorithm.
|
|
20
|
+
message: Optional custom error message.
|
|
21
|
+
"""
|
|
22
|
+
self.algorithm = algorithm
|
|
23
|
+
default = f"Encryption failed using {algorithm}"
|
|
24
|
+
super().__init__(message or default)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class DecryptionError(SecurityError):
|
|
28
|
+
"""Raised when data decryption fails."""
|
|
29
|
+
|
|
30
|
+
def __init__(self, algorithm: str, message: str | None = None) -> None:
|
|
31
|
+
"""Initialize DecryptionError.
|
|
32
|
+
|
|
33
|
+
Args:
|
|
34
|
+
algorithm: Name of the decryption algorithm.
|
|
35
|
+
message: Optional custom error message.
|
|
36
|
+
"""
|
|
37
|
+
self.algorithm = algorithm
|
|
38
|
+
default = f"Decryption failed using {algorithm}"
|
|
39
|
+
super().__init__(message or default)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class InvalidTokenError(SecurityError):
|
|
43
|
+
"""Raised when an authentication token is invalid or expired."""
|
|
44
|
+
|
|
45
|
+
def __init__(
|
|
46
|
+
self,
|
|
47
|
+
token: str | None = None,
|
|
48
|
+
message: str | None = None,
|
|
49
|
+
) -> None:
|
|
50
|
+
"""Initialize InvalidTokenError.
|
|
51
|
+
|
|
52
|
+
Args:
|
|
53
|
+
token: The problematic token.
|
|
54
|
+
message: Optional custom error message.
|
|
55
|
+
"""
|
|
56
|
+
self.token = token
|
|
57
|
+
default = "Invalid authentication token"
|
|
58
|
+
if token:
|
|
59
|
+
default += f": {token}"
|
|
60
|
+
super().__init__(message or default)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
__all__ = [
|
|
64
|
+
"SecurityError",
|
|
65
|
+
"EncryptionError",
|
|
66
|
+
"DecryptionError",
|
|
67
|
+
"InvalidTokenError",
|
|
68
|
+
]
|
|
@@ -0,0 +1,397 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: DataExcept
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A Python package providing structured, easily-extendable custom exception types.
|
|
5
|
+
License: MIT
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Keywords: exceptions,errors,logging
|
|
8
|
+
Author: Diogo Ribeiro
|
|
9
|
+
Author-email: dfr@esmad.ipp.pt
|
|
10
|
+
Maintainer: Diogo Ribeiro
|
|
11
|
+
Maintainer-email: diogo.debastos.ribeiro@gmail.com
|
|
12
|
+
Requires-Python: >=3.10,<3.14
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
20
|
+
Requires-Dist: tomli ; python_version < "3.11"
|
|
21
|
+
Project-URL: Changelog, https://github.com/DiogoRibeiro7/DataExcept/releases
|
|
22
|
+
Project-URL: Documentation, https://diogoribeiro7.github.io/DataExcept/
|
|
23
|
+
Project-URL: Homepage, https://github.com/DiogoRibeiro7/DataExcept
|
|
24
|
+
Project-URL: Issues, https://github.com/DiogoRibeiro7/DataExcept/issues
|
|
25
|
+
Project-URL: Repository, https://github.com/DiogoRibeiro7/DataExcept
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
28
|
+
# DataExcept
|
|
29
|
+
|
|
30
|
+
[](https://github.com/DiogoRibeiro7/DataExcept/actions/workflows/ci.yml) [](https://badge.fury.io/py/DataExcept) [](https://pypi.org/project/DataExcept/) [](https://diogoribeiro7.github.io/DataExcept/htmlcov/) [](https://diogoribeiro7.github.io/DataExcept/) [](https://opensource.org/licenses/MIT) [](https://github.com/psf/black) [](https://mypy-lang.org/)
|
|
31
|
+
|
|
32
|
+
**DataExcept** is a production-ready Python library that provides **structured, hierarchical exception classes** specifically designed for **data science**, **machine learning**, and **data engineering** workflows. Stop debugging generic `ValueError`s and `RuntimeError`s -- get meaningful, actionable error messages that help you understand exactly what went wrong in your data pipeline.
|
|
33
|
+
|
|
34
|
+
## ๐ Why DataExcept?
|
|
35
|
+
|
|
36
|
+
โ Without DataExcept | โ
With DataExcept
|
|
37
|
+
------------------------------- | -------------------------------------------------------------------------------------
|
|
38
|
+
`ValueError: Invalid value` | `DataValidationError: Invalid value for 'age': -1`
|
|
39
|
+
`RuntimeError: Training failed` | `ConvergenceError: Model 'RandomForest' failed to converge after 100 iterations`
|
|
40
|
+
`Exception: Prediction error` | `ModelInferenceError: Inference failed for model 'CNN': CUDA out of memory`
|
|
41
|
+
`KeyError: column not found` | `MissingColumnError: Missing required column 'customer_id' in DataFrame 'sales_data'`
|
|
42
|
+
|
|
43
|
+
## ๐ฏ Key Features
|
|
44
|
+
|
|
45
|
+
- **๐๏ธ Hierarchical Structure**: Catch specific errors or broad categories
|
|
46
|
+
- **๐ Data Science Focused**: 40+ exceptions covering ML pipelines, feature engineering, model training
|
|
47
|
+
- **๐ง Production Ready**: Comprehensive logging helpers and error context
|
|
48
|
+
- **๐ Academic Quality**: Proper documentation, type hints, and citation support
|
|
49
|
+
- **๐ Python 3.10+**: Modern Python with full type safety
|
|
50
|
+
- **๐งช Well Tested**: Broad test suite with comprehensive edge case handling (see the coverage badge above)
|
|
51
|
+
|
|
52
|
+
## ๐ฆ Quick Installation
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
pip install DataExcept
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
For development:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
git clone https://github.com/DiogoRibeiro7/DataExcept.git
|
|
62
|
+
cd DataExcept
|
|
63
|
+
poetry install
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
## ๐โโ๏ธ Quick Start
|
|
67
|
+
|
|
68
|
+
### Basic Usage
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
from dataexcept import ValidationError, ModelTrainingError
|
|
72
|
+
from dataexcept.datascience_exceptions import DataLoadingError
|
|
73
|
+
import pandas as pd
|
|
74
|
+
|
|
75
|
+
# Data validation with context
|
|
76
|
+
def validate_dataframe(df: pd.DataFrame) -> None:
|
|
77
|
+
if 'customer_id' not in df.columns:
|
|
78
|
+
raise ValidationError(
|
|
79
|
+
field='customer_id',
|
|
80
|
+
value=list(df.columns),
|
|
81
|
+
message="Customer ID column is required for processing"
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
# Model training with specific error types
|
|
85
|
+
def train_model(model_type: str, epochs: int) -> None:
|
|
86
|
+
try:
|
|
87
|
+
# Your training code here
|
|
88
|
+
if epochs > 1000:
|
|
89
|
+
raise ModelTrainingError(
|
|
90
|
+
model_type=model_type,
|
|
91
|
+
epoch=epochs,
|
|
92
|
+
message=f"Training {model_type} exceeded reasonable epoch limit"
|
|
93
|
+
)
|
|
94
|
+
except Exception as e:
|
|
95
|
+
# Wrap unknown errors with context
|
|
96
|
+
raise ModelTrainingError(model_type, message=f"Unexpected error: {e}")
|
|
97
|
+
|
|
98
|
+
# File operations with detailed context
|
|
99
|
+
def load_dataset(file_path: str) -> pd.DataFrame:
|
|
100
|
+
try:
|
|
101
|
+
return pd.read_csv(file_path)
|
|
102
|
+
except FileNotFoundError as e:
|
|
103
|
+
raise DataLoadingError(source=file_path, original=e)
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
### Exception Hierarchies
|
|
107
|
+
|
|
108
|
+
```python
|
|
109
|
+
from dataexcept import JobError
|
|
110
|
+
from dataexcept.datascience_exceptions import ModelTrainingError, ConvergenceError
|
|
111
|
+
|
|
112
|
+
try:
|
|
113
|
+
# Your ML pipeline
|
|
114
|
+
train_complex_model()
|
|
115
|
+
except ConvergenceError:
|
|
116
|
+
# Handle specific convergence issues
|
|
117
|
+
logger.warning("Model didn't converge, trying with different parameters")
|
|
118
|
+
train_with_fallback_params()
|
|
119
|
+
except ModelTrainingError:
|
|
120
|
+
# Handle any training-related error
|
|
121
|
+
logger.error("Training failed, falling back to simpler model")
|
|
122
|
+
train_simple_model()
|
|
123
|
+
except JobError:
|
|
124
|
+
# Handle any job-related error
|
|
125
|
+
logger.error("Job failed, notifying administrators")
|
|
126
|
+
send_alert()
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
## ๐๏ธ Exception Categories
|
|
130
|
+
|
|
131
|
+
### ๐ Data Science & ML
|
|
132
|
+
|
|
133
|
+
```python
|
|
134
|
+
from dataexcept.datascience_exceptions import *
|
|
135
|
+
|
|
136
|
+
# Data ingestion and validation
|
|
137
|
+
DataLoadingError("data.csv", FileNotFoundError())
|
|
138
|
+
DataValidationError("age", -5, "Age cannot be negative")
|
|
139
|
+
MissingDataError("income", "Required for credit scoring")
|
|
140
|
+
|
|
141
|
+
# Feature engineering and preprocessing
|
|
142
|
+
FeatureEngineeringError("log_transform", "Cannot take log of negative values")
|
|
143
|
+
DataNormalizationError("StandardScaler", "Division by zero in variance calculation")
|
|
144
|
+
DataImbalanceError(ratio=0.05, threshold=0.1)
|
|
145
|
+
|
|
146
|
+
# Model training and evaluation
|
|
147
|
+
ModelTrainingError("RandomForest", epoch=45)
|
|
148
|
+
ConvergenceError("GradientBoosting", iterations=1000)
|
|
149
|
+
OverfittingError(train_metric=0.98, val_metric=0.65)
|
|
150
|
+
BiasDetectionError("gender", bias_score=0.15, threshold=0.1)
|
|
151
|
+
|
|
152
|
+
# Model deployment and inference
|
|
153
|
+
ModelInferenceError("CNN", RuntimeError("CUDA out of memory"))
|
|
154
|
+
ModelCompatibilityError("2.1.0", "1.8.0")
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
### ๐ง Data Engineering & ETL
|
|
158
|
+
|
|
159
|
+
```python
|
|
160
|
+
from dataexcept.dataengineering_exceptions import *
|
|
161
|
+
|
|
162
|
+
ETLJobError("daily_customer_pipeline")
|
|
163
|
+
SchemaEvolutionError("v2.1", reason="Incompatible column type change")
|
|
164
|
+
DataTransformationError("currency_conversion", "Invalid exchange rate")
|
|
165
|
+
BatchProcessingError("batch_2023_11_13", original=TimeoutError())
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
### ๐ผ Pandas Operations
|
|
169
|
+
|
|
170
|
+
```python
|
|
171
|
+
from dataexcept.pandas_exceptions import *
|
|
172
|
+
|
|
173
|
+
MissingColumnError("customer_id", dataframe="sales_df")
|
|
174
|
+
DtypeMismatchError("revenue", expected=["float64", "int64"], found="object")
|
|
175
|
+
MergeKeyError(["customer_id"], ["cust_id"])
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
### ๐ Infrastructure & Networking
|
|
179
|
+
|
|
180
|
+
```python
|
|
181
|
+
from dataexcept.network_exceptions import *
|
|
182
|
+
from dataexcept.database_exceptions import *
|
|
183
|
+
|
|
184
|
+
HostUnreachableError("api.example.com")
|
|
185
|
+
DatabaseConnectionError("postgresql://prod-db:5432/analytics")
|
|
186
|
+
QueryExecutionError("SELECT * FROM large_table", original=TimeoutError())
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
## ๐ Advanced Features
|
|
190
|
+
|
|
191
|
+
### Smart Logging Integration
|
|
192
|
+
|
|
193
|
+
```python
|
|
194
|
+
from dataexcept.logging_helpers import log_and_raise, log_exception
|
|
195
|
+
import logging
|
|
196
|
+
|
|
197
|
+
logger = logging.getLogger(__name__)
|
|
198
|
+
|
|
199
|
+
# Context manager for automatic logging
|
|
200
|
+
with log_and_raise(logger=logger, context={"job_id": "ETL_001", "batch": "2023-11-13"}):
|
|
201
|
+
process_daily_batch()
|
|
202
|
+
|
|
203
|
+
# Manual exception logging with context
|
|
204
|
+
try:
|
|
205
|
+
risky_operation()
|
|
206
|
+
except Exception as exc:
|
|
207
|
+
log_exception(
|
|
208
|
+
exc,
|
|
209
|
+
logger=logger,
|
|
210
|
+
context={"user_id": "12345", "operation": "feature_extraction"}
|
|
211
|
+
)
|
|
212
|
+
raise
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
### Command Line Interface
|
|
216
|
+
|
|
217
|
+
```bash
|
|
218
|
+
# List all available exception classes
|
|
219
|
+
$ dataexcept list
|
|
220
|
+
JobError
|
|
221
|
+
ValidationError
|
|
222
|
+
DataScienceError
|
|
223
|
+
ModelTrainingError
|
|
224
|
+
... (40+ more)
|
|
225
|
+
|
|
226
|
+
# Check version
|
|
227
|
+
$ dataexcept --version
|
|
228
|
+
dataexcept 0.1.0
|
|
229
|
+
```
|
|
230
|
+
|
|
231
|
+
## ๐ฏ Use Cases
|
|
232
|
+
|
|
233
|
+
### ๐ญ Production ML Pipelines
|
|
234
|
+
|
|
235
|
+
- **Model Training**: Distinguish between convergence issues, data problems, and infrastructure failures
|
|
236
|
+
- **Feature Engineering**: Track which transformation steps fail and why
|
|
237
|
+
- **Model Serving**: Provide actionable error messages for inference failures
|
|
238
|
+
- **Data Drift**: Alert when model assumptions are violated
|
|
239
|
+
|
|
240
|
+
### ๐ Data Engineering
|
|
241
|
+
|
|
242
|
+
- **ETL Pipelines**: Clear error categorization for debugging complex data flows
|
|
243
|
+
- **Data Quality**: Structured validation errors with field-level context
|
|
244
|
+
- **Schema Evolution**: Track migration failures and compatibility issues
|
|
245
|
+
- **Batch Processing**: Identify whether failures are data-related or system-related
|
|
246
|
+
|
|
247
|
+
### ๐ฌ Research & Academia
|
|
248
|
+
|
|
249
|
+
- **Reproducible Experiments**: Consistent error handling across research codebases
|
|
250
|
+
- **Citation Support**: Proper academic attribution with CITATION.cff
|
|
251
|
+
- **Documentation**: Auto-generated API docs with comprehensive examples
|
|
252
|
+
|
|
253
|
+
## ๐ Real-World Example
|
|
254
|
+
|
|
255
|
+
```python
|
|
256
|
+
"""
|
|
257
|
+
Complete ML pipeline with DataExcept error handling
|
|
258
|
+
"""
|
|
259
|
+
import pandas as pd
|
|
260
|
+
from sklearn.ensemble import RandomForestClassifier
|
|
261
|
+
from dataexcept import ValidationError
|
|
262
|
+
from dataexcept.datascience_exceptions import *
|
|
263
|
+
from dataexcept.pandas_exceptions import *
|
|
264
|
+
from dataexcept.logging_helpers import log_and_raise
|
|
265
|
+
import logging
|
|
266
|
+
|
|
267
|
+
def ml_pipeline(data_path: str, target_col: str):
|
|
268
|
+
logger = logging.getLogger(__name__)
|
|
269
|
+
|
|
270
|
+
with log_and_raise(logger=logger, context={"pipeline": "customer_churn"}):
|
|
271
|
+
# 1\. Data Loading
|
|
272
|
+
try:
|
|
273
|
+
df = pd.read_csv(data_path)
|
|
274
|
+
except FileNotFoundError as e:
|
|
275
|
+
raise DataLoadingError(source=data_path, original=e)
|
|
276
|
+
|
|
277
|
+
# 2\. Data Validation
|
|
278
|
+
if target_col not in df.columns:
|
|
279
|
+
raise MissingColumnError(target_col, dataframe="training_data")
|
|
280
|
+
|
|
281
|
+
if df[target_col].dtype not in ['int64', 'bool']:
|
|
282
|
+
raise DtypeMismatchError(
|
|
283
|
+
target_col,
|
|
284
|
+
expected=['int64', 'bool'],
|
|
285
|
+
found=str(df[target_col].dtype)
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
# 3\. Data Quality Checks
|
|
289
|
+
missing_ratio = df.isnull().sum().sum() / (df.shape[0] * df.shape[1])
|
|
290
|
+
if missing_ratio > 0.3:
|
|
291
|
+
raise DataValidationError(
|
|
292
|
+
field="missing_data_ratio",
|
|
293
|
+
value=missing_ratio,
|
|
294
|
+
message=f"Dataset has {missing_ratio:.1%} missing values, exceeds 30% threshold"
|
|
295
|
+
)
|
|
296
|
+
|
|
297
|
+
# 4\. Class Imbalance Check
|
|
298
|
+
class_ratio = df[target_col].value_counts().min() / df[target_col].value_counts().max()
|
|
299
|
+
if class_ratio < 0.1:
|
|
300
|
+
raise DataImbalanceError(ratio=class_ratio, threshold=0.1)
|
|
301
|
+
|
|
302
|
+
# 5\. Feature Engineering
|
|
303
|
+
try:
|
|
304
|
+
df['log_revenue'] = np.log(df['revenue'] + 1)
|
|
305
|
+
except Exception as e:
|
|
306
|
+
raise FeatureEngineeringError("log_transform", cause=str(e))
|
|
307
|
+
|
|
308
|
+
# 6\. Model Training
|
|
309
|
+
try:
|
|
310
|
+
model = RandomForestClassifier(n_estimators=100)
|
|
311
|
+
X = df.drop(columns=[target_col])
|
|
312
|
+
y = df[target_col]
|
|
313
|
+
model.fit(X, y)
|
|
314
|
+
except Exception as e:
|
|
315
|
+
raise ModelTrainingError("RandomForest", message=f"Training failed: {e}")
|
|
316
|
+
|
|
317
|
+
# 7\. Model Validation
|
|
318
|
+
train_score = model.score(X, y)
|
|
319
|
+
if train_score < 0.6:
|
|
320
|
+
raise UnderfittingError(train_metric=train_score, threshold=0.6)
|
|
321
|
+
|
|
322
|
+
return model
|
|
323
|
+
|
|
324
|
+
# Usage
|
|
325
|
+
if __name__ == "__main__":
|
|
326
|
+
try:
|
|
327
|
+
model = ml_pipeline("customer_data.csv", "churned")
|
|
328
|
+
print("โ
Pipeline completed successfully!")
|
|
329
|
+
except DataLoadingError as e:
|
|
330
|
+
print(f"โ Data loading failed: {e}")
|
|
331
|
+
except MissingColumnError as e:
|
|
332
|
+
print(f"โ Schema validation failed: {e}")
|
|
333
|
+
except DataImbalanceError as e:
|
|
334
|
+
print(f"โ ๏ธ Data quality issue: {e}")
|
|
335
|
+
except ModelTrainingError as e:
|
|
336
|
+
print(f"โ Model training failed: {e}")
|
|
337
|
+
except Exception as e:
|
|
338
|
+
print(f"๐ฅ Unexpected error: {e}")
|
|
339
|
+
```
|
|
340
|
+
|
|
341
|
+
## ๐ค Contributing
|
|
342
|
+
|
|
343
|
+
We welcome contributions! See our [Contributing Guide](CONTRIBUTING.md) for details.
|
|
344
|
+
|
|
345
|
+
```bash
|
|
346
|
+
# Development setup
|
|
347
|
+
git clone https://github.com/DiogoRibeiro7/DataExcept.git
|
|
348
|
+
cd DataExcept
|
|
349
|
+
make install # poetry install --with dev,docs
|
|
350
|
+
pre-commit install
|
|
351
|
+
|
|
352
|
+
make check # lint, formatting, mypy and tests - everything CI runs
|
|
353
|
+
make help # list all targets
|
|
354
|
+
```
|
|
355
|
+
|
|
356
|
+
Please also read the [Code of Conduct](CODE_OF_CONDUCT.md). Security issues go
|
|
357
|
+
through [SECURITY.md](SECURITY.md), not the public issue tracker.
|
|
358
|
+
|
|
359
|
+
## ๐ Documentation
|
|
360
|
+
|
|
361
|
+
- **Full Documentation**: [diogoribeiro7.github.io/DataExcept](https://diogoribeiro7.github.io/DataExcept/)
|
|
362
|
+
- **API Reference**: [API Docs](https://diogoribeiro7.github.io/DataExcept/api/)
|
|
363
|
+
- **Advanced Usage**: [Advanced Guide](https://diogoribeiro7.github.io/DataExcept/advanced_usage/)
|
|
364
|
+
- **CLI Reference**: [CLI Guide](https://diogoribeiro7.github.io/DataExcept/cli/)
|
|
365
|
+
- **Changelog**: [CHANGELOG.md](CHANGELOG.md)
|
|
366
|
+
|
|
367
|
+
## ๐ Citation
|
|
368
|
+
|
|
369
|
+
If you use DataExcept in your research, please cite it:
|
|
370
|
+
|
|
371
|
+
```bibtex
|
|
372
|
+
@software{ribeiro_dataexcept_2025,
|
|
373
|
+
author = {Ribeiro, Diogo},
|
|
374
|
+
title = {DataExcept: Structured Exception Handling for Data Science},
|
|
375
|
+
url = {https://github.com/DiogoRibeiro7/DataExcept},
|
|
376
|
+
version = {0.1.0},
|
|
377
|
+
year = {2025},
|
|
378
|
+
publisher = {GitHub}
|
|
379
|
+
}
|
|
380
|
+
```
|
|
381
|
+
|
|
382
|
+
## ๐ License
|
|
383
|
+
|
|
384
|
+
This project is licensed under the MIT License - see the [LICENSE](LICENSE) file for details.
|
|
385
|
+
|
|
386
|
+
## ๐ About the Author
|
|
387
|
+
|
|
388
|
+
**Diogo Ribeiro** is a Lead Data Scientist at Mysense.ai and researcher/instructor at ESMAD (Instituto Politรฉcnico do Porto). With expertise in machine learning, statistical analysis, and production ML systems, he created DataExcept to solve real-world error handling challenges in data science workflows.
|
|
389
|
+
|
|
390
|
+
- ๐ **ORCID**: [0009-0001-2022-7072](https://orcid.org/0009-0001-2022-7072)
|
|
391
|
+
- ๐ **Website**: [diogoribeiro7.github.io](https://diogoribeiro7.github.io/)
|
|
392
|
+
- ๐ข **Affiliation**: ESMAD - Instituto Politรฉcnico do Porto
|
|
393
|
+
|
|
394
|
+
--------------------------------------------------------------------------------
|
|
395
|
+
|
|
396
|
+
โญ **Star this repo** if DataExcept helps you build better data pipelines!
|
|
397
|
+
|