DataExcept 0.3.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. {dataexcept-0.3.0 → dataexcept-0.4.0}/CHANGELOG.md +102 -1
  2. {dataexcept-0.3.0 → dataexcept-0.4.0}/CITATION.cff +1 -1
  3. {dataexcept-0.3.0 → dataexcept-0.4.0}/PKG-INFO +11 -12
  4. {dataexcept-0.3.0 → dataexcept-0.4.0}/README.md +10 -11
  5. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/__init__.py +12 -1
  6. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/__main__.py +7 -1
  7. dataexcept-0.4.0/dataexcept/_validation.py +22 -0
  8. dataexcept-0.4.0/dataexcept/base.py +67 -0
  9. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/database_exceptions.py +7 -3
  10. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/dataengineering_exceptions.py +3 -1
  11. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/datascience_exceptions/base.py +3 -1
  12. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/datascience_exceptions/ingestion.py +5 -4
  13. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/datascience_exceptions/operations.py +4 -3
  14. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/datascience_exceptions/training.py +9 -8
  15. dataexcept-0.4.0/dataexcept/exceptions/base.py +7 -0
  16. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/exceptions/notification.py +6 -4
  17. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/io_exceptions.py +3 -1
  18. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/network_exceptions.py +3 -1
  19. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/pandas_exceptions.py +3 -1
  20. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/pipeline_exceptions.py +6 -3
  21. dataexcept-0.4.0/dataexcept/redaction.py +114 -0
  22. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/security_exceptions.py +8 -3
  23. {dataexcept-0.3.0 → dataexcept-0.4.0}/pyproject.toml +21 -2
  24. dataexcept-0.3.0/dataexcept/exceptions/base.py +0 -4
  25. {dataexcept-0.3.0 → dataexcept-0.4.0}/LICENSE +0 -0
  26. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/_deprecation.py +0 -0
  27. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/datascience_exceptions/__init__.py +0 -0
  28. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/exceptions/__init__.py +0 -0
  29. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/exceptions/authentication.py +0 -0
  30. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/exceptions/configuration.py +0 -0
  31. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/exceptions/external.py +0 -0
  32. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/exceptions/lifecycle.py +0 -0
  33. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/exceptions/parsing.py +0 -0
  34. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/exceptions/scheduling.py +0 -0
  35. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/exceptions/validation.py +0 -0
  36. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/job_exceptions.py +0 -0
  37. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/logging_helpers.py +0 -0
  38. {dataexcept-0.3.0 → dataexcept-0.4.0}/dataexcept/py.typed +0 -0
@@ -7,6 +7,106 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.4.0] - 2026-08-24
11
+
12
+ ### Added
13
+
14
+ - **`DataExceptError`, the root of the hierarchy.** Every exception the package
15
+ raises now derives from it, so one clause catches the whole library:
16
+ ```python
17
+ except DataExceptError:
18
+ ...
19
+ ```
20
+ The nine domain roots (`JobError`, `DataScienceError`, `PipelineError` and
21
+ the rest) sit beneath it and still catch only their own domain, so granular
22
+ handling is unchanged.
23
+
24
+ ### Changed
25
+
26
+ - `SECURITY.md` claimed 0.1.x was the supported version, and `CHECKLIST.md`
27
+ still described an uncommitted lockfile, 30 of 96 top-level exports, 109 tests
28
+ and 85% coverage. Both now match the project.
29
+ - **Coverage is now measured honestly, and gated.** `coverage run -m pytest`
30
+ had no `source` setting, so it counted the tests and examples in the
31
+ denominator — and test files are by definition fully executed. The reported
32
+ figure was inflated: 86% when the package alone was at 79%.
33
+ `[tool.coverage.run]` now restricts measurement to `dataexcept` and enables
34
+ branch coverage, and `fail_under = 91` stops it regressing. The honest number
35
+ is **92%**; the README, checklist and roadmap now quote that rather than the
36
+ inflated one.
37
+ - The CLI is exercised in-process as well as through a subprocess. The
38
+ subprocess tests verify real invocation but coverage cannot see inside them,
39
+ which left `__main__.py` reporting 23% despite being tested. It now reports
40
+ 93%.
41
+
42
+ ### Fixed
43
+
44
+ - **NumPy scalars are accepted where a number is expected.** Eleven validations
45
+ used `isinstance(value, (int, float))`, which rejects `numpy.float32` and
46
+ `numpy.int64` — a poor answer from a library aimed at data science. They now
47
+ use `numbers.Real`, which NumPy registers its scalar types with, so this needs
48
+ no dependency on NumPy. Arrays and non-numbers are still rejected.
49
+ - **A wrapped exception is now chained.** Constructors that take an underlying
50
+ exception recorded it on an attribute but never set `__cause__`, so a
51
+ traceback did not show what actually failed. `DataExceptError` now mirrors it,
52
+ and Python prints "The above exception was the direct cause of the following
53
+ exception" as if `raise ... from` had been used.
54
+ - `from dataexcept import *` failed under `-W error::DeprecationWarning`,
55
+ because `__all__` listed the deprecated `job_exceptions` module. It is no
56
+ longer advertised there; it remains importable until 1.0.0.
57
+ - A redundant `global` declaration in `examples/lambda_main.py` raised three
58
+ `F824` warnings. CI linted only `dataexcept` and `tests` with flake8 while the
59
+ formatters covered `examples` and `scripts`; flake8 now covers all four.
60
+ - **Four exceptions discarded the reason for the failure.**
61
+ `DataLoadingError`, `MissingDataError`, `ModelSerializationError` and
62
+ `DeploymentError` rendered only their identifying attribute —
63
+ `[DataLoadingError] orders.csv` — while "invalid utf-8", "disk full" or
64
+ "permission denied" sat unseen in `args[0]`. Since logging uses `str(exc)`,
65
+ the part a reader needs never reached the log. All four now render the full
66
+ message, and a test asserts no class drops it.
67
+ - **Exceptions can now cross a process boundary.** They could not be pickled:
68
+ most constructors take several arguments while `Exception.args` holds only
69
+ the rendered message, and the default protocol replays `args` through
70
+ `__init__`. Of 98 classes, 39 raised `TypeError` on unpickling and a further
71
+ 47 came back with different state — only 10 round-tripped exactly. Raising
72
+ one inside a `ProcessPoolExecutor` killed the pool with `BrokenProcessPool`.
73
+ `DataExceptError.__reduce__` restores `args` and `__dict__` directly instead
74
+ of replaying `__init__`. All 97 constructible classes now round-trip with
75
+ identical type, message and attributes, covered by a test per class plus a
76
+ real process-pool test.
77
+ - The stability policy and README both claimed `except JobError:` catches
78
+ "anything else this library raises". It did not — there were nine
79
+ disconnected trees under `Exception`, so `JobError` caught neither
80
+ `ModelTrainingError` nor `PipelineError` nor `DatabaseError`. The claim is
81
+ now true of `DataExceptError`, and the docs say which base covers what.
82
+
83
+ ### Security
84
+
85
+ - **A release now has to prove where it came from.** The release workflow
86
+ checked only that the tag text matched `pyproject.toml`, so a tag pushed to
87
+ an unreviewed branch could reach the PyPI publishing job. It now refuses to
88
+ build unless the tagged commit is reachable from `main` and the same checks
89
+ branch protection requires are green on that exact commit.
90
+ - The built wheel is tested before it is published. A new `verify-wheel` job
91
+ installs the artifact, deletes the source package so nothing can import it by
92
+ accident, and runs the whole suite against what will actually be uploaded.
93
+ `scripts/check_wheel.py` then asserts the distribution ships `py.typed` and a
94
+ complete `__all__` — a file can be present in the repository and missing from
95
+ the artifact.
96
+ - **Credentials are no longer written into exception messages.** `log_exception`
97
+ logs `str(exc)`, so a failed connection put the database password in the log.
98
+ `InvalidTokenError` embedded the whole token; `DatabaseConnectionError` the
99
+ whole connection URL including username and password; `WebhookError` and
100
+ `ApiError` the URL including any signing or key parameter.
101
+ These are now redacted before being stored or rendered, so the raw value is
102
+ absent from the message, the attributes and a pickle of the exception. A
103
+ secret renders as `***(1a2b3c4d)` — a truncated SHA-256, so the same bad
104
+ credential failing repeatedly stays correlatable in a log without appearing
105
+ in it. Host, port and path survive, because those are what make the error
106
+ actionable.
107
+ `QueryExecutionError` still embeds the SQL it is given; SECURITY.md now says
108
+ so explicitly rather than leaving it to be discovered.
109
+
10
110
  ## [0.3.0] - 2026-08-24
11
111
 
12
112
  ### Added
@@ -168,7 +268,8 @@ First public release.
168
268
  - Published to PyPI via OIDC trusted publishing; no long-lived API token is
169
269
  involved in a release.
170
270
 
171
- [Unreleased]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.3.0...HEAD
271
+ [Unreleased]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.4.0...HEAD
272
+ [0.4.0]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.3.0...v0.4.0
172
273
  [0.3.0]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.2.1...v0.3.0
173
274
  [0.2.1]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.2.0...v0.2.1
174
275
  [0.2.0]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.1.0...v0.2.0
@@ -1,7 +1,7 @@
1
1
  cff-version: 1.2.0
2
2
  message: "If you use this software, please cite it using the following metadata."
3
3
  title: "DataExcept"
4
- version: "0.3.0"
4
+ version: "0.4.0"
5
5
  authors:
6
6
  - family-names: "Ribeiro"
7
7
  given-names: "Diogo"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: DataExcept
3
- Version: 0.3.0
3
+ Version: 0.4.0
4
4
  Summary: A Python package providing structured, easily-extendable custom exception types.
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -41,13 +41,13 @@ Description-Content-Type: text/markdown
41
41
 
42
42
  ## 🎯 Key Features
43
43
 
44
- - **🏗️ Hierarchical Structure**: Catch specific errors or broad categories
44
+ - **🏗️ Hierarchical Structure**: Catch one specific error, a whole domain, or everything via `DataExceptError`
45
45
  - **📦 One Import**: Every exception is available from `dataexcept` directly, or from its domain module — same objects either way
46
- - **📊 Data Science Focused**: 98 exception classes covering ML pipelines, feature engineering, model training
47
- - **🔧 Production Ready**: Comprehensive logging helpers and error context
46
+ - **📊 Data Science Focused**: 99 exception classes covering ML pipelines, feature engineering, model training
47
+ - **🔧 Production Ready**: Logging helpers, error context, and exceptions that survive a process boundary intact
48
48
  - **📚 Academic Quality**: Proper documentation, type hints, and citation support
49
49
  - **🐍 Python 3.10+**: Modern Python with full type safety
50
- - **🧪 Well Tested**: Broad test suite with comprehensive edge case handling (see the coverage badge above)
50
+ - **🧪 Well Tested**: 690+ tests at 92% branch coverage of the package, gated in CI
51
51
 
52
52
  ## 📦 Quick Installation
53
53
 
@@ -105,8 +105,7 @@ def load_dataset(file_path: str) -> pd.DataFrame:
105
105
  ### Exception Hierarchies
106
106
 
107
107
  ```python
108
- from dataexcept import JobError
109
- from dataexcept.datascience_exceptions import ModelTrainingError, ConvergenceError
108
+ from dataexcept import ConvergenceError, DataExceptError, ModelTrainingError
110
109
 
111
110
  try:
112
111
  # Your ML pipeline
@@ -119,8 +118,8 @@ except ModelTrainingError:
119
118
  # Handle any training-related error
120
119
  logger.error("Training failed, falling back to simpler model")
121
120
  train_simple_model()
122
- except JobError:
123
- # Handle any job-related error
121
+ except DataExceptError:
122
+ # Handle anything else DataExcept raised
124
123
  logger.error("Job failed, notifying administrators")
125
124
  send_alert()
126
125
  ```
@@ -214,7 +213,7 @@ except Exception as exc:
214
213
  ### Command Line Interface
215
214
 
216
215
  ```bash
217
- # List every exception class the package exports (98 of them, alphabetically)
216
+ # List every exception class the package exports (99 of them, alphabetically)
218
217
  $ dataexcept list
219
218
  ApiError
220
219
  AuthenticationError
@@ -225,7 +224,7 @@ BiasDetectionError
225
224
 
226
225
  # Check version
227
226
  $ dataexcept --version
228
- dataexcept 0.3.0
227
+ dataexcept 0.4.0
229
228
  ```
230
229
 
231
230
  ## 🎯 Use Cases
@@ -375,7 +374,7 @@ If you use DataExcept in your research, please cite it:
375
374
  author = {Ribeiro, Diogo},
376
375
  title = {DataExcept: Structured Exception Handling for Data Science},
377
376
  url = {https://github.com/DiogoRibeiro7/DataExcept},
378
- version = {0.3.0},
377
+ version = {0.4.0},
379
378
  year = {2026},
380
379
  publisher = {GitHub}
381
380
  }
@@ -15,13 +15,13 @@
15
15
 
16
16
  ## 🎯 Key Features
17
17
 
18
- - **🏗️ Hierarchical Structure**: Catch specific errors or broad categories
18
+ - **🏗️ Hierarchical Structure**: Catch one specific error, a whole domain, or everything via `DataExceptError`
19
19
  - **📦 One Import**: Every exception is available from `dataexcept` directly, or from its domain module — same objects either way
20
- - **📊 Data Science Focused**: 98 exception classes covering ML pipelines, feature engineering, model training
21
- - **🔧 Production Ready**: Comprehensive logging helpers and error context
20
+ - **📊 Data Science Focused**: 99 exception classes covering ML pipelines, feature engineering, model training
21
+ - **🔧 Production Ready**: Logging helpers, error context, and exceptions that survive a process boundary intact
22
22
  - **📚 Academic Quality**: Proper documentation, type hints, and citation support
23
23
  - **🐍 Python 3.10+**: Modern Python with full type safety
24
- - **🧪 Well Tested**: Broad test suite with comprehensive edge case handling (see the coverage badge above)
24
+ - **🧪 Well Tested**: 690+ tests at 92% branch coverage of the package, gated in CI
25
25
 
26
26
  ## 📦 Quick Installation
27
27
 
@@ -79,8 +79,7 @@ def load_dataset(file_path: str) -> pd.DataFrame:
79
79
  ### Exception Hierarchies
80
80
 
81
81
  ```python
82
- from dataexcept import JobError
83
- from dataexcept.datascience_exceptions import ModelTrainingError, ConvergenceError
82
+ from dataexcept import ConvergenceError, DataExceptError, ModelTrainingError
84
83
 
85
84
  try:
86
85
  # Your ML pipeline
@@ -93,8 +92,8 @@ except ModelTrainingError:
93
92
  # Handle any training-related error
94
93
  logger.error("Training failed, falling back to simpler model")
95
94
  train_simple_model()
96
- except JobError:
97
- # Handle any job-related error
95
+ except DataExceptError:
96
+ # Handle anything else DataExcept raised
98
97
  logger.error("Job failed, notifying administrators")
99
98
  send_alert()
100
99
  ```
@@ -188,7 +187,7 @@ except Exception as exc:
188
187
  ### Command Line Interface
189
188
 
190
189
  ```bash
191
- # List every exception class the package exports (98 of them, alphabetically)
190
+ # List every exception class the package exports (99 of them, alphabetically)
192
191
  $ dataexcept list
193
192
  ApiError
194
193
  AuthenticationError
@@ -199,7 +198,7 @@ BiasDetectionError
199
198
 
200
199
  # Check version
201
200
  $ dataexcept --version
202
- dataexcept 0.3.0
201
+ dataexcept 0.4.0
203
202
  ```
204
203
 
205
204
  ## 🎯 Use Cases
@@ -349,7 +348,7 @@ If you use DataExcept in your research, please cite it:
349
348
  author = {Ribeiro, Diogo},
350
349
  title = {DataExcept: Structured Exception Handling for Data Science},
351
350
  url = {https://github.com/DiogoRibeiro7/DataExcept},
352
- version = {0.3.0},
351
+ version = {0.4.0},
353
352
  year = {2026},
354
353
  publisher = {GitHub}
355
354
  }
@@ -4,6 +4,12 @@ Every exception the package defines is importable straight from here::
4
4
 
5
5
  from dataexcept import ValidationError, ModelTrainingError
6
6
 
7
+ They all derive from :class:`DataExceptError`, so one clause catches anything
8
+ this package raises::
9
+
10
+ except DataExceptError:
11
+ ...
12
+
7
13
  The domain modules (``datascience_exceptions``, ``pipeline_exceptions`` and so
8
14
  on) remain importable and export the same objects, so both spellings work and
9
15
  refer to the same classes.
@@ -32,6 +38,7 @@ from . import ( # noqa: F401
32
38
  security_exceptions,
33
39
  )
34
40
  from ._deprecation import resolve_deprecated
41
+ from .base import DataExceptError
35
42
  from .database_exceptions import (
36
43
  DatabaseConnectionError,
37
44
  DatabaseError,
@@ -156,6 +163,8 @@ from .security_exceptions import (
156
163
  )
157
164
 
158
165
  __all__ = [
166
+ # The root of the hierarchy: catches anything this package raises.
167
+ "DataExceptError",
159
168
  # Every exception class the package defines.
160
169
  "ApiError",
161
170
  "AuthenticationError",
@@ -261,12 +270,14 @@ __all__ = [
261
270
  "log_exception",
262
271
  "log_then_raise",
263
272
  # Domain modules, for callers who prefer a qualified import.
273
+ # job_exceptions is deliberately absent: it is deprecated, and listing it
274
+ # makes `from dataexcept import *` emit a DeprecationWarning, which turns
275
+ # into an error under -W error::DeprecationWarning. It stays importable.
264
276
  "database_exceptions",
265
277
  "dataengineering_exceptions",
266
278
  "datascience_exceptions",
267
279
  "exceptions",
268
280
  "io_exceptions",
269
- "job_exceptions",
270
281
  "logging_helpers",
271
282
  "network_exceptions",
272
283
  "pandas_exceptions",
@@ -21,11 +21,17 @@ _DEPRECATED_MODULES = frozenset({"dataexcept.job_exceptions"})
21
21
  def _iter_exception_modules() -> Iterable[ModuleType]:
22
22
  """Yield every non-deprecated submodule that explicitly defines ``__all__``."""
23
23
  allowed_suffixes = ("exceptions", "_exceptions")
24
+ # dataexcept.base holds DataExceptError, the root of the hierarchy, and
25
+ # does not match the suffix rule.
26
+ always_include = {"dataexcept.base"}
24
27
 
25
28
  for module_info in pkgutil.walk_packages(
26
29
  _PKG_PATH, prefix="dataexcept.", onerror=lambda name: None
27
30
  ):
28
- if not module_info.name.endswith(allowed_suffixes):
31
+ if (
32
+ not module_info.name.endswith(allowed_suffixes)
33
+ and module_info.name not in always_include
34
+ ):
29
35
  continue
30
36
  if module_info.name in _DEPRECATED_MODULES:
31
37
  continue
@@ -0,0 +1,22 @@
1
+ """Small runtime checks shared by the exception constructors."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import numbers
6
+
7
+ __all__ = ["is_number"]
8
+
9
+
10
+ def is_number(value: object) -> bool:
11
+ """Return True for any real number, including NumPy scalars.
12
+
13
+ ``isinstance(value, (int, float))`` rejects ``numpy.float32`` and
14
+ ``numpy.int64``, which is a poor answer from a library aimed at data
15
+ science. ``numbers.Real`` accepts them because NumPy registers its scalar
16
+ types with the ABC, and it needs no dependency on NumPy to do so.
17
+
18
+ This is a function rather than an inline ``isinstance`` because narrowing a
19
+ value to ``numbers.Real`` defeats mypy's inference for the rest of the
20
+ enclosing class.
21
+ """
22
+ return isinstance(value, numbers.Real)
@@ -0,0 +1,67 @@
1
+ """The root of the DataExcept exception hierarchy.
2
+
3
+ Every exception this package raises derives from :class:`DataExceptError`, so a
4
+ caller can catch everything the library can raise with a single clause while
5
+ still catching narrowly where it matters::
6
+
7
+ try:
8
+ run_pipeline()
9
+ except ValidationError:
10
+ ... # exactly this failure
11
+ except DataExceptError:
12
+ ... # anything else DataExcept raised
13
+
14
+ The base also gives the whole hierarchy a working serialization contract. Many
15
+ of these exceptions take several constructor arguments while ``Exception.args``
16
+ holds only the rendered message, so the default pickling protocol -- which
17
+ replays ``args`` through ``__init__`` -- could not rebuild them. That made them
18
+ unusable across a process boundary, which is where a data pipeline most needs
19
+ them.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ from typing import Any, Dict, Tuple, Type
25
+
26
+ __all__ = ["DataExceptError"]
27
+
28
+ #: Attribute names used across the package to hold the exception that caused
29
+ #: this one. Checked in order; the first that holds an exception wins.
30
+ _CAUSE_ATTRIBUTES = ("original", "original_exception", "cause")
31
+
32
+
33
+ def _rebuild(
34
+ cls: Type["DataExceptError"], args: Tuple[Any, ...], state: Dict[str, Any]
35
+ ) -> "DataExceptError":
36
+ """Recreate *cls* without replaying its ``__init__``.
37
+
38
+ Constructors here validate and render a message from their arguments;
39
+ replaying them would need the original arguments, which ``args`` does not
40
+ carry. Restoring ``args`` and ``__dict__` directly reproduces the exception
41
+ exactly, including its message and every attribute it recorded.
42
+ """
43
+ exc = cls.__new__(cls)
44
+ Exception.__init__(exc, *args)
45
+ exc.__dict__.update(state)
46
+ return exc
47
+
48
+
49
+ class DataExceptError(Exception):
50
+ """Base class for every exception DataExcept raises."""
51
+
52
+ def __init__(self, *args: Any) -> None:
53
+ super().__init__(*args)
54
+ # Constructors that wrap another exception record it on an attribute.
55
+ # Mirroring it into __cause__ is what makes a traceback print the
56
+ # underlying failure, exactly as `raise ... from exc` would; assigning
57
+ # __cause__ also sets __suppress_context__, as `raise from` does.
58
+ for attribute in _CAUSE_ATTRIBUTES:
59
+ candidate = getattr(self, attribute, None)
60
+ if isinstance(candidate, BaseException):
61
+ self.__cause__ = candidate
62
+ break
63
+
64
+ def __reduce__(
65
+ self,
66
+ ) -> Tuple[Any, Tuple[Type["DataExceptError"], Tuple[Any, ...], Dict[str, Any]]]:
67
+ return (_rebuild, (type(self), self.args, self.__dict__.copy()))
@@ -2,8 +2,11 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ from .base import DataExceptError
6
+ from .redaction import redact_url
5
7
 
6
- class DatabaseError(Exception):
8
+
9
+ class DatabaseError(DataExceptError):
7
10
  """Base exception for database-related errors."""
8
11
 
9
12
  pass
@@ -19,8 +22,9 @@ class DatabaseConnectionError(DatabaseError):
19
22
  db_url: Database connection URL.
20
23
  message: Optional custom error message.
21
24
  """
22
- self.db_url = db_url
23
- default = f"Failed to connect to database at '{db_url}'"
25
+ # A connection URL routinely carries a username and password.
26
+ self.db_url = redact_url(db_url)
27
+ default = f"Failed to connect to database at '{self.db_url}'"
24
28
  super().__init__(message or default)
25
29
 
26
30
 
@@ -4,8 +4,10 @@ from __future__ import annotations
4
4
 
5
5
  from typing import Optional
6
6
 
7
+ from .base import DataExceptError
7
8
 
8
- class DataEngineeringError(Exception):
9
+
10
+ class DataEngineeringError(DataExceptError):
9
11
  """Base exception for data engineering errors."""
10
12
 
11
13
  pass
@@ -2,8 +2,10 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ from ..base import DataExceptError
5
6
 
6
- class DataScienceError(Exception):
7
+
8
+ class DataScienceError(DataExceptError):
7
9
  """Base exception for data science errors."""
8
10
 
9
11
  def __init__(self, message: str) -> None:
@@ -4,6 +4,7 @@ from __future__ import annotations
4
4
 
5
5
  from typing import Any, Optional, Sequence
6
6
 
7
+ from .._validation import is_number
7
8
  from .base import DataScienceError
8
9
 
9
10
 
@@ -30,7 +31,7 @@ class DataLoadingError(DataScienceError):
30
31
  super().__init__(message)
31
32
 
32
33
  def __str__(self) -> str:
33
- return f"[DataLoadingError] {self.source}"
34
+ return f"[DataLoadingError:{self.source}] {self.message}"
34
35
 
35
36
 
36
37
  class DataFormatError(DataScienceError):
@@ -105,7 +106,7 @@ class MissingDataError(DataScienceError):
105
106
  super().__init__(message)
106
107
 
107
108
  def __str__(self) -> str:
108
- return f"[MissingDataError] {self.feature}"
109
+ return f"[MissingDataError:{self.feature}] {self.message}"
109
110
 
110
111
 
111
112
  class OutlierDetectionError(DataScienceError):
@@ -227,9 +228,9 @@ class DataImbalanceError(DataScienceError):
227
228
  def __init__(
228
229
  self, ratio: float, threshold: float, message: Optional[str] = None
229
230
  ) -> None:
230
- if not isinstance(ratio, (int, float)):
231
+ if not is_number(ratio):
231
232
  raise TypeError(f"ratio must be numeric, got {type(ratio).__name__}")
232
- if not isinstance(threshold, (int, float)):
233
+ if not is_number(threshold):
233
234
  raise TypeError(
234
235
  f"threshold must be numeric, got {type(threshold).__name__}"
235
236
  )
@@ -4,6 +4,7 @@ from __future__ import annotations
4
4
 
5
5
  from typing import Any, Optional
6
6
 
7
+ from .._validation import is_number
7
8
  from .base import DataScienceError
8
9
 
9
10
 
@@ -30,7 +31,7 @@ class ModelSerializationError(DataScienceError):
30
31
  super().__init__(message)
31
32
 
32
33
  def __str__(self) -> str:
33
- return f"[ModelSerializationError] {self.path}"
34
+ return f"[ModelSerializationError:{self.path}] {self.message}"
34
35
 
35
36
 
36
37
  class DeploymentError(DataScienceError):
@@ -57,7 +58,7 @@ class DeploymentError(DataScienceError):
57
58
  super().__init__(msg)
58
59
 
59
60
  def __str__(self) -> str:
60
- return f"[DeploymentError] {self.target}"
61
+ return f"[DeploymentError:{self.target}] {self.message}"
61
62
 
62
63
 
63
64
  class DataDriftError(DataScienceError):
@@ -74,7 +75,7 @@ class DataDriftError(DataScienceError):
74
75
  ) -> None:
75
76
  if not isinstance(feature, str):
76
77
  raise TypeError(f"feature must be str, got {type(feature).__name__}")
77
- if not isinstance(drift_score, (int, float)):
78
+ if not is_number(drift_score):
78
79
  raise TypeError(
79
80
  f"drift_score must be number, got {type(drift_score).__name__}"
80
81
  )
@@ -4,6 +4,7 @@ from __future__ import annotations
4
4
 
5
5
  from typing import Any, Optional
6
6
 
7
+ from .._validation import is_number
7
8
  from .base import DataScienceError
8
9
 
9
10
 
@@ -79,7 +80,7 @@ class TrainingTimeoutError(ModelTrainingError):
79
80
  """Raised when model training exceeds a time limit."""
80
81
 
81
82
  def __init__(self, model_type: str, timeout: float) -> None:
82
- if not isinstance(timeout, (int, float)):
83
+ if not is_number(timeout):
83
84
  raise TypeError(f"timeout must be a number, got {type(timeout).__name__}")
84
85
  message = f"Training '{model_type}' exceeded timeout of {timeout} seconds"
85
86
  self.timeout = float(timeout)
@@ -127,7 +128,7 @@ class ModelEvaluationError(DataScienceError):
127
128
  ) -> None:
128
129
  if not isinstance(metric, str):
129
130
  raise TypeError(f"metric must be str, got {type(metric).__name__}")
130
- if not isinstance(value, (int, float)):
131
+ if not is_number(value):
131
132
  raise TypeError(f"value must be number, got {type(value).__name__}")
132
133
 
133
134
  if message is None:
@@ -330,11 +331,11 @@ class OverfittingError(DataScienceError):
330
331
  """
331
332
 
332
333
  def __init__(self, train_metric: float, val_metric: float) -> None:
333
- if not isinstance(train_metric, (int, float)):
334
+ if not is_number(train_metric):
334
335
  raise TypeError(
335
336
  ("train_metric must be numeric, got " f"{type(train_metric).__name__}")
336
337
  )
337
- if not isinstance(val_metric, (int, float)):
338
+ if not is_number(val_metric):
338
339
  raise TypeError(
339
340
  f"val_metric must be numeric, got {type(val_metric).__name__}"
340
341
  )
@@ -360,11 +361,11 @@ class UnderfittingError(DataScienceError):
360
361
  """
361
362
 
362
363
  def __init__(self, train_metric: float, threshold: float) -> None:
363
- if not isinstance(train_metric, (int, float)):
364
+ if not is_number(train_metric):
364
365
  raise TypeError(
365
366
  ("train_metric must be numeric, got " f"{type(train_metric).__name__}")
366
367
  )
367
- if not isinstance(threshold, (int, float)):
368
+ if not is_number(threshold):
368
369
  raise TypeError(
369
370
  f"threshold must be numeric, got {type(threshold).__name__}"
370
371
  )
@@ -426,11 +427,11 @@ class BiasDetectionError(DataScienceError):
426
427
  ) -> None:
427
428
  if not isinstance(feature, str):
428
429
  raise TypeError(f"feature must be str, got {type(feature).__name__}")
429
- if not isinstance(bias_score, (int, float)):
430
+ if not is_number(bias_score):
430
431
  raise TypeError(
431
432
  f"bias_score must be numeric, got {type(bias_score).__name__}"
432
433
  )
433
- if not isinstance(threshold, (int, float)):
434
+ if not is_number(threshold):
434
435
  raise TypeError(
435
436
  f"threshold must be numeric, got {type(threshold).__name__}"
436
437
  )
@@ -0,0 +1,7 @@
1
+ from ..base import DataExceptError
2
+
3
+
4
+ class JobError(DataExceptError):
5
+ """Base exception for all job-related errors."""
6
+
7
+ pass
@@ -1,4 +1,5 @@
1
1
  # notification.py
2
+ from ..redaction import redact_url
2
3
  from .base import JobError
3
4
 
4
5
 
@@ -30,7 +31,7 @@ class EmailError(NotificationError):
30
31
  ):
31
32
  self.recipient = recipient
32
33
  self.subject = subject
33
- self.original_exception = original_exception
34
+ # original_exception is set by NotificationError.__init__ below.
34
35
  msg = f"Email to '{recipient}' with subject '{subject}' failed"
35
36
  if original_exception:
36
37
  msg += f": {original_exception}"
@@ -41,9 +42,10 @@ class WebhookError(NotificationError):
41
42
  """Raised when a webhook POST fails."""
42
43
 
43
44
  def __init__(self, url: str, original_exception: Exception | None = None):
44
- self.url = url
45
- self.original_exception = original_exception
46
- msg = f"Webhook to URL '{url}' failed"
45
+ # Webhook URLs commonly authenticate through a query parameter.
46
+ self.url = redact_url(url)
47
+ # original_exception is set by NotificationError.__init__ below.
48
+ msg = f"Webhook to URL '{self.url}' failed"
47
49
  if original_exception:
48
50
  msg += f": {original_exception}"
49
51
  super().__init__("webhook", original_exception, message=msg)
@@ -2,8 +2,10 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ from .base import DataExceptError
5
6
 
6
- class CustomIOError(Exception):
7
+
8
+ class CustomIOError(DataExceptError):
7
9
  """Base exception for I/O errors."""
8
10
 
9
11
  pass
@@ -2,8 +2,10 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ from .base import DataExceptError
5
6
 
6
- class NetworkError(Exception):
7
+
8
+ class NetworkError(DataExceptError):
7
9
  """Base exception for network-related errors.
8
10
 
9
11
  Example:
@@ -4,8 +4,10 @@ from __future__ import annotations
4
4
 
5
5
  from typing import Optional, Sequence
6
6
 
7
+ from .base import DataExceptError
7
8
 
8
- class PandasError(Exception):
9
+
10
+ class PandasError(DataExceptError):
9
11
  """Base exception for pandas-related errors."""
10
12
 
11
13
 
@@ -5,9 +5,11 @@ from __future__ import annotations
5
5
  from typing import Any, Optional
6
6
 
7
7
  from ._deprecation import resolve_deprecated
8
+ from .base import DataExceptError
9
+ from .redaction import redact_url
8
10
 
9
11
 
10
- class PipelineError(Exception):
12
+ class PipelineError(DataExceptError):
11
13
  """Base exception for pipeline errors."""
12
14
 
13
15
  pass
@@ -147,10 +149,11 @@ class ApiError(PipelineError):
147
149
  status_code: Optional[int] = None,
148
150
  message: Optional[str] = None,
149
151
  ) -> None:
150
- default = f"API call failed: {endpoint}"
152
+ # An endpoint URL may authenticate through a query parameter.
153
+ self.endpoint = redact_url(endpoint)
154
+ default = f"API call failed: {self.endpoint}"
151
155
  if status_code is not None:
152
156
  default += f" (status {status_code})"
153
- self.endpoint = endpoint
154
157
  self.status_code = status_code
155
158
  super().__init__(message or default)
156
159
 
@@ -0,0 +1,114 @@
1
+ """Redaction helpers for values that must not reach a log.
2
+
3
+ Several exceptions here are raised with credentials in hand: an authentication
4
+ token, a database URL carrying a password, a webhook URL with a signing
5
+ parameter. Those values end up in the exception message, and
6
+ :func:`dataexcept.logging_helpers.log_exception` logs ``str(exc)``, so without
7
+ redaction a failed connection writes the password to the log.
8
+
9
+ The aim is to keep an error debuggable while giving up the secret. A redacted
10
+ value carries a short, non-reversible fingerprint, so repeated failures of the
11
+ *same* credential are still recognisable in a log without the credential
12
+ appearing in it.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import hashlib
18
+ from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
19
+
20
+ __all__ = ["fingerprint", "redact_secret", "redact_url"]
21
+
22
+ PLACEHOLDER = "***"
23
+
24
+ #: Query parameters whose value is treated as a secret. Matched case
25
+ #: insensitively against the whole parameter name.
26
+ SENSITIVE_QUERY_PARAMS = frozenset(
27
+ {
28
+ "access_token",
29
+ "api_key",
30
+ "apikey",
31
+ "auth",
32
+ "credential",
33
+ "key",
34
+ "password",
35
+ "private_key",
36
+ "pwd",
37
+ "refresh_token",
38
+ "secret",
39
+ "session",
40
+ "sig",
41
+ "signature",
42
+ "token",
43
+ }
44
+ )
45
+
46
+
47
+ def fingerprint(value: str) -> str:
48
+ """Return a short, one-way fingerprint of *value*.
49
+
50
+ Enough to tell "the same bad token again" from "a different bad token",
51
+ and not enough to recover the token.
52
+ """
53
+ digest = hashlib.sha256(value.encode("utf-8", "replace")).hexdigest()
54
+ return digest[:8]
55
+
56
+
57
+ def redact_secret(value: str | None) -> str | None:
58
+ """Replace a secret with a placeholder and its fingerprint."""
59
+ if value is None:
60
+ return None
61
+ if not value:
62
+ return PLACEHOLDER
63
+ return f"{PLACEHOLDER}({fingerprint(value)})"
64
+
65
+
66
+ def redact_url(url: str | None) -> str | None:
67
+ """Strip credentials and sensitive query parameters from *url*.
68
+
69
+ The scheme, host, port and path are kept, because those are what make the
70
+ error actionable. Anything that authenticates is replaced.
71
+ """
72
+ if not url:
73
+ return url
74
+
75
+ try:
76
+ parts = urlsplit(url)
77
+ except ValueError: # pragma: no cover - urlsplit is extremely permissive
78
+ return PLACEHOLDER
79
+
80
+ if not parts.scheme and not parts.netloc:
81
+ # Not a URL at all; treat the whole thing as sensitive rather than
82
+ # returning it unchanged.
83
+ return url
84
+
85
+ netloc = parts.netloc
86
+ redacted = False
87
+ if "@" in netloc:
88
+ _, _, host = netloc.rpartition("@")
89
+ netloc = f"{PLACEHOLDER}:{PLACEHOLDER}@{host}"
90
+ redacted = True
91
+
92
+ query = parts.query
93
+ if query:
94
+ pairs = parse_qsl(query, keep_blank_values=True)
95
+ if any(key.lower() in SENSITIVE_QUERY_PARAMS for key, _ in pairs):
96
+ query = urlencode(
97
+ [
98
+ (
99
+ key,
100
+ PLACEHOLDER if key.lower() in SENSITIVE_QUERY_PARAMS else value,
101
+ )
102
+ for key, value in pairs
103
+ ],
104
+ # Keep the placeholder legible rather than percent-encoded.
105
+ safe="*",
106
+ )
107
+ redacted = True
108
+
109
+ if not redacted:
110
+ # Nothing sensitive, so hand back exactly what was passed in.
111
+ # Rebuilding would normalise it -- "sqlite://" loses its slashes.
112
+ return url
113
+
114
+ return urlunsplit((parts.scheme, netloc, parts.path, query, parts.fragment))
@@ -2,8 +2,11 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ from .base import DataExceptError
6
+ from .redaction import redact_secret
5
7
 
6
- class SecurityError(Exception):
8
+
9
+ class SecurityError(DataExceptError):
7
10
  """Base exception for security errors."""
8
11
 
9
12
  pass
@@ -53,10 +56,12 @@ class InvalidTokenError(SecurityError):
53
56
  token: The problematic token.
54
57
  message: Optional custom error message.
55
58
  """
56
- self.token = token
59
+ # The raw token is never stored or rendered: this exception is often
60
+ # logged, and the caller already holds the value it passed in.
61
+ self.token = redact_secret(token)
57
62
  default = "Invalid authentication token"
58
63
  if token:
59
- default += f": {token}"
64
+ default += f": {self.token}"
60
65
  super().__init__(message or default)
61
66
 
62
67
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "DataExcept"
3
- version = "0.3.0"
3
+ version = "0.4.0"
4
4
  description = "A Python package providing structured, easily-extendable custom exception types."
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -85,7 +85,9 @@ max-complexity = 8
85
85
  [tool.ruff.lint.per-file-ignores]
86
86
  # assert is how a test asserts, and "/tmp/data" there is a string literal used
87
87
  # as fixture data rather than a path that is ever opened.
88
- "tests/**" = ["S101", "S108", "S603"]
88
+ # S301: the serialization tests unpickle data they created a line earlier.
89
+ # S105/S106: redaction tests need credential-shaped literals to redact.
90
+ "tests/**" = ["S101", "S105", "S106", "S108", "S301", "S603"]
89
91
  # The lambda example asserts to narrow Optionals for the type checker.
90
92
  "examples/**" = ["S101"]
91
93
  # bump_version.py deliberately shells out to poetry, by name, on a maintainer's
@@ -98,3 +100,20 @@ files = ["dataexcept"]
98
100
  no_implicit_optional = true
99
101
  warn_redundant_casts = true
100
102
  warn_unused_configs = true
103
+
104
+ [tool.coverage.run]
105
+ # Without this, `coverage run -m pytest` measures the tests and examples too,
106
+ # which inflates the figure: the test files are by definition fully executed.
107
+ source = ["dataexcept"]
108
+ branch = true
109
+
110
+ [tool.coverage.report]
111
+ # The floor is a ratchet against regression, not a target. Raise it when the
112
+ # real number rises; never lower it to make a build pass.
113
+ fail_under = 91
114
+ show_missing = true
115
+ exclude_also = [
116
+ "if TYPE_CHECKING:",
117
+ "raise NotImplementedError",
118
+ "if __name__ == .__main__.:",
119
+ ]
@@ -1,4 +0,0 @@
1
- class JobError(Exception):
2
- """Base exception for all job-related errors."""
3
-
4
- pass
File without changes