DataExcept 0.4.1__tar.gz → 0.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataexcept-0.4.1 → dataexcept-0.4.2}/CHANGELOG.md +46 -1
- {dataexcept-0.4.1 → dataexcept-0.4.2}/CITATION.cff +1 -1
- {dataexcept-0.4.1 → dataexcept-0.4.2}/PKG-INFO +3 -3
- {dataexcept-0.4.1 → dataexcept-0.4.2}/README.md +2 -2
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/base.py +18 -13
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/datascience_exceptions/training.py +3 -1
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/pipeline_exceptions.py +3 -1
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/redaction.py +49 -10
- {dataexcept-0.4.1 → dataexcept-0.4.2}/pyproject.toml +1 -1
- {dataexcept-0.4.1 → dataexcept-0.4.2}/LICENSE +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/__init__.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/__main__.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/_deprecation.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/_validation.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/database_exceptions.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/dataengineering_exceptions.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/datascience_exceptions/__init__.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/datascience_exceptions/base.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/datascience_exceptions/ingestion.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/datascience_exceptions/operations.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/__init__.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/authentication.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/base.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/configuration.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/external.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/lifecycle.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/notification.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/parsing.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/scheduling.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/validation.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/io_exceptions.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/job_exceptions.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/logging_helpers.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/network_exceptions.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/pandas_exceptions.py +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/py.typed +0 -0
- {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/security_exceptions.py +0 -0
|
@@ -7,6 +7,50 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.4.2] - 2026-08-25
|
|
11
|
+
|
|
12
|
+
### Changed
|
|
13
|
+
|
|
14
|
+
- `SECURITY.md` states two further boundaries rather than leaving them to be
|
|
15
|
+
discovered: state attached to an exception *after* it is constructed is not
|
|
16
|
+
swept, and a URL nested inside a value you pass is rendered redacted but the
|
|
17
|
+
object itself is not rewritten. Walking and rewriting arbitrary caller data
|
|
18
|
+
structures would be a surprising thing for an exception library to do, and
|
|
19
|
+
could not be complete anyway.
|
|
20
|
+
|
|
21
|
+
### Fixed
|
|
22
|
+
|
|
23
|
+
- **Sensitive query parameters are matched by token, not substring.** The
|
|
24
|
+
substring rule was wrong in both directions: it redacted `monkey`, `design`,
|
|
25
|
+
`assign`, `keyword` and `authors`, mangling ordinary debugging information,
|
|
26
|
+
while still missing `passphrase`. Parameter names are now split on
|
|
27
|
+
separators and camelCase and matched word by word, so `X-Amz-Signature`,
|
|
28
|
+
`accessToken` and `client_secret` are caught and `?monkey=bobo` is left
|
|
29
|
+
alone.
|
|
30
|
+
|
|
31
|
+
### Security
|
|
32
|
+
|
|
33
|
+
- **`jwt`, `bearer`, `hmac` and `sas` are recognised as secret parameter
|
|
34
|
+
names.** Checked the token set against the parameters actually used by AWS
|
|
35
|
+
SigV4, Azure SAS, Google Cloud and OAuth 2: the signature and credential
|
|
36
|
+
parameters were already covered, but those four were not. `code`, `state`,
|
|
37
|
+
`nonce` and `client_id` are deliberately still ignored — an OAuth code is a
|
|
38
|
+
secret, but the name is far more often a country code, an HTTP status or a
|
|
39
|
+
discount code, and redacting those would destroy more than it protects.
|
|
40
|
+
- **Three ways a URL could still reach a message, all closed.** The redaction
|
|
41
|
+
boundary scrubbed the message and a handful of named fields, but 18 classes
|
|
42
|
+
interpolate some *other* attribute into `__str__` — a field, a column, a
|
|
43
|
+
resource — and a caller can put a URL in any of them. Every stored string is
|
|
44
|
+
now swept, so no class leaks a credential-bearing URL through its message,
|
|
45
|
+
its attributes or its `args`. A test fills every string argument of every
|
|
46
|
+
class with one and checks all three surfaces.
|
|
47
|
+
- Two classes assigned their attributes *after* calling `super().__init__`, so
|
|
48
|
+
the sweep never saw them. Both now assign first, and a test fails if any
|
|
49
|
+
constructor does it again.
|
|
50
|
+
- The URL pattern carried a `` anchor, so a URL directly following a word
|
|
51
|
+
character was never matched — including `feature_https://...`, the step name
|
|
52
|
+
`FeaturePreprocessingError` builds from its own argument.
|
|
53
|
+
|
|
10
54
|
## [0.4.1] - 2026-08-25
|
|
11
55
|
|
|
12
56
|
### Added
|
|
@@ -385,7 +429,8 @@ First public release.
|
|
|
385
429
|
- Published to PyPI via OIDC trusted publishing; no long-lived API token is
|
|
386
430
|
involved in a release.
|
|
387
431
|
|
|
388
|
-
[Unreleased]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.4.
|
|
432
|
+
[Unreleased]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.4.2...HEAD
|
|
433
|
+
[0.4.2]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.4.1...v0.4.2
|
|
389
434
|
[0.4.1]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.4.0...v0.4.1
|
|
390
435
|
[0.4.0]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.3.0...v0.4.0
|
|
391
436
|
[0.3.0]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.2.1...v0.3.0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: DataExcept
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.2
|
|
4
4
|
Summary: A Python package providing structured, easily-extendable custom exception types.
|
|
5
5
|
License-Expression: MIT
|
|
6
6
|
License-File: LICENSE
|
|
@@ -225,7 +225,7 @@ BiasDetectionError
|
|
|
225
225
|
|
|
226
226
|
# Check version
|
|
227
227
|
$ dataexcept --version
|
|
228
|
-
dataexcept 0.4.
|
|
228
|
+
dataexcept 0.4.2
|
|
229
229
|
```
|
|
230
230
|
|
|
231
231
|
## 🎯 Use Cases
|
|
@@ -375,7 +375,7 @@ If you use DataExcept in your research, please cite it:
|
|
|
375
375
|
author = {Ribeiro, Diogo},
|
|
376
376
|
title = {DataExcept: Structured Exception Handling for Data Science},
|
|
377
377
|
url = {https://github.com/DiogoRibeiro7/DataExcept},
|
|
378
|
-
version = {0.4.
|
|
378
|
+
version = {0.4.2},
|
|
379
379
|
year = {2026},
|
|
380
380
|
publisher = {GitHub}
|
|
381
381
|
}
|
|
@@ -198,7 +198,7 @@ BiasDetectionError
|
|
|
198
198
|
|
|
199
199
|
# Check version
|
|
200
200
|
$ dataexcept --version
|
|
201
|
-
dataexcept 0.4.
|
|
201
|
+
dataexcept 0.4.2
|
|
202
202
|
```
|
|
203
203
|
|
|
204
204
|
## 🎯 Use Cases
|
|
@@ -348,7 +348,7 @@ If you use DataExcept in your research, please cite it:
|
|
|
348
348
|
author = {Ribeiro, Diogo},
|
|
349
349
|
title = {DataExcept: Structured Exception Handling for Data Science},
|
|
350
350
|
url = {https://github.com/DiogoRibeiro7/DataExcept},
|
|
351
|
-
version = {0.4.
|
|
351
|
+
version = {0.4.2},
|
|
352
352
|
year = {2026},
|
|
353
353
|
publisher = {GitHub}
|
|
354
354
|
}
|
|
@@ -138,22 +138,27 @@ class DataExceptError(Exception):
|
|
|
138
138
|
_keep_url_path = True
|
|
139
139
|
|
|
140
140
|
def __init__(self, *args: Any) -> None:
|
|
141
|
-
# One boundary for the whole hierarchy
|
|
142
|
-
#
|
|
143
|
-
# exception quoting the original URL -- it is scrubbed here
|
|
144
|
-
# only the structured argument leaves all three
|
|
141
|
+
# One boundary for the whole hierarchy. Whatever built the message -- a
|
|
142
|
+
# constructor, a caller-supplied `message`, or the text of a wrapped
|
|
143
|
+
# exception quoting the original URL -- it is scrubbed here, because
|
|
144
|
+
# redacting only the structured argument leaves all three routes open.
|
|
145
145
|
keep_path = type(self)._keep_url_path
|
|
146
146
|
if args and isinstance(args[0], str):
|
|
147
147
|
args = (redact_urls_in_text(args[0], keep_path=keep_path),) + args[1:]
|
|
148
|
-
|
|
149
|
-
#
|
|
150
|
-
#
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
148
|
+
|
|
149
|
+
# Many classes store the message on self.message and render *that* in
|
|
150
|
+
# __str__, and 18 interpolate some other attribute -- a field, a
|
|
151
|
+
# column, a resource -- any of which a caller can fill with a URL. So
|
|
152
|
+
# every stored string is swept, not just the message.
|
|
153
|
+
#
|
|
154
|
+
# redact_urls_in_text rather than redact_if_url: a message has the URL
|
|
155
|
+
# embedded in prose, and redact_if_url only handles a value that is
|
|
156
|
+
# wholly a URL. It is a no-op on anything without "://" in it, so
|
|
157
|
+
# ordinary names and file paths are untouched.
|
|
158
|
+
for name, value in list(self.__dict__.items()):
|
|
159
|
+
if isinstance(value, str) and "://" in value:
|
|
160
|
+
self.__dict__[name] = redact_urls_in_text(value, keep_path=keep_path)
|
|
161
|
+
|
|
157
162
|
super().__init__(*args)
|
|
158
163
|
# Constructors that wrap another exception record it on an attribute.
|
|
159
164
|
# Mirroring it into __cause__ is what makes a traceback print the
|
|
@@ -69,8 +69,10 @@ class ConvergenceError(ModelTrainingError):
|
|
|
69
69
|
f"Model '{model_type}' failed to converge after "
|
|
70
70
|
f"{iterations} iterations"
|
|
71
71
|
)
|
|
72
|
-
super().__init__
|
|
72
|
+
# Assigned before super(): DataExceptError.__init__ sweeps the stored
|
|
73
|
+
# strings for URLs, and anything set afterwards escapes that.
|
|
73
74
|
self.iterations = iterations
|
|
75
|
+
super().__init__(model_type=model_type, epoch=None, message=message)
|
|
74
76
|
|
|
75
77
|
def __str__(self) -> str:
|
|
76
78
|
return f"[ConvergenceError] {self.message}"
|
|
@@ -30,9 +30,11 @@ class FeaturePreprocessingError(PreprocessingError):
|
|
|
30
30
|
"""Raised when feature engineering fails."""
|
|
31
31
|
|
|
32
32
|
def __init__(self, feature: str, reason: Optional[str] = None) -> None:
|
|
33
|
-
super().__init__
|
|
33
|
+
# Assigned before super(): DataExceptError.__init__ sweeps the stored
|
|
34
|
+
# strings for URLs, and anything set afterwards escapes that.
|
|
34
35
|
self.feature = feature
|
|
35
36
|
self.reason = reason
|
|
37
|
+
super().__init__(step_name=f"feature_{feature}", details=reason)
|
|
36
38
|
|
|
37
39
|
|
|
38
40
|
class StorageError(PipelineError):
|
|
@@ -42,33 +42,72 @@ PLACEHOLDER = "***"
|
|
|
42
42
|
#: nothing. The structured field is redacted regardless of length.
|
|
43
43
|
MIN_REMOVABLE_SECRET_LENGTH = 8
|
|
44
44
|
|
|
45
|
-
#:
|
|
46
|
-
#:
|
|
47
|
-
#:
|
|
48
|
-
#:
|
|
49
|
-
|
|
45
|
+
#: Parameter-name tokens that mark a value as a secret. A name is split into
|
|
46
|
+
#: tokens on separators and camelCase boundaries, and matched token by token.
|
|
47
|
+
#:
|
|
48
|
+
#: Substring matching was tried first and was wrong in both directions: it
|
|
49
|
+
#: redacted "monkey", "design", "assign", "keyword" and "authors" -- mangling
|
|
50
|
+
#: ordinary debugging information -- while still missing "passphrase". The
|
|
51
|
+
#: point of keeping host, port and path is that the error stays actionable, and
|
|
52
|
+
#: shredding a legitimate query parameter works against that.
|
|
53
|
+
#:
|
|
54
|
+
#: Deliberately absent: "code", "state", "nonce" and "client_id". An OAuth
|
|
55
|
+
#: authorization code is a secret, but "code" is far more often a country
|
|
56
|
+
#: code, an HTTP status or a discount code, and redacting those would destroy
|
|
57
|
+
#: more debugging information than it protects. Checked against the parameter
|
|
58
|
+
#: names used by AWS SigV4, Azure SAS, Google Cloud and OAuth 2.
|
|
59
|
+
SENSITIVE_PARAM_TOKENS = frozenset(
|
|
50
60
|
{
|
|
61
|
+
"apikey",
|
|
51
62
|
"auth",
|
|
63
|
+
"authorization",
|
|
64
|
+
"bearer",
|
|
52
65
|
"credential",
|
|
66
|
+
"credentials",
|
|
67
|
+
"hmac",
|
|
68
|
+
"jwt",
|
|
53
69
|
"key",
|
|
70
|
+
"keys",
|
|
71
|
+
"passphrase",
|
|
54
72
|
"passwd",
|
|
55
73
|
"password",
|
|
56
74
|
"pwd",
|
|
75
|
+
"sas",
|
|
57
76
|
"secret",
|
|
77
|
+
"secrets",
|
|
58
78
|
"session",
|
|
59
79
|
"sig",
|
|
80
|
+
"signature",
|
|
60
81
|
"token",
|
|
82
|
+
"tokens",
|
|
61
83
|
}
|
|
62
84
|
)
|
|
63
85
|
|
|
64
|
-
#:
|
|
65
|
-
#:
|
|
66
|
-
|
|
86
|
+
#: Splits a parameter name into words: on separators, and between a lower-case
|
|
87
|
+
#: or digit character and an upper-case one, so ``accessToken`` yields
|
|
88
|
+
#: ``["access", "token"]``.
|
|
89
|
+
_NAME_TOKENS = re.compile(r"[A-Za-z0-9]+")
|
|
90
|
+
_CAMEL_BOUNDARY = re.compile(r"(?<=[a-z0-9])(?=[A-Z])")
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _tokens(name: str) -> list[str]:
|
|
94
|
+
words: list[str] = []
|
|
95
|
+
for chunk in _NAME_TOKENS.findall(name):
|
|
96
|
+
words.extend(part.lower() for part in _CAMEL_BOUNDARY.split(chunk) if part)
|
|
97
|
+
return words
|
|
67
98
|
|
|
68
99
|
|
|
69
100
|
def _is_sensitive(name: str) -> bool:
|
|
70
|
-
|
|
71
|
-
|
|
101
|
+
return any(token in SENSITIVE_PARAM_TOKENS for token in _tokens(name))
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
#: Finds URLs inside free-form text, so a credential cannot slip through in a
|
|
105
|
+
#: caller-supplied message or in the text of a wrapped exception.
|
|
106
|
+
# No word-boundary anchor: a URL can directly follow a word character, as
|
|
107
|
+
# in the step name "feature_https://..." that FeaturePreprocessingError
|
|
108
|
+
# builds. The scheme character class excludes "_", so a match still starts
|
|
109
|
+
# at the scheme rather than mid-word.
|
|
110
|
+
_URL_IN_TEXT = re.compile(r"[a-zA-Z][a-zA-Z0-9+.\-]*://[^\s'\"<>,;)\]}]+")
|
|
72
111
|
|
|
73
112
|
|
|
74
113
|
def fingerprint(value: str) -> str:
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|