DataExcept 0.4.1__tar.gz → 0.4.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {dataexcept-0.4.1 → dataexcept-0.4.2}/CHANGELOG.md +46 -1
  2. {dataexcept-0.4.1 → dataexcept-0.4.2}/CITATION.cff +1 -1
  3. {dataexcept-0.4.1 → dataexcept-0.4.2}/PKG-INFO +3 -3
  4. {dataexcept-0.4.1 → dataexcept-0.4.2}/README.md +2 -2
  5. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/base.py +18 -13
  6. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/datascience_exceptions/training.py +3 -1
  7. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/pipeline_exceptions.py +3 -1
  8. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/redaction.py +49 -10
  9. {dataexcept-0.4.1 → dataexcept-0.4.2}/pyproject.toml +1 -1
  10. {dataexcept-0.4.1 → dataexcept-0.4.2}/LICENSE +0 -0
  11. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/__init__.py +0 -0
  12. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/__main__.py +0 -0
  13. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/_deprecation.py +0 -0
  14. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/_validation.py +0 -0
  15. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/database_exceptions.py +0 -0
  16. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/dataengineering_exceptions.py +0 -0
  17. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/datascience_exceptions/__init__.py +0 -0
  18. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/datascience_exceptions/base.py +0 -0
  19. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/datascience_exceptions/ingestion.py +0 -0
  20. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/datascience_exceptions/operations.py +0 -0
  21. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/__init__.py +0 -0
  22. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/authentication.py +0 -0
  23. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/base.py +0 -0
  24. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/configuration.py +0 -0
  25. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/external.py +0 -0
  26. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/lifecycle.py +0 -0
  27. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/notification.py +0 -0
  28. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/parsing.py +0 -0
  29. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/scheduling.py +0 -0
  30. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/exceptions/validation.py +0 -0
  31. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/io_exceptions.py +0 -0
  32. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/job_exceptions.py +0 -0
  33. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/logging_helpers.py +0 -0
  34. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/network_exceptions.py +0 -0
  35. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/pandas_exceptions.py +0 -0
  36. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/py.typed +0 -0
  37. {dataexcept-0.4.1 → dataexcept-0.4.2}/dataexcept/security_exceptions.py +0 -0
@@ -7,6 +7,50 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.4.2] - 2026-08-25
11
+
12
+ ### Changed
13
+
14
+ - `SECURITY.md` states two further boundaries rather than leaving them to be
15
+ discovered: state attached to an exception *after* it is constructed is not
16
+ swept, and a URL nested inside a value you pass is rendered redacted but the
17
+ object itself is not rewritten. Walking and rewriting arbitrary caller data
18
+ structures would be a surprising thing for an exception library to do, and
19
+ could not be complete anyway.
20
+
21
+ ### Fixed
22
+
23
+ - **Sensitive query parameters are matched by token, not substring.** The
24
+ substring rule was wrong in both directions: it redacted `monkey`, `design`,
25
+ `assign`, `keyword` and `authors`, mangling ordinary debugging information,
26
+ while still missing `passphrase`. Parameter names are now split on
27
+ separators and camelCase and matched word by word, so `X-Amz-Signature`,
28
+ `accessToken` and `client_secret` are caught and `?monkey=bobo` is left
29
+ alone.
30
+
31
+ ### Security
32
+
33
+ - **`jwt`, `bearer`, `hmac` and `sas` are recognised as secret parameter
34
+ names.** Checked the token set against the parameters actually used by AWS
35
+ SigV4, Azure SAS, Google Cloud and OAuth 2: the signature and credential
36
+ parameters were already covered, but those four were not. `code`, `state`,
37
+ `nonce` and `client_id` are deliberately still ignored — an OAuth code is a
38
+ secret, but the name is far more often a country code, an HTTP status or a
39
+ discount code, and redacting those would destroy more than it protects.
40
+ - **Three ways a URL could still reach a message, all closed.** The redaction
41
+ boundary scrubbed the message and a handful of named fields, but 18 classes
42
+ interpolate some *other* attribute into `__str__` — a field, a column, a
43
+ resource — and a caller can put a URL in any of them. Every stored string is
44
+ now swept, so no class leaks a credential-bearing URL through its message,
45
+ its attributes or its `args`. A test fills every string argument of every
46
+ class with one and checks all three surfaces.
47
+ - Two classes assigned their attributes *after* calling `super().__init__`, so
48
+ the sweep never saw them. Both now assign first, and a test fails if any
49
+ constructor does it again.
50
+ - The URL pattern carried a `` anchor, so a URL directly following a word
51
+ character was never matched — including `feature_https://...`, the step name
52
+ `FeaturePreprocessingError` builds from its own argument.
53
+
10
54
  ## [0.4.1] - 2026-08-25
11
55
 
12
56
  ### Added
@@ -385,7 +429,8 @@ First public release.
385
429
  - Published to PyPI via OIDC trusted publishing; no long-lived API token is
386
430
  involved in a release.
387
431
 
388
- [Unreleased]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.4.1...HEAD
432
+ [Unreleased]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.4.2...HEAD
433
+ [0.4.2]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.4.1...v0.4.2
389
434
  [0.4.1]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.4.0...v0.4.1
390
435
  [0.4.0]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.3.0...v0.4.0
391
436
  [0.3.0]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.2.1...v0.3.0
@@ -1,7 +1,7 @@
1
1
  cff-version: 1.2.0
2
2
  message: "If you use this software, please cite it using the following metadata."
3
3
  title: "DataExcept"
4
- version: "0.4.1"
4
+ version: "0.4.2"
5
5
  authors:
6
6
  - family-names: "Ribeiro"
7
7
  given-names: "Diogo"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: DataExcept
3
- Version: 0.4.1
3
+ Version: 0.4.2
4
4
  Summary: A Python package providing structured, easily-extendable custom exception types.
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -225,7 +225,7 @@ BiasDetectionError
225
225
 
226
226
  # Check version
227
227
  $ dataexcept --version
228
- dataexcept 0.4.1
228
+ dataexcept 0.4.2
229
229
  ```
230
230
 
231
231
  ## 🎯 Use Cases
@@ -375,7 +375,7 @@ If you use DataExcept in your research, please cite it:
375
375
  author = {Ribeiro, Diogo},
376
376
  title = {DataExcept: Structured Exception Handling for Data Science},
377
377
  url = {https://github.com/DiogoRibeiro7/DataExcept},
378
- version = {0.4.1},
378
+ version = {0.4.2},
379
379
  year = {2026},
380
380
  publisher = {GitHub}
381
381
  }
@@ -198,7 +198,7 @@ BiasDetectionError
198
198
 
199
199
  # Check version
200
200
  $ dataexcept --version
201
- dataexcept 0.4.1
201
+ dataexcept 0.4.2
202
202
  ```
203
203
 
204
204
  ## 🎯 Use Cases
@@ -348,7 +348,7 @@ If you use DataExcept in your research, please cite it:
348
348
  author = {Ribeiro, Diogo},
349
349
  title = {DataExcept: Structured Exception Handling for Data Science},
350
350
  url = {https://github.com/DiogoRibeiro7/DataExcept},
351
- version = {0.4.1},
351
+ version = {0.4.2},
352
352
  year = {2026},
353
353
  publisher = {GitHub}
354
354
  }
@@ -138,22 +138,27 @@ class DataExceptError(Exception):
138
138
  _keep_url_path = True
139
139
 
140
140
  def __init__(self, *args: Any) -> None:
141
- # One boundary for the whole hierarchy: whatever built the message --
142
- # a constructor, a caller-supplied `message`, or the text of a wrapped
143
- # exception quoting the original URL -- it is scrubbed here. Redacting
144
- # only the structured argument leaves all three of those routes open.
141
+ # One boundary for the whole hierarchy. Whatever built the message -- a
142
+ # constructor, a caller-supplied `message`, or the text of a wrapped
143
+ # exception quoting the original URL -- it is scrubbed here, because
144
+ # redacting only the structured argument leaves all three routes open.
145
145
  keep_path = type(self)._keep_url_path
146
146
  if args and isinstance(args[0], str):
147
147
  args = (redact_urls_in_text(args[0], keep_path=keep_path),) + args[1:]
148
- # Many classes also store the message on self.message and render *that*
149
- # in __str__, so scrubbing args alone would leave the rendered form
150
- # untouched. Subclasses set it before calling up, so it is here to fix.
151
- stored = self.__dict__.get("message")
152
- if isinstance(stored, str):
153
- # Written straight into __dict__, symmetric with the read above:
154
- # this rewrites state a subclass already stored, rather than the
155
- # base declaring an attribute of its own.
156
- self.__dict__["message"] = redact_urls_in_text(stored, keep_path=keep_path)
148
+
149
+ # Many classes store the message on self.message and render *that* in
150
+ # __str__, and 18 interpolate some other attribute -- a field, a
151
+ # column, a resource -- any of which a caller can fill with a URL. So
152
+ # every stored string is swept, not just the message.
153
+ #
154
+ # redact_urls_in_text rather than redact_if_url: a message has the URL
155
+ # embedded in prose, and redact_if_url only handles a value that is
156
+ # wholly a URL. It is a no-op on anything without "://" in it, so
157
+ # ordinary names and file paths are untouched.
158
+ for name, value in list(self.__dict__.items()):
159
+ if isinstance(value, str) and "://" in value:
160
+ self.__dict__[name] = redact_urls_in_text(value, keep_path=keep_path)
161
+
157
162
  super().__init__(*args)
158
163
  # Constructors that wrap another exception record it on an attribute.
159
164
  # Mirroring it into __cause__ is what makes a traceback print the
@@ -69,8 +69,10 @@ class ConvergenceError(ModelTrainingError):
69
69
  f"Model '{model_type}' failed to converge after "
70
70
  f"{iterations} iterations"
71
71
  )
72
- super().__init__(model_type=model_type, epoch=None, message=message)
72
+ # Assigned before super(): DataExceptError.__init__ sweeps the stored
73
+ # strings for URLs, and anything set afterwards escapes that.
73
74
  self.iterations = iterations
75
+ super().__init__(model_type=model_type, epoch=None, message=message)
74
76
 
75
77
  def __str__(self) -> str:
76
78
  return f"[ConvergenceError] {self.message}"
@@ -30,9 +30,11 @@ class FeaturePreprocessingError(PreprocessingError):
30
30
  """Raised when feature engineering fails."""
31
31
 
32
32
  def __init__(self, feature: str, reason: Optional[str] = None) -> None:
33
- super().__init__(step_name=f"feature_{feature}", details=reason)
33
+ # Assigned before super(): DataExceptError.__init__ sweeps the stored
34
+ # strings for URLs, and anything set afterwards escapes that.
34
35
  self.feature = feature
35
36
  self.reason = reason
37
+ super().__init__(step_name=f"feature_{feature}", details=reason)
36
38
 
37
39
 
38
40
  class StorageError(PipelineError):
@@ -42,33 +42,72 @@ PLACEHOLDER = "***"
42
42
  #: nothing. The structured field is redacted regardless of length.
43
43
  MIN_REMOVABLE_SECRET_LENGTH = 8
44
44
 
45
- #: Substrings that mark a query or fragment parameter as carrying a secret.
46
- #: Matched anywhere in the parameter name, case insensitively, so
47
- #: ``X-Amz-Signature``, ``auth_token`` and ``refresh_token`` are all covered.
48
- #: Over-matching here is harmless; under-matching leaks.
49
- SENSITIVE_PARAM_MARKERS = frozenset(
45
+ #: Parameter-name tokens that mark a value as a secret. A name is split into
46
+ #: tokens on separators and camelCase boundaries, and matched token by token.
47
+ #:
48
+ #: Substring matching was tried first and was wrong in both directions: it
49
+ #: redacted "monkey", "design", "assign", "keyword" and "authors" -- mangling
50
+ #: ordinary debugging information -- while still missing "passphrase". The
51
+ #: point of keeping host, port and path is that the error stays actionable, and
52
+ #: shredding a legitimate query parameter works against that.
53
+ #:
54
+ #: Deliberately absent: "code", "state", "nonce" and "client_id". An OAuth
55
+ #: authorization code is a secret, but "code" is far more often a country
56
+ #: code, an HTTP status or a discount code, and redacting those would destroy
57
+ #: more debugging information than it protects. Checked against the parameter
58
+ #: names used by AWS SigV4, Azure SAS, Google Cloud and OAuth 2.
59
+ SENSITIVE_PARAM_TOKENS = frozenset(
50
60
  {
61
+ "apikey",
51
62
  "auth",
63
+ "authorization",
64
+ "bearer",
52
65
  "credential",
66
+ "credentials",
67
+ "hmac",
68
+ "jwt",
53
69
  "key",
70
+ "keys",
71
+ "passphrase",
54
72
  "passwd",
55
73
  "password",
56
74
  "pwd",
75
+ "sas",
57
76
  "secret",
77
+ "secrets",
58
78
  "session",
59
79
  "sig",
80
+ "signature",
60
81
  "token",
82
+ "tokens",
61
83
  }
62
84
  )
63
85
 
64
- #: Finds URLs inside free-form text, so a credential cannot slip through in a
65
- #: caller-supplied message or in the text of a wrapped exception.
66
- _URL_IN_TEXT = re.compile(r"\b[a-zA-Z][a-zA-Z0-9+.\-]*://[^\s'\"<>,;)\]}]+")
86
+ #: Splits a parameter name into words: on separators, and between a lower-case
87
+ #: or digit character and an upper-case one, so ``accessToken`` yields
88
+ #: ``["access", "token"]``.
89
+ _NAME_TOKENS = re.compile(r"[A-Za-z0-9]+")
90
+ _CAMEL_BOUNDARY = re.compile(r"(?<=[a-z0-9])(?=[A-Z])")
91
+
92
+
93
+ def _tokens(name: str) -> list[str]:
94
+ words: list[str] = []
95
+ for chunk in _NAME_TOKENS.findall(name):
96
+ words.extend(part.lower() for part in _CAMEL_BOUNDARY.split(chunk) if part)
97
+ return words
67
98
 
68
99
 
69
100
  def _is_sensitive(name: str) -> bool:
70
- lowered = name.lower()
71
- return any(marker in lowered for marker in SENSITIVE_PARAM_MARKERS)
101
+ return any(token in SENSITIVE_PARAM_TOKENS for token in _tokens(name))
102
+
103
+
104
+ #: Finds URLs inside free-form text, so a credential cannot slip through in a
105
+ #: caller-supplied message or in the text of a wrapped exception.
106
+ # No word-boundary anchor: a URL can directly follow a word character, as
107
+ # in the step name "feature_https://..." that FeaturePreprocessingError
108
+ # builds. The scheme character class excludes "_", so a match still starts
109
+ # at the scheme rather than mid-word.
110
+ _URL_IN_TEXT = re.compile(r"[a-zA-Z][a-zA-Z0-9+.\-]*://[^\s'\"<>,;)\]}]+")
72
111
 
73
112
 
74
113
  def fingerprint(value: str) -> str:
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "DataExcept"
3
- version = "0.4.1"
3
+ version = "0.4.2"
4
4
  description = "A Python package providing structured, easily-extendable custom exception types."
5
5
  readme = "README.md"
6
6
  license = "MIT"
File without changes