DataExcept 0.4.1__tar.gz → 0.4.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {dataexcept-0.4.1 → dataexcept-0.4.3}/CHANGELOG.md +92 -1
  2. {dataexcept-0.4.1 → dataexcept-0.4.3}/CITATION.cff +1 -1
  3. {dataexcept-0.4.1 → dataexcept-0.4.3}/PKG-INFO +3 -3
  4. {dataexcept-0.4.1 → dataexcept-0.4.3}/README.md +2 -2
  5. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/base.py +18 -13
  6. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/datascience_exceptions/training.py +3 -1
  7. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/logging_helpers.py +56 -7
  8. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/pipeline_exceptions.py +3 -1
  9. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/redaction.py +53 -11
  10. {dataexcept-0.4.1 → dataexcept-0.4.3}/pyproject.toml +2 -1
  11. {dataexcept-0.4.1 → dataexcept-0.4.3}/LICENSE +0 -0
  12. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/__init__.py +0 -0
  13. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/__main__.py +0 -0
  14. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/_deprecation.py +0 -0
  15. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/_validation.py +0 -0
  16. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/database_exceptions.py +0 -0
  17. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/dataengineering_exceptions.py +0 -0
  18. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/datascience_exceptions/__init__.py +0 -0
  19. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/datascience_exceptions/base.py +0 -0
  20. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/datascience_exceptions/ingestion.py +0 -0
  21. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/datascience_exceptions/operations.py +0 -0
  22. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/exceptions/__init__.py +0 -0
  23. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/exceptions/authentication.py +0 -0
  24. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/exceptions/base.py +0 -0
  25. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/exceptions/configuration.py +0 -0
  26. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/exceptions/external.py +0 -0
  27. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/exceptions/lifecycle.py +0 -0
  28. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/exceptions/notification.py +0 -0
  29. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/exceptions/parsing.py +0 -0
  30. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/exceptions/scheduling.py +0 -0
  31. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/exceptions/validation.py +0 -0
  32. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/io_exceptions.py +0 -0
  33. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/job_exceptions.py +0 -0
  34. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/network_exceptions.py +0 -0
  35. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/pandas_exceptions.py +0 -0
  36. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/py.typed +0 -0
  37. {dataexcept-0.4.1 → dataexcept-0.4.3}/dataexcept/security_exceptions.py +0 -0
@@ -7,6 +7,95 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ### Fixed
11
+
12
+ - The release workflow's `verify-wheel` job installed a hand-written list of
13
+ test requirements, which went stale the moment `hypothesis` was added — the
14
+ suite could not even be collected against the built wheel. It now installs
15
+ the dev group from the lock, so the list cannot drift again. The job did its
16
+ job: it failed, and publishing was skipped.
17
+ - `test_no_module_imports_tomllib_without_a_fallback` walked the whole tree, so
18
+ it picked up any virtualenv sitting in it. It now scans the project's own
19
+ source directories by name; site-packages is not ours to police.
20
+
21
+ ## [0.4.3] - 2026-08-25
22
+
23
+ ### Added
24
+
25
+ - Property-based tests over the logging helpers and the CLI: generated context
26
+ values, including hostile ones, must never make `log_exception` raise; the
27
+ normalised context must be strict-JSON encodable; `log_and_raise` must
28
+ re-raise the *same* exception with its traceback intact; and the CLI must
29
+ raise nothing but `SystemExit` for any argument list.
30
+ - **Property-based tests over the exception hierarchy**, using Hypothesis.
31
+ Generated text is fed to every class and four properties are asserted for
32
+ each: construction raises nothing but `TypeError`, a message built from real
33
+ input is non-empty, the exception survives a pickle round trip with its type,
34
+ message and cause intact, and a credential-bearing URL is never rendered.
35
+ Every constructor defect the reviews found — `MergeKeyError("id")` becoming
36
+ `['i', 'd']`, `is_number(True)`, and the one below — was in argument
37
+ handling and was found by inspection rather than by a test.
38
+
39
+ ### Fixed
40
+
41
+ - **`log_exception` could raise, replacing the caller's error with its own.**
42
+ A context value whose `__repr__` raises propagated straight out of the
43
+ logging helper — so a helper called while handling a failure made the failure
44
+ worse. Nothing in context normalisation can raise now; a value that cannot be
45
+ described at all becomes `<unrepresentable TypeName>`.
46
+ - **`NaN` and `Infinity` in a log context produced invalid JSON.**
47
+ `json.dumps` emits them bare, which a strict consumer downstream rejects.
48
+ They are stringified instead.
49
+ - `ApiError`, `DatabaseConnectionError` and `WebhookError` raised
50
+ `AttributeError: 'object' object has no attribute 'decode'` when given a
51
+ non-string, because the redaction helpers passed whatever they were given to
52
+ `urlsplit`. That masked the caller's actual mistake with a message about
53
+ `.decode`. The helpers now return a non-string untouched.
54
+
55
+ ## [0.4.2] - 2026-08-25
56
+
57
+ ### Changed
58
+
59
+ - `SECURITY.md` states two further boundaries rather than leaving them to be
60
+ discovered: state attached to an exception *after* it is constructed is not
61
+ swept, and a URL nested inside a value you pass is rendered redacted but the
62
+ object itself is not rewritten. Walking and rewriting arbitrary caller data
63
+ structures would be a surprising thing for an exception library to do, and
64
+ could not be complete anyway.
65
+
66
+ ### Fixed
67
+
68
+ - **Sensitive query parameters are matched by token, not substring.** The
69
+ substring rule was wrong in both directions: it redacted `monkey`, `design`,
70
+ `assign`, `keyword` and `authors`, mangling ordinary debugging information,
71
+ while still missing `passphrase`. Parameter names are now split on
72
+ separators and camelCase and matched word by word, so `X-Amz-Signature`,
73
+ `accessToken` and `client_secret` are caught and `?monkey=bobo` is left
74
+ alone.
75
+
76
+ ### Security
77
+
78
+ - **`jwt`, `bearer`, `hmac` and `sas` are recognised as secret parameter
79
+ names.** Checked the token set against the parameters actually used by AWS
80
+ SigV4, Azure SAS, Google Cloud and OAuth 2: the signature and credential
81
+ parameters were already covered, but those four were not. `code`, `state`,
82
+ `nonce` and `client_id` are deliberately still ignored — an OAuth code is a
83
+ secret, but the name is far more often a country code, an HTTP status or a
84
+ discount code, and redacting those would destroy more than it protects.
85
+ - **Three ways a URL could still reach a message, all closed.** The redaction
86
+ boundary scrubbed the message and a handful of named fields, but 18 classes
87
+ interpolate some *other* attribute into `__str__` — a field, a column, a
88
+ resource — and a caller can put a URL in any of them. Every stored string is
89
+ now swept, so no class leaks a credential-bearing URL through its message,
90
+ its attributes or its `args`. A test fills every string argument of every
91
+ class with one and checks all three surfaces.
92
+ - Two classes assigned their attributes *after* calling `super().__init__`, so
93
+ the sweep never saw them. Both now assign first, and a test fails if any
94
+ constructor does it again.
95
+ - The URL pattern carried a `` anchor, so a URL directly following a word
96
+ character was never matched — including `feature_https://...`, the step name
97
+ `FeaturePreprocessingError` builds from its own argument.
98
+
10
99
  ## [0.4.1] - 2026-08-25
11
100
 
12
101
  ### Added
@@ -385,7 +474,9 @@ First public release.
385
474
  - Published to PyPI via OIDC trusted publishing; no long-lived API token is
386
475
  involved in a release.
387
476
 
388
- [Unreleased]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.4.1...HEAD
477
+ [Unreleased]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.4.3...HEAD
478
+ [0.4.3]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.4.2...v0.4.3
479
+ [0.4.2]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.4.1...v0.4.2
389
480
  [0.4.1]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.4.0...v0.4.1
390
481
  [0.4.0]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.3.0...v0.4.0
391
482
  [0.3.0]: https://github.com/DiogoRibeiro7/DataExcept/compare/v0.2.1...v0.3.0
@@ -1,7 +1,7 @@
1
1
  cff-version: 1.2.0
2
2
  message: "If you use this software, please cite it using the following metadata."
3
3
  title: "DataExcept"
4
- version: "0.4.1"
4
+ version: "0.4.3"
5
5
  authors:
6
6
  - family-names: "Ribeiro"
7
7
  given-names: "Diogo"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: DataExcept
3
- Version: 0.4.1
3
+ Version: 0.4.3
4
4
  Summary: A Python package providing structured, easily-extendable custom exception types.
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -225,7 +225,7 @@ BiasDetectionError
225
225
 
226
226
  # Check version
227
227
  $ dataexcept --version
228
- dataexcept 0.4.1
228
+ dataexcept 0.4.3
229
229
  ```
230
230
 
231
231
  ## 🎯 Use Cases
@@ -375,7 +375,7 @@ If you use DataExcept in your research, please cite it:
375
375
  author = {Ribeiro, Diogo},
376
376
  title = {DataExcept: Structured Exception Handling for Data Science},
377
377
  url = {https://github.com/DiogoRibeiro7/DataExcept},
378
- version = {0.4.1},
378
+ version = {0.4.3},
379
379
  year = {2026},
380
380
  publisher = {GitHub}
381
381
  }
@@ -198,7 +198,7 @@ BiasDetectionError
198
198
 
199
199
  # Check version
200
200
  $ dataexcept --version
201
- dataexcept 0.4.1
201
+ dataexcept 0.4.3
202
202
  ```
203
203
 
204
204
  ## 🎯 Use Cases
@@ -348,7 +348,7 @@ If you use DataExcept in your research, please cite it:
348
348
  author = {Ribeiro, Diogo},
349
349
  title = {DataExcept: Structured Exception Handling for Data Science},
350
350
  url = {https://github.com/DiogoRibeiro7/DataExcept},
351
- version = {0.4.1},
351
+ version = {0.4.3},
352
352
  year = {2026},
353
353
  publisher = {GitHub}
354
354
  }
@@ -138,22 +138,27 @@ class DataExceptError(Exception):
138
138
  _keep_url_path = True
139
139
 
140
140
  def __init__(self, *args: Any) -> None:
141
- # One boundary for the whole hierarchy: whatever built the message --
142
- # a constructor, a caller-supplied `message`, or the text of a wrapped
143
- # exception quoting the original URL -- it is scrubbed here. Redacting
144
- # only the structured argument leaves all three of those routes open.
141
+ # One boundary for the whole hierarchy. Whatever built the message -- a
142
+ # constructor, a caller-supplied `message`, or the text of a wrapped
143
+ # exception quoting the original URL -- it is scrubbed here, because
144
+ # redacting only the structured argument leaves all three routes open.
145
145
  keep_path = type(self)._keep_url_path
146
146
  if args and isinstance(args[0], str):
147
147
  args = (redact_urls_in_text(args[0], keep_path=keep_path),) + args[1:]
148
- # Many classes also store the message on self.message and render *that*
149
- # in __str__, so scrubbing args alone would leave the rendered form
150
- # untouched. Subclasses set it before calling up, so it is here to fix.
151
- stored = self.__dict__.get("message")
152
- if isinstance(stored, str):
153
- # Written straight into __dict__, symmetric with the read above:
154
- # this rewrites state a subclass already stored, rather than the
155
- # base declaring an attribute of its own.
156
- self.__dict__["message"] = redact_urls_in_text(stored, keep_path=keep_path)
148
+
149
+ # Many classes store the message on self.message and render *that* in
150
+ # __str__, and 18 interpolate some other attribute -- a field, a
151
+ # column, a resource -- any of which a caller can fill with a URL. So
152
+ # every stored string is swept, not just the message.
153
+ #
154
+ # redact_urls_in_text rather than redact_if_url: a message has the URL
155
+ # embedded in prose, and redact_if_url only handles a value that is
156
+ # wholly a URL. It is a no-op on anything without "://" in it, so
157
+ # ordinary names and file paths are untouched.
158
+ for name, value in list(self.__dict__.items()):
159
+ if isinstance(value, str) and "://" in value:
160
+ self.__dict__[name] = redact_urls_in_text(value, keep_path=keep_path)
161
+
157
162
  super().__init__(*args)
158
163
  # Constructors that wrap another exception record it on an attribute.
159
164
  # Mirroring it into __cause__ is what makes a traceback print the
@@ -69,8 +69,10 @@ class ConvergenceError(ModelTrainingError):
69
69
  f"Model '{model_type}' failed to converge after "
70
70
  f"{iterations} iterations"
71
71
  )
72
- super().__init__(model_type=model_type, epoch=None, message=message)
72
+ # Assigned before super(): DataExceptError.__init__ sweeps the stored
73
+ # strings for URLs, and anything set afterwards escapes that.
73
74
  self.iterations = iterations
75
+ super().__init__(model_type=model_type, epoch=None, message=message)
74
76
 
75
77
  def __str__(self) -> str:
76
78
  return f"[ConvergenceError] {self.message}"
@@ -12,6 +12,9 @@ from .redaction import redact_urls_in_text
12
12
 
13
13
  Context = Mapping[str, Any]
14
14
 
15
+ #: Sentinel: the value could not be coerced into anything JSON will take.
16
+ _UNCOERCIBLE = object()
17
+
15
18
  __all__ = [
16
19
  "Context",
17
20
  "log_and_raise",
@@ -20,15 +23,61 @@ __all__ = [
20
23
  ]
21
24
 
22
25
 
23
- def _normalize_context_value(value: Any) -> Any:
26
+ def _is_json_safe(value: Any) -> bool:
27
+ """True if a strict JSON encoder will accept *value* as it stands.
28
+
29
+ ``allow_nan=False`` because ``json.dumps`` otherwise emits bare ``NaN`` and
30
+ ``Infinity``, which are not valid JSON and will be rejected downstream.
31
+ """
24
32
  try:
25
- json.dumps(value)
26
- return value
27
- except TypeError:
33
+ json.dumps(value, allow_nan=False)
34
+ except (TypeError, ValueError):
35
+ return False
36
+ return True
37
+
38
+
39
+ def _coerced(value: Any) -> Any:
40
+ """Round-trip *value* through JSON, stringifying whatever will not encode.
41
+
42
+ Returns ``_UNCOERCIBLE`` rather than raising: this runs while the caller is
43
+ already handling a failure.
44
+ """
45
+ try:
46
+ return json.loads(json.dumps(value, default=str, allow_nan=False))
47
+ except Exception:
48
+ return _UNCOERCIBLE
49
+
50
+
51
+ def _described(value: Any) -> str:
52
+ """Describe *value* without letting it raise.
53
+
54
+ An object may define a ``__repr__`` that raises. Naming the type is the
55
+ most that can be said without invoking anything the object controls.
56
+ """
57
+ try:
58
+ return repr(value)
59
+ except Exception:
28
60
  try:
29
- return json.loads(json.dumps(value, default=str))
30
- except Exception: # pragma: no cover - extremely defensive
31
- return repr(value)
61
+ return f"<unrepresentable {type(value).__name__}>"
62
+ except Exception: # pragma: no cover - a type with a hostile __name__
63
+ return "<unrepresentable>"
64
+
65
+
66
+ def _normalize_context_value(value: Any) -> Any:
67
+ """Return *value* in a form a strict JSON log encoder will accept.
68
+
69
+ Nothing here may raise. This runs while the caller is already handling a
70
+ failure, and an exception escaping would replace their error with one about
71
+ logging it -- so even a hostile ``__repr__`` has to be survivable.
72
+ """
73
+ if _is_json_safe(value):
74
+ return value
75
+
76
+ coerced = _coerced(value)
77
+ if coerced is not _UNCOERCIBLE:
78
+ return coerced
79
+
80
+ return _described(value)
32
81
 
33
82
 
34
83
  def _build_extra(context: Context | None) -> dict[str, Any] | None:
@@ -30,9 +30,11 @@ class FeaturePreprocessingError(PreprocessingError):
30
30
  """Raised when feature engineering fails."""
31
31
 
32
32
  def __init__(self, feature: str, reason: Optional[str] = None) -> None:
33
- super().__init__(step_name=f"feature_{feature}", details=reason)
33
+ # Assigned before super(): DataExceptError.__init__ sweeps the stored
34
+ # strings for URLs, and anything set afterwards escapes that.
34
35
  self.feature = feature
35
36
  self.reason = reason
37
+ super().__init__(step_name=f"feature_{feature}", details=reason)
36
38
 
37
39
 
38
40
  class StorageError(PipelineError):
@@ -42,33 +42,72 @@ PLACEHOLDER = "***"
42
42
  #: nothing. The structured field is redacted regardless of length.
43
43
  MIN_REMOVABLE_SECRET_LENGTH = 8
44
44
 
45
- #: Substrings that mark a query or fragment parameter as carrying a secret.
46
- #: Matched anywhere in the parameter name, case insensitively, so
47
- #: ``X-Amz-Signature``, ``auth_token`` and ``refresh_token`` are all covered.
48
- #: Over-matching here is harmless; under-matching leaks.
49
- SENSITIVE_PARAM_MARKERS = frozenset(
45
+ #: Parameter-name tokens that mark a value as a secret. A name is split into
46
+ #: tokens on separators and camelCase boundaries, and matched token by token.
47
+ #:
48
+ #: Substring matching was tried first and was wrong in both directions: it
49
+ #: redacted "monkey", "design", "assign", "keyword" and "authors" -- mangling
50
+ #: ordinary debugging information -- while still missing "passphrase". The
51
+ #: point of keeping host, port and path is that the error stays actionable, and
52
+ #: shredding a legitimate query parameter works against that.
53
+ #:
54
+ #: Deliberately absent: "code", "state", "nonce" and "client_id". An OAuth
55
+ #: authorization code is a secret, but "code" is far more often a country
56
+ #: code, an HTTP status or a discount code, and redacting those would destroy
57
+ #: more debugging information than it protects. Checked against the parameter
58
+ #: names used by AWS SigV4, Azure SAS, Google Cloud and OAuth 2.
59
+ SENSITIVE_PARAM_TOKENS = frozenset(
50
60
  {
61
+ "apikey",
51
62
  "auth",
63
+ "authorization",
64
+ "bearer",
52
65
  "credential",
66
+ "credentials",
67
+ "hmac",
68
+ "jwt",
53
69
  "key",
70
+ "keys",
71
+ "passphrase",
54
72
  "passwd",
55
73
  "password",
56
74
  "pwd",
75
+ "sas",
57
76
  "secret",
77
+ "secrets",
58
78
  "session",
59
79
  "sig",
80
+ "signature",
60
81
  "token",
82
+ "tokens",
61
83
  }
62
84
  )
63
85
 
64
- #: Finds URLs inside free-form text, so a credential cannot slip through in a
65
- #: caller-supplied message or in the text of a wrapped exception.
66
- _URL_IN_TEXT = re.compile(r"\b[a-zA-Z][a-zA-Z0-9+.\-]*://[^\s'\"<>,;)\]}]+")
86
+ #: Splits a parameter name into words: on separators, and between a lower-case
87
+ #: or digit character and an upper-case one, so ``accessToken`` yields
88
+ #: ``["access", "token"]``.
89
+ _NAME_TOKENS = re.compile(r"[A-Za-z0-9]+")
90
+ _CAMEL_BOUNDARY = re.compile(r"(?<=[a-z0-9])(?=[A-Z])")
91
+
92
+
93
+ def _tokens(name: str) -> list[str]:
94
+ words: list[str] = []
95
+ for chunk in _NAME_TOKENS.findall(name):
96
+ words.extend(part.lower() for part in _CAMEL_BOUNDARY.split(chunk) if part)
97
+ return words
67
98
 
68
99
 
69
100
  def _is_sensitive(name: str) -> bool:
70
- lowered = name.lower()
71
- return any(marker in lowered for marker in SENSITIVE_PARAM_MARKERS)
101
+ return any(token in SENSITIVE_PARAM_TOKENS for token in _tokens(name))
102
+
103
+
104
+ #: Finds URLs inside free-form text, so a credential cannot slip through in a
105
+ #: caller-supplied message or in the text of a wrapped exception.
106
+ # No word-boundary anchor: a URL can directly follow a word character, as
107
+ # in the step name "feature_https://..." that FeaturePreprocessingError
108
+ # builds. The scheme character class excludes "_", so a match still starts
109
+ # at the scheme rather than mid-word.
110
+ _URL_IN_TEXT = re.compile(r"[a-zA-Z][a-zA-Z0-9+.\-]*://[^\s'\"<>,;)\]}]+")
72
111
 
73
112
 
74
113
  def fingerprint(value: str) -> str:
@@ -131,7 +170,10 @@ def redact_url(url: Optional[str], *, keep_path: bool = True) -> Optional[str]:
131
170
  incoming webhook URL is the common case -- Slack, Discord and others put
132
171
  the secret in the path, so preserving it would defeat the point.
133
172
  """
134
- if not url:
173
+ if not url or not isinstance(url, str):
174
+ # Anything that is not a string is handed back untouched. urlsplit
175
+ # would raise AttributeError from inside urllib, masking whatever the
176
+ # caller's real mistake was with a message about `.decode`.
135
177
  return url
136
178
 
137
179
  try:
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "DataExcept"
3
- version = "0.4.1"
3
+ version = "0.4.3"
4
4
  description = "A Python package providing structured, easily-extendable custom exception types."
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -43,6 +43,7 @@ include = [
43
43
  black = "*"
44
44
  coverage = "*"
45
45
  flake8 = "*"
46
+ hypothesis = "*"
46
47
  isort = "*"
47
48
  mypy = "*"
48
49
  pre-commit = "*"
File without changes