cleanframe-engine 0.3.0__tar.gz → 0.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/CHANGELOG.md +63 -0
  2. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/PKG-INFO +2 -2
  3. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/README.md +1 -1
  4. cleanframe_engine-0.3.1/cleanframe/_version.py +1 -0
  5. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/codegen.py +1 -1
  6. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/dates.py +1 -3
  7. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/fingerprint.py +11 -7
  8. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/llm.py +2 -1
  9. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/ops.py +29 -2
  10. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/planner.py +2 -1
  11. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/profile.py +3 -27
  12. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/report.py +11 -6
  13. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/types.py +18 -0
  14. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/validate.py +4 -2
  15. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/installation.md +14 -0
  16. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/pyproject.toml +3 -1
  17. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_audit_hardening.py +1 -2
  18. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_llm_schema.py +2 -2
  19. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_release_hardening.py +10 -0
  20. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave1_hardening.py +2 -3
  21. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave2_vectorization.py +1 -2
  22. cleanframe_engine-0.3.0/cleanframe/_version.py +0 -1
  23. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/.gitignore +0 -0
  24. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/CONTRIBUTING.md +0 -0
  25. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/LICENSE +0 -0
  26. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/SECURITY.md +0 -0
  27. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/__init__.py +0 -0
  28. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/__main__.py +0 -0
  29. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/_util.py +0 -0
  30. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/api.py +0 -0
  31. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/cli.py +0 -0
  32. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/dataio.py +0 -0
  33. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/__init__.py +0 -0
  34. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/base.py +0 -0
  35. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/categories.py +0 -0
  36. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/contacts.py +0 -0
  37. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/currency.py +0 -0
  38. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/dedup.py +0 -0
  39. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/nulls.py +0 -0
  40. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/outliers.py +0 -0
  41. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/schema_mapping.py +0 -0
  42. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/text.py +0 -0
  43. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/units.py +0 -0
  44. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/diff.py +0 -0
  45. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/drift.py +0 -0
  46. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/errors.py +0 -0
  47. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/executor.py +0 -0
  48. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/issues.py +0 -0
  49. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/py.typed +0 -0
  50. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/quality.py +0 -0
  51. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/readfix.py +0 -0
  52. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/recipe.py +0 -0
  53. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/result.py +0 -0
  54. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/schema.py +0 -0
  55. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/streaming.py +0 -0
  56. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/workbook.py +0 -0
  57. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/README.md +0 -0
  58. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/api-reference.md +0 -0
  59. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/architecture.md +0 -0
  60. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/cli.md +0 -0
  61. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/concepts.md +0 -0
  62. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/detectors-and-ops.md +0 -0
  63. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/faq.md +0 -0
  64. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/getting-started.md +0 -0
  65. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/llm.md +0 -0
  66. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/production.md +0 -0
  67. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/recipe-spec.md +0 -0
  68. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/schema-spec.md +0 -0
  69. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/examples/customer.recipe.yaml +0 -0
  70. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/examples/customer.schema.yaml +0 -0
  71. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/examples/messy_customers.csv +0 -0
  72. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/conftest.py +0 -0
  73. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_api_report_cli.py +0 -0
  74. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_cross_platform.py +0 -0
  75. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_drift_codegen.py +0 -0
  76. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_executor_diff.py +0 -0
  77. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_ops.py +0 -0
  78. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_planner.py +0 -0
  79. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_production_safety.py +0 -0
  80. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_profile_detectors.py +0 -0
  81. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_recipe.py +0 -0
  82. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_validate.py +0 -0
  83. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave1_codegen.py +0 -0
  84. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave1_correctness.py +0 -0
  85. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave1_lineage.py +0 -0
  86. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave1_llm.py +0 -0
  87. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave2_scaling.py +0 -0
  88. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave3_selection.py +0 -0
  89. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave3_workbook.py +0 -0
  90. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave4_readfix.py +0 -0
  91. {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave5_streaming.py +0 -0
@@ -7,6 +7,63 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.3.1] — 2026-09-04
11
+
12
+ First follow-up to the 0.3.0 release, from the CI and code-scanning results that
13
+ only appear once a project is public.
14
+
15
+ ### Fixed
16
+
17
+ - **`Severity` comparisons were alphabetical, not ordered.** Only `__lt__` was
18
+ defined, so the `str` mixin answered everything else: `Severity.ERROR >
19
+ Severity.INFO` returned `False` because `"error" < "info"` as text. All four
20
+ comparisons are now defined on the severity rank. Library code always used
21
+ `.rank` explicitly, so nothing internal was affected, but any caller comparing
22
+ severities directly got wrong answers. Found by CodeQL's incomplete-ordering
23
+ query.
24
+ - **`mypy` failed in CI** with `numpy/__init__.pyi:737: error: Type statement is
25
+ only supported in Python 3.12 and greater`. The `python_version = "3.10"` pin
26
+ also governs how dependency stubs are parsed, and numpy 2.5's stubs use syntax
27
+ that needs 3.12. Removed the pin; minimum-version support is still checked by
28
+ ruff's `target-version` and by the Python 3.10 test job.
29
+ - **The wiki sync workflow could not push.** GitHub Actions' built-in token can
30
+ clone a wiki but not write to it, so the job committed and then failed on
31
+ authentication. It now requires the `WIKI_TOKEN` secret and stops up front with
32
+ an explanatory message. `wiki/README.md` claimed no token was needed; corrected.
33
+ - Two module-level import cycles removed: `profile` with `ops`, and `recipe` with
34
+ `validate`. `COMMON_DATE_FORMATS` now lives in `ops`, which owns date parsing,
35
+ and is re-exported from `profile` so every existing import keeps working.
36
+ `validate` imports `ValidationRule` for annotations only.
37
+ - Empty exception handlers in `fingerprint` and `report` replaced with a helper
38
+ that returns whether a cell is missing, so the intent is in the code rather
39
+ than in a comment beside `pass`.
40
+ - Redundant function-level `import re` in the dates detector removed; it now
41
+ reuses the module's compiled pattern instead of recompiling per call.
42
+
43
+ ### Changed
44
+
45
+ - Workflow actions updated: checkout to v7, setup-python to v7, upload-artifact
46
+ to v7, download-artifact to v8, codeql-action to v4. Clears the Node 20
47
+ deprecation warnings.
48
+ - Dependabot groups the `github-actions` ecosystem, so action bumps arrive as one
49
+ pull request instead of one per action.
50
+ - CodeQL analyses the library and skips `tests`. Test fixtures are deliberately
51
+ hostile (catastrophic-backtracking patterns, injection payloads, handlers that
52
+ assert something must not raise) and are not part of the wheel, so reporting
53
+ them buries real findings. Configuration lives in
54
+ `.github/codeql/codeql-config.yml`.
55
+ - Protocol methods on `LLMClient` and `Planner` carry a docstring instead of a
56
+ bare `...`.
57
+ - Installation docs link the package page, show how to pin a version, and mention
58
+ `cleanframe --version`. `CHANGELOG.md` gained the Keep a Changelog link
59
+ definitions now that releases are tagged.
60
+
61
+ ### Verified
62
+
63
+ 335 tests pass on Python 3.10 with the declared minimum pins (pandas 1.5.3,
64
+ numpy 1.23.5), on Python 3.13 with pandas 2.3.3, and on Python 3.13 with pandas
65
+ 3.0.5 and numpy 2.5.2, which is what a fresh `pip install` resolves today.
66
+
10
67
  ## [0.3.0] — 2026-09-04
11
68
 
12
69
  Release-readiness pass driven by a full audit of the CLI, the Python API, packaging and
@@ -233,3 +290,9 @@ now fail loudly, and a few transforms that quietly corrupted data no longer run.
233
290
  - Initial public release: profiler, detectors, rules + optional LLM planner, recipe YAML,
234
291
  deterministic executor, validation/quarantine, cell-level diff, schema drift, HTML reports,
235
292
  codegen, and CLI.
293
+
294
+ [Unreleased]: https://github.com/inboxpraveen/Cleanframe/compare/v0.3.1...HEAD
295
+ [0.3.1]: https://github.com/inboxpraveen/Cleanframe/releases/tag/v0.3.1
296
+ [0.3.0]: https://github.com/inboxpraveen/Cleanframe/releases/tag/v0.3.0
297
+ [0.2.0]: https://github.com/inboxpraveen/Cleanframe/commits/main
298
+ [0.1.0]: https://github.com/inboxpraveen/Cleanframe/commits/main
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cleanframe-engine
3
- Version: 0.3.0
3
+ Version: 0.3.1
4
4
  Summary: The reproducible data-cleaning engine for Python. AI writes the recipe once; the recipe runs forever — deterministic, diffable, reviewable.
5
5
  Project-URL: Homepage, https://github.com/inboxpraveen/Cleanframe
6
6
  Project-URL: Repository, https://github.com/inboxpraveen/Cleanframe
@@ -291,7 +291,7 @@ pip install "cleanframe-engine[all]"
291
291
  pip install "cleanframe-engine @ git+https://github.com/inboxpraveen/Cleanframe"
292
292
  ```
293
293
 
294
- Python 3.10+. The distribution is `cleanframe-engine`; the import package is
294
+ Python 3.10+. The distribution is [`cleanframe-engine`](https://pypi.org/project/cleanframe-engine/); the import package is
295
295
  `cleanframe` (`import cleanframe as cf`), and the CLI is `cleanframe` (or
296
296
  `python -m cleanframe`).
297
297
 
@@ -241,7 +241,7 @@ pip install "cleanframe-engine[all]"
241
241
  pip install "cleanframe-engine @ git+https://github.com/inboxpraveen/Cleanframe"
242
242
  ```
243
243
 
244
- Python 3.10+. The distribution is `cleanframe-engine`; the import package is
244
+ Python 3.10+. The distribution is [`cleanframe-engine`](https://pypi.org/project/cleanframe-engine/); the import package is
245
245
  `cleanframe` (`import cleanframe as cf`), and the CLI is `cleanframe` (or
246
246
  `python -m cleanframe`).
247
247
 
@@ -0,0 +1 @@
1
+ __version__ = "0.3.1"
@@ -45,11 +45,11 @@ def _constants_source() -> str:
45
45
  _KNOWN_CODES,
46
46
  _UNIT_ALIASES,
47
47
  _UNIT_TO_FAMILY,
48
+ COMMON_DATE_FORMATS,
48
49
  CURRENCY_SYMBOLS,
49
50
  DEFAULT_NA_TOKENS,
50
51
  UNIT_FAMILIES,
51
52
  )
52
- from .profile import COMMON_DATE_FORMATS
53
53
 
54
54
  unit_factors = {u: f for fam in UNIT_FAMILIES.values() for u, f in fam.items()}
55
55
  na_tokens = sorted({t.casefold() for t in DEFAULT_NA_TOKENS}) # includes '' (H9)
@@ -175,11 +175,9 @@ def detect_dates(series: pd.Series, ctx: DetectorContext) -> Issues:
175
175
 
176
176
  def _is_ambiguous(values: list[str]) -> bool:
177
177
  """True if no value disambiguates day-vs-month order (all numeric components <= 12)."""
178
- import re
179
-
180
178
  saw_slashlike = False
181
179
  for v in values:
182
- m = re.match(r"^\s*(\d{1,2})[/\-.](\d{1,2})[/\-.]\d{2,4}\s*$", v)
180
+ m = _SLASH_DATE_RE.match(v)
183
181
  if not m:
184
182
  continue
185
183
  saw_slashlike = True
@@ -17,20 +17,24 @@ import pandas as pd
17
17
  DEFAULT_SAMPLE_ROWS = 200
18
18
 
19
19
 
20
+ def _is_missing(value: Any) -> bool:
21
+ """True for None/NaN/NaT. An exotic cell pd.isna cannot judge counts as present."""
22
+ if value is None:
23
+ return True
24
+ try:
25
+ return bool(pd.isna(value)) # raises on array-like; cells are scalars here
26
+ except (TypeError, ValueError):
27
+ return False
28
+
29
+
20
30
  def _canonical(value: Any) -> str:
21
31
  """Render a single cell to a stable string.
22
32
 
23
33
  ``NaN``/``None`` collapse to a single sentinel so that a missing value hashes
24
34
  the same whether it arrived as ``float('nan')``, ``None``, or ``pd.NA``.
25
35
  """
26
- if value is None:
36
+ if _is_missing(value):
27
37
  return "\x00NA\x00"
28
- # pd.isna raises on array-like; cells are scalars here.
29
- try:
30
- if pd.isna(value):
31
- return "\x00NA\x00"
32
- except (TypeError, ValueError):
33
- pass
34
38
  if isinstance(value, float):
35
39
  # repr(float) is round-trippable and stable across platforms in CPython.
36
40
  return repr(value)
@@ -61,7 +61,8 @@ class LLMClient(Protocol):
61
61
 
62
62
  model: str
63
63
 
64
- def complete(self, system: str, user: str, *, max_tokens: int = 2048) -> LLMResponse: ...
64
+ def complete(self, system: str, user: str, *, max_tokens: int = 2048) -> LLMResponse:
65
+ """Return the model's reply plus token accounting."""
65
66
 
66
67
 
67
68
  class AnthropicClient:
@@ -749,6 +749,34 @@ def round_op(series: pd.Series, decimals: int = 0) -> pd.Series:
749
749
  # ---------------------------------------------------------------------------
750
750
  # Date ops
751
751
  # ---------------------------------------------------------------------------
752
+ #: Candidate date formats, ordered most-specific first. Shared with the dates
753
+ #: detector so profiling and planning agree on what "a date" looks like. Pure
754
+ #: all-digit formats are intentionally excluded to avoid classifying plain
755
+ #: integers (``"20240101"``, ``"1200"``) as dates.
756
+ COMMON_DATE_FORMATS = [
757
+ "%Y-%m-%dT%H:%M:%S",
758
+ "%Y-%m-%d %H:%M:%S",
759
+ "%Y-%m-%d",
760
+ "%Y/%m/%d",
761
+ "%d/%m/%Y",
762
+ "%m/%d/%Y",
763
+ "%d-%m-%Y",
764
+ "%m-%d-%Y",
765
+ "%d.%m.%Y",
766
+ "%d/%m/%y",
767
+ "%m/%d/%y",
768
+ "%d-%m-%y",
769
+ "%d %b %Y",
770
+ "%d %B %Y",
771
+ "%b %d, %Y",
772
+ "%B %d, %Y",
773
+ "%d-%b-%Y",
774
+ "%d-%b-%y",
775
+ "%b %d %Y",
776
+ "%d %b %y",
777
+ ]
778
+
779
+
752
780
  def _coerce_parse_date(raw: Any) -> dict:
753
781
  raw = raw or {}
754
782
  if not isinstance(raw, dict):
@@ -819,8 +847,6 @@ def parse_dates_to_datetime(
819
847
  # and NOT order-dependent, unlike a bare format-less pd.to_datetime which locks
820
848
  # onto the first row's inferred format and silently nulls otherwise-valid dates
821
849
  # (and behaves differently across pandas versions).
822
- from .profile import COMMON_DATE_FORMATS
823
-
824
850
  formats = list(COMMON_DATE_FORMATS)
825
851
  flex_fallback = True
826
852
 
@@ -1238,6 +1264,7 @@ __all__ = [
1238
1264
  "apply_frame_op",
1239
1265
  "parse_dates_to_datetime",
1240
1266
  "parse_unit_scalar",
1267
+ "COMMON_DATE_FORMATS",
1241
1268
  "CAST_TARGETS",
1242
1269
  "UNIT_FAMILIES",
1243
1270
  "CURRENCY_SYMBOLS",
@@ -83,7 +83,8 @@ class Planner(Protocol):
83
83
  schema: Any | None = None,
84
84
  mode: Mode | str = Mode.REVIEW,
85
85
  options: dict[str, Any] | None = None,
86
- ) -> Recipe: ...
86
+ ) -> Recipe:
87
+ """Return a recipe for ``df``, honouring the mode's confidence policy."""
87
88
 
88
89
 
89
90
  def _op_key(op: Op) -> tuple[str, str]:
@@ -19,7 +19,7 @@ from typing import Any
19
19
 
20
20
  import pandas as pd
21
21
 
22
- from .ops import CURRENCY_SYMBOLS, parse_unit_scalar
22
+ from .ops import COMMON_DATE_FORMATS, CURRENCY_SYMBOLS, parse_unit_scalar
23
23
 
24
24
  PATTERN_SAMPLE_CAP = 5000
25
25
 
@@ -37,32 +37,8 @@ _TIMESTAMPISH_RE = re.compile(r"\d{4}-\d{2}-\d{2}|\d{1,2}:\d{2}")
37
37
  _DIGIT_RE = re.compile(r"\d")
38
38
  _BOOL_TOKENS = {"true", "false", "yes", "no", "t", "f", "y", "n"}
39
39
 
40
- #: Candidate date formats, ordered most-specific first. Shared with the dates
41
- #: detector so profiling and planning agree on what "a date" looks like. Pure
42
- #: all-digit formats are intentionally excluded to avoid classifying plain
43
- #: integers (``"20240101"``, ``"1200"``) as dates.
44
- COMMON_DATE_FORMATS = [
45
- "%Y-%m-%dT%H:%M:%S",
46
- "%Y-%m-%d %H:%M:%S",
47
- "%Y-%m-%d",
48
- "%Y/%m/%d",
49
- "%d/%m/%Y",
50
- "%m/%d/%Y",
51
- "%d-%m-%Y",
52
- "%m-%d-%Y",
53
- "%d.%m.%Y",
54
- "%d/%m/%y",
55
- "%m/%d/%y",
56
- "%d-%m-%y",
57
- "%d %b %Y",
58
- "%d %B %Y",
59
- "%b %d, %Y",
60
- "%B %d, %Y",
61
- "%d-%b-%Y",
62
- "%d-%b-%y",
63
- "%b %d %Y",
64
- "%d %b %y",
65
- ]
40
+ #: Re-exported from :mod:`cleanframe.ops`, which owns date parsing. Imported here
41
+ #: because profiling and schema inference both classify against the same table.
66
42
 
67
43
  # Column-name hints (lowercased substrings) that nudge ambiguous classifications.
68
44
  _NAME_HINTS = {
@@ -218,14 +218,19 @@ _CLEAN_BODY = """
218
218
  """
219
219
 
220
220
 
221
- def _fmt_cell(value: Any) -> str:
222
- if value is None or (isinstance(value, float) and pd.isna(value)):
223
- return "∅"
221
+ def _is_missing(value: Any) -> bool:
222
+ """True for None/NaN/NaT. An exotic cell pd.isna cannot judge counts as present."""
223
+ if value is None:
224
+ return True
224
225
  try:
225
- if pd.isna(value):
226
- return "∅"
226
+ return bool(pd.isna(value))
227
227
  except (TypeError, ValueError):
228
- pass
228
+ return False
229
+
230
+
231
+ def _fmt_cell(value: Any) -> str:
232
+ if _is_missing(value):
233
+ return "∅"
229
234
  return str(value)
230
235
 
231
236
 
@@ -58,11 +58,29 @@ class Severity(str, Enum):
58
58
  def rank(self) -> int:
59
59
  return {"info": 0, "warning": 1, "error": 2}[self.value]
60
60
 
61
+ # All four comparisons are spelled out. The ``str`` mixin already provides
62
+ # lexicographic versions, so ``functools.total_ordering`` would leave them in
63
+ # place and "error" < "info" would silently compare as text.
61
64
  def __lt__(self, other: object) -> bool:
62
65
  if isinstance(other, Severity):
63
66
  return self.rank < other.rank
64
67
  return NotImplemented
65
68
 
69
+ def __le__(self, other: object) -> bool:
70
+ if isinstance(other, Severity):
71
+ return self.rank <= other.rank
72
+ return NotImplemented
73
+
74
+ def __gt__(self, other: object) -> bool:
75
+ if isinstance(other, Severity):
76
+ return self.rank > other.rank
77
+ return NotImplemented
78
+
79
+ def __ge__(self, other: object) -> bool:
80
+ if isinstance(other, Severity):
81
+ return self.rank >= other.rank
82
+ return NotImplemented
83
+
66
84
 
67
85
  class LLMExposure(str, Enum):
68
86
  """How much of your data an LLM planner is permitted to see.
@@ -24,7 +24,7 @@ import re
24
24
  import warnings
25
25
  from collections.abc import Callable
26
26
  from dataclasses import dataclass, field
27
- from typing import Any
27
+ from typing import TYPE_CHECKING, Any
28
28
 
29
29
  import numpy as np
30
30
  import pandas as pd
@@ -33,9 +33,11 @@ import yaml
33
33
  from ._util import safe_compile_regex
34
34
  from .errors import CleanFrameWarning, RecipeError, ValidationFailure
35
35
  from .profile import EMAIL_RE, URL_RE
36
- from .recipe import ValidationRule
37
36
  from .types import Mode
38
37
 
38
+ if TYPE_CHECKING: # annotations only — importing it at runtime would cycle
39
+ from .recipe import ValidationRule
40
+
39
41
  # ---------------------------------------------------------------------------
40
42
  # Validator registry (named checks)
41
43
  # ---------------------------------------------------------------------------
@@ -15,6 +15,20 @@ The distribution name on PyPI is `cleanframe-engine`; the import package is
15
15
  `cleanframe` (`import cleanframe as cf`), and the CLI is `cleanframe` — or
16
16
  `python -m cleanframe`, an equivalent alias for every invocation.
17
17
 
18
+ Package page: [https://pypi.org/project/cleanframe-engine/](https://pypi.org/project/cleanframe-engine/)
19
+
20
+ Pin a version in a requirements file the usual way:
21
+
22
+ ```
23
+ cleanframe-engine==0.3.1
24
+ ```
25
+
26
+ Check what you got:
27
+
28
+ ```bash
29
+ cleanframe --version
30
+ ```
31
+
18
32
  ### Extras
19
33
 
20
34
  | Extra | Installs | Needed for |
@@ -95,7 +95,9 @@ filterwarnings = [
95
95
  ]
96
96
 
97
97
  [tool.mypy]
98
- python_version = "3.10"
98
+ # No python_version pin: it also governs how dependency stubs are parsed, and
99
+ # numpy's stubs use syntax that only parses on 3.12+. Minimum-version support is
100
+ # covered by ruff's target-version and by the 3.10 test job.
99
101
  ignore_missing_imports = true
100
102
  files = ["cleanframe"]
101
103
 
@@ -124,8 +124,7 @@ def test_openai_base_url_does_not_hijack_openrouter(monkeypatch):
124
124
  monkeypatch.setenv("OPENAI_BASE_URL", "https://evil.example/v1")
125
125
  monkeypatch.setenv("OPENROUTER_API_KEY", "sk-test")
126
126
  client = get_client("openrouter/google/gemma-4-26b-a4b-it")
127
- assert "openrouter.ai" in (client._base_url or "")
128
- assert "evil.example" not in (client._base_url or "")
127
+ assert client._base_url == "https://openrouter.ai/api/v1"
129
128
 
130
129
 
131
130
  def test_openai_base_url_still_applies_to_openai(monkeypatch):
@@ -56,7 +56,7 @@ def test_sample_exposure_anonymizes_and_shuffles():
56
56
  email_col = next(c for c in md["columns"] if c["name"] == "Email")
57
57
  assert email_col["example_values"]
58
58
  assert all("@" in v for v in email_col["example_values"])
59
- assert "example.com" in email_col["example_values"][0]
59
+ assert email_col["example_values"][0] == "user@example.com"
60
60
  # deterministic across calls
61
61
  md2 = build_metadata(df, profile_dataframe(df), Issues(), None, LLMExposure.SAMPLE)
62
62
  assert md["columns"][0]["example_values"] == md2["columns"][0]["example_values"]
@@ -81,7 +81,7 @@ def test_get_client_resolves_openai_compatible_providers(monkeypatch):
81
81
  assert groq._api_key == "groq-key"
82
82
 
83
83
  gemini = get_client("gemini/gemini-2.0-flash") # alias for google
84
- assert "generativelanguage.googleapis.com" in (gemini._base_url or "")
84
+ assert gemini._base_url == "https://generativelanguage.googleapis.com/v1beta/openai/"
85
85
 
86
86
  ollama = get_client("ollama/llama3.2")
87
87
  assert ollama._base_url == "http://localhost:11434/v1"
@@ -187,6 +187,16 @@ def test_unknown_or_wrongly_typed_op_parameters_are_refused(op):
187
187
  Recipe.from_dict({"version": 1, "columns": {"a": {"ops": [op]}}})
188
188
 
189
189
 
190
+ def test_severity_compares_by_rank_not_alphabetically():
191
+ """The str mixin made `ERROR > INFO` compare "error" to "info", which is False."""
192
+ low, mid, high = cf.Severity.INFO, cf.Severity.WARNING, cf.Severity.ERROR
193
+ assert low < mid < high
194
+ assert high > mid > low
195
+ assert low <= low and high >= high
196
+ assert sorted([high, low, mid]) == [low, mid, high]
197
+ assert max(high, low) is high and min(high, low) is low
198
+
199
+
190
200
  def test_round_accepts_both_documented_forms():
191
201
  for raw in ({"round": 2}, {"round": {"decimals": 2}}):
192
202
  recipe = Recipe.from_dict({"version": 1, "columns": {"a": {"ops": [raw]}}})
@@ -6,6 +6,7 @@ or crashes the pipeline. They must fail before the Batch-A fixes and pass after.
6
6
  from __future__ import annotations
7
7
 
8
8
  import io
9
+ from contextlib import suppress
9
10
 
10
11
  import pandas as pd
11
12
  import pytest
@@ -99,10 +100,8 @@ def test_multiindex_columns_do_not_raise_raw_error():
99
100
  """M4: MultiIndex columns must not leak a raw TypeError from .astype(str)."""
100
101
  df = pd.DataFrame([[1, 2], [3, 4]], columns=pd.MultiIndex.from_tuples([("a", "x"), ("a", "y")]))
101
102
  # Either succeeds (flattened labels) or raises a clean CleanFrameError — never a raw TypeError.
102
- try:
103
+ with suppress(CleanFrameError):
103
104
  cf.clean(df)
104
- except CleanFrameError:
105
- pass
106
105
 
107
106
 
108
107
  def test_string_index_with_name_column_does_not_crash():
@@ -7,6 +7,7 @@ from __future__ import annotations
7
7
 
8
8
  import pandas as pd
9
9
 
10
+ from cleanframe import ops
10
11
  from cleanframe.executor import execute
11
12
  from cleanframe.ops import apply_column_op
12
13
  from cleanframe.recipe import Recipe
@@ -36,8 +37,6 @@ _TRICKY = [
36
37
 
37
38
  def _elementwise(op_name, series):
38
39
  """The reference: force the elementwise path by disabling the fast-path."""
39
- import cleanframe.ops as ops
40
-
41
40
  orig = ops._is_pure_string
42
41
  ops._is_pure_string = lambda s: False
43
42
  try:
@@ -1 +0,0 @@
1
- __version__ = "0.3.0"