cleanframe-engine 0.3.0__tar.gz → 0.3.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/CHANGELOG.md +63 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/PKG-INFO +2 -2
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/README.md +1 -1
- cleanframe_engine-0.3.1/cleanframe/_version.py +1 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/codegen.py +1 -1
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/dates.py +1 -3
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/fingerprint.py +11 -7
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/llm.py +2 -1
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/ops.py +29 -2
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/planner.py +2 -1
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/profile.py +3 -27
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/report.py +11 -6
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/types.py +18 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/validate.py +4 -2
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/installation.md +14 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/pyproject.toml +3 -1
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_audit_hardening.py +1 -2
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_llm_schema.py +2 -2
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_release_hardening.py +10 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave1_hardening.py +2 -3
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave2_vectorization.py +1 -2
- cleanframe_engine-0.3.0/cleanframe/_version.py +0 -1
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/.gitignore +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/CONTRIBUTING.md +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/LICENSE +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/SECURITY.md +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/__init__.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/__main__.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/_util.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/api.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/cli.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/dataio.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/__init__.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/base.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/categories.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/contacts.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/currency.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/dedup.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/nulls.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/outliers.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/schema_mapping.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/text.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/detectors/units.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/diff.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/drift.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/errors.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/executor.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/issues.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/py.typed +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/quality.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/readfix.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/recipe.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/result.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/schema.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/streaming.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/cleanframe/workbook.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/README.md +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/api-reference.md +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/architecture.md +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/cli.md +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/concepts.md +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/detectors-and-ops.md +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/faq.md +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/getting-started.md +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/llm.md +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/production.md +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/recipe-spec.md +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/docs/schema-spec.md +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/examples/customer.recipe.yaml +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/examples/customer.schema.yaml +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/examples/messy_customers.csv +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/conftest.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_api_report_cli.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_cross_platform.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_drift_codegen.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_executor_diff.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_ops.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_planner.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_production_safety.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_profile_detectors.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_recipe.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_validate.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave1_codegen.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave1_correctness.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave1_lineage.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave1_llm.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave2_scaling.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave3_selection.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave3_workbook.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave4_readfix.py +0 -0
- {cleanframe_engine-0.3.0 → cleanframe_engine-0.3.1}/tests/test_wave5_streaming.py +0 -0
|
@@ -7,6 +7,63 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.3.1] — 2026-09-04
|
|
11
|
+
|
|
12
|
+
First follow-up to the 0.3.0 release, from the CI and code-scanning results that
|
|
13
|
+
only appear once a project is public.
|
|
14
|
+
|
|
15
|
+
### Fixed
|
|
16
|
+
|
|
17
|
+
- **`Severity` comparisons were alphabetical, not ordered.** Only `__lt__` was
|
|
18
|
+
defined, so the `str` mixin answered everything else: `Severity.ERROR >
|
|
19
|
+
Severity.INFO` returned `False` because `"error" < "info"` as text. All four
|
|
20
|
+
comparisons are now defined on the severity rank. Library code always used
|
|
21
|
+
`.rank` explicitly, so nothing internal was affected, but any caller comparing
|
|
22
|
+
severities directly got wrong answers. Found by CodeQL's incomplete-ordering
|
|
23
|
+
query.
|
|
24
|
+
- **`mypy` failed in CI** with `numpy/__init__.pyi:737: error: Type statement is
|
|
25
|
+
only supported in Python 3.12 and greater`. The `python_version = "3.10"` pin
|
|
26
|
+
also governs how dependency stubs are parsed, and numpy 2.5's stubs use syntax
|
|
27
|
+
that needs 3.12. Removed the pin; minimum-version support is still checked by
|
|
28
|
+
ruff's `target-version` and by the Python 3.10 test job.
|
|
29
|
+
- **The wiki sync workflow could not push.** GitHub Actions' built-in token can
|
|
30
|
+
clone a wiki but not write to it, so the job committed and then failed on
|
|
31
|
+
authentication. It now requires the `WIKI_TOKEN` secret and stops up front with
|
|
32
|
+
an explanatory message. `wiki/README.md` claimed no token was needed; corrected.
|
|
33
|
+
- Two module-level import cycles removed: `profile` with `ops`, and `recipe` with
|
|
34
|
+
`validate`. `COMMON_DATE_FORMATS` now lives in `ops`, which owns date parsing,
|
|
35
|
+
and is re-exported from `profile` so every existing import keeps working.
|
|
36
|
+
`validate` imports `ValidationRule` for annotations only.
|
|
37
|
+
- Empty exception handlers in `fingerprint` and `report` replaced with a helper
|
|
38
|
+
that returns whether a cell is missing, so the intent is in the code rather
|
|
39
|
+
than in a comment beside `pass`.
|
|
40
|
+
- Redundant function-level `import re` in the dates detector removed; it now
|
|
41
|
+
reuses the module's compiled pattern instead of recompiling per call.
|
|
42
|
+
|
|
43
|
+
### Changed
|
|
44
|
+
|
|
45
|
+
- Workflow actions updated: checkout to v7, setup-python to v7, upload-artifact
|
|
46
|
+
to v7, download-artifact to v8, codeql-action to v4. Clears the Node 20
|
|
47
|
+
deprecation warnings.
|
|
48
|
+
- Dependabot groups the `github-actions` ecosystem, so action bumps arrive as one
|
|
49
|
+
pull request instead of one per action.
|
|
50
|
+
- CodeQL analyses the library and skips `tests`. Test fixtures are deliberately
|
|
51
|
+
hostile (catastrophic-backtracking patterns, injection payloads, handlers that
|
|
52
|
+
assert something must not raise) and are not part of the wheel, so reporting
|
|
53
|
+
them buries real findings. Configuration lives in
|
|
54
|
+
`.github/codeql/codeql-config.yml`.
|
|
55
|
+
- Protocol methods on `LLMClient` and `Planner` carry a docstring instead of a
|
|
56
|
+
bare `...`.
|
|
57
|
+
- Installation docs link the package page, show how to pin a version, and mention
|
|
58
|
+
`cleanframe --version`. `CHANGELOG.md` gained the Keep a Changelog link
|
|
59
|
+
definitions now that releases are tagged.
|
|
60
|
+
|
|
61
|
+
### Verified
|
|
62
|
+
|
|
63
|
+
335 tests pass on Python 3.10 with the declared minimum pins (pandas 1.5.3,
|
|
64
|
+
numpy 1.23.5), on Python 3.13 with pandas 2.3.3, and on Python 3.13 with pandas
|
|
65
|
+
3.0.5 and numpy 2.5.2, which is what a fresh `pip install` resolves today.
|
|
66
|
+
|
|
10
67
|
## [0.3.0] — 2026-09-04
|
|
11
68
|
|
|
12
69
|
Release-readiness pass driven by a full audit of the CLI, the Python API, packaging and
|
|
@@ -233,3 +290,9 @@ now fail loudly, and a few transforms that quietly corrupted data no longer run.
|
|
|
233
290
|
- Initial public release: profiler, detectors, rules + optional LLM planner, recipe YAML,
|
|
234
291
|
deterministic executor, validation/quarantine, cell-level diff, schema drift, HTML reports,
|
|
235
292
|
codegen, and CLI.
|
|
293
|
+
|
|
294
|
+
[Unreleased]: https://github.com/inboxpraveen/Cleanframe/compare/v0.3.1...HEAD
|
|
295
|
+
[0.3.1]: https://github.com/inboxpraveen/Cleanframe/releases/tag/v0.3.1
|
|
296
|
+
[0.3.0]: https://github.com/inboxpraveen/Cleanframe/releases/tag/v0.3.0
|
|
297
|
+
[0.2.0]: https://github.com/inboxpraveen/Cleanframe/commits/main
|
|
298
|
+
[0.1.0]: https://github.com/inboxpraveen/Cleanframe/commits/main
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cleanframe-engine
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.1
|
|
4
4
|
Summary: The reproducible data-cleaning engine for Python. AI writes the recipe once; the recipe runs forever — deterministic, diffable, reviewable.
|
|
5
5
|
Project-URL: Homepage, https://github.com/inboxpraveen/Cleanframe
|
|
6
6
|
Project-URL: Repository, https://github.com/inboxpraveen/Cleanframe
|
|
@@ -291,7 +291,7 @@ pip install "cleanframe-engine[all]"
|
|
|
291
291
|
pip install "cleanframe-engine @ git+https://github.com/inboxpraveen/Cleanframe"
|
|
292
292
|
```
|
|
293
293
|
|
|
294
|
-
Python 3.10+. The distribution is `cleanframe-engine
|
|
294
|
+
Python 3.10+. The distribution is [`cleanframe-engine`](https://pypi.org/project/cleanframe-engine/); the import package is
|
|
295
295
|
`cleanframe` (`import cleanframe as cf`), and the CLI is `cleanframe` (or
|
|
296
296
|
`python -m cleanframe`).
|
|
297
297
|
|
|
@@ -241,7 +241,7 @@ pip install "cleanframe-engine[all]"
|
|
|
241
241
|
pip install "cleanframe-engine @ git+https://github.com/inboxpraveen/Cleanframe"
|
|
242
242
|
```
|
|
243
243
|
|
|
244
|
-
Python 3.10+. The distribution is `cleanframe-engine
|
|
244
|
+
Python 3.10+. The distribution is [`cleanframe-engine`](https://pypi.org/project/cleanframe-engine/); the import package is
|
|
245
245
|
`cleanframe` (`import cleanframe as cf`), and the CLI is `cleanframe` (or
|
|
246
246
|
`python -m cleanframe`).
|
|
247
247
|
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.3.1"
|
|
@@ -45,11 +45,11 @@ def _constants_source() -> str:
|
|
|
45
45
|
_KNOWN_CODES,
|
|
46
46
|
_UNIT_ALIASES,
|
|
47
47
|
_UNIT_TO_FAMILY,
|
|
48
|
+
COMMON_DATE_FORMATS,
|
|
48
49
|
CURRENCY_SYMBOLS,
|
|
49
50
|
DEFAULT_NA_TOKENS,
|
|
50
51
|
UNIT_FAMILIES,
|
|
51
52
|
)
|
|
52
|
-
from .profile import COMMON_DATE_FORMATS
|
|
53
53
|
|
|
54
54
|
unit_factors = {u: f for fam in UNIT_FAMILIES.values() for u, f in fam.items()}
|
|
55
55
|
na_tokens = sorted({t.casefold() for t in DEFAULT_NA_TOKENS}) # includes '' (H9)
|
|
@@ -175,11 +175,9 @@ def detect_dates(series: pd.Series, ctx: DetectorContext) -> Issues:
|
|
|
175
175
|
|
|
176
176
|
def _is_ambiguous(values: list[str]) -> bool:
|
|
177
177
|
"""True if no value disambiguates day-vs-month order (all numeric components <= 12)."""
|
|
178
|
-
import re
|
|
179
|
-
|
|
180
178
|
saw_slashlike = False
|
|
181
179
|
for v in values:
|
|
182
|
-
m =
|
|
180
|
+
m = _SLASH_DATE_RE.match(v)
|
|
183
181
|
if not m:
|
|
184
182
|
continue
|
|
185
183
|
saw_slashlike = True
|
|
@@ -17,20 +17,24 @@ import pandas as pd
|
|
|
17
17
|
DEFAULT_SAMPLE_ROWS = 200
|
|
18
18
|
|
|
19
19
|
|
|
20
|
+
def _is_missing(value: Any) -> bool:
|
|
21
|
+
"""True for None/NaN/NaT. An exotic cell pd.isna cannot judge counts as present."""
|
|
22
|
+
if value is None:
|
|
23
|
+
return True
|
|
24
|
+
try:
|
|
25
|
+
return bool(pd.isna(value)) # raises on array-like; cells are scalars here
|
|
26
|
+
except (TypeError, ValueError):
|
|
27
|
+
return False
|
|
28
|
+
|
|
29
|
+
|
|
20
30
|
def _canonical(value: Any) -> str:
|
|
21
31
|
"""Render a single cell to a stable string.
|
|
22
32
|
|
|
23
33
|
``NaN``/``None`` collapse to a single sentinel so that a missing value hashes
|
|
24
34
|
the same whether it arrived as ``float('nan')``, ``None``, or ``pd.NA``.
|
|
25
35
|
"""
|
|
26
|
-
if value
|
|
36
|
+
if _is_missing(value):
|
|
27
37
|
return "\x00NA\x00"
|
|
28
|
-
# pd.isna raises on array-like; cells are scalars here.
|
|
29
|
-
try:
|
|
30
|
-
if pd.isna(value):
|
|
31
|
-
return "\x00NA\x00"
|
|
32
|
-
except (TypeError, ValueError):
|
|
33
|
-
pass
|
|
34
38
|
if isinstance(value, float):
|
|
35
39
|
# repr(float) is round-trippable and stable across platforms in CPython.
|
|
36
40
|
return repr(value)
|
|
@@ -61,7 +61,8 @@ class LLMClient(Protocol):
|
|
|
61
61
|
|
|
62
62
|
model: str
|
|
63
63
|
|
|
64
|
-
def complete(self, system: str, user: str, *, max_tokens: int = 2048) -> LLMResponse:
|
|
64
|
+
def complete(self, system: str, user: str, *, max_tokens: int = 2048) -> LLMResponse:
|
|
65
|
+
"""Return the model's reply plus token accounting."""
|
|
65
66
|
|
|
66
67
|
|
|
67
68
|
class AnthropicClient:
|
|
@@ -749,6 +749,34 @@ def round_op(series: pd.Series, decimals: int = 0) -> pd.Series:
|
|
|
749
749
|
# ---------------------------------------------------------------------------
|
|
750
750
|
# Date ops
|
|
751
751
|
# ---------------------------------------------------------------------------
|
|
752
|
+
#: Candidate date formats, ordered most-specific first. Shared with the dates
|
|
753
|
+
#: detector so profiling and planning agree on what "a date" looks like. Pure
|
|
754
|
+
#: all-digit formats are intentionally excluded to avoid classifying plain
|
|
755
|
+
#: integers (``"20240101"``, ``"1200"``) as dates.
|
|
756
|
+
COMMON_DATE_FORMATS = [
|
|
757
|
+
"%Y-%m-%dT%H:%M:%S",
|
|
758
|
+
"%Y-%m-%d %H:%M:%S",
|
|
759
|
+
"%Y-%m-%d",
|
|
760
|
+
"%Y/%m/%d",
|
|
761
|
+
"%d/%m/%Y",
|
|
762
|
+
"%m/%d/%Y",
|
|
763
|
+
"%d-%m-%Y",
|
|
764
|
+
"%m-%d-%Y",
|
|
765
|
+
"%d.%m.%Y",
|
|
766
|
+
"%d/%m/%y",
|
|
767
|
+
"%m/%d/%y",
|
|
768
|
+
"%d-%m-%y",
|
|
769
|
+
"%d %b %Y",
|
|
770
|
+
"%d %B %Y",
|
|
771
|
+
"%b %d, %Y",
|
|
772
|
+
"%B %d, %Y",
|
|
773
|
+
"%d-%b-%Y",
|
|
774
|
+
"%d-%b-%y",
|
|
775
|
+
"%b %d %Y",
|
|
776
|
+
"%d %b %y",
|
|
777
|
+
]
|
|
778
|
+
|
|
779
|
+
|
|
752
780
|
def _coerce_parse_date(raw: Any) -> dict:
|
|
753
781
|
raw = raw or {}
|
|
754
782
|
if not isinstance(raw, dict):
|
|
@@ -819,8 +847,6 @@ def parse_dates_to_datetime(
|
|
|
819
847
|
# and NOT order-dependent, unlike a bare format-less pd.to_datetime which locks
|
|
820
848
|
# onto the first row's inferred format and silently nulls otherwise-valid dates
|
|
821
849
|
# (and behaves differently across pandas versions).
|
|
822
|
-
from .profile import COMMON_DATE_FORMATS
|
|
823
|
-
|
|
824
850
|
formats = list(COMMON_DATE_FORMATS)
|
|
825
851
|
flex_fallback = True
|
|
826
852
|
|
|
@@ -1238,6 +1264,7 @@ __all__ = [
|
|
|
1238
1264
|
"apply_frame_op",
|
|
1239
1265
|
"parse_dates_to_datetime",
|
|
1240
1266
|
"parse_unit_scalar",
|
|
1267
|
+
"COMMON_DATE_FORMATS",
|
|
1241
1268
|
"CAST_TARGETS",
|
|
1242
1269
|
"UNIT_FAMILIES",
|
|
1243
1270
|
"CURRENCY_SYMBOLS",
|
|
@@ -83,7 +83,8 @@ class Planner(Protocol):
|
|
|
83
83
|
schema: Any | None = None,
|
|
84
84
|
mode: Mode | str = Mode.REVIEW,
|
|
85
85
|
options: dict[str, Any] | None = None,
|
|
86
|
-
) -> Recipe:
|
|
86
|
+
) -> Recipe:
|
|
87
|
+
"""Return a recipe for ``df``, honouring the mode's confidence policy."""
|
|
87
88
|
|
|
88
89
|
|
|
89
90
|
def _op_key(op: Op) -> tuple[str, str]:
|
|
@@ -19,7 +19,7 @@ from typing import Any
|
|
|
19
19
|
|
|
20
20
|
import pandas as pd
|
|
21
21
|
|
|
22
|
-
from .ops import CURRENCY_SYMBOLS, parse_unit_scalar
|
|
22
|
+
from .ops import COMMON_DATE_FORMATS, CURRENCY_SYMBOLS, parse_unit_scalar
|
|
23
23
|
|
|
24
24
|
PATTERN_SAMPLE_CAP = 5000
|
|
25
25
|
|
|
@@ -37,32 +37,8 @@ _TIMESTAMPISH_RE = re.compile(r"\d{4}-\d{2}-\d{2}|\d{1,2}:\d{2}")
|
|
|
37
37
|
_DIGIT_RE = re.compile(r"\d")
|
|
38
38
|
_BOOL_TOKENS = {"true", "false", "yes", "no", "t", "f", "y", "n"}
|
|
39
39
|
|
|
40
|
-
#:
|
|
41
|
-
#:
|
|
42
|
-
#: all-digit formats are intentionally excluded to avoid classifying plain
|
|
43
|
-
#: integers (``"20240101"``, ``"1200"``) as dates.
|
|
44
|
-
COMMON_DATE_FORMATS = [
|
|
45
|
-
"%Y-%m-%dT%H:%M:%S",
|
|
46
|
-
"%Y-%m-%d %H:%M:%S",
|
|
47
|
-
"%Y-%m-%d",
|
|
48
|
-
"%Y/%m/%d",
|
|
49
|
-
"%d/%m/%Y",
|
|
50
|
-
"%m/%d/%Y",
|
|
51
|
-
"%d-%m-%Y",
|
|
52
|
-
"%m-%d-%Y",
|
|
53
|
-
"%d.%m.%Y",
|
|
54
|
-
"%d/%m/%y",
|
|
55
|
-
"%m/%d/%y",
|
|
56
|
-
"%d-%m-%y",
|
|
57
|
-
"%d %b %Y",
|
|
58
|
-
"%d %B %Y",
|
|
59
|
-
"%b %d, %Y",
|
|
60
|
-
"%B %d, %Y",
|
|
61
|
-
"%d-%b-%Y",
|
|
62
|
-
"%d-%b-%y",
|
|
63
|
-
"%b %d %Y",
|
|
64
|
-
"%d %b %y",
|
|
65
|
-
]
|
|
40
|
+
#: Re-exported from :mod:`cleanframe.ops`, which owns date parsing. Imported here
|
|
41
|
+
#: because profiling and schema inference both classify against the same table.
|
|
66
42
|
|
|
67
43
|
# Column-name hints (lowercased substrings) that nudge ambiguous classifications.
|
|
68
44
|
_NAME_HINTS = {
|
|
@@ -218,14 +218,19 @@ _CLEAN_BODY = """
|
|
|
218
218
|
"""
|
|
219
219
|
|
|
220
220
|
|
|
221
|
-
def
|
|
222
|
-
|
|
223
|
-
|
|
221
|
+
def _is_missing(value: Any) -> bool:
|
|
222
|
+
"""True for None/NaN/NaT. An exotic cell pd.isna cannot judge counts as present."""
|
|
223
|
+
if value is None:
|
|
224
|
+
return True
|
|
224
225
|
try:
|
|
225
|
-
|
|
226
|
-
return "∅"
|
|
226
|
+
return bool(pd.isna(value))
|
|
227
227
|
except (TypeError, ValueError):
|
|
228
|
-
|
|
228
|
+
return False
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _fmt_cell(value: Any) -> str:
|
|
232
|
+
if _is_missing(value):
|
|
233
|
+
return "∅"
|
|
229
234
|
return str(value)
|
|
230
235
|
|
|
231
236
|
|
|
@@ -58,11 +58,29 @@ class Severity(str, Enum):
|
|
|
58
58
|
def rank(self) -> int:
|
|
59
59
|
return {"info": 0, "warning": 1, "error": 2}[self.value]
|
|
60
60
|
|
|
61
|
+
# All four comparisons are spelled out. The ``str`` mixin already provides
|
|
62
|
+
# lexicographic versions, so ``functools.total_ordering`` would leave them in
|
|
63
|
+
# place and "error" < "info" would silently compare as text.
|
|
61
64
|
def __lt__(self, other: object) -> bool:
|
|
62
65
|
if isinstance(other, Severity):
|
|
63
66
|
return self.rank < other.rank
|
|
64
67
|
return NotImplemented
|
|
65
68
|
|
|
69
|
+
def __le__(self, other: object) -> bool:
|
|
70
|
+
if isinstance(other, Severity):
|
|
71
|
+
return self.rank <= other.rank
|
|
72
|
+
return NotImplemented
|
|
73
|
+
|
|
74
|
+
def __gt__(self, other: object) -> bool:
|
|
75
|
+
if isinstance(other, Severity):
|
|
76
|
+
return self.rank > other.rank
|
|
77
|
+
return NotImplemented
|
|
78
|
+
|
|
79
|
+
def __ge__(self, other: object) -> bool:
|
|
80
|
+
if isinstance(other, Severity):
|
|
81
|
+
return self.rank >= other.rank
|
|
82
|
+
return NotImplemented
|
|
83
|
+
|
|
66
84
|
|
|
67
85
|
class LLMExposure(str, Enum):
|
|
68
86
|
"""How much of your data an LLM planner is permitted to see.
|
|
@@ -24,7 +24,7 @@ import re
|
|
|
24
24
|
import warnings
|
|
25
25
|
from collections.abc import Callable
|
|
26
26
|
from dataclasses import dataclass, field
|
|
27
|
-
from typing import Any
|
|
27
|
+
from typing import TYPE_CHECKING, Any
|
|
28
28
|
|
|
29
29
|
import numpy as np
|
|
30
30
|
import pandas as pd
|
|
@@ -33,9 +33,11 @@ import yaml
|
|
|
33
33
|
from ._util import safe_compile_regex
|
|
34
34
|
from .errors import CleanFrameWarning, RecipeError, ValidationFailure
|
|
35
35
|
from .profile import EMAIL_RE, URL_RE
|
|
36
|
-
from .recipe import ValidationRule
|
|
37
36
|
from .types import Mode
|
|
38
37
|
|
|
38
|
+
if TYPE_CHECKING: # annotations only — importing it at runtime would cycle
|
|
39
|
+
from .recipe import ValidationRule
|
|
40
|
+
|
|
39
41
|
# ---------------------------------------------------------------------------
|
|
40
42
|
# Validator registry (named checks)
|
|
41
43
|
# ---------------------------------------------------------------------------
|
|
@@ -15,6 +15,20 @@ The distribution name on PyPI is `cleanframe-engine`; the import package is
|
|
|
15
15
|
`cleanframe` (`import cleanframe as cf`), and the CLI is `cleanframe` — or
|
|
16
16
|
`python -m cleanframe`, an equivalent alias for every invocation.
|
|
17
17
|
|
|
18
|
+
Package page: [https://pypi.org/project/cleanframe-engine/](https://pypi.org/project/cleanframe-engine/)
|
|
19
|
+
|
|
20
|
+
Pin a version in a requirements file the usual way:
|
|
21
|
+
|
|
22
|
+
```
|
|
23
|
+
cleanframe-engine==0.3.1
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
Check what you got:
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
cleanframe --version
|
|
30
|
+
```
|
|
31
|
+
|
|
18
32
|
### Extras
|
|
19
33
|
|
|
20
34
|
| Extra | Installs | Needed for |
|
|
@@ -95,7 +95,9 @@ filterwarnings = [
|
|
|
95
95
|
]
|
|
96
96
|
|
|
97
97
|
[tool.mypy]
|
|
98
|
-
python_version
|
|
98
|
+
# No python_version pin: it also governs how dependency stubs are parsed, and
|
|
99
|
+
# numpy's stubs use syntax that only parses on 3.12+. Minimum-version support is
|
|
100
|
+
# covered by ruff's target-version and by the 3.10 test job.
|
|
99
101
|
ignore_missing_imports = true
|
|
100
102
|
files = ["cleanframe"]
|
|
101
103
|
|
|
@@ -124,8 +124,7 @@ def test_openai_base_url_does_not_hijack_openrouter(monkeypatch):
|
|
|
124
124
|
monkeypatch.setenv("OPENAI_BASE_URL", "https://evil.example/v1")
|
|
125
125
|
monkeypatch.setenv("OPENROUTER_API_KEY", "sk-test")
|
|
126
126
|
client = get_client("openrouter/google/gemma-4-26b-a4b-it")
|
|
127
|
-
assert
|
|
128
|
-
assert "evil.example" not in (client._base_url or "")
|
|
127
|
+
assert client._base_url == "https://openrouter.ai/api/v1"
|
|
129
128
|
|
|
130
129
|
|
|
131
130
|
def test_openai_base_url_still_applies_to_openai(monkeypatch):
|
|
@@ -56,7 +56,7 @@ def test_sample_exposure_anonymizes_and_shuffles():
|
|
|
56
56
|
email_col = next(c for c in md["columns"] if c["name"] == "Email")
|
|
57
57
|
assert email_col["example_values"]
|
|
58
58
|
assert all("@" in v for v in email_col["example_values"])
|
|
59
|
-
assert
|
|
59
|
+
assert email_col["example_values"][0] == "user@example.com"
|
|
60
60
|
# deterministic across calls
|
|
61
61
|
md2 = build_metadata(df, profile_dataframe(df), Issues(), None, LLMExposure.SAMPLE)
|
|
62
62
|
assert md["columns"][0]["example_values"] == md2["columns"][0]["example_values"]
|
|
@@ -81,7 +81,7 @@ def test_get_client_resolves_openai_compatible_providers(monkeypatch):
|
|
|
81
81
|
assert groq._api_key == "groq-key"
|
|
82
82
|
|
|
83
83
|
gemini = get_client("gemini/gemini-2.0-flash") # alias for google
|
|
84
|
-
assert "generativelanguage.googleapis.com"
|
|
84
|
+
assert gemini._base_url == "https://generativelanguage.googleapis.com/v1beta/openai/"
|
|
85
85
|
|
|
86
86
|
ollama = get_client("ollama/llama3.2")
|
|
87
87
|
assert ollama._base_url == "http://localhost:11434/v1"
|
|
@@ -187,6 +187,16 @@ def test_unknown_or_wrongly_typed_op_parameters_are_refused(op):
|
|
|
187
187
|
Recipe.from_dict({"version": 1, "columns": {"a": {"ops": [op]}}})
|
|
188
188
|
|
|
189
189
|
|
|
190
|
+
def test_severity_compares_by_rank_not_alphabetically():
|
|
191
|
+
"""The str mixin made `ERROR > INFO` compare "error" to "info", which is False."""
|
|
192
|
+
low, mid, high = cf.Severity.INFO, cf.Severity.WARNING, cf.Severity.ERROR
|
|
193
|
+
assert low < mid < high
|
|
194
|
+
assert high > mid > low
|
|
195
|
+
assert low <= low and high >= high
|
|
196
|
+
assert sorted([high, low, mid]) == [low, mid, high]
|
|
197
|
+
assert max(high, low) is high and min(high, low) is low
|
|
198
|
+
|
|
199
|
+
|
|
190
200
|
def test_round_accepts_both_documented_forms():
|
|
191
201
|
for raw in ({"round": 2}, {"round": {"decimals": 2}}):
|
|
192
202
|
recipe = Recipe.from_dict({"version": 1, "columns": {"a": {"ops": [raw]}}})
|
|
@@ -6,6 +6,7 @@ or crashes the pipeline. They must fail before the Batch-A fixes and pass after.
|
|
|
6
6
|
from __future__ import annotations
|
|
7
7
|
|
|
8
8
|
import io
|
|
9
|
+
from contextlib import suppress
|
|
9
10
|
|
|
10
11
|
import pandas as pd
|
|
11
12
|
import pytest
|
|
@@ -99,10 +100,8 @@ def test_multiindex_columns_do_not_raise_raw_error():
|
|
|
99
100
|
"""M4: MultiIndex columns must not leak a raw TypeError from .astype(str)."""
|
|
100
101
|
df = pd.DataFrame([[1, 2], [3, 4]], columns=pd.MultiIndex.from_tuples([("a", "x"), ("a", "y")]))
|
|
101
102
|
# Either succeeds (flattened labels) or raises a clean CleanFrameError — never a raw TypeError.
|
|
102
|
-
|
|
103
|
+
with suppress(CleanFrameError):
|
|
103
104
|
cf.clean(df)
|
|
104
|
-
except CleanFrameError:
|
|
105
|
-
pass
|
|
106
105
|
|
|
107
106
|
|
|
108
107
|
def test_string_index_with_name_column_does_not_crash():
|
|
@@ -7,6 +7,7 @@ from __future__ import annotations
|
|
|
7
7
|
|
|
8
8
|
import pandas as pd
|
|
9
9
|
|
|
10
|
+
from cleanframe import ops
|
|
10
11
|
from cleanframe.executor import execute
|
|
11
12
|
from cleanframe.ops import apply_column_op
|
|
12
13
|
from cleanframe.recipe import Recipe
|
|
@@ -36,8 +37,6 @@ _TRICKY = [
|
|
|
36
37
|
|
|
37
38
|
def _elementwise(op_name, series):
|
|
38
39
|
"""The reference: force the elementwise path by disabling the fast-path."""
|
|
39
|
-
import cleanframe.ops as ops
|
|
40
|
-
|
|
41
40
|
orig = ops._is_pure_string
|
|
42
41
|
ops._is_pure_string = lambda s: False
|
|
43
42
|
try:
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
__version__ = "0.3.0"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|