cleanframe-engine 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanframe/__init__.py +169 -0
- cleanframe/__main__.py +5 -0
- cleanframe/_util.py +438 -0
- cleanframe/_version.py +1 -0
- cleanframe/api.py +559 -0
- cleanframe/cli.py +688 -0
- cleanframe/codegen.py +666 -0
- cleanframe/dataio.py +506 -0
- cleanframe/detectors/__init__.py +43 -0
- cleanframe/detectors/base.py +221 -0
- cleanframe/detectors/categories.py +199 -0
- cleanframe/detectors/contacts.py +105 -0
- cleanframe/detectors/currency.py +112 -0
- cleanframe/detectors/dates.py +204 -0
- cleanframe/detectors/dedup.py +150 -0
- cleanframe/detectors/nulls.py +109 -0
- cleanframe/detectors/outliers.py +73 -0
- cleanframe/detectors/schema_mapping.py +125 -0
- cleanframe/detectors/text.py +105 -0
- cleanframe/detectors/units.py +86 -0
- cleanframe/diff.py +369 -0
- cleanframe/drift.py +283 -0
- cleanframe/errors.py +66 -0
- cleanframe/executor.py +229 -0
- cleanframe/fingerprint.py +83 -0
- cleanframe/issues.py +186 -0
- cleanframe/llm.py +811 -0
- cleanframe/ops.py +1245 -0
- cleanframe/planner.py +353 -0
- cleanframe/profile.py +413 -0
- cleanframe/py.typed +1 -0
- cleanframe/quality.py +81 -0
- cleanframe/readfix.py +160 -0
- cleanframe/recipe.py +398 -0
- cleanframe/report.py +345 -0
- cleanframe/result.py +144 -0
- cleanframe/schema.py +259 -0
- cleanframe/streaming.py +354 -0
- cleanframe/types.py +119 -0
- cleanframe/validate.py +363 -0
- cleanframe/workbook.py +370 -0
- cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
- cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
- cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
- cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
- cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""Text hygiene: whitespace and casing.
|
|
2
|
+
|
|
3
|
+
These are the safest, most universal fixes, so they run first (low priority
|
|
4
|
+
number). Whitespace trimming is non-lossy and always proposed when present.
|
|
5
|
+
Title-casing is opinionated (it would mangle ``"McDonald"``), so it is proposed
|
|
6
|
+
only for name-like columns and at a confidence low enough that ``strict`` mode
|
|
7
|
+
declines it.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import re
|
|
13
|
+
|
|
14
|
+
import pandas as pd
|
|
15
|
+
|
|
16
|
+
from .._util import is_string_like, sample_non_null
|
|
17
|
+
from ..issues import Issues, _cap_examples
|
|
18
|
+
from ..profile import _name_hint
|
|
19
|
+
from ..types import Op, Severity
|
|
20
|
+
from .base import DetectorContext, detector
|
|
21
|
+
|
|
22
|
+
_DOUBLE_WS = re.compile(r"\s{2,}")
|
|
23
|
+
_ALPHAWORDS = re.compile(r"^[A-Za-z][A-Za-z.'\- ]*$")
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@detector("whitespace", priority=10)
|
|
27
|
+
def detect_whitespace(series: pd.Series, ctx: DetectorContext) -> Issues:
|
|
28
|
+
"""Flag leading/trailing or repeated internal whitespace and propose a trim."""
|
|
29
|
+
issues = Issues()
|
|
30
|
+
cp = ctx.column_profile
|
|
31
|
+
if cp is None or cp.count == 0 or not is_string_like(series):
|
|
32
|
+
return issues
|
|
33
|
+
|
|
34
|
+
strings = [v for v in sample_non_null(series) if isinstance(v, str)]
|
|
35
|
+
trailing = [v for v in strings if v != v.strip()]
|
|
36
|
+
doubles = [v for v in strings if _DOUBLE_WS.search(v)]
|
|
37
|
+
if not trailing and not doubles:
|
|
38
|
+
return issues
|
|
39
|
+
|
|
40
|
+
# collapse_whitespace subsumes a plain strip, so pick one op, not both.
|
|
41
|
+
if doubles:
|
|
42
|
+
op = Op("collapse_whitespace")
|
|
43
|
+
msg = f"{len(set(trailing) | set(doubles))} value(s) have irregular whitespace"
|
|
44
|
+
else:
|
|
45
|
+
op = Op("strip_whitespace")
|
|
46
|
+
msg = f"{len(trailing)} value(s) have leading/trailing whitespace"
|
|
47
|
+
|
|
48
|
+
issues.add(
|
|
49
|
+
"whitespace",
|
|
50
|
+
msg,
|
|
51
|
+
severity=Severity.WARNING,
|
|
52
|
+
confidence=1.0,
|
|
53
|
+
evidence={
|
|
54
|
+
"leading_trailing": len(trailing),
|
|
55
|
+
"internal_doubles": len(doubles),
|
|
56
|
+
"examples": _cap_examples([repr(v) for v in (doubles or trailing)]),
|
|
57
|
+
},
|
|
58
|
+
ops=[op],
|
|
59
|
+
)
|
|
60
|
+
return issues
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@detector("text_case", priority=60)
|
|
64
|
+
def detect_text_case(series: pd.Series, ctx: DetectorContext) -> Issues:
|
|
65
|
+
"""Propose title-casing for inconsistently-cased *name* columns (low confidence)."""
|
|
66
|
+
issues = Issues()
|
|
67
|
+
cp = ctx.column_profile
|
|
68
|
+
if cp is None or cp.count == 0 or cp.semantic_type != "text":
|
|
69
|
+
return issues
|
|
70
|
+
if not _name_hint(ctx.column or "", "date") and not _looks_like_name_column(ctx.column, series):
|
|
71
|
+
return issues
|
|
72
|
+
|
|
73
|
+
strings = [v.strip() for v in sample_non_null(series) if isinstance(v, str)]
|
|
74
|
+
if not strings:
|
|
75
|
+
return issues
|
|
76
|
+
inconsistent = [v for v in strings if v != v.title()]
|
|
77
|
+
frac = len(inconsistent) / len(strings)
|
|
78
|
+
if frac < 0.25:
|
|
79
|
+
return issues
|
|
80
|
+
|
|
81
|
+
issues.add(
|
|
82
|
+
"inconsistent_case",
|
|
83
|
+
f"{len(inconsistent)} value(s) are not in Title Case",
|
|
84
|
+
severity=Severity.INFO,
|
|
85
|
+
confidence=0.55,
|
|
86
|
+
evidence={"count": len(inconsistent), "examples": _cap_examples(inconsistent)},
|
|
87
|
+
ops=[Op("title_case")],
|
|
88
|
+
)
|
|
89
|
+
return issues
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _looks_like_name_column(column: str | None, series: pd.Series) -> bool:
|
|
93
|
+
if column is None:
|
|
94
|
+
return False
|
|
95
|
+
hinted = any(h in str(column).lower() for h in ("name", "city", "state", "country", "title"))
|
|
96
|
+
if not hinted:
|
|
97
|
+
return False
|
|
98
|
+
strings = [v for v in series.dropna().head(50).tolist() if isinstance(v, str)]
|
|
99
|
+
if not strings:
|
|
100
|
+
return False
|
|
101
|
+
alpha = sum(1 for v in strings if _ALPHAWORDS.match(v.strip()))
|
|
102
|
+
return alpha / len(strings) >= 0.8
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
__all__ = ["detect_whitespace", "detect_text_case"]
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""Unit-quantity detection: ``5kg`` / ``5000 g`` / ``5 KG`` → one canonical unit."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections import Counter
|
|
6
|
+
|
|
7
|
+
import pandas as pd
|
|
8
|
+
|
|
9
|
+
from .._util import sample_non_null
|
|
10
|
+
from ..issues import Issues, _cap_examples
|
|
11
|
+
from ..ops import UNIT_FAMILIES, parse_unit_scalar
|
|
12
|
+
from ..types import Op, Severity
|
|
13
|
+
from .base import DetectorContext, detector
|
|
14
|
+
|
|
15
|
+
# Preferred target unit per family (SI-ish, matches README examples for mass).
|
|
16
|
+
_PREFERRED_TARGET = {"mass": "g", "length": "m", "volume": "l"}
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@detector("units", priority=46)
|
|
20
|
+
def detect_units(series: pd.Series, ctx: DetectorContext) -> Issues:
|
|
21
|
+
"""Detect mixed unit strings and propose ``normalize_unit`` to a single base."""
|
|
22
|
+
issues = Issues()
|
|
23
|
+
cp = ctx.column_profile
|
|
24
|
+
if cp is None or cp.count == 0:
|
|
25
|
+
return issues
|
|
26
|
+
if cp.semantic_type not in ("unit", "text", "categorical"):
|
|
27
|
+
return issues
|
|
28
|
+
|
|
29
|
+
parsed: list[tuple[float, str]] = []
|
|
30
|
+
raw_examples: list[str] = []
|
|
31
|
+
sample = sample_non_null(series)
|
|
32
|
+
for v in sample:
|
|
33
|
+
p = parse_unit_scalar(v)
|
|
34
|
+
if p is None:
|
|
35
|
+
continue
|
|
36
|
+
parsed.append(p)
|
|
37
|
+
raw_examples.append(str(v))
|
|
38
|
+
|
|
39
|
+
if len(parsed) < max(2, int(0.5 * max(len(sample), 1))):
|
|
40
|
+
return issues
|
|
41
|
+
|
|
42
|
+
families: Counter[str] = Counter()
|
|
43
|
+
units: Counter[str] = Counter()
|
|
44
|
+
for _, unit in parsed:
|
|
45
|
+
for fam, table in UNIT_FAMILIES.items():
|
|
46
|
+
if unit in table:
|
|
47
|
+
families[fam] += 1
|
|
48
|
+
units[unit] += 1
|
|
49
|
+
break
|
|
50
|
+
|
|
51
|
+
if not families:
|
|
52
|
+
return issues
|
|
53
|
+
family, _ = families.most_common(1)[0]
|
|
54
|
+
distinct = sorted(u for u, _ in units.most_common() if u in UNIT_FAMILIES[family])
|
|
55
|
+
if len(distinct) < 1:
|
|
56
|
+
return issues
|
|
57
|
+
|
|
58
|
+
mixed = len(distinct) > 1
|
|
59
|
+
# When units are mixed, normalize to the family SI base (matches README:
|
|
60
|
+
# 5kg / 5000 g → grams). A single unit keeps that unit.
|
|
61
|
+
if mixed:
|
|
62
|
+
target = _PREFERRED_TARGET[family]
|
|
63
|
+
else:
|
|
64
|
+
target = distinct[0] if distinct else _PREFERRED_TARGET[family]
|
|
65
|
+
|
|
66
|
+
issues.add(
|
|
67
|
+
"mixed_units" if mixed else "unit_format",
|
|
68
|
+
(
|
|
69
|
+
f"Unit quantities stored as text "
|
|
70
|
+
f"({'mixed units ' + str(distinct) if mixed else 'unit ' + distinct[0]}); "
|
|
71
|
+
f"normalize to {target}"
|
|
72
|
+
),
|
|
73
|
+
severity=Severity.WARNING if mixed else Severity.INFO,
|
|
74
|
+
confidence=0.9 if mixed else 0.75,
|
|
75
|
+
evidence={
|
|
76
|
+
"family": family,
|
|
77
|
+
"units": distinct,
|
|
78
|
+
"target": target,
|
|
79
|
+
"examples": _cap_examples(raw_examples),
|
|
80
|
+
},
|
|
81
|
+
ops=[Op("normalize_unit", {"to": target})],
|
|
82
|
+
)
|
|
83
|
+
return issues
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
__all__ = ["detect_units"]
|
cleanframe/diff.py
ADDED
|
@@ -0,0 +1,369 @@
|
|
|
1
|
+
"""Cell-level diff: exactly which values changed, and how.
|
|
2
|
+
|
|
3
|
+
Every transform is tracked. Given the original frame, the cleaned frame, and the
|
|
4
|
+
column lineage (which output column came from which source), :func:`compute_diff`
|
|
5
|
+
produces a :class:`CellDiff` recording each changed cell (``before → after``), the
|
|
6
|
+
columns added/removed/renamed, and the rows dropped and why. This is the lineage
|
|
7
|
+
that makes CleanFrame auditable — "Every changed cell is tracked."
|
|
8
|
+
|
|
9
|
+
Rows are matched by a stable integer id (the original positional index), so cells
|
|
10
|
+
line up correctly even after dedup removes rows.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import math
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
import pandas as pd
|
|
20
|
+
|
|
21
|
+
from ._util import DEFAULT_MAX_DIFF_CHANGES
|
|
22
|
+
from .ops import _is_na
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _equal(a: Any, b: Any) -> bool:
|
|
26
|
+
"""Value equality that treats NaN==NaN as equal and any NaN/value pair as changed."""
|
|
27
|
+
a_na, b_na = _is_na(a), _is_na(b)
|
|
28
|
+
if a_na and b_na:
|
|
29
|
+
return True
|
|
30
|
+
if a_na or b_na:
|
|
31
|
+
return False
|
|
32
|
+
if isinstance(a, float) and isinstance(b, float):
|
|
33
|
+
if math.isnan(a) and math.isnan(b):
|
|
34
|
+
return True
|
|
35
|
+
try:
|
|
36
|
+
return bool(a == b)
|
|
37
|
+
except Exception: # noqa: BLE001 - exotic types compare unequal
|
|
38
|
+
return False
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _changed_mask(before: pd.Series, after: pd.Series) -> pd.Series:
|
|
42
|
+
"""Vectorised element-wise 'changed?' mask with NaN==NaN treated as unchanged."""
|
|
43
|
+
both_na = before.isna().to_numpy() & after.isna().to_numpy()
|
|
44
|
+
try:
|
|
45
|
+
equal = (before.to_numpy() == after.to_numpy())
|
|
46
|
+
except Exception: # noqa: BLE001 - dtype mismatch -> fall back element-wise
|
|
47
|
+
equal = pd.Series(
|
|
48
|
+
[_equal(b, a) for b, a in zip(before.tolist(), after.tolist(), strict=True)],
|
|
49
|
+
index=before.index,
|
|
50
|
+
).to_numpy()
|
|
51
|
+
equal = pd.Series(equal, index=before.index).fillna(False).astype(bool).to_numpy()
|
|
52
|
+
return pd.Series(~(equal | both_na), index=before.index)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass
|
|
56
|
+
class CellChange:
|
|
57
|
+
row_id: int
|
|
58
|
+
column: str
|
|
59
|
+
before: Any
|
|
60
|
+
after: Any
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass
|
|
64
|
+
class CellDiff:
|
|
65
|
+
"""A structured record of everything a recipe changed."""
|
|
66
|
+
|
|
67
|
+
changes: list[CellChange] = field(default_factory=list)
|
|
68
|
+
added_columns: list[str] = field(default_factory=list)
|
|
69
|
+
removed_columns: list[str] = field(default_factory=list)
|
|
70
|
+
renamed_columns: dict[str, str] = field(default_factory=dict) # source -> output
|
|
71
|
+
dropped_rows: list[tuple[int, str]] = field(default_factory=list) # (row_id, reason)
|
|
72
|
+
n_rows_before: int = 0
|
|
73
|
+
n_rows_after: int = 0
|
|
74
|
+
#: True when ``max_changes`` stopped recording further cell edits (summary still exact).
|
|
75
|
+
truncated: bool = False
|
|
76
|
+
#: Total cells that changed, including those not stored when truncated.
|
|
77
|
+
total_changed_cells: int = 0
|
|
78
|
+
#: Every column with at least one change, including changes beyond the cap.
|
|
79
|
+
changed_column_names: list[str] = field(default_factory=list)
|
|
80
|
+
|
|
81
|
+
# -- summaries -------------------------------------------------------
|
|
82
|
+
@property
|
|
83
|
+
def changed_cells(self) -> int:
|
|
84
|
+
return self.total_changed_cells or len(self.changes)
|
|
85
|
+
|
|
86
|
+
@property
|
|
87
|
+
def changed_columns(self) -> list[str]:
|
|
88
|
+
seen: list[str] = []
|
|
89
|
+
for c in self.changes:
|
|
90
|
+
if c.column not in seen:
|
|
91
|
+
seen.append(c.column)
|
|
92
|
+
return seen
|
|
93
|
+
|
|
94
|
+
def changes_by_column(self) -> dict[str, list[CellChange]]:
|
|
95
|
+
out: dict[str, list[CellChange]] = {}
|
|
96
|
+
for c in self.changes:
|
|
97
|
+
out.setdefault(c.column, []).append(c)
|
|
98
|
+
return out
|
|
99
|
+
|
|
100
|
+
def to_frame(self) -> pd.DataFrame:
|
|
101
|
+
"""A tidy DataFrame of changes: ``row_id, column, before, after``."""
|
|
102
|
+
return pd.DataFrame(
|
|
103
|
+
[(c.row_id, c.column, c.before, c.after) for c in self.changes],
|
|
104
|
+
columns=["row_id", "column", "before", "after"],
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
def summary(self) -> dict[str, Any]:
|
|
108
|
+
return {
|
|
109
|
+
"changed_cells": self.changed_cells,
|
|
110
|
+
"changed_columns": len(self.changed_column_names or self.changed_columns),
|
|
111
|
+
"added_columns": list(self.added_columns),
|
|
112
|
+
"removed_columns": list(self.removed_columns),
|
|
113
|
+
"renamed_columns": dict(self.renamed_columns),
|
|
114
|
+
"rows_dropped": len(self.dropped_rows),
|
|
115
|
+
"rows_before": self.n_rows_before,
|
|
116
|
+
"rows_after": self.n_rows_after,
|
|
117
|
+
"truncated": self.truncated,
|
|
118
|
+
"stored_changes": len(self.changes),
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
def is_empty(self) -> bool:
|
|
122
|
+
return not (
|
|
123
|
+
self.changes
|
|
124
|
+
or self.added_columns
|
|
125
|
+
or self.removed_columns
|
|
126
|
+
or self.renamed_columns
|
|
127
|
+
or self.dropped_rows
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
# -- rendering -------------------------------------------------------
|
|
131
|
+
def render(
|
|
132
|
+
self, *, max_per_column: int = 8, color: bool | None = None, ascii: bool = False
|
|
133
|
+
) -> str:
|
|
134
|
+
"""Render a git-diff-style textual summary (also used by ``show``).
|
|
135
|
+
|
|
136
|
+
Pass ``ascii=True`` to substitute plain ASCII for the ``→``/``∅`` glyphs so
|
|
137
|
+
the text is safe to print on a non-UTF-8 console (e.g. a cp1252 Windows
|
|
138
|
+
terminal). :meth:`show` selects this automatically per output stream.
|
|
139
|
+
"""
|
|
140
|
+
return _render_text(self, max_per_column=max_per_column, color=color, ascii=ascii)
|
|
141
|
+
|
|
142
|
+
def show(
|
|
143
|
+
self,
|
|
144
|
+
*,
|
|
145
|
+
max_per_column: int = 8,
|
|
146
|
+
color: bool | None = None,
|
|
147
|
+
stream: Any = None,
|
|
148
|
+
ascii: bool | None = None,
|
|
149
|
+
) -> None:
|
|
150
|
+
"""Print the diff, git-diff style. Never crashes on a non-UTF-8 console.
|
|
151
|
+
|
|
152
|
+
``ascii`` defaults to auto: ASCII glyphs are used when the target stream
|
|
153
|
+
cannot encode ``→``/``∅`` (the default Windows cp1252 console), so library
|
|
154
|
+
users get clean output without the CLI's stdout reconfiguration.
|
|
155
|
+
"""
|
|
156
|
+
import sys
|
|
157
|
+
|
|
158
|
+
out = stream if stream is not None else sys.stdout
|
|
159
|
+
if ascii is None:
|
|
160
|
+
ascii = not _stream_supports_unicode(out)
|
|
161
|
+
text = self.render(max_per_column=max_per_column, color=color, ascii=ascii)
|
|
162
|
+
try:
|
|
163
|
+
print(text, file=out)
|
|
164
|
+
except UnicodeEncodeError:
|
|
165
|
+
# The glyphs are not the only risk: a currency symbol in the *data* is
|
|
166
|
+
# unencodable on a cp1252 console too, so replace whatever is left.
|
|
167
|
+
fallback = self.render(max_per_column=max_per_column, color=color, ascii=True)
|
|
168
|
+
encoding = getattr(out, "encoding", None) or "ascii"
|
|
169
|
+
print(
|
|
170
|
+
fallback.encode(encoding, errors="replace").decode(encoding, errors="replace"),
|
|
171
|
+
file=out,
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
def __repr__(self) -> str: # pragma: no cover - cosmetic
|
|
175
|
+
s = self.summary()
|
|
176
|
+
return (
|
|
177
|
+
f"CellDiff(changed_cells={s['changed_cells']}, "
|
|
178
|
+
f"columns={s['changed_columns']}, rows_dropped={s['rows_dropped']})"
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def compute_diff(
|
|
183
|
+
original: pd.DataFrame,
|
|
184
|
+
cleaned: pd.DataFrame,
|
|
185
|
+
lineage: dict[str, str | None],
|
|
186
|
+
*,
|
|
187
|
+
dropped_rows: list[tuple[int, str]] | None = None,
|
|
188
|
+
dropped_after: pd.DataFrame | None = None,
|
|
189
|
+
original_columns: list[str] | None = None,
|
|
190
|
+
n_rows_before: int | None = None,
|
|
191
|
+
max_changes: int | None = DEFAULT_MAX_DIFF_CHANGES,
|
|
192
|
+
) -> CellDiff:
|
|
193
|
+
"""Compute a :class:`CellDiff`.
|
|
194
|
+
|
|
195
|
+
Parameters
|
|
196
|
+
----------
|
|
197
|
+
original:
|
|
198
|
+
The input frame, indexed by stable row id (positional 0..n-1).
|
|
199
|
+
cleaned:
|
|
200
|
+
The output frame, indexed by the surviving row ids.
|
|
201
|
+
lineage:
|
|
202
|
+
``{output_column: source_column_or_None}``. ``None`` means the column was
|
|
203
|
+
derived/added (no "before" value).
|
|
204
|
+
dropped_rows:
|
|
205
|
+
``(row_id, reason)`` pairs for rows removed during execution.
|
|
206
|
+
dropped_after:
|
|
207
|
+
Post-transform values (indexed by row id) of rows that were later dropped,
|
|
208
|
+
so a value rewrite applied to a row before it was removed is still tracked
|
|
209
|
+
as a changed cell (invariant #5) rather than vanishing with the row.
|
|
210
|
+
original_columns:
|
|
211
|
+
The full list of input column names. Pass this when ``original`` holds only
|
|
212
|
+
a *subset* of columns (the executor snapshots just the op-touched ones to
|
|
213
|
+
halve peak memory); columns absent from ``original`` are treated as
|
|
214
|
+
value-unchanged pass-throughs. Defaults to ``original.columns``.
|
|
215
|
+
max_changes:
|
|
216
|
+
Cap on stored :class:`CellChange` entries. ``None`` stores every change
|
|
217
|
+
(can OOM on large dirty frames). Counts in :attr:`CellDiff.changed_cells`
|
|
218
|
+
remain exact even when the detail list is truncated.
|
|
219
|
+
"""
|
|
220
|
+
diff = CellDiff(
|
|
221
|
+
dropped_rows=list(dropped_rows or []),
|
|
222
|
+
n_rows_before=n_rows_before if n_rows_before is not None else int(len(original)),
|
|
223
|
+
n_rows_after=int(len(cleaned)),
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
all_columns = [str(c) for c in (original_columns if original_columns is not None else original.columns)]
|
|
227
|
+
source_cols = set(all_columns)
|
|
228
|
+
snapshot_cols = set(original.columns.astype(str))
|
|
229
|
+
used_sources: set[str] = set()
|
|
230
|
+
total = 0
|
|
231
|
+
store_cap = max_changes # None = unlimited
|
|
232
|
+
|
|
233
|
+
for out_col in cleaned.columns:
|
|
234
|
+
out_col = str(out_col)
|
|
235
|
+
source = lineage.get(out_col, out_col if out_col in source_cols else None)
|
|
236
|
+
if source is None:
|
|
237
|
+
diff.added_columns.append(out_col)
|
|
238
|
+
continue
|
|
239
|
+
used_sources.add(source)
|
|
240
|
+
if source != out_col:
|
|
241
|
+
diff.renamed_columns[source] = out_col
|
|
242
|
+
|
|
243
|
+
# A source absent from the (possibly partial) snapshot is a value-unchanged
|
|
244
|
+
# pass-through: record the rename, but there are no cell edits to diff.
|
|
245
|
+
if source not in snapshot_cols:
|
|
246
|
+
continue
|
|
247
|
+
|
|
248
|
+
after_series = cleaned[out_col]
|
|
249
|
+
if dropped_after is not None and out_col in dropped_after.columns:
|
|
250
|
+
# Append the post-transform values of dropped rows so their edits are
|
|
251
|
+
# tracked too (the rows still appear separately in dropped_rows).
|
|
252
|
+
after_series = pd.concat([after_series, dropped_after[out_col]])
|
|
253
|
+
before_series = original[source].reindex(after_series.index)
|
|
254
|
+
changed_mask = _changed_mask(before_series, after_series)
|
|
255
|
+
changed_ids = after_series.index[changed_mask.to_numpy()]
|
|
256
|
+
total += int(len(changed_ids))
|
|
257
|
+
if len(changed_ids) and out_col not in diff.changed_column_names:
|
|
258
|
+
diff.changed_column_names.append(out_col)
|
|
259
|
+
|
|
260
|
+
# Cap-aware bulk extraction: slice to the remaining budget FIRST, then pull
|
|
261
|
+
# before/after in one vectorised call each (avoids per-cell .loc — ~44x).
|
|
262
|
+
if store_cap is not None:
|
|
263
|
+
remaining = store_cap - len(diff.changes)
|
|
264
|
+
if remaining <= 0:
|
|
265
|
+
if len(changed_ids):
|
|
266
|
+
diff.truncated = True
|
|
267
|
+
continue
|
|
268
|
+
if len(changed_ids) > remaining:
|
|
269
|
+
diff.truncated = True
|
|
270
|
+
changed_ids = changed_ids[:remaining]
|
|
271
|
+
if len(changed_ids) == 0:
|
|
272
|
+
continue
|
|
273
|
+
befores = before_series.loc[changed_ids].tolist()
|
|
274
|
+
afters = after_series.loc[changed_ids].tolist()
|
|
275
|
+
for row_id, before, after in zip(changed_ids, befores, afters, strict=True):
|
|
276
|
+
diff.changes.append(CellChange(int(row_id), out_col, before, after))
|
|
277
|
+
|
|
278
|
+
for src in all_columns:
|
|
279
|
+
if src not in used_sources:
|
|
280
|
+
diff.removed_columns.append(src)
|
|
281
|
+
|
|
282
|
+
diff.total_changed_cells = total
|
|
283
|
+
return diff
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
# ---------------------------------------------------------------------------
|
|
287
|
+
# text rendering
|
|
288
|
+
# ---------------------------------------------------------------------------
|
|
289
|
+
def _stream_supports_unicode(stream: Any) -> bool:
|
|
290
|
+
"""True if ``stream`` can encode the diff glyphs (``→``/``∅``/``⚠``)."""
|
|
291
|
+
enc = getattr(stream, "encoding", None) or "utf-8"
|
|
292
|
+
try:
|
|
293
|
+
"→∅⚠".encode(enc)
|
|
294
|
+
return True
|
|
295
|
+
except (UnicodeEncodeError, LookupError):
|
|
296
|
+
return False
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def _fmt(value: Any, ascii: bool = False) -> str:
|
|
300
|
+
if _is_na(value):
|
|
301
|
+
return "<NA>" if ascii else "∅"
|
|
302
|
+
if isinstance(value, str):
|
|
303
|
+
return repr(value)
|
|
304
|
+
return str(value)
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _render_text(
|
|
308
|
+
diff: CellDiff, *, max_per_column: int, color: bool | None, ascii: bool = False
|
|
309
|
+
) -> str:
|
|
310
|
+
import os
|
|
311
|
+
import sys
|
|
312
|
+
|
|
313
|
+
arrow = "->" if ascii else "→"
|
|
314
|
+
if color is None:
|
|
315
|
+
color = sys.stdout.isatty() and os.environ.get("NO_COLOR") is None
|
|
316
|
+
red = "\x1b[31m" if color else ""
|
|
317
|
+
green = "\x1b[32m" if color else ""
|
|
318
|
+
dim = "\x1b[2m" if color else ""
|
|
319
|
+
bold = "\x1b[1m" if color else ""
|
|
320
|
+
reset = "\x1b[0m" if color else ""
|
|
321
|
+
|
|
322
|
+
if diff.is_empty():
|
|
323
|
+
return f"{dim}No changes.{reset}"
|
|
324
|
+
|
|
325
|
+
lines: list[str] = []
|
|
326
|
+
s = diff.summary()
|
|
327
|
+
lines.append(
|
|
328
|
+
f"{bold}CleanFrame diff{reset} "
|
|
329
|
+
f"{s['changed_cells']} cell(s) changed in {s['changed_columns']} column(s), "
|
|
330
|
+
f"{s['rows_dropped']} row(s) dropped "
|
|
331
|
+
f"({s['rows_before']} {arrow} {s['rows_after']} rows)"
|
|
332
|
+
+ (
|
|
333
|
+
f" {dim}[detail truncated to {s['stored_changes']}]{reset}"
|
|
334
|
+
if s.get("truncated")
|
|
335
|
+
else ""
|
|
336
|
+
)
|
|
337
|
+
)
|
|
338
|
+
if diff.renamed_columns:
|
|
339
|
+
renames = ", ".join(f"{k} {arrow} {v}" for k, v in diff.renamed_columns.items())
|
|
340
|
+
lines.append(f" {dim}renamed:{reset} {renames}")
|
|
341
|
+
if diff.added_columns:
|
|
342
|
+
lines.append(f" {green}added columns:{reset} {', '.join(diff.added_columns)}")
|
|
343
|
+
if diff.removed_columns:
|
|
344
|
+
lines.append(f" {red}removed columns:{reset} {', '.join(diff.removed_columns)}")
|
|
345
|
+
|
|
346
|
+
for column, changes in diff.changes_by_column().items():
|
|
347
|
+
lines.append("")
|
|
348
|
+
lines.append(f"{bold}{column}{reset} {dim}({len(changes)} changed){reset}")
|
|
349
|
+
for change in changes[:max_per_column]:
|
|
350
|
+
lines.append(
|
|
351
|
+
f" row {change.row_id}: "
|
|
352
|
+
f"{red}- {_fmt(change.before, ascii)}{reset} "
|
|
353
|
+
f"{green}+ {_fmt(change.after, ascii)}{reset}"
|
|
354
|
+
)
|
|
355
|
+
if len(changes) > max_per_column:
|
|
356
|
+
lines.append(f" {dim}… and {len(changes) - max_per_column} more{reset}")
|
|
357
|
+
|
|
358
|
+
if diff.dropped_rows:
|
|
359
|
+
lines.append("")
|
|
360
|
+
reasons: dict[str, int] = {}
|
|
361
|
+
for _, reason in diff.dropped_rows:
|
|
362
|
+
reasons[reason] = reasons.get(reason, 0) + 1
|
|
363
|
+
detail = ", ".join(f"{n} ({r})" for r, n in reasons.items())
|
|
364
|
+
lines.append(f"{red}dropped rows:{reset} {detail}")
|
|
365
|
+
|
|
366
|
+
return "\n".join(lines)
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
__all__ = ["CellDiff", "CellChange", "compute_diff"]
|