cleanframe-engine 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanframe/__init__.py +169 -0
- cleanframe/__main__.py +5 -0
- cleanframe/_util.py +438 -0
- cleanframe/_version.py +1 -0
- cleanframe/api.py +559 -0
- cleanframe/cli.py +688 -0
- cleanframe/codegen.py +666 -0
- cleanframe/dataio.py +506 -0
- cleanframe/detectors/__init__.py +43 -0
- cleanframe/detectors/base.py +221 -0
- cleanframe/detectors/categories.py +199 -0
- cleanframe/detectors/contacts.py +105 -0
- cleanframe/detectors/currency.py +112 -0
- cleanframe/detectors/dates.py +204 -0
- cleanframe/detectors/dedup.py +150 -0
- cleanframe/detectors/nulls.py +109 -0
- cleanframe/detectors/outliers.py +73 -0
- cleanframe/detectors/schema_mapping.py +125 -0
- cleanframe/detectors/text.py +105 -0
- cleanframe/detectors/units.py +86 -0
- cleanframe/diff.py +369 -0
- cleanframe/drift.py +283 -0
- cleanframe/errors.py +66 -0
- cleanframe/executor.py +229 -0
- cleanframe/fingerprint.py +83 -0
- cleanframe/issues.py +186 -0
- cleanframe/llm.py +811 -0
- cleanframe/ops.py +1245 -0
- cleanframe/planner.py +353 -0
- cleanframe/profile.py +413 -0
- cleanframe/py.typed +1 -0
- cleanframe/quality.py +81 -0
- cleanframe/readfix.py +160 -0
- cleanframe/recipe.py +398 -0
- cleanframe/report.py +345 -0
- cleanframe/result.py +144 -0
- cleanframe/schema.py +259 -0
- cleanframe/streaming.py +354 -0
- cleanframe/types.py +119 -0
- cleanframe/validate.py +363 -0
- cleanframe/workbook.py +370 -0
- cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
- cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
- cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
- cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
- cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
cleanframe/drift.py
ADDED
|
@@ -0,0 +1,283 @@
|
|
|
1
|
+
"""Schema-drift detection: catch the file that came back *different*.
|
|
2
|
+
|
|
3
|
+
Before replaying a recipe on next month's file, compare that file against the
|
|
4
|
+
recipe's ``source_fingerprint`` and its declared expectations. If a column was
|
|
5
|
+
renamed, a new one appeared, a dtype shifted, or values stopped matching the
|
|
6
|
+
formats the recipe parses, CleanFrame **stops and tells you** instead of silently
|
|
7
|
+
producing garbage — exactly the failure mode that makes hand-cleaning fragile.
|
|
8
|
+
|
|
9
|
+
Findings carry fuzzy-match suggestions ("94% match to recipe column …") so the fix
|
|
10
|
+
is obvious. By default ``apply_recipe`` / ``cleanframe apply`` **stop** on drift;
|
|
11
|
+
pass ``on_drift="warn"`` / ``--force`` only when you intentionally want to continue.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
|
|
18
|
+
import pandas as pd
|
|
19
|
+
|
|
20
|
+
from ._util import best_match, canonicalize_dtype, similarity
|
|
21
|
+
from .errors import CleanFrameError
|
|
22
|
+
from .fingerprint import DEFAULT_SAMPLE_ROWS, fingerprint_dataframe
|
|
23
|
+
from .llm import _sketch
|
|
24
|
+
from .ops import parse_dates_to_datetime
|
|
25
|
+
from .recipe import Recipe
|
|
26
|
+
from .types import Severity
|
|
27
|
+
|
|
28
|
+
_RENAME_CONFIDENCE = 0.6
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass
|
|
32
|
+
class DriftFinding:
|
|
33
|
+
kind: str
|
|
34
|
+
message: str
|
|
35
|
+
severity: Severity = Severity.WARNING
|
|
36
|
+
column: str | None = None
|
|
37
|
+
suggestion: str | None = None
|
|
38
|
+
evidence: dict = field(default_factory=dict)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass
|
|
42
|
+
class DriftReport:
|
|
43
|
+
findings: list[DriftFinding] = field(default_factory=list)
|
|
44
|
+
source: str | None = None
|
|
45
|
+
|
|
46
|
+
@property
|
|
47
|
+
def has_drift(self) -> bool:
|
|
48
|
+
return any(f.severity.rank >= Severity.WARNING.rank for f in self.findings)
|
|
49
|
+
|
|
50
|
+
@property
|
|
51
|
+
def worst(self) -> Severity | None:
|
|
52
|
+
if not self.findings:
|
|
53
|
+
return None
|
|
54
|
+
return max((f.severity for f in self.findings), key=lambda s: s.rank)
|
|
55
|
+
|
|
56
|
+
def by_kind(self, kind: str) -> list[DriftFinding]:
|
|
57
|
+
return [f for f in self.findings if f.kind == kind]
|
|
58
|
+
|
|
59
|
+
def render(self) -> str:
|
|
60
|
+
if not self.findings:
|
|
61
|
+
where = f" in {self.source}" if self.source else ""
|
|
62
|
+
return f"✓ No schema drift detected{where}."
|
|
63
|
+
where = f" in {self.source}" if self.source else ""
|
|
64
|
+
lines = [f"⚠ Schema drift detected{where}"]
|
|
65
|
+
for f in self.findings:
|
|
66
|
+
lines.append(f" • {f.message}")
|
|
67
|
+
if f.suggestion:
|
|
68
|
+
lines.append(f" ↳ {f.suggestion}")
|
|
69
|
+
lines.append(
|
|
70
|
+
" Run `cleanframe suggest <file> --recipe <recipe.yaml> --update` to review a patch."
|
|
71
|
+
)
|
|
72
|
+
return "\n".join(lines)
|
|
73
|
+
|
|
74
|
+
def __repr__(self) -> str: # pragma: no cover - cosmetic
|
|
75
|
+
return f"DriftReport({len(self.findings)} finding(s), has_drift={self.has_drift})"
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def detect_drift(df: pd.DataFrame, recipe: Recipe, *, source: str | None = None) -> DriftReport:
|
|
79
|
+
"""Compare an incoming frame against the expectations baked into ``recipe``."""
|
|
80
|
+
report = DriftReport(source=source)
|
|
81
|
+
fp = recipe.source_fingerprint or {}
|
|
82
|
+
if not isinstance(fp, dict):
|
|
83
|
+
raise CleanFrameError(
|
|
84
|
+
f"recipe source_fingerprint must be a mapping, got {type(fp).__name__}."
|
|
85
|
+
)
|
|
86
|
+
raw_cols = fp.get("column_names") or []
|
|
87
|
+
if not isinstance(raw_cols, (list, tuple)):
|
|
88
|
+
raise CleanFrameError("recipe source_fingerprint.column_names must be a list.")
|
|
89
|
+
raw_dtypes = fp.get("dtypes") or {}
|
|
90
|
+
if not isinstance(raw_dtypes, dict):
|
|
91
|
+
raise CleanFrameError("recipe source_fingerprint.dtypes must be a mapping.")
|
|
92
|
+
expected_cols: list[str] = [str(c) for c in raw_cols]
|
|
93
|
+
expected_dtypes: dict[str, str] = {str(k): str(v) for k, v in raw_dtypes.items()}
|
|
94
|
+
actual_cols = [str(c) for c in df.columns]
|
|
95
|
+
|
|
96
|
+
# A fingerprint without column names (hand-written recipe) says nothing about
|
|
97
|
+
# structure; comparing against an empty list would call every column new.
|
|
98
|
+
missing = [c for c in expected_cols if c not in actual_cols] if expected_cols else []
|
|
99
|
+
new = [c for c in actual_cols if c not in expected_cols] if expected_cols else []
|
|
100
|
+
|
|
101
|
+
# -- new columns: try to explain each as a rename ------------------
|
|
102
|
+
for col in new:
|
|
103
|
+
candidate, score = best_match(col, missing + [c.output_name for c in recipe.columns])
|
|
104
|
+
if candidate and score >= _RENAME_CONFIDENCE:
|
|
105
|
+
report.findings.append(
|
|
106
|
+
DriftFinding(
|
|
107
|
+
kind="renamed_column",
|
|
108
|
+
message=f'Column "{col}" is new — {score:.0%} match to recipe column "{candidate}"',
|
|
109
|
+
severity=Severity.WARNING,
|
|
110
|
+
column=col,
|
|
111
|
+
suggestion=f'If "{col}" replaces "{candidate}", update the recipe to map it.',
|
|
112
|
+
evidence={"match": candidate, "score": round(score, 3)},
|
|
113
|
+
)
|
|
114
|
+
)
|
|
115
|
+
else:
|
|
116
|
+
report.findings.append(
|
|
117
|
+
DriftFinding(
|
|
118
|
+
kind="new_column",
|
|
119
|
+
message=f'Column "{col}" is new and unrecognised',
|
|
120
|
+
severity=Severity.WARNING,
|
|
121
|
+
column=col,
|
|
122
|
+
suggestion="Confirm the new column is expected, or drop/ignore it before apply.",
|
|
123
|
+
evidence={},
|
|
124
|
+
)
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
# -- missing columns not explained by a rename --------------------
|
|
128
|
+
explained = {f.evidence.get("match") for f in report.by_kind("renamed_column")}
|
|
129
|
+
for col in missing:
|
|
130
|
+
if col in explained:
|
|
131
|
+
continue
|
|
132
|
+
report.findings.append(
|
|
133
|
+
DriftFinding(
|
|
134
|
+
kind="missing_column",
|
|
135
|
+
message=f'Column "{col}" expected by the recipe is missing',
|
|
136
|
+
severity=Severity.WARNING,
|
|
137
|
+
column=col,
|
|
138
|
+
suggestion=_closest_hint(col, new),
|
|
139
|
+
evidence={},
|
|
140
|
+
)
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
# -- dtype changes (canonical families — object vs str is not drift) -
|
|
144
|
+
for col in actual_cols:
|
|
145
|
+
if col not in expected_dtypes:
|
|
146
|
+
continue
|
|
147
|
+
was = expected_dtypes[col]
|
|
148
|
+
now = str(df[col].dtype)
|
|
149
|
+
if canonicalize_dtype(was) == canonicalize_dtype(now):
|
|
150
|
+
continue
|
|
151
|
+
report.findings.append(
|
|
152
|
+
DriftFinding(
|
|
153
|
+
kind="dtype_change",
|
|
154
|
+
message=f'Column "{col}" changed dtype ({was} → {now})',
|
|
155
|
+
severity=Severity.WARNING,
|
|
156
|
+
column=col,
|
|
157
|
+
suggestion="Re-plan or update casts if ops assume the old dtype.",
|
|
158
|
+
evidence={"was": was, "now": now},
|
|
159
|
+
)
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
# -- content fingerprint (row_count / hash_sample) ------------------
|
|
163
|
+
_detect_content_drift(df, fp, report)
|
|
164
|
+
|
|
165
|
+
# -- value-level format drift (declared parse_date formats) --------
|
|
166
|
+
_detect_format_drift(df, recipe, report)
|
|
167
|
+
|
|
168
|
+
return report
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _detect_content_drift(df: pd.DataFrame, fp: dict, report: DriftReport) -> None:
|
|
172
|
+
"""Compare stored content fingerprint fields when present.
|
|
173
|
+
|
|
174
|
+
Row-count and sample-hash changes are expected for monthly files, so they are
|
|
175
|
+
INFO (visible in the report, do not stop apply). Schema / dtype / format
|
|
176
|
+
findings remain the apply-blocking signals. Same row count + different hash is
|
|
177
|
+
called out in the message so operators can spot a possible wrong-file swap.
|
|
178
|
+
"""
|
|
179
|
+
expected_rows = fp.get("row_count")
|
|
180
|
+
expected_hash = fp.get("hash_sample")
|
|
181
|
+
if expected_rows is None and expected_hash is None:
|
|
182
|
+
return
|
|
183
|
+
|
|
184
|
+
try:
|
|
185
|
+
sample_rows = int(fp.get("sampled_rows") or DEFAULT_SAMPLE_ROWS)
|
|
186
|
+
except (TypeError, ValueError):
|
|
187
|
+
sample_rows = DEFAULT_SAMPLE_ROWS
|
|
188
|
+
actual_fp = fingerprint_dataframe(df, sample_rows=sample_rows)
|
|
189
|
+
actual_rows = actual_fp["row_count"]
|
|
190
|
+
actual_hash = actual_fp["hash_sample"]
|
|
191
|
+
|
|
192
|
+
try:
|
|
193
|
+
rows_changed = expected_rows is not None and int(expected_rows) != int(actual_rows)
|
|
194
|
+
except (TypeError, ValueError):
|
|
195
|
+
rows_changed = False
|
|
196
|
+
expected_rows = None
|
|
197
|
+
hash_changed = expected_hash is not None and str(expected_hash) != str(actual_hash)
|
|
198
|
+
|
|
199
|
+
if rows_changed:
|
|
200
|
+
report.findings.append(
|
|
201
|
+
DriftFinding(
|
|
202
|
+
kind="row_count_change",
|
|
203
|
+
message=f"Row count changed ({expected_rows} → {actual_rows})",
|
|
204
|
+
severity=Severity.INFO,
|
|
205
|
+
evidence={"was": int(expected_rows), "now": int(actual_rows)}, # type: ignore[arg-type]
|
|
206
|
+
)
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
if hash_changed:
|
|
210
|
+
same_n = bool(expected_rows is not None and int(expected_rows) == int(actual_rows))
|
|
211
|
+
report.findings.append(
|
|
212
|
+
DriftFinding(
|
|
213
|
+
kind="content_hash_change",
|
|
214
|
+
message=(
|
|
215
|
+
"Leading-row content hash changed "
|
|
216
|
+
f"({expected_hash} → {actual_hash})"
|
|
217
|
+
),
|
|
218
|
+
severity=Severity.INFO,
|
|
219
|
+
suggestion=(
|
|
220
|
+
"Same row count but different sample values — confirm this is "
|
|
221
|
+
"the intended file (not a schema-matched swap)."
|
|
222
|
+
if same_n
|
|
223
|
+
else None
|
|
224
|
+
),
|
|
225
|
+
evidence={"was": str(expected_hash), "now": str(actual_hash)},
|
|
226
|
+
)
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _detect_format_drift(df: pd.DataFrame, recipe: Recipe, report: DriftReport) -> None:
|
|
231
|
+
for col_recipe in recipe.columns:
|
|
232
|
+
src = col_recipe.source
|
|
233
|
+
if src not in df.columns:
|
|
234
|
+
continue
|
|
235
|
+
for op in col_recipe.ops:
|
|
236
|
+
if op.name != "parse_date":
|
|
237
|
+
continue
|
|
238
|
+
formats = op.params.get("formats")
|
|
239
|
+
if not formats:
|
|
240
|
+
continue
|
|
241
|
+
series = df[src].dropna()
|
|
242
|
+
if series.empty:
|
|
243
|
+
continue
|
|
244
|
+
parsed = parse_dates_to_datetime(
|
|
245
|
+
series,
|
|
246
|
+
list(formats),
|
|
247
|
+
dayfirst=bool(op.params.get("dayfirst", False)),
|
|
248
|
+
yearfirst=bool(op.params.get("yearfirst", False)),
|
|
249
|
+
)
|
|
250
|
+
unmatched = series[parsed.isna()]
|
|
251
|
+
if len(unmatched):
|
|
252
|
+
examples = [str(v) for v in unmatched.head(3).tolist()]
|
|
253
|
+
sketches = sorted({_sketch(e) for e in examples})
|
|
254
|
+
report.findings.append(
|
|
255
|
+
DriftFinding(
|
|
256
|
+
kind="format_drift",
|
|
257
|
+
message=(
|
|
258
|
+
f'{len(unmatched)} value(s) in "{src}" match no allowed date format '
|
|
259
|
+
f"(new: {examples[0]!r})"
|
|
260
|
+
),
|
|
261
|
+
severity=Severity.WARNING,
|
|
262
|
+
column=src,
|
|
263
|
+
suggestion=f"New pattern {sketches[0]!r}; add its format to the recipe.",
|
|
264
|
+
evidence={
|
|
265
|
+
"count": int(len(unmatched)),
|
|
266
|
+
"examples": examples,
|
|
267
|
+
"sketches": sketches,
|
|
268
|
+
},
|
|
269
|
+
)
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _closest_hint(col: str, candidates: list[str]) -> str | None:
|
|
274
|
+
if not candidates:
|
|
275
|
+
return None
|
|
276
|
+
best = max(candidates, key=lambda c: similarity(col, c))
|
|
277
|
+
score = similarity(col, best)
|
|
278
|
+
if score >= 0.4:
|
|
279
|
+
return f'Did a new column "{best}" ({score:.0%} match) replace it?'
|
|
280
|
+
return None
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
__all__ = ["detect_drift", "DriftReport", "DriftFinding"]
|
cleanframe/errors.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""Exception hierarchy for CleanFrame.
|
|
2
|
+
|
|
3
|
+
Every error CleanFrame raises deliberately derives from :class:`CleanFrameError`,
|
|
4
|
+
so callers can ``except cleanframe.CleanFrameError`` to catch anything the library
|
|
5
|
+
raises on purpose without also swallowing unrelated bugs.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class CleanFrameError(Exception):
|
|
12
|
+
"""Base class for all errors raised deliberately by CleanFrame."""
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class CleanFrameWarning(UserWarning):
|
|
16
|
+
"""Category for every warning CleanFrame raises deliberately.
|
|
17
|
+
|
|
18
|
+
Silence the library's advisory output with
|
|
19
|
+
``warnings.simplefilter("ignore", cleanframe.CleanFrameWarning)``.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class RecipeError(CleanFrameError):
|
|
24
|
+
"""A recipe is malformed, references an unknown op, or fails to load."""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class OpError(CleanFrameError):
|
|
28
|
+
"""An op could not be applied (bad params, or a runtime failure in pandas)."""
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class ExecutionError(CleanFrameError):
|
|
32
|
+
"""Applying a recipe to a dataframe failed."""
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class ValidationFailure(CleanFrameError):
|
|
36
|
+
"""A validation rule failed under a policy that raises (e.g. ``on_fail: error``
|
|
37
|
+
or ``mode="strict"``)."""
|
|
38
|
+
|
|
39
|
+
def __init__(self, message: str, failures: list | None = None) -> None:
|
|
40
|
+
super().__init__(message)
|
|
41
|
+
self.failures = failures or []
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class DriftError(CleanFrameError):
|
|
45
|
+
"""Schema drift was detected under a policy that raises (e.g. ``mode="strict"``)."""
|
|
46
|
+
|
|
47
|
+
def __init__(self, message: str, report=None) -> None: # noqa: ANN001 - avoid import cycle
|
|
48
|
+
super().__init__(message)
|
|
49
|
+
self.report = report
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class SchemaError(CleanFrameError):
|
|
53
|
+
"""A target schema is malformed or cannot be satisfied."""
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class LLMError(CleanFrameError):
|
|
57
|
+
"""The LLM planner could not produce a valid recipe (misconfigured, no key,
|
|
58
|
+
budget exceeded, or invalid output)."""
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class BudgetExceeded(LLMError):
|
|
62
|
+
"""Planning was aborted because it would exceed ``max_tokens_budget``."""
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class OutputError(CleanFrameError):
|
|
66
|
+
"""An output file could not be written (bad path, permissions, or engine)."""
|
cleanframe/executor.py
ADDED
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
"""The executor: replay a recipe on a dataframe, deterministically.
|
|
2
|
+
|
|
3
|
+
This is the "Pandas executes" half of the promise. It runs a recipe in fixed
|
|
4
|
+
phases — column ops → renames → frame ops → validation — tracking column lineage
|
|
5
|
+
and a stable row id throughout so a complete :class:`~cleanframe.diff.CellDiff`
|
|
6
|
+
can be computed at the end. No AI, no network, no randomness: the same recipe and
|
|
7
|
+
the same frame always produce the same output, the same diff, and the same
|
|
8
|
+
quarantine.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import warnings
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
|
|
16
|
+
import pandas as pd
|
|
17
|
+
|
|
18
|
+
from ._util import DEFAULT_MAX_DIFF_CHANGES
|
|
19
|
+
from .diff import CellDiff, compute_diff
|
|
20
|
+
from .errors import CleanFrameWarning, ExecutionError
|
|
21
|
+
from .ops import apply_column_op, apply_frame_op
|
|
22
|
+
from .recipe import Recipe
|
|
23
|
+
from .types import Mode
|
|
24
|
+
from .validate import ValidationResult, apply_validations
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class ExecutionResult:
|
|
29
|
+
"""Everything replaying a recipe produced."""
|
|
30
|
+
|
|
31
|
+
dataframe: pd.DataFrame
|
|
32
|
+
diff: CellDiff
|
|
33
|
+
quarantine: pd.DataFrame = field(default_factory=pd.DataFrame)
|
|
34
|
+
validation_results: list[ValidationResult] = field(default_factory=list)
|
|
35
|
+
lineage: dict[str, str | None] = field(default_factory=dict)
|
|
36
|
+
log: list[str] = field(default_factory=list)
|
|
37
|
+
|
|
38
|
+
@property
|
|
39
|
+
def has_quarantine(self) -> bool:
|
|
40
|
+
return not self.quarantine.empty
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def execute(
|
|
44
|
+
recipe: Recipe,
|
|
45
|
+
df: pd.DataFrame,
|
|
46
|
+
*,
|
|
47
|
+
mode: Mode | str = Mode.REVIEW,
|
|
48
|
+
max_diff_changes: int | None = DEFAULT_MAX_DIFF_CHANGES,
|
|
49
|
+
) -> ExecutionResult:
|
|
50
|
+
"""Apply ``recipe`` to ``df`` and return the cleaned frame plus full lineage.
|
|
51
|
+
|
|
52
|
+
Parameters
|
|
53
|
+
----------
|
|
54
|
+
max_diff_changes:
|
|
55
|
+
Cap on stored cell-level diff entries (default 100_000). Pass ``None`` to
|
|
56
|
+
store every change. Counts remain exact when truncated.
|
|
57
|
+
|
|
58
|
+
Memory model
|
|
59
|
+
------------
|
|
60
|
+
Peak extra allocation is roughly ``input + (op-touched columns snapshot) +
|
|
61
|
+
(up to max_diff_changes CellChange records)`` — NOT two full frames. Only the
|
|
62
|
+
columns an op can change are snapshotted for the diff; pass-through columns are
|
|
63
|
+
read back from the final frame. So a 500 MB frame cleans in well under 2x RAM,
|
|
64
|
+
and the diff detail is bounded by ``max_diff_changes`` regardless of frame size.
|
|
65
|
+
"""
|
|
66
|
+
from ._util import ensure_string_columns
|
|
67
|
+
|
|
68
|
+
mode = Mode.coerce(mode)
|
|
69
|
+
# Stable positional row id (survives renames and row drops for the diff).
|
|
70
|
+
work = ensure_string_columns(df).reset_index(drop=True)
|
|
71
|
+
n_rows_before = int(len(work))
|
|
72
|
+
original_columns = [str(c) for c in work.columns]
|
|
73
|
+
# Snapshot ONLY the columns whose *values* an op can change (column-op sources +
|
|
74
|
+
# validation targets), not the whole frame. Pass-through columns are read back
|
|
75
|
+
# from the final frame in the diff, which roughly halves peak memory on wide or
|
|
76
|
+
# large inputs. This set is complete: the only value-mutating paths are column
|
|
77
|
+
# ops (need a ColumnRecipe with ops) and the ``null`` validation policy.
|
|
78
|
+
rename_reverse = {c.rename_to: c.source for c in recipe.columns if c.rename_to}
|
|
79
|
+
snapshot_sources = {c.source for c in recipe.columns if c.ops}
|
|
80
|
+
for v in recipe.validations:
|
|
81
|
+
snapshot_sources.add(rename_reverse.get(v.column, v.column))
|
|
82
|
+
snapshot_cols = [c for c in original_columns if c in snapshot_sources]
|
|
83
|
+
original = work[snapshot_cols].copy()
|
|
84
|
+
log: list[str] = []
|
|
85
|
+
dropped_rows: list[tuple[int, str]] = []
|
|
86
|
+
nulled: dict[str, int] = {}
|
|
87
|
+
|
|
88
|
+
# source_of[current_column] -> original source column, or None if derived.
|
|
89
|
+
source_of: dict[str, str | None] = {str(c): str(c) for c in work.columns}
|
|
90
|
+
|
|
91
|
+
# -- Phase 1: column ops --------------------------------------------
|
|
92
|
+
for col_recipe in recipe.columns:
|
|
93
|
+
src = col_recipe.source
|
|
94
|
+
if src not in work.columns:
|
|
95
|
+
msg = f"recipe references column {src!r}, which is not in the data"
|
|
96
|
+
if mode is Mode.STRICT:
|
|
97
|
+
raise ExecutionError(msg + " (strict mode)")
|
|
98
|
+
log.append("skipped: " + msg)
|
|
99
|
+
warnings.warn(
|
|
100
|
+
f"CleanFrame: skipped recipe column {src!r} — not present in the data. "
|
|
101
|
+
"Use mode='strict' to fail instead, or re-plan / suggest_update for drift.",
|
|
102
|
+
CleanFrameWarning,
|
|
103
|
+
stacklevel=2,
|
|
104
|
+
)
|
|
105
|
+
continue
|
|
106
|
+
|
|
107
|
+
series = work[src]
|
|
108
|
+
na_before = int(series.isna().sum())
|
|
109
|
+
emitted: dict[str, pd.Series] = {}
|
|
110
|
+
for op in col_recipe.ops:
|
|
111
|
+
result = apply_column_op(op, series)
|
|
112
|
+
series = result.series
|
|
113
|
+
for name, extra in result.emit.items():
|
|
114
|
+
if name in emitted:
|
|
115
|
+
raise ExecutionError(
|
|
116
|
+
f"Column {src!r} emits {name!r} more than once; rename one target."
|
|
117
|
+
)
|
|
118
|
+
emitted[name] = extra
|
|
119
|
+
work[src] = series
|
|
120
|
+
na_after = int(series.isna().sum())
|
|
121
|
+
if na_after > na_before:
|
|
122
|
+
nulled[src] = na_after - na_before
|
|
123
|
+
for name, extra in emitted.items():
|
|
124
|
+
if name in source_of and source_of[name] is None:
|
|
125
|
+
# Another op this run already emitted this derived column — two ops
|
|
126
|
+
# writing the same output name would silently clobber each other
|
|
127
|
+
# (the rename phase already guards this class of collision).
|
|
128
|
+
raise ExecutionError(
|
|
129
|
+
f"Two ops emit the same derived column {name!r}; rename one target."
|
|
130
|
+
)
|
|
131
|
+
# If this emit overwrites a pre-existing column that wasn't in the partial
|
|
132
|
+
# snapshot, capture its still-original value now (before the overwrite) so
|
|
133
|
+
# the clobber is tracked in the diff — robust to any emit op.
|
|
134
|
+
if name in original_columns and name not in original.columns:
|
|
135
|
+
original[name] = work[name].copy()
|
|
136
|
+
work[name] = extra.reindex(work.index)
|
|
137
|
+
if name not in source_of:
|
|
138
|
+
source_of[name] = None # brand-new derived column, no "before"
|
|
139
|
+
# else: `name` is an existing source column being overwritten — keep its
|
|
140
|
+
# lineage so every clobbered cell is tracked in the diff, and an
|
|
141
|
+
# idempotent re-emit of identical values registers as no change.
|
|
142
|
+
log.append(f"{src}: emitted derived column {name!r}")
|
|
143
|
+
|
|
144
|
+
if nulled:
|
|
145
|
+
detail = ", ".join(f"{col} ({n})" for col, n in sorted(nulled.items()))
|
|
146
|
+
msg = f"values became missing because an op could not parse them: {detail}"
|
|
147
|
+
log.append(msg)
|
|
148
|
+
warnings.warn("CleanFrame: " + msg, CleanFrameWarning, stacklevel=2)
|
|
149
|
+
|
|
150
|
+
# -- Phase 2: renames -----------------------------------------------
|
|
151
|
+
rename_map = {
|
|
152
|
+
c.source: c.rename_to
|
|
153
|
+
for c in recipe.columns
|
|
154
|
+
if c.rename_to and c.source in work.columns
|
|
155
|
+
}
|
|
156
|
+
if rename_map:
|
|
157
|
+
targets = list(rename_map.values())
|
|
158
|
+
survivors = [str(c) for c in work.columns if c not in rename_map]
|
|
159
|
+
clash = (set(targets) & set(survivors)) | {t for t in targets if targets.count(t) > 1}
|
|
160
|
+
if clash:
|
|
161
|
+
raise ExecutionError(f"Recipe renames collide on output name(s): {sorted(clash)}")
|
|
162
|
+
work = work.rename(columns=rename_map)
|
|
163
|
+
for src, dst in rename_map.items():
|
|
164
|
+
source_of[dst] = source_of.pop(src, src)
|
|
165
|
+
|
|
166
|
+
# Post-transform values of rows that later get dropped, so a value rewrite on a
|
|
167
|
+
# row that dedup/validation removes is still tracked in the diff (invariant #5).
|
|
168
|
+
# Only the dropped rows are snapshotted, not the whole frame.
|
|
169
|
+
dropped_snaps: list[pd.DataFrame] = []
|
|
170
|
+
|
|
171
|
+
# -- Phase 3: frame ops (dedup, drop_columns, …) --------------------
|
|
172
|
+
for op in recipe.frame_ops:
|
|
173
|
+
prev = work
|
|
174
|
+
work = apply_frame_op(op, prev)
|
|
175
|
+
dropped = set(prev.index) - set(work.index)
|
|
176
|
+
if dropped:
|
|
177
|
+
ids = sorted(dropped)
|
|
178
|
+
for rid in ids:
|
|
179
|
+
dropped_rows.append((int(rid), op.name))
|
|
180
|
+
dropped_snaps.append(prev.loc[ids])
|
|
181
|
+
log.append(f"{op.name}: dropped {len(dropped)} row(s)")
|
|
182
|
+
# a frame op can also remove columns (drop_columns); keep lineage tidy
|
|
183
|
+
for name in list(source_of):
|
|
184
|
+
if name not in work.columns:
|
|
185
|
+
source_of.pop(name, None)
|
|
186
|
+
|
|
187
|
+
# -- Phase 4: validation --------------------------------------------
|
|
188
|
+
pre_validation = work
|
|
189
|
+
outcome = apply_validations(work, recipe.validations, mode)
|
|
190
|
+
work = outcome.dataframe
|
|
191
|
+
if outcome.removed_rows:
|
|
192
|
+
present = [rid for rid, _ in outcome.removed_rows if rid in pre_validation.index]
|
|
193
|
+
if present:
|
|
194
|
+
dropped_snaps.append(pre_validation.loc[present])
|
|
195
|
+
dropped_rows.extend(outcome.removed_rows)
|
|
196
|
+
log.extend(outcome.log)
|
|
197
|
+
|
|
198
|
+
dropped_after = pd.concat(dropped_snaps) if dropped_snaps else None
|
|
199
|
+
|
|
200
|
+
# -- Diff -----------------------------------------------------------
|
|
201
|
+
diff = compute_diff(
|
|
202
|
+
original,
|
|
203
|
+
work,
|
|
204
|
+
source_of,
|
|
205
|
+
dropped_rows=dropped_rows,
|
|
206
|
+
dropped_after=dropped_after,
|
|
207
|
+
original_columns=original_columns,
|
|
208
|
+
n_rows_before=n_rows_before,
|
|
209
|
+
max_changes=max_diff_changes,
|
|
210
|
+
)
|
|
211
|
+
if diff.truncated:
|
|
212
|
+
msg = (
|
|
213
|
+
f"diff detail truncated to {len(diff.changes)} of {diff.changed_cells} "
|
|
214
|
+
"changed cells (raise max_diff_changes or pass None for a full lineage)"
|
|
215
|
+
)
|
|
216
|
+
log.append(msg)
|
|
217
|
+
warnings.warn("CleanFrame: " + msg, CleanFrameWarning, stacklevel=2)
|
|
218
|
+
|
|
219
|
+
return ExecutionResult(
|
|
220
|
+
dataframe=work,
|
|
221
|
+
diff=diff,
|
|
222
|
+
quarantine=outcome.quarantine,
|
|
223
|
+
validation_results=outcome.results,
|
|
224
|
+
lineage=source_of,
|
|
225
|
+
log=log,
|
|
226
|
+
)
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
__all__ = ["execute", "ExecutionResult"]
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""Deterministic hashing and dataframe fingerprints.
|
|
2
|
+
|
|
3
|
+
Everything here is pure and reproducible: the same input always yields the same
|
|
4
|
+
digest, on any machine, in any process. That property is what lets a recipe's
|
|
5
|
+
``source_fingerprint`` be compared reliably months later for drift detection.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import hashlib
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
import pandas as pd
|
|
14
|
+
|
|
15
|
+
#: How many leading rows feed the content hash. Fixed so the fingerprint is
|
|
16
|
+
#: stable regardless of total file size.
|
|
17
|
+
DEFAULT_SAMPLE_ROWS = 200
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _canonical(value: Any) -> str:
|
|
21
|
+
"""Render a single cell to a stable string.
|
|
22
|
+
|
|
23
|
+
``NaN``/``None`` collapse to a single sentinel so that a missing value hashes
|
|
24
|
+
the same whether it arrived as ``float('nan')``, ``None``, or ``pd.NA``.
|
|
25
|
+
"""
|
|
26
|
+
if value is None:
|
|
27
|
+
return "\x00NA\x00"
|
|
28
|
+
# pd.isna raises on array-like; cells are scalars here.
|
|
29
|
+
try:
|
|
30
|
+
if pd.isna(value):
|
|
31
|
+
return "\x00NA\x00"
|
|
32
|
+
except (TypeError, ValueError):
|
|
33
|
+
pass
|
|
34
|
+
if isinstance(value, float):
|
|
35
|
+
# repr(float) is round-trippable and stable across platforms in CPython.
|
|
36
|
+
return repr(value)
|
|
37
|
+
return str(value)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def stable_hash(*parts: Any, length: int | None = None) -> str:
|
|
41
|
+
"""Hash an arbitrary sequence of parts into a hex digest.
|
|
42
|
+
|
|
43
|
+
Parts are joined with a delimiter that cannot appear in :func:`_canonical`
|
|
44
|
+
output, so ``("a", "bc")`` and ``("ab", "c")`` never collide.
|
|
45
|
+
"""
|
|
46
|
+
hasher = hashlib.sha256()
|
|
47
|
+
for part in parts:
|
|
48
|
+
hasher.update(_canonical(part).encode("utf-8"))
|
|
49
|
+
hasher.update(b"\x1f") # unit separator delimiter
|
|
50
|
+
digest = hasher.hexdigest()
|
|
51
|
+
return digest[:length] if length else digest
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def fingerprint_dataframe(df: pd.DataFrame, sample_rows: int = DEFAULT_SAMPLE_ROWS) -> dict:
|
|
55
|
+
"""Build a compact, comparable fingerprint of a dataframe's shape and content.
|
|
56
|
+
|
|
57
|
+
The returned dict is plain JSON/YAML-serialisable and is stored verbatim in a
|
|
58
|
+
recipe's ``source_fingerprint``. It intentionally records *structure* (column
|
|
59
|
+
names, order, dtypes) plus a *content* hash of the leading rows — enough to
|
|
60
|
+
detect drift without embedding the data itself.
|
|
61
|
+
"""
|
|
62
|
+
columns = [str(c) for c in df.columns]
|
|
63
|
+
dtypes = {str(c): str(df[c].dtype) for c in df.columns}
|
|
64
|
+
|
|
65
|
+
head = df.head(sample_rows)
|
|
66
|
+
hasher = hashlib.sha256()
|
|
67
|
+
# Column names participate in the content hash so a rename alone changes it.
|
|
68
|
+
for col in columns:
|
|
69
|
+
hasher.update(col.encode("utf-8"))
|
|
70
|
+
hasher.update(b"\x1e") # record separator
|
|
71
|
+
series = head[col]
|
|
72
|
+
for value in series.tolist():
|
|
73
|
+
hasher.update(_canonical(value).encode("utf-8"))
|
|
74
|
+
hasher.update(b"\x1f")
|
|
75
|
+
|
|
76
|
+
return {
|
|
77
|
+
"columns": len(columns),
|
|
78
|
+
"column_names": columns,
|
|
79
|
+
"dtypes": dtypes,
|
|
80
|
+
"row_count": int(len(df)),
|
|
81
|
+
"sampled_rows": int(len(head)),
|
|
82
|
+
"hash_sample": hasher.hexdigest()[:16],
|
|
83
|
+
}
|