cleanframe-engine 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. cleanframe/__init__.py +169 -0
  2. cleanframe/__main__.py +5 -0
  3. cleanframe/_util.py +438 -0
  4. cleanframe/_version.py +1 -0
  5. cleanframe/api.py +559 -0
  6. cleanframe/cli.py +688 -0
  7. cleanframe/codegen.py +666 -0
  8. cleanframe/dataio.py +506 -0
  9. cleanframe/detectors/__init__.py +43 -0
  10. cleanframe/detectors/base.py +221 -0
  11. cleanframe/detectors/categories.py +199 -0
  12. cleanframe/detectors/contacts.py +105 -0
  13. cleanframe/detectors/currency.py +112 -0
  14. cleanframe/detectors/dates.py +204 -0
  15. cleanframe/detectors/dedup.py +150 -0
  16. cleanframe/detectors/nulls.py +109 -0
  17. cleanframe/detectors/outliers.py +73 -0
  18. cleanframe/detectors/schema_mapping.py +125 -0
  19. cleanframe/detectors/text.py +105 -0
  20. cleanframe/detectors/units.py +86 -0
  21. cleanframe/diff.py +369 -0
  22. cleanframe/drift.py +283 -0
  23. cleanframe/errors.py +66 -0
  24. cleanframe/executor.py +229 -0
  25. cleanframe/fingerprint.py +83 -0
  26. cleanframe/issues.py +186 -0
  27. cleanframe/llm.py +811 -0
  28. cleanframe/ops.py +1245 -0
  29. cleanframe/planner.py +353 -0
  30. cleanframe/profile.py +413 -0
  31. cleanframe/py.typed +1 -0
  32. cleanframe/quality.py +81 -0
  33. cleanframe/readfix.py +160 -0
  34. cleanframe/recipe.py +398 -0
  35. cleanframe/report.py +345 -0
  36. cleanframe/result.py +144 -0
  37. cleanframe/schema.py +259 -0
  38. cleanframe/streaming.py +354 -0
  39. cleanframe/types.py +119 -0
  40. cleanframe/validate.py +363 -0
  41. cleanframe/workbook.py +370 -0
  42. cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
  43. cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
  44. cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
  45. cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
  46. cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
cleanframe/drift.py ADDED
@@ -0,0 +1,283 @@
1
+ """Schema-drift detection: catch the file that came back *different*.
2
+
3
+ Before replaying a recipe on next month's file, compare that file against the
4
+ recipe's ``source_fingerprint`` and its declared expectations. If a column was
5
+ renamed, a new one appeared, a dtype shifted, or values stopped matching the
6
+ formats the recipe parses, CleanFrame **stops and tells you** instead of silently
7
+ producing garbage — exactly the failure mode that makes hand-cleaning fragile.
8
+
9
+ Findings carry fuzzy-match suggestions ("94% match to recipe column …") so the fix
10
+ is obvious. By default ``apply_recipe`` / ``cleanframe apply`` **stop** on drift;
11
+ pass ``on_drift="warn"`` / ``--force`` only when you intentionally want to continue.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ from dataclasses import dataclass, field
17
+
18
+ import pandas as pd
19
+
20
+ from ._util import best_match, canonicalize_dtype, similarity
21
+ from .errors import CleanFrameError
22
+ from .fingerprint import DEFAULT_SAMPLE_ROWS, fingerprint_dataframe
23
+ from .llm import _sketch
24
+ from .ops import parse_dates_to_datetime
25
+ from .recipe import Recipe
26
+ from .types import Severity
27
+
28
+ _RENAME_CONFIDENCE = 0.6
29
+
30
+
31
+ @dataclass
32
+ class DriftFinding:
33
+ kind: str
34
+ message: str
35
+ severity: Severity = Severity.WARNING
36
+ column: str | None = None
37
+ suggestion: str | None = None
38
+ evidence: dict = field(default_factory=dict)
39
+
40
+
41
+ @dataclass
42
+ class DriftReport:
43
+ findings: list[DriftFinding] = field(default_factory=list)
44
+ source: str | None = None
45
+
46
+ @property
47
+ def has_drift(self) -> bool:
48
+ return any(f.severity.rank >= Severity.WARNING.rank for f in self.findings)
49
+
50
+ @property
51
+ def worst(self) -> Severity | None:
52
+ if not self.findings:
53
+ return None
54
+ return max((f.severity for f in self.findings), key=lambda s: s.rank)
55
+
56
+ def by_kind(self, kind: str) -> list[DriftFinding]:
57
+ return [f for f in self.findings if f.kind == kind]
58
+
59
+ def render(self) -> str:
60
+ if not self.findings:
61
+ where = f" in {self.source}" if self.source else ""
62
+ return f"✓ No schema drift detected{where}."
63
+ where = f" in {self.source}" if self.source else ""
64
+ lines = [f"⚠ Schema drift detected{where}"]
65
+ for f in self.findings:
66
+ lines.append(f" • {f.message}")
67
+ if f.suggestion:
68
+ lines.append(f" ↳ {f.suggestion}")
69
+ lines.append(
70
+ " Run `cleanframe suggest <file> --recipe <recipe.yaml> --update` to review a patch."
71
+ )
72
+ return "\n".join(lines)
73
+
74
+ def __repr__(self) -> str: # pragma: no cover - cosmetic
75
+ return f"DriftReport({len(self.findings)} finding(s), has_drift={self.has_drift})"
76
+
77
+
78
+ def detect_drift(df: pd.DataFrame, recipe: Recipe, *, source: str | None = None) -> DriftReport:
79
+ """Compare an incoming frame against the expectations baked into ``recipe``."""
80
+ report = DriftReport(source=source)
81
+ fp = recipe.source_fingerprint or {}
82
+ if not isinstance(fp, dict):
83
+ raise CleanFrameError(
84
+ f"recipe source_fingerprint must be a mapping, got {type(fp).__name__}."
85
+ )
86
+ raw_cols = fp.get("column_names") or []
87
+ if not isinstance(raw_cols, (list, tuple)):
88
+ raise CleanFrameError("recipe source_fingerprint.column_names must be a list.")
89
+ raw_dtypes = fp.get("dtypes") or {}
90
+ if not isinstance(raw_dtypes, dict):
91
+ raise CleanFrameError("recipe source_fingerprint.dtypes must be a mapping.")
92
+ expected_cols: list[str] = [str(c) for c in raw_cols]
93
+ expected_dtypes: dict[str, str] = {str(k): str(v) for k, v in raw_dtypes.items()}
94
+ actual_cols = [str(c) for c in df.columns]
95
+
96
+ # A fingerprint without column names (hand-written recipe) says nothing about
97
+ # structure; comparing against an empty list would call every column new.
98
+ missing = [c for c in expected_cols if c not in actual_cols] if expected_cols else []
99
+ new = [c for c in actual_cols if c not in expected_cols] if expected_cols else []
100
+
101
+ # -- new columns: try to explain each as a rename ------------------
102
+ for col in new:
103
+ candidate, score = best_match(col, missing + [c.output_name for c in recipe.columns])
104
+ if candidate and score >= _RENAME_CONFIDENCE:
105
+ report.findings.append(
106
+ DriftFinding(
107
+ kind="renamed_column",
108
+ message=f'Column "{col}" is new — {score:.0%} match to recipe column "{candidate}"',
109
+ severity=Severity.WARNING,
110
+ column=col,
111
+ suggestion=f'If "{col}" replaces "{candidate}", update the recipe to map it.',
112
+ evidence={"match": candidate, "score": round(score, 3)},
113
+ )
114
+ )
115
+ else:
116
+ report.findings.append(
117
+ DriftFinding(
118
+ kind="new_column",
119
+ message=f'Column "{col}" is new and unrecognised',
120
+ severity=Severity.WARNING,
121
+ column=col,
122
+ suggestion="Confirm the new column is expected, or drop/ignore it before apply.",
123
+ evidence={},
124
+ )
125
+ )
126
+
127
+ # -- missing columns not explained by a rename --------------------
128
+ explained = {f.evidence.get("match") for f in report.by_kind("renamed_column")}
129
+ for col in missing:
130
+ if col in explained:
131
+ continue
132
+ report.findings.append(
133
+ DriftFinding(
134
+ kind="missing_column",
135
+ message=f'Column "{col}" expected by the recipe is missing',
136
+ severity=Severity.WARNING,
137
+ column=col,
138
+ suggestion=_closest_hint(col, new),
139
+ evidence={},
140
+ )
141
+ )
142
+
143
+ # -- dtype changes (canonical families — object vs str is not drift) -
144
+ for col in actual_cols:
145
+ if col not in expected_dtypes:
146
+ continue
147
+ was = expected_dtypes[col]
148
+ now = str(df[col].dtype)
149
+ if canonicalize_dtype(was) == canonicalize_dtype(now):
150
+ continue
151
+ report.findings.append(
152
+ DriftFinding(
153
+ kind="dtype_change",
154
+ message=f'Column "{col}" changed dtype ({was} → {now})',
155
+ severity=Severity.WARNING,
156
+ column=col,
157
+ suggestion="Re-plan or update casts if ops assume the old dtype.",
158
+ evidence={"was": was, "now": now},
159
+ )
160
+ )
161
+
162
+ # -- content fingerprint (row_count / hash_sample) ------------------
163
+ _detect_content_drift(df, fp, report)
164
+
165
+ # -- value-level format drift (declared parse_date formats) --------
166
+ _detect_format_drift(df, recipe, report)
167
+
168
+ return report
169
+
170
+
171
+ def _detect_content_drift(df: pd.DataFrame, fp: dict, report: DriftReport) -> None:
172
+ """Compare stored content fingerprint fields when present.
173
+
174
+ Row-count and sample-hash changes are expected for monthly files, so they are
175
+ INFO (visible in the report, do not stop apply). Schema / dtype / format
176
+ findings remain the apply-blocking signals. Same row count + different hash is
177
+ called out in the message so operators can spot a possible wrong-file swap.
178
+ """
179
+ expected_rows = fp.get("row_count")
180
+ expected_hash = fp.get("hash_sample")
181
+ if expected_rows is None and expected_hash is None:
182
+ return
183
+
184
+ try:
185
+ sample_rows = int(fp.get("sampled_rows") or DEFAULT_SAMPLE_ROWS)
186
+ except (TypeError, ValueError):
187
+ sample_rows = DEFAULT_SAMPLE_ROWS
188
+ actual_fp = fingerprint_dataframe(df, sample_rows=sample_rows)
189
+ actual_rows = actual_fp["row_count"]
190
+ actual_hash = actual_fp["hash_sample"]
191
+
192
+ try:
193
+ rows_changed = expected_rows is not None and int(expected_rows) != int(actual_rows)
194
+ except (TypeError, ValueError):
195
+ rows_changed = False
196
+ expected_rows = None
197
+ hash_changed = expected_hash is not None and str(expected_hash) != str(actual_hash)
198
+
199
+ if rows_changed:
200
+ report.findings.append(
201
+ DriftFinding(
202
+ kind="row_count_change",
203
+ message=f"Row count changed ({expected_rows} → {actual_rows})",
204
+ severity=Severity.INFO,
205
+ evidence={"was": int(expected_rows), "now": int(actual_rows)}, # type: ignore[arg-type]
206
+ )
207
+ )
208
+
209
+ if hash_changed:
210
+ same_n = bool(expected_rows is not None and int(expected_rows) == int(actual_rows))
211
+ report.findings.append(
212
+ DriftFinding(
213
+ kind="content_hash_change",
214
+ message=(
215
+ "Leading-row content hash changed "
216
+ f"({expected_hash} → {actual_hash})"
217
+ ),
218
+ severity=Severity.INFO,
219
+ suggestion=(
220
+ "Same row count but different sample values — confirm this is "
221
+ "the intended file (not a schema-matched swap)."
222
+ if same_n
223
+ else None
224
+ ),
225
+ evidence={"was": str(expected_hash), "now": str(actual_hash)},
226
+ )
227
+ )
228
+
229
+
230
+ def _detect_format_drift(df: pd.DataFrame, recipe: Recipe, report: DriftReport) -> None:
231
+ for col_recipe in recipe.columns:
232
+ src = col_recipe.source
233
+ if src not in df.columns:
234
+ continue
235
+ for op in col_recipe.ops:
236
+ if op.name != "parse_date":
237
+ continue
238
+ formats = op.params.get("formats")
239
+ if not formats:
240
+ continue
241
+ series = df[src].dropna()
242
+ if series.empty:
243
+ continue
244
+ parsed = parse_dates_to_datetime(
245
+ series,
246
+ list(formats),
247
+ dayfirst=bool(op.params.get("dayfirst", False)),
248
+ yearfirst=bool(op.params.get("yearfirst", False)),
249
+ )
250
+ unmatched = series[parsed.isna()]
251
+ if len(unmatched):
252
+ examples = [str(v) for v in unmatched.head(3).tolist()]
253
+ sketches = sorted({_sketch(e) for e in examples})
254
+ report.findings.append(
255
+ DriftFinding(
256
+ kind="format_drift",
257
+ message=(
258
+ f'{len(unmatched)} value(s) in "{src}" match no allowed date format '
259
+ f"(new: {examples[0]!r})"
260
+ ),
261
+ severity=Severity.WARNING,
262
+ column=src,
263
+ suggestion=f"New pattern {sketches[0]!r}; add its format to the recipe.",
264
+ evidence={
265
+ "count": int(len(unmatched)),
266
+ "examples": examples,
267
+ "sketches": sketches,
268
+ },
269
+ )
270
+ )
271
+
272
+
273
+ def _closest_hint(col: str, candidates: list[str]) -> str | None:
274
+ if not candidates:
275
+ return None
276
+ best = max(candidates, key=lambda c: similarity(col, c))
277
+ score = similarity(col, best)
278
+ if score >= 0.4:
279
+ return f'Did a new column "{best}" ({score:.0%} match) replace it?'
280
+ return None
281
+
282
+
283
+ __all__ = ["detect_drift", "DriftReport", "DriftFinding"]
cleanframe/errors.py ADDED
@@ -0,0 +1,66 @@
1
+ """Exception hierarchy for CleanFrame.
2
+
3
+ Every error CleanFrame raises deliberately derives from :class:`CleanFrameError`,
4
+ so callers can ``except cleanframe.CleanFrameError`` to catch anything the library
5
+ raises on purpose without also swallowing unrelated bugs.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+
11
+ class CleanFrameError(Exception):
12
+ """Base class for all errors raised deliberately by CleanFrame."""
13
+
14
+
15
+ class CleanFrameWarning(UserWarning):
16
+ """Category for every warning CleanFrame raises deliberately.
17
+
18
+ Silence the library's advisory output with
19
+ ``warnings.simplefilter("ignore", cleanframe.CleanFrameWarning)``.
20
+ """
21
+
22
+
23
+ class RecipeError(CleanFrameError):
24
+ """A recipe is malformed, references an unknown op, or fails to load."""
25
+
26
+
27
+ class OpError(CleanFrameError):
28
+ """An op could not be applied (bad params, or a runtime failure in pandas)."""
29
+
30
+
31
+ class ExecutionError(CleanFrameError):
32
+ """Applying a recipe to a dataframe failed."""
33
+
34
+
35
+ class ValidationFailure(CleanFrameError):
36
+ """A validation rule failed under a policy that raises (e.g. ``on_fail: error``
37
+ or ``mode="strict"``)."""
38
+
39
+ def __init__(self, message: str, failures: list | None = None) -> None:
40
+ super().__init__(message)
41
+ self.failures = failures or []
42
+
43
+
44
+ class DriftError(CleanFrameError):
45
+ """Schema drift was detected under a policy that raises (e.g. ``mode="strict"``)."""
46
+
47
+ def __init__(self, message: str, report=None) -> None: # noqa: ANN001 - avoid import cycle
48
+ super().__init__(message)
49
+ self.report = report
50
+
51
+
52
+ class SchemaError(CleanFrameError):
53
+ """A target schema is malformed or cannot be satisfied."""
54
+
55
+
56
+ class LLMError(CleanFrameError):
57
+ """The LLM planner could not produce a valid recipe (misconfigured, no key,
58
+ budget exceeded, or invalid output)."""
59
+
60
+
61
+ class BudgetExceeded(LLMError):
62
+ """Planning was aborted because it would exceed ``max_tokens_budget``."""
63
+
64
+
65
+ class OutputError(CleanFrameError):
66
+ """An output file could not be written (bad path, permissions, or engine)."""
cleanframe/executor.py ADDED
@@ -0,0 +1,229 @@
1
+ """The executor: replay a recipe on a dataframe, deterministically.
2
+
3
+ This is the "Pandas executes" half of the promise. It runs a recipe in fixed
4
+ phases — column ops → renames → frame ops → validation — tracking column lineage
5
+ and a stable row id throughout so a complete :class:`~cleanframe.diff.CellDiff`
6
+ can be computed at the end. No AI, no network, no randomness: the same recipe and
7
+ the same frame always produce the same output, the same diff, and the same
8
+ quarantine.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import warnings
14
+ from dataclasses import dataclass, field
15
+
16
+ import pandas as pd
17
+
18
+ from ._util import DEFAULT_MAX_DIFF_CHANGES
19
+ from .diff import CellDiff, compute_diff
20
+ from .errors import CleanFrameWarning, ExecutionError
21
+ from .ops import apply_column_op, apply_frame_op
22
+ from .recipe import Recipe
23
+ from .types import Mode
24
+ from .validate import ValidationResult, apply_validations
25
+
26
+
27
+ @dataclass
28
+ class ExecutionResult:
29
+ """Everything replaying a recipe produced."""
30
+
31
+ dataframe: pd.DataFrame
32
+ diff: CellDiff
33
+ quarantine: pd.DataFrame = field(default_factory=pd.DataFrame)
34
+ validation_results: list[ValidationResult] = field(default_factory=list)
35
+ lineage: dict[str, str | None] = field(default_factory=dict)
36
+ log: list[str] = field(default_factory=list)
37
+
38
+ @property
39
+ def has_quarantine(self) -> bool:
40
+ return not self.quarantine.empty
41
+
42
+
43
+ def execute(
44
+ recipe: Recipe,
45
+ df: pd.DataFrame,
46
+ *,
47
+ mode: Mode | str = Mode.REVIEW,
48
+ max_diff_changes: int | None = DEFAULT_MAX_DIFF_CHANGES,
49
+ ) -> ExecutionResult:
50
+ """Apply ``recipe`` to ``df`` and return the cleaned frame plus full lineage.
51
+
52
+ Parameters
53
+ ----------
54
+ max_diff_changes:
55
+ Cap on stored cell-level diff entries (default 100_000). Pass ``None`` to
56
+ store every change. Counts remain exact when truncated.
57
+
58
+ Memory model
59
+ ------------
60
+ Peak extra allocation is roughly ``input + (op-touched columns snapshot) +
61
+ (up to max_diff_changes CellChange records)`` — NOT two full frames. Only the
62
+ columns an op can change are snapshotted for the diff; pass-through columns are
63
+ read back from the final frame. So a 500 MB frame cleans in well under 2x RAM,
64
+ and the diff detail is bounded by ``max_diff_changes`` regardless of frame size.
65
+ """
66
+ from ._util import ensure_string_columns
67
+
68
+ mode = Mode.coerce(mode)
69
+ # Stable positional row id (survives renames and row drops for the diff).
70
+ work = ensure_string_columns(df).reset_index(drop=True)
71
+ n_rows_before = int(len(work))
72
+ original_columns = [str(c) for c in work.columns]
73
+ # Snapshot ONLY the columns whose *values* an op can change (column-op sources +
74
+ # validation targets), not the whole frame. Pass-through columns are read back
75
+ # from the final frame in the diff, which roughly halves peak memory on wide or
76
+ # large inputs. This set is complete: the only value-mutating paths are column
77
+ # ops (need a ColumnRecipe with ops) and the ``null`` validation policy.
78
+ rename_reverse = {c.rename_to: c.source for c in recipe.columns if c.rename_to}
79
+ snapshot_sources = {c.source for c in recipe.columns if c.ops}
80
+ for v in recipe.validations:
81
+ snapshot_sources.add(rename_reverse.get(v.column, v.column))
82
+ snapshot_cols = [c for c in original_columns if c in snapshot_sources]
83
+ original = work[snapshot_cols].copy()
84
+ log: list[str] = []
85
+ dropped_rows: list[tuple[int, str]] = []
86
+ nulled: dict[str, int] = {}
87
+
88
+ # source_of[current_column] -> original source column, or None if derived.
89
+ source_of: dict[str, str | None] = {str(c): str(c) for c in work.columns}
90
+
91
+ # -- Phase 1: column ops --------------------------------------------
92
+ for col_recipe in recipe.columns:
93
+ src = col_recipe.source
94
+ if src not in work.columns:
95
+ msg = f"recipe references column {src!r}, which is not in the data"
96
+ if mode is Mode.STRICT:
97
+ raise ExecutionError(msg + " (strict mode)")
98
+ log.append("skipped: " + msg)
99
+ warnings.warn(
100
+ f"CleanFrame: skipped recipe column {src!r} — not present in the data. "
101
+ "Use mode='strict' to fail instead, or re-plan / suggest_update for drift.",
102
+ CleanFrameWarning,
103
+ stacklevel=2,
104
+ )
105
+ continue
106
+
107
+ series = work[src]
108
+ na_before = int(series.isna().sum())
109
+ emitted: dict[str, pd.Series] = {}
110
+ for op in col_recipe.ops:
111
+ result = apply_column_op(op, series)
112
+ series = result.series
113
+ for name, extra in result.emit.items():
114
+ if name in emitted:
115
+ raise ExecutionError(
116
+ f"Column {src!r} emits {name!r} more than once; rename one target."
117
+ )
118
+ emitted[name] = extra
119
+ work[src] = series
120
+ na_after = int(series.isna().sum())
121
+ if na_after > na_before:
122
+ nulled[src] = na_after - na_before
123
+ for name, extra in emitted.items():
124
+ if name in source_of and source_of[name] is None:
125
+ # Another op this run already emitted this derived column — two ops
126
+ # writing the same output name would silently clobber each other
127
+ # (the rename phase already guards this class of collision).
128
+ raise ExecutionError(
129
+ f"Two ops emit the same derived column {name!r}; rename one target."
130
+ )
131
+ # If this emit overwrites a pre-existing column that wasn't in the partial
132
+ # snapshot, capture its still-original value now (before the overwrite) so
133
+ # the clobber is tracked in the diff — robust to any emit op.
134
+ if name in original_columns and name not in original.columns:
135
+ original[name] = work[name].copy()
136
+ work[name] = extra.reindex(work.index)
137
+ if name not in source_of:
138
+ source_of[name] = None # brand-new derived column, no "before"
139
+ # else: `name` is an existing source column being overwritten — keep its
140
+ # lineage so every clobbered cell is tracked in the diff, and an
141
+ # idempotent re-emit of identical values registers as no change.
142
+ log.append(f"{src}: emitted derived column {name!r}")
143
+
144
+ if nulled:
145
+ detail = ", ".join(f"{col} ({n})" for col, n in sorted(nulled.items()))
146
+ msg = f"values became missing because an op could not parse them: {detail}"
147
+ log.append(msg)
148
+ warnings.warn("CleanFrame: " + msg, CleanFrameWarning, stacklevel=2)
149
+
150
+ # -- Phase 2: renames -----------------------------------------------
151
+ rename_map = {
152
+ c.source: c.rename_to
153
+ for c in recipe.columns
154
+ if c.rename_to and c.source in work.columns
155
+ }
156
+ if rename_map:
157
+ targets = list(rename_map.values())
158
+ survivors = [str(c) for c in work.columns if c not in rename_map]
159
+ clash = (set(targets) & set(survivors)) | {t for t in targets if targets.count(t) > 1}
160
+ if clash:
161
+ raise ExecutionError(f"Recipe renames collide on output name(s): {sorted(clash)}")
162
+ work = work.rename(columns=rename_map)
163
+ for src, dst in rename_map.items():
164
+ source_of[dst] = source_of.pop(src, src)
165
+
166
+ # Post-transform values of rows that later get dropped, so a value rewrite on a
167
+ # row that dedup/validation removes is still tracked in the diff (invariant #5).
168
+ # Only the dropped rows are snapshotted, not the whole frame.
169
+ dropped_snaps: list[pd.DataFrame] = []
170
+
171
+ # -- Phase 3: frame ops (dedup, drop_columns, …) --------------------
172
+ for op in recipe.frame_ops:
173
+ prev = work
174
+ work = apply_frame_op(op, prev)
175
+ dropped = set(prev.index) - set(work.index)
176
+ if dropped:
177
+ ids = sorted(dropped)
178
+ for rid in ids:
179
+ dropped_rows.append((int(rid), op.name))
180
+ dropped_snaps.append(prev.loc[ids])
181
+ log.append(f"{op.name}: dropped {len(dropped)} row(s)")
182
+ # a frame op can also remove columns (drop_columns); keep lineage tidy
183
+ for name in list(source_of):
184
+ if name not in work.columns:
185
+ source_of.pop(name, None)
186
+
187
+ # -- Phase 4: validation --------------------------------------------
188
+ pre_validation = work
189
+ outcome = apply_validations(work, recipe.validations, mode)
190
+ work = outcome.dataframe
191
+ if outcome.removed_rows:
192
+ present = [rid for rid, _ in outcome.removed_rows if rid in pre_validation.index]
193
+ if present:
194
+ dropped_snaps.append(pre_validation.loc[present])
195
+ dropped_rows.extend(outcome.removed_rows)
196
+ log.extend(outcome.log)
197
+
198
+ dropped_after = pd.concat(dropped_snaps) if dropped_snaps else None
199
+
200
+ # -- Diff -----------------------------------------------------------
201
+ diff = compute_diff(
202
+ original,
203
+ work,
204
+ source_of,
205
+ dropped_rows=dropped_rows,
206
+ dropped_after=dropped_after,
207
+ original_columns=original_columns,
208
+ n_rows_before=n_rows_before,
209
+ max_changes=max_diff_changes,
210
+ )
211
+ if diff.truncated:
212
+ msg = (
213
+ f"diff detail truncated to {len(diff.changes)} of {diff.changed_cells} "
214
+ "changed cells (raise max_diff_changes or pass None for a full lineage)"
215
+ )
216
+ log.append(msg)
217
+ warnings.warn("CleanFrame: " + msg, CleanFrameWarning, stacklevel=2)
218
+
219
+ return ExecutionResult(
220
+ dataframe=work,
221
+ diff=diff,
222
+ quarantine=outcome.quarantine,
223
+ validation_results=outcome.results,
224
+ lineage=source_of,
225
+ log=log,
226
+ )
227
+
228
+
229
+ __all__ = ["execute", "ExecutionResult"]
@@ -0,0 +1,83 @@
1
+ """Deterministic hashing and dataframe fingerprints.
2
+
3
+ Everything here is pure and reproducible: the same input always yields the same
4
+ digest, on any machine, in any process. That property is what lets a recipe's
5
+ ``source_fingerprint`` be compared reliably months later for drift detection.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import hashlib
11
+ from typing import Any
12
+
13
+ import pandas as pd
14
+
15
+ #: How many leading rows feed the content hash. Fixed so the fingerprint is
16
+ #: stable regardless of total file size.
17
+ DEFAULT_SAMPLE_ROWS = 200
18
+
19
+
20
+ def _canonical(value: Any) -> str:
21
+ """Render a single cell to a stable string.
22
+
23
+ ``NaN``/``None`` collapse to a single sentinel so that a missing value hashes
24
+ the same whether it arrived as ``float('nan')``, ``None``, or ``pd.NA``.
25
+ """
26
+ if value is None:
27
+ return "\x00NA\x00"
28
+ # pd.isna raises on array-like; cells are scalars here.
29
+ try:
30
+ if pd.isna(value):
31
+ return "\x00NA\x00"
32
+ except (TypeError, ValueError):
33
+ pass
34
+ if isinstance(value, float):
35
+ # repr(float) is round-trippable and stable across platforms in CPython.
36
+ return repr(value)
37
+ return str(value)
38
+
39
+
40
+ def stable_hash(*parts: Any, length: int | None = None) -> str:
41
+ """Hash an arbitrary sequence of parts into a hex digest.
42
+
43
+ Parts are joined with a delimiter that cannot appear in :func:`_canonical`
44
+ output, so ``("a", "bc")`` and ``("ab", "c")`` never collide.
45
+ """
46
+ hasher = hashlib.sha256()
47
+ for part in parts:
48
+ hasher.update(_canonical(part).encode("utf-8"))
49
+ hasher.update(b"\x1f") # unit separator delimiter
50
+ digest = hasher.hexdigest()
51
+ return digest[:length] if length else digest
52
+
53
+
54
+ def fingerprint_dataframe(df: pd.DataFrame, sample_rows: int = DEFAULT_SAMPLE_ROWS) -> dict:
55
+ """Build a compact, comparable fingerprint of a dataframe's shape and content.
56
+
57
+ The returned dict is plain JSON/YAML-serialisable and is stored verbatim in a
58
+ recipe's ``source_fingerprint``. It intentionally records *structure* (column
59
+ names, order, dtypes) plus a *content* hash of the leading rows — enough to
60
+ detect drift without embedding the data itself.
61
+ """
62
+ columns = [str(c) for c in df.columns]
63
+ dtypes = {str(c): str(df[c].dtype) for c in df.columns}
64
+
65
+ head = df.head(sample_rows)
66
+ hasher = hashlib.sha256()
67
+ # Column names participate in the content hash so a rename alone changes it.
68
+ for col in columns:
69
+ hasher.update(col.encode("utf-8"))
70
+ hasher.update(b"\x1e") # record separator
71
+ series = head[col]
72
+ for value in series.tolist():
73
+ hasher.update(_canonical(value).encode("utf-8"))
74
+ hasher.update(b"\x1f")
75
+
76
+ return {
77
+ "columns": len(columns),
78
+ "column_names": columns,
79
+ "dtypes": dtypes,
80
+ "row_count": int(len(df)),
81
+ "sampled_rows": int(len(head)),
82
+ "hash_sample": hasher.hexdigest()[:16],
83
+ }