cleanframe-engine 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. cleanframe/__init__.py +169 -0
  2. cleanframe/__main__.py +5 -0
  3. cleanframe/_util.py +438 -0
  4. cleanframe/_version.py +1 -0
  5. cleanframe/api.py +559 -0
  6. cleanframe/cli.py +688 -0
  7. cleanframe/codegen.py +666 -0
  8. cleanframe/dataio.py +506 -0
  9. cleanframe/detectors/__init__.py +43 -0
  10. cleanframe/detectors/base.py +221 -0
  11. cleanframe/detectors/categories.py +199 -0
  12. cleanframe/detectors/contacts.py +105 -0
  13. cleanframe/detectors/currency.py +112 -0
  14. cleanframe/detectors/dates.py +204 -0
  15. cleanframe/detectors/dedup.py +150 -0
  16. cleanframe/detectors/nulls.py +109 -0
  17. cleanframe/detectors/outliers.py +73 -0
  18. cleanframe/detectors/schema_mapping.py +125 -0
  19. cleanframe/detectors/text.py +105 -0
  20. cleanframe/detectors/units.py +86 -0
  21. cleanframe/diff.py +369 -0
  22. cleanframe/drift.py +283 -0
  23. cleanframe/errors.py +66 -0
  24. cleanframe/executor.py +229 -0
  25. cleanframe/fingerprint.py +83 -0
  26. cleanframe/issues.py +186 -0
  27. cleanframe/llm.py +811 -0
  28. cleanframe/ops.py +1245 -0
  29. cleanframe/planner.py +353 -0
  30. cleanframe/profile.py +413 -0
  31. cleanframe/py.typed +1 -0
  32. cleanframe/quality.py +81 -0
  33. cleanframe/readfix.py +160 -0
  34. cleanframe/recipe.py +398 -0
  35. cleanframe/report.py +345 -0
  36. cleanframe/result.py +144 -0
  37. cleanframe/schema.py +259 -0
  38. cleanframe/streaming.py +354 -0
  39. cleanframe/types.py +119 -0
  40. cleanframe/validate.py +363 -0
  41. cleanframe/workbook.py +370 -0
  42. cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
  43. cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
  44. cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
  45. cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
  46. cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,105 @@
1
+ """Text hygiene: whitespace and casing.
2
+
3
+ These are the safest, most universal fixes, so they run first (low priority
4
+ number). Whitespace trimming is non-lossy and always proposed when present.
5
+ Title-casing is opinionated (it would mangle ``"McDonald"``), so it is proposed
6
+ only for name-like columns and at a confidence low enough that ``strict`` mode
7
+ declines it.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import re
13
+
14
+ import pandas as pd
15
+
16
+ from .._util import is_string_like, sample_non_null
17
+ from ..issues import Issues, _cap_examples
18
+ from ..profile import _name_hint
19
+ from ..types import Op, Severity
20
+ from .base import DetectorContext, detector
21
+
22
+ _DOUBLE_WS = re.compile(r"\s{2,}")
23
+ _ALPHAWORDS = re.compile(r"^[A-Za-z][A-Za-z.'\- ]*$")
24
+
25
+
26
+ @detector("whitespace", priority=10)
27
+ def detect_whitespace(series: pd.Series, ctx: DetectorContext) -> Issues:
28
+ """Flag leading/trailing or repeated internal whitespace and propose a trim."""
29
+ issues = Issues()
30
+ cp = ctx.column_profile
31
+ if cp is None or cp.count == 0 or not is_string_like(series):
32
+ return issues
33
+
34
+ strings = [v for v in sample_non_null(series) if isinstance(v, str)]
35
+ trailing = [v for v in strings if v != v.strip()]
36
+ doubles = [v for v in strings if _DOUBLE_WS.search(v)]
37
+ if not trailing and not doubles:
38
+ return issues
39
+
40
+ # collapse_whitespace subsumes a plain strip, so pick one op, not both.
41
+ if doubles:
42
+ op = Op("collapse_whitespace")
43
+ msg = f"{len(set(trailing) | set(doubles))} value(s) have irregular whitespace"
44
+ else:
45
+ op = Op("strip_whitespace")
46
+ msg = f"{len(trailing)} value(s) have leading/trailing whitespace"
47
+
48
+ issues.add(
49
+ "whitespace",
50
+ msg,
51
+ severity=Severity.WARNING,
52
+ confidence=1.0,
53
+ evidence={
54
+ "leading_trailing": len(trailing),
55
+ "internal_doubles": len(doubles),
56
+ "examples": _cap_examples([repr(v) for v in (doubles or trailing)]),
57
+ },
58
+ ops=[op],
59
+ )
60
+ return issues
61
+
62
+
63
+ @detector("text_case", priority=60)
64
+ def detect_text_case(series: pd.Series, ctx: DetectorContext) -> Issues:
65
+ """Propose title-casing for inconsistently-cased *name* columns (low confidence)."""
66
+ issues = Issues()
67
+ cp = ctx.column_profile
68
+ if cp is None or cp.count == 0 or cp.semantic_type != "text":
69
+ return issues
70
+ if not _name_hint(ctx.column or "", "date") and not _looks_like_name_column(ctx.column, series):
71
+ return issues
72
+
73
+ strings = [v.strip() for v in sample_non_null(series) if isinstance(v, str)]
74
+ if not strings:
75
+ return issues
76
+ inconsistent = [v for v in strings if v != v.title()]
77
+ frac = len(inconsistent) / len(strings)
78
+ if frac < 0.25:
79
+ return issues
80
+
81
+ issues.add(
82
+ "inconsistent_case",
83
+ f"{len(inconsistent)} value(s) are not in Title Case",
84
+ severity=Severity.INFO,
85
+ confidence=0.55,
86
+ evidence={"count": len(inconsistent), "examples": _cap_examples(inconsistent)},
87
+ ops=[Op("title_case")],
88
+ )
89
+ return issues
90
+
91
+
92
+ def _looks_like_name_column(column: str | None, series: pd.Series) -> bool:
93
+ if column is None:
94
+ return False
95
+ hinted = any(h in str(column).lower() for h in ("name", "city", "state", "country", "title"))
96
+ if not hinted:
97
+ return False
98
+ strings = [v for v in series.dropna().head(50).tolist() if isinstance(v, str)]
99
+ if not strings:
100
+ return False
101
+ alpha = sum(1 for v in strings if _ALPHAWORDS.match(v.strip()))
102
+ return alpha / len(strings) >= 0.8
103
+
104
+
105
+ __all__ = ["detect_whitespace", "detect_text_case"]
@@ -0,0 +1,86 @@
1
+ """Unit-quantity detection: ``5kg`` / ``5000 g`` / ``5 KG`` → one canonical unit."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections import Counter
6
+
7
+ import pandas as pd
8
+
9
+ from .._util import sample_non_null
10
+ from ..issues import Issues, _cap_examples
11
+ from ..ops import UNIT_FAMILIES, parse_unit_scalar
12
+ from ..types import Op, Severity
13
+ from .base import DetectorContext, detector
14
+
15
+ # Preferred target unit per family (SI-ish, matches README examples for mass).
16
+ _PREFERRED_TARGET = {"mass": "g", "length": "m", "volume": "l"}
17
+
18
+
19
+ @detector("units", priority=46)
20
+ def detect_units(series: pd.Series, ctx: DetectorContext) -> Issues:
21
+ """Detect mixed unit strings and propose ``normalize_unit`` to a single base."""
22
+ issues = Issues()
23
+ cp = ctx.column_profile
24
+ if cp is None or cp.count == 0:
25
+ return issues
26
+ if cp.semantic_type not in ("unit", "text", "categorical"):
27
+ return issues
28
+
29
+ parsed: list[tuple[float, str]] = []
30
+ raw_examples: list[str] = []
31
+ sample = sample_non_null(series)
32
+ for v in sample:
33
+ p = parse_unit_scalar(v)
34
+ if p is None:
35
+ continue
36
+ parsed.append(p)
37
+ raw_examples.append(str(v))
38
+
39
+ if len(parsed) < max(2, int(0.5 * max(len(sample), 1))):
40
+ return issues
41
+
42
+ families: Counter[str] = Counter()
43
+ units: Counter[str] = Counter()
44
+ for _, unit in parsed:
45
+ for fam, table in UNIT_FAMILIES.items():
46
+ if unit in table:
47
+ families[fam] += 1
48
+ units[unit] += 1
49
+ break
50
+
51
+ if not families:
52
+ return issues
53
+ family, _ = families.most_common(1)[0]
54
+ distinct = sorted(u for u, _ in units.most_common() if u in UNIT_FAMILIES[family])
55
+ if len(distinct) < 1:
56
+ return issues
57
+
58
+ mixed = len(distinct) > 1
59
+ # When units are mixed, normalize to the family SI base (matches README:
60
+ # 5kg / 5000 g → grams). A single unit keeps that unit.
61
+ if mixed:
62
+ target = _PREFERRED_TARGET[family]
63
+ else:
64
+ target = distinct[0] if distinct else _PREFERRED_TARGET[family]
65
+
66
+ issues.add(
67
+ "mixed_units" if mixed else "unit_format",
68
+ (
69
+ f"Unit quantities stored as text "
70
+ f"({'mixed units ' + str(distinct) if mixed else 'unit ' + distinct[0]}); "
71
+ f"normalize to {target}"
72
+ ),
73
+ severity=Severity.WARNING if mixed else Severity.INFO,
74
+ confidence=0.9 if mixed else 0.75,
75
+ evidence={
76
+ "family": family,
77
+ "units": distinct,
78
+ "target": target,
79
+ "examples": _cap_examples(raw_examples),
80
+ },
81
+ ops=[Op("normalize_unit", {"to": target})],
82
+ )
83
+ return issues
84
+
85
+
86
+ __all__ = ["detect_units"]
cleanframe/diff.py ADDED
@@ -0,0 +1,369 @@
1
+ """Cell-level diff: exactly which values changed, and how.
2
+
3
+ Every transform is tracked. Given the original frame, the cleaned frame, and the
4
+ column lineage (which output column came from which source), :func:`compute_diff`
5
+ produces a :class:`CellDiff` recording each changed cell (``before → after``), the
6
+ columns added/removed/renamed, and the rows dropped and why. This is the lineage
7
+ that makes CleanFrame auditable — "Every changed cell is tracked."
8
+
9
+ Rows are matched by a stable integer id (the original positional index), so cells
10
+ line up correctly even after dedup removes rows.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import math
16
+ from dataclasses import dataclass, field
17
+ from typing import Any
18
+
19
+ import pandas as pd
20
+
21
+ from ._util import DEFAULT_MAX_DIFF_CHANGES
22
+ from .ops import _is_na
23
+
24
+
25
+ def _equal(a: Any, b: Any) -> bool:
26
+ """Value equality that treats NaN==NaN as equal and any NaN/value pair as changed."""
27
+ a_na, b_na = _is_na(a), _is_na(b)
28
+ if a_na and b_na:
29
+ return True
30
+ if a_na or b_na:
31
+ return False
32
+ if isinstance(a, float) and isinstance(b, float):
33
+ if math.isnan(a) and math.isnan(b):
34
+ return True
35
+ try:
36
+ return bool(a == b)
37
+ except Exception: # noqa: BLE001 - exotic types compare unequal
38
+ return False
39
+
40
+
41
+ def _changed_mask(before: pd.Series, after: pd.Series) -> pd.Series:
42
+ """Vectorised element-wise 'changed?' mask with NaN==NaN treated as unchanged."""
43
+ both_na = before.isna().to_numpy() & after.isna().to_numpy()
44
+ try:
45
+ equal = (before.to_numpy() == after.to_numpy())
46
+ except Exception: # noqa: BLE001 - dtype mismatch -> fall back element-wise
47
+ equal = pd.Series(
48
+ [_equal(b, a) for b, a in zip(before.tolist(), after.tolist(), strict=True)],
49
+ index=before.index,
50
+ ).to_numpy()
51
+ equal = pd.Series(equal, index=before.index).fillna(False).astype(bool).to_numpy()
52
+ return pd.Series(~(equal | both_na), index=before.index)
53
+
54
+
55
+ @dataclass
56
+ class CellChange:
57
+ row_id: int
58
+ column: str
59
+ before: Any
60
+ after: Any
61
+
62
+
63
+ @dataclass
64
+ class CellDiff:
65
+ """A structured record of everything a recipe changed."""
66
+
67
+ changes: list[CellChange] = field(default_factory=list)
68
+ added_columns: list[str] = field(default_factory=list)
69
+ removed_columns: list[str] = field(default_factory=list)
70
+ renamed_columns: dict[str, str] = field(default_factory=dict) # source -> output
71
+ dropped_rows: list[tuple[int, str]] = field(default_factory=list) # (row_id, reason)
72
+ n_rows_before: int = 0
73
+ n_rows_after: int = 0
74
+ #: True when ``max_changes`` stopped recording further cell edits (summary still exact).
75
+ truncated: bool = False
76
+ #: Total cells that changed, including those not stored when truncated.
77
+ total_changed_cells: int = 0
78
+ #: Every column with at least one change, including changes beyond the cap.
79
+ changed_column_names: list[str] = field(default_factory=list)
80
+
81
+ # -- summaries -------------------------------------------------------
82
+ @property
83
+ def changed_cells(self) -> int:
84
+ return self.total_changed_cells or len(self.changes)
85
+
86
+ @property
87
+ def changed_columns(self) -> list[str]:
88
+ seen: list[str] = []
89
+ for c in self.changes:
90
+ if c.column not in seen:
91
+ seen.append(c.column)
92
+ return seen
93
+
94
+ def changes_by_column(self) -> dict[str, list[CellChange]]:
95
+ out: dict[str, list[CellChange]] = {}
96
+ for c in self.changes:
97
+ out.setdefault(c.column, []).append(c)
98
+ return out
99
+
100
+ def to_frame(self) -> pd.DataFrame:
101
+ """A tidy DataFrame of changes: ``row_id, column, before, after``."""
102
+ return pd.DataFrame(
103
+ [(c.row_id, c.column, c.before, c.after) for c in self.changes],
104
+ columns=["row_id", "column", "before", "after"],
105
+ )
106
+
107
+ def summary(self) -> dict[str, Any]:
108
+ return {
109
+ "changed_cells": self.changed_cells,
110
+ "changed_columns": len(self.changed_column_names or self.changed_columns),
111
+ "added_columns": list(self.added_columns),
112
+ "removed_columns": list(self.removed_columns),
113
+ "renamed_columns": dict(self.renamed_columns),
114
+ "rows_dropped": len(self.dropped_rows),
115
+ "rows_before": self.n_rows_before,
116
+ "rows_after": self.n_rows_after,
117
+ "truncated": self.truncated,
118
+ "stored_changes": len(self.changes),
119
+ }
120
+
121
+ def is_empty(self) -> bool:
122
+ return not (
123
+ self.changes
124
+ or self.added_columns
125
+ or self.removed_columns
126
+ or self.renamed_columns
127
+ or self.dropped_rows
128
+ )
129
+
130
+ # -- rendering -------------------------------------------------------
131
+ def render(
132
+ self, *, max_per_column: int = 8, color: bool | None = None, ascii: bool = False
133
+ ) -> str:
134
+ """Render a git-diff-style textual summary (also used by ``show``).
135
+
136
+ Pass ``ascii=True`` to substitute plain ASCII for the ``→``/``∅`` glyphs so
137
+ the text is safe to print on a non-UTF-8 console (e.g. a cp1252 Windows
138
+ terminal). :meth:`show` selects this automatically per output stream.
139
+ """
140
+ return _render_text(self, max_per_column=max_per_column, color=color, ascii=ascii)
141
+
142
+ def show(
143
+ self,
144
+ *,
145
+ max_per_column: int = 8,
146
+ color: bool | None = None,
147
+ stream: Any = None,
148
+ ascii: bool | None = None,
149
+ ) -> None:
150
+ """Print the diff, git-diff style. Never crashes on a non-UTF-8 console.
151
+
152
+ ``ascii`` defaults to auto: ASCII glyphs are used when the target stream
153
+ cannot encode ``→``/``∅`` (the default Windows cp1252 console), so library
154
+ users get clean output without the CLI's stdout reconfiguration.
155
+ """
156
+ import sys
157
+
158
+ out = stream if stream is not None else sys.stdout
159
+ if ascii is None:
160
+ ascii = not _stream_supports_unicode(out)
161
+ text = self.render(max_per_column=max_per_column, color=color, ascii=ascii)
162
+ try:
163
+ print(text, file=out)
164
+ except UnicodeEncodeError:
165
+ # The glyphs are not the only risk: a currency symbol in the *data* is
166
+ # unencodable on a cp1252 console too, so replace whatever is left.
167
+ fallback = self.render(max_per_column=max_per_column, color=color, ascii=True)
168
+ encoding = getattr(out, "encoding", None) or "ascii"
169
+ print(
170
+ fallback.encode(encoding, errors="replace").decode(encoding, errors="replace"),
171
+ file=out,
172
+ )
173
+
174
+ def __repr__(self) -> str: # pragma: no cover - cosmetic
175
+ s = self.summary()
176
+ return (
177
+ f"CellDiff(changed_cells={s['changed_cells']}, "
178
+ f"columns={s['changed_columns']}, rows_dropped={s['rows_dropped']})"
179
+ )
180
+
181
+
182
+ def compute_diff(
183
+ original: pd.DataFrame,
184
+ cleaned: pd.DataFrame,
185
+ lineage: dict[str, str | None],
186
+ *,
187
+ dropped_rows: list[tuple[int, str]] | None = None,
188
+ dropped_after: pd.DataFrame | None = None,
189
+ original_columns: list[str] | None = None,
190
+ n_rows_before: int | None = None,
191
+ max_changes: int | None = DEFAULT_MAX_DIFF_CHANGES,
192
+ ) -> CellDiff:
193
+ """Compute a :class:`CellDiff`.
194
+
195
+ Parameters
196
+ ----------
197
+ original:
198
+ The input frame, indexed by stable row id (positional 0..n-1).
199
+ cleaned:
200
+ The output frame, indexed by the surviving row ids.
201
+ lineage:
202
+ ``{output_column: source_column_or_None}``. ``None`` means the column was
203
+ derived/added (no "before" value).
204
+ dropped_rows:
205
+ ``(row_id, reason)`` pairs for rows removed during execution.
206
+ dropped_after:
207
+ Post-transform values (indexed by row id) of rows that were later dropped,
208
+ so a value rewrite applied to a row before it was removed is still tracked
209
+ as a changed cell (invariant #5) rather than vanishing with the row.
210
+ original_columns:
211
+ The full list of input column names. Pass this when ``original`` holds only
212
+ a *subset* of columns (the executor snapshots just the op-touched ones to
213
+ halve peak memory); columns absent from ``original`` are treated as
214
+ value-unchanged pass-throughs. Defaults to ``original.columns``.
215
+ max_changes:
216
+ Cap on stored :class:`CellChange` entries. ``None`` stores every change
217
+ (can OOM on large dirty frames). Counts in :attr:`CellDiff.changed_cells`
218
+ remain exact even when the detail list is truncated.
219
+ """
220
+ diff = CellDiff(
221
+ dropped_rows=list(dropped_rows or []),
222
+ n_rows_before=n_rows_before if n_rows_before is not None else int(len(original)),
223
+ n_rows_after=int(len(cleaned)),
224
+ )
225
+
226
+ all_columns = [str(c) for c in (original_columns if original_columns is not None else original.columns)]
227
+ source_cols = set(all_columns)
228
+ snapshot_cols = set(original.columns.astype(str))
229
+ used_sources: set[str] = set()
230
+ total = 0
231
+ store_cap = max_changes # None = unlimited
232
+
233
+ for out_col in cleaned.columns:
234
+ out_col = str(out_col)
235
+ source = lineage.get(out_col, out_col if out_col in source_cols else None)
236
+ if source is None:
237
+ diff.added_columns.append(out_col)
238
+ continue
239
+ used_sources.add(source)
240
+ if source != out_col:
241
+ diff.renamed_columns[source] = out_col
242
+
243
+ # A source absent from the (possibly partial) snapshot is a value-unchanged
244
+ # pass-through: record the rename, but there are no cell edits to diff.
245
+ if source not in snapshot_cols:
246
+ continue
247
+
248
+ after_series = cleaned[out_col]
249
+ if dropped_after is not None and out_col in dropped_after.columns:
250
+ # Append the post-transform values of dropped rows so their edits are
251
+ # tracked too (the rows still appear separately in dropped_rows).
252
+ after_series = pd.concat([after_series, dropped_after[out_col]])
253
+ before_series = original[source].reindex(after_series.index)
254
+ changed_mask = _changed_mask(before_series, after_series)
255
+ changed_ids = after_series.index[changed_mask.to_numpy()]
256
+ total += int(len(changed_ids))
257
+ if len(changed_ids) and out_col not in diff.changed_column_names:
258
+ diff.changed_column_names.append(out_col)
259
+
260
+ # Cap-aware bulk extraction: slice to the remaining budget FIRST, then pull
261
+ # before/after in one vectorised call each (avoids per-cell .loc — ~44x).
262
+ if store_cap is not None:
263
+ remaining = store_cap - len(diff.changes)
264
+ if remaining <= 0:
265
+ if len(changed_ids):
266
+ diff.truncated = True
267
+ continue
268
+ if len(changed_ids) > remaining:
269
+ diff.truncated = True
270
+ changed_ids = changed_ids[:remaining]
271
+ if len(changed_ids) == 0:
272
+ continue
273
+ befores = before_series.loc[changed_ids].tolist()
274
+ afters = after_series.loc[changed_ids].tolist()
275
+ for row_id, before, after in zip(changed_ids, befores, afters, strict=True):
276
+ diff.changes.append(CellChange(int(row_id), out_col, before, after))
277
+
278
+ for src in all_columns:
279
+ if src not in used_sources:
280
+ diff.removed_columns.append(src)
281
+
282
+ diff.total_changed_cells = total
283
+ return diff
284
+
285
+
286
+ # ---------------------------------------------------------------------------
287
+ # text rendering
288
+ # ---------------------------------------------------------------------------
289
+ def _stream_supports_unicode(stream: Any) -> bool:
290
+ """True if ``stream`` can encode the diff glyphs (``→``/``∅``/``⚠``)."""
291
+ enc = getattr(stream, "encoding", None) or "utf-8"
292
+ try:
293
+ "→∅⚠".encode(enc)
294
+ return True
295
+ except (UnicodeEncodeError, LookupError):
296
+ return False
297
+
298
+
299
+ def _fmt(value: Any, ascii: bool = False) -> str:
300
+ if _is_na(value):
301
+ return "<NA>" if ascii else "∅"
302
+ if isinstance(value, str):
303
+ return repr(value)
304
+ return str(value)
305
+
306
+
307
+ def _render_text(
308
+ diff: CellDiff, *, max_per_column: int, color: bool | None, ascii: bool = False
309
+ ) -> str:
310
+ import os
311
+ import sys
312
+
313
+ arrow = "->" if ascii else "→"
314
+ if color is None:
315
+ color = sys.stdout.isatty() and os.environ.get("NO_COLOR") is None
316
+ red = "\x1b[31m" if color else ""
317
+ green = "\x1b[32m" if color else ""
318
+ dim = "\x1b[2m" if color else ""
319
+ bold = "\x1b[1m" if color else ""
320
+ reset = "\x1b[0m" if color else ""
321
+
322
+ if diff.is_empty():
323
+ return f"{dim}No changes.{reset}"
324
+
325
+ lines: list[str] = []
326
+ s = diff.summary()
327
+ lines.append(
328
+ f"{bold}CleanFrame diff{reset} "
329
+ f"{s['changed_cells']} cell(s) changed in {s['changed_columns']} column(s), "
330
+ f"{s['rows_dropped']} row(s) dropped "
331
+ f"({s['rows_before']} {arrow} {s['rows_after']} rows)"
332
+ + (
333
+ f" {dim}[detail truncated to {s['stored_changes']}]{reset}"
334
+ if s.get("truncated")
335
+ else ""
336
+ )
337
+ )
338
+ if diff.renamed_columns:
339
+ renames = ", ".join(f"{k} {arrow} {v}" for k, v in diff.renamed_columns.items())
340
+ lines.append(f" {dim}renamed:{reset} {renames}")
341
+ if diff.added_columns:
342
+ lines.append(f" {green}added columns:{reset} {', '.join(diff.added_columns)}")
343
+ if diff.removed_columns:
344
+ lines.append(f" {red}removed columns:{reset} {', '.join(diff.removed_columns)}")
345
+
346
+ for column, changes in diff.changes_by_column().items():
347
+ lines.append("")
348
+ lines.append(f"{bold}{column}{reset} {dim}({len(changes)} changed){reset}")
349
+ for change in changes[:max_per_column]:
350
+ lines.append(
351
+ f" row {change.row_id}: "
352
+ f"{red}- {_fmt(change.before, ascii)}{reset} "
353
+ f"{green}+ {_fmt(change.after, ascii)}{reset}"
354
+ )
355
+ if len(changes) > max_per_column:
356
+ lines.append(f" {dim}… and {len(changes) - max_per_column} more{reset}")
357
+
358
+ if diff.dropped_rows:
359
+ lines.append("")
360
+ reasons: dict[str, int] = {}
361
+ for _, reason in diff.dropped_rows:
362
+ reasons[reason] = reasons.get(reason, 0) + 1
363
+ detail = ", ".join(f"{n} ({r})" for r, n in reasons.items())
364
+ lines.append(f"{red}dropped rows:{reset} {detail}")
365
+
366
+ return "\n".join(lines)
367
+
368
+
369
+ __all__ = ["CellDiff", "CellChange", "compute_diff"]