cleanframe-engine 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. cleanframe/__init__.py +169 -0
  2. cleanframe/__main__.py +5 -0
  3. cleanframe/_util.py +438 -0
  4. cleanframe/_version.py +1 -0
  5. cleanframe/api.py +559 -0
  6. cleanframe/cli.py +688 -0
  7. cleanframe/codegen.py +666 -0
  8. cleanframe/dataio.py +506 -0
  9. cleanframe/detectors/__init__.py +43 -0
  10. cleanframe/detectors/base.py +221 -0
  11. cleanframe/detectors/categories.py +199 -0
  12. cleanframe/detectors/contacts.py +105 -0
  13. cleanframe/detectors/currency.py +112 -0
  14. cleanframe/detectors/dates.py +204 -0
  15. cleanframe/detectors/dedup.py +150 -0
  16. cleanframe/detectors/nulls.py +109 -0
  17. cleanframe/detectors/outliers.py +73 -0
  18. cleanframe/detectors/schema_mapping.py +125 -0
  19. cleanframe/detectors/text.py +105 -0
  20. cleanframe/detectors/units.py +86 -0
  21. cleanframe/diff.py +369 -0
  22. cleanframe/drift.py +283 -0
  23. cleanframe/errors.py +66 -0
  24. cleanframe/executor.py +229 -0
  25. cleanframe/fingerprint.py +83 -0
  26. cleanframe/issues.py +186 -0
  27. cleanframe/llm.py +811 -0
  28. cleanframe/ops.py +1245 -0
  29. cleanframe/planner.py +353 -0
  30. cleanframe/profile.py +413 -0
  31. cleanframe/py.typed +1 -0
  32. cleanframe/quality.py +81 -0
  33. cleanframe/readfix.py +160 -0
  34. cleanframe/recipe.py +398 -0
  35. cleanframe/report.py +345 -0
  36. cleanframe/result.py +144 -0
  37. cleanframe/schema.py +259 -0
  38. cleanframe/streaming.py +354 -0
  39. cleanframe/types.py +119 -0
  40. cleanframe/validate.py +363 -0
  41. cleanframe/workbook.py +370 -0
  42. cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
  43. cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
  44. cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
  45. cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
  46. cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,204 @@
1
+ """Date detection and format inference.
2
+
3
+ Given a column the profiler flagged as dates, this figures out *which* formats are
4
+ actually present (a file mixing ``31/01/2024``, ``2024-01-31``, and ``1 Jan 2024``
5
+ is the whole point) and proposes a ``parse_date`` op that normalises them to ISO.
6
+
7
+ Ambiguous day/month order (``05/06/2024``) is resolved from the data when any value
8
+ disambiguates (a component > 12); otherwise it honours the ``dayfirst`` option and
9
+ records the assumption in the issue's evidence rather than guessing silently.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import re
15
+
16
+ import pandas as pd
17
+
18
+ from .._util import sample_non_null
19
+ from ..issues import Issues, _cap_examples
20
+ from ..profile import COMMON_DATE_FORMATS, _looks_date
21
+ from ..types import Op, Severity
22
+ from .base import DetectorContext, detector
23
+
24
+ _DAYFIRST_FORMATS = {"%d/%m/%Y", "%d-%m-%Y", "%d.%m.%Y", "%d/%m/%y", "%d-%m-%y"}
25
+ _MONTHFIRST_FORMATS = {"%m/%d/%Y", "%m-%d-%Y", "%m/%d/%y"}
26
+ _SLASH_DATE_RE = re.compile(r"^\s*(\d{1,2})[/\-.](\d{1,2})[/\-.]\d{2,4}\s*$")
27
+
28
+
29
+ def _infer_direction(values: list[str]) -> bool | None:
30
+ """Derive day-first vs month-first from any value with a component > 12.
31
+
32
+ A value like ``01/13/2024`` (second component 13 > 12) proves the column is
33
+ month-first; ``13/01/2024`` proves day-first. Returns ``True`` (day-first),
34
+ ``False`` (month-first), or ``None`` when nothing disambiguates or the column
35
+ genuinely mixes both orders. Feeding this into :func:`_infer_formats` stops the
36
+ greedy cover from producing two contradictory ``d/m`` + ``m/d`` formats that
37
+ silently swap day and month on the ambiguous (both ≤ 12) values.
38
+ """
39
+ saw_dayfirst = saw_monthfirst = False
40
+ for v in values:
41
+ m = _SLASH_DATE_RE.match(v)
42
+ if not m:
43
+ continue
44
+ a, b = int(m.group(1)), int(m.group(2))
45
+ if a > 12 and b <= 12:
46
+ saw_dayfirst = True
47
+ elif b > 12 and a <= 12:
48
+ saw_monthfirst = True
49
+ if saw_dayfirst and not saw_monthfirst:
50
+ return True
51
+ if saw_monthfirst and not saw_dayfirst:
52
+ return False
53
+ return None
54
+
55
+
56
+ def _ordered_formats(dayfirst: bool | None) -> list[str]:
57
+ """COMMON_DATE_FORMATS, reordered so the preferred d/m-vs-m/d variant wins ties."""
58
+ if dayfirst is False:
59
+ # Prefer month-first: move month-first formats ahead of their day-first twins.
60
+ monthfirst = [f for f in COMMON_DATE_FORMATS if f in _MONTHFIRST_FORMATS]
61
+ rest = [f for f in COMMON_DATE_FORMATS if f not in _MONTHFIRST_FORMATS]
62
+ return monthfirst + rest
63
+ return list(COMMON_DATE_FORMATS)
64
+
65
+
66
+ def _infer_formats(values: list[str], dayfirst: bool | None) -> tuple[list[str], int]:
67
+ """Greedy minimal cover of ``values`` by common formats. Returns (formats, unparsed)."""
68
+ dateish = [v for v in values if _looks_date(v)]
69
+ if not dateish:
70
+ return [], 0
71
+ ser = pd.Series(dateish, dtype="object")
72
+ covered = pd.Series(False, index=ser.index)
73
+ kept: list[str] = []
74
+ for fmt in _ordered_formats(dayfirst):
75
+ pending = ser[~covered]
76
+ if pending.empty:
77
+ break
78
+ parsed = pd.to_datetime(pending, format=fmt, errors="coerce")
79
+ hit = parsed.notna()
80
+ if hit.any():
81
+ kept.append(fmt)
82
+ covered.loc[pending.index[hit.to_numpy()]] = True
83
+ kept = _reconcile_slash_formats(kept, dayfirst)
84
+ # Recount against the formats actually kept: dropping the conflicting d/m vs m/d
85
+ # family leaves values that no longer parse, and reporting 0 would hide the NaTs.
86
+ covered = pd.Series(False, index=ser.index)
87
+ for fmt in kept:
88
+ pending = ser[~covered]
89
+ if pending.empty:
90
+ break
91
+ parsed = pd.to_datetime(pending, format=fmt, errors="coerce")
92
+ covered.loc[pending.index[parsed.notna().to_numpy()]] = True
93
+ unparsed = int((~covered).sum())
94
+ return kept, unparsed
95
+
96
+
97
+ def _reconcile_slash_formats(formats: list[str], dayfirst: bool | None) -> list[str]:
98
+ """Never keep both day-first and month-first slash formats in one recipe.
99
+
100
+ Ambiguous values (both components ≤ 12) would silently swap day/month depending
101
+ on which format is tried first. Prefer the inferred/option direction; default
102
+ to day-first when nothing disambiguates (matches COMMON_DATE_FORMATS order).
103
+ """
104
+ has_day = any(f in _DAYFIRST_FORMATS for f in formats)
105
+ has_month = any(f in _MONTHFIRST_FORMATS for f in formats)
106
+ if not (has_day and has_month):
107
+ return formats
108
+ prefer_day = True if dayfirst is None else bool(dayfirst)
109
+ drop = _MONTHFIRST_FORMATS if prefer_day else _DAYFIRST_FORMATS
110
+ return [f for f in formats if f not in drop]
111
+
112
+
113
+ @detector("dates", priority=40)
114
+ def detect_dates(series: pd.Series, ctx: DetectorContext) -> Issues:
115
+ """Detect mixed / non-ISO date formats and propose normalising them to ISO."""
116
+ issues = Issues()
117
+ cp = ctx.column_profile
118
+ if cp is None or cp.count == 0 or cp.semantic_type not in ("date", "datetime"):
119
+ return issues
120
+ # Already a proper datetime dtype — nothing to normalise.
121
+ if pd.api.types.is_datetime64_any_dtype(series.dtype):
122
+ return issues
123
+
124
+ values = [v if isinstance(v, str) else str(v) for v in sample_non_null(series)]
125
+ dayfirst_opt = ctx.option("dayfirst")
126
+ if dayfirst_opt is None:
127
+ # Derive day/month order from the data when the user hasn't pinned it, so
128
+ # the greedy cover uses one consistent slash format for the whole column.
129
+ dayfirst_opt = _infer_direction(values)
130
+ formats, unparsed = _infer_formats(values, dayfirst_opt)
131
+ if not formats:
132
+ # Looked like dates to the profiler but nothing parses cleanly.
133
+ issues.add(
134
+ "unparseable_dates",
135
+ "Column looks date-like but matches no known format",
136
+ severity=Severity.WARNING,
137
+ confidence=0.5,
138
+ evidence={"examples": _cap_examples(values)},
139
+ )
140
+ return issues
141
+
142
+ already_iso = formats == ["%Y-%m-%d"] and unparsed == 0
143
+ if already_iso:
144
+ return issues
145
+
146
+ dayfirst = formats[0] in _DAYFIRST_FORMATS
147
+ ambiguous = _is_ambiguous(values)
148
+ kind = "mixed_date_formats" if len(formats) > 1 else "nonstandard_date_format"
149
+ sev = Severity.WARNING if (len(formats) > 1 or unparsed) else Severity.INFO
150
+ # Preserve time-of-day if any inferred format carries one (don't truncate to date).
151
+ output = "%Y-%m-%dT%H:%M:%S" if any("%H" in f for f in formats) else "%Y-%m-%d"
152
+
153
+ evidence = {
154
+ "formats_found": formats,
155
+ "unparsed": unparsed,
156
+ "examples": _cap_examples(values),
157
+ }
158
+ if ambiguous:
159
+ evidence["ambiguous_day_month"] = True
160
+ evidence["assumed"] = "day-first" if dayfirst else "month-first"
161
+ if any("%y" in f for f in formats):
162
+ # 2-digit years use strptime's 1969–2068 century pivot — surface it.
163
+ evidence["two_digit_year_pivot"] = "1969-2068"
164
+
165
+ issues.add(
166
+ kind,
167
+ _message(formats, unparsed, ambiguous, dayfirst),
168
+ severity=sev,
169
+ confidence=0.9 if not ambiguous else 0.75,
170
+ evidence=evidence,
171
+ ops=[Op("parse_date", {"formats": formats, "dayfirst": dayfirst, "output": output})],
172
+ )
173
+ return issues
174
+
175
+
176
+ def _is_ambiguous(values: list[str]) -> bool:
177
+ """True if no value disambiguates day-vs-month order (all numeric components <= 12)."""
178
+ import re
179
+
180
+ saw_slashlike = False
181
+ for v in values:
182
+ m = re.match(r"^\s*(\d{1,2})[/\-.](\d{1,2})[/\-.]\d{2,4}\s*$", v)
183
+ if not m:
184
+ continue
185
+ saw_slashlike = True
186
+ a, b = int(m.group(1)), int(m.group(2))
187
+ if a > 12 or b > 12:
188
+ return False
189
+ return saw_slashlike
190
+
191
+
192
+ def _message(formats: list[str], unparsed: int, ambiguous: bool, dayfirst: bool) -> str:
193
+ if len(formats) > 1:
194
+ base = f"{len(formats)} different date formats present"
195
+ else:
196
+ base = f"Dates use non-ISO format {formats[0]!r}"
197
+ if unparsed:
198
+ base += f"; {unparsed} value(s) unparseable"
199
+ if ambiguous:
200
+ base += f" (day/month order ambiguous — assuming {'day' if dayfirst else 'month'}-first)"
201
+ return base
202
+
203
+
204
+ __all__ = ["detect_dates"]
@@ -0,0 +1,150 @@
1
+ """Duplicate-row detection — exact and fuzzy.
2
+
3
+ Exact full-row duplicates are safe to drop, so we propose a ``dedup`` op with high
4
+ confidence. *Key* duplicates (two rows sharing an email/id but differing elsewhere)
5
+ and *fuzzy* near-duplicates on name-like columns are only *reported* with
6
+ reviewable pair evidence — silently collapsing them would lose data.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from difflib import SequenceMatcher
12
+
13
+ import pandas as pd
14
+
15
+ from .._util import normalize_key
16
+ from ..issues import Issues, _cap_examples
17
+ from ..types import Op, Severity
18
+ from .base import DetectorContext, detector
19
+
20
+ _KEY_HINTS = ("email", "id", "uuid", "guid")
21
+ _NAME_HINTS = ("name", "customer", "client", "company", "vendor", "title")
22
+ _FUZZY_THRESHOLD = 0.88
23
+ _FUZZY_MAX_ROWS = 200 # pairwise cost; bound for determinism + speed
24
+
25
+
26
+ def _key_columns(ctx: DetectorContext) -> list[str]:
27
+ keys: list[str] = []
28
+ for cp in ctx.profile.columns:
29
+ low = cp.name.lower()
30
+ if cp.semantic_type in ("email", "id") or any(h in low for h in _KEY_HINTS):
31
+ keys.append(cp.name)
32
+ return keys
33
+
34
+
35
+ def _fuzzy_columns(ctx: DetectorContext) -> list[str]:
36
+ cols: list[str] = []
37
+ for cp in ctx.profile.columns:
38
+ low = cp.name.lower()
39
+ if cp.semantic_type in ("text", "categorical", "id") and any(
40
+ h in low for h in _NAME_HINTS
41
+ ):
42
+ cols.append(cp.name)
43
+ return cols
44
+
45
+
46
+ def _fuzzy_pairs(df: pd.DataFrame, column: str) -> list[dict]:
47
+ """Return near-duplicate (i, j, score, values) pairs, sorted deterministically."""
48
+ if column not in df.columns or len(df) < 2:
49
+ return []
50
+ # Cap pairwise work; take a stable head so results don't depend on shuffle.
51
+ work = df.head(_FUZZY_MAX_ROWS)
52
+ # Use positional ids (0..n-1) not the index label — labels may be strings,
53
+ # Timestamps, or anything else, and int(label) would crash on them.
54
+ values = [
55
+ (pos, str(v).strip())
56
+ for pos, (_idx, v) in enumerate(work[column].items())
57
+ if pd.notna(v)
58
+ ]
59
+ pairs: list[dict] = []
60
+ for i in range(len(values)):
61
+ idx_i, a = values[i]
62
+ if not a:
63
+ continue
64
+ na = normalize_key(a)
65
+ for j in range(i + 1, len(values)):
66
+ idx_j, b = values[j]
67
+ if not b or a == b:
68
+ continue
69
+ # Skip exact-normalized matches — those are category/casing, not fuzzy.
70
+ if na == normalize_key(b):
71
+ continue
72
+ score = SequenceMatcher(None, a.casefold(), b.casefold()).ratio()
73
+ if score >= _FUZZY_THRESHOLD:
74
+ pairs.append(
75
+ {
76
+ "rows": [idx_i, idx_j],
77
+ "values": [a, b],
78
+ "score": round(score, 3),
79
+ "proposal": {"keep": idx_i, "drop": idx_j},
80
+ }
81
+ )
82
+ pairs.sort(key=lambda p: (-p["score"], p["rows"][0], p["rows"][1]))
83
+ return pairs
84
+
85
+
86
+ @detector("dedup", scope="frame", priority=80)
87
+ def detect_duplicates(df: pd.DataFrame, ctx: DetectorContext) -> Issues:
88
+ """Propose dropping exact duplicate rows; report key/fuzzy duplicates for review."""
89
+ issues = Issues()
90
+ if len(df) == 0:
91
+ return issues
92
+
93
+ # Reuse the profiler's exact-duplicate count instead of hashing the whole frame
94
+ # a second time (the profiler already computed df.duplicated() for every clean).
95
+ exact = int(ctx.profile.duplicate_row_count)
96
+ if exact:
97
+ issues.add(
98
+ "duplicate_rows",
99
+ f"{exact} exact duplicate row(s)",
100
+ severity=Severity.WARNING,
101
+ confidence=1.0,
102
+ evidence={"count": exact},
103
+ ops=[Op("dedup", {"keep": "first"})],
104
+ )
105
+
106
+ # Key duplicates beyond the exact ones — report, don't auto-merge.
107
+ for key in _key_columns(ctx):
108
+ if key not in df.columns:
109
+ continue
110
+ dupe_mask = df[key].dropna().duplicated(keep=False)
111
+ n = int(dupe_mask.sum())
112
+ if n > exact:
113
+ examples = (
114
+ df.loc[df[key].duplicated(keep=False), key]
115
+ .dropna()
116
+ .drop_duplicates()
117
+ .head(5)
118
+ .tolist()
119
+ )
120
+ issues.add(
121
+ "duplicate_keys",
122
+ f"Column {key!r} has {n} row(s) sharing a value that should be unique",
123
+ severity=Severity.WARNING,
124
+ column=key,
125
+ confidence=0.6,
126
+ evidence={"count": n, "examples": _cap_examples(examples)},
127
+ )
128
+
129
+ # Fuzzy near-duplicates on name-like columns — reviewable merge proposals only.
130
+ for col in _fuzzy_columns(ctx):
131
+ pairs = _fuzzy_pairs(df, col)
132
+ if not pairs:
133
+ continue
134
+ issues.add(
135
+ "fuzzy_duplicates",
136
+ f"{len(pairs)} near-duplicate pair(s) in {col!r} (fuzzy match ≥ {_FUZZY_THRESHOLD:.0%})",
137
+ severity=Severity.INFO,
138
+ column=col,
139
+ confidence=0.55, # below auto threshold — review mode surfaces it
140
+ evidence={
141
+ "count": len(pairs),
142
+ "threshold": _FUZZY_THRESHOLD,
143
+ "pairs": pairs[:10],
144
+ },
145
+ # No ops — merge is a human decision; evidence carries keep/drop proposals.
146
+ )
147
+ return issues
148
+
149
+
150
+ __all__ = ["detect_duplicates"]
@@ -0,0 +1,109 @@
1
+ """Missing values — real and disguised.
2
+
3
+ Two distinct jobs:
4
+
5
+ * **Disguised nulls** (``"NA"``, ``"-"``, ``"unknown"``) are a *fix*: we propose a
6
+ ``to_na`` op to turn them into real NaN so downstream parsing and counts are
7
+ honest.
8
+ * **Actually-missing values** are *reported, never imputed*. CleanFrame will tell
9
+ you a column is 40% empty; it will not silently invent values. (Users who want
10
+ imputation add an explicit ``fill_na`` op themselves.)
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import pandas as pd
16
+
17
+ from .._util import is_string_like, looks_like_code_values, sample_non_null
18
+ from ..issues import Issues
19
+ from ..ops import DEFAULT_NA_TOKENS
20
+ from ..types import Op, Severity
21
+ from .base import DetectorContext, detector
22
+
23
+ _NA_LOOKUP = {t.strip().casefold() for t in DEFAULT_NA_TOKENS if t}
24
+ #: Tokens that are frequently *legitimate* data (an allergy of "none", a status of
25
+ #: "unknown", a placeholder "-"). We surface these for review but do NOT auto-convert
26
+ #: them below strict review, so a real category is never silently turned into NaN.
27
+ _AMBIGUOUS_NA = {"none", "nil", "-", "--", "?", "unknown"}
28
+
29
+
30
+ @detector("nulls", priority=20)
31
+ def detect_nulls(series: pd.Series, ctx: DetectorContext) -> Issues:
32
+ """Convert disguised nulls to NaN; report genuinely-missing and structural oddities."""
33
+ issues = Issues()
34
+ cp = ctx.column_profile
35
+ if cp is None:
36
+ return issues
37
+
38
+ # 1) Disguised nulls in string columns -> fixable.
39
+ if is_string_like(series):
40
+ sample = sample_non_null(series)
41
+ disguised: dict[str, int] = {}
42
+ for v in sample:
43
+ if isinstance(v, str) and v.strip().casefold() in _NA_LOOKUP:
44
+ disguised[v] = disguised.get(v, 0) + 1
45
+ if disguised:
46
+ ambiguous_tokens = set(_AMBIGUOUS_NA)
47
+ if looks_like_code_values(sample):
48
+ ambiguous_tokens |= {"na", "nan", "nil", "none"}
49
+ unambiguous = sorted(
50
+ v for v in disguised if v.strip().casefold() not in ambiguous_tokens
51
+ )
52
+ ambiguous = sorted(v for v in disguised if v.strip().casefold() in ambiguous_tokens)
53
+ # Unambiguous null tokens ('', 'n/a', 'null', …) — safe to convert.
54
+ if unambiguous:
55
+ n = sum(disguised[t] for t in unambiguous)
56
+ issues.add(
57
+ "disguised_nulls",
58
+ f"{n} value(s) are disguised nulls ({', '.join(map(repr, unambiguous[:4]))})",
59
+ severity=Severity.WARNING,
60
+ confidence=1.0,
61
+ evidence={"count": n, "tokens": {t: disguised[t] for t in unambiguous}},
62
+ ops=[Op("to_na", {"tokens": unambiguous})],
63
+ )
64
+ # Ambiguous tokens that may be legitimate data — report, don't auto-convert.
65
+ if ambiguous:
66
+ n = sum(disguised[t] for t in ambiguous)
67
+ issues.add(
68
+ "disguised_nulls",
69
+ f"{n} value(s) may be disguised nulls "
70
+ f"({', '.join(map(repr, ambiguous[:4]))}) — review before converting",
71
+ severity=Severity.INFO,
72
+ confidence=0.45, # below the review threshold: surfaced, not auto-applied
73
+ evidence={"count": n, "tokens": {t: disguised[t] for t in ambiguous}, "ambiguous": True},
74
+ ops=[Op("to_na", {"tokens": ambiguous})],
75
+ )
76
+
77
+ # 2) Genuinely missing values -> report only.
78
+ if cp.null_count:
79
+ sev = Severity.WARNING if cp.null_fraction >= 0.2 else Severity.INFO
80
+ issues.add(
81
+ "missing_values",
82
+ f"{cp.null_count} missing value(s) ({cp.null_fraction:.0%} of the column)",
83
+ severity=sev,
84
+ confidence=1.0,
85
+ evidence={"null_count": cp.null_count, "null_fraction": round(cp.null_fraction, 4)},
86
+ )
87
+
88
+ # 3) Structural oddities worth surfacing.
89
+ if cp.count == 0:
90
+ issues.add(
91
+ "empty_column",
92
+ "Column is entirely empty",
93
+ severity=Severity.WARNING,
94
+ confidence=1.0,
95
+ evidence={},
96
+ )
97
+ elif cp.is_constant:
98
+ only = cp.sample_values[0] if cp.sample_values else None
99
+ issues.add(
100
+ "constant_column",
101
+ f"Column has a single value throughout ({only!r})",
102
+ severity=Severity.INFO,
103
+ confidence=1.0,
104
+ evidence={"value": only},
105
+ )
106
+ return issues
107
+
108
+
109
+ __all__ = ["detect_nulls"]
@@ -0,0 +1,73 @@
1
+ """Outlier detection — flagged with evidence, never auto-"fixed".
2
+
3
+ Uses a classic Tukey IQR fence on numeric columns. Findings carry the fence
4
+ bounds and example values so a human (or a downstream validator) can decide;
5
+ no ops are attached, matching the README / CONTRIBUTING invariant that outliers
6
+ are detected, never silently corrected.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import pandas as pd
12
+
13
+ from ..issues import Issues, _cap_examples
14
+ from ..types import Severity
15
+ from .base import DetectorContext, detector
16
+
17
+ _MIN_VALUES = 8
18
+ _IQR_K = 1.5
19
+
20
+
21
+ @detector("outliers", priority=70)
22
+ def detect_outliers(series: pd.Series, ctx: DetectorContext) -> Issues:
23
+ """Flag numeric outliers via the IQR fence. Report only — never propose a fix."""
24
+ issues = Issues()
25
+ cp = ctx.column_profile
26
+ if cp is None or cp.count < _MIN_VALUES:
27
+ return issues
28
+ if cp.semantic_type not in ("integer", "float", "currency"):
29
+ # Also accept already-numeric dtypes even if semantic type is currency-as-float.
30
+ if not (
31
+ pd.api.types.is_numeric_dtype(series.dtype)
32
+ and not pd.api.types.is_bool_dtype(series.dtype)
33
+ ):
34
+ return issues
35
+
36
+ numeric = pd.to_numeric(series, errors="coerce").dropna()
37
+ if len(numeric) < _MIN_VALUES:
38
+ return issues
39
+
40
+ q1 = float(numeric.quantile(0.25))
41
+ q3 = float(numeric.quantile(0.75))
42
+ iqr = q3 - q1
43
+ if iqr == 0:
44
+ return issues
45
+ low = q1 - _IQR_K * iqr
46
+ high = q3 + _IQR_K * iqr
47
+ mask = (numeric < low) | (numeric > high)
48
+ n = int(mask.sum())
49
+ if n == 0:
50
+ return issues
51
+
52
+ # Slice to the cap BEFORE materialising — a column with millions of outliers must
53
+ # not build a giant throwaway Python list just to keep 5 examples.
54
+ examples = numeric[mask].head(5).tolist()
55
+ issues.add(
56
+ "outliers",
57
+ f"{n} outlier value(s) outside IQR fence [{low:.4g}, {high:.4g}]",
58
+ severity=Severity.INFO,
59
+ confidence=0.85,
60
+ evidence={
61
+ "count": n,
62
+ "low": low,
63
+ "high": high,
64
+ "q1": q1,
65
+ "q3": q3,
66
+ "examples": _cap_examples(examples),
67
+ },
68
+ # Intentionally no ops — outliers are never auto-fixed.
69
+ )
70
+ return issues
71
+
72
+
73
+ __all__ = ["detect_outliers"]
@@ -0,0 +1,125 @@
1
+ """Map a messy file onto a target schema — with confidence scores.
2
+
3
+ Runs only when the caller supplied a target schema. Performs a deterministic,
4
+ greedy 1:1 assignment of schema columns to source columns using fuzzy name
5
+ similarity nudged by type compatibility, then emits rename proposals (strong
6
+ matches), review-flagged weak matches, and reports for schema columns with no
7
+ home and source columns with no schema slot.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import pandas as pd
13
+
14
+ from .._util import similarity
15
+ from ..issues import Issues
16
+ from ..types import Severity
17
+ from .base import DetectorContext, detector
18
+
19
+ #: Which source semantic types satisfy a given schema dtype (None = anything).
20
+ _COMPAT: dict[str, set[str]] = {
21
+ "integer": {"integer", "float", "currency"},
22
+ "float": {"float", "integer", "currency"},
23
+ "currency": {"currency", "float", "integer"},
24
+ "date": {"date", "datetime"},
25
+ "datetime": {"datetime", "date"},
26
+ "email": {"email", "text", "id"},
27
+ "phone": {"phone", "text"},
28
+ "url": {"url", "text"},
29
+ "category": {"categorical", "text", "id"},
30
+ "boolean": {"boolean"},
31
+ }
32
+
33
+ _STRONG = 0.6
34
+ _WEAK = 0.4
35
+
36
+
37
+ def _type_ok(schema_dtype: str, source_semantic: str) -> bool:
38
+ allowed = _COMPAT.get(schema_dtype)
39
+ return True if allowed is None else source_semantic in allowed
40
+
41
+
42
+ @detector("schema_mapping", scope="frame", priority=5, requires_schema=True)
43
+ def detect_schema_mapping(df: pd.DataFrame, ctx: DetectorContext) -> Issues:
44
+ """Map source columns onto the target schema by fuzzy name + type, with confidence."""
45
+ issues = Issues()
46
+ schema = ctx.schema
47
+ if schema is None:
48
+ return issues
49
+
50
+ sources = [str(c) for c in df.columns]
51
+ used: set[str] = set()
52
+
53
+ for scol in schema.columns:
54
+ scored: list[tuple[float, float, str]] = []
55
+ for src in sources:
56
+ if src in used:
57
+ continue
58
+ sim = similarity(scol.name, src)
59
+ sp = ctx.profile.column(src)
60
+ compat = _type_ok(scol.dtype, sp.semantic_type) if sp else True
61
+ score = 1.0 if (src == scol.name or src in (scol.aliases or [])) else sim
62
+ if not compat:
63
+ score -= 0.15
64
+ scored.append((score, sim, src))
65
+
66
+ if not scored:
67
+ issues.add(
68
+ "missing_schema_column",
69
+ f"Schema column {scol.name!r} has no candidate in the data",
70
+ severity=Severity.WARNING,
71
+ confidence=1.0,
72
+ evidence={"target": scol.name},
73
+ )
74
+ continue
75
+
76
+ scored.sort(key=lambda t: (-t[0], t[2])) # best score, then name asc -> deterministic
77
+ best_score, best_sim, best_src = scored[0]
78
+
79
+ if best_score >= _STRONG:
80
+ used.add(best_src)
81
+ if best_src != scol.name:
82
+ issues.add(
83
+ "schema_mapping",
84
+ f"Map {best_src!r} → {scol.name!r} ({best_sim:.0%} name match)",
85
+ severity=Severity.INFO,
86
+ column=best_src,
87
+ confidence=round(best_sim, 3),
88
+ evidence={"target": scol.name, "score": round(best_sim, 3)},
89
+ rename_to=scol.name,
90
+ )
91
+ elif best_score >= _WEAK:
92
+ used.add(best_src)
93
+ issues.add(
94
+ "weak_schema_match",
95
+ f"{best_src!r} only weakly matches schema column {scol.name!r} ({best_sim:.0%})",
96
+ severity=Severity.WARNING,
97
+ column=best_src,
98
+ confidence=round(best_sim, 3),
99
+ evidence={"target": scol.name, "score": round(best_sim, 3)},
100
+ rename_to=scol.name,
101
+ )
102
+ else:
103
+ issues.add(
104
+ "missing_schema_column",
105
+ f"Schema column {scol.name!r} has no good match "
106
+ f"(closest {best_src!r} at {best_sim:.0%})",
107
+ severity=Severity.WARNING,
108
+ confidence=1.0,
109
+ evidence={"target": scol.name, "closest": best_src, "score": round(best_sim, 3)},
110
+ )
111
+
112
+ for src in sources:
113
+ if src not in used:
114
+ issues.add(
115
+ "extra_source_column",
116
+ f"Source column {src!r} is not in the target schema",
117
+ severity=Severity.INFO,
118
+ column=src,
119
+ confidence=1.0,
120
+ evidence={},
121
+ )
122
+ return issues
123
+
124
+
125
+ __all__ = ["detect_schema_mapping"]