cleanframe-engine 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanframe/__init__.py +169 -0
- cleanframe/__main__.py +5 -0
- cleanframe/_util.py +438 -0
- cleanframe/_version.py +1 -0
- cleanframe/api.py +559 -0
- cleanframe/cli.py +688 -0
- cleanframe/codegen.py +666 -0
- cleanframe/dataio.py +506 -0
- cleanframe/detectors/__init__.py +43 -0
- cleanframe/detectors/base.py +221 -0
- cleanframe/detectors/categories.py +199 -0
- cleanframe/detectors/contacts.py +105 -0
- cleanframe/detectors/currency.py +112 -0
- cleanframe/detectors/dates.py +204 -0
- cleanframe/detectors/dedup.py +150 -0
- cleanframe/detectors/nulls.py +109 -0
- cleanframe/detectors/outliers.py +73 -0
- cleanframe/detectors/schema_mapping.py +125 -0
- cleanframe/detectors/text.py +105 -0
- cleanframe/detectors/units.py +86 -0
- cleanframe/diff.py +369 -0
- cleanframe/drift.py +283 -0
- cleanframe/errors.py +66 -0
- cleanframe/executor.py +229 -0
- cleanframe/fingerprint.py +83 -0
- cleanframe/issues.py +186 -0
- cleanframe/llm.py +811 -0
- cleanframe/ops.py +1245 -0
- cleanframe/planner.py +353 -0
- cleanframe/profile.py +413 -0
- cleanframe/py.typed +1 -0
- cleanframe/quality.py +81 -0
- cleanframe/readfix.py +160 -0
- cleanframe/recipe.py +398 -0
- cleanframe/report.py +345 -0
- cleanframe/result.py +144 -0
- cleanframe/schema.py +259 -0
- cleanframe/streaming.py +354 -0
- cleanframe/types.py +119 -0
- cleanframe/validate.py +363 -0
- cleanframe/workbook.py +370 -0
- cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
- cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
- cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
- cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
- cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
"""Date detection and format inference.
|
|
2
|
+
|
|
3
|
+
Given a column the profiler flagged as dates, this figures out *which* formats are
|
|
4
|
+
actually present (a file mixing ``31/01/2024``, ``2024-01-31``, and ``1 Jan 2024``
|
|
5
|
+
is the whole point) and proposes a ``parse_date`` op that normalises them to ISO.
|
|
6
|
+
|
|
7
|
+
Ambiguous day/month order (``05/06/2024``) is resolved from the data when any value
|
|
8
|
+
disambiguates (a component > 12); otherwise it honours the ``dayfirst`` option and
|
|
9
|
+
records the assumption in the issue's evidence rather than guessing silently.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import re
|
|
15
|
+
|
|
16
|
+
import pandas as pd
|
|
17
|
+
|
|
18
|
+
from .._util import sample_non_null
|
|
19
|
+
from ..issues import Issues, _cap_examples
|
|
20
|
+
from ..profile import COMMON_DATE_FORMATS, _looks_date
|
|
21
|
+
from ..types import Op, Severity
|
|
22
|
+
from .base import DetectorContext, detector
|
|
23
|
+
|
|
24
|
+
_DAYFIRST_FORMATS = {"%d/%m/%Y", "%d-%m-%Y", "%d.%m.%Y", "%d/%m/%y", "%d-%m-%y"}
|
|
25
|
+
_MONTHFIRST_FORMATS = {"%m/%d/%Y", "%m-%d-%Y", "%m/%d/%y"}
|
|
26
|
+
_SLASH_DATE_RE = re.compile(r"^\s*(\d{1,2})[/\-.](\d{1,2})[/\-.]\d{2,4}\s*$")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _infer_direction(values: list[str]) -> bool | None:
|
|
30
|
+
"""Derive day-first vs month-first from any value with a component > 12.
|
|
31
|
+
|
|
32
|
+
A value like ``01/13/2024`` (second component 13 > 12) proves the column is
|
|
33
|
+
month-first; ``13/01/2024`` proves day-first. Returns ``True`` (day-first),
|
|
34
|
+
``False`` (month-first), or ``None`` when nothing disambiguates or the column
|
|
35
|
+
genuinely mixes both orders. Feeding this into :func:`_infer_formats` stops the
|
|
36
|
+
greedy cover from producing two contradictory ``d/m`` + ``m/d`` formats that
|
|
37
|
+
silently swap day and month on the ambiguous (both ≤ 12) values.
|
|
38
|
+
"""
|
|
39
|
+
saw_dayfirst = saw_monthfirst = False
|
|
40
|
+
for v in values:
|
|
41
|
+
m = _SLASH_DATE_RE.match(v)
|
|
42
|
+
if not m:
|
|
43
|
+
continue
|
|
44
|
+
a, b = int(m.group(1)), int(m.group(2))
|
|
45
|
+
if a > 12 and b <= 12:
|
|
46
|
+
saw_dayfirst = True
|
|
47
|
+
elif b > 12 and a <= 12:
|
|
48
|
+
saw_monthfirst = True
|
|
49
|
+
if saw_dayfirst and not saw_monthfirst:
|
|
50
|
+
return True
|
|
51
|
+
if saw_monthfirst and not saw_dayfirst:
|
|
52
|
+
return False
|
|
53
|
+
return None
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _ordered_formats(dayfirst: bool | None) -> list[str]:
|
|
57
|
+
"""COMMON_DATE_FORMATS, reordered so the preferred d/m-vs-m/d variant wins ties."""
|
|
58
|
+
if dayfirst is False:
|
|
59
|
+
# Prefer month-first: move month-first formats ahead of their day-first twins.
|
|
60
|
+
monthfirst = [f for f in COMMON_DATE_FORMATS if f in _MONTHFIRST_FORMATS]
|
|
61
|
+
rest = [f for f in COMMON_DATE_FORMATS if f not in _MONTHFIRST_FORMATS]
|
|
62
|
+
return monthfirst + rest
|
|
63
|
+
return list(COMMON_DATE_FORMATS)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _infer_formats(values: list[str], dayfirst: bool | None) -> tuple[list[str], int]:
|
|
67
|
+
"""Greedy minimal cover of ``values`` by common formats. Returns (formats, unparsed)."""
|
|
68
|
+
dateish = [v for v in values if _looks_date(v)]
|
|
69
|
+
if not dateish:
|
|
70
|
+
return [], 0
|
|
71
|
+
ser = pd.Series(dateish, dtype="object")
|
|
72
|
+
covered = pd.Series(False, index=ser.index)
|
|
73
|
+
kept: list[str] = []
|
|
74
|
+
for fmt in _ordered_formats(dayfirst):
|
|
75
|
+
pending = ser[~covered]
|
|
76
|
+
if pending.empty:
|
|
77
|
+
break
|
|
78
|
+
parsed = pd.to_datetime(pending, format=fmt, errors="coerce")
|
|
79
|
+
hit = parsed.notna()
|
|
80
|
+
if hit.any():
|
|
81
|
+
kept.append(fmt)
|
|
82
|
+
covered.loc[pending.index[hit.to_numpy()]] = True
|
|
83
|
+
kept = _reconcile_slash_formats(kept, dayfirst)
|
|
84
|
+
# Recount against the formats actually kept: dropping the conflicting d/m vs m/d
|
|
85
|
+
# family leaves values that no longer parse, and reporting 0 would hide the NaTs.
|
|
86
|
+
covered = pd.Series(False, index=ser.index)
|
|
87
|
+
for fmt in kept:
|
|
88
|
+
pending = ser[~covered]
|
|
89
|
+
if pending.empty:
|
|
90
|
+
break
|
|
91
|
+
parsed = pd.to_datetime(pending, format=fmt, errors="coerce")
|
|
92
|
+
covered.loc[pending.index[parsed.notna().to_numpy()]] = True
|
|
93
|
+
unparsed = int((~covered).sum())
|
|
94
|
+
return kept, unparsed
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _reconcile_slash_formats(formats: list[str], dayfirst: bool | None) -> list[str]:
|
|
98
|
+
"""Never keep both day-first and month-first slash formats in one recipe.
|
|
99
|
+
|
|
100
|
+
Ambiguous values (both components ≤ 12) would silently swap day/month depending
|
|
101
|
+
on which format is tried first. Prefer the inferred/option direction; default
|
|
102
|
+
to day-first when nothing disambiguates (matches COMMON_DATE_FORMATS order).
|
|
103
|
+
"""
|
|
104
|
+
has_day = any(f in _DAYFIRST_FORMATS for f in formats)
|
|
105
|
+
has_month = any(f in _MONTHFIRST_FORMATS for f in formats)
|
|
106
|
+
if not (has_day and has_month):
|
|
107
|
+
return formats
|
|
108
|
+
prefer_day = True if dayfirst is None else bool(dayfirst)
|
|
109
|
+
drop = _MONTHFIRST_FORMATS if prefer_day else _DAYFIRST_FORMATS
|
|
110
|
+
return [f for f in formats if f not in drop]
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
@detector("dates", priority=40)
|
|
114
|
+
def detect_dates(series: pd.Series, ctx: DetectorContext) -> Issues:
|
|
115
|
+
"""Detect mixed / non-ISO date formats and propose normalising them to ISO."""
|
|
116
|
+
issues = Issues()
|
|
117
|
+
cp = ctx.column_profile
|
|
118
|
+
if cp is None or cp.count == 0 or cp.semantic_type not in ("date", "datetime"):
|
|
119
|
+
return issues
|
|
120
|
+
# Already a proper datetime dtype — nothing to normalise.
|
|
121
|
+
if pd.api.types.is_datetime64_any_dtype(series.dtype):
|
|
122
|
+
return issues
|
|
123
|
+
|
|
124
|
+
values = [v if isinstance(v, str) else str(v) for v in sample_non_null(series)]
|
|
125
|
+
dayfirst_opt = ctx.option("dayfirst")
|
|
126
|
+
if dayfirst_opt is None:
|
|
127
|
+
# Derive day/month order from the data when the user hasn't pinned it, so
|
|
128
|
+
# the greedy cover uses one consistent slash format for the whole column.
|
|
129
|
+
dayfirst_opt = _infer_direction(values)
|
|
130
|
+
formats, unparsed = _infer_formats(values, dayfirst_opt)
|
|
131
|
+
if not formats:
|
|
132
|
+
# Looked like dates to the profiler but nothing parses cleanly.
|
|
133
|
+
issues.add(
|
|
134
|
+
"unparseable_dates",
|
|
135
|
+
"Column looks date-like but matches no known format",
|
|
136
|
+
severity=Severity.WARNING,
|
|
137
|
+
confidence=0.5,
|
|
138
|
+
evidence={"examples": _cap_examples(values)},
|
|
139
|
+
)
|
|
140
|
+
return issues
|
|
141
|
+
|
|
142
|
+
already_iso = formats == ["%Y-%m-%d"] and unparsed == 0
|
|
143
|
+
if already_iso:
|
|
144
|
+
return issues
|
|
145
|
+
|
|
146
|
+
dayfirst = formats[0] in _DAYFIRST_FORMATS
|
|
147
|
+
ambiguous = _is_ambiguous(values)
|
|
148
|
+
kind = "mixed_date_formats" if len(formats) > 1 else "nonstandard_date_format"
|
|
149
|
+
sev = Severity.WARNING if (len(formats) > 1 or unparsed) else Severity.INFO
|
|
150
|
+
# Preserve time-of-day if any inferred format carries one (don't truncate to date).
|
|
151
|
+
output = "%Y-%m-%dT%H:%M:%S" if any("%H" in f for f in formats) else "%Y-%m-%d"
|
|
152
|
+
|
|
153
|
+
evidence = {
|
|
154
|
+
"formats_found": formats,
|
|
155
|
+
"unparsed": unparsed,
|
|
156
|
+
"examples": _cap_examples(values),
|
|
157
|
+
}
|
|
158
|
+
if ambiguous:
|
|
159
|
+
evidence["ambiguous_day_month"] = True
|
|
160
|
+
evidence["assumed"] = "day-first" if dayfirst else "month-first"
|
|
161
|
+
if any("%y" in f for f in formats):
|
|
162
|
+
# 2-digit years use strptime's 1969–2068 century pivot — surface it.
|
|
163
|
+
evidence["two_digit_year_pivot"] = "1969-2068"
|
|
164
|
+
|
|
165
|
+
issues.add(
|
|
166
|
+
kind,
|
|
167
|
+
_message(formats, unparsed, ambiguous, dayfirst),
|
|
168
|
+
severity=sev,
|
|
169
|
+
confidence=0.9 if not ambiguous else 0.75,
|
|
170
|
+
evidence=evidence,
|
|
171
|
+
ops=[Op("parse_date", {"formats": formats, "dayfirst": dayfirst, "output": output})],
|
|
172
|
+
)
|
|
173
|
+
return issues
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _is_ambiguous(values: list[str]) -> bool:
|
|
177
|
+
"""True if no value disambiguates day-vs-month order (all numeric components <= 12)."""
|
|
178
|
+
import re
|
|
179
|
+
|
|
180
|
+
saw_slashlike = False
|
|
181
|
+
for v in values:
|
|
182
|
+
m = re.match(r"^\s*(\d{1,2})[/\-.](\d{1,2})[/\-.]\d{2,4}\s*$", v)
|
|
183
|
+
if not m:
|
|
184
|
+
continue
|
|
185
|
+
saw_slashlike = True
|
|
186
|
+
a, b = int(m.group(1)), int(m.group(2))
|
|
187
|
+
if a > 12 or b > 12:
|
|
188
|
+
return False
|
|
189
|
+
return saw_slashlike
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def _message(formats: list[str], unparsed: int, ambiguous: bool, dayfirst: bool) -> str:
|
|
193
|
+
if len(formats) > 1:
|
|
194
|
+
base = f"{len(formats)} different date formats present"
|
|
195
|
+
else:
|
|
196
|
+
base = f"Dates use non-ISO format {formats[0]!r}"
|
|
197
|
+
if unparsed:
|
|
198
|
+
base += f"; {unparsed} value(s) unparseable"
|
|
199
|
+
if ambiguous:
|
|
200
|
+
base += f" (day/month order ambiguous — assuming {'day' if dayfirst else 'month'}-first)"
|
|
201
|
+
return base
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
__all__ = ["detect_dates"]
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
"""Duplicate-row detection — exact and fuzzy.
|
|
2
|
+
|
|
3
|
+
Exact full-row duplicates are safe to drop, so we propose a ``dedup`` op with high
|
|
4
|
+
confidence. *Key* duplicates (two rows sharing an email/id but differing elsewhere)
|
|
5
|
+
and *fuzzy* near-duplicates on name-like columns are only *reported* with
|
|
6
|
+
reviewable pair evidence — silently collapsing them would lose data.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from difflib import SequenceMatcher
|
|
12
|
+
|
|
13
|
+
import pandas as pd
|
|
14
|
+
|
|
15
|
+
from .._util import normalize_key
|
|
16
|
+
from ..issues import Issues, _cap_examples
|
|
17
|
+
from ..types import Op, Severity
|
|
18
|
+
from .base import DetectorContext, detector
|
|
19
|
+
|
|
20
|
+
_KEY_HINTS = ("email", "id", "uuid", "guid")
|
|
21
|
+
_NAME_HINTS = ("name", "customer", "client", "company", "vendor", "title")
|
|
22
|
+
_FUZZY_THRESHOLD = 0.88
|
|
23
|
+
_FUZZY_MAX_ROWS = 200 # pairwise cost; bound for determinism + speed
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _key_columns(ctx: DetectorContext) -> list[str]:
|
|
27
|
+
keys: list[str] = []
|
|
28
|
+
for cp in ctx.profile.columns:
|
|
29
|
+
low = cp.name.lower()
|
|
30
|
+
if cp.semantic_type in ("email", "id") or any(h in low for h in _KEY_HINTS):
|
|
31
|
+
keys.append(cp.name)
|
|
32
|
+
return keys
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _fuzzy_columns(ctx: DetectorContext) -> list[str]:
|
|
36
|
+
cols: list[str] = []
|
|
37
|
+
for cp in ctx.profile.columns:
|
|
38
|
+
low = cp.name.lower()
|
|
39
|
+
if cp.semantic_type in ("text", "categorical", "id") and any(
|
|
40
|
+
h in low for h in _NAME_HINTS
|
|
41
|
+
):
|
|
42
|
+
cols.append(cp.name)
|
|
43
|
+
return cols
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _fuzzy_pairs(df: pd.DataFrame, column: str) -> list[dict]:
|
|
47
|
+
"""Return near-duplicate (i, j, score, values) pairs, sorted deterministically."""
|
|
48
|
+
if column not in df.columns or len(df) < 2:
|
|
49
|
+
return []
|
|
50
|
+
# Cap pairwise work; take a stable head so results don't depend on shuffle.
|
|
51
|
+
work = df.head(_FUZZY_MAX_ROWS)
|
|
52
|
+
# Use positional ids (0..n-1) not the index label — labels may be strings,
|
|
53
|
+
# Timestamps, or anything else, and int(label) would crash on them.
|
|
54
|
+
values = [
|
|
55
|
+
(pos, str(v).strip())
|
|
56
|
+
for pos, (_idx, v) in enumerate(work[column].items())
|
|
57
|
+
if pd.notna(v)
|
|
58
|
+
]
|
|
59
|
+
pairs: list[dict] = []
|
|
60
|
+
for i in range(len(values)):
|
|
61
|
+
idx_i, a = values[i]
|
|
62
|
+
if not a:
|
|
63
|
+
continue
|
|
64
|
+
na = normalize_key(a)
|
|
65
|
+
for j in range(i + 1, len(values)):
|
|
66
|
+
idx_j, b = values[j]
|
|
67
|
+
if not b or a == b:
|
|
68
|
+
continue
|
|
69
|
+
# Skip exact-normalized matches — those are category/casing, not fuzzy.
|
|
70
|
+
if na == normalize_key(b):
|
|
71
|
+
continue
|
|
72
|
+
score = SequenceMatcher(None, a.casefold(), b.casefold()).ratio()
|
|
73
|
+
if score >= _FUZZY_THRESHOLD:
|
|
74
|
+
pairs.append(
|
|
75
|
+
{
|
|
76
|
+
"rows": [idx_i, idx_j],
|
|
77
|
+
"values": [a, b],
|
|
78
|
+
"score": round(score, 3),
|
|
79
|
+
"proposal": {"keep": idx_i, "drop": idx_j},
|
|
80
|
+
}
|
|
81
|
+
)
|
|
82
|
+
pairs.sort(key=lambda p: (-p["score"], p["rows"][0], p["rows"][1]))
|
|
83
|
+
return pairs
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@detector("dedup", scope="frame", priority=80)
|
|
87
|
+
def detect_duplicates(df: pd.DataFrame, ctx: DetectorContext) -> Issues:
|
|
88
|
+
"""Propose dropping exact duplicate rows; report key/fuzzy duplicates for review."""
|
|
89
|
+
issues = Issues()
|
|
90
|
+
if len(df) == 0:
|
|
91
|
+
return issues
|
|
92
|
+
|
|
93
|
+
# Reuse the profiler's exact-duplicate count instead of hashing the whole frame
|
|
94
|
+
# a second time (the profiler already computed df.duplicated() for every clean).
|
|
95
|
+
exact = int(ctx.profile.duplicate_row_count)
|
|
96
|
+
if exact:
|
|
97
|
+
issues.add(
|
|
98
|
+
"duplicate_rows",
|
|
99
|
+
f"{exact} exact duplicate row(s)",
|
|
100
|
+
severity=Severity.WARNING,
|
|
101
|
+
confidence=1.0,
|
|
102
|
+
evidence={"count": exact},
|
|
103
|
+
ops=[Op("dedup", {"keep": "first"})],
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
# Key duplicates beyond the exact ones — report, don't auto-merge.
|
|
107
|
+
for key in _key_columns(ctx):
|
|
108
|
+
if key not in df.columns:
|
|
109
|
+
continue
|
|
110
|
+
dupe_mask = df[key].dropna().duplicated(keep=False)
|
|
111
|
+
n = int(dupe_mask.sum())
|
|
112
|
+
if n > exact:
|
|
113
|
+
examples = (
|
|
114
|
+
df.loc[df[key].duplicated(keep=False), key]
|
|
115
|
+
.dropna()
|
|
116
|
+
.drop_duplicates()
|
|
117
|
+
.head(5)
|
|
118
|
+
.tolist()
|
|
119
|
+
)
|
|
120
|
+
issues.add(
|
|
121
|
+
"duplicate_keys",
|
|
122
|
+
f"Column {key!r} has {n} row(s) sharing a value that should be unique",
|
|
123
|
+
severity=Severity.WARNING,
|
|
124
|
+
column=key,
|
|
125
|
+
confidence=0.6,
|
|
126
|
+
evidence={"count": n, "examples": _cap_examples(examples)},
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
# Fuzzy near-duplicates on name-like columns — reviewable merge proposals only.
|
|
130
|
+
for col in _fuzzy_columns(ctx):
|
|
131
|
+
pairs = _fuzzy_pairs(df, col)
|
|
132
|
+
if not pairs:
|
|
133
|
+
continue
|
|
134
|
+
issues.add(
|
|
135
|
+
"fuzzy_duplicates",
|
|
136
|
+
f"{len(pairs)} near-duplicate pair(s) in {col!r} (fuzzy match ≥ {_FUZZY_THRESHOLD:.0%})",
|
|
137
|
+
severity=Severity.INFO,
|
|
138
|
+
column=col,
|
|
139
|
+
confidence=0.55, # below auto threshold — review mode surfaces it
|
|
140
|
+
evidence={
|
|
141
|
+
"count": len(pairs),
|
|
142
|
+
"threshold": _FUZZY_THRESHOLD,
|
|
143
|
+
"pairs": pairs[:10],
|
|
144
|
+
},
|
|
145
|
+
# No ops — merge is a human decision; evidence carries keep/drop proposals.
|
|
146
|
+
)
|
|
147
|
+
return issues
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
__all__ = ["detect_duplicates"]
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""Missing values — real and disguised.
|
|
2
|
+
|
|
3
|
+
Two distinct jobs:
|
|
4
|
+
|
|
5
|
+
* **Disguised nulls** (``"NA"``, ``"-"``, ``"unknown"``) are a *fix*: we propose a
|
|
6
|
+
``to_na`` op to turn them into real NaN so downstream parsing and counts are
|
|
7
|
+
honest.
|
|
8
|
+
* **Actually-missing values** are *reported, never imputed*. CleanFrame will tell
|
|
9
|
+
you a column is 40% empty; it will not silently invent values. (Users who want
|
|
10
|
+
imputation add an explicit ``fill_na`` op themselves.)
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import pandas as pd
|
|
16
|
+
|
|
17
|
+
from .._util import is_string_like, looks_like_code_values, sample_non_null
|
|
18
|
+
from ..issues import Issues
|
|
19
|
+
from ..ops import DEFAULT_NA_TOKENS
|
|
20
|
+
from ..types import Op, Severity
|
|
21
|
+
from .base import DetectorContext, detector
|
|
22
|
+
|
|
23
|
+
_NA_LOOKUP = {t.strip().casefold() for t in DEFAULT_NA_TOKENS if t}
|
|
24
|
+
#: Tokens that are frequently *legitimate* data (an allergy of "none", a status of
|
|
25
|
+
#: "unknown", a placeholder "-"). We surface these for review but do NOT auto-convert
|
|
26
|
+
#: them below strict review, so a real category is never silently turned into NaN.
|
|
27
|
+
_AMBIGUOUS_NA = {"none", "nil", "-", "--", "?", "unknown"}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@detector("nulls", priority=20)
|
|
31
|
+
def detect_nulls(series: pd.Series, ctx: DetectorContext) -> Issues:
|
|
32
|
+
"""Convert disguised nulls to NaN; report genuinely-missing and structural oddities."""
|
|
33
|
+
issues = Issues()
|
|
34
|
+
cp = ctx.column_profile
|
|
35
|
+
if cp is None:
|
|
36
|
+
return issues
|
|
37
|
+
|
|
38
|
+
# 1) Disguised nulls in string columns -> fixable.
|
|
39
|
+
if is_string_like(series):
|
|
40
|
+
sample = sample_non_null(series)
|
|
41
|
+
disguised: dict[str, int] = {}
|
|
42
|
+
for v in sample:
|
|
43
|
+
if isinstance(v, str) and v.strip().casefold() in _NA_LOOKUP:
|
|
44
|
+
disguised[v] = disguised.get(v, 0) + 1
|
|
45
|
+
if disguised:
|
|
46
|
+
ambiguous_tokens = set(_AMBIGUOUS_NA)
|
|
47
|
+
if looks_like_code_values(sample):
|
|
48
|
+
ambiguous_tokens |= {"na", "nan", "nil", "none"}
|
|
49
|
+
unambiguous = sorted(
|
|
50
|
+
v for v in disguised if v.strip().casefold() not in ambiguous_tokens
|
|
51
|
+
)
|
|
52
|
+
ambiguous = sorted(v for v in disguised if v.strip().casefold() in ambiguous_tokens)
|
|
53
|
+
# Unambiguous null tokens ('', 'n/a', 'null', …) — safe to convert.
|
|
54
|
+
if unambiguous:
|
|
55
|
+
n = sum(disguised[t] for t in unambiguous)
|
|
56
|
+
issues.add(
|
|
57
|
+
"disguised_nulls",
|
|
58
|
+
f"{n} value(s) are disguised nulls ({', '.join(map(repr, unambiguous[:4]))})",
|
|
59
|
+
severity=Severity.WARNING,
|
|
60
|
+
confidence=1.0,
|
|
61
|
+
evidence={"count": n, "tokens": {t: disguised[t] for t in unambiguous}},
|
|
62
|
+
ops=[Op("to_na", {"tokens": unambiguous})],
|
|
63
|
+
)
|
|
64
|
+
# Ambiguous tokens that may be legitimate data — report, don't auto-convert.
|
|
65
|
+
if ambiguous:
|
|
66
|
+
n = sum(disguised[t] for t in ambiguous)
|
|
67
|
+
issues.add(
|
|
68
|
+
"disguised_nulls",
|
|
69
|
+
f"{n} value(s) may be disguised nulls "
|
|
70
|
+
f"({', '.join(map(repr, ambiguous[:4]))}) — review before converting",
|
|
71
|
+
severity=Severity.INFO,
|
|
72
|
+
confidence=0.45, # below the review threshold: surfaced, not auto-applied
|
|
73
|
+
evidence={"count": n, "tokens": {t: disguised[t] for t in ambiguous}, "ambiguous": True},
|
|
74
|
+
ops=[Op("to_na", {"tokens": ambiguous})],
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
# 2) Genuinely missing values -> report only.
|
|
78
|
+
if cp.null_count:
|
|
79
|
+
sev = Severity.WARNING if cp.null_fraction >= 0.2 else Severity.INFO
|
|
80
|
+
issues.add(
|
|
81
|
+
"missing_values",
|
|
82
|
+
f"{cp.null_count} missing value(s) ({cp.null_fraction:.0%} of the column)",
|
|
83
|
+
severity=sev,
|
|
84
|
+
confidence=1.0,
|
|
85
|
+
evidence={"null_count": cp.null_count, "null_fraction": round(cp.null_fraction, 4)},
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
# 3) Structural oddities worth surfacing.
|
|
89
|
+
if cp.count == 0:
|
|
90
|
+
issues.add(
|
|
91
|
+
"empty_column",
|
|
92
|
+
"Column is entirely empty",
|
|
93
|
+
severity=Severity.WARNING,
|
|
94
|
+
confidence=1.0,
|
|
95
|
+
evidence={},
|
|
96
|
+
)
|
|
97
|
+
elif cp.is_constant:
|
|
98
|
+
only = cp.sample_values[0] if cp.sample_values else None
|
|
99
|
+
issues.add(
|
|
100
|
+
"constant_column",
|
|
101
|
+
f"Column has a single value throughout ({only!r})",
|
|
102
|
+
severity=Severity.INFO,
|
|
103
|
+
confidence=1.0,
|
|
104
|
+
evidence={"value": only},
|
|
105
|
+
)
|
|
106
|
+
return issues
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
__all__ = ["detect_nulls"]
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Outlier detection — flagged with evidence, never auto-"fixed".
|
|
2
|
+
|
|
3
|
+
Uses a classic Tukey IQR fence on numeric columns. Findings carry the fence
|
|
4
|
+
bounds and example values so a human (or a downstream validator) can decide;
|
|
5
|
+
no ops are attached, matching the README / CONTRIBUTING invariant that outliers
|
|
6
|
+
are detected, never silently corrected.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import pandas as pd
|
|
12
|
+
|
|
13
|
+
from ..issues import Issues, _cap_examples
|
|
14
|
+
from ..types import Severity
|
|
15
|
+
from .base import DetectorContext, detector
|
|
16
|
+
|
|
17
|
+
_MIN_VALUES = 8
|
|
18
|
+
_IQR_K = 1.5
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@detector("outliers", priority=70)
|
|
22
|
+
def detect_outliers(series: pd.Series, ctx: DetectorContext) -> Issues:
|
|
23
|
+
"""Flag numeric outliers via the IQR fence. Report only — never propose a fix."""
|
|
24
|
+
issues = Issues()
|
|
25
|
+
cp = ctx.column_profile
|
|
26
|
+
if cp is None or cp.count < _MIN_VALUES:
|
|
27
|
+
return issues
|
|
28
|
+
if cp.semantic_type not in ("integer", "float", "currency"):
|
|
29
|
+
# Also accept already-numeric dtypes even if semantic type is currency-as-float.
|
|
30
|
+
if not (
|
|
31
|
+
pd.api.types.is_numeric_dtype(series.dtype)
|
|
32
|
+
and not pd.api.types.is_bool_dtype(series.dtype)
|
|
33
|
+
):
|
|
34
|
+
return issues
|
|
35
|
+
|
|
36
|
+
numeric = pd.to_numeric(series, errors="coerce").dropna()
|
|
37
|
+
if len(numeric) < _MIN_VALUES:
|
|
38
|
+
return issues
|
|
39
|
+
|
|
40
|
+
q1 = float(numeric.quantile(0.25))
|
|
41
|
+
q3 = float(numeric.quantile(0.75))
|
|
42
|
+
iqr = q3 - q1
|
|
43
|
+
if iqr == 0:
|
|
44
|
+
return issues
|
|
45
|
+
low = q1 - _IQR_K * iqr
|
|
46
|
+
high = q3 + _IQR_K * iqr
|
|
47
|
+
mask = (numeric < low) | (numeric > high)
|
|
48
|
+
n = int(mask.sum())
|
|
49
|
+
if n == 0:
|
|
50
|
+
return issues
|
|
51
|
+
|
|
52
|
+
# Slice to the cap BEFORE materialising — a column with millions of outliers must
|
|
53
|
+
# not build a giant throwaway Python list just to keep 5 examples.
|
|
54
|
+
examples = numeric[mask].head(5).tolist()
|
|
55
|
+
issues.add(
|
|
56
|
+
"outliers",
|
|
57
|
+
f"{n} outlier value(s) outside IQR fence [{low:.4g}, {high:.4g}]",
|
|
58
|
+
severity=Severity.INFO,
|
|
59
|
+
confidence=0.85,
|
|
60
|
+
evidence={
|
|
61
|
+
"count": n,
|
|
62
|
+
"low": low,
|
|
63
|
+
"high": high,
|
|
64
|
+
"q1": q1,
|
|
65
|
+
"q3": q3,
|
|
66
|
+
"examples": _cap_examples(examples),
|
|
67
|
+
},
|
|
68
|
+
# Intentionally no ops — outliers are never auto-fixed.
|
|
69
|
+
)
|
|
70
|
+
return issues
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
__all__ = ["detect_outliers"]
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
"""Map a messy file onto a target schema — with confidence scores.
|
|
2
|
+
|
|
3
|
+
Runs only when the caller supplied a target schema. Performs a deterministic,
|
|
4
|
+
greedy 1:1 assignment of schema columns to source columns using fuzzy name
|
|
5
|
+
similarity nudged by type compatibility, then emits rename proposals (strong
|
|
6
|
+
matches), review-flagged weak matches, and reports for schema columns with no
|
|
7
|
+
home and source columns with no schema slot.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import pandas as pd
|
|
13
|
+
|
|
14
|
+
from .._util import similarity
|
|
15
|
+
from ..issues import Issues
|
|
16
|
+
from ..types import Severity
|
|
17
|
+
from .base import DetectorContext, detector
|
|
18
|
+
|
|
19
|
+
#: Which source semantic types satisfy a given schema dtype (None = anything).
|
|
20
|
+
_COMPAT: dict[str, set[str]] = {
|
|
21
|
+
"integer": {"integer", "float", "currency"},
|
|
22
|
+
"float": {"float", "integer", "currency"},
|
|
23
|
+
"currency": {"currency", "float", "integer"},
|
|
24
|
+
"date": {"date", "datetime"},
|
|
25
|
+
"datetime": {"datetime", "date"},
|
|
26
|
+
"email": {"email", "text", "id"},
|
|
27
|
+
"phone": {"phone", "text"},
|
|
28
|
+
"url": {"url", "text"},
|
|
29
|
+
"category": {"categorical", "text", "id"},
|
|
30
|
+
"boolean": {"boolean"},
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
_STRONG = 0.6
|
|
34
|
+
_WEAK = 0.4
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _type_ok(schema_dtype: str, source_semantic: str) -> bool:
|
|
38
|
+
allowed = _COMPAT.get(schema_dtype)
|
|
39
|
+
return True if allowed is None else source_semantic in allowed
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@detector("schema_mapping", scope="frame", priority=5, requires_schema=True)
|
|
43
|
+
def detect_schema_mapping(df: pd.DataFrame, ctx: DetectorContext) -> Issues:
|
|
44
|
+
"""Map source columns onto the target schema by fuzzy name + type, with confidence."""
|
|
45
|
+
issues = Issues()
|
|
46
|
+
schema = ctx.schema
|
|
47
|
+
if schema is None:
|
|
48
|
+
return issues
|
|
49
|
+
|
|
50
|
+
sources = [str(c) for c in df.columns]
|
|
51
|
+
used: set[str] = set()
|
|
52
|
+
|
|
53
|
+
for scol in schema.columns:
|
|
54
|
+
scored: list[tuple[float, float, str]] = []
|
|
55
|
+
for src in sources:
|
|
56
|
+
if src in used:
|
|
57
|
+
continue
|
|
58
|
+
sim = similarity(scol.name, src)
|
|
59
|
+
sp = ctx.profile.column(src)
|
|
60
|
+
compat = _type_ok(scol.dtype, sp.semantic_type) if sp else True
|
|
61
|
+
score = 1.0 if (src == scol.name or src in (scol.aliases or [])) else sim
|
|
62
|
+
if not compat:
|
|
63
|
+
score -= 0.15
|
|
64
|
+
scored.append((score, sim, src))
|
|
65
|
+
|
|
66
|
+
if not scored:
|
|
67
|
+
issues.add(
|
|
68
|
+
"missing_schema_column",
|
|
69
|
+
f"Schema column {scol.name!r} has no candidate in the data",
|
|
70
|
+
severity=Severity.WARNING,
|
|
71
|
+
confidence=1.0,
|
|
72
|
+
evidence={"target": scol.name},
|
|
73
|
+
)
|
|
74
|
+
continue
|
|
75
|
+
|
|
76
|
+
scored.sort(key=lambda t: (-t[0], t[2])) # best score, then name asc -> deterministic
|
|
77
|
+
best_score, best_sim, best_src = scored[0]
|
|
78
|
+
|
|
79
|
+
if best_score >= _STRONG:
|
|
80
|
+
used.add(best_src)
|
|
81
|
+
if best_src != scol.name:
|
|
82
|
+
issues.add(
|
|
83
|
+
"schema_mapping",
|
|
84
|
+
f"Map {best_src!r} → {scol.name!r} ({best_sim:.0%} name match)",
|
|
85
|
+
severity=Severity.INFO,
|
|
86
|
+
column=best_src,
|
|
87
|
+
confidence=round(best_sim, 3),
|
|
88
|
+
evidence={"target": scol.name, "score": round(best_sim, 3)},
|
|
89
|
+
rename_to=scol.name,
|
|
90
|
+
)
|
|
91
|
+
elif best_score >= _WEAK:
|
|
92
|
+
used.add(best_src)
|
|
93
|
+
issues.add(
|
|
94
|
+
"weak_schema_match",
|
|
95
|
+
f"{best_src!r} only weakly matches schema column {scol.name!r} ({best_sim:.0%})",
|
|
96
|
+
severity=Severity.WARNING,
|
|
97
|
+
column=best_src,
|
|
98
|
+
confidence=round(best_sim, 3),
|
|
99
|
+
evidence={"target": scol.name, "score": round(best_sim, 3)},
|
|
100
|
+
rename_to=scol.name,
|
|
101
|
+
)
|
|
102
|
+
else:
|
|
103
|
+
issues.add(
|
|
104
|
+
"missing_schema_column",
|
|
105
|
+
f"Schema column {scol.name!r} has no good match "
|
|
106
|
+
f"(closest {best_src!r} at {best_sim:.0%})",
|
|
107
|
+
severity=Severity.WARNING,
|
|
108
|
+
confidence=1.0,
|
|
109
|
+
evidence={"target": scol.name, "closest": best_src, "score": round(best_sim, 3)},
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
for src in sources:
|
|
113
|
+
if src not in used:
|
|
114
|
+
issues.add(
|
|
115
|
+
"extra_source_column",
|
|
116
|
+
f"Source column {src!r} is not in the target schema",
|
|
117
|
+
severity=Severity.INFO,
|
|
118
|
+
column=src,
|
|
119
|
+
confidence=1.0,
|
|
120
|
+
evidence={},
|
|
121
|
+
)
|
|
122
|
+
return issues
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
__all__ = ["detect_schema_mapping"]
|