cleanframe-engine 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanframe/__init__.py +169 -0
- cleanframe/__main__.py +5 -0
- cleanframe/_util.py +438 -0
- cleanframe/_version.py +1 -0
- cleanframe/api.py +559 -0
- cleanframe/cli.py +688 -0
- cleanframe/codegen.py +666 -0
- cleanframe/dataio.py +506 -0
- cleanframe/detectors/__init__.py +43 -0
- cleanframe/detectors/base.py +221 -0
- cleanframe/detectors/categories.py +199 -0
- cleanframe/detectors/contacts.py +105 -0
- cleanframe/detectors/currency.py +112 -0
- cleanframe/detectors/dates.py +204 -0
- cleanframe/detectors/dedup.py +150 -0
- cleanframe/detectors/nulls.py +109 -0
- cleanframe/detectors/outliers.py +73 -0
- cleanframe/detectors/schema_mapping.py +125 -0
- cleanframe/detectors/text.py +105 -0
- cleanframe/detectors/units.py +86 -0
- cleanframe/diff.py +369 -0
- cleanframe/drift.py +283 -0
- cleanframe/errors.py +66 -0
- cleanframe/executor.py +229 -0
- cleanframe/fingerprint.py +83 -0
- cleanframe/issues.py +186 -0
- cleanframe/llm.py +811 -0
- cleanframe/ops.py +1245 -0
- cleanframe/planner.py +353 -0
- cleanframe/profile.py +413 -0
- cleanframe/py.typed +1 -0
- cleanframe/quality.py +81 -0
- cleanframe/readfix.py +160 -0
- cleanframe/recipe.py +398 -0
- cleanframe/report.py +345 -0
- cleanframe/result.py +144 -0
- cleanframe/schema.py +259 -0
- cleanframe/streaming.py +354 -0
- cleanframe/types.py +119 -0
- cleanframe/validate.py +363 -0
- cleanframe/workbook.py +370 -0
- cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
- cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
- cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
- cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
- cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
"""The detector plugin system.
|
|
2
|
+
|
|
3
|
+
A **detector** inspects data and returns :class:`~cleanframe.issues.Issues`, each
|
|
4
|
+
optionally carrying a :class:`~cleanframe.issues.Proposal` (the fix). Detectors are
|
|
5
|
+
the extension point CleanFrame is built around — the community owns the long tail
|
|
6
|
+
of messy-data weirdness by writing ~15-line detectors::
|
|
7
|
+
|
|
8
|
+
@cf.detector("iban")
|
|
9
|
+
def detect_iban(series):
|
|
10
|
+
issues = cf.Issues()
|
|
11
|
+
...
|
|
12
|
+
return issues
|
|
13
|
+
|
|
14
|
+
Signatures are flexible. A detector takes the data object for its scope
|
|
15
|
+
(``series`` for column detectors, ``df`` for frame detectors) and, optionally, a
|
|
16
|
+
second :class:`DetectorContext` argument when it needs the column name, the
|
|
17
|
+
profile, or a target schema. The runner introspects the signature and passes what
|
|
18
|
+
you ask for, so the minimal form above just works.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import inspect
|
|
24
|
+
from collections.abc import Callable, Iterable
|
|
25
|
+
from dataclasses import dataclass
|
|
26
|
+
from typing import Any
|
|
27
|
+
|
|
28
|
+
import pandas as pd
|
|
29
|
+
|
|
30
|
+
from ..issues import Issue, Issues
|
|
31
|
+
from ..profile import ColumnProfile, DataFrameProfile
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class DetectorContext:
|
|
36
|
+
"""Everything a detector might want beyond the raw data object.
|
|
37
|
+
|
|
38
|
+
Column detectors get ``column``/``series``/``column_profile`` set; frame
|
|
39
|
+
detectors get them as ``None``. ``schema`` is present only when the caller
|
|
40
|
+
passed a target schema. ``options`` carries user knobs (region for phones, a
|
|
41
|
+
seed alias map for categories, thresholds, …).
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
df: pd.DataFrame
|
|
45
|
+
profile: DataFrameProfile
|
|
46
|
+
column: str | None = None
|
|
47
|
+
series: pd.Series | None = None
|
|
48
|
+
column_profile: ColumnProfile | None = None
|
|
49
|
+
schema: Any | None = None
|
|
50
|
+
options: dict[str, Any] | None = None
|
|
51
|
+
|
|
52
|
+
def option(self, key: str, default: Any = None) -> Any:
|
|
53
|
+
return (self.options or {}).get(key, default)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass
|
|
57
|
+
class DetectorSpec:
|
|
58
|
+
name: str
|
|
59
|
+
func: Callable[..., Any]
|
|
60
|
+
scope: str # "column" | "frame"
|
|
61
|
+
priority: int
|
|
62
|
+
requires_schema: bool
|
|
63
|
+
wants_ctx: bool
|
|
64
|
+
doc: str = ""
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
DETECTOR_REGISTRY: dict[str, DetectorSpec] = {}
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _wants_ctx(func: Callable) -> bool:
|
|
71
|
+
"""Does this detector accept a second (context) argument?"""
|
|
72
|
+
try:
|
|
73
|
+
params = [
|
|
74
|
+
p
|
|
75
|
+
for p in inspect.signature(func).parameters.values()
|
|
76
|
+
if p.kind in (p.POSITIONAL_ONLY, p.POSITIONAL_OR_KEYWORD, p.KEYWORD_ONLY)
|
|
77
|
+
]
|
|
78
|
+
except (TypeError, ValueError): # pragma: no cover - builtins without signatures
|
|
79
|
+
return False
|
|
80
|
+
if any(p.name == "ctx" for p in params):
|
|
81
|
+
return True
|
|
82
|
+
return len(params) >= 2
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def detector(
|
|
86
|
+
name: str,
|
|
87
|
+
*,
|
|
88
|
+
scope: str = "column",
|
|
89
|
+
priority: int = 100,
|
|
90
|
+
requires_schema: bool = False,
|
|
91
|
+
) -> Callable[[Callable], Callable]:
|
|
92
|
+
"""Register a detector under ``name``.
|
|
93
|
+
|
|
94
|
+
Parameters
|
|
95
|
+
----------
|
|
96
|
+
scope:
|
|
97
|
+
``"column"`` (called once per column with its Series) or ``"frame"``
|
|
98
|
+
(called once with the whole DataFrame).
|
|
99
|
+
priority:
|
|
100
|
+
Lower runs earlier. Only affects the *order issues are reported*; the
|
|
101
|
+
planner orders the resulting ops canonically, so priority is about
|
|
102
|
+
presentation, not correctness.
|
|
103
|
+
requires_schema:
|
|
104
|
+
If true, the detector is skipped unless a target schema was supplied.
|
|
105
|
+
"""
|
|
106
|
+
|
|
107
|
+
if scope not in ("column", "frame"):
|
|
108
|
+
raise ValueError(f"scope must be 'column' or 'frame', got {scope!r}")
|
|
109
|
+
|
|
110
|
+
def decorator(func: Callable) -> Callable:
|
|
111
|
+
if name in DETECTOR_REGISTRY:
|
|
112
|
+
raise ValueError(f"Detector {name!r} is already registered.")
|
|
113
|
+
DETECTOR_REGISTRY[name] = DetectorSpec(
|
|
114
|
+
name=name,
|
|
115
|
+
func=func,
|
|
116
|
+
scope=scope,
|
|
117
|
+
priority=priority,
|
|
118
|
+
requires_schema=requires_schema,
|
|
119
|
+
wants_ctx=_wants_ctx(func),
|
|
120
|
+
doc=func.__doc__ or "",
|
|
121
|
+
)
|
|
122
|
+
return func
|
|
123
|
+
|
|
124
|
+
return decorator
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def unregister_detector(name: str) -> None:
|
|
128
|
+
"""Remove a detector (used by tests and for overriding a built-in)."""
|
|
129
|
+
DETECTOR_REGISTRY.pop(name, None)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def list_detectors() -> list[str]:
|
|
133
|
+
return sorted(DETECTOR_REGISTRY)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _normalize_result(result: Any) -> Iterable[Issue]:
|
|
137
|
+
if result is None:
|
|
138
|
+
return []
|
|
139
|
+
if isinstance(result, Issues):
|
|
140
|
+
return list(result)
|
|
141
|
+
if isinstance(result, Issue):
|
|
142
|
+
return [result]
|
|
143
|
+
if isinstance(result, (list, tuple)):
|
|
144
|
+
return [r for r in result if isinstance(r, Issue)]
|
|
145
|
+
raise TypeError(
|
|
146
|
+
f"A detector must return Issues / Issue / list / None, got {type(result).__name__}."
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _invoke(spec: DetectorSpec, data: Any, ctx: DetectorContext) -> Iterable[Issue]:
|
|
151
|
+
result = spec.func(data, ctx) if spec.wants_ctx else spec.func(data)
|
|
152
|
+
return _normalize_result(result)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def run_detectors(
|
|
156
|
+
df: pd.DataFrame,
|
|
157
|
+
*,
|
|
158
|
+
profile: DataFrameProfile | None = None,
|
|
159
|
+
schema: Any | None = None,
|
|
160
|
+
options: dict[str, Any] | None = None,
|
|
161
|
+
only: Iterable[str] | None = None,
|
|
162
|
+
) -> Issues:
|
|
163
|
+
"""Run every applicable detector and return the aggregated, stamped issues.
|
|
164
|
+
|
|
165
|
+
Deterministic: detectors run in ``(priority, name)`` order and columns in frame
|
|
166
|
+
order. Each issue is stamped with its detector name (and column, for column
|
|
167
|
+
detectors) if the detector didn't set them.
|
|
168
|
+
"""
|
|
169
|
+
from .._util import ensure_string_columns
|
|
170
|
+
from ..profile import profile_dataframe # local import avoids a cycle at module load
|
|
171
|
+
|
|
172
|
+
df = ensure_string_columns(df)
|
|
173
|
+
profile = profile or profile_dataframe(df)
|
|
174
|
+
options = options or {}
|
|
175
|
+
only_set = set(only) if only is not None else None
|
|
176
|
+
issues = Issues()
|
|
177
|
+
|
|
178
|
+
specs = sorted(DETECTOR_REGISTRY.values(), key=lambda s: (s.priority, s.name))
|
|
179
|
+
for spec in specs:
|
|
180
|
+
if only_set is not None and spec.name not in only_set:
|
|
181
|
+
continue
|
|
182
|
+
if spec.requires_schema and schema is None:
|
|
183
|
+
continue
|
|
184
|
+
|
|
185
|
+
if spec.scope == "column":
|
|
186
|
+
for col in df.columns:
|
|
187
|
+
col_name = str(col)
|
|
188
|
+
ctx = DetectorContext(
|
|
189
|
+
df=df,
|
|
190
|
+
profile=profile,
|
|
191
|
+
column=col_name,
|
|
192
|
+
series=df[col],
|
|
193
|
+
column_profile=profile.column(col_name),
|
|
194
|
+
schema=schema,
|
|
195
|
+
options=options,
|
|
196
|
+
)
|
|
197
|
+
for issue in _invoke(spec, df[col], ctx):
|
|
198
|
+
issue.detector = issue.detector or spec.name
|
|
199
|
+
if issue.column is None:
|
|
200
|
+
issue.column = col_name
|
|
201
|
+
issues.append(issue)
|
|
202
|
+
else:
|
|
203
|
+
ctx = DetectorContext(
|
|
204
|
+
df=df, profile=profile, schema=schema, options=options
|
|
205
|
+
)
|
|
206
|
+
for issue in _invoke(spec, df, ctx):
|
|
207
|
+
issue.detector = issue.detector or spec.name
|
|
208
|
+
issues.append(issue)
|
|
209
|
+
|
|
210
|
+
return issues
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
__all__ = [
|
|
214
|
+
"DetectorContext",
|
|
215
|
+
"DetectorSpec",
|
|
216
|
+
"DETECTOR_REGISTRY",
|
|
217
|
+
"detector",
|
|
218
|
+
"unregister_detector",
|
|
219
|
+
"list_detectors",
|
|
220
|
+
"run_detectors",
|
|
221
|
+
]
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
"""Category canonicalisation.
|
|
2
|
+
|
|
3
|
+
Finds values that *mean the same thing* but are spelled differently and proposes a
|
|
4
|
+
``normalize_values`` mapping to a single canonical spelling. Two mechanisms, both
|
|
5
|
+
deterministic:
|
|
6
|
+
|
|
7
|
+
* **Exact-after-normalisation** clusters (``"Bengaluru"``/``"bengaluru "``/
|
|
8
|
+
``"BENGALURU"``) — high confidence.
|
|
9
|
+
* **Fuzzy typo** clusters (``"Banglore"`` vs ``"Bangalore"``) via a similarity
|
|
10
|
+
threshold — lower confidence.
|
|
11
|
+
|
|
12
|
+
Semantic abbreviations (``"BLR"`` → ``"Bangalore"``) are deliberately *out of
|
|
13
|
+
scope* for rules — that's domain knowledge for a human or the LLM planner. The
|
|
14
|
+
canonical spelling for each cluster is the most frequent original (ties broken
|
|
15
|
+
alphabetically), so results never depend on row order.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import re
|
|
21
|
+
|
|
22
|
+
import pandas as pd
|
|
23
|
+
|
|
24
|
+
from .._util import is_string_like, normalize_key, sample_non_null, similarity
|
|
25
|
+
from ..issues import Issues
|
|
26
|
+
from ..types import Op, Severity
|
|
27
|
+
from .base import DetectorContext, detector
|
|
28
|
+
|
|
29
|
+
#: Above this cardinality a column is treated as free text, not categories.
|
|
30
|
+
_MAX_CATEGORY_CARDINALITY = 60
|
|
31
|
+
_FUZZY_THRESHOLD = 0.86
|
|
32
|
+
#: A fuzzy typo cluster only merges when the variant is this much rarer than the
|
|
33
|
+
#: canonical it would fold into. This stops two *both-frequent* look-alikes
|
|
34
|
+
#: (``insured``/``uninsured``, ``activated``/``deactivated``) from collapsing —
|
|
35
|
+
#: a real typo is nearly always far rarer than the correct spelling.
|
|
36
|
+
_FUZZY_MERGE_MAX_RATIO = 0.34
|
|
37
|
+
_WS = re.compile(r"\s+")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _clean(value: str) -> str:
|
|
41
|
+
"""Whitespace-normalise a value so category clustering sees post-trim spellings.
|
|
42
|
+
|
|
43
|
+
Categories run *after* whitespace ops in a recipe, so clustering on the cleaned
|
|
44
|
+
form keeps the ``normalize_values`` map keys matching the data at that point and
|
|
45
|
+
prevents a whitespace-messy spelling from ever being chosen as canonical.
|
|
46
|
+
"""
|
|
47
|
+
return _WS.sub(" ", value).strip()
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _casing_rank(s: str) -> int:
|
|
51
|
+
"""Tie-break spellings: short CODE > Title > Mixed > lower > UPPER.
|
|
52
|
+
|
|
53
|
+
Short all-caps tokens (``CA``, ``NY``, ``INR``) are codes whose canonical form
|
|
54
|
+
*is* upper case; anything longer reads better title-cased.
|
|
55
|
+
"""
|
|
56
|
+
if s.isupper() and len(s.strip()) <= 3:
|
|
57
|
+
return 0
|
|
58
|
+
if s.istitle():
|
|
59
|
+
return 1
|
|
60
|
+
if not s.isupper() and not s.islower():
|
|
61
|
+
return 2
|
|
62
|
+
if s.islower():
|
|
63
|
+
return 3
|
|
64
|
+
return 4
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _canonical(spellings: dict[str, int]) -> str:
|
|
68
|
+
"""Canonical spelling: most frequent, then nicest casing, then alphabetical.
|
|
69
|
+
|
|
70
|
+
All three keys are deterministic, so the choice never depends on row order.
|
|
71
|
+
"""
|
|
72
|
+
return sorted(spellings.items(), key=lambda kv: (-kv[1], _casing_rank(kv[0]), kv[0]))[0][0]
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
#: A negated spelling is a different category, however similar it looks.
|
|
76
|
+
_NEGATION_PREFIXES = (
|
|
77
|
+
"un", "non", "in", "im", "ir", "il", "dis", "de", "re", "anti", "no", "not",
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _is_negation_pair(a: str, b: str) -> bool:
|
|
82
|
+
"""True when one spelling is the other plus a negation/repetition prefix."""
|
|
83
|
+
x, y = normalize_key(a), normalize_key(b)
|
|
84
|
+
if x == y:
|
|
85
|
+
return False
|
|
86
|
+
longer, shorter = (x, y) if len(x) > len(y) else (y, x)
|
|
87
|
+
if not shorter or not longer.endswith(shorter):
|
|
88
|
+
return False
|
|
89
|
+
return longer[: -len(shorter)] in _NEGATION_PREFIXES
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _cluster(counts: dict[str, int], seed_map: dict[str, str] | None) -> dict[str, str]:
|
|
93
|
+
"""Return a mapping ``{variant: canonical}`` covering only values that change."""
|
|
94
|
+
# 1) group by aggressive normalisation key
|
|
95
|
+
groups: dict[str, dict[str, int]] = {}
|
|
96
|
+
for value, n in counts.items():
|
|
97
|
+
key = normalize_key(value)
|
|
98
|
+
if not key:
|
|
99
|
+
continue # punctuation-only values ("-", "?") are not category variants
|
|
100
|
+
groups.setdefault(key, {})[value] = n
|
|
101
|
+
|
|
102
|
+
canon_of_group = {key: _canonical(spellings) for key, spellings in groups.items()}
|
|
103
|
+
|
|
104
|
+
# 2) fuzzy-merge whole groups whose canonical spellings are near-identical.
|
|
105
|
+
# Deterministic: iterate group keys in sorted order, attach to the first
|
|
106
|
+
# (largest, then alphabetically-first) existing cluster within threshold.
|
|
107
|
+
ordered = sorted(groups, key=lambda k: (-sum(groups[k].values()), k))
|
|
108
|
+
cluster_rep: dict[str, str] = {}
|
|
109
|
+
reps: list[str] = []
|
|
110
|
+
for key in ordered:
|
|
111
|
+
canon = canon_of_group[key]
|
|
112
|
+
key_count = sum(groups[key].values())
|
|
113
|
+
match = None
|
|
114
|
+
for rep in reps:
|
|
115
|
+
# Only fold this group into `rep` if it is BOTH near-identical in
|
|
116
|
+
# spelling AND rare relative to it — otherwise two distinct, comparably
|
|
117
|
+
# frequent categories would be silently merged (semantic inversion).
|
|
118
|
+
rep_count = sum(groups[rep].values())
|
|
119
|
+
if _is_negation_pair(canon, canon_of_group[rep]):
|
|
120
|
+
continue
|
|
121
|
+
if (
|
|
122
|
+
similarity(canon, canon_of_group[rep]) >= _FUZZY_THRESHOLD
|
|
123
|
+
and key_count <= rep_count * _FUZZY_MERGE_MAX_RATIO
|
|
124
|
+
):
|
|
125
|
+
match = rep
|
|
126
|
+
break
|
|
127
|
+
if match is None:
|
|
128
|
+
reps.append(key)
|
|
129
|
+
cluster_rep[key] = key
|
|
130
|
+
else:
|
|
131
|
+
cluster_rep[key] = match
|
|
132
|
+
|
|
133
|
+
# 3) build variant -> canonical, honouring any seed overrides first.
|
|
134
|
+
mapping: dict[str, str] = {}
|
|
135
|
+
for key, spellings in groups.items():
|
|
136
|
+
rep_key = cluster_rep[key]
|
|
137
|
+
canonical = canon_of_group[rep_key]
|
|
138
|
+
for value in spellings:
|
|
139
|
+
target = (seed_map or {}).get(value, canonical)
|
|
140
|
+
if value != target:
|
|
141
|
+
mapping[value] = target
|
|
142
|
+
# Sort keys so the emitted recipe is byte-identical regardless of input row
|
|
143
|
+
# order (the cleaned data is order-independent either way).
|
|
144
|
+
return dict(sorted(mapping.items()))
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
@detector("categories", priority=50)
|
|
148
|
+
def detect_categories(series: pd.Series, ctx: DetectorContext) -> Issues:
|
|
149
|
+
"""Cluster case/whitespace/typo variants of a category into one canonical value."""
|
|
150
|
+
issues = Issues()
|
|
151
|
+
cp = ctx.column_profile
|
|
152
|
+
if cp is None or cp.count == 0:
|
|
153
|
+
return issues
|
|
154
|
+
if not is_string_like(series):
|
|
155
|
+
return issues
|
|
156
|
+
# Only meaningful for low-cardinality columns.
|
|
157
|
+
if cp.unique_count > _MAX_CATEGORY_CARDINALITY or cp.semantic_type in (
|
|
158
|
+
"email", "phone", "url", "currency", "date", "datetime", "id",
|
|
159
|
+
):
|
|
160
|
+
return issues
|
|
161
|
+
|
|
162
|
+
counts: dict[str, int] = {}
|
|
163
|
+
for v in sample_non_null(series):
|
|
164
|
+
if isinstance(v, str):
|
|
165
|
+
cleaned = _clean(v)
|
|
166
|
+
counts[cleaned] = counts.get(cleaned, 0) + 1
|
|
167
|
+
if len(counts) < 2:
|
|
168
|
+
return issues
|
|
169
|
+
|
|
170
|
+
seed_map = ctx.option("category_map", {}).get(ctx.column) if ctx.option("category_map") else None
|
|
171
|
+
mapping = _cluster(counts, seed_map)
|
|
172
|
+
if not mapping:
|
|
173
|
+
return issues
|
|
174
|
+
|
|
175
|
+
distinct_before = len(counts)
|
|
176
|
+
distinct_after = len({counts_key if counts_key not in mapping else mapping[counts_key]
|
|
177
|
+
for counts_key in counts})
|
|
178
|
+
# Confidence: purely case/space merges are safe; fuzzy typo merges less so.
|
|
179
|
+
fuzzy = any(normalize_key(k) != normalize_key(v) for k, v in mapping.items())
|
|
180
|
+
confidence = 0.7 if fuzzy else 0.9
|
|
181
|
+
|
|
182
|
+
issues.add(
|
|
183
|
+
"category_variants",
|
|
184
|
+
f"{len(mapping)} value(s) are variants of other categories "
|
|
185
|
+
f"({distinct_before} → {distinct_after} distinct)",
|
|
186
|
+
severity=Severity.WARNING,
|
|
187
|
+
confidence=confidence,
|
|
188
|
+
evidence={
|
|
189
|
+
"mapping": mapping,
|
|
190
|
+
"distinct_before": distinct_before,
|
|
191
|
+
"distinct_after": distinct_after,
|
|
192
|
+
"fuzzy": fuzzy,
|
|
193
|
+
},
|
|
194
|
+
ops=[Op("normalize_values", {"map": mapping})],
|
|
195
|
+
)
|
|
196
|
+
return issues
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
__all__ = ["detect_categories"]
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""Email and phone detectors.
|
|
2
|
+
|
|
3
|
+
Emails get the safe, universally-correct normalisation (trim + lowercase) plus an
|
|
4
|
+
invalid-address count. Phones get best-effort separator/country-code normalisation
|
|
5
|
+
driven by the ``phone_country_code`` option. Neither drops data — validation rules
|
|
6
|
+
(added by the planner for these semantic types) decide what happens to values that
|
|
7
|
+
still fail.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import re
|
|
13
|
+
|
|
14
|
+
import pandas as pd
|
|
15
|
+
|
|
16
|
+
from .._util import sample_non_null
|
|
17
|
+
from ..issues import Issues, _cap_examples
|
|
18
|
+
from ..profile import EMAIL_RE, _name_hint
|
|
19
|
+
from ..types import Op, Severity
|
|
20
|
+
from .base import DetectorContext, detector
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@detector("emails", priority=45)
|
|
24
|
+
def detect_emails(series: pd.Series, ctx: DetectorContext) -> Issues:
|
|
25
|
+
"""Normalise emails (trim + lowercase) and flag invalid addresses."""
|
|
26
|
+
issues = Issues()
|
|
27
|
+
cp = ctx.column_profile
|
|
28
|
+
if cp is None or cp.count == 0:
|
|
29
|
+
return issues
|
|
30
|
+
if cp.semantic_type != "email" and not _name_hint(ctx.column or "", "email"):
|
|
31
|
+
return issues
|
|
32
|
+
|
|
33
|
+
values = [v for v in sample_non_null(series) if isinstance(v, str)]
|
|
34
|
+
if not values:
|
|
35
|
+
return issues
|
|
36
|
+
|
|
37
|
+
invalid = [v for v in values if not EMAIL_RE.match(v.strip().lower())]
|
|
38
|
+
needs_norm = [v for v in values if v != v.strip().lower()]
|
|
39
|
+
|
|
40
|
+
if needs_norm:
|
|
41
|
+
issues.add(
|
|
42
|
+
"email_normalization",
|
|
43
|
+
f"{len(needs_norm)} email(s) need trimming/lowercasing",
|
|
44
|
+
severity=Severity.INFO,
|
|
45
|
+
confidence=0.95,
|
|
46
|
+
evidence={"count": len(needs_norm), "examples": _cap_examples(needs_norm)},
|
|
47
|
+
ops=[Op("normalize_email")],
|
|
48
|
+
)
|
|
49
|
+
if invalid:
|
|
50
|
+
issues.add(
|
|
51
|
+
"invalid_emails",
|
|
52
|
+
f"{len(invalid)} value(s) are not valid email addresses",
|
|
53
|
+
severity=Severity.ERROR if len(invalid) / len(values) > 0.02 else Severity.WARNING,
|
|
54
|
+
confidence=1.0,
|
|
55
|
+
evidence={"count": len(invalid), "examples": _cap_examples(invalid)},
|
|
56
|
+
)
|
|
57
|
+
return issues
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@detector("phones", priority=45)
|
|
61
|
+
def detect_phones(series: pd.Series, ctx: DetectorContext) -> Issues:
|
|
62
|
+
"""Normalise phone-number formatting and flag implausible digit counts."""
|
|
63
|
+
issues = Issues()
|
|
64
|
+
cp = ctx.column_profile
|
|
65
|
+
if cp is None or cp.count == 0:
|
|
66
|
+
return issues
|
|
67
|
+
if cp.semantic_type != "phone" and not _name_hint(ctx.column or "", "phone"):
|
|
68
|
+
return issues
|
|
69
|
+
|
|
70
|
+
values = [v if isinstance(v, str) else str(v) for v in sample_non_null(series)]
|
|
71
|
+
if not values:
|
|
72
|
+
return issues
|
|
73
|
+
|
|
74
|
+
digit_counts = [len(re.sub(r"\D", "", v)) for v in values]
|
|
75
|
+
invalid = [v for v, d in zip(values, digit_counts, strict=False) if not (7 <= d <= 15)]
|
|
76
|
+
# "Needs normalisation" = contains separators or lacks a leading +.
|
|
77
|
+
messy = [v for v in values if re.search(r"[ ()\-.]", v) or not v.strip().startswith("+")]
|
|
78
|
+
|
|
79
|
+
country_code = ctx.option("phone_country_code") or ctx.option("region")
|
|
80
|
+
if messy:
|
|
81
|
+
params = {"default_country_code": country_code} if country_code else {}
|
|
82
|
+
issues.add(
|
|
83
|
+
"phone_normalization",
|
|
84
|
+
f"{len(messy)} phone number(s) have inconsistent formatting",
|
|
85
|
+
severity=Severity.INFO,
|
|
86
|
+
confidence=0.8,
|
|
87
|
+
evidence={
|
|
88
|
+
"count": len(messy),
|
|
89
|
+
"examples": _cap_examples(messy),
|
|
90
|
+
"country_code": country_code,
|
|
91
|
+
},
|
|
92
|
+
ops=[Op("normalize_phone", params)],
|
|
93
|
+
)
|
|
94
|
+
if invalid:
|
|
95
|
+
issues.add(
|
|
96
|
+
"invalid_phones",
|
|
97
|
+
f"{len(invalid)} value(s) have an implausible number of digits",
|
|
98
|
+
severity=Severity.WARNING,
|
|
99
|
+
confidence=0.9,
|
|
100
|
+
evidence={"count": len(invalid), "examples": _cap_examples(invalid)},
|
|
101
|
+
)
|
|
102
|
+
return issues
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
__all__ = ["detect_emails", "detect_phones"]
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""Currency / money-column detection.
|
|
2
|
+
|
|
3
|
+
Turns ``₹1,20,000`` / ``$1,200`` / ``1200 INR`` into a typed float. When a column
|
|
4
|
+
holds a single currency, the code is folded into the column name (``amount`` →
|
|
5
|
+
``amount_inr``, matching the README). When a column *mixes* currencies — where the
|
|
6
|
+
amount alone would be meaningless — it additionally splits out a ``*_currency``
|
|
7
|
+
column so no information is lost.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import re
|
|
13
|
+
|
|
14
|
+
import numpy as np
|
|
15
|
+
import pandas as pd
|
|
16
|
+
|
|
17
|
+
from .._util import sample_non_null, snake_case, token_set
|
|
18
|
+
from ..issues import Issues, _cap_examples
|
|
19
|
+
from ..ops import _detect_currency_scalar, _parse_number_scalar
|
|
20
|
+
from ..types import Op, Severity
|
|
21
|
+
from .base import DetectorContext, detector
|
|
22
|
+
|
|
23
|
+
_EU_GROUPED_RE = re.compile(r"\d{1,3}(?:\.\d{3})+,\d+")
|
|
24
|
+
_US_GROUPED_RE = re.compile(r"\d{1,3}(?:,\d{3})+")
|
|
25
|
+
_COMMA_DECIMAL_RE = re.compile(r"\d,\d{1,2}(?!\d)")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _decimal_convention(values: list[str]) -> dict[str, str]:
|
|
29
|
+
"""Infer European ``1.234,56`` grouping. Empty dict means the pandas default.
|
|
30
|
+
|
|
31
|
+
Parsing ``€1.200,50`` with the default convention yields 1.2005 while reporting
|
|
32
|
+
nothing unparseable, so the convention has to be decided from the values.
|
|
33
|
+
"""
|
|
34
|
+
eu_grouped = sum(1 for v in values if _EU_GROUPED_RE.search(v))
|
|
35
|
+
us_grouped = sum(1 for v in values if _US_GROUPED_RE.search(v))
|
|
36
|
+
if us_grouped:
|
|
37
|
+
return {}
|
|
38
|
+
if eu_grouped:
|
|
39
|
+
return {"decimal": ",", "thousands": "."}
|
|
40
|
+
comma_decimal = sum(1 for v in values if _COMMA_DECIMAL_RE.search(v))
|
|
41
|
+
if comma_decimal and not any("." in v for v in values):
|
|
42
|
+
return {"decimal": ",", "thousands": "."}
|
|
43
|
+
return {}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@detector("currency", priority=45)
|
|
47
|
+
def detect_currency(series: pd.Series, ctx: DetectorContext) -> Issues:
|
|
48
|
+
"""Detect money-as-text columns and propose parsing to a float (+ currency split)."""
|
|
49
|
+
issues = Issues()
|
|
50
|
+
cp = ctx.column_profile
|
|
51
|
+
if cp is None or cp.count == 0 or cp.semantic_type != "currency":
|
|
52
|
+
return issues
|
|
53
|
+
|
|
54
|
+
values = [v if isinstance(v, str) else str(v) for v in sample_non_null(series)]
|
|
55
|
+
codes = {c for c in (_detect_currency_scalar(v, None) for v in values) if isinstance(c, str)}
|
|
56
|
+
|
|
57
|
+
convention = _decimal_convention(values)
|
|
58
|
+
decimal = convention.get("decimal", ".")
|
|
59
|
+
thousands = convention.get("thousands", ",")
|
|
60
|
+
# How many non-null values fail to become a number? (report, don't hide.)
|
|
61
|
+
unparsed = sum(
|
|
62
|
+
1 for v in values if np.isnan(_parse_number_scalar(v, decimal, thousands, []))
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
snake = snake_case(ctx.column or series.name or "amount")
|
|
66
|
+
ops: list[Op] = []
|
|
67
|
+
rename_to: str | None = None
|
|
68
|
+
|
|
69
|
+
if len(codes) == 1:
|
|
70
|
+
code = next(iter(codes))
|
|
71
|
+
# Fold the currency into the name unless it's already there.
|
|
72
|
+
if code.lower() not in token_set(snake):
|
|
73
|
+
rename_to = f"{snake}_{code.lower()}"
|
|
74
|
+
else:
|
|
75
|
+
rename_to = snake
|
|
76
|
+
ops = [Op("parse_number", dict(convention))]
|
|
77
|
+
currency_note = f"single currency {code}"
|
|
78
|
+
else:
|
|
79
|
+
rename_to = snake
|
|
80
|
+
target = f"{snake}_currency"
|
|
81
|
+
ops = [Op("extract_currency", {"to": target}), Op("parse_number", dict(convention))]
|
|
82
|
+
currency_note = (
|
|
83
|
+
f"mixed currencies {sorted(codes)} — splitting out `{target}`"
|
|
84
|
+
if codes
|
|
85
|
+
else "no explicit currency code"
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
if rename_to == (ctx.column or series.name):
|
|
89
|
+
rename_to = None
|
|
90
|
+
|
|
91
|
+
sev = Severity.WARNING if unparsed else Severity.INFO
|
|
92
|
+
evidence = {
|
|
93
|
+
"currencies": sorted(codes),
|
|
94
|
+
"unparsed": unparsed,
|
|
95
|
+
"examples": _cap_examples(values),
|
|
96
|
+
}
|
|
97
|
+
if convention:
|
|
98
|
+
evidence["decimal_convention"] = "european"
|
|
99
|
+
issues.add(
|
|
100
|
+
"currency_format",
|
|
101
|
+
f"Money column stored as text ({currency_note})"
|
|
102
|
+
+ (f"; {unparsed} value(s) unparseable" if unparsed else ""),
|
|
103
|
+
severity=sev,
|
|
104
|
+
confidence=0.95,
|
|
105
|
+
evidence=evidence,
|
|
106
|
+
ops=ops,
|
|
107
|
+
rename_to=rename_to,
|
|
108
|
+
)
|
|
109
|
+
return issues
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
__all__ = ["detect_currency"]
|