cleanframe-engine 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. cleanframe/__init__.py +169 -0
  2. cleanframe/__main__.py +5 -0
  3. cleanframe/_util.py +438 -0
  4. cleanframe/_version.py +1 -0
  5. cleanframe/api.py +559 -0
  6. cleanframe/cli.py +688 -0
  7. cleanframe/codegen.py +666 -0
  8. cleanframe/dataio.py +506 -0
  9. cleanframe/detectors/__init__.py +43 -0
  10. cleanframe/detectors/base.py +221 -0
  11. cleanframe/detectors/categories.py +199 -0
  12. cleanframe/detectors/contacts.py +105 -0
  13. cleanframe/detectors/currency.py +112 -0
  14. cleanframe/detectors/dates.py +204 -0
  15. cleanframe/detectors/dedup.py +150 -0
  16. cleanframe/detectors/nulls.py +109 -0
  17. cleanframe/detectors/outliers.py +73 -0
  18. cleanframe/detectors/schema_mapping.py +125 -0
  19. cleanframe/detectors/text.py +105 -0
  20. cleanframe/detectors/units.py +86 -0
  21. cleanframe/diff.py +369 -0
  22. cleanframe/drift.py +283 -0
  23. cleanframe/errors.py +66 -0
  24. cleanframe/executor.py +229 -0
  25. cleanframe/fingerprint.py +83 -0
  26. cleanframe/issues.py +186 -0
  27. cleanframe/llm.py +811 -0
  28. cleanframe/ops.py +1245 -0
  29. cleanframe/planner.py +353 -0
  30. cleanframe/profile.py +413 -0
  31. cleanframe/py.typed +1 -0
  32. cleanframe/quality.py +81 -0
  33. cleanframe/readfix.py +160 -0
  34. cleanframe/recipe.py +398 -0
  35. cleanframe/report.py +345 -0
  36. cleanframe/result.py +144 -0
  37. cleanframe/schema.py +259 -0
  38. cleanframe/streaming.py +354 -0
  39. cleanframe/types.py +119 -0
  40. cleanframe/validate.py +363 -0
  41. cleanframe/workbook.py +370 -0
  42. cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
  43. cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
  44. cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
  45. cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
  46. cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,221 @@
1
+ """The detector plugin system.
2
+
3
+ A **detector** inspects data and returns :class:`~cleanframe.issues.Issues`, each
4
+ optionally carrying a :class:`~cleanframe.issues.Proposal` (the fix). Detectors are
5
+ the extension point CleanFrame is built around — the community owns the long tail
6
+ of messy-data weirdness by writing ~15-line detectors::
7
+
8
+ @cf.detector("iban")
9
+ def detect_iban(series):
10
+ issues = cf.Issues()
11
+ ...
12
+ return issues
13
+
14
+ Signatures are flexible. A detector takes the data object for its scope
15
+ (``series`` for column detectors, ``df`` for frame detectors) and, optionally, a
16
+ second :class:`DetectorContext` argument when it needs the column name, the
17
+ profile, or a target schema. The runner introspects the signature and passes what
18
+ you ask for, so the minimal form above just works.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import inspect
24
+ from collections.abc import Callable, Iterable
25
+ from dataclasses import dataclass
26
+ from typing import Any
27
+
28
+ import pandas as pd
29
+
30
+ from ..issues import Issue, Issues
31
+ from ..profile import ColumnProfile, DataFrameProfile
32
+
33
+
34
+ @dataclass
35
+ class DetectorContext:
36
+ """Everything a detector might want beyond the raw data object.
37
+
38
+ Column detectors get ``column``/``series``/``column_profile`` set; frame
39
+ detectors get them as ``None``. ``schema`` is present only when the caller
40
+ passed a target schema. ``options`` carries user knobs (region for phones, a
41
+ seed alias map for categories, thresholds, …).
42
+ """
43
+
44
+ df: pd.DataFrame
45
+ profile: DataFrameProfile
46
+ column: str | None = None
47
+ series: pd.Series | None = None
48
+ column_profile: ColumnProfile | None = None
49
+ schema: Any | None = None
50
+ options: dict[str, Any] | None = None
51
+
52
+ def option(self, key: str, default: Any = None) -> Any:
53
+ return (self.options or {}).get(key, default)
54
+
55
+
56
+ @dataclass
57
+ class DetectorSpec:
58
+ name: str
59
+ func: Callable[..., Any]
60
+ scope: str # "column" | "frame"
61
+ priority: int
62
+ requires_schema: bool
63
+ wants_ctx: bool
64
+ doc: str = ""
65
+
66
+
67
+ DETECTOR_REGISTRY: dict[str, DetectorSpec] = {}
68
+
69
+
70
+ def _wants_ctx(func: Callable) -> bool:
71
+ """Does this detector accept a second (context) argument?"""
72
+ try:
73
+ params = [
74
+ p
75
+ for p in inspect.signature(func).parameters.values()
76
+ if p.kind in (p.POSITIONAL_ONLY, p.POSITIONAL_OR_KEYWORD, p.KEYWORD_ONLY)
77
+ ]
78
+ except (TypeError, ValueError): # pragma: no cover - builtins without signatures
79
+ return False
80
+ if any(p.name == "ctx" for p in params):
81
+ return True
82
+ return len(params) >= 2
83
+
84
+
85
+ def detector(
86
+ name: str,
87
+ *,
88
+ scope: str = "column",
89
+ priority: int = 100,
90
+ requires_schema: bool = False,
91
+ ) -> Callable[[Callable], Callable]:
92
+ """Register a detector under ``name``.
93
+
94
+ Parameters
95
+ ----------
96
+ scope:
97
+ ``"column"`` (called once per column with its Series) or ``"frame"``
98
+ (called once with the whole DataFrame).
99
+ priority:
100
+ Lower runs earlier. Only affects the *order issues are reported*; the
101
+ planner orders the resulting ops canonically, so priority is about
102
+ presentation, not correctness.
103
+ requires_schema:
104
+ If true, the detector is skipped unless a target schema was supplied.
105
+ """
106
+
107
+ if scope not in ("column", "frame"):
108
+ raise ValueError(f"scope must be 'column' or 'frame', got {scope!r}")
109
+
110
+ def decorator(func: Callable) -> Callable:
111
+ if name in DETECTOR_REGISTRY:
112
+ raise ValueError(f"Detector {name!r} is already registered.")
113
+ DETECTOR_REGISTRY[name] = DetectorSpec(
114
+ name=name,
115
+ func=func,
116
+ scope=scope,
117
+ priority=priority,
118
+ requires_schema=requires_schema,
119
+ wants_ctx=_wants_ctx(func),
120
+ doc=func.__doc__ or "",
121
+ )
122
+ return func
123
+
124
+ return decorator
125
+
126
+
127
+ def unregister_detector(name: str) -> None:
128
+ """Remove a detector (used by tests and for overriding a built-in)."""
129
+ DETECTOR_REGISTRY.pop(name, None)
130
+
131
+
132
+ def list_detectors() -> list[str]:
133
+ return sorted(DETECTOR_REGISTRY)
134
+
135
+
136
+ def _normalize_result(result: Any) -> Iterable[Issue]:
137
+ if result is None:
138
+ return []
139
+ if isinstance(result, Issues):
140
+ return list(result)
141
+ if isinstance(result, Issue):
142
+ return [result]
143
+ if isinstance(result, (list, tuple)):
144
+ return [r for r in result if isinstance(r, Issue)]
145
+ raise TypeError(
146
+ f"A detector must return Issues / Issue / list / None, got {type(result).__name__}."
147
+ )
148
+
149
+
150
+ def _invoke(spec: DetectorSpec, data: Any, ctx: DetectorContext) -> Iterable[Issue]:
151
+ result = spec.func(data, ctx) if spec.wants_ctx else spec.func(data)
152
+ return _normalize_result(result)
153
+
154
+
155
+ def run_detectors(
156
+ df: pd.DataFrame,
157
+ *,
158
+ profile: DataFrameProfile | None = None,
159
+ schema: Any | None = None,
160
+ options: dict[str, Any] | None = None,
161
+ only: Iterable[str] | None = None,
162
+ ) -> Issues:
163
+ """Run every applicable detector and return the aggregated, stamped issues.
164
+
165
+ Deterministic: detectors run in ``(priority, name)`` order and columns in frame
166
+ order. Each issue is stamped with its detector name (and column, for column
167
+ detectors) if the detector didn't set them.
168
+ """
169
+ from .._util import ensure_string_columns
170
+ from ..profile import profile_dataframe # local import avoids a cycle at module load
171
+
172
+ df = ensure_string_columns(df)
173
+ profile = profile or profile_dataframe(df)
174
+ options = options or {}
175
+ only_set = set(only) if only is not None else None
176
+ issues = Issues()
177
+
178
+ specs = sorted(DETECTOR_REGISTRY.values(), key=lambda s: (s.priority, s.name))
179
+ for spec in specs:
180
+ if only_set is not None and spec.name not in only_set:
181
+ continue
182
+ if spec.requires_schema and schema is None:
183
+ continue
184
+
185
+ if spec.scope == "column":
186
+ for col in df.columns:
187
+ col_name = str(col)
188
+ ctx = DetectorContext(
189
+ df=df,
190
+ profile=profile,
191
+ column=col_name,
192
+ series=df[col],
193
+ column_profile=profile.column(col_name),
194
+ schema=schema,
195
+ options=options,
196
+ )
197
+ for issue in _invoke(spec, df[col], ctx):
198
+ issue.detector = issue.detector or spec.name
199
+ if issue.column is None:
200
+ issue.column = col_name
201
+ issues.append(issue)
202
+ else:
203
+ ctx = DetectorContext(
204
+ df=df, profile=profile, schema=schema, options=options
205
+ )
206
+ for issue in _invoke(spec, df, ctx):
207
+ issue.detector = issue.detector or spec.name
208
+ issues.append(issue)
209
+
210
+ return issues
211
+
212
+
213
+ __all__ = [
214
+ "DetectorContext",
215
+ "DetectorSpec",
216
+ "DETECTOR_REGISTRY",
217
+ "detector",
218
+ "unregister_detector",
219
+ "list_detectors",
220
+ "run_detectors",
221
+ ]
@@ -0,0 +1,199 @@
1
+ """Category canonicalisation.
2
+
3
+ Finds values that *mean the same thing* but are spelled differently and proposes a
4
+ ``normalize_values`` mapping to a single canonical spelling. Two mechanisms, both
5
+ deterministic:
6
+
7
+ * **Exact-after-normalisation** clusters (``"Bengaluru"``/``"bengaluru "``/
8
+ ``"BENGALURU"``) — high confidence.
9
+ * **Fuzzy typo** clusters (``"Banglore"`` vs ``"Bangalore"``) via a similarity
10
+ threshold — lower confidence.
11
+
12
+ Semantic abbreviations (``"BLR"`` → ``"Bangalore"``) are deliberately *out of
13
+ scope* for rules — that's domain knowledge for a human or the LLM planner. The
14
+ canonical spelling for each cluster is the most frequent original (ties broken
15
+ alphabetically), so results never depend on row order.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import re
21
+
22
+ import pandas as pd
23
+
24
+ from .._util import is_string_like, normalize_key, sample_non_null, similarity
25
+ from ..issues import Issues
26
+ from ..types import Op, Severity
27
+ from .base import DetectorContext, detector
28
+
29
+ #: Above this cardinality a column is treated as free text, not categories.
30
+ _MAX_CATEGORY_CARDINALITY = 60
31
+ _FUZZY_THRESHOLD = 0.86
32
+ #: A fuzzy typo cluster only merges when the variant is this much rarer than the
33
+ #: canonical it would fold into. This stops two *both-frequent* look-alikes
34
+ #: (``insured``/``uninsured``, ``activated``/``deactivated``) from collapsing —
35
+ #: a real typo is nearly always far rarer than the correct spelling.
36
+ _FUZZY_MERGE_MAX_RATIO = 0.34
37
+ _WS = re.compile(r"\s+")
38
+
39
+
40
+ def _clean(value: str) -> str:
41
+ """Whitespace-normalise a value so category clustering sees post-trim spellings.
42
+
43
+ Categories run *after* whitespace ops in a recipe, so clustering on the cleaned
44
+ form keeps the ``normalize_values`` map keys matching the data at that point and
45
+ prevents a whitespace-messy spelling from ever being chosen as canonical.
46
+ """
47
+ return _WS.sub(" ", value).strip()
48
+
49
+
50
+ def _casing_rank(s: str) -> int:
51
+ """Tie-break spellings: short CODE > Title > Mixed > lower > UPPER.
52
+
53
+ Short all-caps tokens (``CA``, ``NY``, ``INR``) are codes whose canonical form
54
+ *is* upper case; anything longer reads better title-cased.
55
+ """
56
+ if s.isupper() and len(s.strip()) <= 3:
57
+ return 0
58
+ if s.istitle():
59
+ return 1
60
+ if not s.isupper() and not s.islower():
61
+ return 2
62
+ if s.islower():
63
+ return 3
64
+ return 4
65
+
66
+
67
+ def _canonical(spellings: dict[str, int]) -> str:
68
+ """Canonical spelling: most frequent, then nicest casing, then alphabetical.
69
+
70
+ All three keys are deterministic, so the choice never depends on row order.
71
+ """
72
+ return sorted(spellings.items(), key=lambda kv: (-kv[1], _casing_rank(kv[0]), kv[0]))[0][0]
73
+
74
+
75
+ #: A negated spelling is a different category, however similar it looks.
76
+ _NEGATION_PREFIXES = (
77
+ "un", "non", "in", "im", "ir", "il", "dis", "de", "re", "anti", "no", "not",
78
+ )
79
+
80
+
81
+ def _is_negation_pair(a: str, b: str) -> bool:
82
+ """True when one spelling is the other plus a negation/repetition prefix."""
83
+ x, y = normalize_key(a), normalize_key(b)
84
+ if x == y:
85
+ return False
86
+ longer, shorter = (x, y) if len(x) > len(y) else (y, x)
87
+ if not shorter or not longer.endswith(shorter):
88
+ return False
89
+ return longer[: -len(shorter)] in _NEGATION_PREFIXES
90
+
91
+
92
+ def _cluster(counts: dict[str, int], seed_map: dict[str, str] | None) -> dict[str, str]:
93
+ """Return a mapping ``{variant: canonical}`` covering only values that change."""
94
+ # 1) group by aggressive normalisation key
95
+ groups: dict[str, dict[str, int]] = {}
96
+ for value, n in counts.items():
97
+ key = normalize_key(value)
98
+ if not key:
99
+ continue # punctuation-only values ("-", "?") are not category variants
100
+ groups.setdefault(key, {})[value] = n
101
+
102
+ canon_of_group = {key: _canonical(spellings) for key, spellings in groups.items()}
103
+
104
+ # 2) fuzzy-merge whole groups whose canonical spellings are near-identical.
105
+ # Deterministic: iterate group keys in sorted order, attach to the first
106
+ # (largest, then alphabetically-first) existing cluster within threshold.
107
+ ordered = sorted(groups, key=lambda k: (-sum(groups[k].values()), k))
108
+ cluster_rep: dict[str, str] = {}
109
+ reps: list[str] = []
110
+ for key in ordered:
111
+ canon = canon_of_group[key]
112
+ key_count = sum(groups[key].values())
113
+ match = None
114
+ for rep in reps:
115
+ # Only fold this group into `rep` if it is BOTH near-identical in
116
+ # spelling AND rare relative to it — otherwise two distinct, comparably
117
+ # frequent categories would be silently merged (semantic inversion).
118
+ rep_count = sum(groups[rep].values())
119
+ if _is_negation_pair(canon, canon_of_group[rep]):
120
+ continue
121
+ if (
122
+ similarity(canon, canon_of_group[rep]) >= _FUZZY_THRESHOLD
123
+ and key_count <= rep_count * _FUZZY_MERGE_MAX_RATIO
124
+ ):
125
+ match = rep
126
+ break
127
+ if match is None:
128
+ reps.append(key)
129
+ cluster_rep[key] = key
130
+ else:
131
+ cluster_rep[key] = match
132
+
133
+ # 3) build variant -> canonical, honouring any seed overrides first.
134
+ mapping: dict[str, str] = {}
135
+ for key, spellings in groups.items():
136
+ rep_key = cluster_rep[key]
137
+ canonical = canon_of_group[rep_key]
138
+ for value in spellings:
139
+ target = (seed_map or {}).get(value, canonical)
140
+ if value != target:
141
+ mapping[value] = target
142
+ # Sort keys so the emitted recipe is byte-identical regardless of input row
143
+ # order (the cleaned data is order-independent either way).
144
+ return dict(sorted(mapping.items()))
145
+
146
+
147
+ @detector("categories", priority=50)
148
+ def detect_categories(series: pd.Series, ctx: DetectorContext) -> Issues:
149
+ """Cluster case/whitespace/typo variants of a category into one canonical value."""
150
+ issues = Issues()
151
+ cp = ctx.column_profile
152
+ if cp is None or cp.count == 0:
153
+ return issues
154
+ if not is_string_like(series):
155
+ return issues
156
+ # Only meaningful for low-cardinality columns.
157
+ if cp.unique_count > _MAX_CATEGORY_CARDINALITY or cp.semantic_type in (
158
+ "email", "phone", "url", "currency", "date", "datetime", "id",
159
+ ):
160
+ return issues
161
+
162
+ counts: dict[str, int] = {}
163
+ for v in sample_non_null(series):
164
+ if isinstance(v, str):
165
+ cleaned = _clean(v)
166
+ counts[cleaned] = counts.get(cleaned, 0) + 1
167
+ if len(counts) < 2:
168
+ return issues
169
+
170
+ seed_map = ctx.option("category_map", {}).get(ctx.column) if ctx.option("category_map") else None
171
+ mapping = _cluster(counts, seed_map)
172
+ if not mapping:
173
+ return issues
174
+
175
+ distinct_before = len(counts)
176
+ distinct_after = len({counts_key if counts_key not in mapping else mapping[counts_key]
177
+ for counts_key in counts})
178
+ # Confidence: purely case/space merges are safe; fuzzy typo merges less so.
179
+ fuzzy = any(normalize_key(k) != normalize_key(v) for k, v in mapping.items())
180
+ confidence = 0.7 if fuzzy else 0.9
181
+
182
+ issues.add(
183
+ "category_variants",
184
+ f"{len(mapping)} value(s) are variants of other categories "
185
+ f"({distinct_before} → {distinct_after} distinct)",
186
+ severity=Severity.WARNING,
187
+ confidence=confidence,
188
+ evidence={
189
+ "mapping": mapping,
190
+ "distinct_before": distinct_before,
191
+ "distinct_after": distinct_after,
192
+ "fuzzy": fuzzy,
193
+ },
194
+ ops=[Op("normalize_values", {"map": mapping})],
195
+ )
196
+ return issues
197
+
198
+
199
+ __all__ = ["detect_categories"]
@@ -0,0 +1,105 @@
1
+ """Email and phone detectors.
2
+
3
+ Emails get the safe, universally-correct normalisation (trim + lowercase) plus an
4
+ invalid-address count. Phones get best-effort separator/country-code normalisation
5
+ driven by the ``phone_country_code`` option. Neither drops data — validation rules
6
+ (added by the planner for these semantic types) decide what happens to values that
7
+ still fail.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import re
13
+
14
+ import pandas as pd
15
+
16
+ from .._util import sample_non_null
17
+ from ..issues import Issues, _cap_examples
18
+ from ..profile import EMAIL_RE, _name_hint
19
+ from ..types import Op, Severity
20
+ from .base import DetectorContext, detector
21
+
22
+
23
+ @detector("emails", priority=45)
24
+ def detect_emails(series: pd.Series, ctx: DetectorContext) -> Issues:
25
+ """Normalise emails (trim + lowercase) and flag invalid addresses."""
26
+ issues = Issues()
27
+ cp = ctx.column_profile
28
+ if cp is None or cp.count == 0:
29
+ return issues
30
+ if cp.semantic_type != "email" and not _name_hint(ctx.column or "", "email"):
31
+ return issues
32
+
33
+ values = [v for v in sample_non_null(series) if isinstance(v, str)]
34
+ if not values:
35
+ return issues
36
+
37
+ invalid = [v for v in values if not EMAIL_RE.match(v.strip().lower())]
38
+ needs_norm = [v for v in values if v != v.strip().lower()]
39
+
40
+ if needs_norm:
41
+ issues.add(
42
+ "email_normalization",
43
+ f"{len(needs_norm)} email(s) need trimming/lowercasing",
44
+ severity=Severity.INFO,
45
+ confidence=0.95,
46
+ evidence={"count": len(needs_norm), "examples": _cap_examples(needs_norm)},
47
+ ops=[Op("normalize_email")],
48
+ )
49
+ if invalid:
50
+ issues.add(
51
+ "invalid_emails",
52
+ f"{len(invalid)} value(s) are not valid email addresses",
53
+ severity=Severity.ERROR if len(invalid) / len(values) > 0.02 else Severity.WARNING,
54
+ confidence=1.0,
55
+ evidence={"count": len(invalid), "examples": _cap_examples(invalid)},
56
+ )
57
+ return issues
58
+
59
+
60
+ @detector("phones", priority=45)
61
+ def detect_phones(series: pd.Series, ctx: DetectorContext) -> Issues:
62
+ """Normalise phone-number formatting and flag implausible digit counts."""
63
+ issues = Issues()
64
+ cp = ctx.column_profile
65
+ if cp is None or cp.count == 0:
66
+ return issues
67
+ if cp.semantic_type != "phone" and not _name_hint(ctx.column or "", "phone"):
68
+ return issues
69
+
70
+ values = [v if isinstance(v, str) else str(v) for v in sample_non_null(series)]
71
+ if not values:
72
+ return issues
73
+
74
+ digit_counts = [len(re.sub(r"\D", "", v)) for v in values]
75
+ invalid = [v for v, d in zip(values, digit_counts, strict=False) if not (7 <= d <= 15)]
76
+ # "Needs normalisation" = contains separators or lacks a leading +.
77
+ messy = [v for v in values if re.search(r"[ ()\-.]", v) or not v.strip().startswith("+")]
78
+
79
+ country_code = ctx.option("phone_country_code") or ctx.option("region")
80
+ if messy:
81
+ params = {"default_country_code": country_code} if country_code else {}
82
+ issues.add(
83
+ "phone_normalization",
84
+ f"{len(messy)} phone number(s) have inconsistent formatting",
85
+ severity=Severity.INFO,
86
+ confidence=0.8,
87
+ evidence={
88
+ "count": len(messy),
89
+ "examples": _cap_examples(messy),
90
+ "country_code": country_code,
91
+ },
92
+ ops=[Op("normalize_phone", params)],
93
+ )
94
+ if invalid:
95
+ issues.add(
96
+ "invalid_phones",
97
+ f"{len(invalid)} value(s) have an implausible number of digits",
98
+ severity=Severity.WARNING,
99
+ confidence=0.9,
100
+ evidence={"count": len(invalid), "examples": _cap_examples(invalid)},
101
+ )
102
+ return issues
103
+
104
+
105
+ __all__ = ["detect_emails", "detect_phones"]
@@ -0,0 +1,112 @@
1
+ """Currency / money-column detection.
2
+
3
+ Turns ``₹1,20,000`` / ``$1,200`` / ``1200 INR`` into a typed float. When a column
4
+ holds a single currency, the code is folded into the column name (``amount`` →
5
+ ``amount_inr``, matching the README). When a column *mixes* currencies — where the
6
+ amount alone would be meaningless — it additionally splits out a ``*_currency``
7
+ column so no information is lost.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import re
13
+
14
+ import numpy as np
15
+ import pandas as pd
16
+
17
+ from .._util import sample_non_null, snake_case, token_set
18
+ from ..issues import Issues, _cap_examples
19
+ from ..ops import _detect_currency_scalar, _parse_number_scalar
20
+ from ..types import Op, Severity
21
+ from .base import DetectorContext, detector
22
+
23
+ _EU_GROUPED_RE = re.compile(r"\d{1,3}(?:\.\d{3})+,\d+")
24
+ _US_GROUPED_RE = re.compile(r"\d{1,3}(?:,\d{3})+")
25
+ _COMMA_DECIMAL_RE = re.compile(r"\d,\d{1,2}(?!\d)")
26
+
27
+
28
+ def _decimal_convention(values: list[str]) -> dict[str, str]:
29
+ """Infer European ``1.234,56`` grouping. Empty dict means the pandas default.
30
+
31
+ Parsing ``€1.200,50`` with the default convention yields 1.2005 while reporting
32
+ nothing unparseable, so the convention has to be decided from the values.
33
+ """
34
+ eu_grouped = sum(1 for v in values if _EU_GROUPED_RE.search(v))
35
+ us_grouped = sum(1 for v in values if _US_GROUPED_RE.search(v))
36
+ if us_grouped:
37
+ return {}
38
+ if eu_grouped:
39
+ return {"decimal": ",", "thousands": "."}
40
+ comma_decimal = sum(1 for v in values if _COMMA_DECIMAL_RE.search(v))
41
+ if comma_decimal and not any("." in v for v in values):
42
+ return {"decimal": ",", "thousands": "."}
43
+ return {}
44
+
45
+
46
+ @detector("currency", priority=45)
47
+ def detect_currency(series: pd.Series, ctx: DetectorContext) -> Issues:
48
+ """Detect money-as-text columns and propose parsing to a float (+ currency split)."""
49
+ issues = Issues()
50
+ cp = ctx.column_profile
51
+ if cp is None or cp.count == 0 or cp.semantic_type != "currency":
52
+ return issues
53
+
54
+ values = [v if isinstance(v, str) else str(v) for v in sample_non_null(series)]
55
+ codes = {c for c in (_detect_currency_scalar(v, None) for v in values) if isinstance(c, str)}
56
+
57
+ convention = _decimal_convention(values)
58
+ decimal = convention.get("decimal", ".")
59
+ thousands = convention.get("thousands", ",")
60
+ # How many non-null values fail to become a number? (report, don't hide.)
61
+ unparsed = sum(
62
+ 1 for v in values if np.isnan(_parse_number_scalar(v, decimal, thousands, []))
63
+ )
64
+
65
+ snake = snake_case(ctx.column or series.name or "amount")
66
+ ops: list[Op] = []
67
+ rename_to: str | None = None
68
+
69
+ if len(codes) == 1:
70
+ code = next(iter(codes))
71
+ # Fold the currency into the name unless it's already there.
72
+ if code.lower() not in token_set(snake):
73
+ rename_to = f"{snake}_{code.lower()}"
74
+ else:
75
+ rename_to = snake
76
+ ops = [Op("parse_number", dict(convention))]
77
+ currency_note = f"single currency {code}"
78
+ else:
79
+ rename_to = snake
80
+ target = f"{snake}_currency"
81
+ ops = [Op("extract_currency", {"to": target}), Op("parse_number", dict(convention))]
82
+ currency_note = (
83
+ f"mixed currencies {sorted(codes)} — splitting out `{target}`"
84
+ if codes
85
+ else "no explicit currency code"
86
+ )
87
+
88
+ if rename_to == (ctx.column or series.name):
89
+ rename_to = None
90
+
91
+ sev = Severity.WARNING if unparsed else Severity.INFO
92
+ evidence = {
93
+ "currencies": sorted(codes),
94
+ "unparsed": unparsed,
95
+ "examples": _cap_examples(values),
96
+ }
97
+ if convention:
98
+ evidence["decimal_convention"] = "european"
99
+ issues.add(
100
+ "currency_format",
101
+ f"Money column stored as text ({currency_note})"
102
+ + (f"; {unparsed} value(s) unparseable" if unparsed else ""),
103
+ severity=sev,
104
+ confidence=0.95,
105
+ evidence=evidence,
106
+ ops=ops,
107
+ rename_to=rename_to,
108
+ )
109
+ return issues
110
+
111
+
112
+ __all__ = ["detect_currency"]