cleanframe-engine 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanframe/__init__.py +169 -0
- cleanframe/__main__.py +5 -0
- cleanframe/_util.py +438 -0
- cleanframe/_version.py +1 -0
- cleanframe/api.py +559 -0
- cleanframe/cli.py +688 -0
- cleanframe/codegen.py +666 -0
- cleanframe/dataio.py +506 -0
- cleanframe/detectors/__init__.py +43 -0
- cleanframe/detectors/base.py +221 -0
- cleanframe/detectors/categories.py +199 -0
- cleanframe/detectors/contacts.py +105 -0
- cleanframe/detectors/currency.py +112 -0
- cleanframe/detectors/dates.py +204 -0
- cleanframe/detectors/dedup.py +150 -0
- cleanframe/detectors/nulls.py +109 -0
- cleanframe/detectors/outliers.py +73 -0
- cleanframe/detectors/schema_mapping.py +125 -0
- cleanframe/detectors/text.py +105 -0
- cleanframe/detectors/units.py +86 -0
- cleanframe/diff.py +369 -0
- cleanframe/drift.py +283 -0
- cleanframe/errors.py +66 -0
- cleanframe/executor.py +229 -0
- cleanframe/fingerprint.py +83 -0
- cleanframe/issues.py +186 -0
- cleanframe/llm.py +811 -0
- cleanframe/ops.py +1245 -0
- cleanframe/planner.py +353 -0
- cleanframe/profile.py +413 -0
- cleanframe/py.typed +1 -0
- cleanframe/quality.py +81 -0
- cleanframe/readfix.py +160 -0
- cleanframe/recipe.py +398 -0
- cleanframe/report.py +345 -0
- cleanframe/result.py +144 -0
- cleanframe/schema.py +259 -0
- cleanframe/streaming.py +354 -0
- cleanframe/types.py +119 -0
- cleanframe/validate.py +363 -0
- cleanframe/workbook.py +370 -0
- cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
- cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
- cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
- cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
- cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
cleanframe/codegen.py
ADDED
|
@@ -0,0 +1,666 @@
|
|
|
1
|
+
"""Export a recipe to standalone, readable pandas — no CleanFrame dependency.
|
|
2
|
+
|
|
3
|
+
``result.code.save("clean_customers.py")`` produces a plain ``clean(df)`` function
|
|
4
|
+
you can read, diff, and drop into a pipeline that never imports CleanFrame.
|
|
5
|
+
|
|
6
|
+
Fidelity is a load-bearing invariant: the generated code must reproduce the
|
|
7
|
+
executor's output *exactly*. To keep the two from drifting, the lookup tables
|
|
8
|
+
(currency symbols, NA tokens, unit factors, date formats) are rendered here from
|
|
9
|
+
the single source of truth in :mod:`cleanframe.ops`, and the emitted helper bodies
|
|
10
|
+
mirror the executor's scalar logic. ``tests/test_wave1_codegen.py`` locks this by
|
|
11
|
+
running representative recipes through both paths and asserting frame equality.
|
|
12
|
+
|
|
13
|
+
Validation is reproduced for the built-in checks (quarantine/drop filter the
|
|
14
|
+
primary frame; ``null`` blanks the cell; ``error`` raises). The quarantine *side*
|
|
15
|
+
frame is a CleanFrame runtime concept and cannot round-trip through a ``df -> df``
|
|
16
|
+
function, so only the cleaned output is reproduced.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import re
|
|
22
|
+
from collections.abc import Callable
|
|
23
|
+
|
|
24
|
+
from .errors import CleanFrameError
|
|
25
|
+
from .recipe import Recipe, ValidationRule
|
|
26
|
+
from .types import Op
|
|
27
|
+
|
|
28
|
+
_UNSAFE_COMMENT_RE = re.compile(r"[\r\n]+")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _comment(text: object) -> str:
|
|
32
|
+
"""One-line, code-safe rendering of user text for a generated comment.
|
|
33
|
+
|
|
34
|
+
A column name containing a newline would otherwise continue the generated module
|
|
35
|
+
on the next line — outside the comment — as executable code.
|
|
36
|
+
"""
|
|
37
|
+
return _UNSAFE_COMMENT_RE.sub(" ", str(text))[:120]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
# ---------------------------------------------------------------------------
|
|
41
|
+
# Constant tables rendered from the executor's single source of truth
|
|
42
|
+
# ---------------------------------------------------------------------------
|
|
43
|
+
def _constants_source() -> str:
|
|
44
|
+
from .ops import (
|
|
45
|
+
_KNOWN_CODES,
|
|
46
|
+
_UNIT_ALIASES,
|
|
47
|
+
_UNIT_TO_FAMILY,
|
|
48
|
+
CURRENCY_SYMBOLS,
|
|
49
|
+
DEFAULT_NA_TOKENS,
|
|
50
|
+
UNIT_FAMILIES,
|
|
51
|
+
)
|
|
52
|
+
from .profile import COMMON_DATE_FORMATS
|
|
53
|
+
|
|
54
|
+
unit_factors = {u: f for fam in UNIT_FAMILIES.values() for u, f in fam.items()}
|
|
55
|
+
na_tokens = sorted({t.casefold() for t in DEFAULT_NA_TOKENS}) # includes '' (H9)
|
|
56
|
+
return "\n".join(
|
|
57
|
+
[
|
|
58
|
+
f"_NA_TOKENS = set({na_tokens!r})",
|
|
59
|
+
f"_CURRENCY_SYMBOLS = {dict(CURRENCY_SYMBOLS)!r}",
|
|
60
|
+
f"_KNOWN_CODES = set({sorted(_KNOWN_CODES)!r})",
|
|
61
|
+
f"_UNIT_FACTORS = {unit_factors!r}",
|
|
62
|
+
f"_UNIT_FAMILY = {dict(_UNIT_TO_FAMILY)!r}",
|
|
63
|
+
f"_UNIT_ALIASES = {dict(_UNIT_ALIASES)!r}",
|
|
64
|
+
f"_COMMON_DATE_FORMATS = {list(COMMON_DATE_FORMATS)!r}",
|
|
65
|
+
r"_CODE_RE = re.compile(r'\b([A-Z]{3})\b')",
|
|
66
|
+
r"_UNIT_VALUE_RE = re.compile(r'^\s*([+-]?\d+(?:[.,]\d+)?)\s*([A-Za-z]+)\s*$')",
|
|
67
|
+
r"_PHONE_EXT_RE = re.compile(r'[\s,;]*(?:ext|extn|x|#)\.?\s*\d+\s*$', re.IGNORECASE)",
|
|
68
|
+
r"_NUMBER_TOKEN_RE = re.compile(r'[+-]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][+-]?\d+)?')",
|
|
69
|
+
r"_STRICT_NUM_RE = re.compile(r'^[+-]?\d+(\.\d+)?$')",
|
|
70
|
+
r"_EMAIL_RE = re.compile(r'^[^@\s]+@[^@\s]+\.[^@\s]+$')",
|
|
71
|
+
r"_URL_RE = re.compile(r'^(https?://|www\.)\S+$', re.IGNORECASE)",
|
|
72
|
+
"_TRUE_TOKENS = {'true', 't', 'yes', 'y', '1'}",
|
|
73
|
+
"_FALSE_TOKENS = {'false', 'f', 'no', 'n', '0'}",
|
|
74
|
+
]
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
_DOC_AND_IMPORTS = '''"""Auto-generated by CleanFrame. Deterministic, dependency-free pandas.
|
|
79
|
+
|
|
80
|
+
Edit freely — this file has no third-party dependency. Regenerate with
|
|
81
|
+
``result.code.save(...)`` if you change the recipe.
|
|
82
|
+
"""
|
|
83
|
+
|
|
84
|
+
import re
|
|
85
|
+
import warnings
|
|
86
|
+
|
|
87
|
+
import numpy as np
|
|
88
|
+
import pandas as pd
|
|
89
|
+
'''
|
|
90
|
+
|
|
91
|
+
# Helper bodies mirror cleanframe.ops / cleanframe.validate scalar logic exactly.
|
|
92
|
+
_HELPERS = '''
|
|
93
|
+
def _smap(series, fn):
|
|
94
|
+
"""Apply fn to string cells only; leave NaN and non-strings untouched."""
|
|
95
|
+
return series.map(lambda v: fn(v) if isinstance(v, str) else v)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _to_na(series, tokens=None):
|
|
99
|
+
toks = _NA_TOKENS if tokens is None else {str(t).strip().casefold() for t in tokens}
|
|
100
|
+
return _smap(series, lambda v: np.nan if v.strip().casefold() in toks else v)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _parse_number(series, decimal=".", thousands=",", symbols=()):
|
|
104
|
+
def one(v):
|
|
105
|
+
if v is None or (isinstance(v, float) and np.isnan(v)) or isinstance(v, bool):
|
|
106
|
+
return np.nan
|
|
107
|
+
if isinstance(v, (int, float)):
|
|
108
|
+
return float(v)
|
|
109
|
+
s = str(v).strip().replace("\\u2212", "-")
|
|
110
|
+
if s == "":
|
|
111
|
+
return np.nan
|
|
112
|
+
neg = s.startswith("(") and s.endswith(")")
|
|
113
|
+
if neg:
|
|
114
|
+
s = s[1:-1]
|
|
115
|
+
for sym in symbols:
|
|
116
|
+
s = s.replace(sym, "")
|
|
117
|
+
if thousands:
|
|
118
|
+
s = s.replace(thousands, "")
|
|
119
|
+
if decimal != ".":
|
|
120
|
+
s = s.replace(decimal, ".")
|
|
121
|
+
s = s.strip()
|
|
122
|
+
trailing_minus = s.endswith("-")
|
|
123
|
+
m = _NUMBER_TOKEN_RE.search(s)
|
|
124
|
+
if not m:
|
|
125
|
+
return np.nan
|
|
126
|
+
token = m.group(0)
|
|
127
|
+
leftover = s[: m.start()] + s[m.end() :]
|
|
128
|
+
if any(ch.isdigit() for ch in leftover):
|
|
129
|
+
return np.nan
|
|
130
|
+
try:
|
|
131
|
+
result = float(token)
|
|
132
|
+
except ValueError:
|
|
133
|
+
return np.nan
|
|
134
|
+
if neg:
|
|
135
|
+
result = -abs(result)
|
|
136
|
+
elif trailing_minus and not token.startswith("-"):
|
|
137
|
+
result = -result
|
|
138
|
+
return result
|
|
139
|
+
|
|
140
|
+
return series.map(one)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _parse_dates_to_datetime(series, formats=None, dayfirst=False, yearfirst=False):
|
|
144
|
+
flex = False
|
|
145
|
+
if not formats:
|
|
146
|
+
formats = _COMMON_DATE_FORMATS
|
|
147
|
+
flex = True
|
|
148
|
+
result = pd.Series(pd.NaT, index=series.index, dtype="datetime64[ns]")
|
|
149
|
+
for fmt in formats:
|
|
150
|
+
mask = result.isna() & series.notna()
|
|
151
|
+
if not mask.any():
|
|
152
|
+
break
|
|
153
|
+
result.loc[mask] = pd.to_datetime(series[mask], format=fmt, errors="coerce")
|
|
154
|
+
if flex:
|
|
155
|
+
remaining = result.isna() & series.notna()
|
|
156
|
+
if remaining.any():
|
|
157
|
+
with warnings.catch_warnings():
|
|
158
|
+
warnings.simplefilter("ignore")
|
|
159
|
+
result.loc[remaining] = pd.to_datetime(
|
|
160
|
+
series[remaining], errors="coerce", dayfirst=dayfirst, yearfirst=yearfirst
|
|
161
|
+
)
|
|
162
|
+
return result
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _parse_date(series, formats=None, dayfirst=False, yearfirst=False, output="%Y-%m-%d"):
|
|
166
|
+
dt = _parse_dates_to_datetime(series, formats, dayfirst=dayfirst, yearfirst=yearfirst)
|
|
167
|
+
if str(output).lower() in ("datetime", "raw", "none"):
|
|
168
|
+
return dt
|
|
169
|
+
fmt = "%Y-%m-%d" if str(output).lower() in ("iso", "date") else output
|
|
170
|
+
return dt.dt.strftime(fmt).where(dt.notna(), np.nan)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _detect_currency(v, default=None):
|
|
174
|
+
if v is None or (isinstance(v, float) and np.isnan(v)):
|
|
175
|
+
return default if default is not None else np.nan
|
|
176
|
+
s = str(v)
|
|
177
|
+
for sym, code in _CURRENCY_SYMBOLS.items():
|
|
178
|
+
if sym in s:
|
|
179
|
+
return code
|
|
180
|
+
m = _CODE_RE.search(s.upper())
|
|
181
|
+
if m and m.group(1) in _KNOWN_CODES:
|
|
182
|
+
return m.group(1)
|
|
183
|
+
return default if default is not None else np.nan
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _phone_text(v):
|
|
187
|
+
"""Stringify a phone cell without inventing digits (a float column has a '.0')."""
|
|
188
|
+
if isinstance(v, float) and float(v).is_integer():
|
|
189
|
+
return str(int(v))
|
|
190
|
+
return str(v)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _normalize_phone(series, default_country_code=None):
|
|
194
|
+
def one(v):
|
|
195
|
+
if v is None or (isinstance(v, float) and np.isnan(v)):
|
|
196
|
+
return v
|
|
197
|
+
if isinstance(v, bool):
|
|
198
|
+
return np.nan
|
|
199
|
+
s = _PHONE_EXT_RE.sub("", _phone_text(v))
|
|
200
|
+
plus = s.strip().startswith("+")
|
|
201
|
+
digits = re.sub(r"\\D", "", s)
|
|
202
|
+
if not digits:
|
|
203
|
+
return np.nan
|
|
204
|
+
if plus:
|
|
205
|
+
return "+" + digits # noqa: RET504
|
|
206
|
+
if default_country_code:
|
|
207
|
+
cc = re.sub(r"\\D", "", str(default_country_code))
|
|
208
|
+
if cc and digits.startswith(cc):
|
|
209
|
+
return "+" + digits
|
|
210
|
+
return "+" + cc + digits.lstrip("0")
|
|
211
|
+
return digits
|
|
212
|
+
|
|
213
|
+
return series.map(one)
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def _parse_unit_scalar(v):
|
|
217
|
+
if v is None or (isinstance(v, float) and np.isnan(v)):
|
|
218
|
+
return None
|
|
219
|
+
if isinstance(v, (int, float, bool)):
|
|
220
|
+
return None
|
|
221
|
+
m = _UNIT_VALUE_RE.match(str(v))
|
|
222
|
+
if not m:
|
|
223
|
+
return None
|
|
224
|
+
num_s, unit_s = m.group(1), m.group(2).casefold()
|
|
225
|
+
unit_s = _UNIT_ALIASES.get(unit_s, unit_s)
|
|
226
|
+
if unit_s not in _UNIT_FAMILY:
|
|
227
|
+
return None
|
|
228
|
+
try:
|
|
229
|
+
return float(num_s.replace(",", ".")), unit_s
|
|
230
|
+
except ValueError:
|
|
231
|
+
return None
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _normalize_unit(series, to="g", emit_unit_column=False):
|
|
235
|
+
to = _UNIT_ALIASES.get(str(to).casefold(), str(to).casefold())
|
|
236
|
+
target_fam = _UNIT_FAMILY[to]
|
|
237
|
+
target_factor = _UNIT_FACTORS[to]
|
|
238
|
+
amounts, units = [], []
|
|
239
|
+
for v in series.tolist():
|
|
240
|
+
parsed = _parse_unit_scalar(v)
|
|
241
|
+
if parsed is None:
|
|
242
|
+
if isinstance(v, (int, float)) and not isinstance(v, bool) and not (isinstance(v, float) and np.isnan(v)):
|
|
243
|
+
amounts.append(float(v))
|
|
244
|
+
units.append(to)
|
|
245
|
+
elif isinstance(v, str) and _STRICT_NUM_RE.match(v.strip().replace(",", ".")):
|
|
246
|
+
try:
|
|
247
|
+
amounts.append(float(v.replace(",", ".").strip()))
|
|
248
|
+
units.append(to)
|
|
249
|
+
except ValueError:
|
|
250
|
+
amounts.append(np.nan)
|
|
251
|
+
units.append(None)
|
|
252
|
+
else:
|
|
253
|
+
amounts.append(np.nan)
|
|
254
|
+
units.append(None)
|
|
255
|
+
continue
|
|
256
|
+
amount, unit = parsed
|
|
257
|
+
if _UNIT_FAMILY[unit] != target_fam:
|
|
258
|
+
amounts.append(np.nan)
|
|
259
|
+
units.append(unit)
|
|
260
|
+
continue
|
|
261
|
+
amounts.append(amount * _UNIT_FACTORS[unit] / target_factor)
|
|
262
|
+
units.append(unit)
|
|
263
|
+
out = pd.Series(amounts, index=series.index, dtype="float64")
|
|
264
|
+
if emit_unit_column:
|
|
265
|
+
return out, pd.Series(units, index=series.index)
|
|
266
|
+
return out
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _to_bool(series):
|
|
270
|
+
def one(v):
|
|
271
|
+
if v is None or (isinstance(v, float) and np.isnan(v)):
|
|
272
|
+
return pd.NA
|
|
273
|
+
if isinstance(v, bool):
|
|
274
|
+
return v
|
|
275
|
+
t = str(v).strip().casefold()
|
|
276
|
+
if t in _TRUE_TOKENS:
|
|
277
|
+
return True
|
|
278
|
+
if t in _FALSE_TOKENS:
|
|
279
|
+
return False
|
|
280
|
+
return pd.NA
|
|
281
|
+
|
|
282
|
+
return series.map(one).astype("boolean")
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
# -- validation pass-masks (True = passes), mirroring cleanframe.validate ------
|
|
286
|
+
def _v_not_null(s):
|
|
287
|
+
return s.notna()
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def _v_unique(s):
|
|
291
|
+
return ~(s.duplicated(keep=False) & s.notna())
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def _v_valid_email(s):
|
|
295
|
+
return s.isna() | s.map(lambda v: bool(_EMAIL_RE.match(str(v).strip().lower())))
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def _v_valid_url(s):
|
|
299
|
+
return s.isna() | s.map(lambda v: bool(_URL_RE.match(str(v).strip())))
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _v_valid_phone(s):
|
|
303
|
+
return s.isna() | s.map(lambda v: 7 <= len(re.sub(r"\\D", "", str(v))) <= 15)
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
_CMP_OPS = {
|
|
307
|
+
">=": lambda a, b: a >= b, "<=": lambda a, b: a <= b, ">": lambda a, b: a > b,
|
|
308
|
+
"<": lambda a, b: a < b, "==": lambda a, b: a == b, "!=": lambda a, b: a != b,
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def _v_cmp(s, op, threshold):
|
|
313
|
+
numeric = pd.to_numeric(s, errors="coerce")
|
|
314
|
+
satisfies = _CMP_OPS[op](numeric, threshold).fillna(False).astype(bool)
|
|
315
|
+
return s.isna() | (numeric.notna() & satisfies)
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _v_in(s, values):
|
|
319
|
+
_BOOL_SPELLINGS = {
|
|
320
|
+
True: ("True", "true", "TRUE", "Yes", "yes", "YES", "On", "on", "Y", "y", "1"),
|
|
321
|
+
False: ("False", "false", "FALSE", "No", "no", "NO", "Off", "off", "N", "n", "0"),
|
|
322
|
+
}
|
|
323
|
+
as_str = set()
|
|
324
|
+
for v in values:
|
|
325
|
+
if isinstance(v, bool):
|
|
326
|
+
as_str.update(_BOOL_SPELLINGS[v])
|
|
327
|
+
else:
|
|
328
|
+
as_str.add(str(v))
|
|
329
|
+
return s.isna() | s.isin(values) | s.astype(str).isin(as_str)
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def _v_matches(s, pattern):
|
|
333
|
+
if len(pattern) > 500:
|
|
334
|
+
raise ValueError(f"Regex pattern exceeds limit of 500 characters ({len(pattern)}).")
|
|
335
|
+
compiled = re.compile(pattern)
|
|
336
|
+
return s.isna() | s.map(lambda v: bool(compiled.search(str(v))))
|
|
337
|
+
'''
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def _col(name: str) -> str:
|
|
341
|
+
return f"df[{name!r}]"
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
# -- per-op code emitters ----------------------------------------------------
|
|
345
|
+
def _remove_symbols(params: dict, c: str) -> list[str]:
|
|
346
|
+
chain = "".join(f".replace({s!r}, '')" for s in params.get("symbols", []))
|
|
347
|
+
return [f" {_col(c)} = _smap({_col(c)}, lambda v: v{chain})"]
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def _replace(params: dict, c: str) -> list[str]:
|
|
351
|
+
pattern = params["pattern"]
|
|
352
|
+
if params.get("regex", True):
|
|
353
|
+
# Fail at export time with the same ReDoS / length guards the executor uses.
|
|
354
|
+
from ._util import safe_compile_regex
|
|
355
|
+
|
|
356
|
+
safe_compile_regex(pattern)
|
|
357
|
+
return [
|
|
358
|
+
f" {_col(c)} = _smap({_col(c)}, "
|
|
359
|
+
f"lambda v: re.sub({pattern!r}, {params.get('repl', '')!r}, v))"
|
|
360
|
+
]
|
|
361
|
+
return [
|
|
362
|
+
f" {_col(c)} = _smap({_col(c)}, "
|
|
363
|
+
f"lambda v: v.replace({pattern!r}, {params.get('repl', '')!r}))"
|
|
364
|
+
]
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def _cast(params: dict, c: str) -> list[str]:
|
|
368
|
+
to = str(params["to"]).lower()
|
|
369
|
+
col = _col(c)
|
|
370
|
+
expr = {
|
|
371
|
+
"float": f"pd.to_numeric({col}, errors='coerce').astype('float64')",
|
|
372
|
+
"float64": f"pd.to_numeric({col}, errors='coerce').astype('float64')",
|
|
373
|
+
"number": f"pd.to_numeric({col}, errors='coerce').astype('float64')",
|
|
374
|
+
"int": f"pd.to_numeric({col}, errors='coerce').round().astype('Int64')",
|
|
375
|
+
"integer": f"pd.to_numeric({col}, errors='coerce').round().astype('Int64')",
|
|
376
|
+
"int64": f"pd.to_numeric({col}, errors='coerce').round().astype('Int64')",
|
|
377
|
+
"string": f"{col}.astype('string')",
|
|
378
|
+
"str": f"{col}.astype('string')",
|
|
379
|
+
"text": f"{col}.astype('string')",
|
|
380
|
+
"bool": f"_to_bool({col})",
|
|
381
|
+
"boolean": f"_to_bool({col})",
|
|
382
|
+
"datetime": f"_parse_dates_to_datetime({col})",
|
|
383
|
+
"date": f"_parse_dates_to_datetime({col})",
|
|
384
|
+
"category": f"{col}.astype('category')",
|
|
385
|
+
}.get(to)
|
|
386
|
+
if expr is None:
|
|
387
|
+
return [f" # NOTE: cast to {to!r} not reproduced"]
|
|
388
|
+
return [f" {col} = {expr}"]
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def _normalize_values(params: dict, c: str) -> list[str]:
|
|
392
|
+
mapping = params.get("map", {})
|
|
393
|
+
if params.get("case_insensitive"):
|
|
394
|
+
return [
|
|
395
|
+
f" _m = {{str(k).strip().casefold(): v for k, v in {mapping!r}.items()}}",
|
|
396
|
+
f" {_col(c)} = _smap({_col(c)}, lambda v: _m.get(v.strip().casefold(), v))",
|
|
397
|
+
]
|
|
398
|
+
return [
|
|
399
|
+
f" _m = {mapping!r}",
|
|
400
|
+
f" {_col(c)} = {_col(c)}.map(lambda v: _m.get(v, _m.get(str(v), v)) "
|
|
401
|
+
f"if isinstance(v, str) else v)",
|
|
402
|
+
]
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def _parse_date_gen(params: dict, c: str) -> list[str]:
|
|
406
|
+
args = []
|
|
407
|
+
if params.get("formats"):
|
|
408
|
+
args.append(f"formats={list(params['formats'])!r}")
|
|
409
|
+
if params.get("dayfirst"):
|
|
410
|
+
args.append("dayfirst=True")
|
|
411
|
+
if params.get("yearfirst"):
|
|
412
|
+
args.append("yearfirst=True")
|
|
413
|
+
out = params.get("output", "%Y-%m-%d")
|
|
414
|
+
if out != "%Y-%m-%d":
|
|
415
|
+
args.append(f"output={out!r}")
|
|
416
|
+
tail = (", " + ", ".join(args)) if args else ""
|
|
417
|
+
return [f" {_col(c)} = _parse_date({_col(c)}{tail})"]
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def _parse_number_gen(params: dict, c: str) -> list[str]:
|
|
421
|
+
args = []
|
|
422
|
+
if params.get("decimal", ".") != ".":
|
|
423
|
+
args.append(f"decimal={params['decimal']!r}")
|
|
424
|
+
if params.get("thousands", ",") != ",":
|
|
425
|
+
args.append(f"thousands={params['thousands']!r}")
|
|
426
|
+
if params.get("symbols"):
|
|
427
|
+
args.append(f"symbols={list(params['symbols'])!r}")
|
|
428
|
+
tail = (", " + ", ".join(args)) if args else ""
|
|
429
|
+
return [f" {_col(c)} = _parse_number({_col(c)}{tail})"]
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def _extract_currency_gen(params: dict, c: str) -> list[str]:
|
|
433
|
+
to = params.get("to") or f"{c}_currency"
|
|
434
|
+
default = params.get("default")
|
|
435
|
+
return [f" df[{to!r}] = {_col(c)}.map(lambda v: _detect_currency(v, {default!r}))"]
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
def _normalize_unit_gen(params: dict, c: str) -> list[str]:
|
|
439
|
+
to = params.get("to", "g")
|
|
440
|
+
emit = params.get("emit_unit_column")
|
|
441
|
+
if emit:
|
|
442
|
+
return [
|
|
443
|
+
f" {_col(c)}, df[{emit!r}] = _normalize_unit("
|
|
444
|
+
f"{_col(c)}, to={to!r}, emit_unit_column=True)"
|
|
445
|
+
]
|
|
446
|
+
return [f" {_col(c)} = _normalize_unit({_col(c)}, to={to!r})"]
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
def _fill_na_gen(params: dict, c: str) -> list[str]:
|
|
450
|
+
col = _col(c)
|
|
451
|
+
strategy = params.get("strategy")
|
|
452
|
+
if strategy is None:
|
|
453
|
+
return [f" {col} = {col}.fillna({params.get('value')!r})"]
|
|
454
|
+
strategy = str(strategy).lower()
|
|
455
|
+
if strategy == "mean":
|
|
456
|
+
return [f" {col} = {col}.fillna(pd.to_numeric({col}, errors='coerce').mean())"]
|
|
457
|
+
if strategy == "median":
|
|
458
|
+
return [f" {col} = {col}.fillna(pd.to_numeric({col}, errors='coerce').median())"]
|
|
459
|
+
if strategy == "mode":
|
|
460
|
+
return [
|
|
461
|
+
f" _modes = {col}.dropna().mode()",
|
|
462
|
+
f" {col} = {col} if len(_modes) == 0 else {col}.fillna(sorted(_modes.tolist(), key=str)[0])",
|
|
463
|
+
]
|
|
464
|
+
if strategy in ("ffill", "pad"):
|
|
465
|
+
return [f" {col} = {col}.ffill()"]
|
|
466
|
+
if strategy in ("bfill", "backfill"):
|
|
467
|
+
return [f" {col} = {col}.bfill()"]
|
|
468
|
+
if strategy == "zero":
|
|
469
|
+
return [f" {col} = {col}.fillna(0)"]
|
|
470
|
+
if strategy == "empty":
|
|
471
|
+
return [f" {col} = {col}.fillna('')"]
|
|
472
|
+
return [f" # NOTE: unknown fill_na strategy {strategy!r} not reproduced"]
|
|
473
|
+
|
|
474
|
+
|
|
475
|
+
_SIMPLE: dict[str, str] = {
|
|
476
|
+
"strip_whitespace": "lambda v: v.strip()",
|
|
477
|
+
"collapse_whitespace": "lambda v: re.sub(r'\\s+', ' ', v).strip()",
|
|
478
|
+
"lowercase": "lambda v: v.lower()",
|
|
479
|
+
"uppercase": "lambda v: v.upper()",
|
|
480
|
+
"title_case": "lambda v: re.sub(r'\\s+', ' ', v).strip().title()",
|
|
481
|
+
"capitalize": "lambda v: v.strip().capitalize()",
|
|
482
|
+
"normalize_email": "lambda v: v.strip().lower()",
|
|
483
|
+
}
|
|
484
|
+
|
|
485
|
+
_EMITTERS: dict[str, Callable[[dict, str], list[str]]] = {
|
|
486
|
+
"remove_symbols": _remove_symbols,
|
|
487
|
+
"replace": _replace,
|
|
488
|
+
"cast": _cast,
|
|
489
|
+
"normalize_values": _normalize_values,
|
|
490
|
+
"parse_date": _parse_date_gen,
|
|
491
|
+
"parse_number": _parse_number_gen,
|
|
492
|
+
"extract_currency": _extract_currency_gen,
|
|
493
|
+
"normalize_unit": _normalize_unit_gen,
|
|
494
|
+
}
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
def _emit_column_op(op: Op, source: str, unsupported: list[str] | None = None) -> list[str]:
|
|
498
|
+
if op.name in _SIMPLE:
|
|
499
|
+
return [f" {_col(source)} = _smap({_col(source)}, {_SIMPLE[op.name]})"]
|
|
500
|
+
if op.name in _EMITTERS:
|
|
501
|
+
return _EMITTERS[op.name](op.params, source)
|
|
502
|
+
if op.name == "to_na":
|
|
503
|
+
tokens = op.params.get("tokens")
|
|
504
|
+
arg = f", tokens={list(tokens)!r}" if tokens else ""
|
|
505
|
+
return [f" {_col(source)} = _to_na({_col(source)}{arg})"]
|
|
506
|
+
if op.name == "normalize_phone":
|
|
507
|
+
cc = op.params.get("default_country_code")
|
|
508
|
+
arg = f"default_country_code={cc!r}" if cc else ""
|
|
509
|
+
return [f" {_col(source)} = _normalize_phone({_col(source)}{', ' + arg if arg else ''})"]
|
|
510
|
+
if op.name == "fill_na":
|
|
511
|
+
return _fill_na_gen(op.params, source)
|
|
512
|
+
if op.name == "round":
|
|
513
|
+
return [
|
|
514
|
+
f" {_col(source)} = pd.to_numeric({_col(source)}, errors='coerce')"
|
|
515
|
+
f".round({op.params.get('decimals', 0)})"
|
|
516
|
+
]
|
|
517
|
+
if unsupported is not None:
|
|
518
|
+
unsupported.append(f"op {op.name!r} on column {source!r}")
|
|
519
|
+
return [f" # NOTE: op {_comment(op.name)!r} is not reproduced by codegen"]
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
# -- validation emission -----------------------------------------------------
|
|
523
|
+
_NAMED_VALIDATORS = {"not_null", "unique", "valid_email", "valid_url", "valid_phone"}
|
|
524
|
+
_CMP_RE = re.compile(r"^(>=|<=|==|!=|>|<)\s*(-?\d+(?:\.\d+)?)$")
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
def _membership_values(rule: ValidationRule) -> list:
|
|
528
|
+
values = rule.params.get("values")
|
|
529
|
+
if values is not None:
|
|
530
|
+
return [values] if isinstance(values, str) else list(values)
|
|
531
|
+
import yaml
|
|
532
|
+
|
|
533
|
+
rest = rule.check[2:].strip()
|
|
534
|
+
parsed = yaml.load(rest, Loader=yaml.BaseLoader) if rest else []
|
|
535
|
+
return list(parsed) if isinstance(parsed, (list, tuple)) else [parsed]
|
|
536
|
+
|
|
537
|
+
|
|
538
|
+
def _mask_expr(rule: ValidationRule) -> str | None:
|
|
539
|
+
check = rule.check.strip()
|
|
540
|
+
col = _col(rule.column)
|
|
541
|
+
if check in _NAMED_VALIDATORS:
|
|
542
|
+
return f"_v_{check}({col})"
|
|
543
|
+
cmp = _CMP_RE.match(check)
|
|
544
|
+
if cmp:
|
|
545
|
+
return f"_v_cmp({col}, {cmp.group(1)!r}, {float(cmp.group(2))!r})"
|
|
546
|
+
if check == "in" or check.startswith("in ") or check.startswith("in["):
|
|
547
|
+
return f"_v_in({col}, {_membership_values(rule)!r})"
|
|
548
|
+
if check.startswith("matches") or check.startswith("regex"):
|
|
549
|
+
pattern = rule.params.get("pattern") or re.sub(r"^(matches|regex):?\s*", "", check)
|
|
550
|
+
from ._util import safe_compile_regex
|
|
551
|
+
|
|
552
|
+
safe_compile_regex(str(pattern))
|
|
553
|
+
return f"_v_matches({col}, {pattern!r})"
|
|
554
|
+
return None
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
def _emit_validations(rules: list[ValidationRule], unsupported: list[str]) -> list[str]:
|
|
558
|
+
"""Emit ALL rules evaluated against one snapshot, then remove the union once.
|
|
559
|
+
|
|
560
|
+
Mirrors :func:`cleanframe.validate.apply_validations`: every rule's pass-mask is
|
|
561
|
+
computed on the same (pre-null, pre-drop) frame, so non-row-local checks like
|
|
562
|
+
``unique`` see all rows — applying rules sequentially would evaluate ``unique``
|
|
563
|
+
over the already-shrunk frame and diverge from the executor.
|
|
564
|
+
"""
|
|
565
|
+
lines = [" # --- validation (all masks over one snapshot; union removed once) ---"]
|
|
566
|
+
lines.append(" _remove = pd.Series(False, index=df.index)")
|
|
567
|
+
null_targets: list[tuple[int, str]] = []
|
|
568
|
+
idx = 0
|
|
569
|
+
for rule in rules:
|
|
570
|
+
expr = _mask_expr(rule)
|
|
571
|
+
label = f"{rule.column}:{rule.check}"
|
|
572
|
+
if expr is None:
|
|
573
|
+
unsupported.append(f"validation {label}")
|
|
574
|
+
lines.append(
|
|
575
|
+
f" # NOTE: validation {_comment(label)} (custom check) not reproduced"
|
|
576
|
+
)
|
|
577
|
+
continue
|
|
578
|
+
lines.append(
|
|
579
|
+
f" _pass{idx} = ({expr}).astype(bool) # {_comment(label)} "
|
|
580
|
+
f"(on_fail={rule.on_fail})"
|
|
581
|
+
)
|
|
582
|
+
if rule.on_fail in ("quarantine", "drop"):
|
|
583
|
+
lines.append(f" _remove = _remove | ~_pass{idx}")
|
|
584
|
+
elif rule.on_fail == "error":
|
|
585
|
+
lines.append(f" if (~_pass{idx}).any():")
|
|
586
|
+
lines.append(f" raise ValueError({f'validation failed: {label}'!r})")
|
|
587
|
+
elif rule.on_fail == "null":
|
|
588
|
+
null_targets.append((idx, rule.column))
|
|
589
|
+
# warn -> reported only; nothing to enforce here
|
|
590
|
+
idx += 1
|
|
591
|
+
# Null-blank against the snapshot (masks were all computed pre-null), then drop once.
|
|
592
|
+
for i, col in null_targets:
|
|
593
|
+
lines.append(f" df.loc[~_pass{i}, {col!r}] = np.nan")
|
|
594
|
+
lines.append(" df = df[~_remove]")
|
|
595
|
+
return lines
|
|
596
|
+
|
|
597
|
+
|
|
598
|
+
def generate_code(
|
|
599
|
+
recipe: Recipe, func_name: str = "clean", *, allow_partial: bool = False
|
|
600
|
+
) -> str:
|
|
601
|
+
"""Render ``recipe`` to a standalone pandas module defining ``func_name(df)``.
|
|
602
|
+
|
|
603
|
+
Raises :class:`~cleanframe.errors.CleanFrameError` when the recipe uses a custom
|
|
604
|
+
op or check the exporter cannot reproduce, because silently dropping a step would
|
|
605
|
+
make the exported code disagree with the executor. Pass ``allow_partial=True`` to
|
|
606
|
+
accept a partial export whose gaps are marked with ``# NOTE`` comments.
|
|
607
|
+
"""
|
|
608
|
+
if not isinstance(recipe, Recipe):
|
|
609
|
+
raise CleanFrameError(
|
|
610
|
+
f"generate_code expects a Recipe, got {type(recipe).__name__}. Load it with "
|
|
611
|
+
"cleanframe.Recipe.load() first."
|
|
612
|
+
)
|
|
613
|
+
unsupported: list[str] = []
|
|
614
|
+
lines: list[str] = [_DOC_AND_IMPORTS, _constants_source(), _HELPERS, ""]
|
|
615
|
+
lines.append(f"def {func_name}(df):")
|
|
616
|
+
lines.append(' """Clean a DataFrame according to the exported recipe. Returns a new frame."""')
|
|
617
|
+
lines.append(" df = df.copy()")
|
|
618
|
+
|
|
619
|
+
for col_recipe in recipe.columns:
|
|
620
|
+
if not col_recipe.ops and not col_recipe.rename_to:
|
|
621
|
+
continue
|
|
622
|
+
lines.append("")
|
|
623
|
+
lines.append(f" # --- {_comment(col_recipe.source)} ---")
|
|
624
|
+
for op in col_recipe.ops:
|
|
625
|
+
lines.extend(_emit_column_op(op, col_recipe.source, unsupported))
|
|
626
|
+
|
|
627
|
+
renames = {c.source: c.rename_to for c in recipe.columns if c.rename_to}
|
|
628
|
+
if renames:
|
|
629
|
+
lines.append("")
|
|
630
|
+
lines.append(f" df = df.rename(columns={renames!r})")
|
|
631
|
+
|
|
632
|
+
for op in recipe.frame_ops:
|
|
633
|
+
lines.append("")
|
|
634
|
+
if op.name == "dedup":
|
|
635
|
+
subset = op.params.get("subset")
|
|
636
|
+
keep = op.params.get("keep", "first")
|
|
637
|
+
if op.params.get("ignore_case"):
|
|
638
|
+
cols_expr = f"{subset!r}" if subset else "list(df.columns)"
|
|
639
|
+
keep_val = keep if keep is not False else False
|
|
640
|
+
lines.append(
|
|
641
|
+
f" _key = df[{cols_expr}].apply("
|
|
642
|
+
"lambda s: s.map(lambda v: v.strip().casefold() if isinstance(v, str) else v))"
|
|
643
|
+
)
|
|
644
|
+
lines.append(f" df = df[~_key.duplicated(keep={keep_val!r})]")
|
|
645
|
+
else:
|
|
646
|
+
lines.append(f" df = df.drop_duplicates(subset={subset!r}, keep={keep!r})")
|
|
647
|
+
elif op.name == "drop_columns":
|
|
648
|
+
lines.append(f" df = df.drop(columns={op.params.get('columns', [])!r}, errors='ignore')")
|
|
649
|
+
|
|
650
|
+
if recipe.validations:
|
|
651
|
+
lines.append("")
|
|
652
|
+
lines.extend(_emit_validations(recipe.validations, unsupported))
|
|
653
|
+
|
|
654
|
+
lines.append("")
|
|
655
|
+
lines.append(" return df")
|
|
656
|
+
lines.append("")
|
|
657
|
+
if unsupported and not allow_partial:
|
|
658
|
+
raise CleanFrameError(
|
|
659
|
+
"Cannot export this recipe as standalone pandas: "
|
|
660
|
+
f"{'; '.join(unsupported)} has no code equivalent. Pass allow_partial=True "
|
|
661
|
+
"to export the rest with the gaps marked."
|
|
662
|
+
)
|
|
663
|
+
return "\n".join(lines)
|
|
664
|
+
|
|
665
|
+
|
|
666
|
+
__all__ = ["generate_code"]
|