cleanframe-engine 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. cleanframe/__init__.py +169 -0
  2. cleanframe/__main__.py +5 -0
  3. cleanframe/_util.py +438 -0
  4. cleanframe/_version.py +1 -0
  5. cleanframe/api.py +559 -0
  6. cleanframe/cli.py +688 -0
  7. cleanframe/codegen.py +666 -0
  8. cleanframe/dataio.py +506 -0
  9. cleanframe/detectors/__init__.py +43 -0
  10. cleanframe/detectors/base.py +221 -0
  11. cleanframe/detectors/categories.py +199 -0
  12. cleanframe/detectors/contacts.py +105 -0
  13. cleanframe/detectors/currency.py +112 -0
  14. cleanframe/detectors/dates.py +204 -0
  15. cleanframe/detectors/dedup.py +150 -0
  16. cleanframe/detectors/nulls.py +109 -0
  17. cleanframe/detectors/outliers.py +73 -0
  18. cleanframe/detectors/schema_mapping.py +125 -0
  19. cleanframe/detectors/text.py +105 -0
  20. cleanframe/detectors/units.py +86 -0
  21. cleanframe/diff.py +369 -0
  22. cleanframe/drift.py +283 -0
  23. cleanframe/errors.py +66 -0
  24. cleanframe/executor.py +229 -0
  25. cleanframe/fingerprint.py +83 -0
  26. cleanframe/issues.py +186 -0
  27. cleanframe/llm.py +811 -0
  28. cleanframe/ops.py +1245 -0
  29. cleanframe/planner.py +353 -0
  30. cleanframe/profile.py +413 -0
  31. cleanframe/py.typed +1 -0
  32. cleanframe/quality.py +81 -0
  33. cleanframe/readfix.py +160 -0
  34. cleanframe/recipe.py +398 -0
  35. cleanframe/report.py +345 -0
  36. cleanframe/result.py +144 -0
  37. cleanframe/schema.py +259 -0
  38. cleanframe/streaming.py +354 -0
  39. cleanframe/types.py +119 -0
  40. cleanframe/validate.py +363 -0
  41. cleanframe/workbook.py +370 -0
  42. cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
  43. cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
  44. cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
  45. cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
  46. cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
cleanframe/codegen.py ADDED
@@ -0,0 +1,666 @@
1
+ """Export a recipe to standalone, readable pandas — no CleanFrame dependency.
2
+
3
+ ``result.code.save("clean_customers.py")`` produces a plain ``clean(df)`` function
4
+ you can read, diff, and drop into a pipeline that never imports CleanFrame.
5
+
6
+ Fidelity is a load-bearing invariant: the generated code must reproduce the
7
+ executor's output *exactly*. To keep the two from drifting, the lookup tables
8
+ (currency symbols, NA tokens, unit factors, date formats) are rendered here from
9
+ the single source of truth in :mod:`cleanframe.ops`, and the emitted helper bodies
10
+ mirror the executor's scalar logic. ``tests/test_wave1_codegen.py`` locks this by
11
+ running representative recipes through both paths and asserting frame equality.
12
+
13
+ Validation is reproduced for the built-in checks (quarantine/drop filter the
14
+ primary frame; ``null`` blanks the cell; ``error`` raises). The quarantine *side*
15
+ frame is a CleanFrame runtime concept and cannot round-trip through a ``df -> df``
16
+ function, so only the cleaned output is reproduced.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import re
22
+ from collections.abc import Callable
23
+
24
+ from .errors import CleanFrameError
25
+ from .recipe import Recipe, ValidationRule
26
+ from .types import Op
27
+
28
+ _UNSAFE_COMMENT_RE = re.compile(r"[\r\n]+")
29
+
30
+
31
+ def _comment(text: object) -> str:
32
+ """One-line, code-safe rendering of user text for a generated comment.
33
+
34
+ A column name containing a newline would otherwise continue the generated module
35
+ on the next line — outside the comment — as executable code.
36
+ """
37
+ return _UNSAFE_COMMENT_RE.sub(" ", str(text))[:120]
38
+
39
+
40
+ # ---------------------------------------------------------------------------
41
+ # Constant tables rendered from the executor's single source of truth
42
+ # ---------------------------------------------------------------------------
43
+ def _constants_source() -> str:
44
+ from .ops import (
45
+ _KNOWN_CODES,
46
+ _UNIT_ALIASES,
47
+ _UNIT_TO_FAMILY,
48
+ CURRENCY_SYMBOLS,
49
+ DEFAULT_NA_TOKENS,
50
+ UNIT_FAMILIES,
51
+ )
52
+ from .profile import COMMON_DATE_FORMATS
53
+
54
+ unit_factors = {u: f for fam in UNIT_FAMILIES.values() for u, f in fam.items()}
55
+ na_tokens = sorted({t.casefold() for t in DEFAULT_NA_TOKENS}) # includes '' (H9)
56
+ return "\n".join(
57
+ [
58
+ f"_NA_TOKENS = set({na_tokens!r})",
59
+ f"_CURRENCY_SYMBOLS = {dict(CURRENCY_SYMBOLS)!r}",
60
+ f"_KNOWN_CODES = set({sorted(_KNOWN_CODES)!r})",
61
+ f"_UNIT_FACTORS = {unit_factors!r}",
62
+ f"_UNIT_FAMILY = {dict(_UNIT_TO_FAMILY)!r}",
63
+ f"_UNIT_ALIASES = {dict(_UNIT_ALIASES)!r}",
64
+ f"_COMMON_DATE_FORMATS = {list(COMMON_DATE_FORMATS)!r}",
65
+ r"_CODE_RE = re.compile(r'\b([A-Z]{3})\b')",
66
+ r"_UNIT_VALUE_RE = re.compile(r'^\s*([+-]?\d+(?:[.,]\d+)?)\s*([A-Za-z]+)\s*$')",
67
+ r"_PHONE_EXT_RE = re.compile(r'[\s,;]*(?:ext|extn|x|#)\.?\s*\d+\s*$', re.IGNORECASE)",
68
+ r"_NUMBER_TOKEN_RE = re.compile(r'[+-]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][+-]?\d+)?')",
69
+ r"_STRICT_NUM_RE = re.compile(r'^[+-]?\d+(\.\d+)?$')",
70
+ r"_EMAIL_RE = re.compile(r'^[^@\s]+@[^@\s]+\.[^@\s]+$')",
71
+ r"_URL_RE = re.compile(r'^(https?://|www\.)\S+$', re.IGNORECASE)",
72
+ "_TRUE_TOKENS = {'true', 't', 'yes', 'y', '1'}",
73
+ "_FALSE_TOKENS = {'false', 'f', 'no', 'n', '0'}",
74
+ ]
75
+ )
76
+
77
+
78
+ _DOC_AND_IMPORTS = '''"""Auto-generated by CleanFrame. Deterministic, dependency-free pandas.
79
+
80
+ Edit freely — this file has no third-party dependency. Regenerate with
81
+ ``result.code.save(...)`` if you change the recipe.
82
+ """
83
+
84
+ import re
85
+ import warnings
86
+
87
+ import numpy as np
88
+ import pandas as pd
89
+ '''
90
+
91
+ # Helper bodies mirror cleanframe.ops / cleanframe.validate scalar logic exactly.
92
+ _HELPERS = '''
93
+ def _smap(series, fn):
94
+ """Apply fn to string cells only; leave NaN and non-strings untouched."""
95
+ return series.map(lambda v: fn(v) if isinstance(v, str) else v)
96
+
97
+
98
+ def _to_na(series, tokens=None):
99
+ toks = _NA_TOKENS if tokens is None else {str(t).strip().casefold() for t in tokens}
100
+ return _smap(series, lambda v: np.nan if v.strip().casefold() in toks else v)
101
+
102
+
103
+ def _parse_number(series, decimal=".", thousands=",", symbols=()):
104
+ def one(v):
105
+ if v is None or (isinstance(v, float) and np.isnan(v)) or isinstance(v, bool):
106
+ return np.nan
107
+ if isinstance(v, (int, float)):
108
+ return float(v)
109
+ s = str(v).strip().replace("\\u2212", "-")
110
+ if s == "":
111
+ return np.nan
112
+ neg = s.startswith("(") and s.endswith(")")
113
+ if neg:
114
+ s = s[1:-1]
115
+ for sym in symbols:
116
+ s = s.replace(sym, "")
117
+ if thousands:
118
+ s = s.replace(thousands, "")
119
+ if decimal != ".":
120
+ s = s.replace(decimal, ".")
121
+ s = s.strip()
122
+ trailing_minus = s.endswith("-")
123
+ m = _NUMBER_TOKEN_RE.search(s)
124
+ if not m:
125
+ return np.nan
126
+ token = m.group(0)
127
+ leftover = s[: m.start()] + s[m.end() :]
128
+ if any(ch.isdigit() for ch in leftover):
129
+ return np.nan
130
+ try:
131
+ result = float(token)
132
+ except ValueError:
133
+ return np.nan
134
+ if neg:
135
+ result = -abs(result)
136
+ elif trailing_minus and not token.startswith("-"):
137
+ result = -result
138
+ return result
139
+
140
+ return series.map(one)
141
+
142
+
143
+ def _parse_dates_to_datetime(series, formats=None, dayfirst=False, yearfirst=False):
144
+ flex = False
145
+ if not formats:
146
+ formats = _COMMON_DATE_FORMATS
147
+ flex = True
148
+ result = pd.Series(pd.NaT, index=series.index, dtype="datetime64[ns]")
149
+ for fmt in formats:
150
+ mask = result.isna() & series.notna()
151
+ if not mask.any():
152
+ break
153
+ result.loc[mask] = pd.to_datetime(series[mask], format=fmt, errors="coerce")
154
+ if flex:
155
+ remaining = result.isna() & series.notna()
156
+ if remaining.any():
157
+ with warnings.catch_warnings():
158
+ warnings.simplefilter("ignore")
159
+ result.loc[remaining] = pd.to_datetime(
160
+ series[remaining], errors="coerce", dayfirst=dayfirst, yearfirst=yearfirst
161
+ )
162
+ return result
163
+
164
+
165
+ def _parse_date(series, formats=None, dayfirst=False, yearfirst=False, output="%Y-%m-%d"):
166
+ dt = _parse_dates_to_datetime(series, formats, dayfirst=dayfirst, yearfirst=yearfirst)
167
+ if str(output).lower() in ("datetime", "raw", "none"):
168
+ return dt
169
+ fmt = "%Y-%m-%d" if str(output).lower() in ("iso", "date") else output
170
+ return dt.dt.strftime(fmt).where(dt.notna(), np.nan)
171
+
172
+
173
+ def _detect_currency(v, default=None):
174
+ if v is None or (isinstance(v, float) and np.isnan(v)):
175
+ return default if default is not None else np.nan
176
+ s = str(v)
177
+ for sym, code in _CURRENCY_SYMBOLS.items():
178
+ if sym in s:
179
+ return code
180
+ m = _CODE_RE.search(s.upper())
181
+ if m and m.group(1) in _KNOWN_CODES:
182
+ return m.group(1)
183
+ return default if default is not None else np.nan
184
+
185
+
186
+ def _phone_text(v):
187
+ """Stringify a phone cell without inventing digits (a float column has a '.0')."""
188
+ if isinstance(v, float) and float(v).is_integer():
189
+ return str(int(v))
190
+ return str(v)
191
+
192
+
193
+ def _normalize_phone(series, default_country_code=None):
194
+ def one(v):
195
+ if v is None or (isinstance(v, float) and np.isnan(v)):
196
+ return v
197
+ if isinstance(v, bool):
198
+ return np.nan
199
+ s = _PHONE_EXT_RE.sub("", _phone_text(v))
200
+ plus = s.strip().startswith("+")
201
+ digits = re.sub(r"\\D", "", s)
202
+ if not digits:
203
+ return np.nan
204
+ if plus:
205
+ return "+" + digits # noqa: RET504
206
+ if default_country_code:
207
+ cc = re.sub(r"\\D", "", str(default_country_code))
208
+ if cc and digits.startswith(cc):
209
+ return "+" + digits
210
+ return "+" + cc + digits.lstrip("0")
211
+ return digits
212
+
213
+ return series.map(one)
214
+
215
+
216
+ def _parse_unit_scalar(v):
217
+ if v is None or (isinstance(v, float) and np.isnan(v)):
218
+ return None
219
+ if isinstance(v, (int, float, bool)):
220
+ return None
221
+ m = _UNIT_VALUE_RE.match(str(v))
222
+ if not m:
223
+ return None
224
+ num_s, unit_s = m.group(1), m.group(2).casefold()
225
+ unit_s = _UNIT_ALIASES.get(unit_s, unit_s)
226
+ if unit_s not in _UNIT_FAMILY:
227
+ return None
228
+ try:
229
+ return float(num_s.replace(",", ".")), unit_s
230
+ except ValueError:
231
+ return None
232
+
233
+
234
+ def _normalize_unit(series, to="g", emit_unit_column=False):
235
+ to = _UNIT_ALIASES.get(str(to).casefold(), str(to).casefold())
236
+ target_fam = _UNIT_FAMILY[to]
237
+ target_factor = _UNIT_FACTORS[to]
238
+ amounts, units = [], []
239
+ for v in series.tolist():
240
+ parsed = _parse_unit_scalar(v)
241
+ if parsed is None:
242
+ if isinstance(v, (int, float)) and not isinstance(v, bool) and not (isinstance(v, float) and np.isnan(v)):
243
+ amounts.append(float(v))
244
+ units.append(to)
245
+ elif isinstance(v, str) and _STRICT_NUM_RE.match(v.strip().replace(",", ".")):
246
+ try:
247
+ amounts.append(float(v.replace(",", ".").strip()))
248
+ units.append(to)
249
+ except ValueError:
250
+ amounts.append(np.nan)
251
+ units.append(None)
252
+ else:
253
+ amounts.append(np.nan)
254
+ units.append(None)
255
+ continue
256
+ amount, unit = parsed
257
+ if _UNIT_FAMILY[unit] != target_fam:
258
+ amounts.append(np.nan)
259
+ units.append(unit)
260
+ continue
261
+ amounts.append(amount * _UNIT_FACTORS[unit] / target_factor)
262
+ units.append(unit)
263
+ out = pd.Series(amounts, index=series.index, dtype="float64")
264
+ if emit_unit_column:
265
+ return out, pd.Series(units, index=series.index)
266
+ return out
267
+
268
+
269
+ def _to_bool(series):
270
+ def one(v):
271
+ if v is None or (isinstance(v, float) and np.isnan(v)):
272
+ return pd.NA
273
+ if isinstance(v, bool):
274
+ return v
275
+ t = str(v).strip().casefold()
276
+ if t in _TRUE_TOKENS:
277
+ return True
278
+ if t in _FALSE_TOKENS:
279
+ return False
280
+ return pd.NA
281
+
282
+ return series.map(one).astype("boolean")
283
+
284
+
285
+ # -- validation pass-masks (True = passes), mirroring cleanframe.validate ------
286
+ def _v_not_null(s):
287
+ return s.notna()
288
+
289
+
290
+ def _v_unique(s):
291
+ return ~(s.duplicated(keep=False) & s.notna())
292
+
293
+
294
+ def _v_valid_email(s):
295
+ return s.isna() | s.map(lambda v: bool(_EMAIL_RE.match(str(v).strip().lower())))
296
+
297
+
298
+ def _v_valid_url(s):
299
+ return s.isna() | s.map(lambda v: bool(_URL_RE.match(str(v).strip())))
300
+
301
+
302
+ def _v_valid_phone(s):
303
+ return s.isna() | s.map(lambda v: 7 <= len(re.sub(r"\\D", "", str(v))) <= 15)
304
+
305
+
306
+ _CMP_OPS = {
307
+ ">=": lambda a, b: a >= b, "<=": lambda a, b: a <= b, ">": lambda a, b: a > b,
308
+ "<": lambda a, b: a < b, "==": lambda a, b: a == b, "!=": lambda a, b: a != b,
309
+ }
310
+
311
+
312
+ def _v_cmp(s, op, threshold):
313
+ numeric = pd.to_numeric(s, errors="coerce")
314
+ satisfies = _CMP_OPS[op](numeric, threshold).fillna(False).astype(bool)
315
+ return s.isna() | (numeric.notna() & satisfies)
316
+
317
+
318
+ def _v_in(s, values):
319
+ _BOOL_SPELLINGS = {
320
+ True: ("True", "true", "TRUE", "Yes", "yes", "YES", "On", "on", "Y", "y", "1"),
321
+ False: ("False", "false", "FALSE", "No", "no", "NO", "Off", "off", "N", "n", "0"),
322
+ }
323
+ as_str = set()
324
+ for v in values:
325
+ if isinstance(v, bool):
326
+ as_str.update(_BOOL_SPELLINGS[v])
327
+ else:
328
+ as_str.add(str(v))
329
+ return s.isna() | s.isin(values) | s.astype(str).isin(as_str)
330
+
331
+
332
+ def _v_matches(s, pattern):
333
+ if len(pattern) > 500:
334
+ raise ValueError(f"Regex pattern exceeds limit of 500 characters ({len(pattern)}).")
335
+ compiled = re.compile(pattern)
336
+ return s.isna() | s.map(lambda v: bool(compiled.search(str(v))))
337
+ '''
338
+
339
+
340
+ def _col(name: str) -> str:
341
+ return f"df[{name!r}]"
342
+
343
+
344
+ # -- per-op code emitters ----------------------------------------------------
345
+ def _remove_symbols(params: dict, c: str) -> list[str]:
346
+ chain = "".join(f".replace({s!r}, '')" for s in params.get("symbols", []))
347
+ return [f" {_col(c)} = _smap({_col(c)}, lambda v: v{chain})"]
348
+
349
+
350
+ def _replace(params: dict, c: str) -> list[str]:
351
+ pattern = params["pattern"]
352
+ if params.get("regex", True):
353
+ # Fail at export time with the same ReDoS / length guards the executor uses.
354
+ from ._util import safe_compile_regex
355
+
356
+ safe_compile_regex(pattern)
357
+ return [
358
+ f" {_col(c)} = _smap({_col(c)}, "
359
+ f"lambda v: re.sub({pattern!r}, {params.get('repl', '')!r}, v))"
360
+ ]
361
+ return [
362
+ f" {_col(c)} = _smap({_col(c)}, "
363
+ f"lambda v: v.replace({pattern!r}, {params.get('repl', '')!r}))"
364
+ ]
365
+
366
+
367
+ def _cast(params: dict, c: str) -> list[str]:
368
+ to = str(params["to"]).lower()
369
+ col = _col(c)
370
+ expr = {
371
+ "float": f"pd.to_numeric({col}, errors='coerce').astype('float64')",
372
+ "float64": f"pd.to_numeric({col}, errors='coerce').astype('float64')",
373
+ "number": f"pd.to_numeric({col}, errors='coerce').astype('float64')",
374
+ "int": f"pd.to_numeric({col}, errors='coerce').round().astype('Int64')",
375
+ "integer": f"pd.to_numeric({col}, errors='coerce').round().astype('Int64')",
376
+ "int64": f"pd.to_numeric({col}, errors='coerce').round().astype('Int64')",
377
+ "string": f"{col}.astype('string')",
378
+ "str": f"{col}.astype('string')",
379
+ "text": f"{col}.astype('string')",
380
+ "bool": f"_to_bool({col})",
381
+ "boolean": f"_to_bool({col})",
382
+ "datetime": f"_parse_dates_to_datetime({col})",
383
+ "date": f"_parse_dates_to_datetime({col})",
384
+ "category": f"{col}.astype('category')",
385
+ }.get(to)
386
+ if expr is None:
387
+ return [f" # NOTE: cast to {to!r} not reproduced"]
388
+ return [f" {col} = {expr}"]
389
+
390
+
391
+ def _normalize_values(params: dict, c: str) -> list[str]:
392
+ mapping = params.get("map", {})
393
+ if params.get("case_insensitive"):
394
+ return [
395
+ f" _m = {{str(k).strip().casefold(): v for k, v in {mapping!r}.items()}}",
396
+ f" {_col(c)} = _smap({_col(c)}, lambda v: _m.get(v.strip().casefold(), v))",
397
+ ]
398
+ return [
399
+ f" _m = {mapping!r}",
400
+ f" {_col(c)} = {_col(c)}.map(lambda v: _m.get(v, _m.get(str(v), v)) "
401
+ f"if isinstance(v, str) else v)",
402
+ ]
403
+
404
+
405
+ def _parse_date_gen(params: dict, c: str) -> list[str]:
406
+ args = []
407
+ if params.get("formats"):
408
+ args.append(f"formats={list(params['formats'])!r}")
409
+ if params.get("dayfirst"):
410
+ args.append("dayfirst=True")
411
+ if params.get("yearfirst"):
412
+ args.append("yearfirst=True")
413
+ out = params.get("output", "%Y-%m-%d")
414
+ if out != "%Y-%m-%d":
415
+ args.append(f"output={out!r}")
416
+ tail = (", " + ", ".join(args)) if args else ""
417
+ return [f" {_col(c)} = _parse_date({_col(c)}{tail})"]
418
+
419
+
420
+ def _parse_number_gen(params: dict, c: str) -> list[str]:
421
+ args = []
422
+ if params.get("decimal", ".") != ".":
423
+ args.append(f"decimal={params['decimal']!r}")
424
+ if params.get("thousands", ",") != ",":
425
+ args.append(f"thousands={params['thousands']!r}")
426
+ if params.get("symbols"):
427
+ args.append(f"symbols={list(params['symbols'])!r}")
428
+ tail = (", " + ", ".join(args)) if args else ""
429
+ return [f" {_col(c)} = _parse_number({_col(c)}{tail})"]
430
+
431
+
432
+ def _extract_currency_gen(params: dict, c: str) -> list[str]:
433
+ to = params.get("to") or f"{c}_currency"
434
+ default = params.get("default")
435
+ return [f" df[{to!r}] = {_col(c)}.map(lambda v: _detect_currency(v, {default!r}))"]
436
+
437
+
438
+ def _normalize_unit_gen(params: dict, c: str) -> list[str]:
439
+ to = params.get("to", "g")
440
+ emit = params.get("emit_unit_column")
441
+ if emit:
442
+ return [
443
+ f" {_col(c)}, df[{emit!r}] = _normalize_unit("
444
+ f"{_col(c)}, to={to!r}, emit_unit_column=True)"
445
+ ]
446
+ return [f" {_col(c)} = _normalize_unit({_col(c)}, to={to!r})"]
447
+
448
+
449
+ def _fill_na_gen(params: dict, c: str) -> list[str]:
450
+ col = _col(c)
451
+ strategy = params.get("strategy")
452
+ if strategy is None:
453
+ return [f" {col} = {col}.fillna({params.get('value')!r})"]
454
+ strategy = str(strategy).lower()
455
+ if strategy == "mean":
456
+ return [f" {col} = {col}.fillna(pd.to_numeric({col}, errors='coerce').mean())"]
457
+ if strategy == "median":
458
+ return [f" {col} = {col}.fillna(pd.to_numeric({col}, errors='coerce').median())"]
459
+ if strategy == "mode":
460
+ return [
461
+ f" _modes = {col}.dropna().mode()",
462
+ f" {col} = {col} if len(_modes) == 0 else {col}.fillna(sorted(_modes.tolist(), key=str)[0])",
463
+ ]
464
+ if strategy in ("ffill", "pad"):
465
+ return [f" {col} = {col}.ffill()"]
466
+ if strategy in ("bfill", "backfill"):
467
+ return [f" {col} = {col}.bfill()"]
468
+ if strategy == "zero":
469
+ return [f" {col} = {col}.fillna(0)"]
470
+ if strategy == "empty":
471
+ return [f" {col} = {col}.fillna('')"]
472
+ return [f" # NOTE: unknown fill_na strategy {strategy!r} not reproduced"]
473
+
474
+
475
+ _SIMPLE: dict[str, str] = {
476
+ "strip_whitespace": "lambda v: v.strip()",
477
+ "collapse_whitespace": "lambda v: re.sub(r'\\s+', ' ', v).strip()",
478
+ "lowercase": "lambda v: v.lower()",
479
+ "uppercase": "lambda v: v.upper()",
480
+ "title_case": "lambda v: re.sub(r'\\s+', ' ', v).strip().title()",
481
+ "capitalize": "lambda v: v.strip().capitalize()",
482
+ "normalize_email": "lambda v: v.strip().lower()",
483
+ }
484
+
485
+ _EMITTERS: dict[str, Callable[[dict, str], list[str]]] = {
486
+ "remove_symbols": _remove_symbols,
487
+ "replace": _replace,
488
+ "cast": _cast,
489
+ "normalize_values": _normalize_values,
490
+ "parse_date": _parse_date_gen,
491
+ "parse_number": _parse_number_gen,
492
+ "extract_currency": _extract_currency_gen,
493
+ "normalize_unit": _normalize_unit_gen,
494
+ }
495
+
496
+
497
+ def _emit_column_op(op: Op, source: str, unsupported: list[str] | None = None) -> list[str]:
498
+ if op.name in _SIMPLE:
499
+ return [f" {_col(source)} = _smap({_col(source)}, {_SIMPLE[op.name]})"]
500
+ if op.name in _EMITTERS:
501
+ return _EMITTERS[op.name](op.params, source)
502
+ if op.name == "to_na":
503
+ tokens = op.params.get("tokens")
504
+ arg = f", tokens={list(tokens)!r}" if tokens else ""
505
+ return [f" {_col(source)} = _to_na({_col(source)}{arg})"]
506
+ if op.name == "normalize_phone":
507
+ cc = op.params.get("default_country_code")
508
+ arg = f"default_country_code={cc!r}" if cc else ""
509
+ return [f" {_col(source)} = _normalize_phone({_col(source)}{', ' + arg if arg else ''})"]
510
+ if op.name == "fill_na":
511
+ return _fill_na_gen(op.params, source)
512
+ if op.name == "round":
513
+ return [
514
+ f" {_col(source)} = pd.to_numeric({_col(source)}, errors='coerce')"
515
+ f".round({op.params.get('decimals', 0)})"
516
+ ]
517
+ if unsupported is not None:
518
+ unsupported.append(f"op {op.name!r} on column {source!r}")
519
+ return [f" # NOTE: op {_comment(op.name)!r} is not reproduced by codegen"]
520
+
521
+
522
+ # -- validation emission -----------------------------------------------------
523
+ _NAMED_VALIDATORS = {"not_null", "unique", "valid_email", "valid_url", "valid_phone"}
524
+ _CMP_RE = re.compile(r"^(>=|<=|==|!=|>|<)\s*(-?\d+(?:\.\d+)?)$")
525
+
526
+
527
+ def _membership_values(rule: ValidationRule) -> list:
528
+ values = rule.params.get("values")
529
+ if values is not None:
530
+ return [values] if isinstance(values, str) else list(values)
531
+ import yaml
532
+
533
+ rest = rule.check[2:].strip()
534
+ parsed = yaml.load(rest, Loader=yaml.BaseLoader) if rest else []
535
+ return list(parsed) if isinstance(parsed, (list, tuple)) else [parsed]
536
+
537
+
538
+ def _mask_expr(rule: ValidationRule) -> str | None:
539
+ check = rule.check.strip()
540
+ col = _col(rule.column)
541
+ if check in _NAMED_VALIDATORS:
542
+ return f"_v_{check}({col})"
543
+ cmp = _CMP_RE.match(check)
544
+ if cmp:
545
+ return f"_v_cmp({col}, {cmp.group(1)!r}, {float(cmp.group(2))!r})"
546
+ if check == "in" or check.startswith("in ") or check.startswith("in["):
547
+ return f"_v_in({col}, {_membership_values(rule)!r})"
548
+ if check.startswith("matches") or check.startswith("regex"):
549
+ pattern = rule.params.get("pattern") or re.sub(r"^(matches|regex):?\s*", "", check)
550
+ from ._util import safe_compile_regex
551
+
552
+ safe_compile_regex(str(pattern))
553
+ return f"_v_matches({col}, {pattern!r})"
554
+ return None
555
+
556
+
557
+ def _emit_validations(rules: list[ValidationRule], unsupported: list[str]) -> list[str]:
558
+ """Emit ALL rules evaluated against one snapshot, then remove the union once.
559
+
560
+ Mirrors :func:`cleanframe.validate.apply_validations`: every rule's pass-mask is
561
+ computed on the same (pre-null, pre-drop) frame, so non-row-local checks like
562
+ ``unique`` see all rows — applying rules sequentially would evaluate ``unique``
563
+ over the already-shrunk frame and diverge from the executor.
564
+ """
565
+ lines = [" # --- validation (all masks over one snapshot; union removed once) ---"]
566
+ lines.append(" _remove = pd.Series(False, index=df.index)")
567
+ null_targets: list[tuple[int, str]] = []
568
+ idx = 0
569
+ for rule in rules:
570
+ expr = _mask_expr(rule)
571
+ label = f"{rule.column}:{rule.check}"
572
+ if expr is None:
573
+ unsupported.append(f"validation {label}")
574
+ lines.append(
575
+ f" # NOTE: validation {_comment(label)} (custom check) not reproduced"
576
+ )
577
+ continue
578
+ lines.append(
579
+ f" _pass{idx} = ({expr}).astype(bool) # {_comment(label)} "
580
+ f"(on_fail={rule.on_fail})"
581
+ )
582
+ if rule.on_fail in ("quarantine", "drop"):
583
+ lines.append(f" _remove = _remove | ~_pass{idx}")
584
+ elif rule.on_fail == "error":
585
+ lines.append(f" if (~_pass{idx}).any():")
586
+ lines.append(f" raise ValueError({f'validation failed: {label}'!r})")
587
+ elif rule.on_fail == "null":
588
+ null_targets.append((idx, rule.column))
589
+ # warn -> reported only; nothing to enforce here
590
+ idx += 1
591
+ # Null-blank against the snapshot (masks were all computed pre-null), then drop once.
592
+ for i, col in null_targets:
593
+ lines.append(f" df.loc[~_pass{i}, {col!r}] = np.nan")
594
+ lines.append(" df = df[~_remove]")
595
+ return lines
596
+
597
+
598
+ def generate_code(
599
+ recipe: Recipe, func_name: str = "clean", *, allow_partial: bool = False
600
+ ) -> str:
601
+ """Render ``recipe`` to a standalone pandas module defining ``func_name(df)``.
602
+
603
+ Raises :class:`~cleanframe.errors.CleanFrameError` when the recipe uses a custom
604
+ op or check the exporter cannot reproduce, because silently dropping a step would
605
+ make the exported code disagree with the executor. Pass ``allow_partial=True`` to
606
+ accept a partial export whose gaps are marked with ``# NOTE`` comments.
607
+ """
608
+ if not isinstance(recipe, Recipe):
609
+ raise CleanFrameError(
610
+ f"generate_code expects a Recipe, got {type(recipe).__name__}. Load it with "
611
+ "cleanframe.Recipe.load() first."
612
+ )
613
+ unsupported: list[str] = []
614
+ lines: list[str] = [_DOC_AND_IMPORTS, _constants_source(), _HELPERS, ""]
615
+ lines.append(f"def {func_name}(df):")
616
+ lines.append(' """Clean a DataFrame according to the exported recipe. Returns a new frame."""')
617
+ lines.append(" df = df.copy()")
618
+
619
+ for col_recipe in recipe.columns:
620
+ if not col_recipe.ops and not col_recipe.rename_to:
621
+ continue
622
+ lines.append("")
623
+ lines.append(f" # --- {_comment(col_recipe.source)} ---")
624
+ for op in col_recipe.ops:
625
+ lines.extend(_emit_column_op(op, col_recipe.source, unsupported))
626
+
627
+ renames = {c.source: c.rename_to for c in recipe.columns if c.rename_to}
628
+ if renames:
629
+ lines.append("")
630
+ lines.append(f" df = df.rename(columns={renames!r})")
631
+
632
+ for op in recipe.frame_ops:
633
+ lines.append("")
634
+ if op.name == "dedup":
635
+ subset = op.params.get("subset")
636
+ keep = op.params.get("keep", "first")
637
+ if op.params.get("ignore_case"):
638
+ cols_expr = f"{subset!r}" if subset else "list(df.columns)"
639
+ keep_val = keep if keep is not False else False
640
+ lines.append(
641
+ f" _key = df[{cols_expr}].apply("
642
+ "lambda s: s.map(lambda v: v.strip().casefold() if isinstance(v, str) else v))"
643
+ )
644
+ lines.append(f" df = df[~_key.duplicated(keep={keep_val!r})]")
645
+ else:
646
+ lines.append(f" df = df.drop_duplicates(subset={subset!r}, keep={keep!r})")
647
+ elif op.name == "drop_columns":
648
+ lines.append(f" df = df.drop(columns={op.params.get('columns', [])!r}, errors='ignore')")
649
+
650
+ if recipe.validations:
651
+ lines.append("")
652
+ lines.extend(_emit_validations(recipe.validations, unsupported))
653
+
654
+ lines.append("")
655
+ lines.append(" return df")
656
+ lines.append("")
657
+ if unsupported and not allow_partial:
658
+ raise CleanFrameError(
659
+ "Cannot export this recipe as standalone pandas: "
660
+ f"{'; '.join(unsupported)} has no code equivalent. Pass allow_partial=True "
661
+ "to export the rest with the gaps marked."
662
+ )
663
+ return "\n".join(lines)
664
+
665
+
666
+ __all__ = ["generate_code"]