cleanframe-engine 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanframe/__init__.py +169 -0
- cleanframe/__main__.py +5 -0
- cleanframe/_util.py +438 -0
- cleanframe/_version.py +1 -0
- cleanframe/api.py +559 -0
- cleanframe/cli.py +688 -0
- cleanframe/codegen.py +666 -0
- cleanframe/dataio.py +506 -0
- cleanframe/detectors/__init__.py +43 -0
- cleanframe/detectors/base.py +221 -0
- cleanframe/detectors/categories.py +199 -0
- cleanframe/detectors/contacts.py +105 -0
- cleanframe/detectors/currency.py +112 -0
- cleanframe/detectors/dates.py +204 -0
- cleanframe/detectors/dedup.py +150 -0
- cleanframe/detectors/nulls.py +109 -0
- cleanframe/detectors/outliers.py +73 -0
- cleanframe/detectors/schema_mapping.py +125 -0
- cleanframe/detectors/text.py +105 -0
- cleanframe/detectors/units.py +86 -0
- cleanframe/diff.py +369 -0
- cleanframe/drift.py +283 -0
- cleanframe/errors.py +66 -0
- cleanframe/executor.py +229 -0
- cleanframe/fingerprint.py +83 -0
- cleanframe/issues.py +186 -0
- cleanframe/llm.py +811 -0
- cleanframe/ops.py +1245 -0
- cleanframe/planner.py +353 -0
- cleanframe/profile.py +413 -0
- cleanframe/py.typed +1 -0
- cleanframe/quality.py +81 -0
- cleanframe/readfix.py +160 -0
- cleanframe/recipe.py +398 -0
- cleanframe/report.py +345 -0
- cleanframe/result.py +144 -0
- cleanframe/schema.py +259 -0
- cleanframe/streaming.py +354 -0
- cleanframe/types.py +119 -0
- cleanframe/validate.py +363 -0
- cleanframe/workbook.py +370 -0
- cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
- cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
- cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
- cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
- cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
cleanframe/cli.py
ADDED
|
@@ -0,0 +1,688 @@
|
|
|
1
|
+
"""The ``cleanframe`` command-line interface.
|
|
2
|
+
|
|
3
|
+
Subcommands mirror the library:
|
|
4
|
+
|
|
5
|
+
* ``report FILE`` — write an HTML profiling report.
|
|
6
|
+
* ``clean FILE`` — plan + clean; save recipe, code, cleaned data, report.
|
|
7
|
+
* ``apply FILE --recipe R`` — replay a recipe (with drift check).
|
|
8
|
+
* ``suggest FILE --recipe R`` — show drift and optionally patch the recipe.
|
|
9
|
+
* ``infer-schema FILE`` — draft a target schema.
|
|
10
|
+
* ``detectors`` / ``ops`` — list what's available.
|
|
11
|
+
|
|
12
|
+
Exit codes: ``0`` success, ``1`` data/recipe/output error, ``2`` usage error,
|
|
13
|
+
``3`` stopped on schema drift, ``4`` validation failed, ``70`` internal error,
|
|
14
|
+
``130`` interrupted. Pass ``--debug`` (or set ``CLEANFRAME_DEBUG=1``) to print a
|
|
15
|
+
traceback for an internal error instead of a one-line message.
|
|
16
|
+
|
|
17
|
+
Stdout is switched to UTF-8 so currency symbols and diff glyphs render on any
|
|
18
|
+
terminal (notably Windows consoles).
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import argparse
|
|
24
|
+
import os
|
|
25
|
+
import sys
|
|
26
|
+
import warnings
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
|
|
29
|
+
from ._version import __version__
|
|
30
|
+
from .errors import CleanFrameError, CleanFrameWarning, DriftError, ValidationFailure
|
|
31
|
+
|
|
32
|
+
EXIT_OK = 0
|
|
33
|
+
EXIT_ERROR = 1
|
|
34
|
+
EXIT_USAGE = 2
|
|
35
|
+
EXIT_DRIFT = 3
|
|
36
|
+
EXIT_VALIDATION = 4
|
|
37
|
+
EXIT_INTERNAL = 70
|
|
38
|
+
EXIT_INTERRUPT = 130
|
|
39
|
+
|
|
40
|
+
_ISSUE_URL = "https://github.com/inboxpraveen/Cleanframe/issues"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _reconfigure_stdout() -> None:
|
|
44
|
+
for stream in (sys.stdout, sys.stderr):
|
|
45
|
+
try:
|
|
46
|
+
stream.reconfigure(encoding="utf-8") # type: ignore[union-attr]
|
|
47
|
+
except (AttributeError, ValueError): # pragma: no cover - older/odd streams
|
|
48
|
+
pass
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _emit(text: str, *, stream=None) -> None:
|
|
52
|
+
"""Print a line, degrading gracefully on a console that cannot encode it."""
|
|
53
|
+
target = stream or sys.stdout
|
|
54
|
+
try:
|
|
55
|
+
print(text, file=target)
|
|
56
|
+
except UnicodeEncodeError: # pragma: no cover - depends on console codepage
|
|
57
|
+
encoding = getattr(target, "encoding", None) or "ascii"
|
|
58
|
+
print(text.encode(encoding, "replace").decode(encoding, "replace"), file=target)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _install_warning_format() -> None:
|
|
62
|
+
"""Show advisories as one readable line instead of a file path and source echo."""
|
|
63
|
+
|
|
64
|
+
def show(message, category, filename, lineno, file=None, line=None): # noqa: ANN001
|
|
65
|
+
text = str(message)
|
|
66
|
+
if text.startswith("CleanFrame:"):
|
|
67
|
+
text = text[len("CleanFrame:") :].strip()
|
|
68
|
+
elif not issubclass(category, CleanFrameWarning):
|
|
69
|
+
text = f"{category.__name__}: {text}"
|
|
70
|
+
_emit(f"⚠ {text}", stream=file or sys.stderr)
|
|
71
|
+
|
|
72
|
+
warnings.showwarning = show
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _debug_enabled(args: argparse.Namespace | None = None) -> bool:
|
|
76
|
+
if args is not None and getattr(args, "debug", False):
|
|
77
|
+
return True
|
|
78
|
+
return os.environ.get("CLEANFRAME_DEBUG", "").strip() not in ("", "0", "false", "False")
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
# ---------------------------------------------------------------------------
|
|
82
|
+
# argument helpers
|
|
83
|
+
# ---------------------------------------------------------------------------
|
|
84
|
+
def _nonneg_int(text: str) -> int:
|
|
85
|
+
try:
|
|
86
|
+
value = int(text)
|
|
87
|
+
except ValueError:
|
|
88
|
+
raise argparse.ArgumentTypeError(f"expected a whole number, got {text!r}") from None
|
|
89
|
+
if value < 0:
|
|
90
|
+
raise argparse.ArgumentTypeError(f"must be 0 or more, got {value}")
|
|
91
|
+
return value
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _positive_int(text: str) -> int:
|
|
95
|
+
try:
|
|
96
|
+
value = int(text)
|
|
97
|
+
except ValueError:
|
|
98
|
+
raise argparse.ArgumentTypeError(f"expected a whole number, got {text!r}") from None
|
|
99
|
+
if value < 1:
|
|
100
|
+
raise argparse.ArgumentTypeError(f"must be 1 or more, got {value}")
|
|
101
|
+
return value
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _default_out(file: str, suffix: str) -> Path:
|
|
105
|
+
return Path(file).with_suffix(suffix)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _add_selection_args(parser: argparse.ArgumentParser, *, sheet: bool = True) -> None:
|
|
109
|
+
if sheet:
|
|
110
|
+
parser.add_argument(
|
|
111
|
+
"--sheet",
|
|
112
|
+
help="Excel sheet name, or #N for a 0-based index (e.g. --sheet '#0')",
|
|
113
|
+
)
|
|
114
|
+
parser.add_argument("--columns", help="comma-separated column subset to read")
|
|
115
|
+
parser.add_argument("--nrows", type=_nonneg_int, help="read only the first N data rows")
|
|
116
|
+
parser.add_argument(
|
|
117
|
+
"--skiprows",
|
|
118
|
+
type=_nonneg_int,
|
|
119
|
+
help="skip the first N data rows (the header row is always kept)",
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _add_read_args(parser: argparse.ArgumentParser) -> None:
|
|
124
|
+
parser.add_argument("--sep", help="field delimiter, overriding auto-detection")
|
|
125
|
+
parser.add_argument("--encoding", help="file encoding, overriding auto-detection")
|
|
126
|
+
parser.add_argument(
|
|
127
|
+
"--text",
|
|
128
|
+
action="store_true",
|
|
129
|
+
help="read every field verbatim (keeps leading zeros, 'NA' text and 1e5 exact)",
|
|
130
|
+
)
|
|
131
|
+
parser.add_argument(
|
|
132
|
+
"--no-correct", action="store_true", help="disable read-time format auto-detection"
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _selection_kwargs(args: argparse.Namespace) -> dict:
|
|
137
|
+
out: dict = {}
|
|
138
|
+
sheet = getattr(args, "sheet", None)
|
|
139
|
+
if sheet is not None:
|
|
140
|
+
# Digits alone are sheet *names* (e.g. "2024"). Use "#0" / "#1" for indices.
|
|
141
|
+
if sheet.startswith("#") and sheet[1:].lstrip("-").isdigit():
|
|
142
|
+
out["sheet"] = int(sheet[1:])
|
|
143
|
+
else:
|
|
144
|
+
out["sheet"] = sheet
|
|
145
|
+
cols = getattr(args, "columns", None)
|
|
146
|
+
if cols:
|
|
147
|
+
out["columns"] = [c.strip() for c in cols.split(",") if c.strip()]
|
|
148
|
+
for name in ("nrows", "skiprows"):
|
|
149
|
+
if getattr(args, name, None) is not None:
|
|
150
|
+
out[name] = getattr(args, name)
|
|
151
|
+
return out
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _read_kwargs(args: argparse.Namespace) -> dict:
|
|
155
|
+
out: dict = {}
|
|
156
|
+
if getattr(args, "sep", None):
|
|
157
|
+
out["sep"] = args.sep
|
|
158
|
+
if getattr(args, "encoding", None):
|
|
159
|
+
out["encoding"] = args.encoding
|
|
160
|
+
if getattr(args, "text", False):
|
|
161
|
+
out["text"] = True
|
|
162
|
+
if hasattr(args, "no_correct"):
|
|
163
|
+
out["correct_format"] = not args.no_correct
|
|
164
|
+
return out
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _print_log(args: argparse.Namespace, log: list[str]) -> None:
|
|
168
|
+
if getattr(args, "verbose", False) and log:
|
|
169
|
+
_emit("")
|
|
170
|
+
for line in log:
|
|
171
|
+
_emit(f" · {line}")
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _reserve_outputs(args: argparse.Namespace, *attrs: str) -> None:
|
|
175
|
+
"""Validate output paths up front so a bad one fails before any work is done."""
|
|
176
|
+
from ._util import check_output_target
|
|
177
|
+
|
|
178
|
+
for attr in attrs:
|
|
179
|
+
target = getattr(args, attr, None)
|
|
180
|
+
if target:
|
|
181
|
+
check_output_target(target, args.file, overwrite=getattr(args, "overwrite", False))
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _reject_unsupported(mode: str, args: argparse.Namespace, names: dict[str, str]) -> None:
|
|
185
|
+
"""Fail loudly for flags that would otherwise be silently ignored."""
|
|
186
|
+
given = [flag for attr, flag in names.items() if getattr(args, attr, None)]
|
|
187
|
+
if given:
|
|
188
|
+
raise CleanFrameError(
|
|
189
|
+
f"{', '.join(given)} {'is' if len(given) == 1 else 'are'} not supported in "
|
|
190
|
+
f"{mode}. Re-run without {'it' if len(given) == 1 else 'them'}."
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
# ---------------------------------------------------------------------------
|
|
195
|
+
# command handlers
|
|
196
|
+
# ---------------------------------------------------------------------------
|
|
197
|
+
def _cmd_report(args: argparse.Namespace) -> int:
|
|
198
|
+
from . import report as _report
|
|
199
|
+
from ._util import check_output_target
|
|
200
|
+
|
|
201
|
+
rep = _report(args.file, schema=args.schema, **_selection_kwargs(args), **_read_kwargs(args))
|
|
202
|
+
out = Path(args.out) if args.out else _default_out(args.file, ".report.html")
|
|
203
|
+
rep.save(check_output_target(out, args.file))
|
|
204
|
+
_emit(f"✓ Report written to {out}")
|
|
205
|
+
q = rep.quality
|
|
206
|
+
if q:
|
|
207
|
+
_emit(f" Quality score: {q.score}/100 (grade {q.grade} — {q.label})")
|
|
208
|
+
if args.open:
|
|
209
|
+
import webbrowser
|
|
210
|
+
|
|
211
|
+
webbrowser.open(out.resolve().as_uri())
|
|
212
|
+
return EXIT_OK
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _is_multisheet_workbook(file: str, selection: dict) -> bool:
|
|
216
|
+
"""A multi-sheet .xlsx with no explicit --sheet -> clean every tab (workbook mode)."""
|
|
217
|
+
path = Path(file)
|
|
218
|
+
if path.suffix.lower() not in (".xlsx", ".xls", ".xlsm") or "sheet" in selection:
|
|
219
|
+
return False
|
|
220
|
+
try:
|
|
221
|
+
from .dataio import excel_sheet_names
|
|
222
|
+
|
|
223
|
+
return len(excel_sheet_names(path)) > 1
|
|
224
|
+
except CleanFrameError:
|
|
225
|
+
return False
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _out_dir_targets(args: argparse.Namespace, *, workbook: bool) -> None:
|
|
229
|
+
"""Expand --out-dir into the individual artifact paths."""
|
|
230
|
+
directory = Path(args.out_dir)
|
|
231
|
+
stem = Path(args.file).stem
|
|
232
|
+
args.recipe = args.recipe or str(directory / f"{stem}.recipe.yaml")
|
|
233
|
+
if workbook:
|
|
234
|
+
# A workbook produces one recipe and one rewritten workbook, nothing else.
|
|
235
|
+
args.out = args.out or str(directory / f"{stem}.clean.xlsx")
|
|
236
|
+
return
|
|
237
|
+
args.out = args.out or str(directory / f"{stem}.clean.csv")
|
|
238
|
+
args.code = args.code or str(directory / f"{stem}.py")
|
|
239
|
+
args.report = args.report or str(directory / f"{stem}.report.html")
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def _cmd_clean_workbook(args: argparse.Namespace) -> int:
|
|
243
|
+
from .workbook import clean_workbook
|
|
244
|
+
|
|
245
|
+
_reject_unsupported(
|
|
246
|
+
"workbook mode (every sheet is cleaned into one recipe)",
|
|
247
|
+
args,
|
|
248
|
+
{
|
|
249
|
+
"code": "--code", "report": "--report", "quarantine": "--quarantine",
|
|
250
|
+
"nrows": "--nrows", "skiprows": "--skiprows", "sep": "--sep",
|
|
251
|
+
"encoding": "--encoding",
|
|
252
|
+
},
|
|
253
|
+
)
|
|
254
|
+
_reserve_outputs(args, "recipe")
|
|
255
|
+
selection = _selection_kwargs(args)
|
|
256
|
+
result = clean_workbook(
|
|
257
|
+
args.file,
|
|
258
|
+
target_schema=args.schema,
|
|
259
|
+
llm=args.llm,
|
|
260
|
+
mode=args.mode,
|
|
261
|
+
max_tokens_budget=args.max_tokens,
|
|
262
|
+
llm_exposure=args.llm_exposure,
|
|
263
|
+
llm_fallback=not args.no_llm_fallback,
|
|
264
|
+
text=args.text,
|
|
265
|
+
**{k: v for k, v in selection.items() if k == "columns"},
|
|
266
|
+
)
|
|
267
|
+
recipe_out = Path(args.recipe) if args.recipe else _default_out(args.file, ".recipe.yaml")
|
|
268
|
+
result.save_recipe(recipe_out)
|
|
269
|
+
_emit(f"✓ Workbook recipe → {recipe_out} ({len(result.sheets)} sheet(s) cleaned)")
|
|
270
|
+
if args.out:
|
|
271
|
+
result.save_data(args.out, overwrite=bool(args.overwrite))
|
|
272
|
+
_emit(f"✓ Cleaned workbook → {args.out}")
|
|
273
|
+
_emit("")
|
|
274
|
+
_emit(result.summary())
|
|
275
|
+
if not args.out:
|
|
276
|
+
_emit("\n (pass --out cleaned.xlsx to write every cleaned sheet back)")
|
|
277
|
+
return EXIT_OK
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def _cmd_clean(args: argparse.Namespace) -> int:
|
|
281
|
+
from . import clean as _clean
|
|
282
|
+
from .dataio import write_frame
|
|
283
|
+
|
|
284
|
+
selection = _selection_kwargs(args)
|
|
285
|
+
workbook = _is_multisheet_workbook(args.file, selection)
|
|
286
|
+
if args.out_dir:
|
|
287
|
+
_out_dir_targets(args, workbook=workbook)
|
|
288
|
+
if workbook:
|
|
289
|
+
return _cmd_clean_workbook(args)
|
|
290
|
+
|
|
291
|
+
_reserve_outputs(args, "recipe", "out", "code", "report", "quarantine")
|
|
292
|
+
result = _clean(
|
|
293
|
+
args.file,
|
|
294
|
+
target_schema=args.schema,
|
|
295
|
+
llm=args.llm,
|
|
296
|
+
mode=args.mode,
|
|
297
|
+
max_tokens_budget=args.max_tokens,
|
|
298
|
+
llm_exposure=args.llm_exposure,
|
|
299
|
+
llm_fallback=not args.no_llm_fallback,
|
|
300
|
+
**selection,
|
|
301
|
+
**_read_kwargs(args),
|
|
302
|
+
)
|
|
303
|
+
recipe_out = Path(args.recipe) if args.recipe else _default_out(args.file, ".recipe.yaml")
|
|
304
|
+
result.recipe.save(recipe_out)
|
|
305
|
+
_emit(f"✓ Recipe → {recipe_out}")
|
|
306
|
+
|
|
307
|
+
if args.out:
|
|
308
|
+
write_frame(result.dataframe, args.out, source=args.file, overwrite=args.overwrite)
|
|
309
|
+
_emit(f"✓ Cleaned → {args.out} ({len(result.dataframe)} rows)")
|
|
310
|
+
if args.code:
|
|
311
|
+
result.code.save(args.code)
|
|
312
|
+
_emit(f"✓ Code → {args.code}")
|
|
313
|
+
if args.report:
|
|
314
|
+
result.report(args.report)
|
|
315
|
+
_emit(f"✓ Report → {args.report}")
|
|
316
|
+
if result.has_quarantine and args.quarantine:
|
|
317
|
+
write_frame(result.quarantine, args.quarantine, source=args.file, overwrite=args.overwrite)
|
|
318
|
+
_emit(f"✓ Quarantine → {args.quarantine} ({len(result.quarantine)} rows)")
|
|
319
|
+
|
|
320
|
+
_emit("")
|
|
321
|
+
result.diff.show()
|
|
322
|
+
if result.has_quarantine and not args.quarantine:
|
|
323
|
+
_emit(
|
|
324
|
+
f"\n⚠ {len(result.quarantine)} row(s) quarantined "
|
|
325
|
+
"(pass --quarantine FILE to save them)."
|
|
326
|
+
)
|
|
327
|
+
_print_log(args, result.log)
|
|
328
|
+
return EXIT_OK
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def _drift_stop(args: argparse.Namespace, exc: DriftError, action: str) -> int:
|
|
332
|
+
_emit(exc.report.render() if exc.report is not None else str(exc))
|
|
333
|
+
_emit("")
|
|
334
|
+
if args.mode == "strict":
|
|
335
|
+
_emit(f"Stopped — strict mode never {action} a drifted file. Re-plan the recipe, or")
|
|
336
|
+
_emit(f" cleanframe suggest {args.file} --recipe {args.recipe} --update")
|
|
337
|
+
else:
|
|
338
|
+
_emit(f"Stopped — re-run with --force to {action} anyway, or")
|
|
339
|
+
_emit(f" cleanframe suggest {args.file} --recipe {args.recipe} --update")
|
|
340
|
+
return EXIT_DRIFT
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def _cmd_apply_workbook(args: argparse.Namespace, recipe) -> int:
|
|
344
|
+
from .workbook import apply_workbook
|
|
345
|
+
|
|
346
|
+
_reject_unsupported(
|
|
347
|
+
"workbook mode (the recipe's per-sheet read: sections govern selection)",
|
|
348
|
+
args,
|
|
349
|
+
{
|
|
350
|
+
"report": "--report", "quarantine": "--quarantine", "sheet": "--sheet",
|
|
351
|
+
"columns": "--columns", "nrows": "--nrows", "skiprows": "--skiprows",
|
|
352
|
+
"sep": "--sep", "encoding": "--encoding", "text": "--text",
|
|
353
|
+
},
|
|
354
|
+
)
|
|
355
|
+
on_drift = "ignore" if args.force else "error"
|
|
356
|
+
try:
|
|
357
|
+
result = apply_workbook(
|
|
358
|
+
args.file, recipe, mode=args.mode,
|
|
359
|
+
check_drift=not args.no_drift_check, on_drift=on_drift,
|
|
360
|
+
)
|
|
361
|
+
except DriftError as exc:
|
|
362
|
+
return _drift_stop(args, exc, "apply")
|
|
363
|
+
out = Path(args.out) if args.out else _default_out(args.file, ".clean.xlsx")
|
|
364
|
+
result.save_data(out, overwrite=bool(args.overwrite))
|
|
365
|
+
_emit(f"✓ Cleaned workbook → {out}")
|
|
366
|
+
_emit("")
|
|
367
|
+
_emit(result.summary())
|
|
368
|
+
return EXIT_OK
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def _cmd_apply_stream(args: argparse.Namespace, recipe) -> int:
|
|
372
|
+
from .streaming import stream_apply
|
|
373
|
+
|
|
374
|
+
_reject_unsupported(
|
|
375
|
+
"streaming mode (--chunksize); the recipe's read: section governs selection",
|
|
376
|
+
args,
|
|
377
|
+
{"report": "--report", "sheet": "--sheet", "columns": "--columns", "nrows": "--nrows",
|
|
378
|
+
"skiprows": "--skiprows", "sep": "--sep", "encoding": "--encoding", "text": "--text"},
|
|
379
|
+
)
|
|
380
|
+
_reserve_outputs(args, "out", "quarantine")
|
|
381
|
+
out = Path(args.out) if args.out else _default_out(args.file, ".clean.csv")
|
|
382
|
+
try:
|
|
383
|
+
summary = stream_apply(
|
|
384
|
+
recipe, args.file, out, chunksize=args.chunksize, mode=args.mode,
|
|
385
|
+
quarantine_path=args.quarantine, overwrite=args.overwrite,
|
|
386
|
+
check_drift=not args.no_drift_check,
|
|
387
|
+
on_drift="ignore" if args.force else "error",
|
|
388
|
+
)
|
|
389
|
+
except DriftError as exc:
|
|
390
|
+
return _drift_stop(args, exc, "stream")
|
|
391
|
+
_emit(f"✓ Streamed → {out}")
|
|
392
|
+
_emit(summary.render())
|
|
393
|
+
return EXIT_OK
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def _cmd_apply(args: argparse.Namespace) -> int:
|
|
397
|
+
from . import apply_recipe
|
|
398
|
+
from .dataio import write_frame
|
|
399
|
+
from .workbook import WorkbookRecipe, load_recipe
|
|
400
|
+
|
|
401
|
+
loaded = load_recipe(args.recipe)
|
|
402
|
+
if isinstance(loaded, WorkbookRecipe):
|
|
403
|
+
return _cmd_apply_workbook(args, loaded)
|
|
404
|
+
if args.chunksize:
|
|
405
|
+
return _cmd_apply_stream(args, loaded)
|
|
406
|
+
|
|
407
|
+
_reserve_outputs(args, "out", "report", "quarantine")
|
|
408
|
+
try:
|
|
409
|
+
result = apply_recipe(
|
|
410
|
+
args.file,
|
|
411
|
+
loaded,
|
|
412
|
+
mode=args.mode,
|
|
413
|
+
check_drift=not args.no_drift_check,
|
|
414
|
+
on_drift="ignore" if args.force else "error",
|
|
415
|
+
**_selection_kwargs(args),
|
|
416
|
+
**{k: v for k, v in _read_kwargs(args).items() if k != "correct_format"},
|
|
417
|
+
)
|
|
418
|
+
except DriftError as exc:
|
|
419
|
+
return _drift_stop(args, exc, "apply")
|
|
420
|
+
|
|
421
|
+
if result.drift is not None and result.drift.has_drift:
|
|
422
|
+
_emit(result.drift.render())
|
|
423
|
+
_emit("")
|
|
424
|
+
out = Path(args.out) if args.out else _default_out(args.file, ".clean.csv")
|
|
425
|
+
write_frame(result.dataframe, out, source=args.file, overwrite=args.overwrite)
|
|
426
|
+
_emit(f"✓ Cleaned → {out} ({len(result.dataframe)} rows)")
|
|
427
|
+
if args.report:
|
|
428
|
+
result.report(args.report)
|
|
429
|
+
_emit(f"✓ Report → {args.report}")
|
|
430
|
+
if result.has_quarantine:
|
|
431
|
+
if args.quarantine:
|
|
432
|
+
write_frame(
|
|
433
|
+
result.quarantine, args.quarantine, source=args.file, overwrite=args.overwrite
|
|
434
|
+
)
|
|
435
|
+
_emit(f"✓ Quarantine → {args.quarantine} ({len(result.quarantine)} rows)")
|
|
436
|
+
else:
|
|
437
|
+
_emit(
|
|
438
|
+
f"⚠ {len(result.quarantine)} row(s) quarantined by validation "
|
|
439
|
+
"(pass --quarantine FILE to save them)."
|
|
440
|
+
)
|
|
441
|
+
_emit("")
|
|
442
|
+
result.diff.show()
|
|
443
|
+
_print_log(args, result.log)
|
|
444
|
+
return EXIT_OK
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def _cmd_suggest(args: argparse.Namespace) -> int:
|
|
448
|
+
from . import suggest_update
|
|
449
|
+
|
|
450
|
+
patched, drift = suggest_update(args.file, args.recipe, **_selection_kwargs(args))
|
|
451
|
+
_emit(drift.render())
|
|
452
|
+
if not drift.has_drift:
|
|
453
|
+
if args.update:
|
|
454
|
+
_emit("\nNothing to patch — the recipe already matches this file.")
|
|
455
|
+
return EXIT_OK
|
|
456
|
+
|
|
457
|
+
if not args.update:
|
|
458
|
+
_emit("\nRe-run with --update to write a patched recipe.")
|
|
459
|
+
return EXIT_DRIFT
|
|
460
|
+
|
|
461
|
+
src = Path(args.recipe)
|
|
462
|
+
if args.out:
|
|
463
|
+
write_to: Path = Path(args.out)
|
|
464
|
+
elif args.in_place:
|
|
465
|
+
write_to = src
|
|
466
|
+
else:
|
|
467
|
+
# Default: write a sibling patched file — never clobber the recipe silently.
|
|
468
|
+
write_to = src.with_name(src.stem + ".patched.yaml")
|
|
469
|
+
patched.save(write_to)
|
|
470
|
+
_emit(f"\n✓ Patched recipe written to {write_to}")
|
|
471
|
+
changes = patched.meta.get("patched_for_drift")
|
|
472
|
+
if isinstance(changes, list) and changes:
|
|
473
|
+
for change in changes:
|
|
474
|
+
_emit(f" • {change}")
|
|
475
|
+
return EXIT_OK
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
def _cmd_infer_schema(args: argparse.Namespace) -> int:
|
|
479
|
+
from . import infer_schema
|
|
480
|
+
from ._util import check_output_target
|
|
481
|
+
|
|
482
|
+
schema = infer_schema(
|
|
483
|
+
args.file, name=args.name, **_selection_kwargs(args), **_read_kwargs(args)
|
|
484
|
+
)
|
|
485
|
+
out = Path(args.out) if args.out else _default_out(args.file, ".schema.yaml")
|
|
486
|
+
schema.save(check_output_target(out, args.file))
|
|
487
|
+
_emit(f"✓ Schema ({len(schema.columns)} columns) → {out}")
|
|
488
|
+
return EXIT_OK
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
def _cmd_detectors(args: argparse.Namespace) -> int:
|
|
492
|
+
from .detectors import DETECTOR_REGISTRY, list_detectors
|
|
493
|
+
|
|
494
|
+
for name in list_detectors():
|
|
495
|
+
spec = DETECTOR_REGISTRY[name]
|
|
496
|
+
doc = (spec.doc or "").strip().splitlines()[0] if spec.doc else ""
|
|
497
|
+
_emit(f" {name:16s} [{spec.scope}] {doc}")
|
|
498
|
+
return EXIT_OK
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
def _cmd_ops(args: argparse.Namespace) -> int:
|
|
502
|
+
from .ops import OP_REGISTRY, list_ops
|
|
503
|
+
|
|
504
|
+
for name in list_ops():
|
|
505
|
+
spec = OP_REGISTRY[name]
|
|
506
|
+
doc = (spec.doc or "").strip().splitlines()[0] if spec.doc else ""
|
|
507
|
+
_emit(f" {name:20s} [{spec.scope}] {doc}")
|
|
508
|
+
return EXIT_OK
|
|
509
|
+
|
|
510
|
+
|
|
511
|
+
# ---------------------------------------------------------------------------
|
|
512
|
+
# parser
|
|
513
|
+
# ---------------------------------------------------------------------------
|
|
514
|
+
_MODE_HELP = (
|
|
515
|
+
"review (default: surface everything for approval), auto (unattended: only "
|
|
516
|
+
"higher-confidence fixes), strict (fail on drift or validation failures)"
|
|
517
|
+
)
|
|
518
|
+
|
|
519
|
+
|
|
520
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
521
|
+
common = argparse.ArgumentParser(add_help=False)
|
|
522
|
+
common.add_argument(
|
|
523
|
+
"--verbose", "-v", action="store_true", default=argparse.SUPPRESS,
|
|
524
|
+
help="print the run log (skipped columns, quarantine reasons, parse losses)",
|
|
525
|
+
)
|
|
526
|
+
common.add_argument(
|
|
527
|
+
"--debug", action="store_true", default=argparse.SUPPRESS,
|
|
528
|
+
help="print a traceback on an internal error",
|
|
529
|
+
)
|
|
530
|
+
|
|
531
|
+
parser = argparse.ArgumentParser(
|
|
532
|
+
prog="cleanframe",
|
|
533
|
+
description="The reproducible data-cleaning engine. Profile, clean, replay, detect drift.",
|
|
534
|
+
)
|
|
535
|
+
parser.add_argument("--version", action="version", version=f"cleanframe {__version__}")
|
|
536
|
+
parser.add_argument("--verbose", "-v", action="store_true", help=argparse.SUPPRESS)
|
|
537
|
+
parser.add_argument("--debug", action="store_true", help=argparse.SUPPRESS)
|
|
538
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
539
|
+
|
|
540
|
+
p = sub.add_parser("report", help="write an HTML profiling report", parents=[common])
|
|
541
|
+
p.add_argument("file")
|
|
542
|
+
p.add_argument("--out", "-o", help="output .html (default: <file>.report.html)")
|
|
543
|
+
p.add_argument("--schema", help="target schema YAML (adds mapping diagnostics)")
|
|
544
|
+
p.add_argument("--open", action="store_true", help="open the report in a browser")
|
|
545
|
+
_add_read_args(p)
|
|
546
|
+
_add_selection_args(p)
|
|
547
|
+
p.set_defaults(func=_cmd_report)
|
|
548
|
+
|
|
549
|
+
p = sub.add_parser("clean", help="plan and clean a file", parents=[common])
|
|
550
|
+
p.add_argument("file")
|
|
551
|
+
p.add_argument("--recipe", help="recipe output (default: <file>.recipe.yaml)")
|
|
552
|
+
p.add_argument("--out", "-o", help="cleaned data output (csv/tsv/xlsx/parquet/json)")
|
|
553
|
+
p.add_argument(
|
|
554
|
+
"--out-dir",
|
|
555
|
+
help="write recipe, cleaned data, code and report into this directory",
|
|
556
|
+
)
|
|
557
|
+
p.add_argument("--code", help="export standalone pandas to this .py")
|
|
558
|
+
p.add_argument("--report", help="write an HTML diff report here")
|
|
559
|
+
p.add_argument("--quarantine", help="write quarantined rows to this file")
|
|
560
|
+
p.add_argument("--schema", help="target schema YAML")
|
|
561
|
+
p.add_argument(
|
|
562
|
+
"--llm",
|
|
563
|
+
help="LLM planner as provider/model, e.g. openrouter/anthropic/claude-sonnet-4, "
|
|
564
|
+
"groq/llama-3.3-70b-versatile, anthropic/claude-sonnet-4-6",
|
|
565
|
+
)
|
|
566
|
+
p.add_argument("--max-tokens", type=_positive_int, default=None, help="LLM token budget cap")
|
|
567
|
+
p.add_argument(
|
|
568
|
+
"--llm-exposure",
|
|
569
|
+
default="metadata",
|
|
570
|
+
choices=["none", "metadata", "sample"],
|
|
571
|
+
help="what the LLM may see (default: metadata — never raw cells)",
|
|
572
|
+
)
|
|
573
|
+
p.add_argument(
|
|
574
|
+
"--no-llm-fallback",
|
|
575
|
+
action="store_true",
|
|
576
|
+
help="fail instead of degrading to the rules planner when an LLM call fails",
|
|
577
|
+
)
|
|
578
|
+
p.add_argument("--mode", default="review", choices=["review", "auto", "strict"], help=_MODE_HELP)
|
|
579
|
+
p.add_argument(
|
|
580
|
+
"--overwrite",
|
|
581
|
+
action="store_true",
|
|
582
|
+
help="allow writing output over the input file (loses the original)",
|
|
583
|
+
)
|
|
584
|
+
_add_read_args(p)
|
|
585
|
+
_add_selection_args(p)
|
|
586
|
+
p.set_defaults(func=_cmd_clean)
|
|
587
|
+
|
|
588
|
+
p = sub.add_parser("apply", help="replay a saved recipe (no LLM)", parents=[common])
|
|
589
|
+
p.add_argument("file")
|
|
590
|
+
p.add_argument("--recipe", required=True, help="recipe YAML to replay")
|
|
591
|
+
p.add_argument(
|
|
592
|
+
"--out", "-o",
|
|
593
|
+
help="cleaned data output (default: <file>.clean.csv, or .clean.xlsx for a workbook recipe)",
|
|
594
|
+
)
|
|
595
|
+
p.add_argument("--report", help="write an HTML diff report here")
|
|
596
|
+
p.add_argument("--mode", default="review", choices=["review", "auto", "strict"], help=_MODE_HELP)
|
|
597
|
+
p.add_argument("--no-drift-check", action="store_true", help="skip schema-drift check")
|
|
598
|
+
p.add_argument(
|
|
599
|
+
"--force",
|
|
600
|
+
action="store_true",
|
|
601
|
+
help="apply even when schema drift is detected (default: stop)",
|
|
602
|
+
)
|
|
603
|
+
p.add_argument(
|
|
604
|
+
"--overwrite",
|
|
605
|
+
action="store_true",
|
|
606
|
+
help="allow writing output over the input file (loses the original)",
|
|
607
|
+
)
|
|
608
|
+
p.add_argument(
|
|
609
|
+
"--chunksize",
|
|
610
|
+
type=_positive_int,
|
|
611
|
+
help="stream the CSV in chunks of N rows (out-of-core; row-independent recipes only)",
|
|
612
|
+
)
|
|
613
|
+
p.add_argument("--quarantine", help="write quarantined rows to this file")
|
|
614
|
+
p.add_argument("--sep", help="field delimiter, overriding the recipe's read: section")
|
|
615
|
+
p.add_argument("--encoding", help="file encoding, overriding the recipe's read: section")
|
|
616
|
+
p.add_argument("--text", action="store_true", help="read every field verbatim")
|
|
617
|
+
_add_selection_args(p)
|
|
618
|
+
p.set_defaults(func=_cmd_apply)
|
|
619
|
+
|
|
620
|
+
p = sub.add_parser(
|
|
621
|
+
"suggest", help="detect drift and optionally patch the recipe", parents=[common]
|
|
622
|
+
)
|
|
623
|
+
p.add_argument("file")
|
|
624
|
+
p.add_argument("--recipe", required=True, help="recipe YAML to check")
|
|
625
|
+
p.add_argument("--update", action="store_true", help="write a patched recipe")
|
|
626
|
+
p.add_argument(
|
|
627
|
+
"--out", "-o",
|
|
628
|
+
help="where to write the patched recipe (default: <recipe>.patched.yaml)",
|
|
629
|
+
)
|
|
630
|
+
p.add_argument(
|
|
631
|
+
"--in-place",
|
|
632
|
+
action="store_true",
|
|
633
|
+
help="with --update, overwrite the recipe file (default writes *.patched.yaml)",
|
|
634
|
+
)
|
|
635
|
+
_add_selection_args(p)
|
|
636
|
+
p.set_defaults(func=_cmd_suggest)
|
|
637
|
+
|
|
638
|
+
p = sub.add_parser("infer-schema", help="draft a target schema from a file", parents=[common])
|
|
639
|
+
p.add_argument("file")
|
|
640
|
+
p.add_argument("--out", "-o", help="schema output (default: <file>.schema.yaml)")
|
|
641
|
+
p.add_argument("--name", help="schema name")
|
|
642
|
+
_add_read_args(p)
|
|
643
|
+
_add_selection_args(p)
|
|
644
|
+
p.set_defaults(func=_cmd_infer_schema)
|
|
645
|
+
|
|
646
|
+
sub.add_parser("detectors", help="list available detectors", parents=[common]).set_defaults(
|
|
647
|
+
func=_cmd_detectors
|
|
648
|
+
)
|
|
649
|
+
sub.add_parser("ops", help="list available ops", parents=[common]).set_defaults(func=_cmd_ops)
|
|
650
|
+
|
|
651
|
+
return parser
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
def main(argv: list[str] | None = None) -> int:
|
|
655
|
+
_reconfigure_stdout()
|
|
656
|
+
_install_warning_format()
|
|
657
|
+
parser = build_parser()
|
|
658
|
+
args = parser.parse_args(argv)
|
|
659
|
+
try:
|
|
660
|
+
return args.func(args)
|
|
661
|
+
except ValidationFailure as exc:
|
|
662
|
+
_emit(f"✗ {exc}", stream=sys.stderr)
|
|
663
|
+
return EXIT_VALIDATION
|
|
664
|
+
except DriftError as exc:
|
|
665
|
+
_emit(f"✗ {exc}", stream=sys.stderr)
|
|
666
|
+
return EXIT_DRIFT
|
|
667
|
+
except CleanFrameError as exc:
|
|
668
|
+
_emit(f"✗ {exc}", stream=sys.stderr)
|
|
669
|
+
return EXIT_ERROR
|
|
670
|
+
except KeyboardInterrupt: # pragma: no cover
|
|
671
|
+
_emit("Interrupted.", stream=sys.stderr)
|
|
672
|
+
return EXIT_INTERRUPT
|
|
673
|
+
except BrokenPipeError: # pragma: no cover - piped into head/less
|
|
674
|
+
return EXIT_OK
|
|
675
|
+
except Exception as exc: # noqa: BLE001 - the CLI must not hand users a traceback
|
|
676
|
+
if _debug_enabled(args):
|
|
677
|
+
raise
|
|
678
|
+
_emit(
|
|
679
|
+
f"✗ Internal error: {type(exc).__name__}: {exc}\n"
|
|
680
|
+
f" This is a bug. Re-run with --debug for a traceback, then report it at "
|
|
681
|
+
f"{_ISSUE_URL}",
|
|
682
|
+
stream=sys.stderr,
|
|
683
|
+
)
|
|
684
|
+
return EXIT_INTERNAL
|
|
685
|
+
|
|
686
|
+
|
|
687
|
+
if __name__ == "__main__": # pragma: no cover
|
|
688
|
+
raise SystemExit(main())
|