cleanframe-engine 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. cleanframe/__init__.py +169 -0
  2. cleanframe/__main__.py +5 -0
  3. cleanframe/_util.py +438 -0
  4. cleanframe/_version.py +1 -0
  5. cleanframe/api.py +559 -0
  6. cleanframe/cli.py +688 -0
  7. cleanframe/codegen.py +666 -0
  8. cleanframe/dataio.py +506 -0
  9. cleanframe/detectors/__init__.py +43 -0
  10. cleanframe/detectors/base.py +221 -0
  11. cleanframe/detectors/categories.py +199 -0
  12. cleanframe/detectors/contacts.py +105 -0
  13. cleanframe/detectors/currency.py +112 -0
  14. cleanframe/detectors/dates.py +204 -0
  15. cleanframe/detectors/dedup.py +150 -0
  16. cleanframe/detectors/nulls.py +109 -0
  17. cleanframe/detectors/outliers.py +73 -0
  18. cleanframe/detectors/schema_mapping.py +125 -0
  19. cleanframe/detectors/text.py +105 -0
  20. cleanframe/detectors/units.py +86 -0
  21. cleanframe/diff.py +369 -0
  22. cleanframe/drift.py +283 -0
  23. cleanframe/errors.py +66 -0
  24. cleanframe/executor.py +229 -0
  25. cleanframe/fingerprint.py +83 -0
  26. cleanframe/issues.py +186 -0
  27. cleanframe/llm.py +811 -0
  28. cleanframe/ops.py +1245 -0
  29. cleanframe/planner.py +353 -0
  30. cleanframe/profile.py +413 -0
  31. cleanframe/py.typed +1 -0
  32. cleanframe/quality.py +81 -0
  33. cleanframe/readfix.py +160 -0
  34. cleanframe/recipe.py +398 -0
  35. cleanframe/report.py +345 -0
  36. cleanframe/result.py +144 -0
  37. cleanframe/schema.py +259 -0
  38. cleanframe/streaming.py +354 -0
  39. cleanframe/types.py +119 -0
  40. cleanframe/validate.py +363 -0
  41. cleanframe/workbook.py +370 -0
  42. cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
  43. cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
  44. cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
  45. cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
  46. cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
cleanframe/cli.py ADDED
@@ -0,0 +1,688 @@
1
+ """The ``cleanframe`` command-line interface.
2
+
3
+ Subcommands mirror the library:
4
+
5
+ * ``report FILE`` — write an HTML profiling report.
6
+ * ``clean FILE`` — plan + clean; save recipe, code, cleaned data, report.
7
+ * ``apply FILE --recipe R`` — replay a recipe (with drift check).
8
+ * ``suggest FILE --recipe R`` — show drift and optionally patch the recipe.
9
+ * ``infer-schema FILE`` — draft a target schema.
10
+ * ``detectors`` / ``ops`` — list what's available.
11
+
12
+ Exit codes: ``0`` success, ``1`` data/recipe/output error, ``2`` usage error,
13
+ ``3`` stopped on schema drift, ``4`` validation failed, ``70`` internal error,
14
+ ``130`` interrupted. Pass ``--debug`` (or set ``CLEANFRAME_DEBUG=1``) to print a
15
+ traceback for an internal error instead of a one-line message.
16
+
17
+ Stdout is switched to UTF-8 so currency symbols and diff glyphs render on any
18
+ terminal (notably Windows consoles).
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import argparse
24
+ import os
25
+ import sys
26
+ import warnings
27
+ from pathlib import Path
28
+
29
+ from ._version import __version__
30
+ from .errors import CleanFrameError, CleanFrameWarning, DriftError, ValidationFailure
31
+
32
+ EXIT_OK = 0
33
+ EXIT_ERROR = 1
34
+ EXIT_USAGE = 2
35
+ EXIT_DRIFT = 3
36
+ EXIT_VALIDATION = 4
37
+ EXIT_INTERNAL = 70
38
+ EXIT_INTERRUPT = 130
39
+
40
+ _ISSUE_URL = "https://github.com/inboxpraveen/Cleanframe/issues"
41
+
42
+
43
+ def _reconfigure_stdout() -> None:
44
+ for stream in (sys.stdout, sys.stderr):
45
+ try:
46
+ stream.reconfigure(encoding="utf-8") # type: ignore[union-attr]
47
+ except (AttributeError, ValueError): # pragma: no cover - older/odd streams
48
+ pass
49
+
50
+
51
+ def _emit(text: str, *, stream=None) -> None:
52
+ """Print a line, degrading gracefully on a console that cannot encode it."""
53
+ target = stream or sys.stdout
54
+ try:
55
+ print(text, file=target)
56
+ except UnicodeEncodeError: # pragma: no cover - depends on console codepage
57
+ encoding = getattr(target, "encoding", None) or "ascii"
58
+ print(text.encode(encoding, "replace").decode(encoding, "replace"), file=target)
59
+
60
+
61
+ def _install_warning_format() -> None:
62
+ """Show advisories as one readable line instead of a file path and source echo."""
63
+
64
+ def show(message, category, filename, lineno, file=None, line=None): # noqa: ANN001
65
+ text = str(message)
66
+ if text.startswith("CleanFrame:"):
67
+ text = text[len("CleanFrame:") :].strip()
68
+ elif not issubclass(category, CleanFrameWarning):
69
+ text = f"{category.__name__}: {text}"
70
+ _emit(f"⚠ {text}", stream=file or sys.stderr)
71
+
72
+ warnings.showwarning = show
73
+
74
+
75
+ def _debug_enabled(args: argparse.Namespace | None = None) -> bool:
76
+ if args is not None and getattr(args, "debug", False):
77
+ return True
78
+ return os.environ.get("CLEANFRAME_DEBUG", "").strip() not in ("", "0", "false", "False")
79
+
80
+
81
+ # ---------------------------------------------------------------------------
82
+ # argument helpers
83
+ # ---------------------------------------------------------------------------
84
+ def _nonneg_int(text: str) -> int:
85
+ try:
86
+ value = int(text)
87
+ except ValueError:
88
+ raise argparse.ArgumentTypeError(f"expected a whole number, got {text!r}") from None
89
+ if value < 0:
90
+ raise argparse.ArgumentTypeError(f"must be 0 or more, got {value}")
91
+ return value
92
+
93
+
94
+ def _positive_int(text: str) -> int:
95
+ try:
96
+ value = int(text)
97
+ except ValueError:
98
+ raise argparse.ArgumentTypeError(f"expected a whole number, got {text!r}") from None
99
+ if value < 1:
100
+ raise argparse.ArgumentTypeError(f"must be 1 or more, got {value}")
101
+ return value
102
+
103
+
104
+ def _default_out(file: str, suffix: str) -> Path:
105
+ return Path(file).with_suffix(suffix)
106
+
107
+
108
+ def _add_selection_args(parser: argparse.ArgumentParser, *, sheet: bool = True) -> None:
109
+ if sheet:
110
+ parser.add_argument(
111
+ "--sheet",
112
+ help="Excel sheet name, or #N for a 0-based index (e.g. --sheet '#0')",
113
+ )
114
+ parser.add_argument("--columns", help="comma-separated column subset to read")
115
+ parser.add_argument("--nrows", type=_nonneg_int, help="read only the first N data rows")
116
+ parser.add_argument(
117
+ "--skiprows",
118
+ type=_nonneg_int,
119
+ help="skip the first N data rows (the header row is always kept)",
120
+ )
121
+
122
+
123
+ def _add_read_args(parser: argparse.ArgumentParser) -> None:
124
+ parser.add_argument("--sep", help="field delimiter, overriding auto-detection")
125
+ parser.add_argument("--encoding", help="file encoding, overriding auto-detection")
126
+ parser.add_argument(
127
+ "--text",
128
+ action="store_true",
129
+ help="read every field verbatim (keeps leading zeros, 'NA' text and 1e5 exact)",
130
+ )
131
+ parser.add_argument(
132
+ "--no-correct", action="store_true", help="disable read-time format auto-detection"
133
+ )
134
+
135
+
136
+ def _selection_kwargs(args: argparse.Namespace) -> dict:
137
+ out: dict = {}
138
+ sheet = getattr(args, "sheet", None)
139
+ if sheet is not None:
140
+ # Digits alone are sheet *names* (e.g. "2024"). Use "#0" / "#1" for indices.
141
+ if sheet.startswith("#") and sheet[1:].lstrip("-").isdigit():
142
+ out["sheet"] = int(sheet[1:])
143
+ else:
144
+ out["sheet"] = sheet
145
+ cols = getattr(args, "columns", None)
146
+ if cols:
147
+ out["columns"] = [c.strip() for c in cols.split(",") if c.strip()]
148
+ for name in ("nrows", "skiprows"):
149
+ if getattr(args, name, None) is not None:
150
+ out[name] = getattr(args, name)
151
+ return out
152
+
153
+
154
+ def _read_kwargs(args: argparse.Namespace) -> dict:
155
+ out: dict = {}
156
+ if getattr(args, "sep", None):
157
+ out["sep"] = args.sep
158
+ if getattr(args, "encoding", None):
159
+ out["encoding"] = args.encoding
160
+ if getattr(args, "text", False):
161
+ out["text"] = True
162
+ if hasattr(args, "no_correct"):
163
+ out["correct_format"] = not args.no_correct
164
+ return out
165
+
166
+
167
+ def _print_log(args: argparse.Namespace, log: list[str]) -> None:
168
+ if getattr(args, "verbose", False) and log:
169
+ _emit("")
170
+ for line in log:
171
+ _emit(f" · {line}")
172
+
173
+
174
+ def _reserve_outputs(args: argparse.Namespace, *attrs: str) -> None:
175
+ """Validate output paths up front so a bad one fails before any work is done."""
176
+ from ._util import check_output_target
177
+
178
+ for attr in attrs:
179
+ target = getattr(args, attr, None)
180
+ if target:
181
+ check_output_target(target, args.file, overwrite=getattr(args, "overwrite", False))
182
+
183
+
184
+ def _reject_unsupported(mode: str, args: argparse.Namespace, names: dict[str, str]) -> None:
185
+ """Fail loudly for flags that would otherwise be silently ignored."""
186
+ given = [flag for attr, flag in names.items() if getattr(args, attr, None)]
187
+ if given:
188
+ raise CleanFrameError(
189
+ f"{', '.join(given)} {'is' if len(given) == 1 else 'are'} not supported in "
190
+ f"{mode}. Re-run without {'it' if len(given) == 1 else 'them'}."
191
+ )
192
+
193
+
194
+ # ---------------------------------------------------------------------------
195
+ # command handlers
196
+ # ---------------------------------------------------------------------------
197
+ def _cmd_report(args: argparse.Namespace) -> int:
198
+ from . import report as _report
199
+ from ._util import check_output_target
200
+
201
+ rep = _report(args.file, schema=args.schema, **_selection_kwargs(args), **_read_kwargs(args))
202
+ out = Path(args.out) if args.out else _default_out(args.file, ".report.html")
203
+ rep.save(check_output_target(out, args.file))
204
+ _emit(f"✓ Report written to {out}")
205
+ q = rep.quality
206
+ if q:
207
+ _emit(f" Quality score: {q.score}/100 (grade {q.grade} — {q.label})")
208
+ if args.open:
209
+ import webbrowser
210
+
211
+ webbrowser.open(out.resolve().as_uri())
212
+ return EXIT_OK
213
+
214
+
215
+ def _is_multisheet_workbook(file: str, selection: dict) -> bool:
216
+ """A multi-sheet .xlsx with no explicit --sheet -> clean every tab (workbook mode)."""
217
+ path = Path(file)
218
+ if path.suffix.lower() not in (".xlsx", ".xls", ".xlsm") or "sheet" in selection:
219
+ return False
220
+ try:
221
+ from .dataio import excel_sheet_names
222
+
223
+ return len(excel_sheet_names(path)) > 1
224
+ except CleanFrameError:
225
+ return False
226
+
227
+
228
+ def _out_dir_targets(args: argparse.Namespace, *, workbook: bool) -> None:
229
+ """Expand --out-dir into the individual artifact paths."""
230
+ directory = Path(args.out_dir)
231
+ stem = Path(args.file).stem
232
+ args.recipe = args.recipe or str(directory / f"{stem}.recipe.yaml")
233
+ if workbook:
234
+ # A workbook produces one recipe and one rewritten workbook, nothing else.
235
+ args.out = args.out or str(directory / f"{stem}.clean.xlsx")
236
+ return
237
+ args.out = args.out or str(directory / f"{stem}.clean.csv")
238
+ args.code = args.code or str(directory / f"{stem}.py")
239
+ args.report = args.report or str(directory / f"{stem}.report.html")
240
+
241
+
242
+ def _cmd_clean_workbook(args: argparse.Namespace) -> int:
243
+ from .workbook import clean_workbook
244
+
245
+ _reject_unsupported(
246
+ "workbook mode (every sheet is cleaned into one recipe)",
247
+ args,
248
+ {
249
+ "code": "--code", "report": "--report", "quarantine": "--quarantine",
250
+ "nrows": "--nrows", "skiprows": "--skiprows", "sep": "--sep",
251
+ "encoding": "--encoding",
252
+ },
253
+ )
254
+ _reserve_outputs(args, "recipe")
255
+ selection = _selection_kwargs(args)
256
+ result = clean_workbook(
257
+ args.file,
258
+ target_schema=args.schema,
259
+ llm=args.llm,
260
+ mode=args.mode,
261
+ max_tokens_budget=args.max_tokens,
262
+ llm_exposure=args.llm_exposure,
263
+ llm_fallback=not args.no_llm_fallback,
264
+ text=args.text,
265
+ **{k: v for k, v in selection.items() if k == "columns"},
266
+ )
267
+ recipe_out = Path(args.recipe) if args.recipe else _default_out(args.file, ".recipe.yaml")
268
+ result.save_recipe(recipe_out)
269
+ _emit(f"✓ Workbook recipe → {recipe_out} ({len(result.sheets)} sheet(s) cleaned)")
270
+ if args.out:
271
+ result.save_data(args.out, overwrite=bool(args.overwrite))
272
+ _emit(f"✓ Cleaned workbook → {args.out}")
273
+ _emit("")
274
+ _emit(result.summary())
275
+ if not args.out:
276
+ _emit("\n (pass --out cleaned.xlsx to write every cleaned sheet back)")
277
+ return EXIT_OK
278
+
279
+
280
+ def _cmd_clean(args: argparse.Namespace) -> int:
281
+ from . import clean as _clean
282
+ from .dataio import write_frame
283
+
284
+ selection = _selection_kwargs(args)
285
+ workbook = _is_multisheet_workbook(args.file, selection)
286
+ if args.out_dir:
287
+ _out_dir_targets(args, workbook=workbook)
288
+ if workbook:
289
+ return _cmd_clean_workbook(args)
290
+
291
+ _reserve_outputs(args, "recipe", "out", "code", "report", "quarantine")
292
+ result = _clean(
293
+ args.file,
294
+ target_schema=args.schema,
295
+ llm=args.llm,
296
+ mode=args.mode,
297
+ max_tokens_budget=args.max_tokens,
298
+ llm_exposure=args.llm_exposure,
299
+ llm_fallback=not args.no_llm_fallback,
300
+ **selection,
301
+ **_read_kwargs(args),
302
+ )
303
+ recipe_out = Path(args.recipe) if args.recipe else _default_out(args.file, ".recipe.yaml")
304
+ result.recipe.save(recipe_out)
305
+ _emit(f"✓ Recipe → {recipe_out}")
306
+
307
+ if args.out:
308
+ write_frame(result.dataframe, args.out, source=args.file, overwrite=args.overwrite)
309
+ _emit(f"✓ Cleaned → {args.out} ({len(result.dataframe)} rows)")
310
+ if args.code:
311
+ result.code.save(args.code)
312
+ _emit(f"✓ Code → {args.code}")
313
+ if args.report:
314
+ result.report(args.report)
315
+ _emit(f"✓ Report → {args.report}")
316
+ if result.has_quarantine and args.quarantine:
317
+ write_frame(result.quarantine, args.quarantine, source=args.file, overwrite=args.overwrite)
318
+ _emit(f"✓ Quarantine → {args.quarantine} ({len(result.quarantine)} rows)")
319
+
320
+ _emit("")
321
+ result.diff.show()
322
+ if result.has_quarantine and not args.quarantine:
323
+ _emit(
324
+ f"\n⚠ {len(result.quarantine)} row(s) quarantined "
325
+ "(pass --quarantine FILE to save them)."
326
+ )
327
+ _print_log(args, result.log)
328
+ return EXIT_OK
329
+
330
+
331
+ def _drift_stop(args: argparse.Namespace, exc: DriftError, action: str) -> int:
332
+ _emit(exc.report.render() if exc.report is not None else str(exc))
333
+ _emit("")
334
+ if args.mode == "strict":
335
+ _emit(f"Stopped — strict mode never {action} a drifted file. Re-plan the recipe, or")
336
+ _emit(f" cleanframe suggest {args.file} --recipe {args.recipe} --update")
337
+ else:
338
+ _emit(f"Stopped — re-run with --force to {action} anyway, or")
339
+ _emit(f" cleanframe suggest {args.file} --recipe {args.recipe} --update")
340
+ return EXIT_DRIFT
341
+
342
+
343
+ def _cmd_apply_workbook(args: argparse.Namespace, recipe) -> int:
344
+ from .workbook import apply_workbook
345
+
346
+ _reject_unsupported(
347
+ "workbook mode (the recipe's per-sheet read: sections govern selection)",
348
+ args,
349
+ {
350
+ "report": "--report", "quarantine": "--quarantine", "sheet": "--sheet",
351
+ "columns": "--columns", "nrows": "--nrows", "skiprows": "--skiprows",
352
+ "sep": "--sep", "encoding": "--encoding", "text": "--text",
353
+ },
354
+ )
355
+ on_drift = "ignore" if args.force else "error"
356
+ try:
357
+ result = apply_workbook(
358
+ args.file, recipe, mode=args.mode,
359
+ check_drift=not args.no_drift_check, on_drift=on_drift,
360
+ )
361
+ except DriftError as exc:
362
+ return _drift_stop(args, exc, "apply")
363
+ out = Path(args.out) if args.out else _default_out(args.file, ".clean.xlsx")
364
+ result.save_data(out, overwrite=bool(args.overwrite))
365
+ _emit(f"✓ Cleaned workbook → {out}")
366
+ _emit("")
367
+ _emit(result.summary())
368
+ return EXIT_OK
369
+
370
+
371
+ def _cmd_apply_stream(args: argparse.Namespace, recipe) -> int:
372
+ from .streaming import stream_apply
373
+
374
+ _reject_unsupported(
375
+ "streaming mode (--chunksize); the recipe's read: section governs selection",
376
+ args,
377
+ {"report": "--report", "sheet": "--sheet", "columns": "--columns", "nrows": "--nrows",
378
+ "skiprows": "--skiprows", "sep": "--sep", "encoding": "--encoding", "text": "--text"},
379
+ )
380
+ _reserve_outputs(args, "out", "quarantine")
381
+ out = Path(args.out) if args.out else _default_out(args.file, ".clean.csv")
382
+ try:
383
+ summary = stream_apply(
384
+ recipe, args.file, out, chunksize=args.chunksize, mode=args.mode,
385
+ quarantine_path=args.quarantine, overwrite=args.overwrite,
386
+ check_drift=not args.no_drift_check,
387
+ on_drift="ignore" if args.force else "error",
388
+ )
389
+ except DriftError as exc:
390
+ return _drift_stop(args, exc, "stream")
391
+ _emit(f"✓ Streamed → {out}")
392
+ _emit(summary.render())
393
+ return EXIT_OK
394
+
395
+
396
+ def _cmd_apply(args: argparse.Namespace) -> int:
397
+ from . import apply_recipe
398
+ from .dataio import write_frame
399
+ from .workbook import WorkbookRecipe, load_recipe
400
+
401
+ loaded = load_recipe(args.recipe)
402
+ if isinstance(loaded, WorkbookRecipe):
403
+ return _cmd_apply_workbook(args, loaded)
404
+ if args.chunksize:
405
+ return _cmd_apply_stream(args, loaded)
406
+
407
+ _reserve_outputs(args, "out", "report", "quarantine")
408
+ try:
409
+ result = apply_recipe(
410
+ args.file,
411
+ loaded,
412
+ mode=args.mode,
413
+ check_drift=not args.no_drift_check,
414
+ on_drift="ignore" if args.force else "error",
415
+ **_selection_kwargs(args),
416
+ **{k: v for k, v in _read_kwargs(args).items() if k != "correct_format"},
417
+ )
418
+ except DriftError as exc:
419
+ return _drift_stop(args, exc, "apply")
420
+
421
+ if result.drift is not None and result.drift.has_drift:
422
+ _emit(result.drift.render())
423
+ _emit("")
424
+ out = Path(args.out) if args.out else _default_out(args.file, ".clean.csv")
425
+ write_frame(result.dataframe, out, source=args.file, overwrite=args.overwrite)
426
+ _emit(f"✓ Cleaned → {out} ({len(result.dataframe)} rows)")
427
+ if args.report:
428
+ result.report(args.report)
429
+ _emit(f"✓ Report → {args.report}")
430
+ if result.has_quarantine:
431
+ if args.quarantine:
432
+ write_frame(
433
+ result.quarantine, args.quarantine, source=args.file, overwrite=args.overwrite
434
+ )
435
+ _emit(f"✓ Quarantine → {args.quarantine} ({len(result.quarantine)} rows)")
436
+ else:
437
+ _emit(
438
+ f"⚠ {len(result.quarantine)} row(s) quarantined by validation "
439
+ "(pass --quarantine FILE to save them)."
440
+ )
441
+ _emit("")
442
+ result.diff.show()
443
+ _print_log(args, result.log)
444
+ return EXIT_OK
445
+
446
+
447
+ def _cmd_suggest(args: argparse.Namespace) -> int:
448
+ from . import suggest_update
449
+
450
+ patched, drift = suggest_update(args.file, args.recipe, **_selection_kwargs(args))
451
+ _emit(drift.render())
452
+ if not drift.has_drift:
453
+ if args.update:
454
+ _emit("\nNothing to patch — the recipe already matches this file.")
455
+ return EXIT_OK
456
+
457
+ if not args.update:
458
+ _emit("\nRe-run with --update to write a patched recipe.")
459
+ return EXIT_DRIFT
460
+
461
+ src = Path(args.recipe)
462
+ if args.out:
463
+ write_to: Path = Path(args.out)
464
+ elif args.in_place:
465
+ write_to = src
466
+ else:
467
+ # Default: write a sibling patched file — never clobber the recipe silently.
468
+ write_to = src.with_name(src.stem + ".patched.yaml")
469
+ patched.save(write_to)
470
+ _emit(f"\n✓ Patched recipe written to {write_to}")
471
+ changes = patched.meta.get("patched_for_drift")
472
+ if isinstance(changes, list) and changes:
473
+ for change in changes:
474
+ _emit(f" • {change}")
475
+ return EXIT_OK
476
+
477
+
478
+ def _cmd_infer_schema(args: argparse.Namespace) -> int:
479
+ from . import infer_schema
480
+ from ._util import check_output_target
481
+
482
+ schema = infer_schema(
483
+ args.file, name=args.name, **_selection_kwargs(args), **_read_kwargs(args)
484
+ )
485
+ out = Path(args.out) if args.out else _default_out(args.file, ".schema.yaml")
486
+ schema.save(check_output_target(out, args.file))
487
+ _emit(f"✓ Schema ({len(schema.columns)} columns) → {out}")
488
+ return EXIT_OK
489
+
490
+
491
+ def _cmd_detectors(args: argparse.Namespace) -> int:
492
+ from .detectors import DETECTOR_REGISTRY, list_detectors
493
+
494
+ for name in list_detectors():
495
+ spec = DETECTOR_REGISTRY[name]
496
+ doc = (spec.doc or "").strip().splitlines()[0] if spec.doc else ""
497
+ _emit(f" {name:16s} [{spec.scope}] {doc}")
498
+ return EXIT_OK
499
+
500
+
501
+ def _cmd_ops(args: argparse.Namespace) -> int:
502
+ from .ops import OP_REGISTRY, list_ops
503
+
504
+ for name in list_ops():
505
+ spec = OP_REGISTRY[name]
506
+ doc = (spec.doc or "").strip().splitlines()[0] if spec.doc else ""
507
+ _emit(f" {name:20s} [{spec.scope}] {doc}")
508
+ return EXIT_OK
509
+
510
+
511
+ # ---------------------------------------------------------------------------
512
+ # parser
513
+ # ---------------------------------------------------------------------------
514
+ _MODE_HELP = (
515
+ "review (default: surface everything for approval), auto (unattended: only "
516
+ "higher-confidence fixes), strict (fail on drift or validation failures)"
517
+ )
518
+
519
+
520
+ def build_parser() -> argparse.ArgumentParser:
521
+ common = argparse.ArgumentParser(add_help=False)
522
+ common.add_argument(
523
+ "--verbose", "-v", action="store_true", default=argparse.SUPPRESS,
524
+ help="print the run log (skipped columns, quarantine reasons, parse losses)",
525
+ )
526
+ common.add_argument(
527
+ "--debug", action="store_true", default=argparse.SUPPRESS,
528
+ help="print a traceback on an internal error",
529
+ )
530
+
531
+ parser = argparse.ArgumentParser(
532
+ prog="cleanframe",
533
+ description="The reproducible data-cleaning engine. Profile, clean, replay, detect drift.",
534
+ )
535
+ parser.add_argument("--version", action="version", version=f"cleanframe {__version__}")
536
+ parser.add_argument("--verbose", "-v", action="store_true", help=argparse.SUPPRESS)
537
+ parser.add_argument("--debug", action="store_true", help=argparse.SUPPRESS)
538
+ sub = parser.add_subparsers(dest="command", required=True)
539
+
540
+ p = sub.add_parser("report", help="write an HTML profiling report", parents=[common])
541
+ p.add_argument("file")
542
+ p.add_argument("--out", "-o", help="output .html (default: <file>.report.html)")
543
+ p.add_argument("--schema", help="target schema YAML (adds mapping diagnostics)")
544
+ p.add_argument("--open", action="store_true", help="open the report in a browser")
545
+ _add_read_args(p)
546
+ _add_selection_args(p)
547
+ p.set_defaults(func=_cmd_report)
548
+
549
+ p = sub.add_parser("clean", help="plan and clean a file", parents=[common])
550
+ p.add_argument("file")
551
+ p.add_argument("--recipe", help="recipe output (default: <file>.recipe.yaml)")
552
+ p.add_argument("--out", "-o", help="cleaned data output (csv/tsv/xlsx/parquet/json)")
553
+ p.add_argument(
554
+ "--out-dir",
555
+ help="write recipe, cleaned data, code and report into this directory",
556
+ )
557
+ p.add_argument("--code", help="export standalone pandas to this .py")
558
+ p.add_argument("--report", help="write an HTML diff report here")
559
+ p.add_argument("--quarantine", help="write quarantined rows to this file")
560
+ p.add_argument("--schema", help="target schema YAML")
561
+ p.add_argument(
562
+ "--llm",
563
+ help="LLM planner as provider/model, e.g. openrouter/anthropic/claude-sonnet-4, "
564
+ "groq/llama-3.3-70b-versatile, anthropic/claude-sonnet-4-6",
565
+ )
566
+ p.add_argument("--max-tokens", type=_positive_int, default=None, help="LLM token budget cap")
567
+ p.add_argument(
568
+ "--llm-exposure",
569
+ default="metadata",
570
+ choices=["none", "metadata", "sample"],
571
+ help="what the LLM may see (default: metadata — never raw cells)",
572
+ )
573
+ p.add_argument(
574
+ "--no-llm-fallback",
575
+ action="store_true",
576
+ help="fail instead of degrading to the rules planner when an LLM call fails",
577
+ )
578
+ p.add_argument("--mode", default="review", choices=["review", "auto", "strict"], help=_MODE_HELP)
579
+ p.add_argument(
580
+ "--overwrite",
581
+ action="store_true",
582
+ help="allow writing output over the input file (loses the original)",
583
+ )
584
+ _add_read_args(p)
585
+ _add_selection_args(p)
586
+ p.set_defaults(func=_cmd_clean)
587
+
588
+ p = sub.add_parser("apply", help="replay a saved recipe (no LLM)", parents=[common])
589
+ p.add_argument("file")
590
+ p.add_argument("--recipe", required=True, help="recipe YAML to replay")
591
+ p.add_argument(
592
+ "--out", "-o",
593
+ help="cleaned data output (default: <file>.clean.csv, or .clean.xlsx for a workbook recipe)",
594
+ )
595
+ p.add_argument("--report", help="write an HTML diff report here")
596
+ p.add_argument("--mode", default="review", choices=["review", "auto", "strict"], help=_MODE_HELP)
597
+ p.add_argument("--no-drift-check", action="store_true", help="skip schema-drift check")
598
+ p.add_argument(
599
+ "--force",
600
+ action="store_true",
601
+ help="apply even when schema drift is detected (default: stop)",
602
+ )
603
+ p.add_argument(
604
+ "--overwrite",
605
+ action="store_true",
606
+ help="allow writing output over the input file (loses the original)",
607
+ )
608
+ p.add_argument(
609
+ "--chunksize",
610
+ type=_positive_int,
611
+ help="stream the CSV in chunks of N rows (out-of-core; row-independent recipes only)",
612
+ )
613
+ p.add_argument("--quarantine", help="write quarantined rows to this file")
614
+ p.add_argument("--sep", help="field delimiter, overriding the recipe's read: section")
615
+ p.add_argument("--encoding", help="file encoding, overriding the recipe's read: section")
616
+ p.add_argument("--text", action="store_true", help="read every field verbatim")
617
+ _add_selection_args(p)
618
+ p.set_defaults(func=_cmd_apply)
619
+
620
+ p = sub.add_parser(
621
+ "suggest", help="detect drift and optionally patch the recipe", parents=[common]
622
+ )
623
+ p.add_argument("file")
624
+ p.add_argument("--recipe", required=True, help="recipe YAML to check")
625
+ p.add_argument("--update", action="store_true", help="write a patched recipe")
626
+ p.add_argument(
627
+ "--out", "-o",
628
+ help="where to write the patched recipe (default: <recipe>.patched.yaml)",
629
+ )
630
+ p.add_argument(
631
+ "--in-place",
632
+ action="store_true",
633
+ help="with --update, overwrite the recipe file (default writes *.patched.yaml)",
634
+ )
635
+ _add_selection_args(p)
636
+ p.set_defaults(func=_cmd_suggest)
637
+
638
+ p = sub.add_parser("infer-schema", help="draft a target schema from a file", parents=[common])
639
+ p.add_argument("file")
640
+ p.add_argument("--out", "-o", help="schema output (default: <file>.schema.yaml)")
641
+ p.add_argument("--name", help="schema name")
642
+ _add_read_args(p)
643
+ _add_selection_args(p)
644
+ p.set_defaults(func=_cmd_infer_schema)
645
+
646
+ sub.add_parser("detectors", help="list available detectors", parents=[common]).set_defaults(
647
+ func=_cmd_detectors
648
+ )
649
+ sub.add_parser("ops", help="list available ops", parents=[common]).set_defaults(func=_cmd_ops)
650
+
651
+ return parser
652
+
653
+
654
+ def main(argv: list[str] | None = None) -> int:
655
+ _reconfigure_stdout()
656
+ _install_warning_format()
657
+ parser = build_parser()
658
+ args = parser.parse_args(argv)
659
+ try:
660
+ return args.func(args)
661
+ except ValidationFailure as exc:
662
+ _emit(f"✗ {exc}", stream=sys.stderr)
663
+ return EXIT_VALIDATION
664
+ except DriftError as exc:
665
+ _emit(f"✗ {exc}", stream=sys.stderr)
666
+ return EXIT_DRIFT
667
+ except CleanFrameError as exc:
668
+ _emit(f"✗ {exc}", stream=sys.stderr)
669
+ return EXIT_ERROR
670
+ except KeyboardInterrupt: # pragma: no cover
671
+ _emit("Interrupted.", stream=sys.stderr)
672
+ return EXIT_INTERRUPT
673
+ except BrokenPipeError: # pragma: no cover - piped into head/less
674
+ return EXIT_OK
675
+ except Exception as exc: # noqa: BLE001 - the CLI must not hand users a traceback
676
+ if _debug_enabled(args):
677
+ raise
678
+ _emit(
679
+ f"✗ Internal error: {type(exc).__name__}: {exc}\n"
680
+ f" This is a bug. Re-run with --debug for a traceback, then report it at "
681
+ f"{_ISSUE_URL}",
682
+ stream=sys.stderr,
683
+ )
684
+ return EXIT_INTERNAL
685
+
686
+
687
+ if __name__ == "__main__": # pragma: no cover
688
+ raise SystemExit(main())