cleanframe-engine 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. cleanframe/__init__.py +169 -0
  2. cleanframe/__main__.py +5 -0
  3. cleanframe/_util.py +438 -0
  4. cleanframe/_version.py +1 -0
  5. cleanframe/api.py +559 -0
  6. cleanframe/cli.py +688 -0
  7. cleanframe/codegen.py +666 -0
  8. cleanframe/dataio.py +506 -0
  9. cleanframe/detectors/__init__.py +43 -0
  10. cleanframe/detectors/base.py +221 -0
  11. cleanframe/detectors/categories.py +199 -0
  12. cleanframe/detectors/contacts.py +105 -0
  13. cleanframe/detectors/currency.py +112 -0
  14. cleanframe/detectors/dates.py +204 -0
  15. cleanframe/detectors/dedup.py +150 -0
  16. cleanframe/detectors/nulls.py +109 -0
  17. cleanframe/detectors/outliers.py +73 -0
  18. cleanframe/detectors/schema_mapping.py +125 -0
  19. cleanframe/detectors/text.py +105 -0
  20. cleanframe/detectors/units.py +86 -0
  21. cleanframe/diff.py +369 -0
  22. cleanframe/drift.py +283 -0
  23. cleanframe/errors.py +66 -0
  24. cleanframe/executor.py +229 -0
  25. cleanframe/fingerprint.py +83 -0
  26. cleanframe/issues.py +186 -0
  27. cleanframe/llm.py +811 -0
  28. cleanframe/ops.py +1245 -0
  29. cleanframe/planner.py +353 -0
  30. cleanframe/profile.py +413 -0
  31. cleanframe/py.typed +1 -0
  32. cleanframe/quality.py +81 -0
  33. cleanframe/readfix.py +160 -0
  34. cleanframe/recipe.py +398 -0
  35. cleanframe/report.py +345 -0
  36. cleanframe/result.py +144 -0
  37. cleanframe/schema.py +259 -0
  38. cleanframe/streaming.py +354 -0
  39. cleanframe/types.py +119 -0
  40. cleanframe/validate.py +363 -0
  41. cleanframe/workbook.py +370 -0
  42. cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
  43. cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
  44. cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
  45. cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
  46. cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
cleanframe/dataio.py ADDED
@@ -0,0 +1,506 @@
1
+ """Reading and writing dataframes by file extension.
2
+
3
+ A thin, predictable wrapper so the CLI and API accept a path anywhere a dataframe
4
+ is expected. CSV/TSV/Excel/Parquet/JSON are dispatched by suffix. Reading uses
5
+ pandas' default NA handling (blank/``NA``/``null`` → NaN); CleanFrame's detectors
6
+ then catch the *disguised* nulls pandas leaves behind (``unknown``, ``-``, ``?``).
7
+
8
+ Cross-platform defaults:
9
+
10
+ * Paths go through :class:`pathlib.Path` (Windows / POSIX alike).
11
+ * CSV/TSV reads use ``utf-8-sig`` so Excel/Notepad BOMs on Windows don't break.
12
+ * CSV/TSV writes use UTF-8 + ``\\n`` line endings (not OS-dependent ``\\r\\n``).
13
+ * Parent directories are created automatically on write.
14
+ * Formula-like cells are sanitised on CSV/TSV export by default.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import csv as _csv
20
+ import os
21
+ import re as _re
22
+ from pathlib import Path
23
+ from typing import Any
24
+
25
+ import pandas as pd
26
+
27
+ from ._util import (
28
+ check_output_target,
29
+ looks_like_code_values,
30
+ sanitize_dataframe_for_csv,
31
+ sanitize_dataframe_for_spreadsheet,
32
+ )
33
+ from .errors import CleanFrameError, OutputError
34
+
35
+ #: Encoding for CSV/TSV. ``utf-8-sig`` accepts a BOM and writes plain UTF-8 when
36
+ #: pandas strips the sig on read; we still pass ``encoding="utf-8"`` on write.
37
+ _CSV_READ_ENCODING = "utf-8-sig"
38
+ _CSV_WRITE_ENCODING = "utf-8"
39
+
40
+
41
+ _EXCEL_SUFFIXES = (".xlsx", ".xls", ".xlsm")
42
+ #: Read as delimited text. Anything else is refused rather than parsed as CSV.
43
+ _CSV_SUFFIXES = (".csv", ".txt", ".tsv", ".dat", "")
44
+ _TYPED_SUFFIXES = (".parquet", ".json")
45
+
46
+
47
+ def _detail(exc: BaseException) -> str:
48
+ return str(exc).strip().rstrip(".")
49
+
50
+
51
+ def _text_kwargs() -> dict[str, Any]:
52
+ """Read every field verbatim: no numeric coercion, no invented nulls.
53
+
54
+ Only an empty field becomes NaN, so leading zeros, ``NA``/``None`` tokens and
55
+ ``1e5`` survive read-time inference and stay visible in the diff.
56
+ """
57
+ return {"dtype": str, "keep_default_na": False, "na_values": [""]}
58
+
59
+
60
+ def _pandas_skiprows(skiprows: int | list[int] | None, blank_lines: int):
61
+ """Translate public ``skiprows`` into pandas line indices.
62
+
63
+ Public semantics are data rows, never file lines: ``skiprows=2`` drops the first
64
+ two records and keeps the header (a bare pandas ``skiprows=2`` would eat it), and
65
+ a list holds 1-based data-row numbers. ``blank_lines`` are the empty lines the
66
+ format corrector found above the header.
67
+ """
68
+ lines = list(range(blank_lines))
69
+ if skiprows is None:
70
+ return lines or None
71
+ if isinstance(skiprows, bool):
72
+ raise CleanFrameError("skiprows must be a row count or a list of row numbers.")
73
+ if isinstance(skiprows, int):
74
+ if skiprows < 0:
75
+ raise CleanFrameError(f"skiprows must be 0 or more, got {skiprows}.")
76
+ lines += [blank_lines + i for i in range(1, skiprows + 1)]
77
+ elif isinstance(skiprows, (list, tuple)):
78
+ bad = [r for r in skiprows if isinstance(r, bool) or not isinstance(r, int) or r < 1]
79
+ if bad:
80
+ raise CleanFrameError(
81
+ f"skiprows entries must be data-row numbers of 1 or more, got {bad}."
82
+ )
83
+ lines += [blank_lines + int(r) for r in skiprows]
84
+ else:
85
+ raise CleanFrameError(
86
+ f"skiprows must be a row count or a list of row numbers, got "
87
+ f"{type(skiprows).__name__}."
88
+ )
89
+ return lines or None
90
+
91
+
92
+ def _check_csv_header(path: Path, encoding: str, sep: str | None, blank_lines: int) -> None:
93
+ """Refuse duplicate or blank CSV headers instead of letting pandas rename them.
94
+
95
+ pandas turns a repeated ``name`` into ``name.1`` and a blank one into
96
+ ``Unnamed: 3``, so the recipe would key lineage off a label the file never had.
97
+ """
98
+ try:
99
+ with open(path, encoding=encoding, newline="") as fh:
100
+ for _ in range(blank_lines):
101
+ fh.readline()
102
+ line = fh.readline()
103
+ except (OSError, UnicodeDecodeError, LookupError):
104
+ return
105
+ if not line.strip():
106
+ return
107
+ fields = next(_csv.reader([line.rstrip("\r\n")], delimiter=sep or ","), None)
108
+ if not fields or len(fields) < 2:
109
+ return
110
+ names = [f.strip() for f in fields]
111
+ if any(not n for n in names):
112
+ raise CleanFrameError(
113
+ f"{path.name} has an empty column name in its header row "
114
+ f"(position {names.index('') + 1} of {len(names)}). Name every column, or "
115
+ "select the ones you need with columns=."
116
+ )
117
+ seen: dict[str, int] = {}
118
+ for n in names:
119
+ seen[n] = seen.get(n, 0) + 1
120
+ dups = sorted(n for n, c in seen.items() if c > 1)
121
+ if dups:
122
+ raise CleanFrameError(
123
+ f"{path.name} has duplicate column name(s) {dups} in its header row. "
124
+ "CleanFrame needs unique names — rename them in the file, or select a "
125
+ "subset with columns=."
126
+ )
127
+
128
+
129
+ def excel_sheet_names(path: str | Path) -> list[str]:
130
+ """Return a workbook's sheet names in file order (raises CleanFrameError on error)."""
131
+ path = Path(path)
132
+ try:
133
+ with pd.ExcelFile(path) as workbook:
134
+ return list(workbook.sheet_names)
135
+ except ImportError as exc: # pragma: no cover - optional engine missing
136
+ raise CleanFrameError(
137
+ f"Reading Excel requires openpyxl ({_detail(exc)}). "
138
+ "Try `pip install cleanframe-engine[excel]`."
139
+ ) from exc
140
+ except CleanFrameError:
141
+ raise
142
+ except Exception as exc: # noqa: BLE001
143
+ raise CleanFrameError(
144
+ f"Could not open workbook {path.name}: {_detail(exc)}."
145
+ ) from exc
146
+
147
+
148
+ def _apply_row_slice(df: pd.DataFrame, nrows: int | None, skiprows) -> pd.DataFrame:
149
+ """Row selection for formats with no header line (parquet/json).
150
+
151
+ Uses the same public semantics as the CSV path: an int drops that many leading
152
+ data rows, a list names 1-based data rows.
153
+ """
154
+ if skiprows is not None:
155
+ if isinstance(skiprows, int) and not isinstance(skiprows, bool):
156
+ if skiprows < 0:
157
+ raise CleanFrameError(f"skiprows must be 0 or more, got {skiprows}.")
158
+ df = df.iloc[skiprows:]
159
+ elif isinstance(skiprows, (list, tuple)):
160
+ drop = {int(r) - 1 for r in skiprows}
161
+ df = df.iloc[[i for i in range(len(df)) if i not in drop]]
162
+ else:
163
+ raise CleanFrameError(
164
+ "skiprows must be a row count or a list of row numbers, got "
165
+ f"{type(skiprows).__name__}."
166
+ )
167
+ if nrows is not None:
168
+ if not isinstance(nrows, int) or isinstance(nrows, bool) or nrows < 0:
169
+ raise CleanFrameError(f"nrows must be 0 or more, got {nrows!r}.")
170
+ df = df.head(nrows)
171
+ return df
172
+
173
+
174
+ _LEADING_ZERO_RE = _re.compile(r"^[+-]?0\d")
175
+ _SCI_RE = _re.compile(r"^[+-]?\d+(?:\.\d+)?[eE][+-]?\d+$")
176
+ _BOOL_TEXT = frozenset({"true", "false", "yes", "no"})
177
+ #: Spellings that mean "missing" in most datasets. Reading them as NaN is what the
178
+ #: file asked for, so it is not worth a warning — unless the column holds short
179
+ #: codes, where ``NA`` is Namibia. ``None``/``nil`` stay reportable everywhere.
180
+ _PLAIN_NULL_TEXT = frozenset(
181
+ {"n/a", "na", "null", "nan", "-nan", "<na>", "#n/a", "#na", "#n/a n/a"}
182
+ )
183
+
184
+
185
+ def compare_for_losses(df: pd.DataFrame, raw: pd.DataFrame) -> dict[str, str]:
186
+ """Columns where the coerced frame differs from the verbatim one, and how."""
187
+ losses: dict[str, str] = {}
188
+ rows = min(len(raw), len(df))
189
+ for col in df.columns:
190
+ if col not in raw.columns:
191
+ continue
192
+ coerced = df[col].to_numpy()
193
+ literal = raw[col].to_numpy()
194
+ code_column = looks_like_code_values(literal[:rows])
195
+ for i in range(rows):
196
+ text = literal[i]
197
+ if not isinstance(text, str) or not text.strip():
198
+ continue
199
+ token = text.strip()
200
+ value = coerced[i]
201
+ try:
202
+ missing = bool(pd.isna(value))
203
+ except (TypeError, ValueError): # pragma: no cover - exotic cell
204
+ continue
205
+ if missing:
206
+ if token.casefold() in _PLAIN_NULL_TEXT and not code_column:
207
+ continue
208
+ losses[str(col)] = f"{token!r} was read as a missing value"
209
+ elif isinstance(value, bool) and token.casefold() in _BOOL_TEXT:
210
+ losses[str(col)] = f"{token!r} was read as the boolean {value}"
211
+ elif not isinstance(value, str) and _LEADING_ZERO_RE.match(token):
212
+ losses[str(col)] = f"{token!r} lost its leading zero(s) and became {value}"
213
+ elif not isinstance(value, str) and _SCI_RE.match(token):
214
+ losses[str(col)] = f"{token!r} was rewritten as {value}"
215
+ else:
216
+ continue
217
+ break
218
+ return losses
219
+
220
+
221
+ def inference_losses(path: str | Path, df: pd.DataFrame, *, limit: int = 200, **read_kwargs):
222
+ """Columns whose values pandas changed while reading ``path``, and how.
223
+
224
+ Read-time coercion happens before CleanFrame sees the data, so no diff can show
225
+ it: a leading-zero ZIP, a ``None`` token or a country code of ``NA`` is already
226
+ gone. A bounded verbatim re-read makes the loss visible.
227
+ """
228
+ try:
229
+ raw = read_frame(path, nrows=limit, text=True, **read_kwargs)
230
+ except CleanFrameError:
231
+ return {}
232
+ return compare_for_losses(df, raw)
233
+
234
+
235
+ def read_frame(
236
+ path: str | Path,
237
+ *,
238
+ sheet: str | int | None = None,
239
+ columns: list[str] | None = None,
240
+ nrows: int | None = None,
241
+ skiprows: int | list[int] | None = None,
242
+ blank_lines: int = 0,
243
+ text: bool = False,
244
+ **kwargs,
245
+ ) -> pd.DataFrame:
246
+ """Read a single dataframe, dispatching on file extension.
247
+
248
+ Parameters
249
+ ----------
250
+ sheet:
251
+ Excel only — a sheet name or 0-based index. If a workbook has more than one
252
+ sheet and none is chosen, a :class:`~cleanframe.errors.CleanFrameError` is
253
+ raised (never silently pick the first). Use :func:`cleanframe.clean_workbook`
254
+ to clean every sheet.
255
+ columns / nrows / skiprows:
256
+ Select a subset of columns (``usecols``) and/or a row range. ``columns`` is a
257
+ *filter*, not a reorder — output keeps file order. ``skiprows`` counts *data
258
+ rows*, never file lines: an int drops that many leading records and keeps the
259
+ header; a list names 1-based data rows. Under ``skiprows``/``nrows`` the diff's
260
+ ``row_id`` is relative to the loaded slice, not the physical file line.
261
+ text:
262
+ Read every field as a string (CSV/Excel). Pandas otherwise infers types while
263
+ reading, which drops leading zeros, turns ``NA``/``None`` text into NaN and
264
+ rewrites ``1e5`` — losses no diff can show because they happen before CleanFrame
265
+ sees the data.
266
+
267
+ Any failure pandas/pyarrow would surface as a raw traceback is re-raised as a
268
+ :class:`~cleanframe.errors.CleanFrameError` with an actionable hint.
269
+ """
270
+ path = Path(path)
271
+ if not path.exists():
272
+ raise CleanFrameError(f"Input file not found: {path}")
273
+ if not path.is_file():
274
+ raise CleanFrameError(f"Input path is not a file (is it a directory?): {path}")
275
+ if path.stat().st_size == 0:
276
+ raise CleanFrameError(f"Input file is empty (no columns to parse): {path}")
277
+ suffix = path.suffix.lower()
278
+ is_excel = suffix in _EXCEL_SUFFIXES
279
+ if sheet is not None and not is_excel:
280
+ raise CleanFrameError(f"sheet= is only valid for Excel files, not {suffix or 'this file'}.")
281
+ if not is_excel and suffix not in _CSV_SUFFIXES and suffix not in _TYPED_SUFFIXES:
282
+ raise CleanFrameError(
283
+ f"Unsupported input format {suffix!r} for {path.name}. CleanFrame reads "
284
+ ".csv/.txt/.tsv/.dat, .xlsx/.xlsm/.xls, .parquet and .json. Convert the file, "
285
+ "or rename it to the extension matching its contents."
286
+ )
287
+ try:
288
+ if is_excel:
289
+ if sheet is not None:
290
+ names = excel_sheet_names(path)
291
+ if isinstance(sheet, str) and sheet not in names:
292
+ hint = (
293
+ " A bare number is read as a sheet *name*; use the CLI form "
294
+ f"--sheet '#{sheet}' (or sheet={sheet} in Python) for a positional index."
295
+ if sheet.isdigit()
296
+ else ""
297
+ )
298
+ raise CleanFrameError(
299
+ f"Sheet {sheet!r} not found in {path.name}. "
300
+ f"Available sheets: {names}.{hint}"
301
+ )
302
+ if isinstance(sheet, int) and not -len(names) <= sheet < len(names):
303
+ raise CleanFrameError(
304
+ f"Sheet index {sheet} is out of range for {path.name}, which has "
305
+ f"{len(names)} sheet(s): {names}."
306
+ )
307
+ if sheet is None:
308
+ names = excel_sheet_names(path)
309
+ if len(names) > 1:
310
+ raise CleanFrameError(
311
+ f"{path.name} has {len(names)} sheets ({names}). Pass sheet='NAME' "
312
+ "(or a 0-based index) to pick one, or use cleanframe.clean_workbook() "
313
+ "/ the `cleanframe clean` CLI, which cleans every sheet."
314
+ )
315
+ sheet = 0
316
+ xl_kwargs = dict(kwargs)
317
+ xl_kwargs.pop("encoding", None)
318
+ xl_kwargs.pop("sep", None)
319
+ if text:
320
+ for key, value in _text_kwargs().items():
321
+ xl_kwargs.setdefault(key, value)
322
+ if columns is not None:
323
+ available = list(pd.read_excel(path, sheet_name=sheet, nrows=0).columns)
324
+ missing = [c for c in columns if c not in available]
325
+ if missing:
326
+ raise CleanFrameError(
327
+ f"Requested column(s) not found in {path.name}: {missing}. "
328
+ f"Available: {available}."
329
+ )
330
+ xl_kwargs["usecols"] = list(columns)
331
+ if nrows is not None:
332
+ xl_kwargs["nrows"] = nrows
333
+ xl_skiprows = _pandas_skiprows(skiprows, blank_lines)
334
+ if xl_skiprows is not None:
335
+ xl_kwargs["skiprows"] = xl_skiprows
336
+ df = pd.read_excel(path, sheet_name=sheet, **xl_kwargs)
337
+ if isinstance(df, dict):
338
+ raise CleanFrameError("sheet must select a single sheet (a name or index).")
339
+ return df
340
+ if suffix == ".parquet":
341
+ df = pd.read_parquet(path, columns=list(columns) if columns else None, **kwargs)
342
+ return _apply_row_slice(df, nrows, skiprows)
343
+ if suffix == ".json":
344
+ kwargs.setdefault("encoding", "utf-8")
345
+ df = pd.read_json(path, **kwargs)
346
+ if columns is not None:
347
+ missing = [c for c in columns if c not in df.columns]
348
+ if missing:
349
+ raise CleanFrameError(
350
+ f"Requested column(s) not found in {path.name}: {missing}. "
351
+ f"Available: {list(df.columns)}."
352
+ )
353
+ df = df[list(columns)]
354
+ return _apply_row_slice(df, nrows, skiprows)
355
+ # CSV family (.csv/.txt/.tsv/.dat and extension-less files).
356
+ kwargs.setdefault("encoding", _CSV_READ_ENCODING)
357
+ if suffix == ".tsv":
358
+ kwargs.setdefault("sep", "\t")
359
+ if text:
360
+ for key, value in _text_kwargs().items():
361
+ kwargs.setdefault(key, value)
362
+ _check_csv_header(path, kwargs["encoding"], kwargs.get("sep"), blank_lines)
363
+ # index_col=False: a row with one field too many (a stray trailing delimiter)
364
+ # otherwise becomes the index and shifts every column one place left.
365
+ kwargs.setdefault("index_col", False)
366
+ csv_skiprows = _pandas_skiprows(skiprows, blank_lines)
367
+ if csv_skiprows is not None:
368
+ kwargs["skiprows"] = csv_skiprows
369
+ if columns is not None:
370
+ header_kwargs = {
371
+ k: v for k, v in kwargs.items() if k in ("encoding", "sep", "skiprows", "index_col")
372
+ }
373
+ available = list(pd.read_csv(path, nrows=0, **header_kwargs).columns)
374
+ missing = [c for c in columns if c not in available]
375
+ if missing:
376
+ raise CleanFrameError(
377
+ f"Requested column(s) not found in {path.name}: {missing}. "
378
+ f"Available: {available}."
379
+ )
380
+ kwargs["usecols"] = list(columns)
381
+ if nrows is not None:
382
+ if not isinstance(nrows, int) or isinstance(nrows, bool) or nrows < 0:
383
+ raise CleanFrameError(f"nrows must be 0 or more, got {nrows!r}.")
384
+ kwargs["nrows"] = nrows
385
+ return pd.read_csv(path, **kwargs)
386
+ except ImportError as exc: # pragma: no cover - optional engine missing
387
+ hint = "Try `pip install cleanframe-engine[excel]` for Excel support."
388
+ if suffix == ".parquet":
389
+ hint = "Try `pip install cleanframe-engine[parquet]` (pyarrow) for Parquet support."
390
+ elif suffix == ".xls":
391
+ hint = "Legacy .xls needs xlrd: `pip install xlrd` (.xlsx needs only openpyxl)."
392
+ raise CleanFrameError(
393
+ f"Reading {suffix} requires an extra engine ({_detail(exc)}). {hint}"
394
+ ) from exc
395
+ except CleanFrameError:
396
+ raise
397
+ except UnicodeDecodeError as exc:
398
+ raise CleanFrameError(
399
+ f"Could not decode {path.name} as UTF-8 ({exc}). It may be saved as "
400
+ "Latin-1 / Windows-1252 / UTF-16 (common for Excel 'Save as CSV' on "
401
+ "Windows). Read it with cleanframe.read_frame(..., encoding='cp1252'), "
402
+ "let `clean`/`report` detect the encoding for you, or record it in the "
403
+ "recipe's read: section (encoding: cp1252) for replay."
404
+ ) from exc
405
+ except (pd.errors.ParserError, pd.errors.EmptyDataError) as exc:
406
+ raise CleanFrameError(
407
+ f"Could not parse {path.name}: {type(exc).__name__}: {_detail(exc)}. "
408
+ "Check the delimiter/quoting, that the file matches its extension, and "
409
+ "that it is not truncated."
410
+ ) from exc
411
+ except OSError as exc:
412
+ raise CleanFrameError(f"Could not read {path}: {_detail(exc)}.") from exc
413
+ except Exception as exc: # noqa: BLE001 - IO boundary: surface any failure cleanly
414
+ raise CleanFrameError(
415
+ f"Could not read {path.name}: {type(exc).__name__}: {_detail(exc)}."
416
+ ) from exc
417
+
418
+
419
+ def _write_dispatch(
420
+ df: pd.DataFrame, target: Path, suffix: str, sanitize_csv: bool, kwargs: dict
421
+ ) -> None:
422
+ if suffix == ".parquet":
423
+ df.to_parquet(target, index=False, **kwargs)
424
+ return
425
+ if suffix == ".json":
426
+ kwargs.setdefault("force_ascii", False)
427
+ df.to_json(target, orient="records", indent=2, **kwargs)
428
+ return
429
+ if suffix in (".xlsx", ".xlsm"):
430
+ out = sanitize_dataframe_for_spreadsheet(df) if sanitize_csv else df
431
+ kwargs.setdefault("engine", "openpyxl")
432
+ out.to_excel(target, index=False, **kwargs)
433
+ return
434
+ out = sanitize_dataframe_for_csv(df) if sanitize_csv else df
435
+ kwargs.setdefault("encoding", _CSV_WRITE_ENCODING)
436
+ kwargs.setdefault("lineterminator", "\n")
437
+ kwargs.setdefault("sep", "\t" if suffix == ".tsv" else ",")
438
+ out.to_csv(target, index=False, **kwargs)
439
+
440
+
441
+ def write_frame(
442
+ df: pd.DataFrame,
443
+ path: str | Path,
444
+ *,
445
+ sanitize_csv: bool = True,
446
+ source: str | Path | None = None,
447
+ overwrite: bool = False,
448
+ **kwargs,
449
+ ) -> Path:
450
+ """Write a dataframe, dispatching on file extension. Never writes the index.
451
+
452
+ The frame is written to a temporary sibling and moved into place, so a failure
453
+ part-way through leaves the previous file intact rather than a truncated one.
454
+
455
+ Parameters
456
+ ----------
457
+ sanitize_csv:
458
+ When ``True`` (default), string cells and headers that look like spreadsheet
459
+ formulas (leading ``=``, ``@``, or ``+``/``-`` followed by a non-number) are
460
+ escaped before CSV/TSV/Excel export. Set ``False`` only when you intentionally
461
+ need raw formula cells.
462
+ source / overwrite:
463
+ ``source`` is the path the data was read from; writing back over it needs
464
+ ``overwrite=True`` so the original is never destroyed by accident.
465
+ """
466
+ if not isinstance(df, pd.DataFrame):
467
+ raise CleanFrameError(f"write_frame expects a DataFrame, got {type(df).__name__}.")
468
+ path = check_output_target(path, source, overwrite=overwrite)
469
+ suffix = path.suffix.lower()
470
+ if suffix == ".xls":
471
+ raise OutputError(
472
+ "Writing legacy .xls is not supported: pandas emits .xlsx bytes, which "
473
+ "Excel refuses under an .xls name. Write .xlsx instead."
474
+ )
475
+ tmp = path.with_name(path.name + ".cf-tmp")
476
+ try:
477
+ _write_dispatch(df, tmp, suffix, sanitize_csv, kwargs)
478
+ os.replace(tmp, path)
479
+ except ImportError as exc: # pragma: no cover - optional engine missing
480
+ tmp.unlink(missing_ok=True)
481
+ extra = "parquet" if suffix == ".parquet" else "excel"
482
+ raise OutputError(
483
+ f"Writing {suffix} requires an extra engine ({_detail(exc)}). "
484
+ f"Try `pip install cleanframe-engine[{extra}]`."
485
+ ) from exc
486
+ except CleanFrameError:
487
+ tmp.unlink(missing_ok=True)
488
+ raise
489
+ except OSError as exc:
490
+ tmp.unlink(missing_ok=True)
491
+ raise OutputError(f"Could not write {path}: {_detail(exc)}.") from exc
492
+ except Exception as exc: # noqa: BLE001 - IO boundary
493
+ tmp.unlink(missing_ok=True)
494
+ raise OutputError(
495
+ f"Could not write {path.name}: {type(exc).__name__}: {_detail(exc)}."
496
+ ) from exc
497
+ return path
498
+
499
+
500
+ __all__ = [
501
+ "read_frame",
502
+ "write_frame",
503
+ "excel_sheet_names",
504
+ "inference_losses",
505
+ "compare_for_losses",
506
+ ]
@@ -0,0 +1,43 @@
1
+ """Detector registry and the built-in detector suite.
2
+
3
+ Importing this package registers every built-in detector (importing the modules
4
+ runs their ``@detector`` decorators). Third-party detectors register themselves the
5
+ same way — by being imported — so ``import cleanframe`` wires up the built-ins and
6
+ users add their own with ``@cf.detector(...)``.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ # Importing each module triggers its @detector registrations. Ordered by the
12
+ # priority they run at, for readability only.
13
+ from . import (
14
+ categories, # noqa: E402,F401 priority 50
15
+ contacts, # noqa: E402,F401 priority 45
16
+ currency, # noqa: E402,F401 priority 45
17
+ dates, # noqa: E402,F401 priority 40
18
+ dedup, # noqa: E402,F401 priority 80
19
+ nulls, # noqa: E402,F401 priority 20
20
+ outliers, # noqa: E402,F401 priority 70
21
+ schema_mapping, # noqa: E402,F401 priority 5
22
+ text, # noqa: E402,F401 priority 10, 60
23
+ units, # noqa: E402,F401 priority 46
24
+ )
25
+ from .base import (
26
+ DETECTOR_REGISTRY,
27
+ DetectorContext,
28
+ DetectorSpec,
29
+ detector,
30
+ list_detectors,
31
+ run_detectors,
32
+ unregister_detector,
33
+ )
34
+
35
+ __all__ = [
36
+ "DETECTOR_REGISTRY",
37
+ "DetectorContext",
38
+ "DetectorSpec",
39
+ "detector",
40
+ "list_detectors",
41
+ "run_detectors",
42
+ "unregister_detector",
43
+ ]