cleanframe-engine 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanframe/__init__.py +169 -0
- cleanframe/__main__.py +5 -0
- cleanframe/_util.py +438 -0
- cleanframe/_version.py +1 -0
- cleanframe/api.py +559 -0
- cleanframe/cli.py +688 -0
- cleanframe/codegen.py +666 -0
- cleanframe/dataio.py +506 -0
- cleanframe/detectors/__init__.py +43 -0
- cleanframe/detectors/base.py +221 -0
- cleanframe/detectors/categories.py +199 -0
- cleanframe/detectors/contacts.py +105 -0
- cleanframe/detectors/currency.py +112 -0
- cleanframe/detectors/dates.py +204 -0
- cleanframe/detectors/dedup.py +150 -0
- cleanframe/detectors/nulls.py +109 -0
- cleanframe/detectors/outliers.py +73 -0
- cleanframe/detectors/schema_mapping.py +125 -0
- cleanframe/detectors/text.py +105 -0
- cleanframe/detectors/units.py +86 -0
- cleanframe/diff.py +369 -0
- cleanframe/drift.py +283 -0
- cleanframe/errors.py +66 -0
- cleanframe/executor.py +229 -0
- cleanframe/fingerprint.py +83 -0
- cleanframe/issues.py +186 -0
- cleanframe/llm.py +811 -0
- cleanframe/ops.py +1245 -0
- cleanframe/planner.py +353 -0
- cleanframe/profile.py +413 -0
- cleanframe/py.typed +1 -0
- cleanframe/quality.py +81 -0
- cleanframe/readfix.py +160 -0
- cleanframe/recipe.py +398 -0
- cleanframe/report.py +345 -0
- cleanframe/result.py +144 -0
- cleanframe/schema.py +259 -0
- cleanframe/streaming.py +354 -0
- cleanframe/types.py +119 -0
- cleanframe/validate.py +363 -0
- cleanframe/workbook.py +370 -0
- cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
- cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
- cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
- cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
- cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
cleanframe/dataio.py
ADDED
|
@@ -0,0 +1,506 @@
|
|
|
1
|
+
"""Reading and writing dataframes by file extension.
|
|
2
|
+
|
|
3
|
+
A thin, predictable wrapper so the CLI and API accept a path anywhere a dataframe
|
|
4
|
+
is expected. CSV/TSV/Excel/Parquet/JSON are dispatched by suffix. Reading uses
|
|
5
|
+
pandas' default NA handling (blank/``NA``/``null`` → NaN); CleanFrame's detectors
|
|
6
|
+
then catch the *disguised* nulls pandas leaves behind (``unknown``, ``-``, ``?``).
|
|
7
|
+
|
|
8
|
+
Cross-platform defaults:
|
|
9
|
+
|
|
10
|
+
* Paths go through :class:`pathlib.Path` (Windows / POSIX alike).
|
|
11
|
+
* CSV/TSV reads use ``utf-8-sig`` so Excel/Notepad BOMs on Windows don't break.
|
|
12
|
+
* CSV/TSV writes use UTF-8 + ``\\n`` line endings (not OS-dependent ``\\r\\n``).
|
|
13
|
+
* Parent directories are created automatically on write.
|
|
14
|
+
* Formula-like cells are sanitised on CSV/TSV export by default.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import csv as _csv
|
|
20
|
+
import os
|
|
21
|
+
import re as _re
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
import pandas as pd
|
|
26
|
+
|
|
27
|
+
from ._util import (
|
|
28
|
+
check_output_target,
|
|
29
|
+
looks_like_code_values,
|
|
30
|
+
sanitize_dataframe_for_csv,
|
|
31
|
+
sanitize_dataframe_for_spreadsheet,
|
|
32
|
+
)
|
|
33
|
+
from .errors import CleanFrameError, OutputError
|
|
34
|
+
|
|
35
|
+
#: Encoding for CSV/TSV. ``utf-8-sig`` accepts a BOM and writes plain UTF-8 when
|
|
36
|
+
#: pandas strips the sig on read; we still pass ``encoding="utf-8"`` on write.
|
|
37
|
+
_CSV_READ_ENCODING = "utf-8-sig"
|
|
38
|
+
_CSV_WRITE_ENCODING = "utf-8"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
_EXCEL_SUFFIXES = (".xlsx", ".xls", ".xlsm")
|
|
42
|
+
#: Read as delimited text. Anything else is refused rather than parsed as CSV.
|
|
43
|
+
_CSV_SUFFIXES = (".csv", ".txt", ".tsv", ".dat", "")
|
|
44
|
+
_TYPED_SUFFIXES = (".parquet", ".json")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _detail(exc: BaseException) -> str:
|
|
48
|
+
return str(exc).strip().rstrip(".")
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _text_kwargs() -> dict[str, Any]:
|
|
52
|
+
"""Read every field verbatim: no numeric coercion, no invented nulls.
|
|
53
|
+
|
|
54
|
+
Only an empty field becomes NaN, so leading zeros, ``NA``/``None`` tokens and
|
|
55
|
+
``1e5`` survive read-time inference and stay visible in the diff.
|
|
56
|
+
"""
|
|
57
|
+
return {"dtype": str, "keep_default_na": False, "na_values": [""]}
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _pandas_skiprows(skiprows: int | list[int] | None, blank_lines: int):
|
|
61
|
+
"""Translate public ``skiprows`` into pandas line indices.
|
|
62
|
+
|
|
63
|
+
Public semantics are data rows, never file lines: ``skiprows=2`` drops the first
|
|
64
|
+
two records and keeps the header (a bare pandas ``skiprows=2`` would eat it), and
|
|
65
|
+
a list holds 1-based data-row numbers. ``blank_lines`` are the empty lines the
|
|
66
|
+
format corrector found above the header.
|
|
67
|
+
"""
|
|
68
|
+
lines = list(range(blank_lines))
|
|
69
|
+
if skiprows is None:
|
|
70
|
+
return lines or None
|
|
71
|
+
if isinstance(skiprows, bool):
|
|
72
|
+
raise CleanFrameError("skiprows must be a row count or a list of row numbers.")
|
|
73
|
+
if isinstance(skiprows, int):
|
|
74
|
+
if skiprows < 0:
|
|
75
|
+
raise CleanFrameError(f"skiprows must be 0 or more, got {skiprows}.")
|
|
76
|
+
lines += [blank_lines + i for i in range(1, skiprows + 1)]
|
|
77
|
+
elif isinstance(skiprows, (list, tuple)):
|
|
78
|
+
bad = [r for r in skiprows if isinstance(r, bool) or not isinstance(r, int) or r < 1]
|
|
79
|
+
if bad:
|
|
80
|
+
raise CleanFrameError(
|
|
81
|
+
f"skiprows entries must be data-row numbers of 1 or more, got {bad}."
|
|
82
|
+
)
|
|
83
|
+
lines += [blank_lines + int(r) for r in skiprows]
|
|
84
|
+
else:
|
|
85
|
+
raise CleanFrameError(
|
|
86
|
+
f"skiprows must be a row count or a list of row numbers, got "
|
|
87
|
+
f"{type(skiprows).__name__}."
|
|
88
|
+
)
|
|
89
|
+
return lines or None
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _check_csv_header(path: Path, encoding: str, sep: str | None, blank_lines: int) -> None:
|
|
93
|
+
"""Refuse duplicate or blank CSV headers instead of letting pandas rename them.
|
|
94
|
+
|
|
95
|
+
pandas turns a repeated ``name`` into ``name.1`` and a blank one into
|
|
96
|
+
``Unnamed: 3``, so the recipe would key lineage off a label the file never had.
|
|
97
|
+
"""
|
|
98
|
+
try:
|
|
99
|
+
with open(path, encoding=encoding, newline="") as fh:
|
|
100
|
+
for _ in range(blank_lines):
|
|
101
|
+
fh.readline()
|
|
102
|
+
line = fh.readline()
|
|
103
|
+
except (OSError, UnicodeDecodeError, LookupError):
|
|
104
|
+
return
|
|
105
|
+
if not line.strip():
|
|
106
|
+
return
|
|
107
|
+
fields = next(_csv.reader([line.rstrip("\r\n")], delimiter=sep or ","), None)
|
|
108
|
+
if not fields or len(fields) < 2:
|
|
109
|
+
return
|
|
110
|
+
names = [f.strip() for f in fields]
|
|
111
|
+
if any(not n for n in names):
|
|
112
|
+
raise CleanFrameError(
|
|
113
|
+
f"{path.name} has an empty column name in its header row "
|
|
114
|
+
f"(position {names.index('') + 1} of {len(names)}). Name every column, or "
|
|
115
|
+
"select the ones you need with columns=."
|
|
116
|
+
)
|
|
117
|
+
seen: dict[str, int] = {}
|
|
118
|
+
for n in names:
|
|
119
|
+
seen[n] = seen.get(n, 0) + 1
|
|
120
|
+
dups = sorted(n for n, c in seen.items() if c > 1)
|
|
121
|
+
if dups:
|
|
122
|
+
raise CleanFrameError(
|
|
123
|
+
f"{path.name} has duplicate column name(s) {dups} in its header row. "
|
|
124
|
+
"CleanFrame needs unique names — rename them in the file, or select a "
|
|
125
|
+
"subset with columns=."
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def excel_sheet_names(path: str | Path) -> list[str]:
|
|
130
|
+
"""Return a workbook's sheet names in file order (raises CleanFrameError on error)."""
|
|
131
|
+
path = Path(path)
|
|
132
|
+
try:
|
|
133
|
+
with pd.ExcelFile(path) as workbook:
|
|
134
|
+
return list(workbook.sheet_names)
|
|
135
|
+
except ImportError as exc: # pragma: no cover - optional engine missing
|
|
136
|
+
raise CleanFrameError(
|
|
137
|
+
f"Reading Excel requires openpyxl ({_detail(exc)}). "
|
|
138
|
+
"Try `pip install cleanframe-engine[excel]`."
|
|
139
|
+
) from exc
|
|
140
|
+
except CleanFrameError:
|
|
141
|
+
raise
|
|
142
|
+
except Exception as exc: # noqa: BLE001
|
|
143
|
+
raise CleanFrameError(
|
|
144
|
+
f"Could not open workbook {path.name}: {_detail(exc)}."
|
|
145
|
+
) from exc
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _apply_row_slice(df: pd.DataFrame, nrows: int | None, skiprows) -> pd.DataFrame:
|
|
149
|
+
"""Row selection for formats with no header line (parquet/json).
|
|
150
|
+
|
|
151
|
+
Uses the same public semantics as the CSV path: an int drops that many leading
|
|
152
|
+
data rows, a list names 1-based data rows.
|
|
153
|
+
"""
|
|
154
|
+
if skiprows is not None:
|
|
155
|
+
if isinstance(skiprows, int) and not isinstance(skiprows, bool):
|
|
156
|
+
if skiprows < 0:
|
|
157
|
+
raise CleanFrameError(f"skiprows must be 0 or more, got {skiprows}.")
|
|
158
|
+
df = df.iloc[skiprows:]
|
|
159
|
+
elif isinstance(skiprows, (list, tuple)):
|
|
160
|
+
drop = {int(r) - 1 for r in skiprows}
|
|
161
|
+
df = df.iloc[[i for i in range(len(df)) if i not in drop]]
|
|
162
|
+
else:
|
|
163
|
+
raise CleanFrameError(
|
|
164
|
+
"skiprows must be a row count or a list of row numbers, got "
|
|
165
|
+
f"{type(skiprows).__name__}."
|
|
166
|
+
)
|
|
167
|
+
if nrows is not None:
|
|
168
|
+
if not isinstance(nrows, int) or isinstance(nrows, bool) or nrows < 0:
|
|
169
|
+
raise CleanFrameError(f"nrows must be 0 or more, got {nrows!r}.")
|
|
170
|
+
df = df.head(nrows)
|
|
171
|
+
return df
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
_LEADING_ZERO_RE = _re.compile(r"^[+-]?0\d")
|
|
175
|
+
_SCI_RE = _re.compile(r"^[+-]?\d+(?:\.\d+)?[eE][+-]?\d+$")
|
|
176
|
+
_BOOL_TEXT = frozenset({"true", "false", "yes", "no"})
|
|
177
|
+
#: Spellings that mean "missing" in most datasets. Reading them as NaN is what the
|
|
178
|
+
#: file asked for, so it is not worth a warning — unless the column holds short
|
|
179
|
+
#: codes, where ``NA`` is Namibia. ``None``/``nil`` stay reportable everywhere.
|
|
180
|
+
_PLAIN_NULL_TEXT = frozenset(
|
|
181
|
+
{"n/a", "na", "null", "nan", "-nan", "<na>", "#n/a", "#na", "#n/a n/a"}
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def compare_for_losses(df: pd.DataFrame, raw: pd.DataFrame) -> dict[str, str]:
|
|
186
|
+
"""Columns where the coerced frame differs from the verbatim one, and how."""
|
|
187
|
+
losses: dict[str, str] = {}
|
|
188
|
+
rows = min(len(raw), len(df))
|
|
189
|
+
for col in df.columns:
|
|
190
|
+
if col not in raw.columns:
|
|
191
|
+
continue
|
|
192
|
+
coerced = df[col].to_numpy()
|
|
193
|
+
literal = raw[col].to_numpy()
|
|
194
|
+
code_column = looks_like_code_values(literal[:rows])
|
|
195
|
+
for i in range(rows):
|
|
196
|
+
text = literal[i]
|
|
197
|
+
if not isinstance(text, str) or not text.strip():
|
|
198
|
+
continue
|
|
199
|
+
token = text.strip()
|
|
200
|
+
value = coerced[i]
|
|
201
|
+
try:
|
|
202
|
+
missing = bool(pd.isna(value))
|
|
203
|
+
except (TypeError, ValueError): # pragma: no cover - exotic cell
|
|
204
|
+
continue
|
|
205
|
+
if missing:
|
|
206
|
+
if token.casefold() in _PLAIN_NULL_TEXT and not code_column:
|
|
207
|
+
continue
|
|
208
|
+
losses[str(col)] = f"{token!r} was read as a missing value"
|
|
209
|
+
elif isinstance(value, bool) and token.casefold() in _BOOL_TEXT:
|
|
210
|
+
losses[str(col)] = f"{token!r} was read as the boolean {value}"
|
|
211
|
+
elif not isinstance(value, str) and _LEADING_ZERO_RE.match(token):
|
|
212
|
+
losses[str(col)] = f"{token!r} lost its leading zero(s) and became {value}"
|
|
213
|
+
elif not isinstance(value, str) and _SCI_RE.match(token):
|
|
214
|
+
losses[str(col)] = f"{token!r} was rewritten as {value}"
|
|
215
|
+
else:
|
|
216
|
+
continue
|
|
217
|
+
break
|
|
218
|
+
return losses
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def inference_losses(path: str | Path, df: pd.DataFrame, *, limit: int = 200, **read_kwargs):
|
|
222
|
+
"""Columns whose values pandas changed while reading ``path``, and how.
|
|
223
|
+
|
|
224
|
+
Read-time coercion happens before CleanFrame sees the data, so no diff can show
|
|
225
|
+
it: a leading-zero ZIP, a ``None`` token or a country code of ``NA`` is already
|
|
226
|
+
gone. A bounded verbatim re-read makes the loss visible.
|
|
227
|
+
"""
|
|
228
|
+
try:
|
|
229
|
+
raw = read_frame(path, nrows=limit, text=True, **read_kwargs)
|
|
230
|
+
except CleanFrameError:
|
|
231
|
+
return {}
|
|
232
|
+
return compare_for_losses(df, raw)
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def read_frame(
|
|
236
|
+
path: str | Path,
|
|
237
|
+
*,
|
|
238
|
+
sheet: str | int | None = None,
|
|
239
|
+
columns: list[str] | None = None,
|
|
240
|
+
nrows: int | None = None,
|
|
241
|
+
skiprows: int | list[int] | None = None,
|
|
242
|
+
blank_lines: int = 0,
|
|
243
|
+
text: bool = False,
|
|
244
|
+
**kwargs,
|
|
245
|
+
) -> pd.DataFrame:
|
|
246
|
+
"""Read a single dataframe, dispatching on file extension.
|
|
247
|
+
|
|
248
|
+
Parameters
|
|
249
|
+
----------
|
|
250
|
+
sheet:
|
|
251
|
+
Excel only — a sheet name or 0-based index. If a workbook has more than one
|
|
252
|
+
sheet and none is chosen, a :class:`~cleanframe.errors.CleanFrameError` is
|
|
253
|
+
raised (never silently pick the first). Use :func:`cleanframe.clean_workbook`
|
|
254
|
+
to clean every sheet.
|
|
255
|
+
columns / nrows / skiprows:
|
|
256
|
+
Select a subset of columns (``usecols``) and/or a row range. ``columns`` is a
|
|
257
|
+
*filter*, not a reorder — output keeps file order. ``skiprows`` counts *data
|
|
258
|
+
rows*, never file lines: an int drops that many leading records and keeps the
|
|
259
|
+
header; a list names 1-based data rows. Under ``skiprows``/``nrows`` the diff's
|
|
260
|
+
``row_id`` is relative to the loaded slice, not the physical file line.
|
|
261
|
+
text:
|
|
262
|
+
Read every field as a string (CSV/Excel). Pandas otherwise infers types while
|
|
263
|
+
reading, which drops leading zeros, turns ``NA``/``None`` text into NaN and
|
|
264
|
+
rewrites ``1e5`` — losses no diff can show because they happen before CleanFrame
|
|
265
|
+
sees the data.
|
|
266
|
+
|
|
267
|
+
Any failure pandas/pyarrow would surface as a raw traceback is re-raised as a
|
|
268
|
+
:class:`~cleanframe.errors.CleanFrameError` with an actionable hint.
|
|
269
|
+
"""
|
|
270
|
+
path = Path(path)
|
|
271
|
+
if not path.exists():
|
|
272
|
+
raise CleanFrameError(f"Input file not found: {path}")
|
|
273
|
+
if not path.is_file():
|
|
274
|
+
raise CleanFrameError(f"Input path is not a file (is it a directory?): {path}")
|
|
275
|
+
if path.stat().st_size == 0:
|
|
276
|
+
raise CleanFrameError(f"Input file is empty (no columns to parse): {path}")
|
|
277
|
+
suffix = path.suffix.lower()
|
|
278
|
+
is_excel = suffix in _EXCEL_SUFFIXES
|
|
279
|
+
if sheet is not None and not is_excel:
|
|
280
|
+
raise CleanFrameError(f"sheet= is only valid for Excel files, not {suffix or 'this file'}.")
|
|
281
|
+
if not is_excel and suffix not in _CSV_SUFFIXES and suffix not in _TYPED_SUFFIXES:
|
|
282
|
+
raise CleanFrameError(
|
|
283
|
+
f"Unsupported input format {suffix!r} for {path.name}. CleanFrame reads "
|
|
284
|
+
".csv/.txt/.tsv/.dat, .xlsx/.xlsm/.xls, .parquet and .json. Convert the file, "
|
|
285
|
+
"or rename it to the extension matching its contents."
|
|
286
|
+
)
|
|
287
|
+
try:
|
|
288
|
+
if is_excel:
|
|
289
|
+
if sheet is not None:
|
|
290
|
+
names = excel_sheet_names(path)
|
|
291
|
+
if isinstance(sheet, str) and sheet not in names:
|
|
292
|
+
hint = (
|
|
293
|
+
" A bare number is read as a sheet *name*; use the CLI form "
|
|
294
|
+
f"--sheet '#{sheet}' (or sheet={sheet} in Python) for a positional index."
|
|
295
|
+
if sheet.isdigit()
|
|
296
|
+
else ""
|
|
297
|
+
)
|
|
298
|
+
raise CleanFrameError(
|
|
299
|
+
f"Sheet {sheet!r} not found in {path.name}. "
|
|
300
|
+
f"Available sheets: {names}.{hint}"
|
|
301
|
+
)
|
|
302
|
+
if isinstance(sheet, int) and not -len(names) <= sheet < len(names):
|
|
303
|
+
raise CleanFrameError(
|
|
304
|
+
f"Sheet index {sheet} is out of range for {path.name}, which has "
|
|
305
|
+
f"{len(names)} sheet(s): {names}."
|
|
306
|
+
)
|
|
307
|
+
if sheet is None:
|
|
308
|
+
names = excel_sheet_names(path)
|
|
309
|
+
if len(names) > 1:
|
|
310
|
+
raise CleanFrameError(
|
|
311
|
+
f"{path.name} has {len(names)} sheets ({names}). Pass sheet='NAME' "
|
|
312
|
+
"(or a 0-based index) to pick one, or use cleanframe.clean_workbook() "
|
|
313
|
+
"/ the `cleanframe clean` CLI, which cleans every sheet."
|
|
314
|
+
)
|
|
315
|
+
sheet = 0
|
|
316
|
+
xl_kwargs = dict(kwargs)
|
|
317
|
+
xl_kwargs.pop("encoding", None)
|
|
318
|
+
xl_kwargs.pop("sep", None)
|
|
319
|
+
if text:
|
|
320
|
+
for key, value in _text_kwargs().items():
|
|
321
|
+
xl_kwargs.setdefault(key, value)
|
|
322
|
+
if columns is not None:
|
|
323
|
+
available = list(pd.read_excel(path, sheet_name=sheet, nrows=0).columns)
|
|
324
|
+
missing = [c for c in columns if c not in available]
|
|
325
|
+
if missing:
|
|
326
|
+
raise CleanFrameError(
|
|
327
|
+
f"Requested column(s) not found in {path.name}: {missing}. "
|
|
328
|
+
f"Available: {available}."
|
|
329
|
+
)
|
|
330
|
+
xl_kwargs["usecols"] = list(columns)
|
|
331
|
+
if nrows is not None:
|
|
332
|
+
xl_kwargs["nrows"] = nrows
|
|
333
|
+
xl_skiprows = _pandas_skiprows(skiprows, blank_lines)
|
|
334
|
+
if xl_skiprows is not None:
|
|
335
|
+
xl_kwargs["skiprows"] = xl_skiprows
|
|
336
|
+
df = pd.read_excel(path, sheet_name=sheet, **xl_kwargs)
|
|
337
|
+
if isinstance(df, dict):
|
|
338
|
+
raise CleanFrameError("sheet must select a single sheet (a name or index).")
|
|
339
|
+
return df
|
|
340
|
+
if suffix == ".parquet":
|
|
341
|
+
df = pd.read_parquet(path, columns=list(columns) if columns else None, **kwargs)
|
|
342
|
+
return _apply_row_slice(df, nrows, skiprows)
|
|
343
|
+
if suffix == ".json":
|
|
344
|
+
kwargs.setdefault("encoding", "utf-8")
|
|
345
|
+
df = pd.read_json(path, **kwargs)
|
|
346
|
+
if columns is not None:
|
|
347
|
+
missing = [c for c in columns if c not in df.columns]
|
|
348
|
+
if missing:
|
|
349
|
+
raise CleanFrameError(
|
|
350
|
+
f"Requested column(s) not found in {path.name}: {missing}. "
|
|
351
|
+
f"Available: {list(df.columns)}."
|
|
352
|
+
)
|
|
353
|
+
df = df[list(columns)]
|
|
354
|
+
return _apply_row_slice(df, nrows, skiprows)
|
|
355
|
+
# CSV family (.csv/.txt/.tsv/.dat and extension-less files).
|
|
356
|
+
kwargs.setdefault("encoding", _CSV_READ_ENCODING)
|
|
357
|
+
if suffix == ".tsv":
|
|
358
|
+
kwargs.setdefault("sep", "\t")
|
|
359
|
+
if text:
|
|
360
|
+
for key, value in _text_kwargs().items():
|
|
361
|
+
kwargs.setdefault(key, value)
|
|
362
|
+
_check_csv_header(path, kwargs["encoding"], kwargs.get("sep"), blank_lines)
|
|
363
|
+
# index_col=False: a row with one field too many (a stray trailing delimiter)
|
|
364
|
+
# otherwise becomes the index and shifts every column one place left.
|
|
365
|
+
kwargs.setdefault("index_col", False)
|
|
366
|
+
csv_skiprows = _pandas_skiprows(skiprows, blank_lines)
|
|
367
|
+
if csv_skiprows is not None:
|
|
368
|
+
kwargs["skiprows"] = csv_skiprows
|
|
369
|
+
if columns is not None:
|
|
370
|
+
header_kwargs = {
|
|
371
|
+
k: v for k, v in kwargs.items() if k in ("encoding", "sep", "skiprows", "index_col")
|
|
372
|
+
}
|
|
373
|
+
available = list(pd.read_csv(path, nrows=0, **header_kwargs).columns)
|
|
374
|
+
missing = [c for c in columns if c not in available]
|
|
375
|
+
if missing:
|
|
376
|
+
raise CleanFrameError(
|
|
377
|
+
f"Requested column(s) not found in {path.name}: {missing}. "
|
|
378
|
+
f"Available: {available}."
|
|
379
|
+
)
|
|
380
|
+
kwargs["usecols"] = list(columns)
|
|
381
|
+
if nrows is not None:
|
|
382
|
+
if not isinstance(nrows, int) or isinstance(nrows, bool) or nrows < 0:
|
|
383
|
+
raise CleanFrameError(f"nrows must be 0 or more, got {nrows!r}.")
|
|
384
|
+
kwargs["nrows"] = nrows
|
|
385
|
+
return pd.read_csv(path, **kwargs)
|
|
386
|
+
except ImportError as exc: # pragma: no cover - optional engine missing
|
|
387
|
+
hint = "Try `pip install cleanframe-engine[excel]` for Excel support."
|
|
388
|
+
if suffix == ".parquet":
|
|
389
|
+
hint = "Try `pip install cleanframe-engine[parquet]` (pyarrow) for Parquet support."
|
|
390
|
+
elif suffix == ".xls":
|
|
391
|
+
hint = "Legacy .xls needs xlrd: `pip install xlrd` (.xlsx needs only openpyxl)."
|
|
392
|
+
raise CleanFrameError(
|
|
393
|
+
f"Reading {suffix} requires an extra engine ({_detail(exc)}). {hint}"
|
|
394
|
+
) from exc
|
|
395
|
+
except CleanFrameError:
|
|
396
|
+
raise
|
|
397
|
+
except UnicodeDecodeError as exc:
|
|
398
|
+
raise CleanFrameError(
|
|
399
|
+
f"Could not decode {path.name} as UTF-8 ({exc}). It may be saved as "
|
|
400
|
+
"Latin-1 / Windows-1252 / UTF-16 (common for Excel 'Save as CSV' on "
|
|
401
|
+
"Windows). Read it with cleanframe.read_frame(..., encoding='cp1252'), "
|
|
402
|
+
"let `clean`/`report` detect the encoding for you, or record it in the "
|
|
403
|
+
"recipe's read: section (encoding: cp1252) for replay."
|
|
404
|
+
) from exc
|
|
405
|
+
except (pd.errors.ParserError, pd.errors.EmptyDataError) as exc:
|
|
406
|
+
raise CleanFrameError(
|
|
407
|
+
f"Could not parse {path.name}: {type(exc).__name__}: {_detail(exc)}. "
|
|
408
|
+
"Check the delimiter/quoting, that the file matches its extension, and "
|
|
409
|
+
"that it is not truncated."
|
|
410
|
+
) from exc
|
|
411
|
+
except OSError as exc:
|
|
412
|
+
raise CleanFrameError(f"Could not read {path}: {_detail(exc)}.") from exc
|
|
413
|
+
except Exception as exc: # noqa: BLE001 - IO boundary: surface any failure cleanly
|
|
414
|
+
raise CleanFrameError(
|
|
415
|
+
f"Could not read {path.name}: {type(exc).__name__}: {_detail(exc)}."
|
|
416
|
+
) from exc
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def _write_dispatch(
|
|
420
|
+
df: pd.DataFrame, target: Path, suffix: str, sanitize_csv: bool, kwargs: dict
|
|
421
|
+
) -> None:
|
|
422
|
+
if suffix == ".parquet":
|
|
423
|
+
df.to_parquet(target, index=False, **kwargs)
|
|
424
|
+
return
|
|
425
|
+
if suffix == ".json":
|
|
426
|
+
kwargs.setdefault("force_ascii", False)
|
|
427
|
+
df.to_json(target, orient="records", indent=2, **kwargs)
|
|
428
|
+
return
|
|
429
|
+
if suffix in (".xlsx", ".xlsm"):
|
|
430
|
+
out = sanitize_dataframe_for_spreadsheet(df) if sanitize_csv else df
|
|
431
|
+
kwargs.setdefault("engine", "openpyxl")
|
|
432
|
+
out.to_excel(target, index=False, **kwargs)
|
|
433
|
+
return
|
|
434
|
+
out = sanitize_dataframe_for_csv(df) if sanitize_csv else df
|
|
435
|
+
kwargs.setdefault("encoding", _CSV_WRITE_ENCODING)
|
|
436
|
+
kwargs.setdefault("lineterminator", "\n")
|
|
437
|
+
kwargs.setdefault("sep", "\t" if suffix == ".tsv" else ",")
|
|
438
|
+
out.to_csv(target, index=False, **kwargs)
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
def write_frame(
|
|
442
|
+
df: pd.DataFrame,
|
|
443
|
+
path: str | Path,
|
|
444
|
+
*,
|
|
445
|
+
sanitize_csv: bool = True,
|
|
446
|
+
source: str | Path | None = None,
|
|
447
|
+
overwrite: bool = False,
|
|
448
|
+
**kwargs,
|
|
449
|
+
) -> Path:
|
|
450
|
+
"""Write a dataframe, dispatching on file extension. Never writes the index.
|
|
451
|
+
|
|
452
|
+
The frame is written to a temporary sibling and moved into place, so a failure
|
|
453
|
+
part-way through leaves the previous file intact rather than a truncated one.
|
|
454
|
+
|
|
455
|
+
Parameters
|
|
456
|
+
----------
|
|
457
|
+
sanitize_csv:
|
|
458
|
+
When ``True`` (default), string cells and headers that look like spreadsheet
|
|
459
|
+
formulas (leading ``=``, ``@``, or ``+``/``-`` followed by a non-number) are
|
|
460
|
+
escaped before CSV/TSV/Excel export. Set ``False`` only when you intentionally
|
|
461
|
+
need raw formula cells.
|
|
462
|
+
source / overwrite:
|
|
463
|
+
``source`` is the path the data was read from; writing back over it needs
|
|
464
|
+
``overwrite=True`` so the original is never destroyed by accident.
|
|
465
|
+
"""
|
|
466
|
+
if not isinstance(df, pd.DataFrame):
|
|
467
|
+
raise CleanFrameError(f"write_frame expects a DataFrame, got {type(df).__name__}.")
|
|
468
|
+
path = check_output_target(path, source, overwrite=overwrite)
|
|
469
|
+
suffix = path.suffix.lower()
|
|
470
|
+
if suffix == ".xls":
|
|
471
|
+
raise OutputError(
|
|
472
|
+
"Writing legacy .xls is not supported: pandas emits .xlsx bytes, which "
|
|
473
|
+
"Excel refuses under an .xls name. Write .xlsx instead."
|
|
474
|
+
)
|
|
475
|
+
tmp = path.with_name(path.name + ".cf-tmp")
|
|
476
|
+
try:
|
|
477
|
+
_write_dispatch(df, tmp, suffix, sanitize_csv, kwargs)
|
|
478
|
+
os.replace(tmp, path)
|
|
479
|
+
except ImportError as exc: # pragma: no cover - optional engine missing
|
|
480
|
+
tmp.unlink(missing_ok=True)
|
|
481
|
+
extra = "parquet" if suffix == ".parquet" else "excel"
|
|
482
|
+
raise OutputError(
|
|
483
|
+
f"Writing {suffix} requires an extra engine ({_detail(exc)}). "
|
|
484
|
+
f"Try `pip install cleanframe-engine[{extra}]`."
|
|
485
|
+
) from exc
|
|
486
|
+
except CleanFrameError:
|
|
487
|
+
tmp.unlink(missing_ok=True)
|
|
488
|
+
raise
|
|
489
|
+
except OSError as exc:
|
|
490
|
+
tmp.unlink(missing_ok=True)
|
|
491
|
+
raise OutputError(f"Could not write {path}: {_detail(exc)}.") from exc
|
|
492
|
+
except Exception as exc: # noqa: BLE001 - IO boundary
|
|
493
|
+
tmp.unlink(missing_ok=True)
|
|
494
|
+
raise OutputError(
|
|
495
|
+
f"Could not write {path.name}: {type(exc).__name__}: {_detail(exc)}."
|
|
496
|
+
) from exc
|
|
497
|
+
return path
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
__all__ = [
|
|
501
|
+
"read_frame",
|
|
502
|
+
"write_frame",
|
|
503
|
+
"excel_sheet_names",
|
|
504
|
+
"inference_losses",
|
|
505
|
+
"compare_for_losses",
|
|
506
|
+
]
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""Detector registry and the built-in detector suite.
|
|
2
|
+
|
|
3
|
+
Importing this package registers every built-in detector (importing the modules
|
|
4
|
+
runs their ``@detector`` decorators). Third-party detectors register themselves the
|
|
5
|
+
same way — by being imported — so ``import cleanframe`` wires up the built-ins and
|
|
6
|
+
users add their own with ``@cf.detector(...)``.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
# Importing each module triggers its @detector registrations. Ordered by the
|
|
12
|
+
# priority they run at, for readability only.
|
|
13
|
+
from . import (
|
|
14
|
+
categories, # noqa: E402,F401 priority 50
|
|
15
|
+
contacts, # noqa: E402,F401 priority 45
|
|
16
|
+
currency, # noqa: E402,F401 priority 45
|
|
17
|
+
dates, # noqa: E402,F401 priority 40
|
|
18
|
+
dedup, # noqa: E402,F401 priority 80
|
|
19
|
+
nulls, # noqa: E402,F401 priority 20
|
|
20
|
+
outliers, # noqa: E402,F401 priority 70
|
|
21
|
+
schema_mapping, # noqa: E402,F401 priority 5
|
|
22
|
+
text, # noqa: E402,F401 priority 10, 60
|
|
23
|
+
units, # noqa: E402,F401 priority 46
|
|
24
|
+
)
|
|
25
|
+
from .base import (
|
|
26
|
+
DETECTOR_REGISTRY,
|
|
27
|
+
DetectorContext,
|
|
28
|
+
DetectorSpec,
|
|
29
|
+
detector,
|
|
30
|
+
list_detectors,
|
|
31
|
+
run_detectors,
|
|
32
|
+
unregister_detector,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
__all__ = [
|
|
36
|
+
"DETECTOR_REGISTRY",
|
|
37
|
+
"DetectorContext",
|
|
38
|
+
"DetectorSpec",
|
|
39
|
+
"detector",
|
|
40
|
+
"list_detectors",
|
|
41
|
+
"run_detectors",
|
|
42
|
+
"unregister_detector",
|
|
43
|
+
]
|