cleanframe-engine 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanframe/__init__.py +169 -0
- cleanframe/__main__.py +5 -0
- cleanframe/_util.py +438 -0
- cleanframe/_version.py +1 -0
- cleanframe/api.py +559 -0
- cleanframe/cli.py +688 -0
- cleanframe/codegen.py +666 -0
- cleanframe/dataio.py +506 -0
- cleanframe/detectors/__init__.py +43 -0
- cleanframe/detectors/base.py +221 -0
- cleanframe/detectors/categories.py +199 -0
- cleanframe/detectors/contacts.py +105 -0
- cleanframe/detectors/currency.py +112 -0
- cleanframe/detectors/dates.py +204 -0
- cleanframe/detectors/dedup.py +150 -0
- cleanframe/detectors/nulls.py +109 -0
- cleanframe/detectors/outliers.py +73 -0
- cleanframe/detectors/schema_mapping.py +125 -0
- cleanframe/detectors/text.py +105 -0
- cleanframe/detectors/units.py +86 -0
- cleanframe/diff.py +369 -0
- cleanframe/drift.py +283 -0
- cleanframe/errors.py +66 -0
- cleanframe/executor.py +229 -0
- cleanframe/fingerprint.py +83 -0
- cleanframe/issues.py +186 -0
- cleanframe/llm.py +811 -0
- cleanframe/ops.py +1245 -0
- cleanframe/planner.py +353 -0
- cleanframe/profile.py +413 -0
- cleanframe/py.typed +1 -0
- cleanframe/quality.py +81 -0
- cleanframe/readfix.py +160 -0
- cleanframe/recipe.py +398 -0
- cleanframe/report.py +345 -0
- cleanframe/result.py +144 -0
- cleanframe/schema.py +259 -0
- cleanframe/streaming.py +354 -0
- cleanframe/types.py +119 -0
- cleanframe/validate.py +363 -0
- cleanframe/workbook.py +370 -0
- cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
- cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
- cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
- cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
- cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
cleanframe/api.py
ADDED
|
@@ -0,0 +1,559 @@
|
|
|
1
|
+
"""The high-level API: ``clean``, ``report``, ``apply_recipe``, ``suggest_update``.
|
|
2
|
+
|
|
3
|
+
These stitch the pipeline stages (profile → detect → plan → execute) into the few
|
|
4
|
+
calls most users ever touch. Every function accepts either a DataFrame or a path,
|
|
5
|
+
and returns rich result objects rather than bare frames so the recipe, diff, and
|
|
6
|
+
report are always one attribute away.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import warnings
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
import pandas as pd
|
|
16
|
+
|
|
17
|
+
from .dataio import read_frame
|
|
18
|
+
from .detectors import run_detectors
|
|
19
|
+
from .drift import DriftReport, detect_drift
|
|
20
|
+
from .errors import CleanFrameError, CleanFrameWarning, DriftError
|
|
21
|
+
from .executor import execute
|
|
22
|
+
from .issues import Issues
|
|
23
|
+
from .planner import Planner, RulesPlanner
|
|
24
|
+
from .profile import profile_dataframe
|
|
25
|
+
from .quality import quality_score
|
|
26
|
+
from .recipe import Recipe
|
|
27
|
+
from .result import CleanResult, Report, build_profile_report_object
|
|
28
|
+
from .schema import Schema
|
|
29
|
+
from .schema import infer_schema as _infer_schema
|
|
30
|
+
from .types import Mode
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
# ---------------------------------------------------------------------------
|
|
34
|
+
# input coercion
|
|
35
|
+
# ---------------------------------------------------------------------------
|
|
36
|
+
def _read_binding(sheet, columns, nrows, skiprows) -> dict[str, Any]: # noqa: D401
|
|
37
|
+
"""Collect the non-default read/selection options into a dict (empty if none)."""
|
|
38
|
+
binding = {"sheet": sheet, "columns": columns, "nrows": nrows, "skiprows": skiprows}
|
|
39
|
+
return {k: v for k, v in binding.items() if v is not None}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _read_input(
|
|
43
|
+
data, source, *, sheet, columns, nrows, skiprows, correct_format, text=False,
|
|
44
|
+
sep=None, encoding=None, warn=False,
|
|
45
|
+
) -> tuple[pd.DataFrame, str | None, dict[str, Any], list[str]]:
|
|
46
|
+
"""Read ``data`` with optional CSV format auto-correction.
|
|
47
|
+
|
|
48
|
+
Returns ``(df, source, read_binding, notes)`` where ``read_binding`` is the full
|
|
49
|
+
selection+format binding to record in a recipe's ``read:`` section.
|
|
50
|
+
"""
|
|
51
|
+
fmt_kwargs: dict[str, Any] = {}
|
|
52
|
+
notes: list[str] = []
|
|
53
|
+
if correct_format and isinstance(data, (str, Path)):
|
|
54
|
+
from .readfix import detect_csv_options, is_csv_family
|
|
55
|
+
|
|
56
|
+
# Only sniff a file that is actually there: read_frame owns the
|
|
57
|
+
# missing-file / directory / empty-file messages.
|
|
58
|
+
if is_csv_family(data) and Path(data).is_file() and Path(data).stat().st_size:
|
|
59
|
+
_opts, report_ = detect_csv_options(data) # raises on ambiguous delimiter
|
|
60
|
+
fmt_kwargs = report_.as_read_binding() # {encoding?, sep?, blank_lines?}
|
|
61
|
+
notes = report_.notes
|
|
62
|
+
if warn and notes:
|
|
63
|
+
warnings.warn(
|
|
64
|
+
"CleanFrame: read-time format correction — " + "; ".join(notes),
|
|
65
|
+
CleanFrameWarning,
|
|
66
|
+
stacklevel=3,
|
|
67
|
+
)
|
|
68
|
+
# An explicit sep/encoding always wins over what detection guessed.
|
|
69
|
+
fmt_kwargs.update({k: v for k, v in (("sep", sep), ("encoding", encoding)) if v is not None})
|
|
70
|
+
df, source = _as_frame(
|
|
71
|
+
data, source, sheet=sheet, columns=columns, nrows=nrows, skiprows=skiprows,
|
|
72
|
+
text=text, **fmt_kwargs,
|
|
73
|
+
)
|
|
74
|
+
if correct_format and not text and isinstance(data, (str, Path)):
|
|
75
|
+
from .dataio import inference_losses
|
|
76
|
+
|
|
77
|
+
if Path(data).is_file() and Path(data).suffix.lower() not in (".parquet", ".json"):
|
|
78
|
+
losses = inference_losses(
|
|
79
|
+
data, df, sheet=sheet, columns=columns, skiprows=skiprows, **fmt_kwargs
|
|
80
|
+
)
|
|
81
|
+
if losses:
|
|
82
|
+
detail = "; ".join(f"{col}: {why}" for col, why in sorted(losses.items()))
|
|
83
|
+
notes = [*notes, f"type inference changed values on read — {detail}"]
|
|
84
|
+
if warn:
|
|
85
|
+
warnings.warn(
|
|
86
|
+
"CleanFrame: pandas type inference changed values while reading "
|
|
87
|
+
f"({detail}). Pass text=True to read every field verbatim.",
|
|
88
|
+
CleanFrameWarning,
|
|
89
|
+
stacklevel=3,
|
|
90
|
+
)
|
|
91
|
+
binding = _read_binding(sheet, columns, nrows, skiprows)
|
|
92
|
+
if text:
|
|
93
|
+
binding["text"] = True
|
|
94
|
+
binding.update(fmt_kwargs)
|
|
95
|
+
return df, source, binding, notes
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _as_frame(
|
|
99
|
+
data: pd.DataFrame | str | Path,
|
|
100
|
+
source: str | None,
|
|
101
|
+
*,
|
|
102
|
+
sheet=None,
|
|
103
|
+
columns=None,
|
|
104
|
+
nrows=None,
|
|
105
|
+
skiprows=None,
|
|
106
|
+
text: bool = False,
|
|
107
|
+
**read_kwargs,
|
|
108
|
+
) -> tuple[pd.DataFrame, str | None]:
|
|
109
|
+
from ._util import ensure_string_columns
|
|
110
|
+
|
|
111
|
+
if isinstance(data, pd.DataFrame):
|
|
112
|
+
if sheet is not None or nrows is not None or skiprows is not None:
|
|
113
|
+
raise CleanFrameError(
|
|
114
|
+
"sheet=/nrows=/skiprows= selection applies to file inputs only, "
|
|
115
|
+
"not an in-memory DataFrame. Slice the DataFrame yourself first."
|
|
116
|
+
)
|
|
117
|
+
df = data
|
|
118
|
+
if columns is not None: # a column projection is well-defined on a DataFrame
|
|
119
|
+
missing = [c for c in columns if c not in df.columns]
|
|
120
|
+
if missing:
|
|
121
|
+
raise CleanFrameError(
|
|
122
|
+
f"Requested column(s) not found: {missing}. "
|
|
123
|
+
f"Available: {list(df.columns)}."
|
|
124
|
+
)
|
|
125
|
+
df = df[list(columns)]
|
|
126
|
+
return ensure_string_columns(df), source # read_kwargs (encoding/sep) are file-only
|
|
127
|
+
if isinstance(data, (str, Path)):
|
|
128
|
+
df = read_frame(
|
|
129
|
+
data, sheet=sheet, columns=columns, nrows=nrows, skiprows=skiprows,
|
|
130
|
+
text=text, **read_kwargs,
|
|
131
|
+
)
|
|
132
|
+
return ensure_string_columns(df), source or str(data)
|
|
133
|
+
raise CleanFrameError(f"Expected a DataFrame or file path, got {type(data).__name__}.")
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _resolve_schema(schema: Any) -> Schema | None:
|
|
137
|
+
if schema is None or isinstance(schema, Schema):
|
|
138
|
+
return schema
|
|
139
|
+
if isinstance(schema, (str, Path)):
|
|
140
|
+
return Schema.load(schema)
|
|
141
|
+
if isinstance(schema, dict):
|
|
142
|
+
return Schema.from_dict(schema)
|
|
143
|
+
raise CleanFrameError(f"Unsupported schema type: {type(schema).__name__}.")
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _resolve_recipe(recipe: Any) -> Recipe:
|
|
147
|
+
if isinstance(recipe, Recipe):
|
|
148
|
+
return recipe
|
|
149
|
+
if isinstance(recipe, (str, Path)):
|
|
150
|
+
return Recipe.load(recipe)
|
|
151
|
+
if isinstance(recipe, dict):
|
|
152
|
+
return Recipe.from_dict(recipe)
|
|
153
|
+
raise CleanFrameError(f"Unsupported recipe type: {type(recipe).__name__}.")
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _resolve_planner(
|
|
157
|
+
planner: Planner | None,
|
|
158
|
+
llm: Any,
|
|
159
|
+
llm_exposure: str,
|
|
160
|
+
max_tokens_budget: int | None,
|
|
161
|
+
llm_fallback: bool = True,
|
|
162
|
+
) -> Planner:
|
|
163
|
+
if planner is not None:
|
|
164
|
+
if not hasattr(planner, "plan"):
|
|
165
|
+
raise CleanFrameError(
|
|
166
|
+
f"planner must have a .plan() method, got {type(planner).__name__}."
|
|
167
|
+
)
|
|
168
|
+
return planner
|
|
169
|
+
if llm is None:
|
|
170
|
+
return RulesPlanner()
|
|
171
|
+
from .types import LLMExposure
|
|
172
|
+
|
|
173
|
+
if LLMExposure(str(llm_exposure)) is LLMExposure.NONE:
|
|
174
|
+
# "none" means nothing leaves the machine, so there is no request to make.
|
|
175
|
+
warnings.warn(
|
|
176
|
+
"CleanFrame: llm_exposure='none' keeps everything local, so the LLM was not "
|
|
177
|
+
"called — planning with deterministic rules instead.",
|
|
178
|
+
CleanFrameWarning,
|
|
179
|
+
stacklevel=3,
|
|
180
|
+
)
|
|
181
|
+
return RulesPlanner()
|
|
182
|
+
from .llm import LLMPlanner, get_client
|
|
183
|
+
|
|
184
|
+
if isinstance(llm, str):
|
|
185
|
+
client = get_client(llm)
|
|
186
|
+
elif hasattr(llm, "complete"):
|
|
187
|
+
client = llm
|
|
188
|
+
else:
|
|
189
|
+
raise CleanFrameError(
|
|
190
|
+
"llm must be a 'provider/model' string or an object with a .complete() method."
|
|
191
|
+
)
|
|
192
|
+
return LLMPlanner(
|
|
193
|
+
client,
|
|
194
|
+
exposure=llm_exposure,
|
|
195
|
+
max_tokens_budget=max_tokens_budget,
|
|
196
|
+
fallback="rules" if llm_fallback else None,
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
_ON_DRIFT = ("error", "warn", "ignore")
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _check_on_drift(on_drift: Any) -> str:
|
|
204
|
+
if on_drift not in _ON_DRIFT:
|
|
205
|
+
raise CleanFrameError(
|
|
206
|
+
f"on_drift must be one of {list(_ON_DRIFT)}, got {on_drift!r}. A typo here "
|
|
207
|
+
"would silently disable the drift guard."
|
|
208
|
+
)
|
|
209
|
+
return on_drift
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _check_options(options: Any) -> dict[str, Any]:
|
|
213
|
+
if options is None:
|
|
214
|
+
return {}
|
|
215
|
+
if not isinstance(options, dict):
|
|
216
|
+
raise CleanFrameError(
|
|
217
|
+
f"options must be a mapping of detector knobs, got {type(options).__name__}."
|
|
218
|
+
)
|
|
219
|
+
cap = options.get("max_diff_changes", 0)
|
|
220
|
+
if cap is not None and (not isinstance(cap, int) or isinstance(cap, bool) or cap < 0):
|
|
221
|
+
raise CleanFrameError(
|
|
222
|
+
f"options['max_diff_changes'] must be a non-negative int or None, got {cap!r}."
|
|
223
|
+
)
|
|
224
|
+
return dict(options)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
# ---------------------------------------------------------------------------
|
|
228
|
+
# clean
|
|
229
|
+
# ---------------------------------------------------------------------------
|
|
230
|
+
def clean(
|
|
231
|
+
data: pd.DataFrame | str | Path,
|
|
232
|
+
*,
|
|
233
|
+
target_schema: Any = None,
|
|
234
|
+
schema: Any = None,
|
|
235
|
+
llm: Any = None,
|
|
236
|
+
mode: Mode | str = Mode.REVIEW,
|
|
237
|
+
options: dict[str, Any] | None = None,
|
|
238
|
+
planner: Planner | None = None,
|
|
239
|
+
max_tokens_budget: int | None = None,
|
|
240
|
+
llm_exposure: str = "metadata",
|
|
241
|
+
source: str | None = None,
|
|
242
|
+
sheet: str | int | None = None,
|
|
243
|
+
columns: list[str] | None = None,
|
|
244
|
+
nrows: int | None = None,
|
|
245
|
+
skiprows: int | list[int] | None = None,
|
|
246
|
+
correct_format: bool = True,
|
|
247
|
+
text: bool = False,
|
|
248
|
+
sep: str | None = None,
|
|
249
|
+
encoding: str | None = None,
|
|
250
|
+
llm_fallback: bool = True,
|
|
251
|
+
) -> CleanResult:
|
|
252
|
+
"""Profile, plan, and clean ``data`` — the main entry point.
|
|
253
|
+
|
|
254
|
+
``sheet``/``columns``/``nrows``/``skiprows`` select part of a file (see
|
|
255
|
+
:func:`read_frame`); the selection is recorded in the recipe's ``read:`` section
|
|
256
|
+
so :func:`apply_recipe` re-reads the same slice. For a multi-sheet workbook, pass
|
|
257
|
+
``sheet=`` or use :func:`clean_workbook` to clean every sheet.
|
|
258
|
+
|
|
259
|
+
``correct_format`` (default ``True``) auto-detects a CSV-family file's encoding
|
|
260
|
+
and delimiter at read time (e.g. a ``;``-separated or cp1252 file), warns, and
|
|
261
|
+
pins the choice into the recipe's ``read:`` section for deterministic replay. An
|
|
262
|
+
ambiguous delimiter raises rather than guessing. It also reports values that
|
|
263
|
+
pandas' type inference changed while reading (a leading-zero ZIP, an ``NA``
|
|
264
|
+
token); pass ``text=True`` to read every field verbatim instead.
|
|
265
|
+
|
|
266
|
+
``llm_fallback`` (default ``True``) keeps the documented behaviour of degrading
|
|
267
|
+
to the rules planner when an LLM call fails. Pass ``False`` to make such a
|
|
268
|
+
failure raise instead of quietly producing a rules-only recipe.
|
|
269
|
+
|
|
270
|
+
The returned frame is indexed 0..n-1: replaying a recipe re-keys rows to the
|
|
271
|
+
stable positional row ids the diff and quarantine refer to.
|
|
272
|
+
|
|
273
|
+
Parameters
|
|
274
|
+
----------
|
|
275
|
+
data:
|
|
276
|
+
A DataFrame or a path to a CSV/Excel/Parquet/JSON file.
|
|
277
|
+
target_schema / schema:
|
|
278
|
+
Optional target :class:`~cleanframe.schema.Schema`, path, or dict. Drives
|
|
279
|
+
column mapping and validation synthesis.
|
|
280
|
+
llm:
|
|
281
|
+
``None`` (rules-only, default), a ``"provider/model"`` string, or any object
|
|
282
|
+
with a ``.complete()`` method. The LLM only writes the recipe; it never sees
|
|
283
|
+
raw data (see :mod:`cleanframe.llm`).
|
|
284
|
+
mode:
|
|
285
|
+
``"review"`` (default), ``"auto"``, or ``"strict"``.
|
|
286
|
+
max_tokens_budget:
|
|
287
|
+
Hard cap that aborts LLM planning before it gets expensive.
|
|
288
|
+
|
|
289
|
+
Returns
|
|
290
|
+
-------
|
|
291
|
+
CleanResult
|
|
292
|
+
Cleaned dataframe plus recipe, diff, quarantine, issues, and report.
|
|
293
|
+
"""
|
|
294
|
+
df, source, read_binding, read_notes = _read_input(
|
|
295
|
+
data, source, sheet=sheet, columns=columns, nrows=nrows, skiprows=skiprows,
|
|
296
|
+
correct_format=correct_format, text=text, sep=sep, encoding=encoding, warn=True,
|
|
297
|
+
)
|
|
298
|
+
schema_obj = _resolve_schema(target_schema if target_schema is not None else schema)
|
|
299
|
+
options = _check_options(options)
|
|
300
|
+
|
|
301
|
+
profile = profile_dataframe(df)
|
|
302
|
+
issues = run_detectors(df, profile=profile, schema=schema_obj, options=options)
|
|
303
|
+
the_planner = _resolve_planner(
|
|
304
|
+
planner, llm, llm_exposure, max_tokens_budget, llm_fallback=llm_fallback
|
|
305
|
+
)
|
|
306
|
+
recipe = the_planner.plan(df, profile, issues, schema=schema_obj, mode=mode, options=options)
|
|
307
|
+
|
|
308
|
+
# Record the read/selection + format binding so `apply` re-reads identically.
|
|
309
|
+
if read_binding:
|
|
310
|
+
recipe.read = {**(recipe.read or {}), **read_binding}
|
|
311
|
+
|
|
312
|
+
from ._util import DEFAULT_MAX_DIFF_CHANGES
|
|
313
|
+
|
|
314
|
+
max_diff = options.get("max_diff_changes", DEFAULT_MAX_DIFF_CHANGES)
|
|
315
|
+
exec_result = execute(recipe, df, mode=mode, max_diff_changes=max_diff)
|
|
316
|
+
quality = quality_score(profile, issues)
|
|
317
|
+
|
|
318
|
+
return CleanResult(
|
|
319
|
+
dataframe=exec_result.dataframe,
|
|
320
|
+
recipe=recipe,
|
|
321
|
+
diff=exec_result.diff,
|
|
322
|
+
quarantine=exec_result.quarantine,
|
|
323
|
+
issues=issues,
|
|
324
|
+
profile=profile,
|
|
325
|
+
validation_results=exec_result.validation_results,
|
|
326
|
+
quality=quality,
|
|
327
|
+
source=source,
|
|
328
|
+
log=[*(f"read-fix: {n}" for n in read_notes), *exec_result.log],
|
|
329
|
+
)
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
# ---------------------------------------------------------------------------
|
|
333
|
+
# report
|
|
334
|
+
# ---------------------------------------------------------------------------
|
|
335
|
+
def report(
|
|
336
|
+
data: pd.DataFrame | str | Path,
|
|
337
|
+
*,
|
|
338
|
+
schema: Any = None,
|
|
339
|
+
options: dict[str, Any] | None = None,
|
|
340
|
+
source: str | None = None,
|
|
341
|
+
sheet: str | int | None = None,
|
|
342
|
+
columns: list[str] | None = None,
|
|
343
|
+
nrows: int | None = None,
|
|
344
|
+
skiprows: int | list[int] | None = None,
|
|
345
|
+
correct_format: bool = True,
|
|
346
|
+
text: bool = False,
|
|
347
|
+
sep: str | None = None,
|
|
348
|
+
encoding: str | None = None,
|
|
349
|
+
) -> Report:
|
|
350
|
+
"""Profile ``data`` and return an HTML :class:`~cleanframe.result.Report` (no changes made)."""
|
|
351
|
+
df, source, _, _ = _read_input(
|
|
352
|
+
data, source, sheet=sheet, columns=columns, nrows=nrows, skiprows=skiprows,
|
|
353
|
+
correct_format=correct_format, text=text, sep=sep, encoding=encoding, warn=True,
|
|
354
|
+
)
|
|
355
|
+
schema_obj = _resolve_schema(schema)
|
|
356
|
+
profile = profile_dataframe(df)
|
|
357
|
+
issues = run_detectors(df, profile=profile, schema=schema_obj, options=_check_options(options))
|
|
358
|
+
quality = quality_score(profile, issues)
|
|
359
|
+
return build_profile_report_object(profile, issues, source=source, quality=quality)
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
# ---------------------------------------------------------------------------
|
|
363
|
+
# apply (replay)
|
|
364
|
+
# ---------------------------------------------------------------------------
|
|
365
|
+
def apply_recipe(
|
|
366
|
+
data: pd.DataFrame | str | Path,
|
|
367
|
+
recipe: Recipe | str | Path | dict,
|
|
368
|
+
*,
|
|
369
|
+
mode: Mode | str = Mode.REVIEW,
|
|
370
|
+
check_drift: bool = True,
|
|
371
|
+
on_drift: str = "error",
|
|
372
|
+
source: str | None = None,
|
|
373
|
+
sheet: str | int | None = None,
|
|
374
|
+
columns: list[str] | None = None,
|
|
375
|
+
nrows: int | None = None,
|
|
376
|
+
skiprows: int | list[int] | None = None,
|
|
377
|
+
text: bool = False,
|
|
378
|
+
sep: str | None = None,
|
|
379
|
+
encoding: str | None = None,
|
|
380
|
+
) -> CleanResult:
|
|
381
|
+
"""Replay a saved recipe on new data — deterministic, no LLM.
|
|
382
|
+
|
|
383
|
+
If ``check_drift`` and the incoming schema drifted, ``on_drift`` decides:
|
|
384
|
+
``"error"`` (default — raise :class:`~cleanframe.errors.DriftError` so nothing
|
|
385
|
+
is silently corrupted), ``"warn"`` (attach the report and warn, then continue),
|
|
386
|
+
or ``"ignore"``. ``strict`` mode always raises on drift.
|
|
387
|
+
|
|
388
|
+
The selection used to plan the recipe (its ``read:`` section) is re-applied when
|
|
389
|
+
``data`` is a path, unless overridden by an explicit ``sheet``/``columns``/etc.
|
|
390
|
+
"""
|
|
391
|
+
recipe_obj = _resolve_recipe(recipe)
|
|
392
|
+
mode = Mode.coerce(mode)
|
|
393
|
+
_check_on_drift(on_drift)
|
|
394
|
+
|
|
395
|
+
# Precedence: explicit call args > recipe-recorded read binding > whole file.
|
|
396
|
+
call = _read_binding(sheet, columns, nrows, skiprows)
|
|
397
|
+
call.update({k: v for k, v in (("sep", sep), ("encoding", encoding)) if v is not None})
|
|
398
|
+
if text:
|
|
399
|
+
call["text"] = True
|
|
400
|
+
effective = {**(recipe_obj.read or {}), **call}
|
|
401
|
+
if isinstance(data, pd.DataFrame) and effective:
|
|
402
|
+
unreplayable = {k for k in effective if k in ("sheet", "nrows", "skiprows")}
|
|
403
|
+
if unreplayable and not call:
|
|
404
|
+
warnings.warn(
|
|
405
|
+
f"CleanFrame: recipe's recorded read binding {sorted(unreplayable)} cannot "
|
|
406
|
+
"replay against an in-memory DataFrame; ignoring it.",
|
|
407
|
+
CleanFrameWarning,
|
|
408
|
+
stacklevel=2,
|
|
409
|
+
)
|
|
410
|
+
effective = {k: v for k, v in effective.items() if k == "columns"}
|
|
411
|
+
df, source = _as_frame(data, source, **effective)
|
|
412
|
+
|
|
413
|
+
drift: DriftReport | None = None
|
|
414
|
+
if check_drift and not recipe_obj.source_fingerprint:
|
|
415
|
+
warnings.warn(
|
|
416
|
+
"CleanFrame: recipe has no source_fingerprint — drift check skipped. "
|
|
417
|
+
"Re-plan or stamp a fingerprint for production replay.",
|
|
418
|
+
CleanFrameWarning,
|
|
419
|
+
stacklevel=2,
|
|
420
|
+
)
|
|
421
|
+
if check_drift and recipe_obj.source_fingerprint:
|
|
422
|
+
drift = detect_drift(df, recipe_obj, source=source)
|
|
423
|
+
if drift.has_drift:
|
|
424
|
+
if on_drift == "error" or mode is Mode.STRICT:
|
|
425
|
+
raise DriftError(drift.render(), report=drift)
|
|
426
|
+
if on_drift == "warn":
|
|
427
|
+
warnings.warn("CleanFrame: " + drift.render(), CleanFrameWarning, stacklevel=2)
|
|
428
|
+
|
|
429
|
+
exec_result = execute(recipe_obj, df, mode=mode)
|
|
430
|
+
log = list(exec_result.log)
|
|
431
|
+
if drift is not None and drift.has_drift:
|
|
432
|
+
log.insert(0, "drift detected: " + "; ".join(f.message for f in drift.findings))
|
|
433
|
+
|
|
434
|
+
return CleanResult(
|
|
435
|
+
dataframe=exec_result.dataframe,
|
|
436
|
+
recipe=recipe_obj,
|
|
437
|
+
diff=exec_result.diff,
|
|
438
|
+
quarantine=exec_result.quarantine,
|
|
439
|
+
issues=Issues(),
|
|
440
|
+
profile=None,
|
|
441
|
+
validation_results=exec_result.validation_results,
|
|
442
|
+
quality=None,
|
|
443
|
+
source=source,
|
|
444
|
+
log=log,
|
|
445
|
+
drift=drift,
|
|
446
|
+
)
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
# ---------------------------------------------------------------------------
|
|
450
|
+
# suggest --update (drift patch)
|
|
451
|
+
# ---------------------------------------------------------------------------
|
|
452
|
+
def suggest_update(
|
|
453
|
+
data: pd.DataFrame | str | Path,
|
|
454
|
+
recipe: Recipe | str | Path | dict,
|
|
455
|
+
*,
|
|
456
|
+
out: str | Path | None = None,
|
|
457
|
+
source: str | None = None,
|
|
458
|
+
sheet: str | int | None = None,
|
|
459
|
+
columns: list[str] | None = None,
|
|
460
|
+
nrows: int | None = None,
|
|
461
|
+
skiprows: int | list[int] | None = None,
|
|
462
|
+
) -> tuple[Recipe, DriftReport]:
|
|
463
|
+
"""Return a recipe patched to accommodate drift in ``data``, plus the drift report.
|
|
464
|
+
|
|
465
|
+
Applies safe, mechanical patches: repoint a renamed column to its new source,
|
|
466
|
+
and teach ``parse_date`` any new date formats that appeared. Structural
|
|
467
|
+
additions are reported but not auto-adopted — those are a human's call.
|
|
468
|
+
|
|
469
|
+
The recipe's recorded ``read:`` binding (sheet, delimiter, encoding, selection)
|
|
470
|
+
is re-applied, so a workbook or ``;``-separated file is read the same way it was
|
|
471
|
+
planned instead of reporting every column as drifted.
|
|
472
|
+
"""
|
|
473
|
+
original = _resolve_recipe(recipe)
|
|
474
|
+
call = _read_binding(sheet, columns, nrows, skiprows)
|
|
475
|
+
effective = {**(original.read or {}), **call}
|
|
476
|
+
if isinstance(data, pd.DataFrame):
|
|
477
|
+
effective = {k: v for k, v in effective.items() if k == "columns"}
|
|
478
|
+
df, source = _as_frame(data, source, **effective)
|
|
479
|
+
report_ = detect_drift(df, original, source=source)
|
|
480
|
+
patched = _patch_recipe_for_drift(original, report_, df)
|
|
481
|
+
if out is not None:
|
|
482
|
+
patched.save(out)
|
|
483
|
+
return patched, report_
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
def _patch_recipe_for_drift(recipe: Recipe, report: DriftReport, df: pd.DataFrame) -> Recipe:
|
|
487
|
+
from ._util import sample_non_null
|
|
488
|
+
from .detectors.dates import _infer_formats
|
|
489
|
+
from .fingerprint import fingerprint_dataframe
|
|
490
|
+
|
|
491
|
+
patched = Recipe.from_dict(recipe.to_dict()) # deep copy
|
|
492
|
+
patched.source_fingerprint = recipe.source_fingerprint
|
|
493
|
+
changes: list[str] = []
|
|
494
|
+
|
|
495
|
+
# 1) renamed columns -> repoint the recipe column's source
|
|
496
|
+
for finding in report.by_kind("renamed_column"):
|
|
497
|
+
new_col = finding.column
|
|
498
|
+
if new_col is None:
|
|
499
|
+
continue
|
|
500
|
+
matched = finding.evidence.get("match")
|
|
501
|
+
for col in patched.columns:
|
|
502
|
+
if col.output_name == matched or col.source == matched or col.rename_to == matched:
|
|
503
|
+
changes.append(f"repointed {col.source!r} → {new_col!r}")
|
|
504
|
+
col.source = new_col
|
|
505
|
+
break
|
|
506
|
+
|
|
507
|
+
# 2) new date formats -> extend the parse_date op
|
|
508
|
+
for finding in report.by_kind("format_drift"):
|
|
509
|
+
src = finding.column
|
|
510
|
+
if src is None:
|
|
511
|
+
continue
|
|
512
|
+
column = next((c for c in patched.columns if c.source == src), None)
|
|
513
|
+
if column is None:
|
|
514
|
+
continue
|
|
515
|
+
for op in column.ops:
|
|
516
|
+
if op.name != "parse_date" or src not in df.columns:
|
|
517
|
+
continue
|
|
518
|
+
existing = list(op.params.get("formats") or [])
|
|
519
|
+
new_formats, _ = _infer_formats(
|
|
520
|
+
[str(v) for v in sample_non_null(df[src])], op.params.get("dayfirst")
|
|
521
|
+
)
|
|
522
|
+
added = [f for f in new_formats if f not in existing]
|
|
523
|
+
if added:
|
|
524
|
+
op.params["formats"] = existing + added
|
|
525
|
+
changes.append(f"added date format(s) {added} to {src!r}")
|
|
526
|
+
|
|
527
|
+
if df is not None:
|
|
528
|
+
patched.source_fingerprint = fingerprint_dataframe(df)
|
|
529
|
+
patched.stamp_meta(patched_for_drift=changes or "no automatic patch applied")
|
|
530
|
+
return patched
|
|
531
|
+
|
|
532
|
+
|
|
533
|
+
# re-export
|
|
534
|
+
def infer_schema(
|
|
535
|
+
df: pd.DataFrame | str | Path,
|
|
536
|
+
name: str | None = None,
|
|
537
|
+
*,
|
|
538
|
+
sheet: str | int | None = None,
|
|
539
|
+
columns: list[str] | None = None,
|
|
540
|
+
nrows: int | None = None,
|
|
541
|
+
skiprows: int | list[int] | None = None,
|
|
542
|
+
correct_format: bool = True,
|
|
543
|
+
text: bool = False,
|
|
544
|
+
sep: str | None = None,
|
|
545
|
+
encoding: str | None = None,
|
|
546
|
+
) -> Schema:
|
|
547
|
+
"""Infer a target :class:`~cleanframe.schema.Schema` from data.
|
|
548
|
+
|
|
549
|
+
See :func:`cleanframe.schema.infer_schema`. Like :func:`clean`, a CSV-family
|
|
550
|
+
file's encoding and delimiter are detected unless ``correct_format=False``.
|
|
551
|
+
"""
|
|
552
|
+
frame, _, _, _ = _read_input(
|
|
553
|
+
df, None, sheet=sheet, columns=columns, nrows=nrows, skiprows=skiprows,
|
|
554
|
+
correct_format=correct_format, text=text, sep=sep, encoding=encoding,
|
|
555
|
+
)
|
|
556
|
+
return _infer_schema(frame, name=name)
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
__all__ = ["clean", "report", "apply_recipe", "suggest_update", "infer_schema"]
|