cleanframe-engine 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. cleanframe/__init__.py +169 -0
  2. cleanframe/__main__.py +5 -0
  3. cleanframe/_util.py +438 -0
  4. cleanframe/_version.py +1 -0
  5. cleanframe/api.py +559 -0
  6. cleanframe/cli.py +688 -0
  7. cleanframe/codegen.py +666 -0
  8. cleanframe/dataio.py +506 -0
  9. cleanframe/detectors/__init__.py +43 -0
  10. cleanframe/detectors/base.py +221 -0
  11. cleanframe/detectors/categories.py +199 -0
  12. cleanframe/detectors/contacts.py +105 -0
  13. cleanframe/detectors/currency.py +112 -0
  14. cleanframe/detectors/dates.py +204 -0
  15. cleanframe/detectors/dedup.py +150 -0
  16. cleanframe/detectors/nulls.py +109 -0
  17. cleanframe/detectors/outliers.py +73 -0
  18. cleanframe/detectors/schema_mapping.py +125 -0
  19. cleanframe/detectors/text.py +105 -0
  20. cleanframe/detectors/units.py +86 -0
  21. cleanframe/diff.py +369 -0
  22. cleanframe/drift.py +283 -0
  23. cleanframe/errors.py +66 -0
  24. cleanframe/executor.py +229 -0
  25. cleanframe/fingerprint.py +83 -0
  26. cleanframe/issues.py +186 -0
  27. cleanframe/llm.py +811 -0
  28. cleanframe/ops.py +1245 -0
  29. cleanframe/planner.py +353 -0
  30. cleanframe/profile.py +413 -0
  31. cleanframe/py.typed +1 -0
  32. cleanframe/quality.py +81 -0
  33. cleanframe/readfix.py +160 -0
  34. cleanframe/recipe.py +398 -0
  35. cleanframe/report.py +345 -0
  36. cleanframe/result.py +144 -0
  37. cleanframe/schema.py +259 -0
  38. cleanframe/streaming.py +354 -0
  39. cleanframe/types.py +119 -0
  40. cleanframe/validate.py +363 -0
  41. cleanframe/workbook.py +370 -0
  42. cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
  43. cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
  44. cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
  45. cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
  46. cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
cleanframe/api.py ADDED
@@ -0,0 +1,559 @@
1
+ """The high-level API: ``clean``, ``report``, ``apply_recipe``, ``suggest_update``.
2
+
3
+ These stitch the pipeline stages (profile → detect → plan → execute) into the few
4
+ calls most users ever touch. Every function accepts either a DataFrame or a path,
5
+ and returns rich result objects rather than bare frames so the recipe, diff, and
6
+ report are always one attribute away.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import warnings
12
+ from pathlib import Path
13
+ from typing import Any
14
+
15
+ import pandas as pd
16
+
17
+ from .dataio import read_frame
18
+ from .detectors import run_detectors
19
+ from .drift import DriftReport, detect_drift
20
+ from .errors import CleanFrameError, CleanFrameWarning, DriftError
21
+ from .executor import execute
22
+ from .issues import Issues
23
+ from .planner import Planner, RulesPlanner
24
+ from .profile import profile_dataframe
25
+ from .quality import quality_score
26
+ from .recipe import Recipe
27
+ from .result import CleanResult, Report, build_profile_report_object
28
+ from .schema import Schema
29
+ from .schema import infer_schema as _infer_schema
30
+ from .types import Mode
31
+
32
+
33
+ # ---------------------------------------------------------------------------
34
+ # input coercion
35
+ # ---------------------------------------------------------------------------
36
+ def _read_binding(sheet, columns, nrows, skiprows) -> dict[str, Any]: # noqa: D401
37
+ """Collect the non-default read/selection options into a dict (empty if none)."""
38
+ binding = {"sheet": sheet, "columns": columns, "nrows": nrows, "skiprows": skiprows}
39
+ return {k: v for k, v in binding.items() if v is not None}
40
+
41
+
42
+ def _read_input(
43
+ data, source, *, sheet, columns, nrows, skiprows, correct_format, text=False,
44
+ sep=None, encoding=None, warn=False,
45
+ ) -> tuple[pd.DataFrame, str | None, dict[str, Any], list[str]]:
46
+ """Read ``data`` with optional CSV format auto-correction.
47
+
48
+ Returns ``(df, source, read_binding, notes)`` where ``read_binding`` is the full
49
+ selection+format binding to record in a recipe's ``read:`` section.
50
+ """
51
+ fmt_kwargs: dict[str, Any] = {}
52
+ notes: list[str] = []
53
+ if correct_format and isinstance(data, (str, Path)):
54
+ from .readfix import detect_csv_options, is_csv_family
55
+
56
+ # Only sniff a file that is actually there: read_frame owns the
57
+ # missing-file / directory / empty-file messages.
58
+ if is_csv_family(data) and Path(data).is_file() and Path(data).stat().st_size:
59
+ _opts, report_ = detect_csv_options(data) # raises on ambiguous delimiter
60
+ fmt_kwargs = report_.as_read_binding() # {encoding?, sep?, blank_lines?}
61
+ notes = report_.notes
62
+ if warn and notes:
63
+ warnings.warn(
64
+ "CleanFrame: read-time format correction — " + "; ".join(notes),
65
+ CleanFrameWarning,
66
+ stacklevel=3,
67
+ )
68
+ # An explicit sep/encoding always wins over what detection guessed.
69
+ fmt_kwargs.update({k: v for k, v in (("sep", sep), ("encoding", encoding)) if v is not None})
70
+ df, source = _as_frame(
71
+ data, source, sheet=sheet, columns=columns, nrows=nrows, skiprows=skiprows,
72
+ text=text, **fmt_kwargs,
73
+ )
74
+ if correct_format and not text and isinstance(data, (str, Path)):
75
+ from .dataio import inference_losses
76
+
77
+ if Path(data).is_file() and Path(data).suffix.lower() not in (".parquet", ".json"):
78
+ losses = inference_losses(
79
+ data, df, sheet=sheet, columns=columns, skiprows=skiprows, **fmt_kwargs
80
+ )
81
+ if losses:
82
+ detail = "; ".join(f"{col}: {why}" for col, why in sorted(losses.items()))
83
+ notes = [*notes, f"type inference changed values on read — {detail}"]
84
+ if warn:
85
+ warnings.warn(
86
+ "CleanFrame: pandas type inference changed values while reading "
87
+ f"({detail}). Pass text=True to read every field verbatim.",
88
+ CleanFrameWarning,
89
+ stacklevel=3,
90
+ )
91
+ binding = _read_binding(sheet, columns, nrows, skiprows)
92
+ if text:
93
+ binding["text"] = True
94
+ binding.update(fmt_kwargs)
95
+ return df, source, binding, notes
96
+
97
+
98
+ def _as_frame(
99
+ data: pd.DataFrame | str | Path,
100
+ source: str | None,
101
+ *,
102
+ sheet=None,
103
+ columns=None,
104
+ nrows=None,
105
+ skiprows=None,
106
+ text: bool = False,
107
+ **read_kwargs,
108
+ ) -> tuple[pd.DataFrame, str | None]:
109
+ from ._util import ensure_string_columns
110
+
111
+ if isinstance(data, pd.DataFrame):
112
+ if sheet is not None or nrows is not None or skiprows is not None:
113
+ raise CleanFrameError(
114
+ "sheet=/nrows=/skiprows= selection applies to file inputs only, "
115
+ "not an in-memory DataFrame. Slice the DataFrame yourself first."
116
+ )
117
+ df = data
118
+ if columns is not None: # a column projection is well-defined on a DataFrame
119
+ missing = [c for c in columns if c not in df.columns]
120
+ if missing:
121
+ raise CleanFrameError(
122
+ f"Requested column(s) not found: {missing}. "
123
+ f"Available: {list(df.columns)}."
124
+ )
125
+ df = df[list(columns)]
126
+ return ensure_string_columns(df), source # read_kwargs (encoding/sep) are file-only
127
+ if isinstance(data, (str, Path)):
128
+ df = read_frame(
129
+ data, sheet=sheet, columns=columns, nrows=nrows, skiprows=skiprows,
130
+ text=text, **read_kwargs,
131
+ )
132
+ return ensure_string_columns(df), source or str(data)
133
+ raise CleanFrameError(f"Expected a DataFrame or file path, got {type(data).__name__}.")
134
+
135
+
136
+ def _resolve_schema(schema: Any) -> Schema | None:
137
+ if schema is None or isinstance(schema, Schema):
138
+ return schema
139
+ if isinstance(schema, (str, Path)):
140
+ return Schema.load(schema)
141
+ if isinstance(schema, dict):
142
+ return Schema.from_dict(schema)
143
+ raise CleanFrameError(f"Unsupported schema type: {type(schema).__name__}.")
144
+
145
+
146
+ def _resolve_recipe(recipe: Any) -> Recipe:
147
+ if isinstance(recipe, Recipe):
148
+ return recipe
149
+ if isinstance(recipe, (str, Path)):
150
+ return Recipe.load(recipe)
151
+ if isinstance(recipe, dict):
152
+ return Recipe.from_dict(recipe)
153
+ raise CleanFrameError(f"Unsupported recipe type: {type(recipe).__name__}.")
154
+
155
+
156
+ def _resolve_planner(
157
+ planner: Planner | None,
158
+ llm: Any,
159
+ llm_exposure: str,
160
+ max_tokens_budget: int | None,
161
+ llm_fallback: bool = True,
162
+ ) -> Planner:
163
+ if planner is not None:
164
+ if not hasattr(planner, "plan"):
165
+ raise CleanFrameError(
166
+ f"planner must have a .plan() method, got {type(planner).__name__}."
167
+ )
168
+ return planner
169
+ if llm is None:
170
+ return RulesPlanner()
171
+ from .types import LLMExposure
172
+
173
+ if LLMExposure(str(llm_exposure)) is LLMExposure.NONE:
174
+ # "none" means nothing leaves the machine, so there is no request to make.
175
+ warnings.warn(
176
+ "CleanFrame: llm_exposure='none' keeps everything local, so the LLM was not "
177
+ "called — planning with deterministic rules instead.",
178
+ CleanFrameWarning,
179
+ stacklevel=3,
180
+ )
181
+ return RulesPlanner()
182
+ from .llm import LLMPlanner, get_client
183
+
184
+ if isinstance(llm, str):
185
+ client = get_client(llm)
186
+ elif hasattr(llm, "complete"):
187
+ client = llm
188
+ else:
189
+ raise CleanFrameError(
190
+ "llm must be a 'provider/model' string or an object with a .complete() method."
191
+ )
192
+ return LLMPlanner(
193
+ client,
194
+ exposure=llm_exposure,
195
+ max_tokens_budget=max_tokens_budget,
196
+ fallback="rules" if llm_fallback else None,
197
+ )
198
+
199
+
200
+ _ON_DRIFT = ("error", "warn", "ignore")
201
+
202
+
203
+ def _check_on_drift(on_drift: Any) -> str:
204
+ if on_drift not in _ON_DRIFT:
205
+ raise CleanFrameError(
206
+ f"on_drift must be one of {list(_ON_DRIFT)}, got {on_drift!r}. A typo here "
207
+ "would silently disable the drift guard."
208
+ )
209
+ return on_drift
210
+
211
+
212
+ def _check_options(options: Any) -> dict[str, Any]:
213
+ if options is None:
214
+ return {}
215
+ if not isinstance(options, dict):
216
+ raise CleanFrameError(
217
+ f"options must be a mapping of detector knobs, got {type(options).__name__}."
218
+ )
219
+ cap = options.get("max_diff_changes", 0)
220
+ if cap is not None and (not isinstance(cap, int) or isinstance(cap, bool) or cap < 0):
221
+ raise CleanFrameError(
222
+ f"options['max_diff_changes'] must be a non-negative int or None, got {cap!r}."
223
+ )
224
+ return dict(options)
225
+
226
+
227
+ # ---------------------------------------------------------------------------
228
+ # clean
229
+ # ---------------------------------------------------------------------------
230
+ def clean(
231
+ data: pd.DataFrame | str | Path,
232
+ *,
233
+ target_schema: Any = None,
234
+ schema: Any = None,
235
+ llm: Any = None,
236
+ mode: Mode | str = Mode.REVIEW,
237
+ options: dict[str, Any] | None = None,
238
+ planner: Planner | None = None,
239
+ max_tokens_budget: int | None = None,
240
+ llm_exposure: str = "metadata",
241
+ source: str | None = None,
242
+ sheet: str | int | None = None,
243
+ columns: list[str] | None = None,
244
+ nrows: int | None = None,
245
+ skiprows: int | list[int] | None = None,
246
+ correct_format: bool = True,
247
+ text: bool = False,
248
+ sep: str | None = None,
249
+ encoding: str | None = None,
250
+ llm_fallback: bool = True,
251
+ ) -> CleanResult:
252
+ """Profile, plan, and clean ``data`` — the main entry point.
253
+
254
+ ``sheet``/``columns``/``nrows``/``skiprows`` select part of a file (see
255
+ :func:`read_frame`); the selection is recorded in the recipe's ``read:`` section
256
+ so :func:`apply_recipe` re-reads the same slice. For a multi-sheet workbook, pass
257
+ ``sheet=`` or use :func:`clean_workbook` to clean every sheet.
258
+
259
+ ``correct_format`` (default ``True``) auto-detects a CSV-family file's encoding
260
+ and delimiter at read time (e.g. a ``;``-separated or cp1252 file), warns, and
261
+ pins the choice into the recipe's ``read:`` section for deterministic replay. An
262
+ ambiguous delimiter raises rather than guessing. It also reports values that
263
+ pandas' type inference changed while reading (a leading-zero ZIP, an ``NA``
264
+ token); pass ``text=True`` to read every field verbatim instead.
265
+
266
+ ``llm_fallback`` (default ``True``) keeps the documented behaviour of degrading
267
+ to the rules planner when an LLM call fails. Pass ``False`` to make such a
268
+ failure raise instead of quietly producing a rules-only recipe.
269
+
270
+ The returned frame is indexed 0..n-1: replaying a recipe re-keys rows to the
271
+ stable positional row ids the diff and quarantine refer to.
272
+
273
+ Parameters
274
+ ----------
275
+ data:
276
+ A DataFrame or a path to a CSV/Excel/Parquet/JSON file.
277
+ target_schema / schema:
278
+ Optional target :class:`~cleanframe.schema.Schema`, path, or dict. Drives
279
+ column mapping and validation synthesis.
280
+ llm:
281
+ ``None`` (rules-only, default), a ``"provider/model"`` string, or any object
282
+ with a ``.complete()`` method. The LLM only writes the recipe; it never sees
283
+ raw data (see :mod:`cleanframe.llm`).
284
+ mode:
285
+ ``"review"`` (default), ``"auto"``, or ``"strict"``.
286
+ max_tokens_budget:
287
+ Hard cap that aborts LLM planning before it gets expensive.
288
+
289
+ Returns
290
+ -------
291
+ CleanResult
292
+ Cleaned dataframe plus recipe, diff, quarantine, issues, and report.
293
+ """
294
+ df, source, read_binding, read_notes = _read_input(
295
+ data, source, sheet=sheet, columns=columns, nrows=nrows, skiprows=skiprows,
296
+ correct_format=correct_format, text=text, sep=sep, encoding=encoding, warn=True,
297
+ )
298
+ schema_obj = _resolve_schema(target_schema if target_schema is not None else schema)
299
+ options = _check_options(options)
300
+
301
+ profile = profile_dataframe(df)
302
+ issues = run_detectors(df, profile=profile, schema=schema_obj, options=options)
303
+ the_planner = _resolve_planner(
304
+ planner, llm, llm_exposure, max_tokens_budget, llm_fallback=llm_fallback
305
+ )
306
+ recipe = the_planner.plan(df, profile, issues, schema=schema_obj, mode=mode, options=options)
307
+
308
+ # Record the read/selection + format binding so `apply` re-reads identically.
309
+ if read_binding:
310
+ recipe.read = {**(recipe.read or {}), **read_binding}
311
+
312
+ from ._util import DEFAULT_MAX_DIFF_CHANGES
313
+
314
+ max_diff = options.get("max_diff_changes", DEFAULT_MAX_DIFF_CHANGES)
315
+ exec_result = execute(recipe, df, mode=mode, max_diff_changes=max_diff)
316
+ quality = quality_score(profile, issues)
317
+
318
+ return CleanResult(
319
+ dataframe=exec_result.dataframe,
320
+ recipe=recipe,
321
+ diff=exec_result.diff,
322
+ quarantine=exec_result.quarantine,
323
+ issues=issues,
324
+ profile=profile,
325
+ validation_results=exec_result.validation_results,
326
+ quality=quality,
327
+ source=source,
328
+ log=[*(f"read-fix: {n}" for n in read_notes), *exec_result.log],
329
+ )
330
+
331
+
332
+ # ---------------------------------------------------------------------------
333
+ # report
334
+ # ---------------------------------------------------------------------------
335
+ def report(
336
+ data: pd.DataFrame | str | Path,
337
+ *,
338
+ schema: Any = None,
339
+ options: dict[str, Any] | None = None,
340
+ source: str | None = None,
341
+ sheet: str | int | None = None,
342
+ columns: list[str] | None = None,
343
+ nrows: int | None = None,
344
+ skiprows: int | list[int] | None = None,
345
+ correct_format: bool = True,
346
+ text: bool = False,
347
+ sep: str | None = None,
348
+ encoding: str | None = None,
349
+ ) -> Report:
350
+ """Profile ``data`` and return an HTML :class:`~cleanframe.result.Report` (no changes made)."""
351
+ df, source, _, _ = _read_input(
352
+ data, source, sheet=sheet, columns=columns, nrows=nrows, skiprows=skiprows,
353
+ correct_format=correct_format, text=text, sep=sep, encoding=encoding, warn=True,
354
+ )
355
+ schema_obj = _resolve_schema(schema)
356
+ profile = profile_dataframe(df)
357
+ issues = run_detectors(df, profile=profile, schema=schema_obj, options=_check_options(options))
358
+ quality = quality_score(profile, issues)
359
+ return build_profile_report_object(profile, issues, source=source, quality=quality)
360
+
361
+
362
+ # ---------------------------------------------------------------------------
363
+ # apply (replay)
364
+ # ---------------------------------------------------------------------------
365
+ def apply_recipe(
366
+ data: pd.DataFrame | str | Path,
367
+ recipe: Recipe | str | Path | dict,
368
+ *,
369
+ mode: Mode | str = Mode.REVIEW,
370
+ check_drift: bool = True,
371
+ on_drift: str = "error",
372
+ source: str | None = None,
373
+ sheet: str | int | None = None,
374
+ columns: list[str] | None = None,
375
+ nrows: int | None = None,
376
+ skiprows: int | list[int] | None = None,
377
+ text: bool = False,
378
+ sep: str | None = None,
379
+ encoding: str | None = None,
380
+ ) -> CleanResult:
381
+ """Replay a saved recipe on new data — deterministic, no LLM.
382
+
383
+ If ``check_drift`` and the incoming schema drifted, ``on_drift`` decides:
384
+ ``"error"`` (default — raise :class:`~cleanframe.errors.DriftError` so nothing
385
+ is silently corrupted), ``"warn"`` (attach the report and warn, then continue),
386
+ or ``"ignore"``. ``strict`` mode always raises on drift.
387
+
388
+ The selection used to plan the recipe (its ``read:`` section) is re-applied when
389
+ ``data`` is a path, unless overridden by an explicit ``sheet``/``columns``/etc.
390
+ """
391
+ recipe_obj = _resolve_recipe(recipe)
392
+ mode = Mode.coerce(mode)
393
+ _check_on_drift(on_drift)
394
+
395
+ # Precedence: explicit call args > recipe-recorded read binding > whole file.
396
+ call = _read_binding(sheet, columns, nrows, skiprows)
397
+ call.update({k: v for k, v in (("sep", sep), ("encoding", encoding)) if v is not None})
398
+ if text:
399
+ call["text"] = True
400
+ effective = {**(recipe_obj.read or {}), **call}
401
+ if isinstance(data, pd.DataFrame) and effective:
402
+ unreplayable = {k for k in effective if k in ("sheet", "nrows", "skiprows")}
403
+ if unreplayable and not call:
404
+ warnings.warn(
405
+ f"CleanFrame: recipe's recorded read binding {sorted(unreplayable)} cannot "
406
+ "replay against an in-memory DataFrame; ignoring it.",
407
+ CleanFrameWarning,
408
+ stacklevel=2,
409
+ )
410
+ effective = {k: v for k, v in effective.items() if k == "columns"}
411
+ df, source = _as_frame(data, source, **effective)
412
+
413
+ drift: DriftReport | None = None
414
+ if check_drift and not recipe_obj.source_fingerprint:
415
+ warnings.warn(
416
+ "CleanFrame: recipe has no source_fingerprint — drift check skipped. "
417
+ "Re-plan or stamp a fingerprint for production replay.",
418
+ CleanFrameWarning,
419
+ stacklevel=2,
420
+ )
421
+ if check_drift and recipe_obj.source_fingerprint:
422
+ drift = detect_drift(df, recipe_obj, source=source)
423
+ if drift.has_drift:
424
+ if on_drift == "error" or mode is Mode.STRICT:
425
+ raise DriftError(drift.render(), report=drift)
426
+ if on_drift == "warn":
427
+ warnings.warn("CleanFrame: " + drift.render(), CleanFrameWarning, stacklevel=2)
428
+
429
+ exec_result = execute(recipe_obj, df, mode=mode)
430
+ log = list(exec_result.log)
431
+ if drift is not None and drift.has_drift:
432
+ log.insert(0, "drift detected: " + "; ".join(f.message for f in drift.findings))
433
+
434
+ return CleanResult(
435
+ dataframe=exec_result.dataframe,
436
+ recipe=recipe_obj,
437
+ diff=exec_result.diff,
438
+ quarantine=exec_result.quarantine,
439
+ issues=Issues(),
440
+ profile=None,
441
+ validation_results=exec_result.validation_results,
442
+ quality=None,
443
+ source=source,
444
+ log=log,
445
+ drift=drift,
446
+ )
447
+
448
+
449
+ # ---------------------------------------------------------------------------
450
+ # suggest --update (drift patch)
451
+ # ---------------------------------------------------------------------------
452
+ def suggest_update(
453
+ data: pd.DataFrame | str | Path,
454
+ recipe: Recipe | str | Path | dict,
455
+ *,
456
+ out: str | Path | None = None,
457
+ source: str | None = None,
458
+ sheet: str | int | None = None,
459
+ columns: list[str] | None = None,
460
+ nrows: int | None = None,
461
+ skiprows: int | list[int] | None = None,
462
+ ) -> tuple[Recipe, DriftReport]:
463
+ """Return a recipe patched to accommodate drift in ``data``, plus the drift report.
464
+
465
+ Applies safe, mechanical patches: repoint a renamed column to its new source,
466
+ and teach ``parse_date`` any new date formats that appeared. Structural
467
+ additions are reported but not auto-adopted — those are a human's call.
468
+
469
+ The recipe's recorded ``read:`` binding (sheet, delimiter, encoding, selection)
470
+ is re-applied, so a workbook or ``;``-separated file is read the same way it was
471
+ planned instead of reporting every column as drifted.
472
+ """
473
+ original = _resolve_recipe(recipe)
474
+ call = _read_binding(sheet, columns, nrows, skiprows)
475
+ effective = {**(original.read or {}), **call}
476
+ if isinstance(data, pd.DataFrame):
477
+ effective = {k: v for k, v in effective.items() if k == "columns"}
478
+ df, source = _as_frame(data, source, **effective)
479
+ report_ = detect_drift(df, original, source=source)
480
+ patched = _patch_recipe_for_drift(original, report_, df)
481
+ if out is not None:
482
+ patched.save(out)
483
+ return patched, report_
484
+
485
+
486
+ def _patch_recipe_for_drift(recipe: Recipe, report: DriftReport, df: pd.DataFrame) -> Recipe:
487
+ from ._util import sample_non_null
488
+ from .detectors.dates import _infer_formats
489
+ from .fingerprint import fingerprint_dataframe
490
+
491
+ patched = Recipe.from_dict(recipe.to_dict()) # deep copy
492
+ patched.source_fingerprint = recipe.source_fingerprint
493
+ changes: list[str] = []
494
+
495
+ # 1) renamed columns -> repoint the recipe column's source
496
+ for finding in report.by_kind("renamed_column"):
497
+ new_col = finding.column
498
+ if new_col is None:
499
+ continue
500
+ matched = finding.evidence.get("match")
501
+ for col in patched.columns:
502
+ if col.output_name == matched or col.source == matched or col.rename_to == matched:
503
+ changes.append(f"repointed {col.source!r} → {new_col!r}")
504
+ col.source = new_col
505
+ break
506
+
507
+ # 2) new date formats -> extend the parse_date op
508
+ for finding in report.by_kind("format_drift"):
509
+ src = finding.column
510
+ if src is None:
511
+ continue
512
+ column = next((c for c in patched.columns if c.source == src), None)
513
+ if column is None:
514
+ continue
515
+ for op in column.ops:
516
+ if op.name != "parse_date" or src not in df.columns:
517
+ continue
518
+ existing = list(op.params.get("formats") or [])
519
+ new_formats, _ = _infer_formats(
520
+ [str(v) for v in sample_non_null(df[src])], op.params.get("dayfirst")
521
+ )
522
+ added = [f for f in new_formats if f not in existing]
523
+ if added:
524
+ op.params["formats"] = existing + added
525
+ changes.append(f"added date format(s) {added} to {src!r}")
526
+
527
+ if df is not None:
528
+ patched.source_fingerprint = fingerprint_dataframe(df)
529
+ patched.stamp_meta(patched_for_drift=changes or "no automatic patch applied")
530
+ return patched
531
+
532
+
533
+ # re-export
534
+ def infer_schema(
535
+ df: pd.DataFrame | str | Path,
536
+ name: str | None = None,
537
+ *,
538
+ sheet: str | int | None = None,
539
+ columns: list[str] | None = None,
540
+ nrows: int | None = None,
541
+ skiprows: int | list[int] | None = None,
542
+ correct_format: bool = True,
543
+ text: bool = False,
544
+ sep: str | None = None,
545
+ encoding: str | None = None,
546
+ ) -> Schema:
547
+ """Infer a target :class:`~cleanframe.schema.Schema` from data.
548
+
549
+ See :func:`cleanframe.schema.infer_schema`. Like :func:`clean`, a CSV-family
550
+ file's encoding and delimiter are detected unless ``correct_format=False``.
551
+ """
552
+ frame, _, _, _ = _read_input(
553
+ df, None, sheet=sheet, columns=columns, nrows=nrows, skiprows=skiprows,
554
+ correct_format=correct_format, text=text, sep=sep, encoding=encoding,
555
+ )
556
+ return _infer_schema(frame, name=name)
557
+
558
+
559
+ __all__ = ["clean", "report", "apply_recipe", "suggest_update", "infer_schema"]