tabalyst 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tabalyst/analysis.py ADDED
@@ -0,0 +1,786 @@
1
+ """Pure column and dataset analysis. Raw cells are never rewritten."""
2
+
3
+ import math
4
+ import random
5
+ import re
6
+ from collections import Counter
7
+ from dataclasses import dataclass
8
+ from datetime import UTC, date, datetime
9
+ from itertools import combinations, islice
10
+
11
+ import pandas as pd
12
+
13
+ from tabalyst.config import AnalysisConfig
14
+ from tabalyst.ingestion import CsvDataset
15
+ from tabalyst.models import (
16
+ ColumnProfile,
17
+ DatasetProfile,
18
+ DatasetSummary,
19
+ DateBreakdownItem,
20
+ DateFormatCount,
21
+ DateProfile,
22
+ EnumCandidate,
23
+ Issue,
24
+ NormalizationStats,
25
+ NumericStats,
26
+ PreviewRow,
27
+ StringLengthDistribution,
28
+ StringLengthExample,
29
+ StringProfile,
30
+ ValueOccurrence,
31
+ ValueProfile,
32
+ )
33
+
34
+ INTEGER = re.compile(r"[+-]?(?:0|[1-9][0-9]*)")
35
+ NUMBER = re.compile(
36
+ r"[+-]?(?:(?:0|[1-9][0-9]*)(?:\.[0-9]*)?|\.[0-9]+)(?:[eE][+-]?[0-9]+)?"
37
+ )
38
+ INTERNAL_HORIZONTAL_WHITESPACE = re.compile(r"[^\S\r\n]+")
39
+
40
+
41
+ def percent(count: int, total: int) -> float:
42
+ return round(100 * count / total, 2) if total else 0.0
43
+
44
+
45
+ def value_type(value: str) -> str:
46
+ if value.casefold() in {"true", "false"}:
47
+ return "boolean"
48
+ if INTEGER.fullmatch(value):
49
+ return "integer"
50
+ if NUMBER.fullmatch(value):
51
+ return "number"
52
+ return "text"
53
+
54
+
55
+ def missing_mask(values: pd.Series, config: AnalysisConfig) -> pd.Series:
56
+ return values.str.strip().isin({marker.strip() for marker in config.missing_values})
57
+
58
+
59
+ def normalize_values(
60
+ values: pd.Series, config: AnalysisConfig
61
+ ) -> tuple[pd.Series, NormalizationStats]:
62
+ """Normalize analysis values while counting each changed cell per operation."""
63
+ normalized = values
64
+ trim_count = 0
65
+ collapse_count = 0
66
+ if config.normalization.trim:
67
+ trimmed = normalized.str.strip()
68
+ trim_count = int((trimmed != normalized).sum())
69
+ normalized = trimmed
70
+ if config.normalization.collapse_internal_whitespace:
71
+ collapsed = normalized.str.replace(
72
+ INTERNAL_HORIZONTAL_WHITESPACE, " ", regex=True
73
+ )
74
+ collapse_count = int((collapsed != normalized).sum())
75
+ normalized = collapsed
76
+ return normalized, NormalizationStats(
77
+ trim_count=trim_count,
78
+ trim_percent=percent(trim_count, len(values)),
79
+ collapse_internal_whitespace_count=collapse_count,
80
+ collapse_internal_whitespace_percent=percent(collapse_count, len(values)),
81
+ )
82
+
83
+
84
+ @dataclass(frozen=True)
85
+ class DateValueResult:
86
+ state: str
87
+ order: str | None = None
88
+ separator: str | None = None
89
+ format: str | None = None
90
+ possible_orders: tuple[str, ...] = ()
91
+ possible_formats: tuple[str, ...] = ()
92
+ error: str | None = None
93
+
94
+
95
+ def valid_calendar_date(order: str, first: int, second: int, third: int) -> bool:
96
+ if order == "YMD":
97
+ year, month, day = first, second, third
98
+ elif order == "MDY":
99
+ month, day, year = first, second, third
100
+ else:
101
+ day, month, year = first, second, third
102
+ try:
103
+ date(year, month, day)
104
+ except ValueError:
105
+ return False
106
+ return True
107
+
108
+
109
+ def date_format_name(
110
+ order: str,
111
+ first: str,
112
+ second: str,
113
+ third: str,
114
+ separator: str,
115
+ ) -> str:
116
+ """Describe component order, separator and zero-padding explicitly."""
117
+ month = "MM" if len(second if order == "YMD" else first) == 2 else "M"
118
+ if order == "YMD":
119
+ day = "DD" if len(third) == 2 else "D"
120
+ parts = ("YYYY", month, day)
121
+ elif order == "MDY":
122
+ day = "DD" if len(second) == 2 else "D"
123
+ parts = (month, day, "YYYY")
124
+ else:
125
+ day = "DD" if len(first) == 2 else "D"
126
+ month = "MM" if len(second) == 2 else "M"
127
+ parts = (day, month, "YYYY")
128
+ return separator.join(parts)
129
+
130
+
131
+ def classify_date_value(value: str, config: AnalysisConfig) -> DateValueResult:
132
+ """Recognize configured numeric date structures without fuzzy interpretation."""
133
+ settings = config.date_detection
134
+ if not settings.enabled:
135
+ return DateValueResult("not_date")
136
+ separators = "".join(re.escape(separator) for separator in settings.separators)
137
+ match = re.fullmatch(
138
+ rf"(?P<a>\d{{1,4}})(?P<s1>[{separators}])(?P<b>\d{{1,4}})"
139
+ rf"(?P<s2>[{separators}])(?P<c>\d{{1,4}})",
140
+ value,
141
+ )
142
+ if not match:
143
+ return DateValueResult("not_date")
144
+ first_text, second_text, third_text = (
145
+ match.group("a"),
146
+ match.group("b"),
147
+ match.group("c"),
148
+ )
149
+ first, second, third = map(int, (first_text, second_text, third_text))
150
+ separator = match.group("s1")
151
+ year_first = len(first_text) == 4
152
+ year_last = len(third_text) == 4
153
+ if year_first and len(second_text) <= 2 and len(third_text) <= 2:
154
+ possible_orders = ("YMD",)
155
+ elif year_last and len(first_text) <= 2 and len(second_text) <= 2:
156
+ possible_orders = ("MDY", "DMY")
157
+ elif year_first:
158
+ return DateValueResult("invalid", error="invalid_component_width")
159
+ else:
160
+ return DateValueResult("not_date")
161
+ if match.group("s1") != match.group("s2"):
162
+ return DateValueResult("invalid", error="mixed_separators")
163
+ allowed_orders = [order for order in possible_orders if order in settings.orders]
164
+ if not allowed_orders:
165
+ return DateValueResult("invalid", error="unsupported_order")
166
+ valid_orders = tuple(
167
+ order
168
+ for order in allowed_orders
169
+ if valid_calendar_date(order, first, second, third)
170
+ )
171
+ if not valid_orders:
172
+ return DateValueResult("invalid", error="invalid_calendar_date")
173
+ if len(valid_orders) == 1:
174
+ order = valid_orders[0]
175
+ return DateValueResult(
176
+ "valid",
177
+ order=order,
178
+ separator=separator,
179
+ format=date_format_name(
180
+ order, first_text, second_text, third_text, separator
181
+ ),
182
+ )
183
+ return DateValueResult(
184
+ "ambiguous",
185
+ separator=separator,
186
+ possible_orders=valid_orders,
187
+ possible_formats=tuple(
188
+ date_format_name(
189
+ order, first_text, second_text, third_text, separator
190
+ )
191
+ for order in valid_orders
192
+ ),
193
+ )
194
+
195
+
196
+ def build_date_profile(
197
+ frequencies: list[tuple[str, int]], config: AnalysisConfig
198
+ ) -> tuple[DateProfile | None, set[str]]:
199
+ """Summarize strict date formats and resolve ambiguity only from clear evidence."""
200
+ if not config.date_detection.enabled:
201
+ return None, set()
202
+ classified = {
203
+ value: classify_date_value(value, config) for value, _ in frequencies
204
+ }
205
+ if not any(result.state != "not_date" for result in classified.values()):
206
+ return None, set()
207
+
208
+ evidence = Counter()
209
+ for value, count in frequencies:
210
+ result = classified[value]
211
+ if result.state == "valid" and result.order in {"MDY", "DMY"}:
212
+ evidence[result.order] += count
213
+ resolved_order = config.date_detection.ambiguous_order
214
+ resolution_source = "config" if resolved_order else None
215
+ if not resolved_order:
216
+ if evidence["MDY"] and not evidence["DMY"]:
217
+ resolved_order = "MDY"
218
+ resolution_source = "column"
219
+ elif evidence["DMY"] and not evidence["MDY"]:
220
+ resolved_order = "DMY"
221
+ resolution_source = "column"
222
+
223
+ valid_values: set[str] = set()
224
+ formats: Counter[tuple[str, str, str]] = Counter()
225
+ ambiguous_formats: Counter[str] = Counter()
226
+ errors: Counter[str] = Counter()
227
+ valid_count = 0
228
+ ambiguous_count = 0
229
+ invalid_count = 0
230
+ not_date_count = 0
231
+ for value, count in frequencies:
232
+ result = classified[value]
233
+ order = result.order
234
+ format_name = result.format
235
+ if (
236
+ result.state == "ambiguous"
237
+ and resolved_order in result.possible_orders
238
+ ):
239
+ order = resolved_order
240
+ format_name = result.possible_formats[
241
+ result.possible_orders.index(str(resolved_order))
242
+ ]
243
+ if result.state == "valid" or order:
244
+ valid_values.add(value)
245
+ valid_count += count
246
+ formats[(str(format_name), str(order), str(result.separator))] += count
247
+ elif result.state == "ambiguous":
248
+ ambiguous_count += count
249
+ ambiguous_formats[
250
+ "Ambiguous: " + " or ".join(result.possible_formats)
251
+ ] += count
252
+ elif result.state == "invalid":
253
+ invalid_count += count
254
+ errors[str(result.error)] += count
255
+ else:
256
+ not_date_count += count
257
+
258
+ total = sum(count for _, count in frequencies)
259
+ if valid_count == total:
260
+ status = "valid" if len(formats) == 1 else "multiple_formats"
261
+ elif ambiguous_count == total:
262
+ status = "ambiguous"
263
+ elif invalid_count == total:
264
+ status = "invalid"
265
+ else:
266
+ status = "mixed"
267
+ format_items = [
268
+ DateFormatCount(
269
+ format=format_name,
270
+ order=order,
271
+ separator=separator,
272
+ count=count,
273
+ percent=percent(count, total),
274
+ iso=format_name == "YYYY-MM-DD",
275
+ )
276
+ for (format_name, order, separator), count in sorted(
277
+ formats.items(), key=lambda item: (-item[1], item[0])
278
+ )
279
+ ]
280
+ date_breakdown = [
281
+ DateBreakdownItem(
282
+ label=item.format,
283
+ category="valid",
284
+ count=item.count,
285
+ percent=item.percent,
286
+ )
287
+ for item in format_items
288
+ ]
289
+ date_breakdown.extend(
290
+ DateBreakdownItem(
291
+ label=label,
292
+ category="ambiguous",
293
+ count=count,
294
+ percent=percent(count, total),
295
+ )
296
+ for label, count in ambiguous_formats.items()
297
+ )
298
+ date_breakdown.sort(key=lambda item: (-item.count, item.label))
299
+ breakdown = date_breakdown + [
300
+ DateBreakdownItem(
301
+ label="Invalid date",
302
+ category="invalid",
303
+ count=invalid_count,
304
+ percent=percent(invalid_count, total),
305
+ ),
306
+ DateBreakdownItem(
307
+ label="Not a date",
308
+ category="not_date",
309
+ count=not_date_count,
310
+ percent=percent(not_date_count, total),
311
+ ),
312
+ ]
313
+ return (
314
+ DateProfile(
315
+ status=status,
316
+ valid_count=valid_count,
317
+ ambiguous_count=ambiguous_count,
318
+ invalid_date_count=invalid_count,
319
+ not_date_count=not_date_count,
320
+ resolved_ambiguous_order=resolved_order,
321
+ ambiguous_order_source=resolution_source,
322
+ formats=format_items,
323
+ format_count=len(format_items),
324
+ breakdown=breakdown,
325
+ errors=dict(errors),
326
+ ),
327
+ valid_values,
328
+ )
329
+
330
+
331
+ def infer_column_type(
332
+ counts: Counter[str],
333
+ date_profile: DateProfile | None,
334
+ total: int,
335
+ config: AnalysisConfig,
336
+ ) -> tuple[str, str | None, float, int | None, float | None]:
337
+ """Infer a dominant type and quantify values outside the accepted family."""
338
+ if not total:
339
+ return "empty", None, 1.0, 0, 0.0
340
+ threshold = config.type_inference.minimum_confidence
341
+ candidates = [
342
+ ("integer", counts["integer"]),
343
+ ("number", counts["integer"] + counts["number"]),
344
+ ]
345
+ for inferred_type, accepted in candidates:
346
+ confidence = accepted / total
347
+ if confidence >= threshold:
348
+ errors = total - accepted
349
+ return inferred_type, None, confidence, errors, percent(errors, total)
350
+
351
+ date_accepted = (
352
+ date_profile.valid_count + date_profile.ambiguous_count
353
+ if date_profile
354
+ else 0
355
+ )
356
+ date_confidence = date_accepted / total
357
+ if date_profile and date_confidence >= threshold:
358
+ inferred_type = (
359
+ "date"
360
+ if date_profile.format_count == 1
361
+ and date_profile.ambiguous_count == 0
362
+ else "mixed"
363
+ )
364
+ errors = total - date_profile.valid_count
365
+ return inferred_type, "date", date_confidence, errors, percent(errors, total)
366
+
367
+ for inferred_type in ("boolean", "text"):
368
+ accepted = counts[inferred_type]
369
+ confidence = accepted / total
370
+ if confidence >= threshold:
371
+ errors = total - accepted
372
+ return inferred_type, None, confidence, errors, percent(errors, total)
373
+
374
+ family_counts = [
375
+ counts["integer"],
376
+ counts["integer"] + counts["number"],
377
+ date_accepted,
378
+ counts["boolean"],
379
+ counts["text"],
380
+ ]
381
+ return "mixed", None, max(family_counts) / total, None, None
382
+
383
+
384
+ def build_string_profile(
385
+ present: pd.Series,
386
+ frequencies: list[tuple[str, int]],
387
+ inferred_type: str,
388
+ config: AnalysisConfig,
389
+ ) -> StringProfile | None:
390
+ """Summarize text lengths and retain bounded examples for shorter values."""
391
+ if inferred_type != "text" or present.empty:
392
+ return None
393
+ lengths = present.str.len()
394
+ minimum = int(lengths.min())
395
+ maximum = int(lengths.max())
396
+ settings = config.string_analysis
397
+ if maximum <= settings.very_short_max_length:
398
+ status = "very_short"
399
+ elif maximum <= config.string_analysis.short_max_length:
400
+ status = "short"
401
+ elif maximum <= config.string_analysis.medium_max_length:
402
+ status = "medium"
403
+ elif maximum <= config.string_analysis.long_max_length:
404
+ status = "long"
405
+ else:
406
+ status = "very_long"
407
+ distribution = []
408
+ if maximum <= settings.length_distribution_max_length:
409
+ for length in sorted({int(value) for value in lengths}):
410
+ matching = [
411
+ (value, count) for value, count in frequencies if len(value) == length
412
+ ]
413
+ distribution.append(
414
+ StringLengthDistribution(
415
+ length=length,
416
+ count=sum(count for _, count in matching),
417
+ percent=percent(sum(count for _, count in matching), len(present)),
418
+ distinct_count=len(matching),
419
+ examples=[
420
+ StringLengthExample(value=value, count=count)
421
+ for value, count in matching[: settings.examples_per_length]
422
+ ],
423
+ )
424
+ )
425
+ return StringProfile(
426
+ status=status,
427
+ present_count=len(present),
428
+ minimum_length=minimum,
429
+ maximum_length=maximum,
430
+ mean_length=round(float(lengths.mean()), 2),
431
+ median_length=round(float(lengths.median()), 2),
432
+ distinct_length_count=int(lengths.nunique()),
433
+ fixed_length=minimum if minimum == maximum else None,
434
+ length_distribution=distribution,
435
+ )
436
+
437
+
438
+ def normalized_edit_distance(left: str, right: str) -> float:
439
+ """Return Levenshtein distance normalized to the longer string."""
440
+ if left == right:
441
+ return 0.0
442
+ if not left or not right:
443
+ return 1.0
444
+ if len(left) < len(right):
445
+ left, right = right, left
446
+ previous = list(range(len(right) + 1))
447
+ for row, left_char in enumerate(left, start=1):
448
+ current = [row]
449
+ for column, right_char in enumerate(right, start=1):
450
+ current.append(
451
+ min(
452
+ current[-1] + 1,
453
+ previous[column] + 1,
454
+ previous[column - 1] + (left_char != right_char),
455
+ )
456
+ )
457
+ previous = current
458
+ return previous[-1] / max(len(left), len(right))
459
+
460
+
461
+ def select_diverse_values(values: list[str], limit: int) -> list[str]:
462
+ """Select a deterministic farthest-first subset from sampled short strings."""
463
+ if len(values) <= limit:
464
+ return values
465
+ if limit == 1:
466
+ return values[:1]
467
+ distances: dict[tuple[int, int], float] = {}
468
+
469
+ def distance(left: int, right: int) -> float:
470
+ pair = (min(left, right), max(left, right))
471
+ if pair not in distances:
472
+ distances[pair] = normalized_edit_distance(values[left], values[right])
473
+ return distances[pair]
474
+
475
+ first, second = max(
476
+ combinations(range(len(values)), 2), key=lambda pair: distance(*pair)
477
+ )
478
+ selected = [first, second]
479
+ remaining = set(range(len(values))) - set(selected)
480
+ while len(selected) < limit and remaining:
481
+ next_index = max(
482
+ remaining,
483
+ key=lambda candidate: (
484
+ min(distance(candidate, chosen) for chosen in selected),
485
+ -candidate,
486
+ ),
487
+ )
488
+ selected.append(next_index)
489
+ remaining.remove(next_index)
490
+ return [values[index] for index in selected]
491
+
492
+
493
+ def build_value_profile(
494
+ frequencies: list[tuple[str, int]],
495
+ *,
496
+ inferred_type: str,
497
+ position: int,
498
+ config: AnalysisConfig,
499
+ ) -> ValueProfile:
500
+ """Build a bounded, reproducible value representation for the global JSON."""
501
+ settings = config.value_examples
502
+ if len(frequencies) <= settings.full_distribution_max_distinct:
503
+ return ValueProfile(
504
+ selection="complete",
505
+ sampled_distinct_count=len(frequencies),
506
+ values=[ValueOccurrence(value=value, count=count) for value, count in frequencies],
507
+ )
508
+
509
+ population = sorted(value for value, _ in frequencies)
510
+ sample_size = min(settings.candidate_sample_size, len(population))
511
+ candidates = random.Random(settings.random_seed + position).sample(
512
+ population, sample_size
513
+ )
514
+ counts = dict(frequencies)
515
+ lengths = sorted(len(value) for value in candidates)
516
+ percentile_index = max(
517
+ 0, math.ceil(settings.short_text_percentile * len(lengths)) - 1
518
+ )
519
+ short_text = (
520
+ inferred_type == "text"
521
+ and lengths[percentile_index] <= settings.short_text_max_length
522
+ )
523
+ if short_text:
524
+ short_candidates = [
525
+ value
526
+ for value in candidates
527
+ if len(value) <= settings.short_text_max_length
528
+ ]
529
+ selected = select_diverse_values(
530
+ short_candidates, settings.short_text_result_size
531
+ )
532
+ frequent = [
533
+ value for value, _ in frequencies[: settings.inline_display_size]
534
+ ]
535
+ selected = (frequent + [value for value in selected if value not in frequent])[
536
+ : settings.short_text_result_size
537
+ ]
538
+ return ValueProfile(
539
+ selection="diverse_sample",
540
+ sampled_distinct_count=sample_size,
541
+ values=sorted(
542
+ [ValueOccurrence(value=value, count=counts[value]) for value in selected],
543
+ key=lambda item: (-item.count, item.value),
544
+ ),
545
+ )
546
+
547
+ selected = candidates[: settings.long_text_result_size]
548
+ frequent = [value for value, _ in frequencies[: settings.inline_display_size]]
549
+ selected = (frequent + [value for value in selected if value not in frequent])[
550
+ : settings.long_text_result_size
551
+ ]
552
+ values = []
553
+ for value in selected:
554
+ truncated = len(value) > settings.long_text_truncate_at
555
+ display = (
556
+ value[: settings.long_text_truncate_at] + settings.truncation_suffix
557
+ if truncated
558
+ else value
559
+ )
560
+ values.append(
561
+ ValueOccurrence(value=display, count=counts[value], truncated=truncated)
562
+ )
563
+ return ValueProfile(
564
+ selection="random_sample",
565
+ sampled_distinct_count=sample_size,
566
+ values=sorted(values, key=lambda item: (-item.count, item.value)),
567
+ )
568
+
569
+
570
+ def infer_enum_candidate(
571
+ present: pd.Series,
572
+ *,
573
+ inferred_type: str,
574
+ row_count: int,
575
+ config: AnalysisConfig,
576
+ ) -> EnumCandidate | None:
577
+ """Classify low-cardinality text as an enum candidate, not a physical type."""
578
+ settings = config.enum_detection
579
+ if (
580
+ not settings.enabled
581
+ or inferred_type not in settings.eligible_types
582
+ or row_count < settings.minimum_row_count
583
+ ):
584
+ return None
585
+ observed = (
586
+ present.nunique()
587
+ if settings.case_sensitive
588
+ else present.str.casefold().nunique()
589
+ )
590
+ if not 0 < observed <= settings.maximum_distinct_values:
591
+ return None
592
+ coverage = percent(len(present), row_count)
593
+ return EnumCandidate(
594
+ observed_distinct_count=int(observed),
595
+ non_missing_count=len(present),
596
+ coverage_percent=coverage,
597
+ confidence=round(len(present) / row_count, 4) if row_count else 0.0,
598
+ )
599
+
600
+
601
+ def analyze_column(
602
+ values: pd.Series,
603
+ *,
604
+ name: str | None = None,
605
+ position: int = 1,
606
+ config: AnalysisConfig | None = None,
607
+ ) -> ColumnProfile:
608
+ """Profile one raw string column, also usable independently of CSV ingestion."""
609
+ config = config or AnalysisConfig()
610
+ if not all(isinstance(value, str) for value in values):
611
+ raise ValueError("Column analysis expects raw strings, without null objects.")
612
+ normalized, normalization = normalize_values(values, config)
613
+ absent = missing_mask(normalized, config)
614
+ present = normalized[~absent]
615
+ frequencies = sorted(
616
+ ((str(value), int(count)) for value, count in present.value_counts().items()),
617
+ key=lambda item: (-item[1], item[0]),
618
+ )
619
+ date_profile, valid_date_values = build_date_profile(frequencies, config)
620
+ counts: Counter[str] = Counter()
621
+ for value, count in frequencies:
622
+ counts["date" if value in valid_date_values else value_type(value)] += count
623
+ inferred, semantic_type, type_confidence, type_error_count, type_error_percent = (
624
+ infer_column_type(counts, date_profile, len(present), config)
625
+ )
626
+
627
+ numeric = None
628
+ if inferred in {"integer", "number"}:
629
+ accepted_types = {"integer"} if inferred == "integer" else {"integer", "number"}
630
+ numeric_values = present[present.map(lambda value: value_type(value) in accepted_types)]
631
+ numbers = pd.to_numeric(numeric_values, errors="coerce").astype(float)
632
+ stats = [numbers.min(), numbers.max(), numbers.mean(), numbers.median()]
633
+ if numbers.notna().all() and all(math.isfinite(value) for value in stats):
634
+ numeric = NumericStats(
635
+ minimum=stats[0], maximum=stats[1], mean=stats[2], median=stats[3]
636
+ )
637
+ value_profile = build_value_profile(
638
+ frequencies,
639
+ inferred_type=inferred,
640
+ position=position,
641
+ config=config,
642
+ )
643
+ enum = infer_enum_candidate(
644
+ present,
645
+ inferred_type=inferred,
646
+ row_count=len(values),
647
+ config=config,
648
+ )
649
+ if enum and semantic_type is None:
650
+ semantic_type = "enum"
651
+ string_profile = build_string_profile(present, frequencies, inferred, config)
652
+ return ColumnProfile(
653
+ id=f"column_{position}",
654
+ name=name if name is not None else str(values.name or ""),
655
+ position=position,
656
+ inferred_type=inferred,
657
+ type_counts=dict(counts),
658
+ type_confidence=round(type_confidence, 4),
659
+ type_error_count=type_error_count,
660
+ type_error_percent=type_error_percent,
661
+ missing_count=int(absent.sum()),
662
+ missing_percent=percent(int(absent.sum()), len(values)),
663
+ normalization=normalization,
664
+ distinct_count=len(frequencies),
665
+ examples=[
666
+ item.value
667
+ for item in value_profile.values[: config.value_examples.inline_display_size]
668
+ ],
669
+ value_profile=value_profile,
670
+ semantic_type=semantic_type,
671
+ enum=enum,
672
+ date_profile=date_profile,
673
+ string_profile=string_profile,
674
+ numeric=numeric,
675
+ )
676
+
677
+
678
+ def analyze_dataset(dataset: CsvDataset, config: AnalysisConfig) -> DatasetProfile:
679
+ frame = dataset.frame
680
+ columns = [
681
+ analyze_column(frame.iloc[:, i], name=name, position=i + 1, config=config)
682
+ for i, name in enumerate(dataset.headers)
683
+ ]
684
+ absent = frame.apply(lambda values: missing_mask(values, config))
685
+ duplicates = frame.duplicated(keep="first")
686
+ missing_count = sum(column.missing_count for column in columns)
687
+ trim_count = sum(column.normalization.trim_count for column in columns)
688
+ collapse_count = sum(
689
+ column.normalization.collapse_internal_whitespace_count for column in columns
690
+ )
691
+ empty = [column.id for column in columns if column.inferred_type == "empty"]
692
+ constant = [column.id for column in columns if column.distinct_count == 1]
693
+ mixed = [column.id for column in columns if column.inferred_type == "mixed"]
694
+ issues = []
695
+
696
+ def add_issue(
697
+ code, message, count, ids=(), rows=(), severity="warning", always=False
698
+ ):
699
+ if count or always:
700
+ issues.append(
701
+ Issue(
702
+ code=code,
703
+ severity=severity,
704
+ message=message,
705
+ count=count,
706
+ column_ids=list(ids),
707
+ row_numbers=list(islice(rows, 10)),
708
+ )
709
+ )
710
+
711
+ add_issue(
712
+ "duplicate_rows",
713
+ "Duplicate rows beyond their first occurrence",
714
+ int(duplicates.sum()),
715
+ rows=(i + 1 for i, value in enumerate(duplicates) if value),
716
+ )
717
+ add_issue(
718
+ "missing_values",
719
+ "Missing cells",
720
+ missing_count,
721
+ [c.id for c in columns if c.missing_count],
722
+ (i + 1 for i, value in enumerate(absent.any(axis=1)) if value),
723
+ )
724
+ add_issue("empty_columns", "Columns without any present values", len(empty), empty)
725
+ add_issue(
726
+ "trimmed_cells",
727
+ "Cells changed by trimming surrounding whitespace",
728
+ trim_count,
729
+ [c.id for c in columns if c.normalization.trim_count],
730
+ severity="info",
731
+ always=True,
732
+ )
733
+ add_issue(
734
+ "collapsed_whitespace",
735
+ "Cells changed by collapsing repeated internal whitespace",
736
+ collapse_count,
737
+ [c.id for c in columns if c.normalization.collapse_internal_whitespace_count],
738
+ severity="info",
739
+ always=True,
740
+ )
741
+ add_issue(
742
+ "constant_columns",
743
+ "Columns with one distinct present value",
744
+ len(constant),
745
+ constant,
746
+ severity="info",
747
+ )
748
+ add_issue("mixed_types", "Columns with mixed value types", len(mixed), mixed)
749
+ header_counts = Counter(dataset.headers)
750
+ bad_headers = [
751
+ c.id for c in columns if not c.name.strip() or header_counts[c.name] > 1
752
+ ]
753
+ add_issue(
754
+ "ambiguous_headers",
755
+ "Blank or repeated column names",
756
+ len(bad_headers),
757
+ bad_headers,
758
+ )
759
+
760
+ return DatasetProfile(
761
+ generated_at=datetime.now(UTC),
762
+ processing_seconds=0.0,
763
+ source=dataset.source,
764
+ config=config,
765
+ summary=DatasetSummary(
766
+ row_count=len(frame),
767
+ column_count=len(columns),
768
+ cell_count=frame.size,
769
+ missing_count=missing_count,
770
+ missing_percent=percent(missing_count, frame.size),
771
+ trim_count=trim_count,
772
+ collapse_internal_whitespace_count=collapse_count,
773
+ duplicate_row_count=int(duplicates.sum()),
774
+ empty_row_count=int(absent.all(axis=1).sum()),
775
+ empty_column_count=len(empty),
776
+ constant_column_count=len(constant),
777
+ ),
778
+ columns=columns,
779
+ issues=issues,
780
+ preview=[
781
+ PreviewRow(row_number=i + 1, values=list(row))
782
+ for i, row in enumerate(
783
+ frame.head(config.preview_rows).itertuples(index=False, name=None)
784
+ )
785
+ ],
786
+ )