tabalyst 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tabalyst/__init__.py +20 -0
- tabalyst/__main__.py +6 -0
- tabalyst/_version.py +23 -0
- tabalyst/analysis.py +786 -0
- tabalyst/cli.py +208 -0
- tabalyst/config.py +214 -0
- tabalyst/errors.py +17 -0
- tabalyst/execution_log.py +102 -0
- tabalyst/ingestion.py +86 -0
- tabalyst/models.py +174 -0
- tabalyst/reporting.py +54 -0
- tabalyst/service.py +134 -0
- tabalyst/static/report.js +500 -0
- tabalyst/static/theme.css +411 -0
- tabalyst/templates/report.html +318 -0
- tabalyst-0.1.0.dist-info/METADATA +208 -0
- tabalyst-0.1.0.dist-info/RECORD +21 -0
- tabalyst-0.1.0.dist-info/WHEEL +5 -0
- tabalyst-0.1.0.dist-info/entry_points.txt +2 -0
- tabalyst-0.1.0.dist-info/licenses/LICENSE +21 -0
- tabalyst-0.1.0.dist-info/top_level.txt +1 -0
tabalyst/analysis.py
ADDED
|
@@ -0,0 +1,786 @@
|
|
|
1
|
+
"""Pure column and dataset analysis. Raw cells are never rewritten."""
|
|
2
|
+
|
|
3
|
+
import math
|
|
4
|
+
import random
|
|
5
|
+
import re
|
|
6
|
+
from collections import Counter
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from datetime import UTC, date, datetime
|
|
9
|
+
from itertools import combinations, islice
|
|
10
|
+
|
|
11
|
+
import pandas as pd
|
|
12
|
+
|
|
13
|
+
from tabalyst.config import AnalysisConfig
|
|
14
|
+
from tabalyst.ingestion import CsvDataset
|
|
15
|
+
from tabalyst.models import (
|
|
16
|
+
ColumnProfile,
|
|
17
|
+
DatasetProfile,
|
|
18
|
+
DatasetSummary,
|
|
19
|
+
DateBreakdownItem,
|
|
20
|
+
DateFormatCount,
|
|
21
|
+
DateProfile,
|
|
22
|
+
EnumCandidate,
|
|
23
|
+
Issue,
|
|
24
|
+
NormalizationStats,
|
|
25
|
+
NumericStats,
|
|
26
|
+
PreviewRow,
|
|
27
|
+
StringLengthDistribution,
|
|
28
|
+
StringLengthExample,
|
|
29
|
+
StringProfile,
|
|
30
|
+
ValueOccurrence,
|
|
31
|
+
ValueProfile,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
INTEGER = re.compile(r"[+-]?(?:0|[1-9][0-9]*)")
|
|
35
|
+
NUMBER = re.compile(
|
|
36
|
+
r"[+-]?(?:(?:0|[1-9][0-9]*)(?:\.[0-9]*)?|\.[0-9]+)(?:[eE][+-]?[0-9]+)?"
|
|
37
|
+
)
|
|
38
|
+
INTERNAL_HORIZONTAL_WHITESPACE = re.compile(r"[^\S\r\n]+")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def percent(count: int, total: int) -> float:
|
|
42
|
+
return round(100 * count / total, 2) if total else 0.0
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def value_type(value: str) -> str:
|
|
46
|
+
if value.casefold() in {"true", "false"}:
|
|
47
|
+
return "boolean"
|
|
48
|
+
if INTEGER.fullmatch(value):
|
|
49
|
+
return "integer"
|
|
50
|
+
if NUMBER.fullmatch(value):
|
|
51
|
+
return "number"
|
|
52
|
+
return "text"
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def missing_mask(values: pd.Series, config: AnalysisConfig) -> pd.Series:
|
|
56
|
+
return values.str.strip().isin({marker.strip() for marker in config.missing_values})
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def normalize_values(
|
|
60
|
+
values: pd.Series, config: AnalysisConfig
|
|
61
|
+
) -> tuple[pd.Series, NormalizationStats]:
|
|
62
|
+
"""Normalize analysis values while counting each changed cell per operation."""
|
|
63
|
+
normalized = values
|
|
64
|
+
trim_count = 0
|
|
65
|
+
collapse_count = 0
|
|
66
|
+
if config.normalization.trim:
|
|
67
|
+
trimmed = normalized.str.strip()
|
|
68
|
+
trim_count = int((trimmed != normalized).sum())
|
|
69
|
+
normalized = trimmed
|
|
70
|
+
if config.normalization.collapse_internal_whitespace:
|
|
71
|
+
collapsed = normalized.str.replace(
|
|
72
|
+
INTERNAL_HORIZONTAL_WHITESPACE, " ", regex=True
|
|
73
|
+
)
|
|
74
|
+
collapse_count = int((collapsed != normalized).sum())
|
|
75
|
+
normalized = collapsed
|
|
76
|
+
return normalized, NormalizationStats(
|
|
77
|
+
trim_count=trim_count,
|
|
78
|
+
trim_percent=percent(trim_count, len(values)),
|
|
79
|
+
collapse_internal_whitespace_count=collapse_count,
|
|
80
|
+
collapse_internal_whitespace_percent=percent(collapse_count, len(values)),
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
@dataclass(frozen=True)
|
|
85
|
+
class DateValueResult:
|
|
86
|
+
state: str
|
|
87
|
+
order: str | None = None
|
|
88
|
+
separator: str | None = None
|
|
89
|
+
format: str | None = None
|
|
90
|
+
possible_orders: tuple[str, ...] = ()
|
|
91
|
+
possible_formats: tuple[str, ...] = ()
|
|
92
|
+
error: str | None = None
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def valid_calendar_date(order: str, first: int, second: int, third: int) -> bool:
|
|
96
|
+
if order == "YMD":
|
|
97
|
+
year, month, day = first, second, third
|
|
98
|
+
elif order == "MDY":
|
|
99
|
+
month, day, year = first, second, third
|
|
100
|
+
else:
|
|
101
|
+
day, month, year = first, second, third
|
|
102
|
+
try:
|
|
103
|
+
date(year, month, day)
|
|
104
|
+
except ValueError:
|
|
105
|
+
return False
|
|
106
|
+
return True
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def date_format_name(
|
|
110
|
+
order: str,
|
|
111
|
+
first: str,
|
|
112
|
+
second: str,
|
|
113
|
+
third: str,
|
|
114
|
+
separator: str,
|
|
115
|
+
) -> str:
|
|
116
|
+
"""Describe component order, separator and zero-padding explicitly."""
|
|
117
|
+
month = "MM" if len(second if order == "YMD" else first) == 2 else "M"
|
|
118
|
+
if order == "YMD":
|
|
119
|
+
day = "DD" if len(third) == 2 else "D"
|
|
120
|
+
parts = ("YYYY", month, day)
|
|
121
|
+
elif order == "MDY":
|
|
122
|
+
day = "DD" if len(second) == 2 else "D"
|
|
123
|
+
parts = (month, day, "YYYY")
|
|
124
|
+
else:
|
|
125
|
+
day = "DD" if len(first) == 2 else "D"
|
|
126
|
+
month = "MM" if len(second) == 2 else "M"
|
|
127
|
+
parts = (day, month, "YYYY")
|
|
128
|
+
return separator.join(parts)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def classify_date_value(value: str, config: AnalysisConfig) -> DateValueResult:
|
|
132
|
+
"""Recognize configured numeric date structures without fuzzy interpretation."""
|
|
133
|
+
settings = config.date_detection
|
|
134
|
+
if not settings.enabled:
|
|
135
|
+
return DateValueResult("not_date")
|
|
136
|
+
separators = "".join(re.escape(separator) for separator in settings.separators)
|
|
137
|
+
match = re.fullmatch(
|
|
138
|
+
rf"(?P<a>\d{{1,4}})(?P<s1>[{separators}])(?P<b>\d{{1,4}})"
|
|
139
|
+
rf"(?P<s2>[{separators}])(?P<c>\d{{1,4}})",
|
|
140
|
+
value,
|
|
141
|
+
)
|
|
142
|
+
if not match:
|
|
143
|
+
return DateValueResult("not_date")
|
|
144
|
+
first_text, second_text, third_text = (
|
|
145
|
+
match.group("a"),
|
|
146
|
+
match.group("b"),
|
|
147
|
+
match.group("c"),
|
|
148
|
+
)
|
|
149
|
+
first, second, third = map(int, (first_text, second_text, third_text))
|
|
150
|
+
separator = match.group("s1")
|
|
151
|
+
year_first = len(first_text) == 4
|
|
152
|
+
year_last = len(third_text) == 4
|
|
153
|
+
if year_first and len(second_text) <= 2 and len(third_text) <= 2:
|
|
154
|
+
possible_orders = ("YMD",)
|
|
155
|
+
elif year_last and len(first_text) <= 2 and len(second_text) <= 2:
|
|
156
|
+
possible_orders = ("MDY", "DMY")
|
|
157
|
+
elif year_first:
|
|
158
|
+
return DateValueResult("invalid", error="invalid_component_width")
|
|
159
|
+
else:
|
|
160
|
+
return DateValueResult("not_date")
|
|
161
|
+
if match.group("s1") != match.group("s2"):
|
|
162
|
+
return DateValueResult("invalid", error="mixed_separators")
|
|
163
|
+
allowed_orders = [order for order in possible_orders if order in settings.orders]
|
|
164
|
+
if not allowed_orders:
|
|
165
|
+
return DateValueResult("invalid", error="unsupported_order")
|
|
166
|
+
valid_orders = tuple(
|
|
167
|
+
order
|
|
168
|
+
for order in allowed_orders
|
|
169
|
+
if valid_calendar_date(order, first, second, third)
|
|
170
|
+
)
|
|
171
|
+
if not valid_orders:
|
|
172
|
+
return DateValueResult("invalid", error="invalid_calendar_date")
|
|
173
|
+
if len(valid_orders) == 1:
|
|
174
|
+
order = valid_orders[0]
|
|
175
|
+
return DateValueResult(
|
|
176
|
+
"valid",
|
|
177
|
+
order=order,
|
|
178
|
+
separator=separator,
|
|
179
|
+
format=date_format_name(
|
|
180
|
+
order, first_text, second_text, third_text, separator
|
|
181
|
+
),
|
|
182
|
+
)
|
|
183
|
+
return DateValueResult(
|
|
184
|
+
"ambiguous",
|
|
185
|
+
separator=separator,
|
|
186
|
+
possible_orders=valid_orders,
|
|
187
|
+
possible_formats=tuple(
|
|
188
|
+
date_format_name(
|
|
189
|
+
order, first_text, second_text, third_text, separator
|
|
190
|
+
)
|
|
191
|
+
for order in valid_orders
|
|
192
|
+
),
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def build_date_profile(
|
|
197
|
+
frequencies: list[tuple[str, int]], config: AnalysisConfig
|
|
198
|
+
) -> tuple[DateProfile | None, set[str]]:
|
|
199
|
+
"""Summarize strict date formats and resolve ambiguity only from clear evidence."""
|
|
200
|
+
if not config.date_detection.enabled:
|
|
201
|
+
return None, set()
|
|
202
|
+
classified = {
|
|
203
|
+
value: classify_date_value(value, config) for value, _ in frequencies
|
|
204
|
+
}
|
|
205
|
+
if not any(result.state != "not_date" for result in classified.values()):
|
|
206
|
+
return None, set()
|
|
207
|
+
|
|
208
|
+
evidence = Counter()
|
|
209
|
+
for value, count in frequencies:
|
|
210
|
+
result = classified[value]
|
|
211
|
+
if result.state == "valid" and result.order in {"MDY", "DMY"}:
|
|
212
|
+
evidence[result.order] += count
|
|
213
|
+
resolved_order = config.date_detection.ambiguous_order
|
|
214
|
+
resolution_source = "config" if resolved_order else None
|
|
215
|
+
if not resolved_order:
|
|
216
|
+
if evidence["MDY"] and not evidence["DMY"]:
|
|
217
|
+
resolved_order = "MDY"
|
|
218
|
+
resolution_source = "column"
|
|
219
|
+
elif evidence["DMY"] and not evidence["MDY"]:
|
|
220
|
+
resolved_order = "DMY"
|
|
221
|
+
resolution_source = "column"
|
|
222
|
+
|
|
223
|
+
valid_values: set[str] = set()
|
|
224
|
+
formats: Counter[tuple[str, str, str]] = Counter()
|
|
225
|
+
ambiguous_formats: Counter[str] = Counter()
|
|
226
|
+
errors: Counter[str] = Counter()
|
|
227
|
+
valid_count = 0
|
|
228
|
+
ambiguous_count = 0
|
|
229
|
+
invalid_count = 0
|
|
230
|
+
not_date_count = 0
|
|
231
|
+
for value, count in frequencies:
|
|
232
|
+
result = classified[value]
|
|
233
|
+
order = result.order
|
|
234
|
+
format_name = result.format
|
|
235
|
+
if (
|
|
236
|
+
result.state == "ambiguous"
|
|
237
|
+
and resolved_order in result.possible_orders
|
|
238
|
+
):
|
|
239
|
+
order = resolved_order
|
|
240
|
+
format_name = result.possible_formats[
|
|
241
|
+
result.possible_orders.index(str(resolved_order))
|
|
242
|
+
]
|
|
243
|
+
if result.state == "valid" or order:
|
|
244
|
+
valid_values.add(value)
|
|
245
|
+
valid_count += count
|
|
246
|
+
formats[(str(format_name), str(order), str(result.separator))] += count
|
|
247
|
+
elif result.state == "ambiguous":
|
|
248
|
+
ambiguous_count += count
|
|
249
|
+
ambiguous_formats[
|
|
250
|
+
"Ambiguous: " + " or ".join(result.possible_formats)
|
|
251
|
+
] += count
|
|
252
|
+
elif result.state == "invalid":
|
|
253
|
+
invalid_count += count
|
|
254
|
+
errors[str(result.error)] += count
|
|
255
|
+
else:
|
|
256
|
+
not_date_count += count
|
|
257
|
+
|
|
258
|
+
total = sum(count for _, count in frequencies)
|
|
259
|
+
if valid_count == total:
|
|
260
|
+
status = "valid" if len(formats) == 1 else "multiple_formats"
|
|
261
|
+
elif ambiguous_count == total:
|
|
262
|
+
status = "ambiguous"
|
|
263
|
+
elif invalid_count == total:
|
|
264
|
+
status = "invalid"
|
|
265
|
+
else:
|
|
266
|
+
status = "mixed"
|
|
267
|
+
format_items = [
|
|
268
|
+
DateFormatCount(
|
|
269
|
+
format=format_name,
|
|
270
|
+
order=order,
|
|
271
|
+
separator=separator,
|
|
272
|
+
count=count,
|
|
273
|
+
percent=percent(count, total),
|
|
274
|
+
iso=format_name == "YYYY-MM-DD",
|
|
275
|
+
)
|
|
276
|
+
for (format_name, order, separator), count in sorted(
|
|
277
|
+
formats.items(), key=lambda item: (-item[1], item[0])
|
|
278
|
+
)
|
|
279
|
+
]
|
|
280
|
+
date_breakdown = [
|
|
281
|
+
DateBreakdownItem(
|
|
282
|
+
label=item.format,
|
|
283
|
+
category="valid",
|
|
284
|
+
count=item.count,
|
|
285
|
+
percent=item.percent,
|
|
286
|
+
)
|
|
287
|
+
for item in format_items
|
|
288
|
+
]
|
|
289
|
+
date_breakdown.extend(
|
|
290
|
+
DateBreakdownItem(
|
|
291
|
+
label=label,
|
|
292
|
+
category="ambiguous",
|
|
293
|
+
count=count,
|
|
294
|
+
percent=percent(count, total),
|
|
295
|
+
)
|
|
296
|
+
for label, count in ambiguous_formats.items()
|
|
297
|
+
)
|
|
298
|
+
date_breakdown.sort(key=lambda item: (-item.count, item.label))
|
|
299
|
+
breakdown = date_breakdown + [
|
|
300
|
+
DateBreakdownItem(
|
|
301
|
+
label="Invalid date",
|
|
302
|
+
category="invalid",
|
|
303
|
+
count=invalid_count,
|
|
304
|
+
percent=percent(invalid_count, total),
|
|
305
|
+
),
|
|
306
|
+
DateBreakdownItem(
|
|
307
|
+
label="Not a date",
|
|
308
|
+
category="not_date",
|
|
309
|
+
count=not_date_count,
|
|
310
|
+
percent=percent(not_date_count, total),
|
|
311
|
+
),
|
|
312
|
+
]
|
|
313
|
+
return (
|
|
314
|
+
DateProfile(
|
|
315
|
+
status=status,
|
|
316
|
+
valid_count=valid_count,
|
|
317
|
+
ambiguous_count=ambiguous_count,
|
|
318
|
+
invalid_date_count=invalid_count,
|
|
319
|
+
not_date_count=not_date_count,
|
|
320
|
+
resolved_ambiguous_order=resolved_order,
|
|
321
|
+
ambiguous_order_source=resolution_source,
|
|
322
|
+
formats=format_items,
|
|
323
|
+
format_count=len(format_items),
|
|
324
|
+
breakdown=breakdown,
|
|
325
|
+
errors=dict(errors),
|
|
326
|
+
),
|
|
327
|
+
valid_values,
|
|
328
|
+
)
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def infer_column_type(
|
|
332
|
+
counts: Counter[str],
|
|
333
|
+
date_profile: DateProfile | None,
|
|
334
|
+
total: int,
|
|
335
|
+
config: AnalysisConfig,
|
|
336
|
+
) -> tuple[str, str | None, float, int | None, float | None]:
|
|
337
|
+
"""Infer a dominant type and quantify values outside the accepted family."""
|
|
338
|
+
if not total:
|
|
339
|
+
return "empty", None, 1.0, 0, 0.0
|
|
340
|
+
threshold = config.type_inference.minimum_confidence
|
|
341
|
+
candidates = [
|
|
342
|
+
("integer", counts["integer"]),
|
|
343
|
+
("number", counts["integer"] + counts["number"]),
|
|
344
|
+
]
|
|
345
|
+
for inferred_type, accepted in candidates:
|
|
346
|
+
confidence = accepted / total
|
|
347
|
+
if confidence >= threshold:
|
|
348
|
+
errors = total - accepted
|
|
349
|
+
return inferred_type, None, confidence, errors, percent(errors, total)
|
|
350
|
+
|
|
351
|
+
date_accepted = (
|
|
352
|
+
date_profile.valid_count + date_profile.ambiguous_count
|
|
353
|
+
if date_profile
|
|
354
|
+
else 0
|
|
355
|
+
)
|
|
356
|
+
date_confidence = date_accepted / total
|
|
357
|
+
if date_profile and date_confidence >= threshold:
|
|
358
|
+
inferred_type = (
|
|
359
|
+
"date"
|
|
360
|
+
if date_profile.format_count == 1
|
|
361
|
+
and date_profile.ambiguous_count == 0
|
|
362
|
+
else "mixed"
|
|
363
|
+
)
|
|
364
|
+
errors = total - date_profile.valid_count
|
|
365
|
+
return inferred_type, "date", date_confidence, errors, percent(errors, total)
|
|
366
|
+
|
|
367
|
+
for inferred_type in ("boolean", "text"):
|
|
368
|
+
accepted = counts[inferred_type]
|
|
369
|
+
confidence = accepted / total
|
|
370
|
+
if confidence >= threshold:
|
|
371
|
+
errors = total - accepted
|
|
372
|
+
return inferred_type, None, confidence, errors, percent(errors, total)
|
|
373
|
+
|
|
374
|
+
family_counts = [
|
|
375
|
+
counts["integer"],
|
|
376
|
+
counts["integer"] + counts["number"],
|
|
377
|
+
date_accepted,
|
|
378
|
+
counts["boolean"],
|
|
379
|
+
counts["text"],
|
|
380
|
+
]
|
|
381
|
+
return "mixed", None, max(family_counts) / total, None, None
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def build_string_profile(
|
|
385
|
+
present: pd.Series,
|
|
386
|
+
frequencies: list[tuple[str, int]],
|
|
387
|
+
inferred_type: str,
|
|
388
|
+
config: AnalysisConfig,
|
|
389
|
+
) -> StringProfile | None:
|
|
390
|
+
"""Summarize text lengths and retain bounded examples for shorter values."""
|
|
391
|
+
if inferred_type != "text" or present.empty:
|
|
392
|
+
return None
|
|
393
|
+
lengths = present.str.len()
|
|
394
|
+
minimum = int(lengths.min())
|
|
395
|
+
maximum = int(lengths.max())
|
|
396
|
+
settings = config.string_analysis
|
|
397
|
+
if maximum <= settings.very_short_max_length:
|
|
398
|
+
status = "very_short"
|
|
399
|
+
elif maximum <= config.string_analysis.short_max_length:
|
|
400
|
+
status = "short"
|
|
401
|
+
elif maximum <= config.string_analysis.medium_max_length:
|
|
402
|
+
status = "medium"
|
|
403
|
+
elif maximum <= config.string_analysis.long_max_length:
|
|
404
|
+
status = "long"
|
|
405
|
+
else:
|
|
406
|
+
status = "very_long"
|
|
407
|
+
distribution = []
|
|
408
|
+
if maximum <= settings.length_distribution_max_length:
|
|
409
|
+
for length in sorted({int(value) for value in lengths}):
|
|
410
|
+
matching = [
|
|
411
|
+
(value, count) for value, count in frequencies if len(value) == length
|
|
412
|
+
]
|
|
413
|
+
distribution.append(
|
|
414
|
+
StringLengthDistribution(
|
|
415
|
+
length=length,
|
|
416
|
+
count=sum(count for _, count in matching),
|
|
417
|
+
percent=percent(sum(count for _, count in matching), len(present)),
|
|
418
|
+
distinct_count=len(matching),
|
|
419
|
+
examples=[
|
|
420
|
+
StringLengthExample(value=value, count=count)
|
|
421
|
+
for value, count in matching[: settings.examples_per_length]
|
|
422
|
+
],
|
|
423
|
+
)
|
|
424
|
+
)
|
|
425
|
+
return StringProfile(
|
|
426
|
+
status=status,
|
|
427
|
+
present_count=len(present),
|
|
428
|
+
minimum_length=minimum,
|
|
429
|
+
maximum_length=maximum,
|
|
430
|
+
mean_length=round(float(lengths.mean()), 2),
|
|
431
|
+
median_length=round(float(lengths.median()), 2),
|
|
432
|
+
distinct_length_count=int(lengths.nunique()),
|
|
433
|
+
fixed_length=minimum if minimum == maximum else None,
|
|
434
|
+
length_distribution=distribution,
|
|
435
|
+
)
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
def normalized_edit_distance(left: str, right: str) -> float:
|
|
439
|
+
"""Return Levenshtein distance normalized to the longer string."""
|
|
440
|
+
if left == right:
|
|
441
|
+
return 0.0
|
|
442
|
+
if not left or not right:
|
|
443
|
+
return 1.0
|
|
444
|
+
if len(left) < len(right):
|
|
445
|
+
left, right = right, left
|
|
446
|
+
previous = list(range(len(right) + 1))
|
|
447
|
+
for row, left_char in enumerate(left, start=1):
|
|
448
|
+
current = [row]
|
|
449
|
+
for column, right_char in enumerate(right, start=1):
|
|
450
|
+
current.append(
|
|
451
|
+
min(
|
|
452
|
+
current[-1] + 1,
|
|
453
|
+
previous[column] + 1,
|
|
454
|
+
previous[column - 1] + (left_char != right_char),
|
|
455
|
+
)
|
|
456
|
+
)
|
|
457
|
+
previous = current
|
|
458
|
+
return previous[-1] / max(len(left), len(right))
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def select_diverse_values(values: list[str], limit: int) -> list[str]:
|
|
462
|
+
"""Select a deterministic farthest-first subset from sampled short strings."""
|
|
463
|
+
if len(values) <= limit:
|
|
464
|
+
return values
|
|
465
|
+
if limit == 1:
|
|
466
|
+
return values[:1]
|
|
467
|
+
distances: dict[tuple[int, int], float] = {}
|
|
468
|
+
|
|
469
|
+
def distance(left: int, right: int) -> float:
|
|
470
|
+
pair = (min(left, right), max(left, right))
|
|
471
|
+
if pair not in distances:
|
|
472
|
+
distances[pair] = normalized_edit_distance(values[left], values[right])
|
|
473
|
+
return distances[pair]
|
|
474
|
+
|
|
475
|
+
first, second = max(
|
|
476
|
+
combinations(range(len(values)), 2), key=lambda pair: distance(*pair)
|
|
477
|
+
)
|
|
478
|
+
selected = [first, second]
|
|
479
|
+
remaining = set(range(len(values))) - set(selected)
|
|
480
|
+
while len(selected) < limit and remaining:
|
|
481
|
+
next_index = max(
|
|
482
|
+
remaining,
|
|
483
|
+
key=lambda candidate: (
|
|
484
|
+
min(distance(candidate, chosen) for chosen in selected),
|
|
485
|
+
-candidate,
|
|
486
|
+
),
|
|
487
|
+
)
|
|
488
|
+
selected.append(next_index)
|
|
489
|
+
remaining.remove(next_index)
|
|
490
|
+
return [values[index] for index in selected]
|
|
491
|
+
|
|
492
|
+
|
|
493
|
+
def build_value_profile(
|
|
494
|
+
frequencies: list[tuple[str, int]],
|
|
495
|
+
*,
|
|
496
|
+
inferred_type: str,
|
|
497
|
+
position: int,
|
|
498
|
+
config: AnalysisConfig,
|
|
499
|
+
) -> ValueProfile:
|
|
500
|
+
"""Build a bounded, reproducible value representation for the global JSON."""
|
|
501
|
+
settings = config.value_examples
|
|
502
|
+
if len(frequencies) <= settings.full_distribution_max_distinct:
|
|
503
|
+
return ValueProfile(
|
|
504
|
+
selection="complete",
|
|
505
|
+
sampled_distinct_count=len(frequencies),
|
|
506
|
+
values=[ValueOccurrence(value=value, count=count) for value, count in frequencies],
|
|
507
|
+
)
|
|
508
|
+
|
|
509
|
+
population = sorted(value for value, _ in frequencies)
|
|
510
|
+
sample_size = min(settings.candidate_sample_size, len(population))
|
|
511
|
+
candidates = random.Random(settings.random_seed + position).sample(
|
|
512
|
+
population, sample_size
|
|
513
|
+
)
|
|
514
|
+
counts = dict(frequencies)
|
|
515
|
+
lengths = sorted(len(value) for value in candidates)
|
|
516
|
+
percentile_index = max(
|
|
517
|
+
0, math.ceil(settings.short_text_percentile * len(lengths)) - 1
|
|
518
|
+
)
|
|
519
|
+
short_text = (
|
|
520
|
+
inferred_type == "text"
|
|
521
|
+
and lengths[percentile_index] <= settings.short_text_max_length
|
|
522
|
+
)
|
|
523
|
+
if short_text:
|
|
524
|
+
short_candidates = [
|
|
525
|
+
value
|
|
526
|
+
for value in candidates
|
|
527
|
+
if len(value) <= settings.short_text_max_length
|
|
528
|
+
]
|
|
529
|
+
selected = select_diverse_values(
|
|
530
|
+
short_candidates, settings.short_text_result_size
|
|
531
|
+
)
|
|
532
|
+
frequent = [
|
|
533
|
+
value for value, _ in frequencies[: settings.inline_display_size]
|
|
534
|
+
]
|
|
535
|
+
selected = (frequent + [value for value in selected if value not in frequent])[
|
|
536
|
+
: settings.short_text_result_size
|
|
537
|
+
]
|
|
538
|
+
return ValueProfile(
|
|
539
|
+
selection="diverse_sample",
|
|
540
|
+
sampled_distinct_count=sample_size,
|
|
541
|
+
values=sorted(
|
|
542
|
+
[ValueOccurrence(value=value, count=counts[value]) for value in selected],
|
|
543
|
+
key=lambda item: (-item.count, item.value),
|
|
544
|
+
),
|
|
545
|
+
)
|
|
546
|
+
|
|
547
|
+
selected = candidates[: settings.long_text_result_size]
|
|
548
|
+
frequent = [value for value, _ in frequencies[: settings.inline_display_size]]
|
|
549
|
+
selected = (frequent + [value for value in selected if value not in frequent])[
|
|
550
|
+
: settings.long_text_result_size
|
|
551
|
+
]
|
|
552
|
+
values = []
|
|
553
|
+
for value in selected:
|
|
554
|
+
truncated = len(value) > settings.long_text_truncate_at
|
|
555
|
+
display = (
|
|
556
|
+
value[: settings.long_text_truncate_at] + settings.truncation_suffix
|
|
557
|
+
if truncated
|
|
558
|
+
else value
|
|
559
|
+
)
|
|
560
|
+
values.append(
|
|
561
|
+
ValueOccurrence(value=display, count=counts[value], truncated=truncated)
|
|
562
|
+
)
|
|
563
|
+
return ValueProfile(
|
|
564
|
+
selection="random_sample",
|
|
565
|
+
sampled_distinct_count=sample_size,
|
|
566
|
+
values=sorted(values, key=lambda item: (-item.count, item.value)),
|
|
567
|
+
)
|
|
568
|
+
|
|
569
|
+
|
|
570
|
+
def infer_enum_candidate(
|
|
571
|
+
present: pd.Series,
|
|
572
|
+
*,
|
|
573
|
+
inferred_type: str,
|
|
574
|
+
row_count: int,
|
|
575
|
+
config: AnalysisConfig,
|
|
576
|
+
) -> EnumCandidate | None:
|
|
577
|
+
"""Classify low-cardinality text as an enum candidate, not a physical type."""
|
|
578
|
+
settings = config.enum_detection
|
|
579
|
+
if (
|
|
580
|
+
not settings.enabled
|
|
581
|
+
or inferred_type not in settings.eligible_types
|
|
582
|
+
or row_count < settings.minimum_row_count
|
|
583
|
+
):
|
|
584
|
+
return None
|
|
585
|
+
observed = (
|
|
586
|
+
present.nunique()
|
|
587
|
+
if settings.case_sensitive
|
|
588
|
+
else present.str.casefold().nunique()
|
|
589
|
+
)
|
|
590
|
+
if not 0 < observed <= settings.maximum_distinct_values:
|
|
591
|
+
return None
|
|
592
|
+
coverage = percent(len(present), row_count)
|
|
593
|
+
return EnumCandidate(
|
|
594
|
+
observed_distinct_count=int(observed),
|
|
595
|
+
non_missing_count=len(present),
|
|
596
|
+
coverage_percent=coverage,
|
|
597
|
+
confidence=round(len(present) / row_count, 4) if row_count else 0.0,
|
|
598
|
+
)
|
|
599
|
+
|
|
600
|
+
|
|
601
|
+
def analyze_column(
|
|
602
|
+
values: pd.Series,
|
|
603
|
+
*,
|
|
604
|
+
name: str | None = None,
|
|
605
|
+
position: int = 1,
|
|
606
|
+
config: AnalysisConfig | None = None,
|
|
607
|
+
) -> ColumnProfile:
|
|
608
|
+
"""Profile one raw string column, also usable independently of CSV ingestion."""
|
|
609
|
+
config = config or AnalysisConfig()
|
|
610
|
+
if not all(isinstance(value, str) for value in values):
|
|
611
|
+
raise ValueError("Column analysis expects raw strings, without null objects.")
|
|
612
|
+
normalized, normalization = normalize_values(values, config)
|
|
613
|
+
absent = missing_mask(normalized, config)
|
|
614
|
+
present = normalized[~absent]
|
|
615
|
+
frequencies = sorted(
|
|
616
|
+
((str(value), int(count)) for value, count in present.value_counts().items()),
|
|
617
|
+
key=lambda item: (-item[1], item[0]),
|
|
618
|
+
)
|
|
619
|
+
date_profile, valid_date_values = build_date_profile(frequencies, config)
|
|
620
|
+
counts: Counter[str] = Counter()
|
|
621
|
+
for value, count in frequencies:
|
|
622
|
+
counts["date" if value in valid_date_values else value_type(value)] += count
|
|
623
|
+
inferred, semantic_type, type_confidence, type_error_count, type_error_percent = (
|
|
624
|
+
infer_column_type(counts, date_profile, len(present), config)
|
|
625
|
+
)
|
|
626
|
+
|
|
627
|
+
numeric = None
|
|
628
|
+
if inferred in {"integer", "number"}:
|
|
629
|
+
accepted_types = {"integer"} if inferred == "integer" else {"integer", "number"}
|
|
630
|
+
numeric_values = present[present.map(lambda value: value_type(value) in accepted_types)]
|
|
631
|
+
numbers = pd.to_numeric(numeric_values, errors="coerce").astype(float)
|
|
632
|
+
stats = [numbers.min(), numbers.max(), numbers.mean(), numbers.median()]
|
|
633
|
+
if numbers.notna().all() and all(math.isfinite(value) for value in stats):
|
|
634
|
+
numeric = NumericStats(
|
|
635
|
+
minimum=stats[0], maximum=stats[1], mean=stats[2], median=stats[3]
|
|
636
|
+
)
|
|
637
|
+
value_profile = build_value_profile(
|
|
638
|
+
frequencies,
|
|
639
|
+
inferred_type=inferred,
|
|
640
|
+
position=position,
|
|
641
|
+
config=config,
|
|
642
|
+
)
|
|
643
|
+
enum = infer_enum_candidate(
|
|
644
|
+
present,
|
|
645
|
+
inferred_type=inferred,
|
|
646
|
+
row_count=len(values),
|
|
647
|
+
config=config,
|
|
648
|
+
)
|
|
649
|
+
if enum and semantic_type is None:
|
|
650
|
+
semantic_type = "enum"
|
|
651
|
+
string_profile = build_string_profile(present, frequencies, inferred, config)
|
|
652
|
+
return ColumnProfile(
|
|
653
|
+
id=f"column_{position}",
|
|
654
|
+
name=name if name is not None else str(values.name or ""),
|
|
655
|
+
position=position,
|
|
656
|
+
inferred_type=inferred,
|
|
657
|
+
type_counts=dict(counts),
|
|
658
|
+
type_confidence=round(type_confidence, 4),
|
|
659
|
+
type_error_count=type_error_count,
|
|
660
|
+
type_error_percent=type_error_percent,
|
|
661
|
+
missing_count=int(absent.sum()),
|
|
662
|
+
missing_percent=percent(int(absent.sum()), len(values)),
|
|
663
|
+
normalization=normalization,
|
|
664
|
+
distinct_count=len(frequencies),
|
|
665
|
+
examples=[
|
|
666
|
+
item.value
|
|
667
|
+
for item in value_profile.values[: config.value_examples.inline_display_size]
|
|
668
|
+
],
|
|
669
|
+
value_profile=value_profile,
|
|
670
|
+
semantic_type=semantic_type,
|
|
671
|
+
enum=enum,
|
|
672
|
+
date_profile=date_profile,
|
|
673
|
+
string_profile=string_profile,
|
|
674
|
+
numeric=numeric,
|
|
675
|
+
)
|
|
676
|
+
|
|
677
|
+
|
|
678
|
+
def analyze_dataset(dataset: CsvDataset, config: AnalysisConfig) -> DatasetProfile:
|
|
679
|
+
frame = dataset.frame
|
|
680
|
+
columns = [
|
|
681
|
+
analyze_column(frame.iloc[:, i], name=name, position=i + 1, config=config)
|
|
682
|
+
for i, name in enumerate(dataset.headers)
|
|
683
|
+
]
|
|
684
|
+
absent = frame.apply(lambda values: missing_mask(values, config))
|
|
685
|
+
duplicates = frame.duplicated(keep="first")
|
|
686
|
+
missing_count = sum(column.missing_count for column in columns)
|
|
687
|
+
trim_count = sum(column.normalization.trim_count for column in columns)
|
|
688
|
+
collapse_count = sum(
|
|
689
|
+
column.normalization.collapse_internal_whitespace_count for column in columns
|
|
690
|
+
)
|
|
691
|
+
empty = [column.id for column in columns if column.inferred_type == "empty"]
|
|
692
|
+
constant = [column.id for column in columns if column.distinct_count == 1]
|
|
693
|
+
mixed = [column.id for column in columns if column.inferred_type == "mixed"]
|
|
694
|
+
issues = []
|
|
695
|
+
|
|
696
|
+
def add_issue(
|
|
697
|
+
code, message, count, ids=(), rows=(), severity="warning", always=False
|
|
698
|
+
):
|
|
699
|
+
if count or always:
|
|
700
|
+
issues.append(
|
|
701
|
+
Issue(
|
|
702
|
+
code=code,
|
|
703
|
+
severity=severity,
|
|
704
|
+
message=message,
|
|
705
|
+
count=count,
|
|
706
|
+
column_ids=list(ids),
|
|
707
|
+
row_numbers=list(islice(rows, 10)),
|
|
708
|
+
)
|
|
709
|
+
)
|
|
710
|
+
|
|
711
|
+
add_issue(
|
|
712
|
+
"duplicate_rows",
|
|
713
|
+
"Duplicate rows beyond their first occurrence",
|
|
714
|
+
int(duplicates.sum()),
|
|
715
|
+
rows=(i + 1 for i, value in enumerate(duplicates) if value),
|
|
716
|
+
)
|
|
717
|
+
add_issue(
|
|
718
|
+
"missing_values",
|
|
719
|
+
"Missing cells",
|
|
720
|
+
missing_count,
|
|
721
|
+
[c.id for c in columns if c.missing_count],
|
|
722
|
+
(i + 1 for i, value in enumerate(absent.any(axis=1)) if value),
|
|
723
|
+
)
|
|
724
|
+
add_issue("empty_columns", "Columns without any present values", len(empty), empty)
|
|
725
|
+
add_issue(
|
|
726
|
+
"trimmed_cells",
|
|
727
|
+
"Cells changed by trimming surrounding whitespace",
|
|
728
|
+
trim_count,
|
|
729
|
+
[c.id for c in columns if c.normalization.trim_count],
|
|
730
|
+
severity="info",
|
|
731
|
+
always=True,
|
|
732
|
+
)
|
|
733
|
+
add_issue(
|
|
734
|
+
"collapsed_whitespace",
|
|
735
|
+
"Cells changed by collapsing repeated internal whitespace",
|
|
736
|
+
collapse_count,
|
|
737
|
+
[c.id for c in columns if c.normalization.collapse_internal_whitespace_count],
|
|
738
|
+
severity="info",
|
|
739
|
+
always=True,
|
|
740
|
+
)
|
|
741
|
+
add_issue(
|
|
742
|
+
"constant_columns",
|
|
743
|
+
"Columns with one distinct present value",
|
|
744
|
+
len(constant),
|
|
745
|
+
constant,
|
|
746
|
+
severity="info",
|
|
747
|
+
)
|
|
748
|
+
add_issue("mixed_types", "Columns with mixed value types", len(mixed), mixed)
|
|
749
|
+
header_counts = Counter(dataset.headers)
|
|
750
|
+
bad_headers = [
|
|
751
|
+
c.id for c in columns if not c.name.strip() or header_counts[c.name] > 1
|
|
752
|
+
]
|
|
753
|
+
add_issue(
|
|
754
|
+
"ambiguous_headers",
|
|
755
|
+
"Blank or repeated column names",
|
|
756
|
+
len(bad_headers),
|
|
757
|
+
bad_headers,
|
|
758
|
+
)
|
|
759
|
+
|
|
760
|
+
return DatasetProfile(
|
|
761
|
+
generated_at=datetime.now(UTC),
|
|
762
|
+
processing_seconds=0.0,
|
|
763
|
+
source=dataset.source,
|
|
764
|
+
config=config,
|
|
765
|
+
summary=DatasetSummary(
|
|
766
|
+
row_count=len(frame),
|
|
767
|
+
column_count=len(columns),
|
|
768
|
+
cell_count=frame.size,
|
|
769
|
+
missing_count=missing_count,
|
|
770
|
+
missing_percent=percent(missing_count, frame.size),
|
|
771
|
+
trim_count=trim_count,
|
|
772
|
+
collapse_internal_whitespace_count=collapse_count,
|
|
773
|
+
duplicate_row_count=int(duplicates.sum()),
|
|
774
|
+
empty_row_count=int(absent.all(axis=1).sum()),
|
|
775
|
+
empty_column_count=len(empty),
|
|
776
|
+
constant_column_count=len(constant),
|
|
777
|
+
),
|
|
778
|
+
columns=columns,
|
|
779
|
+
issues=issues,
|
|
780
|
+
preview=[
|
|
781
|
+
PreviewRow(row_number=i + 1, values=list(row))
|
|
782
|
+
for i, row in enumerate(
|
|
783
|
+
frame.head(config.preview_rows).itertuples(index=False, name=None)
|
|
784
|
+
)
|
|
785
|
+
],
|
|
786
|
+
)
|