tabalyst 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tabalyst/__init__.py +20 -0
- tabalyst/__main__.py +6 -0
- tabalyst/_version.py +23 -0
- tabalyst/analysis.py +786 -0
- tabalyst/cli.py +208 -0
- tabalyst/config.py +214 -0
- tabalyst/errors.py +17 -0
- tabalyst/execution_log.py +102 -0
- tabalyst/ingestion.py +86 -0
- tabalyst/models.py +174 -0
- tabalyst/reporting.py +54 -0
- tabalyst/service.py +134 -0
- tabalyst/static/report.js +500 -0
- tabalyst/static/theme.css +411 -0
- tabalyst/templates/report.html +318 -0
- tabalyst-0.1.0.dist-info/METADATA +208 -0
- tabalyst-0.1.0.dist-info/RECORD +21 -0
- tabalyst-0.1.0.dist-info/WHEEL +5 -0
- tabalyst-0.1.0.dist-info/entry_points.txt +2 -0
- tabalyst-0.1.0.dist-info/licenses/LICENSE +21 -0
- tabalyst-0.1.0.dist-info/top_level.txt +1 -0
tabalyst/models.py
ADDED
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
"""Serializable analysis results, independent from presentation."""
|
|
2
|
+
|
|
3
|
+
from datetime import datetime
|
|
4
|
+
from typing import Literal
|
|
5
|
+
|
|
6
|
+
from pydantic import BaseModel, ConfigDict, FiniteFloat
|
|
7
|
+
|
|
8
|
+
from tabalyst.config import AnalysisConfig
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class ResultModel(BaseModel):
|
|
12
|
+
model_config = ConfigDict(extra="forbid")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class SourceInfo(ResultModel):
|
|
16
|
+
filename: str
|
|
17
|
+
size_bytes: int
|
|
18
|
+
sha256: str
|
|
19
|
+
encoding: str
|
|
20
|
+
delimiter: str
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class NumericStats(ResultModel):
|
|
24
|
+
minimum: FiniteFloat
|
|
25
|
+
maximum: FiniteFloat
|
|
26
|
+
mean: FiniteFloat
|
|
27
|
+
median: FiniteFloat
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class NormalizationStats(ResultModel):
|
|
31
|
+
trim_count: int
|
|
32
|
+
trim_percent: float
|
|
33
|
+
collapse_internal_whitespace_count: int
|
|
34
|
+
collapse_internal_whitespace_percent: float
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class DateFormatCount(ResultModel):
|
|
38
|
+
format: str
|
|
39
|
+
order: Literal["YMD", "MDY", "DMY"]
|
|
40
|
+
separator: str
|
|
41
|
+
count: int
|
|
42
|
+
percent: float
|
|
43
|
+
iso: bool = False
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class DateBreakdownItem(ResultModel):
|
|
47
|
+
label: str
|
|
48
|
+
category: Literal["valid", "ambiguous", "invalid", "not_date"]
|
|
49
|
+
count: int
|
|
50
|
+
percent: float
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class DateProfile(ResultModel):
|
|
54
|
+
status: Literal["valid", "multiple_formats", "mixed", "ambiguous", "invalid"]
|
|
55
|
+
valid_count: int
|
|
56
|
+
ambiguous_count: int
|
|
57
|
+
invalid_date_count: int
|
|
58
|
+
not_date_count: int
|
|
59
|
+
resolved_ambiguous_order: Literal["MDY", "DMY"] | None = None
|
|
60
|
+
ambiguous_order_source: Literal["config", "column"] | None = None
|
|
61
|
+
formats: list[DateFormatCount]
|
|
62
|
+
format_count: int
|
|
63
|
+
breakdown: list[DateBreakdownItem]
|
|
64
|
+
errors: dict[str, int]
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class StringLengthExample(ResultModel):
|
|
68
|
+
value: str
|
|
69
|
+
count: int
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class StringLengthDistribution(ResultModel):
|
|
73
|
+
length: int
|
|
74
|
+
count: int
|
|
75
|
+
percent: float
|
|
76
|
+
distinct_count: int
|
|
77
|
+
examples: list[StringLengthExample]
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class StringProfile(ResultModel):
|
|
81
|
+
status: Literal["very_short", "short", "medium", "long", "very_long"]
|
|
82
|
+
present_count: int
|
|
83
|
+
minimum_length: int
|
|
84
|
+
maximum_length: int
|
|
85
|
+
mean_length: FiniteFloat
|
|
86
|
+
median_length: FiniteFloat
|
|
87
|
+
distinct_length_count: int
|
|
88
|
+
fixed_length: int | None = None
|
|
89
|
+
length_distribution: list[StringLengthDistribution]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class ValueOccurrence(ResultModel):
|
|
93
|
+
value: str
|
|
94
|
+
count: int
|
|
95
|
+
truncated: bool = False
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class ValueProfile(ResultModel):
|
|
99
|
+
selection: Literal["complete", "diverse_sample", "random_sample"]
|
|
100
|
+
sampled_distinct_count: int
|
|
101
|
+
values: list[ValueOccurrence]
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
class EnumCandidate(ResultModel):
|
|
105
|
+
status: Literal["candidate"] = "candidate"
|
|
106
|
+
observed_distinct_count: int
|
|
107
|
+
non_missing_count: int
|
|
108
|
+
coverage_percent: float
|
|
109
|
+
confidence: float
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
class ColumnProfile(ResultModel):
|
|
113
|
+
id: str
|
|
114
|
+
name: str
|
|
115
|
+
position: int
|
|
116
|
+
inferred_type: Literal[
|
|
117
|
+
"empty", "boolean", "integer", "number", "date", "text", "mixed"
|
|
118
|
+
]
|
|
119
|
+
type_counts: dict[str, int]
|
|
120
|
+
type_confidence: float
|
|
121
|
+
type_error_count: int | None
|
|
122
|
+
type_error_percent: float | None
|
|
123
|
+
missing_count: int
|
|
124
|
+
missing_percent: float
|
|
125
|
+
normalization: NormalizationStats
|
|
126
|
+
distinct_count: int
|
|
127
|
+
examples: list[str]
|
|
128
|
+
value_profile: ValueProfile
|
|
129
|
+
semantic_type: Literal["enum", "date"] | None = None
|
|
130
|
+
enum: EnumCandidate | None = None
|
|
131
|
+
date_profile: DateProfile | None = None
|
|
132
|
+
string_profile: StringProfile | None = None
|
|
133
|
+
numeric: NumericStats | None = None
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
class DatasetSummary(ResultModel):
|
|
137
|
+
row_count: int
|
|
138
|
+
column_count: int
|
|
139
|
+
cell_count: int
|
|
140
|
+
missing_count: int
|
|
141
|
+
missing_percent: float
|
|
142
|
+
trim_count: int
|
|
143
|
+
collapse_internal_whitespace_count: int
|
|
144
|
+
duplicate_row_count: int
|
|
145
|
+
empty_row_count: int
|
|
146
|
+
empty_column_count: int
|
|
147
|
+
constant_column_count: int
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
class Issue(ResultModel):
|
|
151
|
+
code: str
|
|
152
|
+
severity: Literal["info", "warning"]
|
|
153
|
+
message: str
|
|
154
|
+
count: int
|
|
155
|
+
column_ids: list[str]
|
|
156
|
+
row_numbers: list[int]
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class PreviewRow(ResultModel):
|
|
160
|
+
row_number: int
|
|
161
|
+
values: list[str]
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
class DatasetProfile(ResultModel):
|
|
165
|
+
format_version: Literal["0.1.0a"] = "0.1.0a"
|
|
166
|
+
format_revision: Literal[1] = 1
|
|
167
|
+
generated_at: datetime
|
|
168
|
+
processing_seconds: FiniteFloat
|
|
169
|
+
source: SourceInfo
|
|
170
|
+
config: AnalysisConfig
|
|
171
|
+
summary: DatasetSummary
|
|
172
|
+
columns: list[ColumnProfile]
|
|
173
|
+
issues: list[Issue]
|
|
174
|
+
preview: list[PreviewRow]
|
tabalyst/reporting.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Render self-contained analysis results as an interactive HTML report."""
|
|
2
|
+
|
|
3
|
+
from collections import Counter
|
|
4
|
+
from importlib.resources import files
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from jinja2 import Environment, StrictUndefined
|
|
8
|
+
|
|
9
|
+
from tabalyst._version import get_version
|
|
10
|
+
from tabalyst.models import DatasetProfile
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def format_number(number: float) -> str:
|
|
14
|
+
"""Keep ordinary values readable and compact only genuinely large magnitudes."""
|
|
15
|
+
if abs(number) > 10**10:
|
|
16
|
+
return f"{number:.4e}"
|
|
17
|
+
return f"{number:,.4f}".rstrip("0").rstrip(".")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def format_seconds(seconds: float) -> str:
|
|
21
|
+
"""Display elapsed time with no more than two decimal places."""
|
|
22
|
+
return f"{seconds:.2f}".rstrip("0").rstrip(".")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def format_size(size_bytes: int) -> str:
|
|
26
|
+
"""Display source size in KB, switching to MB at one mebibyte."""
|
|
27
|
+
if size_bytes >= 1024**2:
|
|
28
|
+
return f"{size_bytes / 1024**2:.1f} MB"
|
|
29
|
+
return f"{size_bytes / 1024:.1f} KB"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def load_profile(path: str | Path) -> DatasetProfile:
|
|
33
|
+
"""Validate and load the experimental JSON profile format."""
|
|
34
|
+
return DatasetProfile.model_validate_json(Path(path).read_text(encoding="utf-8"))
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def render_report(profile: DatasetProfile) -> str:
|
|
38
|
+
"""Render exclusively from serialized analysis results; CSV access is unnecessary."""
|
|
39
|
+
resources = files("tabalyst")
|
|
40
|
+
environment = Environment(autoescape=True, undefined=StrictUndefined)
|
|
41
|
+
environment.filters["count"] = lambda number: f"{number:,}"
|
|
42
|
+
environment.filters["number"] = format_number
|
|
43
|
+
environment.filters["seconds"] = format_seconds
|
|
44
|
+
environment.filters["size"] = format_size
|
|
45
|
+
template = environment.from_string(
|
|
46
|
+
resources.joinpath("templates/report.html").read_text(encoding="utf-8")
|
|
47
|
+
)
|
|
48
|
+
return template.render(
|
|
49
|
+
report=profile,
|
|
50
|
+
tabalyst_version=get_version(),
|
|
51
|
+
types=Counter(column.inferred_type for column in profile.columns),
|
|
52
|
+
theme_css=resources.joinpath("static/theme.css").read_text(encoding="utf-8"),
|
|
53
|
+
report_js=resources.joinpath("static/report.js").read_text(encoding="utf-8"),
|
|
54
|
+
)
|
tabalyst/service.py
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
"""Public analysis boundary shared by Python callers and the CLI."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from collections.abc import Sequence
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from time import perf_counter
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from tabalyst.analysis import analyze_dataset
|
|
10
|
+
from tabalyst.config import AnalysisConfig, resolve_config
|
|
11
|
+
from tabalyst.errors import ConfigurationError, InputError, ReportError
|
|
12
|
+
from tabalyst.execution_log import (
|
|
13
|
+
EXECUTION_LOG_NAME,
|
|
14
|
+
append_execution,
|
|
15
|
+
build_execution_entry,
|
|
16
|
+
)
|
|
17
|
+
from tabalyst.ingestion import read_csv
|
|
18
|
+
from tabalyst.models import DatasetProfile
|
|
19
|
+
from tabalyst.reporting import load_profile, render_report
|
|
20
|
+
|
|
21
|
+
ConfigPath = str | Path | Sequence[str | Path]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def analyze_csv(
|
|
25
|
+
path: str | Path, config: AnalysisConfig | None = None
|
|
26
|
+
) -> DatasetProfile:
|
|
27
|
+
"""Read and analyze one CSV without coupling callers to the CLI."""
|
|
28
|
+
config = config or AnalysisConfig()
|
|
29
|
+
started = perf_counter()
|
|
30
|
+
profile = analyze_dataset(read_csv(Path(path), config.csv), config)
|
|
31
|
+
profile.processing_seconds = round(perf_counter() - started, 4)
|
|
32
|
+
return profile
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _config_paths(config_path: ConfigPath | None) -> list[Path]:
|
|
36
|
+
if config_path is None:
|
|
37
|
+
return []
|
|
38
|
+
if isinstance(config_path, (str, Path)):
|
|
39
|
+
paths = [Path(config_path)]
|
|
40
|
+
else:
|
|
41
|
+
paths = [Path(path) for path in config_path]
|
|
42
|
+
for path in paths:
|
|
43
|
+
if not path.exists():
|
|
44
|
+
raise ConfigurationError(f"Configuration file does not exist: {path}")
|
|
45
|
+
if not path.is_file():
|
|
46
|
+
raise ConfigurationError(f"Configuration path is not a file: {path}")
|
|
47
|
+
return paths
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _validate_paths(
|
|
51
|
+
source: Path, report: Path, config_paths: list[Path]
|
|
52
|
+
) -> tuple[Path, Path]:
|
|
53
|
+
if not source.exists():
|
|
54
|
+
raise InputError(f"CSV file does not exist: {source}")
|
|
55
|
+
if not source.is_file():
|
|
56
|
+
raise InputError(f"CSV path is not a file: {source}")
|
|
57
|
+
if report.suffix.lower() != ".html":
|
|
58
|
+
raise InputError("report_path must be a complete filename ending in .html")
|
|
59
|
+
if report.exists() and report.is_dir():
|
|
60
|
+
raise InputError(f"Report path is a directory: {report}")
|
|
61
|
+
|
|
62
|
+
json_report = report.with_suffix(".json")
|
|
63
|
+
execution_report = report.parent / EXECUTION_LOG_NAME
|
|
64
|
+
inputs = {source.resolve(), *(path.resolve() for path in config_paths)}
|
|
65
|
+
outputs = [report.resolve(), json_report.resolve(), execution_report.resolve()]
|
|
66
|
+
if len(set(outputs)) != len(outputs):
|
|
67
|
+
raise InputError("HTML and JSON report paths must be different.")
|
|
68
|
+
for output in outputs:
|
|
69
|
+
if output in inputs:
|
|
70
|
+
raise InputError(f"Report would overwrite an input file: {output}")
|
|
71
|
+
return json_report, execution_report
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _write_text(path: Path, content: str) -> None:
|
|
75
|
+
try:
|
|
76
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
77
|
+
path.write_text(content, encoding="utf-8", newline="\n")
|
|
78
|
+
except OSError as exc:
|
|
79
|
+
raise ReportError(f"Cannot write report file {path}: {exc}") from exc
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def analyze(
|
|
83
|
+
csv_path: str | Path,
|
|
84
|
+
report_path: str | Path,
|
|
85
|
+
separator: str | None = None,
|
|
86
|
+
encoding: str | None = None,
|
|
87
|
+
config_path: ConfigPath | None = None,
|
|
88
|
+
) -> dict[str, Any]:
|
|
89
|
+
"""Analyze one CSV, write sibling JSON/HTML reports and return the result.
|
|
90
|
+
|
|
91
|
+
Explicit ``separator`` and ``encoding`` values override configuration-file
|
|
92
|
+
values, which in turn override Tabalyst's built-in defaults.
|
|
93
|
+
"""
|
|
94
|
+
started = perf_counter()
|
|
95
|
+
source = Path(csv_path)
|
|
96
|
+
report = Path(report_path)
|
|
97
|
+
config_paths = _config_paths(config_path)
|
|
98
|
+
json_report, execution_report = _validate_paths(source, report, config_paths)
|
|
99
|
+
config = resolve_config(
|
|
100
|
+
config_paths,
|
|
101
|
+
separator=separator,
|
|
102
|
+
encoding=encoding,
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
try:
|
|
106
|
+
profile = analyze_csv(source, config)
|
|
107
|
+
except InputError:
|
|
108
|
+
raise
|
|
109
|
+
except OSError as exc:
|
|
110
|
+
raise InputError(f"Cannot read CSV file {source}: {exc}") from exc
|
|
111
|
+
|
|
112
|
+
result = profile.model_dump(mode="json")
|
|
113
|
+
json_text = json.dumps(result, indent=2, ensure_ascii=False) + "\n"
|
|
114
|
+
_write_text(json_report, json_text)
|
|
115
|
+
try:
|
|
116
|
+
html = render_report(load_profile(json_report))
|
|
117
|
+
except OSError as exc:
|
|
118
|
+
raise ReportError(f"Cannot render HTML report {report}: {exc}") from exc
|
|
119
|
+
_write_text(report, html)
|
|
120
|
+
try:
|
|
121
|
+
entry = build_execution_entry(
|
|
122
|
+
source=source,
|
|
123
|
+
html_output=report,
|
|
124
|
+
json_output=json_report,
|
|
125
|
+
profile=profile,
|
|
126
|
+
total_seconds=perf_counter() - started,
|
|
127
|
+
git_directory=Path.cwd(),
|
|
128
|
+
)
|
|
129
|
+
append_execution(execution_report, entry)
|
|
130
|
+
except (OSError, ValueError) as exc:
|
|
131
|
+
raise ReportError(
|
|
132
|
+
f"Cannot update execution history {execution_report}: {exc}"
|
|
133
|
+
) from exc
|
|
134
|
+
return result
|