tabalyst 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tabalyst/models.py ADDED
@@ -0,0 +1,174 @@
1
+ """Serializable analysis results, independent from presentation."""
2
+
3
+ from datetime import datetime
4
+ from typing import Literal
5
+
6
+ from pydantic import BaseModel, ConfigDict, FiniteFloat
7
+
8
+ from tabalyst.config import AnalysisConfig
9
+
10
+
11
+ class ResultModel(BaseModel):
12
+ model_config = ConfigDict(extra="forbid")
13
+
14
+
15
+ class SourceInfo(ResultModel):
16
+ filename: str
17
+ size_bytes: int
18
+ sha256: str
19
+ encoding: str
20
+ delimiter: str
21
+
22
+
23
+ class NumericStats(ResultModel):
24
+ minimum: FiniteFloat
25
+ maximum: FiniteFloat
26
+ mean: FiniteFloat
27
+ median: FiniteFloat
28
+
29
+
30
+ class NormalizationStats(ResultModel):
31
+ trim_count: int
32
+ trim_percent: float
33
+ collapse_internal_whitespace_count: int
34
+ collapse_internal_whitespace_percent: float
35
+
36
+
37
+ class DateFormatCount(ResultModel):
38
+ format: str
39
+ order: Literal["YMD", "MDY", "DMY"]
40
+ separator: str
41
+ count: int
42
+ percent: float
43
+ iso: bool = False
44
+
45
+
46
+ class DateBreakdownItem(ResultModel):
47
+ label: str
48
+ category: Literal["valid", "ambiguous", "invalid", "not_date"]
49
+ count: int
50
+ percent: float
51
+
52
+
53
+ class DateProfile(ResultModel):
54
+ status: Literal["valid", "multiple_formats", "mixed", "ambiguous", "invalid"]
55
+ valid_count: int
56
+ ambiguous_count: int
57
+ invalid_date_count: int
58
+ not_date_count: int
59
+ resolved_ambiguous_order: Literal["MDY", "DMY"] | None = None
60
+ ambiguous_order_source: Literal["config", "column"] | None = None
61
+ formats: list[DateFormatCount]
62
+ format_count: int
63
+ breakdown: list[DateBreakdownItem]
64
+ errors: dict[str, int]
65
+
66
+
67
+ class StringLengthExample(ResultModel):
68
+ value: str
69
+ count: int
70
+
71
+
72
+ class StringLengthDistribution(ResultModel):
73
+ length: int
74
+ count: int
75
+ percent: float
76
+ distinct_count: int
77
+ examples: list[StringLengthExample]
78
+
79
+
80
+ class StringProfile(ResultModel):
81
+ status: Literal["very_short", "short", "medium", "long", "very_long"]
82
+ present_count: int
83
+ minimum_length: int
84
+ maximum_length: int
85
+ mean_length: FiniteFloat
86
+ median_length: FiniteFloat
87
+ distinct_length_count: int
88
+ fixed_length: int | None = None
89
+ length_distribution: list[StringLengthDistribution]
90
+
91
+
92
+ class ValueOccurrence(ResultModel):
93
+ value: str
94
+ count: int
95
+ truncated: bool = False
96
+
97
+
98
+ class ValueProfile(ResultModel):
99
+ selection: Literal["complete", "diverse_sample", "random_sample"]
100
+ sampled_distinct_count: int
101
+ values: list[ValueOccurrence]
102
+
103
+
104
+ class EnumCandidate(ResultModel):
105
+ status: Literal["candidate"] = "candidate"
106
+ observed_distinct_count: int
107
+ non_missing_count: int
108
+ coverage_percent: float
109
+ confidence: float
110
+
111
+
112
+ class ColumnProfile(ResultModel):
113
+ id: str
114
+ name: str
115
+ position: int
116
+ inferred_type: Literal[
117
+ "empty", "boolean", "integer", "number", "date", "text", "mixed"
118
+ ]
119
+ type_counts: dict[str, int]
120
+ type_confidence: float
121
+ type_error_count: int | None
122
+ type_error_percent: float | None
123
+ missing_count: int
124
+ missing_percent: float
125
+ normalization: NormalizationStats
126
+ distinct_count: int
127
+ examples: list[str]
128
+ value_profile: ValueProfile
129
+ semantic_type: Literal["enum", "date"] | None = None
130
+ enum: EnumCandidate | None = None
131
+ date_profile: DateProfile | None = None
132
+ string_profile: StringProfile | None = None
133
+ numeric: NumericStats | None = None
134
+
135
+
136
+ class DatasetSummary(ResultModel):
137
+ row_count: int
138
+ column_count: int
139
+ cell_count: int
140
+ missing_count: int
141
+ missing_percent: float
142
+ trim_count: int
143
+ collapse_internal_whitespace_count: int
144
+ duplicate_row_count: int
145
+ empty_row_count: int
146
+ empty_column_count: int
147
+ constant_column_count: int
148
+
149
+
150
+ class Issue(ResultModel):
151
+ code: str
152
+ severity: Literal["info", "warning"]
153
+ message: str
154
+ count: int
155
+ column_ids: list[str]
156
+ row_numbers: list[int]
157
+
158
+
159
+ class PreviewRow(ResultModel):
160
+ row_number: int
161
+ values: list[str]
162
+
163
+
164
+ class DatasetProfile(ResultModel):
165
+ format_version: Literal["0.1.0a"] = "0.1.0a"
166
+ format_revision: Literal[1] = 1
167
+ generated_at: datetime
168
+ processing_seconds: FiniteFloat
169
+ source: SourceInfo
170
+ config: AnalysisConfig
171
+ summary: DatasetSummary
172
+ columns: list[ColumnProfile]
173
+ issues: list[Issue]
174
+ preview: list[PreviewRow]
tabalyst/reporting.py ADDED
@@ -0,0 +1,54 @@
1
+ """Render self-contained analysis results as an interactive HTML report."""
2
+
3
+ from collections import Counter
4
+ from importlib.resources import files
5
+ from pathlib import Path
6
+
7
+ from jinja2 import Environment, StrictUndefined
8
+
9
+ from tabalyst._version import get_version
10
+ from tabalyst.models import DatasetProfile
11
+
12
+
13
+ def format_number(number: float) -> str:
14
+ """Keep ordinary values readable and compact only genuinely large magnitudes."""
15
+ if abs(number) > 10**10:
16
+ return f"{number:.4e}"
17
+ return f"{number:,.4f}".rstrip("0").rstrip(".")
18
+
19
+
20
+ def format_seconds(seconds: float) -> str:
21
+ """Display elapsed time with no more than two decimal places."""
22
+ return f"{seconds:.2f}".rstrip("0").rstrip(".")
23
+
24
+
25
+ def format_size(size_bytes: int) -> str:
26
+ """Display source size in KB, switching to MB at one mebibyte."""
27
+ if size_bytes >= 1024**2:
28
+ return f"{size_bytes / 1024**2:.1f} MB"
29
+ return f"{size_bytes / 1024:.1f} KB"
30
+
31
+
32
+ def load_profile(path: str | Path) -> DatasetProfile:
33
+ """Validate and load the experimental JSON profile format."""
34
+ return DatasetProfile.model_validate_json(Path(path).read_text(encoding="utf-8"))
35
+
36
+
37
+ def render_report(profile: DatasetProfile) -> str:
38
+ """Render exclusively from serialized analysis results; CSV access is unnecessary."""
39
+ resources = files("tabalyst")
40
+ environment = Environment(autoescape=True, undefined=StrictUndefined)
41
+ environment.filters["count"] = lambda number: f"{number:,}"
42
+ environment.filters["number"] = format_number
43
+ environment.filters["seconds"] = format_seconds
44
+ environment.filters["size"] = format_size
45
+ template = environment.from_string(
46
+ resources.joinpath("templates/report.html").read_text(encoding="utf-8")
47
+ )
48
+ return template.render(
49
+ report=profile,
50
+ tabalyst_version=get_version(),
51
+ types=Counter(column.inferred_type for column in profile.columns),
52
+ theme_css=resources.joinpath("static/theme.css").read_text(encoding="utf-8"),
53
+ report_js=resources.joinpath("static/report.js").read_text(encoding="utf-8"),
54
+ )
tabalyst/service.py ADDED
@@ -0,0 +1,134 @@
1
+ """Public analysis boundary shared by Python callers and the CLI."""
2
+
3
+ import json
4
+ from collections.abc import Sequence
5
+ from pathlib import Path
6
+ from time import perf_counter
7
+ from typing import Any
8
+
9
+ from tabalyst.analysis import analyze_dataset
10
+ from tabalyst.config import AnalysisConfig, resolve_config
11
+ from tabalyst.errors import ConfigurationError, InputError, ReportError
12
+ from tabalyst.execution_log import (
13
+ EXECUTION_LOG_NAME,
14
+ append_execution,
15
+ build_execution_entry,
16
+ )
17
+ from tabalyst.ingestion import read_csv
18
+ from tabalyst.models import DatasetProfile
19
+ from tabalyst.reporting import load_profile, render_report
20
+
21
+ ConfigPath = str | Path | Sequence[str | Path]
22
+
23
+
24
+ def analyze_csv(
25
+ path: str | Path, config: AnalysisConfig | None = None
26
+ ) -> DatasetProfile:
27
+ """Read and analyze one CSV without coupling callers to the CLI."""
28
+ config = config or AnalysisConfig()
29
+ started = perf_counter()
30
+ profile = analyze_dataset(read_csv(Path(path), config.csv), config)
31
+ profile.processing_seconds = round(perf_counter() - started, 4)
32
+ return profile
33
+
34
+
35
+ def _config_paths(config_path: ConfigPath | None) -> list[Path]:
36
+ if config_path is None:
37
+ return []
38
+ if isinstance(config_path, (str, Path)):
39
+ paths = [Path(config_path)]
40
+ else:
41
+ paths = [Path(path) for path in config_path]
42
+ for path in paths:
43
+ if not path.exists():
44
+ raise ConfigurationError(f"Configuration file does not exist: {path}")
45
+ if not path.is_file():
46
+ raise ConfigurationError(f"Configuration path is not a file: {path}")
47
+ return paths
48
+
49
+
50
+ def _validate_paths(
51
+ source: Path, report: Path, config_paths: list[Path]
52
+ ) -> tuple[Path, Path]:
53
+ if not source.exists():
54
+ raise InputError(f"CSV file does not exist: {source}")
55
+ if not source.is_file():
56
+ raise InputError(f"CSV path is not a file: {source}")
57
+ if report.suffix.lower() != ".html":
58
+ raise InputError("report_path must be a complete filename ending in .html")
59
+ if report.exists() and report.is_dir():
60
+ raise InputError(f"Report path is a directory: {report}")
61
+
62
+ json_report = report.with_suffix(".json")
63
+ execution_report = report.parent / EXECUTION_LOG_NAME
64
+ inputs = {source.resolve(), *(path.resolve() for path in config_paths)}
65
+ outputs = [report.resolve(), json_report.resolve(), execution_report.resolve()]
66
+ if len(set(outputs)) != len(outputs):
67
+ raise InputError("HTML and JSON report paths must be different.")
68
+ for output in outputs:
69
+ if output in inputs:
70
+ raise InputError(f"Report would overwrite an input file: {output}")
71
+ return json_report, execution_report
72
+
73
+
74
+ def _write_text(path: Path, content: str) -> None:
75
+ try:
76
+ path.parent.mkdir(parents=True, exist_ok=True)
77
+ path.write_text(content, encoding="utf-8", newline="\n")
78
+ except OSError as exc:
79
+ raise ReportError(f"Cannot write report file {path}: {exc}") from exc
80
+
81
+
82
+ def analyze(
83
+ csv_path: str | Path,
84
+ report_path: str | Path,
85
+ separator: str | None = None,
86
+ encoding: str | None = None,
87
+ config_path: ConfigPath | None = None,
88
+ ) -> dict[str, Any]:
89
+ """Analyze one CSV, write sibling JSON/HTML reports and return the result.
90
+
91
+ Explicit ``separator`` and ``encoding`` values override configuration-file
92
+ values, which in turn override Tabalyst's built-in defaults.
93
+ """
94
+ started = perf_counter()
95
+ source = Path(csv_path)
96
+ report = Path(report_path)
97
+ config_paths = _config_paths(config_path)
98
+ json_report, execution_report = _validate_paths(source, report, config_paths)
99
+ config = resolve_config(
100
+ config_paths,
101
+ separator=separator,
102
+ encoding=encoding,
103
+ )
104
+
105
+ try:
106
+ profile = analyze_csv(source, config)
107
+ except InputError:
108
+ raise
109
+ except OSError as exc:
110
+ raise InputError(f"Cannot read CSV file {source}: {exc}") from exc
111
+
112
+ result = profile.model_dump(mode="json")
113
+ json_text = json.dumps(result, indent=2, ensure_ascii=False) + "\n"
114
+ _write_text(json_report, json_text)
115
+ try:
116
+ html = render_report(load_profile(json_report))
117
+ except OSError as exc:
118
+ raise ReportError(f"Cannot render HTML report {report}: {exc}") from exc
119
+ _write_text(report, html)
120
+ try:
121
+ entry = build_execution_entry(
122
+ source=source,
123
+ html_output=report,
124
+ json_output=json_report,
125
+ profile=profile,
126
+ total_seconds=perf_counter() - started,
127
+ git_directory=Path.cwd(),
128
+ )
129
+ append_execution(execution_report, entry)
130
+ except (OSError, ValueError) as exc:
131
+ raise ReportError(
132
+ f"Cannot update execution history {execution_report}: {exc}"
133
+ ) from exc
134
+ return result