tabalyst 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tabalyst/__init__.py +20 -0
- tabalyst/__main__.py +6 -0
- tabalyst/_version.py +23 -0
- tabalyst/analysis.py +786 -0
- tabalyst/cli.py +208 -0
- tabalyst/config.py +214 -0
- tabalyst/errors.py +17 -0
- tabalyst/execution_log.py +102 -0
- tabalyst/ingestion.py +86 -0
- tabalyst/models.py +174 -0
- tabalyst/reporting.py +54 -0
- tabalyst/service.py +134 -0
- tabalyst/static/report.js +500 -0
- tabalyst/static/theme.css +411 -0
- tabalyst/templates/report.html +318 -0
- tabalyst-0.1.0.dist-info/METADATA +208 -0
- tabalyst-0.1.0.dist-info/RECORD +21 -0
- tabalyst-0.1.0.dist-info/WHEEL +5 -0
- tabalyst-0.1.0.dist-info/entry_points.txt +2 -0
- tabalyst-0.1.0.dist-info/licenses/LICENSE +21 -0
- tabalyst-0.1.0.dist-info/top_level.txt +1 -0
tabalyst/cli.py
ADDED
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
"""Thin command-line adapters over the public Tabalyst API."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import sys
|
|
5
|
+
import tempfile
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Annotated
|
|
8
|
+
|
|
9
|
+
import typer
|
|
10
|
+
|
|
11
|
+
from tabalyst import __version__
|
|
12
|
+
from tabalyst.config import load_config
|
|
13
|
+
from tabalyst.errors import TabalystError
|
|
14
|
+
from tabalyst.execution_log import EXECUTION_LOG_NAME
|
|
15
|
+
from tabalyst.reporting import load_profile, render_report
|
|
16
|
+
from tabalyst.service import analyze as analyze_report
|
|
17
|
+
|
|
18
|
+
PROJECT_CONFIG_NAME = "tabalyst.json"
|
|
19
|
+
|
|
20
|
+
# The official 0.1.0 interface is a single command with two positional paths.
|
|
21
|
+
official_app = typer.Typer(
|
|
22
|
+
add_completion=False,
|
|
23
|
+
help="Analyze CSV data and generate sibling JSON and HTML reports.",
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
# Keep the alpha subcommands available during the transition to the stable CLI.
|
|
27
|
+
app = typer.Typer(
|
|
28
|
+
add_completion=False,
|
|
29
|
+
no_args_is_help=True,
|
|
30
|
+
help="Legacy alpha commands retained for compatibility.",
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def analysis_config_paths(explicit: list[Path]) -> list[Path]:
|
|
35
|
+
"""Put the current project's automatic config before explicit overrides."""
|
|
36
|
+
project_config = Path.cwd() / PROJECT_CONFIG_NAME
|
|
37
|
+
paths = [project_config] if project_config.is_file() else []
|
|
38
|
+
known = {path.resolve() for path in paths}
|
|
39
|
+
for path in explicit:
|
|
40
|
+
if path.resolve() not in known:
|
|
41
|
+
paths.append(path)
|
|
42
|
+
known.add(path.resolve())
|
|
43
|
+
return paths
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def check_outputs(inputs: list[Path], outputs: list[Path]) -> None:
|
|
47
|
+
"""Protect legacy commands from overwriting inputs or aliased outputs."""
|
|
48
|
+
resolved = [path.resolve() for path in outputs]
|
|
49
|
+
if len(set(resolved)) != len(resolved):
|
|
50
|
+
raise ValueError("Output paths must be different.")
|
|
51
|
+
for output in outputs:
|
|
52
|
+
if output.resolve() in {path.resolve() for path in inputs}:
|
|
53
|
+
raise ValueError(f"Output would overwrite an input file: {output}")
|
|
54
|
+
if output.exists() and any(output.samefile(path) for path in inputs):
|
|
55
|
+
raise ValueError(f"Output would overwrite an input file: {output}")
|
|
56
|
+
if output.is_dir():
|
|
57
|
+
raise ValueError(f"Output is a directory: {output}")
|
|
58
|
+
if (
|
|
59
|
+
len(outputs) > 1
|
|
60
|
+
and all(path.exists() for path in outputs)
|
|
61
|
+
and outputs[0].samefile(outputs[1])
|
|
62
|
+
):
|
|
63
|
+
raise ValueError("Outputs refer to the same file.")
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def write_output(path: Path, content: str) -> None:
|
|
67
|
+
"""Write a UTF-8 text artifact for the legacy render command."""
|
|
68
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
69
|
+
path.write_text(content, encoding="utf-8", newline="\n")
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _print_success(result: dict, report_path: Path) -> None:
|
|
73
|
+
summary = result["summary"]
|
|
74
|
+
typer.echo(
|
|
75
|
+
f"Analyzed {summary['row_count']:,} rows and "
|
|
76
|
+
f"{summary['column_count']} columns."
|
|
77
|
+
)
|
|
78
|
+
typer.echo(f"JSON: {report_path.with_suffix('.json').resolve()}")
|
|
79
|
+
typer.echo(f"HTML: {report_path.resolve()}")
|
|
80
|
+
execution_path = (report_path.parent / EXECUTION_LOG_NAME).resolve()
|
|
81
|
+
typer.echo(f"Execution history: {execution_path}")
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _version_callback(value: bool) -> None:
|
|
85
|
+
if value:
|
|
86
|
+
typer.echo(__version__)
|
|
87
|
+
raise typer.Exit()
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@official_app.command()
|
|
91
|
+
def run(
|
|
92
|
+
csv_path: Annotated[Path, typer.Argument(help="CSV source file.")],
|
|
93
|
+
report_path: Annotated[
|
|
94
|
+
Path, typer.Argument(help="Complete output filename ending in .html.")
|
|
95
|
+
],
|
|
96
|
+
separator: Annotated[
|
|
97
|
+
str | None, typer.Option("--separator", help="One-character CSV separator.")
|
|
98
|
+
] = None,
|
|
99
|
+
encoding: Annotated[
|
|
100
|
+
str | None, typer.Option("--encoding", help="CSV text encoding.")
|
|
101
|
+
] = None,
|
|
102
|
+
config: Annotated[
|
|
103
|
+
Path | None, typer.Option("--config", help="JSON configuration file.")
|
|
104
|
+
] = None,
|
|
105
|
+
show_version: Annotated[
|
|
106
|
+
bool,
|
|
107
|
+
typer.Option(
|
|
108
|
+
"--version",
|
|
109
|
+
help="Show the installed Tabalyst version and exit.",
|
|
110
|
+
is_eager=True,
|
|
111
|
+
callback=_version_callback,
|
|
112
|
+
),
|
|
113
|
+
] = False,
|
|
114
|
+
) -> None:
|
|
115
|
+
"""Analyze CSV_PATH into REPORT_PATH and its sibling JSON profile."""
|
|
116
|
+
try:
|
|
117
|
+
result = analyze_report(
|
|
118
|
+
csv_path,
|
|
119
|
+
report_path,
|
|
120
|
+
separator=separator,
|
|
121
|
+
encoding=encoding,
|
|
122
|
+
config_path=config,
|
|
123
|
+
)
|
|
124
|
+
except TabalystError as exc:
|
|
125
|
+
typer.echo(f"Error: {exc}", err=True)
|
|
126
|
+
raise typer.Exit(1) from exc
|
|
127
|
+
_print_success(result, report_path)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
@app.command("analyze")
|
|
131
|
+
def analyze_legacy(
|
|
132
|
+
source: Annotated[Path, typer.Argument(exists=True, dir_okay=False, readable=True)],
|
|
133
|
+
output: Annotated[Path, typer.Option("--output", "-o")] = Path("report.html"),
|
|
134
|
+
delimiter: Annotated[
|
|
135
|
+
str | None, typer.Option("--delimiter", "--separator")
|
|
136
|
+
] = None,
|
|
137
|
+
encoding: Annotated[str | None, typer.Option("--encoding")] = None,
|
|
138
|
+
preview_rows: Annotated[
|
|
139
|
+
int | None, typer.Option("--preview-rows", min=0, max=100)
|
|
140
|
+
] = None,
|
|
141
|
+
config: Annotated[
|
|
142
|
+
list[Path] | None, typer.Option("--config", exists=True, dir_okay=False)
|
|
143
|
+
] = None,
|
|
144
|
+
) -> None:
|
|
145
|
+
"""Run the alpha command and retain its cumulative execution history."""
|
|
146
|
+
json_output = output.with_suffix(".json")
|
|
147
|
+
execution_output = output.parent / EXECUTION_LOG_NAME
|
|
148
|
+
temporary_config: Path | None = None
|
|
149
|
+
try:
|
|
150
|
+
config_paths = analysis_config_paths(config or [])
|
|
151
|
+
check_outputs(
|
|
152
|
+
[source, *config_paths], [output, json_output, execution_output]
|
|
153
|
+
)
|
|
154
|
+
effective_config = config_paths
|
|
155
|
+
if preview_rows is not None:
|
|
156
|
+
settings = load_config(config_paths).model_dump()
|
|
157
|
+
settings["preview_rows"] = preview_rows
|
|
158
|
+
with tempfile.NamedTemporaryFile(
|
|
159
|
+
"w", encoding="utf-8", suffix=".json", delete=False
|
|
160
|
+
) as stream:
|
|
161
|
+
json.dump(settings, stream, indent=2, ensure_ascii=False)
|
|
162
|
+
stream.write("\n")
|
|
163
|
+
temporary_config = Path(stream.name)
|
|
164
|
+
effective_config = [temporary_config]
|
|
165
|
+
|
|
166
|
+
result = analyze_report(
|
|
167
|
+
source,
|
|
168
|
+
output,
|
|
169
|
+
separator=delimiter,
|
|
170
|
+
encoding=encoding,
|
|
171
|
+
config_path=effective_config,
|
|
172
|
+
)
|
|
173
|
+
except (TabalystError, OSError, ValueError, LookupError) as exc:
|
|
174
|
+
typer.echo(f"Error: {exc}", err=True)
|
|
175
|
+
raise typer.Exit(1) from exc
|
|
176
|
+
finally:
|
|
177
|
+
if temporary_config is not None:
|
|
178
|
+
temporary_config.unlink(missing_ok=True)
|
|
179
|
+
|
|
180
|
+
_print_success(result, output)
|
|
181
|
+
if config_paths:
|
|
182
|
+
typer.echo(
|
|
183
|
+
"Configuration: "
|
|
184
|
+
+ ", ".join(str(path.resolve()) for path in config_paths)
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
@app.command()
|
|
189
|
+
def render(
|
|
190
|
+
source: Annotated[Path, typer.Argument(exists=True, dir_okay=False, readable=True)],
|
|
191
|
+
output: Annotated[Path, typer.Option("--output", "-o")] = Path("report.html"),
|
|
192
|
+
) -> None:
|
|
193
|
+
"""Regenerate HTML from a report JSON profile without rereading the CSV."""
|
|
194
|
+
try:
|
|
195
|
+
check_outputs([source], [output])
|
|
196
|
+
write_output(output, render_report(load_profile(source)))
|
|
197
|
+
except (OSError, ValueError) as exc:
|
|
198
|
+
typer.echo(f"Error: {exc}", err=True)
|
|
199
|
+
raise typer.Exit(1) from exc
|
|
200
|
+
typer.echo(f"HTML: {output.resolve()}")
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def main() -> None:
|
|
204
|
+
"""Dispatch stable syntax while preserving the two alpha subcommands."""
|
|
205
|
+
if len(sys.argv) > 1 and sys.argv[1] in {"analyze", "render"}:
|
|
206
|
+
app()
|
|
207
|
+
else:
|
|
208
|
+
official_app()
|
tabalyst/config.py
ADDED
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
"""Experimental configuration. Later files override earlier settings."""
|
|
2
|
+
|
|
3
|
+
import codecs
|
|
4
|
+
from collections.abc import Iterable
|
|
5
|
+
from itertools import pairwise
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Literal
|
|
8
|
+
|
|
9
|
+
from pydantic import (
|
|
10
|
+
BaseModel,
|
|
11
|
+
ConfigDict,
|
|
12
|
+
Field,
|
|
13
|
+
ValidationError,
|
|
14
|
+
field_validator,
|
|
15
|
+
model_validator,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
from tabalyst.errors import ConfigurationError
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class CsvConfig(BaseModel):
|
|
22
|
+
model_config = ConfigDict(extra="forbid")
|
|
23
|
+
|
|
24
|
+
encoding: str = "utf-8-sig"
|
|
25
|
+
delimiter: str = ","
|
|
26
|
+
|
|
27
|
+
@field_validator("encoding")
|
|
28
|
+
@classmethod
|
|
29
|
+
def known_encoding(cls, value: str) -> str:
|
|
30
|
+
try:
|
|
31
|
+
codecs.lookup(value)
|
|
32
|
+
except LookupError as exc:
|
|
33
|
+
raise ValueError(f"Unknown encoding: {value}") from exc
|
|
34
|
+
return value
|
|
35
|
+
|
|
36
|
+
@field_validator("delimiter")
|
|
37
|
+
@classmethod
|
|
38
|
+
def single_separator(cls, value: str) -> str:
|
|
39
|
+
if len(value) != 1 or value in '\r\n"\0':
|
|
40
|
+
raise ValueError(
|
|
41
|
+
"Delimiter must be one character, excluding quotes/NUL/newlines"
|
|
42
|
+
)
|
|
43
|
+
return value
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class ValueExamplesConfig(BaseModel):
|
|
47
|
+
model_config = ConfigDict(extra="forbid")
|
|
48
|
+
|
|
49
|
+
full_distribution_max_distinct: int = Field(default=50, ge=1)
|
|
50
|
+
candidate_sample_size: int = Field(default=100, ge=1)
|
|
51
|
+
short_text_max_length: int = Field(default=20, ge=1)
|
|
52
|
+
short_text_percentile: float = Field(default=0.95, gt=0, le=1)
|
|
53
|
+
short_text_result_size: int = Field(default=20, ge=1)
|
|
54
|
+
long_text_result_size: int = Field(default=20, ge=1)
|
|
55
|
+
long_text_truncate_at: int = Field(default=30, ge=1)
|
|
56
|
+
truncation_suffix: str = "..."
|
|
57
|
+
inline_display_size: int = Field(default=3, ge=1)
|
|
58
|
+
random_seed: int = 42
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class NormalizationConfig(BaseModel):
|
|
62
|
+
model_config = ConfigDict(extra="forbid")
|
|
63
|
+
|
|
64
|
+
trim: bool = True
|
|
65
|
+
collapse_internal_whitespace: bool = True
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class DateDetectionConfig(BaseModel):
|
|
69
|
+
model_config = ConfigDict(extra="forbid")
|
|
70
|
+
|
|
71
|
+
enabled: bool = True
|
|
72
|
+
orders: list[Literal["YMD", "MDY", "DMY"]] = Field(
|
|
73
|
+
default_factory=lambda: ["YMD", "MDY", "DMY"]
|
|
74
|
+
)
|
|
75
|
+
separators: list[str] = Field(default_factory=lambda: ["-", "/", "."])
|
|
76
|
+
ambiguous_order: Literal["MDY", "DMY"] | None = None
|
|
77
|
+
|
|
78
|
+
@field_validator("separators")
|
|
79
|
+
@classmethod
|
|
80
|
+
def valid_separators(cls, values: list[str]) -> list[str]:
|
|
81
|
+
if not values or any(
|
|
82
|
+
len(value) != 1 or value.isdigit() or value in "\r\n" for value in values
|
|
83
|
+
):
|
|
84
|
+
raise ValueError("Date separators must be non-digit single characters")
|
|
85
|
+
if len(set(values)) != len(values):
|
|
86
|
+
raise ValueError("Date separators must be unique")
|
|
87
|
+
return values
|
|
88
|
+
|
|
89
|
+
@model_validator(mode="after")
|
|
90
|
+
def ambiguous_order_must_be_enabled(self):
|
|
91
|
+
if self.ambiguous_order and self.ambiguous_order not in self.orders:
|
|
92
|
+
raise ValueError("ambiguous_order must also appear in date orders")
|
|
93
|
+
return self
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class TypeInferenceConfig(BaseModel):
|
|
97
|
+
model_config = ConfigDict(extra="forbid")
|
|
98
|
+
|
|
99
|
+
minimum_confidence: float = Field(default=0.95, gt=0.5, le=1)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class StringAnalysisConfig(BaseModel):
|
|
103
|
+
model_config = ConfigDict(extra="forbid")
|
|
104
|
+
|
|
105
|
+
very_short_max_length: int = Field(default=5, ge=1)
|
|
106
|
+
short_max_length: int = Field(default=20, ge=1)
|
|
107
|
+
medium_max_length: int = Field(default=50, ge=1)
|
|
108
|
+
long_max_length: int = Field(default=255, ge=1)
|
|
109
|
+
length_distribution_max_length: int = Field(default=50, ge=1)
|
|
110
|
+
examples_per_length: int = Field(default=10, ge=1)
|
|
111
|
+
|
|
112
|
+
@model_validator(mode="after")
|
|
113
|
+
def length_thresholds_must_increase(self):
|
|
114
|
+
thresholds = [
|
|
115
|
+
self.very_short_max_length,
|
|
116
|
+
self.short_max_length,
|
|
117
|
+
self.medium_max_length,
|
|
118
|
+
self.long_max_length,
|
|
119
|
+
]
|
|
120
|
+
if any(left >= right for left, right in pairwise(thresholds)):
|
|
121
|
+
raise ValueError("String length thresholds must increase")
|
|
122
|
+
if self.length_distribution_max_length > self.medium_max_length:
|
|
123
|
+
raise ValueError(
|
|
124
|
+
"length_distribution_max_length cannot exceed medium_max_length"
|
|
125
|
+
)
|
|
126
|
+
return self
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
class EnumDetectionConfig(BaseModel):
|
|
130
|
+
model_config = ConfigDict(extra="forbid")
|
|
131
|
+
|
|
132
|
+
enabled: bool = True
|
|
133
|
+
minimum_row_count: int = Field(default=500, ge=1)
|
|
134
|
+
maximum_distinct_values: int = Field(default=49, ge=1)
|
|
135
|
+
eligible_types: list[str] = Field(default_factory=lambda: ["text"])
|
|
136
|
+
case_sensitive: bool = True
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class AnalysisConfig(BaseModel):
|
|
140
|
+
model_config = ConfigDict(extra="forbid")
|
|
141
|
+
|
|
142
|
+
csv: CsvConfig = Field(default_factory=CsvConfig)
|
|
143
|
+
missing_values: list[str] = Field(default_factory=lambda: [""])
|
|
144
|
+
preview_rows: int = Field(default=10, ge=0, le=100)
|
|
145
|
+
normalization: NormalizationConfig = Field(default_factory=NormalizationConfig)
|
|
146
|
+
date_detection: DateDetectionConfig = Field(default_factory=DateDetectionConfig)
|
|
147
|
+
type_inference: TypeInferenceConfig = Field(default_factory=TypeInferenceConfig)
|
|
148
|
+
string_analysis: StringAnalysisConfig = Field(default_factory=StringAnalysisConfig)
|
|
149
|
+
value_examples: ValueExamplesConfig = Field(default_factory=ValueExamplesConfig)
|
|
150
|
+
enum_detection: EnumDetectionConfig = Field(default_factory=EnumDetectionConfig)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _validation_message(exc: ValidationError) -> str:
|
|
154
|
+
first = exc.errors(include_url=False)[0]
|
|
155
|
+
location = ".".join(str(part) for part in first["loc"])
|
|
156
|
+
message = first["msg"]
|
|
157
|
+
return f"{location}: {message}" if location else message
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def load_config(paths: Iterable[Path]) -> AnalysisConfig:
|
|
161
|
+
"""Load and recursively merge strict JSON configuration files."""
|
|
162
|
+
|
|
163
|
+
def merge(base: dict, override: dict) -> dict:
|
|
164
|
+
result = dict(base)
|
|
165
|
+
for key, value in override.items():
|
|
166
|
+
if isinstance(value, dict) and isinstance(result.get(key), dict):
|
|
167
|
+
result[key] = merge(result[key], value)
|
|
168
|
+
else:
|
|
169
|
+
result[key] = value
|
|
170
|
+
return result
|
|
171
|
+
|
|
172
|
+
merged = {}
|
|
173
|
+
for path in paths:
|
|
174
|
+
try:
|
|
175
|
+
content = path.read_text(encoding="utf-8-sig")
|
|
176
|
+
except OSError as exc:
|
|
177
|
+
raise ConfigurationError(
|
|
178
|
+
f"Cannot read configuration file {path}: {exc}"
|
|
179
|
+
) from exc
|
|
180
|
+
try:
|
|
181
|
+
current = AnalysisConfig.model_validate_json(content).model_dump(
|
|
182
|
+
exclude_unset=True
|
|
183
|
+
)
|
|
184
|
+
except ValidationError as exc:
|
|
185
|
+
raise ConfigurationError(
|
|
186
|
+
f"Invalid configuration in {path}: {_validation_message(exc)}"
|
|
187
|
+
) from exc
|
|
188
|
+
merged = merge(merged, current)
|
|
189
|
+
try:
|
|
190
|
+
return AnalysisConfig.model_validate(merged)
|
|
191
|
+
except ValidationError as exc:
|
|
192
|
+
raise ConfigurationError(
|
|
193
|
+
f"Invalid merged configuration: {_validation_message(exc)}"
|
|
194
|
+
) from exc
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def resolve_config(
|
|
198
|
+
paths: Iterable[Path] = (),
|
|
199
|
+
*,
|
|
200
|
+
separator: str | None = None,
|
|
201
|
+
encoding: str | None = None,
|
|
202
|
+
) -> AnalysisConfig:
|
|
203
|
+
"""Resolve defaults, files and explicit values in one shared layer."""
|
|
204
|
+
settings = load_config(paths).model_dump()
|
|
205
|
+
if separator is not None:
|
|
206
|
+
settings["csv"]["delimiter"] = separator
|
|
207
|
+
if encoding is not None:
|
|
208
|
+
settings["csv"]["encoding"] = encoding
|
|
209
|
+
try:
|
|
210
|
+
return AnalysisConfig.model_validate(settings)
|
|
211
|
+
except ValidationError as exc:
|
|
212
|
+
raise ConfigurationError(
|
|
213
|
+
f"Invalid configuration: {_validation_message(exc)}"
|
|
214
|
+
) from exc
|
tabalyst/errors.py
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""Public exceptions raised by Tabalyst."""
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class TabalystError(Exception):
|
|
5
|
+
"""Base class for expected Tabalyst failures."""
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class InputError(TabalystError, ValueError):
|
|
9
|
+
"""The CSV input or requested report path is invalid."""
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class ConfigurationError(TabalystError, ValueError):
|
|
13
|
+
"""A configuration file or explicit configuration value is invalid."""
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class ReportError(TabalystError):
|
|
17
|
+
"""Tabalyst could not write or render the requested report."""
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""Persistent performance history for successful CLI analyses."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import subprocess
|
|
5
|
+
from datetime import UTC, datetime
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from tabalyst._version import get_version
|
|
10
|
+
from tabalyst.models import DatasetProfile
|
|
11
|
+
|
|
12
|
+
EXECUTION_LOG_NAME = "executions.json"
|
|
13
|
+
EXECUTION_SCHEMA_VERSION = "1.0"
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def tabalyst_version() -> str:
|
|
17
|
+
"""Return the package version for execution-history compatibility."""
|
|
18
|
+
return get_version()
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def git_state(directory: Path) -> dict[str, Any]:
|
|
22
|
+
"""Describe the current checkout without requiring Git in production."""
|
|
23
|
+
try:
|
|
24
|
+
commit = subprocess.run(
|
|
25
|
+
["git", "rev-parse", "HEAD"],
|
|
26
|
+
cwd=directory,
|
|
27
|
+
check=True,
|
|
28
|
+
capture_output=True,
|
|
29
|
+
text=True,
|
|
30
|
+
timeout=2,
|
|
31
|
+
).stdout.strip()
|
|
32
|
+
dirty = bool(
|
|
33
|
+
subprocess.run(
|
|
34
|
+
["git", "status", "--porcelain"],
|
|
35
|
+
cwd=directory,
|
|
36
|
+
check=True,
|
|
37
|
+
capture_output=True,
|
|
38
|
+
text=True,
|
|
39
|
+
timeout=2,
|
|
40
|
+
).stdout.strip()
|
|
41
|
+
)
|
|
42
|
+
except (OSError, subprocess.SubprocessError):
|
|
43
|
+
return {"available": False, "commit": None, "dirty": None, "state": None}
|
|
44
|
+
suffix = "+working" if dirty else ""
|
|
45
|
+
return {
|
|
46
|
+
"available": True,
|
|
47
|
+
"commit": commit,
|
|
48
|
+
"dirty": dirty,
|
|
49
|
+
"state": f"{commit[:8]}{suffix}",
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def build_execution_entry(
|
|
54
|
+
*,
|
|
55
|
+
source: Path,
|
|
56
|
+
html_output: Path,
|
|
57
|
+
json_output: Path,
|
|
58
|
+
profile: DatasetProfile,
|
|
59
|
+
total_seconds: float,
|
|
60
|
+
git_directory: Path,
|
|
61
|
+
) -> dict[str, Any]:
|
|
62
|
+
"""Build one portable history item using names relative to its output folder."""
|
|
63
|
+
return {
|
|
64
|
+
"timestamp": datetime.now(UTC).isoformat().replace("+00:00", "Z"),
|
|
65
|
+
"tabalyst_version": tabalyst_version(),
|
|
66
|
+
"source_file": source.name,
|
|
67
|
+
"html_file": html_output.name,
|
|
68
|
+
"json_file": json_output.name,
|
|
69
|
+
"rows": profile.summary.row_count,
|
|
70
|
+
"columns": profile.summary.column_count,
|
|
71
|
+
"analysis_seconds": profile.processing_seconds,
|
|
72
|
+
"total_seconds": round(total_seconds, 4),
|
|
73
|
+
"git": git_state(git_directory),
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def append_execution(path: Path, entry: dict[str, Any]) -> None:
|
|
78
|
+
"""Append an item and atomically replace the JSON history file."""
|
|
79
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
80
|
+
if path.exists():
|
|
81
|
+
history = json.loads(path.read_text(encoding="utf-8"))
|
|
82
|
+
if not isinstance(history, dict) or not isinstance(
|
|
83
|
+
history.get("executions"), list
|
|
84
|
+
):
|
|
85
|
+
raise ValueError(f"Invalid execution history: {path}")
|
|
86
|
+
if history.get("schema_version") != EXECUTION_SCHEMA_VERSION:
|
|
87
|
+
raise ValueError(
|
|
88
|
+
f"Unsupported execution history schema: {path}"
|
|
89
|
+
)
|
|
90
|
+
else:
|
|
91
|
+
history = {
|
|
92
|
+
"schema_version": EXECUTION_SCHEMA_VERSION,
|
|
93
|
+
"executions": [],
|
|
94
|
+
}
|
|
95
|
+
history["executions"].append(entry)
|
|
96
|
+
temporary = path.with_name(f".{path.name}.tmp")
|
|
97
|
+
temporary.write_text(
|
|
98
|
+
json.dumps(history, indent=2, ensure_ascii=False) + "\n",
|
|
99
|
+
encoding="utf-8",
|
|
100
|
+
newline="\n",
|
|
101
|
+
)
|
|
102
|
+
temporary.replace(path)
|
tabalyst/ingestion.py
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""Validate CSV structure before asking pandas to load the raw strings."""
|
|
2
|
+
|
|
3
|
+
import csv
|
|
4
|
+
import hashlib
|
|
5
|
+
import io
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import pandas as pd
|
|
10
|
+
|
|
11
|
+
from tabalyst.config import CsvConfig
|
|
12
|
+
from tabalyst.errors import InputError
|
|
13
|
+
from tabalyst.models import SourceInfo
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class CsvInputError(InputError):
|
|
17
|
+
pass
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass
|
|
21
|
+
class CsvDataset:
|
|
22
|
+
frame: pd.DataFrame
|
|
23
|
+
headers: list[str]
|
|
24
|
+
source: SourceInfo
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def read_csv(path: Path, config: CsvConfig) -> CsvDataset:
|
|
28
|
+
raw = path.read_bytes()
|
|
29
|
+
try:
|
|
30
|
+
content = raw.decode(config.encoding)
|
|
31
|
+
except UnicodeError as exc:
|
|
32
|
+
raise CsvInputError(
|
|
33
|
+
f"Cannot decode {path.name} as {config.encoding}. Specify --encoding."
|
|
34
|
+
) from exc
|
|
35
|
+
if "\0" in content:
|
|
36
|
+
raise CsvInputError("CSV contains NUL characters; verify the file encoding.")
|
|
37
|
+
|
|
38
|
+
reader = csv.reader(
|
|
39
|
+
io.StringIO(content, newline=""), delimiter=config.delimiter, strict=True
|
|
40
|
+
)
|
|
41
|
+
row_count = 0
|
|
42
|
+
try:
|
|
43
|
+
headers = next(reader, None)
|
|
44
|
+
if not headers:
|
|
45
|
+
raise CsvInputError("CSV is empty or has no header on its first record.")
|
|
46
|
+
for row_count, row in enumerate(reader, start=1):
|
|
47
|
+
if len(row) != len(headers):
|
|
48
|
+
raise CsvInputError(
|
|
49
|
+
f"Data record {row_count} (ending at physical line {reader.line_num}): "
|
|
50
|
+
f"expected {len(headers)} fields, found {len(row)}. "
|
|
51
|
+
"Verify the delimiter and quoting. No rows were skipped."
|
|
52
|
+
)
|
|
53
|
+
except csv.Error as exc:
|
|
54
|
+
raise CsvInputError(
|
|
55
|
+
f"Invalid CSV near physical line {reader.line_num}: {exc}"
|
|
56
|
+
) from exc
|
|
57
|
+
|
|
58
|
+
# Internal IDs preserve duplicate/blank headers without pandas renaming them.
|
|
59
|
+
ids = [f"column_{position}" for position in range(1, len(headers) + 1)]
|
|
60
|
+
frame = pd.read_csv(
|
|
61
|
+
io.StringIO(content, newline=""),
|
|
62
|
+
sep=config.delimiter,
|
|
63
|
+
engine="python",
|
|
64
|
+
header=0,
|
|
65
|
+
names=ids,
|
|
66
|
+
dtype=str,
|
|
67
|
+
na_filter=False,
|
|
68
|
+
keep_default_na=False,
|
|
69
|
+
skip_blank_lines=False,
|
|
70
|
+
on_bad_lines="error",
|
|
71
|
+
)
|
|
72
|
+
if len(frame) != row_count:
|
|
73
|
+
raise CsvInputError(
|
|
74
|
+
"CSV record count changed during parsing; analysis was stopped."
|
|
75
|
+
)
|
|
76
|
+
return CsvDataset(
|
|
77
|
+
frame=frame,
|
|
78
|
+
headers=headers,
|
|
79
|
+
source=SourceInfo(
|
|
80
|
+
filename=path.name,
|
|
81
|
+
size_bytes=len(raw),
|
|
82
|
+
sha256=hashlib.sha256(raw).hexdigest(),
|
|
83
|
+
encoding=config.encoding,
|
|
84
|
+
delimiter=config.delimiter,
|
|
85
|
+
),
|
|
86
|
+
)
|