pyautostat 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. pyautostat-0.1.0/.github/workflows/ci.yml +72 -0
  2. pyautostat-0.1.0/.github/workflows/publish.yml +92 -0
  3. pyautostat-0.1.0/.gitignore +16 -0
  4. pyautostat-0.1.0/.pre-commit-config.yaml +24 -0
  5. pyautostat-0.1.0/AGENTS.md +128 -0
  6. pyautostat-0.1.0/API_REFERENCE.md +158 -0
  7. pyautostat-0.1.0/CHANGELOG.md +81 -0
  8. pyautostat-0.1.0/LICENSE +21 -0
  9. pyautostat-0.1.0/PKG-INFO +146 -0
  10. pyautostat-0.1.0/README.md +108 -0
  11. pyautostat-0.1.0/ROADMAP.md +18 -0
  12. pyautostat-0.1.0/examples/README.md +58 -0
  13. pyautostat-0.1.0/examples/example_usage.py +373 -0
  14. pyautostat-0.1.0/pyproject.toml +79 -0
  15. pyautostat-0.1.0/src/pyautostat/__init__.py +33 -0
  16. pyautostat-0.1.0/src/pyautostat/analyzer.py +1039 -0
  17. pyautostat-0.1.0/src/pyautostat/categorical.py +170 -0
  18. pyautostat-0.1.0/src/pyautostat/detection.py +175 -0
  19. pyautostat-0.1.0/src/pyautostat/exceptions.py +29 -0
  20. pyautostat-0.1.0/src/pyautostat/insights.py +266 -0
  21. pyautostat-0.1.0/src/pyautostat/report.py +864 -0
  22. pyautostat-0.1.0/tests/conftest.py +90 -0
  23. pyautostat-0.1.0/tests/test_analyzer_correlation.py +21 -0
  24. pyautostat-0.1.0/tests/test_analyzer_descriptive.py +34 -0
  25. pyautostat-0.1.0/tests/test_analyzer_distributions.py +20 -0
  26. pyautostat-0.1.0/tests/test_analyzer_effect_size_bands.py +28 -0
  27. pyautostat-0.1.0/tests/test_analyzer_histograms.py +15 -0
  28. pyautostat-0.1.0/tests/test_analyzer_hypothesis.py +139 -0
  29. pyautostat-0.1.0/tests/test_analyzer_missing_and_quality.py +28 -0
  30. pyautostat-0.1.0/tests/test_analyzer_normality.py +23 -0
  31. pyautostat-0.1.0/tests/test_analyzer_outliers.py +20 -0
  32. pyautostat-0.1.0/tests/test_analyzer_overview.py +12 -0
  33. pyautostat-0.1.0/tests/test_categorical.py +96 -0
  34. pyautostat-0.1.0/tests/test_detection.py +86 -0
  35. pyautostat-0.1.0/tests/test_examples.py +88 -0
  36. pyautostat-0.1.0/tests/test_exceptions.py +56 -0
  37. pyautostat-0.1.0/tests/test_insights.py +66 -0
  38. pyautostat-0.1.0/tests/test_report.py +83 -0
  39. pyautostat-0.1.0/tests/test_report_interactive.py +80 -0
  40. pyautostat-0.1.0/tests/test_robustness.py +197 -0
@@ -0,0 +1,72 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ branches: [main, master]
6
+ pull_request:
7
+ branches: [main, master]
8
+ workflow_dispatch:
9
+
10
+ jobs:
11
+ lint:
12
+ name: Lint & type-check
13
+ runs-on: ubuntu-latest
14
+ steps:
15
+ - uses: actions/checkout@v4
16
+ - uses: actions/setup-python@v5
17
+ with:
18
+ python-version: "3.12"
19
+ cache: pip
20
+ cache-dependency-path: pyproject.toml
21
+ - name: Install package with dev extras
22
+ run: pip install -e ".[dev]"
23
+ - name: ruff check
24
+ run: ruff check src tests
25
+ - name: ruff format --check
26
+ run: ruff format --check src tests
27
+ - name: mypy
28
+ run: mypy src/pyautostat
29
+
30
+ test:
31
+ name: Test (Python ${{ matrix.python-version }}, ${{ matrix.os }})
32
+ needs: lint
33
+ runs-on: ${{ matrix.os }}
34
+ strategy:
35
+ fail-fast: false
36
+ matrix:
37
+ os: [ubuntu-latest, windows-latest]
38
+ python-version: ["3.10", "3.11", "3.12", "3.13"]
39
+ steps:
40
+ - uses: actions/checkout@v4
41
+ - uses: actions/setup-python@v5
42
+ with:
43
+ python-version: ${{ matrix.python-version }}
44
+ cache: pip
45
+ cache-dependency-path: pyproject.toml
46
+ - name: Install package with dev extras
47
+ run: pip install -e ".[dev]"
48
+ - name: pytest
49
+ run: pytest -q --cov=pyautostat --cov-report=term-missing --cov-fail-under=90
50
+
51
+ build:
52
+ name: Build distribution
53
+ needs: lint
54
+ runs-on: ubuntu-latest
55
+ steps:
56
+ - uses: actions/checkout@v4
57
+ - uses: actions/setup-python@v5
58
+ with:
59
+ python-version: "3.12"
60
+ cache: pip
61
+ cache-dependency-path: pyproject.toml
62
+ - name: Install build tools
63
+ run: pip install build twine
64
+ - name: Build sdist and wheel
65
+ run: python -m build
66
+ - name: Check distribution metadata
67
+ run: twine check dist/*
68
+ - name: Upload build artifacts
69
+ uses: actions/upload-artifact@v4
70
+ with:
71
+ name: dist
72
+ path: dist/
@@ -0,0 +1,92 @@
1
+ name: Publish PyAutoStat to PyPI
2
+
3
+ on:
4
+ release:
5
+ types: [published]
6
+
7
+ jobs:
8
+ build:
9
+ runs-on: ubuntu-latest
10
+ permissions:
11
+ contents: read
12
+
13
+ steps:
14
+ - name: Checkout code
15
+ uses: actions/checkout@v4
16
+ with:
17
+ persist-credentials: false
18
+
19
+ - name: Set up Python
20
+ uses: actions/setup-python@v5
21
+ with:
22
+ python-version: "3.12"
23
+
24
+ - name: Install build tools
25
+ run: python -m pip install --upgrade build twine
26
+
27
+ - name: Verify release version
28
+ shell: bash
29
+ run: |
30
+ python - <<'PY'
31
+ import ast
32
+ import os
33
+ from pathlib import Path
34
+
35
+ tree = ast.parse(
36
+ Path("src/pyautostat/__init__.py").read_text(
37
+ encoding="utf-8"
38
+ )
39
+ )
40
+ version = next(
41
+ ast.literal_eval(node.value)
42
+ for node in tree.body
43
+ if isinstance(node, ast.Assign)
44
+ and any(
45
+ isinstance(target, ast.Name)
46
+ and target.id == "__version__"
47
+ for target in node.targets
48
+ )
49
+ )
50
+
51
+ expected = f"v{version}"
52
+ actual = os.environ["GITHUB_REF_NAME"]
53
+
54
+ if actual != expected:
55
+ raise SystemExit(
56
+ f"Tag {actual!r} does not match {expected!r}"
57
+ )
58
+ PY
59
+
60
+ - name: Build package
61
+ run: python -m build
62
+
63
+ - name: Check distributions
64
+ run: python -m twine check dist/*
65
+
66
+ - name: Upload distributions
67
+ uses: actions/upload-artifact@v4
68
+ with:
69
+ name: release-distributions
70
+ path: dist/
71
+ if-no-files-found: error
72
+
73
+ publish:
74
+ needs: build
75
+ runs-on: ubuntu-latest
76
+
77
+ environment:
78
+ name: pypi
79
+ url: https://pypi.org/project/pyautostat/
80
+
81
+ permissions:
82
+ id-token: write
83
+
84
+ steps:
85
+ - name: Download distributions
86
+ uses: actions/download-artifact@v4
87
+ with:
88
+ name: release-distributions
89
+ path: dist/
90
+
91
+ - name: Publish to PyPI
92
+ uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,16 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *.egg-info/
4
+ .eggs/
5
+ build/
6
+ dist/
7
+ reports/
8
+ .pytest_cache/
9
+ .mypy_cache/
10
+ .ruff_cache/
11
+ .coverage
12
+ htmlcov/
13
+ .venv/
14
+ venv/
15
+ .vscode/
16
+ .idea/
@@ -0,0 +1,24 @@
1
+ repos:
2
+ - repo: https://github.com/pre-commit/pre-commit-hooks
3
+ rev: v5.0.0
4
+ hooks:
5
+ - id: trailing-whitespace
6
+ - id: end-of-file-fixer
7
+ - id: check-yaml
8
+ - id: check-toml
9
+ - id: check-added-large-files
10
+ - id: check-merge-conflict
11
+
12
+ - repo: https://github.com/astral-sh/ruff-pre-commit
13
+ rev: v0.16.8
14
+ hooks:
15
+ - id: ruff
16
+ args: [--fix]
17
+ - id: ruff-format
18
+
19
+ - repo: https://github.com/pre-commit/mirrors-mypy
20
+ rev: v2.3.1
21
+ hooks:
22
+ - id: mypy
23
+ files: ^src/
24
+ additional_dependencies: [pandas, numpy, scipy]
@@ -0,0 +1,128 @@
1
+ # PyAutoStat development guide
2
+
3
+ This file gives future contributors and coding agents the repository context and
4
+ working rules for PyAutoStat. Follow the user's current request first. Keep this
5
+ guide aligned with the code as the package evolves.
6
+
7
+ ## Project context
8
+
9
+ - PyAutoStat is a Python package for automated statistical analysis of pandas
10
+ DataFrames, aimed at researchers and analysts. The package is currently an
11
+ alpha release (`0.1.0` in `src/pyautostat/__init__.py`).
12
+ - The package uses a `src` layout and is built with Hatchling. Python `>=3.10`
13
+ is supported. Runtime dependencies are pandas, NumPy, and SciPy; Plotly is an
14
+ optional dependency for interactive reports. `pyproject.toml` is the source
15
+ of truth for package metadata, dependencies, and tool configuration.
16
+ - Public imports are defined in `src/pyautostat/__init__.py`. Preserve them and
17
+ the shape of existing result dictionaries unless an intentional API change
18
+ has been requested and documented.
19
+ - `StatisticalAnalyzer` in `analyzer.py` copies the input DataFrame.
20
+ `analyze_all()` returns overview, descriptive, normality, outlier,
21
+ correlation, missing-data, data-quality, distribution, column-role,
22
+ column-type, histogram, and `analysis_warnings` sections. `hypothesis_tests()` is a separate
23
+ operation; its results are not added to `analyze_all()` automatically.
24
+ - `categorical_association()` is a separate independent-sample chi-square
25
+ comparison with Cramér's V. It requires expected cell counts of at least
26
+ five. Cohen's h is available for 2x2 tables with an explicit success value.
27
+ - `detection.py` provides advisory column-type and name-based role heuristics,
28
+ including contact-format and missingness hints.
29
+ These suggestions do not alter the analysis or exclude columns.
30
+ - `InsightEngine` in `insights.py` converts analysis results into severity-rated
31
+ findings and recommendations. `ReportGenerator` in `report.py` exports dict,
32
+ JSON, CSV, static HTML, and optional Plotly HTML.
33
+ - `examples/example_usage.py` is the runnable end-to-end feature showcase;
34
+ `examples/README.md` maps public features to its sections. `ROADMAP.md`
35
+ records product goals and distinguishes shipped features from future work. Tests live
36
+ under `tests/`; CI runs tests with coverage on Linux and Windows, and runs
37
+ lint, formatting, mypy, and a package build on Linux.
38
+
39
+ ## Source of truth and scope
40
+
41
+ - Read the relevant source and tests before changing behavior. `ROADMAP.md`
42
+ contains proposed work, not a specification to apply wholesale. In particular,
43
+ do not assume that streaming/chunking, post-hoc tests, imputation, regulatory
44
+ compliance, or publication-ready formatting are implemented.
45
+ - Keep feature claims in `README.md`, `API_REFERENCE.md`, and `examples/README.md`
46
+ consistent with shipped behavior. Treat performance and market claims as
47
+ unverified unless measured or sourced.
48
+ - Make focused changes. Update public documentation and `CHANGELOG.md` when a
49
+ user-visible API, result schema, behavior, dependency, or supported Python
50
+ version changes.
51
+
52
+ ## Statistical and data-handling standards
53
+
54
+ - Prefer well-defined methods from SciPy, NumPy, and pandas. State the test,
55
+ assumptions, sample sizes, effect-size definition, confidence level, and
56
+ direction of comparisons where applicable. Do not equate a p-value with the
57
+ probability that a hypothesis is true, or a nonsignificant result with proof
58
+ of no effect.
59
+ - Do not silently change an analysis based on heuristic column detection.
60
+ Keep user-selected group and value columns explicit. Validate their types,
61
+ distinct groups, and usable observations before running a test.
62
+ - Handle small samples, all-missing columns, constant values, non-finite
63
+ values, and zero denominators deliberately. Return a documented unavailable
64
+ result or raise a specific `PyAutoStatError` subclass with guidance; do not
65
+ emit misleading statistics or silently return an empty result.
66
+ - Check the minimum sample size and other preconditions for each SciPy test.
67
+ D'Agostino-Pearson's `normaltest` needs at least eight observations; the
68
+ current implementation skips smaller samples. `hypothesis_tests(auto)`
69
+ selects between standard ANOVA and Kruskal-Wallis for three or more groups
70
+ using per-group normality and Levene screens. These screens are heuristics,
71
+ and Kruskal-Wallis requires at least five usable observations per group.
72
+ - Preserve the meaning of missing values, group ordering, and paired versus
73
+ independent observations. Do not describe the current independent-group
74
+ tests as suitable for paired before/after data.
75
+ - Avoid global warning suppression in new code. Handle expected numerical
76
+ warnings locally and make invalid or undefined outputs clear to callers.
77
+ - Escape user-supplied labels and findings before inserting them into HTML.
78
+ Reports may contain untrusted DataFrame column names or values.
79
+
80
+ ## Implementation conventions
81
+
82
+ - Keep runtime imports limited to declared dependencies. Import optional
83
+ dependencies inside the feature that needs them and provide a clear install
84
+ message when absent.
85
+ - Use type hints for new or substantially changed public code. Keep functions
86
+ small enough to explain their statistical purpose, and document public
87
+ parameters, return structures, units, and exceptions.
88
+ - Preserve input DataFrames unless a public API explicitly promises mutation.
89
+ Prefer deterministic behavior and local random generators with fixed seeds
90
+ in examples and tests.
91
+ - Keep exports portable across supported Python versions and operating
92
+ systems. Use `pathlib` for new path handling, UTF-8 for text files, and
93
+ avoid assuming output directories already exist without documenting it.
94
+ - Respect the existing Ruff configuration (`E`, `F`, `I`, `UP`, `B`, line
95
+ length 100) and mypy configuration in `pyproject.toml`. Do not add broad
96
+ lint/type ignores to work around a local issue.
97
+
98
+ ## Verification
99
+
100
+ - Add or update focused tests when statistical behavior, result schemas, or
101
+ error handling changes. Compare numerical results with trusted SciPy or
102
+ pandas calculations, and test relevant boundary cases rather than only
103
+ asserting that a key exists. Do not add tests for prose-only or trivial
104
+ reversible changes.
105
+ - Run the checks relevant to the change. The CI commands are:
106
+
107
+ ```text
108
+ python -m ruff check src tests
109
+ python -m ruff format --check src tests
110
+ python -m mypy src/pyautostat
111
+ python -m pytest -q --cov=pyautostat --cov-report=term-missing --cov-fail-under=90
112
+ python -m build
113
+ python -m twine check dist/*
114
+ ```
115
+
116
+ - Install development dependencies with `python -m pip install -e ".[dev]"`
117
+ when the environment permits. Report any check that could not run and why;
118
+ do not claim it passed. CI targets Python 3.10 through 3.13 on Linux and
119
+ Windows. The mypy target is set to Python 3.12 because of NumPy stub syntax.
120
+ - Keep tests independent of network access. Plotly is optional at runtime;
121
+ the interactive HTML currently references Plotly JavaScript from a CDN.
122
+
123
+ ## Before finishing a change
124
+
125
+ - Confirm the public behavior and examples agree with the implementation.
126
+ - Summarize what changed, why, which checks ran, and any remaining limitation.
127
+ - Do not treat this workspace as a Git checkout unless `.git` is present; a
128
+ source snapshot may have no repository metadata.
@@ -0,0 +1,158 @@
1
+ # PyAutoStat API reference
2
+
3
+ This page describes the public API in `pyautostat`. The [README](README.md) has a short start-to-finish example; the [examples guide](examples/README.md) runs every feature and shows its output.
4
+
5
+ ## Imports
6
+
7
+ ```python
8
+ from pyautostat import (
9
+ InsightEngine,
10
+ ReportGenerator,
11
+ StatisticalAnalyzer,
12
+ detect_column_types,
13
+ suggest_column_roles,
14
+ )
15
+ ```
16
+
17
+ All documented exception classes are also exported from `pyautostat`.
18
+
19
+ ## `StatisticalAnalyzer`
20
+
21
+ ### Construction and full analysis
22
+
23
+ ```python
24
+ analyzer = StatisticalAnalyzer(df)
25
+ analysis = analyzer.analyze_all()
26
+ ```
27
+
28
+ `df` must be a nonempty pandas DataFrame with unique, nonempty string column names and scalar values. Numeric values must be finite and real; missing values are allowed. The analyzer copies the DataFrame and exposes `df`, `numeric_cols`, `categorical_cols`, and `all_results`.
29
+
30
+ `analyze_all()` returns a dictionary with these sections:
31
+
32
+ | Key | Contents |
33
+ | --- | --- |
34
+ | `overview` | Shape, row and column counts, memory use, column names and dtypes |
35
+ | `descriptive` | Per-numeric-column count, center, spread, quantiles, skewness and kurtosis |
36
+ | `normality` | Shapiro-Wilk, D'Agostino-Pearson and Anderson-Darling when their preconditions are met |
37
+ | `outliers` | IQR, Z-score and median absolute deviation summaries |
38
+ | `correlation` | Pearson, Spearman and Kendall matrices; pairwise Pearson p-values |
39
+ | `missing_data` | Missing counts and percentages by column and overall |
40
+ | `data_quality` | Completeness, per-column uniqueness and duplicate rows |
41
+ | `distributions` | Skewness and kurtosis descriptions, range and a histogram-based bimodality hint |
42
+ | `column_roles` | Advisory roles derived from column names |
43
+ | `column_types` | Advisory types and missingness hints |
44
+ | `histograms` | Precomputed bin edges and counts for numeric columns |
45
+ | `analysis_warnings` | Records with `code`, `section`, `column` and `message` for skipped or undefined calculations |
46
+
47
+ An unavailable numeric result is `None`. Some tests are skipped for all-missing, constant or short columns; inspect `analysis_warnings`. Correlation uses pairwise nonmissing observations. Only Pearson pairs include p-values.
48
+
49
+ ### Independent group comparisons
50
+
51
+ ```python
52
+ result = analyzer.hypothesis_tests(
53
+ group_col="group",
54
+ value_col="outcome",
55
+ test_type="auto",
56
+ confidence_level=0.95,
57
+ bootstrap_samples=499,
58
+ random_state=0,
59
+ )
60
+ ```
61
+
62
+ `test_type` can be `auto`, `ttest`, `mannwhitney`, `anova` or `kruskal`. Automatic selection uses per-group normality and Levene variance screens. It chooses a t-test or Mann-Whitney U for two groups, and one-way ANOVA or Kruskal-Wallis for three or more groups. Explicitly selected tests are retained, with assumption warnings where appropriate.
63
+
64
+ The result contains `test`, `statistic`, `p_value`, `groups`, `assumptions` and `effect_size`. The assumptions include per-group normality, Levene results, selection reason and warnings. The effect-size record contains its name, value, interpretation and a percentile bootstrap confidence interval. T-tests also return an analytical `confidence_interval` for the first minus second group's mean; a t-test result includes `equal_variance`.
65
+
66
+ Rows missing a group or outcome are excluded. Every group needs at least two usable numeric values. Kruskal-Wallis requires at least five per group for its approximation. Use `bootstrap_samples=0` to omit effect-size intervals, or an integer of at least 100 for resampling. `random_state` controls a local random generator.
67
+
68
+ These tests assume independent observations. The automatic screens cannot verify study design, and bootstrap intervals do not adjust for multiple testing.
69
+
70
+ ### Categorical association
71
+
72
+ ```python
73
+ association = analyzer.categorical_association(
74
+ group_col="group",
75
+ outcome_col="response",
76
+ success_value="yes",
77
+ confidence_level=0.95,
78
+ bootstrap_samples=499,
79
+ random_state=0,
80
+ )
81
+ ```
82
+
83
+ This method runs a Pearson chi-square test of independence without continuity correction. It excludes rows missing either selected value, preserves first-appearance category order, and requires every expected cell count to be at least five.
84
+
85
+ The result includes `test`, `statistic`, `p_value`, `degrees_of_freedom`, `groups`, `outcomes`, `observed_counts`, `expected_counts`, `sample_size`, `assumptions` and Cramér's V in `effect_size`. For a two-by-two table with an explicit `success_value`, it also includes `cohens_h`. Positive Cohen's h means the first group has the higher success proportion. Set `bootstrap_samples=0` to omit bootstrap intervals.
86
+
87
+ ## Column detection
88
+
89
+ ```python
90
+ types = detect_column_types(df)
91
+ roles = suggest_column_roles(df)
92
+ ```
93
+
94
+ `detect_column_types()` returns one record per column with `detected_type`, `notes`, `missing_count` and `missing_percentage`. It recognizes numeric categories, continuous numbers, booleans, datetimes, date-like text and common email, URL and phone formats. Contact and date checks sample up to 20 nonmissing values.
95
+
96
+ `suggest_column_roles()` returns `role`, `reason` and `suggested_action` for each column. Roles include identifier, target, datetime, economic, measurement and unknown. Both helpers are advisory: they do not transform data or select test columns.
97
+
98
+ ## `InsightEngine`
99
+
100
+ ```python
101
+ engine = InsightEngine(analysis)
102
+ findings = engine.generate_insights()
103
+ summary = engine.get_summary()
104
+ ```
105
+
106
+ `generate_insights()` returns a list of findings with category, severity, finding and recommendations. `get_summary()` generates findings if needed and returns `total_insights`, `high_severity`, `medium_severity`, `low_severity` and `insights`.
107
+
108
+ ## `ReportGenerator`
109
+
110
+ ```python
111
+ report = ReportGenerator(
112
+ analysis_results=analysis,
113
+ insights=summary,
114
+ hypothesis_results=[result, association],
115
+ )
116
+
117
+ payload = report.to_dict()
118
+ json_text = report.to_json()
119
+ csv_tables = report.to_csv()
120
+ html_text = report.to_html()
121
+ interactive_html = report.to_interactive_html() # requires the "report" extra
122
+ ```
123
+
124
+ `insights` and `hypothesis_results` are optional. A comparison can be one result dictionary or a list of results. Pass a path to write a format instead of returning its content:
125
+
126
+ ```python
127
+ report.to_json("analysis.json", pretty=True)
128
+ report.to_csv("csv_reports")
129
+ report.to_html("analysis.html", title="Analysis")
130
+ report.to_interactive_html("interactive.html", title="Interactive analysis")
131
+ ```
132
+
133
+ | Method | Without path | With path |
134
+ | --- | --- | --- |
135
+ | `to_dict()` | Dictionary with timestamp, analysis, insights and optionally hypothesis tests | N/A |
136
+ | `to_json(filepath=None, pretty=True)` | JSON string | Writes JSON and returns a status message |
137
+ | `to_csv(output_dir=None)` | Dictionary of DataFrames | Writes CSV files and returns the tables |
138
+ | `to_html(filepath=None, title=...)` | Static HTML string | Writes HTML and returns a status message |
139
+ | `to_interactive_html(filepath=None, title=...)` | Plotly HTML string | Writes HTML and returns a status message |
140
+
141
+ CSV tables include descriptive statistics, outliers, missing data and insights. When comparisons are supplied, they also include hypothesis tests. Written CSV protects formula-like text for spreadsheet use; in-memory DataFrames preserve the original values. JSON represents unavailable or non-finite numbers as `null`. Report HTML escapes data-derived text. Interactive HTML requires Plotly and loads Plotly JavaScript from a CDN when opened.
142
+
143
+ ## Errors
144
+
145
+ All package-specific errors derive from `PyAutoStatError`:
146
+
147
+ | Exception | Typical cause |
148
+ | --- | --- |
149
+ | `InvalidDataError` | Unsupported DataFrame shape, names, values or outcome type |
150
+ | `ColumnNotFoundError` | Requested column is absent |
151
+ | `InsufficientGroupsError` | Too few usable groups |
152
+ | `InsufficientDataError` | Too few usable observations or undefined test result |
153
+ | `InvalidTestError` | Unknown or incompatible test request |
154
+ | `ReportError` | A report file could not be created or written |
155
+
156
+ `analysis_warnings` records skipped descriptive calculations without aborting the full analysis.
157
+
158
+ For working code and output files, see [the feature showcase](examples/README.md).
@@ -0,0 +1,81 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented in this file.
4
+
5
+ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/).
6
+
7
+ ## [Unreleased]
8
+
9
+ ### Added
10
+
11
+ - Initial packaging as `pyautostat` (src layout, pyproject.toml, tests, CI).
12
+ - `PyAutoStatError` and friendly, specific error messages for invalid data,
13
+ unknown columns, unknown `test_type`, and too few groups in `hypothesis_tests()`.
14
+ - Effect sizes and confidence intervals in `hypothesis_tests()`: Cohen's d + CI
15
+ for t-tests, rank-biserial correlation for Mann-Whitney, eta-squared for
16
+ ANOVA, epsilon-squared for Kruskal-Wallis, each with a small/medium/large
17
+ interpretation.
18
+ - `pyautostat.detect_column_types()` and `pyautostat.suggest_column_roles()`:
19
+ heuristics for numeric-but-categorical columns, date-like text, and
20
+ name-based role suggestions (identifier/target/datetime/economic). Both
21
+ are included automatically in `analyze_all()`.
22
+ - `ReportGenerator.to_interactive_html()`: Plotly-based report with a
23
+ hoverable correlation heatmap and zoomable per-column histograms.
24
+ Requires the `report` extra (`pip install pyautostat[report]`).
25
+ - Phase 1: contact-format and missingness hints in `detect_column_types()`;
26
+ measurement and action hints in `suggest_column_roles()`.
27
+ - Detailed per-group normality and Levene assumption statuses, selection
28
+ reasons, and configurable deterministic percentile bootstrap intervals for
29
+ hypothesis effect sizes. Reports can include hypothesis results in JSON,
30
+ HTML, interactive HTML, and CSV.
31
+ - Categorical chi-square association with Cramér's V, expected-count checks,
32
+ and optional Cohen's h for a named success outcome in a 2x2 table; both
33
+ effect sizes support bootstrap intervals.
34
+ - Collapsible interactive report sections, table filtering and sorting, and
35
+ chart rendering when a section is opened.
36
+
37
+ ### Fixed
38
+
39
+ - `hypothesis_tests()` no longer treats a missing-value row as its own group.
40
+ - Validate DataFrame labels, scalar values, finite real numeric input, hypothesis
41
+ test compatibility, usable group sizes, and constant or undefined test data.
42
+ - Skip normality tests that lack the required sample size or variation; preserve
43
+ finite/undefined results explicitly and report reasons in `analysis_warnings`.
44
+ - Use pairwise observations for correlation p-values and represent undefined
45
+ coefficients, outlier metrics, and descriptive statistics as `None`.
46
+ - Keep extreme finite values from aborting histogram peak analysis; flag
47
+ numerical underflow and reject undefined effect sizes or confidence intervals.
48
+ - Escape untrusted HTML report text, emit standards-compliant JSON, protect
49
+ written CSV files from spreadsheet formulas, and wrap file errors in
50
+ `ReportError`. Removed global warning suppression.
51
+
52
+ ### Changed
53
+
54
+ - Renamed the package from `autostat` to `pyautostat` everywhere (import
55
+ path, PyPI distribution name, docs). `AutoStatError` is now `PyAutoStatError`.
56
+ - Dropped Python 3.9 support (EOL); the package now requires Python >=3.10.
57
+ - Three-or-more-group `auto` comparisons now choose ANOVA only when normality
58
+ and equal-variance screens pass; otherwise they choose Kruskal-Wallis.
59
+ - Rank-biserial correlation now follows the first-versus-second group direction;
60
+ negative sample epsilon-squared estimates are truncated at zero.
61
+
62
+ ### Chore
63
+
64
+ - Prepared the README for PyPI and removed generated report files from version
65
+ control; the example script recreates them on demand.
66
+ - Consolidated repository documentation into a focused README, API reference,
67
+ roadmap, changelog, contributor guide, and examples guide; removed redundant
68
+ and unverified marketing documents.
69
+ - Reworked the runnable example to cover all public analysis and export
70
+ features, optional Plotly output, and representative bad-data errors; added
71
+ an examples guide and CLI smoke tests.
72
+ - Added `ruff check`, `ruff format`, and `mypy` (clean on `src/`) with a
73
+ pre-commit config, and a GitHub Actions CI workflow that lints,
74
+ type-checks, runs the test suite on Python 3.10-3.13 across Linux and
75
+ Windows, and builds/validates the sdist and wheel with `twine check`.
76
+
77
+ ## [0.1.0] - 2026-09-21
78
+
79
+ - Statistical analysis (descriptive stats, normality, outliers, correlation, hypothesis testing).
80
+ - Automated insights engine.
81
+ - Report export to JSON, HTML, CSV, and dict.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Koushik Chandra Maji
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.