pyautostat 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pyautostat-0.1.0/.github/workflows/ci.yml +72 -0
- pyautostat-0.1.0/.github/workflows/publish.yml +92 -0
- pyautostat-0.1.0/.gitignore +16 -0
- pyautostat-0.1.0/.pre-commit-config.yaml +24 -0
- pyautostat-0.1.0/AGENTS.md +128 -0
- pyautostat-0.1.0/API_REFERENCE.md +158 -0
- pyautostat-0.1.0/CHANGELOG.md +81 -0
- pyautostat-0.1.0/LICENSE +21 -0
- pyautostat-0.1.0/PKG-INFO +146 -0
- pyautostat-0.1.0/README.md +108 -0
- pyautostat-0.1.0/ROADMAP.md +18 -0
- pyautostat-0.1.0/examples/README.md +58 -0
- pyautostat-0.1.0/examples/example_usage.py +373 -0
- pyautostat-0.1.0/pyproject.toml +79 -0
- pyautostat-0.1.0/src/pyautostat/__init__.py +33 -0
- pyautostat-0.1.0/src/pyautostat/analyzer.py +1039 -0
- pyautostat-0.1.0/src/pyautostat/categorical.py +170 -0
- pyautostat-0.1.0/src/pyautostat/detection.py +175 -0
- pyautostat-0.1.0/src/pyautostat/exceptions.py +29 -0
- pyautostat-0.1.0/src/pyautostat/insights.py +266 -0
- pyautostat-0.1.0/src/pyautostat/report.py +864 -0
- pyautostat-0.1.0/tests/conftest.py +90 -0
- pyautostat-0.1.0/tests/test_analyzer_correlation.py +21 -0
- pyautostat-0.1.0/tests/test_analyzer_descriptive.py +34 -0
- pyautostat-0.1.0/tests/test_analyzer_distributions.py +20 -0
- pyautostat-0.1.0/tests/test_analyzer_effect_size_bands.py +28 -0
- pyautostat-0.1.0/tests/test_analyzer_histograms.py +15 -0
- pyautostat-0.1.0/tests/test_analyzer_hypothesis.py +139 -0
- pyautostat-0.1.0/tests/test_analyzer_missing_and_quality.py +28 -0
- pyautostat-0.1.0/tests/test_analyzer_normality.py +23 -0
- pyautostat-0.1.0/tests/test_analyzer_outliers.py +20 -0
- pyautostat-0.1.0/tests/test_analyzer_overview.py +12 -0
- pyautostat-0.1.0/tests/test_categorical.py +96 -0
- pyautostat-0.1.0/tests/test_detection.py +86 -0
- pyautostat-0.1.0/tests/test_examples.py +88 -0
- pyautostat-0.1.0/tests/test_exceptions.py +56 -0
- pyautostat-0.1.0/tests/test_insights.py +66 -0
- pyautostat-0.1.0/tests/test_report.py +83 -0
- pyautostat-0.1.0/tests/test_report_interactive.py +80 -0
- pyautostat-0.1.0/tests/test_robustness.py +197 -0
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main, master]
|
|
6
|
+
pull_request:
|
|
7
|
+
branches: [main, master]
|
|
8
|
+
workflow_dispatch:
|
|
9
|
+
|
|
10
|
+
jobs:
|
|
11
|
+
lint:
|
|
12
|
+
name: Lint & type-check
|
|
13
|
+
runs-on: ubuntu-latest
|
|
14
|
+
steps:
|
|
15
|
+
- uses: actions/checkout@v4
|
|
16
|
+
- uses: actions/setup-python@v5
|
|
17
|
+
with:
|
|
18
|
+
python-version: "3.12"
|
|
19
|
+
cache: pip
|
|
20
|
+
cache-dependency-path: pyproject.toml
|
|
21
|
+
- name: Install package with dev extras
|
|
22
|
+
run: pip install -e ".[dev]"
|
|
23
|
+
- name: ruff check
|
|
24
|
+
run: ruff check src tests
|
|
25
|
+
- name: ruff format --check
|
|
26
|
+
run: ruff format --check src tests
|
|
27
|
+
- name: mypy
|
|
28
|
+
run: mypy src/pyautostat
|
|
29
|
+
|
|
30
|
+
test:
|
|
31
|
+
name: Test (Python ${{ matrix.python-version }}, ${{ matrix.os }})
|
|
32
|
+
needs: lint
|
|
33
|
+
runs-on: ${{ matrix.os }}
|
|
34
|
+
strategy:
|
|
35
|
+
fail-fast: false
|
|
36
|
+
matrix:
|
|
37
|
+
os: [ubuntu-latest, windows-latest]
|
|
38
|
+
python-version: ["3.10", "3.11", "3.12", "3.13"]
|
|
39
|
+
steps:
|
|
40
|
+
- uses: actions/checkout@v4
|
|
41
|
+
- uses: actions/setup-python@v5
|
|
42
|
+
with:
|
|
43
|
+
python-version: ${{ matrix.python-version }}
|
|
44
|
+
cache: pip
|
|
45
|
+
cache-dependency-path: pyproject.toml
|
|
46
|
+
- name: Install package with dev extras
|
|
47
|
+
run: pip install -e ".[dev]"
|
|
48
|
+
- name: pytest
|
|
49
|
+
run: pytest -q --cov=pyautostat --cov-report=term-missing --cov-fail-under=90
|
|
50
|
+
|
|
51
|
+
build:
|
|
52
|
+
name: Build distribution
|
|
53
|
+
needs: lint
|
|
54
|
+
runs-on: ubuntu-latest
|
|
55
|
+
steps:
|
|
56
|
+
- uses: actions/checkout@v4
|
|
57
|
+
- uses: actions/setup-python@v5
|
|
58
|
+
with:
|
|
59
|
+
python-version: "3.12"
|
|
60
|
+
cache: pip
|
|
61
|
+
cache-dependency-path: pyproject.toml
|
|
62
|
+
- name: Install build tools
|
|
63
|
+
run: pip install build twine
|
|
64
|
+
- name: Build sdist and wheel
|
|
65
|
+
run: python -m build
|
|
66
|
+
- name: Check distribution metadata
|
|
67
|
+
run: twine check dist/*
|
|
68
|
+
- name: Upload build artifacts
|
|
69
|
+
uses: actions/upload-artifact@v4
|
|
70
|
+
with:
|
|
71
|
+
name: dist
|
|
72
|
+
path: dist/
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
name: Publish PyAutoStat to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
release:
|
|
5
|
+
types: [published]
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
build:
|
|
9
|
+
runs-on: ubuntu-latest
|
|
10
|
+
permissions:
|
|
11
|
+
contents: read
|
|
12
|
+
|
|
13
|
+
steps:
|
|
14
|
+
- name: Checkout code
|
|
15
|
+
uses: actions/checkout@v4
|
|
16
|
+
with:
|
|
17
|
+
persist-credentials: false
|
|
18
|
+
|
|
19
|
+
- name: Set up Python
|
|
20
|
+
uses: actions/setup-python@v5
|
|
21
|
+
with:
|
|
22
|
+
python-version: "3.12"
|
|
23
|
+
|
|
24
|
+
- name: Install build tools
|
|
25
|
+
run: python -m pip install --upgrade build twine
|
|
26
|
+
|
|
27
|
+
- name: Verify release version
|
|
28
|
+
shell: bash
|
|
29
|
+
run: |
|
|
30
|
+
python - <<'PY'
|
|
31
|
+
import ast
|
|
32
|
+
import os
|
|
33
|
+
from pathlib import Path
|
|
34
|
+
|
|
35
|
+
tree = ast.parse(
|
|
36
|
+
Path("src/pyautostat/__init__.py").read_text(
|
|
37
|
+
encoding="utf-8"
|
|
38
|
+
)
|
|
39
|
+
)
|
|
40
|
+
version = next(
|
|
41
|
+
ast.literal_eval(node.value)
|
|
42
|
+
for node in tree.body
|
|
43
|
+
if isinstance(node, ast.Assign)
|
|
44
|
+
and any(
|
|
45
|
+
isinstance(target, ast.Name)
|
|
46
|
+
and target.id == "__version__"
|
|
47
|
+
for target in node.targets
|
|
48
|
+
)
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
expected = f"v{version}"
|
|
52
|
+
actual = os.environ["GITHUB_REF_NAME"]
|
|
53
|
+
|
|
54
|
+
if actual != expected:
|
|
55
|
+
raise SystemExit(
|
|
56
|
+
f"Tag {actual!r} does not match {expected!r}"
|
|
57
|
+
)
|
|
58
|
+
PY
|
|
59
|
+
|
|
60
|
+
- name: Build package
|
|
61
|
+
run: python -m build
|
|
62
|
+
|
|
63
|
+
- name: Check distributions
|
|
64
|
+
run: python -m twine check dist/*
|
|
65
|
+
|
|
66
|
+
- name: Upload distributions
|
|
67
|
+
uses: actions/upload-artifact@v4
|
|
68
|
+
with:
|
|
69
|
+
name: release-distributions
|
|
70
|
+
path: dist/
|
|
71
|
+
if-no-files-found: error
|
|
72
|
+
|
|
73
|
+
publish:
|
|
74
|
+
needs: build
|
|
75
|
+
runs-on: ubuntu-latest
|
|
76
|
+
|
|
77
|
+
environment:
|
|
78
|
+
name: pypi
|
|
79
|
+
url: https://pypi.org/project/pyautostat/
|
|
80
|
+
|
|
81
|
+
permissions:
|
|
82
|
+
id-token: write
|
|
83
|
+
|
|
84
|
+
steps:
|
|
85
|
+
- name: Download distributions
|
|
86
|
+
uses: actions/download-artifact@v4
|
|
87
|
+
with:
|
|
88
|
+
name: release-distributions
|
|
89
|
+
path: dist/
|
|
90
|
+
|
|
91
|
+
- name: Publish to PyPI
|
|
92
|
+
uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
repos:
|
|
2
|
+
- repo: https://github.com/pre-commit/pre-commit-hooks
|
|
3
|
+
rev: v5.0.0
|
|
4
|
+
hooks:
|
|
5
|
+
- id: trailing-whitespace
|
|
6
|
+
- id: end-of-file-fixer
|
|
7
|
+
- id: check-yaml
|
|
8
|
+
- id: check-toml
|
|
9
|
+
- id: check-added-large-files
|
|
10
|
+
- id: check-merge-conflict
|
|
11
|
+
|
|
12
|
+
- repo: https://github.com/astral-sh/ruff-pre-commit
|
|
13
|
+
rev: v0.16.8
|
|
14
|
+
hooks:
|
|
15
|
+
- id: ruff
|
|
16
|
+
args: [--fix]
|
|
17
|
+
- id: ruff-format
|
|
18
|
+
|
|
19
|
+
- repo: https://github.com/pre-commit/mirrors-mypy
|
|
20
|
+
rev: v2.3.1
|
|
21
|
+
hooks:
|
|
22
|
+
- id: mypy
|
|
23
|
+
files: ^src/
|
|
24
|
+
additional_dependencies: [pandas, numpy, scipy]
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
# PyAutoStat development guide
|
|
2
|
+
|
|
3
|
+
This file gives future contributors and coding agents the repository context and
|
|
4
|
+
working rules for PyAutoStat. Follow the user's current request first. Keep this
|
|
5
|
+
guide aligned with the code as the package evolves.
|
|
6
|
+
|
|
7
|
+
## Project context
|
|
8
|
+
|
|
9
|
+
- PyAutoStat is a Python package for automated statistical analysis of pandas
|
|
10
|
+
DataFrames, aimed at researchers and analysts. The package is currently an
|
|
11
|
+
alpha release (`0.1.0` in `src/pyautostat/__init__.py`).
|
|
12
|
+
- The package uses a `src` layout and is built with Hatchling. Python `>=3.10`
|
|
13
|
+
is supported. Runtime dependencies are pandas, NumPy, and SciPy; Plotly is an
|
|
14
|
+
optional dependency for interactive reports. `pyproject.toml` is the source
|
|
15
|
+
of truth for package metadata, dependencies, and tool configuration.
|
|
16
|
+
- Public imports are defined in `src/pyautostat/__init__.py`. Preserve them and
|
|
17
|
+
the shape of existing result dictionaries unless an intentional API change
|
|
18
|
+
has been requested and documented.
|
|
19
|
+
- `StatisticalAnalyzer` in `analyzer.py` copies the input DataFrame.
|
|
20
|
+
`analyze_all()` returns overview, descriptive, normality, outlier,
|
|
21
|
+
correlation, missing-data, data-quality, distribution, column-role,
|
|
22
|
+
column-type, histogram, and `analysis_warnings` sections. `hypothesis_tests()` is a separate
|
|
23
|
+
operation; its results are not added to `analyze_all()` automatically.
|
|
24
|
+
- `categorical_association()` is a separate independent-sample chi-square
|
|
25
|
+
comparison with Cramér's V. It requires expected cell counts of at least
|
|
26
|
+
five. Cohen's h is available for 2x2 tables with an explicit success value.
|
|
27
|
+
- `detection.py` provides advisory column-type and name-based role heuristics,
|
|
28
|
+
including contact-format and missingness hints.
|
|
29
|
+
These suggestions do not alter the analysis or exclude columns.
|
|
30
|
+
- `InsightEngine` in `insights.py` converts analysis results into severity-rated
|
|
31
|
+
findings and recommendations. `ReportGenerator` in `report.py` exports dict,
|
|
32
|
+
JSON, CSV, static HTML, and optional Plotly HTML.
|
|
33
|
+
- `examples/example_usage.py` is the runnable end-to-end feature showcase;
|
|
34
|
+
`examples/README.md` maps public features to its sections. `ROADMAP.md`
|
|
35
|
+
records product goals and distinguishes shipped features from future work. Tests live
|
|
36
|
+
under `tests/`; CI runs tests with coverage on Linux and Windows, and runs
|
|
37
|
+
lint, formatting, mypy, and a package build on Linux.
|
|
38
|
+
|
|
39
|
+
## Source of truth and scope
|
|
40
|
+
|
|
41
|
+
- Read the relevant source and tests before changing behavior. `ROADMAP.md`
|
|
42
|
+
contains proposed work, not a specification to apply wholesale. In particular,
|
|
43
|
+
do not assume that streaming/chunking, post-hoc tests, imputation, regulatory
|
|
44
|
+
compliance, or publication-ready formatting are implemented.
|
|
45
|
+
- Keep feature claims in `README.md`, `API_REFERENCE.md`, and `examples/README.md`
|
|
46
|
+
consistent with shipped behavior. Treat performance and market claims as
|
|
47
|
+
unverified unless measured or sourced.
|
|
48
|
+
- Make focused changes. Update public documentation and `CHANGELOG.md` when a
|
|
49
|
+
user-visible API, result schema, behavior, dependency, or supported Python
|
|
50
|
+
version changes.
|
|
51
|
+
|
|
52
|
+
## Statistical and data-handling standards
|
|
53
|
+
|
|
54
|
+
- Prefer well-defined methods from SciPy, NumPy, and pandas. State the test,
|
|
55
|
+
assumptions, sample sizes, effect-size definition, confidence level, and
|
|
56
|
+
direction of comparisons where applicable. Do not equate a p-value with the
|
|
57
|
+
probability that a hypothesis is true, or a nonsignificant result with proof
|
|
58
|
+
of no effect.
|
|
59
|
+
- Do not silently change an analysis based on heuristic column detection.
|
|
60
|
+
Keep user-selected group and value columns explicit. Validate their types,
|
|
61
|
+
distinct groups, and usable observations before running a test.
|
|
62
|
+
- Handle small samples, all-missing columns, constant values, non-finite
|
|
63
|
+
values, and zero denominators deliberately. Return a documented unavailable
|
|
64
|
+
result or raise a specific `PyAutoStatError` subclass with guidance; do not
|
|
65
|
+
emit misleading statistics or silently return an empty result.
|
|
66
|
+
- Check the minimum sample size and other preconditions for each SciPy test.
|
|
67
|
+
D'Agostino-Pearson's `normaltest` needs at least eight observations; the
|
|
68
|
+
current implementation skips smaller samples. `hypothesis_tests(auto)`
|
|
69
|
+
selects between standard ANOVA and Kruskal-Wallis for three or more groups
|
|
70
|
+
using per-group normality and Levene screens. These screens are heuristics,
|
|
71
|
+
and Kruskal-Wallis requires at least five usable observations per group.
|
|
72
|
+
- Preserve the meaning of missing values, group ordering, and paired versus
|
|
73
|
+
independent observations. Do not describe the current independent-group
|
|
74
|
+
tests as suitable for paired before/after data.
|
|
75
|
+
- Avoid global warning suppression in new code. Handle expected numerical
|
|
76
|
+
warnings locally and make invalid or undefined outputs clear to callers.
|
|
77
|
+
- Escape user-supplied labels and findings before inserting them into HTML.
|
|
78
|
+
Reports may contain untrusted DataFrame column names or values.
|
|
79
|
+
|
|
80
|
+
## Implementation conventions
|
|
81
|
+
|
|
82
|
+
- Keep runtime imports limited to declared dependencies. Import optional
|
|
83
|
+
dependencies inside the feature that needs them and provide a clear install
|
|
84
|
+
message when absent.
|
|
85
|
+
- Use type hints for new or substantially changed public code. Keep functions
|
|
86
|
+
small enough to explain their statistical purpose, and document public
|
|
87
|
+
parameters, return structures, units, and exceptions.
|
|
88
|
+
- Preserve input DataFrames unless a public API explicitly promises mutation.
|
|
89
|
+
Prefer deterministic behavior and local random generators with fixed seeds
|
|
90
|
+
in examples and tests.
|
|
91
|
+
- Keep exports portable across supported Python versions and operating
|
|
92
|
+
systems. Use `pathlib` for new path handling, UTF-8 for text files, and
|
|
93
|
+
avoid assuming output directories already exist without documenting it.
|
|
94
|
+
- Respect the existing Ruff configuration (`E`, `F`, `I`, `UP`, `B`, line
|
|
95
|
+
length 100) and mypy configuration in `pyproject.toml`. Do not add broad
|
|
96
|
+
lint/type ignores to work around a local issue.
|
|
97
|
+
|
|
98
|
+
## Verification
|
|
99
|
+
|
|
100
|
+
- Add or update focused tests when statistical behavior, result schemas, or
|
|
101
|
+
error handling changes. Compare numerical results with trusted SciPy or
|
|
102
|
+
pandas calculations, and test relevant boundary cases rather than only
|
|
103
|
+
asserting that a key exists. Do not add tests for prose-only or trivial
|
|
104
|
+
reversible changes.
|
|
105
|
+
- Run the checks relevant to the change. The CI commands are:
|
|
106
|
+
|
|
107
|
+
```text
|
|
108
|
+
python -m ruff check src tests
|
|
109
|
+
python -m ruff format --check src tests
|
|
110
|
+
python -m mypy src/pyautostat
|
|
111
|
+
python -m pytest -q --cov=pyautostat --cov-report=term-missing --cov-fail-under=90
|
|
112
|
+
python -m build
|
|
113
|
+
python -m twine check dist/*
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
- Install development dependencies with `python -m pip install -e ".[dev]"`
|
|
117
|
+
when the environment permits. Report any check that could not run and why;
|
|
118
|
+
do not claim it passed. CI targets Python 3.10 through 3.13 on Linux and
|
|
119
|
+
Windows. The mypy target is set to Python 3.12 because of NumPy stub syntax.
|
|
120
|
+
- Keep tests independent of network access. Plotly is optional at runtime;
|
|
121
|
+
the interactive HTML currently references Plotly JavaScript from a CDN.
|
|
122
|
+
|
|
123
|
+
## Before finishing a change
|
|
124
|
+
|
|
125
|
+
- Confirm the public behavior and examples agree with the implementation.
|
|
126
|
+
- Summarize what changed, why, which checks ran, and any remaining limitation.
|
|
127
|
+
- Do not treat this workspace as a Git checkout unless `.git` is present; a
|
|
128
|
+
source snapshot may have no repository metadata.
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
# PyAutoStat API reference
|
|
2
|
+
|
|
3
|
+
This page describes the public API in `pyautostat`. The [README](README.md) has a short start-to-finish example; the [examples guide](examples/README.md) runs every feature and shows its output.
|
|
4
|
+
|
|
5
|
+
## Imports
|
|
6
|
+
|
|
7
|
+
```python
|
|
8
|
+
from pyautostat import (
|
|
9
|
+
InsightEngine,
|
|
10
|
+
ReportGenerator,
|
|
11
|
+
StatisticalAnalyzer,
|
|
12
|
+
detect_column_types,
|
|
13
|
+
suggest_column_roles,
|
|
14
|
+
)
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
All documented exception classes are also exported from `pyautostat`.
|
|
18
|
+
|
|
19
|
+
## `StatisticalAnalyzer`
|
|
20
|
+
|
|
21
|
+
### Construction and full analysis
|
|
22
|
+
|
|
23
|
+
```python
|
|
24
|
+
analyzer = StatisticalAnalyzer(df)
|
|
25
|
+
analysis = analyzer.analyze_all()
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
`df` must be a nonempty pandas DataFrame with unique, nonempty string column names and scalar values. Numeric values must be finite and real; missing values are allowed. The analyzer copies the DataFrame and exposes `df`, `numeric_cols`, `categorical_cols`, and `all_results`.
|
|
29
|
+
|
|
30
|
+
`analyze_all()` returns a dictionary with these sections:
|
|
31
|
+
|
|
32
|
+
| Key | Contents |
|
|
33
|
+
| --- | --- |
|
|
34
|
+
| `overview` | Shape, row and column counts, memory use, column names and dtypes |
|
|
35
|
+
| `descriptive` | Per-numeric-column count, center, spread, quantiles, skewness and kurtosis |
|
|
36
|
+
| `normality` | Shapiro-Wilk, D'Agostino-Pearson and Anderson-Darling when their preconditions are met |
|
|
37
|
+
| `outliers` | IQR, Z-score and median absolute deviation summaries |
|
|
38
|
+
| `correlation` | Pearson, Spearman and Kendall matrices; pairwise Pearson p-values |
|
|
39
|
+
| `missing_data` | Missing counts and percentages by column and overall |
|
|
40
|
+
| `data_quality` | Completeness, per-column uniqueness and duplicate rows |
|
|
41
|
+
| `distributions` | Skewness and kurtosis descriptions, range and a histogram-based bimodality hint |
|
|
42
|
+
| `column_roles` | Advisory roles derived from column names |
|
|
43
|
+
| `column_types` | Advisory types and missingness hints |
|
|
44
|
+
| `histograms` | Precomputed bin edges and counts for numeric columns |
|
|
45
|
+
| `analysis_warnings` | Records with `code`, `section`, `column` and `message` for skipped or undefined calculations |
|
|
46
|
+
|
|
47
|
+
An unavailable numeric result is `None`. Some tests are skipped for all-missing, constant or short columns; inspect `analysis_warnings`. Correlation uses pairwise nonmissing observations. Only Pearson pairs include p-values.
|
|
48
|
+
|
|
49
|
+
### Independent group comparisons
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
result = analyzer.hypothesis_tests(
|
|
53
|
+
group_col="group",
|
|
54
|
+
value_col="outcome",
|
|
55
|
+
test_type="auto",
|
|
56
|
+
confidence_level=0.95,
|
|
57
|
+
bootstrap_samples=499,
|
|
58
|
+
random_state=0,
|
|
59
|
+
)
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
`test_type` can be `auto`, `ttest`, `mannwhitney`, `anova` or `kruskal`. Automatic selection uses per-group normality and Levene variance screens. It chooses a t-test or Mann-Whitney U for two groups, and one-way ANOVA or Kruskal-Wallis for three or more groups. Explicitly selected tests are retained, with assumption warnings where appropriate.
|
|
63
|
+
|
|
64
|
+
The result contains `test`, `statistic`, `p_value`, `groups`, `assumptions` and `effect_size`. The assumptions include per-group normality, Levene results, selection reason and warnings. The effect-size record contains its name, value, interpretation and a percentile bootstrap confidence interval. T-tests also return an analytical `confidence_interval` for the first minus second group's mean; a t-test result includes `equal_variance`.
|
|
65
|
+
|
|
66
|
+
Rows missing a group or outcome are excluded. Every group needs at least two usable numeric values. Kruskal-Wallis requires at least five per group for its approximation. Use `bootstrap_samples=0` to omit effect-size intervals, or an integer of at least 100 for resampling. `random_state` controls a local random generator.
|
|
67
|
+
|
|
68
|
+
These tests assume independent observations. The automatic screens cannot verify study design, and bootstrap intervals do not adjust for multiple testing.
|
|
69
|
+
|
|
70
|
+
### Categorical association
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
association = analyzer.categorical_association(
|
|
74
|
+
group_col="group",
|
|
75
|
+
outcome_col="response",
|
|
76
|
+
success_value="yes",
|
|
77
|
+
confidence_level=0.95,
|
|
78
|
+
bootstrap_samples=499,
|
|
79
|
+
random_state=0,
|
|
80
|
+
)
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
This method runs a Pearson chi-square test of independence without continuity correction. It excludes rows missing either selected value, preserves first-appearance category order, and requires every expected cell count to be at least five.
|
|
84
|
+
|
|
85
|
+
The result includes `test`, `statistic`, `p_value`, `degrees_of_freedom`, `groups`, `outcomes`, `observed_counts`, `expected_counts`, `sample_size`, `assumptions` and Cramér's V in `effect_size`. For a two-by-two table with an explicit `success_value`, it also includes `cohens_h`. Positive Cohen's h means the first group has the higher success proportion. Set `bootstrap_samples=0` to omit bootstrap intervals.
|
|
86
|
+
|
|
87
|
+
## Column detection
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
types = detect_column_types(df)
|
|
91
|
+
roles = suggest_column_roles(df)
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
`detect_column_types()` returns one record per column with `detected_type`, `notes`, `missing_count` and `missing_percentage`. It recognizes numeric categories, continuous numbers, booleans, datetimes, date-like text and common email, URL and phone formats. Contact and date checks sample up to 20 nonmissing values.
|
|
95
|
+
|
|
96
|
+
`suggest_column_roles()` returns `role`, `reason` and `suggested_action` for each column. Roles include identifier, target, datetime, economic, measurement and unknown. Both helpers are advisory: they do not transform data or select test columns.
|
|
97
|
+
|
|
98
|
+
## `InsightEngine`
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
engine = InsightEngine(analysis)
|
|
102
|
+
findings = engine.generate_insights()
|
|
103
|
+
summary = engine.get_summary()
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
`generate_insights()` returns a list of findings with category, severity, finding and recommendations. `get_summary()` generates findings if needed and returns `total_insights`, `high_severity`, `medium_severity`, `low_severity` and `insights`.
|
|
107
|
+
|
|
108
|
+
## `ReportGenerator`
|
|
109
|
+
|
|
110
|
+
```python
|
|
111
|
+
report = ReportGenerator(
|
|
112
|
+
analysis_results=analysis,
|
|
113
|
+
insights=summary,
|
|
114
|
+
hypothesis_results=[result, association],
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
payload = report.to_dict()
|
|
118
|
+
json_text = report.to_json()
|
|
119
|
+
csv_tables = report.to_csv()
|
|
120
|
+
html_text = report.to_html()
|
|
121
|
+
interactive_html = report.to_interactive_html() # requires the "report" extra
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
`insights` and `hypothesis_results` are optional. A comparison can be one result dictionary or a list of results. Pass a path to write a format instead of returning its content:
|
|
125
|
+
|
|
126
|
+
```python
|
|
127
|
+
report.to_json("analysis.json", pretty=True)
|
|
128
|
+
report.to_csv("csv_reports")
|
|
129
|
+
report.to_html("analysis.html", title="Analysis")
|
|
130
|
+
report.to_interactive_html("interactive.html", title="Interactive analysis")
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
| Method | Without path | With path |
|
|
134
|
+
| --- | --- | --- |
|
|
135
|
+
| `to_dict()` | Dictionary with timestamp, analysis, insights and optionally hypothesis tests | N/A |
|
|
136
|
+
| `to_json(filepath=None, pretty=True)` | JSON string | Writes JSON and returns a status message |
|
|
137
|
+
| `to_csv(output_dir=None)` | Dictionary of DataFrames | Writes CSV files and returns the tables |
|
|
138
|
+
| `to_html(filepath=None, title=...)` | Static HTML string | Writes HTML and returns a status message |
|
|
139
|
+
| `to_interactive_html(filepath=None, title=...)` | Plotly HTML string | Writes HTML and returns a status message |
|
|
140
|
+
|
|
141
|
+
CSV tables include descriptive statistics, outliers, missing data and insights. When comparisons are supplied, they also include hypothesis tests. Written CSV protects formula-like text for spreadsheet use; in-memory DataFrames preserve the original values. JSON represents unavailable or non-finite numbers as `null`. Report HTML escapes data-derived text. Interactive HTML requires Plotly and loads Plotly JavaScript from a CDN when opened.
|
|
142
|
+
|
|
143
|
+
## Errors
|
|
144
|
+
|
|
145
|
+
All package-specific errors derive from `PyAutoStatError`:
|
|
146
|
+
|
|
147
|
+
| Exception | Typical cause |
|
|
148
|
+
| --- | --- |
|
|
149
|
+
| `InvalidDataError` | Unsupported DataFrame shape, names, values or outcome type |
|
|
150
|
+
| `ColumnNotFoundError` | Requested column is absent |
|
|
151
|
+
| `InsufficientGroupsError` | Too few usable groups |
|
|
152
|
+
| `InsufficientDataError` | Too few usable observations or undefined test result |
|
|
153
|
+
| `InvalidTestError` | Unknown or incompatible test request |
|
|
154
|
+
| `ReportError` | A report file could not be created or written |
|
|
155
|
+
|
|
156
|
+
`analysis_warnings` records skipped descriptive calculations without aborting the full analysis.
|
|
157
|
+
|
|
158
|
+
For working code and output files, see [the feature showcase](examples/README.md).
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented in this file.
|
|
4
|
+
|
|
5
|
+
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/).
|
|
6
|
+
|
|
7
|
+
## [Unreleased]
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- Initial packaging as `pyautostat` (src layout, pyproject.toml, tests, CI).
|
|
12
|
+
- `PyAutoStatError` and friendly, specific error messages for invalid data,
|
|
13
|
+
unknown columns, unknown `test_type`, and too few groups in `hypothesis_tests()`.
|
|
14
|
+
- Effect sizes and confidence intervals in `hypothesis_tests()`: Cohen's d + CI
|
|
15
|
+
for t-tests, rank-biserial correlation for Mann-Whitney, eta-squared for
|
|
16
|
+
ANOVA, epsilon-squared for Kruskal-Wallis, each with a small/medium/large
|
|
17
|
+
interpretation.
|
|
18
|
+
- `pyautostat.detect_column_types()` and `pyautostat.suggest_column_roles()`:
|
|
19
|
+
heuristics for numeric-but-categorical columns, date-like text, and
|
|
20
|
+
name-based role suggestions (identifier/target/datetime/economic). Both
|
|
21
|
+
are included automatically in `analyze_all()`.
|
|
22
|
+
- `ReportGenerator.to_interactive_html()`: Plotly-based report with a
|
|
23
|
+
hoverable correlation heatmap and zoomable per-column histograms.
|
|
24
|
+
Requires the `report` extra (`pip install pyautostat[report]`).
|
|
25
|
+
- Phase 1: contact-format and missingness hints in `detect_column_types()`;
|
|
26
|
+
measurement and action hints in `suggest_column_roles()`.
|
|
27
|
+
- Detailed per-group normality and Levene assumption statuses, selection
|
|
28
|
+
reasons, and configurable deterministic percentile bootstrap intervals for
|
|
29
|
+
hypothesis effect sizes. Reports can include hypothesis results in JSON,
|
|
30
|
+
HTML, interactive HTML, and CSV.
|
|
31
|
+
- Categorical chi-square association with Cramér's V, expected-count checks,
|
|
32
|
+
and optional Cohen's h for a named success outcome in a 2x2 table; both
|
|
33
|
+
effect sizes support bootstrap intervals.
|
|
34
|
+
- Collapsible interactive report sections, table filtering and sorting, and
|
|
35
|
+
chart rendering when a section is opened.
|
|
36
|
+
|
|
37
|
+
### Fixed
|
|
38
|
+
|
|
39
|
+
- `hypothesis_tests()` no longer treats a missing-value row as its own group.
|
|
40
|
+
- Validate DataFrame labels, scalar values, finite real numeric input, hypothesis
|
|
41
|
+
test compatibility, usable group sizes, and constant or undefined test data.
|
|
42
|
+
- Skip normality tests that lack the required sample size or variation; preserve
|
|
43
|
+
finite/undefined results explicitly and report reasons in `analysis_warnings`.
|
|
44
|
+
- Use pairwise observations for correlation p-values and represent undefined
|
|
45
|
+
coefficients, outlier metrics, and descriptive statistics as `None`.
|
|
46
|
+
- Keep extreme finite values from aborting histogram peak analysis; flag
|
|
47
|
+
numerical underflow and reject undefined effect sizes or confidence intervals.
|
|
48
|
+
- Escape untrusted HTML report text, emit standards-compliant JSON, protect
|
|
49
|
+
written CSV files from spreadsheet formulas, and wrap file errors in
|
|
50
|
+
`ReportError`. Removed global warning suppression.
|
|
51
|
+
|
|
52
|
+
### Changed
|
|
53
|
+
|
|
54
|
+
- Renamed the package from `autostat` to `pyautostat` everywhere (import
|
|
55
|
+
path, PyPI distribution name, docs). `AutoStatError` is now `PyAutoStatError`.
|
|
56
|
+
- Dropped Python 3.9 support (EOL); the package now requires Python >=3.10.
|
|
57
|
+
- Three-or-more-group `auto` comparisons now choose ANOVA only when normality
|
|
58
|
+
and equal-variance screens pass; otherwise they choose Kruskal-Wallis.
|
|
59
|
+
- Rank-biserial correlation now follows the first-versus-second group direction;
|
|
60
|
+
negative sample epsilon-squared estimates are truncated at zero.
|
|
61
|
+
|
|
62
|
+
### Chore
|
|
63
|
+
|
|
64
|
+
- Prepared the README for PyPI and removed generated report files from version
|
|
65
|
+
control; the example script recreates them on demand.
|
|
66
|
+
- Consolidated repository documentation into a focused README, API reference,
|
|
67
|
+
roadmap, changelog, contributor guide, and examples guide; removed redundant
|
|
68
|
+
and unverified marketing documents.
|
|
69
|
+
- Reworked the runnable example to cover all public analysis and export
|
|
70
|
+
features, optional Plotly output, and representative bad-data errors; added
|
|
71
|
+
an examples guide and CLI smoke tests.
|
|
72
|
+
- Added `ruff check`, `ruff format`, and `mypy` (clean on `src/`) with a
|
|
73
|
+
pre-commit config, and a GitHub Actions CI workflow that lints,
|
|
74
|
+
type-checks, runs the test suite on Python 3.10-3.13 across Linux and
|
|
75
|
+
Windows, and builds/validates the sdist and wheel with `twine check`.
|
|
76
|
+
|
|
77
|
+
## [0.1.0] - 2026-09-21
|
|
78
|
+
|
|
79
|
+
- Statistical analysis (descriptive stats, normality, outliers, correlation, hypothesis testing).
|
|
80
|
+
- Automated insights engine.
|
|
81
|
+
- Report export to JSON, HTML, CSV, and dict.
|
pyautostat-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Koushik Chandra Maji
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|