signapy 0.1.0a1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- signapy-0.1.0a1/.github/workflows/ci.yml +71 -0
- signapy-0.1.0a1/.github/workflows/publish.yml +85 -0
- signapy-0.1.0a1/.gitignore +29 -0
- signapy-0.1.0a1/CHANGELOG.md +10 -0
- signapy-0.1.0a1/LICENSE +21 -0
- signapy-0.1.0a1/PKG-INFO +250 -0
- signapy-0.1.0a1/README.md +217 -0
- signapy-0.1.0a1/docs/architecture.md +688 -0
- signapy-0.1.0a1/docs/metrics.md +605 -0
- signapy-0.1.0a1/examples/binary_categorical_discovery.py +82 -0
- signapy-0.1.0a1/examples/mixed_categorical_continuous_discovery.py +119 -0
- signapy-0.1.0a1/examples/ordinal_binary_discovery.py +82 -0
- signapy-0.1.0a1/pyproject.toml +77 -0
- signapy-0.1.0a1/src/signapy/__init__.py +13 -0
- signapy-0.1.0a1/src/signapy/discovery/__init__.py +345 -0
- signapy-0.1.0a1/src/signapy/discovery/feature.py +311 -0
- signapy-0.1.0a1/src/signapy/discovery/value.py +61 -0
- signapy-0.1.0a1/src/signapy/metrics/__init__.py +24 -0
- signapy-0.1.0a1/src/signapy/metrics/association.py +603 -0
- signapy-0.1.0a1/src/signapy/metrics/lift.py +158 -0
- signapy-0.1.0a1/src/signapy/metrics/significance.py +106 -0
- signapy-0.1.0a1/src/signapy/profiling/__init__.py +9 -0
- signapy-0.1.0a1/src/signapy/profiling/types.py +39 -0
- signapy-0.1.0a1/src/signapy/py.typed +0 -0
- signapy-0.1.0a1/src/signapy/results/__init__.py +9 -0
- signapy-0.1.0a1/src/signapy/results/models.py +148 -0
- signapy-0.1.0a1/tests/test_association.py +126 -0
- signapy-0.1.0a1/tests/test_discovery.py +296 -0
- signapy-0.1.0a1/tests/test_lift.py +161 -0
- signapy-0.1.0a1/tests/test_mixed_discovery.py +428 -0
- signapy-0.1.0a1/tests/test_ordinal_discovery.py +309 -0
- signapy-0.1.0a1/tests/test_package.py +14 -0
- signapy-0.1.0a1/tests/test_point_biserial.py +322 -0
- signapy-0.1.0a1/tests/test_results.py +104 -0
- signapy-0.1.0a1/tests/test_significance.py +109 -0
- signapy-0.1.0a1/tests/test_spearman.py +312 -0
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
concurrency:
|
|
12
|
+
group: ${{ github.workflow }}-${{ github.ref }}
|
|
13
|
+
cancel-in-progress: true
|
|
14
|
+
|
|
15
|
+
jobs:
|
|
16
|
+
lint:
|
|
17
|
+
runs-on: ubuntu-latest
|
|
18
|
+
steps:
|
|
19
|
+
- uses: actions/checkout@v4
|
|
20
|
+
- uses: actions/setup-python@v5
|
|
21
|
+
with:
|
|
22
|
+
python-version: "3.12"
|
|
23
|
+
- name: Install package with dev dependencies
|
|
24
|
+
run: python -m pip install --upgrade pip && python -m pip install -e ".[dev]"
|
|
25
|
+
- name: Lint
|
|
26
|
+
run: ruff check .
|
|
27
|
+
- name: Check formatting
|
|
28
|
+
run: ruff format --check .
|
|
29
|
+
|
|
30
|
+
test:
|
|
31
|
+
runs-on: ubuntu-latest
|
|
32
|
+
strategy:
|
|
33
|
+
fail-fast: false
|
|
34
|
+
matrix:
|
|
35
|
+
python-version: ["3.10", "3.11", "3.12", "3.13", "3.14"]
|
|
36
|
+
steps:
|
|
37
|
+
- uses: actions/checkout@v4
|
|
38
|
+
- uses: actions/setup-python@v5
|
|
39
|
+
with:
|
|
40
|
+
python-version: ${{ matrix.python-version }}
|
|
41
|
+
- name: Install package with dev dependencies
|
|
42
|
+
run: python -m pip install --upgrade pip && python -m pip install -e ".[dev]"
|
|
43
|
+
- name: Verify import
|
|
44
|
+
run: python -c "import signapy; print(signapy.__version__)"
|
|
45
|
+
- name: Run tests
|
|
46
|
+
run: pytest
|
|
47
|
+
|
|
48
|
+
package:
|
|
49
|
+
runs-on: ubuntu-latest
|
|
50
|
+
steps:
|
|
51
|
+
- uses: actions/checkout@v4
|
|
52
|
+
- uses: actions/setup-python@v5
|
|
53
|
+
with:
|
|
54
|
+
python-version: "3.12"
|
|
55
|
+
- name: Build and validate distributions
|
|
56
|
+
run: |
|
|
57
|
+
python -m pip install --upgrade pip build twine
|
|
58
|
+
python -m build
|
|
59
|
+
python -m twine check dist/*
|
|
60
|
+
- name: Smoke-test the wheel in a clean environment
|
|
61
|
+
run: |
|
|
62
|
+
python -m venv "$RUNNER_TEMP/wheel-test"
|
|
63
|
+
"$RUNNER_TEMP/wheel-test/bin/python" -m pip install dist/*.whl
|
|
64
|
+
cd "$RUNNER_TEMP"
|
|
65
|
+
"$RUNNER_TEMP/wheel-test/bin/python" -c "import signapy; print(signapy.__version__); assert callable(signapy.discover)"
|
|
66
|
+
- name: Smoke-test the source distribution in a clean environment
|
|
67
|
+
run: |
|
|
68
|
+
python -m venv "$RUNNER_TEMP/sdist-test"
|
|
69
|
+
cd "$RUNNER_TEMP"
|
|
70
|
+
"$RUNNER_TEMP/sdist-test/bin/python" -m pip install "$GITHUB_WORKSPACE"/dist/*.tar.gz
|
|
71
|
+
"$RUNNER_TEMP/sdist-test/bin/python" -c "import signapy; print(signapy.__version__); assert callable(signapy.discover)"
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
name: Publish to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
release:
|
|
5
|
+
types: [published]
|
|
6
|
+
workflow_dispatch:
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
jobs:
|
|
12
|
+
build:
|
|
13
|
+
runs-on: ubuntu-latest
|
|
14
|
+
|
|
15
|
+
steps:
|
|
16
|
+
- name: Check out repository
|
|
17
|
+
uses: actions/checkout@v6
|
|
18
|
+
with:
|
|
19
|
+
persist-credentials: false
|
|
20
|
+
|
|
21
|
+
- name: Set up Python
|
|
22
|
+
uses: actions/setup-python@v6
|
|
23
|
+
with:
|
|
24
|
+
python-version: "3.12"
|
|
25
|
+
|
|
26
|
+
- name: Install build tools
|
|
27
|
+
run: python -m pip install build twine
|
|
28
|
+
|
|
29
|
+
- name: Build distributions
|
|
30
|
+
run: python -m build
|
|
31
|
+
|
|
32
|
+
- name: Validate distributions
|
|
33
|
+
run: python -m twine check dist/*
|
|
34
|
+
|
|
35
|
+
- name: Upload distributions
|
|
36
|
+
uses: actions/upload-artifact@v5
|
|
37
|
+
with:
|
|
38
|
+
name: python-package-distributions
|
|
39
|
+
path: dist/
|
|
40
|
+
|
|
41
|
+
publish:
|
|
42
|
+
if: github.event_name == 'release'
|
|
43
|
+
needs: build
|
|
44
|
+
runs-on: ubuntu-latest
|
|
45
|
+
|
|
46
|
+
environment:
|
|
47
|
+
name: pypi
|
|
48
|
+
url: https://pypi.org/p/signapy
|
|
49
|
+
|
|
50
|
+
permissions:
|
|
51
|
+
id-token: write
|
|
52
|
+
|
|
53
|
+
steps:
|
|
54
|
+
- name: Download distributions
|
|
55
|
+
uses: actions/download-artifact@v6
|
|
56
|
+
with:
|
|
57
|
+
name: python-package-distributions
|
|
58
|
+
path: dist/
|
|
59
|
+
|
|
60
|
+
- name: Publish distributions
|
|
61
|
+
uses: pypa/gh-action-pypi-publish@release/v1
|
|
62
|
+
|
|
63
|
+
publish-test:
|
|
64
|
+
if: github.event_name == 'workflow_dispatch'
|
|
65
|
+
needs: build
|
|
66
|
+
runs-on: ubuntu-latest
|
|
67
|
+
|
|
68
|
+
environment:
|
|
69
|
+
name: testpypi
|
|
70
|
+
url: https://test.pypi.org/p/signapy
|
|
71
|
+
|
|
72
|
+
permissions:
|
|
73
|
+
id-token: write
|
|
74
|
+
|
|
75
|
+
steps:
|
|
76
|
+
- name: Download distributions
|
|
77
|
+
uses: actions/download-artifact@v6
|
|
78
|
+
with:
|
|
79
|
+
name: python-package-distributions
|
|
80
|
+
path: dist/
|
|
81
|
+
|
|
82
|
+
- name: Publish distributions
|
|
83
|
+
uses: pypa/gh-action-pypi-publish@release/v1
|
|
84
|
+
with:
|
|
85
|
+
repository-url: https://test.pypi.org/legacy/
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# Byte-compiled / cache
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
.pytest_cache/
|
|
5
|
+
.ruff_cache/
|
|
6
|
+
.mypy_cache/
|
|
7
|
+
|
|
8
|
+
# Build artifacts
|
|
9
|
+
build/
|
|
10
|
+
dist/
|
|
11
|
+
*.egg-info/
|
|
12
|
+
.eggs/
|
|
13
|
+
|
|
14
|
+
# Virtual environments
|
|
15
|
+
.venv/
|
|
16
|
+
venv/
|
|
17
|
+
env/
|
|
18
|
+
|
|
19
|
+
# Coverage
|
|
20
|
+
.coverage
|
|
21
|
+
.coverage.*
|
|
22
|
+
htmlcov/
|
|
23
|
+
coverage.xml
|
|
24
|
+
|
|
25
|
+
# Notebooks / editors / OS
|
|
26
|
+
.ipynb_checkpoints/
|
|
27
|
+
.idea/
|
|
28
|
+
.vscode/
|
|
29
|
+
.DS_Store
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.1.0a1 — first public alpha
|
|
4
|
+
|
|
5
|
+
Initial release. The API is experimental and may change.
|
|
6
|
+
|
|
7
|
+
- `signapy.discover` for a binary target against categorical, continuous, and
|
|
8
|
+
ordinal features.
|
|
9
|
+
- Evidence to find, quantify, localize, and validate predictive signal; see
|
|
10
|
+
`docs/metrics.md` for interpretation and `docs/architecture.md` for design.
|
signapy-0.1.0a1/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Carlos Meisel
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
signapy-0.1.0a1/PKG-INFO
ADDED
|
@@ -0,0 +1,250 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: signapy
|
|
3
|
+
Version: 0.1.0a1
|
|
4
|
+
Summary: Mixed categorical, continuous, and ordinal feature discovery for labeled datasets.
|
|
5
|
+
Project-URL: Homepage, https://github.com/Carmeisel101/signapy
|
|
6
|
+
Project-URL: Repository, https://github.com/Carmeisel101/signapy
|
|
7
|
+
Project-URL: Issues, https://github.com/Carmeisel101/signapy/issues
|
|
8
|
+
Author: Carlos Meisel
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: data science,exploratory data analysis,feature discovery,statistics
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
23
|
+
Classifier: Typing :: Typed
|
|
24
|
+
Requires-Python: >=3.10
|
|
25
|
+
Requires-Dist: numpy>=1.23
|
|
26
|
+
Requires-Dist: pandas>=2.0
|
|
27
|
+
Requires-Dist: scipy>=1.10
|
|
28
|
+
Provides-Extra: dev
|
|
29
|
+
Requires-Dist: packaging>=23.0; extra == 'dev'
|
|
30
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
31
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
32
|
+
Description-Content-Type: text/markdown
|
|
33
|
+
|
|
34
|
+
# SignaPy
|
|
35
|
+
|
|
36
|
+
**SignaPy is an exploratory feature-discovery library for labeled datasets.**
|
|
37
|
+
|
|
38
|
+
Long-term mission: *find, quantify, localize, and validate predictive signal.*
|
|
39
|
+
|
|
40
|
+
> **Status: alpha (`0.1.0a1`).** `signapy.discover()` works against
|
|
41
|
+
> **binary targets** with **categorical, continuous, and ordinal features**
|
|
42
|
+
> today. Everything under [Planned next](#planned-next) — other
|
|
43
|
+
> feature/target types, interactions, stability, and more — is not
|
|
44
|
+
> implemented yet.
|
|
45
|
+
|
|
46
|
+
## The problem
|
|
47
|
+
|
|
48
|
+
Before modeling, data scientists want to know which features carry signal about
|
|
49
|
+
the target, and where within those features that signal lives. Today this means
|
|
50
|
+
picking the right test for each feature/target combination (Cramér's V? Phi?
|
|
51
|
+
Spearman? point-biserial?), running it by hand, and stitching the output together.
|
|
52
|
+
|
|
53
|
+
SignaPy aims to make that a single, task-aware workflow. It uses the feature
|
|
54
|
+
types you declare (or the categorical-only default), chooses an appropriate
|
|
55
|
+
method, and returns structured evidence. You shouldn't need to know ahead of
|
|
56
|
+
time which statistical test applies.
|
|
57
|
+
|
|
58
|
+
SignaPy discovers evidence of predictive signal.
|
|
59
|
+
It does not decide what your model should use.
|
|
60
|
+
|
|
61
|
+
It is **not** an AutoML framework, a feature-selection oracle, or a modeling
|
|
62
|
+
library, and it does not drop features based on p-values.
|
|
63
|
+
|
|
64
|
+
## Discovery hierarchy
|
|
65
|
+
|
|
66
|
+
| Level | Question |
|
|
67
|
+
| --- | --- |
|
|
68
|
+
| Profiling | What data am I working with? |
|
|
69
|
+
| Feature-level discovery | Does this feature contain signal about the target? |
|
|
70
|
+
| Value-level discovery | Where within this feature does the signal live? |
|
|
71
|
+
| Interaction discovery *(future)* | Does signal emerge from combinations of features or values? |
|
|
72
|
+
| Stability analysis *(future)* | Does the signal hold across samples, folds, time, or subgroups? |
|
|
73
|
+
|
|
74
|
+
For example, feature-level discovery might find that `acquisition_channel` is
|
|
75
|
+
associated with conversion. Value-level discovery then shows *where*:
|
|
76
|
+
|
|
77
|
+
```text
|
|
78
|
+
referral -> lift 1.91
|
|
79
|
+
organic -> lift 1.28
|
|
80
|
+
paid -> lift 0.73
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
## Installation (development)
|
|
84
|
+
|
|
85
|
+
SignaPy supports Python 3.10 to 3.14. Clone the repository and install its
|
|
86
|
+
development dependencies:
|
|
87
|
+
|
|
88
|
+
```bash
|
|
89
|
+
git clone <repo-url> signapy
|
|
90
|
+
cd signapy
|
|
91
|
+
python -m venv .venv
|
|
92
|
+
source .venv/bin/activate
|
|
93
|
+
pip install -e ".[dev]"
|
|
94
|
+
pytest
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Runtime dependencies are `numpy`, `pandas`, and `scipy`. SignaPy doesn't
|
|
98
|
+
depend on any ML framework.
|
|
99
|
+
|
|
100
|
+
## Usage
|
|
101
|
+
|
|
102
|
+
Categorical-only (every non-target column analyzed automatically):
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
import signapy
|
|
106
|
+
|
|
107
|
+
report = signapy.discover(df, target="converted", positive_class=True)
|
|
108
|
+
|
|
109
|
+
report.features # feature-level results
|
|
110
|
+
report.feature("acquisition_channel") # one feature's evidence
|
|
111
|
+
report.feature("acquisition_channel").values # value-level results
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
Mixed categorical and continuous — pass `feature_types` to select exactly
|
|
115
|
+
which columns to analyze and declare each one's semantic type:
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
from signapy.profiling import FeatureType
|
|
119
|
+
|
|
120
|
+
report = signapy.discover(
|
|
121
|
+
df,
|
|
122
|
+
target="outcome",
|
|
123
|
+
positive_class=True,
|
|
124
|
+
feature_types={
|
|
125
|
+
"category_context": FeatureType.CATEGORICAL,
|
|
126
|
+
"continuous_measurement": FeatureType.CONTINUOUS,
|
|
127
|
+
},
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
report.feature("continuous_measurement").effect_size # point-biserial coefficient
|
|
131
|
+
report.feature("continuous_measurement").p_value # its p-value
|
|
132
|
+
report.feature("continuous_measurement").values # always None (see below)
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
An **ordinal** feature needs an explicit order — SignaPy never infers one,
|
|
136
|
+
not from the values and never alphabetically — declared via an ordered
|
|
137
|
+
pandas `Categorical`:
|
|
138
|
+
|
|
139
|
+
```python
|
|
140
|
+
df["severity"] = pd.Categorical(
|
|
141
|
+
df["severity"],
|
|
142
|
+
categories=["low", "medium", "high"],
|
|
143
|
+
ordered=True,
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
report = signapy.discover(
|
|
147
|
+
df,
|
|
148
|
+
target="converted",
|
|
149
|
+
positive_class=True,
|
|
150
|
+
feature_types={"severity": FeatureType.ORDINAL},
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
report.feature("severity").effect_size # Spearman's rho
|
|
154
|
+
report.feature(
|
|
155
|
+
"severity"
|
|
156
|
+
).values # support/rate/baseline/lift, in low->medium->high order
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
Pandas converts any value omitted from `categories=[...]` to a missing value;
|
|
160
|
+
under SignaPy's missing-data policy, that row is then excluded from this
|
|
161
|
+
feature's analysis. Make sure the declaration includes every intended level.
|
|
162
|
+
|
|
163
|
+
`discover()` supports **categorical, continuous, and ordinal features
|
|
164
|
+
against a binary target** (see [v0.1 scope](#v01-scope) below):
|
|
165
|
+
|
|
166
|
+
- Without `feature_types` (the default), every non-target column of `df`
|
|
167
|
+
with an `object`, pandas `string`, or pandas `category` dtype is analyzed
|
|
168
|
+
as categorical; other dtypes raise a clear error.
|
|
169
|
+
- With `feature_types`, only the named columns are analyzed, each as its
|
|
170
|
+
declared type (`FeatureType.CATEGORICAL`, `FeatureType.CONTINUOUS`, or
|
|
171
|
+
`FeatureType.ORDINAL` — the only three supported so far). **Numeric
|
|
172
|
+
columns are never silently treated as continuous, and categorical columns
|
|
173
|
+
are never silently treated as ordinal** — an integer column might be a
|
|
174
|
+
measured quantity, an ordinal level, or a category identifier, and
|
|
175
|
+
SignaPy doesn't guess which; declaring it via `feature_types` is
|
|
176
|
+
required, and an ordinal declaration additionally requires the column to
|
|
177
|
+
already be an ordered `Categorical`.
|
|
178
|
+
- Categorical features get Cramér's V, a chi-square test, and per-category
|
|
179
|
+
`values` (support, target rate, baseline rate, lift), in order of first
|
|
180
|
+
appearance. Continuous features get a point-biserial correlation, its
|
|
181
|
+
p-value, and group counts/means in `details` — but **`values` is always
|
|
182
|
+
`None`** for continuous features: localizing *where within a continuous
|
|
183
|
+
range* the signal lives needs binning, which SignaPy doesn't do
|
|
184
|
+
automatically (see [`docs/architecture.md`](docs/architecture.md)).
|
|
185
|
+
Ordinal features get a Spearman rank correlation, its p-value, and the
|
|
186
|
+
same per-level `values` as categorical — but in the feature's *declared*
|
|
187
|
+
category order, not order of first appearance.
|
|
188
|
+
|
|
189
|
+
See [`docs/metrics.md`](docs/metrics.md) for how to choose between and
|
|
190
|
+
interpret `effect_size`/`p_value` across all three feature types
|
|
191
|
+
(including why Spearman, not a naive Pearson correlation on level codes, is
|
|
192
|
+
the right tool for ordinal data), and
|
|
193
|
+
[`examples/binary_categorical_discovery.py`](examples/binary_categorical_discovery.py) /
|
|
194
|
+
[`examples/mixed_categorical_continuous_discovery.py`](examples/mixed_categorical_continuous_discovery.py) /
|
|
195
|
+
[`examples/ordinal_binary_discovery.py`](examples/ordinal_binary_discovery.py)
|
|
196
|
+
for runnable, printed examples.
|
|
197
|
+
|
|
198
|
+
Also available: the low-level, pure statistical building blocks in
|
|
199
|
+
`signapy.metrics` (`association.cramers_v`, `association.point_biserial`,
|
|
200
|
+
`association.spearman_rho`, `significance.chi_square`,
|
|
201
|
+
`lift.categorical_lift`) that `discover()` is built on, the result models
|
|
202
|
+
(`signapy.results.FeatureResult`, `ValueResult`, `DiscoveryReport`), and the
|
|
203
|
+
semantic type enums (`signapy.profiling.FeatureType`, `TargetType`).
|
|
204
|
+
|
|
205
|
+
`signapy.discover()` characterizes univariate statistical relationships. It
|
|
206
|
+
does not train a classifier, select features automatically, or evaluate how
|
|
207
|
+
features perform in combination — that judgment, and any modeling built on
|
|
208
|
+
this evidence, stays yours.
|
|
209
|
+
|
|
210
|
+
## v0.1 scope
|
|
211
|
+
|
|
212
|
+
Implemented so far:
|
|
213
|
+
|
|
214
|
+
- **Target:** binary classification
|
|
215
|
+
- **Categorical features:** Cramér's V (effect size), chi-square test
|
|
216
|
+
(p-value), and per-category support/target rate/baseline rate/lift
|
|
217
|
+
- **Continuous features:** point-biserial correlation (effect size and
|
|
218
|
+
p-value) and group counts/means; no value-level localization yet
|
|
219
|
+
- **Ordinal features:** Spearman rank correlation (effect size and
|
|
220
|
+
p-value), and per-level support/target rate/baseline rate/lift in
|
|
221
|
+
declared category order; requires an explicit ordered `Categorical`
|
|
222
|
+
- **Feature selection:** explicit, via the `feature_types` argument;
|
|
223
|
+
numeric/boolean columns are never inferred as a semantic type, and an
|
|
224
|
+
order is never inferred for ordinal columns
|
|
225
|
+
|
|
226
|
+
See [`docs/architecture.md`](docs/architecture.md) for the full design,
|
|
227
|
+
including the missing-data and dtype policies this slice follows.
|
|
228
|
+
|
|
229
|
+
## Planned next
|
|
230
|
+
|
|
231
|
+
Not yet implemented: boolean features, semantic type overrides (e.g.
|
|
232
|
+
integer-coded categories), multiclass or continuous targets,
|
|
233
|
+
continuous-feature value-level localization (binning — unlike ordinal,
|
|
234
|
+
which got this via its declared levels), additional value-level metrics
|
|
235
|
+
(risk difference, odds ratio, confidence intervals), interaction discovery,
|
|
236
|
+
and stability analysis. See the
|
|
237
|
+
[discovery hierarchy](#discovery-hierarchy) above and
|
|
238
|
+
[`docs/architecture.md`](docs/architecture.md) for how these fit the overall
|
|
239
|
+
design.
|
|
240
|
+
|
|
241
|
+
## Versioning
|
|
242
|
+
|
|
243
|
+
SignaPy uses [PEP 440](https://peps.python.org/pep-0440/) versions and will
|
|
244
|
+
follow [semantic versioning](https://semver.org/) for releases. Pre-release
|
|
245
|
+
development builds use `.devN` suffixes. The current alpha release is
|
|
246
|
+
`0.1.0a1`. The version is defined once, in `src/signapy/__init__.py`.
|
|
247
|
+
|
|
248
|
+
## License
|
|
249
|
+
|
|
250
|
+
MIT. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
# SignaPy
|
|
2
|
+
|
|
3
|
+
**SignaPy is an exploratory feature-discovery library for labeled datasets.**
|
|
4
|
+
|
|
5
|
+
Long-term mission: *find, quantify, localize, and validate predictive signal.*
|
|
6
|
+
|
|
7
|
+
> **Status: alpha (`0.1.0a1`).** `signapy.discover()` works against
|
|
8
|
+
> **binary targets** with **categorical, continuous, and ordinal features**
|
|
9
|
+
> today. Everything under [Planned next](#planned-next) — other
|
|
10
|
+
> feature/target types, interactions, stability, and more — is not
|
|
11
|
+
> implemented yet.
|
|
12
|
+
|
|
13
|
+
## The problem
|
|
14
|
+
|
|
15
|
+
Before modeling, data scientists want to know which features carry signal about
|
|
16
|
+
the target, and where within those features that signal lives. Today this means
|
|
17
|
+
picking the right test for each feature/target combination (Cramér's V? Phi?
|
|
18
|
+
Spearman? point-biserial?), running it by hand, and stitching the output together.
|
|
19
|
+
|
|
20
|
+
SignaPy aims to make that a single, task-aware workflow. It uses the feature
|
|
21
|
+
types you declare (or the categorical-only default), chooses an appropriate
|
|
22
|
+
method, and returns structured evidence. You shouldn't need to know ahead of
|
|
23
|
+
time which statistical test applies.
|
|
24
|
+
|
|
25
|
+
SignaPy discovers evidence of predictive signal.
|
|
26
|
+
It does not decide what your model should use.
|
|
27
|
+
|
|
28
|
+
It is **not** an AutoML framework, a feature-selection oracle, or a modeling
|
|
29
|
+
library, and it does not drop features based on p-values.
|
|
30
|
+
|
|
31
|
+
## Discovery hierarchy
|
|
32
|
+
|
|
33
|
+
| Level | Question |
|
|
34
|
+
| --- | --- |
|
|
35
|
+
| Profiling | What data am I working with? |
|
|
36
|
+
| Feature-level discovery | Does this feature contain signal about the target? |
|
|
37
|
+
| Value-level discovery | Where within this feature does the signal live? |
|
|
38
|
+
| Interaction discovery *(future)* | Does signal emerge from combinations of features or values? |
|
|
39
|
+
| Stability analysis *(future)* | Does the signal hold across samples, folds, time, or subgroups? |
|
|
40
|
+
|
|
41
|
+
For example, feature-level discovery might find that `acquisition_channel` is
|
|
42
|
+
associated with conversion. Value-level discovery then shows *where*:
|
|
43
|
+
|
|
44
|
+
```text
|
|
45
|
+
referral -> lift 1.91
|
|
46
|
+
organic -> lift 1.28
|
|
47
|
+
paid -> lift 0.73
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
## Installation (development)
|
|
51
|
+
|
|
52
|
+
SignaPy supports Python 3.10 to 3.14. Clone the repository and install its
|
|
53
|
+
development dependencies:
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
git clone <repo-url> signapy
|
|
57
|
+
cd signapy
|
|
58
|
+
python -m venv .venv
|
|
59
|
+
source .venv/bin/activate
|
|
60
|
+
pip install -e ".[dev]"
|
|
61
|
+
pytest
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Runtime dependencies are `numpy`, `pandas`, and `scipy`. SignaPy doesn't
|
|
65
|
+
depend on any ML framework.
|
|
66
|
+
|
|
67
|
+
## Usage
|
|
68
|
+
|
|
69
|
+
Categorical-only (every non-target column analyzed automatically):
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
import signapy
|
|
73
|
+
|
|
74
|
+
report = signapy.discover(df, target="converted", positive_class=True)
|
|
75
|
+
|
|
76
|
+
report.features # feature-level results
|
|
77
|
+
report.feature("acquisition_channel") # one feature's evidence
|
|
78
|
+
report.feature("acquisition_channel").values # value-level results
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
Mixed categorical and continuous — pass `feature_types` to select exactly
|
|
82
|
+
which columns to analyze and declare each one's semantic type:
|
|
83
|
+
|
|
84
|
+
```python
|
|
85
|
+
from signapy.profiling import FeatureType
|
|
86
|
+
|
|
87
|
+
report = signapy.discover(
|
|
88
|
+
df,
|
|
89
|
+
target="outcome",
|
|
90
|
+
positive_class=True,
|
|
91
|
+
feature_types={
|
|
92
|
+
"category_context": FeatureType.CATEGORICAL,
|
|
93
|
+
"continuous_measurement": FeatureType.CONTINUOUS,
|
|
94
|
+
},
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
report.feature("continuous_measurement").effect_size # point-biserial coefficient
|
|
98
|
+
report.feature("continuous_measurement").p_value # its p-value
|
|
99
|
+
report.feature("continuous_measurement").values # always None (see below)
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
An **ordinal** feature needs an explicit order — SignaPy never infers one,
|
|
103
|
+
not from the values and never alphabetically — declared via an ordered
|
|
104
|
+
pandas `Categorical`:
|
|
105
|
+
|
|
106
|
+
```python
|
|
107
|
+
df["severity"] = pd.Categorical(
|
|
108
|
+
df["severity"],
|
|
109
|
+
categories=["low", "medium", "high"],
|
|
110
|
+
ordered=True,
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
report = signapy.discover(
|
|
114
|
+
df,
|
|
115
|
+
target="converted",
|
|
116
|
+
positive_class=True,
|
|
117
|
+
feature_types={"severity": FeatureType.ORDINAL},
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
report.feature("severity").effect_size # Spearman's rho
|
|
121
|
+
report.feature(
|
|
122
|
+
"severity"
|
|
123
|
+
).values # support/rate/baseline/lift, in low->medium->high order
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
Pandas converts any value omitted from `categories=[...]` to a missing value;
|
|
127
|
+
under SignaPy's missing-data policy, that row is then excluded from this
|
|
128
|
+
feature's analysis. Make sure the declaration includes every intended level.
|
|
129
|
+
|
|
130
|
+
`discover()` supports **categorical, continuous, and ordinal features
|
|
131
|
+
against a binary target** (see [v0.1 scope](#v01-scope) below):
|
|
132
|
+
|
|
133
|
+
- Without `feature_types` (the default), every non-target column of `df`
|
|
134
|
+
with an `object`, pandas `string`, or pandas `category` dtype is analyzed
|
|
135
|
+
as categorical; other dtypes raise a clear error.
|
|
136
|
+
- With `feature_types`, only the named columns are analyzed, each as its
|
|
137
|
+
declared type (`FeatureType.CATEGORICAL`, `FeatureType.CONTINUOUS`, or
|
|
138
|
+
`FeatureType.ORDINAL` — the only three supported so far). **Numeric
|
|
139
|
+
columns are never silently treated as continuous, and categorical columns
|
|
140
|
+
are never silently treated as ordinal** — an integer column might be a
|
|
141
|
+
measured quantity, an ordinal level, or a category identifier, and
|
|
142
|
+
SignaPy doesn't guess which; declaring it via `feature_types` is
|
|
143
|
+
required, and an ordinal declaration additionally requires the column to
|
|
144
|
+
already be an ordered `Categorical`.
|
|
145
|
+
- Categorical features get Cramér's V, a chi-square test, and per-category
|
|
146
|
+
`values` (support, target rate, baseline rate, lift), in order of first
|
|
147
|
+
appearance. Continuous features get a point-biserial correlation, its
|
|
148
|
+
p-value, and group counts/means in `details` — but **`values` is always
|
|
149
|
+
`None`** for continuous features: localizing *where within a continuous
|
|
150
|
+
range* the signal lives needs binning, which SignaPy doesn't do
|
|
151
|
+
automatically (see [`docs/architecture.md`](docs/architecture.md)).
|
|
152
|
+
Ordinal features get a Spearman rank correlation, its p-value, and the
|
|
153
|
+
same per-level `values` as categorical — but in the feature's *declared*
|
|
154
|
+
category order, not order of first appearance.
|
|
155
|
+
|
|
156
|
+
See [`docs/metrics.md`](docs/metrics.md) for how to choose between and
|
|
157
|
+
interpret `effect_size`/`p_value` across all three feature types
|
|
158
|
+
(including why Spearman, not a naive Pearson correlation on level codes, is
|
|
159
|
+
the right tool for ordinal data), and
|
|
160
|
+
[`examples/binary_categorical_discovery.py`](examples/binary_categorical_discovery.py) /
|
|
161
|
+
[`examples/mixed_categorical_continuous_discovery.py`](examples/mixed_categorical_continuous_discovery.py) /
|
|
162
|
+
[`examples/ordinal_binary_discovery.py`](examples/ordinal_binary_discovery.py)
|
|
163
|
+
for runnable, printed examples.
|
|
164
|
+
|
|
165
|
+
Also available: the low-level, pure statistical building blocks in
|
|
166
|
+
`signapy.metrics` (`association.cramers_v`, `association.point_biserial`,
|
|
167
|
+
`association.spearman_rho`, `significance.chi_square`,
|
|
168
|
+
`lift.categorical_lift`) that `discover()` is built on, the result models
|
|
169
|
+
(`signapy.results.FeatureResult`, `ValueResult`, `DiscoveryReport`), and the
|
|
170
|
+
semantic type enums (`signapy.profiling.FeatureType`, `TargetType`).
|
|
171
|
+
|
|
172
|
+
`signapy.discover()` characterizes univariate statistical relationships. It
|
|
173
|
+
does not train a classifier, select features automatically, or evaluate how
|
|
174
|
+
features perform in combination — that judgment, and any modeling built on
|
|
175
|
+
this evidence, stays yours.
|
|
176
|
+
|
|
177
|
+
## v0.1 scope
|
|
178
|
+
|
|
179
|
+
Implemented so far:
|
|
180
|
+
|
|
181
|
+
- **Target:** binary classification
|
|
182
|
+
- **Categorical features:** Cramér's V (effect size), chi-square test
|
|
183
|
+
(p-value), and per-category support/target rate/baseline rate/lift
|
|
184
|
+
- **Continuous features:** point-biserial correlation (effect size and
|
|
185
|
+
p-value) and group counts/means; no value-level localization yet
|
|
186
|
+
- **Ordinal features:** Spearman rank correlation (effect size and
|
|
187
|
+
p-value), and per-level support/target rate/baseline rate/lift in
|
|
188
|
+
declared category order; requires an explicit ordered `Categorical`
|
|
189
|
+
- **Feature selection:** explicit, via the `feature_types` argument;
|
|
190
|
+
numeric/boolean columns are never inferred as a semantic type, and an
|
|
191
|
+
order is never inferred for ordinal columns
|
|
192
|
+
|
|
193
|
+
See [`docs/architecture.md`](docs/architecture.md) for the full design,
|
|
194
|
+
including the missing-data and dtype policies this slice follows.
|
|
195
|
+
|
|
196
|
+
## Planned next
|
|
197
|
+
|
|
198
|
+
Not yet implemented: boolean features, semantic type overrides (e.g.
|
|
199
|
+
integer-coded categories), multiclass or continuous targets,
|
|
200
|
+
continuous-feature value-level localization (binning — unlike ordinal,
|
|
201
|
+
which got this via its declared levels), additional value-level metrics
|
|
202
|
+
(risk difference, odds ratio, confidence intervals), interaction discovery,
|
|
203
|
+
and stability analysis. See the
|
|
204
|
+
[discovery hierarchy](#discovery-hierarchy) above and
|
|
205
|
+
[`docs/architecture.md`](docs/architecture.md) for how these fit the overall
|
|
206
|
+
design.
|
|
207
|
+
|
|
208
|
+
## Versioning
|
|
209
|
+
|
|
210
|
+
SignaPy uses [PEP 440](https://peps.python.org/pep-0440/) versions and will
|
|
211
|
+
follow [semantic versioning](https://semver.org/) for releases. Pre-release
|
|
212
|
+
development builds use `.devN` suffixes. The current alpha release is
|
|
213
|
+
`0.1.0a1`. The version is defined once, in `src/signapy/__init__.py`.
|
|
214
|
+
|
|
215
|
+
## License
|
|
216
|
+
|
|
217
|
+
MIT. See [LICENSE](LICENSE).
|