tmc-pbp 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tmc_pbp-0.1.0/LICENSE +21 -0
- tmc_pbp-0.1.0/PKG-INFO +176 -0
- tmc_pbp-0.1.0/README.md +147 -0
- tmc_pbp-0.1.0/__init__.py +24 -0
- tmc_pbp-0.1.0/build_pbp.sh +28 -0
- tmc_pbp-0.1.0/cli.py +268 -0
- tmc_pbp-0.1.0/codec.py +2016 -0
- tmc_pbp-0.1.0/codec_c.py +307 -0
- tmc_pbp-0.1.0/core.py +374 -0
- tmc_pbp-0.1.0/core_c.py +250 -0
- tmc_pbp-0.1.0/edge_refine.py +207 -0
- tmc_pbp-0.1.0/evaluate.py +199 -0
- tmc_pbp-0.1.0/fast_degree.py +847 -0
- tmc_pbp-0.1.0/inference.py +615 -0
- tmc_pbp-0.1.0/mesh_advanced.py +231 -0
- tmc_pbp-0.1.0/mesh_applications.py +321 -0
- tmc_pbp-0.1.0/mesh_depth.py +300 -0
- tmc_pbp-0.1.0/mesh_gen.py +1258 -0
- tmc_pbp-0.1.0/mesh_topology.py +289 -0
- tmc_pbp-0.1.0/pbp_codec.c +291 -0
- tmc_pbp-0.1.0/pbp_core.c +735 -0
- tmc_pbp-0.1.0/pbp_core.h +182 -0
- tmc_pbp-0.1.0/perm_red.py +2349 -0
- tmc_pbp-0.1.0/pertpy_integration.py +929 -0
- tmc_pbp-0.1.0/pipeline.py +167 -0
- tmc_pbp-0.1.0/poly_viz.py +429 -0
- tmc_pbp-0.1.0/pyproject.toml +40 -0
- tmc_pbp-0.1.0/server.py +372 -0
- tmc_pbp-0.1.0/setup.cfg +4 -0
- tmc_pbp-0.1.0/test_codec.py +509 -0
- tmc_pbp-0.1.0/test_core_c.py +284 -0
- tmc_pbp-0.1.0/test_images.py +655 -0
- tmc_pbp-0.1.0/test_pbp.py +26 -0
- tmc_pbp-0.1.0/tmc_pbp.egg-info/PKG-INFO +176 -0
- tmc_pbp-0.1.0/tmc_pbp.egg-info/SOURCES.txt +65 -0
- tmc_pbp-0.1.0/tmc_pbp.egg-info/dependency_links.txt +1 -0
- tmc_pbp-0.1.0/tmc_pbp.egg-info/entry_points.txt +2 -0
- tmc_pbp-0.1.0/tmc_pbp.egg-info/requires.txt +21 -0
- tmc_pbp-0.1.0/tmc_pbp.egg-info/top_level.txt +1 -0
tmc_pbp-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Tendai Mapungwana Chikake
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
tmc_pbp-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tmc-pbp
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Pseudo-Boolean Polynomial decomposition for data analysis
|
|
5
|
+
Author-email: Tendai Mapungwana Chikake <tendaichikake@phystech.edu>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Repository, https://github.com/Tenfleques/tmc-pbp
|
|
8
|
+
Requires-Python: >=3.10
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Requires-Dist: numpy
|
|
12
|
+
Requires-Dist: pandas
|
|
13
|
+
Requires-Dist: scipy
|
|
14
|
+
Requires-Dist: bitarray
|
|
15
|
+
Requires-Dist: scikit-learn
|
|
16
|
+
Provides-Extra: viz
|
|
17
|
+
Requires-Dist: matplotlib; extra == "viz"
|
|
18
|
+
Provides-Extra: images
|
|
19
|
+
Requires-Dist: opencv-python; extra == "images"
|
|
20
|
+
Requires-Dist: Pillow; extra == "images"
|
|
21
|
+
Provides-Extra: server
|
|
22
|
+
Requires-Dist: flask; extra == "server"
|
|
23
|
+
Provides-Extra: all
|
|
24
|
+
Requires-Dist: matplotlib; extra == "all"
|
|
25
|
+
Requires-Dist: opencv-python; extra == "all"
|
|
26
|
+
Requires-Dist: Pillow; extra == "all"
|
|
27
|
+
Requires-Dist: flask; extra == "all"
|
|
28
|
+
Dynamic: license-file
|
|
29
|
+
|
|
30
|
+
# tmc-pbp
|
|
31
|
+
|
|
32
|
+
Pseudo-Boolean Polynomial (PBP) decomposition for data analysis. Training-free, deterministic algebraic decomposition of data matrices into multilinear polynomials over binary variables.
|
|
33
|
+
|
|
34
|
+
## What it does
|
|
35
|
+
|
|
36
|
+
Given a data matrix (rows = variables, columns = observations), PBP decomposes it into a multilinear polynomial where each term represents a specific combination of variables and its coefficient quantifies the interaction strength. The decomposition is:
|
|
37
|
+
|
|
38
|
+
- **Training-free** -- no learned parameters, no optimization
|
|
39
|
+
- **Deterministic** -- same input always produces the same output
|
|
40
|
+
- **Fast** -- sub-millisecond per sample, 15K genes in 1.7 seconds
|
|
41
|
+
- **Interpretable** -- each coefficient names a specific variable interaction
|
|
42
|
+
- **Mathematically equivalent** to the Walsh-Hadamard spectral transform
|
|
43
|
+
|
|
44
|
+
## Install
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
pip install tmc-pbp # core (numpy, pandas, scipy, bitarray, scikit-learn)
|
|
48
|
+
pip install "tmc-pbp[all]" # plus matplotlib, OpenCV, Pillow and Flask for the viz, image and server modules
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
The package is imported as `pbp`. A pure-Python core is always available (`pbp.core`).
|
|
52
|
+
The optional C backend (`pbp.core_c`) is shipped as source; compile it once in the installed
|
|
53
|
+
package directory to enable it:
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
bash "$(python -c 'import pbp, os; print(os.path.dirname(pbp.__file__))')/build_pbp.sh"
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## Quick start
|
|
60
|
+
|
|
61
|
+
### Python API
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
from pbp.core import create_pbp, pbp_vector
|
|
65
|
+
|
|
66
|
+
import numpy as np
|
|
67
|
+
matrix = np.array([
|
|
68
|
+
[5.2, 4.8, 3.1, 2.0], # Gene A
|
|
69
|
+
[3.0, 3.5, 4.2, 4.8], # Gene B
|
|
70
|
+
[1.0, 1.5, 2.8, 4.5], # Gene C
|
|
71
|
+
])
|
|
72
|
+
|
|
73
|
+
# Full decomposition -> DataFrame with (y, coeffs, degree)
|
|
74
|
+
pbp = create_pbp(matrix)
|
|
75
|
+
|
|
76
|
+
# Fixed-length vector representation (2^m - 1 elements)
|
|
77
|
+
vec = pbp_vector(matrix)
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
### CLI
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
pbp analyze matrix.csv --output results/ # Full decomposition + energy profile
|
|
84
|
+
pbp vector matrix.csv # Fixed-length PBP vector
|
|
85
|
+
pbp hasse matrix.csv --format dot # Hasse diagram (Graphviz DOT)
|
|
86
|
+
pbp info matrix.csv # Matrix stats
|
|
87
|
+
pbp version # Package version
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
### Web application
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
bash pbp/webapp/run.sh
|
|
94
|
+
# Open http://localhost:8430
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
6-tab interface covering decomposition, epistasis, quorum sensing, sequential inference, structural comparison, and anomaly detection. 25 API endpoints. OpenAPI docs at `/docs`.
|
|
98
|
+
|
|
99
|
+
## Modules
|
|
100
|
+
|
|
101
|
+
| Module | Purpose |
|
|
102
|
+
|--------|---------|
|
|
103
|
+
| `core.py` | Canonical PBP decomposition. `create_pbp()`, `pbp_vector()`, permutation/coefficient/variable matrices. Bitarray-based, no size limit. |
|
|
104
|
+
| `inference.py` | PBP-DAG sequential inference. Autoregressive sampling through the Hasse diagram using PBP coefficients as Boltzmann energies. Greedy, stochastic, and beam search modes. |
|
|
105
|
+
| `evaluate.py` | Evaluation metrics (Spearman, Kendall, NDE), bootstrap CIs, and baseline predictors (expression magnitude, PCA order, random). |
|
|
106
|
+
| `pertpy_integration.py` | Pertpy/scverse-compatible functions: `pbp_distance()`, `pbp_score()`, `pbp_classify_type()`. Works with AnnData objects or plain numpy arrays. |
|
|
107
|
+
| `cli.py` | Command-line interface. `pbp analyze\|vector\|hasse\|info\|version`. |
|
|
108
|
+
| `pipeline.py` | High-level pipeline utilities. |
|
|
109
|
+
| `server.py` | Legacy Flask server for DAG inference visualization. |
|
|
110
|
+
| `webapp/` | FastAPI web application (25 endpoints, single-page frontend with Chart.js + D3.js). |
|
|
111
|
+
|
|
112
|
+
## Applications
|
|
113
|
+
|
|
114
|
+
### Genetic epistasis (Perturb-seq)
|
|
115
|
+
PBP spectral profiles characterize interaction structure in combinatorial perturbation screens. Degree-wise energy separates interaction type from magnitude. Validated on 4 datasets from 4 labs (Norman, Wessels, Dixit, Joung).
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
from pbp.pertpy_integration import pbp_score, pbp_classify_type
|
|
119
|
+
scores = pbp_score(adata, groupby='perturbation', reference='control')
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
### Quorum sensing signal integration
|
|
123
|
+
Decompose factorial QS experiments into main effects (a1, a2) and interaction (a12) per gene. Gate classification: AND, OR, antagonistic, mixed.
|
|
124
|
+
|
|
125
|
+
### Sequential inference
|
|
126
|
+
Predict activation order from expression matrices using greedy sampling through the PBP Hasse diagram.
|
|
127
|
+
|
|
128
|
+
```python
|
|
129
|
+
from pbp.inference import pbp_dag_sample
|
|
130
|
+
result = pbp_dag_sample(pbp_df, m, mode='greedy')
|
|
131
|
+
print(result['sequence']) # Predicted activation order
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
### Anomaly detection
|
|
135
|
+
PBP total energy flags samples with disrupted interaction structure. Interpretable: identifies which specific interactions are affected.
|
|
136
|
+
|
|
137
|
+
### Structural comparison
|
|
138
|
+
Pairwise Hasse diagram distance compares interaction architectures across samples, conditions, or organisms.
|
|
139
|
+
|
|
140
|
+
## Key functions
|
|
141
|
+
|
|
142
|
+
| Function | What it does |
|
|
143
|
+
|----------|-------------|
|
|
144
|
+
| `create_pbp(matrix)` | Full PBP decomposition -> DataFrame with monomial index, coefficient, degree |
|
|
145
|
+
| `pbp_vector(matrix)` | Fixed-length vector (2^m - 1 coefficients) for machine learning |
|
|
146
|
+
| `pbp_dag_sample(pbp, m)` | Greedy/stochastic/beam sequential inference through Hasse diagram |
|
|
147
|
+
| `evaluate_sequence(pred, gt)` | Spearman rho, Kendall tau, top-k accuracy, NDE |
|
|
148
|
+
| `pbp_distance(adata, groupby)` | Pairwise structural distance (Pertpy-compatible) |
|
|
149
|
+
| `pbp_score(adata, groupby)` | Spectral profile + interaction fraction per group |
|
|
150
|
+
| `pbp_classify_type(adata, groupby)` | Synergistic/antagonistic/additive classification |
|
|
151
|
+
|
|
152
|
+
## Mathematical identity
|
|
153
|
+
|
|
154
|
+
PBP decomposition is mathematically identical to:
|
|
155
|
+
- **Walsh-Hadamard spectral transform** (Fourier analysis on {0,1}^m)
|
|
156
|
+
- **Epistatic interaction coefficients** (Poelwijk et al. 2019)
|
|
157
|
+
- **Multilinear extension** of pseudo-Boolean functions (Hammer & Rudeanu 1968, Boros & Hammer 2002)
|
|
158
|
+
|
|
159
|
+
## Citation
|
|
160
|
+
|
|
161
|
+
```bibtex
|
|
162
|
+
@article{chikake2025compoptics,
|
|
163
|
+
author = {Chikake, Tendai M. and Goldengorin, Boris I. and Pardalos, Panos M.},
|
|
164
|
+
title = {Pseudo-Boolean Polynomial Approach to Solving Computer Vision Tasks},
|
|
165
|
+
journal = {Computer Optics},
|
|
166
|
+
volume = {49},
|
|
167
|
+
number = {6},
|
|
168
|
+
pages = {1191--1201},
|
|
169
|
+
year = {2025},
|
|
170
|
+
doi = {10.18287/COJ1815},
|
|
171
|
+
}
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
## License
|
|
175
|
+
|
|
176
|
+
MIT
|
tmc_pbp-0.1.0/README.md
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
# tmc-pbp
|
|
2
|
+
|
|
3
|
+
Pseudo-Boolean Polynomial (PBP) decomposition for data analysis. Training-free, deterministic algebraic decomposition of data matrices into multilinear polynomials over binary variables.
|
|
4
|
+
|
|
5
|
+
## What it does
|
|
6
|
+
|
|
7
|
+
Given a data matrix (rows = variables, columns = observations), PBP decomposes it into a multilinear polynomial where each term represents a specific combination of variables and its coefficient quantifies the interaction strength. The decomposition is:
|
|
8
|
+
|
|
9
|
+
- **Training-free** -- no learned parameters, no optimization
|
|
10
|
+
- **Deterministic** -- same input always produces the same output
|
|
11
|
+
- **Fast** -- sub-millisecond per sample, 15K genes in 1.7 seconds
|
|
12
|
+
- **Interpretable** -- each coefficient names a specific variable interaction
|
|
13
|
+
- **Mathematically equivalent** to the Walsh-Hadamard spectral transform
|
|
14
|
+
|
|
15
|
+
## Install
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
pip install tmc-pbp # core (numpy, pandas, scipy, bitarray, scikit-learn)
|
|
19
|
+
pip install "tmc-pbp[all]" # plus matplotlib, OpenCV, Pillow and Flask for the viz, image and server modules
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
The package is imported as `pbp`. A pure-Python core is always available (`pbp.core`).
|
|
23
|
+
The optional C backend (`pbp.core_c`) is shipped as source; compile it once in the installed
|
|
24
|
+
package directory to enable it:
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
bash "$(python -c 'import pbp, os; print(os.path.dirname(pbp.__file__))')/build_pbp.sh"
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
## Quick start
|
|
31
|
+
|
|
32
|
+
### Python API
|
|
33
|
+
|
|
34
|
+
```python
|
|
35
|
+
from pbp.core import create_pbp, pbp_vector
|
|
36
|
+
|
|
37
|
+
import numpy as np
|
|
38
|
+
matrix = np.array([
|
|
39
|
+
[5.2, 4.8, 3.1, 2.0], # Gene A
|
|
40
|
+
[3.0, 3.5, 4.2, 4.8], # Gene B
|
|
41
|
+
[1.0, 1.5, 2.8, 4.5], # Gene C
|
|
42
|
+
])
|
|
43
|
+
|
|
44
|
+
# Full decomposition -> DataFrame with (y, coeffs, degree)
|
|
45
|
+
pbp = create_pbp(matrix)
|
|
46
|
+
|
|
47
|
+
# Fixed-length vector representation (2^m - 1 elements)
|
|
48
|
+
vec = pbp_vector(matrix)
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
### CLI
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
pbp analyze matrix.csv --output results/ # Full decomposition + energy profile
|
|
55
|
+
pbp vector matrix.csv # Fixed-length PBP vector
|
|
56
|
+
pbp hasse matrix.csv --format dot # Hasse diagram (Graphviz DOT)
|
|
57
|
+
pbp info matrix.csv # Matrix stats
|
|
58
|
+
pbp version # Package version
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
### Web application
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
bash pbp/webapp/run.sh
|
|
65
|
+
# Open http://localhost:8430
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
6-tab interface covering decomposition, epistasis, quorum sensing, sequential inference, structural comparison, and anomaly detection. 25 API endpoints. OpenAPI docs at `/docs`.
|
|
69
|
+
|
|
70
|
+
## Modules
|
|
71
|
+
|
|
72
|
+
| Module | Purpose |
|
|
73
|
+
|--------|---------|
|
|
74
|
+
| `core.py` | Canonical PBP decomposition. `create_pbp()`, `pbp_vector()`, permutation/coefficient/variable matrices. Bitarray-based, no size limit. |
|
|
75
|
+
| `inference.py` | PBP-DAG sequential inference. Autoregressive sampling through the Hasse diagram using PBP coefficients as Boltzmann energies. Greedy, stochastic, and beam search modes. |
|
|
76
|
+
| `evaluate.py` | Evaluation metrics (Spearman, Kendall, NDE), bootstrap CIs, and baseline predictors (expression magnitude, PCA order, random). |
|
|
77
|
+
| `pertpy_integration.py` | Pertpy/scverse-compatible functions: `pbp_distance()`, `pbp_score()`, `pbp_classify_type()`. Works with AnnData objects or plain numpy arrays. |
|
|
78
|
+
| `cli.py` | Command-line interface. `pbp analyze\|vector\|hasse\|info\|version`. |
|
|
79
|
+
| `pipeline.py` | High-level pipeline utilities. |
|
|
80
|
+
| `server.py` | Legacy Flask server for DAG inference visualization. |
|
|
81
|
+
| `webapp/` | FastAPI web application (25 endpoints, single-page frontend with Chart.js + D3.js). |
|
|
82
|
+
|
|
83
|
+
## Applications
|
|
84
|
+
|
|
85
|
+
### Genetic epistasis (Perturb-seq)
|
|
86
|
+
PBP spectral profiles characterize interaction structure in combinatorial perturbation screens. Degree-wise energy separates interaction type from magnitude. Validated on 4 datasets from 4 labs (Norman, Wessels, Dixit, Joung).
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
from pbp.pertpy_integration import pbp_score, pbp_classify_type
|
|
90
|
+
scores = pbp_score(adata, groupby='perturbation', reference='control')
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
### Quorum sensing signal integration
|
|
94
|
+
Decompose factorial QS experiments into main effects (a1, a2) and interaction (a12) per gene. Gate classification: AND, OR, antagonistic, mixed.
|
|
95
|
+
|
|
96
|
+
### Sequential inference
|
|
97
|
+
Predict activation order from expression matrices using greedy sampling through the PBP Hasse diagram.
|
|
98
|
+
|
|
99
|
+
```python
|
|
100
|
+
from pbp.inference import pbp_dag_sample
|
|
101
|
+
result = pbp_dag_sample(pbp_df, m, mode='greedy')
|
|
102
|
+
print(result['sequence']) # Predicted activation order
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
### Anomaly detection
|
|
106
|
+
PBP total energy flags samples with disrupted interaction structure. Interpretable: identifies which specific interactions are affected.
|
|
107
|
+
|
|
108
|
+
### Structural comparison
|
|
109
|
+
Pairwise Hasse diagram distance compares interaction architectures across samples, conditions, or organisms.
|
|
110
|
+
|
|
111
|
+
## Key functions
|
|
112
|
+
|
|
113
|
+
| Function | What it does |
|
|
114
|
+
|----------|-------------|
|
|
115
|
+
| `create_pbp(matrix)` | Full PBP decomposition -> DataFrame with monomial index, coefficient, degree |
|
|
116
|
+
| `pbp_vector(matrix)` | Fixed-length vector (2^m - 1 coefficients) for machine learning |
|
|
117
|
+
| `pbp_dag_sample(pbp, m)` | Greedy/stochastic/beam sequential inference through Hasse diagram |
|
|
118
|
+
| `evaluate_sequence(pred, gt)` | Spearman rho, Kendall tau, top-k accuracy, NDE |
|
|
119
|
+
| `pbp_distance(adata, groupby)` | Pairwise structural distance (Pertpy-compatible) |
|
|
120
|
+
| `pbp_score(adata, groupby)` | Spectral profile + interaction fraction per group |
|
|
121
|
+
| `pbp_classify_type(adata, groupby)` | Synergistic/antagonistic/additive classification |
|
|
122
|
+
|
|
123
|
+
## Mathematical identity
|
|
124
|
+
|
|
125
|
+
PBP decomposition is mathematically identical to:
|
|
126
|
+
- **Walsh-Hadamard spectral transform** (Fourier analysis on {0,1}^m)
|
|
127
|
+
- **Epistatic interaction coefficients** (Poelwijk et al. 2019)
|
|
128
|
+
- **Multilinear extension** of pseudo-Boolean functions (Hammer & Rudeanu 1968, Boros & Hammer 2002)
|
|
129
|
+
|
|
130
|
+
## Citation
|
|
131
|
+
|
|
132
|
+
```bibtex
|
|
133
|
+
@article{chikake2025compoptics,
|
|
134
|
+
author = {Chikake, Tendai M. and Goldengorin, Boris I. and Pardalos, Panos M.},
|
|
135
|
+
title = {Pseudo-Boolean Polynomial Approach to Solving Computer Vision Tasks},
|
|
136
|
+
journal = {Computer Optics},
|
|
137
|
+
volume = {49},
|
|
138
|
+
number = {6},
|
|
139
|
+
pages = {1191--1201},
|
|
140
|
+
year = {2025},
|
|
141
|
+
doi = {10.18287/COJ1815},
|
|
142
|
+
}
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
## License
|
|
146
|
+
|
|
147
|
+
MIT
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""
|
|
2
|
+
pbp-polynomials: Pseudo-Boolean Polynomial decomposition for data analysis.
|
|
3
|
+
|
|
4
|
+
Training-free, deterministic algebraic decomposition of data matrices into
|
|
5
|
+
multilinear polynomials over binary variables. Applications: genetic epistasis
|
|
6
|
+
characterization, quorum sensing signal decomposition, sequential inference,
|
|
7
|
+
anomaly detection.
|
|
8
|
+
|
|
9
|
+
Usage:
|
|
10
|
+
from pbp.core import create_pbp, pbp_vector
|
|
11
|
+
from pbp.inference import pbp_dag_sample
|
|
12
|
+
from pbp.evaluate import evaluate_sequence
|
|
13
|
+
from pbp.pertpy_integration import pbp_distance, pbp_score, pbp_classify_type
|
|
14
|
+
|
|
15
|
+
CLI:
|
|
16
|
+
pbp analyze matrix.csv --output results/
|
|
17
|
+
pbp vector matrix.csv
|
|
18
|
+
pbp hasse matrix.csv --output diagram.json
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
__version__ = "0.1.0"
|
|
22
|
+
__author__ = "Tendai Chikake"
|
|
23
|
+
|
|
24
|
+
from pbp.core import create_pbp, pbp_vector
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
#!/bin/bash
|
|
2
|
+
# Build PBP shared libraries (core + codec)
|
|
3
|
+
# Usage: bash pbp/build_pbp.sh
|
|
4
|
+
|
|
5
|
+
set -e
|
|
6
|
+
cd "$(dirname "$0")"
|
|
7
|
+
|
|
8
|
+
CC="${CC:-cc}"
|
|
9
|
+
CFLAGS="-O2 -Wall -Wextra -std=c99 -fPIC"
|
|
10
|
+
|
|
11
|
+
case "$(uname -s)" in
|
|
12
|
+
Darwin)
|
|
13
|
+
SOEXT="dylib"
|
|
14
|
+
SOFLAGS="-dynamiclib"
|
|
15
|
+
;;
|
|
16
|
+
*)
|
|
17
|
+
SOEXT="so"
|
|
18
|
+
SOFLAGS="-shared"
|
|
19
|
+
;;
|
|
20
|
+
esac
|
|
21
|
+
|
|
22
|
+
echo "Compiling pbp_core.c -> libpbp_core.${SOEXT}"
|
|
23
|
+
$CC $CFLAGS $SOFLAGS -o "libpbp_core.${SOEXT}" pbp_core.c -lm
|
|
24
|
+
echo "Done: pbp/libpbp_core.${SOEXT}"
|
|
25
|
+
|
|
26
|
+
echo "Compiling pbp_codec.c -> libpbp_codec.${SOEXT}"
|
|
27
|
+
$CC $CFLAGS $SOFLAGS -o "libpbp_codec.${SOEXT}" pbp_codec.c -lm
|
|
28
|
+
echo "Done: pbp/libpbp_codec.${SOEXT}"
|
tmc_pbp-0.1.0/cli.py
ADDED
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
pbp — Command-line interface for pseudo-Boolean polynomial analysis.
|
|
4
|
+
|
|
5
|
+
Usage:
|
|
6
|
+
pbp analyze matrix.csv [--output DIR] [--format csv|json]
|
|
7
|
+
pbp vector matrix.csv [--output FILE]
|
|
8
|
+
pbp hasse matrix.csv [--output FILE] [--format json|dot]
|
|
9
|
+
pbp info matrix.csv
|
|
10
|
+
pbp version
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import argparse
|
|
14
|
+
import sys
|
|
15
|
+
import os
|
|
16
|
+
import io
|
|
17
|
+
import json
|
|
18
|
+
import time
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
import pandas as pd
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _quiet_pbp(matrix, **kwargs):
|
|
25
|
+
"""Run create_pbp without stdout noise."""
|
|
26
|
+
from pbp.core import create_pbp
|
|
27
|
+
old = sys.stdout
|
|
28
|
+
sys.stdout = io.StringIO()
|
|
29
|
+
try:
|
|
30
|
+
return create_pbp(matrix, **kwargs)
|
|
31
|
+
finally:
|
|
32
|
+
sys.stdout = old
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _load_matrix(path):
|
|
36
|
+
"""Load a matrix from CSV/TSV. Rows = variables, columns = conditions."""
|
|
37
|
+
if path.endswith('.tsv') or path.endswith('.txt'):
|
|
38
|
+
df = pd.read_csv(path, sep='\t')
|
|
39
|
+
else:
|
|
40
|
+
df = pd.read_csv(path)
|
|
41
|
+
|
|
42
|
+
# If first column is non-numeric (labels), use as index
|
|
43
|
+
if df.iloc[:, 0].dtype == object:
|
|
44
|
+
df = df.set_index(df.columns[0])
|
|
45
|
+
|
|
46
|
+
return df.values.astype(float), df.index.tolist(), df.columns.tolist()
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def cmd_analyze(args):
|
|
50
|
+
"""Full PBP analysis: decomposition + spectral profile + summary."""
|
|
51
|
+
matrix, row_labels, col_labels = _load_matrix(args.input)
|
|
52
|
+
m, n = matrix.shape
|
|
53
|
+
|
|
54
|
+
print(f"Matrix: {m} variables x {n} conditions")
|
|
55
|
+
print(f"Variables: {row_labels[:5]}{'...' if m > 5 else ''}")
|
|
56
|
+
|
|
57
|
+
t0 = time.time()
|
|
58
|
+
pbp = _quiet_pbp(matrix)
|
|
59
|
+
elapsed = time.time() - t0
|
|
60
|
+
|
|
61
|
+
print(f"PBP computed in {elapsed*1000:.1f} ms")
|
|
62
|
+
print(f" Monomials: {len(pbp)}")
|
|
63
|
+
print(f" Max degree: {int(pbp['degree'].max())}")
|
|
64
|
+
print(f" Non-zero coefficients: {(pbp['coeffs'] != 0).sum()}")
|
|
65
|
+
|
|
66
|
+
# Degree-wise energy
|
|
67
|
+
energy = pbp.groupby('degree')['coeffs'].apply(lambda x: (x**2).sum())
|
|
68
|
+
total_energy = energy.sum()
|
|
69
|
+
print(f"\n Degree-wise energy:")
|
|
70
|
+
for d, e in energy.items():
|
|
71
|
+
pct = 100 * e / total_energy if total_energy > 0 else 0
|
|
72
|
+
print(f" d={int(d)}: {e:.4f} ({pct:.1f}%)")
|
|
73
|
+
|
|
74
|
+
# PBP vector
|
|
75
|
+
from pbp.core import pbp_vector
|
|
76
|
+
vec = pbp_vector(matrix)
|
|
77
|
+
print(f"\n PBP vector: {len(vec)} elements")
|
|
78
|
+
|
|
79
|
+
# Save output
|
|
80
|
+
outdir = args.output or '.'
|
|
81
|
+
os.makedirs(outdir, exist_ok=True)
|
|
82
|
+
|
|
83
|
+
if args.format == 'json':
|
|
84
|
+
result = {
|
|
85
|
+
'matrix_shape': [m, n],
|
|
86
|
+
'row_labels': row_labels,
|
|
87
|
+
'col_labels': col_labels,
|
|
88
|
+
'n_monomials': len(pbp),
|
|
89
|
+
'max_degree': int(pbp['degree'].max()),
|
|
90
|
+
'elapsed_ms': round(elapsed * 1000, 1),
|
|
91
|
+
'degree_energy': {int(k): round(float(v), 6) for k, v in energy.items()},
|
|
92
|
+
'total_energy': round(float(total_energy), 6),
|
|
93
|
+
'pbp_vector': vec.tolist(),
|
|
94
|
+
}
|
|
95
|
+
outpath = os.path.join(outdir, 'pbp_analysis.json')
|
|
96
|
+
with open(outpath, 'w') as f:
|
|
97
|
+
json.dump(result, f, indent=2)
|
|
98
|
+
else:
|
|
99
|
+
# Save decomposition
|
|
100
|
+
outpath = os.path.join(outdir, 'pbp_decomposition.csv')
|
|
101
|
+
pbp.to_csv(outpath, index=False)
|
|
102
|
+
|
|
103
|
+
# Save vector
|
|
104
|
+
vecpath = os.path.join(outdir, 'pbp_vector.csv')
|
|
105
|
+
pd.DataFrame({'index': range(len(vec)), 'coefficient': vec}).to_csv(vecpath, index=False)
|
|
106
|
+
|
|
107
|
+
# Save energy
|
|
108
|
+
epath = os.path.join(outdir, 'pbp_energy.csv')
|
|
109
|
+
energy_df = pd.DataFrame({
|
|
110
|
+
'degree': energy.index.astype(int),
|
|
111
|
+
'energy': energy.values,
|
|
112
|
+
'fraction': energy.values / total_energy if total_energy > 0 else 0,
|
|
113
|
+
})
|
|
114
|
+
energy_df.to_csv(epath, index=False)
|
|
115
|
+
|
|
116
|
+
print(f"\n Output saved to {outdir}/")
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def cmd_vector(args):
|
|
120
|
+
"""Compute and output the PBP vector (fixed-length representation)."""
|
|
121
|
+
matrix, row_labels, _ = _load_matrix(args.input)
|
|
122
|
+
from pbp.core import pbp_vector
|
|
123
|
+
vec = pbp_vector(matrix)
|
|
124
|
+
|
|
125
|
+
if args.output:
|
|
126
|
+
pd.DataFrame({'index': range(len(vec)), 'coefficient': vec}).to_csv(args.output, index=False)
|
|
127
|
+
print(f"PBP vector ({len(vec)} elements) saved to {args.output}")
|
|
128
|
+
else:
|
|
129
|
+
for i, v in enumerate(vec):
|
|
130
|
+
if v != 0:
|
|
131
|
+
print(f" [{i}] = {v:.6f}")
|
|
132
|
+
print(f"\n {len(vec)} elements, {(vec != 0).sum()} non-zero")
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def cmd_hasse(args):
|
|
136
|
+
"""Export Hasse diagram structure."""
|
|
137
|
+
matrix, row_labels, _ = _load_matrix(args.input)
|
|
138
|
+
m = matrix.shape[0]
|
|
139
|
+
|
|
140
|
+
pbp = _quiet_pbp(matrix)
|
|
141
|
+
|
|
142
|
+
# Build Hasse structure
|
|
143
|
+
nodes = []
|
|
144
|
+
for _, row in pbp.iterrows():
|
|
145
|
+
y = int(row['y'])
|
|
146
|
+
d = int(row['degree'])
|
|
147
|
+
c = float(row['coeffs'])
|
|
148
|
+
bits = []
|
|
149
|
+
val = y
|
|
150
|
+
i = 0
|
|
151
|
+
while val:
|
|
152
|
+
if val & 1:
|
|
153
|
+
bits.append(i)
|
|
154
|
+
val >>= 1
|
|
155
|
+
i += 1
|
|
156
|
+
label = ','.join(str(b) for b in bits) if bits else '∅'
|
|
157
|
+
if row_labels and all(isinstance(r, str) for r in row_labels):
|
|
158
|
+
named = ','.join(row_labels[b] for b in bits) if bits else '∅'
|
|
159
|
+
else:
|
|
160
|
+
named = label
|
|
161
|
+
nodes.append({
|
|
162
|
+
'id': y, 'degree': d, 'coefficient': c,
|
|
163
|
+
'variables': bits, 'label': label, 'named': named,
|
|
164
|
+
})
|
|
165
|
+
|
|
166
|
+
# Build edges (subset relations between adjacent degrees)
|
|
167
|
+
edges = []
|
|
168
|
+
by_degree = {}
|
|
169
|
+
for node in nodes:
|
|
170
|
+
by_degree.setdefault(node['degree'], []).append(node)
|
|
171
|
+
|
|
172
|
+
max_d = max(by_degree.keys()) if by_degree else 0
|
|
173
|
+
for d in range(max_d):
|
|
174
|
+
for lo in by_degree.get(d, []):
|
|
175
|
+
for hi in by_degree.get(d + 1, []):
|
|
176
|
+
if (lo['id'] & hi['id']) == lo['id']:
|
|
177
|
+
edges.append({'from': lo['id'], 'to': hi['id']})
|
|
178
|
+
|
|
179
|
+
result = {'nodes': nodes, 'edges': edges, 'm': m, 'labels': row_labels}
|
|
180
|
+
|
|
181
|
+
if args.format == 'dot':
|
|
182
|
+
# Graphviz DOT format
|
|
183
|
+
lines = ['digraph Hasse {', ' rankdir=TB;']
|
|
184
|
+
for node in nodes:
|
|
185
|
+
color = '#D55E00' if node['coefficient'] > 0 else '#0072B2'
|
|
186
|
+
lines.append(f' n{node["id"]} [label="{node["named"]}\\n{node["coefficient"]:.2f}" '
|
|
187
|
+
f'style=filled fillcolor="{color}40"];')
|
|
188
|
+
for edge in edges:
|
|
189
|
+
lines.append(f' n{edge["from"]} -> n{edge["to"]};')
|
|
190
|
+
lines.append('}')
|
|
191
|
+
out = '\n'.join(lines)
|
|
192
|
+
else:
|
|
193
|
+
out = json.dumps(result, indent=2)
|
|
194
|
+
|
|
195
|
+
if args.output:
|
|
196
|
+
with open(args.output, 'w') as f:
|
|
197
|
+
f.write(out)
|
|
198
|
+
print(f"Hasse diagram ({len(nodes)} nodes, {len(edges)} edges) saved to {args.output}")
|
|
199
|
+
else:
|
|
200
|
+
print(out)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def cmd_info(args):
|
|
204
|
+
"""Quick matrix info without full decomposition."""
|
|
205
|
+
matrix, row_labels, col_labels = _load_matrix(args.input)
|
|
206
|
+
m, n = matrix.shape
|
|
207
|
+
print(f"Matrix: {m} variables x {n} conditions")
|
|
208
|
+
print(f"Variables: {row_labels}")
|
|
209
|
+
print(f"Conditions: {col_labels}")
|
|
210
|
+
print(f"PBP vector length: {2**m - 1}")
|
|
211
|
+
print(f"Max monomials: {2**m}")
|
|
212
|
+
print(f"Value range: [{matrix.min():.4f}, {matrix.max():.4f}]")
|
|
213
|
+
print(f"Mean: {matrix.mean():.4f}, Std: {matrix.std():.4f}")
|
|
214
|
+
if m > 20:
|
|
215
|
+
print(f"WARNING: m={m} will produce {2**m:,} monomials. Consider reshaping.")
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def cmd_version(args):
|
|
219
|
+
from pbp import __version__
|
|
220
|
+
print(f"pbp-polynomials {__version__}")
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def main():
|
|
224
|
+
parser = argparse.ArgumentParser(
|
|
225
|
+
prog='pbp',
|
|
226
|
+
description='Pseudo-Boolean Polynomial analysis of data matrices',
|
|
227
|
+
)
|
|
228
|
+
subparsers = parser.add_subparsers(dest='command', help='Available commands')
|
|
229
|
+
|
|
230
|
+
# analyze
|
|
231
|
+
p_analyze = subparsers.add_parser('analyze', help='Full PBP analysis')
|
|
232
|
+
p_analyze.add_argument('input', help='Input matrix (CSV/TSV)')
|
|
233
|
+
p_analyze.add_argument('--output', '-o', help='Output directory')
|
|
234
|
+
p_analyze.add_argument('--format', choices=['csv', 'json'], default='csv')
|
|
235
|
+
p_analyze.set_defaults(func=cmd_analyze)
|
|
236
|
+
|
|
237
|
+
# vector
|
|
238
|
+
p_vector = subparsers.add_parser('vector', help='Compute PBP vector')
|
|
239
|
+
p_vector.add_argument('input', help='Input matrix (CSV/TSV)')
|
|
240
|
+
p_vector.add_argument('--output', '-o', help='Output CSV file')
|
|
241
|
+
p_vector.set_defaults(func=cmd_vector)
|
|
242
|
+
|
|
243
|
+
# hasse
|
|
244
|
+
p_hasse = subparsers.add_parser('hasse', help='Export Hasse diagram')
|
|
245
|
+
p_hasse.add_argument('input', help='Input matrix (CSV/TSV)')
|
|
246
|
+
p_hasse.add_argument('--output', '-o', help='Output file')
|
|
247
|
+
p_hasse.add_argument('--format', choices=['json', 'dot'], default='json')
|
|
248
|
+
p_hasse.set_defaults(func=cmd_hasse)
|
|
249
|
+
|
|
250
|
+
# info
|
|
251
|
+
p_info = subparsers.add_parser('info', help='Matrix info')
|
|
252
|
+
p_info.add_argument('input', help='Input matrix (CSV/TSV)')
|
|
253
|
+
p_info.set_defaults(func=cmd_info)
|
|
254
|
+
|
|
255
|
+
# version
|
|
256
|
+
p_version = subparsers.add_parser('version', help='Show version')
|
|
257
|
+
p_version.set_defaults(func=cmd_version)
|
|
258
|
+
|
|
259
|
+
args = parser.parse_args()
|
|
260
|
+
if not args.command:
|
|
261
|
+
parser.print_help()
|
|
262
|
+
sys.exit(1)
|
|
263
|
+
|
|
264
|
+
args.func(args)
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
if __name__ == '__main__':
|
|
268
|
+
main()
|