tmc-pbp 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tmc_pbp-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Tendai Mapungwana Chikake
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
tmc_pbp-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,176 @@
1
+ Metadata-Version: 2.4
2
+ Name: tmc-pbp
3
+ Version: 0.1.0
4
+ Summary: Pseudo-Boolean Polynomial decomposition for data analysis
5
+ Author-email: Tendai Mapungwana Chikake <tendaichikake@phystech.edu>
6
+ License: MIT
7
+ Project-URL: Repository, https://github.com/Tenfleques/tmc-pbp
8
+ Requires-Python: >=3.10
9
+ Description-Content-Type: text/markdown
10
+ License-File: LICENSE
11
+ Requires-Dist: numpy
12
+ Requires-Dist: pandas
13
+ Requires-Dist: scipy
14
+ Requires-Dist: bitarray
15
+ Requires-Dist: scikit-learn
16
+ Provides-Extra: viz
17
+ Requires-Dist: matplotlib; extra == "viz"
18
+ Provides-Extra: images
19
+ Requires-Dist: opencv-python; extra == "images"
20
+ Requires-Dist: Pillow; extra == "images"
21
+ Provides-Extra: server
22
+ Requires-Dist: flask; extra == "server"
23
+ Provides-Extra: all
24
+ Requires-Dist: matplotlib; extra == "all"
25
+ Requires-Dist: opencv-python; extra == "all"
26
+ Requires-Dist: Pillow; extra == "all"
27
+ Requires-Dist: flask; extra == "all"
28
+ Dynamic: license-file
29
+
30
+ # tmc-pbp
31
+
32
+ Pseudo-Boolean Polynomial (PBP) decomposition for data analysis. Training-free, deterministic algebraic decomposition of data matrices into multilinear polynomials over binary variables.
33
+
34
+ ## What it does
35
+
36
+ Given a data matrix (rows = variables, columns = observations), PBP decomposes it into a multilinear polynomial where each term represents a specific combination of variables and its coefficient quantifies the interaction strength. The decomposition is:
37
+
38
+ - **Training-free** -- no learned parameters, no optimization
39
+ - **Deterministic** -- same input always produces the same output
40
+ - **Fast** -- sub-millisecond per sample, 15K genes in 1.7 seconds
41
+ - **Interpretable** -- each coefficient names a specific variable interaction
42
+ - **Mathematically equivalent** to the Walsh-Hadamard spectral transform
43
+
44
+ ## Install
45
+
46
+ ```bash
47
+ pip install tmc-pbp # core (numpy, pandas, scipy, bitarray, scikit-learn)
48
+ pip install "tmc-pbp[all]" # plus matplotlib, OpenCV, Pillow and Flask for the viz, image and server modules
49
+ ```
50
+
51
+ The package is imported as `pbp`. A pure-Python core is always available (`pbp.core`).
52
+ The optional C backend (`pbp.core_c`) is shipped as source; compile it once in the installed
53
+ package directory to enable it:
54
+
55
+ ```bash
56
+ bash "$(python -c 'import pbp, os; print(os.path.dirname(pbp.__file__))')/build_pbp.sh"
57
+ ```
58
+
59
+ ## Quick start
60
+
61
+ ### Python API
62
+
63
+ ```python
64
+ from pbp.core import create_pbp, pbp_vector
65
+
66
+ import numpy as np
67
+ matrix = np.array([
68
+ [5.2, 4.8, 3.1, 2.0], # Gene A
69
+ [3.0, 3.5, 4.2, 4.8], # Gene B
70
+ [1.0, 1.5, 2.8, 4.5], # Gene C
71
+ ])
72
+
73
+ # Full decomposition -> DataFrame with (y, coeffs, degree)
74
+ pbp = create_pbp(matrix)
75
+
76
+ # Fixed-length vector representation (2^m - 1 elements)
77
+ vec = pbp_vector(matrix)
78
+ ```
79
+
80
+ ### CLI
81
+
82
+ ```bash
83
+ pbp analyze matrix.csv --output results/ # Full decomposition + energy profile
84
+ pbp vector matrix.csv # Fixed-length PBP vector
85
+ pbp hasse matrix.csv --format dot # Hasse diagram (Graphviz DOT)
86
+ pbp info matrix.csv # Matrix stats
87
+ pbp version # Package version
88
+ ```
89
+
90
+ ### Web application
91
+
92
+ ```bash
93
+ bash pbp/webapp/run.sh
94
+ # Open http://localhost:8430
95
+ ```
96
+
97
+ 6-tab interface covering decomposition, epistasis, quorum sensing, sequential inference, structural comparison, and anomaly detection. 25 API endpoints. OpenAPI docs at `/docs`.
98
+
99
+ ## Modules
100
+
101
+ | Module | Purpose |
102
+ |--------|---------|
103
+ | `core.py` | Canonical PBP decomposition. `create_pbp()`, `pbp_vector()`, permutation/coefficient/variable matrices. Bitarray-based, no size limit. |
104
+ | `inference.py` | PBP-DAG sequential inference. Autoregressive sampling through the Hasse diagram using PBP coefficients as Boltzmann energies. Greedy, stochastic, and beam search modes. |
105
+ | `evaluate.py` | Evaluation metrics (Spearman, Kendall, NDE), bootstrap CIs, and baseline predictors (expression magnitude, PCA order, random). |
106
+ | `pertpy_integration.py` | Pertpy/scverse-compatible functions: `pbp_distance()`, `pbp_score()`, `pbp_classify_type()`. Works with AnnData objects or plain numpy arrays. |
107
+ | `cli.py` | Command-line interface. `pbp analyze\|vector\|hasse\|info\|version`. |
108
+ | `pipeline.py` | High-level pipeline utilities. |
109
+ | `server.py` | Legacy Flask server for DAG inference visualization. |
110
+ | `webapp/` | FastAPI web application (25 endpoints, single-page frontend with Chart.js + D3.js). |
111
+
112
+ ## Applications
113
+
114
+ ### Genetic epistasis (Perturb-seq)
115
+ PBP spectral profiles characterize interaction structure in combinatorial perturbation screens. Degree-wise energy separates interaction type from magnitude. Validated on 4 datasets from 4 labs (Norman, Wessels, Dixit, Joung).
116
+
117
+ ```python
118
+ from pbp.pertpy_integration import pbp_score, pbp_classify_type
119
+ scores = pbp_score(adata, groupby='perturbation', reference='control')
120
+ ```
121
+
122
+ ### Quorum sensing signal integration
123
+ Decompose factorial QS experiments into main effects (a1, a2) and interaction (a12) per gene. Gate classification: AND, OR, antagonistic, mixed.
124
+
125
+ ### Sequential inference
126
+ Predict activation order from expression matrices using greedy sampling through the PBP Hasse diagram.
127
+
128
+ ```python
129
+ from pbp.inference import pbp_dag_sample
130
+ result = pbp_dag_sample(pbp_df, m, mode='greedy')
131
+ print(result['sequence']) # Predicted activation order
132
+ ```
133
+
134
+ ### Anomaly detection
135
+ PBP total energy flags samples with disrupted interaction structure. Interpretable: identifies which specific interactions are affected.
136
+
137
+ ### Structural comparison
138
+ Pairwise Hasse diagram distance compares interaction architectures across samples, conditions, or organisms.
139
+
140
+ ## Key functions
141
+
142
+ | Function | What it does |
143
+ |----------|-------------|
144
+ | `create_pbp(matrix)` | Full PBP decomposition -> DataFrame with monomial index, coefficient, degree |
145
+ | `pbp_vector(matrix)` | Fixed-length vector (2^m - 1 coefficients) for machine learning |
146
+ | `pbp_dag_sample(pbp, m)` | Greedy/stochastic/beam sequential inference through Hasse diagram |
147
+ | `evaluate_sequence(pred, gt)` | Spearman rho, Kendall tau, top-k accuracy, NDE |
148
+ | `pbp_distance(adata, groupby)` | Pairwise structural distance (Pertpy-compatible) |
149
+ | `pbp_score(adata, groupby)` | Spectral profile + interaction fraction per group |
150
+ | `pbp_classify_type(adata, groupby)` | Synergistic/antagonistic/additive classification |
151
+
152
+ ## Mathematical identity
153
+
154
+ PBP decomposition is mathematically identical to:
155
+ - **Walsh-Hadamard spectral transform** (Fourier analysis on {0,1}^m)
156
+ - **Epistatic interaction coefficients** (Poelwijk et al. 2019)
157
+ - **Multilinear extension** of pseudo-Boolean functions (Hammer & Rudeanu 1968, Boros & Hammer 2002)
158
+
159
+ ## Citation
160
+
161
+ ```bibtex
162
+ @article{chikake2025compoptics,
163
+ author = {Chikake, Tendai M. and Goldengorin, Boris I. and Pardalos, Panos M.},
164
+ title = {Pseudo-Boolean Polynomial Approach to Solving Computer Vision Tasks},
165
+ journal = {Computer Optics},
166
+ volume = {49},
167
+ number = {6},
168
+ pages = {1191--1201},
169
+ year = {2025},
170
+ doi = {10.18287/COJ1815},
171
+ }
172
+ ```
173
+
174
+ ## License
175
+
176
+ MIT
@@ -0,0 +1,147 @@
1
+ # tmc-pbp
2
+
3
+ Pseudo-Boolean Polynomial (PBP) decomposition for data analysis. Training-free, deterministic algebraic decomposition of data matrices into multilinear polynomials over binary variables.
4
+
5
+ ## What it does
6
+
7
+ Given a data matrix (rows = variables, columns = observations), PBP decomposes it into a multilinear polynomial where each term represents a specific combination of variables and its coefficient quantifies the interaction strength. The decomposition is:
8
+
9
+ - **Training-free** -- no learned parameters, no optimization
10
+ - **Deterministic** -- same input always produces the same output
11
+ - **Fast** -- sub-millisecond per sample, 15K genes in 1.7 seconds
12
+ - **Interpretable** -- each coefficient names a specific variable interaction
13
+ - **Mathematically equivalent** to the Walsh-Hadamard spectral transform
14
+
15
+ ## Install
16
+
17
+ ```bash
18
+ pip install tmc-pbp # core (numpy, pandas, scipy, bitarray, scikit-learn)
19
+ pip install "tmc-pbp[all]" # plus matplotlib, OpenCV, Pillow and Flask for the viz, image and server modules
20
+ ```
21
+
22
+ The package is imported as `pbp`. A pure-Python core is always available (`pbp.core`).
23
+ The optional C backend (`pbp.core_c`) is shipped as source; compile it once in the installed
24
+ package directory to enable it:
25
+
26
+ ```bash
27
+ bash "$(python -c 'import pbp, os; print(os.path.dirname(pbp.__file__))')/build_pbp.sh"
28
+ ```
29
+
30
+ ## Quick start
31
+
32
+ ### Python API
33
+
34
+ ```python
35
+ from pbp.core import create_pbp, pbp_vector
36
+
37
+ import numpy as np
38
+ matrix = np.array([
39
+ [5.2, 4.8, 3.1, 2.0], # Gene A
40
+ [3.0, 3.5, 4.2, 4.8], # Gene B
41
+ [1.0, 1.5, 2.8, 4.5], # Gene C
42
+ ])
43
+
44
+ # Full decomposition -> DataFrame with (y, coeffs, degree)
45
+ pbp = create_pbp(matrix)
46
+
47
+ # Fixed-length vector representation (2^m - 1 elements)
48
+ vec = pbp_vector(matrix)
49
+ ```
50
+
51
+ ### CLI
52
+
53
+ ```bash
54
+ pbp analyze matrix.csv --output results/ # Full decomposition + energy profile
55
+ pbp vector matrix.csv # Fixed-length PBP vector
56
+ pbp hasse matrix.csv --format dot # Hasse diagram (Graphviz DOT)
57
+ pbp info matrix.csv # Matrix stats
58
+ pbp version # Package version
59
+ ```
60
+
61
+ ### Web application
62
+
63
+ ```bash
64
+ bash pbp/webapp/run.sh
65
+ # Open http://localhost:8430
66
+ ```
67
+
68
+ 6-tab interface covering decomposition, epistasis, quorum sensing, sequential inference, structural comparison, and anomaly detection. 25 API endpoints. OpenAPI docs at `/docs`.
69
+
70
+ ## Modules
71
+
72
+ | Module | Purpose |
73
+ |--------|---------|
74
+ | `core.py` | Canonical PBP decomposition. `create_pbp()`, `pbp_vector()`, permutation/coefficient/variable matrices. Bitarray-based, no size limit. |
75
+ | `inference.py` | PBP-DAG sequential inference. Autoregressive sampling through the Hasse diagram using PBP coefficients as Boltzmann energies. Greedy, stochastic, and beam search modes. |
76
+ | `evaluate.py` | Evaluation metrics (Spearman, Kendall, NDE), bootstrap CIs, and baseline predictors (expression magnitude, PCA order, random). |
77
+ | `pertpy_integration.py` | Pertpy/scverse-compatible functions: `pbp_distance()`, `pbp_score()`, `pbp_classify_type()`. Works with AnnData objects or plain numpy arrays. |
78
+ | `cli.py` | Command-line interface. `pbp analyze\|vector\|hasse\|info\|version`. |
79
+ | `pipeline.py` | High-level pipeline utilities. |
80
+ | `server.py` | Legacy Flask server for DAG inference visualization. |
81
+ | `webapp/` | FastAPI web application (25 endpoints, single-page frontend with Chart.js + D3.js). |
82
+
83
+ ## Applications
84
+
85
+ ### Genetic epistasis (Perturb-seq)
86
+ PBP spectral profiles characterize interaction structure in combinatorial perturbation screens. Degree-wise energy separates interaction type from magnitude. Validated on 4 datasets from 4 labs (Norman, Wessels, Dixit, Joung).
87
+
88
+ ```python
89
+ from pbp.pertpy_integration import pbp_score, pbp_classify_type
90
+ scores = pbp_score(adata, groupby='perturbation', reference='control')
91
+ ```
92
+
93
+ ### Quorum sensing signal integration
94
+ Decompose factorial QS experiments into main effects (a1, a2) and interaction (a12) per gene. Gate classification: AND, OR, antagonistic, mixed.
95
+
96
+ ### Sequential inference
97
+ Predict activation order from expression matrices using greedy sampling through the PBP Hasse diagram.
98
+
99
+ ```python
100
+ from pbp.inference import pbp_dag_sample
101
+ result = pbp_dag_sample(pbp_df, m, mode='greedy')
102
+ print(result['sequence']) # Predicted activation order
103
+ ```
104
+
105
+ ### Anomaly detection
106
+ PBP total energy flags samples with disrupted interaction structure. Interpretable: identifies which specific interactions are affected.
107
+
108
+ ### Structural comparison
109
+ Pairwise Hasse diagram distance compares interaction architectures across samples, conditions, or organisms.
110
+
111
+ ## Key functions
112
+
113
+ | Function | What it does |
114
+ |----------|-------------|
115
+ | `create_pbp(matrix)` | Full PBP decomposition -> DataFrame with monomial index, coefficient, degree |
116
+ | `pbp_vector(matrix)` | Fixed-length vector (2^m - 1 coefficients) for machine learning |
117
+ | `pbp_dag_sample(pbp, m)` | Greedy/stochastic/beam sequential inference through Hasse diagram |
118
+ | `evaluate_sequence(pred, gt)` | Spearman rho, Kendall tau, top-k accuracy, NDE |
119
+ | `pbp_distance(adata, groupby)` | Pairwise structural distance (Pertpy-compatible) |
120
+ | `pbp_score(adata, groupby)` | Spectral profile + interaction fraction per group |
121
+ | `pbp_classify_type(adata, groupby)` | Synergistic/antagonistic/additive classification |
122
+
123
+ ## Mathematical identity
124
+
125
+ PBP decomposition is mathematically identical to:
126
+ - **Walsh-Hadamard spectral transform** (Fourier analysis on {0,1}^m)
127
+ - **Epistatic interaction coefficients** (Poelwijk et al. 2019)
128
+ - **Multilinear extension** of pseudo-Boolean functions (Hammer & Rudeanu 1968, Boros & Hammer 2002)
129
+
130
+ ## Citation
131
+
132
+ ```bibtex
133
+ @article{chikake2025compoptics,
134
+ author = {Chikake, Tendai M. and Goldengorin, Boris I. and Pardalos, Panos M.},
135
+ title = {Pseudo-Boolean Polynomial Approach to Solving Computer Vision Tasks},
136
+ journal = {Computer Optics},
137
+ volume = {49},
138
+ number = {6},
139
+ pages = {1191--1201},
140
+ year = {2025},
141
+ doi = {10.18287/COJ1815},
142
+ }
143
+ ```
144
+
145
+ ## License
146
+
147
+ MIT
@@ -0,0 +1,24 @@
1
+ """
2
+ pbp-polynomials: Pseudo-Boolean Polynomial decomposition for data analysis.
3
+
4
+ Training-free, deterministic algebraic decomposition of data matrices into
5
+ multilinear polynomials over binary variables. Applications: genetic epistasis
6
+ characterization, quorum sensing signal decomposition, sequential inference,
7
+ anomaly detection.
8
+
9
+ Usage:
10
+ from pbp.core import create_pbp, pbp_vector
11
+ from pbp.inference import pbp_dag_sample
12
+ from pbp.evaluate import evaluate_sequence
13
+ from pbp.pertpy_integration import pbp_distance, pbp_score, pbp_classify_type
14
+
15
+ CLI:
16
+ pbp analyze matrix.csv --output results/
17
+ pbp vector matrix.csv
18
+ pbp hasse matrix.csv --output diagram.json
19
+ """
20
+
21
+ __version__ = "0.1.0"
22
+ __author__ = "Tendai Chikake"
23
+
24
+ from pbp.core import create_pbp, pbp_vector
@@ -0,0 +1,28 @@
1
+ #!/bin/bash
2
+ # Build PBP shared libraries (core + codec)
3
+ # Usage: bash pbp/build_pbp.sh
4
+
5
+ set -e
6
+ cd "$(dirname "$0")"
7
+
8
+ CC="${CC:-cc}"
9
+ CFLAGS="-O2 -Wall -Wextra -std=c99 -fPIC"
10
+
11
+ case "$(uname -s)" in
12
+ Darwin)
13
+ SOEXT="dylib"
14
+ SOFLAGS="-dynamiclib"
15
+ ;;
16
+ *)
17
+ SOEXT="so"
18
+ SOFLAGS="-shared"
19
+ ;;
20
+ esac
21
+
22
+ echo "Compiling pbp_core.c -> libpbp_core.${SOEXT}"
23
+ $CC $CFLAGS $SOFLAGS -o "libpbp_core.${SOEXT}" pbp_core.c -lm
24
+ echo "Done: pbp/libpbp_core.${SOEXT}"
25
+
26
+ echo "Compiling pbp_codec.c -> libpbp_codec.${SOEXT}"
27
+ $CC $CFLAGS $SOFLAGS -o "libpbp_codec.${SOEXT}" pbp_codec.c -lm
28
+ echo "Done: pbp/libpbp_codec.${SOEXT}"
tmc_pbp-0.1.0/cli.py ADDED
@@ -0,0 +1,268 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ pbp — Command-line interface for pseudo-Boolean polynomial analysis.
4
+
5
+ Usage:
6
+ pbp analyze matrix.csv [--output DIR] [--format csv|json]
7
+ pbp vector matrix.csv [--output FILE]
8
+ pbp hasse matrix.csv [--output FILE] [--format json|dot]
9
+ pbp info matrix.csv
10
+ pbp version
11
+ """
12
+
13
+ import argparse
14
+ import sys
15
+ import os
16
+ import io
17
+ import json
18
+ import time
19
+
20
+ import numpy as np
21
+ import pandas as pd
22
+
23
+
24
+ def _quiet_pbp(matrix, **kwargs):
25
+ """Run create_pbp without stdout noise."""
26
+ from pbp.core import create_pbp
27
+ old = sys.stdout
28
+ sys.stdout = io.StringIO()
29
+ try:
30
+ return create_pbp(matrix, **kwargs)
31
+ finally:
32
+ sys.stdout = old
33
+
34
+
35
+ def _load_matrix(path):
36
+ """Load a matrix from CSV/TSV. Rows = variables, columns = conditions."""
37
+ if path.endswith('.tsv') or path.endswith('.txt'):
38
+ df = pd.read_csv(path, sep='\t')
39
+ else:
40
+ df = pd.read_csv(path)
41
+
42
+ # If first column is non-numeric (labels), use as index
43
+ if df.iloc[:, 0].dtype == object:
44
+ df = df.set_index(df.columns[0])
45
+
46
+ return df.values.astype(float), df.index.tolist(), df.columns.tolist()
47
+
48
+
49
+ def cmd_analyze(args):
50
+ """Full PBP analysis: decomposition + spectral profile + summary."""
51
+ matrix, row_labels, col_labels = _load_matrix(args.input)
52
+ m, n = matrix.shape
53
+
54
+ print(f"Matrix: {m} variables x {n} conditions")
55
+ print(f"Variables: {row_labels[:5]}{'...' if m > 5 else ''}")
56
+
57
+ t0 = time.time()
58
+ pbp = _quiet_pbp(matrix)
59
+ elapsed = time.time() - t0
60
+
61
+ print(f"PBP computed in {elapsed*1000:.1f} ms")
62
+ print(f" Monomials: {len(pbp)}")
63
+ print(f" Max degree: {int(pbp['degree'].max())}")
64
+ print(f" Non-zero coefficients: {(pbp['coeffs'] != 0).sum()}")
65
+
66
+ # Degree-wise energy
67
+ energy = pbp.groupby('degree')['coeffs'].apply(lambda x: (x**2).sum())
68
+ total_energy = energy.sum()
69
+ print(f"\n Degree-wise energy:")
70
+ for d, e in energy.items():
71
+ pct = 100 * e / total_energy if total_energy > 0 else 0
72
+ print(f" d={int(d)}: {e:.4f} ({pct:.1f}%)")
73
+
74
+ # PBP vector
75
+ from pbp.core import pbp_vector
76
+ vec = pbp_vector(matrix)
77
+ print(f"\n PBP vector: {len(vec)} elements")
78
+
79
+ # Save output
80
+ outdir = args.output or '.'
81
+ os.makedirs(outdir, exist_ok=True)
82
+
83
+ if args.format == 'json':
84
+ result = {
85
+ 'matrix_shape': [m, n],
86
+ 'row_labels': row_labels,
87
+ 'col_labels': col_labels,
88
+ 'n_monomials': len(pbp),
89
+ 'max_degree': int(pbp['degree'].max()),
90
+ 'elapsed_ms': round(elapsed * 1000, 1),
91
+ 'degree_energy': {int(k): round(float(v), 6) for k, v in energy.items()},
92
+ 'total_energy': round(float(total_energy), 6),
93
+ 'pbp_vector': vec.tolist(),
94
+ }
95
+ outpath = os.path.join(outdir, 'pbp_analysis.json')
96
+ with open(outpath, 'w') as f:
97
+ json.dump(result, f, indent=2)
98
+ else:
99
+ # Save decomposition
100
+ outpath = os.path.join(outdir, 'pbp_decomposition.csv')
101
+ pbp.to_csv(outpath, index=False)
102
+
103
+ # Save vector
104
+ vecpath = os.path.join(outdir, 'pbp_vector.csv')
105
+ pd.DataFrame({'index': range(len(vec)), 'coefficient': vec}).to_csv(vecpath, index=False)
106
+
107
+ # Save energy
108
+ epath = os.path.join(outdir, 'pbp_energy.csv')
109
+ energy_df = pd.DataFrame({
110
+ 'degree': energy.index.astype(int),
111
+ 'energy': energy.values,
112
+ 'fraction': energy.values / total_energy if total_energy > 0 else 0,
113
+ })
114
+ energy_df.to_csv(epath, index=False)
115
+
116
+ print(f"\n Output saved to {outdir}/")
117
+
118
+
119
+ def cmd_vector(args):
120
+ """Compute and output the PBP vector (fixed-length representation)."""
121
+ matrix, row_labels, _ = _load_matrix(args.input)
122
+ from pbp.core import pbp_vector
123
+ vec = pbp_vector(matrix)
124
+
125
+ if args.output:
126
+ pd.DataFrame({'index': range(len(vec)), 'coefficient': vec}).to_csv(args.output, index=False)
127
+ print(f"PBP vector ({len(vec)} elements) saved to {args.output}")
128
+ else:
129
+ for i, v in enumerate(vec):
130
+ if v != 0:
131
+ print(f" [{i}] = {v:.6f}")
132
+ print(f"\n {len(vec)} elements, {(vec != 0).sum()} non-zero")
133
+
134
+
135
+ def cmd_hasse(args):
136
+ """Export Hasse diagram structure."""
137
+ matrix, row_labels, _ = _load_matrix(args.input)
138
+ m = matrix.shape[0]
139
+
140
+ pbp = _quiet_pbp(matrix)
141
+
142
+ # Build Hasse structure
143
+ nodes = []
144
+ for _, row in pbp.iterrows():
145
+ y = int(row['y'])
146
+ d = int(row['degree'])
147
+ c = float(row['coeffs'])
148
+ bits = []
149
+ val = y
150
+ i = 0
151
+ while val:
152
+ if val & 1:
153
+ bits.append(i)
154
+ val >>= 1
155
+ i += 1
156
+ label = ','.join(str(b) for b in bits) if bits else '∅'
157
+ if row_labels and all(isinstance(r, str) for r in row_labels):
158
+ named = ','.join(row_labels[b] for b in bits) if bits else '∅'
159
+ else:
160
+ named = label
161
+ nodes.append({
162
+ 'id': y, 'degree': d, 'coefficient': c,
163
+ 'variables': bits, 'label': label, 'named': named,
164
+ })
165
+
166
+ # Build edges (subset relations between adjacent degrees)
167
+ edges = []
168
+ by_degree = {}
169
+ for node in nodes:
170
+ by_degree.setdefault(node['degree'], []).append(node)
171
+
172
+ max_d = max(by_degree.keys()) if by_degree else 0
173
+ for d in range(max_d):
174
+ for lo in by_degree.get(d, []):
175
+ for hi in by_degree.get(d + 1, []):
176
+ if (lo['id'] & hi['id']) == lo['id']:
177
+ edges.append({'from': lo['id'], 'to': hi['id']})
178
+
179
+ result = {'nodes': nodes, 'edges': edges, 'm': m, 'labels': row_labels}
180
+
181
+ if args.format == 'dot':
182
+ # Graphviz DOT format
183
+ lines = ['digraph Hasse {', ' rankdir=TB;']
184
+ for node in nodes:
185
+ color = '#D55E00' if node['coefficient'] > 0 else '#0072B2'
186
+ lines.append(f' n{node["id"]} [label="{node["named"]}\\n{node["coefficient"]:.2f}" '
187
+ f'style=filled fillcolor="{color}40"];')
188
+ for edge in edges:
189
+ lines.append(f' n{edge["from"]} -> n{edge["to"]};')
190
+ lines.append('}')
191
+ out = '\n'.join(lines)
192
+ else:
193
+ out = json.dumps(result, indent=2)
194
+
195
+ if args.output:
196
+ with open(args.output, 'w') as f:
197
+ f.write(out)
198
+ print(f"Hasse diagram ({len(nodes)} nodes, {len(edges)} edges) saved to {args.output}")
199
+ else:
200
+ print(out)
201
+
202
+
203
+ def cmd_info(args):
204
+ """Quick matrix info without full decomposition."""
205
+ matrix, row_labels, col_labels = _load_matrix(args.input)
206
+ m, n = matrix.shape
207
+ print(f"Matrix: {m} variables x {n} conditions")
208
+ print(f"Variables: {row_labels}")
209
+ print(f"Conditions: {col_labels}")
210
+ print(f"PBP vector length: {2**m - 1}")
211
+ print(f"Max monomials: {2**m}")
212
+ print(f"Value range: [{matrix.min():.4f}, {matrix.max():.4f}]")
213
+ print(f"Mean: {matrix.mean():.4f}, Std: {matrix.std():.4f}")
214
+ if m > 20:
215
+ print(f"WARNING: m={m} will produce {2**m:,} monomials. Consider reshaping.")
216
+
217
+
218
+ def cmd_version(args):
219
+ from pbp import __version__
220
+ print(f"pbp-polynomials {__version__}")
221
+
222
+
223
+ def main():
224
+ parser = argparse.ArgumentParser(
225
+ prog='pbp',
226
+ description='Pseudo-Boolean Polynomial analysis of data matrices',
227
+ )
228
+ subparsers = parser.add_subparsers(dest='command', help='Available commands')
229
+
230
+ # analyze
231
+ p_analyze = subparsers.add_parser('analyze', help='Full PBP analysis')
232
+ p_analyze.add_argument('input', help='Input matrix (CSV/TSV)')
233
+ p_analyze.add_argument('--output', '-o', help='Output directory')
234
+ p_analyze.add_argument('--format', choices=['csv', 'json'], default='csv')
235
+ p_analyze.set_defaults(func=cmd_analyze)
236
+
237
+ # vector
238
+ p_vector = subparsers.add_parser('vector', help='Compute PBP vector')
239
+ p_vector.add_argument('input', help='Input matrix (CSV/TSV)')
240
+ p_vector.add_argument('--output', '-o', help='Output CSV file')
241
+ p_vector.set_defaults(func=cmd_vector)
242
+
243
+ # hasse
244
+ p_hasse = subparsers.add_parser('hasse', help='Export Hasse diagram')
245
+ p_hasse.add_argument('input', help='Input matrix (CSV/TSV)')
246
+ p_hasse.add_argument('--output', '-o', help='Output file')
247
+ p_hasse.add_argument('--format', choices=['json', 'dot'], default='json')
248
+ p_hasse.set_defaults(func=cmd_hasse)
249
+
250
+ # info
251
+ p_info = subparsers.add_parser('info', help='Matrix info')
252
+ p_info.add_argument('input', help='Input matrix (CSV/TSV)')
253
+ p_info.set_defaults(func=cmd_info)
254
+
255
+ # version
256
+ p_version = subparsers.add_parser('version', help='Show version')
257
+ p_version.set_defaults(func=cmd_version)
258
+
259
+ args = parser.parse_args()
260
+ if not args.command:
261
+ parser.print_help()
262
+ sys.exit(1)
263
+
264
+ args.func(args)
265
+
266
+
267
+ if __name__ == '__main__':
268
+ main()