algebrax 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
algebrax/__init__.py ADDED
@@ -0,0 +1,257 @@
1
+ """
2
+ The `algebrax` namespace provides a comprehensive suite of mathematical operations
3
+ optimized for sparse, dictionary-based data structures.
4
+
5
+ It treats Python's native `dict` (and `Mapping`) as a first-class mathematical object,
6
+ enabling Linear Algebra, Set Theory, Graph Theory, and Probability operations directly on
7
+ sparse data without conversion to dense arrays.
8
+
9
+ Comparison with Other Libraries
10
+ -------------------------------
11
+
12
+ 1. **`scipy.sparse`**:
13
+ - **Domain**: Numerical Linear Algebra.
14
+ - **Pros**: Industry standard, extremely fast (C/Fortran backend).
15
+ - **Cons**: Keys must be integers; requires conversion from dicts; heavy dependency.
16
+ - **Use Case**: Large-scale numerical simulations (e.g., Finite Element Method).
17
+
18
+ 2. **`numpy`**:
19
+ - **Domain**: Dense Numerical Arrays.
20
+ - **Pros**: Universal standard for dense data.
21
+ - **Cons**: Inefficient for sparse data (O(N^2) memory); integer indices only.
22
+ - **Use Case**: Image processing, dense tensors.
23
+
24
+ 3. **`pandas`**:
25
+ - **Domain**: Tabular Data Analysis.
26
+ - **Pros**: Excellent for time-series and labeled data.
27
+ - **Cons**: Not optimized for general mathematical algebra (e.g., matrix multiplication).
28
+ - **Use Case**: Data cleaning, ETL, statistical analysis.
29
+
30
+ 4. **`algebrax` (This Library)**:
31
+ - **Domain**: Symbolic/Sparse Algebra on Mappings.
32
+ - **Pros**:
33
+ - **Symbolic Keys**: Works with `str`, `tuple`, or any hashable object (e.g., graphs with string nodes).
34
+ - **Zero-Dependency**: Pure Python.
35
+ - **Functional**: Composable API (`combine`, `compose`).
36
+ - **Cons**: Slower than C-based libraries for massive numerical computations.
37
+ - **Use Case**: NLP (word vectors), Knowledge Graphs, Item-Item similarity, small-to-medium sparse matrices.
38
+
39
+ Definitions & Criteria
40
+ ----------------------
41
+
42
+ * **Sparse**: Data where the number of non-zero elements ($k$) is significantly smaller than the total capacity ($N$).
43
+ * *Criterion*: Density ($k/N$) < 0.05 (5%).
44
+ * **Dense**: Data where most elements are non-zero.
45
+ * *Criterion*: Density > 0.5 (50%).
46
+ * **Lightweight**: Minimal memory overhead and startup time.
47
+ * *Criterion*: Import time < 10ms; Memory overhead < 1KB per object (beyond data).
48
+ * **Symbolic**: Keys represent semantic entities (e.g., "User_123", "Product_X") rather than contiguous memory
49
+ offsets (0, 1, 2).
50
+
51
+ Modules
52
+ -------
53
+
54
+ * **`matrix`**: Linear Algebra (Core & Academic).
55
+ * **`lattice`**: Set/Fuzzy Logic (Union, Intersection).
56
+ * **`analysis`**: Vector Calculus on Graphs (Gradient, Laplacian).
57
+ * **`probability`**: Bayesian/Markov Inference.
58
+ * **`transforms`**: Signal Processing (DFT, Convolution).
59
+ * **`automata`**: Finite State Machines.
60
+ * **`group`**: Permutations.
61
+ * **`sparsity`**: Metrics and checks.
62
+ * **`semiring`**: Generalized algebra (Tropical, Boolean, String).
63
+ * **`trie`**: Algebraic Tries (Sparse Tensors).
64
+ * **`typing`**: Type aliases for sparse/dense structures.
65
+ """
66
+
67
+ from algebrax.analysis import (
68
+ divergence,
69
+ gaussian_kernel,
70
+ gradient,
71
+ laplacian,
72
+ ollivier_ricci_curvature,
73
+ )
74
+ from algebrax.automata import (
75
+ dfa_step,
76
+ nfa_step,
77
+ simulate_dfa,
78
+ simulate_nfa,
79
+ )
80
+ from algebrax.converters import (
81
+ dense_to_sparse_matrix,
82
+ dense_to_sparse_vector,
83
+ sparse_to_dense_matrix,
84
+ sparse_to_dense_vector,
85
+ )
86
+ from algebrax.group import compose, invert, signature
87
+ from algebrax.lattice import (
88
+ average,
89
+ combine,
90
+ difference,
91
+ exclude,
92
+ exclusive,
93
+ geometric_mean,
94
+ harmonic_mean,
95
+ join,
96
+ mask,
97
+ meet,
98
+ product,
99
+ ratio,
100
+ symmetric_difference,
101
+ )
102
+ from algebrax.matrix import (
103
+ add,
104
+ adjoint,
105
+ cofactor,
106
+ determinant,
107
+ dot,
108
+ eigen_centrality,
109
+ inner,
110
+ inverse,
111
+ kronecker_delta,
112
+ mat_vec,
113
+ power,
114
+ trace,
115
+ transpose,
116
+ vec_mat,
117
+ )
118
+ from algebrax.probability import (
119
+ bayes_update,
120
+ cross_entropy,
121
+ entropy,
122
+ expected_value,
123
+ kl_divergence,
124
+ kurtosis,
125
+ marginalize,
126
+ markov_steady_state,
127
+ markov_step,
128
+ mode,
129
+ mutual_information,
130
+ normalize,
131
+ skewness,
132
+ variance,
133
+ )
134
+ from algebrax.semiring import (
135
+ BooleanSemiring,
136
+ BottleneckSemiring,
137
+ LogSemiring,
138
+ ReliabilitySemiring,
139
+ Semiring,
140
+ StandardSemiring,
141
+ StringSemiring,
142
+ TropicalSemiring,
143
+ ViterbiSemiring,
144
+ )
145
+ from algebrax.sparsity import (
146
+ deepness,
147
+ density,
148
+ is_sparse,
149
+ sparsity,
150
+ uniformness,
151
+ wideness,
152
+ )
153
+ from algebrax.transforms import (
154
+ box_counting_dimension,
155
+ convolve,
156
+ dft,
157
+ hilbert,
158
+ idft,
159
+ lorentz_boost,
160
+ z_transform,
161
+ )
162
+ from algebrax.trie import AlgebraicTrie
163
+ from algebrax.typing import (
164
+ DenseMatrix,
165
+ DenseVector,
166
+ SparseMatrix,
167
+ SparseTensor,
168
+ SparseVector,
169
+ )
170
+
171
+ __all__ = [
172
+ 'AlgebraicTrie',
173
+ 'BooleanSemiring',
174
+ 'BottleneckSemiring',
175
+ 'DenseMatrix',
176
+ 'DenseVector',
177
+ 'LogSemiring',
178
+ 'ReliabilitySemiring',
179
+ 'Semiring',
180
+ 'SparseMatrix',
181
+ 'SparseTensor',
182
+ 'SparseVector',
183
+ 'StandardSemiring',
184
+ 'StringSemiring',
185
+ 'TropicalSemiring',
186
+ 'ViterbiSemiring',
187
+ 'add',
188
+ 'adjoint',
189
+ 'average',
190
+ 'bayes_update',
191
+ 'box_counting_dimension',
192
+ 'cofactor',
193
+ 'combine',
194
+ 'compose',
195
+ 'convolve',
196
+ 'cross_entropy',
197
+ 'deepness',
198
+ 'dense_to_sparse_matrix',
199
+ 'dense_to_sparse_vector',
200
+ 'density',
201
+ 'determinant',
202
+ 'dfa_step',
203
+ 'dft',
204
+ 'difference',
205
+ 'divergence',
206
+ 'dot',
207
+ 'eigen_centrality',
208
+ 'entropy',
209
+ 'exclude',
210
+ 'exclusive',
211
+ 'expected_value',
212
+ 'gaussian_kernel',
213
+ 'geometric_mean',
214
+ 'gradient',
215
+ 'harmonic_mean',
216
+ 'hilbert',
217
+ 'idft',
218
+ 'inner',
219
+ 'inverse',
220
+ 'invert',
221
+ 'is_sparse',
222
+ 'join',
223
+ 'kl_divergence',
224
+ 'kronecker_delta',
225
+ 'kurtosis',
226
+ 'laplacian',
227
+ 'lorentz_boost',
228
+ 'marginalize',
229
+ 'markov_steady_state',
230
+ 'markov_step',
231
+ 'mask',
232
+ 'mat_vec',
233
+ 'meet',
234
+ 'mode',
235
+ 'mutual_information',
236
+ 'nfa_step',
237
+ 'normalize',
238
+ 'ollivier_ricci_curvature',
239
+ 'power',
240
+ 'product',
241
+ 'ratio',
242
+ 'signature',
243
+ 'simulate_dfa',
244
+ 'simulate_nfa',
245
+ 'skewness',
246
+ 'sparse_to_dense_matrix',
247
+ 'sparse_to_dense_vector',
248
+ 'sparsity',
249
+ 'symmetric_difference',
250
+ 'trace',
251
+ 'transpose',
252
+ 'uniformness',
253
+ 'variance',
254
+ 'vec_mat',
255
+ 'wideness',
256
+ 'z_transform',
257
+ ]
algebrax/analysis.py ADDED
@@ -0,0 +1,253 @@
1
+ import math
2
+ from collections import defaultdict
3
+ from collections.abc import Iterable, Mapping
4
+
5
+ from algebrax.semiring import Semiring, StandardSemiring
6
+ from algebrax.typing import K, N, SparseMatrix, SparseVector
7
+
8
+ __all__ = [
9
+ 'divergence',
10
+ 'fenchel_legendre_transform',
11
+ 'gaussian_kernel',
12
+ 'gradient',
13
+ 'laplacian',
14
+ 'ollivier_ricci_curvature',
15
+ ]
16
+
17
+
18
+ def divergence(flow: SparseMatrix) -> SparseVector:
19
+ """
20
+ Compute the discrete divergence of a 1-form (flow/edge signals).
21
+ Maps edges (matrix) to nodes (vector).
22
+ div(F)_i = sum_j (F_ij)
23
+
24
+ This corresponds to the adjoint of the gradient (d*).
25
+
26
+ Args:
27
+ flow: A matrix representing flow between nodes.
28
+ Positive F_ij implies flow from i to j.
29
+ (Note: Convention varies, sometimes it's net flow *out*).
30
+
31
+ Returns:
32
+ A vector representing the net flow out of each node.
33
+ """
34
+ result = defaultdict(int)
35
+ for u, neighbors in flow.items():
36
+ for v, val in neighbors.items():
37
+ # Flow u -> v counts as positive divergence for u
38
+ result[u] += val
39
+ # And negative divergence for v (if the matrix is not skew-symmetric stored)
40
+ # If the matrix is fully stored (both u->v and v->u), we just sum rows.
41
+ # If it's sparse/upper-triangular, we need to handle the other side.
42
+ # Let's assume the matrix represents the 1-form fully or we treat it as directed.
43
+ # Standard divergence is row_sum - col_sum?
44
+ # If F is skew-symmetric (F_ij = -F_ji), then row_sum is sufficient.
45
+ # If F is just weights, we usually define div at i as sum(w_ij) - sum(w_ji).
46
+ result[v] -= val
47
+
48
+ return dict(result)
49
+
50
+
51
+ def fenchel_legendre_transform(
52
+ signal: SparseVector[K, N],
53
+ slope: N,
54
+ semiring: Semiring[N] | None = None,
55
+ ) -> N:
56
+ """
57
+ Compute the discrete Fenchel-Legendre transform (Slope Transform) of a signal at a specific slope.
58
+ This is the Tropical/Idempotent analog of the Fourier Transform.
59
+
60
+ For Min-Plus Semiring (Tropical):
61
+ (F*)(s) = sup_x { <s, x> - F(x) }
62
+ But in Min-Plus algebra terms (where * is +, + is min):
63
+ This often corresponds to the convex conjugate.
64
+
65
+ In the context of "Tropical Fourier Transform" for a periodic signal f:
66
+ F(s) = min_x ( f(x) - s*x ) (or similar depending on convention)
67
+
68
+ Here we implement the standard convex conjugate definition:
69
+ f*(s) = sup_x ( s*x - f(x) )
70
+
71
+ Args:
72
+ signal: The input signal (mapping from index/position to value).
73
+ slope: The slope parameter (dual variable).
74
+ semiring: The algebraic structure.
75
+ If Tropical (Min-Plus), we use min/plus logic?
76
+ Actually, the Fenchel transform is usually defined over the standard ring for the values.
77
+ If we are strictly in Tropical Semiring, "integration" is min.
78
+
79
+ Returns:
80
+ The value of the transform at the given slope.
81
+ """
82
+ # Standard Fenchel-Legendre: max(s*x - f(x))
83
+ # This detects "slope content".
84
+
85
+ # If the signal is sparse, we iterate over defined points.
86
+ # We assume K (keys) are numeric (positions).
87
+
88
+ max_val = float('-inf')
89
+
90
+ for x, fx in signal.items():
91
+ # We assume x is numeric (time/space index)
92
+ if not isinstance(x, (int, float)):
93
+ continue
94
+
95
+ # val = s*x - f(x)
96
+ val = slope * x - fx
97
+ if val > max_val:
98
+ max_val = val
99
+
100
+ return max_val
101
+
102
+
103
+ def gaussian_kernel(distance_matrix: SparseMatrix, sigma: float = 1.0, threshold: float = 1e-6) -> SparseMatrix:
104
+ """
105
+ Compute the Gaussian (RBF) kernel from a distance matrix.
106
+ K_ij = exp(-d_ij^2 / (2 * sigma^2))
107
+
108
+ This transforms a distance metric into a similarity (adjacency) matrix,
109
+ often used for spectral clustering or diffusion maps.
110
+
111
+ Args:
112
+ distance_matrix: A sparse matrix of distances between nodes.
113
+ sigma: The bandwidth parameter (standard deviation).
114
+ threshold: Minimum value to retain in the sparse output.
115
+
116
+ Returns:
117
+ A sparse similarity matrix.
118
+ """
119
+ result = {}
120
+ denom = 2 * sigma * sigma
121
+
122
+ for u, neighbors in distance_matrix.items():
123
+ row = {}
124
+ for v, dist in neighbors.items():
125
+ val = math.exp(-(dist * dist) / denom)
126
+ if val > threshold:
127
+ row[v] = val
128
+ if row:
129
+ result[u] = row
130
+
131
+ return result
132
+
133
+
134
+ def gradient(field: SparseVector, graph: Mapping[K, Iterable[K]]) -> SparseMatrix:
135
+ """
136
+ Compute the discrete gradient (exterior derivative d0) of a 0-form (node signals).
137
+ Maps nodes (vector) to edges (matrix).
138
+ grad(f)_ij = f(j) - f(i)
139
+
140
+ Args:
141
+ field: A vector of values at nodes.
142
+ graph: Adjacency list defining the edges (topology).
143
+
144
+ Returns:
145
+ A matrix (1-form) representing the gradient along edges.
146
+ """
147
+ result = {}
148
+ for u, neighbors in graph.items():
149
+ if u not in field:
150
+ continue
151
+
152
+ val_u = field[u]
153
+ row = {}
154
+ for v in neighbors:
155
+ if v in field:
156
+ # d f(u, v) = f(v) - f(u)
157
+ row[v] = field[v] - val_u
158
+
159
+ if row:
160
+ result[u] = row
161
+ return result
162
+
163
+
164
+ def laplacian(field: SparseVector, graph: SparseMatrix) -> SparseVector:
165
+ """
166
+ Compute the combinatorial Laplacian of a scalar field.
167
+ L = D - A (for unweighted) or L f = div(grad f).
168
+
169
+ Delta f_i = sum_{j ~ i} w_ij * (f_i - f_j)
170
+
171
+ Args:
172
+ field: A vector of values at nodes.
173
+ graph: Adjacency matrix (weighted).
174
+
175
+ Returns:
176
+ A vector representing the Laplacian at each node.
177
+ """
178
+ # L = div(grad(f))
179
+ # But calculating grad then div is expensive (creates intermediate matrix).
180
+ # Direct calculation:
181
+ result = defaultdict(int)
182
+
183
+ for u, neighbors in graph.items():
184
+ if u not in field:
185
+ continue
186
+
187
+ val_u = field[u]
188
+ # Degree (weighted)
189
+ # For standard Laplacian, we sum w_ij * (f_u - f_v)
190
+
191
+ local_sum = 0
192
+ for v, weight in neighbors.items():
193
+ if v in field:
194
+ diff = val_u - field[v]
195
+ local_sum += weight * diff
196
+
197
+ if local_sum != 0:
198
+ result[u] = local_sum
199
+
200
+ return dict(result)
201
+
202
+
203
+ def ollivier_ricci_curvature(graph: SparseMatrix) -> dict[tuple[K, K], float]:
204
+ """
205
+ Compute the Ollivier-Ricci Curvature for edges in a graph.
206
+ Ric(xy) = 1 - W_1(m_x, m_y) / d(x, y)
207
+
208
+ Where W_1 is the Wasserstein distance (Earth Mover's Distance) between
209
+ probability measures m_x and m_y defined around nodes x and y.
210
+
211
+ Note: This is a simplified implementation approximating W_1 for sparse graphs.
212
+ Full computation requires a linear programming solver (e.g., scipy.optimize).
213
+ Here we use a greedy approximation or simple overlap for efficiency in pure Python.
214
+
215
+ Approximation: Jaccard-like overlap of neighborhoods.
216
+ Ric(xy) approx 1 - (1 - |N(x) n N(y)| / |N(x) u N(y)|) ... this is rough.
217
+
218
+ Let's implement a basic version based on "Forman-Ricci Curvature" which is
219
+ much faster and purely combinatorial (O(N) instead of O(N^3)).
220
+
221
+ Forman-Ricci Curvature for edge e=(u, v):
222
+ F(e) = 4 - deg(u) - deg(v)
223
+
224
+ Args:
225
+ graph: Adjacency matrix (weighted).
226
+
227
+ Returns:
228
+ A dictionary mapping edges (u, v) to their curvature values.
229
+ """
230
+ curvature = {}
231
+ degrees = {u: len(neighbors) for u, neighbors in graph.items()}
232
+
233
+ for u, neighbors in graph.items():
234
+ for v in neighbors:
235
+ if u < v: # Undirected edge, process once
236
+ # Forman-Ricci Curvature (Combinatorial)
237
+ # F(e) = 4 - deg(u) - deg(v)
238
+ # This is a very rough proxy for "manifold curvature".
239
+ # Negative values -> Hyperbolic (Tree-like, Expander)
240
+ # Positive values -> Spherical (Clique-like, Cluster)
241
+ # Zero -> Euclidean (Grid)
242
+
243
+ # Refined Forman (accounting for triangles/weights is better but complex)
244
+ # Let's stick to the basic combinatorial definition.
245
+ deg_u = degrees.get(u, 0)
246
+ deg_v = degrees.get(v, 0)
247
+
248
+ # Standard Forman formula for unweighted graphs
249
+ k = 4 - deg_u - deg_v
250
+
251
+ curvature[(u, v)] = k
252
+
253
+ return curvature
algebrax/automata.py ADDED
@@ -0,0 +1,131 @@
1
+ from collections import defaultdict
2
+ from collections.abc import Mapping
3
+
4
+ from algebrax.typing import A, S
5
+
6
+ __all__ = [
7
+ 'dfa_step',
8
+ 'nfa_step',
9
+ 'simulate_dfa',
10
+ 'simulate_nfa',
11
+ ]
12
+
13
+
14
+ def dfa_step(
15
+ current_state: S,
16
+ symbol: A,
17
+ transitions: Mapping[S, Mapping[A, S]],
18
+ ) -> S | None:
19
+ """
20
+ Perform a single step in a Deterministic Finite Automaton (DFA).
21
+ delta: S x A -> S
22
+
23
+ Args:
24
+ current_state: The current state.
25
+ symbol: The input symbol.
26
+ transitions: The transition function {state: {symbol: next_state}}.
27
+
28
+ Returns:
29
+ The next state, or None if the transition is undefined (implicit sink state).
30
+ """
31
+ if current_state in transitions:
32
+ return transitions[current_state].get(symbol)
33
+ return None
34
+
35
+
36
+ def nfa_step(
37
+ current_states: Mapping[S, float],
38
+ symbol: A,
39
+ transitions: Mapping[S, Mapping[A, Mapping[S, float]]],
40
+ ) -> dict[S, float]:
41
+ """
42
+ Perform a single step in a Nondeterministic Finite Automaton (NFA) or Probabilistic Automaton.
43
+ delta: S x A -> P(S)
44
+
45
+ This handles both standard NFAs (where values are 1.0) and Probabilistic Automata
46
+ (where values are probabilities).
47
+
48
+ Args:
49
+ current_states: A mapping of current states to their weights/probabilities.
50
+ symbol: The input symbol.
51
+ transitions: The transition function {state: {symbol: {next_state: weight}}}.
52
+
53
+ Returns:
54
+ A mapping of next states to their accumulated weights.
55
+ """
56
+ next_states = defaultdict(float)
57
+
58
+ for state, weight in current_states.items():
59
+ if state in transitions:
60
+ symbol_transitions = transitions[state].get(symbol)
61
+ if symbol_transitions:
62
+ for next_state, trans_weight in symbol_transitions.items():
63
+ next_states[next_state] += weight * trans_weight
64
+
65
+ return dict(next_states)
66
+
67
+
68
+ def simulate_dfa(
69
+ start_state: S,
70
+ input_sequence: Mapping[int, A],
71
+ transitions: Mapping[S, Mapping[A, S]],
72
+ ) -> S | None:
73
+ """
74
+ Simulate a DFA on an input sequence.
75
+
76
+ Args:
77
+ start_state: The initial state.
78
+ input_sequence: A mapping {time_step: symbol} or list of symbols.
79
+ transitions: The transition function.
80
+
81
+ Returns:
82
+ The final state, or None if the machine crashed.
83
+ """
84
+ current = start_state
85
+
86
+ # Handle both dicts (sparse sequence) and lists/iterables
87
+ if isinstance(input_sequence, Mapping):
88
+ # Sort by time index
89
+ steps = sorted(input_sequence.keys())
90
+ sequence = (input_sequence[k] for k in steps)
91
+ else:
92
+ sequence = input_sequence
93
+
94
+ for symbol in sequence:
95
+ current = dfa_step(current, symbol, transitions)
96
+ if current is None:
97
+ return None
98
+
99
+ return current
100
+
101
+
102
+ def simulate_nfa(
103
+ start_states: Mapping[S, float],
104
+ input_sequence: Mapping[int, A],
105
+ transitions: Mapping[S, Mapping[A, Mapping[S, float]]],
106
+ ) -> dict[S, float]:
107
+ """
108
+ Simulate an NFA or Probabilistic Automaton on an input sequence.
109
+
110
+ Args:
111
+ start_states: Initial distribution of states.
112
+ input_sequence: A mapping {time_step: symbol} or list of symbols.
113
+ transitions: The transition function.
114
+
115
+ Returns:
116
+ The final distribution of states.
117
+ """
118
+ current = start_states
119
+
120
+ if isinstance(input_sequence, Mapping):
121
+ steps = sorted(input_sequence.keys())
122
+ sequence = (input_sequence[k] for k in steps)
123
+ else:
124
+ sequence = input_sequence
125
+
126
+ for symbol in sequence:
127
+ current = nfa_step(current, symbol, transitions)
128
+ if not current:
129
+ break
130
+
131
+ return current