algebrax 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- algebrax/__init__.py +257 -0
- algebrax/analysis.py +253 -0
- algebrax/automata.py +131 -0
- algebrax/converters.py +392 -0
- algebrax/group.py +71 -0
- algebrax/lattice.py +409 -0
- algebrax/matrix/__init__.py +35 -0
- algebrax/matrix/academic.py +282 -0
- algebrax/matrix/core.py +443 -0
- algebrax/probability.py +383 -0
- algebrax/semiring.py +937 -0
- algebrax/sparsity.py +171 -0
- algebrax/transforms.py +347 -0
- algebrax/trie.py +147 -0
- algebrax/typing.py +72 -0
- algebrax-0.1.0.dist-info/METADATA +49 -0
- algebrax-0.1.0.dist-info/RECORD +19 -0
- algebrax-0.1.0.dist-info/WHEEL +4 -0
- algebrax-0.1.0.dist-info/licenses/LICENSE +21 -0
algebrax/__init__.py
ADDED
|
@@ -0,0 +1,257 @@
|
|
|
1
|
+
"""
|
|
2
|
+
The `algebrax` namespace provides a comprehensive suite of mathematical operations
|
|
3
|
+
optimized for sparse, dictionary-based data structures.
|
|
4
|
+
|
|
5
|
+
It treats Python's native `dict` (and `Mapping`) as a first-class mathematical object,
|
|
6
|
+
enabling Linear Algebra, Set Theory, Graph Theory, and Probability operations directly on
|
|
7
|
+
sparse data without conversion to dense arrays.
|
|
8
|
+
|
|
9
|
+
Comparison with Other Libraries
|
|
10
|
+
-------------------------------
|
|
11
|
+
|
|
12
|
+
1. **`scipy.sparse`**:
|
|
13
|
+
- **Domain**: Numerical Linear Algebra.
|
|
14
|
+
- **Pros**: Industry standard, extremely fast (C/Fortran backend).
|
|
15
|
+
- **Cons**: Keys must be integers; requires conversion from dicts; heavy dependency.
|
|
16
|
+
- **Use Case**: Large-scale numerical simulations (e.g., Finite Element Method).
|
|
17
|
+
|
|
18
|
+
2. **`numpy`**:
|
|
19
|
+
- **Domain**: Dense Numerical Arrays.
|
|
20
|
+
- **Pros**: Universal standard for dense data.
|
|
21
|
+
- **Cons**: Inefficient for sparse data (O(N^2) memory); integer indices only.
|
|
22
|
+
- **Use Case**: Image processing, dense tensors.
|
|
23
|
+
|
|
24
|
+
3. **`pandas`**:
|
|
25
|
+
- **Domain**: Tabular Data Analysis.
|
|
26
|
+
- **Pros**: Excellent for time-series and labeled data.
|
|
27
|
+
- **Cons**: Not optimized for general mathematical algebra (e.g., matrix multiplication).
|
|
28
|
+
- **Use Case**: Data cleaning, ETL, statistical analysis.
|
|
29
|
+
|
|
30
|
+
4. **`algebrax` (This Library)**:
|
|
31
|
+
- **Domain**: Symbolic/Sparse Algebra on Mappings.
|
|
32
|
+
- **Pros**:
|
|
33
|
+
- **Symbolic Keys**: Works with `str`, `tuple`, or any hashable object (e.g., graphs with string nodes).
|
|
34
|
+
- **Zero-Dependency**: Pure Python.
|
|
35
|
+
- **Functional**: Composable API (`combine`, `compose`).
|
|
36
|
+
- **Cons**: Slower than C-based libraries for massive numerical computations.
|
|
37
|
+
- **Use Case**: NLP (word vectors), Knowledge Graphs, Item-Item similarity, small-to-medium sparse matrices.
|
|
38
|
+
|
|
39
|
+
Definitions & Criteria
|
|
40
|
+
----------------------
|
|
41
|
+
|
|
42
|
+
* **Sparse**: Data where the number of non-zero elements ($k$) is significantly smaller than the total capacity ($N$).
|
|
43
|
+
* *Criterion*: Density ($k/N$) < 0.05 (5%).
|
|
44
|
+
* **Dense**: Data where most elements are non-zero.
|
|
45
|
+
* *Criterion*: Density > 0.5 (50%).
|
|
46
|
+
* **Lightweight**: Minimal memory overhead and startup time.
|
|
47
|
+
* *Criterion*: Import time < 10ms; Memory overhead < 1KB per object (beyond data).
|
|
48
|
+
* **Symbolic**: Keys represent semantic entities (e.g., "User_123", "Product_X") rather than contiguous memory
|
|
49
|
+
offsets (0, 1, 2).
|
|
50
|
+
|
|
51
|
+
Modules
|
|
52
|
+
-------
|
|
53
|
+
|
|
54
|
+
* **`matrix`**: Linear Algebra (Core & Academic).
|
|
55
|
+
* **`lattice`**: Set/Fuzzy Logic (Union, Intersection).
|
|
56
|
+
* **`analysis`**: Vector Calculus on Graphs (Gradient, Laplacian).
|
|
57
|
+
* **`probability`**: Bayesian/Markov Inference.
|
|
58
|
+
* **`transforms`**: Signal Processing (DFT, Convolution).
|
|
59
|
+
* **`automata`**: Finite State Machines.
|
|
60
|
+
* **`group`**: Permutations.
|
|
61
|
+
* **`sparsity`**: Metrics and checks.
|
|
62
|
+
* **`semiring`**: Generalized algebra (Tropical, Boolean, String).
|
|
63
|
+
* **`trie`**: Algebraic Tries (Sparse Tensors).
|
|
64
|
+
* **`typing`**: Type aliases for sparse/dense structures.
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
from algebrax.analysis import (
|
|
68
|
+
divergence,
|
|
69
|
+
gaussian_kernel,
|
|
70
|
+
gradient,
|
|
71
|
+
laplacian,
|
|
72
|
+
ollivier_ricci_curvature,
|
|
73
|
+
)
|
|
74
|
+
from algebrax.automata import (
|
|
75
|
+
dfa_step,
|
|
76
|
+
nfa_step,
|
|
77
|
+
simulate_dfa,
|
|
78
|
+
simulate_nfa,
|
|
79
|
+
)
|
|
80
|
+
from algebrax.converters import (
|
|
81
|
+
dense_to_sparse_matrix,
|
|
82
|
+
dense_to_sparse_vector,
|
|
83
|
+
sparse_to_dense_matrix,
|
|
84
|
+
sparse_to_dense_vector,
|
|
85
|
+
)
|
|
86
|
+
from algebrax.group import compose, invert, signature
|
|
87
|
+
from algebrax.lattice import (
|
|
88
|
+
average,
|
|
89
|
+
combine,
|
|
90
|
+
difference,
|
|
91
|
+
exclude,
|
|
92
|
+
exclusive,
|
|
93
|
+
geometric_mean,
|
|
94
|
+
harmonic_mean,
|
|
95
|
+
join,
|
|
96
|
+
mask,
|
|
97
|
+
meet,
|
|
98
|
+
product,
|
|
99
|
+
ratio,
|
|
100
|
+
symmetric_difference,
|
|
101
|
+
)
|
|
102
|
+
from algebrax.matrix import (
|
|
103
|
+
add,
|
|
104
|
+
adjoint,
|
|
105
|
+
cofactor,
|
|
106
|
+
determinant,
|
|
107
|
+
dot,
|
|
108
|
+
eigen_centrality,
|
|
109
|
+
inner,
|
|
110
|
+
inverse,
|
|
111
|
+
kronecker_delta,
|
|
112
|
+
mat_vec,
|
|
113
|
+
power,
|
|
114
|
+
trace,
|
|
115
|
+
transpose,
|
|
116
|
+
vec_mat,
|
|
117
|
+
)
|
|
118
|
+
from algebrax.probability import (
|
|
119
|
+
bayes_update,
|
|
120
|
+
cross_entropy,
|
|
121
|
+
entropy,
|
|
122
|
+
expected_value,
|
|
123
|
+
kl_divergence,
|
|
124
|
+
kurtosis,
|
|
125
|
+
marginalize,
|
|
126
|
+
markov_steady_state,
|
|
127
|
+
markov_step,
|
|
128
|
+
mode,
|
|
129
|
+
mutual_information,
|
|
130
|
+
normalize,
|
|
131
|
+
skewness,
|
|
132
|
+
variance,
|
|
133
|
+
)
|
|
134
|
+
from algebrax.semiring import (
|
|
135
|
+
BooleanSemiring,
|
|
136
|
+
BottleneckSemiring,
|
|
137
|
+
LogSemiring,
|
|
138
|
+
ReliabilitySemiring,
|
|
139
|
+
Semiring,
|
|
140
|
+
StandardSemiring,
|
|
141
|
+
StringSemiring,
|
|
142
|
+
TropicalSemiring,
|
|
143
|
+
ViterbiSemiring,
|
|
144
|
+
)
|
|
145
|
+
from algebrax.sparsity import (
|
|
146
|
+
deepness,
|
|
147
|
+
density,
|
|
148
|
+
is_sparse,
|
|
149
|
+
sparsity,
|
|
150
|
+
uniformness,
|
|
151
|
+
wideness,
|
|
152
|
+
)
|
|
153
|
+
from algebrax.transforms import (
|
|
154
|
+
box_counting_dimension,
|
|
155
|
+
convolve,
|
|
156
|
+
dft,
|
|
157
|
+
hilbert,
|
|
158
|
+
idft,
|
|
159
|
+
lorentz_boost,
|
|
160
|
+
z_transform,
|
|
161
|
+
)
|
|
162
|
+
from algebrax.trie import AlgebraicTrie
|
|
163
|
+
from algebrax.typing import (
|
|
164
|
+
DenseMatrix,
|
|
165
|
+
DenseVector,
|
|
166
|
+
SparseMatrix,
|
|
167
|
+
SparseTensor,
|
|
168
|
+
SparseVector,
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
__all__ = [
|
|
172
|
+
'AlgebraicTrie',
|
|
173
|
+
'BooleanSemiring',
|
|
174
|
+
'BottleneckSemiring',
|
|
175
|
+
'DenseMatrix',
|
|
176
|
+
'DenseVector',
|
|
177
|
+
'LogSemiring',
|
|
178
|
+
'ReliabilitySemiring',
|
|
179
|
+
'Semiring',
|
|
180
|
+
'SparseMatrix',
|
|
181
|
+
'SparseTensor',
|
|
182
|
+
'SparseVector',
|
|
183
|
+
'StandardSemiring',
|
|
184
|
+
'StringSemiring',
|
|
185
|
+
'TropicalSemiring',
|
|
186
|
+
'ViterbiSemiring',
|
|
187
|
+
'add',
|
|
188
|
+
'adjoint',
|
|
189
|
+
'average',
|
|
190
|
+
'bayes_update',
|
|
191
|
+
'box_counting_dimension',
|
|
192
|
+
'cofactor',
|
|
193
|
+
'combine',
|
|
194
|
+
'compose',
|
|
195
|
+
'convolve',
|
|
196
|
+
'cross_entropy',
|
|
197
|
+
'deepness',
|
|
198
|
+
'dense_to_sparse_matrix',
|
|
199
|
+
'dense_to_sparse_vector',
|
|
200
|
+
'density',
|
|
201
|
+
'determinant',
|
|
202
|
+
'dfa_step',
|
|
203
|
+
'dft',
|
|
204
|
+
'difference',
|
|
205
|
+
'divergence',
|
|
206
|
+
'dot',
|
|
207
|
+
'eigen_centrality',
|
|
208
|
+
'entropy',
|
|
209
|
+
'exclude',
|
|
210
|
+
'exclusive',
|
|
211
|
+
'expected_value',
|
|
212
|
+
'gaussian_kernel',
|
|
213
|
+
'geometric_mean',
|
|
214
|
+
'gradient',
|
|
215
|
+
'harmonic_mean',
|
|
216
|
+
'hilbert',
|
|
217
|
+
'idft',
|
|
218
|
+
'inner',
|
|
219
|
+
'inverse',
|
|
220
|
+
'invert',
|
|
221
|
+
'is_sparse',
|
|
222
|
+
'join',
|
|
223
|
+
'kl_divergence',
|
|
224
|
+
'kronecker_delta',
|
|
225
|
+
'kurtosis',
|
|
226
|
+
'laplacian',
|
|
227
|
+
'lorentz_boost',
|
|
228
|
+
'marginalize',
|
|
229
|
+
'markov_steady_state',
|
|
230
|
+
'markov_step',
|
|
231
|
+
'mask',
|
|
232
|
+
'mat_vec',
|
|
233
|
+
'meet',
|
|
234
|
+
'mode',
|
|
235
|
+
'mutual_information',
|
|
236
|
+
'nfa_step',
|
|
237
|
+
'normalize',
|
|
238
|
+
'ollivier_ricci_curvature',
|
|
239
|
+
'power',
|
|
240
|
+
'product',
|
|
241
|
+
'ratio',
|
|
242
|
+
'signature',
|
|
243
|
+
'simulate_dfa',
|
|
244
|
+
'simulate_nfa',
|
|
245
|
+
'skewness',
|
|
246
|
+
'sparse_to_dense_matrix',
|
|
247
|
+
'sparse_to_dense_vector',
|
|
248
|
+
'sparsity',
|
|
249
|
+
'symmetric_difference',
|
|
250
|
+
'trace',
|
|
251
|
+
'transpose',
|
|
252
|
+
'uniformness',
|
|
253
|
+
'variance',
|
|
254
|
+
'vec_mat',
|
|
255
|
+
'wideness',
|
|
256
|
+
'z_transform',
|
|
257
|
+
]
|
algebrax/analysis.py
ADDED
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
import math
|
|
2
|
+
from collections import defaultdict
|
|
3
|
+
from collections.abc import Iterable, Mapping
|
|
4
|
+
|
|
5
|
+
from algebrax.semiring import Semiring, StandardSemiring
|
|
6
|
+
from algebrax.typing import K, N, SparseMatrix, SparseVector
|
|
7
|
+
|
|
8
|
+
__all__ = [
|
|
9
|
+
'divergence',
|
|
10
|
+
'fenchel_legendre_transform',
|
|
11
|
+
'gaussian_kernel',
|
|
12
|
+
'gradient',
|
|
13
|
+
'laplacian',
|
|
14
|
+
'ollivier_ricci_curvature',
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def divergence(flow: SparseMatrix) -> SparseVector:
|
|
19
|
+
"""
|
|
20
|
+
Compute the discrete divergence of a 1-form (flow/edge signals).
|
|
21
|
+
Maps edges (matrix) to nodes (vector).
|
|
22
|
+
div(F)_i = sum_j (F_ij)
|
|
23
|
+
|
|
24
|
+
This corresponds to the adjoint of the gradient (d*).
|
|
25
|
+
|
|
26
|
+
Args:
|
|
27
|
+
flow: A matrix representing flow between nodes.
|
|
28
|
+
Positive F_ij implies flow from i to j.
|
|
29
|
+
(Note: Convention varies, sometimes it's net flow *out*).
|
|
30
|
+
|
|
31
|
+
Returns:
|
|
32
|
+
A vector representing the net flow out of each node.
|
|
33
|
+
"""
|
|
34
|
+
result = defaultdict(int)
|
|
35
|
+
for u, neighbors in flow.items():
|
|
36
|
+
for v, val in neighbors.items():
|
|
37
|
+
# Flow u -> v counts as positive divergence for u
|
|
38
|
+
result[u] += val
|
|
39
|
+
# And negative divergence for v (if the matrix is not skew-symmetric stored)
|
|
40
|
+
# If the matrix is fully stored (both u->v and v->u), we just sum rows.
|
|
41
|
+
# If it's sparse/upper-triangular, we need to handle the other side.
|
|
42
|
+
# Let's assume the matrix represents the 1-form fully or we treat it as directed.
|
|
43
|
+
# Standard divergence is row_sum - col_sum?
|
|
44
|
+
# If F is skew-symmetric (F_ij = -F_ji), then row_sum is sufficient.
|
|
45
|
+
# If F is just weights, we usually define div at i as sum(w_ij) - sum(w_ji).
|
|
46
|
+
result[v] -= val
|
|
47
|
+
|
|
48
|
+
return dict(result)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def fenchel_legendre_transform(
|
|
52
|
+
signal: SparseVector[K, N],
|
|
53
|
+
slope: N,
|
|
54
|
+
semiring: Semiring[N] | None = None,
|
|
55
|
+
) -> N:
|
|
56
|
+
"""
|
|
57
|
+
Compute the discrete Fenchel-Legendre transform (Slope Transform) of a signal at a specific slope.
|
|
58
|
+
This is the Tropical/Idempotent analog of the Fourier Transform.
|
|
59
|
+
|
|
60
|
+
For Min-Plus Semiring (Tropical):
|
|
61
|
+
(F*)(s) = sup_x { <s, x> - F(x) }
|
|
62
|
+
But in Min-Plus algebra terms (where * is +, + is min):
|
|
63
|
+
This often corresponds to the convex conjugate.
|
|
64
|
+
|
|
65
|
+
In the context of "Tropical Fourier Transform" for a periodic signal f:
|
|
66
|
+
F(s) = min_x ( f(x) - s*x ) (or similar depending on convention)
|
|
67
|
+
|
|
68
|
+
Here we implement the standard convex conjugate definition:
|
|
69
|
+
f*(s) = sup_x ( s*x - f(x) )
|
|
70
|
+
|
|
71
|
+
Args:
|
|
72
|
+
signal: The input signal (mapping from index/position to value).
|
|
73
|
+
slope: The slope parameter (dual variable).
|
|
74
|
+
semiring: The algebraic structure.
|
|
75
|
+
If Tropical (Min-Plus), we use min/plus logic?
|
|
76
|
+
Actually, the Fenchel transform is usually defined over the standard ring for the values.
|
|
77
|
+
If we are strictly in Tropical Semiring, "integration" is min.
|
|
78
|
+
|
|
79
|
+
Returns:
|
|
80
|
+
The value of the transform at the given slope.
|
|
81
|
+
"""
|
|
82
|
+
# Standard Fenchel-Legendre: max(s*x - f(x))
|
|
83
|
+
# This detects "slope content".
|
|
84
|
+
|
|
85
|
+
# If the signal is sparse, we iterate over defined points.
|
|
86
|
+
# We assume K (keys) are numeric (positions).
|
|
87
|
+
|
|
88
|
+
max_val = float('-inf')
|
|
89
|
+
|
|
90
|
+
for x, fx in signal.items():
|
|
91
|
+
# We assume x is numeric (time/space index)
|
|
92
|
+
if not isinstance(x, (int, float)):
|
|
93
|
+
continue
|
|
94
|
+
|
|
95
|
+
# val = s*x - f(x)
|
|
96
|
+
val = slope * x - fx
|
|
97
|
+
if val > max_val:
|
|
98
|
+
max_val = val
|
|
99
|
+
|
|
100
|
+
return max_val
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def gaussian_kernel(distance_matrix: SparseMatrix, sigma: float = 1.0, threshold: float = 1e-6) -> SparseMatrix:
|
|
104
|
+
"""
|
|
105
|
+
Compute the Gaussian (RBF) kernel from a distance matrix.
|
|
106
|
+
K_ij = exp(-d_ij^2 / (2 * sigma^2))
|
|
107
|
+
|
|
108
|
+
This transforms a distance metric into a similarity (adjacency) matrix,
|
|
109
|
+
often used for spectral clustering or diffusion maps.
|
|
110
|
+
|
|
111
|
+
Args:
|
|
112
|
+
distance_matrix: A sparse matrix of distances between nodes.
|
|
113
|
+
sigma: The bandwidth parameter (standard deviation).
|
|
114
|
+
threshold: Minimum value to retain in the sparse output.
|
|
115
|
+
|
|
116
|
+
Returns:
|
|
117
|
+
A sparse similarity matrix.
|
|
118
|
+
"""
|
|
119
|
+
result = {}
|
|
120
|
+
denom = 2 * sigma * sigma
|
|
121
|
+
|
|
122
|
+
for u, neighbors in distance_matrix.items():
|
|
123
|
+
row = {}
|
|
124
|
+
for v, dist in neighbors.items():
|
|
125
|
+
val = math.exp(-(dist * dist) / denom)
|
|
126
|
+
if val > threshold:
|
|
127
|
+
row[v] = val
|
|
128
|
+
if row:
|
|
129
|
+
result[u] = row
|
|
130
|
+
|
|
131
|
+
return result
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def gradient(field: SparseVector, graph: Mapping[K, Iterable[K]]) -> SparseMatrix:
|
|
135
|
+
"""
|
|
136
|
+
Compute the discrete gradient (exterior derivative d0) of a 0-form (node signals).
|
|
137
|
+
Maps nodes (vector) to edges (matrix).
|
|
138
|
+
grad(f)_ij = f(j) - f(i)
|
|
139
|
+
|
|
140
|
+
Args:
|
|
141
|
+
field: A vector of values at nodes.
|
|
142
|
+
graph: Adjacency list defining the edges (topology).
|
|
143
|
+
|
|
144
|
+
Returns:
|
|
145
|
+
A matrix (1-form) representing the gradient along edges.
|
|
146
|
+
"""
|
|
147
|
+
result = {}
|
|
148
|
+
for u, neighbors in graph.items():
|
|
149
|
+
if u not in field:
|
|
150
|
+
continue
|
|
151
|
+
|
|
152
|
+
val_u = field[u]
|
|
153
|
+
row = {}
|
|
154
|
+
for v in neighbors:
|
|
155
|
+
if v in field:
|
|
156
|
+
# d f(u, v) = f(v) - f(u)
|
|
157
|
+
row[v] = field[v] - val_u
|
|
158
|
+
|
|
159
|
+
if row:
|
|
160
|
+
result[u] = row
|
|
161
|
+
return result
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def laplacian(field: SparseVector, graph: SparseMatrix) -> SparseVector:
|
|
165
|
+
"""
|
|
166
|
+
Compute the combinatorial Laplacian of a scalar field.
|
|
167
|
+
L = D - A (for unweighted) or L f = div(grad f).
|
|
168
|
+
|
|
169
|
+
Delta f_i = sum_{j ~ i} w_ij * (f_i - f_j)
|
|
170
|
+
|
|
171
|
+
Args:
|
|
172
|
+
field: A vector of values at nodes.
|
|
173
|
+
graph: Adjacency matrix (weighted).
|
|
174
|
+
|
|
175
|
+
Returns:
|
|
176
|
+
A vector representing the Laplacian at each node.
|
|
177
|
+
"""
|
|
178
|
+
# L = div(grad(f))
|
|
179
|
+
# But calculating grad then div is expensive (creates intermediate matrix).
|
|
180
|
+
# Direct calculation:
|
|
181
|
+
result = defaultdict(int)
|
|
182
|
+
|
|
183
|
+
for u, neighbors in graph.items():
|
|
184
|
+
if u not in field:
|
|
185
|
+
continue
|
|
186
|
+
|
|
187
|
+
val_u = field[u]
|
|
188
|
+
# Degree (weighted)
|
|
189
|
+
# For standard Laplacian, we sum w_ij * (f_u - f_v)
|
|
190
|
+
|
|
191
|
+
local_sum = 0
|
|
192
|
+
for v, weight in neighbors.items():
|
|
193
|
+
if v in field:
|
|
194
|
+
diff = val_u - field[v]
|
|
195
|
+
local_sum += weight * diff
|
|
196
|
+
|
|
197
|
+
if local_sum != 0:
|
|
198
|
+
result[u] = local_sum
|
|
199
|
+
|
|
200
|
+
return dict(result)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def ollivier_ricci_curvature(graph: SparseMatrix) -> dict[tuple[K, K], float]:
|
|
204
|
+
"""
|
|
205
|
+
Compute the Ollivier-Ricci Curvature for edges in a graph.
|
|
206
|
+
Ric(xy) = 1 - W_1(m_x, m_y) / d(x, y)
|
|
207
|
+
|
|
208
|
+
Where W_1 is the Wasserstein distance (Earth Mover's Distance) between
|
|
209
|
+
probability measures m_x and m_y defined around nodes x and y.
|
|
210
|
+
|
|
211
|
+
Note: This is a simplified implementation approximating W_1 for sparse graphs.
|
|
212
|
+
Full computation requires a linear programming solver (e.g., scipy.optimize).
|
|
213
|
+
Here we use a greedy approximation or simple overlap for efficiency in pure Python.
|
|
214
|
+
|
|
215
|
+
Approximation: Jaccard-like overlap of neighborhoods.
|
|
216
|
+
Ric(xy) approx 1 - (1 - |N(x) n N(y)| / |N(x) u N(y)|) ... this is rough.
|
|
217
|
+
|
|
218
|
+
Let's implement a basic version based on "Forman-Ricci Curvature" which is
|
|
219
|
+
much faster and purely combinatorial (O(N) instead of O(N^3)).
|
|
220
|
+
|
|
221
|
+
Forman-Ricci Curvature for edge e=(u, v):
|
|
222
|
+
F(e) = 4 - deg(u) - deg(v)
|
|
223
|
+
|
|
224
|
+
Args:
|
|
225
|
+
graph: Adjacency matrix (weighted).
|
|
226
|
+
|
|
227
|
+
Returns:
|
|
228
|
+
A dictionary mapping edges (u, v) to their curvature values.
|
|
229
|
+
"""
|
|
230
|
+
curvature = {}
|
|
231
|
+
degrees = {u: len(neighbors) for u, neighbors in graph.items()}
|
|
232
|
+
|
|
233
|
+
for u, neighbors in graph.items():
|
|
234
|
+
for v in neighbors:
|
|
235
|
+
if u < v: # Undirected edge, process once
|
|
236
|
+
# Forman-Ricci Curvature (Combinatorial)
|
|
237
|
+
# F(e) = 4 - deg(u) - deg(v)
|
|
238
|
+
# This is a very rough proxy for "manifold curvature".
|
|
239
|
+
# Negative values -> Hyperbolic (Tree-like, Expander)
|
|
240
|
+
# Positive values -> Spherical (Clique-like, Cluster)
|
|
241
|
+
# Zero -> Euclidean (Grid)
|
|
242
|
+
|
|
243
|
+
# Refined Forman (accounting for triangles/weights is better but complex)
|
|
244
|
+
# Let's stick to the basic combinatorial definition.
|
|
245
|
+
deg_u = degrees.get(u, 0)
|
|
246
|
+
deg_v = degrees.get(v, 0)
|
|
247
|
+
|
|
248
|
+
# Standard Forman formula for unweighted graphs
|
|
249
|
+
k = 4 - deg_u - deg_v
|
|
250
|
+
|
|
251
|
+
curvature[(u, v)] = k
|
|
252
|
+
|
|
253
|
+
return curvature
|
algebrax/automata.py
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
from collections import defaultdict
|
|
2
|
+
from collections.abc import Mapping
|
|
3
|
+
|
|
4
|
+
from algebrax.typing import A, S
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
'dfa_step',
|
|
8
|
+
'nfa_step',
|
|
9
|
+
'simulate_dfa',
|
|
10
|
+
'simulate_nfa',
|
|
11
|
+
]
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def dfa_step(
|
|
15
|
+
current_state: S,
|
|
16
|
+
symbol: A,
|
|
17
|
+
transitions: Mapping[S, Mapping[A, S]],
|
|
18
|
+
) -> S | None:
|
|
19
|
+
"""
|
|
20
|
+
Perform a single step in a Deterministic Finite Automaton (DFA).
|
|
21
|
+
delta: S x A -> S
|
|
22
|
+
|
|
23
|
+
Args:
|
|
24
|
+
current_state: The current state.
|
|
25
|
+
symbol: The input symbol.
|
|
26
|
+
transitions: The transition function {state: {symbol: next_state}}.
|
|
27
|
+
|
|
28
|
+
Returns:
|
|
29
|
+
The next state, or None if the transition is undefined (implicit sink state).
|
|
30
|
+
"""
|
|
31
|
+
if current_state in transitions:
|
|
32
|
+
return transitions[current_state].get(symbol)
|
|
33
|
+
return None
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def nfa_step(
|
|
37
|
+
current_states: Mapping[S, float],
|
|
38
|
+
symbol: A,
|
|
39
|
+
transitions: Mapping[S, Mapping[A, Mapping[S, float]]],
|
|
40
|
+
) -> dict[S, float]:
|
|
41
|
+
"""
|
|
42
|
+
Perform a single step in a Nondeterministic Finite Automaton (NFA) or Probabilistic Automaton.
|
|
43
|
+
delta: S x A -> P(S)
|
|
44
|
+
|
|
45
|
+
This handles both standard NFAs (where values are 1.0) and Probabilistic Automata
|
|
46
|
+
(where values are probabilities).
|
|
47
|
+
|
|
48
|
+
Args:
|
|
49
|
+
current_states: A mapping of current states to their weights/probabilities.
|
|
50
|
+
symbol: The input symbol.
|
|
51
|
+
transitions: The transition function {state: {symbol: {next_state: weight}}}.
|
|
52
|
+
|
|
53
|
+
Returns:
|
|
54
|
+
A mapping of next states to their accumulated weights.
|
|
55
|
+
"""
|
|
56
|
+
next_states = defaultdict(float)
|
|
57
|
+
|
|
58
|
+
for state, weight in current_states.items():
|
|
59
|
+
if state in transitions:
|
|
60
|
+
symbol_transitions = transitions[state].get(symbol)
|
|
61
|
+
if symbol_transitions:
|
|
62
|
+
for next_state, trans_weight in symbol_transitions.items():
|
|
63
|
+
next_states[next_state] += weight * trans_weight
|
|
64
|
+
|
|
65
|
+
return dict(next_states)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def simulate_dfa(
|
|
69
|
+
start_state: S,
|
|
70
|
+
input_sequence: Mapping[int, A],
|
|
71
|
+
transitions: Mapping[S, Mapping[A, S]],
|
|
72
|
+
) -> S | None:
|
|
73
|
+
"""
|
|
74
|
+
Simulate a DFA on an input sequence.
|
|
75
|
+
|
|
76
|
+
Args:
|
|
77
|
+
start_state: The initial state.
|
|
78
|
+
input_sequence: A mapping {time_step: symbol} or list of symbols.
|
|
79
|
+
transitions: The transition function.
|
|
80
|
+
|
|
81
|
+
Returns:
|
|
82
|
+
The final state, or None if the machine crashed.
|
|
83
|
+
"""
|
|
84
|
+
current = start_state
|
|
85
|
+
|
|
86
|
+
# Handle both dicts (sparse sequence) and lists/iterables
|
|
87
|
+
if isinstance(input_sequence, Mapping):
|
|
88
|
+
# Sort by time index
|
|
89
|
+
steps = sorted(input_sequence.keys())
|
|
90
|
+
sequence = (input_sequence[k] for k in steps)
|
|
91
|
+
else:
|
|
92
|
+
sequence = input_sequence
|
|
93
|
+
|
|
94
|
+
for symbol in sequence:
|
|
95
|
+
current = dfa_step(current, symbol, transitions)
|
|
96
|
+
if current is None:
|
|
97
|
+
return None
|
|
98
|
+
|
|
99
|
+
return current
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def simulate_nfa(
|
|
103
|
+
start_states: Mapping[S, float],
|
|
104
|
+
input_sequence: Mapping[int, A],
|
|
105
|
+
transitions: Mapping[S, Mapping[A, Mapping[S, float]]],
|
|
106
|
+
) -> dict[S, float]:
|
|
107
|
+
"""
|
|
108
|
+
Simulate an NFA or Probabilistic Automaton on an input sequence.
|
|
109
|
+
|
|
110
|
+
Args:
|
|
111
|
+
start_states: Initial distribution of states.
|
|
112
|
+
input_sequence: A mapping {time_step: symbol} or list of symbols.
|
|
113
|
+
transitions: The transition function.
|
|
114
|
+
|
|
115
|
+
Returns:
|
|
116
|
+
The final distribution of states.
|
|
117
|
+
"""
|
|
118
|
+
current = start_states
|
|
119
|
+
|
|
120
|
+
if isinstance(input_sequence, Mapping):
|
|
121
|
+
steps = sorted(input_sequence.keys())
|
|
122
|
+
sequence = (input_sequence[k] for k in steps)
|
|
123
|
+
else:
|
|
124
|
+
sequence = input_sequence
|
|
125
|
+
|
|
126
|
+
for symbol in sequence:
|
|
127
|
+
current = nfa_step(current, symbol, transitions)
|
|
128
|
+
if not current:
|
|
129
|
+
break
|
|
130
|
+
|
|
131
|
+
return current
|