sofic 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sofic/__init__.py +185 -0
- sofic/automata/__init__.py +207 -0
- sofic/automata/_config_simulation.py +40 -0
- sofic/automata/active.py +611 -0
- sofic/automata/alergia.py +222 -0
- sofic/automata/algorithms.py +376 -0
- sofic/automata/atomaton.py +58 -0
- sofic/automata/base.py +161 -0
- sofic/automata/buchi.py +23 -0
- sofic/automata/buchi_simulation.py +67 -0
- sofic/automata/canonical_dual.py +18 -0
- sofic/automata/canonical_extraction.py +122 -0
- sofic/automata/dfa.py +85 -0
- sofic/automata/dfasat.py +195 -0
- sofic/automata/edsm.py +219 -0
- sofic/automata/enumeration.py +44 -0
- sofic/automata/icdfa.py +421 -0
- sofic/automata/idfa.py +363 -0
- sofic/automata/languages/__init__.py +39 -0
- sofic/automata/languages/_quotient_utils.py +64 -0
- sofic/automata/languages/atoms.py +31 -0
- sofic/automata/languages/automaton_ops.py +243 -0
- sofic/automata/languages/base.py +67 -0
- sofic/automata/languages/operations.py +78 -0
- sofic/automata/languages/quotients.py +66 -0
- sofic/automata/languages/residuals.py +25 -0
- sofic/automata/learning.py +79 -0
- sofic/automata/nfa.py +39 -0
- sofic/automata/nwa.py +343 -0
- sofic/automata/nwa_simulation.py +56 -0
- sofic/automata/observation.py +40 -0
- sofic/automata/papni.py +301 -0
- sofic/automata/regex.py +128 -0
- sofic/automata/rfsa.py +35 -0
- sofic/automata/rpni.py +193 -0
- sofic/automata/subsequential.py +201 -0
- sofic/automata/transducer_operations.py +350 -0
- sofic/automata/transducer_simulation.py +150 -0
- sofic/automata/transducers.py +365 -0
- sofic/automata/unifilar.py +107 -0
- sofic/automata/vpa.py +1373 -0
- sofic/automata/vpa_simulation.py +53 -0
- sofic/base.py +153 -0
- sofic/core.py +47 -0
- sofic/examples/__init__.py +86 -0
- sofic/examples/epsilon_machines.py +1089 -0
- sofic/examples/processes.py +1491 -0
- sofic/examples/shifts.py +144 -0
- sofic/exceptions.py +33 -0
- sofic/generators/__init__.py +115 -0
- sofic/generators/_word_measures.py +94 -0
- sofic/generators/alternative_complexity.py +104 -0
- sofic/generators/base.py +327 -0
- sofic/generators/bidirectional_construction.py +717 -0
- sofic/generators/bidirectional_epsilon_machine.py +689 -0
- sofic/generators/block_convergence.py +668 -0
- sofic/generators/block_entropy.py +578 -0
- sofic/generators/channel_measures.py +75 -0
- sofic/generators/conversions.py +182 -0
- sofic/generators/directional_flow.py +245 -0
- sofic/generators/edge_emissions.py +36 -0
- sofic/generators/edge_machine.py +178 -0
- sofic/generators/epsilon_construction.py +193 -0
- sofic/generators/epsilon_inference.py +703 -0
- sofic/generators/epsilon_machine.py +557 -0
- sofic/generators/epsilon_transducer.py +168 -0
- sofic/generators/epsilon_transducer_construction.py +185 -0
- sofic/generators/epsilon_transducer_inference.py +499 -0
- sofic/generators/hmm_inference.py +719 -0
- sofic/generators/information_diagram.py +428 -0
- sofic/generators/lumping.py +447 -0
- sofic/generators/markov.py +100 -0
- sofic/generators/mealy.py +156 -0
- sofic/generators/measures.py +257 -0
- sofic/generators/minimal_generative_model.py +821 -0
- sofic/generators/mixed_state.py +250 -0
- sofic/generators/mixed_state_construction.py +163 -0
- sofic/generators/moore.py +75 -0
- sofic/generators/nmachine.py +78 -0
- sofic/generators/nmachine_construction.py +70 -0
- sofic/generators/pfa.py +100 -0
- sofic/generators/prob.py +291 -0
- sofic/generators/process_equivalence.py +207 -0
- sofic/generators/quasi_inference.py +74 -0
- sofic/generators/quasi_realization.py +97 -0
- sofic/generators/reversal.py +66 -0
- sofic/generators/stack_hmm.py +426 -0
- sofic/generators/stack_inference.py +509 -0
- sofic/generators/stationary.py +134 -0
- sofic/generators/stochastic.py +65 -0
- sofic/generators/synchronization.py +407 -0
- sofic/generators/topological_epsilon_enumeration.py +349 -0
- sofic/generators/words.py +226 -0
- sofic/graph.py +135 -0
- sofic/indexing.py +31 -0
- sofic/inference/__init__.py +45 -0
- sofic/inference/bayesian/__init__.py +68 -0
- sofic/inference/bayesian/comparison.py +199 -0
- sofic/inference/bayesian/counts.py +219 -0
- sofic/inference/bayesian/diversity.py +254 -0
- sofic/inference/bayesian/epsilon.py +270 -0
- sofic/inference/bayesian/hdp_hmm.py +340 -0
- sofic/inference/bayesian/markov.py +294 -0
- sofic/inference/bayesian/pymc_backend.py +71 -0
- sofic/inference/bayesian/stack_hmm.py +215 -0
- sofic/inference/model_selection.py +365 -0
- sofic/inference/spectral.py +564 -0
- sofic/operations.py +16 -0
- sofic/properties.py +339 -0
- sofic/serialization.py +450 -0
- sofic/shifts/__init__.py +48 -0
- sofic/shifts/algorithms.py +84 -0
- sofic/shifts/base.py +49 -0
- sofic/shifts/cover_construction.py +76 -0
- sofic/shifts/covers.py +47 -0
- sofic/shifts/dyck_algorithms.py +100 -0
- sofic/shifts/dyck_enumeration.py +275 -0
- sofic/shifts/markov_dyck.py +172 -0
- sofic/shifts/parry_construction.py +82 -0
- sofic/shifts/sft.py +104 -0
- sofic/shifts/sft_construction.py +52 -0
- sofic/shifts/sliding_block_code.py +156 -0
- sofic/shifts/sofic.py +111 -0
- sofic/shifts/sofic_dyck.py +110 -0
- sofic/shifts/sofic_relation.py +64 -0
- sofic/shifts/textile.py +104 -0
- sofic/shifts/tmc.py +46 -0
- sofic/shifts/tmc_construction.py +58 -0
- sofic/shifts/topological_anatomy.py +150 -0
- sofic/states.py +27 -0
- sofic/testing/__init__.py +8 -0
- sofic/testing/strategies.py +154 -0
- sofic/viz/__init__.py +16 -0
- sofic/viz/_context.py +345 -0
- sofic/viz/_edge.py +216 -0
- sofic/viz/_format.py +89 -0
- sofic/viz/_labels.py +34 -0
- sofic/viz/_names.py +17 -0
- sofic/viz/_rational.py +20 -0
- sofic/viz/_tikz_compile.py +177 -0
- sofic/viz/_tikz_format.py +122 -0
- sofic/viz/_tikz_layout.py +218 -0
- sofic/viz/assets/vaucanson.tikz +71 -0
- sofic/viz/graphviz.py +158 -0
- sofic/viz/idiagram.py +350 -0
- sofic/viz/tikz.py +381 -0
- sofic-0.1.0.dist-info/METADATA +444 -0
- sofic-0.1.0.dist-info/RECORD +150 -0
- sofic-0.1.0.dist-info/WHEEL +4 -0
- sofic-0.1.0.dist-info/licenses/LICENSE.txt +29 -0
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
"""Probabilistic-automaton learning via the ALERGIA state-merging algorithm.
|
|
2
|
+
|
|
3
|
+
ALERGIA :cite:`Carrasco1994` learns a probabilistic deterministic automaton from
|
|
4
|
+
unlabeled positive strings by merging states of a frequency prefix-tree acceptor
|
|
5
|
+
whenever a Hoeffding-bound test cannot distinguish their outgoing (and recursive)
|
|
6
|
+
transition statistics. It is the stochastic, unlabeled counterpart of RPNI/EDSM
|
|
7
|
+
and a state-merging alternative to Causal-State Splitting Reconstruction
|
|
8
|
+
(:func:`sofic.generators.epsilon_inference.cssr`).
|
|
9
|
+
|
|
10
|
+
The learned automaton is returned as a
|
|
11
|
+
:class:`~sofic.generators.pfa.ProbabilisticFiniteAutomaton` describing the
|
|
12
|
+
symbol-generation process: per-state transition probabilities are renormalized
|
|
13
|
+
over the alphabet (the string-termination mass of the underlying PDFA is
|
|
14
|
+
dropped), so each state's outgoing masses sum to one, matching sofic's
|
|
15
|
+
row-stochastic generator convention.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import math
|
|
21
|
+
from collections.abc import Hashable, Sequence
|
|
22
|
+
from dataclasses import dataclass
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
from sofic.generators.pfa import ProbabilisticFiniteAutomaton
|
|
26
|
+
|
|
27
|
+
__all__ = ["learn_pfa_alergia"]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class _FPTA:
|
|
32
|
+
"""Frequency prefix-tree acceptor with union-find over its nodes."""
|
|
33
|
+
|
|
34
|
+
parent: list[int]
|
|
35
|
+
count: list[int] # arrivals at the block
|
|
36
|
+
final: list[int] # strings terminating in the block
|
|
37
|
+
tfreq: list[dict[Any, int]] # symbol -> transition frequency out of the block
|
|
38
|
+
tchild: list[dict[Any, int]] # symbol -> child node id
|
|
39
|
+
|
|
40
|
+
def find(self, node: int) -> int:
|
|
41
|
+
root = node
|
|
42
|
+
while self.parent[root] != root:
|
|
43
|
+
root = self.parent[root]
|
|
44
|
+
while self.parent[node] != root:
|
|
45
|
+
self.parent[node], node = root, self.parent[node]
|
|
46
|
+
return root
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _build_fpta(samples: Sequence[Sequence[Any]]) -> tuple[_FPTA, tuple[Any, ...]]:
|
|
50
|
+
count = [0]
|
|
51
|
+
final = [0]
|
|
52
|
+
tfreq: list[dict[Any, int]] = [{}]
|
|
53
|
+
tchild: list[dict[Any, int]] = [{}]
|
|
54
|
+
alphabet: set[Any] = set()
|
|
55
|
+
|
|
56
|
+
def ensure(node: int, symbol: Any) -> int:
|
|
57
|
+
if symbol not in tchild[node]:
|
|
58
|
+
tchild[node][symbol] = len(count)
|
|
59
|
+
tfreq[node][symbol] = 0
|
|
60
|
+
count.append(0)
|
|
61
|
+
final.append(0)
|
|
62
|
+
tfreq.append({})
|
|
63
|
+
tchild.append({})
|
|
64
|
+
return tchild[node][symbol]
|
|
65
|
+
|
|
66
|
+
for word in samples:
|
|
67
|
+
node = 0
|
|
68
|
+
count[0] += 1
|
|
69
|
+
for symbol in word:
|
|
70
|
+
alphabet.add(symbol)
|
|
71
|
+
child = ensure(node, symbol)
|
|
72
|
+
tfreq[node][symbol] += 1
|
|
73
|
+
count[child] += 1
|
|
74
|
+
node = child
|
|
75
|
+
final[node] += 1
|
|
76
|
+
|
|
77
|
+
fpta = _FPTA(parent=list(range(len(count))), count=count, final=final, tfreq=tfreq, tchild=tchild)
|
|
78
|
+
return fpta, tuple(sorted(alphabet, key=repr))
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _hoeffding_compatible(f1: int, n1: int, f2: int, n2: int, alpha: float) -> bool:
|
|
82
|
+
"""Hoeffding-bound two-proportion test (compatible when within the bound)."""
|
|
83
|
+
if n1 == 0 or n2 == 0:
|
|
84
|
+
return True
|
|
85
|
+
bound = math.sqrt(0.5 * math.log(2.0 / alpha)) * (1.0 / math.sqrt(n1) + 1.0 / math.sqrt(n2))
|
|
86
|
+
return abs(f1 / n1 - f2 / n2) <= bound
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _compatible(fpta: _FPTA, a: int, b: int, alpha: float, seen: set[tuple[int, int]]) -> bool:
|
|
90
|
+
"""Recursive ALERGIA compatibility of blocks ``a`` and ``b``."""
|
|
91
|
+
a, b = fpta.find(a), fpta.find(b)
|
|
92
|
+
if a == b:
|
|
93
|
+
return True
|
|
94
|
+
key = (a, b) if a < b else (b, a)
|
|
95
|
+
if key in seen:
|
|
96
|
+
return True
|
|
97
|
+
seen.add(key)
|
|
98
|
+
|
|
99
|
+
na, nb = fpta.count[a], fpta.count[b]
|
|
100
|
+
if not _hoeffding_compatible(fpta.final[a], na, fpta.final[b], nb, alpha):
|
|
101
|
+
return False
|
|
102
|
+
symbols = set(fpta.tfreq[a]) | set(fpta.tfreq[b])
|
|
103
|
+
for symbol in symbols:
|
|
104
|
+
fa = fpta.tfreq[a].get(symbol, 0)
|
|
105
|
+
fb = fpta.tfreq[b].get(symbol, 0)
|
|
106
|
+
if not _hoeffding_compatible(fa, na, fb, nb, alpha):
|
|
107
|
+
return False
|
|
108
|
+
for symbol in set(fpta.tchild[a]) & set(fpta.tchild[b]):
|
|
109
|
+
if not _compatible(fpta, fpta.tchild[a][symbol], fpta.tchild[b][symbol], alpha, seen):
|
|
110
|
+
return False
|
|
111
|
+
return True
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _merge(fpta: _FPTA, red: int, blue: int) -> None:
|
|
115
|
+
"""Fold ``blue`` into ``red``, accumulating counts and recursing on shared symbols."""
|
|
116
|
+
stack: list[tuple[int, int]] = [(red, blue)]
|
|
117
|
+
while stack:
|
|
118
|
+
left, right = stack.pop()
|
|
119
|
+
x, y = fpta.find(left), fpta.find(right)
|
|
120
|
+
if x == y:
|
|
121
|
+
continue
|
|
122
|
+
fpta.parent[y] = x
|
|
123
|
+
fpta.count[x] += fpta.count[y]
|
|
124
|
+
fpta.final[x] += fpta.final[y]
|
|
125
|
+
for symbol, freq in fpta.tfreq[y].items():
|
|
126
|
+
if symbol in fpta.tchild[x]:
|
|
127
|
+
fpta.tfreq[x][symbol] += freq
|
|
128
|
+
stack.append((fpta.tchild[x][symbol], fpta.tchild[y][symbol]))
|
|
129
|
+
else:
|
|
130
|
+
fpta.tchild[x][symbol] = fpta.tchild[y][symbol]
|
|
131
|
+
fpta.tfreq[x][symbol] = freq
|
|
132
|
+
fpta.tfreq[y] = {}
|
|
133
|
+
fpta.tchild[y] = {}
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _to_pfa(fpta: _FPTA, red: list[int], alphabet: Sequence[Any]) -> ProbabilisticFiniteAutomaton:
|
|
137
|
+
reps = sorted({fpta.find(r) for r in red})
|
|
138
|
+
rep_to_label: dict[int, Hashable] = {rep: f"q{index}" for index, rep in enumerate(reps)}
|
|
139
|
+
|
|
140
|
+
pfa = ProbabilisticFiniteAutomaton(
|
|
141
|
+
initial_distribution={rep_to_label[fpta.find(0)]: 1.0},
|
|
142
|
+
output_alphabet=frozenset(alphabet),
|
|
143
|
+
)
|
|
144
|
+
for name in rep_to_label.values():
|
|
145
|
+
pfa.graph.add_state(name)
|
|
146
|
+
|
|
147
|
+
for rep in reps:
|
|
148
|
+
name = rep_to_label[rep]
|
|
149
|
+
total = sum(fpta.tfreq[rep].get(symbol, 0) for symbol in fpta.tchild[rep])
|
|
150
|
+
if total <= 0:
|
|
151
|
+
continue # pure terminal block: no outgoing edges (allowed)
|
|
152
|
+
for symbol in sorted(fpta.tchild[rep], key=repr):
|
|
153
|
+
freq = fpta.tfreq[rep].get(symbol, 0)
|
|
154
|
+
if freq <= 0:
|
|
155
|
+
continue
|
|
156
|
+
target = rep_to_label[fpta.find(fpta.tchild[rep][symbol])]
|
|
157
|
+
pfa.add_transition(name, target, symbol, freq / total)
|
|
158
|
+
pfa.validate()
|
|
159
|
+
return pfa
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def learn_pfa_alergia(
|
|
163
|
+
samples: Sequence[Sequence[Any]],
|
|
164
|
+
*,
|
|
165
|
+
alpha: float = 0.05,
|
|
166
|
+
) -> ProbabilisticFiniteAutomaton:
|
|
167
|
+
"""Learn a probabilistic finite automaton from positive strings by ALERGIA.
|
|
168
|
+
|
|
169
|
+
Parameters
|
|
170
|
+
----------
|
|
171
|
+
samples
|
|
172
|
+
Observed strings drawn from the target process (e.g. realizations, or a
|
|
173
|
+
long sequence split into windows).
|
|
174
|
+
alpha
|
|
175
|
+
Significance level of the Hoeffding compatibility test. Smaller ``alpha``
|
|
176
|
+
merges more aggressively (fewer states); larger ``alpha`` is more
|
|
177
|
+
conservative.
|
|
178
|
+
|
|
179
|
+
Returns
|
|
180
|
+
-------
|
|
181
|
+
ProbabilisticFiniteAutomaton
|
|
182
|
+
A row-stochastic generator for the symbol process, with per-state
|
|
183
|
+
transition probabilities estimated from the merged frequencies
|
|
184
|
+
:cite:`Carrasco1994`.
|
|
185
|
+
"""
|
|
186
|
+
strings = [tuple(word) for word in samples]
|
|
187
|
+
if not strings:
|
|
188
|
+
raise ValueError("at least one sample string is required")
|
|
189
|
+
if not 0.0 < alpha < 1.0:
|
|
190
|
+
raise ValueError("alpha must lie in (0, 1)")
|
|
191
|
+
|
|
192
|
+
fpta, alphabet = _build_fpta(strings)
|
|
193
|
+
|
|
194
|
+
red: list[int] = [0]
|
|
195
|
+
while True:
|
|
196
|
+
red_reps = {fpta.find(r) for r in red}
|
|
197
|
+
blue: int | None = None
|
|
198
|
+
for r in red:
|
|
199
|
+
rep = fpta.find(r)
|
|
200
|
+
for symbol in sorted(fpta.tchild[rep], key=repr):
|
|
201
|
+
child = fpta.find(fpta.tchild[rep][symbol])
|
|
202
|
+
if child not in red_reps:
|
|
203
|
+
blue = child
|
|
204
|
+
break
|
|
205
|
+
if blue is not None:
|
|
206
|
+
break
|
|
207
|
+
if blue is None:
|
|
208
|
+
break
|
|
209
|
+
|
|
210
|
+
merged = False
|
|
211
|
+
for r in red:
|
|
212
|
+
rep = fpta.find(r)
|
|
213
|
+
if rep == blue:
|
|
214
|
+
continue
|
|
215
|
+
if _compatible(fpta, rep, blue, alpha, set()):
|
|
216
|
+
_merge(fpta, rep, blue)
|
|
217
|
+
merged = True
|
|
218
|
+
break
|
|
219
|
+
if not merged:
|
|
220
|
+
red.append(blue)
|
|
221
|
+
|
|
222
|
+
return _to_pfa(fpta, red, alphabet)
|
|
@@ -0,0 +1,376 @@
|
|
|
1
|
+
"""Automata constructions: reverse, determinize, and DFA minimization."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections import deque
|
|
6
|
+
from collections.abc import Hashable, Sequence
|
|
7
|
+
from typing import Any, Literal, TypeVar
|
|
8
|
+
|
|
9
|
+
from sofic.automata.base import LabeledAutomaton
|
|
10
|
+
from sofic.automata.dfa import DFA
|
|
11
|
+
from sofic.automata.nfa import NFA
|
|
12
|
+
from sofic.graph import ATTR_SYMBOL, EPSILON
|
|
13
|
+
|
|
14
|
+
MinimizationAlgorithm = Literal["hopcroft", "moore", "brzozowski"]
|
|
15
|
+
|
|
16
|
+
_TRAP = object()
|
|
17
|
+
|
|
18
|
+
L = TypeVar("L", bound=LabeledAutomaton)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def trim(aut: L) -> L: # noqa: UP047 - keep Python 3.11 compatibility.
|
|
22
|
+
"""Remove states not reachable from initials or not coaccessible to acceptors."""
|
|
23
|
+
reachable = _forward_reachable(aut)
|
|
24
|
+
coaccessible = _backward_coaccessible(aut)
|
|
25
|
+
keep = reachable & coaccessible
|
|
26
|
+
|
|
27
|
+
result = aut.copy()
|
|
28
|
+
for state in list(result.states()):
|
|
29
|
+
if state not in keep:
|
|
30
|
+
result.graph.nx.remove_node(state)
|
|
31
|
+
|
|
32
|
+
result.initial_states = frozenset(s for s in aut.initial_states if s in keep)
|
|
33
|
+
result.accepting_states = frozenset(s for s in aut.accepting_states if s in keep)
|
|
34
|
+
return result
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def complete(dfa: DFA, alphabet: frozenset[Any] | None = None) -> DFA:
|
|
38
|
+
"""Add a trap state so every state has one outgoing transition per symbol."""
|
|
39
|
+
symbols = alphabet if alphabet is not None else _effective_alphabet(dfa)
|
|
40
|
+
result = dfa.copy()
|
|
41
|
+
trap = _TRAP
|
|
42
|
+
if trap not in result.graph.nx:
|
|
43
|
+
result.graph.add_state(trap)
|
|
44
|
+
|
|
45
|
+
for state in list(result.states()):
|
|
46
|
+
if state == trap:
|
|
47
|
+
continue
|
|
48
|
+
for symbol in symbols:
|
|
49
|
+
if not result.delta(state, symbol):
|
|
50
|
+
result.add_transition(state, trap, symbol)
|
|
51
|
+
|
|
52
|
+
for symbol in symbols:
|
|
53
|
+
result.add_transition(trap, trap, symbol)
|
|
54
|
+
|
|
55
|
+
if symbols:
|
|
56
|
+
result.input_alphabet = frozenset(symbols) | result.input_alphabet
|
|
57
|
+
return result
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def reverse(aut: NFA | DFA) -> NFA:
|
|
61
|
+
"""Return an NFA recognizing the reversed language.
|
|
62
|
+
|
|
63
|
+
For a generic :class:`~sofic.base.StateMachine`, use
|
|
64
|
+
:func:`~sofic.operations.reverse` instead.
|
|
65
|
+
"""
|
|
66
|
+
return aut.reverse()
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def determinize(nfa: NFA, *, alphabet: frozenset[Any] | None = None) -> DFA:
|
|
70
|
+
"""Subset construction with epsilon closure."""
|
|
71
|
+
symbols = alphabet if alphabet is not None else _effective_alphabet(nfa)
|
|
72
|
+
start = frozenset(nfa.epsilon_closure(set(nfa.initial_states)))
|
|
73
|
+
|
|
74
|
+
subsets: dict[frozenset[Hashable], frozenset[Hashable]] = {start: start}
|
|
75
|
+
queue: deque[frozenset[Hashable]] = deque([start])
|
|
76
|
+
edges: list[tuple[frozenset[Hashable], frozenset[Hashable], Any]] = []
|
|
77
|
+
|
|
78
|
+
while queue:
|
|
79
|
+
current = queue.popleft()
|
|
80
|
+
for symbol in symbols:
|
|
81
|
+
next_raw: set[Hashable] = set()
|
|
82
|
+
for state in current:
|
|
83
|
+
next_raw.update(nfa.delta(state, symbol))
|
|
84
|
+
target = frozenset(nfa.epsilon_closure(next_raw))
|
|
85
|
+
edges.append((current, target, symbol))
|
|
86
|
+
if target not in subsets:
|
|
87
|
+
subsets[target] = target
|
|
88
|
+
queue.append(target)
|
|
89
|
+
|
|
90
|
+
dfa = DFA(
|
|
91
|
+
input_alphabet=symbols,
|
|
92
|
+
initial_states=frozenset({start}),
|
|
93
|
+
accepting_states=frozenset(s for s in subsets if s & nfa.accepting_states),
|
|
94
|
+
)
|
|
95
|
+
for subset in subsets:
|
|
96
|
+
dfa.graph.add_state(subset)
|
|
97
|
+
for source, target, symbol in edges:
|
|
98
|
+
dfa.add_transition(source, target, symbol)
|
|
99
|
+
return dfa
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def minimize(
|
|
103
|
+
aut: DFA | NFA,
|
|
104
|
+
*,
|
|
105
|
+
algorithm: MinimizationAlgorithm = "hopcroft",
|
|
106
|
+
alphabet: frozenset[Any] | None = None,
|
|
107
|
+
) -> DFA:
|
|
108
|
+
"""Return a minimal DFA equivalent to ``aut``."""
|
|
109
|
+
if algorithm == "brzozowski":
|
|
110
|
+
return minimize_brzozowski(aut, alphabet=alphabet)
|
|
111
|
+
dfa = aut if isinstance(aut, DFA) else determinize(aut, alphabet=alphabet)
|
|
112
|
+
if algorithm == "hopcroft":
|
|
113
|
+
return minimize_hopcroft(dfa, alphabet=alphabet)
|
|
114
|
+
if algorithm == "moore":
|
|
115
|
+
return minimize_moore(dfa, alphabet=alphabet)
|
|
116
|
+
raise ValueError(f"unknown minimization algorithm {algorithm!r}")
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def minimize_brzozowski(
|
|
120
|
+
aut: DFA | NFA,
|
|
121
|
+
*,
|
|
122
|
+
alphabet: frozenset[Any] | None = None,
|
|
123
|
+
) -> DFA:
|
|
124
|
+
"""Minimize via Brzozowski double reversal: det(rev(det(rev(A))))."""
|
|
125
|
+
nfa = aut if isinstance(aut, NFA) else _dfa_as_nfa(aut)
|
|
126
|
+
return determinize(trim(nfa).reverse().determinize(alphabet=alphabet).reverse().determinize(alphabet=alphabet))
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def minimize_moore(dfa: DFA, *, alphabet: frozenset[Any] | None = None) -> DFA:
|
|
130
|
+
"""Minimize a DFA using Moore (1961) partition refinement."""
|
|
131
|
+
symbols = alphabet if alphabet is not None else _effective_alphabet(dfa)
|
|
132
|
+
work = complete(trim(dfa), symbols)
|
|
133
|
+
states = sorted(work.states(), key=repr)
|
|
134
|
+
if not states:
|
|
135
|
+
return work
|
|
136
|
+
|
|
137
|
+
partition = _initial_partition(states, work.accepting_states)
|
|
138
|
+
changed = True
|
|
139
|
+
while changed:
|
|
140
|
+
changed = False
|
|
141
|
+
new_partition: list[set[Hashable]] = []
|
|
142
|
+
for block in partition:
|
|
143
|
+
refined: list[set[Hashable]] = [set(block)]
|
|
144
|
+
for symbol in sorted(symbols, key=repr):
|
|
145
|
+
next_refined: list[set[Hashable]] = []
|
|
146
|
+
for piece in refined:
|
|
147
|
+
groups: dict[int, set[Hashable]] = {}
|
|
148
|
+
for state in piece:
|
|
149
|
+
successor = _dfa_successor(work, state, symbol)
|
|
150
|
+
index = -1 if successor is None else _block_index(partition, successor)
|
|
151
|
+
groups.setdefault(index, set()).add(state)
|
|
152
|
+
next_refined.extend(groups.values())
|
|
153
|
+
refined = next_refined
|
|
154
|
+
if len(refined) > 1:
|
|
155
|
+
changed = True
|
|
156
|
+
new_partition.extend(refined)
|
|
157
|
+
partition = new_partition
|
|
158
|
+
|
|
159
|
+
return _quotient_from_partition(work, partition, symbols)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def minimize_hopcroft(dfa: DFA, *, alphabet: frozenset[Any] | None = None) -> DFA:
|
|
163
|
+
"""Minimize a DFA using Hopcroft's algorithm."""
|
|
164
|
+
symbols = alphabet if alphabet is not None else _effective_alphabet(dfa)
|
|
165
|
+
work = complete(trim(dfa), symbols)
|
|
166
|
+
states = sorted(work.states(), key=repr)
|
|
167
|
+
if not states:
|
|
168
|
+
return work
|
|
169
|
+
|
|
170
|
+
accepting = set(work.accepting_states)
|
|
171
|
+
partition: list[set[Hashable]] = []
|
|
172
|
+
if accepting:
|
|
173
|
+
partition.append(accepting & set(states))
|
|
174
|
+
non_accepting = set(states) - accepting
|
|
175
|
+
if non_accepting:
|
|
176
|
+
partition.append(non_accepting)
|
|
177
|
+
|
|
178
|
+
pred = _inverse_transitions(work, states, symbols)
|
|
179
|
+
worklist: list[set[Hashable]] = [block.copy() for block in partition]
|
|
180
|
+
|
|
181
|
+
while worklist:
|
|
182
|
+
focus = worklist.pop()
|
|
183
|
+
for symbol in symbols:
|
|
184
|
+
predecessors: set[Hashable] = set()
|
|
185
|
+
for state in focus:
|
|
186
|
+
predecessors.update(pred[state][symbol])
|
|
187
|
+
refined_partition: list[set[Hashable]] = []
|
|
188
|
+
for block in partition:
|
|
189
|
+
intersection = block & predecessors
|
|
190
|
+
difference = block - predecessors
|
|
191
|
+
if intersection and difference:
|
|
192
|
+
refined_partition.append(intersection)
|
|
193
|
+
refined_partition.append(difference)
|
|
194
|
+
if block in worklist:
|
|
195
|
+
worklist.remove(block)
|
|
196
|
+
worklist.append(intersection)
|
|
197
|
+
worklist.append(difference)
|
|
198
|
+
else:
|
|
199
|
+
if len(intersection) <= len(difference):
|
|
200
|
+
worklist.append(intersection)
|
|
201
|
+
else:
|
|
202
|
+
worklist.append(difference)
|
|
203
|
+
else:
|
|
204
|
+
refined_partition.append(block)
|
|
205
|
+
partition = refined_partition
|
|
206
|
+
|
|
207
|
+
return _quotient_from_partition(work, partition, symbols)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def equivalent(
|
|
211
|
+
aut1: LabeledAutomaton,
|
|
212
|
+
aut2: LabeledAutomaton,
|
|
213
|
+
alphabet: frozenset[Any],
|
|
214
|
+
) -> bool:
|
|
215
|
+
"""Return whether two automata recognize the same language over ``alphabet``."""
|
|
216
|
+
d1 = minimize(_to_nfa(aut1), alphabet=alphabet, algorithm="hopcroft")
|
|
217
|
+
d2 = minimize(_to_nfa(aut2), alphabet=alphabet, algorithm="hopcroft")
|
|
218
|
+
return _isomorphic_minimal_dfa(d1, d2, alphabet)
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _to_nfa(aut: LabeledAutomaton) -> NFA:
|
|
222
|
+
if isinstance(aut, NFA):
|
|
223
|
+
return aut
|
|
224
|
+
if isinstance(aut, DFA):
|
|
225
|
+
return _dfa_as_nfa(aut)
|
|
226
|
+
raise TypeError(f"unsupported automaton type {type(aut)!r}")
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _dfa_as_nfa(dfa: DFA) -> NFA:
|
|
230
|
+
nfa = NFA(
|
|
231
|
+
input_alphabet=dfa.input_alphabet,
|
|
232
|
+
initial_states=dfa.initial_states,
|
|
233
|
+
accepting_states=dfa.accepting_states,
|
|
234
|
+
graph=dfa.graph.copy(),
|
|
235
|
+
)
|
|
236
|
+
return nfa
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _effective_alphabet(aut: LabeledAutomaton) -> frozenset[Any]:
|
|
240
|
+
symbols = {symbol for symbol in aut.input_alphabet if symbol is not EPSILON}
|
|
241
|
+
if symbols:
|
|
242
|
+
return frozenset(symbols)
|
|
243
|
+
for transition in aut.transitions():
|
|
244
|
+
symbol = transition.data.get(ATTR_SYMBOL)
|
|
245
|
+
if symbol is not None and symbol is not EPSILON:
|
|
246
|
+
symbols.add(symbol)
|
|
247
|
+
return frozenset(symbols)
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _forward_reachable(aut: LabeledAutomaton) -> set[Hashable]:
|
|
251
|
+
if not aut.initial_states:
|
|
252
|
+
return set()
|
|
253
|
+
seed = set(aut.epsilon_closure(set(aut.initial_states)))
|
|
254
|
+
return set(aut.graph.forward_reachable(seed))
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def _backward_coaccessible(aut: LabeledAutomaton) -> set[Hashable]:
|
|
258
|
+
predecessors: dict[Hashable, set[Hashable]] = {state: set() for state in aut.states()}
|
|
259
|
+
for transition in aut.transitions():
|
|
260
|
+
predecessors.setdefault(transition.target, set()).add(transition.source)
|
|
261
|
+
|
|
262
|
+
coaccessible = set(aut.accepting_states)
|
|
263
|
+
queue = deque(coaccessible)
|
|
264
|
+
while queue:
|
|
265
|
+
state = queue.popleft()
|
|
266
|
+
for predecessor in predecessors.get(state, ()):
|
|
267
|
+
if predecessor not in coaccessible:
|
|
268
|
+
coaccessible.add(predecessor)
|
|
269
|
+
queue.append(predecessor)
|
|
270
|
+
return coaccessible
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _dfa_successor(dfa: DFA, state: Hashable, symbol: Any) -> Hashable | None:
|
|
274
|
+
successors = dfa.delta(state, symbol)
|
|
275
|
+
if not successors:
|
|
276
|
+
return None
|
|
277
|
+
return next(iter(successors))
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def _initial_partition(states: Sequence[Hashable], accepting: frozenset[Hashable]) -> list[set[Hashable]]:
|
|
281
|
+
accepting_block = set(accepting) & set(states)
|
|
282
|
+
non_accepting = set(states) - accepting_block
|
|
283
|
+
partition: list[set[Hashable]] = []
|
|
284
|
+
if accepting_block:
|
|
285
|
+
partition.append(accepting_block)
|
|
286
|
+
if non_accepting:
|
|
287
|
+
partition.append(non_accepting)
|
|
288
|
+
return partition
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _block_index(partition: list[set[Hashable]], state: Hashable) -> int:
|
|
292
|
+
for index, block in enumerate(partition):
|
|
293
|
+
if state in block:
|
|
294
|
+
return index
|
|
295
|
+
raise KeyError(state)
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def _inverse_transitions(
|
|
299
|
+
dfa: DFA,
|
|
300
|
+
states: Sequence[Hashable],
|
|
301
|
+
symbols: frozenset[Any],
|
|
302
|
+
) -> dict[Hashable, dict[Any, set[Hashable]]]:
|
|
303
|
+
pred: dict[Hashable, dict[Any, set[Hashable]]] = {state: {symbol: set() for symbol in symbols} for state in states}
|
|
304
|
+
for state in states:
|
|
305
|
+
for symbol in symbols:
|
|
306
|
+
target = _dfa_successor(dfa, state, symbol)
|
|
307
|
+
if target is not None and target in pred:
|
|
308
|
+
pred[target][symbol].add(state)
|
|
309
|
+
return pred
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def _quotient_from_partition(dfa: DFA, partition: list[set[Hashable]], symbols: frozenset[Any]) -> DFA:
|
|
313
|
+
blocks = [block for block in partition if block]
|
|
314
|
+
block_of = {state: index for index, block in enumerate(blocks) for state in block}
|
|
315
|
+
representatives = [min(block, key=repr) for block in blocks]
|
|
316
|
+
rep_for_block = {index: frozenset({representatives[index]}) for index in range(len(blocks))}
|
|
317
|
+
|
|
318
|
+
initial_block = block_of[next(iter(dfa.initial_states))] if dfa.initial_states else 0
|
|
319
|
+
accepting_blocks = frozenset(
|
|
320
|
+
rep_for_block[index] for index, block in enumerate(blocks) if block & dfa.accepting_states
|
|
321
|
+
)
|
|
322
|
+
|
|
323
|
+
result = DFA(
|
|
324
|
+
input_alphabet=symbols,
|
|
325
|
+
initial_states=frozenset({rep_for_block[initial_block]}),
|
|
326
|
+
accepting_states=accepting_blocks,
|
|
327
|
+
)
|
|
328
|
+
for index, _rep in enumerate(representatives):
|
|
329
|
+
result.graph.add_state(rep_for_block[index])
|
|
330
|
+
|
|
331
|
+
for index, rep in enumerate(representatives):
|
|
332
|
+
source = rep_for_block[index]
|
|
333
|
+
for symbol in symbols:
|
|
334
|
+
target_state = _dfa_successor(dfa, rep, symbol)
|
|
335
|
+
if target_state is None:
|
|
336
|
+
continue
|
|
337
|
+
target_block = block_of[target_state]
|
|
338
|
+
result.add_transition(source, rep_for_block[target_block], symbol)
|
|
339
|
+
|
|
340
|
+
return trim(result)
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def _isomorphic_minimal_dfa(d1: DFA, d2: DFA, alphabet: frozenset[Any]) -> bool:
|
|
344
|
+
states1 = list(d1.states())
|
|
345
|
+
states2 = list(d2.states())
|
|
346
|
+
if len(states1) != len(states2):
|
|
347
|
+
return False
|
|
348
|
+
if not d1.initial_states or not d2.initial_states:
|
|
349
|
+
return not d1.initial_states and not d2.initial_states
|
|
350
|
+
|
|
351
|
+
start1 = next(iter(d1.initial_states))
|
|
352
|
+
start2 = next(iter(d2.initial_states))
|
|
353
|
+
if (start1 in d1.accepting_states) != (start2 in d2.accepting_states):
|
|
354
|
+
return False
|
|
355
|
+
|
|
356
|
+
mapping: dict[Hashable, Hashable] = {start1: start2}
|
|
357
|
+
queue = deque([start1])
|
|
358
|
+
while queue:
|
|
359
|
+
state1 = queue.popleft()
|
|
360
|
+
state2 = mapping[state1]
|
|
361
|
+
for symbol in sorted(alphabet, key=repr):
|
|
362
|
+
succ1 = _dfa_successor(d1, state1, symbol)
|
|
363
|
+
succ2 = _dfa_successor(d2, state2, symbol)
|
|
364
|
+
if succ1 is None and succ2 is None:
|
|
365
|
+
continue
|
|
366
|
+
if succ1 is None or succ2 is None:
|
|
367
|
+
return False
|
|
368
|
+
if succ1 in mapping:
|
|
369
|
+
if mapping[succ1] != succ2:
|
|
370
|
+
return False
|
|
371
|
+
else:
|
|
372
|
+
if (succ1 in d1.accepting_states) != (succ2 in d2.accepting_states):
|
|
373
|
+
return False
|
|
374
|
+
mapping[succ1] = succ2
|
|
375
|
+
queue.append(succ1)
|
|
376
|
+
return len(mapping) == len(states1)
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Atomic and átomaton NFA presentations."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import TYPE_CHECKING, Any
|
|
6
|
+
|
|
7
|
+
from sofic.automata.dfa import DFA
|
|
8
|
+
from sofic.automata.languages.base import RegularLanguage
|
|
9
|
+
from sofic.automata.nfa import NFA
|
|
10
|
+
|
|
11
|
+
if TYPE_CHECKING:
|
|
12
|
+
from sofic.automata.observation import ObservationTable
|
|
13
|
+
from sofic.automata.rfsa import CanonicalRFSA
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class AtomicAutomaton(NFA):
|
|
17
|
+
"""NFA whose states accept unions of atoms."""
|
|
18
|
+
|
|
19
|
+
def validate(self) -> None:
|
|
20
|
+
super().validate()
|
|
21
|
+
# Phase 2: verify right languages are unions of atoms
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class Atomaton(AtomicAutomaton):
|
|
25
|
+
"""Canonical átomaton whose states are all atoms of L."""
|
|
26
|
+
|
|
27
|
+
@classmethod
|
|
28
|
+
def from_language(cls, language: RegularLanguage | NFA, **kwargs: Any) -> Atomaton:
|
|
29
|
+
from sofic.automata.canonical_extraction import atomaton_from_language
|
|
30
|
+
|
|
31
|
+
return atomaton_from_language(language)
|
|
32
|
+
|
|
33
|
+
def to_minimal_dfa_via_double_reversal(self) -> DFA:
|
|
34
|
+
from sofic.automata.algorithms import minimize
|
|
35
|
+
|
|
36
|
+
return minimize(self, algorithm="brzozowski")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class MaximizedPrimeAtomaton(AtomicAutomaton):
|
|
40
|
+
"""Maximized prime átomaton — dual of the canonical RFSA."""
|
|
41
|
+
|
|
42
|
+
@classmethod
|
|
43
|
+
def from_language(cls, language: RegularLanguage | NFA, **kwargs: Any) -> MaximizedPrimeAtomaton:
|
|
44
|
+
from sofic.automata.canonical_extraction import maximized_prime_atomaton_from_language
|
|
45
|
+
|
|
46
|
+
return maximized_prime_atomaton_from_language(language)
|
|
47
|
+
|
|
48
|
+
@classmethod
|
|
49
|
+
def from_observation_table(cls, table: ObservationTable, **kwargs: Any) -> MaximizedPrimeAtomaton:
|
|
50
|
+
from sofic.automata.canonical_extraction import observation_to_maximized_prime_atomaton
|
|
51
|
+
|
|
52
|
+
return observation_to_maximized_prime_atomaton(table)
|
|
53
|
+
|
|
54
|
+
@classmethod
|
|
55
|
+
def from_canonical_rfsa(cls, rfsa: CanonicalRFSA, **kwargs: Any) -> MaximizedPrimeAtomaton:
|
|
56
|
+
from sofic.automata.canonical_dual import dual_atomaton_from_rfsa
|
|
57
|
+
|
|
58
|
+
return dual_atomaton_from_rfsa(rfsa)
|