sofic 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (150) hide show
  1. sofic/__init__.py +185 -0
  2. sofic/automata/__init__.py +207 -0
  3. sofic/automata/_config_simulation.py +40 -0
  4. sofic/automata/active.py +611 -0
  5. sofic/automata/alergia.py +222 -0
  6. sofic/automata/algorithms.py +376 -0
  7. sofic/automata/atomaton.py +58 -0
  8. sofic/automata/base.py +161 -0
  9. sofic/automata/buchi.py +23 -0
  10. sofic/automata/buchi_simulation.py +67 -0
  11. sofic/automata/canonical_dual.py +18 -0
  12. sofic/automata/canonical_extraction.py +122 -0
  13. sofic/automata/dfa.py +85 -0
  14. sofic/automata/dfasat.py +195 -0
  15. sofic/automata/edsm.py +219 -0
  16. sofic/automata/enumeration.py +44 -0
  17. sofic/automata/icdfa.py +421 -0
  18. sofic/automata/idfa.py +363 -0
  19. sofic/automata/languages/__init__.py +39 -0
  20. sofic/automata/languages/_quotient_utils.py +64 -0
  21. sofic/automata/languages/atoms.py +31 -0
  22. sofic/automata/languages/automaton_ops.py +243 -0
  23. sofic/automata/languages/base.py +67 -0
  24. sofic/automata/languages/operations.py +78 -0
  25. sofic/automata/languages/quotients.py +66 -0
  26. sofic/automata/languages/residuals.py +25 -0
  27. sofic/automata/learning.py +79 -0
  28. sofic/automata/nfa.py +39 -0
  29. sofic/automata/nwa.py +343 -0
  30. sofic/automata/nwa_simulation.py +56 -0
  31. sofic/automata/observation.py +40 -0
  32. sofic/automata/papni.py +301 -0
  33. sofic/automata/regex.py +128 -0
  34. sofic/automata/rfsa.py +35 -0
  35. sofic/automata/rpni.py +193 -0
  36. sofic/automata/subsequential.py +201 -0
  37. sofic/automata/transducer_operations.py +350 -0
  38. sofic/automata/transducer_simulation.py +150 -0
  39. sofic/automata/transducers.py +365 -0
  40. sofic/automata/unifilar.py +107 -0
  41. sofic/automata/vpa.py +1373 -0
  42. sofic/automata/vpa_simulation.py +53 -0
  43. sofic/base.py +153 -0
  44. sofic/core.py +47 -0
  45. sofic/examples/__init__.py +86 -0
  46. sofic/examples/epsilon_machines.py +1089 -0
  47. sofic/examples/processes.py +1491 -0
  48. sofic/examples/shifts.py +144 -0
  49. sofic/exceptions.py +33 -0
  50. sofic/generators/__init__.py +115 -0
  51. sofic/generators/_word_measures.py +94 -0
  52. sofic/generators/alternative_complexity.py +104 -0
  53. sofic/generators/base.py +327 -0
  54. sofic/generators/bidirectional_construction.py +717 -0
  55. sofic/generators/bidirectional_epsilon_machine.py +689 -0
  56. sofic/generators/block_convergence.py +668 -0
  57. sofic/generators/block_entropy.py +578 -0
  58. sofic/generators/channel_measures.py +75 -0
  59. sofic/generators/conversions.py +182 -0
  60. sofic/generators/directional_flow.py +245 -0
  61. sofic/generators/edge_emissions.py +36 -0
  62. sofic/generators/edge_machine.py +178 -0
  63. sofic/generators/epsilon_construction.py +193 -0
  64. sofic/generators/epsilon_inference.py +703 -0
  65. sofic/generators/epsilon_machine.py +557 -0
  66. sofic/generators/epsilon_transducer.py +168 -0
  67. sofic/generators/epsilon_transducer_construction.py +185 -0
  68. sofic/generators/epsilon_transducer_inference.py +499 -0
  69. sofic/generators/hmm_inference.py +719 -0
  70. sofic/generators/information_diagram.py +428 -0
  71. sofic/generators/lumping.py +447 -0
  72. sofic/generators/markov.py +100 -0
  73. sofic/generators/mealy.py +156 -0
  74. sofic/generators/measures.py +257 -0
  75. sofic/generators/minimal_generative_model.py +821 -0
  76. sofic/generators/mixed_state.py +250 -0
  77. sofic/generators/mixed_state_construction.py +163 -0
  78. sofic/generators/moore.py +75 -0
  79. sofic/generators/nmachine.py +78 -0
  80. sofic/generators/nmachine_construction.py +70 -0
  81. sofic/generators/pfa.py +100 -0
  82. sofic/generators/prob.py +291 -0
  83. sofic/generators/process_equivalence.py +207 -0
  84. sofic/generators/quasi_inference.py +74 -0
  85. sofic/generators/quasi_realization.py +97 -0
  86. sofic/generators/reversal.py +66 -0
  87. sofic/generators/stack_hmm.py +426 -0
  88. sofic/generators/stack_inference.py +509 -0
  89. sofic/generators/stationary.py +134 -0
  90. sofic/generators/stochastic.py +65 -0
  91. sofic/generators/synchronization.py +407 -0
  92. sofic/generators/topological_epsilon_enumeration.py +349 -0
  93. sofic/generators/words.py +226 -0
  94. sofic/graph.py +135 -0
  95. sofic/indexing.py +31 -0
  96. sofic/inference/__init__.py +45 -0
  97. sofic/inference/bayesian/__init__.py +68 -0
  98. sofic/inference/bayesian/comparison.py +199 -0
  99. sofic/inference/bayesian/counts.py +219 -0
  100. sofic/inference/bayesian/diversity.py +254 -0
  101. sofic/inference/bayesian/epsilon.py +270 -0
  102. sofic/inference/bayesian/hdp_hmm.py +340 -0
  103. sofic/inference/bayesian/markov.py +294 -0
  104. sofic/inference/bayesian/pymc_backend.py +71 -0
  105. sofic/inference/bayesian/stack_hmm.py +215 -0
  106. sofic/inference/model_selection.py +365 -0
  107. sofic/inference/spectral.py +564 -0
  108. sofic/operations.py +16 -0
  109. sofic/properties.py +339 -0
  110. sofic/serialization.py +450 -0
  111. sofic/shifts/__init__.py +48 -0
  112. sofic/shifts/algorithms.py +84 -0
  113. sofic/shifts/base.py +49 -0
  114. sofic/shifts/cover_construction.py +76 -0
  115. sofic/shifts/covers.py +47 -0
  116. sofic/shifts/dyck_algorithms.py +100 -0
  117. sofic/shifts/dyck_enumeration.py +275 -0
  118. sofic/shifts/markov_dyck.py +172 -0
  119. sofic/shifts/parry_construction.py +82 -0
  120. sofic/shifts/sft.py +104 -0
  121. sofic/shifts/sft_construction.py +52 -0
  122. sofic/shifts/sliding_block_code.py +156 -0
  123. sofic/shifts/sofic.py +111 -0
  124. sofic/shifts/sofic_dyck.py +110 -0
  125. sofic/shifts/sofic_relation.py +64 -0
  126. sofic/shifts/textile.py +104 -0
  127. sofic/shifts/tmc.py +46 -0
  128. sofic/shifts/tmc_construction.py +58 -0
  129. sofic/shifts/topological_anatomy.py +150 -0
  130. sofic/states.py +27 -0
  131. sofic/testing/__init__.py +8 -0
  132. sofic/testing/strategies.py +154 -0
  133. sofic/viz/__init__.py +16 -0
  134. sofic/viz/_context.py +345 -0
  135. sofic/viz/_edge.py +216 -0
  136. sofic/viz/_format.py +89 -0
  137. sofic/viz/_labels.py +34 -0
  138. sofic/viz/_names.py +17 -0
  139. sofic/viz/_rational.py +20 -0
  140. sofic/viz/_tikz_compile.py +177 -0
  141. sofic/viz/_tikz_format.py +122 -0
  142. sofic/viz/_tikz_layout.py +218 -0
  143. sofic/viz/assets/vaucanson.tikz +71 -0
  144. sofic/viz/graphviz.py +158 -0
  145. sofic/viz/idiagram.py +350 -0
  146. sofic/viz/tikz.py +381 -0
  147. sofic-0.1.0.dist-info/METADATA +444 -0
  148. sofic-0.1.0.dist-info/RECORD +150 -0
  149. sofic-0.1.0.dist-info/WHEEL +4 -0
  150. sofic-0.1.0.dist-info/licenses/LICENSE.txt +29 -0
@@ -0,0 +1,222 @@
1
+ """Probabilistic-automaton learning via the ALERGIA state-merging algorithm.
2
+
3
+ ALERGIA :cite:`Carrasco1994` learns a probabilistic deterministic automaton from
4
+ unlabeled positive strings by merging states of a frequency prefix-tree acceptor
5
+ whenever a Hoeffding-bound test cannot distinguish their outgoing (and recursive)
6
+ transition statistics. It is the stochastic, unlabeled counterpart of RPNI/EDSM
7
+ and a state-merging alternative to Causal-State Splitting Reconstruction
8
+ (:func:`sofic.generators.epsilon_inference.cssr`).
9
+
10
+ The learned automaton is returned as a
11
+ :class:`~sofic.generators.pfa.ProbabilisticFiniteAutomaton` describing the
12
+ symbol-generation process: per-state transition probabilities are renormalized
13
+ over the alphabet (the string-termination mass of the underlying PDFA is
14
+ dropped), so each state's outgoing masses sum to one, matching sofic's
15
+ row-stochastic generator convention.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import math
21
+ from collections.abc import Hashable, Sequence
22
+ from dataclasses import dataclass
23
+ from typing import Any
24
+
25
+ from sofic.generators.pfa import ProbabilisticFiniteAutomaton
26
+
27
+ __all__ = ["learn_pfa_alergia"]
28
+
29
+
30
+ @dataclass
31
+ class _FPTA:
32
+ """Frequency prefix-tree acceptor with union-find over its nodes."""
33
+
34
+ parent: list[int]
35
+ count: list[int] # arrivals at the block
36
+ final: list[int] # strings terminating in the block
37
+ tfreq: list[dict[Any, int]] # symbol -> transition frequency out of the block
38
+ tchild: list[dict[Any, int]] # symbol -> child node id
39
+
40
+ def find(self, node: int) -> int:
41
+ root = node
42
+ while self.parent[root] != root:
43
+ root = self.parent[root]
44
+ while self.parent[node] != root:
45
+ self.parent[node], node = root, self.parent[node]
46
+ return root
47
+
48
+
49
+ def _build_fpta(samples: Sequence[Sequence[Any]]) -> tuple[_FPTA, tuple[Any, ...]]:
50
+ count = [0]
51
+ final = [0]
52
+ tfreq: list[dict[Any, int]] = [{}]
53
+ tchild: list[dict[Any, int]] = [{}]
54
+ alphabet: set[Any] = set()
55
+
56
+ def ensure(node: int, symbol: Any) -> int:
57
+ if symbol not in tchild[node]:
58
+ tchild[node][symbol] = len(count)
59
+ tfreq[node][symbol] = 0
60
+ count.append(0)
61
+ final.append(0)
62
+ tfreq.append({})
63
+ tchild.append({})
64
+ return tchild[node][symbol]
65
+
66
+ for word in samples:
67
+ node = 0
68
+ count[0] += 1
69
+ for symbol in word:
70
+ alphabet.add(symbol)
71
+ child = ensure(node, symbol)
72
+ tfreq[node][symbol] += 1
73
+ count[child] += 1
74
+ node = child
75
+ final[node] += 1
76
+
77
+ fpta = _FPTA(parent=list(range(len(count))), count=count, final=final, tfreq=tfreq, tchild=tchild)
78
+ return fpta, tuple(sorted(alphabet, key=repr))
79
+
80
+
81
+ def _hoeffding_compatible(f1: int, n1: int, f2: int, n2: int, alpha: float) -> bool:
82
+ """Hoeffding-bound two-proportion test (compatible when within the bound)."""
83
+ if n1 == 0 or n2 == 0:
84
+ return True
85
+ bound = math.sqrt(0.5 * math.log(2.0 / alpha)) * (1.0 / math.sqrt(n1) + 1.0 / math.sqrt(n2))
86
+ return abs(f1 / n1 - f2 / n2) <= bound
87
+
88
+
89
+ def _compatible(fpta: _FPTA, a: int, b: int, alpha: float, seen: set[tuple[int, int]]) -> bool:
90
+ """Recursive ALERGIA compatibility of blocks ``a`` and ``b``."""
91
+ a, b = fpta.find(a), fpta.find(b)
92
+ if a == b:
93
+ return True
94
+ key = (a, b) if a < b else (b, a)
95
+ if key in seen:
96
+ return True
97
+ seen.add(key)
98
+
99
+ na, nb = fpta.count[a], fpta.count[b]
100
+ if not _hoeffding_compatible(fpta.final[a], na, fpta.final[b], nb, alpha):
101
+ return False
102
+ symbols = set(fpta.tfreq[a]) | set(fpta.tfreq[b])
103
+ for symbol in symbols:
104
+ fa = fpta.tfreq[a].get(symbol, 0)
105
+ fb = fpta.tfreq[b].get(symbol, 0)
106
+ if not _hoeffding_compatible(fa, na, fb, nb, alpha):
107
+ return False
108
+ for symbol in set(fpta.tchild[a]) & set(fpta.tchild[b]):
109
+ if not _compatible(fpta, fpta.tchild[a][symbol], fpta.tchild[b][symbol], alpha, seen):
110
+ return False
111
+ return True
112
+
113
+
114
+ def _merge(fpta: _FPTA, red: int, blue: int) -> None:
115
+ """Fold ``blue`` into ``red``, accumulating counts and recursing on shared symbols."""
116
+ stack: list[tuple[int, int]] = [(red, blue)]
117
+ while stack:
118
+ left, right = stack.pop()
119
+ x, y = fpta.find(left), fpta.find(right)
120
+ if x == y:
121
+ continue
122
+ fpta.parent[y] = x
123
+ fpta.count[x] += fpta.count[y]
124
+ fpta.final[x] += fpta.final[y]
125
+ for symbol, freq in fpta.tfreq[y].items():
126
+ if symbol in fpta.tchild[x]:
127
+ fpta.tfreq[x][symbol] += freq
128
+ stack.append((fpta.tchild[x][symbol], fpta.tchild[y][symbol]))
129
+ else:
130
+ fpta.tchild[x][symbol] = fpta.tchild[y][symbol]
131
+ fpta.tfreq[x][symbol] = freq
132
+ fpta.tfreq[y] = {}
133
+ fpta.tchild[y] = {}
134
+
135
+
136
+ def _to_pfa(fpta: _FPTA, red: list[int], alphabet: Sequence[Any]) -> ProbabilisticFiniteAutomaton:
137
+ reps = sorted({fpta.find(r) for r in red})
138
+ rep_to_label: dict[int, Hashable] = {rep: f"q{index}" for index, rep in enumerate(reps)}
139
+
140
+ pfa = ProbabilisticFiniteAutomaton(
141
+ initial_distribution={rep_to_label[fpta.find(0)]: 1.0},
142
+ output_alphabet=frozenset(alphabet),
143
+ )
144
+ for name in rep_to_label.values():
145
+ pfa.graph.add_state(name)
146
+
147
+ for rep in reps:
148
+ name = rep_to_label[rep]
149
+ total = sum(fpta.tfreq[rep].get(symbol, 0) for symbol in fpta.tchild[rep])
150
+ if total <= 0:
151
+ continue # pure terminal block: no outgoing edges (allowed)
152
+ for symbol in sorted(fpta.tchild[rep], key=repr):
153
+ freq = fpta.tfreq[rep].get(symbol, 0)
154
+ if freq <= 0:
155
+ continue
156
+ target = rep_to_label[fpta.find(fpta.tchild[rep][symbol])]
157
+ pfa.add_transition(name, target, symbol, freq / total)
158
+ pfa.validate()
159
+ return pfa
160
+
161
+
162
+ def learn_pfa_alergia(
163
+ samples: Sequence[Sequence[Any]],
164
+ *,
165
+ alpha: float = 0.05,
166
+ ) -> ProbabilisticFiniteAutomaton:
167
+ """Learn a probabilistic finite automaton from positive strings by ALERGIA.
168
+
169
+ Parameters
170
+ ----------
171
+ samples
172
+ Observed strings drawn from the target process (e.g. realizations, or a
173
+ long sequence split into windows).
174
+ alpha
175
+ Significance level of the Hoeffding compatibility test. Smaller ``alpha``
176
+ merges more aggressively (fewer states); larger ``alpha`` is more
177
+ conservative.
178
+
179
+ Returns
180
+ -------
181
+ ProbabilisticFiniteAutomaton
182
+ A row-stochastic generator for the symbol process, with per-state
183
+ transition probabilities estimated from the merged frequencies
184
+ :cite:`Carrasco1994`.
185
+ """
186
+ strings = [tuple(word) for word in samples]
187
+ if not strings:
188
+ raise ValueError("at least one sample string is required")
189
+ if not 0.0 < alpha < 1.0:
190
+ raise ValueError("alpha must lie in (0, 1)")
191
+
192
+ fpta, alphabet = _build_fpta(strings)
193
+
194
+ red: list[int] = [0]
195
+ while True:
196
+ red_reps = {fpta.find(r) for r in red}
197
+ blue: int | None = None
198
+ for r in red:
199
+ rep = fpta.find(r)
200
+ for symbol in sorted(fpta.tchild[rep], key=repr):
201
+ child = fpta.find(fpta.tchild[rep][symbol])
202
+ if child not in red_reps:
203
+ blue = child
204
+ break
205
+ if blue is not None:
206
+ break
207
+ if blue is None:
208
+ break
209
+
210
+ merged = False
211
+ for r in red:
212
+ rep = fpta.find(r)
213
+ if rep == blue:
214
+ continue
215
+ if _compatible(fpta, rep, blue, alpha, set()):
216
+ _merge(fpta, rep, blue)
217
+ merged = True
218
+ break
219
+ if not merged:
220
+ red.append(blue)
221
+
222
+ return _to_pfa(fpta, red, alphabet)
@@ -0,0 +1,376 @@
1
+ """Automata constructions: reverse, determinize, and DFA minimization."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections import deque
6
+ from collections.abc import Hashable, Sequence
7
+ from typing import Any, Literal, TypeVar
8
+
9
+ from sofic.automata.base import LabeledAutomaton
10
+ from sofic.automata.dfa import DFA
11
+ from sofic.automata.nfa import NFA
12
+ from sofic.graph import ATTR_SYMBOL, EPSILON
13
+
14
+ MinimizationAlgorithm = Literal["hopcroft", "moore", "brzozowski"]
15
+
16
+ _TRAP = object()
17
+
18
+ L = TypeVar("L", bound=LabeledAutomaton)
19
+
20
+
21
+ def trim(aut: L) -> L: # noqa: UP047 - keep Python 3.11 compatibility.
22
+ """Remove states not reachable from initials or not coaccessible to acceptors."""
23
+ reachable = _forward_reachable(aut)
24
+ coaccessible = _backward_coaccessible(aut)
25
+ keep = reachable & coaccessible
26
+
27
+ result = aut.copy()
28
+ for state in list(result.states()):
29
+ if state not in keep:
30
+ result.graph.nx.remove_node(state)
31
+
32
+ result.initial_states = frozenset(s for s in aut.initial_states if s in keep)
33
+ result.accepting_states = frozenset(s for s in aut.accepting_states if s in keep)
34
+ return result
35
+
36
+
37
+ def complete(dfa: DFA, alphabet: frozenset[Any] | None = None) -> DFA:
38
+ """Add a trap state so every state has one outgoing transition per symbol."""
39
+ symbols = alphabet if alphabet is not None else _effective_alphabet(dfa)
40
+ result = dfa.copy()
41
+ trap = _TRAP
42
+ if trap not in result.graph.nx:
43
+ result.graph.add_state(trap)
44
+
45
+ for state in list(result.states()):
46
+ if state == trap:
47
+ continue
48
+ for symbol in symbols:
49
+ if not result.delta(state, symbol):
50
+ result.add_transition(state, trap, symbol)
51
+
52
+ for symbol in symbols:
53
+ result.add_transition(trap, trap, symbol)
54
+
55
+ if symbols:
56
+ result.input_alphabet = frozenset(symbols) | result.input_alphabet
57
+ return result
58
+
59
+
60
+ def reverse(aut: NFA | DFA) -> NFA:
61
+ """Return an NFA recognizing the reversed language.
62
+
63
+ For a generic :class:`~sofic.base.StateMachine`, use
64
+ :func:`~sofic.operations.reverse` instead.
65
+ """
66
+ return aut.reverse()
67
+
68
+
69
+ def determinize(nfa: NFA, *, alphabet: frozenset[Any] | None = None) -> DFA:
70
+ """Subset construction with epsilon closure."""
71
+ symbols = alphabet if alphabet is not None else _effective_alphabet(nfa)
72
+ start = frozenset(nfa.epsilon_closure(set(nfa.initial_states)))
73
+
74
+ subsets: dict[frozenset[Hashable], frozenset[Hashable]] = {start: start}
75
+ queue: deque[frozenset[Hashable]] = deque([start])
76
+ edges: list[tuple[frozenset[Hashable], frozenset[Hashable], Any]] = []
77
+
78
+ while queue:
79
+ current = queue.popleft()
80
+ for symbol in symbols:
81
+ next_raw: set[Hashable] = set()
82
+ for state in current:
83
+ next_raw.update(nfa.delta(state, symbol))
84
+ target = frozenset(nfa.epsilon_closure(next_raw))
85
+ edges.append((current, target, symbol))
86
+ if target not in subsets:
87
+ subsets[target] = target
88
+ queue.append(target)
89
+
90
+ dfa = DFA(
91
+ input_alphabet=symbols,
92
+ initial_states=frozenset({start}),
93
+ accepting_states=frozenset(s for s in subsets if s & nfa.accepting_states),
94
+ )
95
+ for subset in subsets:
96
+ dfa.graph.add_state(subset)
97
+ for source, target, symbol in edges:
98
+ dfa.add_transition(source, target, symbol)
99
+ return dfa
100
+
101
+
102
+ def minimize(
103
+ aut: DFA | NFA,
104
+ *,
105
+ algorithm: MinimizationAlgorithm = "hopcroft",
106
+ alphabet: frozenset[Any] | None = None,
107
+ ) -> DFA:
108
+ """Return a minimal DFA equivalent to ``aut``."""
109
+ if algorithm == "brzozowski":
110
+ return minimize_brzozowski(aut, alphabet=alphabet)
111
+ dfa = aut if isinstance(aut, DFA) else determinize(aut, alphabet=alphabet)
112
+ if algorithm == "hopcroft":
113
+ return minimize_hopcroft(dfa, alphabet=alphabet)
114
+ if algorithm == "moore":
115
+ return minimize_moore(dfa, alphabet=alphabet)
116
+ raise ValueError(f"unknown minimization algorithm {algorithm!r}")
117
+
118
+
119
+ def minimize_brzozowski(
120
+ aut: DFA | NFA,
121
+ *,
122
+ alphabet: frozenset[Any] | None = None,
123
+ ) -> DFA:
124
+ """Minimize via Brzozowski double reversal: det(rev(det(rev(A))))."""
125
+ nfa = aut if isinstance(aut, NFA) else _dfa_as_nfa(aut)
126
+ return determinize(trim(nfa).reverse().determinize(alphabet=alphabet).reverse().determinize(alphabet=alphabet))
127
+
128
+
129
+ def minimize_moore(dfa: DFA, *, alphabet: frozenset[Any] | None = None) -> DFA:
130
+ """Minimize a DFA using Moore (1961) partition refinement."""
131
+ symbols = alphabet if alphabet is not None else _effective_alphabet(dfa)
132
+ work = complete(trim(dfa), symbols)
133
+ states = sorted(work.states(), key=repr)
134
+ if not states:
135
+ return work
136
+
137
+ partition = _initial_partition(states, work.accepting_states)
138
+ changed = True
139
+ while changed:
140
+ changed = False
141
+ new_partition: list[set[Hashable]] = []
142
+ for block in partition:
143
+ refined: list[set[Hashable]] = [set(block)]
144
+ for symbol in sorted(symbols, key=repr):
145
+ next_refined: list[set[Hashable]] = []
146
+ for piece in refined:
147
+ groups: dict[int, set[Hashable]] = {}
148
+ for state in piece:
149
+ successor = _dfa_successor(work, state, symbol)
150
+ index = -1 if successor is None else _block_index(partition, successor)
151
+ groups.setdefault(index, set()).add(state)
152
+ next_refined.extend(groups.values())
153
+ refined = next_refined
154
+ if len(refined) > 1:
155
+ changed = True
156
+ new_partition.extend(refined)
157
+ partition = new_partition
158
+
159
+ return _quotient_from_partition(work, partition, symbols)
160
+
161
+
162
+ def minimize_hopcroft(dfa: DFA, *, alphabet: frozenset[Any] | None = None) -> DFA:
163
+ """Minimize a DFA using Hopcroft's algorithm."""
164
+ symbols = alphabet if alphabet is not None else _effective_alphabet(dfa)
165
+ work = complete(trim(dfa), symbols)
166
+ states = sorted(work.states(), key=repr)
167
+ if not states:
168
+ return work
169
+
170
+ accepting = set(work.accepting_states)
171
+ partition: list[set[Hashable]] = []
172
+ if accepting:
173
+ partition.append(accepting & set(states))
174
+ non_accepting = set(states) - accepting
175
+ if non_accepting:
176
+ partition.append(non_accepting)
177
+
178
+ pred = _inverse_transitions(work, states, symbols)
179
+ worklist: list[set[Hashable]] = [block.copy() for block in partition]
180
+
181
+ while worklist:
182
+ focus = worklist.pop()
183
+ for symbol in symbols:
184
+ predecessors: set[Hashable] = set()
185
+ for state in focus:
186
+ predecessors.update(pred[state][symbol])
187
+ refined_partition: list[set[Hashable]] = []
188
+ for block in partition:
189
+ intersection = block & predecessors
190
+ difference = block - predecessors
191
+ if intersection and difference:
192
+ refined_partition.append(intersection)
193
+ refined_partition.append(difference)
194
+ if block in worklist:
195
+ worklist.remove(block)
196
+ worklist.append(intersection)
197
+ worklist.append(difference)
198
+ else:
199
+ if len(intersection) <= len(difference):
200
+ worklist.append(intersection)
201
+ else:
202
+ worklist.append(difference)
203
+ else:
204
+ refined_partition.append(block)
205
+ partition = refined_partition
206
+
207
+ return _quotient_from_partition(work, partition, symbols)
208
+
209
+
210
+ def equivalent(
211
+ aut1: LabeledAutomaton,
212
+ aut2: LabeledAutomaton,
213
+ alphabet: frozenset[Any],
214
+ ) -> bool:
215
+ """Return whether two automata recognize the same language over ``alphabet``."""
216
+ d1 = minimize(_to_nfa(aut1), alphabet=alphabet, algorithm="hopcroft")
217
+ d2 = minimize(_to_nfa(aut2), alphabet=alphabet, algorithm="hopcroft")
218
+ return _isomorphic_minimal_dfa(d1, d2, alphabet)
219
+
220
+
221
+ def _to_nfa(aut: LabeledAutomaton) -> NFA:
222
+ if isinstance(aut, NFA):
223
+ return aut
224
+ if isinstance(aut, DFA):
225
+ return _dfa_as_nfa(aut)
226
+ raise TypeError(f"unsupported automaton type {type(aut)!r}")
227
+
228
+
229
+ def _dfa_as_nfa(dfa: DFA) -> NFA:
230
+ nfa = NFA(
231
+ input_alphabet=dfa.input_alphabet,
232
+ initial_states=dfa.initial_states,
233
+ accepting_states=dfa.accepting_states,
234
+ graph=dfa.graph.copy(),
235
+ )
236
+ return nfa
237
+
238
+
239
+ def _effective_alphabet(aut: LabeledAutomaton) -> frozenset[Any]:
240
+ symbols = {symbol for symbol in aut.input_alphabet if symbol is not EPSILON}
241
+ if symbols:
242
+ return frozenset(symbols)
243
+ for transition in aut.transitions():
244
+ symbol = transition.data.get(ATTR_SYMBOL)
245
+ if symbol is not None and symbol is not EPSILON:
246
+ symbols.add(symbol)
247
+ return frozenset(symbols)
248
+
249
+
250
+ def _forward_reachable(aut: LabeledAutomaton) -> set[Hashable]:
251
+ if not aut.initial_states:
252
+ return set()
253
+ seed = set(aut.epsilon_closure(set(aut.initial_states)))
254
+ return set(aut.graph.forward_reachable(seed))
255
+
256
+
257
+ def _backward_coaccessible(aut: LabeledAutomaton) -> set[Hashable]:
258
+ predecessors: dict[Hashable, set[Hashable]] = {state: set() for state in aut.states()}
259
+ for transition in aut.transitions():
260
+ predecessors.setdefault(transition.target, set()).add(transition.source)
261
+
262
+ coaccessible = set(aut.accepting_states)
263
+ queue = deque(coaccessible)
264
+ while queue:
265
+ state = queue.popleft()
266
+ for predecessor in predecessors.get(state, ()):
267
+ if predecessor not in coaccessible:
268
+ coaccessible.add(predecessor)
269
+ queue.append(predecessor)
270
+ return coaccessible
271
+
272
+
273
+ def _dfa_successor(dfa: DFA, state: Hashable, symbol: Any) -> Hashable | None:
274
+ successors = dfa.delta(state, symbol)
275
+ if not successors:
276
+ return None
277
+ return next(iter(successors))
278
+
279
+
280
+ def _initial_partition(states: Sequence[Hashable], accepting: frozenset[Hashable]) -> list[set[Hashable]]:
281
+ accepting_block = set(accepting) & set(states)
282
+ non_accepting = set(states) - accepting_block
283
+ partition: list[set[Hashable]] = []
284
+ if accepting_block:
285
+ partition.append(accepting_block)
286
+ if non_accepting:
287
+ partition.append(non_accepting)
288
+ return partition
289
+
290
+
291
+ def _block_index(partition: list[set[Hashable]], state: Hashable) -> int:
292
+ for index, block in enumerate(partition):
293
+ if state in block:
294
+ return index
295
+ raise KeyError(state)
296
+
297
+
298
+ def _inverse_transitions(
299
+ dfa: DFA,
300
+ states: Sequence[Hashable],
301
+ symbols: frozenset[Any],
302
+ ) -> dict[Hashable, dict[Any, set[Hashable]]]:
303
+ pred: dict[Hashable, dict[Any, set[Hashable]]] = {state: {symbol: set() for symbol in symbols} for state in states}
304
+ for state in states:
305
+ for symbol in symbols:
306
+ target = _dfa_successor(dfa, state, symbol)
307
+ if target is not None and target in pred:
308
+ pred[target][symbol].add(state)
309
+ return pred
310
+
311
+
312
+ def _quotient_from_partition(dfa: DFA, partition: list[set[Hashable]], symbols: frozenset[Any]) -> DFA:
313
+ blocks = [block for block in partition if block]
314
+ block_of = {state: index for index, block in enumerate(blocks) for state in block}
315
+ representatives = [min(block, key=repr) for block in blocks]
316
+ rep_for_block = {index: frozenset({representatives[index]}) for index in range(len(blocks))}
317
+
318
+ initial_block = block_of[next(iter(dfa.initial_states))] if dfa.initial_states else 0
319
+ accepting_blocks = frozenset(
320
+ rep_for_block[index] for index, block in enumerate(blocks) if block & dfa.accepting_states
321
+ )
322
+
323
+ result = DFA(
324
+ input_alphabet=symbols,
325
+ initial_states=frozenset({rep_for_block[initial_block]}),
326
+ accepting_states=accepting_blocks,
327
+ )
328
+ for index, _rep in enumerate(representatives):
329
+ result.graph.add_state(rep_for_block[index])
330
+
331
+ for index, rep in enumerate(representatives):
332
+ source = rep_for_block[index]
333
+ for symbol in symbols:
334
+ target_state = _dfa_successor(dfa, rep, symbol)
335
+ if target_state is None:
336
+ continue
337
+ target_block = block_of[target_state]
338
+ result.add_transition(source, rep_for_block[target_block], symbol)
339
+
340
+ return trim(result)
341
+
342
+
343
+ def _isomorphic_minimal_dfa(d1: DFA, d2: DFA, alphabet: frozenset[Any]) -> bool:
344
+ states1 = list(d1.states())
345
+ states2 = list(d2.states())
346
+ if len(states1) != len(states2):
347
+ return False
348
+ if not d1.initial_states or not d2.initial_states:
349
+ return not d1.initial_states and not d2.initial_states
350
+
351
+ start1 = next(iter(d1.initial_states))
352
+ start2 = next(iter(d2.initial_states))
353
+ if (start1 in d1.accepting_states) != (start2 in d2.accepting_states):
354
+ return False
355
+
356
+ mapping: dict[Hashable, Hashable] = {start1: start2}
357
+ queue = deque([start1])
358
+ while queue:
359
+ state1 = queue.popleft()
360
+ state2 = mapping[state1]
361
+ for symbol in sorted(alphabet, key=repr):
362
+ succ1 = _dfa_successor(d1, state1, symbol)
363
+ succ2 = _dfa_successor(d2, state2, symbol)
364
+ if succ1 is None and succ2 is None:
365
+ continue
366
+ if succ1 is None or succ2 is None:
367
+ return False
368
+ if succ1 in mapping:
369
+ if mapping[succ1] != succ2:
370
+ return False
371
+ else:
372
+ if (succ1 in d1.accepting_states) != (succ2 in d2.accepting_states):
373
+ return False
374
+ mapping[succ1] = succ2
375
+ queue.append(succ1)
376
+ return len(mapping) == len(states1)
@@ -0,0 +1,58 @@
1
+ """Atomic and átomaton NFA presentations."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import TYPE_CHECKING, Any
6
+
7
+ from sofic.automata.dfa import DFA
8
+ from sofic.automata.languages.base import RegularLanguage
9
+ from sofic.automata.nfa import NFA
10
+
11
+ if TYPE_CHECKING:
12
+ from sofic.automata.observation import ObservationTable
13
+ from sofic.automata.rfsa import CanonicalRFSA
14
+
15
+
16
+ class AtomicAutomaton(NFA):
17
+ """NFA whose states accept unions of atoms."""
18
+
19
+ def validate(self) -> None:
20
+ super().validate()
21
+ # Phase 2: verify right languages are unions of atoms
22
+
23
+
24
+ class Atomaton(AtomicAutomaton):
25
+ """Canonical átomaton whose states are all atoms of L."""
26
+
27
+ @classmethod
28
+ def from_language(cls, language: RegularLanguage | NFA, **kwargs: Any) -> Atomaton:
29
+ from sofic.automata.canonical_extraction import atomaton_from_language
30
+
31
+ return atomaton_from_language(language)
32
+
33
+ def to_minimal_dfa_via_double_reversal(self) -> DFA:
34
+ from sofic.automata.algorithms import minimize
35
+
36
+ return minimize(self, algorithm="brzozowski")
37
+
38
+
39
+ class MaximizedPrimeAtomaton(AtomicAutomaton):
40
+ """Maximized prime átomaton — dual of the canonical RFSA."""
41
+
42
+ @classmethod
43
+ def from_language(cls, language: RegularLanguage | NFA, **kwargs: Any) -> MaximizedPrimeAtomaton:
44
+ from sofic.automata.canonical_extraction import maximized_prime_atomaton_from_language
45
+
46
+ return maximized_prime_atomaton_from_language(language)
47
+
48
+ @classmethod
49
+ def from_observation_table(cls, table: ObservationTable, **kwargs: Any) -> MaximizedPrimeAtomaton:
50
+ from sofic.automata.canonical_extraction import observation_to_maximized_prime_atomaton
51
+
52
+ return observation_to_maximized_prime_atomaton(table)
53
+
54
+ @classmethod
55
+ def from_canonical_rfsa(cls, rfsa: CanonicalRFSA, **kwargs: Any) -> MaximizedPrimeAtomaton:
56
+ from sofic.automata.canonical_dual import dual_atomaton_from_rfsa
57
+
58
+ return dual_atomaton_from_rfsa(rfsa)