openprocess 0.7.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (173) hide show
  1. cpnpy/__init__.py +46 -0
  2. openprocess/__init__.py +57 -0
  3. openprocess/analysis/__init__.py +0 -0
  4. openprocess/analysis/state_space.py +521 -0
  5. openprocess/analysis/state_space_process.py +251 -0
  6. openprocess/cli.py +742 -0
  7. openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/answer.pnml +31 -0
  8. openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/question.md +38 -0
  9. openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/answer.pnml +27 -0
  10. openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/net.pnml +29 -0
  11. openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/question.md +65 -0
  12. openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/log.txt +1 -0
  13. openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/question.md +67 -0
  14. openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/question.md +75 -0
  15. openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/ts.txt +5 -0
  16. openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/net.pnml +27 -0
  17. openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/question.md +65 -0
  18. openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/log.txt +1 -0
  19. openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/question.md +62 -0
  20. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/log.txt +1 -0
  21. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m1.pnml +36 -0
  22. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m2.pnml +28 -0
  23. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m3.pnml +30 -0
  24. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/net.pnml +36 -0
  25. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/question.md +69 -0
  26. openprocess/exercises/pack.md +14 -0
  27. openprocess/flow/__init__.py +50 -0
  28. openprocess/flow/box.py +466 -0
  29. openprocess/flow/boxes/__init__.py +7 -0
  30. openprocess/flow/boxes/check.py +119 -0
  31. openprocess/flow/boxes/compare.py +16 -0
  32. openprocess/flow/boxes/cpn.py +53 -0
  33. openprocess/flow/boxes/discover.py +124 -0
  34. openprocess/flow/boxes/filter.py +80 -0
  35. openprocess/flow/boxes/input.py +124 -0
  36. openprocess/flow/boxes/output.py +52 -0
  37. openprocess/flow/boxes/predict.py +186 -0
  38. openprocess/flow/boxes/science.py +159 -0
  39. openprocess/flow/boxes/sweeps.py +18 -0
  40. openprocess/flow/convert.py +187 -0
  41. openprocess/flow/datasets.py +198 -0
  42. openprocess/flow/explain.py +115 -0
  43. openprocess/flow/library.py +222 -0
  44. openprocess/flow/record.py +385 -0
  45. openprocess/flow/runner.py +357 -0
  46. openprocess/flow/sweep.py +92 -0
  47. openprocess/flow/types.py +290 -0
  48. openprocess/flow/workflow.py +628 -0
  49. openprocess/gui/__init__.py +0 -0
  50. openprocess/gui/app.py +90 -0
  51. openprocess/gui/arc_editing.py +295 -0
  52. openprocess/gui/canvas.py +1414 -0
  53. openprocess/gui/flow/__init__.py +8 -0
  54. openprocess/gui/flow/canvas.py +854 -0
  55. openprocess/gui/flow/page.py +972 -0
  56. openprocess/gui/flow/templates.py +131 -0
  57. openprocess/gui/flow/viewers.py +665 -0
  58. openprocess/gui/items.py +1275 -0
  59. openprocess/gui/learn/answer_boxes.py +978 -0
  60. openprocess/gui/learn/concealment.py +91 -0
  61. openprocess/gui/learn/mode.py +1181 -0
  62. openprocess/gui/panning.py +241 -0
  63. openprocess/gui/resources/openprocess-icon.png +0 -0
  64. openprocess/gui/studio/__init__.py +1 -0
  65. openprocess/gui/studio/__main__.py +3 -0
  66. openprocess/gui/studio/app.py +4031 -0
  67. openprocess/gui/studio/charts.py +115 -0
  68. openprocess/gui/studio/compare_page.py +487 -0
  69. openprocess/gui/studio/cpn_page.py +1858 -0
  70. openprocess/gui/studio/definition_view.py +284 -0
  71. openprocess/gui/studio/derivation_view.py +421 -0
  72. openprocess/gui/studio/documents.py +152 -0
  73. openprocess/gui/studio/dotted_chart.py +1401 -0
  74. openprocess/gui/studio/file_dialogs.py +143 -0
  75. openprocess/gui/studio/filter_dialog.py +247 -0
  76. openprocess/gui/studio/graph_builders.py +176 -0
  77. openprocess/gui/studio/graph_view.py +682 -0
  78. openprocess/gui/studio/instances.py +413 -0
  79. openprocess/gui/studio/log_editor.py +675 -0
  80. openprocess/gui/studio/log_page.py +800 -0
  81. openprocess/gui/studio/markdown_view.py +127 -0
  82. openprocess/gui/studio/mathtext.py +260 -0
  83. openprocess/gui/studio/ml_highlighter.py +75 -0
  84. openprocess/gui/studio/model_page.py +760 -0
  85. openprocess/gui/studio/net_comparison.py +124 -0
  86. openprocess/gui/studio/notes_overlay.py +275 -0
  87. openprocess/gui/studio/petri_page.py +844 -0
  88. openprocess/gui/studio/regions_view.py +502 -0
  89. openprocess/gui/studio/sidebar.py +149 -0
  90. openprocess/gui/studio/style.py +503 -0
  91. openprocess/gui/studio/tool_icons.py +134 -0
  92. openprocess/gui/studio/updates.py +439 -0
  93. openprocess/gui/studio/widgets.py +899 -0
  94. openprocess/gui/studio/workers.py +60 -0
  95. openprocess/gui/studio/workspace.py +447 -0
  96. openprocess/gui/theme.py +394 -0
  97. openprocess/gui/tidy.py +86 -0
  98. openprocess/io/__init__.py +0 -0
  99. openprocess/io/cpn_reader.py +389 -0
  100. openprocess/io/cpn_writer.py +357 -0
  101. openprocess/learn/__init__.py +23 -0
  102. openprocess/learn/answers.py +188 -0
  103. openprocess/learn/checks.py +953 -0
  104. openprocess/learn/computed.py +1180 -0
  105. openprocess/learn/context.py +145 -0
  106. openprocess/learn/exam.py +169 -0
  107. openprocess/learn/exercise-packs.md +325 -0
  108. openprocess/learn/importer.py +216 -0
  109. openprocess/learn/notation.py +474 -0
  110. openprocess/learn/pack.py +511 -0
  111. openprocess/learn/sheet.py +296 -0
  112. openprocess/mining/__init__.py +73 -0
  113. openprocess/mining/analysis.py +689 -0
  114. openprocess/mining/columns.py +282 -0
  115. openprocess/mining/compare_nets.py +246 -0
  116. openprocess/mining/conformance/__init__.py +0 -0
  117. openprocess/mining/conformance/alignments.py +263 -0
  118. openprocess/mining/conformance/quality.py +145 -0
  119. openprocess/mining/conformance/token_replay.py +252 -0
  120. openprocess/mining/csv_import.py +222 -0
  121. openprocess/mining/definitions.py +584 -0
  122. openprocess/mining/dfg.py +187 -0
  123. openprocess/mining/discovery/__init__.py +0 -0
  124. openprocess/mining/discovery/alpha.py +168 -0
  125. openprocess/mining/discovery/heuristics.py +332 -0
  126. openprocess/mining/discovery/inductive.py +477 -0
  127. openprocess/mining/discovery/state_regions.py +62 -0
  128. openprocess/mining/filtering.py +237 -0
  129. openprocess/mining/footprint.py +183 -0
  130. openprocess/mining/invariants.py +191 -0
  131. openprocess/mining/layout.py +279 -0
  132. openprocess/mining/log.py +364 -0
  133. openprocess/mining/petrinet.py +354 -0
  134. openprocess/mining/playout.py +75 -0
  135. openprocess/mining/pm4py_bridge.py +82 -0
  136. openprocess/mining/pnml.py +223 -0
  137. openprocess/mining/processtree.py +216 -0
  138. openprocess/mining/regions.py +476 -0
  139. openprocess/mining/stats.py +160 -0
  140. openprocess/mining/structure.py +374 -0
  141. openprocess/mining/transition_system.py +409 -0
  142. openprocess/mining/xes.py +399 -0
  143. openprocess/ml/__init__.py +0 -0
  144. openprocess/ml/ast_nodes.py +332 -0
  145. openprocess/ml/builtins.py +364 -0
  146. openprocess/ml/colorsets.py +522 -0
  147. openprocess/ml/errors.py +60 -0
  148. openprocess/ml/evaluator.py +754 -0
  149. openprocess/ml/lexer.py +277 -0
  150. openprocess/ml/multiset.py +417 -0
  151. openprocess/ml/parser.py +737 -0
  152. openprocess/ml/values.py +319 -0
  153. openprocess/model/__init__.py +0 -0
  154. openprocess/model/declarations.py +617 -0
  155. openprocess/model/examples.py +98 -0
  156. openprocess/model/net.py +701 -0
  157. openprocess/model/plain.py +192 -0
  158. openprocess/references.py +280 -0
  159. openprocess/sim/__init__.py +0 -0
  160. openprocess/sim/binding.py +620 -0
  161. openprocess/sim/export.py +66 -0
  162. openprocess/sim/simulator.py +315 -0
  163. openprocess/teaching/__init__.py +4 -0
  164. openprocess/teaching/answers.py +4 -0
  165. openprocess/teaching/checks.py +5 -0
  166. openprocess/teaching/pack.py +4 -0
  167. openprocess/teaching/sheet.py +4 -0
  168. openprocess-0.7.0.dist-info/METADATA +927 -0
  169. openprocess-0.7.0.dist-info/RECORD +173 -0
  170. openprocess-0.7.0.dist-info/WHEEL +5 -0
  171. openprocess-0.7.0.dist-info/entry_points.txt +6 -0
  172. openprocess-0.7.0.dist-info/licenses/LICENSE +21 -0
  173. openprocess-0.7.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,477 @@
1
+ """The Inductive Miner (IM) and Inductive Miner – infrequent (IMf).
2
+
3
+ Reading: Leemans, Fahland & van der Aalst, "Discovering block-structured
4
+ process models from event logs" (course reading 4).
5
+
6
+ The idea: divide and conquer
7
+ ----------------------------
8
+ Look at the directly-follows graph of the log and try to find a **cut**: a
9
+ partition of the activities into groups ``Σ1, ..., Σn`` that matches one of
10
+ the four process-tree operators. If one is found, **split** the log into one
11
+ sub-log per group, recurse on each, and combine the results under that
12
+ operator. Recursion stops at **base cases** (a log with one activity, or
13
+ only empty traces). When no cut exists, a **fall-through** produces a
14
+ general but still correct model.
15
+
16
+ The four cuts, in the order they are tried
17
+ ------------------------------------------
18
+ × exclusive choice
19
+ The groups are the *connected components* of the DFG (ignoring edge
20
+ direction). No trace mixes two groups.
21
+ → sequence
22
+ Every activity of an earlier group can reach every activity of a later
23
+ group, and never the other way round. Computed from the strongly
24
+ connected components, merging components that are mutually unreachable.
25
+ ∧ parallel
26
+ Between two different groups, *every* pair of activities directly
27
+ follows each other in both directions, and every group contains a start
28
+ and an end activity. Computed as the connected components of the
29
+ *negated* DFG (an edge where the pair is **not** in both directions).
30
+ ↺ loop
31
+ The *body* group contains all start and end activities; each *redo*
32
+ group is entered only from end activities and exits only to start
33
+ activities.
34
+
35
+ Guarantees (IM): the result always *fits* the log perfectly and is always
36
+ *sound*. The price is that fall-throughs may overgeneralise.
37
+
38
+ IMf: handling infrequent behaviour
39
+ ----------------------------------
40
+ IM treats one odd trace the same as a thousand normal ones. IMf adds a
41
+ **noise threshold** ``f`` in [0, 1]: when no cut is found on the full DFG,
42
+ edges whose frequency is below ``f`` × (the most frequent outgoing edge of
43
+ the same activity) are removed and cut detection is retried. The log split
44
+ is then made *filtering-aware*: events that do not fit the chosen cut are
45
+ dropped. Fitness is no longer guaranteed to be 1, but the models are much
46
+ simpler. ``f = 0`` gives plain IM.
47
+ """
48
+
49
+ from __future__ import annotations
50
+
51
+ from collections import Counter, deque
52
+ from dataclasses import dataclass, field
53
+
54
+ from ..dfg import DFG, dfg_from_simple_log
55
+ from ..log import SimpleLog
56
+ from ..petrinet import PetriNet
57
+ from ..processtree import Operator, ProcessTree, to_petri_net
58
+
59
+
60
+ @dataclass
61
+ class Step:
62
+ """One recursion step, structured so the app can typeset it.
63
+
64
+ ``kind`` is ``"cut"``, ``"base"``, ``"fall"`` (fall-through) or
65
+ ``"filter"`` (IMf noise handling). ``operator`` is the process-tree
66
+ symbol produced (→ × ∧ ↺), ``groups`` the partition found by a cut, and
67
+ ``result`` the leaf of a base case.
68
+ """
69
+
70
+ depth: int
71
+ kind: str
72
+ text: str
73
+ operator: str | None = None
74
+ groups: list[list[str]] | None = None
75
+ result: str | None = None
76
+ detail: str | None = None
77
+ filtered: bool = False
78
+
79
+
80
+ @dataclass
81
+ class InductiveResult:
82
+ tree: ProcessTree
83
+ net: PetriNet
84
+ #: One line per recursion step: which cut or fall-through was applied.
85
+ trace_of_steps: list[str] = field(default_factory=list)
86
+ #: The same steps, structured (see :class:`Step`).
87
+ steps: list[Step] = field(default_factory=list)
88
+
89
+
90
+ def inductive_miner(log: SimpleLog, noise_threshold: float = 0.0,
91
+ name: str | None = None) -> InductiveResult:
92
+ """Discover a process tree (and its Petri net) with IM / IMf."""
93
+ miner = _Miner(noise_threshold)
94
+ tree = miner.mine(Counter({t: n for t, n in log.items() if n > 0}), depth=0).simplified()
95
+ label = name or ("IMf" if noise_threshold > 0 else "IM")
96
+ net = to_petri_net(tree, label)
97
+ net.info["algorithm"] = (f"Inductive Miner – infrequent (f = {noise_threshold:g})"
98
+ if noise_threshold > 0 else "Inductive Miner")
99
+ net.info["process tree"] = str(tree)
100
+ return InductiveResult(tree, net, miner.steps, miner.records)
101
+
102
+
103
+ # ---------------------------------------------------------------------------
104
+ # The recursion
105
+ # ---------------------------------------------------------------------------
106
+ class _Miner:
107
+ def __init__(self, noise_threshold: float) -> None:
108
+ self.f = noise_threshold
109
+ self.steps: list[str] = []
110
+ self.records: list[Step] = []
111
+
112
+ def note(self, depth: int, text: str, kind: str = "fall", **extra) -> None:
113
+ self.steps.append(" " * depth + text)
114
+ self.records.append(Step(depth, kind, text, **extra))
115
+
116
+ def mine(self, log: SimpleLog, depth: int) -> ProcessTree:
117
+ activities = sorted({a for trace in log for a in trace})
118
+ total = sum(log.values())
119
+ empty = log.get((), 0)
120
+
121
+ # ---- base cases --------------------------------------------------
122
+ if total == 0 or not activities:
123
+ self.note(depth, "base case: only empty traces → τ", "base", result="τ")
124
+ return ProcessTree.tau()
125
+ if len(activities) == 1 and empty == 0 and all(len(t) == 1 for t in log):
126
+ self.note(depth, f"base case: single activity → {activities[0]}", "base",
127
+ result=activities[0])
128
+ return ProcessTree.leaf(activities[0])
129
+
130
+ # ---- empty traces --------------------------------------------------
131
+ if empty:
132
+ if self.f > 0 and empty / total < self.f:
133
+ self.note(depth, f"IMf: ignoring {empty} infrequent empty trace(s)", "filter",
134
+ detail=f"{empty} of {total} traces are empty, below f = {self.f:g}")
135
+ log = Counter({t: n for t, n in log.items() if t})
136
+ return self.mine(log, depth)
137
+ self.note(depth, f"fall-through: empty traces → ×(τ, …) [{empty} empty]",
138
+ operator="×", detail=f"{empty} empty trace(s): the part may be skipped")
139
+ rest = Counter({t: n for t, n in log.items() if t})
140
+ return ProcessTree.node(Operator.XOR, [ProcessTree.tau(), self.mine(rest, depth + 1)])
141
+
142
+ # ---- cuts on the full DFG ---------------------------------------------
143
+ dfg = dfg_from_simple_log(log)
144
+ found = self.find_cut(dfg)
145
+ filtered = False
146
+ if found is None and self.f > 0:
147
+ found = self.find_cut(filter_dfg(dfg, self.f))
148
+ filtered = found is not None
149
+ if found is not None:
150
+ operator, groups = found
151
+ described = " | ".join("{" + ", ".join(sorted(g)) + "}" for g in groups)
152
+ self.note(depth, f"{operator.value} cut{' (on filtered DFG)' if filtered else ''}: "
153
+ f"{described}", "cut", operator=operator.value,
154
+ groups=[sorted(g) for g in groups], filtered=filtered)
155
+ sublogs = split_log(log, operator, groups, dfg)
156
+ children = [self.mine(sub, depth + 1) for sub in sublogs]
157
+ if operator is Operator.LOOP and len(children) > 2:
158
+ # ↺(body, r1, r2, …) is kept as a loop with several redos.
159
+ return ProcessTree.node(Operator.LOOP, children)
160
+ return ProcessTree.node(operator, children)
161
+
162
+ # ---- fall-throughs ---------------------------------------------------
163
+ return self.fall_through(log, dfg, activities, depth)
164
+
165
+ # -------------------------------------------------------------------------
166
+ def find_cut(self, dfg: DFG):
167
+ for finder, operator in ((xor_cut, Operator.XOR), (sequence_cut, Operator.SEQUENCE),
168
+ (parallel_cut, Operator.PARALLEL), (loop_cut, Operator.LOOP)):
169
+ groups = finder(dfg)
170
+ if groups is not None and len(groups) > 1:
171
+ return operator, groups
172
+ return None
173
+
174
+ def fall_through(self, log: SimpleLog, dfg: DFG, activities: list[str],
175
+ depth: int) -> ProcessTree:
176
+ # 1. Activity once per trace: it can be put in parallel with the rest.
177
+ for activity in activities:
178
+ if all(trace.count(activity) == 1 for trace in log):
179
+ self.note(depth, f"fall-through: '{activity}' occurs once per trace → ∧({activity}, …)",
180
+ operator="∧", detail=f"{activity} occurs exactly once in every trace")
181
+ rest = _project_out(log, activity)
182
+ return ProcessTree.node(Operator.PARALLEL,
183
+ [ProcessTree.leaf(activity), self.mine(rest, depth + 1)])
184
+
185
+ # 2. Activity concurrent: removing one activity makes a cut appear.
186
+ if len(activities) > 2:
187
+ for activity in activities:
188
+ rest = _project_out(log, activity)
189
+ if self.find_cut(dfg_from_simple_log(rest)) is not None:
190
+ self.note(depth, f"fall-through: activity concurrent '{activity}' → ∧({activity}, …)",
191
+ operator="∧", detail=f"without {activity} a cut exists")
192
+ return ProcessTree.node(Operator.PARALLEL, [
193
+ self.mine(_keep_only(log, {activity}), depth + 1),
194
+ self.mine(rest, depth + 1)])
195
+
196
+ # 3. Strict τ-loop: split traces where an end activity is followed by
197
+ # a start activity.
198
+ split = _split_on(log, lambda a, b: a in dfg.end and b in dfg.start)
199
+ if sum(split.values()) > sum(log.values()):
200
+ self.note(depth, "fall-through: strict τ-loop → ↺(…, τ)", operator="↺",
201
+ detail="split traces where an end activity is followed by a start activity")
202
+ return ProcessTree.node(Operator.LOOP, [self.mine(split, depth + 1), ProcessTree.tau()])
203
+
204
+ # 4. τ-loop: split traces before every start activity.
205
+ split = _split_on(log, lambda a, b: b in dfg.start)
206
+ if sum(split.values()) > sum(log.values()):
207
+ self.note(depth, "fall-through: τ-loop → ↺(…, τ)", operator="↺",
208
+ detail="split traces before every start activity")
209
+ return ProcessTree.node(Operator.LOOP, [self.mine(split, depth + 1), ProcessTree.tau()])
210
+
211
+ # 5. Flower model: anything goes, in any order, any number of times.
212
+ self.note(depth, "fall-through: flower model ↺(τ, ×(" + ", ".join(activities) + "))",
213
+ operator="↺", detail="nothing else applies: allow any order (flower model)")
214
+ return ProcessTree.node(Operator.LOOP, [
215
+ ProcessTree.tau(),
216
+ ProcessTree.node(Operator.XOR, [ProcessTree.leaf(a) for a in activities])
217
+ if len(activities) > 1 else ProcessTree.leaf(activities[0])])
218
+
219
+
220
+ # ---------------------------------------------------------------------------
221
+ # Graph helpers
222
+ # ---------------------------------------------------------------------------
223
+ def _undirected_components(nodes: list[str], connected) -> list[set[str]]:
224
+ """Connected components where ``connected(a, b)`` says whether a–b touch."""
225
+ remaining = set(nodes)
226
+ components = []
227
+ while remaining:
228
+ seed = min(remaining)
229
+ component = {seed}
230
+ queue = deque([seed])
231
+ remaining.discard(seed)
232
+ while queue:
233
+ node = queue.popleft()
234
+ for other in list(remaining):
235
+ if connected(node, other):
236
+ remaining.discard(other)
237
+ component.add(other)
238
+ queue.append(other)
239
+ components.append(component)
240
+ return components
241
+
242
+
243
+ def _reachability(dfg: DFG) -> dict[str, set[str]]:
244
+ """For every activity, the activities reachable from it via ≥ 1 edge."""
245
+ succ = {a: set() for a in dfg.activities}
246
+ for a, b in dfg.edges:
247
+ succ[a].add(b)
248
+ reach = {}
249
+ for start in dfg.activities:
250
+ seen: set[str] = set()
251
+ queue = deque(succ[start])
252
+ while queue:
253
+ node = queue.popleft()
254
+ if node not in seen:
255
+ seen.add(node)
256
+ queue.extend(succ[node])
257
+ reach[start] = seen
258
+ return reach
259
+
260
+
261
+ # ---------------------------------------------------------------------------
262
+ # Cut detection
263
+ # ---------------------------------------------------------------------------
264
+ def xor_cut(dfg: DFG) -> list[set[str]] | None:
265
+ nodes = sorted(dfg.activities)
266
+ edges = set(dfg.edges)
267
+ groups = _undirected_components(nodes, lambda a, b: (a, b) in edges or (b, a) in edges)
268
+ return groups if len(groups) > 1 else None
269
+
270
+
271
+ def sequence_cut(dfg: DFG) -> list[set[str]] | None:
272
+ reach = _reachability(dfg)
273
+ nodes = sorted(dfg.activities)
274
+ # Strongly connected components: a and b together iff each reaches the other.
275
+ groups: list[set[str]] = []
276
+ for node in nodes:
277
+ for group in groups:
278
+ other = next(iter(group))
279
+ if (other in reach[node] and node in reach[other]):
280
+ group.add(node)
281
+ break
282
+ else:
283
+ groups.append({node})
284
+
285
+ def reaches(g1: set[str], g2: set[str]) -> bool:
286
+ return any(b in reach[a] for a in g1 for b in g2)
287
+
288
+ # Merge groups that are mutually unreachable (neither can reach the other):
289
+ # they must sit in the same position of the sequence.
290
+ merged = True
291
+ while merged:
292
+ merged = False
293
+ for i in range(len(groups)):
294
+ for j in range(i + 1, len(groups)):
295
+ if not reaches(groups[i], groups[j]) and not reaches(groups[j], groups[i]):
296
+ groups[i] |= groups.pop(j)
297
+ merged = True
298
+ break
299
+ if merged:
300
+ break
301
+ if len(groups) < 2:
302
+ return None
303
+ # Order: g1 before g2 iff g1 reaches g2. Sorting by "number of groups
304
+ # reachable from me" (descending) gives a topological order.
305
+ # (Scores are computed first: during list.sort() the list looks empty to
306
+ # a key function that refers back to it.)
307
+ score = [sum(1 for other in groups if other is not g and reaches(g, other)) for g in groups]
308
+ groups = [g for _, g in sorted(zip(score, groups), key=lambda item: -item[0])]
309
+ # Validate: every earlier group reaches every later one and never back.
310
+ for i in range(len(groups)):
311
+ for j in range(i + 1, len(groups)):
312
+ if not reaches(groups[i], groups[j]) or reaches(groups[j], groups[i]):
313
+ return None
314
+ return groups
315
+
316
+
317
+ def parallel_cut(dfg: DFG) -> list[set[str]] | None:
318
+ nodes = sorted(dfg.activities)
319
+ edges = set(dfg.edges)
320
+ # Negated graph: connect a, b unless they directly follow in both directions.
321
+ groups = _undirected_components(
322
+ nodes, lambda a, b: not ((a, b) in edges and (b, a) in edges))
323
+ if len(groups) < 2:
324
+ return None
325
+ # Every group needs a start and an end activity; merge groups that lack one.
326
+ good = [g for g in groups if g & set(dfg.start) and g & set(dfg.end)]
327
+ bad = [g for g in groups if not (g & set(dfg.start) and g & set(dfg.end))]
328
+ if not good:
329
+ return None
330
+ for group in bad:
331
+ good[0] |= group
332
+ if len(good) < 2:
333
+ return None
334
+ # Re-check the defining property after merging.
335
+ for i, g1 in enumerate(good):
336
+ for g2 in good[i + 1:]:
337
+ if not all((a, b) in edges and (b, a) in edges for a in g1 for b in g2):
338
+ return None
339
+ return good
340
+
341
+
342
+ def loop_cut(dfg: DFG) -> list[set[str]] | None:
343
+ starts, ends = set(dfg.start), set(dfg.end)
344
+ body = starts | ends
345
+ others = sorted(set(dfg.activities) - body)
346
+ if not others:
347
+ return None
348
+ edges = set(dfg.edges)
349
+ # Components of the graph *without* the start/end activities.
350
+ redo_candidates = _undirected_components(
351
+ others, lambda a, b: (a, b) in edges or (b, a) in edges)
352
+
353
+ body = set(body)
354
+ redos: list[set[str]] = []
355
+ for group in redo_candidates:
356
+ belongs_to_body = False
357
+ for a in group:
358
+ # Entered from a start activity (other than via an end) → body.
359
+ if any((s, a) in edges for s in starts - ends):
360
+ belongs_to_body = True
361
+ # Leaves to an end activity (other than a start) → body.
362
+ if any((a, e) in edges for e in ends - starts):
363
+ belongs_to_body = True
364
+ # If some end activity leads to a, every end activity must.
365
+ if any((e, a) in edges for e in ends) and not all((e, a) in edges for e in ends):
366
+ belongs_to_body = True
367
+ # If a leads to some start activity, it must lead to all of them.
368
+ if any((a, s) in edges for s in starts) and not all((a, s) in edges for s in starts):
369
+ belongs_to_body = True
370
+ if belongs_to_body:
371
+ body |= group
372
+ else:
373
+ redos.append(group)
374
+ # Activities inside the body that are fed only by body nodes are fine;
375
+ # redo groups must actually connect end → redo → start.
376
+ redos = [g for g in redos
377
+ if any((e, a) in edges for e in ends for a in g)
378
+ and any((a, s) in edges for a in g for s in starts)]
379
+ leftover = set(dfg.activities) - body - set().union(*redos) if redos else set()
380
+ body |= leftover
381
+ if not redos:
382
+ return None
383
+ return [body] + redos
384
+
385
+
386
+ def filter_dfg(dfg: DFG, threshold: float) -> DFG:
387
+ """IMf filtering: drop edges below ``threshold`` × strongest outgoing edge."""
388
+ strongest: dict[str, int] = Counter()
389
+ for (a, _), n in dfg.edges.items():
390
+ strongest[a] = max(strongest[a], n)
391
+ for a, n in dfg.end.items():
392
+ strongest[a] = max(strongest[a], n)
393
+ result = DFG(activities=Counter(dfg.activities), trace_count=dfg.trace_count)
394
+ result.edges = Counter({(a, b): n for (a, b), n in dfg.edges.items()
395
+ if n >= threshold * strongest[a]})
396
+ max_start = max(dfg.start.values(), default=0)
397
+ max_end = max(dfg.end.values(), default=0)
398
+ result.start = Counter({a: n for a, n in dfg.start.items() if n >= threshold * max_start})
399
+ result.end = Counter({a: n for a, n in dfg.end.items() if n >= threshold * strongest[a]
400
+ or n >= threshold * max_end})
401
+ return result
402
+
403
+
404
+ # ---------------------------------------------------------------------------
405
+ # Log splitting
406
+ # ---------------------------------------------------------------------------
407
+ def _project_out(log: SimpleLog, activity: str) -> SimpleLog:
408
+ result: SimpleLog = Counter()
409
+ for trace, n in log.items():
410
+ result[tuple(a for a in trace if a != activity)] += n
411
+ return result
412
+
413
+
414
+ def _keep_only(log: SimpleLog, keep: set[str]) -> SimpleLog:
415
+ result: SimpleLog = Counter()
416
+ for trace, n in log.items():
417
+ result[tuple(a for a in trace if a in keep)] += n
418
+ return result
419
+
420
+
421
+ def _split_on(log: SimpleLog, boundary) -> SimpleLog:
422
+ """Cut every trace between consecutive a, b where ``boundary(a, b)``."""
423
+ result: SimpleLog = Counter()
424
+ for trace, n in log.items():
425
+ piece: list[str] = []
426
+ for index, activity in enumerate(trace):
427
+ if piece and boundary(trace[index - 1], activity):
428
+ result[tuple(piece)] += n
429
+ piece = []
430
+ piece.append(activity)
431
+ result[tuple(piece)] += n
432
+ return result
433
+
434
+
435
+ def split_log(log: SimpleLog, operator: Operator, groups: list[set[str]],
436
+ dfg: DFG) -> list[SimpleLog]:
437
+ """Divide the log over the groups of a cut."""
438
+ sublogs: list[SimpleLog] = [Counter() for _ in groups]
439
+ if operator is Operator.XOR:
440
+ # Each trace goes to the group it overlaps most with; with IM every
441
+ # trace lies entirely in one group, with IMf stray events are dropped.
442
+ for trace, n in log.items():
443
+ best = max(range(len(groups)), key=lambda i: sum(a in groups[i] for a in trace))
444
+ sublogs[best][tuple(a for a in trace if a in groups[best])] += n
445
+ elif operator in (Operator.SEQUENCE, Operator.PARALLEL):
446
+ # Projection on each group. For a sequence cut found on the full DFG
447
+ # this is exactly the split into consecutive segments.
448
+ for trace, n in log.items():
449
+ for i, group in enumerate(groups):
450
+ sublogs[i][tuple(a for a in trace if a in group)] += n
451
+ elif operator is Operator.LOOP:
452
+ # Cut every trace into alternating body / redo segments. A body
453
+ # segment always comes first and last; if a trace starts or ends in
454
+ # a redo group (only possible with noise), an empty body segment is
455
+ # implied, which becomes an empty trace in the body log.
456
+ group_of = {a: i for i, g in enumerate(groups) for a in g}
457
+ for trace, n in log.items():
458
+ # 1. Chop the trace into maximal runs belonging to one group.
459
+ runs: list[tuple[int, list[str]]] = []
460
+ for activity in trace:
461
+ g = group_of.get(activity, 0)
462
+ if runs and runs[-1][0] == g:
463
+ runs[-1][1].append(activity)
464
+ else:
465
+ runs.append((g, [activity]))
466
+ # 2. Enforce body, redo, body, redo, ..., body.
467
+ normalised: list[tuple[int, list[str]]] = []
468
+ for g, run in runs:
469
+ previous_is_redo = normalised and normalised[-1][0] != 0
470
+ if g != 0 and (not normalised or previous_is_redo):
471
+ normalised.append((0, []))
472
+ normalised.append((g, run))
473
+ if not normalised or normalised[-1][0] != 0:
474
+ normalised.append((0, []))
475
+ for g, run in normalised:
476
+ sublogs[g][tuple(run)] += n
477
+ return [Counter({t: c for t, c in sub.items() if c}) for sub in sublogs]
@@ -0,0 +1,62 @@
1
+ """Discovery with state-based regions: log → transition system → Petri net.
2
+
3
+ The standard example of *two-phase* discovery (van der Aalst, *Process
4
+ Mining*, Section 7.4):
5
+
6
+ 1. build a **transition system** from the log with a state function
7
+ (:func:`~openprocess.mining.transition_system.transition_system_from_log`):
8
+ the prefix, postfix or both of every event, as a sequence, multiset or
9
+ set, over a horizon of the last ``k`` events or all of them;
10
+ 2. find its **regions** and **minimal regions**
11
+ (:func:`~openprocess.mining.regions.analyse_regions`), and check whether it is
12
+ **elementary** (state separation and forward closure);
13
+ 3. **synthesise** a Petri net with one place per minimal region
14
+ (:func:`~openprocess.mining.regions.synthesise`). For an elementary
15
+ transition system the net's reachability graph is isomorphic to it.
16
+
17
+ The abstraction is the knob: a coarser one (a set, a short horizon) merges
18
+ states and so generalises, a finer one (the full sequence) only allows the
19
+ log. The result keeps every step so the app can show the derivation.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ from dataclasses import dataclass, field
25
+
26
+ from ..log import SimpleLog
27
+ from ..petrinet import PetriNet
28
+ from ..regions import RegionAnalysis, Synthesis, analyse_regions, synthesise
29
+ from ..transition_system import TransitionSystem, transition_system_from_log
30
+
31
+
32
+ @dataclass
33
+ class RegionResult:
34
+ ts: TransitionSystem
35
+ analysis: RegionAnalysis
36
+ synthesis: Synthesis
37
+ warnings: list[str] = field(default_factory=list)
38
+
39
+ @property
40
+ def net(self) -> PetriNet | None:
41
+ return self.synthesis.net
42
+
43
+
44
+ def region_result(ts: TransitionSystem) -> RegionResult:
45
+ """Regions and synthesis for a transition system already built (or typed)."""
46
+ analysis = analyse_regions(ts)
47
+ synthesis = synthesise(ts, analysis)
48
+ return RegionResult(ts, analysis, synthesis, list(synthesis.warnings))
49
+
50
+
51
+ def region_miner(log: SimpleLog, direction: str = "prefix", representation: str = "set",
52
+ horizon: int | None = None) -> RegionResult:
53
+ """Discover a Petri net from ``log`` through state-based regions.
54
+
55
+ Raises ``ValueError`` when no net can be made (several initial states, or
56
+ a transition system too large to search for regions).
57
+ """
58
+ ts = transition_system_from_log(log, direction, representation, horizon)
59
+ result = region_result(ts)
60
+ if result.net is None:
61
+ raise ValueError(" ".join(result.warnings) or "No net could be synthesised.")
62
+ return result