openprocess 0.7.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (173) hide show
  1. cpnpy/__init__.py +46 -0
  2. openprocess/__init__.py +57 -0
  3. openprocess/analysis/__init__.py +0 -0
  4. openprocess/analysis/state_space.py +521 -0
  5. openprocess/analysis/state_space_process.py +251 -0
  6. openprocess/cli.py +742 -0
  7. openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/answer.pnml +31 -0
  8. openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/question.md +38 -0
  9. openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/answer.pnml +27 -0
  10. openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/net.pnml +29 -0
  11. openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/question.md +65 -0
  12. openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/log.txt +1 -0
  13. openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/question.md +67 -0
  14. openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/question.md +75 -0
  15. openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/ts.txt +5 -0
  16. openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/net.pnml +27 -0
  17. openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/question.md +65 -0
  18. openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/log.txt +1 -0
  19. openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/question.md +62 -0
  20. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/log.txt +1 -0
  21. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m1.pnml +36 -0
  22. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m2.pnml +28 -0
  23. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m3.pnml +30 -0
  24. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/net.pnml +36 -0
  25. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/question.md +69 -0
  26. openprocess/exercises/pack.md +14 -0
  27. openprocess/flow/__init__.py +50 -0
  28. openprocess/flow/box.py +466 -0
  29. openprocess/flow/boxes/__init__.py +7 -0
  30. openprocess/flow/boxes/check.py +119 -0
  31. openprocess/flow/boxes/compare.py +16 -0
  32. openprocess/flow/boxes/cpn.py +53 -0
  33. openprocess/flow/boxes/discover.py +124 -0
  34. openprocess/flow/boxes/filter.py +80 -0
  35. openprocess/flow/boxes/input.py +124 -0
  36. openprocess/flow/boxes/output.py +52 -0
  37. openprocess/flow/boxes/predict.py +186 -0
  38. openprocess/flow/boxes/science.py +159 -0
  39. openprocess/flow/boxes/sweeps.py +18 -0
  40. openprocess/flow/convert.py +187 -0
  41. openprocess/flow/datasets.py +198 -0
  42. openprocess/flow/explain.py +115 -0
  43. openprocess/flow/library.py +222 -0
  44. openprocess/flow/record.py +385 -0
  45. openprocess/flow/runner.py +357 -0
  46. openprocess/flow/sweep.py +92 -0
  47. openprocess/flow/types.py +290 -0
  48. openprocess/flow/workflow.py +628 -0
  49. openprocess/gui/__init__.py +0 -0
  50. openprocess/gui/app.py +90 -0
  51. openprocess/gui/arc_editing.py +295 -0
  52. openprocess/gui/canvas.py +1414 -0
  53. openprocess/gui/flow/__init__.py +8 -0
  54. openprocess/gui/flow/canvas.py +854 -0
  55. openprocess/gui/flow/page.py +972 -0
  56. openprocess/gui/flow/templates.py +131 -0
  57. openprocess/gui/flow/viewers.py +665 -0
  58. openprocess/gui/items.py +1275 -0
  59. openprocess/gui/learn/answer_boxes.py +978 -0
  60. openprocess/gui/learn/concealment.py +91 -0
  61. openprocess/gui/learn/mode.py +1181 -0
  62. openprocess/gui/panning.py +241 -0
  63. openprocess/gui/resources/openprocess-icon.png +0 -0
  64. openprocess/gui/studio/__init__.py +1 -0
  65. openprocess/gui/studio/__main__.py +3 -0
  66. openprocess/gui/studio/app.py +4031 -0
  67. openprocess/gui/studio/charts.py +115 -0
  68. openprocess/gui/studio/compare_page.py +487 -0
  69. openprocess/gui/studio/cpn_page.py +1858 -0
  70. openprocess/gui/studio/definition_view.py +284 -0
  71. openprocess/gui/studio/derivation_view.py +421 -0
  72. openprocess/gui/studio/documents.py +152 -0
  73. openprocess/gui/studio/dotted_chart.py +1401 -0
  74. openprocess/gui/studio/file_dialogs.py +143 -0
  75. openprocess/gui/studio/filter_dialog.py +247 -0
  76. openprocess/gui/studio/graph_builders.py +176 -0
  77. openprocess/gui/studio/graph_view.py +682 -0
  78. openprocess/gui/studio/instances.py +413 -0
  79. openprocess/gui/studio/log_editor.py +675 -0
  80. openprocess/gui/studio/log_page.py +800 -0
  81. openprocess/gui/studio/markdown_view.py +127 -0
  82. openprocess/gui/studio/mathtext.py +260 -0
  83. openprocess/gui/studio/ml_highlighter.py +75 -0
  84. openprocess/gui/studio/model_page.py +760 -0
  85. openprocess/gui/studio/net_comparison.py +124 -0
  86. openprocess/gui/studio/notes_overlay.py +275 -0
  87. openprocess/gui/studio/petri_page.py +844 -0
  88. openprocess/gui/studio/regions_view.py +502 -0
  89. openprocess/gui/studio/sidebar.py +149 -0
  90. openprocess/gui/studio/style.py +503 -0
  91. openprocess/gui/studio/tool_icons.py +134 -0
  92. openprocess/gui/studio/updates.py +439 -0
  93. openprocess/gui/studio/widgets.py +899 -0
  94. openprocess/gui/studio/workers.py +60 -0
  95. openprocess/gui/studio/workspace.py +447 -0
  96. openprocess/gui/theme.py +394 -0
  97. openprocess/gui/tidy.py +86 -0
  98. openprocess/io/__init__.py +0 -0
  99. openprocess/io/cpn_reader.py +389 -0
  100. openprocess/io/cpn_writer.py +357 -0
  101. openprocess/learn/__init__.py +23 -0
  102. openprocess/learn/answers.py +188 -0
  103. openprocess/learn/checks.py +953 -0
  104. openprocess/learn/computed.py +1180 -0
  105. openprocess/learn/context.py +145 -0
  106. openprocess/learn/exam.py +169 -0
  107. openprocess/learn/exercise-packs.md +325 -0
  108. openprocess/learn/importer.py +216 -0
  109. openprocess/learn/notation.py +474 -0
  110. openprocess/learn/pack.py +511 -0
  111. openprocess/learn/sheet.py +296 -0
  112. openprocess/mining/__init__.py +73 -0
  113. openprocess/mining/analysis.py +689 -0
  114. openprocess/mining/columns.py +282 -0
  115. openprocess/mining/compare_nets.py +246 -0
  116. openprocess/mining/conformance/__init__.py +0 -0
  117. openprocess/mining/conformance/alignments.py +263 -0
  118. openprocess/mining/conformance/quality.py +145 -0
  119. openprocess/mining/conformance/token_replay.py +252 -0
  120. openprocess/mining/csv_import.py +222 -0
  121. openprocess/mining/definitions.py +584 -0
  122. openprocess/mining/dfg.py +187 -0
  123. openprocess/mining/discovery/__init__.py +0 -0
  124. openprocess/mining/discovery/alpha.py +168 -0
  125. openprocess/mining/discovery/heuristics.py +332 -0
  126. openprocess/mining/discovery/inductive.py +477 -0
  127. openprocess/mining/discovery/state_regions.py +62 -0
  128. openprocess/mining/filtering.py +237 -0
  129. openprocess/mining/footprint.py +183 -0
  130. openprocess/mining/invariants.py +191 -0
  131. openprocess/mining/layout.py +279 -0
  132. openprocess/mining/log.py +364 -0
  133. openprocess/mining/petrinet.py +354 -0
  134. openprocess/mining/playout.py +75 -0
  135. openprocess/mining/pm4py_bridge.py +82 -0
  136. openprocess/mining/pnml.py +223 -0
  137. openprocess/mining/processtree.py +216 -0
  138. openprocess/mining/regions.py +476 -0
  139. openprocess/mining/stats.py +160 -0
  140. openprocess/mining/structure.py +374 -0
  141. openprocess/mining/transition_system.py +409 -0
  142. openprocess/mining/xes.py +399 -0
  143. openprocess/ml/__init__.py +0 -0
  144. openprocess/ml/ast_nodes.py +332 -0
  145. openprocess/ml/builtins.py +364 -0
  146. openprocess/ml/colorsets.py +522 -0
  147. openprocess/ml/errors.py +60 -0
  148. openprocess/ml/evaluator.py +754 -0
  149. openprocess/ml/lexer.py +277 -0
  150. openprocess/ml/multiset.py +417 -0
  151. openprocess/ml/parser.py +737 -0
  152. openprocess/ml/values.py +319 -0
  153. openprocess/model/__init__.py +0 -0
  154. openprocess/model/declarations.py +617 -0
  155. openprocess/model/examples.py +98 -0
  156. openprocess/model/net.py +701 -0
  157. openprocess/model/plain.py +192 -0
  158. openprocess/references.py +280 -0
  159. openprocess/sim/__init__.py +0 -0
  160. openprocess/sim/binding.py +620 -0
  161. openprocess/sim/export.py +66 -0
  162. openprocess/sim/simulator.py +315 -0
  163. openprocess/teaching/__init__.py +4 -0
  164. openprocess/teaching/answers.py +4 -0
  165. openprocess/teaching/checks.py +5 -0
  166. openprocess/teaching/pack.py +4 -0
  167. openprocess/teaching/sheet.py +4 -0
  168. openprocess-0.7.0.dist-info/METADATA +927 -0
  169. openprocess-0.7.0.dist-info/RECORD +173 -0
  170. openprocess-0.7.0.dist-info/WHEEL +5 -0
  171. openprocess-0.7.0.dist-info/entry_points.txt +6 -0
  172. openprocess-0.7.0.dist-info/licenses/LICENSE +21 -0
  173. openprocess-0.7.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,187 @@
1
+ """Directly-follows graphs (DFGs) -- the "process map" of tools like Disco.
2
+
3
+ Definition
4
+ ----------
5
+ Activity ``b`` *directly follows* ``a`` in a log, written ``a >_L b``, iff
6
+ some trace contains ``a`` immediately followed by ``b``. The DFG has one
7
+ node per activity and an edge ``a -> b`` weighted by *how often* that
8
+ happened, plus two artificial nodes ▶ (start) and ■ (end) with edges to the
9
+ first and from the last activity of every trace.
10
+
11
+ A DFG is not a Petri net: it has no concurrency, and every path through it is
12
+ allowed, so it usually *overgeneralises* (reading: Leemans, Poppe & Wynn,
13
+ "Directly-follows-based process mining"). It is nonetheless the most useful
14
+ first look at a log, and the Inductive Miner family works directly on it.
15
+
16
+ Performance
17
+ -----------
18
+ When events carry timestamps, each edge also records the time elapsed between
19
+ ``a`` and the ``b`` that directly followed it. Mean and median are shown in
20
+ performance mode.
21
+
22
+ Simplification
23
+ --------------
24
+ Real logs produce "spaghetti". :meth:`DFG.simplified` keeps a fraction of
25
+ the activities (most frequent first) and a fraction of the edges, while
26
+ guaranteeing that every kept activity stays connected: each keeps its most
27
+ frequent incoming and outgoing edge. This is the same idea as Disco's two
28
+ sliders.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ import statistics
34
+ from collections import Counter, defaultdict
35
+ from dataclasses import dataclass, field
36
+
37
+ from .log import Classifier, EventLog, SimpleLog
38
+
39
+ START = "▶"
40
+ END = "■"
41
+
42
+
43
+ @dataclass
44
+ class DFG:
45
+ activities: Counter = field(default_factory=Counter) # activity -> events
46
+ edges: Counter = field(default_factory=Counter) # (a, b) -> count
47
+ start: Counter = field(default_factory=Counter) # activity -> traces
48
+ end: Counter = field(default_factory=Counter)
49
+ durations: dict[tuple[str, str], list[float]] = field(default_factory=dict)
50
+ trace_count: int = 0
51
+ empty_traces: int = 0
52
+
53
+ # -- queries --------------------------------------------------------
54
+ def successors(self, activity: str) -> set[str]:
55
+ return {b for (a, b) in self.edges if a == activity}
56
+
57
+ def predecessors(self, activity: str) -> set[str]:
58
+ return {a for (a, b) in self.edges if b == activity}
59
+
60
+ def mean_duration(self, edge: tuple[str, str]) -> float | None:
61
+ values = self.durations.get(edge)
62
+ return statistics.fmean(values) if values else None
63
+
64
+ def median_duration(self, edge: tuple[str, str]) -> float | None:
65
+ values = self.durations.get(edge)
66
+ return statistics.median(values) if values else None
67
+
68
+ def all_edges(self) -> list[tuple[str, str, int]]:
69
+ """Edges including the artificial start/end ones, as (a, b, count)."""
70
+ edges = [(START, a, n) for a, n in self.start.items()]
71
+ edges += [(a, b, n) for (a, b), n in self.edges.items()]
72
+ edges += [(a, END, n) for a, n in self.end.items()]
73
+ return edges
74
+
75
+ # -- simplification ------------------------------------------------
76
+ def simplified(self, activity_fraction: float = 1.0, edge_fraction: float = 1.0) -> "DFG":
77
+ """Keep the top ``activity_fraction`` of activities and ``edge_fraction`` of edges.
78
+
79
+ Fractions are in [0, 1]. At least one activity is always kept.
80
+ Removing activities *projects* them out: we do not reconnect their
81
+ neighbours, because that would invent directly-follows pairs that
82
+ never occurred. Edges touching removed activities disappear.
83
+ """
84
+ ranked = [a for a, _ in sorted(self.activities.items(), key=lambda x: (-x[1], x[0]))]
85
+ keep_count = max(1, round(len(ranked) * activity_fraction))
86
+ kept = set(ranked[:keep_count])
87
+
88
+ candidates = [(a, b, n) for (a, b, n) in self.all_edges()
89
+ if (a in kept or a == START) and (b in kept or b == END)]
90
+ candidates.sort(key=lambda edge: (-edge[2], edge[0], edge[1]))
91
+ keep_edges = set((a, b) for a, b, _ in candidates[:round(len(candidates) * edge_fraction)])
92
+
93
+ # Connectivity guarantee: best incoming and best outgoing per activity.
94
+ for activity in kept:
95
+ incoming = [e for e in candidates if e[1] == activity]
96
+ outgoing = [e for e in candidates if e[0] == activity]
97
+ if incoming:
98
+ keep_edges.add(incoming[0][:2])
99
+ if outgoing:
100
+ keep_edges.add(outgoing[0][:2])
101
+
102
+ result = DFG(trace_count=self.trace_count, empty_traces=self.empty_traces)
103
+ result.activities = Counter({a: self.activities[a] for a in kept})
104
+ for a, b, n in candidates:
105
+ if (a, b) not in keep_edges:
106
+ continue
107
+ if a == START:
108
+ result.start[b] = n
109
+ elif b == END:
110
+ result.end[a] = n
111
+ else:
112
+ result.edges[(a, b)] = n
113
+ if (a, b) in self.durations:
114
+ result.durations[(a, b)] = self.durations[(a, b)]
115
+ return result
116
+
117
+
118
+ def dfg_from_simple_log(log: SimpleLog) -> DFG:
119
+ """DFG of a multiset of sequences (no performance information)."""
120
+ dfg = DFG()
121
+ for sequence, count in log.items():
122
+ dfg.trace_count += count
123
+ if not sequence:
124
+ dfg.empty_traces += count
125
+ continue
126
+ for activity in sequence:
127
+ dfg.activities[activity] += count
128
+ dfg.start[sequence[0]] += count
129
+ dfg.end[sequence[-1]] += count
130
+ for a, b in zip(sequence, sequence[1:]):
131
+ dfg.edges[(a, b)] += count
132
+ return dfg
133
+
134
+
135
+ def discover_dfg(log: EventLog | SimpleLog, classifier: Classifier | None = None) -> DFG:
136
+ """DFG of a log; with an :class:`EventLog`, edge durations are recorded too."""
137
+ if not isinstance(log, EventLog):
138
+ return dfg_from_simple_log(log)
139
+
140
+ classifier = classifier or log.default_classifier()
141
+ if log.columnar:
142
+ return _dfg_from_columns(log, classifier)
143
+ dfg = DFG()
144
+ durations: dict[tuple[str, str], list[float]] = defaultdict(list)
145
+ for trace in log.traces:
146
+ dfg.trace_count += 1
147
+ events = [e for e in trace.events if classifier.accepts(e)]
148
+ if not events:
149
+ dfg.empty_traces += 1
150
+ continue
151
+ labels = [classifier.label(e) for e in events]
152
+ dfg.activities.update(labels)
153
+ dfg.start[labels[0]] += 1
154
+ dfg.end[labels[-1]] += 1
155
+ for (first, second), (a, b) in zip(zip(events, events[1:]), zip(labels, labels[1:])):
156
+ dfg.edges[(a, b)] += 1
157
+ if first.timestamp is not None and second.timestamp is not None:
158
+ durations[(a, b)].append((second.timestamp - first.timestamp).total_seconds())
159
+ dfg.durations = dict(durations)
160
+ return dfg
161
+
162
+
163
+ def _dfg_from_columns(log: EventLog, classifier: Classifier) -> DFG:
164
+ """The same counts and durations, from a columnar log's arrays."""
165
+ from .columns import NO_TIME
166
+ store = log.traces.store
167
+ dfg = DFG()
168
+ durations: dict[tuple[str, str], list[float]] = defaultdict(list)
169
+ sequences = store.sequences(classifier.keys, classifier.lifecycles)
170
+ wanted = None if classifier.lifecycles is None else {v.lower() for v in classifier.lifecycles}
171
+ lifecycle, micros, offsets = store.lifecycle, store.micros, store.offsets
172
+ for case, labels in enumerate(sequences):
173
+ dfg.trace_count += 1
174
+ if not labels:
175
+ dfg.empty_traces += 1
176
+ continue
177
+ dfg.activities.update(labels)
178
+ dfg.start[labels[0]] += 1
179
+ dfg.end[labels[-1]] += 1
180
+ kept = [i for i in range(offsets[case], offsets[case + 1])
181
+ if wanted is None or lifecycle.codes[i] < 0 or lifecycle.values[lifecycle.codes[i]].lower() in wanted]
182
+ for (a, b), (i, j) in zip(zip(labels, labels[1:]), zip(kept, kept[1:])):
183
+ dfg.edges[(a, b)] += 1
184
+ if micros[i] != NO_TIME and micros[j] != NO_TIME:
185
+ durations[(a, b)].append((micros[j] - micros[i]) / 1_000_000)
186
+ dfg.durations = dict(durations)
187
+ return dfg
File without changes
@@ -0,0 +1,168 @@
1
+ """The α-algorithm (van der Aalst, Weijters & Maruster, 2004).
2
+
3
+ The eight steps (book, Definition 6.4), for a log L over activities T:
4
+
5
+ 1. ``T_L = { t | t occurs in some trace }`` -- the transitions
6
+ 2. ``T_I = { t | t is the first activity of some trace }`` -- start activities
7
+ 3. ``T_O = { t | t is the last activity of some trace }`` -- end activities
8
+ 4. ``X_L = { (A, B) | A ⊆ T_L, A ≠ ∅, B ⊆ T_L, B ≠ ∅,
9
+ ∀a∈A ∀b∈B: a →_L b, ∀a1,a2∈A: a1 #_L a2, ∀b1,b2∈B: b1 #_L b2 }``
10
+ 5. ``Y_L = { (A, B) ∈ X_L | no (A', B') ∈ X_L with A ⊆ A', B ⊆ B', (A,B) ≠ (A',B') }``
11
+ -- keep only the *maximal* pairs
12
+ 6. ``P_L = { p_(A,B) | (A,B) ∈ Y_L } ∪ { i_L, o_L }``
13
+ 7. ``F_L`` connects every ``a ∈ A`` to ``p_(A,B)``, ``p_(A,B)`` to every ``b ∈ B``,
14
+ ``i_L`` to every start activity and every end activity to ``o_L``
15
+ 8. ``α(L) = (P_L, T_L, F_L)``
16
+
17
+ Intuition for step 4: a place between A and B must be *fed* by the
18
+ activities in A and *consumed* by those in B. Every a must cause every b
19
+ (``→``), and no two members of A (or of B) may ever be seen together in
20
+ either order (``#``) -- if they were, they would be concurrent, and a single
21
+ place would wrongly force a choice between them. ``a1 # a2`` with
22
+ ``a1 = a2`` forces ``a # a``, which is why an activity in a length-one loop
23
+ (``a > a``) can never be in any pair.
24
+
25
+ Known limitations, reported by :func:`alpha_miner` as warnings:
26
+
27
+ * **length-one loops** (``a > a``) -- the activity ends up disconnected;
28
+ * **length-two loops** (``a b a``) -- ``a > b`` and ``b > a`` make them look
29
+ parallel, so the loop is lost;
30
+ * non-free-choice constructs and duplicate or silent activities cannot be
31
+ represented at all.
32
+
33
+ The result object keeps every intermediate set so the app can show the
34
+ derivation step by step -- exactly how exam questions ask for it.
35
+ """
36
+
37
+ from __future__ import annotations
38
+
39
+ from dataclasses import dataclass, field
40
+
41
+ from ..footprint import Footprint, footprint_of_log
42
+ from ..log import SimpleLog
43
+ from ..petrinet import Marking, PetriNet
44
+
45
+ Pair = tuple[frozenset[str], frozenset[str]]
46
+
47
+
48
+ @dataclass
49
+ class AlphaResult:
50
+ footprint: Footprint
51
+ T_L: list[str]
52
+ T_I: list[str]
53
+ T_O: list[str]
54
+ X_L: list[Pair]
55
+ Y_L: list[Pair]
56
+ net: PetriNet
57
+ warnings: list[str] = field(default_factory=list)
58
+
59
+ def steps(self) -> list[tuple[str, str]]:
60
+ """The derivation as (step, content) rows, in textbook notation."""
61
+ def s(items) -> str:
62
+ return "{" + ", ".join(sorted(items)) + "}"
63
+
64
+ def pairs(collection: list[Pair]) -> str:
65
+ if not collection:
66
+ return "∅"
67
+ return "{ " + ", ".join(f"({s(a)}, {s(b)})" for a, b in collection) + " }"
68
+
69
+ places = ["i_L"] + [f"p({s(a)},{s(b)})" for a, b in self.Y_L] + ["o_L"]
70
+ return [
71
+ ("1. T_L — all activities", s(self.T_L)),
72
+ ("2. T_I — start activities", s(self.T_I)),
73
+ ("3. T_O — end activities", s(self.T_O)),
74
+ ("4. X_L — candidate (A, B) pairs", pairs(self.X_L)),
75
+ ("5. Y_L — maximal pairs", pairs(self.Y_L)),
76
+ ("6. P_L — places", "{" + ", ".join(places) + "}"),
77
+ ("7. F_L — arcs", f"{len(self.net.arcs)} arcs"),
78
+ ("8. α(L) = (P_L, T_L, F_L)", self.net.summary()),
79
+ ]
80
+
81
+
82
+ def _pair_key(pair: Pair) -> tuple:
83
+ return (len(pair[0]) + len(pair[1]), sorted(pair[0]), sorted(pair[1]))
84
+
85
+
86
+ def alpha_miner(log: SimpleLog, name: str = "α(L)") -> AlphaResult:
87
+ """Run the α-algorithm on a multiset of activity sequences."""
88
+ fp = footprint_of_log(log)
89
+ T_L = fp.activities
90
+ T_I = sorted({trace[0] for trace in log if trace})
91
+ T_O = sorted({trace[-1] for trace in log if trace})
92
+
93
+ def independent(group: frozenset[str]) -> bool:
94
+ """All members pairwise # -- including each with itself."""
95
+ return all(fp.choice(x, y) for x in group for y in group)
96
+
97
+ def causal(a_set: frozenset[str], b_set: frozenset[str]) -> bool:
98
+ return all(fp.causal(a, b) for a in a_set for b in b_set)
99
+
100
+ # Step 4. Grow pairs from the singleton pairs ({a},{b}) with a → b.
101
+ # Any (A, B) in X_L can be reached this way by adding one element at a
102
+ # time, because every sub-pair of a valid pair is itself valid. The
103
+ # search is exponential in the worst case, but course-sized logs have at
104
+ # most a few dozen activities, so this is instantaneous in practice.
105
+ seeds = {(frozenset({a}), frozenset({b}))
106
+ for a in T_L for b in T_L
107
+ if fp.causal(a, b) and fp.choice(a, a) and fp.choice(b, b)}
108
+ X_L: set[Pair] = set(seeds)
109
+ frontier = list(seeds)
110
+ while frontier:
111
+ next_frontier = []
112
+ for a_set, b_set in frontier:
113
+ for extra in T_L:
114
+ if extra not in a_set:
115
+ grown_a = a_set | {extra}
116
+ candidate = (grown_a, b_set)
117
+ if candidate not in X_L and independent(grown_a) and causal(grown_a, b_set):
118
+ X_L.add(candidate)
119
+ next_frontier.append(candidate)
120
+ if extra not in b_set:
121
+ grown_b = b_set | {extra}
122
+ candidate = (a_set, grown_b)
123
+ if candidate not in X_L and independent(grown_b) and causal(a_set, grown_b):
124
+ X_L.add(candidate)
125
+ next_frontier.append(candidate)
126
+ frontier = next_frontier
127
+
128
+ # Step 5. Keep only the maximal pairs.
129
+ Y_L = [pair for pair in X_L
130
+ if not any(pair != other and pair[0] <= other[0] and pair[1] <= other[1]
131
+ for other in X_L)]
132
+
133
+ X_sorted = sorted(X_L, key=_pair_key)
134
+ Y_sorted = sorted(Y_L, key=_pair_key)
135
+
136
+ # Steps 6-8. Build the net.
137
+ net = PetriNet(name)
138
+ transitions = {a: net.add_transition(a, id=f"t_{a}") for a in T_L}
139
+ source = net.add_place("i_L", id="i_L")
140
+ sink = net.add_place("o_L", id="o_L")
141
+ for number, (a_set, b_set) in enumerate(Y_sorted, 1):
142
+ label = "p({" + ",".join(sorted(a_set)) + "},{" + ",".join(sorted(b_set)) + "})"
143
+ place = net.add_place(label, id=f"p{number}")
144
+ for a in a_set:
145
+ net.add_arc(transitions[a], place)
146
+ for b in b_set:
147
+ net.add_arc(place, transitions[b])
148
+ for a in T_I:
149
+ net.add_arc(source, transitions[a])
150
+ for a in T_O:
151
+ net.add_arc(transitions[a], sink)
152
+ net.initial_marking = Marking({source.id: 1})
153
+ net.final_marking = Marking({sink.id: 1})
154
+ net.info["algorithm"] = "α-algorithm"
155
+
156
+ warnings = []
157
+ loops1 = [a for a in T_L if (a, a) in fp.directly_follows]
158
+ if loops1:
159
+ warnings.append("Length-one loops detected (" + ", ".join(loops1) + "). The "
160
+ "α-algorithm cannot discover these; the activities end up "
161
+ "without input/output places.")
162
+ loops2 = sorted({tuple(sorted((trace[i], trace[i + 1])))
163
+ for trace in log for i in range(len(trace) - 2)
164
+ if trace[i] == trace[i + 2] and trace[i] != trace[i + 1]})
165
+ if loops2:
166
+ warnings.append("Length-two loops detected (" + ", ".join(f"{a}–{b}" for a, b in loops2)
167
+ + "). The α-algorithm treats these pairs as parallel.")
168
+ return AlphaResult(fp, T_L, T_I, T_O, X_sorted, Y_sorted, net, warnings)
@@ -0,0 +1,332 @@
1
+ """Heuristics Miner: frequency-aware dependency graphs (Weijters & van der Aalst).
2
+
3
+ Where the α-algorithm asks the yes/no question "does a ever directly precede
4
+ b?", the Heuristics Miner asks *how strongly* the log supports ``a`` causing
5
+ ``b``, using counts ``|a >_L b|`` (how often b directly follows a):
6
+
7
+ Dependency measure, for ``a ≠ b``::
8
+
9
+ |a > b| − |b > a|
10
+ a ⇒ b = ─────────────────── value in (−1, 1)
11
+ |a > b| + |b > a| + 1
12
+
13
+ It is close to 1 when b follows a often and a almost never follows b, close
14
+ to 0 when both orders are equally common (concurrency, or noise), negative
15
+ when the dependency is the other way round. The ``+ 1`` damps small counts:
16
+ one occurrence gives only 0.5, a hundred give 0.99.
17
+
18
+ Length-one loop (``a`` directly followed by itself)::
19
+
20
+ a ⇒ a = |a > a| / (|a > a| + 1)
21
+
22
+ Length-two loop (``a b a`` patterns, counted as ``|a >> b|``)::
23
+
24
+ a ⇒2 b = (|a >> b| + |b >> a|) / (|a >> b| + |b >> a| + 1)
25
+
26
+ An edge ``a → b`` is kept when ``a ⇒ b`` reaches the *dependency threshold*
27
+ and ``|a > b|`` reaches the *positive observations* threshold. With
28
+ *all tasks connected* on (the default), every activity additionally keeps its
29
+ single best incoming and outgoing edge, so no activity is left floating.
30
+
31
+ The output of :func:`heuristics_miner` is a *dependency graph*: which
32
+ activity causes which, without saying whether a fork is a choice (XOR) or
33
+ runs in parallel (AND).
34
+
35
+ From dependency graph to Petri net (:func:`heuristics_net`)
36
+ -----------------------------------------------------------
37
+ The split/join semantics come from the log, as a **causal net** (C-net,
38
+ van der Aalst, *Process Mining*, 2016, Section 7.3). Every activity gets a
39
+ set of *input bindings* and *output bindings*: which of its predecessors it
40
+ waited for, and which of its successors it enabled. Replaying each trace
41
+ on the dependency graph gives them:
42
+
43
+ * the output binding of ``a`` at position ``i`` is every successor ``b`` of
44
+ ``a`` that occurs after ``i`` with no other ``a`` in between (its first
45
+ such occurrence);
46
+ * the input binding of ``b`` at position ``j`` is every predecessor ``a``
47
+ of ``b`` that occurs before ``j`` with no other ``b`` in between.
48
+
49
+ These two rules match up exactly: ``b`` is in ``a``'s output binding iff
50
+ ``a`` is in the input binding of that occurrence of ``b``. A virtual start
51
+ activity precedes the first event of every trace and a virtual end activity
52
+ follows the last one. So after ``a`` in ``[⟨a,b,c,d⟩³, ⟨a,c,b,d⟩², ⟨a,e,d⟩]``,
53
+ ``{b, c}`` (5 times) and ``{e}`` (once) are the output bindings: b and c
54
+ run in parallel, as an alternative to e.
55
+
56
+ The C-net becomes a Petri net in the standard way: a place for every arc
57
+ ``(a, b)`` of the dependency graph, a transition per activity, and where an
58
+ activity has several bindings on a side, a silent transition per binding
59
+ that takes (or fills) exactly the places of that binding. Silent steps that
60
+ make no difference to the behaviour are then removed.
61
+ """
62
+
63
+ from __future__ import annotations
64
+
65
+ from collections import Counter
66
+ from dataclasses import dataclass, field
67
+
68
+ from ..dfg import DFG, dfg_from_simple_log
69
+ from ..log import SimpleLog
70
+
71
+
72
+ @dataclass
73
+ class DependencyGraph:
74
+ activities: Counter
75
+ #: (a, b) -> (dependency value, |a > b|) for every kept edge
76
+ edges: dict[tuple[str, str], tuple[float, int]] = field(default_factory=dict)
77
+ start: Counter = field(default_factory=Counter)
78
+ end: Counter = field(default_factory=Counter)
79
+ #: dependency value of every ordered pair that directly follows at least once
80
+ all_dependencies: dict[tuple[str, str], float] = field(default_factory=dict)
81
+
82
+
83
+ def dependency(dfg: DFG, a: str, b: str) -> float:
84
+ ab, ba = dfg.edges.get((a, b), 0), dfg.edges.get((b, a), 0)
85
+ if a == b:
86
+ return ab / (ab + 1)
87
+ return (ab - ba) / (ab + ba + 1)
88
+
89
+
90
+ def heuristics_miner(log: SimpleLog, dependency_threshold: float = 0.5,
91
+ min_observations: int = 1, loop_two_threshold: float = 0.5,
92
+ all_tasks_connected: bool = True) -> DependencyGraph:
93
+ dfg = dfg_from_simple_log(log)
94
+
95
+ # |a >> b|: occurrences of the pattern a b a.
96
+ two_loops: Counter = Counter()
97
+ for trace, n in log.items():
98
+ for i in range(len(trace) - 2):
99
+ if trace[i] == trace[i + 2] and trace[i] != trace[i + 1]:
100
+ two_loops[(trace[i], trace[i + 1])] += n
101
+
102
+ graph = DependencyGraph(Counter(dfg.activities), start=Counter(dfg.start),
103
+ end=Counter(dfg.end))
104
+ for (a, b), count in dfg.edges.items():
105
+ value = dependency(dfg, a, b)
106
+ graph.all_dependencies[(a, b)] = value
107
+ if value >= dependency_threshold and count >= min_observations:
108
+ graph.edges[(a, b)] = (value, count)
109
+
110
+ # Length-two loops: a b a with both a⇒a and b⇒b weak (otherwise they are
111
+ # already length-one loops) keeps both directions a→b and b→a.
112
+ for (a, b), count in two_loops.items():
113
+ mutual = count + two_loops.get((b, a), 0)
114
+ value = mutual / (mutual + 1)
115
+ if value >= loop_two_threshold:
116
+ for x, y in ((a, b), (b, a)):
117
+ if (x, y) in dfg.edges:
118
+ graph.edges[(x, y)] = (value, dfg.edges[(x, y)])
119
+
120
+ if all_tasks_connected:
121
+ for activity in dfg.activities:
122
+ outgoing = [(dependency(dfg, activity, b), b) for (x, b) in dfg.edges
123
+ if x == activity and b != activity]
124
+ incoming = [(dependency(dfg, a, activity), a) for (a, y) in dfg.edges
125
+ if y == activity and a != activity]
126
+ if outgoing and activity not in dfg.end:
127
+ value, best = max(outgoing)
128
+ graph.edges.setdefault((activity, best), (value, dfg.edges[(activity, best)]))
129
+ if incoming and activity not in dfg.start:
130
+ value, best = max(incoming)
131
+ graph.edges.setdefault((best, activity), (value, dfg.edges[(best, activity)]))
132
+ return graph
133
+
134
+
135
+ # ---------------------------------------------------------------------------
136
+ # Causal nets and their Petri nets
137
+ # ---------------------------------------------------------------------------
138
+ #: The virtual activities before the first and after the last event.
139
+ START, END = "▶ start", "■ end"
140
+
141
+
142
+ @dataclass
143
+ class CausalNet:
144
+ """A dependency graph with, for every activity, its observed input and
145
+ output bindings (frozensets of activities) and how often each occurred."""
146
+
147
+ graph: DependencyGraph
148
+ inputs: dict[str, Counter] = field(default_factory=dict)
149
+ outputs: dict[str, Counter] = field(default_factory=dict)
150
+
151
+
152
+ @dataclass
153
+ class HeuristicsResult:
154
+ """What the Heuristics Miner found: the graph, the C-net, the Petri net."""
155
+
156
+ net: "PetriNet"
157
+ causal: CausalNet
158
+ dependency_threshold: float
159
+
160
+ @property
161
+ def graph(self) -> DependencyGraph:
162
+ return self.causal.graph
163
+
164
+
165
+ def causal_net(log: SimpleLog, graph: DependencyGraph) -> CausalNet:
166
+ """Learn the input and output bindings by replaying ``log`` on ``graph``."""
167
+ successors: dict[str, set[str]] = {}
168
+ predecessors: dict[str, set[str]] = {}
169
+ for a, b in graph.edges:
170
+ successors.setdefault(a, set()).add(b)
171
+ predecessors.setdefault(b, set()).add(a)
172
+ net = CausalNet(graph)
173
+
174
+ def record(side: dict[str, Counter], activity: str, binding: set[str], n: int) -> None:
175
+ if binding:
176
+ side.setdefault(activity, Counter())[frozenset(binding)] += n
177
+
178
+ for trace, n in log.items():
179
+ if not trace:
180
+ record(net.outputs, START, {END}, n)
181
+ record(net.inputs, END, {START}, n)
182
+ continue
183
+ record(net.outputs, START, {trace[0]}, n)
184
+ record(net.inputs, END, {trace[-1]}, n)
185
+ for i, a in enumerate(trace):
186
+ outputs = {END} if i == len(trace) - 1 else set()
187
+ for b in successors.get(a, ()):
188
+ if b == a: # a self-loop: only a a
189
+ if i + 1 < len(trace) and trace[i + 1] == a:
190
+ outputs.add(a)
191
+ continue
192
+ for later in trace[i + 1:]:
193
+ if later == a:
194
+ break
195
+ if later == b:
196
+ outputs.add(b)
197
+ break
198
+ record(net.outputs, a, outputs, n)
199
+ inputs = {START} if i == 0 else set()
200
+ for c in predecessors.get(a, ()):
201
+ if c == a:
202
+ if i > 0 and trace[i - 1] == a:
203
+ inputs.add(a)
204
+ continue
205
+ for earlier in reversed(trace[:i]):
206
+ if earlier == a:
207
+ break
208
+ if earlier == c:
209
+ inputs.add(c)
210
+ break
211
+ record(net.inputs, a, inputs, n)
212
+ return net
213
+
214
+
215
+ def causal_net_to_petri_net(causal: CausalNet, name: str = "Heuristics net") -> "PetriNet":
216
+ """The Petri net of a C-net: arc places, activity transitions, and a
217
+ silent transition per binding where an activity has more than one."""
218
+ from ..petrinet import Marking, PetriNet
219
+ net = PetriNet(name)
220
+ activities = set(causal.inputs) | set(causal.outputs)
221
+ places: dict[tuple[str, str], str] = {}
222
+
223
+ def arc_place(a: str, b: str) -> str:
224
+ if (a, b) not in places:
225
+ places[(a, b)] = net.add_place(f"({a},{b})").id
226
+ return places[(a, b)]
227
+
228
+ transitions: dict[str, str] = {}
229
+ for activity in sorted(activities, key=lambda x: (x != START, x == END, x)):
230
+ silent = activity in (START, END)
231
+ transitions[activity] = net.add_transition(None if silent else activity,
232
+ name=activity).id
233
+ source, sink = net.add_place("start"), net.add_place("end")
234
+ net.add_arc(source.id, transitions[START])
235
+ net.add_arc(transitions[END], sink.id)
236
+
237
+ for activity, bindings in sorted(causal.inputs.items()):
238
+ target = transitions[activity]
239
+ if len(bindings) == 1:
240
+ for before in next(iter(bindings)):
241
+ net.add_arc(arc_place(before, activity), target)
242
+ continue
243
+ joined = net.add_place(f"in({activity})").id
244
+ net.add_arc(joined, target)
245
+ for binding in sorted(bindings, key=sorted):
246
+ join = net.add_transition(
247
+ None, name=f"τ {{{', '.join(sorted(binding))}}} → {activity}").id
248
+ for before in sorted(binding):
249
+ net.add_arc(arc_place(before, activity), join)
250
+ net.add_arc(join, joined)
251
+ for activity, bindings in sorted(causal.outputs.items()):
252
+ origin = transitions[activity]
253
+ if len(bindings) == 1:
254
+ for after in next(iter(bindings)):
255
+ net.add_arc(origin, arc_place(activity, after))
256
+ continue
257
+ split = net.add_place(f"out({activity})").id
258
+ net.add_arc(origin, split)
259
+ for binding in sorted(bindings, key=sorted):
260
+ fork = net.add_transition(
261
+ None, name=f"τ {activity} → {{{', '.join(sorted(binding))}}}").id
262
+ net.add_arc(split, fork)
263
+ for after in sorted(binding):
264
+ net.add_arc(fork, arc_place(activity, after))
265
+ net.initial_marking = Marking({source.id: 1})
266
+ net.final_marking = Marking({sink.id: 1})
267
+ _remove_redundant_silent_steps(net)
268
+ return net
269
+
270
+
271
+ def _remove_redundant_silent_steps(net) -> None:
272
+ """Remove τ transitions that only pass one token from one place to the next.
273
+
274
+ Two classic reductions, both keeping the visible behaviour:
275
+
276
+ * τ is the only consumer of its input place p: drop τ and p, and let
277
+ whatever filled p fill τ's output place instead;
278
+ * τ is the only producer of its output place q: drop τ and q, and let
279
+ whatever took from q take from τ's input place instead.
280
+
281
+ A τ with one input and one output place is applied only when the arcs
282
+ have weight 1 and the places are different.
283
+ """
284
+ from ..petrinet import Marking
285
+ changed = True
286
+ while changed:
287
+ changed = False
288
+ for transition_id, transition in list(net.transitions.items()):
289
+ if not transition.silent:
290
+ continue
291
+ ins = [a for a in net.arcs if a.target == transition_id]
292
+ outs = [a for a in net.arcs if a.source == transition_id]
293
+ if len(ins) != 1 or len(outs) != 1 or ins[0].weight != 1 or outs[0].weight != 1:
294
+ continue
295
+ p, q = ins[0].source, outs[0].target
296
+ if p == q:
297
+ continue
298
+ consumers_of_p = [a for a in net.arcs if a.source == p]
299
+ producers_of_q = [a for a in net.arcs if a.target == q]
300
+ if len(consumers_of_p) == 1:
301
+ # p only feeds τ: whatever filled p fills q.
302
+ for arc in [a for a in net.arcs if a.target == p]:
303
+ net.add_arc(arc.source, q, arc.weight)
304
+ if net.initial_marking[p]:
305
+ net.initial_marking = Marking({**dict(net.initial_marking.items()),
306
+ q: net.initial_marking[p], p: 0})
307
+ if net.final_marking[p]:
308
+ continue
309
+ net.remove_transition(transition_id)
310
+ net.remove_place(p)
311
+ changed = True
312
+ break
313
+ if len(producers_of_q) == 1 and not net.final_marking[q] \
314
+ and not net.initial_marking[q]:
315
+ # q is only filled by τ: whatever took from q takes from p.
316
+ for arc in [a for a in net.arcs if a.source == q]:
317
+ net.add_arc(p, arc.target, arc.weight)
318
+ net.remove_transition(transition_id)
319
+ net.remove_place(q)
320
+ changed = True
321
+ break
322
+
323
+
324
+ def heuristics_net(log: SimpleLog, dependency_threshold: float = 0.5,
325
+ **options) -> HeuristicsResult:
326
+ """Heuristics Miner all the way to a Petri net (dependency graph, C-net,
327
+ Petri net), without PM4Py. ``options`` go to :func:`heuristics_miner`."""
328
+ graph = heuristics_miner(log, dependency_threshold=dependency_threshold, **options)
329
+ causal = causal_net(log, graph)
330
+ net = causal_net_to_petri_net(causal)
331
+ net.info["algorithm"] = f"Heuristics Miner (dependency ≥ {dependency_threshold:g})"
332
+ return HeuristicsResult(net, causal, dependency_threshold)