openprocess 0.7.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cpnpy/__init__.py +46 -0
- openprocess/__init__.py +57 -0
- openprocess/analysis/__init__.py +0 -0
- openprocess/analysis/state_space.py +521 -0
- openprocess/analysis/state_space_process.py +251 -0
- openprocess/cli.py +742 -0
- openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/answer.pnml +31 -0
- openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/question.md +38 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/answer.pnml +27 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/net.pnml +29 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/question.md +65 -0
- openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/log.txt +1 -0
- openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/question.md +67 -0
- openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/question.md +75 -0
- openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/ts.txt +5 -0
- openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/net.pnml +27 -0
- openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/question.md +65 -0
- openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/log.txt +1 -0
- openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/question.md +62 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/log.txt +1 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m1.pnml +36 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m2.pnml +28 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m3.pnml +30 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/net.pnml +36 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/question.md +69 -0
- openprocess/exercises/pack.md +14 -0
- openprocess/flow/__init__.py +50 -0
- openprocess/flow/box.py +466 -0
- openprocess/flow/boxes/__init__.py +7 -0
- openprocess/flow/boxes/check.py +119 -0
- openprocess/flow/boxes/compare.py +16 -0
- openprocess/flow/boxes/cpn.py +53 -0
- openprocess/flow/boxes/discover.py +124 -0
- openprocess/flow/boxes/filter.py +80 -0
- openprocess/flow/boxes/input.py +124 -0
- openprocess/flow/boxes/output.py +52 -0
- openprocess/flow/boxes/predict.py +186 -0
- openprocess/flow/boxes/science.py +159 -0
- openprocess/flow/boxes/sweeps.py +18 -0
- openprocess/flow/convert.py +187 -0
- openprocess/flow/datasets.py +198 -0
- openprocess/flow/explain.py +115 -0
- openprocess/flow/library.py +222 -0
- openprocess/flow/record.py +385 -0
- openprocess/flow/runner.py +357 -0
- openprocess/flow/sweep.py +92 -0
- openprocess/flow/types.py +290 -0
- openprocess/flow/workflow.py +628 -0
- openprocess/gui/__init__.py +0 -0
- openprocess/gui/app.py +90 -0
- openprocess/gui/arc_editing.py +295 -0
- openprocess/gui/canvas.py +1414 -0
- openprocess/gui/flow/__init__.py +8 -0
- openprocess/gui/flow/canvas.py +854 -0
- openprocess/gui/flow/page.py +972 -0
- openprocess/gui/flow/templates.py +131 -0
- openprocess/gui/flow/viewers.py +665 -0
- openprocess/gui/items.py +1275 -0
- openprocess/gui/learn/answer_boxes.py +978 -0
- openprocess/gui/learn/concealment.py +91 -0
- openprocess/gui/learn/mode.py +1181 -0
- openprocess/gui/panning.py +241 -0
- openprocess/gui/resources/openprocess-icon.png +0 -0
- openprocess/gui/studio/__init__.py +1 -0
- openprocess/gui/studio/__main__.py +3 -0
- openprocess/gui/studio/app.py +4031 -0
- openprocess/gui/studio/charts.py +115 -0
- openprocess/gui/studio/compare_page.py +487 -0
- openprocess/gui/studio/cpn_page.py +1858 -0
- openprocess/gui/studio/definition_view.py +284 -0
- openprocess/gui/studio/derivation_view.py +421 -0
- openprocess/gui/studio/documents.py +152 -0
- openprocess/gui/studio/dotted_chart.py +1401 -0
- openprocess/gui/studio/file_dialogs.py +143 -0
- openprocess/gui/studio/filter_dialog.py +247 -0
- openprocess/gui/studio/graph_builders.py +176 -0
- openprocess/gui/studio/graph_view.py +682 -0
- openprocess/gui/studio/instances.py +413 -0
- openprocess/gui/studio/log_editor.py +675 -0
- openprocess/gui/studio/log_page.py +800 -0
- openprocess/gui/studio/markdown_view.py +127 -0
- openprocess/gui/studio/mathtext.py +260 -0
- openprocess/gui/studio/ml_highlighter.py +75 -0
- openprocess/gui/studio/model_page.py +760 -0
- openprocess/gui/studio/net_comparison.py +124 -0
- openprocess/gui/studio/notes_overlay.py +275 -0
- openprocess/gui/studio/petri_page.py +844 -0
- openprocess/gui/studio/regions_view.py +502 -0
- openprocess/gui/studio/sidebar.py +149 -0
- openprocess/gui/studio/style.py +503 -0
- openprocess/gui/studio/tool_icons.py +134 -0
- openprocess/gui/studio/updates.py +439 -0
- openprocess/gui/studio/widgets.py +899 -0
- openprocess/gui/studio/workers.py +60 -0
- openprocess/gui/studio/workspace.py +447 -0
- openprocess/gui/theme.py +394 -0
- openprocess/gui/tidy.py +86 -0
- openprocess/io/__init__.py +0 -0
- openprocess/io/cpn_reader.py +389 -0
- openprocess/io/cpn_writer.py +357 -0
- openprocess/learn/__init__.py +23 -0
- openprocess/learn/answers.py +188 -0
- openprocess/learn/checks.py +953 -0
- openprocess/learn/computed.py +1180 -0
- openprocess/learn/context.py +145 -0
- openprocess/learn/exam.py +169 -0
- openprocess/learn/exercise-packs.md +325 -0
- openprocess/learn/importer.py +216 -0
- openprocess/learn/notation.py +474 -0
- openprocess/learn/pack.py +511 -0
- openprocess/learn/sheet.py +296 -0
- openprocess/mining/__init__.py +73 -0
- openprocess/mining/analysis.py +689 -0
- openprocess/mining/columns.py +282 -0
- openprocess/mining/compare_nets.py +246 -0
- openprocess/mining/conformance/__init__.py +0 -0
- openprocess/mining/conformance/alignments.py +263 -0
- openprocess/mining/conformance/quality.py +145 -0
- openprocess/mining/conformance/token_replay.py +252 -0
- openprocess/mining/csv_import.py +222 -0
- openprocess/mining/definitions.py +584 -0
- openprocess/mining/dfg.py +187 -0
- openprocess/mining/discovery/__init__.py +0 -0
- openprocess/mining/discovery/alpha.py +168 -0
- openprocess/mining/discovery/heuristics.py +332 -0
- openprocess/mining/discovery/inductive.py +477 -0
- openprocess/mining/discovery/state_regions.py +62 -0
- openprocess/mining/filtering.py +237 -0
- openprocess/mining/footprint.py +183 -0
- openprocess/mining/invariants.py +191 -0
- openprocess/mining/layout.py +279 -0
- openprocess/mining/log.py +364 -0
- openprocess/mining/petrinet.py +354 -0
- openprocess/mining/playout.py +75 -0
- openprocess/mining/pm4py_bridge.py +82 -0
- openprocess/mining/pnml.py +223 -0
- openprocess/mining/processtree.py +216 -0
- openprocess/mining/regions.py +476 -0
- openprocess/mining/stats.py +160 -0
- openprocess/mining/structure.py +374 -0
- openprocess/mining/transition_system.py +409 -0
- openprocess/mining/xes.py +399 -0
- openprocess/ml/__init__.py +0 -0
- openprocess/ml/ast_nodes.py +332 -0
- openprocess/ml/builtins.py +364 -0
- openprocess/ml/colorsets.py +522 -0
- openprocess/ml/errors.py +60 -0
- openprocess/ml/evaluator.py +754 -0
- openprocess/ml/lexer.py +277 -0
- openprocess/ml/multiset.py +417 -0
- openprocess/ml/parser.py +737 -0
- openprocess/ml/values.py +319 -0
- openprocess/model/__init__.py +0 -0
- openprocess/model/declarations.py +617 -0
- openprocess/model/examples.py +98 -0
- openprocess/model/net.py +701 -0
- openprocess/model/plain.py +192 -0
- openprocess/references.py +280 -0
- openprocess/sim/__init__.py +0 -0
- openprocess/sim/binding.py +620 -0
- openprocess/sim/export.py +66 -0
- openprocess/sim/simulator.py +315 -0
- openprocess/teaching/__init__.py +4 -0
- openprocess/teaching/answers.py +4 -0
- openprocess/teaching/checks.py +5 -0
- openprocess/teaching/pack.py +4 -0
- openprocess/teaching/sheet.py +4 -0
- openprocess-0.7.0.dist-info/METADATA +927 -0
- openprocess-0.7.0.dist-info/RECORD +173 -0
- openprocess-0.7.0.dist-info/WHEEL +5 -0
- openprocess-0.7.0.dist-info/entry_points.txt +6 -0
- openprocess-0.7.0.dist-info/licenses/LICENSE +21 -0
- openprocess-0.7.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,477 @@
|
|
|
1
|
+
"""The Inductive Miner (IM) and Inductive Miner – infrequent (IMf).
|
|
2
|
+
|
|
3
|
+
Reading: Leemans, Fahland & van der Aalst, "Discovering block-structured
|
|
4
|
+
process models from event logs" (course reading 4).
|
|
5
|
+
|
|
6
|
+
The idea: divide and conquer
|
|
7
|
+
----------------------------
|
|
8
|
+
Look at the directly-follows graph of the log and try to find a **cut**: a
|
|
9
|
+
partition of the activities into groups ``Σ1, ..., Σn`` that matches one of
|
|
10
|
+
the four process-tree operators. If one is found, **split** the log into one
|
|
11
|
+
sub-log per group, recurse on each, and combine the results under that
|
|
12
|
+
operator. Recursion stops at **base cases** (a log with one activity, or
|
|
13
|
+
only empty traces). When no cut exists, a **fall-through** produces a
|
|
14
|
+
general but still correct model.
|
|
15
|
+
|
|
16
|
+
The four cuts, in the order they are tried
|
|
17
|
+
------------------------------------------
|
|
18
|
+
× exclusive choice
|
|
19
|
+
The groups are the *connected components* of the DFG (ignoring edge
|
|
20
|
+
direction). No trace mixes two groups.
|
|
21
|
+
→ sequence
|
|
22
|
+
Every activity of an earlier group can reach every activity of a later
|
|
23
|
+
group, and never the other way round. Computed from the strongly
|
|
24
|
+
connected components, merging components that are mutually unreachable.
|
|
25
|
+
∧ parallel
|
|
26
|
+
Between two different groups, *every* pair of activities directly
|
|
27
|
+
follows each other in both directions, and every group contains a start
|
|
28
|
+
and an end activity. Computed as the connected components of the
|
|
29
|
+
*negated* DFG (an edge where the pair is **not** in both directions).
|
|
30
|
+
↺ loop
|
|
31
|
+
The *body* group contains all start and end activities; each *redo*
|
|
32
|
+
group is entered only from end activities and exits only to start
|
|
33
|
+
activities.
|
|
34
|
+
|
|
35
|
+
Guarantees (IM): the result always *fits* the log perfectly and is always
|
|
36
|
+
*sound*. The price is that fall-throughs may overgeneralise.
|
|
37
|
+
|
|
38
|
+
IMf: handling infrequent behaviour
|
|
39
|
+
----------------------------------
|
|
40
|
+
IM treats one odd trace the same as a thousand normal ones. IMf adds a
|
|
41
|
+
**noise threshold** ``f`` in [0, 1]: when no cut is found on the full DFG,
|
|
42
|
+
edges whose frequency is below ``f`` × (the most frequent outgoing edge of
|
|
43
|
+
the same activity) are removed and cut detection is retried. The log split
|
|
44
|
+
is then made *filtering-aware*: events that do not fit the chosen cut are
|
|
45
|
+
dropped. Fitness is no longer guaranteed to be 1, but the models are much
|
|
46
|
+
simpler. ``f = 0`` gives plain IM.
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
from __future__ import annotations
|
|
50
|
+
|
|
51
|
+
from collections import Counter, deque
|
|
52
|
+
from dataclasses import dataclass, field
|
|
53
|
+
|
|
54
|
+
from ..dfg import DFG, dfg_from_simple_log
|
|
55
|
+
from ..log import SimpleLog
|
|
56
|
+
from ..petrinet import PetriNet
|
|
57
|
+
from ..processtree import Operator, ProcessTree, to_petri_net
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass
|
|
61
|
+
class Step:
|
|
62
|
+
"""One recursion step, structured so the app can typeset it.
|
|
63
|
+
|
|
64
|
+
``kind`` is ``"cut"``, ``"base"``, ``"fall"`` (fall-through) or
|
|
65
|
+
``"filter"`` (IMf noise handling). ``operator`` is the process-tree
|
|
66
|
+
symbol produced (→ × ∧ ↺), ``groups`` the partition found by a cut, and
|
|
67
|
+
``result`` the leaf of a base case.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
depth: int
|
|
71
|
+
kind: str
|
|
72
|
+
text: str
|
|
73
|
+
operator: str | None = None
|
|
74
|
+
groups: list[list[str]] | None = None
|
|
75
|
+
result: str | None = None
|
|
76
|
+
detail: str | None = None
|
|
77
|
+
filtered: bool = False
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass
|
|
81
|
+
class InductiveResult:
|
|
82
|
+
tree: ProcessTree
|
|
83
|
+
net: PetriNet
|
|
84
|
+
#: One line per recursion step: which cut or fall-through was applied.
|
|
85
|
+
trace_of_steps: list[str] = field(default_factory=list)
|
|
86
|
+
#: The same steps, structured (see :class:`Step`).
|
|
87
|
+
steps: list[Step] = field(default_factory=list)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def inductive_miner(log: SimpleLog, noise_threshold: float = 0.0,
|
|
91
|
+
name: str | None = None) -> InductiveResult:
|
|
92
|
+
"""Discover a process tree (and its Petri net) with IM / IMf."""
|
|
93
|
+
miner = _Miner(noise_threshold)
|
|
94
|
+
tree = miner.mine(Counter({t: n for t, n in log.items() if n > 0}), depth=0).simplified()
|
|
95
|
+
label = name or ("IMf" if noise_threshold > 0 else "IM")
|
|
96
|
+
net = to_petri_net(tree, label)
|
|
97
|
+
net.info["algorithm"] = (f"Inductive Miner – infrequent (f = {noise_threshold:g})"
|
|
98
|
+
if noise_threshold > 0 else "Inductive Miner")
|
|
99
|
+
net.info["process tree"] = str(tree)
|
|
100
|
+
return InductiveResult(tree, net, miner.steps, miner.records)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
# ---------------------------------------------------------------------------
|
|
104
|
+
# The recursion
|
|
105
|
+
# ---------------------------------------------------------------------------
|
|
106
|
+
class _Miner:
|
|
107
|
+
def __init__(self, noise_threshold: float) -> None:
|
|
108
|
+
self.f = noise_threshold
|
|
109
|
+
self.steps: list[str] = []
|
|
110
|
+
self.records: list[Step] = []
|
|
111
|
+
|
|
112
|
+
def note(self, depth: int, text: str, kind: str = "fall", **extra) -> None:
|
|
113
|
+
self.steps.append(" " * depth + text)
|
|
114
|
+
self.records.append(Step(depth, kind, text, **extra))
|
|
115
|
+
|
|
116
|
+
def mine(self, log: SimpleLog, depth: int) -> ProcessTree:
|
|
117
|
+
activities = sorted({a for trace in log for a in trace})
|
|
118
|
+
total = sum(log.values())
|
|
119
|
+
empty = log.get((), 0)
|
|
120
|
+
|
|
121
|
+
# ---- base cases --------------------------------------------------
|
|
122
|
+
if total == 0 or not activities:
|
|
123
|
+
self.note(depth, "base case: only empty traces → τ", "base", result="τ")
|
|
124
|
+
return ProcessTree.tau()
|
|
125
|
+
if len(activities) == 1 and empty == 0 and all(len(t) == 1 for t in log):
|
|
126
|
+
self.note(depth, f"base case: single activity → {activities[0]}", "base",
|
|
127
|
+
result=activities[0])
|
|
128
|
+
return ProcessTree.leaf(activities[0])
|
|
129
|
+
|
|
130
|
+
# ---- empty traces --------------------------------------------------
|
|
131
|
+
if empty:
|
|
132
|
+
if self.f > 0 and empty / total < self.f:
|
|
133
|
+
self.note(depth, f"IMf: ignoring {empty} infrequent empty trace(s)", "filter",
|
|
134
|
+
detail=f"{empty} of {total} traces are empty, below f = {self.f:g}")
|
|
135
|
+
log = Counter({t: n for t, n in log.items() if t})
|
|
136
|
+
return self.mine(log, depth)
|
|
137
|
+
self.note(depth, f"fall-through: empty traces → ×(τ, …) [{empty} empty]",
|
|
138
|
+
operator="×", detail=f"{empty} empty trace(s): the part may be skipped")
|
|
139
|
+
rest = Counter({t: n for t, n in log.items() if t})
|
|
140
|
+
return ProcessTree.node(Operator.XOR, [ProcessTree.tau(), self.mine(rest, depth + 1)])
|
|
141
|
+
|
|
142
|
+
# ---- cuts on the full DFG ---------------------------------------------
|
|
143
|
+
dfg = dfg_from_simple_log(log)
|
|
144
|
+
found = self.find_cut(dfg)
|
|
145
|
+
filtered = False
|
|
146
|
+
if found is None and self.f > 0:
|
|
147
|
+
found = self.find_cut(filter_dfg(dfg, self.f))
|
|
148
|
+
filtered = found is not None
|
|
149
|
+
if found is not None:
|
|
150
|
+
operator, groups = found
|
|
151
|
+
described = " | ".join("{" + ", ".join(sorted(g)) + "}" for g in groups)
|
|
152
|
+
self.note(depth, f"{operator.value} cut{' (on filtered DFG)' if filtered else ''}: "
|
|
153
|
+
f"{described}", "cut", operator=operator.value,
|
|
154
|
+
groups=[sorted(g) for g in groups], filtered=filtered)
|
|
155
|
+
sublogs = split_log(log, operator, groups, dfg)
|
|
156
|
+
children = [self.mine(sub, depth + 1) for sub in sublogs]
|
|
157
|
+
if operator is Operator.LOOP and len(children) > 2:
|
|
158
|
+
# ↺(body, r1, r2, …) is kept as a loop with several redos.
|
|
159
|
+
return ProcessTree.node(Operator.LOOP, children)
|
|
160
|
+
return ProcessTree.node(operator, children)
|
|
161
|
+
|
|
162
|
+
# ---- fall-throughs ---------------------------------------------------
|
|
163
|
+
return self.fall_through(log, dfg, activities, depth)
|
|
164
|
+
|
|
165
|
+
# -------------------------------------------------------------------------
|
|
166
|
+
def find_cut(self, dfg: DFG):
|
|
167
|
+
for finder, operator in ((xor_cut, Operator.XOR), (sequence_cut, Operator.SEQUENCE),
|
|
168
|
+
(parallel_cut, Operator.PARALLEL), (loop_cut, Operator.LOOP)):
|
|
169
|
+
groups = finder(dfg)
|
|
170
|
+
if groups is not None and len(groups) > 1:
|
|
171
|
+
return operator, groups
|
|
172
|
+
return None
|
|
173
|
+
|
|
174
|
+
def fall_through(self, log: SimpleLog, dfg: DFG, activities: list[str],
|
|
175
|
+
depth: int) -> ProcessTree:
|
|
176
|
+
# 1. Activity once per trace: it can be put in parallel with the rest.
|
|
177
|
+
for activity in activities:
|
|
178
|
+
if all(trace.count(activity) == 1 for trace in log):
|
|
179
|
+
self.note(depth, f"fall-through: '{activity}' occurs once per trace → ∧({activity}, …)",
|
|
180
|
+
operator="∧", detail=f"{activity} occurs exactly once in every trace")
|
|
181
|
+
rest = _project_out(log, activity)
|
|
182
|
+
return ProcessTree.node(Operator.PARALLEL,
|
|
183
|
+
[ProcessTree.leaf(activity), self.mine(rest, depth + 1)])
|
|
184
|
+
|
|
185
|
+
# 2. Activity concurrent: removing one activity makes a cut appear.
|
|
186
|
+
if len(activities) > 2:
|
|
187
|
+
for activity in activities:
|
|
188
|
+
rest = _project_out(log, activity)
|
|
189
|
+
if self.find_cut(dfg_from_simple_log(rest)) is not None:
|
|
190
|
+
self.note(depth, f"fall-through: activity concurrent '{activity}' → ∧({activity}, …)",
|
|
191
|
+
operator="∧", detail=f"without {activity} a cut exists")
|
|
192
|
+
return ProcessTree.node(Operator.PARALLEL, [
|
|
193
|
+
self.mine(_keep_only(log, {activity}), depth + 1),
|
|
194
|
+
self.mine(rest, depth + 1)])
|
|
195
|
+
|
|
196
|
+
# 3. Strict τ-loop: split traces where an end activity is followed by
|
|
197
|
+
# a start activity.
|
|
198
|
+
split = _split_on(log, lambda a, b: a in dfg.end and b in dfg.start)
|
|
199
|
+
if sum(split.values()) > sum(log.values()):
|
|
200
|
+
self.note(depth, "fall-through: strict τ-loop → ↺(…, τ)", operator="↺",
|
|
201
|
+
detail="split traces where an end activity is followed by a start activity")
|
|
202
|
+
return ProcessTree.node(Operator.LOOP, [self.mine(split, depth + 1), ProcessTree.tau()])
|
|
203
|
+
|
|
204
|
+
# 4. τ-loop: split traces before every start activity.
|
|
205
|
+
split = _split_on(log, lambda a, b: b in dfg.start)
|
|
206
|
+
if sum(split.values()) > sum(log.values()):
|
|
207
|
+
self.note(depth, "fall-through: τ-loop → ↺(…, τ)", operator="↺",
|
|
208
|
+
detail="split traces before every start activity")
|
|
209
|
+
return ProcessTree.node(Operator.LOOP, [self.mine(split, depth + 1), ProcessTree.tau()])
|
|
210
|
+
|
|
211
|
+
# 5. Flower model: anything goes, in any order, any number of times.
|
|
212
|
+
self.note(depth, "fall-through: flower model ↺(τ, ×(" + ", ".join(activities) + "))",
|
|
213
|
+
operator="↺", detail="nothing else applies: allow any order (flower model)")
|
|
214
|
+
return ProcessTree.node(Operator.LOOP, [
|
|
215
|
+
ProcessTree.tau(),
|
|
216
|
+
ProcessTree.node(Operator.XOR, [ProcessTree.leaf(a) for a in activities])
|
|
217
|
+
if len(activities) > 1 else ProcessTree.leaf(activities[0])])
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
# ---------------------------------------------------------------------------
|
|
221
|
+
# Graph helpers
|
|
222
|
+
# ---------------------------------------------------------------------------
|
|
223
|
+
def _undirected_components(nodes: list[str], connected) -> list[set[str]]:
|
|
224
|
+
"""Connected components where ``connected(a, b)`` says whether a–b touch."""
|
|
225
|
+
remaining = set(nodes)
|
|
226
|
+
components = []
|
|
227
|
+
while remaining:
|
|
228
|
+
seed = min(remaining)
|
|
229
|
+
component = {seed}
|
|
230
|
+
queue = deque([seed])
|
|
231
|
+
remaining.discard(seed)
|
|
232
|
+
while queue:
|
|
233
|
+
node = queue.popleft()
|
|
234
|
+
for other in list(remaining):
|
|
235
|
+
if connected(node, other):
|
|
236
|
+
remaining.discard(other)
|
|
237
|
+
component.add(other)
|
|
238
|
+
queue.append(other)
|
|
239
|
+
components.append(component)
|
|
240
|
+
return components
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _reachability(dfg: DFG) -> dict[str, set[str]]:
|
|
244
|
+
"""For every activity, the activities reachable from it via ≥ 1 edge."""
|
|
245
|
+
succ = {a: set() for a in dfg.activities}
|
|
246
|
+
for a, b in dfg.edges:
|
|
247
|
+
succ[a].add(b)
|
|
248
|
+
reach = {}
|
|
249
|
+
for start in dfg.activities:
|
|
250
|
+
seen: set[str] = set()
|
|
251
|
+
queue = deque(succ[start])
|
|
252
|
+
while queue:
|
|
253
|
+
node = queue.popleft()
|
|
254
|
+
if node not in seen:
|
|
255
|
+
seen.add(node)
|
|
256
|
+
queue.extend(succ[node])
|
|
257
|
+
reach[start] = seen
|
|
258
|
+
return reach
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
# ---------------------------------------------------------------------------
|
|
262
|
+
# Cut detection
|
|
263
|
+
# ---------------------------------------------------------------------------
|
|
264
|
+
def xor_cut(dfg: DFG) -> list[set[str]] | None:
|
|
265
|
+
nodes = sorted(dfg.activities)
|
|
266
|
+
edges = set(dfg.edges)
|
|
267
|
+
groups = _undirected_components(nodes, lambda a, b: (a, b) in edges or (b, a) in edges)
|
|
268
|
+
return groups if len(groups) > 1 else None
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def sequence_cut(dfg: DFG) -> list[set[str]] | None:
|
|
272
|
+
reach = _reachability(dfg)
|
|
273
|
+
nodes = sorted(dfg.activities)
|
|
274
|
+
# Strongly connected components: a and b together iff each reaches the other.
|
|
275
|
+
groups: list[set[str]] = []
|
|
276
|
+
for node in nodes:
|
|
277
|
+
for group in groups:
|
|
278
|
+
other = next(iter(group))
|
|
279
|
+
if (other in reach[node] and node in reach[other]):
|
|
280
|
+
group.add(node)
|
|
281
|
+
break
|
|
282
|
+
else:
|
|
283
|
+
groups.append({node})
|
|
284
|
+
|
|
285
|
+
def reaches(g1: set[str], g2: set[str]) -> bool:
|
|
286
|
+
return any(b in reach[a] for a in g1 for b in g2)
|
|
287
|
+
|
|
288
|
+
# Merge groups that are mutually unreachable (neither can reach the other):
|
|
289
|
+
# they must sit in the same position of the sequence.
|
|
290
|
+
merged = True
|
|
291
|
+
while merged:
|
|
292
|
+
merged = False
|
|
293
|
+
for i in range(len(groups)):
|
|
294
|
+
for j in range(i + 1, len(groups)):
|
|
295
|
+
if not reaches(groups[i], groups[j]) and not reaches(groups[j], groups[i]):
|
|
296
|
+
groups[i] |= groups.pop(j)
|
|
297
|
+
merged = True
|
|
298
|
+
break
|
|
299
|
+
if merged:
|
|
300
|
+
break
|
|
301
|
+
if len(groups) < 2:
|
|
302
|
+
return None
|
|
303
|
+
# Order: g1 before g2 iff g1 reaches g2. Sorting by "number of groups
|
|
304
|
+
# reachable from me" (descending) gives a topological order.
|
|
305
|
+
# (Scores are computed first: during list.sort() the list looks empty to
|
|
306
|
+
# a key function that refers back to it.)
|
|
307
|
+
score = [sum(1 for other in groups if other is not g and reaches(g, other)) for g in groups]
|
|
308
|
+
groups = [g for _, g in sorted(zip(score, groups), key=lambda item: -item[0])]
|
|
309
|
+
# Validate: every earlier group reaches every later one and never back.
|
|
310
|
+
for i in range(len(groups)):
|
|
311
|
+
for j in range(i + 1, len(groups)):
|
|
312
|
+
if not reaches(groups[i], groups[j]) or reaches(groups[j], groups[i]):
|
|
313
|
+
return None
|
|
314
|
+
return groups
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def parallel_cut(dfg: DFG) -> list[set[str]] | None:
|
|
318
|
+
nodes = sorted(dfg.activities)
|
|
319
|
+
edges = set(dfg.edges)
|
|
320
|
+
# Negated graph: connect a, b unless they directly follow in both directions.
|
|
321
|
+
groups = _undirected_components(
|
|
322
|
+
nodes, lambda a, b: not ((a, b) in edges and (b, a) in edges))
|
|
323
|
+
if len(groups) < 2:
|
|
324
|
+
return None
|
|
325
|
+
# Every group needs a start and an end activity; merge groups that lack one.
|
|
326
|
+
good = [g for g in groups if g & set(dfg.start) and g & set(dfg.end)]
|
|
327
|
+
bad = [g for g in groups if not (g & set(dfg.start) and g & set(dfg.end))]
|
|
328
|
+
if not good:
|
|
329
|
+
return None
|
|
330
|
+
for group in bad:
|
|
331
|
+
good[0] |= group
|
|
332
|
+
if len(good) < 2:
|
|
333
|
+
return None
|
|
334
|
+
# Re-check the defining property after merging.
|
|
335
|
+
for i, g1 in enumerate(good):
|
|
336
|
+
for g2 in good[i + 1:]:
|
|
337
|
+
if not all((a, b) in edges and (b, a) in edges for a in g1 for b in g2):
|
|
338
|
+
return None
|
|
339
|
+
return good
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def loop_cut(dfg: DFG) -> list[set[str]] | None:
|
|
343
|
+
starts, ends = set(dfg.start), set(dfg.end)
|
|
344
|
+
body = starts | ends
|
|
345
|
+
others = sorted(set(dfg.activities) - body)
|
|
346
|
+
if not others:
|
|
347
|
+
return None
|
|
348
|
+
edges = set(dfg.edges)
|
|
349
|
+
# Components of the graph *without* the start/end activities.
|
|
350
|
+
redo_candidates = _undirected_components(
|
|
351
|
+
others, lambda a, b: (a, b) in edges or (b, a) in edges)
|
|
352
|
+
|
|
353
|
+
body = set(body)
|
|
354
|
+
redos: list[set[str]] = []
|
|
355
|
+
for group in redo_candidates:
|
|
356
|
+
belongs_to_body = False
|
|
357
|
+
for a in group:
|
|
358
|
+
# Entered from a start activity (other than via an end) → body.
|
|
359
|
+
if any((s, a) in edges for s in starts - ends):
|
|
360
|
+
belongs_to_body = True
|
|
361
|
+
# Leaves to an end activity (other than a start) → body.
|
|
362
|
+
if any((a, e) in edges for e in ends - starts):
|
|
363
|
+
belongs_to_body = True
|
|
364
|
+
# If some end activity leads to a, every end activity must.
|
|
365
|
+
if any((e, a) in edges for e in ends) and not all((e, a) in edges for e in ends):
|
|
366
|
+
belongs_to_body = True
|
|
367
|
+
# If a leads to some start activity, it must lead to all of them.
|
|
368
|
+
if any((a, s) in edges for s in starts) and not all((a, s) in edges for s in starts):
|
|
369
|
+
belongs_to_body = True
|
|
370
|
+
if belongs_to_body:
|
|
371
|
+
body |= group
|
|
372
|
+
else:
|
|
373
|
+
redos.append(group)
|
|
374
|
+
# Activities inside the body that are fed only by body nodes are fine;
|
|
375
|
+
# redo groups must actually connect end → redo → start.
|
|
376
|
+
redos = [g for g in redos
|
|
377
|
+
if any((e, a) in edges for e in ends for a in g)
|
|
378
|
+
and any((a, s) in edges for a in g for s in starts)]
|
|
379
|
+
leftover = set(dfg.activities) - body - set().union(*redos) if redos else set()
|
|
380
|
+
body |= leftover
|
|
381
|
+
if not redos:
|
|
382
|
+
return None
|
|
383
|
+
return [body] + redos
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
def filter_dfg(dfg: DFG, threshold: float) -> DFG:
|
|
387
|
+
"""IMf filtering: drop edges below ``threshold`` × strongest outgoing edge."""
|
|
388
|
+
strongest: dict[str, int] = Counter()
|
|
389
|
+
for (a, _), n in dfg.edges.items():
|
|
390
|
+
strongest[a] = max(strongest[a], n)
|
|
391
|
+
for a, n in dfg.end.items():
|
|
392
|
+
strongest[a] = max(strongest[a], n)
|
|
393
|
+
result = DFG(activities=Counter(dfg.activities), trace_count=dfg.trace_count)
|
|
394
|
+
result.edges = Counter({(a, b): n for (a, b), n in dfg.edges.items()
|
|
395
|
+
if n >= threshold * strongest[a]})
|
|
396
|
+
max_start = max(dfg.start.values(), default=0)
|
|
397
|
+
max_end = max(dfg.end.values(), default=0)
|
|
398
|
+
result.start = Counter({a: n for a, n in dfg.start.items() if n >= threshold * max_start})
|
|
399
|
+
result.end = Counter({a: n for a, n in dfg.end.items() if n >= threshold * strongest[a]
|
|
400
|
+
or n >= threshold * max_end})
|
|
401
|
+
return result
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
# ---------------------------------------------------------------------------
|
|
405
|
+
# Log splitting
|
|
406
|
+
# ---------------------------------------------------------------------------
|
|
407
|
+
def _project_out(log: SimpleLog, activity: str) -> SimpleLog:
|
|
408
|
+
result: SimpleLog = Counter()
|
|
409
|
+
for trace, n in log.items():
|
|
410
|
+
result[tuple(a for a in trace if a != activity)] += n
|
|
411
|
+
return result
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def _keep_only(log: SimpleLog, keep: set[str]) -> SimpleLog:
|
|
415
|
+
result: SimpleLog = Counter()
|
|
416
|
+
for trace, n in log.items():
|
|
417
|
+
result[tuple(a for a in trace if a in keep)] += n
|
|
418
|
+
return result
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
def _split_on(log: SimpleLog, boundary) -> SimpleLog:
|
|
422
|
+
"""Cut every trace between consecutive a, b where ``boundary(a, b)``."""
|
|
423
|
+
result: SimpleLog = Counter()
|
|
424
|
+
for trace, n in log.items():
|
|
425
|
+
piece: list[str] = []
|
|
426
|
+
for index, activity in enumerate(trace):
|
|
427
|
+
if piece and boundary(trace[index - 1], activity):
|
|
428
|
+
result[tuple(piece)] += n
|
|
429
|
+
piece = []
|
|
430
|
+
piece.append(activity)
|
|
431
|
+
result[tuple(piece)] += n
|
|
432
|
+
return result
|
|
433
|
+
|
|
434
|
+
|
|
435
|
+
def split_log(log: SimpleLog, operator: Operator, groups: list[set[str]],
|
|
436
|
+
dfg: DFG) -> list[SimpleLog]:
|
|
437
|
+
"""Divide the log over the groups of a cut."""
|
|
438
|
+
sublogs: list[SimpleLog] = [Counter() for _ in groups]
|
|
439
|
+
if operator is Operator.XOR:
|
|
440
|
+
# Each trace goes to the group it overlaps most with; with IM every
|
|
441
|
+
# trace lies entirely in one group, with IMf stray events are dropped.
|
|
442
|
+
for trace, n in log.items():
|
|
443
|
+
best = max(range(len(groups)), key=lambda i: sum(a in groups[i] for a in trace))
|
|
444
|
+
sublogs[best][tuple(a for a in trace if a in groups[best])] += n
|
|
445
|
+
elif operator in (Operator.SEQUENCE, Operator.PARALLEL):
|
|
446
|
+
# Projection on each group. For a sequence cut found on the full DFG
|
|
447
|
+
# this is exactly the split into consecutive segments.
|
|
448
|
+
for trace, n in log.items():
|
|
449
|
+
for i, group in enumerate(groups):
|
|
450
|
+
sublogs[i][tuple(a for a in trace if a in group)] += n
|
|
451
|
+
elif operator is Operator.LOOP:
|
|
452
|
+
# Cut every trace into alternating body / redo segments. A body
|
|
453
|
+
# segment always comes first and last; if a trace starts or ends in
|
|
454
|
+
# a redo group (only possible with noise), an empty body segment is
|
|
455
|
+
# implied, which becomes an empty trace in the body log.
|
|
456
|
+
group_of = {a: i for i, g in enumerate(groups) for a in g}
|
|
457
|
+
for trace, n in log.items():
|
|
458
|
+
# 1. Chop the trace into maximal runs belonging to one group.
|
|
459
|
+
runs: list[tuple[int, list[str]]] = []
|
|
460
|
+
for activity in trace:
|
|
461
|
+
g = group_of.get(activity, 0)
|
|
462
|
+
if runs and runs[-1][0] == g:
|
|
463
|
+
runs[-1][1].append(activity)
|
|
464
|
+
else:
|
|
465
|
+
runs.append((g, [activity]))
|
|
466
|
+
# 2. Enforce body, redo, body, redo, ..., body.
|
|
467
|
+
normalised: list[tuple[int, list[str]]] = []
|
|
468
|
+
for g, run in runs:
|
|
469
|
+
previous_is_redo = normalised and normalised[-1][0] != 0
|
|
470
|
+
if g != 0 and (not normalised or previous_is_redo):
|
|
471
|
+
normalised.append((0, []))
|
|
472
|
+
normalised.append((g, run))
|
|
473
|
+
if not normalised or normalised[-1][0] != 0:
|
|
474
|
+
normalised.append((0, []))
|
|
475
|
+
for g, run in normalised:
|
|
476
|
+
sublogs[g][tuple(run)] += n
|
|
477
|
+
return [Counter({t: c for t, c in sub.items() if c}) for sub in sublogs]
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""Discovery with state-based regions: log → transition system → Petri net.
|
|
2
|
+
|
|
3
|
+
The standard example of *two-phase* discovery (van der Aalst, *Process
|
|
4
|
+
Mining*, Section 7.4):
|
|
5
|
+
|
|
6
|
+
1. build a **transition system** from the log with a state function
|
|
7
|
+
(:func:`~openprocess.mining.transition_system.transition_system_from_log`):
|
|
8
|
+
the prefix, postfix or both of every event, as a sequence, multiset or
|
|
9
|
+
set, over a horizon of the last ``k`` events or all of them;
|
|
10
|
+
2. find its **regions** and **minimal regions**
|
|
11
|
+
(:func:`~openprocess.mining.regions.analyse_regions`), and check whether it is
|
|
12
|
+
**elementary** (state separation and forward closure);
|
|
13
|
+
3. **synthesise** a Petri net with one place per minimal region
|
|
14
|
+
(:func:`~openprocess.mining.regions.synthesise`). For an elementary
|
|
15
|
+
transition system the net's reachability graph is isomorphic to it.
|
|
16
|
+
|
|
17
|
+
The abstraction is the knob: a coarser one (a set, a short horizon) merges
|
|
18
|
+
states and so generalises, a finer one (the full sequence) only allows the
|
|
19
|
+
log. The result keeps every step so the app can show the derivation.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
from dataclasses import dataclass, field
|
|
25
|
+
|
|
26
|
+
from ..log import SimpleLog
|
|
27
|
+
from ..petrinet import PetriNet
|
|
28
|
+
from ..regions import RegionAnalysis, Synthesis, analyse_regions, synthesise
|
|
29
|
+
from ..transition_system import TransitionSystem, transition_system_from_log
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class RegionResult:
|
|
34
|
+
ts: TransitionSystem
|
|
35
|
+
analysis: RegionAnalysis
|
|
36
|
+
synthesis: Synthesis
|
|
37
|
+
warnings: list[str] = field(default_factory=list)
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
def net(self) -> PetriNet | None:
|
|
41
|
+
return self.synthesis.net
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def region_result(ts: TransitionSystem) -> RegionResult:
|
|
45
|
+
"""Regions and synthesis for a transition system already built (or typed)."""
|
|
46
|
+
analysis = analyse_regions(ts)
|
|
47
|
+
synthesis = synthesise(ts, analysis)
|
|
48
|
+
return RegionResult(ts, analysis, synthesis, list(synthesis.warnings))
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def region_miner(log: SimpleLog, direction: str = "prefix", representation: str = "set",
|
|
52
|
+
horizon: int | None = None) -> RegionResult:
|
|
53
|
+
"""Discover a Petri net from ``log`` through state-based regions.
|
|
54
|
+
|
|
55
|
+
Raises ``ValueError`` when no net can be made (several initial states, or
|
|
56
|
+
a transition system too large to search for regions).
|
|
57
|
+
"""
|
|
58
|
+
ts = transition_system_from_log(log, direction, representation, horizon)
|
|
59
|
+
result = region_result(ts)
|
|
60
|
+
if result.net is None:
|
|
61
|
+
raise ValueError(" ".join(result.warnings) or "No net could be synthesised.")
|
|
62
|
+
return result
|