openprocess 0.7.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cpnpy/__init__.py +46 -0
- openprocess/__init__.py +57 -0
- openprocess/analysis/__init__.py +0 -0
- openprocess/analysis/state_space.py +521 -0
- openprocess/analysis/state_space_process.py +251 -0
- openprocess/cli.py +742 -0
- openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/answer.pnml +31 -0
- openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/question.md +38 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/answer.pnml +27 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/net.pnml +29 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/question.md +65 -0
- openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/log.txt +1 -0
- openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/question.md +67 -0
- openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/question.md +75 -0
- openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/ts.txt +5 -0
- openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/net.pnml +27 -0
- openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/question.md +65 -0
- openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/log.txt +1 -0
- openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/question.md +62 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/log.txt +1 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m1.pnml +36 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m2.pnml +28 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m3.pnml +30 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/net.pnml +36 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/question.md +69 -0
- openprocess/exercises/pack.md +14 -0
- openprocess/flow/__init__.py +50 -0
- openprocess/flow/box.py +466 -0
- openprocess/flow/boxes/__init__.py +7 -0
- openprocess/flow/boxes/check.py +119 -0
- openprocess/flow/boxes/compare.py +16 -0
- openprocess/flow/boxes/cpn.py +53 -0
- openprocess/flow/boxes/discover.py +124 -0
- openprocess/flow/boxes/filter.py +80 -0
- openprocess/flow/boxes/input.py +124 -0
- openprocess/flow/boxes/output.py +52 -0
- openprocess/flow/boxes/predict.py +186 -0
- openprocess/flow/boxes/science.py +159 -0
- openprocess/flow/boxes/sweeps.py +18 -0
- openprocess/flow/convert.py +187 -0
- openprocess/flow/datasets.py +198 -0
- openprocess/flow/explain.py +115 -0
- openprocess/flow/library.py +222 -0
- openprocess/flow/record.py +385 -0
- openprocess/flow/runner.py +357 -0
- openprocess/flow/sweep.py +92 -0
- openprocess/flow/types.py +290 -0
- openprocess/flow/workflow.py +628 -0
- openprocess/gui/__init__.py +0 -0
- openprocess/gui/app.py +90 -0
- openprocess/gui/arc_editing.py +295 -0
- openprocess/gui/canvas.py +1414 -0
- openprocess/gui/flow/__init__.py +8 -0
- openprocess/gui/flow/canvas.py +854 -0
- openprocess/gui/flow/page.py +972 -0
- openprocess/gui/flow/templates.py +131 -0
- openprocess/gui/flow/viewers.py +665 -0
- openprocess/gui/items.py +1275 -0
- openprocess/gui/learn/answer_boxes.py +978 -0
- openprocess/gui/learn/concealment.py +91 -0
- openprocess/gui/learn/mode.py +1181 -0
- openprocess/gui/panning.py +241 -0
- openprocess/gui/resources/openprocess-icon.png +0 -0
- openprocess/gui/studio/__init__.py +1 -0
- openprocess/gui/studio/__main__.py +3 -0
- openprocess/gui/studio/app.py +4031 -0
- openprocess/gui/studio/charts.py +115 -0
- openprocess/gui/studio/compare_page.py +487 -0
- openprocess/gui/studio/cpn_page.py +1858 -0
- openprocess/gui/studio/definition_view.py +284 -0
- openprocess/gui/studio/derivation_view.py +421 -0
- openprocess/gui/studio/documents.py +152 -0
- openprocess/gui/studio/dotted_chart.py +1401 -0
- openprocess/gui/studio/file_dialogs.py +143 -0
- openprocess/gui/studio/filter_dialog.py +247 -0
- openprocess/gui/studio/graph_builders.py +176 -0
- openprocess/gui/studio/graph_view.py +682 -0
- openprocess/gui/studio/instances.py +413 -0
- openprocess/gui/studio/log_editor.py +675 -0
- openprocess/gui/studio/log_page.py +800 -0
- openprocess/gui/studio/markdown_view.py +127 -0
- openprocess/gui/studio/mathtext.py +260 -0
- openprocess/gui/studio/ml_highlighter.py +75 -0
- openprocess/gui/studio/model_page.py +760 -0
- openprocess/gui/studio/net_comparison.py +124 -0
- openprocess/gui/studio/notes_overlay.py +275 -0
- openprocess/gui/studio/petri_page.py +844 -0
- openprocess/gui/studio/regions_view.py +502 -0
- openprocess/gui/studio/sidebar.py +149 -0
- openprocess/gui/studio/style.py +503 -0
- openprocess/gui/studio/tool_icons.py +134 -0
- openprocess/gui/studio/updates.py +439 -0
- openprocess/gui/studio/widgets.py +899 -0
- openprocess/gui/studio/workers.py +60 -0
- openprocess/gui/studio/workspace.py +447 -0
- openprocess/gui/theme.py +394 -0
- openprocess/gui/tidy.py +86 -0
- openprocess/io/__init__.py +0 -0
- openprocess/io/cpn_reader.py +389 -0
- openprocess/io/cpn_writer.py +357 -0
- openprocess/learn/__init__.py +23 -0
- openprocess/learn/answers.py +188 -0
- openprocess/learn/checks.py +953 -0
- openprocess/learn/computed.py +1180 -0
- openprocess/learn/context.py +145 -0
- openprocess/learn/exam.py +169 -0
- openprocess/learn/exercise-packs.md +325 -0
- openprocess/learn/importer.py +216 -0
- openprocess/learn/notation.py +474 -0
- openprocess/learn/pack.py +511 -0
- openprocess/learn/sheet.py +296 -0
- openprocess/mining/__init__.py +73 -0
- openprocess/mining/analysis.py +689 -0
- openprocess/mining/columns.py +282 -0
- openprocess/mining/compare_nets.py +246 -0
- openprocess/mining/conformance/__init__.py +0 -0
- openprocess/mining/conformance/alignments.py +263 -0
- openprocess/mining/conformance/quality.py +145 -0
- openprocess/mining/conformance/token_replay.py +252 -0
- openprocess/mining/csv_import.py +222 -0
- openprocess/mining/definitions.py +584 -0
- openprocess/mining/dfg.py +187 -0
- openprocess/mining/discovery/__init__.py +0 -0
- openprocess/mining/discovery/alpha.py +168 -0
- openprocess/mining/discovery/heuristics.py +332 -0
- openprocess/mining/discovery/inductive.py +477 -0
- openprocess/mining/discovery/state_regions.py +62 -0
- openprocess/mining/filtering.py +237 -0
- openprocess/mining/footprint.py +183 -0
- openprocess/mining/invariants.py +191 -0
- openprocess/mining/layout.py +279 -0
- openprocess/mining/log.py +364 -0
- openprocess/mining/petrinet.py +354 -0
- openprocess/mining/playout.py +75 -0
- openprocess/mining/pm4py_bridge.py +82 -0
- openprocess/mining/pnml.py +223 -0
- openprocess/mining/processtree.py +216 -0
- openprocess/mining/regions.py +476 -0
- openprocess/mining/stats.py +160 -0
- openprocess/mining/structure.py +374 -0
- openprocess/mining/transition_system.py +409 -0
- openprocess/mining/xes.py +399 -0
- openprocess/ml/__init__.py +0 -0
- openprocess/ml/ast_nodes.py +332 -0
- openprocess/ml/builtins.py +364 -0
- openprocess/ml/colorsets.py +522 -0
- openprocess/ml/errors.py +60 -0
- openprocess/ml/evaluator.py +754 -0
- openprocess/ml/lexer.py +277 -0
- openprocess/ml/multiset.py +417 -0
- openprocess/ml/parser.py +737 -0
- openprocess/ml/values.py +319 -0
- openprocess/model/__init__.py +0 -0
- openprocess/model/declarations.py +617 -0
- openprocess/model/examples.py +98 -0
- openprocess/model/net.py +701 -0
- openprocess/model/plain.py +192 -0
- openprocess/references.py +280 -0
- openprocess/sim/__init__.py +0 -0
- openprocess/sim/binding.py +620 -0
- openprocess/sim/export.py +66 -0
- openprocess/sim/simulator.py +315 -0
- openprocess/teaching/__init__.py +4 -0
- openprocess/teaching/answers.py +4 -0
- openprocess/teaching/checks.py +5 -0
- openprocess/teaching/pack.py +4 -0
- openprocess/teaching/sheet.py +4 -0
- openprocess-0.7.0.dist-info/METADATA +927 -0
- openprocess-0.7.0.dist-info/RECORD +173 -0
- openprocess-0.7.0.dist-info/WHEEL +5 -0
- openprocess-0.7.0.dist-info/entry_points.txt +6 -0
- openprocess-0.7.0.dist-info/licenses/LICENSE +21 -0
- openprocess-0.7.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,476 @@
|
|
|
1
|
+
"""State-based regions: from a transition system to a Petri net.
|
|
2
|
+
|
|
3
|
+
Definitions (Cortadella et al. 1998; van der Aalst, *Process Mining*, Section 7.4.2)
|
|
4
|
+
--------------------------------------------------------------------------------------
|
|
5
|
+
Let ``TS = (S, E, T, s_in)`` be a transition system.
|
|
6
|
+
|
|
7
|
+
region
|
|
8
|
+
A set of states ``R ⊆ S`` such that, for every event ``e``, *all*
|
|
9
|
+
transitions labelled ``e`` do the same thing with respect to ``R``:
|
|
10
|
+
they all **enter** it (start outside, end inside), all **exit** it
|
|
11
|
+
(start inside, end outside), or all **do not cross** it (both ends
|
|
12
|
+
inside, or both outside). ``∅`` and ``S`` are the trivial regions. The
|
|
13
|
+
complement of a region is a region too (enter and exit swap).
|
|
14
|
+
minimal region
|
|
15
|
+
A non-empty region that contains no smaller non-empty region.
|
|
16
|
+
pre-region / post-region
|
|
17
|
+
``R`` is a pre-region of ``e`` if ``e`` exits ``R``, a post-region if
|
|
18
|
+
``e`` enters it. A region is a place in disguise: its pre-events put a
|
|
19
|
+
token in, its post-events take it out.
|
|
20
|
+
generalised excitation region
|
|
21
|
+
``GER(e)``: the states in which ``e`` is enabled.
|
|
22
|
+
elementary transition system
|
|
23
|
+
One that satisfies
|
|
24
|
+
*state separation* -- any two distinct states are told apart by some
|
|
25
|
+
region (one is inside it, the other is not) -- and
|
|
26
|
+
*forward closure* -- for every event, the intersection of its
|
|
27
|
+
pre-regions is exactly ``GER(e)``, i.e. the places before ``e`` are
|
|
28
|
+
marked only where ``e`` really is enabled.
|
|
29
|
+
|
|
30
|
+
Synthesis
|
|
31
|
+
---------
|
|
32
|
+
One place per minimal region, an arc ``R → e`` for each pre-region and
|
|
33
|
+
``e → R`` for each post-region, and a token in every minimal region that
|
|
34
|
+
contains ``s_in``. For an elementary transition system the reachability
|
|
35
|
+
graph of the result is isomorphic to the transition system (checked by
|
|
36
|
+
:func:`synthesise`).
|
|
37
|
+
|
|
38
|
+
Finding the regions
|
|
39
|
+
-------------------
|
|
40
|
+
Course-sized transition systems are solved exactly: every assignment of the
|
|
41
|
+
states to inside/outside is explored, abandoning an assignment as soon as two
|
|
42
|
+
transitions with the same event cross it differently. That is the same as
|
|
43
|
+
checking every subset of states, only much faster, and it is limited to
|
|
44
|
+
:data:`MAX_STATES` states (and :data:`MAX_REGIONS` regions); the result says
|
|
45
|
+
when a limit cut it short. (Larger systems need the expansion algorithm of
|
|
46
|
+
Cortadella et al., which is not implemented.)
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
from __future__ import annotations
|
|
50
|
+
|
|
51
|
+
from collections import deque
|
|
52
|
+
from dataclasses import dataclass, field
|
|
53
|
+
|
|
54
|
+
from .petrinet import Marking, PetriNet
|
|
55
|
+
from .transition_system import TransitionSystem, isomorphic, transition_system_of_graph
|
|
56
|
+
|
|
57
|
+
#: Larger transition systems are not searched for regions.
|
|
58
|
+
MAX_STATES = 26
|
|
59
|
+
#: Stop collecting regions after this many (the result says so).
|
|
60
|
+
MAX_REGIONS = 50_000
|
|
61
|
+
|
|
62
|
+
ENTER, EXIT, INSIDE, OUTSIDE = "enter", "exit", "inside", "outside"
|
|
63
|
+
#: ``inside`` and ``outside`` are both "does not cross".
|
|
64
|
+
NO_CROSS = "no cross"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def crossing(region: frozenset[str] | set[str], source: str, target: str) -> str:
|
|
68
|
+
"""What one transition does with respect to ``region``."""
|
|
69
|
+
a, b = source in region, target in region
|
|
70
|
+
if a and b:
|
|
71
|
+
return INSIDE
|
|
72
|
+
if not a and not b:
|
|
73
|
+
return OUTSIDE
|
|
74
|
+
return EXIT if a else ENTER
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _kind(relation: str) -> str:
|
|
78
|
+
return NO_CROSS if relation in (INSIDE, OUTSIDE) else relation
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def format_states(states, order: list[str] | None = None) -> str:
|
|
82
|
+
"""``{s0, s2, s3}`` in the transition system's own order."""
|
|
83
|
+
items = list(states)
|
|
84
|
+
if order is not None:
|
|
85
|
+
position = {s: i for i, s in enumerate(order)}
|
|
86
|
+
items.sort(key=lambda s: position.get(s, len(position)))
|
|
87
|
+
else:
|
|
88
|
+
items.sort()
|
|
89
|
+
return "{" + ", ".join(items) + "}"
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
# ---------------------------------------------------------------------------
|
|
93
|
+
# Is this a region?
|
|
94
|
+
# ---------------------------------------------------------------------------
|
|
95
|
+
@dataclass
|
|
96
|
+
class RegionCheck:
|
|
97
|
+
states: frozenset[str]
|
|
98
|
+
is_region: bool
|
|
99
|
+
#: For a region: what each event does (enter, exit or no cross).
|
|
100
|
+
events: dict[str, str] = field(default_factory=dict)
|
|
101
|
+
#: When it is not: the event that breaks it, and two of its transitions
|
|
102
|
+
#: that cross the set differently, with what each does.
|
|
103
|
+
event: str | None = None
|
|
104
|
+
conflict: list[tuple[str, str, str]] = field(default_factory=list)
|
|
105
|
+
unknown: list[str] = field(default_factory=list)
|
|
106
|
+
#: The empty set or every state.
|
|
107
|
+
trivial: bool = False
|
|
108
|
+
|
|
109
|
+
def explanation(self, order: list[str] | None = None) -> str:
|
|
110
|
+
shown = format_states(self.states, order)
|
|
111
|
+
if self.unknown:
|
|
112
|
+
return "Not states of this transition system: " + ", ".join(self.unknown) + "."
|
|
113
|
+
if self.is_region:
|
|
114
|
+
trivial = " (a trivial region)" if self.trivial else ""
|
|
115
|
+
parts = [f"{e} {what}s" if what != NO_CROSS else f"{e} does not cross"
|
|
116
|
+
for e, what in self.events.items()]
|
|
117
|
+
return f"{shown} is a region{trivial}: " + ", ".join(parts) + "."
|
|
118
|
+
(s1, t1, r1), (s2, t2, r2) = self.conflict[:2]
|
|
119
|
+
|
|
120
|
+
def said(relation: str) -> str:
|
|
121
|
+
return {ENTER: "enters", EXIT: "exits", INSIDE: "does not cross (stays inside)",
|
|
122
|
+
OUTSIDE: "does not cross (stays outside)"}[relation]
|
|
123
|
+
return (f"{shown} is not a region: event {self.event} has {s1} -{self.event}-> {t1}, "
|
|
124
|
+
f"which {said(r1)}, but {s2} -{self.event}-> {t2}, which {said(r2)}.")
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def check_region(ts: TransitionSystem, states) -> RegionCheck:
|
|
128
|
+
"""Is ``states`` a region of ``ts``? If not, which event breaks it and how."""
|
|
129
|
+
region = frozenset(states)
|
|
130
|
+
unknown = sorted(region - set(ts.states))
|
|
131
|
+
if unknown:
|
|
132
|
+
return RegionCheck(region, False, unknown=unknown)
|
|
133
|
+
events: dict[str, str] = {}
|
|
134
|
+
for event in ts.events:
|
|
135
|
+
first: tuple[str, str, str] | None = None
|
|
136
|
+
for source, target in ts.transitions_of(event):
|
|
137
|
+
relation = crossing(region, source, target)
|
|
138
|
+
if first is None:
|
|
139
|
+
first = (source, target, relation)
|
|
140
|
+
events[event] = _kind(relation)
|
|
141
|
+
elif _kind(relation) != _kind(first[2]):
|
|
142
|
+
return RegionCheck(region, False, event=event,
|
|
143
|
+
conflict=[first, (source, target, relation)])
|
|
144
|
+
return RegionCheck(region, True, events=events,
|
|
145
|
+
trivial=not region or region == frozenset(ts.states))
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
# ---------------------------------------------------------------------------
|
|
149
|
+
# All regions
|
|
150
|
+
# ---------------------------------------------------------------------------
|
|
151
|
+
@dataclass
|
|
152
|
+
class EventRegions:
|
|
153
|
+
"""Everything about one event's place in the net."""
|
|
154
|
+
|
|
155
|
+
event: str
|
|
156
|
+
ger: frozenset[str]
|
|
157
|
+
pre: list[frozenset[str]] # every pre-region (e exits)
|
|
158
|
+
post: list[frozenset[str]] # every post-region (e enters)
|
|
159
|
+
minimal_pre: list[frozenset[str]]
|
|
160
|
+
minimal_post: list[frozenset[str]]
|
|
161
|
+
#: ⋂ pre(e); S when e has no pre-region.
|
|
162
|
+
intersection: frozenset[str]
|
|
163
|
+
|
|
164
|
+
@property
|
|
165
|
+
def forward_closed(self) -> bool:
|
|
166
|
+
return self.intersection == self.ger
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
@dataclass
|
|
170
|
+
class Separation:
|
|
171
|
+
"""Two states no region tells apart, and why."""
|
|
172
|
+
|
|
173
|
+
first: str
|
|
174
|
+
second: str
|
|
175
|
+
#: Events whose transitions force every region to contain both or neither,
|
|
176
|
+
#: as (event, shared neighbour, "target" | "source").
|
|
177
|
+
reasons: list[tuple[str, str, str]] = field(default_factory=list)
|
|
178
|
+
|
|
179
|
+
def explanation(self) -> str:
|
|
180
|
+
a, b = self.first, self.second
|
|
181
|
+
if not self.reasons:
|
|
182
|
+
return (f"{a} and {b} cannot be separated: every region contains both of them "
|
|
183
|
+
"or neither.")
|
|
184
|
+
lines = []
|
|
185
|
+
for event, other, role in self.reasons:
|
|
186
|
+
if role == "target":
|
|
187
|
+
lines.append(f"{a} -{event}-> {other} and {b} -{event}-> {other}: a region "
|
|
188
|
+
f"with {a} but not {b} would have one {event} step stay put and "
|
|
189
|
+
f"the other cross it, whether {other} is inside or not")
|
|
190
|
+
else:
|
|
191
|
+
lines.append(f"{other} -{event}-> {a} and {other} -{event}-> {b}: a region "
|
|
192
|
+
f"with {a} but not {b} would have one {event} step stay put and "
|
|
193
|
+
f"the other cross it, whether {other} is inside or not")
|
|
194
|
+
return f"{a} and {b} cannot be separated. " + "; ".join(lines) + "."
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
@dataclass
|
|
198
|
+
class RegionAnalysis:
|
|
199
|
+
ts: TransitionSystem
|
|
200
|
+
#: Every non-trivial region, smallest first.
|
|
201
|
+
regions: list[frozenset[str]] = field(default_factory=list)
|
|
202
|
+
minimal: list[frozenset[str]] = field(default_factory=list)
|
|
203
|
+
by_event: dict[str, EventRegions] = field(default_factory=dict)
|
|
204
|
+
#: Pairs of distinct states no region separates.
|
|
205
|
+
inseparable: list[Separation] = field(default_factory=list)
|
|
206
|
+
#: The search stopped early (too many states or regions): verdicts are not proofs.
|
|
207
|
+
truncated: bool = False
|
|
208
|
+
#: Why the search did not run or stopped.
|
|
209
|
+
limit_message: str = ""
|
|
210
|
+
|
|
211
|
+
@property
|
|
212
|
+
def state_separation(self) -> bool | None:
|
|
213
|
+
return None if self.truncated else not self.inseparable
|
|
214
|
+
|
|
215
|
+
@property
|
|
216
|
+
def forward_closure(self) -> bool | None:
|
|
217
|
+
return None if self.truncated else all(r.forward_closed for r in self.by_event.values())
|
|
218
|
+
|
|
219
|
+
@property
|
|
220
|
+
def elementary(self) -> bool | None:
|
|
221
|
+
if self.state_separation is False or self.forward_closure is False:
|
|
222
|
+
return False
|
|
223
|
+
if self.state_separation is None or self.forward_closure is None:
|
|
224
|
+
return None
|
|
225
|
+
return True
|
|
226
|
+
|
|
227
|
+
def closure_failures(self) -> list[tuple[str, frozenset[str]]]:
|
|
228
|
+
"""Each event whose pre-regions meet outside GER(e), with those extra states."""
|
|
229
|
+
return [(event, r.intersection - r.ger) for event, r in self.by_event.items()
|
|
230
|
+
if not r.forward_closed]
|
|
231
|
+
|
|
232
|
+
def name_of(self, region: frozenset[str]) -> str:
|
|
233
|
+
"""``r3`` for the third minimal region (places are named so)."""
|
|
234
|
+
try:
|
|
235
|
+
return f"r{self.minimal.index(region) + 1}"
|
|
236
|
+
except ValueError:
|
|
237
|
+
return format_states(region, self.ts.states)
|
|
238
|
+
|
|
239
|
+
def region_text(self, region: frozenset[str]) -> str:
|
|
240
|
+
return format_states(region, self.ts.states)
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _enumerate(ts: TransitionSystem, limit: int) -> tuple[list[int], bool]:
|
|
244
|
+
"""Every non-trivial region as a bitmask over ``ts.states`` (and whether cut short)."""
|
|
245
|
+
states = ts.states
|
|
246
|
+
index = {s: i for i, s in enumerate(states)}
|
|
247
|
+
n = len(states)
|
|
248
|
+
# Visit states breadth-first from the start, so neighbours are decided
|
|
249
|
+
# close together and conflicts show up early.
|
|
250
|
+
order: list[int] = []
|
|
251
|
+
seen: set[int] = set()
|
|
252
|
+
adjacency: dict[int, set[int]] = {i: set() for i in range(n)}
|
|
253
|
+
for s, _, t in ts.transitions:
|
|
254
|
+
adjacency[index[s]].add(index[t])
|
|
255
|
+
adjacency[index[t]].add(index[s])
|
|
256
|
+
for root in [index[s] for s in ts.initial if s in index] + list(range(n)):
|
|
257
|
+
if root in seen:
|
|
258
|
+
continue
|
|
259
|
+
seen.add(root)
|
|
260
|
+
queue = deque([root])
|
|
261
|
+
while queue:
|
|
262
|
+
node = queue.popleft()
|
|
263
|
+
order.append(node)
|
|
264
|
+
for other in sorted(adjacency[node]):
|
|
265
|
+
if other not in seen:
|
|
266
|
+
seen.add(other)
|
|
267
|
+
queue.append(other)
|
|
268
|
+
events = ts.events
|
|
269
|
+
event_index = {e: k for k, e in enumerate(events)}
|
|
270
|
+
# For each state: the transitions touching it, as (event no., source, target).
|
|
271
|
+
touching: dict[int, list[tuple[int, int, int]]] = {i: [] for i in range(n)}
|
|
272
|
+
for s, e, t in ts.transitions:
|
|
273
|
+
step = (event_index[e], index[s], index[t])
|
|
274
|
+
touching[index[s]].append(step)
|
|
275
|
+
if t != s:
|
|
276
|
+
touching[index[t]].append(step)
|
|
277
|
+
value = [-1] * n
|
|
278
|
+
# counts[event][kind]: decided transitions of the event of each kind (0 enter, 1 exit, 2 no).
|
|
279
|
+
counts = [[0, 0, 0] for _ in events]
|
|
280
|
+
found: list[int] = []
|
|
281
|
+
full = (1 << n) - 1
|
|
282
|
+
stopped = [False]
|
|
283
|
+
|
|
284
|
+
def kind(source: int, target: int) -> int:
|
|
285
|
+
a, b = value[source], value[target]
|
|
286
|
+
if a == b:
|
|
287
|
+
return 2
|
|
288
|
+
return 0 if b == 1 else 1
|
|
289
|
+
|
|
290
|
+
def assign(position: int, mask: int) -> None:
|
|
291
|
+
if stopped[0]:
|
|
292
|
+
return
|
|
293
|
+
if position == n:
|
|
294
|
+
if mask and mask != full:
|
|
295
|
+
found.append(mask)
|
|
296
|
+
if len(found) >= limit:
|
|
297
|
+
stopped[0] = True
|
|
298
|
+
return
|
|
299
|
+
state = order[position]
|
|
300
|
+
for choice in (0, 1):
|
|
301
|
+
value[state] = choice
|
|
302
|
+
changed = []
|
|
303
|
+
ok = True
|
|
304
|
+
for event, source, target in touching[state]:
|
|
305
|
+
if value[source] < 0 or value[target] < 0:
|
|
306
|
+
continue
|
|
307
|
+
k = kind(source, target)
|
|
308
|
+
counts[event][k] += 1
|
|
309
|
+
changed.append((event, k))
|
|
310
|
+
tally = counts[event]
|
|
311
|
+
if (tally[0] > 0) + (tally[1] > 0) + (tally[2] > 0) > 1:
|
|
312
|
+
ok = False
|
|
313
|
+
break
|
|
314
|
+
if ok:
|
|
315
|
+
assign(position + 1, mask | (1 << state) if choice else mask)
|
|
316
|
+
for event, k in changed:
|
|
317
|
+
counts[event][k] -= 1
|
|
318
|
+
value[state] = -1
|
|
319
|
+
if stopped[0]:
|
|
320
|
+
return
|
|
321
|
+
|
|
322
|
+
assign(0, 0)
|
|
323
|
+
return found, stopped[0]
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def analyse_regions(ts: TransitionSystem, max_states: int = MAX_STATES,
|
|
327
|
+
max_regions: int = MAX_REGIONS) -> RegionAnalysis:
|
|
328
|
+
"""Regions, minimal regions, pre-/post-regions, GER and the elementary checks."""
|
|
329
|
+
analysis = RegionAnalysis(ts)
|
|
330
|
+
states = ts.states
|
|
331
|
+
if len(states) > max_states:
|
|
332
|
+
analysis.truncated = True
|
|
333
|
+
analysis.limit_message = (f"The transition system has {len(states)} states; regions "
|
|
334
|
+
f"are only searched for up to {max_states}.")
|
|
335
|
+
return analysis
|
|
336
|
+
masks, cut = _enumerate(ts, max_regions)
|
|
337
|
+
if cut:
|
|
338
|
+
analysis.truncated = True
|
|
339
|
+
analysis.limit_message = (f"Stopped after {max_regions:,} regions; the lists are "
|
|
340
|
+
"incomplete and the verdicts are not proofs.")
|
|
341
|
+
masks.sort(key=lambda m: (bin(m).count("1"), m))
|
|
342
|
+
|
|
343
|
+
def as_set(mask: int) -> frozenset[str]:
|
|
344
|
+
return frozenset(s for i, s in enumerate(states) if mask >> i & 1)
|
|
345
|
+
|
|
346
|
+
minimal_masks: list[int] = []
|
|
347
|
+
for mask in masks:
|
|
348
|
+
if not any(m & mask == m for m in minimal_masks):
|
|
349
|
+
minimal_masks.append(mask)
|
|
350
|
+
analysis.regions = [as_set(m) for m in masks]
|
|
351
|
+
analysis.minimal = [as_set(m) for m in minimal_masks]
|
|
352
|
+
minimal_set = set(minimal_masks)
|
|
353
|
+
|
|
354
|
+
index = {s: i for i, s in enumerate(states)}
|
|
355
|
+
every = (1 << len(states)) - 1
|
|
356
|
+
for event in ts.events:
|
|
357
|
+
steps = [(index[s], index[t]) for s, t in ts.transitions_of(event)]
|
|
358
|
+
ger_mask = 0
|
|
359
|
+
for source, _ in steps:
|
|
360
|
+
ger_mask |= 1 << source
|
|
361
|
+
pre, post = [], []
|
|
362
|
+
for mask in masks:
|
|
363
|
+
source, target = steps[0]
|
|
364
|
+
a, b = mask >> source & 1, mask >> target & 1
|
|
365
|
+
if a and not b:
|
|
366
|
+
pre.append(mask)
|
|
367
|
+
elif b and not a:
|
|
368
|
+
post.append(mask)
|
|
369
|
+
intersection = every
|
|
370
|
+
for mask in pre:
|
|
371
|
+
intersection &= mask
|
|
372
|
+
analysis.by_event[event] = EventRegions(
|
|
373
|
+
event, as_set(ger_mask), [as_set(m) for m in pre], [as_set(m) for m in post],
|
|
374
|
+
[as_set(m) for m in pre if m in minimal_set],
|
|
375
|
+
[as_set(m) for m in post if m in minimal_set], as_set(intersection))
|
|
376
|
+
|
|
377
|
+
# State separation: two states are told apart by some region unless every
|
|
378
|
+
# region contains both or neither, i.e. they have the same "signature".
|
|
379
|
+
signature: dict[str, int] = {s: 0 for s in states}
|
|
380
|
+
for number, mask in enumerate(masks):
|
|
381
|
+
for i, s in enumerate(states):
|
|
382
|
+
if mask >> i & 1:
|
|
383
|
+
signature[s] |= 1 << number
|
|
384
|
+
groups: dict[int, list[str]] = {}
|
|
385
|
+
for s in states:
|
|
386
|
+
groups.setdefault(signature[s], []).append(s)
|
|
387
|
+
for group in groups.values():
|
|
388
|
+
for i, a in enumerate(group):
|
|
389
|
+
for b in group[i + 1:]:
|
|
390
|
+
analysis.inseparable.append(Separation(a, b, _separation_reasons(ts, a, b)))
|
|
391
|
+
return analysis
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
def _separation_reasons(ts: TransitionSystem, a: str, b: str) -> list[tuple[str, str, str]]:
|
|
395
|
+
"""Events that go from ``a`` and ``b`` to one state (or come from one state):
|
|
396
|
+
those two steps would cross a region containing only one of them differently."""
|
|
397
|
+
reasons = []
|
|
398
|
+
for event in ts.events:
|
|
399
|
+
steps = ts.transitions_of(event)
|
|
400
|
+
from_a = {t for s, t in steps if s == a}
|
|
401
|
+
from_b = {t for s, t in steps if s == b}
|
|
402
|
+
for other in sorted(from_a & from_b):
|
|
403
|
+
reasons.append((event, other, "target"))
|
|
404
|
+
into_a = {s for s, t in steps if t == a}
|
|
405
|
+
into_b = {s for s, t in steps if t == b}
|
|
406
|
+
for other in sorted(into_a & into_b):
|
|
407
|
+
reasons.append((event, other, "source"))
|
|
408
|
+
return reasons
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
# ---------------------------------------------------------------------------
|
|
412
|
+
# Synthesis
|
|
413
|
+
# ---------------------------------------------------------------------------
|
|
414
|
+
@dataclass
|
|
415
|
+
class Synthesis:
|
|
416
|
+
analysis: RegionAnalysis
|
|
417
|
+
net: PetriNet | None
|
|
418
|
+
#: Is the net's reachability graph isomorphic to the transition system?
|
|
419
|
+
#: None when it could not be checked.
|
|
420
|
+
isomorphic: bool | None = None
|
|
421
|
+
warnings: list[str] = field(default_factory=list)
|
|
422
|
+
#: Place id -> its minimal region.
|
|
423
|
+
places: dict[str, frozenset[str]] = field(default_factory=dict)
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def synthesise(ts: TransitionSystem, analysis: RegionAnalysis | None = None,
|
|
427
|
+
name: str | None = None) -> Synthesis:
|
|
428
|
+
"""The Petri net of the minimal regions (see the module docstring)."""
|
|
429
|
+
analysis = analysis or analyse_regions(ts)
|
|
430
|
+
result = Synthesis(analysis, None)
|
|
431
|
+
if analysis.limit_message and not analysis.regions:
|
|
432
|
+
result.warnings.append(analysis.limit_message)
|
|
433
|
+
return result
|
|
434
|
+
start = ts.initial_state
|
|
435
|
+
if start is None:
|
|
436
|
+
result.warnings.append(
|
|
437
|
+
f"The transition system has {len(ts.initial)} initial states; region synthesis "
|
|
438
|
+
"needs exactly one (the postfix gives every trace its own start, so use the "
|
|
439
|
+
"prefix).")
|
|
440
|
+
return result
|
|
441
|
+
if analysis.truncated:
|
|
442
|
+
result.warnings.append(analysis.limit_message)
|
|
443
|
+
net = PetriNet(name or f"Regions · {ts.name}")
|
|
444
|
+
net.info["algorithm"] = "State-based regions"
|
|
445
|
+
if ts.abstraction:
|
|
446
|
+
net.info["abstraction"] = ts.abstraction
|
|
447
|
+
transition_of = {}
|
|
448
|
+
for event in ts.events:
|
|
449
|
+
transition_of[event] = net.add_transition(event, id=f"t_{len(transition_of) + 1}").id
|
|
450
|
+
for number, region in enumerate(analysis.minimal, 1):
|
|
451
|
+
place = net.add_place(f"r{number}", id=f"r{number}")
|
|
452
|
+
result.places[place.id] = region
|
|
453
|
+
for event, regions in analysis.by_event.items():
|
|
454
|
+
if region in regions.minimal_pre:
|
|
455
|
+
net.add_arc(place.id, transition_of[event])
|
|
456
|
+
if region in regions.minimal_post:
|
|
457
|
+
net.add_arc(transition_of[event], place.id)
|
|
458
|
+
net.initial_marking = Marking({p: 1 for p, region in result.places.items() if start in region})
|
|
459
|
+
# The final marking: when every final state stands for the same marking.
|
|
460
|
+
finals = {Marking({p: 1 for p, region in result.places.items() if end in region})
|
|
461
|
+
for end in ts.final}
|
|
462
|
+
if len(finals) == 1:
|
|
463
|
+
net.final_marking = finals.pop()
|
|
464
|
+
loops = sorted({e for s, e, t in ts.transitions if s == t})
|
|
465
|
+
if loops:
|
|
466
|
+
result.warnings.append("Self-loops (" + ", ".join(loops) + ") cross no region, so "
|
|
467
|
+
"the net cannot show them.")
|
|
468
|
+
if analysis.elementary is False:
|
|
469
|
+
result.warnings.append("The transition system is not elementary, so the net does not "
|
|
470
|
+
"behave exactly like it (it may allow more).")
|
|
471
|
+
result.net = net
|
|
472
|
+
from .analysis import reachability_graph
|
|
473
|
+
graph = reachability_graph(net, max_states=max(2_000, 20 * len(ts.states)))
|
|
474
|
+
if not graph.truncated:
|
|
475
|
+
result.isomorphic = isomorphic(transition_system_of_graph(graph), ts)
|
|
476
|
+
return result
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
"""Descriptive statistics of an event log.
|
|
2
|
+
|
|
3
|
+
Before discovering anything it pays to *look* at the log: how many cases,
|
|
4
|
+
how many distinct behaviours (variants), which activities start and end
|
|
5
|
+
cases, how long cases take. This module computes those numbers once so the
|
|
6
|
+
views can display them without recomputing.
|
|
7
|
+
|
|
8
|
+
Variants
|
|
9
|
+
--------
|
|
10
|
+
A *variant* is a distinct activity sequence. Two cases belong to the same
|
|
11
|
+
variant iff they have exactly the same sequence under the chosen classifier.
|
|
12
|
+
The number of variants relative to the number of cases is a quick measure of
|
|
13
|
+
how structured the process is: 1 variant means perfectly repetitive; one
|
|
14
|
+
variant per case means every case is unique (a "spaghetti" process).
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import statistics
|
|
20
|
+
from collections import Counter
|
|
21
|
+
from dataclasses import dataclass, field
|
|
22
|
+
from datetime import datetime, timedelta
|
|
23
|
+
|
|
24
|
+
from .log import Classifier, EventLog
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class Variant:
|
|
29
|
+
"""One distinct sequence, with the indices of the traces that follow it."""
|
|
30
|
+
|
|
31
|
+
sequence: tuple[str, ...]
|
|
32
|
+
trace_indices: list[int] = field(default_factory=list)
|
|
33
|
+
|
|
34
|
+
@property
|
|
35
|
+
def count(self) -> int:
|
|
36
|
+
return len(self.trace_indices)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class ActivityStats:
|
|
41
|
+
name: str
|
|
42
|
+
occurrences: int # total number of events with this label
|
|
43
|
+
cases: int # number of cases containing it at least once
|
|
44
|
+
as_start: int # cases starting with it
|
|
45
|
+
as_end: int # cases ending with it
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass
|
|
49
|
+
class LogSummary:
|
|
50
|
+
"""Everything the overview page shows."""
|
|
51
|
+
|
|
52
|
+
case_count: int
|
|
53
|
+
event_count: int # events *kept* by the classifier
|
|
54
|
+
variants: list[Variant] # most frequent first
|
|
55
|
+
activities: list[ActivityStats] # most frequent first
|
|
56
|
+
start: datetime | None # earliest timestamp
|
|
57
|
+
end: datetime | None # latest timestamp
|
|
58
|
+
case_durations: list[timedelta] # one per case that has timestamps
|
|
59
|
+
events_per_case: Counter # length -> number of cases
|
|
60
|
+
|
|
61
|
+
@property
|
|
62
|
+
def variant_count(self) -> int:
|
|
63
|
+
return len(self.variants)
|
|
64
|
+
|
|
65
|
+
@property
|
|
66
|
+
def activity_count(self) -> int:
|
|
67
|
+
return len(self.activities)
|
|
68
|
+
|
|
69
|
+
@property
|
|
70
|
+
def mean_case_duration(self) -> timedelta | None:
|
|
71
|
+
if not self.case_durations:
|
|
72
|
+
return None
|
|
73
|
+
seconds = statistics.fmean(d.total_seconds() for d in self.case_durations)
|
|
74
|
+
return timedelta(seconds=seconds)
|
|
75
|
+
|
|
76
|
+
@property
|
|
77
|
+
def median_case_duration(self) -> timedelta | None:
|
|
78
|
+
if not self.case_durations:
|
|
79
|
+
return None
|
|
80
|
+
return timedelta(seconds=statistics.median(
|
|
81
|
+
d.total_seconds() for d in self.case_durations))
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def summarise(log: EventLog, classifier: Classifier | None = None) -> LogSummary:
|
|
85
|
+
"""Compute a :class:`LogSummary` in one pass over the log."""
|
|
86
|
+
classifier = classifier or log.default_classifier()
|
|
87
|
+
|
|
88
|
+
by_sequence: dict[tuple[str, ...], Variant] = {}
|
|
89
|
+
occurrences: Counter = Counter()
|
|
90
|
+
cases_with: Counter = Counter()
|
|
91
|
+
starts: Counter = Counter()
|
|
92
|
+
ends: Counter = Counter()
|
|
93
|
+
durations: list[timedelta] = []
|
|
94
|
+
lengths: Counter = Counter()
|
|
95
|
+
first: datetime | None = None
|
|
96
|
+
last: datetime | None = None
|
|
97
|
+
kept_events = 0
|
|
98
|
+
|
|
99
|
+
for index, trace in enumerate(log.traces):
|
|
100
|
+
sequence = log.sequence(trace, classifier)
|
|
101
|
+
by_sequence.setdefault(sequence, Variant(sequence)).trace_indices.append(index)
|
|
102
|
+
occurrences.update(sequence)
|
|
103
|
+
cases_with.update(set(sequence))
|
|
104
|
+
kept_events += len(sequence)
|
|
105
|
+
lengths[len(sequence)] += 1
|
|
106
|
+
if sequence:
|
|
107
|
+
starts[sequence[0]] += 1
|
|
108
|
+
ends[sequence[-1]] += 1
|
|
109
|
+
|
|
110
|
+
# Timestamps are taken from *all* events, not only the classified
|
|
111
|
+
# ones: a case's duration does not change with the classifier.
|
|
112
|
+
stamps = [e.timestamp for e in trace.events if e.timestamp is not None]
|
|
113
|
+
if stamps:
|
|
114
|
+
low, high = min(stamps), max(stamps)
|
|
115
|
+
durations.append(high - low)
|
|
116
|
+
first = low if first is None or low < first else first
|
|
117
|
+
last = high if last is None or high > last else last
|
|
118
|
+
|
|
119
|
+
variants = sorted(by_sequence.values(), key=lambda v: (-v.count, v.sequence))
|
|
120
|
+
activities = sorted(
|
|
121
|
+
(ActivityStats(name, occurrences[name], cases_with[name], starts[name], ends[name])
|
|
122
|
+
for name in occurrences),
|
|
123
|
+
key=lambda a: (-a.occurrences, a.name))
|
|
124
|
+
|
|
125
|
+
return LogSummary(
|
|
126
|
+
case_count=len(log.traces),
|
|
127
|
+
event_count=kept_events,
|
|
128
|
+
variants=variants,
|
|
129
|
+
activities=activities,
|
|
130
|
+
start=first,
|
|
131
|
+
end=last,
|
|
132
|
+
case_durations=durations,
|
|
133
|
+
events_per_case=lengths,
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def format_duration(delta: timedelta | float | None) -> str:
|
|
138
|
+
"""Human-friendly duration: ``3d 4h``, ``12m 5s``, ``850ms``.
|
|
139
|
+
|
|
140
|
+
Two units at most -- enough precision to compare, short enough to scan.
|
|
141
|
+
"""
|
|
142
|
+
if delta is None:
|
|
143
|
+
return "–"
|
|
144
|
+
seconds = delta.total_seconds() if isinstance(delta, timedelta) else float(delta)
|
|
145
|
+
if seconds < 0:
|
|
146
|
+
return "-" + format_duration(-seconds)
|
|
147
|
+
if seconds < 1:
|
|
148
|
+
return f"{seconds * 1000:.0f}ms"
|
|
149
|
+
units = [("y", 365 * 86400), ("d", 86400), ("h", 3600), ("m", 60), ("s", 1)]
|
|
150
|
+
parts: list[str] = []
|
|
151
|
+
remaining = seconds
|
|
152
|
+
for suffix, size in units:
|
|
153
|
+
if remaining >= size or parts:
|
|
154
|
+
amount = int(remaining // size)
|
|
155
|
+
remaining -= amount * size
|
|
156
|
+
if amount or parts:
|
|
157
|
+
parts.append(f"{amount}{suffix}")
|
|
158
|
+
if len(parts) == 2:
|
|
159
|
+
break
|
|
160
|
+
return " ".join(p for p in parts if not p.startswith("0")) or "0s"
|