openprocess 0.7.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cpnpy/__init__.py +46 -0
- openprocess/__init__.py +57 -0
- openprocess/analysis/__init__.py +0 -0
- openprocess/analysis/state_space.py +521 -0
- openprocess/analysis/state_space_process.py +251 -0
- openprocess/cli.py +742 -0
- openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/answer.pnml +31 -0
- openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/question.md +38 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/answer.pnml +27 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/net.pnml +29 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/question.md +65 -0
- openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/log.txt +1 -0
- openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/question.md +67 -0
- openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/question.md +75 -0
- openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/ts.txt +5 -0
- openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/net.pnml +27 -0
- openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/question.md +65 -0
- openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/log.txt +1 -0
- openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/question.md +62 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/log.txt +1 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m1.pnml +36 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m2.pnml +28 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m3.pnml +30 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/net.pnml +36 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/question.md +69 -0
- openprocess/exercises/pack.md +14 -0
- openprocess/flow/__init__.py +50 -0
- openprocess/flow/box.py +466 -0
- openprocess/flow/boxes/__init__.py +7 -0
- openprocess/flow/boxes/check.py +119 -0
- openprocess/flow/boxes/compare.py +16 -0
- openprocess/flow/boxes/cpn.py +53 -0
- openprocess/flow/boxes/discover.py +124 -0
- openprocess/flow/boxes/filter.py +80 -0
- openprocess/flow/boxes/input.py +124 -0
- openprocess/flow/boxes/output.py +52 -0
- openprocess/flow/boxes/predict.py +186 -0
- openprocess/flow/boxes/science.py +159 -0
- openprocess/flow/boxes/sweeps.py +18 -0
- openprocess/flow/convert.py +187 -0
- openprocess/flow/datasets.py +198 -0
- openprocess/flow/explain.py +115 -0
- openprocess/flow/library.py +222 -0
- openprocess/flow/record.py +385 -0
- openprocess/flow/runner.py +357 -0
- openprocess/flow/sweep.py +92 -0
- openprocess/flow/types.py +290 -0
- openprocess/flow/workflow.py +628 -0
- openprocess/gui/__init__.py +0 -0
- openprocess/gui/app.py +90 -0
- openprocess/gui/arc_editing.py +295 -0
- openprocess/gui/canvas.py +1414 -0
- openprocess/gui/flow/__init__.py +8 -0
- openprocess/gui/flow/canvas.py +854 -0
- openprocess/gui/flow/page.py +972 -0
- openprocess/gui/flow/templates.py +131 -0
- openprocess/gui/flow/viewers.py +665 -0
- openprocess/gui/items.py +1275 -0
- openprocess/gui/learn/answer_boxes.py +978 -0
- openprocess/gui/learn/concealment.py +91 -0
- openprocess/gui/learn/mode.py +1181 -0
- openprocess/gui/panning.py +241 -0
- openprocess/gui/resources/openprocess-icon.png +0 -0
- openprocess/gui/studio/__init__.py +1 -0
- openprocess/gui/studio/__main__.py +3 -0
- openprocess/gui/studio/app.py +4031 -0
- openprocess/gui/studio/charts.py +115 -0
- openprocess/gui/studio/compare_page.py +487 -0
- openprocess/gui/studio/cpn_page.py +1858 -0
- openprocess/gui/studio/definition_view.py +284 -0
- openprocess/gui/studio/derivation_view.py +421 -0
- openprocess/gui/studio/documents.py +152 -0
- openprocess/gui/studio/dotted_chart.py +1401 -0
- openprocess/gui/studio/file_dialogs.py +143 -0
- openprocess/gui/studio/filter_dialog.py +247 -0
- openprocess/gui/studio/graph_builders.py +176 -0
- openprocess/gui/studio/graph_view.py +682 -0
- openprocess/gui/studio/instances.py +413 -0
- openprocess/gui/studio/log_editor.py +675 -0
- openprocess/gui/studio/log_page.py +800 -0
- openprocess/gui/studio/markdown_view.py +127 -0
- openprocess/gui/studio/mathtext.py +260 -0
- openprocess/gui/studio/ml_highlighter.py +75 -0
- openprocess/gui/studio/model_page.py +760 -0
- openprocess/gui/studio/net_comparison.py +124 -0
- openprocess/gui/studio/notes_overlay.py +275 -0
- openprocess/gui/studio/petri_page.py +844 -0
- openprocess/gui/studio/regions_view.py +502 -0
- openprocess/gui/studio/sidebar.py +149 -0
- openprocess/gui/studio/style.py +503 -0
- openprocess/gui/studio/tool_icons.py +134 -0
- openprocess/gui/studio/updates.py +439 -0
- openprocess/gui/studio/widgets.py +899 -0
- openprocess/gui/studio/workers.py +60 -0
- openprocess/gui/studio/workspace.py +447 -0
- openprocess/gui/theme.py +394 -0
- openprocess/gui/tidy.py +86 -0
- openprocess/io/__init__.py +0 -0
- openprocess/io/cpn_reader.py +389 -0
- openprocess/io/cpn_writer.py +357 -0
- openprocess/learn/__init__.py +23 -0
- openprocess/learn/answers.py +188 -0
- openprocess/learn/checks.py +953 -0
- openprocess/learn/computed.py +1180 -0
- openprocess/learn/context.py +145 -0
- openprocess/learn/exam.py +169 -0
- openprocess/learn/exercise-packs.md +325 -0
- openprocess/learn/importer.py +216 -0
- openprocess/learn/notation.py +474 -0
- openprocess/learn/pack.py +511 -0
- openprocess/learn/sheet.py +296 -0
- openprocess/mining/__init__.py +73 -0
- openprocess/mining/analysis.py +689 -0
- openprocess/mining/columns.py +282 -0
- openprocess/mining/compare_nets.py +246 -0
- openprocess/mining/conformance/__init__.py +0 -0
- openprocess/mining/conformance/alignments.py +263 -0
- openprocess/mining/conformance/quality.py +145 -0
- openprocess/mining/conformance/token_replay.py +252 -0
- openprocess/mining/csv_import.py +222 -0
- openprocess/mining/definitions.py +584 -0
- openprocess/mining/dfg.py +187 -0
- openprocess/mining/discovery/__init__.py +0 -0
- openprocess/mining/discovery/alpha.py +168 -0
- openprocess/mining/discovery/heuristics.py +332 -0
- openprocess/mining/discovery/inductive.py +477 -0
- openprocess/mining/discovery/state_regions.py +62 -0
- openprocess/mining/filtering.py +237 -0
- openprocess/mining/footprint.py +183 -0
- openprocess/mining/invariants.py +191 -0
- openprocess/mining/layout.py +279 -0
- openprocess/mining/log.py +364 -0
- openprocess/mining/petrinet.py +354 -0
- openprocess/mining/playout.py +75 -0
- openprocess/mining/pm4py_bridge.py +82 -0
- openprocess/mining/pnml.py +223 -0
- openprocess/mining/processtree.py +216 -0
- openprocess/mining/regions.py +476 -0
- openprocess/mining/stats.py +160 -0
- openprocess/mining/structure.py +374 -0
- openprocess/mining/transition_system.py +409 -0
- openprocess/mining/xes.py +399 -0
- openprocess/ml/__init__.py +0 -0
- openprocess/ml/ast_nodes.py +332 -0
- openprocess/ml/builtins.py +364 -0
- openprocess/ml/colorsets.py +522 -0
- openprocess/ml/errors.py +60 -0
- openprocess/ml/evaluator.py +754 -0
- openprocess/ml/lexer.py +277 -0
- openprocess/ml/multiset.py +417 -0
- openprocess/ml/parser.py +737 -0
- openprocess/ml/values.py +319 -0
- openprocess/model/__init__.py +0 -0
- openprocess/model/declarations.py +617 -0
- openprocess/model/examples.py +98 -0
- openprocess/model/net.py +701 -0
- openprocess/model/plain.py +192 -0
- openprocess/references.py +280 -0
- openprocess/sim/__init__.py +0 -0
- openprocess/sim/binding.py +620 -0
- openprocess/sim/export.py +66 -0
- openprocess/sim/simulator.py +315 -0
- openprocess/teaching/__init__.py +4 -0
- openprocess/teaching/answers.py +4 -0
- openprocess/teaching/checks.py +5 -0
- openprocess/teaching/pack.py +4 -0
- openprocess/teaching/sheet.py +4 -0
- openprocess-0.7.0.dist-info/METADATA +927 -0
- openprocess-0.7.0.dist-info/RECORD +173 -0
- openprocess-0.7.0.dist-info/WHEEL +5 -0
- openprocess-0.7.0.dist-info/entry_points.txt +6 -0
- openprocess-0.7.0.dist-info/licenses/LICENSE +21 -0
- openprocess-0.7.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,237 @@
|
|
|
1
|
+
"""Filtering event logs: keep the part of the log you want to look at.
|
|
2
|
+
|
|
3
|
+
Real logs are messy: rare variants, noise activities, cases cut off by the
|
|
4
|
+
extraction window. Filtering is therefore the usual first step before
|
|
5
|
+
discovery, as in ProM ("Filter Log using Simple Heuristics") and Disco. Every
|
|
6
|
+
filter here returns a **new** log and never changes the original, so you can
|
|
7
|
+
always compare the filtered log with the one it came from.
|
|
8
|
+
|
|
9
|
+
The filters, in the order :func:`apply_filters` applies them:
|
|
10
|
+
|
|
11
|
+
1. **Time frame** -- keep cases *contained in*, *started in*, or
|
|
12
|
+
*intersecting* a period. Cases without timestamps cannot be placed in
|
|
13
|
+
time and are dropped.
|
|
14
|
+
2. **Start and end activities** -- keep cases that start (end) with one of
|
|
15
|
+
the chosen activities, e.g. to drop cases cut off by the extraction.
|
|
16
|
+
3. **Activities** -- three modes, as in Disco:
|
|
17
|
+
|
|
18
|
+
* *keep events*: remove every event of the other activities (a projection;
|
|
19
|
+
cases keep their other events);
|
|
20
|
+
* *mandatory*: keep cases that contain at least one of the activities;
|
|
21
|
+
* *forbidden*: keep cases that contain none of them.
|
|
22
|
+
|
|
23
|
+
4. **Case length** -- keep cases with between ``min`` and ``max`` events.
|
|
24
|
+
5. **Variants** -- keep the most frequent variants, either the top ``k`` or
|
|
25
|
+
as many as needed to cover a percentage of the cases.
|
|
26
|
+
|
|
27
|
+
Activities, variants and lengths are seen through the log's *classifier*
|
|
28
|
+
(see :mod:`.log`), so they match what the rest of the app shows.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
from collections import Counter
|
|
34
|
+
from dataclasses import dataclass, field
|
|
35
|
+
from datetime import datetime
|
|
36
|
+
|
|
37
|
+
from .log import KEY_NAME, Classifier, EventLog, Trace
|
|
38
|
+
|
|
39
|
+
TIME_MODES = ("contained", "started", "intersecting")
|
|
40
|
+
ACTIVITY_MODES = ("keep events", "mandatory", "forbidden")
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass
|
|
44
|
+
class FilterSettings:
|
|
45
|
+
"""Which filters to apply. ``None`` (or empty) means "do not filter on this"."""
|
|
46
|
+
|
|
47
|
+
#: ``(from, to, mode)`` with ``mode`` in :data:`TIME_MODES`.
|
|
48
|
+
time_frame: tuple[datetime, datetime, str] | None = None
|
|
49
|
+
start_activities: set[str] | None = None
|
|
50
|
+
end_activities: set[str] | None = None
|
|
51
|
+
#: ``(activities, mode)`` with ``mode`` in :data:`ACTIVITY_MODES`.
|
|
52
|
+
activities: tuple[set[str], str] | None = None
|
|
53
|
+
#: ``(minimum, maximum)`` number of events (after the activity filter).
|
|
54
|
+
case_length: tuple[int, int] | None = None
|
|
55
|
+
#: Keep the variants covering this percentage of the cases (0-100].
|
|
56
|
+
variant_coverage: float | None = None
|
|
57
|
+
#: Or: keep the ``k`` most frequent variants.
|
|
58
|
+
top_variants: int | None = None
|
|
59
|
+
|
|
60
|
+
def describe(self) -> list[str]:
|
|
61
|
+
"""One line per active filter, for the record kept in the new log."""
|
|
62
|
+
lines: list[str] = []
|
|
63
|
+
if self.time_frame is not None:
|
|
64
|
+
start, end, mode = self.time_frame
|
|
65
|
+
lines.append(f"cases {mode} {start:%Y-%m-%d %H:%M} – {end:%Y-%m-%d %H:%M}")
|
|
66
|
+
if self.start_activities is not None:
|
|
67
|
+
lines.append("start with " + ", ".join(sorted(self.start_activities)))
|
|
68
|
+
if self.end_activities is not None:
|
|
69
|
+
lines.append("end with " + ", ".join(sorted(self.end_activities)))
|
|
70
|
+
if self.activities is not None:
|
|
71
|
+
names, mode = self.activities
|
|
72
|
+
lines.append(f"activities ({mode}): " + ", ".join(sorted(names)))
|
|
73
|
+
if self.case_length is not None:
|
|
74
|
+
lines.append(f"{self.case_length[0]} to {self.case_length[1]} events per case")
|
|
75
|
+
if self.variant_coverage is not None:
|
|
76
|
+
lines.append(f"most frequent variants covering {self.variant_coverage:g}% of cases")
|
|
77
|
+
if self.top_variants is not None:
|
|
78
|
+
lines.append(f"the {self.top_variants} most frequent variants")
|
|
79
|
+
return lines
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
# ---------------------------------------------------------------------------
|
|
83
|
+
# Single filters (each returns a new log)
|
|
84
|
+
# ---------------------------------------------------------------------------
|
|
85
|
+
def _derived(log: EventLog, traces: list[Trace]) -> EventLog:
|
|
86
|
+
result = log.filtered([])
|
|
87
|
+
result.traces = traces
|
|
88
|
+
return result
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _span(trace: Trace) -> tuple[datetime, datetime] | None:
|
|
92
|
+
stamps = [event.timestamp for event in trace if event.timestamp is not None]
|
|
93
|
+
return (min(stamps), max(stamps)) if stamps else None
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def filter_time_frame(log: EventLog, start: datetime, end: datetime,
|
|
97
|
+
mode: str = "contained") -> EventLog:
|
|
98
|
+
"""Cases *contained* in, *started* in, or *intersecting* ``[start, end]``."""
|
|
99
|
+
if mode not in TIME_MODES:
|
|
100
|
+
raise ValueError(f"mode must be one of {TIME_MODES}")
|
|
101
|
+
kept = []
|
|
102
|
+
for trace in log:
|
|
103
|
+
span = _span(trace)
|
|
104
|
+
if span is None:
|
|
105
|
+
continue
|
|
106
|
+
first, last = span
|
|
107
|
+
if (mode == "contained" and start <= first and last <= end) or \
|
|
108
|
+
(mode == "started" and start <= first <= end) or \
|
|
109
|
+
(mode == "intersecting" and first <= end and last >= start):
|
|
110
|
+
kept.append(trace)
|
|
111
|
+
return _derived(log, kept)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def filter_start_activities(log: EventLog, allowed: set[str],
|
|
115
|
+
classifier: Classifier | None = None) -> EventLog:
|
|
116
|
+
"""Cases whose first activity is one of ``allowed``."""
|
|
117
|
+
kept = [trace for trace, sequence in zip(log, log.sequences(classifier))
|
|
118
|
+
if sequence and sequence[0] in allowed]
|
|
119
|
+
return _derived(log, kept)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def filter_end_activities(log: EventLog, allowed: set[str],
|
|
123
|
+
classifier: Classifier | None = None) -> EventLog:
|
|
124
|
+
"""Cases whose last activity is one of ``allowed``."""
|
|
125
|
+
kept = [trace for trace, sequence in zip(log, log.sequences(classifier))
|
|
126
|
+
if sequence and sequence[-1] in allowed]
|
|
127
|
+
return _derived(log, kept)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def filter_activities(log: EventLog, activities: set[str], mode: str = "keep events",
|
|
131
|
+
classifier: Classifier | None = None) -> EventLog:
|
|
132
|
+
"""Project onto ``activities``, or keep cases that must (not) contain them.
|
|
133
|
+
|
|
134
|
+
*keep events* removes the events of every other activity -- including
|
|
135
|
+
events the classifier skips (a *start* event of a kept activity stays, so
|
|
136
|
+
the dotted chart still shows it). Cases left without events are dropped.
|
|
137
|
+
"""
|
|
138
|
+
if mode not in ACTIVITY_MODES:
|
|
139
|
+
raise ValueError(f"mode must be one of {ACTIVITY_MODES}")
|
|
140
|
+
classifier = classifier or log.default_classifier()
|
|
141
|
+
if mode == "keep events":
|
|
142
|
+
kept = []
|
|
143
|
+
for trace in log:
|
|
144
|
+
events = [event for event in trace if classifier.label(event) in activities]
|
|
145
|
+
if any(classifier.accepts(event) for event in events):
|
|
146
|
+
kept.append(Trace(dict(trace.attributes), events))
|
|
147
|
+
return _derived(log, kept)
|
|
148
|
+
kept = []
|
|
149
|
+
for trace, sequence in zip(log, log.sequences(classifier)):
|
|
150
|
+
contains = any(activity in activities for activity in sequence)
|
|
151
|
+
if contains == (mode == "mandatory"):
|
|
152
|
+
kept.append(trace)
|
|
153
|
+
return _derived(log, kept)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def filter_case_length(log: EventLog, minimum: int, maximum: int,
|
|
157
|
+
classifier: Classifier | None = None) -> EventLog:
|
|
158
|
+
"""Cases with between ``minimum`` and ``maximum`` events (inclusive)."""
|
|
159
|
+
kept = [trace for trace, sequence in zip(log, log.sequences(classifier))
|
|
160
|
+
if minimum <= len(sequence) <= maximum]
|
|
161
|
+
return _derived(log, kept)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def top_variants(log: EventLog, classifier: Classifier | None = None, *,
|
|
165
|
+
coverage: float | None = None, count: int | None = None
|
|
166
|
+
) -> list[tuple[tuple[str, ...], int]]:
|
|
167
|
+
"""The most frequent variants: the top ``count``, or just enough of them
|
|
168
|
+
to cover ``coverage`` percent of the cases (always at least one).
|
|
169
|
+
|
|
170
|
+
Variants with the same frequency are taken in the order they first occur
|
|
171
|
+
in the log, so the choice is deterministic. (Keeping every tied variant
|
|
172
|
+
instead would make the filter useless on the many real logs where most
|
|
173
|
+
cases have a variant of their own.)
|
|
174
|
+
"""
|
|
175
|
+
counts = Counter(log.sequences(classifier)) # insertion order = first occurrence
|
|
176
|
+
ranked = sorted(counts.items(), key=lambda item: -item[1]) # stable sort
|
|
177
|
+
if count is not None:
|
|
178
|
+
return ranked[:max(count, 0)]
|
|
179
|
+
if coverage is None:
|
|
180
|
+
return ranked
|
|
181
|
+
needed = sum(counts.values()) * min(max(coverage, 0), 100) / 100
|
|
182
|
+
chosen, covered = [], 0
|
|
183
|
+
for variant, frequency in ranked:
|
|
184
|
+
if chosen and covered >= needed:
|
|
185
|
+
break
|
|
186
|
+
chosen.append((variant, frequency))
|
|
187
|
+
covered += frequency
|
|
188
|
+
return chosen
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def filter_variants(log: EventLog, classifier: Classifier | None = None, *,
|
|
192
|
+
coverage: float | None = None, count: int | None = None) -> EventLog:
|
|
193
|
+
"""Keep the cases of the most frequent variants (see :func:`top_variants`)."""
|
|
194
|
+
keep = {variant for variant, _ in top_variants(log, classifier, coverage=coverage,
|
|
195
|
+
count=count)}
|
|
196
|
+
kept = [trace for trace, sequence in zip(log, log.sequences(classifier))
|
|
197
|
+
if sequence in keep]
|
|
198
|
+
return _derived(log, kept)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
# ---------------------------------------------------------------------------
|
|
202
|
+
# All together
|
|
203
|
+
# ---------------------------------------------------------------------------
|
|
204
|
+
def apply_filters(log: EventLog, settings: FilterSettings,
|
|
205
|
+
classifier: Classifier | None = None, name: str | None = None) -> EventLog:
|
|
206
|
+
"""Apply every active filter of ``settings``, in the order of the module
|
|
207
|
+
docstring, and name the result. The new log records what was done in
|
|
208
|
+
its ``openprocess:filter`` attribute."""
|
|
209
|
+
classifier = classifier or log.default_classifier()
|
|
210
|
+
result = log
|
|
211
|
+
if settings.time_frame is not None:
|
|
212
|
+
result = filter_time_frame(result, *settings.time_frame)
|
|
213
|
+
if settings.start_activities is not None:
|
|
214
|
+
result = filter_start_activities(result, settings.start_activities, classifier)
|
|
215
|
+
if settings.end_activities is not None:
|
|
216
|
+
result = filter_end_activities(result, settings.end_activities, classifier)
|
|
217
|
+
if settings.activities is not None:
|
|
218
|
+
result = filter_activities(result, *settings.activities, classifier=classifier)
|
|
219
|
+
if settings.case_length is not None:
|
|
220
|
+
result = filter_case_length(result, *settings.case_length, classifier=classifier)
|
|
221
|
+
if settings.variant_coverage is not None or settings.top_variants is not None:
|
|
222
|
+
result = filter_variants(result, classifier, coverage=settings.variant_coverage,
|
|
223
|
+
count=settings.top_variants)
|
|
224
|
+
if result is log:
|
|
225
|
+
result = _derived(log, list(log.traces))
|
|
226
|
+
result.attributes = dict(log.attributes)
|
|
227
|
+
result.attributes[KEY_NAME] = name or f"{log.name} (filtered)"
|
|
228
|
+
result.attributes["openprocess:filter"] = "; ".join(settings.describe()) or "none"
|
|
229
|
+
result.source_path = None
|
|
230
|
+
return result
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
__all__ = [
|
|
234
|
+
"ACTIVITY_MODES", "FilterSettings", "TIME_MODES", "apply_filters",
|
|
235
|
+
"filter_activities", "filter_case_length", "filter_end_activities",
|
|
236
|
+
"filter_start_activities", "filter_time_frame", "filter_variants", "top_variants",
|
|
237
|
+
]
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
"""Ordering relations and footprint matrices (van der Aalst, ch. 6.2 and 8.4).
|
|
2
|
+
|
|
3
|
+
From the directly-follows relation ``>_L`` four *log-based ordering
|
|
4
|
+
relations* are derived, for every pair of activities a, b:
|
|
5
|
+
|
|
6
|
+
==================== =========================================== =========
|
|
7
|
+
relation definition symbol
|
|
8
|
+
==================== =========================================== =========
|
|
9
|
+
causality ``a > b`` and **not** ``b > a`` ``→``
|
|
10
|
+
inverse causality ``b > a`` and **not** ``a > b`` ``←``
|
|
11
|
+
parallel ``a > b`` **and** ``b > a`` ``‖``
|
|
12
|
+
choice / unrelated **neither** ``a > b`` nor ``b > a`` ``#``
|
|
13
|
+
==================== =========================================== =========
|
|
14
|
+
|
|
15
|
+
Exactly one of the four holds for each ordered pair, so the relations can be
|
|
16
|
+
tabulated as the **footprint matrix**. Note the diagonal: ``a # a`` unless
|
|
17
|
+
``a`` directly follows itself (a length-one loop), in which case ``a ‖ a``.
|
|
18
|
+
|
|
19
|
+
Frequencies play no role: a pair that directly follows once counts the same
|
|
20
|
+
as one that follows a thousand times. That is exactly why the α-algorithm is
|
|
21
|
+
sensitive to noise.
|
|
22
|
+
|
|
23
|
+
Conformance via footprints (ch. 8.4)
|
|
24
|
+
------------------------------------
|
|
25
|
+
A model has a footprint too (computed from its behaviour). Comparing the log
|
|
26
|
+
footprint with the model footprint cell by cell gives a simple conformance
|
|
27
|
+
measure::
|
|
28
|
+
|
|
29
|
+
fitness_footprint = 1 - (number of differing cells) / (total cells)
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
from collections import deque
|
|
35
|
+
from dataclasses import dataclass
|
|
36
|
+
|
|
37
|
+
from .dfg import DFG, dfg_from_simple_log
|
|
38
|
+
from .log import SimpleLog
|
|
39
|
+
from .petrinet import PetriNet
|
|
40
|
+
|
|
41
|
+
CAUSAL = "→"
|
|
42
|
+
INVERSE = "←"
|
|
43
|
+
PARALLEL = "‖"
|
|
44
|
+
CHOICE = "#"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass
|
|
48
|
+
class Footprint:
|
|
49
|
+
activities: list[str]
|
|
50
|
+
directly_follows: set[tuple[str, str]]
|
|
51
|
+
start: set[str]
|
|
52
|
+
end: set[str]
|
|
53
|
+
|
|
54
|
+
def relation(self, a: str, b: str) -> str:
|
|
55
|
+
forward = (a, b) in self.directly_follows
|
|
56
|
+
backward = (b, a) in self.directly_follows
|
|
57
|
+
if forward and backward:
|
|
58
|
+
return PARALLEL
|
|
59
|
+
if forward:
|
|
60
|
+
return CAUSAL
|
|
61
|
+
if backward:
|
|
62
|
+
return INVERSE
|
|
63
|
+
return CHOICE
|
|
64
|
+
|
|
65
|
+
# Convenience predicates with the textbook names -----------------------
|
|
66
|
+
def causal(self, a: str, b: str) -> bool:
|
|
67
|
+
return self.relation(a, b) == CAUSAL
|
|
68
|
+
|
|
69
|
+
def parallel(self, a: str, b: str) -> bool:
|
|
70
|
+
return self.relation(a, b) == PARALLEL
|
|
71
|
+
|
|
72
|
+
def choice(self, a: str, b: str) -> bool:
|
|
73
|
+
return self.relation(a, b) == CHOICE
|
|
74
|
+
|
|
75
|
+
def matrix(self) -> list[list[str]]:
|
|
76
|
+
return [[self.relation(a, b) for b in self.activities] for a in self.activities]
|
|
77
|
+
|
|
78
|
+
def as_text(self) -> str:
|
|
79
|
+
width = max([len(a) for a in self.activities] + [1])
|
|
80
|
+
header = " " * (width + 1) + " ".join(a.rjust(width) for a in self.activities)
|
|
81
|
+
rows = [header]
|
|
82
|
+
for a, row in zip(self.activities, self.matrix()):
|
|
83
|
+
rows.append(a.rjust(width) + " " + " ".join(cell.rjust(width) for cell in row))
|
|
84
|
+
return "\n".join(rows)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def footprint_from_dfg(dfg: DFG) -> Footprint:
|
|
88
|
+
return Footprint(sorted(dfg.activities), set(dfg.edges), set(dfg.start), set(dfg.end))
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def footprint_of_log(log: SimpleLog) -> Footprint:
|
|
92
|
+
return footprint_from_dfg(dfg_from_simple_log(log))
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def footprint_of_net(net: PetriNet, max_states: int = 50_000) -> Footprint:
|
|
96
|
+
"""The footprint of a model's behaviour.
|
|
97
|
+
|
|
98
|
+
``a > b`` holds in the model iff some reachable marking allows ``a`` and
|
|
99
|
+
then ``b`` with only silent transitions in between. Computed from the
|
|
100
|
+
reachability graph, so the net must be bounded.
|
|
101
|
+
"""
|
|
102
|
+
from .analysis import reachability_graph
|
|
103
|
+
|
|
104
|
+
graph = reachability_graph(net, max_states=max_states)
|
|
105
|
+
if graph.has_omega or graph.truncated:
|
|
106
|
+
raise ValueError("The model's state space is unbounded or too large to "
|
|
107
|
+
"compute its footprint.")
|
|
108
|
+
|
|
109
|
+
def visible_next(state: int) -> set[str]:
|
|
110
|
+
"""Labels that can occur next from ``state``, skipping silent steps."""
|
|
111
|
+
seen, result = {state}, set()
|
|
112
|
+
queue = deque([state])
|
|
113
|
+
while queue:
|
|
114
|
+
current = queue.popleft()
|
|
115
|
+
for transition, target in graph.successors(current):
|
|
116
|
+
label = net.transitions[transition].label
|
|
117
|
+
if label is None:
|
|
118
|
+
if target not in seen:
|
|
119
|
+
seen.add(target)
|
|
120
|
+
queue.append(target)
|
|
121
|
+
else:
|
|
122
|
+
result.add(label)
|
|
123
|
+
return result
|
|
124
|
+
|
|
125
|
+
follows: set[tuple[str, str]] = set()
|
|
126
|
+
for source, transition, target in graph.edges:
|
|
127
|
+
label = net.transitions[transition].label
|
|
128
|
+
if label is None:
|
|
129
|
+
continue
|
|
130
|
+
for successor in visible_next(target):
|
|
131
|
+
follows.add((label, successor))
|
|
132
|
+
start = visible_next(0)
|
|
133
|
+
final_states = [i for i, m in enumerate(graph.states) if m == net.final_marking]
|
|
134
|
+
end: set[str] = set()
|
|
135
|
+
for source, transition, target in graph.edges:
|
|
136
|
+
label = net.transitions[transition].label
|
|
137
|
+
if label is not None and _silently_reaches(graph, net, target, set(final_states)):
|
|
138
|
+
end.add(label)
|
|
139
|
+
return Footprint(sorted(net.labels()), follows, start, end)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _silently_reaches(graph, net, state: int, targets: set[int]) -> bool:
|
|
143
|
+
seen = {state}
|
|
144
|
+
queue = deque([state])
|
|
145
|
+
while queue:
|
|
146
|
+
current = queue.popleft()
|
|
147
|
+
if current in targets:
|
|
148
|
+
return True
|
|
149
|
+
for transition, target in graph.successors(current):
|
|
150
|
+
if net.transitions[transition].label is None and target not in seen:
|
|
151
|
+
seen.add(target)
|
|
152
|
+
queue.append(target)
|
|
153
|
+
return False
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
@dataclass
|
|
157
|
+
class FootprintComparison:
|
|
158
|
+
activities: list[str]
|
|
159
|
+
log_matrix: list[list[str]]
|
|
160
|
+
model_matrix: list[list[str]]
|
|
161
|
+
|
|
162
|
+
@property
|
|
163
|
+
def differences(self) -> list[tuple[str, str, str, str]]:
|
|
164
|
+
"""(a, b, log relation, model relation) for every differing cell."""
|
|
165
|
+
result = []
|
|
166
|
+
for i, a in enumerate(self.activities):
|
|
167
|
+
for j, b in enumerate(self.activities):
|
|
168
|
+
if self.log_matrix[i][j] != self.model_matrix[i][j]:
|
|
169
|
+
result.append((a, b, self.log_matrix[i][j], self.model_matrix[i][j]))
|
|
170
|
+
return result
|
|
171
|
+
|
|
172
|
+
@property
|
|
173
|
+
def fitness(self) -> float:
|
|
174
|
+
cells = len(self.activities) ** 2
|
|
175
|
+
return 1.0 if cells == 0 else 1 - len(self.differences) / cells
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def compare_footprints(log_fp: Footprint, model_fp: Footprint) -> FootprintComparison:
|
|
179
|
+
"""Cell-by-cell comparison over the union of both alphabets."""
|
|
180
|
+
activities = sorted(set(log_fp.activities) | set(model_fp.activities))
|
|
181
|
+
log_matrix = [[log_fp.relation(a, b) for b in activities] for a in activities]
|
|
182
|
+
model_matrix = [[model_fp.relation(a, b) for b in activities] for a in activities]
|
|
183
|
+
return FootprintComparison(activities, log_matrix, model_matrix)
|
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
"""The incidence matrix, place invariants and transition invariants.
|
|
2
|
+
|
|
3
|
+
Linear algebra instead of state spaces
|
|
4
|
+
--------------------------------------
|
|
5
|
+
Everything in :mod:`.analysis` explores markings one by one. Invariants
|
|
6
|
+
answer questions from the *structure* alone, with a little linear algebra,
|
|
7
|
+
so they work even when the state space is huge or infinite.
|
|
8
|
+
|
|
9
|
+
**Incidence matrix.** ``C`` has a row per place and a column per transition;
|
|
10
|
+
``C(p, t) = W(t, p) − W(p, t)`` is what firing ``t`` does to ``p`` (the tokens
|
|
11
|
+
it puts in minus the tokens it takes out). Firing a sequence ``σ`` whose
|
|
12
|
+
Parikh vector ``σ⃗`` counts how often each transition fires gives the
|
|
13
|
+
**marking equation**::
|
|
14
|
+
|
|
15
|
+
M --σ--> M' implies M' = M + C · σ⃗
|
|
16
|
+
|
|
17
|
+
**Place invariant (P-invariant).** A weighting ``y`` of the places with
|
|
18
|
+
``yᵀ · C = 0``. Multiplying the marking equation by ``yᵀ`` gives
|
|
19
|
+
``y · M' = y · M``: the weighted token count is the same in *every* reachable
|
|
20
|
+
marking. In a WF-net, ``i + c1 + c2 + o`` being invariant says "exactly one
|
|
21
|
+
token moves through these places".
|
|
22
|
+
|
|
23
|
+
**Transition invariant (T-invariant).** A firing count ``x`` with
|
|
24
|
+
``C · x = 0``: firing every transition as often as ``x`` says (in any
|
|
25
|
+
enabled order) leads back to the marking you started from. A cycle of the
|
|
26
|
+
net's behaviour.
|
|
27
|
+
|
|
28
|
+
We compute the **minimal semi-positive** invariants (no negative weights, not
|
|
29
|
+
all zero, and no other invariant uses a strict subset of their places or
|
|
30
|
+
transitions). Every semi-positive invariant is a non-negative combination
|
|
31
|
+
of these, so they are *the* invariants to show. The method is Farkas'
|
|
32
|
+
algorithm (Martínez & Silva, 1982): start from ``[C | I]`` and eliminate one
|
|
33
|
+
column of ``C`` at a time by adding pairs of rows with opposite signs.
|
|
34
|
+
|
|
35
|
+
What they tell you
|
|
36
|
+
------------------
|
|
37
|
+
* Every place in some semi-positive P-invariant (**covered by P-invariants**)
|
|
38
|
+
⇒ the net is *structurally bounded*: bounded from any initial marking.
|
|
39
|
+
* A net that is live and bounded is covered by T-invariants, so a transition
|
|
40
|
+
in no T-invariant means "not both live and bounded". For a WF-net, apply
|
|
41
|
+
that to the short-circuited net N̄: by the soundness theorem, a sound
|
|
42
|
+
WF-net's N̄ is covered by T-invariants.
|
|
43
|
+
|
|
44
|
+
The number of minimal invariants can grow exponentially with the size of the
|
|
45
|
+
net. :func:`invariants` stops at ``limit`` intermediate rows and says so.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
from __future__ import annotations
|
|
49
|
+
|
|
50
|
+
from dataclasses import dataclass, field
|
|
51
|
+
from math import gcd
|
|
52
|
+
|
|
53
|
+
from .petrinet import Marking, PetriNet
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class TooManyInvariants(Exception):
|
|
57
|
+
"""Farkas' algorithm needed more than the allowed number of rows."""
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def incidence_matrix(net: PetriNet) -> tuple[list[str], list[str], list[list[int]]]:
|
|
61
|
+
"""``(places, transitions, C)`` with ``C[i][j] = W(t_j, p_i) − W(p_i, t_j)``."""
|
|
62
|
+
places = list(net.places)
|
|
63
|
+
transitions = list(net.transitions)
|
|
64
|
+
row = {p: i for i, p in enumerate(places)}
|
|
65
|
+
column = {t: j for j, t in enumerate(transitions)}
|
|
66
|
+
matrix = [[0] * len(transitions) for _ in places]
|
|
67
|
+
for arc in net.arcs:
|
|
68
|
+
if arc.source in row and arc.target in column: # p -> t: consumed
|
|
69
|
+
matrix[row[arc.source]][column[arc.target]] -= arc.weight
|
|
70
|
+
elif arc.source in column and arc.target in row: # t -> p: produced
|
|
71
|
+
matrix[row[arc.target]][column[arc.source]] += arc.weight
|
|
72
|
+
return places, transitions, matrix
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _normalise(vector: list[int]) -> list[int]:
|
|
76
|
+
divisor = 0
|
|
77
|
+
for value in vector:
|
|
78
|
+
divisor = gcd(divisor, value)
|
|
79
|
+
return [value // divisor for value in vector] if divisor > 1 else vector
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _support(vector: list[int]) -> frozenset[int]:
|
|
83
|
+
return frozenset(i for i, value in enumerate(vector) if value)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _minimal(rows: list[tuple[list[int], list[int]]]) -> list[tuple[list[int], list[int]]]:
|
|
87
|
+
"""Drop duplicate rows and rows whose support strictly contains another's."""
|
|
88
|
+
unique: dict[tuple[int, ...], tuple[list[int], list[int]]] = {}
|
|
89
|
+
for rest, generator in rows:
|
|
90
|
+
unique.setdefault(tuple(rest) + tuple(generator), (rest, generator))
|
|
91
|
+
rows = list(unique.values())
|
|
92
|
+
supports = [_support(generator) for _, generator in rows]
|
|
93
|
+
return [row for row, support in zip(rows, supports)
|
|
94
|
+
if not any(other < support for other in supports)]
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def semi_positive_invariants(matrix: list[list[int]], limit: int = 5_000) -> list[list[int]]:
|
|
98
|
+
"""Minimal semi-positive solutions ``y ≥ 0, y ≠ 0`` of ``yᵀ · matrix = 0``.
|
|
99
|
+
|
|
100
|
+
``matrix`` has one row per variable. Farkas' algorithm: rows are pairs
|
|
101
|
+
(what is left of the matrix row, the combination of variables it stands
|
|
102
|
+
for). Column by column, rows with opposite signs in that column are
|
|
103
|
+
combined so that it cancels, and rows that do not cancel are dropped.
|
|
104
|
+
"""
|
|
105
|
+
count = len(matrix)
|
|
106
|
+
columns = len(matrix[0]) if count else 0
|
|
107
|
+
rows = [(list(matrix[i]), [1 if j == i else 0 for j in range(count)]) for i in range(count)]
|
|
108
|
+
for column in range(columns):
|
|
109
|
+
keep = [row for row in rows if row[0][column] == 0]
|
|
110
|
+
positive = [row for row in rows if row[0][column] > 0]
|
|
111
|
+
negative = [row for row in rows if row[0][column] < 0]
|
|
112
|
+
for up_rest, up_generator in positive:
|
|
113
|
+
for down_rest, down_generator in negative:
|
|
114
|
+
# a·up + b·down cancels the column, with a, b > 0.
|
|
115
|
+
a, b = -down_rest[column], up_rest[column]
|
|
116
|
+
combined = _normalise([a * x + b * y for x, y in zip(
|
|
117
|
+
up_rest + up_generator, down_rest + down_generator)])
|
|
118
|
+
keep.append((combined[:columns], combined[columns:]))
|
|
119
|
+
rows = _minimal(keep)
|
|
120
|
+
if len(rows) > limit:
|
|
121
|
+
raise TooManyInvariants(f"more than {limit} candidate invariants")
|
|
122
|
+
invariants = [_normalise(generator) for _, generator in _minimal(rows) if any(generator)]
|
|
123
|
+
return sorted(invariants, key=lambda v: (sum(1 for x in v if x), [-x for x in v]))
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
@dataclass
|
|
127
|
+
class Invariants:
|
|
128
|
+
"""The incidence matrix and the minimal semi-positive invariants of a net."""
|
|
129
|
+
|
|
130
|
+
net: PetriNet
|
|
131
|
+
places: list[str]
|
|
132
|
+
transitions: list[str]
|
|
133
|
+
#: ``incidence[i][j]``: the effect of transitions[j] on places[i].
|
|
134
|
+
incidence: list[list[int]]
|
|
135
|
+
#: Each P-invariant as ``{place id: weight}`` (weights > 0 only).
|
|
136
|
+
p_invariants: list[dict[str, int]] = field(default_factory=list)
|
|
137
|
+
#: Each T-invariant as ``{transition id: count}`` (counts > 0 only).
|
|
138
|
+
t_invariants: list[dict[str, int]] = field(default_factory=list)
|
|
139
|
+
#: Set when there were too many to compute: the lists are then empty.
|
|
140
|
+
p_truncated: bool = False
|
|
141
|
+
t_truncated: bool = False
|
|
142
|
+
|
|
143
|
+
def uncovered_places(self) -> list[str]:
|
|
144
|
+
covered = {p for invariant in self.p_invariants for p in invariant}
|
|
145
|
+
return [p for p in self.places if p not in covered]
|
|
146
|
+
|
|
147
|
+
def uncovered_transitions(self) -> list[str]:
|
|
148
|
+
covered = {t for invariant in self.t_invariants for t in invariant}
|
|
149
|
+
return [t for t in self.transitions if t not in covered]
|
|
150
|
+
|
|
151
|
+
@property
|
|
152
|
+
def covered_by_p_invariants(self) -> bool | None:
|
|
153
|
+
"""Every place in some P-invariant (⇒ structurally bounded)."""
|
|
154
|
+
return None if self.p_truncated else not self.uncovered_places()
|
|
155
|
+
|
|
156
|
+
@property
|
|
157
|
+
def covered_by_t_invariants(self) -> bool | None:
|
|
158
|
+
return None if self.t_truncated else not self.uncovered_transitions()
|
|
159
|
+
|
|
160
|
+
def token_sum(self, invariant: dict[str, int], marking: Marking | None = None) -> int:
|
|
161
|
+
"""``y · M``: the weighted token count the invariant keeps constant."""
|
|
162
|
+
marking = self.net.initial_marking if marking is None else marking
|
|
163
|
+
return sum(weight * marking[place] for place, weight in invariant.items())
|
|
164
|
+
|
|
165
|
+
def describe(self, invariant: dict[str, int], with_value: bool = False) -> str:
|
|
166
|
+
"""``2·p1 + p2`` (P-invariant) or ``a + b + t*`` (T-invariant), by name."""
|
|
167
|
+
terms = [(f"{weight}·" if weight != 1 else "") + self.net.node_name(node)
|
|
168
|
+
for node, weight in invariant.items()]
|
|
169
|
+
text = " + ".join(terms)
|
|
170
|
+
if with_value:
|
|
171
|
+
text += f" = {self.token_sum(invariant)}"
|
|
172
|
+
return text
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def invariants(net: PetriNet, limit: int = 5_000) -> Invariants:
|
|
176
|
+
"""Incidence matrix plus minimal semi-positive P- and T-invariants."""
|
|
177
|
+
places, transitions, matrix = incidence_matrix(net)
|
|
178
|
+
result = Invariants(net, places, transitions, matrix)
|
|
179
|
+
try:
|
|
180
|
+
for vector in semi_positive_invariants(matrix, limit):
|
|
181
|
+
result.p_invariants.append({p: w for p, w in zip(places, vector) if w})
|
|
182
|
+
except TooManyInvariants:
|
|
183
|
+
result.p_truncated = True
|
|
184
|
+
transposed = [list(column) for column in zip(*matrix)] if places else \
|
|
185
|
+
[[] for _ in transitions]
|
|
186
|
+
try:
|
|
187
|
+
for vector in semi_positive_invariants(transposed, limit):
|
|
188
|
+
result.t_invariants.append({t: w for t, w in zip(transitions, vector) if w})
|
|
189
|
+
except TooManyInvariants:
|
|
190
|
+
result.t_truncated = True
|
|
191
|
+
return result
|