openprocess 0.7.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (173) hide show
  1. cpnpy/__init__.py +46 -0
  2. openprocess/__init__.py +57 -0
  3. openprocess/analysis/__init__.py +0 -0
  4. openprocess/analysis/state_space.py +521 -0
  5. openprocess/analysis/state_space_process.py +251 -0
  6. openprocess/cli.py +742 -0
  7. openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/answer.pnml +31 -0
  8. openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/question.md +38 -0
  9. openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/answer.pnml +27 -0
  10. openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/net.pnml +29 -0
  11. openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/question.md +65 -0
  12. openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/log.txt +1 -0
  13. openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/question.md +67 -0
  14. openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/question.md +75 -0
  15. openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/ts.txt +5 -0
  16. openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/net.pnml +27 -0
  17. openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/question.md +65 -0
  18. openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/log.txt +1 -0
  19. openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/question.md +62 -0
  20. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/log.txt +1 -0
  21. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m1.pnml +36 -0
  22. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m2.pnml +28 -0
  23. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m3.pnml +30 -0
  24. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/net.pnml +36 -0
  25. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/question.md +69 -0
  26. openprocess/exercises/pack.md +14 -0
  27. openprocess/flow/__init__.py +50 -0
  28. openprocess/flow/box.py +466 -0
  29. openprocess/flow/boxes/__init__.py +7 -0
  30. openprocess/flow/boxes/check.py +119 -0
  31. openprocess/flow/boxes/compare.py +16 -0
  32. openprocess/flow/boxes/cpn.py +53 -0
  33. openprocess/flow/boxes/discover.py +124 -0
  34. openprocess/flow/boxes/filter.py +80 -0
  35. openprocess/flow/boxes/input.py +124 -0
  36. openprocess/flow/boxes/output.py +52 -0
  37. openprocess/flow/boxes/predict.py +186 -0
  38. openprocess/flow/boxes/science.py +159 -0
  39. openprocess/flow/boxes/sweeps.py +18 -0
  40. openprocess/flow/convert.py +187 -0
  41. openprocess/flow/datasets.py +198 -0
  42. openprocess/flow/explain.py +115 -0
  43. openprocess/flow/library.py +222 -0
  44. openprocess/flow/record.py +385 -0
  45. openprocess/flow/runner.py +357 -0
  46. openprocess/flow/sweep.py +92 -0
  47. openprocess/flow/types.py +290 -0
  48. openprocess/flow/workflow.py +628 -0
  49. openprocess/gui/__init__.py +0 -0
  50. openprocess/gui/app.py +90 -0
  51. openprocess/gui/arc_editing.py +295 -0
  52. openprocess/gui/canvas.py +1414 -0
  53. openprocess/gui/flow/__init__.py +8 -0
  54. openprocess/gui/flow/canvas.py +854 -0
  55. openprocess/gui/flow/page.py +972 -0
  56. openprocess/gui/flow/templates.py +131 -0
  57. openprocess/gui/flow/viewers.py +665 -0
  58. openprocess/gui/items.py +1275 -0
  59. openprocess/gui/learn/answer_boxes.py +978 -0
  60. openprocess/gui/learn/concealment.py +91 -0
  61. openprocess/gui/learn/mode.py +1181 -0
  62. openprocess/gui/panning.py +241 -0
  63. openprocess/gui/resources/openprocess-icon.png +0 -0
  64. openprocess/gui/studio/__init__.py +1 -0
  65. openprocess/gui/studio/__main__.py +3 -0
  66. openprocess/gui/studio/app.py +4031 -0
  67. openprocess/gui/studio/charts.py +115 -0
  68. openprocess/gui/studio/compare_page.py +487 -0
  69. openprocess/gui/studio/cpn_page.py +1858 -0
  70. openprocess/gui/studio/definition_view.py +284 -0
  71. openprocess/gui/studio/derivation_view.py +421 -0
  72. openprocess/gui/studio/documents.py +152 -0
  73. openprocess/gui/studio/dotted_chart.py +1401 -0
  74. openprocess/gui/studio/file_dialogs.py +143 -0
  75. openprocess/gui/studio/filter_dialog.py +247 -0
  76. openprocess/gui/studio/graph_builders.py +176 -0
  77. openprocess/gui/studio/graph_view.py +682 -0
  78. openprocess/gui/studio/instances.py +413 -0
  79. openprocess/gui/studio/log_editor.py +675 -0
  80. openprocess/gui/studio/log_page.py +800 -0
  81. openprocess/gui/studio/markdown_view.py +127 -0
  82. openprocess/gui/studio/mathtext.py +260 -0
  83. openprocess/gui/studio/ml_highlighter.py +75 -0
  84. openprocess/gui/studio/model_page.py +760 -0
  85. openprocess/gui/studio/net_comparison.py +124 -0
  86. openprocess/gui/studio/notes_overlay.py +275 -0
  87. openprocess/gui/studio/petri_page.py +844 -0
  88. openprocess/gui/studio/regions_view.py +502 -0
  89. openprocess/gui/studio/sidebar.py +149 -0
  90. openprocess/gui/studio/style.py +503 -0
  91. openprocess/gui/studio/tool_icons.py +134 -0
  92. openprocess/gui/studio/updates.py +439 -0
  93. openprocess/gui/studio/widgets.py +899 -0
  94. openprocess/gui/studio/workers.py +60 -0
  95. openprocess/gui/studio/workspace.py +447 -0
  96. openprocess/gui/theme.py +394 -0
  97. openprocess/gui/tidy.py +86 -0
  98. openprocess/io/__init__.py +0 -0
  99. openprocess/io/cpn_reader.py +389 -0
  100. openprocess/io/cpn_writer.py +357 -0
  101. openprocess/learn/__init__.py +23 -0
  102. openprocess/learn/answers.py +188 -0
  103. openprocess/learn/checks.py +953 -0
  104. openprocess/learn/computed.py +1180 -0
  105. openprocess/learn/context.py +145 -0
  106. openprocess/learn/exam.py +169 -0
  107. openprocess/learn/exercise-packs.md +325 -0
  108. openprocess/learn/importer.py +216 -0
  109. openprocess/learn/notation.py +474 -0
  110. openprocess/learn/pack.py +511 -0
  111. openprocess/learn/sheet.py +296 -0
  112. openprocess/mining/__init__.py +73 -0
  113. openprocess/mining/analysis.py +689 -0
  114. openprocess/mining/columns.py +282 -0
  115. openprocess/mining/compare_nets.py +246 -0
  116. openprocess/mining/conformance/__init__.py +0 -0
  117. openprocess/mining/conformance/alignments.py +263 -0
  118. openprocess/mining/conformance/quality.py +145 -0
  119. openprocess/mining/conformance/token_replay.py +252 -0
  120. openprocess/mining/csv_import.py +222 -0
  121. openprocess/mining/definitions.py +584 -0
  122. openprocess/mining/dfg.py +187 -0
  123. openprocess/mining/discovery/__init__.py +0 -0
  124. openprocess/mining/discovery/alpha.py +168 -0
  125. openprocess/mining/discovery/heuristics.py +332 -0
  126. openprocess/mining/discovery/inductive.py +477 -0
  127. openprocess/mining/discovery/state_regions.py +62 -0
  128. openprocess/mining/filtering.py +237 -0
  129. openprocess/mining/footprint.py +183 -0
  130. openprocess/mining/invariants.py +191 -0
  131. openprocess/mining/layout.py +279 -0
  132. openprocess/mining/log.py +364 -0
  133. openprocess/mining/petrinet.py +354 -0
  134. openprocess/mining/playout.py +75 -0
  135. openprocess/mining/pm4py_bridge.py +82 -0
  136. openprocess/mining/pnml.py +223 -0
  137. openprocess/mining/processtree.py +216 -0
  138. openprocess/mining/regions.py +476 -0
  139. openprocess/mining/stats.py +160 -0
  140. openprocess/mining/structure.py +374 -0
  141. openprocess/mining/transition_system.py +409 -0
  142. openprocess/mining/xes.py +399 -0
  143. openprocess/ml/__init__.py +0 -0
  144. openprocess/ml/ast_nodes.py +332 -0
  145. openprocess/ml/builtins.py +364 -0
  146. openprocess/ml/colorsets.py +522 -0
  147. openprocess/ml/errors.py +60 -0
  148. openprocess/ml/evaluator.py +754 -0
  149. openprocess/ml/lexer.py +277 -0
  150. openprocess/ml/multiset.py +417 -0
  151. openprocess/ml/parser.py +737 -0
  152. openprocess/ml/values.py +319 -0
  153. openprocess/model/__init__.py +0 -0
  154. openprocess/model/declarations.py +617 -0
  155. openprocess/model/examples.py +98 -0
  156. openprocess/model/net.py +701 -0
  157. openprocess/model/plain.py +192 -0
  158. openprocess/references.py +280 -0
  159. openprocess/sim/__init__.py +0 -0
  160. openprocess/sim/binding.py +620 -0
  161. openprocess/sim/export.py +66 -0
  162. openprocess/sim/simulator.py +315 -0
  163. openprocess/teaching/__init__.py +4 -0
  164. openprocess/teaching/answers.py +4 -0
  165. openprocess/teaching/checks.py +5 -0
  166. openprocess/teaching/pack.py +4 -0
  167. openprocess/teaching/sheet.py +4 -0
  168. openprocess-0.7.0.dist-info/METADATA +927 -0
  169. openprocess-0.7.0.dist-info/RECORD +173 -0
  170. openprocess-0.7.0.dist-info/WHEEL +5 -0
  171. openprocess-0.7.0.dist-info/entry_points.txt +6 -0
  172. openprocess-0.7.0.dist-info/licenses/LICENSE +21 -0
  173. openprocess-0.7.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,237 @@
1
+ """Filtering event logs: keep the part of the log you want to look at.
2
+
3
+ Real logs are messy: rare variants, noise activities, cases cut off by the
4
+ extraction window. Filtering is therefore the usual first step before
5
+ discovery, as in ProM ("Filter Log using Simple Heuristics") and Disco. Every
6
+ filter here returns a **new** log and never changes the original, so you can
7
+ always compare the filtered log with the one it came from.
8
+
9
+ The filters, in the order :func:`apply_filters` applies them:
10
+
11
+ 1. **Time frame** -- keep cases *contained in*, *started in*, or
12
+ *intersecting* a period. Cases without timestamps cannot be placed in
13
+ time and are dropped.
14
+ 2. **Start and end activities** -- keep cases that start (end) with one of
15
+ the chosen activities, e.g. to drop cases cut off by the extraction.
16
+ 3. **Activities** -- three modes, as in Disco:
17
+
18
+ * *keep events*: remove every event of the other activities (a projection;
19
+ cases keep their other events);
20
+ * *mandatory*: keep cases that contain at least one of the activities;
21
+ * *forbidden*: keep cases that contain none of them.
22
+
23
+ 4. **Case length** -- keep cases with between ``min`` and ``max`` events.
24
+ 5. **Variants** -- keep the most frequent variants, either the top ``k`` or
25
+ as many as needed to cover a percentage of the cases.
26
+
27
+ Activities, variants and lengths are seen through the log's *classifier*
28
+ (see :mod:`.log`), so they match what the rest of the app shows.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ from collections import Counter
34
+ from dataclasses import dataclass, field
35
+ from datetime import datetime
36
+
37
+ from .log import KEY_NAME, Classifier, EventLog, Trace
38
+
39
+ TIME_MODES = ("contained", "started", "intersecting")
40
+ ACTIVITY_MODES = ("keep events", "mandatory", "forbidden")
41
+
42
+
43
+ @dataclass
44
+ class FilterSettings:
45
+ """Which filters to apply. ``None`` (or empty) means "do not filter on this"."""
46
+
47
+ #: ``(from, to, mode)`` with ``mode`` in :data:`TIME_MODES`.
48
+ time_frame: tuple[datetime, datetime, str] | None = None
49
+ start_activities: set[str] | None = None
50
+ end_activities: set[str] | None = None
51
+ #: ``(activities, mode)`` with ``mode`` in :data:`ACTIVITY_MODES`.
52
+ activities: tuple[set[str], str] | None = None
53
+ #: ``(minimum, maximum)`` number of events (after the activity filter).
54
+ case_length: tuple[int, int] | None = None
55
+ #: Keep the variants covering this percentage of the cases (0-100].
56
+ variant_coverage: float | None = None
57
+ #: Or: keep the ``k`` most frequent variants.
58
+ top_variants: int | None = None
59
+
60
+ def describe(self) -> list[str]:
61
+ """One line per active filter, for the record kept in the new log."""
62
+ lines: list[str] = []
63
+ if self.time_frame is not None:
64
+ start, end, mode = self.time_frame
65
+ lines.append(f"cases {mode} {start:%Y-%m-%d %H:%M} – {end:%Y-%m-%d %H:%M}")
66
+ if self.start_activities is not None:
67
+ lines.append("start with " + ", ".join(sorted(self.start_activities)))
68
+ if self.end_activities is not None:
69
+ lines.append("end with " + ", ".join(sorted(self.end_activities)))
70
+ if self.activities is not None:
71
+ names, mode = self.activities
72
+ lines.append(f"activities ({mode}): " + ", ".join(sorted(names)))
73
+ if self.case_length is not None:
74
+ lines.append(f"{self.case_length[0]} to {self.case_length[1]} events per case")
75
+ if self.variant_coverage is not None:
76
+ lines.append(f"most frequent variants covering {self.variant_coverage:g}% of cases")
77
+ if self.top_variants is not None:
78
+ lines.append(f"the {self.top_variants} most frequent variants")
79
+ return lines
80
+
81
+
82
+ # ---------------------------------------------------------------------------
83
+ # Single filters (each returns a new log)
84
+ # ---------------------------------------------------------------------------
85
+ def _derived(log: EventLog, traces: list[Trace]) -> EventLog:
86
+ result = log.filtered([])
87
+ result.traces = traces
88
+ return result
89
+
90
+
91
+ def _span(trace: Trace) -> tuple[datetime, datetime] | None:
92
+ stamps = [event.timestamp for event in trace if event.timestamp is not None]
93
+ return (min(stamps), max(stamps)) if stamps else None
94
+
95
+
96
+ def filter_time_frame(log: EventLog, start: datetime, end: datetime,
97
+ mode: str = "contained") -> EventLog:
98
+ """Cases *contained* in, *started* in, or *intersecting* ``[start, end]``."""
99
+ if mode not in TIME_MODES:
100
+ raise ValueError(f"mode must be one of {TIME_MODES}")
101
+ kept = []
102
+ for trace in log:
103
+ span = _span(trace)
104
+ if span is None:
105
+ continue
106
+ first, last = span
107
+ if (mode == "contained" and start <= first and last <= end) or \
108
+ (mode == "started" and start <= first <= end) or \
109
+ (mode == "intersecting" and first <= end and last >= start):
110
+ kept.append(trace)
111
+ return _derived(log, kept)
112
+
113
+
114
+ def filter_start_activities(log: EventLog, allowed: set[str],
115
+ classifier: Classifier | None = None) -> EventLog:
116
+ """Cases whose first activity is one of ``allowed``."""
117
+ kept = [trace for trace, sequence in zip(log, log.sequences(classifier))
118
+ if sequence and sequence[0] in allowed]
119
+ return _derived(log, kept)
120
+
121
+
122
+ def filter_end_activities(log: EventLog, allowed: set[str],
123
+ classifier: Classifier | None = None) -> EventLog:
124
+ """Cases whose last activity is one of ``allowed``."""
125
+ kept = [trace for trace, sequence in zip(log, log.sequences(classifier))
126
+ if sequence and sequence[-1] in allowed]
127
+ return _derived(log, kept)
128
+
129
+
130
+ def filter_activities(log: EventLog, activities: set[str], mode: str = "keep events",
131
+ classifier: Classifier | None = None) -> EventLog:
132
+ """Project onto ``activities``, or keep cases that must (not) contain them.
133
+
134
+ *keep events* removes the events of every other activity -- including
135
+ events the classifier skips (a *start* event of a kept activity stays, so
136
+ the dotted chart still shows it). Cases left without events are dropped.
137
+ """
138
+ if mode not in ACTIVITY_MODES:
139
+ raise ValueError(f"mode must be one of {ACTIVITY_MODES}")
140
+ classifier = classifier or log.default_classifier()
141
+ if mode == "keep events":
142
+ kept = []
143
+ for trace in log:
144
+ events = [event for event in trace if classifier.label(event) in activities]
145
+ if any(classifier.accepts(event) for event in events):
146
+ kept.append(Trace(dict(trace.attributes), events))
147
+ return _derived(log, kept)
148
+ kept = []
149
+ for trace, sequence in zip(log, log.sequences(classifier)):
150
+ contains = any(activity in activities for activity in sequence)
151
+ if contains == (mode == "mandatory"):
152
+ kept.append(trace)
153
+ return _derived(log, kept)
154
+
155
+
156
+ def filter_case_length(log: EventLog, minimum: int, maximum: int,
157
+ classifier: Classifier | None = None) -> EventLog:
158
+ """Cases with between ``minimum`` and ``maximum`` events (inclusive)."""
159
+ kept = [trace for trace, sequence in zip(log, log.sequences(classifier))
160
+ if minimum <= len(sequence) <= maximum]
161
+ return _derived(log, kept)
162
+
163
+
164
+ def top_variants(log: EventLog, classifier: Classifier | None = None, *,
165
+ coverage: float | None = None, count: int | None = None
166
+ ) -> list[tuple[tuple[str, ...], int]]:
167
+ """The most frequent variants: the top ``count``, or just enough of them
168
+ to cover ``coverage`` percent of the cases (always at least one).
169
+
170
+ Variants with the same frequency are taken in the order they first occur
171
+ in the log, so the choice is deterministic. (Keeping every tied variant
172
+ instead would make the filter useless on the many real logs where most
173
+ cases have a variant of their own.)
174
+ """
175
+ counts = Counter(log.sequences(classifier)) # insertion order = first occurrence
176
+ ranked = sorted(counts.items(), key=lambda item: -item[1]) # stable sort
177
+ if count is not None:
178
+ return ranked[:max(count, 0)]
179
+ if coverage is None:
180
+ return ranked
181
+ needed = sum(counts.values()) * min(max(coverage, 0), 100) / 100
182
+ chosen, covered = [], 0
183
+ for variant, frequency in ranked:
184
+ if chosen and covered >= needed:
185
+ break
186
+ chosen.append((variant, frequency))
187
+ covered += frequency
188
+ return chosen
189
+
190
+
191
+ def filter_variants(log: EventLog, classifier: Classifier | None = None, *,
192
+ coverage: float | None = None, count: int | None = None) -> EventLog:
193
+ """Keep the cases of the most frequent variants (see :func:`top_variants`)."""
194
+ keep = {variant for variant, _ in top_variants(log, classifier, coverage=coverage,
195
+ count=count)}
196
+ kept = [trace for trace, sequence in zip(log, log.sequences(classifier))
197
+ if sequence in keep]
198
+ return _derived(log, kept)
199
+
200
+
201
+ # ---------------------------------------------------------------------------
202
+ # All together
203
+ # ---------------------------------------------------------------------------
204
+ def apply_filters(log: EventLog, settings: FilterSettings,
205
+ classifier: Classifier | None = None, name: str | None = None) -> EventLog:
206
+ """Apply every active filter of ``settings``, in the order of the module
207
+ docstring, and name the result. The new log records what was done in
208
+ its ``openprocess:filter`` attribute."""
209
+ classifier = classifier or log.default_classifier()
210
+ result = log
211
+ if settings.time_frame is not None:
212
+ result = filter_time_frame(result, *settings.time_frame)
213
+ if settings.start_activities is not None:
214
+ result = filter_start_activities(result, settings.start_activities, classifier)
215
+ if settings.end_activities is not None:
216
+ result = filter_end_activities(result, settings.end_activities, classifier)
217
+ if settings.activities is not None:
218
+ result = filter_activities(result, *settings.activities, classifier=classifier)
219
+ if settings.case_length is not None:
220
+ result = filter_case_length(result, *settings.case_length, classifier=classifier)
221
+ if settings.variant_coverage is not None or settings.top_variants is not None:
222
+ result = filter_variants(result, classifier, coverage=settings.variant_coverage,
223
+ count=settings.top_variants)
224
+ if result is log:
225
+ result = _derived(log, list(log.traces))
226
+ result.attributes = dict(log.attributes)
227
+ result.attributes[KEY_NAME] = name or f"{log.name} (filtered)"
228
+ result.attributes["openprocess:filter"] = "; ".join(settings.describe()) or "none"
229
+ result.source_path = None
230
+ return result
231
+
232
+
233
+ __all__ = [
234
+ "ACTIVITY_MODES", "FilterSettings", "TIME_MODES", "apply_filters",
235
+ "filter_activities", "filter_case_length", "filter_end_activities",
236
+ "filter_start_activities", "filter_time_frame", "filter_variants", "top_variants",
237
+ ]
@@ -0,0 +1,183 @@
1
+ """Ordering relations and footprint matrices (van der Aalst, ch. 6.2 and 8.4).
2
+
3
+ From the directly-follows relation ``>_L`` four *log-based ordering
4
+ relations* are derived, for every pair of activities a, b:
5
+
6
+ ==================== =========================================== =========
7
+ relation definition symbol
8
+ ==================== =========================================== =========
9
+ causality ``a > b`` and **not** ``b > a`` ``→``
10
+ inverse causality ``b > a`` and **not** ``a > b`` ``←``
11
+ parallel ``a > b`` **and** ``b > a`` ``‖``
12
+ choice / unrelated **neither** ``a > b`` nor ``b > a`` ``#``
13
+ ==================== =========================================== =========
14
+
15
+ Exactly one of the four holds for each ordered pair, so the relations can be
16
+ tabulated as the **footprint matrix**. Note the diagonal: ``a # a`` unless
17
+ ``a`` directly follows itself (a length-one loop), in which case ``a ‖ a``.
18
+
19
+ Frequencies play no role: a pair that directly follows once counts the same
20
+ as one that follows a thousand times. That is exactly why the α-algorithm is
21
+ sensitive to noise.
22
+
23
+ Conformance via footprints (ch. 8.4)
24
+ ------------------------------------
25
+ A model has a footprint too (computed from its behaviour). Comparing the log
26
+ footprint with the model footprint cell by cell gives a simple conformance
27
+ measure::
28
+
29
+ fitness_footprint = 1 - (number of differing cells) / (total cells)
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ from collections import deque
35
+ from dataclasses import dataclass
36
+
37
+ from .dfg import DFG, dfg_from_simple_log
38
+ from .log import SimpleLog
39
+ from .petrinet import PetriNet
40
+
41
+ CAUSAL = "→"
42
+ INVERSE = "←"
43
+ PARALLEL = "‖"
44
+ CHOICE = "#"
45
+
46
+
47
+ @dataclass
48
+ class Footprint:
49
+ activities: list[str]
50
+ directly_follows: set[tuple[str, str]]
51
+ start: set[str]
52
+ end: set[str]
53
+
54
+ def relation(self, a: str, b: str) -> str:
55
+ forward = (a, b) in self.directly_follows
56
+ backward = (b, a) in self.directly_follows
57
+ if forward and backward:
58
+ return PARALLEL
59
+ if forward:
60
+ return CAUSAL
61
+ if backward:
62
+ return INVERSE
63
+ return CHOICE
64
+
65
+ # Convenience predicates with the textbook names -----------------------
66
+ def causal(self, a: str, b: str) -> bool:
67
+ return self.relation(a, b) == CAUSAL
68
+
69
+ def parallel(self, a: str, b: str) -> bool:
70
+ return self.relation(a, b) == PARALLEL
71
+
72
+ def choice(self, a: str, b: str) -> bool:
73
+ return self.relation(a, b) == CHOICE
74
+
75
+ def matrix(self) -> list[list[str]]:
76
+ return [[self.relation(a, b) for b in self.activities] for a in self.activities]
77
+
78
+ def as_text(self) -> str:
79
+ width = max([len(a) for a in self.activities] + [1])
80
+ header = " " * (width + 1) + " ".join(a.rjust(width) for a in self.activities)
81
+ rows = [header]
82
+ for a, row in zip(self.activities, self.matrix()):
83
+ rows.append(a.rjust(width) + " " + " ".join(cell.rjust(width) for cell in row))
84
+ return "\n".join(rows)
85
+
86
+
87
+ def footprint_from_dfg(dfg: DFG) -> Footprint:
88
+ return Footprint(sorted(dfg.activities), set(dfg.edges), set(dfg.start), set(dfg.end))
89
+
90
+
91
+ def footprint_of_log(log: SimpleLog) -> Footprint:
92
+ return footprint_from_dfg(dfg_from_simple_log(log))
93
+
94
+
95
+ def footprint_of_net(net: PetriNet, max_states: int = 50_000) -> Footprint:
96
+ """The footprint of a model's behaviour.
97
+
98
+ ``a > b`` holds in the model iff some reachable marking allows ``a`` and
99
+ then ``b`` with only silent transitions in between. Computed from the
100
+ reachability graph, so the net must be bounded.
101
+ """
102
+ from .analysis import reachability_graph
103
+
104
+ graph = reachability_graph(net, max_states=max_states)
105
+ if graph.has_omega or graph.truncated:
106
+ raise ValueError("The model's state space is unbounded or too large to "
107
+ "compute its footprint.")
108
+
109
+ def visible_next(state: int) -> set[str]:
110
+ """Labels that can occur next from ``state``, skipping silent steps."""
111
+ seen, result = {state}, set()
112
+ queue = deque([state])
113
+ while queue:
114
+ current = queue.popleft()
115
+ for transition, target in graph.successors(current):
116
+ label = net.transitions[transition].label
117
+ if label is None:
118
+ if target not in seen:
119
+ seen.add(target)
120
+ queue.append(target)
121
+ else:
122
+ result.add(label)
123
+ return result
124
+
125
+ follows: set[tuple[str, str]] = set()
126
+ for source, transition, target in graph.edges:
127
+ label = net.transitions[transition].label
128
+ if label is None:
129
+ continue
130
+ for successor in visible_next(target):
131
+ follows.add((label, successor))
132
+ start = visible_next(0)
133
+ final_states = [i for i, m in enumerate(graph.states) if m == net.final_marking]
134
+ end: set[str] = set()
135
+ for source, transition, target in graph.edges:
136
+ label = net.transitions[transition].label
137
+ if label is not None and _silently_reaches(graph, net, target, set(final_states)):
138
+ end.add(label)
139
+ return Footprint(sorted(net.labels()), follows, start, end)
140
+
141
+
142
+ def _silently_reaches(graph, net, state: int, targets: set[int]) -> bool:
143
+ seen = {state}
144
+ queue = deque([state])
145
+ while queue:
146
+ current = queue.popleft()
147
+ if current in targets:
148
+ return True
149
+ for transition, target in graph.successors(current):
150
+ if net.transitions[transition].label is None and target not in seen:
151
+ seen.add(target)
152
+ queue.append(target)
153
+ return False
154
+
155
+
156
+ @dataclass
157
+ class FootprintComparison:
158
+ activities: list[str]
159
+ log_matrix: list[list[str]]
160
+ model_matrix: list[list[str]]
161
+
162
+ @property
163
+ def differences(self) -> list[tuple[str, str, str, str]]:
164
+ """(a, b, log relation, model relation) for every differing cell."""
165
+ result = []
166
+ for i, a in enumerate(self.activities):
167
+ for j, b in enumerate(self.activities):
168
+ if self.log_matrix[i][j] != self.model_matrix[i][j]:
169
+ result.append((a, b, self.log_matrix[i][j], self.model_matrix[i][j]))
170
+ return result
171
+
172
+ @property
173
+ def fitness(self) -> float:
174
+ cells = len(self.activities) ** 2
175
+ return 1.0 if cells == 0 else 1 - len(self.differences) / cells
176
+
177
+
178
+ def compare_footprints(log_fp: Footprint, model_fp: Footprint) -> FootprintComparison:
179
+ """Cell-by-cell comparison over the union of both alphabets."""
180
+ activities = sorted(set(log_fp.activities) | set(model_fp.activities))
181
+ log_matrix = [[log_fp.relation(a, b) for b in activities] for a in activities]
182
+ model_matrix = [[model_fp.relation(a, b) for b in activities] for a in activities]
183
+ return FootprintComparison(activities, log_matrix, model_matrix)
@@ -0,0 +1,191 @@
1
+ """The incidence matrix, place invariants and transition invariants.
2
+
3
+ Linear algebra instead of state spaces
4
+ --------------------------------------
5
+ Everything in :mod:`.analysis` explores markings one by one. Invariants
6
+ answer questions from the *structure* alone, with a little linear algebra,
7
+ so they work even when the state space is huge or infinite.
8
+
9
+ **Incidence matrix.** ``C`` has a row per place and a column per transition;
10
+ ``C(p, t) = W(t, p) − W(p, t)`` is what firing ``t`` does to ``p`` (the tokens
11
+ it puts in minus the tokens it takes out). Firing a sequence ``σ`` whose
12
+ Parikh vector ``σ⃗`` counts how often each transition fires gives the
13
+ **marking equation**::
14
+
15
+ M --σ--> M' implies M' = M + C · σ⃗
16
+
17
+ **Place invariant (P-invariant).** A weighting ``y`` of the places with
18
+ ``yᵀ · C = 0``. Multiplying the marking equation by ``yᵀ`` gives
19
+ ``y · M' = y · M``: the weighted token count is the same in *every* reachable
20
+ marking. In a WF-net, ``i + c1 + c2 + o`` being invariant says "exactly one
21
+ token moves through these places".
22
+
23
+ **Transition invariant (T-invariant).** A firing count ``x`` with
24
+ ``C · x = 0``: firing every transition as often as ``x`` says (in any
25
+ enabled order) leads back to the marking you started from. A cycle of the
26
+ net's behaviour.
27
+
28
+ We compute the **minimal semi-positive** invariants (no negative weights, not
29
+ all zero, and no other invariant uses a strict subset of their places or
30
+ transitions). Every semi-positive invariant is a non-negative combination
31
+ of these, so they are *the* invariants to show. The method is Farkas'
32
+ algorithm (Martínez & Silva, 1982): start from ``[C | I]`` and eliminate one
33
+ column of ``C`` at a time by adding pairs of rows with opposite signs.
34
+
35
+ What they tell you
36
+ ------------------
37
+ * Every place in some semi-positive P-invariant (**covered by P-invariants**)
38
+ ⇒ the net is *structurally bounded*: bounded from any initial marking.
39
+ * A net that is live and bounded is covered by T-invariants, so a transition
40
+ in no T-invariant means "not both live and bounded". For a WF-net, apply
41
+ that to the short-circuited net N̄: by the soundness theorem, a sound
42
+ WF-net's N̄ is covered by T-invariants.
43
+
44
+ The number of minimal invariants can grow exponentially with the size of the
45
+ net. :func:`invariants` stops at ``limit`` intermediate rows and says so.
46
+ """
47
+
48
+ from __future__ import annotations
49
+
50
+ from dataclasses import dataclass, field
51
+ from math import gcd
52
+
53
+ from .petrinet import Marking, PetriNet
54
+
55
+
56
+ class TooManyInvariants(Exception):
57
+ """Farkas' algorithm needed more than the allowed number of rows."""
58
+
59
+
60
+ def incidence_matrix(net: PetriNet) -> tuple[list[str], list[str], list[list[int]]]:
61
+ """``(places, transitions, C)`` with ``C[i][j] = W(t_j, p_i) − W(p_i, t_j)``."""
62
+ places = list(net.places)
63
+ transitions = list(net.transitions)
64
+ row = {p: i for i, p in enumerate(places)}
65
+ column = {t: j for j, t in enumerate(transitions)}
66
+ matrix = [[0] * len(transitions) for _ in places]
67
+ for arc in net.arcs:
68
+ if arc.source in row and arc.target in column: # p -> t: consumed
69
+ matrix[row[arc.source]][column[arc.target]] -= arc.weight
70
+ elif arc.source in column and arc.target in row: # t -> p: produced
71
+ matrix[row[arc.target]][column[arc.source]] += arc.weight
72
+ return places, transitions, matrix
73
+
74
+
75
+ def _normalise(vector: list[int]) -> list[int]:
76
+ divisor = 0
77
+ for value in vector:
78
+ divisor = gcd(divisor, value)
79
+ return [value // divisor for value in vector] if divisor > 1 else vector
80
+
81
+
82
+ def _support(vector: list[int]) -> frozenset[int]:
83
+ return frozenset(i for i, value in enumerate(vector) if value)
84
+
85
+
86
+ def _minimal(rows: list[tuple[list[int], list[int]]]) -> list[tuple[list[int], list[int]]]:
87
+ """Drop duplicate rows and rows whose support strictly contains another's."""
88
+ unique: dict[tuple[int, ...], tuple[list[int], list[int]]] = {}
89
+ for rest, generator in rows:
90
+ unique.setdefault(tuple(rest) + tuple(generator), (rest, generator))
91
+ rows = list(unique.values())
92
+ supports = [_support(generator) for _, generator in rows]
93
+ return [row for row, support in zip(rows, supports)
94
+ if not any(other < support for other in supports)]
95
+
96
+
97
+ def semi_positive_invariants(matrix: list[list[int]], limit: int = 5_000) -> list[list[int]]:
98
+ """Minimal semi-positive solutions ``y ≥ 0, y ≠ 0`` of ``yᵀ · matrix = 0``.
99
+
100
+ ``matrix`` has one row per variable. Farkas' algorithm: rows are pairs
101
+ (what is left of the matrix row, the combination of variables it stands
102
+ for). Column by column, rows with opposite signs in that column are
103
+ combined so that it cancels, and rows that do not cancel are dropped.
104
+ """
105
+ count = len(matrix)
106
+ columns = len(matrix[0]) if count else 0
107
+ rows = [(list(matrix[i]), [1 if j == i else 0 for j in range(count)]) for i in range(count)]
108
+ for column in range(columns):
109
+ keep = [row for row in rows if row[0][column] == 0]
110
+ positive = [row for row in rows if row[0][column] > 0]
111
+ negative = [row for row in rows if row[0][column] < 0]
112
+ for up_rest, up_generator in positive:
113
+ for down_rest, down_generator in negative:
114
+ # a·up + b·down cancels the column, with a, b > 0.
115
+ a, b = -down_rest[column], up_rest[column]
116
+ combined = _normalise([a * x + b * y for x, y in zip(
117
+ up_rest + up_generator, down_rest + down_generator)])
118
+ keep.append((combined[:columns], combined[columns:]))
119
+ rows = _minimal(keep)
120
+ if len(rows) > limit:
121
+ raise TooManyInvariants(f"more than {limit} candidate invariants")
122
+ invariants = [_normalise(generator) for _, generator in _minimal(rows) if any(generator)]
123
+ return sorted(invariants, key=lambda v: (sum(1 for x in v if x), [-x for x in v]))
124
+
125
+
126
+ @dataclass
127
+ class Invariants:
128
+ """The incidence matrix and the minimal semi-positive invariants of a net."""
129
+
130
+ net: PetriNet
131
+ places: list[str]
132
+ transitions: list[str]
133
+ #: ``incidence[i][j]``: the effect of transitions[j] on places[i].
134
+ incidence: list[list[int]]
135
+ #: Each P-invariant as ``{place id: weight}`` (weights > 0 only).
136
+ p_invariants: list[dict[str, int]] = field(default_factory=list)
137
+ #: Each T-invariant as ``{transition id: count}`` (counts > 0 only).
138
+ t_invariants: list[dict[str, int]] = field(default_factory=list)
139
+ #: Set when there were too many to compute: the lists are then empty.
140
+ p_truncated: bool = False
141
+ t_truncated: bool = False
142
+
143
+ def uncovered_places(self) -> list[str]:
144
+ covered = {p for invariant in self.p_invariants for p in invariant}
145
+ return [p for p in self.places if p not in covered]
146
+
147
+ def uncovered_transitions(self) -> list[str]:
148
+ covered = {t for invariant in self.t_invariants for t in invariant}
149
+ return [t for t in self.transitions if t not in covered]
150
+
151
+ @property
152
+ def covered_by_p_invariants(self) -> bool | None:
153
+ """Every place in some P-invariant (⇒ structurally bounded)."""
154
+ return None if self.p_truncated else not self.uncovered_places()
155
+
156
+ @property
157
+ def covered_by_t_invariants(self) -> bool | None:
158
+ return None if self.t_truncated else not self.uncovered_transitions()
159
+
160
+ def token_sum(self, invariant: dict[str, int], marking: Marking | None = None) -> int:
161
+ """``y · M``: the weighted token count the invariant keeps constant."""
162
+ marking = self.net.initial_marking if marking is None else marking
163
+ return sum(weight * marking[place] for place, weight in invariant.items())
164
+
165
+ def describe(self, invariant: dict[str, int], with_value: bool = False) -> str:
166
+ """``2·p1 + p2`` (P-invariant) or ``a + b + t*`` (T-invariant), by name."""
167
+ terms = [(f"{weight}·" if weight != 1 else "") + self.net.node_name(node)
168
+ for node, weight in invariant.items()]
169
+ text = " + ".join(terms)
170
+ if with_value:
171
+ text += f" = {self.token_sum(invariant)}"
172
+ return text
173
+
174
+
175
+ def invariants(net: PetriNet, limit: int = 5_000) -> Invariants:
176
+ """Incidence matrix plus minimal semi-positive P- and T-invariants."""
177
+ places, transitions, matrix = incidence_matrix(net)
178
+ result = Invariants(net, places, transitions, matrix)
179
+ try:
180
+ for vector in semi_positive_invariants(matrix, limit):
181
+ result.p_invariants.append({p: w for p, w in zip(places, vector) if w})
182
+ except TooManyInvariants:
183
+ result.p_truncated = True
184
+ transposed = [list(column) for column in zip(*matrix)] if places else \
185
+ [[] for _ in transitions]
186
+ try:
187
+ for vector in semi_positive_invariants(transposed, limit):
188
+ result.t_invariants.append({t: w for t, w in zip(transitions, vector) if w})
189
+ except TooManyInvariants:
190
+ result.t_truncated = True
191
+ return result