openprocess 0.7.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cpnpy/__init__.py +46 -0
- openprocess/__init__.py +57 -0
- openprocess/analysis/__init__.py +0 -0
- openprocess/analysis/state_space.py +521 -0
- openprocess/analysis/state_space_process.py +251 -0
- openprocess/cli.py +742 -0
- openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/answer.pnml +31 -0
- openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/question.md +38 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/answer.pnml +27 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/net.pnml +29 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/question.md +65 -0
- openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/log.txt +1 -0
- openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/question.md +67 -0
- openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/question.md +75 -0
- openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/ts.txt +5 -0
- openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/net.pnml +27 -0
- openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/question.md +65 -0
- openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/log.txt +1 -0
- openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/question.md +62 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/log.txt +1 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m1.pnml +36 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m2.pnml +28 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m3.pnml +30 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/net.pnml +36 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/question.md +69 -0
- openprocess/exercises/pack.md +14 -0
- openprocess/flow/__init__.py +50 -0
- openprocess/flow/box.py +466 -0
- openprocess/flow/boxes/__init__.py +7 -0
- openprocess/flow/boxes/check.py +119 -0
- openprocess/flow/boxes/compare.py +16 -0
- openprocess/flow/boxes/cpn.py +53 -0
- openprocess/flow/boxes/discover.py +124 -0
- openprocess/flow/boxes/filter.py +80 -0
- openprocess/flow/boxes/input.py +124 -0
- openprocess/flow/boxes/output.py +52 -0
- openprocess/flow/boxes/predict.py +186 -0
- openprocess/flow/boxes/science.py +159 -0
- openprocess/flow/boxes/sweeps.py +18 -0
- openprocess/flow/convert.py +187 -0
- openprocess/flow/datasets.py +198 -0
- openprocess/flow/explain.py +115 -0
- openprocess/flow/library.py +222 -0
- openprocess/flow/record.py +385 -0
- openprocess/flow/runner.py +357 -0
- openprocess/flow/sweep.py +92 -0
- openprocess/flow/types.py +290 -0
- openprocess/flow/workflow.py +628 -0
- openprocess/gui/__init__.py +0 -0
- openprocess/gui/app.py +90 -0
- openprocess/gui/arc_editing.py +295 -0
- openprocess/gui/canvas.py +1414 -0
- openprocess/gui/flow/__init__.py +8 -0
- openprocess/gui/flow/canvas.py +854 -0
- openprocess/gui/flow/page.py +972 -0
- openprocess/gui/flow/templates.py +131 -0
- openprocess/gui/flow/viewers.py +665 -0
- openprocess/gui/items.py +1275 -0
- openprocess/gui/learn/answer_boxes.py +978 -0
- openprocess/gui/learn/concealment.py +91 -0
- openprocess/gui/learn/mode.py +1181 -0
- openprocess/gui/panning.py +241 -0
- openprocess/gui/resources/openprocess-icon.png +0 -0
- openprocess/gui/studio/__init__.py +1 -0
- openprocess/gui/studio/__main__.py +3 -0
- openprocess/gui/studio/app.py +4031 -0
- openprocess/gui/studio/charts.py +115 -0
- openprocess/gui/studio/compare_page.py +487 -0
- openprocess/gui/studio/cpn_page.py +1858 -0
- openprocess/gui/studio/definition_view.py +284 -0
- openprocess/gui/studio/derivation_view.py +421 -0
- openprocess/gui/studio/documents.py +152 -0
- openprocess/gui/studio/dotted_chart.py +1401 -0
- openprocess/gui/studio/file_dialogs.py +143 -0
- openprocess/gui/studio/filter_dialog.py +247 -0
- openprocess/gui/studio/graph_builders.py +176 -0
- openprocess/gui/studio/graph_view.py +682 -0
- openprocess/gui/studio/instances.py +413 -0
- openprocess/gui/studio/log_editor.py +675 -0
- openprocess/gui/studio/log_page.py +800 -0
- openprocess/gui/studio/markdown_view.py +127 -0
- openprocess/gui/studio/mathtext.py +260 -0
- openprocess/gui/studio/ml_highlighter.py +75 -0
- openprocess/gui/studio/model_page.py +760 -0
- openprocess/gui/studio/net_comparison.py +124 -0
- openprocess/gui/studio/notes_overlay.py +275 -0
- openprocess/gui/studio/petri_page.py +844 -0
- openprocess/gui/studio/regions_view.py +502 -0
- openprocess/gui/studio/sidebar.py +149 -0
- openprocess/gui/studio/style.py +503 -0
- openprocess/gui/studio/tool_icons.py +134 -0
- openprocess/gui/studio/updates.py +439 -0
- openprocess/gui/studio/widgets.py +899 -0
- openprocess/gui/studio/workers.py +60 -0
- openprocess/gui/studio/workspace.py +447 -0
- openprocess/gui/theme.py +394 -0
- openprocess/gui/tidy.py +86 -0
- openprocess/io/__init__.py +0 -0
- openprocess/io/cpn_reader.py +389 -0
- openprocess/io/cpn_writer.py +357 -0
- openprocess/learn/__init__.py +23 -0
- openprocess/learn/answers.py +188 -0
- openprocess/learn/checks.py +953 -0
- openprocess/learn/computed.py +1180 -0
- openprocess/learn/context.py +145 -0
- openprocess/learn/exam.py +169 -0
- openprocess/learn/exercise-packs.md +325 -0
- openprocess/learn/importer.py +216 -0
- openprocess/learn/notation.py +474 -0
- openprocess/learn/pack.py +511 -0
- openprocess/learn/sheet.py +296 -0
- openprocess/mining/__init__.py +73 -0
- openprocess/mining/analysis.py +689 -0
- openprocess/mining/columns.py +282 -0
- openprocess/mining/compare_nets.py +246 -0
- openprocess/mining/conformance/__init__.py +0 -0
- openprocess/mining/conformance/alignments.py +263 -0
- openprocess/mining/conformance/quality.py +145 -0
- openprocess/mining/conformance/token_replay.py +252 -0
- openprocess/mining/csv_import.py +222 -0
- openprocess/mining/definitions.py +584 -0
- openprocess/mining/dfg.py +187 -0
- openprocess/mining/discovery/__init__.py +0 -0
- openprocess/mining/discovery/alpha.py +168 -0
- openprocess/mining/discovery/heuristics.py +332 -0
- openprocess/mining/discovery/inductive.py +477 -0
- openprocess/mining/discovery/state_regions.py +62 -0
- openprocess/mining/filtering.py +237 -0
- openprocess/mining/footprint.py +183 -0
- openprocess/mining/invariants.py +191 -0
- openprocess/mining/layout.py +279 -0
- openprocess/mining/log.py +364 -0
- openprocess/mining/petrinet.py +354 -0
- openprocess/mining/playout.py +75 -0
- openprocess/mining/pm4py_bridge.py +82 -0
- openprocess/mining/pnml.py +223 -0
- openprocess/mining/processtree.py +216 -0
- openprocess/mining/regions.py +476 -0
- openprocess/mining/stats.py +160 -0
- openprocess/mining/structure.py +374 -0
- openprocess/mining/transition_system.py +409 -0
- openprocess/mining/xes.py +399 -0
- openprocess/ml/__init__.py +0 -0
- openprocess/ml/ast_nodes.py +332 -0
- openprocess/ml/builtins.py +364 -0
- openprocess/ml/colorsets.py +522 -0
- openprocess/ml/errors.py +60 -0
- openprocess/ml/evaluator.py +754 -0
- openprocess/ml/lexer.py +277 -0
- openprocess/ml/multiset.py +417 -0
- openprocess/ml/parser.py +737 -0
- openprocess/ml/values.py +319 -0
- openprocess/model/__init__.py +0 -0
- openprocess/model/declarations.py +617 -0
- openprocess/model/examples.py +98 -0
- openprocess/model/net.py +701 -0
- openprocess/model/plain.py +192 -0
- openprocess/references.py +280 -0
- openprocess/sim/__init__.py +0 -0
- openprocess/sim/binding.py +620 -0
- openprocess/sim/export.py +66 -0
- openprocess/sim/simulator.py +315 -0
- openprocess/teaching/__init__.py +4 -0
- openprocess/teaching/answers.py +4 -0
- openprocess/teaching/checks.py +5 -0
- openprocess/teaching/pack.py +4 -0
- openprocess/teaching/sheet.py +4 -0
- openprocess-0.7.0.dist-info/METADATA +927 -0
- openprocess-0.7.0.dist-info/RECORD +173 -0
- openprocess-0.7.0.dist-info/WHEEL +5 -0
- openprocess-0.7.0.dist-info/entry_points.txt +6 -0
- openprocess-0.7.0.dist-info/licenses/LICENSE +21 -0
- openprocess-0.7.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
"""Predict boxes: the prediction pipeline of predictive process monitoring.
|
|
2
|
+
|
|
3
|
+
*Prefixes* turns a log into a dataset of prefix encodings with a label
|
|
4
|
+
(the next activity, the remaining time, or the outcome); *Split by time*
|
|
5
|
+
makes a train and a test set by case start, never at random; *Evaluate
|
|
6
|
+
predictions* scores a model's predictions per prefix length. A model of
|
|
7
|
+
your own goes in between as a box that takes a :class:`~..types.Dataset`
|
|
8
|
+
and gives a :class:`~..types.Predictor` (a scikit-learn estimator fits the
|
|
9
|
+
shape; see ``examples/boxes/next_activity_sklearn.py``). *Frequency
|
|
10
|
+
model* is a baseline that needs no library.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from collections import Counter
|
|
16
|
+
from dataclasses import dataclass
|
|
17
|
+
from datetime import datetime
|
|
18
|
+
from typing import Literal
|
|
19
|
+
|
|
20
|
+
from ... import flow
|
|
21
|
+
from ...mining.log import EventLog
|
|
22
|
+
from ..box import box
|
|
23
|
+
from ..types import Dataset, Predictions, Predictor, Table
|
|
24
|
+
|
|
25
|
+
END = "⟂" # the label of "the case ends here"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@box(name="Prefixes", group="Predict")
|
|
29
|
+
def prefixes(log: EventLog, label: Literal["next activity", "remaining time", "outcome"] = "next activity",
|
|
30
|
+
encoding: Literal["index", "one-hot", "frequency"] = "frequency",
|
|
31
|
+
min_prefix: int = 1, max_prefix: int = 20) -> Dataset:
|
|
32
|
+
"""One row per prefix of every case, encoded as features, with the
|
|
33
|
+
label to predict: the next activity (⟂ when the case ends), the
|
|
34
|
+
remaining time in hours, or the outcome (the last activity).
|
|
35
|
+
|
|
36
|
+
encoding: frequency = how often each activity occurred so far, plus the last activity; index = the last k activities as indices; one-hot = the last k activities as 0/1 columns
|
|
37
|
+
min_prefix: the shortest prefix to include
|
|
38
|
+
max_prefix: the longest prefix to include (longer cases give more rows)
|
|
39
|
+
"""
|
|
40
|
+
activities = sorted({a for seq in log.sequences() for a in seq})
|
|
41
|
+
index = {a: i for i, a in enumerate(activities)}
|
|
42
|
+
k = max_prefix if encoding != "frequency" else 1
|
|
43
|
+
if encoding == "frequency":
|
|
44
|
+
names = [f"count {a}" for a in activities] + ["prefix length"] + [f"last is {a}" for a in activities]
|
|
45
|
+
elif encoding == "index":
|
|
46
|
+
names = [f"activity −{i}" for i in range(k, 0, -1)]
|
|
47
|
+
else:
|
|
48
|
+
names = [f"−{i} is {a}" for i in range(k, 0, -1) for a in activities]
|
|
49
|
+
X, y, case_ids, lengths, starts = [], [], [], [], []
|
|
50
|
+
classifier = log.default_classifier()
|
|
51
|
+
for trace in log:
|
|
52
|
+
events = [e for e in trace if classifier.accepts(e)]
|
|
53
|
+
seq = [classifier.label(e) for e in events]
|
|
54
|
+
times = [e.timestamp for e in events]
|
|
55
|
+
start = times[0] if times and times[0] else None
|
|
56
|
+
end_time = times[-1] if times and times[-1] else None
|
|
57
|
+
for n in range(min_prefix, min(len(seq), max_prefix) + 1):
|
|
58
|
+
prefix = seq[:n]
|
|
59
|
+
if encoding == "frequency":
|
|
60
|
+
counts = Counter(prefix)
|
|
61
|
+
row = [counts[a] for a in activities] + [n] + [1 if prefix[-1] == a else 0 for a in activities]
|
|
62
|
+
elif encoding == "index":
|
|
63
|
+
window = ([-1] * k + [index[a] for a in prefix])[-k:]
|
|
64
|
+
row = window
|
|
65
|
+
else:
|
|
66
|
+
window = ([None] * k + prefix)[-k:]
|
|
67
|
+
row = [1 if w == a else 0 for w in window for a in activities]
|
|
68
|
+
if label == "next activity":
|
|
69
|
+
target = seq[n] if n < len(seq) else END
|
|
70
|
+
elif label == "outcome":
|
|
71
|
+
target = seq[-1]
|
|
72
|
+
else:
|
|
73
|
+
if end_time is None or times[n - 1] is None:
|
|
74
|
+
continue
|
|
75
|
+
target = (end_time - times[n - 1]).total_seconds() / 3600
|
|
76
|
+
X.append(row)
|
|
77
|
+
y.append(target)
|
|
78
|
+
case_ids.append(trace.case_id)
|
|
79
|
+
lengths.append(n)
|
|
80
|
+
starts.append(start)
|
|
81
|
+
kind = "regression" if label == "remaining time" else "classification"
|
|
82
|
+
classes = sorted(set(y)) if kind == "classification" else None
|
|
83
|
+
flow.note(f"{len(X)} prefixes from {len(log)} cases, {len(names)} features, label: {label}")
|
|
84
|
+
return Dataset(X, y, names, label, kind, case_ids, lengths, starts, activities, classes,
|
|
85
|
+
name=f"Prefixes of {log.name}")
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
@dataclass
|
|
89
|
+
class Split:
|
|
90
|
+
train: Dataset
|
|
91
|
+
test: Dataset
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
@box(name="Split by time", group="Predict")
|
|
95
|
+
def split_by_time(dataset: Dataset, train_fraction: float = 0.7) -> Split:
|
|
96
|
+
"""Train and test sets by case start: the earliest cases train, the
|
|
97
|
+
latest test, so the model never sees the future. Without timestamps the
|
|
98
|
+
cases are split in log order."""
|
|
99
|
+
cases = list(dict.fromkeys(dataset.case_ids))
|
|
100
|
+
starts = {c: s for c, s in zip(dataset.case_ids, dataset.case_starts)}
|
|
101
|
+
if any(isinstance(starts.get(c), datetime) for c in cases):
|
|
102
|
+
cases.sort(key=lambda c: (starts.get(c) is None, starts.get(c) or datetime.min))
|
|
103
|
+
cut = max(1, int(round(len(cases) * train_fraction)))
|
|
104
|
+
train_cases = set(cases[:cut])
|
|
105
|
+
train = [i for i, c in enumerate(dataset.case_ids) if c in train_cases]
|
|
106
|
+
test = [i for i, c in enumerate(dataset.case_ids) if c not in train_cases]
|
|
107
|
+
flow.note(f"{len(train_cases)} cases ({len(train)} prefixes) train, "
|
|
108
|
+
f"{len(cases) - len(train_cases)} cases ({len(test)} prefixes) test")
|
|
109
|
+
return Split(dataset.subset(train, "train"), dataset.subset(test, "test"))
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
class FrequencyModel(Predictor):
|
|
113
|
+
"""Predicts, for each last activity, what most often came next (or the
|
|
114
|
+
mean remaining time): the baseline every paper has to beat."""
|
|
115
|
+
|
|
116
|
+
name = "Frequency model"
|
|
117
|
+
|
|
118
|
+
def __init__(self) -> None:
|
|
119
|
+
self.table: dict = {}
|
|
120
|
+
self.default = None
|
|
121
|
+
self.kind = "classification"
|
|
122
|
+
|
|
123
|
+
def _key(self, row, dataset: Dataset):
|
|
124
|
+
if dataset.feature_names and dataset.feature_names[-1].startswith("last is "):
|
|
125
|
+
n = len(dataset.activities)
|
|
126
|
+
last = row[-n:]
|
|
127
|
+
return next((a for a, flag in zip(dataset.activities, last) if flag), None)
|
|
128
|
+
return tuple(row[-1:])
|
|
129
|
+
|
|
130
|
+
def fit(self, dataset: Dataset) -> "FrequencyModel":
|
|
131
|
+
self.kind = dataset.kind
|
|
132
|
+
seen: dict = {}
|
|
133
|
+
for row, label in zip(dataset.X, dataset.y):
|
|
134
|
+
seen.setdefault(self._key(row, dataset), []).append(label)
|
|
135
|
+
if dataset.kind == "classification":
|
|
136
|
+
self.table = {key: Counter(labels).most_common(1)[0][0] for key, labels in seen.items()}
|
|
137
|
+
self.default = Counter(dataset.y).most_common(1)[0][0] if dataset.y else None
|
|
138
|
+
else:
|
|
139
|
+
self.table = {key: sum(labels) / len(labels) for key, labels in seen.items()}
|
|
140
|
+
self.default = sum(dataset.y) / len(dataset.y) if dataset.y else 0.0
|
|
141
|
+
return self
|
|
142
|
+
|
|
143
|
+
def predict(self, dataset: Dataset) -> list:
|
|
144
|
+
return [self.table.get(self._key(row, dataset), self.default) for row in dataset.X]
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
@box(name="Frequency model", group="Predict")
|
|
148
|
+
def frequency_model(train: Dataset) -> Predictor:
|
|
149
|
+
"""The baseline: for each last activity, the most frequent next activity
|
|
150
|
+
(or the mean remaining time) in the training set."""
|
|
151
|
+
model = FrequencyModel().fit(train)
|
|
152
|
+
flow.note(f"{len(model.table)} last activities seen in {len(train)} prefixes")
|
|
153
|
+
return model
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
@box(name="Predict", group="Predict")
|
|
157
|
+
def predict(model: Predictor, dataset: Dataset) -> Predictions:
|
|
158
|
+
"""Applies a fitted model to a dataset (the test set)."""
|
|
159
|
+
values = list(model.predict(dataset))
|
|
160
|
+
flow.note(f"{len(values)} predictions by {getattr(model, 'name', type(model).__name__)}")
|
|
161
|
+
return Predictions(values, dataset, getattr(model, "name", type(model).__name__))
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
@box(name="Evaluate predictions", group="Predict")
|
|
165
|
+
def evaluate_predictions(predictions: Predictions) -> Table:
|
|
166
|
+
"""Accuracy (classification) or mean absolute error in hours
|
|
167
|
+
(regression), overall and per prefix length."""
|
|
168
|
+
dataset = predictions.dataset
|
|
169
|
+
if len(predictions.values) != len(dataset):
|
|
170
|
+
raise ValueError("The predictions and the dataset have different lengths")
|
|
171
|
+
by_length: dict[int, list] = {}
|
|
172
|
+
for value, truth, length in zip(predictions.values, dataset.y, dataset.prefix_lengths or [0] * len(dataset)):
|
|
173
|
+
by_length.setdefault(length, []).append((value, truth))
|
|
174
|
+
classification = dataset.kind == "classification"
|
|
175
|
+
|
|
176
|
+
def score(pairs) -> float:
|
|
177
|
+
if classification:
|
|
178
|
+
return sum(1 for v, t in pairs if v == t) / len(pairs)
|
|
179
|
+
return sum(abs(float(v) - float(t)) for v, t in pairs) / len(pairs)
|
|
180
|
+
overall = score(list(zip(predictions.values, dataset.y)))
|
|
181
|
+
metric = "accuracy" if classification else "MAE (hours)"
|
|
182
|
+
rows = [["all", len(dataset), round(overall, 4)]]
|
|
183
|
+
rows += [[str(length), len(pairs), round(score(pairs), 4)] for length, pairs in sorted(by_length.items())]
|
|
184
|
+
flow.note(f"{metric} over {len(dataset)} prefixes: {overall:.3f}")
|
|
185
|
+
return Table(f"{predictions.name}: {metric}", ["prefix length", "prefixes", metric], rows,
|
|
186
|
+
note=f"{metric} of {predictions.name} on {dataset.name}: {overall:.3f}.")
|
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
"""Science boxes: how any Python library plugs in.
|
|
2
|
+
|
|
3
|
+
Each box here is a few lines around one library call, so it reads as a
|
|
4
|
+
recipe: copy it, change three lines, save it in your folder's ``boxes/``.
|
|
5
|
+
They live behind ``pip install openprocess[science]``; without a library the box
|
|
6
|
+
is listed greyed out with "needs pandas" (the ``needs=`` of ``@box``).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import random
|
|
12
|
+
from typing import Literal
|
|
13
|
+
|
|
14
|
+
from ... import flow
|
|
15
|
+
from ...mining.conformance.token_replay import token_replay
|
|
16
|
+
from ...mining.log import EventLog
|
|
17
|
+
from ...mining.petrinet import PetriNet
|
|
18
|
+
from ..box import box
|
|
19
|
+
from ..convert import convert
|
|
20
|
+
from ..types import Figure, Scores, Table
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@box(name="Describe log", group="Science", needs="pandas")
|
|
24
|
+
def describe_log(log: EventLog) -> Table:
|
|
25
|
+
"""pandas' ``describe()`` of the events per case and, when the log has
|
|
26
|
+
timestamps, of the case durations in hours. The log goes in as a
|
|
27
|
+
DataFrame through the converter; the table comes back as a Table."""
|
|
28
|
+
import pandas as pd
|
|
29
|
+
frame = convert(log, "pandas.DataFrame")
|
|
30
|
+
per_case = frame.groupby("case:concept:name", sort=False).size().rename("events per case")
|
|
31
|
+
columns = {"events per case": per_case.describe()}
|
|
32
|
+
if "time:timestamp" in frame.columns and frame["time:timestamp"].notna().any():
|
|
33
|
+
span = frame.groupby("case:concept:name")["time:timestamp"].agg(["min", "max"])
|
|
34
|
+
hours = ((span["max"] - span["min"]).dt.total_seconds() / 3600).rename("case duration (h)")
|
|
35
|
+
columns["case duration (h)"] = hours.describe()
|
|
36
|
+
described = pd.DataFrame(columns)
|
|
37
|
+
flow.note(f"df: {len(frame)} rows × {len(frame.columns)} columns; per_case = df.groupby(case).size()")
|
|
38
|
+
rows = [[stat, *[round(float(described.loc[stat, c]), 3) for c in described.columns]] for stat in described.index]
|
|
39
|
+
return Table("Describe", ["statistic", *described.columns], rows)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@box(name="Bootstrap fitness", group="Science", needs="numpy")
|
|
43
|
+
def bootstrap_fitness(model: PetriNet, log: EventLog, samples: int = 200, seed: int = 0) -> Table:
|
|
44
|
+
"""Fitness with a 95 % bootstrap interval: resample the cases with
|
|
45
|
+
replacement *samples* times and replay each sample, so a number comes
|
|
46
|
+
with its uncertainty.
|
|
47
|
+
|
|
48
|
+
samples: how many resamples (more: a steadier interval)
|
|
49
|
+
seed: the random seed, recorded in the workflow file
|
|
50
|
+
"""
|
|
51
|
+
import numpy as np
|
|
52
|
+
sequences = log.sequences()
|
|
53
|
+
rng = np.random.default_rng(seed)
|
|
54
|
+
values = []
|
|
55
|
+
for _ in range(samples):
|
|
56
|
+
picked = rng.integers(0, len(sequences), len(sequences))
|
|
57
|
+
from collections import Counter
|
|
58
|
+
sample = Counter(sequences[i] for i in picked)
|
|
59
|
+
values.append(token_replay(model, sample).fitness)
|
|
60
|
+
values = np.array(values)
|
|
61
|
+
low, high = np.percentile(values, [2.5, 97.5])
|
|
62
|
+
flow.note(f"{samples} samples of {len(sequences)} cases, seed {seed}; "
|
|
63
|
+
f"min {values.min():.3f}, max {values.max():.3f}")
|
|
64
|
+
table = Table(f"Bootstrap fitness of {model.name}", ["statistic", "fitness"],
|
|
65
|
+
[["mean", round(float(values.mean()), 4)], ["2.5 %", round(float(low), 4)],
|
|
66
|
+
["97.5 %", round(float(high), 4)], ["samples", samples], ["seed", seed]],
|
|
67
|
+
samples=[float(v) for v in values],
|
|
68
|
+
note=f"Fitness of {model.name}: {values.mean():.3f} (95 % interval {low:.3f} to {high:.3f}).")
|
|
69
|
+
return table
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@box(name="Compare samples", group="Science", needs="scipy")
|
|
73
|
+
def compare_samples(a: Table, b: Table,
|
|
74
|
+
test: Literal["Mann–Whitney U", "t-test"] = "Mann–Whitney U") -> Table:
|
|
75
|
+
"""Are two samples (two bootstrap tables) from the same distribution?
|
|
76
|
+
A Mann–Whitney U test, or Welch's t-test, from SciPy."""
|
|
77
|
+
from scipy import stats
|
|
78
|
+
if not a.samples or not b.samples:
|
|
79
|
+
raise ValueError("Both tables need samples (connect two Bootstrap fitness boxes)")
|
|
80
|
+
x, y = a.samples, b.samples
|
|
81
|
+
if test == "t-test":
|
|
82
|
+
result = stats.ttest_ind(x, y, equal_var=False)
|
|
83
|
+
statistic_name = "t"
|
|
84
|
+
else:
|
|
85
|
+
result = stats.mannwhitneyu(x, y, alternative="two-sided")
|
|
86
|
+
statistic_name = "U"
|
|
87
|
+
p = float(result.pvalue)
|
|
88
|
+
mean_a, mean_b = sum(x) / len(x), sum(y) / len(y)
|
|
89
|
+
verdict = (f"The samples differ (p {'< 0.001' if p < 0.001 else f'= {p:.3f}'}): "
|
|
90
|
+
f"{a.name if mean_a > mean_b else b.name} is higher, and the bootstrap says it is not luck."
|
|
91
|
+
if p < 0.05 else f"No difference the test can see (p = {p:.3f}).")
|
|
92
|
+
flow.note(f"A = {a.name} ({len(x)} samples, mean {mean_a:.3f}); B = {b.name} ({len(y)} samples, mean {mean_b:.3f})")
|
|
93
|
+
return Table(test, ["", "value"], [[statistic_name, round(float(result.statistic), 3)],
|
|
94
|
+
["p-value", "< 0.001" if p < 0.001 else round(p, 4)],
|
|
95
|
+
["mean A", round(mean_a, 4)], ["mean B", round(mean_b, 4)]], note=verdict)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
@box(name="Correlate", group="Science", needs="pandas")
|
|
99
|
+
def correlate(table: Table) -> Table:
|
|
100
|
+
"""Correlations between the numeric columns of a table (pandas), e.g.
|
|
101
|
+
a swept noise level against fitness."""
|
|
102
|
+
frame = convert(table, "pandas.DataFrame").select_dtypes("number")
|
|
103
|
+
if frame.shape[1] < 2:
|
|
104
|
+
raise ValueError("The table needs at least two numeric columns")
|
|
105
|
+
matrix = frame.corr()
|
|
106
|
+
rows = [[row, *[round(float(v), 3) for v in matrix.loc[row]]] for row in matrix.index]
|
|
107
|
+
return Table("Correlations", ["", *matrix.columns], rows)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
@box(name="Plot", group="Science", needs="matplotlib")
|
|
111
|
+
def plot(scores: list[Scores] = (), tables: list[Table] = (),
|
|
112
|
+
x: str = "", y: str = "") -> Figure:
|
|
113
|
+
"""Bars per model for scores, histograms for bootstrap tables, or a line
|
|
114
|
+
of column *y* against column *x* of a table (a sweep). The figure is a
|
|
115
|
+
matplotlib figure, converted to SVG; its code is this box's code.
|
|
116
|
+
|
|
117
|
+
x: a column of the table to put on the x axis (lines only)
|
|
118
|
+
y: the column to plot against it
|
|
119
|
+
"""
|
|
120
|
+
import matplotlib
|
|
121
|
+
matplotlib.use("Agg")
|
|
122
|
+
import matplotlib.pyplot as plt
|
|
123
|
+
fig, ax = plt.subplots(figsize=(5.5, 3.2))
|
|
124
|
+
scores, tables = list(scores), list(tables)
|
|
125
|
+
if scores:
|
|
126
|
+
metrics = [m for m in dict.fromkeys(k for s in scores for k in s.metrics)
|
|
127
|
+
if all(isinstance(s.metrics.get(m, 0.0), (int, float)) for s in scores)]
|
|
128
|
+
width = 0.8 / max(1, len(metrics))
|
|
129
|
+
for j, metric in enumerate(metrics):
|
|
130
|
+
ax.bar([i + j * width for i in range(len(scores))], [float(s.metrics.get(metric, 0)) for s in scores],
|
|
131
|
+
width=width, label=metric)
|
|
132
|
+
ax.set_xticks([i + 0.4 - width / 2 for i in range(len(scores))])
|
|
133
|
+
ax.set_xticklabels([s.model for s in scores], rotation=15, ha="right")
|
|
134
|
+
ax.set_ylim(0, 1.05)
|
|
135
|
+
ax.legend(fontsize=8)
|
|
136
|
+
caption = f"{len(scores)} models, {len(metrics)} scores each."
|
|
137
|
+
elif tables and x and y:
|
|
138
|
+
for table in tables:
|
|
139
|
+
ax.plot(table.column(x), table.column(y), marker="o", label=table.name)
|
|
140
|
+
ax.set_xlabel(x)
|
|
141
|
+
ax.set_ylabel(y)
|
|
142
|
+
if len(tables) > 1:
|
|
143
|
+
ax.legend(fontsize=8)
|
|
144
|
+
caption = f"{y} against {x}."
|
|
145
|
+
elif tables:
|
|
146
|
+
for table in tables:
|
|
147
|
+
if table.samples:
|
|
148
|
+
ax.hist(table.samples, bins=12, alpha=0.6, label=table.name)
|
|
149
|
+
ax.set_xlabel("fitness over bootstrap samples")
|
|
150
|
+
ax.legend(fontsize=8)
|
|
151
|
+
caption = f"{len(tables)} bootstrap distribution{'s' if len(tables) != 1 else ''}."
|
|
152
|
+
else:
|
|
153
|
+
raise ValueError("Connect scores, or tables with samples or an x and y column")
|
|
154
|
+
fig.tight_layout()
|
|
155
|
+
figure = convert(fig, Figure)
|
|
156
|
+
plt.close(fig)
|
|
157
|
+
figure.caption = caption
|
|
158
|
+
flow.note(f"matplotlib {matplotlib.__version__}: {caption}")
|
|
159
|
+
return figure
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""Sweep table: the scores of every swept value, stacked."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from ... import flow
|
|
6
|
+
from ..box import box
|
|
7
|
+
from ..types import Scores, Table
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@box(name="Sweep table", group="Sweep")
|
|
11
|
+
def sweep_table(scores: list[Scores]) -> Table:
|
|
12
|
+
"""Collects the scores of every value of a sweep into one table: the
|
|
13
|
+
swept settings as columns (from the scores' context), then the scores.
|
|
14
|
+
Feed it to Plot (with x and y) or Correlate."""
|
|
15
|
+
table = Table.from_scores(scores, "Sweep")
|
|
16
|
+
swept = [c for c in table.columns if c != "model" and any(c in s.context for s in scores)]
|
|
17
|
+
flow.note(f"{len(table)} rows; swept: {', '.join(swept) or 'nothing'}")
|
|
18
|
+
return table
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
"""Converters between OpenProcess's types and other libraries' objects.
|
|
2
|
+
|
|
3
|
+
A box that wraps a pandas, PM4Py or matplotlib function asks for the
|
|
4
|
+
conversion in one line::
|
|
5
|
+
|
|
6
|
+
df = convert(log, "pandas.DataFrame") # one row per event
|
|
7
|
+
return convert(fig, Figure) # a matplotlib figure as SVG
|
|
8
|
+
|
|
9
|
+
Foreign types are named by their dotted path, so this module imports none
|
|
10
|
+
of them until a conversion is asked for. Register your own with
|
|
11
|
+
:func:`register`: a function from one type to another, keyed by the source
|
|
12
|
+
class and the target (a class, or a dotted name).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import importlib
|
|
18
|
+
from typing import Any, Callable
|
|
19
|
+
|
|
20
|
+
from ..mining.log import KEY_NAME, KEY_RESOURCE, KEY_TIME, EventLog, SimpleLog
|
|
21
|
+
from ..mining.petrinet import PetriNet
|
|
22
|
+
from .types import Figure, Table
|
|
23
|
+
|
|
24
|
+
_converters: dict[tuple[str, str], Callable] = {}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _name(target) -> str:
|
|
28
|
+
if isinstance(target, str):
|
|
29
|
+
return target
|
|
30
|
+
return f"{target.__module__}.{target.__qualname__}"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def register(source, target, function: Callable) -> None:
|
|
34
|
+
"""``function(value) -> converted`` for a value of class ``source``."""
|
|
35
|
+
_converters[(_name(source), _name(target))] = function
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _class_names(value) -> list[str]:
|
|
39
|
+
return [f"{cls.__module__}.{cls.__qualname__}" for cls in type(value).__mro__]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def can_convert(value: Any, target) -> bool:
|
|
43
|
+
return any((name, _name(target)) in _converters for name in _class_names(value))
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def convert(value: Any, target):
|
|
47
|
+
"""Convert ``value`` to ``target`` (a class or a dotted name)."""
|
|
48
|
+
wanted = _name(target)
|
|
49
|
+
for name in _class_names(value):
|
|
50
|
+
if name == wanted:
|
|
51
|
+
return value
|
|
52
|
+
function = _converters.get((name, wanted))
|
|
53
|
+
if function is not None:
|
|
54
|
+
return function(value)
|
|
55
|
+
raise TypeError(f"No converter from {type(value).__name__} to {wanted}")
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def converters() -> list[tuple[str, str]]:
|
|
59
|
+
return sorted(_converters)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
# ---------------------------------------------------------------------------
|
|
63
|
+
# Built-in converters
|
|
64
|
+
# ---------------------------------------------------------------------------
|
|
65
|
+
def _simple_to_log(simple: SimpleLog) -> EventLog:
|
|
66
|
+
return EventLog.from_simple_log(simple)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _log_to_simple(log: EventLog) -> SimpleLog:
|
|
70
|
+
return log.simple_log()
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _log_to_dataframe(log: EventLog):
|
|
74
|
+
pd = importlib.import_module("pandas")
|
|
75
|
+
rows = []
|
|
76
|
+
for trace in log:
|
|
77
|
+
case = trace.case_id
|
|
78
|
+
for event in trace:
|
|
79
|
+
row = {"case:concept:name": case}
|
|
80
|
+
row.update(event.attributes)
|
|
81
|
+
rows.append(row)
|
|
82
|
+
frame = pd.DataFrame(rows)
|
|
83
|
+
if KEY_TIME in frame.columns:
|
|
84
|
+
frame[KEY_TIME] = pd.to_datetime(frame[KEY_TIME], utc=True, errors="coerce")
|
|
85
|
+
return frame
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _dataframe_to_log(frame) -> EventLog:
|
|
89
|
+
from ..mining.log import Event, Trace
|
|
90
|
+
case_column = "case:concept:name" if "case:concept:name" in frame.columns else frame.columns[0]
|
|
91
|
+
log = EventLog(attributes={KEY_NAME: "DataFrame"})
|
|
92
|
+
for case, group in frame.groupby(case_column, sort=False):
|
|
93
|
+
trace = Trace({KEY_NAME: str(case)})
|
|
94
|
+
for record in group.drop(columns=[case_column]).to_dict("records"):
|
|
95
|
+
attributes = {k: (v.to_pydatetime() if hasattr(v, "to_pydatetime") else v)
|
|
96
|
+
for k, v in record.items() if v == v} # drop NaN
|
|
97
|
+
trace.events.append(Event(attributes))
|
|
98
|
+
log.traces.append(trace)
|
|
99
|
+
return log
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _table_to_dataframe(table: Table):
|
|
103
|
+
pd = importlib.import_module("pandas")
|
|
104
|
+
return pd.DataFrame(table.rows, columns=table.columns)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _dataframe_to_table(frame) -> Table:
|
|
108
|
+
columns = [str(c) for c in frame.columns]
|
|
109
|
+
rows = [[_plain(v) for v in row] for row in frame.itertuples(index=False, name=None)]
|
|
110
|
+
return Table("DataFrame", columns, rows)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _series_to_table(series) -> Table:
|
|
114
|
+
name = str(series.name or "value")
|
|
115
|
+
return Table(name, ["index", name], [[_plain(k), _plain(v)] for k, v in series.items()])
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _plain(value):
|
|
119
|
+
item = getattr(value, "item", None) # NumPy scalars -> Python
|
|
120
|
+
if callable(item):
|
|
121
|
+
try:
|
|
122
|
+
return item()
|
|
123
|
+
except (ValueError, TypeError):
|
|
124
|
+
return value
|
|
125
|
+
return value
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _ndarray_to_table(array) -> Table:
|
|
129
|
+
rows = array.tolist()
|
|
130
|
+
if rows and not isinstance(rows[0], list):
|
|
131
|
+
rows = [[r] for r in rows]
|
|
132
|
+
width = len(rows[0]) if rows else 0
|
|
133
|
+
return Table("array", [f"c{i}" for i in range(width)], rows)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _mpl_figure_to_figure(fig) -> Figure:
|
|
137
|
+
import io
|
|
138
|
+
buffer = io.StringIO()
|
|
139
|
+
fig.savefig(buffer, format="svg", bbox_inches="tight")
|
|
140
|
+
svg = buffer.getvalue()
|
|
141
|
+
png = io.BytesIO()
|
|
142
|
+
try:
|
|
143
|
+
fig.savefig(png, format="png", dpi=160, bbox_inches="tight")
|
|
144
|
+
png_bytes = png.getvalue()
|
|
145
|
+
except Exception: # noqa: BLE001 - no PNG backend
|
|
146
|
+
png_bytes = None
|
|
147
|
+
title = ""
|
|
148
|
+
try:
|
|
149
|
+
title = fig.axes[0].get_title() if fig.axes else ""
|
|
150
|
+
except Exception: # noqa: BLE001
|
|
151
|
+
pass
|
|
152
|
+
return Figure(title or "Figure", svg=svg, png=png_bytes)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _petri_to_pm4py(net: PetriNet):
|
|
156
|
+
import os
|
|
157
|
+
import tempfile
|
|
158
|
+
pm4py = importlib.import_module("pm4py")
|
|
159
|
+
from ..mining.pnml import write_pnml
|
|
160
|
+
handle, path = tempfile.mkstemp(suffix=".pnml")
|
|
161
|
+
os.close(handle)
|
|
162
|
+
try:
|
|
163
|
+
write_pnml(net, path)
|
|
164
|
+
return pm4py.read_pnml(path)
|
|
165
|
+
finally:
|
|
166
|
+
os.unlink(path)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _pm4py_to_petri(triple) -> PetriNet:
|
|
170
|
+
from ..mining.pm4py_bridge import _to_native
|
|
171
|
+
net, im, fm = triple
|
|
172
|
+
return _to_native(net, im, fm, getattr(net, "name", None) or "PM4Py net")
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
register(SimpleLog, EventLog, _simple_to_log)
|
|
176
|
+
register(EventLog, SimpleLog, _log_to_simple)
|
|
177
|
+
register(EventLog, "pandas.core.frame.DataFrame", _log_to_dataframe)
|
|
178
|
+
register("pandas.core.frame.DataFrame", EventLog, _dataframe_to_log)
|
|
179
|
+
register(Table, "pandas.core.frame.DataFrame", _table_to_dataframe)
|
|
180
|
+
register("pandas.core.frame.DataFrame", Table, _dataframe_to_table)
|
|
181
|
+
register("pandas.core.series.Series", Table, _series_to_table)
|
|
182
|
+
register("numpy.ndarray", Table, _ndarray_to_table)
|
|
183
|
+
register("matplotlib.figure.Figure", Figure, _mpl_figure_to_figure)
|
|
184
|
+
register(PetriNet, "pm4py", _petri_to_pm4py)
|
|
185
|
+
register(tuple, "openprocess.mining.petrinet.PetriNet", _pm4py_to_petri)
|
|
186
|
+
|
|
187
|
+
__all__ = ["can_convert", "convert", "converters", "register"]
|