openprocess 0.7.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (173) hide show
  1. cpnpy/__init__.py +46 -0
  2. openprocess/__init__.py +57 -0
  3. openprocess/analysis/__init__.py +0 -0
  4. openprocess/analysis/state_space.py +521 -0
  5. openprocess/analysis/state_space_process.py +251 -0
  6. openprocess/cli.py +742 -0
  7. openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/answer.pnml +31 -0
  8. openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/question.md +38 -0
  9. openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/answer.pnml +27 -0
  10. openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/net.pnml +29 -0
  11. openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/question.md +65 -0
  12. openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/log.txt +1 -0
  13. openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/question.md +67 -0
  14. openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/question.md +75 -0
  15. openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/ts.txt +5 -0
  16. openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/net.pnml +27 -0
  17. openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/question.md +65 -0
  18. openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/log.txt +1 -0
  19. openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/question.md +62 -0
  20. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/log.txt +1 -0
  21. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m1.pnml +36 -0
  22. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m2.pnml +28 -0
  23. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m3.pnml +30 -0
  24. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/net.pnml +36 -0
  25. openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/question.md +69 -0
  26. openprocess/exercises/pack.md +14 -0
  27. openprocess/flow/__init__.py +50 -0
  28. openprocess/flow/box.py +466 -0
  29. openprocess/flow/boxes/__init__.py +7 -0
  30. openprocess/flow/boxes/check.py +119 -0
  31. openprocess/flow/boxes/compare.py +16 -0
  32. openprocess/flow/boxes/cpn.py +53 -0
  33. openprocess/flow/boxes/discover.py +124 -0
  34. openprocess/flow/boxes/filter.py +80 -0
  35. openprocess/flow/boxes/input.py +124 -0
  36. openprocess/flow/boxes/output.py +52 -0
  37. openprocess/flow/boxes/predict.py +186 -0
  38. openprocess/flow/boxes/science.py +159 -0
  39. openprocess/flow/boxes/sweeps.py +18 -0
  40. openprocess/flow/convert.py +187 -0
  41. openprocess/flow/datasets.py +198 -0
  42. openprocess/flow/explain.py +115 -0
  43. openprocess/flow/library.py +222 -0
  44. openprocess/flow/record.py +385 -0
  45. openprocess/flow/runner.py +357 -0
  46. openprocess/flow/sweep.py +92 -0
  47. openprocess/flow/types.py +290 -0
  48. openprocess/flow/workflow.py +628 -0
  49. openprocess/gui/__init__.py +0 -0
  50. openprocess/gui/app.py +90 -0
  51. openprocess/gui/arc_editing.py +295 -0
  52. openprocess/gui/canvas.py +1414 -0
  53. openprocess/gui/flow/__init__.py +8 -0
  54. openprocess/gui/flow/canvas.py +854 -0
  55. openprocess/gui/flow/page.py +972 -0
  56. openprocess/gui/flow/templates.py +131 -0
  57. openprocess/gui/flow/viewers.py +665 -0
  58. openprocess/gui/items.py +1275 -0
  59. openprocess/gui/learn/answer_boxes.py +978 -0
  60. openprocess/gui/learn/concealment.py +91 -0
  61. openprocess/gui/learn/mode.py +1181 -0
  62. openprocess/gui/panning.py +241 -0
  63. openprocess/gui/resources/openprocess-icon.png +0 -0
  64. openprocess/gui/studio/__init__.py +1 -0
  65. openprocess/gui/studio/__main__.py +3 -0
  66. openprocess/gui/studio/app.py +4031 -0
  67. openprocess/gui/studio/charts.py +115 -0
  68. openprocess/gui/studio/compare_page.py +487 -0
  69. openprocess/gui/studio/cpn_page.py +1858 -0
  70. openprocess/gui/studio/definition_view.py +284 -0
  71. openprocess/gui/studio/derivation_view.py +421 -0
  72. openprocess/gui/studio/documents.py +152 -0
  73. openprocess/gui/studio/dotted_chart.py +1401 -0
  74. openprocess/gui/studio/file_dialogs.py +143 -0
  75. openprocess/gui/studio/filter_dialog.py +247 -0
  76. openprocess/gui/studio/graph_builders.py +176 -0
  77. openprocess/gui/studio/graph_view.py +682 -0
  78. openprocess/gui/studio/instances.py +413 -0
  79. openprocess/gui/studio/log_editor.py +675 -0
  80. openprocess/gui/studio/log_page.py +800 -0
  81. openprocess/gui/studio/markdown_view.py +127 -0
  82. openprocess/gui/studio/mathtext.py +260 -0
  83. openprocess/gui/studio/ml_highlighter.py +75 -0
  84. openprocess/gui/studio/model_page.py +760 -0
  85. openprocess/gui/studio/net_comparison.py +124 -0
  86. openprocess/gui/studio/notes_overlay.py +275 -0
  87. openprocess/gui/studio/petri_page.py +844 -0
  88. openprocess/gui/studio/regions_view.py +502 -0
  89. openprocess/gui/studio/sidebar.py +149 -0
  90. openprocess/gui/studio/style.py +503 -0
  91. openprocess/gui/studio/tool_icons.py +134 -0
  92. openprocess/gui/studio/updates.py +439 -0
  93. openprocess/gui/studio/widgets.py +899 -0
  94. openprocess/gui/studio/workers.py +60 -0
  95. openprocess/gui/studio/workspace.py +447 -0
  96. openprocess/gui/theme.py +394 -0
  97. openprocess/gui/tidy.py +86 -0
  98. openprocess/io/__init__.py +0 -0
  99. openprocess/io/cpn_reader.py +389 -0
  100. openprocess/io/cpn_writer.py +357 -0
  101. openprocess/learn/__init__.py +23 -0
  102. openprocess/learn/answers.py +188 -0
  103. openprocess/learn/checks.py +953 -0
  104. openprocess/learn/computed.py +1180 -0
  105. openprocess/learn/context.py +145 -0
  106. openprocess/learn/exam.py +169 -0
  107. openprocess/learn/exercise-packs.md +325 -0
  108. openprocess/learn/importer.py +216 -0
  109. openprocess/learn/notation.py +474 -0
  110. openprocess/learn/pack.py +511 -0
  111. openprocess/learn/sheet.py +296 -0
  112. openprocess/mining/__init__.py +73 -0
  113. openprocess/mining/analysis.py +689 -0
  114. openprocess/mining/columns.py +282 -0
  115. openprocess/mining/compare_nets.py +246 -0
  116. openprocess/mining/conformance/__init__.py +0 -0
  117. openprocess/mining/conformance/alignments.py +263 -0
  118. openprocess/mining/conformance/quality.py +145 -0
  119. openprocess/mining/conformance/token_replay.py +252 -0
  120. openprocess/mining/csv_import.py +222 -0
  121. openprocess/mining/definitions.py +584 -0
  122. openprocess/mining/dfg.py +187 -0
  123. openprocess/mining/discovery/__init__.py +0 -0
  124. openprocess/mining/discovery/alpha.py +168 -0
  125. openprocess/mining/discovery/heuristics.py +332 -0
  126. openprocess/mining/discovery/inductive.py +477 -0
  127. openprocess/mining/discovery/state_regions.py +62 -0
  128. openprocess/mining/filtering.py +237 -0
  129. openprocess/mining/footprint.py +183 -0
  130. openprocess/mining/invariants.py +191 -0
  131. openprocess/mining/layout.py +279 -0
  132. openprocess/mining/log.py +364 -0
  133. openprocess/mining/petrinet.py +354 -0
  134. openprocess/mining/playout.py +75 -0
  135. openprocess/mining/pm4py_bridge.py +82 -0
  136. openprocess/mining/pnml.py +223 -0
  137. openprocess/mining/processtree.py +216 -0
  138. openprocess/mining/regions.py +476 -0
  139. openprocess/mining/stats.py +160 -0
  140. openprocess/mining/structure.py +374 -0
  141. openprocess/mining/transition_system.py +409 -0
  142. openprocess/mining/xes.py +399 -0
  143. openprocess/ml/__init__.py +0 -0
  144. openprocess/ml/ast_nodes.py +332 -0
  145. openprocess/ml/builtins.py +364 -0
  146. openprocess/ml/colorsets.py +522 -0
  147. openprocess/ml/errors.py +60 -0
  148. openprocess/ml/evaluator.py +754 -0
  149. openprocess/ml/lexer.py +277 -0
  150. openprocess/ml/multiset.py +417 -0
  151. openprocess/ml/parser.py +737 -0
  152. openprocess/ml/values.py +319 -0
  153. openprocess/model/__init__.py +0 -0
  154. openprocess/model/declarations.py +617 -0
  155. openprocess/model/examples.py +98 -0
  156. openprocess/model/net.py +701 -0
  157. openprocess/model/plain.py +192 -0
  158. openprocess/references.py +280 -0
  159. openprocess/sim/__init__.py +0 -0
  160. openprocess/sim/binding.py +620 -0
  161. openprocess/sim/export.py +66 -0
  162. openprocess/sim/simulator.py +315 -0
  163. openprocess/teaching/__init__.py +4 -0
  164. openprocess/teaching/answers.py +4 -0
  165. openprocess/teaching/checks.py +5 -0
  166. openprocess/teaching/pack.py +4 -0
  167. openprocess/teaching/sheet.py +4 -0
  168. openprocess-0.7.0.dist-info/METADATA +927 -0
  169. openprocess-0.7.0.dist-info/RECORD +173 -0
  170. openprocess-0.7.0.dist-info/WHEEL +5 -0
  171. openprocess-0.7.0.dist-info/entry_points.txt +6 -0
  172. openprocess-0.7.0.dist-info/licenses/LICENSE +21 -0
  173. openprocess-0.7.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,186 @@
1
+ """Predict boxes: the prediction pipeline of predictive process monitoring.
2
+
3
+ *Prefixes* turns a log into a dataset of prefix encodings with a label
4
+ (the next activity, the remaining time, or the outcome); *Split by time*
5
+ makes a train and a test set by case start, never at random; *Evaluate
6
+ predictions* scores a model's predictions per prefix length. A model of
7
+ your own goes in between as a box that takes a :class:`~..types.Dataset`
8
+ and gives a :class:`~..types.Predictor` (a scikit-learn estimator fits the
9
+ shape; see ``examples/boxes/next_activity_sklearn.py``). *Frequency
10
+ model* is a baseline that needs no library.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from collections import Counter
16
+ from dataclasses import dataclass
17
+ from datetime import datetime
18
+ from typing import Literal
19
+
20
+ from ... import flow
21
+ from ...mining.log import EventLog
22
+ from ..box import box
23
+ from ..types import Dataset, Predictions, Predictor, Table
24
+
25
+ END = "⟂" # the label of "the case ends here"
26
+
27
+
28
+ @box(name="Prefixes", group="Predict")
29
+ def prefixes(log: EventLog, label: Literal["next activity", "remaining time", "outcome"] = "next activity",
30
+ encoding: Literal["index", "one-hot", "frequency"] = "frequency",
31
+ min_prefix: int = 1, max_prefix: int = 20) -> Dataset:
32
+ """One row per prefix of every case, encoded as features, with the
33
+ label to predict: the next activity (⟂ when the case ends), the
34
+ remaining time in hours, or the outcome (the last activity).
35
+
36
+ encoding: frequency = how often each activity occurred so far, plus the last activity; index = the last k activities as indices; one-hot = the last k activities as 0/1 columns
37
+ min_prefix: the shortest prefix to include
38
+ max_prefix: the longest prefix to include (longer cases give more rows)
39
+ """
40
+ activities = sorted({a for seq in log.sequences() for a in seq})
41
+ index = {a: i for i, a in enumerate(activities)}
42
+ k = max_prefix if encoding != "frequency" else 1
43
+ if encoding == "frequency":
44
+ names = [f"count {a}" for a in activities] + ["prefix length"] + [f"last is {a}" for a in activities]
45
+ elif encoding == "index":
46
+ names = [f"activity −{i}" for i in range(k, 0, -1)]
47
+ else:
48
+ names = [f"−{i} is {a}" for i in range(k, 0, -1) for a in activities]
49
+ X, y, case_ids, lengths, starts = [], [], [], [], []
50
+ classifier = log.default_classifier()
51
+ for trace in log:
52
+ events = [e for e in trace if classifier.accepts(e)]
53
+ seq = [classifier.label(e) for e in events]
54
+ times = [e.timestamp for e in events]
55
+ start = times[0] if times and times[0] else None
56
+ end_time = times[-1] if times and times[-1] else None
57
+ for n in range(min_prefix, min(len(seq), max_prefix) + 1):
58
+ prefix = seq[:n]
59
+ if encoding == "frequency":
60
+ counts = Counter(prefix)
61
+ row = [counts[a] for a in activities] + [n] + [1 if prefix[-1] == a else 0 for a in activities]
62
+ elif encoding == "index":
63
+ window = ([-1] * k + [index[a] for a in prefix])[-k:]
64
+ row = window
65
+ else:
66
+ window = ([None] * k + prefix)[-k:]
67
+ row = [1 if w == a else 0 for w in window for a in activities]
68
+ if label == "next activity":
69
+ target = seq[n] if n < len(seq) else END
70
+ elif label == "outcome":
71
+ target = seq[-1]
72
+ else:
73
+ if end_time is None or times[n - 1] is None:
74
+ continue
75
+ target = (end_time - times[n - 1]).total_seconds() / 3600
76
+ X.append(row)
77
+ y.append(target)
78
+ case_ids.append(trace.case_id)
79
+ lengths.append(n)
80
+ starts.append(start)
81
+ kind = "regression" if label == "remaining time" else "classification"
82
+ classes = sorted(set(y)) if kind == "classification" else None
83
+ flow.note(f"{len(X)} prefixes from {len(log)} cases, {len(names)} features, label: {label}")
84
+ return Dataset(X, y, names, label, kind, case_ids, lengths, starts, activities, classes,
85
+ name=f"Prefixes of {log.name}")
86
+
87
+
88
+ @dataclass
89
+ class Split:
90
+ train: Dataset
91
+ test: Dataset
92
+
93
+
94
+ @box(name="Split by time", group="Predict")
95
+ def split_by_time(dataset: Dataset, train_fraction: float = 0.7) -> Split:
96
+ """Train and test sets by case start: the earliest cases train, the
97
+ latest test, so the model never sees the future. Without timestamps the
98
+ cases are split in log order."""
99
+ cases = list(dict.fromkeys(dataset.case_ids))
100
+ starts = {c: s for c, s in zip(dataset.case_ids, dataset.case_starts)}
101
+ if any(isinstance(starts.get(c), datetime) for c in cases):
102
+ cases.sort(key=lambda c: (starts.get(c) is None, starts.get(c) or datetime.min))
103
+ cut = max(1, int(round(len(cases) * train_fraction)))
104
+ train_cases = set(cases[:cut])
105
+ train = [i for i, c in enumerate(dataset.case_ids) if c in train_cases]
106
+ test = [i for i, c in enumerate(dataset.case_ids) if c not in train_cases]
107
+ flow.note(f"{len(train_cases)} cases ({len(train)} prefixes) train, "
108
+ f"{len(cases) - len(train_cases)} cases ({len(test)} prefixes) test")
109
+ return Split(dataset.subset(train, "train"), dataset.subset(test, "test"))
110
+
111
+
112
+ class FrequencyModel(Predictor):
113
+ """Predicts, for each last activity, what most often came next (or the
114
+ mean remaining time): the baseline every paper has to beat."""
115
+
116
+ name = "Frequency model"
117
+
118
+ def __init__(self) -> None:
119
+ self.table: dict = {}
120
+ self.default = None
121
+ self.kind = "classification"
122
+
123
+ def _key(self, row, dataset: Dataset):
124
+ if dataset.feature_names and dataset.feature_names[-1].startswith("last is "):
125
+ n = len(dataset.activities)
126
+ last = row[-n:]
127
+ return next((a for a, flag in zip(dataset.activities, last) if flag), None)
128
+ return tuple(row[-1:])
129
+
130
+ def fit(self, dataset: Dataset) -> "FrequencyModel":
131
+ self.kind = dataset.kind
132
+ seen: dict = {}
133
+ for row, label in zip(dataset.X, dataset.y):
134
+ seen.setdefault(self._key(row, dataset), []).append(label)
135
+ if dataset.kind == "classification":
136
+ self.table = {key: Counter(labels).most_common(1)[0][0] for key, labels in seen.items()}
137
+ self.default = Counter(dataset.y).most_common(1)[0][0] if dataset.y else None
138
+ else:
139
+ self.table = {key: sum(labels) / len(labels) for key, labels in seen.items()}
140
+ self.default = sum(dataset.y) / len(dataset.y) if dataset.y else 0.0
141
+ return self
142
+
143
+ def predict(self, dataset: Dataset) -> list:
144
+ return [self.table.get(self._key(row, dataset), self.default) for row in dataset.X]
145
+
146
+
147
+ @box(name="Frequency model", group="Predict")
148
+ def frequency_model(train: Dataset) -> Predictor:
149
+ """The baseline: for each last activity, the most frequent next activity
150
+ (or the mean remaining time) in the training set."""
151
+ model = FrequencyModel().fit(train)
152
+ flow.note(f"{len(model.table)} last activities seen in {len(train)} prefixes")
153
+ return model
154
+
155
+
156
+ @box(name="Predict", group="Predict")
157
+ def predict(model: Predictor, dataset: Dataset) -> Predictions:
158
+ """Applies a fitted model to a dataset (the test set)."""
159
+ values = list(model.predict(dataset))
160
+ flow.note(f"{len(values)} predictions by {getattr(model, 'name', type(model).__name__)}")
161
+ return Predictions(values, dataset, getattr(model, "name", type(model).__name__))
162
+
163
+
164
+ @box(name="Evaluate predictions", group="Predict")
165
+ def evaluate_predictions(predictions: Predictions) -> Table:
166
+ """Accuracy (classification) or mean absolute error in hours
167
+ (regression), overall and per prefix length."""
168
+ dataset = predictions.dataset
169
+ if len(predictions.values) != len(dataset):
170
+ raise ValueError("The predictions and the dataset have different lengths")
171
+ by_length: dict[int, list] = {}
172
+ for value, truth, length in zip(predictions.values, dataset.y, dataset.prefix_lengths or [0] * len(dataset)):
173
+ by_length.setdefault(length, []).append((value, truth))
174
+ classification = dataset.kind == "classification"
175
+
176
+ def score(pairs) -> float:
177
+ if classification:
178
+ return sum(1 for v, t in pairs if v == t) / len(pairs)
179
+ return sum(abs(float(v) - float(t)) for v, t in pairs) / len(pairs)
180
+ overall = score(list(zip(predictions.values, dataset.y)))
181
+ metric = "accuracy" if classification else "MAE (hours)"
182
+ rows = [["all", len(dataset), round(overall, 4)]]
183
+ rows += [[str(length), len(pairs), round(score(pairs), 4)] for length, pairs in sorted(by_length.items())]
184
+ flow.note(f"{metric} over {len(dataset)} prefixes: {overall:.3f}")
185
+ return Table(f"{predictions.name}: {metric}", ["prefix length", "prefixes", metric], rows,
186
+ note=f"{metric} of {predictions.name} on {dataset.name}: {overall:.3f}.")
@@ -0,0 +1,159 @@
1
+ """Science boxes: how any Python library plugs in.
2
+
3
+ Each box here is a few lines around one library call, so it reads as a
4
+ recipe: copy it, change three lines, save it in your folder's ``boxes/``.
5
+ They live behind ``pip install openprocess[science]``; without a library the box
6
+ is listed greyed out with "needs pandas" (the ``needs=`` of ``@box``).
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import random
12
+ from typing import Literal
13
+
14
+ from ... import flow
15
+ from ...mining.conformance.token_replay import token_replay
16
+ from ...mining.log import EventLog
17
+ from ...mining.petrinet import PetriNet
18
+ from ..box import box
19
+ from ..convert import convert
20
+ from ..types import Figure, Scores, Table
21
+
22
+
23
+ @box(name="Describe log", group="Science", needs="pandas")
24
+ def describe_log(log: EventLog) -> Table:
25
+ """pandas' ``describe()`` of the events per case and, when the log has
26
+ timestamps, of the case durations in hours. The log goes in as a
27
+ DataFrame through the converter; the table comes back as a Table."""
28
+ import pandas as pd
29
+ frame = convert(log, "pandas.DataFrame")
30
+ per_case = frame.groupby("case:concept:name", sort=False).size().rename("events per case")
31
+ columns = {"events per case": per_case.describe()}
32
+ if "time:timestamp" in frame.columns and frame["time:timestamp"].notna().any():
33
+ span = frame.groupby("case:concept:name")["time:timestamp"].agg(["min", "max"])
34
+ hours = ((span["max"] - span["min"]).dt.total_seconds() / 3600).rename("case duration (h)")
35
+ columns["case duration (h)"] = hours.describe()
36
+ described = pd.DataFrame(columns)
37
+ flow.note(f"df: {len(frame)} rows × {len(frame.columns)} columns; per_case = df.groupby(case).size()")
38
+ rows = [[stat, *[round(float(described.loc[stat, c]), 3) for c in described.columns]] for stat in described.index]
39
+ return Table("Describe", ["statistic", *described.columns], rows)
40
+
41
+
42
+ @box(name="Bootstrap fitness", group="Science", needs="numpy")
43
+ def bootstrap_fitness(model: PetriNet, log: EventLog, samples: int = 200, seed: int = 0) -> Table:
44
+ """Fitness with a 95 % bootstrap interval: resample the cases with
45
+ replacement *samples* times and replay each sample, so a number comes
46
+ with its uncertainty.
47
+
48
+ samples: how many resamples (more: a steadier interval)
49
+ seed: the random seed, recorded in the workflow file
50
+ """
51
+ import numpy as np
52
+ sequences = log.sequences()
53
+ rng = np.random.default_rng(seed)
54
+ values = []
55
+ for _ in range(samples):
56
+ picked = rng.integers(0, len(sequences), len(sequences))
57
+ from collections import Counter
58
+ sample = Counter(sequences[i] for i in picked)
59
+ values.append(token_replay(model, sample).fitness)
60
+ values = np.array(values)
61
+ low, high = np.percentile(values, [2.5, 97.5])
62
+ flow.note(f"{samples} samples of {len(sequences)} cases, seed {seed}; "
63
+ f"min {values.min():.3f}, max {values.max():.3f}")
64
+ table = Table(f"Bootstrap fitness of {model.name}", ["statistic", "fitness"],
65
+ [["mean", round(float(values.mean()), 4)], ["2.5 %", round(float(low), 4)],
66
+ ["97.5 %", round(float(high), 4)], ["samples", samples], ["seed", seed]],
67
+ samples=[float(v) for v in values],
68
+ note=f"Fitness of {model.name}: {values.mean():.3f} (95 % interval {low:.3f} to {high:.3f}).")
69
+ return table
70
+
71
+
72
+ @box(name="Compare samples", group="Science", needs="scipy")
73
+ def compare_samples(a: Table, b: Table,
74
+ test: Literal["Mann–Whitney U", "t-test"] = "Mann–Whitney U") -> Table:
75
+ """Are two samples (two bootstrap tables) from the same distribution?
76
+ A Mann–Whitney U test, or Welch's t-test, from SciPy."""
77
+ from scipy import stats
78
+ if not a.samples or not b.samples:
79
+ raise ValueError("Both tables need samples (connect two Bootstrap fitness boxes)")
80
+ x, y = a.samples, b.samples
81
+ if test == "t-test":
82
+ result = stats.ttest_ind(x, y, equal_var=False)
83
+ statistic_name = "t"
84
+ else:
85
+ result = stats.mannwhitneyu(x, y, alternative="two-sided")
86
+ statistic_name = "U"
87
+ p = float(result.pvalue)
88
+ mean_a, mean_b = sum(x) / len(x), sum(y) / len(y)
89
+ verdict = (f"The samples differ (p {'< 0.001' if p < 0.001 else f'= {p:.3f}'}): "
90
+ f"{a.name if mean_a > mean_b else b.name} is higher, and the bootstrap says it is not luck."
91
+ if p < 0.05 else f"No difference the test can see (p = {p:.3f}).")
92
+ flow.note(f"A = {a.name} ({len(x)} samples, mean {mean_a:.3f}); B = {b.name} ({len(y)} samples, mean {mean_b:.3f})")
93
+ return Table(test, ["", "value"], [[statistic_name, round(float(result.statistic), 3)],
94
+ ["p-value", "< 0.001" if p < 0.001 else round(p, 4)],
95
+ ["mean A", round(mean_a, 4)], ["mean B", round(mean_b, 4)]], note=verdict)
96
+
97
+
98
+ @box(name="Correlate", group="Science", needs="pandas")
99
+ def correlate(table: Table) -> Table:
100
+ """Correlations between the numeric columns of a table (pandas), e.g.
101
+ a swept noise level against fitness."""
102
+ frame = convert(table, "pandas.DataFrame").select_dtypes("number")
103
+ if frame.shape[1] < 2:
104
+ raise ValueError("The table needs at least two numeric columns")
105
+ matrix = frame.corr()
106
+ rows = [[row, *[round(float(v), 3) for v in matrix.loc[row]]] for row in matrix.index]
107
+ return Table("Correlations", ["", *matrix.columns], rows)
108
+
109
+
110
+ @box(name="Plot", group="Science", needs="matplotlib")
111
+ def plot(scores: list[Scores] = (), tables: list[Table] = (),
112
+ x: str = "", y: str = "") -> Figure:
113
+ """Bars per model for scores, histograms for bootstrap tables, or a line
114
+ of column *y* against column *x* of a table (a sweep). The figure is a
115
+ matplotlib figure, converted to SVG; its code is this box's code.
116
+
117
+ x: a column of the table to put on the x axis (lines only)
118
+ y: the column to plot against it
119
+ """
120
+ import matplotlib
121
+ matplotlib.use("Agg")
122
+ import matplotlib.pyplot as plt
123
+ fig, ax = plt.subplots(figsize=(5.5, 3.2))
124
+ scores, tables = list(scores), list(tables)
125
+ if scores:
126
+ metrics = [m for m in dict.fromkeys(k for s in scores for k in s.metrics)
127
+ if all(isinstance(s.metrics.get(m, 0.0), (int, float)) for s in scores)]
128
+ width = 0.8 / max(1, len(metrics))
129
+ for j, metric in enumerate(metrics):
130
+ ax.bar([i + j * width for i in range(len(scores))], [float(s.metrics.get(metric, 0)) for s in scores],
131
+ width=width, label=metric)
132
+ ax.set_xticks([i + 0.4 - width / 2 for i in range(len(scores))])
133
+ ax.set_xticklabels([s.model for s in scores], rotation=15, ha="right")
134
+ ax.set_ylim(0, 1.05)
135
+ ax.legend(fontsize=8)
136
+ caption = f"{len(scores)} models, {len(metrics)} scores each."
137
+ elif tables and x and y:
138
+ for table in tables:
139
+ ax.plot(table.column(x), table.column(y), marker="o", label=table.name)
140
+ ax.set_xlabel(x)
141
+ ax.set_ylabel(y)
142
+ if len(tables) > 1:
143
+ ax.legend(fontsize=8)
144
+ caption = f"{y} against {x}."
145
+ elif tables:
146
+ for table in tables:
147
+ if table.samples:
148
+ ax.hist(table.samples, bins=12, alpha=0.6, label=table.name)
149
+ ax.set_xlabel("fitness over bootstrap samples")
150
+ ax.legend(fontsize=8)
151
+ caption = f"{len(tables)} bootstrap distribution{'s' if len(tables) != 1 else ''}."
152
+ else:
153
+ raise ValueError("Connect scores, or tables with samples or an x and y column")
154
+ fig.tight_layout()
155
+ figure = convert(fig, Figure)
156
+ plt.close(fig)
157
+ figure.caption = caption
158
+ flow.note(f"matplotlib {matplotlib.__version__}: {caption}")
159
+ return figure
@@ -0,0 +1,18 @@
1
+ """Sweep table: the scores of every swept value, stacked."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from ... import flow
6
+ from ..box import box
7
+ from ..types import Scores, Table
8
+
9
+
10
+ @box(name="Sweep table", group="Sweep")
11
+ def sweep_table(scores: list[Scores]) -> Table:
12
+ """Collects the scores of every value of a sweep into one table: the
13
+ swept settings as columns (from the scores' context), then the scores.
14
+ Feed it to Plot (with x and y) or Correlate."""
15
+ table = Table.from_scores(scores, "Sweep")
16
+ swept = [c for c in table.columns if c != "model" and any(c in s.context for s in scores)]
17
+ flow.note(f"{len(table)} rows; swept: {', '.join(swept) or 'nothing'}")
18
+ return table
@@ -0,0 +1,187 @@
1
+ """Converters between OpenProcess's types and other libraries' objects.
2
+
3
+ A box that wraps a pandas, PM4Py or matplotlib function asks for the
4
+ conversion in one line::
5
+
6
+ df = convert(log, "pandas.DataFrame") # one row per event
7
+ return convert(fig, Figure) # a matplotlib figure as SVG
8
+
9
+ Foreign types are named by their dotted path, so this module imports none
10
+ of them until a conversion is asked for. Register your own with
11
+ :func:`register`: a function from one type to another, keyed by the source
12
+ class and the target (a class, or a dotted name).
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import importlib
18
+ from typing import Any, Callable
19
+
20
+ from ..mining.log import KEY_NAME, KEY_RESOURCE, KEY_TIME, EventLog, SimpleLog
21
+ from ..mining.petrinet import PetriNet
22
+ from .types import Figure, Table
23
+
24
+ _converters: dict[tuple[str, str], Callable] = {}
25
+
26
+
27
+ def _name(target) -> str:
28
+ if isinstance(target, str):
29
+ return target
30
+ return f"{target.__module__}.{target.__qualname__}"
31
+
32
+
33
+ def register(source, target, function: Callable) -> None:
34
+ """``function(value) -> converted`` for a value of class ``source``."""
35
+ _converters[(_name(source), _name(target))] = function
36
+
37
+
38
+ def _class_names(value) -> list[str]:
39
+ return [f"{cls.__module__}.{cls.__qualname__}" for cls in type(value).__mro__]
40
+
41
+
42
+ def can_convert(value: Any, target) -> bool:
43
+ return any((name, _name(target)) in _converters for name in _class_names(value))
44
+
45
+
46
+ def convert(value: Any, target):
47
+ """Convert ``value`` to ``target`` (a class or a dotted name)."""
48
+ wanted = _name(target)
49
+ for name in _class_names(value):
50
+ if name == wanted:
51
+ return value
52
+ function = _converters.get((name, wanted))
53
+ if function is not None:
54
+ return function(value)
55
+ raise TypeError(f"No converter from {type(value).__name__} to {wanted}")
56
+
57
+
58
+ def converters() -> list[tuple[str, str]]:
59
+ return sorted(_converters)
60
+
61
+
62
+ # ---------------------------------------------------------------------------
63
+ # Built-in converters
64
+ # ---------------------------------------------------------------------------
65
+ def _simple_to_log(simple: SimpleLog) -> EventLog:
66
+ return EventLog.from_simple_log(simple)
67
+
68
+
69
+ def _log_to_simple(log: EventLog) -> SimpleLog:
70
+ return log.simple_log()
71
+
72
+
73
+ def _log_to_dataframe(log: EventLog):
74
+ pd = importlib.import_module("pandas")
75
+ rows = []
76
+ for trace in log:
77
+ case = trace.case_id
78
+ for event in trace:
79
+ row = {"case:concept:name": case}
80
+ row.update(event.attributes)
81
+ rows.append(row)
82
+ frame = pd.DataFrame(rows)
83
+ if KEY_TIME in frame.columns:
84
+ frame[KEY_TIME] = pd.to_datetime(frame[KEY_TIME], utc=True, errors="coerce")
85
+ return frame
86
+
87
+
88
+ def _dataframe_to_log(frame) -> EventLog:
89
+ from ..mining.log import Event, Trace
90
+ case_column = "case:concept:name" if "case:concept:name" in frame.columns else frame.columns[0]
91
+ log = EventLog(attributes={KEY_NAME: "DataFrame"})
92
+ for case, group in frame.groupby(case_column, sort=False):
93
+ trace = Trace({KEY_NAME: str(case)})
94
+ for record in group.drop(columns=[case_column]).to_dict("records"):
95
+ attributes = {k: (v.to_pydatetime() if hasattr(v, "to_pydatetime") else v)
96
+ for k, v in record.items() if v == v} # drop NaN
97
+ trace.events.append(Event(attributes))
98
+ log.traces.append(trace)
99
+ return log
100
+
101
+
102
+ def _table_to_dataframe(table: Table):
103
+ pd = importlib.import_module("pandas")
104
+ return pd.DataFrame(table.rows, columns=table.columns)
105
+
106
+
107
+ def _dataframe_to_table(frame) -> Table:
108
+ columns = [str(c) for c in frame.columns]
109
+ rows = [[_plain(v) for v in row] for row in frame.itertuples(index=False, name=None)]
110
+ return Table("DataFrame", columns, rows)
111
+
112
+
113
+ def _series_to_table(series) -> Table:
114
+ name = str(series.name or "value")
115
+ return Table(name, ["index", name], [[_plain(k), _plain(v)] for k, v in series.items()])
116
+
117
+
118
+ def _plain(value):
119
+ item = getattr(value, "item", None) # NumPy scalars -> Python
120
+ if callable(item):
121
+ try:
122
+ return item()
123
+ except (ValueError, TypeError):
124
+ return value
125
+ return value
126
+
127
+
128
+ def _ndarray_to_table(array) -> Table:
129
+ rows = array.tolist()
130
+ if rows and not isinstance(rows[0], list):
131
+ rows = [[r] for r in rows]
132
+ width = len(rows[0]) if rows else 0
133
+ return Table("array", [f"c{i}" for i in range(width)], rows)
134
+
135
+
136
+ def _mpl_figure_to_figure(fig) -> Figure:
137
+ import io
138
+ buffer = io.StringIO()
139
+ fig.savefig(buffer, format="svg", bbox_inches="tight")
140
+ svg = buffer.getvalue()
141
+ png = io.BytesIO()
142
+ try:
143
+ fig.savefig(png, format="png", dpi=160, bbox_inches="tight")
144
+ png_bytes = png.getvalue()
145
+ except Exception: # noqa: BLE001 - no PNG backend
146
+ png_bytes = None
147
+ title = ""
148
+ try:
149
+ title = fig.axes[0].get_title() if fig.axes else ""
150
+ except Exception: # noqa: BLE001
151
+ pass
152
+ return Figure(title or "Figure", svg=svg, png=png_bytes)
153
+
154
+
155
+ def _petri_to_pm4py(net: PetriNet):
156
+ import os
157
+ import tempfile
158
+ pm4py = importlib.import_module("pm4py")
159
+ from ..mining.pnml import write_pnml
160
+ handle, path = tempfile.mkstemp(suffix=".pnml")
161
+ os.close(handle)
162
+ try:
163
+ write_pnml(net, path)
164
+ return pm4py.read_pnml(path)
165
+ finally:
166
+ os.unlink(path)
167
+
168
+
169
+ def _pm4py_to_petri(triple) -> PetriNet:
170
+ from ..mining.pm4py_bridge import _to_native
171
+ net, im, fm = triple
172
+ return _to_native(net, im, fm, getattr(net, "name", None) or "PM4Py net")
173
+
174
+
175
+ register(SimpleLog, EventLog, _simple_to_log)
176
+ register(EventLog, SimpleLog, _log_to_simple)
177
+ register(EventLog, "pandas.core.frame.DataFrame", _log_to_dataframe)
178
+ register("pandas.core.frame.DataFrame", EventLog, _dataframe_to_log)
179
+ register(Table, "pandas.core.frame.DataFrame", _table_to_dataframe)
180
+ register("pandas.core.frame.DataFrame", Table, _dataframe_to_table)
181
+ register("pandas.core.series.Series", Table, _series_to_table)
182
+ register("numpy.ndarray", Table, _ndarray_to_table)
183
+ register("matplotlib.figure.Figure", Figure, _mpl_figure_to_figure)
184
+ register(PetriNet, "pm4py", _petri_to_pm4py)
185
+ register(tuple, "openprocess.mining.petrinet.PetriNet", _pm4py_to_petri)
186
+
187
+ __all__ = ["can_convert", "convert", "converters", "register"]