openprocess 0.7.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cpnpy/__init__.py +46 -0
- openprocess/__init__.py +57 -0
- openprocess/analysis/__init__.py +0 -0
- openprocess/analysis/state_space.py +521 -0
- openprocess/analysis/state_space_process.py +251 -0
- openprocess/cli.py +742 -0
- openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/answer.pnml +31 -0
- openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/question.md +38 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/answer.pnml +27 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/net.pnml +29 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/question.md +65 -0
- openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/log.txt +1 -0
- openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/question.md +67 -0
- openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/question.md +75 -0
- openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/ts.txt +5 -0
- openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/net.pnml +27 -0
- openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/question.md +65 -0
- openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/log.txt +1 -0
- openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/question.md +62 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/log.txt +1 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m1.pnml +36 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m2.pnml +28 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m3.pnml +30 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/net.pnml +36 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/question.md +69 -0
- openprocess/exercises/pack.md +14 -0
- openprocess/flow/__init__.py +50 -0
- openprocess/flow/box.py +466 -0
- openprocess/flow/boxes/__init__.py +7 -0
- openprocess/flow/boxes/check.py +119 -0
- openprocess/flow/boxes/compare.py +16 -0
- openprocess/flow/boxes/cpn.py +53 -0
- openprocess/flow/boxes/discover.py +124 -0
- openprocess/flow/boxes/filter.py +80 -0
- openprocess/flow/boxes/input.py +124 -0
- openprocess/flow/boxes/output.py +52 -0
- openprocess/flow/boxes/predict.py +186 -0
- openprocess/flow/boxes/science.py +159 -0
- openprocess/flow/boxes/sweeps.py +18 -0
- openprocess/flow/convert.py +187 -0
- openprocess/flow/datasets.py +198 -0
- openprocess/flow/explain.py +115 -0
- openprocess/flow/library.py +222 -0
- openprocess/flow/record.py +385 -0
- openprocess/flow/runner.py +357 -0
- openprocess/flow/sweep.py +92 -0
- openprocess/flow/types.py +290 -0
- openprocess/flow/workflow.py +628 -0
- openprocess/gui/__init__.py +0 -0
- openprocess/gui/app.py +90 -0
- openprocess/gui/arc_editing.py +295 -0
- openprocess/gui/canvas.py +1414 -0
- openprocess/gui/flow/__init__.py +8 -0
- openprocess/gui/flow/canvas.py +854 -0
- openprocess/gui/flow/page.py +972 -0
- openprocess/gui/flow/templates.py +131 -0
- openprocess/gui/flow/viewers.py +665 -0
- openprocess/gui/items.py +1275 -0
- openprocess/gui/learn/answer_boxes.py +978 -0
- openprocess/gui/learn/concealment.py +91 -0
- openprocess/gui/learn/mode.py +1181 -0
- openprocess/gui/panning.py +241 -0
- openprocess/gui/resources/openprocess-icon.png +0 -0
- openprocess/gui/studio/__init__.py +1 -0
- openprocess/gui/studio/__main__.py +3 -0
- openprocess/gui/studio/app.py +4031 -0
- openprocess/gui/studio/charts.py +115 -0
- openprocess/gui/studio/compare_page.py +487 -0
- openprocess/gui/studio/cpn_page.py +1858 -0
- openprocess/gui/studio/definition_view.py +284 -0
- openprocess/gui/studio/derivation_view.py +421 -0
- openprocess/gui/studio/documents.py +152 -0
- openprocess/gui/studio/dotted_chart.py +1401 -0
- openprocess/gui/studio/file_dialogs.py +143 -0
- openprocess/gui/studio/filter_dialog.py +247 -0
- openprocess/gui/studio/graph_builders.py +176 -0
- openprocess/gui/studio/graph_view.py +682 -0
- openprocess/gui/studio/instances.py +413 -0
- openprocess/gui/studio/log_editor.py +675 -0
- openprocess/gui/studio/log_page.py +800 -0
- openprocess/gui/studio/markdown_view.py +127 -0
- openprocess/gui/studio/mathtext.py +260 -0
- openprocess/gui/studio/ml_highlighter.py +75 -0
- openprocess/gui/studio/model_page.py +760 -0
- openprocess/gui/studio/net_comparison.py +124 -0
- openprocess/gui/studio/notes_overlay.py +275 -0
- openprocess/gui/studio/petri_page.py +844 -0
- openprocess/gui/studio/regions_view.py +502 -0
- openprocess/gui/studio/sidebar.py +149 -0
- openprocess/gui/studio/style.py +503 -0
- openprocess/gui/studio/tool_icons.py +134 -0
- openprocess/gui/studio/updates.py +439 -0
- openprocess/gui/studio/widgets.py +899 -0
- openprocess/gui/studio/workers.py +60 -0
- openprocess/gui/studio/workspace.py +447 -0
- openprocess/gui/theme.py +394 -0
- openprocess/gui/tidy.py +86 -0
- openprocess/io/__init__.py +0 -0
- openprocess/io/cpn_reader.py +389 -0
- openprocess/io/cpn_writer.py +357 -0
- openprocess/learn/__init__.py +23 -0
- openprocess/learn/answers.py +188 -0
- openprocess/learn/checks.py +953 -0
- openprocess/learn/computed.py +1180 -0
- openprocess/learn/context.py +145 -0
- openprocess/learn/exam.py +169 -0
- openprocess/learn/exercise-packs.md +325 -0
- openprocess/learn/importer.py +216 -0
- openprocess/learn/notation.py +474 -0
- openprocess/learn/pack.py +511 -0
- openprocess/learn/sheet.py +296 -0
- openprocess/mining/__init__.py +73 -0
- openprocess/mining/analysis.py +689 -0
- openprocess/mining/columns.py +282 -0
- openprocess/mining/compare_nets.py +246 -0
- openprocess/mining/conformance/__init__.py +0 -0
- openprocess/mining/conformance/alignments.py +263 -0
- openprocess/mining/conformance/quality.py +145 -0
- openprocess/mining/conformance/token_replay.py +252 -0
- openprocess/mining/csv_import.py +222 -0
- openprocess/mining/definitions.py +584 -0
- openprocess/mining/dfg.py +187 -0
- openprocess/mining/discovery/__init__.py +0 -0
- openprocess/mining/discovery/alpha.py +168 -0
- openprocess/mining/discovery/heuristics.py +332 -0
- openprocess/mining/discovery/inductive.py +477 -0
- openprocess/mining/discovery/state_regions.py +62 -0
- openprocess/mining/filtering.py +237 -0
- openprocess/mining/footprint.py +183 -0
- openprocess/mining/invariants.py +191 -0
- openprocess/mining/layout.py +279 -0
- openprocess/mining/log.py +364 -0
- openprocess/mining/petrinet.py +354 -0
- openprocess/mining/playout.py +75 -0
- openprocess/mining/pm4py_bridge.py +82 -0
- openprocess/mining/pnml.py +223 -0
- openprocess/mining/processtree.py +216 -0
- openprocess/mining/regions.py +476 -0
- openprocess/mining/stats.py +160 -0
- openprocess/mining/structure.py +374 -0
- openprocess/mining/transition_system.py +409 -0
- openprocess/mining/xes.py +399 -0
- openprocess/ml/__init__.py +0 -0
- openprocess/ml/ast_nodes.py +332 -0
- openprocess/ml/builtins.py +364 -0
- openprocess/ml/colorsets.py +522 -0
- openprocess/ml/errors.py +60 -0
- openprocess/ml/evaluator.py +754 -0
- openprocess/ml/lexer.py +277 -0
- openprocess/ml/multiset.py +417 -0
- openprocess/ml/parser.py +737 -0
- openprocess/ml/values.py +319 -0
- openprocess/model/__init__.py +0 -0
- openprocess/model/declarations.py +617 -0
- openprocess/model/examples.py +98 -0
- openprocess/model/net.py +701 -0
- openprocess/model/plain.py +192 -0
- openprocess/references.py +280 -0
- openprocess/sim/__init__.py +0 -0
- openprocess/sim/binding.py +620 -0
- openprocess/sim/export.py +66 -0
- openprocess/sim/simulator.py +315 -0
- openprocess/teaching/__init__.py +4 -0
- openprocess/teaching/answers.py +4 -0
- openprocess/teaching/checks.py +5 -0
- openprocess/teaching/pack.py +4 -0
- openprocess/teaching/sheet.py +4 -0
- openprocess-0.7.0.dist-info/METADATA +927 -0
- openprocess-0.7.0.dist-info/RECORD +173 -0
- openprocess-0.7.0.dist-info/WHEEL +5 -0
- openprocess-0.7.0.dist-info/entry_points.txt +6 -0
- openprocess-0.7.0.dist-info/licenses/LICENSE +21 -0
- openprocess-0.7.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
"""Importing event logs from CSV files.
|
|
2
|
+
|
|
3
|
+
A CSV event log has one row per event. Three columns are essential:
|
|
4
|
+
|
|
5
|
+
* **case id** -- which process instance the event belongs to,
|
|
6
|
+
* **activity** -- what happened,
|
|
7
|
+
* **timestamp** -- when (optional, but needed to order events and for any
|
|
8
|
+
time-based analysis).
|
|
9
|
+
|
|
10
|
+
Everything else becomes an ordinary event attribute.
|
|
11
|
+
|
|
12
|
+
Because column names differ per source system, :func:`guess_mapping` proposes
|
|
13
|
+
a mapping from common names (``case:concept:name``, ``Case ID``, ``Activity``,
|
|
14
|
+
``Timestamp``, ...), and the GUI lets the user correct it before importing.
|
|
15
|
+
|
|
16
|
+
Ordering: rows of one case are sorted by timestamp. The sort is *stable*, so
|
|
17
|
+
events with equal (or missing) timestamps keep their file order -- which is
|
|
18
|
+
the only other evidence of order a CSV contains.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import csv
|
|
24
|
+
from dataclasses import dataclass
|
|
25
|
+
from datetime import datetime, timezone
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
|
|
28
|
+
from .log import KEY_LIFECYCLE, KEY_NAME, KEY_RESOURCE, KEY_TIME, Event, EventLog, Trace
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass
|
|
32
|
+
class ColumnMapping:
|
|
33
|
+
"""Which CSV column plays which role. ``None`` means "not present"."""
|
|
34
|
+
|
|
35
|
+
case: str
|
|
36
|
+
activity: str
|
|
37
|
+
timestamp: str | None = None
|
|
38
|
+
resource: str | None = None
|
|
39
|
+
lifecycle: str | None = None
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
_CANDIDATES = {
|
|
43
|
+
"case": ["case:concept:name", "case id", "caseid", "case_id", "case", "trace", "case:id"],
|
|
44
|
+
"activity": ["concept:name", "activity", "activity name", "event", "task", "action"],
|
|
45
|
+
"timestamp": ["time:timestamp", "timestamp", "complete timestamp", "time", "date",
|
|
46
|
+
"end", "end time", "completetime"],
|
|
47
|
+
"resource": ["org:resource", "resource", "user", "performer", "agent"],
|
|
48
|
+
"lifecycle": ["lifecycle:transition", "lifecycle", "transition", "event type"],
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _encoding(path: str | Path) -> str:
|
|
53
|
+
"""UTF-8 (with or without Excel's byte-order mark), else Windows' cp1252.
|
|
54
|
+
|
|
55
|
+
Excel on Windows saves "CSV" in the system's code page, so a log with
|
|
56
|
+
"café" or "Müller" in it is not valid UTF-8; cp1252 reads every byte.
|
|
57
|
+
"""
|
|
58
|
+
try:
|
|
59
|
+
with open(path, encoding="utf-8-sig") as handle:
|
|
60
|
+
while handle.read(1 << 20):
|
|
61
|
+
pass
|
|
62
|
+
return "utf-8-sig"
|
|
63
|
+
except UnicodeDecodeError:
|
|
64
|
+
return "cp1252"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def sniff(path: str | Path, sample_size: int = 64_000) -> tuple[csv.Dialect, list[str]]:
|
|
68
|
+
"""Detect the delimiter and read the header row."""
|
|
69
|
+
with open(path, newline="", encoding=_encoding(path), errors="replace") as handle:
|
|
70
|
+
sample = handle.read(sample_size)
|
|
71
|
+
try:
|
|
72
|
+
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t|")
|
|
73
|
+
except csv.Error:
|
|
74
|
+
dialect = csv.excel # fall back to plain commas
|
|
75
|
+
header = next(csv.reader(sample.splitlines(), dialect), [])
|
|
76
|
+
return dialect, header
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def guess_mapping(header: list[str]) -> ColumnMapping | None:
|
|
80
|
+
"""Propose a mapping by matching column names, case-insensitively."""
|
|
81
|
+
lowered = {name.strip().lower(): name for name in header}
|
|
82
|
+
chosen: dict[str, str | None] = {}
|
|
83
|
+
for role, candidates in _CANDIDATES.items():
|
|
84
|
+
chosen[role] = next((lowered[c] for c in candidates if c in lowered), None)
|
|
85
|
+
if chosen["case"] is None or chosen["activity"] is None:
|
|
86
|
+
return None
|
|
87
|
+
return ColumnMapping(**chosen) # type: ignore[arg-type]
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
_DATE_FORMATS = [
|
|
91
|
+
"%Y-%m-%d %H:%M:%S", "%Y-%m-%d %H:%M", "%Y-%m-%d",
|
|
92
|
+
"%d-%m-%Y %H:%M:%S.%f", "%d-%m-%Y %H:%M:%S", "%d-%m-%Y %H:%M", "%d-%m-%Y",
|
|
93
|
+
"%d/%m/%Y %H:%M:%S.%f", "%d/%m/%Y %H:%M:%S", "%d/%m/%Y %H:%M", "%d/%m/%Y",
|
|
94
|
+
"%Y/%m/%d %H:%M:%S.%f", "%Y/%m/%d %H:%M:%S", "%Y/%m/%d %H:%M", "%Y/%m/%d",
|
|
95
|
+
"%d.%m.%Y %H:%M:%S.%f", "%d.%m.%Y %H:%M:%S", "%d.%m.%Y %H:%M", "%d.%m.%Y",
|
|
96
|
+
]
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def parse_timestamp(text: str) -> datetime | None:
|
|
100
|
+
"""Parse a timestamp in ISO 8601 or one of the common European forms.
|
|
101
|
+
|
|
102
|
+
Day-first is assumed for ``dd-mm-yyyy`` because that is the convention in
|
|
103
|
+
the Netherlands and most of Europe; ISO dates (year first) are
|
|
104
|
+
unambiguous anyway. Returns ``None`` rather than raising, so one odd
|
|
105
|
+
cell does not abort an import.
|
|
106
|
+
"""
|
|
107
|
+
value = text.strip()
|
|
108
|
+
if not value:
|
|
109
|
+
return None
|
|
110
|
+
try:
|
|
111
|
+
from .xes import parse_xes_date
|
|
112
|
+
return parse_xes_date(value.replace(" ", "T", 1) if "T" not in value and
|
|
113
|
+
len(value) > 10 and value[4] == "-" else value)
|
|
114
|
+
except ValueError:
|
|
115
|
+
pass
|
|
116
|
+
for fmt in _DATE_FORMATS:
|
|
117
|
+
try:
|
|
118
|
+
return datetime.strptime(value, fmt).replace(tzinfo=timezone.utc)
|
|
119
|
+
except ValueError:
|
|
120
|
+
continue
|
|
121
|
+
return None
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _typed(text: str):
|
|
125
|
+
"""Best-effort conversion of an ordinary cell to int or float."""
|
|
126
|
+
try:
|
|
127
|
+
return int(text)
|
|
128
|
+
except ValueError:
|
|
129
|
+
pass
|
|
130
|
+
try:
|
|
131
|
+
return float(text)
|
|
132
|
+
except ValueError:
|
|
133
|
+
return text
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def read_csv(path: str | Path, mapping: ColumnMapping | None = None) -> EventLog:
|
|
137
|
+
"""Read a CSV event log. Without a mapping, one is guessed from the header."""
|
|
138
|
+
dialect, header = sniff(path)
|
|
139
|
+
mapping = mapping or guess_mapping(header)
|
|
140
|
+
if mapping is None:
|
|
141
|
+
raise ValueError(
|
|
142
|
+
"Could not identify the case id and activity columns. "
|
|
143
|
+
f"Columns found: {', '.join(header)}")
|
|
144
|
+
|
|
145
|
+
cases: dict[str, list[tuple[int, Event]]] = {}
|
|
146
|
+
with open(path, newline="", encoding=_encoding(path), errors="replace") as handle:
|
|
147
|
+
for row_number, row in enumerate(csv.DictReader(handle, dialect=dialect)):
|
|
148
|
+
case_id = (row.get(mapping.case) or "").strip()
|
|
149
|
+
if not case_id:
|
|
150
|
+
continue
|
|
151
|
+
attributes: dict = {}
|
|
152
|
+
for column, cell in row.items():
|
|
153
|
+
if column is None or cell is None:
|
|
154
|
+
continue
|
|
155
|
+
if column == mapping.case:
|
|
156
|
+
continue
|
|
157
|
+
if column == mapping.activity:
|
|
158
|
+
attributes[KEY_NAME] = cell.strip()
|
|
159
|
+
elif column == mapping.timestamp:
|
|
160
|
+
stamp = parse_timestamp(cell)
|
|
161
|
+
if stamp is not None:
|
|
162
|
+
attributes[KEY_TIME] = stamp
|
|
163
|
+
elif column == mapping.resource:
|
|
164
|
+
attributes[KEY_RESOURCE] = cell.strip()
|
|
165
|
+
elif column == mapping.lifecycle:
|
|
166
|
+
attributes[KEY_LIFECYCLE] = cell.strip().lower()
|
|
167
|
+
elif cell != "":
|
|
168
|
+
attributes[column] = _typed(cell)
|
|
169
|
+
cases.setdefault(case_id, []).append((row_number, Event(attributes)))
|
|
170
|
+
|
|
171
|
+
log = EventLog(attributes={KEY_NAME: Path(path).stem}, source_path=str(path))
|
|
172
|
+
for case_id, rows in cases.items():
|
|
173
|
+
# Sort by time only if every event of the case has a timestamp;
|
|
174
|
+
# otherwise we cannot place the undated ones, and file order is the
|
|
175
|
+
# more honest ordering. Ties keep file order (row number).
|
|
176
|
+
if all(event.timestamp is not None for _, event in rows):
|
|
177
|
+
rows.sort(key=lambda item: (item[1].timestamp, item[0]))
|
|
178
|
+
log.traces.append(Trace({KEY_NAME: case_id}, [event for _, event in rows]))
|
|
179
|
+
return log
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
# ---------------------------------------------------------------------------
|
|
183
|
+
# Writing
|
|
184
|
+
# ---------------------------------------------------------------------------
|
|
185
|
+
def write_csv(log: EventLog, path: str | Path, delimiter: str = ",") -> None:
|
|
186
|
+
"""Write one row per event, in the column names PM4Py and Disco use.
|
|
187
|
+
|
|
188
|
+
``case:concept:name`` (the case id), ``concept:name``, ``time:timestamp``
|
|
189
|
+
(ISO 8601), ``org:resource`` and ``lifecycle:transition`` come first, then
|
|
190
|
+
the other event attributes, then trace attributes as ``case:<name>``.
|
|
191
|
+
:func:`read_csv` recognises the standard columns when the file is opened
|
|
192
|
+
again.
|
|
193
|
+
"""
|
|
194
|
+
first = [KEY_NAME, KEY_TIME, KEY_RESOURCE, KEY_LIFECYCLE]
|
|
195
|
+
event_keys = sorted({key for trace in log for event in trace for key in event.attributes}
|
|
196
|
+
- set(first))
|
|
197
|
+
trace_keys = sorted({key for trace in log for key in trace.attributes} - {KEY_NAME})
|
|
198
|
+
present = [key for key in first
|
|
199
|
+
if any(key in event.attributes for trace in log for event in trace)]
|
|
200
|
+
header = ["case:concept:name"] + present + event_keys + [f"case:{k}" for k in trace_keys]
|
|
201
|
+
|
|
202
|
+
def cell(value) -> str:
|
|
203
|
+
if value is None:
|
|
204
|
+
return ""
|
|
205
|
+
if isinstance(value, datetime):
|
|
206
|
+
return value.isoformat()
|
|
207
|
+
if isinstance(value, bool):
|
|
208
|
+
return "true" if value else "false"
|
|
209
|
+
if isinstance(value, list):
|
|
210
|
+
return "|".join(str(item) for item in value)
|
|
211
|
+
return str(value)
|
|
212
|
+
|
|
213
|
+
with open(path, "w", newline="", encoding="utf-8") as handle:
|
|
214
|
+
writer = csv.writer(handle, delimiter=delimiter)
|
|
215
|
+
writer.writerow(header)
|
|
216
|
+
for trace in log:
|
|
217
|
+
case_values = [cell(trace.attributes.get(key)) for key in trace_keys]
|
|
218
|
+
for event in trace:
|
|
219
|
+
writer.writerow([trace.case_id]
|
|
220
|
+
+ [cell(event.attributes.get(key)) for key in present]
|
|
221
|
+
+ [cell(event.attributes.get(key)) for key in event_keys]
|
|
222
|
+
+ case_values)
|