openprocess 0.7.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cpnpy/__init__.py +46 -0
- openprocess/__init__.py +57 -0
- openprocess/analysis/__init__.py +0 -0
- openprocess/analysis/state_space.py +521 -0
- openprocess/analysis/state_space_process.py +251 -0
- openprocess/cli.py +742 -0
- openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/answer.pnml +31 -0
- openprocess/exercises/1 Petri nets/Exercise 1.1 Order handling/question.md +38 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/answer.pnml +27 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/net.pnml +29 -0
- openprocess/exercises/2 Soundness/Exercise 2.1 Spot the flaw/question.md +65 -0
- openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/log.txt +1 -0
- openprocess/exercises/3 Discovery/Exercise 3.1 The alpha-algorithm/question.md +67 -0
- openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/question.md +75 -0
- openprocess/exercises/4 Regions/Exercise 4.1 Regions of a transition system/ts.txt +5 -0
- openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/net.pnml +27 -0
- openprocess/exercises/5 Markings/Exercise 5.1 Markings and matrices/question.md +65 -0
- openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/log.txt +1 -0
- openprocess/exercises/6 Inductive Miner/Exercise 6.1 Cuts and trees/question.md +62 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/log.txt +1 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m1.pnml +36 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m2.pnml +28 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/m3.pnml +30 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/net.pnml +36 -0
- openprocess/exercises/7 Conformance/Exercise 7.1 Replay, alignments and workflows/question.md +69 -0
- openprocess/exercises/pack.md +14 -0
- openprocess/flow/__init__.py +50 -0
- openprocess/flow/box.py +466 -0
- openprocess/flow/boxes/__init__.py +7 -0
- openprocess/flow/boxes/check.py +119 -0
- openprocess/flow/boxes/compare.py +16 -0
- openprocess/flow/boxes/cpn.py +53 -0
- openprocess/flow/boxes/discover.py +124 -0
- openprocess/flow/boxes/filter.py +80 -0
- openprocess/flow/boxes/input.py +124 -0
- openprocess/flow/boxes/output.py +52 -0
- openprocess/flow/boxes/predict.py +186 -0
- openprocess/flow/boxes/science.py +159 -0
- openprocess/flow/boxes/sweeps.py +18 -0
- openprocess/flow/convert.py +187 -0
- openprocess/flow/datasets.py +198 -0
- openprocess/flow/explain.py +115 -0
- openprocess/flow/library.py +222 -0
- openprocess/flow/record.py +385 -0
- openprocess/flow/runner.py +357 -0
- openprocess/flow/sweep.py +92 -0
- openprocess/flow/types.py +290 -0
- openprocess/flow/workflow.py +628 -0
- openprocess/gui/__init__.py +0 -0
- openprocess/gui/app.py +90 -0
- openprocess/gui/arc_editing.py +295 -0
- openprocess/gui/canvas.py +1414 -0
- openprocess/gui/flow/__init__.py +8 -0
- openprocess/gui/flow/canvas.py +854 -0
- openprocess/gui/flow/page.py +972 -0
- openprocess/gui/flow/templates.py +131 -0
- openprocess/gui/flow/viewers.py +665 -0
- openprocess/gui/items.py +1275 -0
- openprocess/gui/learn/answer_boxes.py +978 -0
- openprocess/gui/learn/concealment.py +91 -0
- openprocess/gui/learn/mode.py +1181 -0
- openprocess/gui/panning.py +241 -0
- openprocess/gui/resources/openprocess-icon.png +0 -0
- openprocess/gui/studio/__init__.py +1 -0
- openprocess/gui/studio/__main__.py +3 -0
- openprocess/gui/studio/app.py +4031 -0
- openprocess/gui/studio/charts.py +115 -0
- openprocess/gui/studio/compare_page.py +487 -0
- openprocess/gui/studio/cpn_page.py +1858 -0
- openprocess/gui/studio/definition_view.py +284 -0
- openprocess/gui/studio/derivation_view.py +421 -0
- openprocess/gui/studio/documents.py +152 -0
- openprocess/gui/studio/dotted_chart.py +1401 -0
- openprocess/gui/studio/file_dialogs.py +143 -0
- openprocess/gui/studio/filter_dialog.py +247 -0
- openprocess/gui/studio/graph_builders.py +176 -0
- openprocess/gui/studio/graph_view.py +682 -0
- openprocess/gui/studio/instances.py +413 -0
- openprocess/gui/studio/log_editor.py +675 -0
- openprocess/gui/studio/log_page.py +800 -0
- openprocess/gui/studio/markdown_view.py +127 -0
- openprocess/gui/studio/mathtext.py +260 -0
- openprocess/gui/studio/ml_highlighter.py +75 -0
- openprocess/gui/studio/model_page.py +760 -0
- openprocess/gui/studio/net_comparison.py +124 -0
- openprocess/gui/studio/notes_overlay.py +275 -0
- openprocess/gui/studio/petri_page.py +844 -0
- openprocess/gui/studio/regions_view.py +502 -0
- openprocess/gui/studio/sidebar.py +149 -0
- openprocess/gui/studio/style.py +503 -0
- openprocess/gui/studio/tool_icons.py +134 -0
- openprocess/gui/studio/updates.py +439 -0
- openprocess/gui/studio/widgets.py +899 -0
- openprocess/gui/studio/workers.py +60 -0
- openprocess/gui/studio/workspace.py +447 -0
- openprocess/gui/theme.py +394 -0
- openprocess/gui/tidy.py +86 -0
- openprocess/io/__init__.py +0 -0
- openprocess/io/cpn_reader.py +389 -0
- openprocess/io/cpn_writer.py +357 -0
- openprocess/learn/__init__.py +23 -0
- openprocess/learn/answers.py +188 -0
- openprocess/learn/checks.py +953 -0
- openprocess/learn/computed.py +1180 -0
- openprocess/learn/context.py +145 -0
- openprocess/learn/exam.py +169 -0
- openprocess/learn/exercise-packs.md +325 -0
- openprocess/learn/importer.py +216 -0
- openprocess/learn/notation.py +474 -0
- openprocess/learn/pack.py +511 -0
- openprocess/learn/sheet.py +296 -0
- openprocess/mining/__init__.py +73 -0
- openprocess/mining/analysis.py +689 -0
- openprocess/mining/columns.py +282 -0
- openprocess/mining/compare_nets.py +246 -0
- openprocess/mining/conformance/__init__.py +0 -0
- openprocess/mining/conformance/alignments.py +263 -0
- openprocess/mining/conformance/quality.py +145 -0
- openprocess/mining/conformance/token_replay.py +252 -0
- openprocess/mining/csv_import.py +222 -0
- openprocess/mining/definitions.py +584 -0
- openprocess/mining/dfg.py +187 -0
- openprocess/mining/discovery/__init__.py +0 -0
- openprocess/mining/discovery/alpha.py +168 -0
- openprocess/mining/discovery/heuristics.py +332 -0
- openprocess/mining/discovery/inductive.py +477 -0
- openprocess/mining/discovery/state_regions.py +62 -0
- openprocess/mining/filtering.py +237 -0
- openprocess/mining/footprint.py +183 -0
- openprocess/mining/invariants.py +191 -0
- openprocess/mining/layout.py +279 -0
- openprocess/mining/log.py +364 -0
- openprocess/mining/petrinet.py +354 -0
- openprocess/mining/playout.py +75 -0
- openprocess/mining/pm4py_bridge.py +82 -0
- openprocess/mining/pnml.py +223 -0
- openprocess/mining/processtree.py +216 -0
- openprocess/mining/regions.py +476 -0
- openprocess/mining/stats.py +160 -0
- openprocess/mining/structure.py +374 -0
- openprocess/mining/transition_system.py +409 -0
- openprocess/mining/xes.py +399 -0
- openprocess/ml/__init__.py +0 -0
- openprocess/ml/ast_nodes.py +332 -0
- openprocess/ml/builtins.py +364 -0
- openprocess/ml/colorsets.py +522 -0
- openprocess/ml/errors.py +60 -0
- openprocess/ml/evaluator.py +754 -0
- openprocess/ml/lexer.py +277 -0
- openprocess/ml/multiset.py +417 -0
- openprocess/ml/parser.py +737 -0
- openprocess/ml/values.py +319 -0
- openprocess/model/__init__.py +0 -0
- openprocess/model/declarations.py +617 -0
- openprocess/model/examples.py +98 -0
- openprocess/model/net.py +701 -0
- openprocess/model/plain.py +192 -0
- openprocess/references.py +280 -0
- openprocess/sim/__init__.py +0 -0
- openprocess/sim/binding.py +620 -0
- openprocess/sim/export.py +66 -0
- openprocess/sim/simulator.py +315 -0
- openprocess/teaching/__init__.py +4 -0
- openprocess/teaching/answers.py +4 -0
- openprocess/teaching/checks.py +5 -0
- openprocess/teaching/pack.py +4 -0
- openprocess/teaching/sheet.py +4 -0
- openprocess-0.7.0.dist-info/METADATA +927 -0
- openprocess-0.7.0.dist-info/RECORD +173 -0
- openprocess-0.7.0.dist-info/WHEEL +5 -0
- openprocess-0.7.0.dist-info/entry_points.txt +6 -0
- openprocess-0.7.0.dist-info/licenses/LICENSE +21 -0
- openprocess-0.7.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,399 @@
|
|
|
1
|
+
"""Reading and writing XES, the IEEE 1849 standard format for event logs.
|
|
2
|
+
|
|
3
|
+
What an XES file looks like
|
|
4
|
+
---------------------------
|
|
5
|
+
::
|
|
6
|
+
|
|
7
|
+
<log xes.version="1.0">
|
|
8
|
+
<extension name="Concept" prefix="concept" uri="..."/>
|
|
9
|
+
<global scope="event"> <string key="concept:name" value="UNKNOWN"/> </global>
|
|
10
|
+
<classifier name="Event Name" keys="concept:name"/>
|
|
11
|
+
<string key="concept:name" value="Random 50 passengers"/> <- log attribute
|
|
12
|
+
<trace>
|
|
13
|
+
<string key="concept:name" value="2050016"/> <- case id
|
|
14
|
+
<event>
|
|
15
|
+
<string key="concept:name" value="Move in"/>
|
|
16
|
+
<date key="time:timestamp" value="3924-10-11T09:00:00+02:00"/>
|
|
17
|
+
<int key="r" value="9"/>
|
|
18
|
+
</event>
|
|
19
|
+
...
|
|
20
|
+
</trace>
|
|
21
|
+
</log>
|
|
22
|
+
|
|
23
|
+
Every attribute element's *tag* is its type: ``string``, ``date``, ``int``,
|
|
24
|
+
``float``, ``boolean``, ``id``, ``list`` or ``container``. Attributes may nest
|
|
25
|
+
(an attribute can carry "meta-attributes"); we keep the value and ignore the
|
|
26
|
+
meta-attributes, except for ``list``/``container`` where the children *are*
|
|
27
|
+
the value.
|
|
28
|
+
|
|
29
|
+
How the reader works
|
|
30
|
+
--------------------
|
|
31
|
+
Logs can be hundreds of megabytes, so we do not build the whole XML tree.
|
|
32
|
+
:func:`xml.etree.ElementTree.iterparse` streams *end* events; when a
|
|
33
|
+
``<trace>`` closes we convert it to a :class:`Trace` and then ``clear()`` the
|
|
34
|
+
element so its memory is released. ``.xes.gz`` files are decompressed on the
|
|
35
|
+
fly.
|
|
36
|
+
|
|
37
|
+
Security: the standard-library expat parser does not fetch external entities,
|
|
38
|
+
so opening a file cannot make the program read local files or the network.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
from __future__ import annotations
|
|
42
|
+
|
|
43
|
+
import gzip
|
|
44
|
+
import io
|
|
45
|
+
from datetime import datetime, timezone
|
|
46
|
+
from pathlib import Path
|
|
47
|
+
from typing import Any, BinaryIO
|
|
48
|
+
from xml.etree import ElementTree as ET
|
|
49
|
+
from xml.sax.saxutils import quoteattr
|
|
50
|
+
|
|
51
|
+
from .log import KEY_LIFECYCLE, KEY_NAME, KEY_RESOURCE, KEY_TIME, Classifier, Event, EventLog, Trace
|
|
52
|
+
|
|
53
|
+
_ATTRIBUTE_TAGS = {"string", "date", "int", "float", "boolean", "id", "list", "container"}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
# ---------------------------------------------------------------------------
|
|
57
|
+
# Value parsing
|
|
58
|
+
# ---------------------------------------------------------------------------
|
|
59
|
+
def parse_xes_date(text: str) -> datetime:
|
|
60
|
+
"""Parse an XES (xs:dateTime) timestamp into an aware ``datetime``.
|
|
61
|
+
|
|
62
|
+
``datetime.fromisoformat`` handles most forms since Python 3.11; we
|
|
63
|
+
normalise the two things older producers emit that it rejects: a trailing
|
|
64
|
+
``Z`` and fractional seconds that are not 3 or 6 digits long.
|
|
65
|
+
"""
|
|
66
|
+
value = text.strip()
|
|
67
|
+
if value.endswith("Z"):
|
|
68
|
+
value = value[:-1] + "+00:00"
|
|
69
|
+
# Normalise fractional seconds to 6 digits.
|
|
70
|
+
if "." in value:
|
|
71
|
+
head, _, rest = value.partition(".")
|
|
72
|
+
digits = ""
|
|
73
|
+
while rest and rest[0].isdigit():
|
|
74
|
+
digits, rest = digits + rest[0], rest[1:]
|
|
75
|
+
value = f"{head}.{(digits + '000000')[:6]}{rest}"
|
|
76
|
+
parsed = datetime.fromisoformat(value)
|
|
77
|
+
# A timestamp without offset is interpreted as UTC so that all timestamps
|
|
78
|
+
# in a log are comparable (mixing naive and aware datetimes raises).
|
|
79
|
+
if parsed.tzinfo is None:
|
|
80
|
+
parsed = parsed.replace(tzinfo=timezone.utc)
|
|
81
|
+
return parsed
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _convert(element: ET.Element) -> Any:
|
|
85
|
+
"""Turn one attribute element into a Python value."""
|
|
86
|
+
tag = element.tag
|
|
87
|
+
raw = element.get("value")
|
|
88
|
+
try:
|
|
89
|
+
if tag == "string" or tag == "id":
|
|
90
|
+
return raw if raw is not None else ""
|
|
91
|
+
if tag == "int":
|
|
92
|
+
return int(raw)
|
|
93
|
+
if tag == "float":
|
|
94
|
+
return float(raw)
|
|
95
|
+
if tag == "boolean":
|
|
96
|
+
return str(raw).strip().lower() == "true"
|
|
97
|
+
if tag == "date":
|
|
98
|
+
return parse_xes_date(raw)
|
|
99
|
+
if tag == "list":
|
|
100
|
+
# A list's items live inside a <values> child (XES 2.0) or,
|
|
101
|
+
# in older files, directly as children.
|
|
102
|
+
holder = element.find("values")
|
|
103
|
+
children = list(holder) if holder is not None else list(element)
|
|
104
|
+
return [_convert(child) for child in children if child.tag in _ATTRIBUTE_TAGS]
|
|
105
|
+
if tag == "container":
|
|
106
|
+
return {child.get("key"): _convert(child) for child in element
|
|
107
|
+
if child.tag in _ATTRIBUTE_TAGS}
|
|
108
|
+
except (TypeError, ValueError):
|
|
109
|
+
# A malformed value is kept as text rather than aborting the import:
|
|
110
|
+
# one bad row should not cost the user the whole log.
|
|
111
|
+
return raw
|
|
112
|
+
return raw
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _attributes_of(element: ET.Element) -> dict[str, Any]:
|
|
116
|
+
"""The direct attribute children of a log, trace or event element."""
|
|
117
|
+
return {child.get("key"): _convert(child)
|
|
118
|
+
for child in element if child.tag in _ATTRIBUTE_TAGS}
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _strip_namespace(tag: str) -> str:
|
|
122
|
+
"""Some producers put XES in an XML namespace; we only care about names."""
|
|
123
|
+
return tag.rsplit("}", 1)[-1] if "}" in tag else tag
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
# ---------------------------------------------------------------------------
|
|
127
|
+
# Reading
|
|
128
|
+
# ---------------------------------------------------------------------------
|
|
129
|
+
def _open(source: str | Path | BinaryIO) -> BinaryIO:
|
|
130
|
+
if hasattr(source, "read"):
|
|
131
|
+
return source # type: ignore[return-value]
|
|
132
|
+
path = Path(source)
|
|
133
|
+
with open(path, "rb") as handle:
|
|
134
|
+
magic = handle.read(2)
|
|
135
|
+
if magic == b"\x1f\x8b": # gzip signature, whatever the extension says
|
|
136
|
+
return gzip.open(path, "rb")
|
|
137
|
+
return open(path, "rb")
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
#: Files larger than this (bytes on disk) are read into columns unless told otherwise.
|
|
141
|
+
COLUMNAR_FROM = 20 * 1024 * 1024
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def read_xes(source: str | Path | BinaryIO, progress=None, columnar: bool | None = None) -> EventLog:
|
|
145
|
+
"""Read an XES (or gzip-compressed XES) file into an :class:`EventLog`.
|
|
146
|
+
|
|
147
|
+
``progress`` is an optional callable receiving the number of traces read
|
|
148
|
+
so far; the GUI uses it to update a progress indicator. ``columnar``
|
|
149
|
+
keeps the events in arrays rather than objects (see
|
|
150
|
+
:mod:`.columns`); by default that is done for files above
|
|
151
|
+
:data:`COLUMNAR_FROM`.
|
|
152
|
+
"""
|
|
153
|
+
if columnar is None:
|
|
154
|
+
try:
|
|
155
|
+
columnar = isinstance(source, (str, Path)) and Path(source).stat().st_size >= COLUMNAR_FROM
|
|
156
|
+
except OSError:
|
|
157
|
+
columnar = False
|
|
158
|
+
if columnar:
|
|
159
|
+
return _read_columnar(source, progress)
|
|
160
|
+
log = EventLog()
|
|
161
|
+
if isinstance(source, (str, Path)):
|
|
162
|
+
log.source_path = str(source)
|
|
163
|
+
|
|
164
|
+
stream = _open(source)
|
|
165
|
+
try:
|
|
166
|
+
# depth tracks where we are: 1 = inside <log>, 2 = inside <trace> etc.
|
|
167
|
+
# We need it to tell a log-level attribute from a trace-level one.
|
|
168
|
+
stack: list[str] = []
|
|
169
|
+
for kind, element in ET.iterparse(stream, events=("start", "end")):
|
|
170
|
+
tag = _strip_namespace(element.tag)
|
|
171
|
+
if kind == "start":
|
|
172
|
+
stack.append(tag)
|
|
173
|
+
continue
|
|
174
|
+
|
|
175
|
+
stack.pop()
|
|
176
|
+
element.tag = tag
|
|
177
|
+
parent = stack[-1] if stack else None
|
|
178
|
+
|
|
179
|
+
if tag == "trace":
|
|
180
|
+
trace = Trace(attributes=_attributes_of(element))
|
|
181
|
+
for child in element:
|
|
182
|
+
if _strip_namespace(child.tag) == "event":
|
|
183
|
+
trace.events.append(Event(_attributes_of(child)))
|
|
184
|
+
log.traces.append(trace)
|
|
185
|
+
element.clear() # release memory: the whole point of streaming
|
|
186
|
+
if progress is not None and len(log.traces) % 500 == 0:
|
|
187
|
+
progress(len(log.traces))
|
|
188
|
+
elif tag == "event" and parent == "trace":
|
|
189
|
+
# Converted when the enclosing trace closes; normalise child
|
|
190
|
+
# tags now so _attributes_of recognises them.
|
|
191
|
+
for child in element.iter():
|
|
192
|
+
child.tag = _strip_namespace(child.tag)
|
|
193
|
+
elif tag in _ATTRIBUTE_TAGS and parent in ("event", "trace", "list",
|
|
194
|
+
"values", "container"):
|
|
195
|
+
element.tag = tag
|
|
196
|
+
elif tag == "extension" and parent == "log":
|
|
197
|
+
log.extensions.append(dict(element.attrib))
|
|
198
|
+
elif tag == "global" and parent == "log":
|
|
199
|
+
for child in element:
|
|
200
|
+
child.tag = _strip_namespace(child.tag)
|
|
201
|
+
scope = element.get("scope", "event")
|
|
202
|
+
target = (log.global_trace_attributes if scope == "trace"
|
|
203
|
+
else log.global_event_attributes)
|
|
204
|
+
target.update(_attributes_of(element))
|
|
205
|
+
elif tag == "classifier" and parent == "log":
|
|
206
|
+
keys = tuple(_split_classifier_keys(element.get("keys", "")))
|
|
207
|
+
if keys:
|
|
208
|
+
log.declared_classifiers.append(
|
|
209
|
+
Classifier(element.get("name") or " + ".join(keys), keys))
|
|
210
|
+
elif tag in _ATTRIBUTE_TAGS and parent == "log":
|
|
211
|
+
log.attributes[element.get("key")] = _convert(element)
|
|
212
|
+
finally:
|
|
213
|
+
if not hasattr(source, "read"):
|
|
214
|
+
stream.close()
|
|
215
|
+
|
|
216
|
+
_apply_event_defaults(log)
|
|
217
|
+
return log
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _read_columnar(source: str | Path | BinaryIO, progress=None) -> EventLog:
|
|
221
|
+
"""The same parse, but events go into a :class:`~.columns.ColumnStore`."""
|
|
222
|
+
from .columns import ColumnStore
|
|
223
|
+
store = ColumnStore()
|
|
224
|
+
fields: dict = {"attributes": {}, "extensions": [], "global_trace_attributes": {},
|
|
225
|
+
"global_event_attributes": {}, "declared_classifiers": []}
|
|
226
|
+
if isinstance(source, (str, Path)):
|
|
227
|
+
fields["source_path"] = str(source)
|
|
228
|
+
stream = _open(source)
|
|
229
|
+
try:
|
|
230
|
+
stack: list[str] = []
|
|
231
|
+
for kind, element in ET.iterparse(stream, events=("start", "end")):
|
|
232
|
+
tag = _strip_namespace(element.tag)
|
|
233
|
+
if kind == "start":
|
|
234
|
+
stack.append(tag)
|
|
235
|
+
continue
|
|
236
|
+
stack.pop()
|
|
237
|
+
element.tag = tag
|
|
238
|
+
parent = stack[-1] if stack else None
|
|
239
|
+
if tag == "event" and parent == "trace":
|
|
240
|
+
activity = stamp = lifecycle = resource = None
|
|
241
|
+
extras, order = [], []
|
|
242
|
+
for child in element:
|
|
243
|
+
child.tag = _strip_namespace(child.tag)
|
|
244
|
+
key = child.get("key")
|
|
245
|
+
order.append(key)
|
|
246
|
+
if key == KEY_NAME:
|
|
247
|
+
activity = child.get("value", "")
|
|
248
|
+
elif key == KEY_TIME:
|
|
249
|
+
try:
|
|
250
|
+
stamp = parse_xes_date(child.get("value", ""))
|
|
251
|
+
except (TypeError, ValueError):
|
|
252
|
+
stamp = None
|
|
253
|
+
elif key == KEY_LIFECYCLE:
|
|
254
|
+
lifecycle = child.get("value", "")
|
|
255
|
+
elif key == KEY_RESOURCE:
|
|
256
|
+
resource = child.get("value", "")
|
|
257
|
+
elif child.tag in _ATTRIBUTE_TAGS:
|
|
258
|
+
for inner in child.iter():
|
|
259
|
+
inner.tag = _strip_namespace(inner.tag)
|
|
260
|
+
extras.append((key, _convert(child)))
|
|
261
|
+
store.add(activity, stamp, lifecycle, resource, extras, order)
|
|
262
|
+
elif tag == "trace":
|
|
263
|
+
for child in element.iter():
|
|
264
|
+
child.tag = _strip_namespace(child.tag)
|
|
265
|
+
store.end_case(_attributes_of(element))
|
|
266
|
+
element.clear()
|
|
267
|
+
if progress is not None and store.case_count % 500 == 0:
|
|
268
|
+
progress(store.case_count)
|
|
269
|
+
elif tag in _ATTRIBUTE_TAGS and parent in ("event", "trace", "list", "values", "container"):
|
|
270
|
+
element.tag = tag
|
|
271
|
+
elif tag == "extension" and parent == "log":
|
|
272
|
+
fields["extensions"].append(dict(element.attrib))
|
|
273
|
+
elif tag == "global" and parent == "log":
|
|
274
|
+
for child in element:
|
|
275
|
+
child.tag = _strip_namespace(child.tag)
|
|
276
|
+
scope = element.get("scope", "event")
|
|
277
|
+
target = fields["global_trace_attributes"] if scope == "trace" else fields["global_event_attributes"]
|
|
278
|
+
target.update(_attributes_of(element))
|
|
279
|
+
elif tag == "classifier" and parent == "log":
|
|
280
|
+
keys = tuple(_split_classifier_keys(element.get("keys", "")))
|
|
281
|
+
if keys:
|
|
282
|
+
fields["declared_classifiers"].append(Classifier(element.get("name") or " + ".join(keys), keys))
|
|
283
|
+
elif tag in _ATTRIBUTE_TAGS and parent == "log":
|
|
284
|
+
fields["attributes"][element.get("key")] = _convert(element)
|
|
285
|
+
finally:
|
|
286
|
+
if not hasattr(source, "read"):
|
|
287
|
+
stream.close()
|
|
288
|
+
return EventLog.from_columns(store, **fields)
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _split_classifier_keys(text: str) -> list[str]:
|
|
292
|
+
"""Classifier keys are space-separated; keys containing spaces are quoted."""
|
|
293
|
+
keys, current, quoted = [], "", False
|
|
294
|
+
for char in text:
|
|
295
|
+
if char == "'":
|
|
296
|
+
quoted = not quoted
|
|
297
|
+
elif char == " " and not quoted:
|
|
298
|
+
if current:
|
|
299
|
+
keys.append(current)
|
|
300
|
+
current = ""
|
|
301
|
+
else:
|
|
302
|
+
current += char
|
|
303
|
+
if current:
|
|
304
|
+
keys.append(current)
|
|
305
|
+
return keys
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def _apply_event_defaults(log: EventLog) -> None:
|
|
309
|
+
"""Nothing to fill in -- globals are *declarations*, not defaults.
|
|
310
|
+
|
|
311
|
+
The XES standard says a global attribute is guaranteed to be present on
|
|
312
|
+
every event; producers that declare globals therefore already write them.
|
|
313
|
+
We deliberately do **not** copy global values ("UNKNOWN", 1970-01-01) into
|
|
314
|
+
events that lack them, because doing so would invent data.
|
|
315
|
+
"""
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def read_xes_string(text: str) -> EventLog:
|
|
319
|
+
"""Parse XES from a string (used by tests)."""
|
|
320
|
+
return read_xes(io.BytesIO(text.encode("utf-8")))
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
# ---------------------------------------------------------------------------
|
|
324
|
+
# Writing
|
|
325
|
+
# ---------------------------------------------------------------------------
|
|
326
|
+
def _format_value(key: str, value: Any, indent: str) -> str:
|
|
327
|
+
"""Serialise one attribute as an XES element."""
|
|
328
|
+
k = quoteattr(str(key))
|
|
329
|
+
if isinstance(value, bool): # bool before int: bool is a subclass of int
|
|
330
|
+
return f'{indent}<boolean key={k} value="{str(value).lower()}"/>\n'
|
|
331
|
+
if isinstance(value, int):
|
|
332
|
+
return f'{indent}<int key={k} value="{value}"/>\n'
|
|
333
|
+
if isinstance(value, float):
|
|
334
|
+
return f'{indent}<float key={k} value="{value!r}"/>\n'
|
|
335
|
+
if isinstance(value, datetime):
|
|
336
|
+
stamp = value if value.tzinfo else value.replace(tzinfo=timezone.utc)
|
|
337
|
+
return f'{indent}<date key={k} value="{stamp.isoformat(timespec="milliseconds")}"/>\n'
|
|
338
|
+
if isinstance(value, list):
|
|
339
|
+
inner = "".join(_format_value(key, item, indent + "\t\t") for item in value)
|
|
340
|
+
return (f"{indent}<list key={k}>\n{indent}\t<values>\n{inner}"
|
|
341
|
+
f"{indent}\t</values>\n{indent}</list>\n")
|
|
342
|
+
if isinstance(value, dict):
|
|
343
|
+
inner = "".join(_format_value(sub, item, indent + "\t") for sub, item in value.items())
|
|
344
|
+
return f"{indent}<container key={k}>\n{inner}{indent}</container>\n"
|
|
345
|
+
return f"{indent}<string key={k} value={quoteattr(str(value))}/>\n"
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
_STANDARD_EXTENSIONS = [
|
|
349
|
+
{"name": "Concept", "prefix": "concept", "uri": "http://www.xes-standard.org/concept.xesext"},
|
|
350
|
+
{"name": "Time", "prefix": "time", "uri": "http://www.xes-standard.org/time.xesext"},
|
|
351
|
+
{"name": "Lifecycle", "prefix": "lifecycle", "uri": "http://www.xes-standard.org/lifecycle.xesext"},
|
|
352
|
+
{"name": "Organizational", "prefix": "org", "uri": "http://www.xes-standard.org/org.xesext"},
|
|
353
|
+
]
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def xes_string(log: EventLog) -> str:
|
|
357
|
+
"""Serialise a log to an XES document string."""
|
|
358
|
+
out = io.StringIO()
|
|
359
|
+
out.write('<?xml version="1.0" encoding="UTF-8" ?>\n')
|
|
360
|
+
out.write('<!-- Written by OpenProcess -->\n')
|
|
361
|
+
out.write('<log xes.version="1.0" xes.features="nested-attributes">\n')
|
|
362
|
+
for ext in (log.extensions or _STANDARD_EXTENSIONS):
|
|
363
|
+
attrs = " ".join(f"{name}={quoteattr(str(value))}" for name, value in ext.items())
|
|
364
|
+
out.write(f"\t<extension {attrs}/>\n")
|
|
365
|
+
for scope, values in (("trace", log.global_trace_attributes),
|
|
366
|
+
("event", log.global_event_attributes)):
|
|
367
|
+
if values:
|
|
368
|
+
out.write(f'\t<global scope="{scope}">\n')
|
|
369
|
+
for key, value in values.items():
|
|
370
|
+
out.write(_format_value(key, value, "\t\t"))
|
|
371
|
+
out.write("\t</global>\n")
|
|
372
|
+
for classifier in log.declared_classifiers:
|
|
373
|
+
keys = " ".join(f"'{k}'" if " " in k else k for k in classifier.keys)
|
|
374
|
+
out.write(f"\t<classifier name={quoteattr(classifier.name)} keys={quoteattr(keys)}/>\n")
|
|
375
|
+
for key, value in log.attributes.items():
|
|
376
|
+
out.write(_format_value(key, value, "\t"))
|
|
377
|
+
for trace in log.traces:
|
|
378
|
+
out.write("\t<trace>\n")
|
|
379
|
+
for key, value in trace.attributes.items():
|
|
380
|
+
out.write(_format_value(key, value, "\t\t"))
|
|
381
|
+
for event in trace.events:
|
|
382
|
+
out.write("\t\t<event>\n")
|
|
383
|
+
for key, value in event.attributes.items():
|
|
384
|
+
out.write(_format_value(key, value, "\t\t\t"))
|
|
385
|
+
out.write("\t\t</event>\n")
|
|
386
|
+
out.write("\t</trace>\n")
|
|
387
|
+
out.write("</log>\n")
|
|
388
|
+
return out.getvalue()
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def write_xes(log: EventLog, path: str | Path) -> None:
|
|
392
|
+
"""Write a log to ``path``; a ``.gz`` suffix produces a compressed file."""
|
|
393
|
+
text = xes_string(log).encode("utf-8")
|
|
394
|
+
path = Path(path)
|
|
395
|
+
if path.suffix == ".gz":
|
|
396
|
+
with gzip.open(path, "wb") as handle:
|
|
397
|
+
handle.write(text)
|
|
398
|
+
else:
|
|
399
|
+
path.write_bytes(text)
|
|
File without changes
|