chronos4pm 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
chronos4pm/__init__.py ADDED
@@ -0,0 +1,46 @@
1
+ """Zero-shot process mining with native PM4Py inputs and an overridable model."""
2
+
3
+ from .api import (
4
+ analyse_completeness,
5
+ analyse_entropy,
6
+ analyse_variety,
7
+ discover_dfg,
8
+ discover_dfg_edges,
9
+ discover_footprints,
10
+ discover_performance_dfg,
11
+ discover_temporal_profile,
12
+ get_end_activities,
13
+ get_event_attribute_values,
14
+ get_performance_bottlenecks,
15
+ get_start_activities,
16
+ get_variants_as_tuples,
17
+ predict_next_activity,
18
+ predict_next_event_time,
19
+ predict_remaining_time,
20
+ )
21
+ from .configuration import CovariateSet, Footprint, ModelConfig
22
+
23
+ MODEL: ModelConfig = ModelConfig()
24
+
25
+ __all__: list[str] = [
26
+ "MODEL",
27
+ "ModelConfig",
28
+ "CovariateSet",
29
+ "Footprint",
30
+ "analyse_completeness",
31
+ "analyse_entropy",
32
+ "analyse_variety",
33
+ "discover_dfg",
34
+ "discover_dfg_edges",
35
+ "discover_footprints",
36
+ "discover_performance_dfg",
37
+ "discover_temporal_profile",
38
+ "get_end_activities",
39
+ "get_event_attribute_values",
40
+ "get_performance_bottlenecks",
41
+ "get_start_activities",
42
+ "get_variants_as_tuples",
43
+ "predict_next_activity",
44
+ "predict_next_event_time",
45
+ "predict_remaining_time",
46
+ ]
chronos4pm/api.py ADDED
@@ -0,0 +1,326 @@
1
+ """PM4Py-style functions returning forecasts over context-supported elements."""
2
+
3
+ from itertools import product
4
+ from typing import Unpack, cast
5
+
6
+ import numpy as np
7
+ from pm4py.objects.log.obj import Trace
8
+
9
+ from .configuration import (
10
+ ACTIVITY_KEY,
11
+ PRESENCE_THRESHOLD,
12
+ RELATIONS,
13
+ CovariateSet,
14
+ Edge,
15
+ Footprint,
16
+ ForecastOptions,
17
+ Key,
18
+ Log,
19
+ PrefixOptions,
20
+ )
21
+ from .predictive import predict
22
+ from .tasks import aggregate
23
+
24
+
25
+ def discover_dfg(
26
+ log: Log,
27
+ *,
28
+ covariates: CovariateSet = "target_only",
29
+ **options: Unpack[ForecastOptions],
30
+ ) -> tuple[dict[Edge, float], dict[str, float], dict[str, float]]:
31
+ """Forecast directly-follows, start and end distributions for the next window.
32
+
33
+ :param log: Completed historical cases (EventLog or DataFrame).
34
+ :param covariates: Automatic target_only, selected or global covariate set.
35
+ :param options: Window count, origin, PM4Py keys and past/future covariates.
36
+ :return: PM4Py-style (dfg, start_activities, end_activities), with probabilities.
37
+ """
38
+ return (
39
+ cast(dict[Edge, float], aggregate(log, "dfg", covariates, options)),
40
+ get_start_activities(log, covariates=covariates, **options),
41
+ get_end_activities(log, covariates=covariates, **options),
42
+ )
43
+
44
+
45
+ def discover_dfg_edges(
46
+ log: Log,
47
+ *,
48
+ covariates: CovariateSet = "target_only",
49
+ **options: Unpack[ForecastOptions],
50
+ ) -> set[Edge]:
51
+ """Forecast edge presence from binary histories using the benchmark threshold.
52
+
53
+ :param log: Completed historical cases.
54
+ :param covariates: Automatic covariate set.
55
+ :param options: Shared aggregate forecast parameters.
56
+ :return: Set of ordered activity pairs with scores at least 0.5.
57
+ """
58
+ scores: dict[Key, float] = aggregate(log, "edges", covariates, options)
59
+ return {
60
+ cast(Edge, edge)
61
+ for edge, score in scores.items()
62
+ if score >= PRESENCE_THRESHOLD
63
+ }
64
+
65
+
66
+ def get_start_activities(
67
+ log: Log,
68
+ *,
69
+ covariates: CovariateSet = "target_only",
70
+ **options: Unpack[ForecastOptions],
71
+ ) -> dict[str, float]:
72
+ """Forecast the next-window start-activity distribution.
73
+
74
+ :param log: Completed historical cases.
75
+ :param covariates: Automatic covariate set.
76
+ :param options: Shared aggregate forecast parameters.
77
+ :return: Activity-to-probability dictionary.
78
+ """
79
+ return cast(dict[str, float], aggregate(log, "start", covariates, options))
80
+
81
+
82
+ def get_end_activities(
83
+ log: Log,
84
+ *,
85
+ covariates: CovariateSet = "target_only",
86
+ **options: Unpack[ForecastOptions],
87
+ ) -> dict[str, float]:
88
+ """Forecast the next-window end-activity distribution.
89
+
90
+ :param log: Completed historical cases.
91
+ :param covariates: Automatic covariate set.
92
+ :param options: Shared aggregate forecast parameters.
93
+ :return: Activity-to-probability dictionary.
94
+ """
95
+ return cast(dict[str, float], aggregate(log, "end", covariates, options))
96
+
97
+
98
+ def get_event_attribute_values(
99
+ log: Log,
100
+ attribute: str = ACTIVITY_KEY,
101
+ *,
102
+ covariates: CovariateSet = "target_only",
103
+ **options: Unpack[ForecastOptions],
104
+ ) -> dict[str, float]:
105
+ """Forecast event mass by activity (or another categorical event attribute).
106
+
107
+ :param log: Completed historical cases.
108
+ :param attribute: Attribute whose values form the target support.
109
+ :param covariates: Automatic covariate set.
110
+ :param options: Shared aggregate forecast parameters.
111
+ :return: Attribute-value-to-probability dictionary with string keys.
112
+ """
113
+ options["activity_key"] = attribute
114
+ return cast(dict[str, float], aggregate(log, "activity", covariates, options))
115
+
116
+
117
+ def get_variants_as_tuples(
118
+ log: Log,
119
+ *,
120
+ covariates: CovariateSet = "target_only",
121
+ **options: Unpack[ForecastOptions],
122
+ ) -> dict[tuple[str, ...], float]:
123
+ """Forecast probabilities of complete trace variants observed in context.
124
+
125
+ :param log: Completed historical cases.
126
+ :param covariates: Automatic covariate set.
127
+ :param options: Shared aggregate forecast parameters.
128
+ :return: Activity-tuple-to-probability dictionary.
129
+ """
130
+ return cast(
131
+ dict[tuple[str, ...], float], aggregate(log, "variants", covariates, options)
132
+ )
133
+
134
+
135
+ def discover_footprints(
136
+ log: Log,
137
+ *,
138
+ covariates: CovariateSet = "target_only",
139
+ **options: Unpack[ForecastOptions],
140
+ ) -> Footprint:
141
+ """Forecast the benchmark's activities, sequence and parallel footprint fields.
142
+
143
+ :param log: Completed historical cases.
144
+ :param covariates: Automatic covariate set.
145
+ :param options: Shared aggregate forecast parameters.
146
+ :return: PM4Py footprint dictionary containing the three forecast fields.
147
+ """
148
+ scores: dict[Key, float] = aggregate(log, "footprints", covariates, options)
149
+ activities: set[str] = {cast(tuple[str, str, str], key)[0] for key in scores}
150
+ result: Footprint = {"activities": activities, "sequence": set(), "parallel": set()}
151
+ for source, target in product(sorted(activities), repeat=2):
152
+ relation: str = RELATIONS[
153
+ int(
154
+ np.argmax(
155
+ [scores[(source, target, relation)] for relation in RELATIONS]
156
+ )
157
+ )
158
+ ]
159
+ if relation == "sequence":
160
+ result["sequence"].add((source, target))
161
+ elif relation == "parallel":
162
+ result["parallel"].add((source, target))
163
+ return result
164
+
165
+
166
+ def discover_performance_dfg(
167
+ log: Log,
168
+ *,
169
+ covariates: CovariateSet = "target_only",
170
+ **options: Unpack[ForecastOptions],
171
+ ) -> tuple[dict[Edge, float], dict[str, float], dict[str, float]]:
172
+ """Forecast mean directly-follows durations and start/end distributions.
173
+
174
+ :param log: Completed historical cases.
175
+ :param covariates: Automatic covariate set.
176
+ :param options: Shared aggregate forecast parameters.
177
+ :return: PM4Py-style tuple; edge values are raw median forecasts in seconds.
178
+ """
179
+ return (
180
+ cast(dict[Edge, float], aggregate(log, "performance", covariates, options)),
181
+ get_start_activities(log, covariates=covariates, **options),
182
+ get_end_activities(log, covariates=covariates, **options),
183
+ )
184
+
185
+
186
+ def get_performance_bottlenecks(
187
+ log: Log,
188
+ *,
189
+ covariates: CovariateSet = "target_only",
190
+ **options: Unpack[ForecastOptions],
191
+ ) -> list[tuple[Edge, float]]:
192
+ """Rank forecast mean edge durations from longest to shortest.
193
+
194
+ :param log: Completed historical cases.
195
+ :param covariates: Automatic covariate set.
196
+ :param options: Shared aggregate forecast parameters.
197
+ :return: (Edge, seconds) pairs with lexical tie-breaking.
198
+ """
199
+ scores: dict[Edge, float] = cast(
200
+ dict[Edge, float], aggregate(log, "bottlenecks", covariates, options)
201
+ )
202
+ return sorted(scores.items(), key=lambda item: (-item[1], item[0]))
203
+
204
+
205
+ def discover_temporal_profile(
206
+ log: Log,
207
+ *,
208
+ covariates: CovariateSet = "target_only",
209
+ **options: Unpack[ForecastOptions],
210
+ ) -> dict[Edge, tuple[float, float]]:
211
+ """Forecast each temporal relation's mean and standard deviation.
212
+
213
+ :param log: Completed historical cases.
214
+ :param covariates: Automatic covariate set.
215
+ :param options: Shared aggregate forecast parameters.
216
+ :return: PM4Py pair-to-(mean, stdev) dictionary in seconds.
217
+ """
218
+ scores: dict[Key, float] = aggregate(log, "temporal", covariates, options)
219
+ pairs: list[Edge] = [
220
+ cast(tuple[Edge, int], key)[0]
221
+ for key in scores
222
+ if cast(tuple[Edge, int], key)[1] == 0
223
+ ]
224
+ return {pair: (scores[(pair, 0)], scores[(pair, 1)]) for pair in pairs}
225
+
226
+
227
+ def analyse_entropy(
228
+ log: Log,
229
+ *,
230
+ covariates: CovariateSet = "target_only",
231
+ **options: Unpack[ForecastOptions],
232
+ ) -> float:
233
+ """Forecast the EBI trace-distribution entropy in bits.
234
+
235
+ :param log: Completed historical cases.
236
+ :param covariates: Automatic covariate set.
237
+ :param options: Shared aggregate forecast parameters.
238
+ :return: Raw next-window scalar forecast.
239
+ """
240
+ return aggregate(log, "entropy", covariates, options)["value"]
241
+
242
+
243
+ def analyse_variety(
244
+ log: Log,
245
+ *,
246
+ covariates: CovariateSet = "target_only",
247
+ **options: Unpack[ForecastOptions],
248
+ ) -> float:
249
+ """Forecast EBI's trace-variety statistic.
250
+
251
+ :param log: Completed historical cases.
252
+ :param covariates: Automatic covariate set.
253
+ :param options: Shared aggregate forecast parameters.
254
+ :return: Raw next-window scalar forecast in EBI units.
255
+ """
256
+ return aggregate(log, "variety", covariates, options)["value"]
257
+
258
+
259
+ def analyse_completeness(
260
+ log: Log,
261
+ *,
262
+ covariates: CovariateSet = "target_only",
263
+ **options: Unpack[ForecastOptions],
264
+ ) -> float:
265
+ """Forecast EBI's event-log completeness statistic.
266
+
267
+ :param log: Completed historical cases.
268
+ :param covariates: Automatic covariate set.
269
+ :param options: Shared aggregate forecast parameters.
270
+ :return: Raw next-window scalar forecast in EBI units.
271
+ """
272
+ return aggregate(log, "completeness", covariates, options)["value"]
273
+
274
+
275
+ def predict_next_activity(
276
+ log: Log,
277
+ prefix: Trace | Log,
278
+ *,
279
+ covariates: CovariateSet = "target_only",
280
+ **options: Unpack[PrefixOptions],
281
+ ) -> str:
282
+ """Predict the next activity from the observed within-case one-hot history.
283
+
284
+ :param log: Completed cases ending before the current case starts.
285
+ :param prefix: One observed case prefix, with at least one event.
286
+ :param covariates: Automatic covariate set.
287
+ :param options: PM4Py keys, attribute allowlist and aligned past covariates.
288
+ :return: Activity label from context support; no end-of-case class is added.
289
+ """
290
+ return cast(str, predict(log, prefix, "next_activity", covariates, options))
291
+
292
+
293
+ def predict_next_event_time(
294
+ log: Log,
295
+ prefix: Trace | Log,
296
+ *,
297
+ covariates: CovariateSet = "target_only",
298
+ **options: Unpack[PrefixOptions],
299
+ ) -> float:
300
+ """Predict the duration from the current observed event to its successor.
301
+
302
+ :param log: Completed cases ending before the current case starts.
303
+ :param prefix: One observed case prefix, with at least two events.
304
+ :param covariates: Automatic covariate set.
305
+ :param options: PM4Py keys, attribute allowlist and aligned past/future series.
306
+ :return: Raw median duration in seconds, not an absolute timestamp.
307
+ """
308
+ return cast(float, predict(log, prefix, "next_time", covariates, options))
309
+
310
+
311
+ def predict_remaining_time(
312
+ log: Log,
313
+ prefix: Trace | Log,
314
+ *,
315
+ covariates: CovariateSet = "target_only",
316
+ **options: Unpack[PrefixOptions],
317
+ ) -> float:
318
+ """Predict remaining cycle time from prior cases at the same prefix position.
319
+
320
+ :param log: Completed cases ending before the current case starts.
321
+ :param prefix: One observed nonterminal case prefix.
322
+ :param covariates: Automatic covariate set.
323
+ :param options: PM4Py keys, attribute allowlist and aligned past/future series.
324
+ :return: Raw median remaining duration in seconds.
325
+ """
326
+ return cast(float, predict(log, prefix, "remaining_time", covariates, options))
@@ -0,0 +1,94 @@
1
+ """Typed parameters shared by the small functional API."""
2
+
3
+ from collections.abc import Mapping, Sequence
4
+ from dataclasses import dataclass
5
+ from typing import Literal, TypeAlias, TypedDict
6
+
7
+ import numpy as np
8
+ import pandas as pd
9
+ from numpy.typing import ArrayLike, NDArray
10
+ from pm4py.objects.log.obj import EventLog
11
+
12
+ Log: TypeAlias = EventLog | pd.DataFrame
13
+ Edge: TypeAlias = tuple[str, str]
14
+ Key: TypeAlias = str | tuple[str, ...] | tuple[Edge, int]
15
+ FloatArray: TypeAlias = NDArray[np.float64]
16
+ Series: TypeAlias = NDArray[np.generic]
17
+ Covariates: TypeAlias = Mapping[str, ArrayLike]
18
+ CovariateSet: TypeAlias = Literal["target_only", "selected", "global"]
19
+ Task: TypeAlias = Literal[
20
+ "dfg",
21
+ "edges",
22
+ "start",
23
+ "end",
24
+ "activity",
25
+ "variants",
26
+ "footprints",
27
+ "performance",
28
+ "bottlenecks",
29
+ "temporal",
30
+ "entropy",
31
+ "variety",
32
+ "completeness",
33
+ "next_activity",
34
+ "next_time",
35
+ "remaining_time",
36
+ ]
37
+ ACTIVITY_KEY: str = "concept:name"
38
+ TIMESTAMP_KEY: str = "time:timestamp"
39
+ CASE_ID_KEY: str = "case:concept:name"
40
+ WINDOW_COUNT: int = 8
41
+ PRESENCE_THRESHOLD: float = 0.5
42
+ RELATIONS: tuple[str, ...] = ("none", "sequence", "parallel")
43
+
44
+
45
+ @dataclass(frozen=True)
46
+ class ModelConfig:
47
+ """Identify one checkpoint and its inference device; no training occurs."""
48
+
49
+ repository_id: str = "autogluon/chronos-2-small"
50
+ revision: str = "ddec01313e50b6bc58ebaa92ede81bc24a3d9f9a"
51
+ device_map: str = "cpu"
52
+ batch_size: int = 256
53
+
54
+ def __post_init__(self) -> None:
55
+ """Reject incomplete model configuration.
56
+
57
+ :return: None.
58
+ """
59
+ if not self.repository_id or not self.revision or self.batch_size < 1:
60
+ raise ValueError(
61
+ "model repository, revision and positive batch_size are required"
62
+ )
63
+
64
+
65
+ class ForecastOptions(TypedDict, total=False):
66
+ """Additional keyword parameters accepted by aggregate functions."""
67
+
68
+ n_windows: int
69
+ forecast_origin: pd.Timestamp
70
+ activity_key: str
71
+ timestamp_key: str
72
+ case_id_key: str
73
+ attribute_allowlist: Sequence[str]
74
+ past_covariates: Covariates
75
+ future_covariates: Covariates
76
+
77
+
78
+ class PrefixOptions(TypedDict, total=False):
79
+ """Additional keyword parameters accepted by prefix functions."""
80
+
81
+ activity_key: str
82
+ timestamp_key: str
83
+ case_id_key: str
84
+ attribute_allowlist: Sequence[str]
85
+ past_covariates: Covariates
86
+ future_covariates: Covariates
87
+
88
+
89
+ class Footprint(TypedDict):
90
+ """The three PM4Py footprint fields forecast by the benchmark."""
91
+
92
+ activities: set[str]
93
+ sequence: set[Edge]
94
+ parallel: set[Edge]