pmprov 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
models/__init__.py ADDED
@@ -0,0 +1,78 @@
1
+ """
2
+ Pydantic data model for Analytic Provenance in Exploratory Process Mining.
3
+
4
+ Derived from: "A Conceptual Model to Enhance Process Mining Analysis with Provenance"
5
+ (ER 2026, blinded for review)
6
+
7
+ Package layout
8
+ --------------
9
+ agents – Agent, AgentType, RuntimeEnvironment
10
+ artifacts – Artifact, ArtifactType, ArtifactState, Delta, DataFrameDelta, ModificationType
11
+ parameters – Parameter, ParameterValueType, and all ParameterValue subclasses / union
12
+ operations – OperationType, StepCategory, Operation, Version
13
+ analysis – AnalysisStep, AnalysisState, StateAbstraction
14
+ history – AnalysisBranch, AnalysisHistory
15
+ pipelines – PipelineFragment, Pipeline
16
+ """
17
+
18
+ from .agents import Agent, AgentType, RuntimeEnvironment
19
+ from .analysis import AnalysisState, AnalysisStep, StateAbstraction
20
+ from .artifacts import (
21
+ Artifact,
22
+ ArtifactState,
23
+ ArtifactType,
24
+ DataFrameDelta,
25
+ Delta,
26
+ ModificationType,
27
+ )
28
+ from .history import AnalysisBranch, AnalysisHistory
29
+ from .operations import Operation, OperationType, StepCategory, Version
30
+ from .parameters import (
31
+ ArtifactStateParameterValue,
32
+ DictParameterValue,
33
+ LambdaParameterValue,
34
+ ListParameterValue,
35
+ Parameter,
36
+ ParameterValue,
37
+ ParameterValueType,
38
+ ScalarParameterValue,
39
+ )
40
+ from .pipelines import Pipeline, PipelineFragment
41
+
42
+ __all__ = [
43
+ # agents
44
+ "AgentType",
45
+ "Agent",
46
+ "RuntimeEnvironment",
47
+ # artifacts
48
+ "ArtifactType",
49
+ "ModificationType",
50
+ "Artifact",
51
+ "ArtifactState",
52
+ "Delta",
53
+ "DataFrameDelta",
54
+ # parameters
55
+ "ParameterValueType",
56
+ "Parameter",
57
+ "ScalarParameterValue",
58
+ "ArtifactStateParameterValue",
59
+ "LambdaParameterValue",
60
+ "ListParameterValue",
61
+ "DictParameterValue",
62
+ "ParameterValue",
63
+ # operations
64
+ "OperationType",
65
+ "StepCategory",
66
+ "Operation",
67
+ "Version",
68
+ # analysis
69
+ "StateAbstraction",
70
+ "AnalysisState",
71
+ "AnalysisStep",
72
+ # history
73
+ "AnalysisBranch",
74
+ "AnalysisHistory",
75
+ # pipelines
76
+ "PipelineFragment",
77
+ "Pipeline",
78
+ ]
models/agents.py ADDED
@@ -0,0 +1,59 @@
1
+ """Agent and execution-environment models."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from enum import Enum
6
+ from typing import Optional
7
+
8
+ from pydantic import BaseModel, Field
9
+
10
+
11
+ class AgentType(str, Enum):
12
+ """Distinguishes human analysts from automated scripts / systems."""
13
+
14
+ HUMAN = "human"
15
+ AUTOMATED = "automated"
16
+
17
+
18
+ class Agent(BaseModel):
19
+ """
20
+ Represents the entity (human or system) that performs an AnalysisStep.
21
+
22
+ UML attributes: agentId, agentType
23
+ """
24
+
25
+ agent_id: str = Field(..., description="Unique identifier for the agent.")
26
+ agent_type: AgentType = Field(
27
+ ..., description="Whether the agent is a human analyst or an automated system."
28
+ )
29
+ username: Optional[str] = Field(
30
+ None, description="OS username of the agent, captured at session start."
31
+ )
32
+
33
+
34
+ class RuntimeEnvironment(BaseModel):
35
+ """
36
+ Captures the execution environment of an AnalysisStep to enable reproducibility (R2).
37
+
38
+ UML attributes: envId, toolVersion, libraryVersions, runtime
39
+ """
40
+
41
+ env_id: str = Field(..., description="Unique identifier for this environment snapshot.")
42
+ tool_version: str = Field(
43
+ ..., description="Version of the primary analysis tool (e.g., Python 3.11)."
44
+ )
45
+ library_versions: dict[str, str] = Field(
46
+ default_factory=dict,
47
+ description="Map of library name → version string (e.g., {'pandas': '2.2.1'}).",
48
+ )
49
+ runtime: Optional[str] = Field(
50
+ None,
51
+ description="Free-form description of the runtime platform (OS, hardware, etc.).",
52
+ )
53
+
54
+
55
+ __all__ = [
56
+ "AgentType",
57
+ "Agent",
58
+ "RuntimeEnvironment",
59
+ ]
models/analysis.py ADDED
@@ -0,0 +1,137 @@
1
+ """AnalysisStep, AnalysisState, and StateAbstraction models."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from datetime import datetime, timezone
6
+ from typing import Any, Optional
7
+
8
+ from pydantic import BaseModel, Field
9
+
10
+
11
+ class StateAbstraction(BaseModel):
12
+ """
13
+ A comparable summary or aggregated representation of an AnalysisState,
14
+ enabling state-level and history-level comparison (R5, R6).
15
+
16
+ UML attributes : abstractionId, type, value, function
17
+ UML relationship:
18
+ - wasAbstractedBy → AnalysisState (via analysis_state_id)
19
+ """
20
+
21
+ abstraction_id: str = Field(..., description="Unique identifier for this abstraction.")
22
+ analysis_state_id: str = Field(
23
+ ...,
24
+ description="FK → AnalysisState.state_id – the state this abstraction summarises.",
25
+ )
26
+ abstraction_type: str = Field(
27
+ ...,
28
+ description=(
29
+ "Category of abstraction (e.g., 'trace_variant_distribution', "
30
+ "'performance_metric', 'feature_vector')."
31
+ ),
32
+ )
33
+ value: Any = Field(
34
+ ...,
35
+ description="The computed abstraction value (scalar, dict, list, …).",
36
+ )
37
+ function: str = Field(
38
+ ...,
39
+ description=(
40
+ "Identifier or source-code reference of the function used to compute "
41
+ "this abstraction (e.g., a qualified Python callable name)."
42
+ ),
43
+ )
44
+
45
+
46
+ class AnalysisState(BaseModel):
47
+ """
48
+ An immutable snapshot of the overall analysis at a specific point in time.
49
+ Acts as a container for one or more ArtifactStates.
50
+
51
+ UML attributes : stateId, createdAt
52
+ UML relationships:
53
+ - usedInput → AnalysisStep (via produced_by_step_id)
54
+ - outputProducedBy → AnalysisStep (via produced_by_step_id)
55
+ - wasDerivedFrom → AnalysisState (via derived_from_state_id)
56
+ - hasActiveState → AnalysisBranch (via branch_id)
57
+ - includes → ArtifactState (1-to-N; represented in ArtifactState.analysis_state_id)
58
+ - wasAbstractedBy → StateAbstraction (1-to-N; represented in StateAbstraction.analysis_state_id)
59
+ """
60
+
61
+ state_id: str = Field(..., description="Unique identifier for this analysis state.")
62
+ history_id: str = Field(
63
+ ...,
64
+ description="FK → AnalysisHistory.history_id – the session this state belongs to.",
65
+ )
66
+ created_at: datetime = Field(
67
+ default_factory=lambda: datetime.now(timezone.utc),
68
+ description="UTC timestamp at which this state was created.",
69
+ )
70
+ branch_id: str = Field(
71
+ ...,
72
+ description="FK → AnalysisBranch.branch_id – the branch this state belongs to.",
73
+ )
74
+ produced_by_step_id: Optional[str] = Field(
75
+ None,
76
+ description=(
77
+ "FK → AnalysisStep.step_id – the step whose output produced this state. "
78
+ "None for the initial ROOT state."
79
+ ),
80
+ )
81
+ derived_from_state_id: Optional[str] = Field(
82
+ None,
83
+ description=(
84
+ "FK → AnalysisState.state_id – the predecessor state in the analysis chain. "
85
+ "None for the ROOT state."
86
+ ),
87
+ )
88
+
89
+
90
+ class AnalysisStep(BaseModel):
91
+ """
92
+ A concrete, immutable execution of an Operation within an AnalysisBranch.
93
+ Bridges two AnalysisStates (input → output) and records the full execution context.
94
+
95
+ UML attributes: stepId, createdAt
96
+ UML relationships:
97
+ - usedInput → AnalysisState (via input_state_id)
98
+ - outputProducedBy → AnalysisState (via output_state_id)
99
+ - performedBy → Agent (via agent_id)
100
+ - executesIn → RuntimeEnvironment (via env_id)
101
+ - executes → Operation (via operation_id)
102
+ - produces → Delta (1-to-N; represented in Delta.root/updated refs)
103
+ - instantiates → ParameterValue (1-to-N; represented in ParameterValue.step_id)
104
+ """
105
+
106
+ step_id: str = Field(..., description="Unique identifier for this analysis step.")
107
+ created_at: datetime = Field(
108
+ default_factory=lambda: datetime.now(timezone.utc),
109
+ description="UTC timestamp at which this step was executed.",
110
+ )
111
+ input_state_id: str = Field(
112
+ ...,
113
+ description="FK → AnalysisState.state_id – the state consumed as input.",
114
+ )
115
+ output_state_id: str = Field(
116
+ ...,
117
+ description="FK → AnalysisState.state_id – the state produced as output.",
118
+ )
119
+ agent_id: str = Field(
120
+ ...,
121
+ description="FK → Agent.agent_id – the agent that performed this step.",
122
+ )
123
+ env_id: str = Field(
124
+ ...,
125
+ description="FK → RuntimeEnvironment.env_id – the runtime environment used.",
126
+ )
127
+ operation_id: str = Field(
128
+ ...,
129
+ description="FK → Operation.operation_id – the operation that was executed.",
130
+ )
131
+
132
+
133
+ __all__ = [
134
+ "StateAbstraction",
135
+ "AnalysisState",
136
+ "AnalysisStep",
137
+ ]
models/artifacts.py ADDED
@@ -0,0 +1,134 @@
1
+ """Artifact, ArtifactState, and Delta models."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from enum import Enum
6
+ from typing import Optional
7
+
8
+ from pydantic import BaseModel, Field
9
+
10
+
11
+ class ArtifactType(str, Enum):
12
+ EVENT_LOG = "event_log"
13
+ PROCESS_MODEL = "process_model"
14
+ KPI_REPORT = "kpi_report"
15
+ OTHER = "other"
16
+
17
+
18
+ class ModificationType(str, Enum):
19
+ """Analyst-facing classification of what a Delta describes."""
20
+ ADDITION = "addition" # new rows or columns added
21
+ REMOVAL = "removal" # rows or columns removed
22
+ RENAMING = "renaming" # columns renamed
23
+ CASTING = "casting" # column dtype changed
24
+ NORMALIZATION = "normalization" # values scaled or standardised (e.g. min-max, z-score)
25
+ ENRICHMENT = "enrichment" # new information derived from external or computed sources
26
+ OBFUSCATION = "obfuscation" # values masked, hashed, or anonymised
27
+ RECALCULATION = "recalculation" # existing values recomputed (e.g. updated formula or parameter)
28
+ OTHER = "other"
29
+
30
+
31
+ class Artifact(BaseModel):
32
+ """
33
+ Conceptual data object (e.g., an event log or process model) that persists
34
+ across the analysis. ArtifactStates are its versioned snapshots.
35
+
36
+ UML attributes: artifactId, name, artifactType
37
+ """
38
+
39
+ artifact_id: str = Field(..., description="Unique identifier for the artifact.")
40
+ name: str = Field(..., description="Human-readable name (e.g., 'RTFM event log').")
41
+ artifact_type: ArtifactType = Field(..., description="Category of the artifact.")
42
+
43
+
44
+ class ArtifactState(BaseModel):
45
+ """
46
+ An immutable snapshot of an Artifact at a specific point in the analysis.
47
+ Multiple ArtifactStates may be *included* in a single AnalysisState.
48
+
49
+ UML attributes : artifactStateId, mimeType, checksum, contentRef, sizeBytes
50
+ UML relationships:
51
+ - includes → AnalysisState (via analysis_state_id)
52
+ - roots → Delta (an ArtifactState is the root of a Delta chain)
53
+ - updates → Delta (an ArtifactState is the updated end of a Delta)
54
+ - snapshots → Artifact (via artifact_id)
55
+ """
56
+
57
+ artifact_state_id: str = Field(..., description="Unique identifier for this snapshot.")
58
+ artifact_id: str = Field(
59
+ ..., description="FK → Artifact.artifact_id – the artifact being snapshotted."
60
+ )
61
+ analysis_state_id: str = Field(
62
+ ...,
63
+ description="FK → AnalysisState.state_id – the analysis state that includes this snapshot.",
64
+ )
65
+ mime_type: str = Field(
66
+ ..., description="MIME type of the stored content (e.g., 'text/csv')."
67
+ )
68
+ checksum: str = Field(
69
+ ..., description="Hash of the content for integrity verification (e.g., SHA-256)."
70
+ )
71
+ content_ref: str = Field(
72
+ ...,
73
+ description="Pointer to the actual stored content (file path, URI, object-store key, …).",
74
+ )
75
+ size_bytes: int = Field(..., description="Size of the stored content in bytes.")
76
+
77
+
78
+ class Delta(BaseModel):
79
+ """
80
+ Abstract base class that records the incremental structural change between
81
+ two ArtifactStates, supporting Data Evolution Transparency (R7).
82
+
83
+ UML attributes : deltaId, modificationType
84
+ UML relationships:
85
+ - roots → ArtifactState (via root_artifact_state_id)
86
+ - updates → ArtifactState (via updated_artifact_state_id)
87
+ """
88
+
89
+ delta_id: str = Field(..., description="Unique identifier for this delta.")
90
+ modification_type: ModificationType = Field(
91
+ ..., description="High-level classification of the change."
92
+ )
93
+ root_artifact_state_id: str = Field(
94
+ ...,
95
+ description="FK → ArtifactState.artifact_state_id – the state *before* the change.",
96
+ )
97
+ updated_artifact_state_id: str = Field(
98
+ ...,
99
+ description="FK → ArtifactState.artifact_state_id – the state *after* the change.",
100
+ )
101
+
102
+
103
+ class DataFrameDelta(Delta):
104
+ """
105
+ Concrete Delta subclass for pandas DataFrame transformations.
106
+
107
+ Extends Delta with DataFrame-specific change attributes used in the
108
+ evaluation scenario (Sect. 6 of the paper).
109
+ """
110
+
111
+ columns_added: list[str] = Field(
112
+ default_factory=list, description="Column names added in this step."
113
+ )
114
+ columns_removed: list[str] = Field(
115
+ default_factory=list, description="Column names removed in this step."
116
+ )
117
+ dtype_changes: dict[str, str] = Field(
118
+ default_factory=dict,
119
+ description="Map of column name → new dtype string for columns whose type changed.",
120
+ )
121
+ rows_delta: Optional[int] = Field(
122
+ None,
123
+ description="Net row count change (positive = rows added, negative = rows removed).",
124
+ )
125
+
126
+
127
+ __all__ = [
128
+ "ArtifactType",
129
+ "ModificationType",
130
+ "Artifact",
131
+ "ArtifactState",
132
+ "Delta",
133
+ "DataFrameDelta",
134
+ ]
models/history.py ADDED
@@ -0,0 +1,72 @@
1
+ """AnalysisBranch and AnalysisHistory models."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from datetime import datetime, timezone
6
+ from typing import Optional
7
+
8
+ from pydantic import BaseModel, Field
9
+
10
+
11
+ class AnalysisBranch(BaseModel):
12
+ """
13
+ An independent analysis path within an AnalysisHistory. Multiple branches
14
+ may share a common ancestor state (R3 – Branching Capability).
15
+
16
+ UML attributes: branchId
17
+ UML relationships:
18
+ - belongs to → AnalysisHistory (via history_id)
19
+ - startsAt → AnalysisState (via starts_at_state_id – divergence point)
20
+ - hasActiveState → AnalysisState (via active_state_id – current tip of the branch)
21
+ """
22
+
23
+ branch_id: str = Field(..., description="Unique identifier for this branch.")
24
+ history_id: str = Field(
25
+ ...,
26
+ description="FK → AnalysisHistory.history_id – the history this branch belongs to.",
27
+ )
28
+ name: str = Field(
29
+ "main",
30
+ description="Human-readable branch label (e.g., 'main', 'experiment-1'). Defaults to 'main'.",
31
+ )
32
+ starts_at_state_id: str = Field(
33
+ ...,
34
+ description=(
35
+ "FK → AnalysisState.state_id – the state at which this branch diverges "
36
+ "(i.e., its common origin with the parent branch)."
37
+ ),
38
+ )
39
+
40
+
41
+ class AnalysisHistory(BaseModel):
42
+ """
43
+ Root container that encapsulates the entire lifecycle of an analytical process,
44
+ including all of its branches.
45
+
46
+ UML attributes: historyId
47
+ UML relationships:
48
+ - aggregates → AnalysisBranch (1-to-N; represented in AnalysisBranch.history_id)
49
+ """
50
+
51
+ history_id: str = Field(..., description="Unique identifier for this analysis history.")
52
+ name: Optional[str] = Field(
53
+ None, description="Optional human-readable title for the analysis session."
54
+ )
55
+ created_at: datetime = Field(
56
+ default_factory=lambda: datetime.now(timezone.utc),
57
+ description="UTC timestamp at which this history was initiated.",
58
+ )
59
+ active_state_id: Optional[str] = Field(
60
+ None,
61
+ description=(
62
+ "FK → AnalysisState.state_id – the most recently produced state in this "
63
+ "history, regardless of branch. Updated after every step so it always "
64
+ "reflects where the analyst currently is."
65
+ ),
66
+ )
67
+
68
+
69
+ __all__ = [
70
+ "AnalysisBranch",
71
+ "AnalysisHistory",
72
+ ]
models/operations.py ADDED
@@ -0,0 +1,99 @@
1
+ """Operation definitions, their classification types, and versioning."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from datetime import datetime, timezone
6
+ from typing import Optional
7
+
8
+ from pydantic import BaseModel, Field
9
+
10
+
11
+ class OperationType(BaseModel):
12
+ """
13
+ Fine-grained classification of an operation (e.g., 'attribute_derivation',
14
+ 'case_filter', 'conformance_check').
15
+
16
+ UML attributes: typeId, name
17
+ """
18
+
19
+ type_id: str = Field(..., description="Unique identifier for this operation type.")
20
+ name: str = Field(..., description="Human-readable label (e.g., 'attribute_derivation').")
21
+
22
+
23
+ class StepCategory(BaseModel):
24
+ """
25
+ Broad grouping of analysis steps (e.g., 'log_enrichment', 'process_discovery',
26
+ 'conformance_checking').
27
+
28
+ UML attributes: categoryId, name
29
+ """
30
+
31
+ category_id: str = Field(..., description="Unique identifier for this step category.")
32
+ name: str = Field(..., description="Human-readable label (e.g., 'log_enrichment').")
33
+
34
+
35
+ class Operation(BaseModel):
36
+ """
37
+ The abstract, reusable definition of an analysis task (e.g., 'filter by threshold',
38
+ 'derive case-centric view'). Concrete executions are recorded as AnalysisSteps.
39
+
40
+ UML attributes: operationId, name
41
+ UML relationships:
42
+ - hasType → OperationType (via operation_type_id)
43
+ - belongsTo → StepCategory (via step_category_id)
44
+ - aggregates → Parameter (1-to-N; represented in Parameter.operation_id)
45
+ """
46
+
47
+ operation_id: str = Field(..., description="Unique identifier for this operation definition.")
48
+ name: str = Field(..., description="Human-readable operation name (e.g., 'add_attribute').")
49
+ operation_type_id: str = Field(
50
+ ...,
51
+ description="FK → OperationType.type_id – the fine-grained type of this operation.",
52
+ )
53
+ step_category_id: Optional[str] = Field(
54
+ None,
55
+ description=(
56
+ "FK → StepCategory.category_id – the broad category this operation belongs to. "
57
+ "Optional if not yet categorised."
58
+ ),
59
+ )
60
+
61
+
62
+ class Version(BaseModel):
63
+ """
64
+ Governs individual operations within a Pipeline, capturing creation timestamps
65
+ and hierarchical lineage to maintain the integrity of reusable components.
66
+
67
+ UML attributes: versionId, number, createdAt
68
+ UML relationship:
69
+ - originatedFrom → Version (via originated_from_version_id – parent version)
70
+ - references → Operation (via operation_id)
71
+ """
72
+
73
+ version_id: str = Field(..., description="Unique identifier for this version record.")
74
+ operation_id: str = Field(
75
+ ...,
76
+ description="FK → Operation.operation_id – the operation this version tracks.",
77
+ )
78
+ number: str = Field(
79
+ ..., description="Semantic version string (e.g., '1.0.0', '2.1.3')."
80
+ )
81
+ created_at: datetime = Field(
82
+ default_factory=lambda: datetime.now(timezone.utc),
83
+ description="UTC timestamp at which this version was created.",
84
+ )
85
+ originated_from_version_id: Optional[str] = Field(
86
+ None,
87
+ description=(
88
+ "FK → Version.version_id – the parent version from which this one was derived. "
89
+ "None for the initial version."
90
+ ),
91
+ )
92
+
93
+
94
+ __all__ = [
95
+ "OperationType",
96
+ "StepCategory",
97
+ "Operation",
98
+ "Version",
99
+ ]