sampling-mining-workflows-dsl 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sampling_mining_workflows_dsl/ __init__.py +0 -0
- sampling_mining_workflows_dsl/CompleteWorkflow.py +23 -0
- sampling_mining_workflows_dsl/Workflow.py +242 -0
- sampling_mining_workflows_dsl/WorkflowBuilder.py +11 -0
- sampling_mining_workflows_dsl/analysis/ChiSquareAnalysis.py +37 -0
- sampling_mining_workflows_dsl/analysis/CochranTest.py +44 -0
- sampling_mining_workflows_dsl/analysis/CochranWorkflowAnalysis.py +37 -0
- sampling_mining_workflows_dsl/analysis/CoverageTest.py +60 -0
- sampling_mining_workflows_dsl/analysis/DistributionWorkflowAnalysis.py +57 -0
- sampling_mining_workflows_dsl/analysis/HistAnalysis.py +201 -0
- sampling_mining_workflows_dsl/analysis/HistWorkflowAnalysis.py +72 -0
- sampling_mining_workflows_dsl/analysis/KSWorkflowAnalysis.py +217 -0
- sampling_mining_workflows_dsl/analysis/WorkflowAnalysis.py +7 -0
- sampling_mining_workflows_dsl/analysis/YamaneTest.py +16 -0
- sampling_mining_workflows_dsl/analysis/YamaneWorkflowAnalysis.py +27 -0
- sampling_mining_workflows_dsl/analysis/kolmogorov_smirnov.py +31 -0
- sampling_mining_workflows_dsl/constraint/BoolComparator.py +23 -0
- sampling_mining_workflows_dsl/constraint/BoolConstraint.py +50 -0
- sampling_mining_workflows_dsl/constraint/BoolConstraintString.py +66 -0
- sampling_mining_workflows_dsl/constraint/Comparator.py +16 -0
- sampling_mining_workflows_dsl/constraint/Constraint.py +23 -0
- sampling_mining_workflows_dsl/constraint/NaturalComparator.py +17 -0
- sampling_mining_workflows_dsl/element/Element.py +45 -0
- sampling_mining_workflows_dsl/element/Loader.py +17 -0
- sampling_mining_workflows_dsl/element/Repository.py +28 -0
- sampling_mining_workflows_dsl/element/Set.py +210 -0
- sampling_mining_workflows_dsl/element/Writer.py +9 -0
- sampling_mining_workflows_dsl/element/loader/CsvLoader.py +67 -0
- sampling_mining_workflows_dsl/element/loader/JsonLoader.py +84 -0
- sampling_mining_workflows_dsl/element/loader/LoaderFactory.py +14 -0
- sampling_mining_workflows_dsl/element/writer/CsvWriter.py +46 -0
- sampling_mining_workflows_dsl/element/writer/JsonWriter.py +58 -0
- sampling_mining_workflows_dsl/element/writer/WriterFactory.py +7 -0
- sampling_mining_workflows_dsl/exec_visualizer/WorkflowVisualizer.py +189 -0
- sampling_mining_workflows_dsl/github_seart/loader.py +17 -0
- sampling_mining_workflows_dsl/github_seart/metadata.py +106 -0
- sampling_mining_workflows_dsl/metadata/Metadata.py +117 -0
- sampling_mining_workflows_dsl/metadata/MetadataBoolean.py +14 -0
- sampling_mining_workflows_dsl/metadata/MetadataDate.py +75 -0
- sampling_mining_workflows_dsl/metadata/MetadataDict.py +33 -0
- sampling_mining_workflows_dsl/metadata/MetadataList.py +32 -0
- sampling_mining_workflows_dsl/metadata/MetadataNumber.py +29 -0
- sampling_mining_workflows_dsl/metadata/MetadataString.py +13 -0
- sampling_mining_workflows_dsl/metadata/MetadataValue.py +24 -0
- sampling_mining_workflows_dsl/operator/Operator.py +165 -0
- sampling_mining_workflows_dsl/operator/OperatorBuilder.py +191 -0
- sampling_mining_workflows_dsl/operator/OperatorFactory.py +76 -0
- sampling_mining_workflows_dsl/operator/clustering/GroupingOperator.py +36 -0
- sampling_mining_workflows_dsl/operator/clustering/SubWorkflowOperatorBuilder.py +76 -0
- sampling_mining_workflows_dsl/operator/selection/SelectionOperator.py +5 -0
- sampling_mining_workflows_dsl/operator/selection/filter/FilterOperator.py +30 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/SamplingOperator.py +11 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/automatic/AutomaticSamplingOperator.py +9 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/automatic/RandomSelectionOperator.py +24 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/automatic/RandomSelectionPartitionOperator.py +62 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/automatic/SystematicRandomSelectionOperator.py +14 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/automatic/SystematicSelectionOperator.py +31 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/manual/InteractiveManualSamplingOperator.py +40 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/manual/ManualSamplingOperator.py +28 -0
- sampling_mining_workflows_dsl/operator/set_algebra/ExternalSetOperator.py +33 -0
- sampling_mining_workflows_dsl/operator/set_algebra/InternalSetOperator.py +47 -0
- sampling_mining_workflows_dsl/operator/set_algebra/SetOperator.py +30 -0
- sampling_mining_workflows_dsl/operator/set_algebra/external_set_operator/DifferenceOperator.py +17 -0
- sampling_mining_workflows_dsl/operator/set_algebra/external_set_operator/IntersectionOperator.py +19 -0
- sampling_mining_workflows_dsl/operator/set_algebra/external_set_operator/UnionOperator.py +17 -0
- sampling_mining_workflows_dsl/operator/set_algebra/internal_set_operator/DifferenceOperator.py +17 -0
- sampling_mining_workflows_dsl/operator/set_algebra/internal_set_operator/IntersectionOperator.py +18 -0
- sampling_mining_workflows_dsl/operator/set_algebra/internal_set_operator/UnionOperator.py +17 -0
- sampling_mining_workflows_dsl/operator/set_algebra/set_operator/DifferenceOperator.py +15 -0
- sampling_mining_workflows_dsl/operator/set_algebra/set_operator/IntersectionOperator.py +18 -0
- sampling_mining_workflows_dsl/operator/set_algebra/set_operator/UnionOperator.py +17 -0
- sampling_mining_workflows_dsl/test/ __init__.py +0 -0
- sampling_mining_workflows_dsl/test/Workflow_simple.py +38 -0
- sampling_mining_workflows_dsl/test/input.json +401 -0
- sampling_mining_workflows_dsl/toolbox.py +42 -0
- sampling_mining_workflows_dsl-0.0.1.dist-info/METADATA +236 -0
- sampling_mining_workflows_dsl-0.0.1.dist-info/RECORD +79 -0
- sampling_mining_workflows_dsl-0.0.1.dist-info/WHEEL +4 -0
- sampling_mining_workflows_dsl-0.0.1.dist-info/licenses/LICENSE.txt +674 -0
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
from typing import TYPE_CHECKING, TypeVar
|
|
2
|
+
|
|
3
|
+
from sampling_mining_workflows_dsl.constraint.Constraint import Constraint
|
|
4
|
+
from sampling_mining_workflows_dsl.element.Element import Element
|
|
5
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
6
|
+
from sampling_mining_workflows_dsl.Workflow import Workflow
|
|
7
|
+
|
|
8
|
+
if TYPE_CHECKING:
|
|
9
|
+
from sampling_mining_workflows_dsl.constraint.BoolConstraint import BoolConstraint
|
|
10
|
+
|
|
11
|
+
T = TypeVar("T")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class BoolConstraintString(Constraint[T]):
|
|
15
|
+
def __init__(
|
|
16
|
+
self,
|
|
17
|
+
workflow: Workflow,
|
|
18
|
+
string_constraint: str,
|
|
19
|
+
*targeted_metadatas: Metadata[T],
|
|
20
|
+
):
|
|
21
|
+
super().__init__(workflow, *targeted_metadatas)
|
|
22
|
+
|
|
23
|
+
self.string_constraint = string_constraint
|
|
24
|
+
self.or_constraint: Constraint | None = None
|
|
25
|
+
self.and_constraint: Constraint | None = None
|
|
26
|
+
|
|
27
|
+
def set_workflow(self, workflow: Workflow):
|
|
28
|
+
self.workflow = workflow
|
|
29
|
+
worflow_metadatas = workflow.get_all_Metadata()
|
|
30
|
+
self.targeted_metadatas = tuple(worflow_metadatas)
|
|
31
|
+
return self
|
|
32
|
+
|
|
33
|
+
def is_satisfied(self, element: Element) -> bool:
|
|
34
|
+
all_metadata = self.workflow.get_all_Metadata()
|
|
35
|
+
|
|
36
|
+
# Retrieve matching Metadata objects by checking if their names exist in the string_constraint
|
|
37
|
+
matching_metadata = [
|
|
38
|
+
metadata
|
|
39
|
+
for metadata in all_metadata
|
|
40
|
+
if metadata.name in self.string_constraint
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
metadata_values = {}
|
|
44
|
+
for metadata in matching_metadata:
|
|
45
|
+
metadata_value = element.get_metadata_value(metadata)
|
|
46
|
+
if metadata_value:
|
|
47
|
+
metadata_values[metadata.name] = metadata_value.get_value()
|
|
48
|
+
|
|
49
|
+
# Evaluate the string_constraint
|
|
50
|
+
constraint_result = False
|
|
51
|
+
try:
|
|
52
|
+
constraint_result = eval(self.string_constraint, {}, metadata_values)
|
|
53
|
+
except Exception as e:
|
|
54
|
+
print(f"Error evaluating string_constraint: {e}")
|
|
55
|
+
return constraint_result
|
|
56
|
+
|
|
57
|
+
def or_(self, other: "BoolConstraint") -> "BoolConstraint":
|
|
58
|
+
self.or_constraint = other
|
|
59
|
+
return other
|
|
60
|
+
|
|
61
|
+
def and_(self, other: "BoolConstraint") -> "BoolConstraint":
|
|
62
|
+
self.and_constraint = other
|
|
63
|
+
return other
|
|
64
|
+
|
|
65
|
+
def get_string_constraint(self) -> str:
|
|
66
|
+
return self.string_constraint
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
from abc import ABC, abstractmethod
|
|
2
|
+
from typing import TypeVar
|
|
3
|
+
|
|
4
|
+
from sampling_mining_workflows_dsl.element.Element import Element
|
|
5
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
6
|
+
|
|
7
|
+
T = TypeVar("T")
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class Comparator[T](ABC):
|
|
11
|
+
def __init__(self, targeted_metadata: Metadata[T]):
|
|
12
|
+
self.targeted_metadata = targeted_metadata
|
|
13
|
+
|
|
14
|
+
@abstractmethod
|
|
15
|
+
def compare(self, a: Element, b: Element) -> int:
|
|
16
|
+
pass
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
import abc
|
|
2
|
+
from typing import TYPE_CHECKING, TypeVar
|
|
3
|
+
|
|
4
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
5
|
+
|
|
6
|
+
if TYPE_CHECKING:
|
|
7
|
+
from sampling_mining_workflows_dsl.Workflow import Workflow
|
|
8
|
+
|
|
9
|
+
T = TypeVar("T")
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class Constraint[T]:
|
|
13
|
+
def __init__(self, workflow: "Workflow", *targeted_metadatas: Metadata[T]):
|
|
14
|
+
self.targeted_metadatas = targeted_metadatas
|
|
15
|
+
self.workflow = workflow
|
|
16
|
+
|
|
17
|
+
@abc.abstractmethod
|
|
18
|
+
def is_satisfied(self, element):
|
|
19
|
+
pass
|
|
20
|
+
|
|
21
|
+
def set_workflow(self, workflow: "Workflow"):
|
|
22
|
+
self.workflow = workflow
|
|
23
|
+
return self
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
from typing import TypeVar
|
|
2
|
+
|
|
3
|
+
from sampling_mining_workflows_dsl.constraint.Comparator import Comparator
|
|
4
|
+
from sampling_mining_workflows_dsl.element.Element import Element
|
|
5
|
+
|
|
6
|
+
T = TypeVar("T")
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class NaturalComparator[T](Comparator[T]):
|
|
10
|
+
"""
|
|
11
|
+
Generic comparator for any type T that supports < and >.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
def compare(self, a: Element, b: Element) -> int:
|
|
15
|
+
va = self.targeted_metadata.extract(a)
|
|
16
|
+
vb = self.targeted_metadata.extract(b)
|
|
17
|
+
return (va > vb) - (va < vb)
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
from abc import ABC, abstractmethod
|
|
2
|
+
from typing import TypeVar
|
|
3
|
+
|
|
4
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
5
|
+
from sampling_mining_workflows_dsl.metadata.MetadataValue import MetadataValue
|
|
6
|
+
|
|
7
|
+
T = TypeVar("T")
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class Element(ABC):
|
|
11
|
+
def __init__(self):
|
|
12
|
+
self.metadata: dict[Metadata, MetadataValue] = {}
|
|
13
|
+
|
|
14
|
+
def __hash__(self) -> int:
|
|
15
|
+
return hash(frozenset(self.metadata.items()))
|
|
16
|
+
|
|
17
|
+
def __eq__(self, other) -> bool:
|
|
18
|
+
if not isinstance(other, Element):
|
|
19
|
+
return False
|
|
20
|
+
return self.metadata == other.metadata
|
|
21
|
+
|
|
22
|
+
def get_metadata_value(self, metadata: Metadata[T]) -> MetadataValue[T]:
|
|
23
|
+
if metadata not in self.metadata:
|
|
24
|
+
raise RuntimeError(f"Missing metadata {metadata.name}")
|
|
25
|
+
return self.metadata[metadata]
|
|
26
|
+
|
|
27
|
+
def add_metadata_values(self, metadata_values: list[MetadataValue]):
|
|
28
|
+
for metadata_value in metadata_values:
|
|
29
|
+
self.metadata[metadata_value.get_metadata()] = metadata_value
|
|
30
|
+
|
|
31
|
+
def add_metadata_value(self, metadata_value: MetadataValue):
|
|
32
|
+
self.metadata[metadata_value.get_metadata()] = metadata_value
|
|
33
|
+
|
|
34
|
+
def get_all_metadata_values(self) -> dict[Metadata, MetadataValue]:
|
|
35
|
+
return self.metadata.copy()
|
|
36
|
+
|
|
37
|
+
def get_raw_metadata_values(self) -> dict[str, T]:
|
|
38
|
+
self
|
|
39
|
+
|
|
40
|
+
def get_id(self) -> str:
|
|
41
|
+
raise NotImplementedError("Subclasses must implement get_id method")
|
|
42
|
+
|
|
43
|
+
@abstractmethod
|
|
44
|
+
def to_string(self, level: int) -> str:
|
|
45
|
+
pass
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
from abc import ABC, abstractmethod
|
|
2
|
+
|
|
3
|
+
from sampling_mining_workflows_dsl.element import Set
|
|
4
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class Loader(ABC):
|
|
8
|
+
def __init__(self, *metadatas: Metadata | None):
|
|
9
|
+
# Assuming the first metadata is the ID, store its name for further use
|
|
10
|
+
self.metadata_id_name = metadatas[0].name
|
|
11
|
+
self.metadatas: dict[str, Metadata] = {}
|
|
12
|
+
for metadata in metadatas:
|
|
13
|
+
self.metadatas[metadata.name] = metadata
|
|
14
|
+
|
|
15
|
+
@abstractmethod
|
|
16
|
+
def load_set(self) -> Set:
|
|
17
|
+
pass
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
from sampling_mining_workflows_dsl.element.Element import Element
|
|
2
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class Repository(Element):
|
|
6
|
+
def __init__(self, id_metadata: Metadata[str]):
|
|
7
|
+
super().__init__()
|
|
8
|
+
self.id = id_metadata
|
|
9
|
+
|
|
10
|
+
def get_id(self) -> str:
|
|
11
|
+
return self.get_metadata_value(self.id).get_value()
|
|
12
|
+
|
|
13
|
+
def __hash__(self) -> int:
|
|
14
|
+
return hash(self.get_id())
|
|
15
|
+
|
|
16
|
+
def __eq__(self, other):
|
|
17
|
+
if not super().__eq__(other):
|
|
18
|
+
return False
|
|
19
|
+
if not isinstance(other, Repository):
|
|
20
|
+
return False
|
|
21
|
+
return self.get_id() == other.get_id()
|
|
22
|
+
|
|
23
|
+
def __str__(self) -> str:
|
|
24
|
+
return self.to_string(0)
|
|
25
|
+
|
|
26
|
+
def to_string(self, level: int = 0) -> str:
|
|
27
|
+
indent = " " * level
|
|
28
|
+
return f"{indent}{self.get_metadata_value(self.id)}"
|
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
import random
|
|
2
|
+
from functools import cmp_to_key
|
|
3
|
+
from collections import OrderedDict
|
|
4
|
+
|
|
5
|
+
from sampling_mining_workflows_dsl.constraint.Comparator import Comparator
|
|
6
|
+
from sampling_mining_workflows_dsl.element.Element import Element
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class Set(Element):
|
|
10
|
+
def __init__(self):
|
|
11
|
+
super().__init__()
|
|
12
|
+
self.elements = OrderedDict()
|
|
13
|
+
self.ids= set()
|
|
14
|
+
self.set_id = None
|
|
15
|
+
|
|
16
|
+
def __hash__(self):
|
|
17
|
+
return hash(tuple(self.elements.items)) + super().__hash__()
|
|
18
|
+
|
|
19
|
+
def __eq__(self, other):
|
|
20
|
+
if not super().__eq__(other):
|
|
21
|
+
return False
|
|
22
|
+
if not isinstance(other, Set):
|
|
23
|
+
return False
|
|
24
|
+
return self.elements == other.elements
|
|
25
|
+
|
|
26
|
+
def get_element_by_index(self, index: int) -> Element:
|
|
27
|
+
if index < 0 or index >= len(self.elements):
|
|
28
|
+
raise IndexError("Index out of range")
|
|
29
|
+
return list(self.elements.values())[index]
|
|
30
|
+
def remove_all_elements(self) -> "Set":
|
|
31
|
+
self.elements.clear()
|
|
32
|
+
self.ids.clear()
|
|
33
|
+
return self
|
|
34
|
+
def add_element(self, element: Element) -> "Set":
|
|
35
|
+
if not element.get_id() in self.ids:
|
|
36
|
+
self.elements[element.get_id()]=element
|
|
37
|
+
self.ids.add(element.get_id())
|
|
38
|
+
else :
|
|
39
|
+
print(f"{element.get_id()} Already present")
|
|
40
|
+
|
|
41
|
+
return self
|
|
42
|
+
|
|
43
|
+
def sort_by_metadata(self, metadata_name: str, comparator: Comparator, reverse=False) -> "Set":
|
|
44
|
+
# Sort the items of the OrderedDict
|
|
45
|
+
sorted_items = sorted(
|
|
46
|
+
self.elements.items(),
|
|
47
|
+
key=cmp_to_key(lambda x, y: comparator.compare(x[1], y[1])),
|
|
48
|
+
reverse=reverse
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
# Rebuild as OrderedDict
|
|
52
|
+
self.elements = OrderedDict(sorted_items)
|
|
53
|
+
return self
|
|
54
|
+
|
|
55
|
+
def get_depth(self) -> int:
|
|
56
|
+
max_depth = 1
|
|
57
|
+
for element in self.elements.values():
|
|
58
|
+
if isinstance(element, Set):
|
|
59
|
+
max_depth = max(max_depth, 1 + element.get_depth())
|
|
60
|
+
return max_depth
|
|
61
|
+
|
|
62
|
+
def union(self, other: "Set") -> "Set":
|
|
63
|
+
for element in other.elements.values():
|
|
64
|
+
self.add_element(element)
|
|
65
|
+
return self
|
|
66
|
+
|
|
67
|
+
def intersection(self, other: "Set") -> "Set":
|
|
68
|
+
common_elements = Set()
|
|
69
|
+
for id, element in self.elements.items():
|
|
70
|
+
if id in other.elements.keys():
|
|
71
|
+
common_elements.add_element(element)
|
|
72
|
+
return common_elements
|
|
73
|
+
|
|
74
|
+
def difference(self, other: "Set") -> "Set":
|
|
75
|
+
"""Return a new set with elements in this set but not in other"""
|
|
76
|
+
diff_elements = Set()
|
|
77
|
+
for id, element in self.elements.items():
|
|
78
|
+
if id not in other.elements.keys():
|
|
79
|
+
diff_elements.add_element(element)
|
|
80
|
+
return diff_elements
|
|
81
|
+
|
|
82
|
+
def symmetric_difference(self, other: "Set") -> "Set":
|
|
83
|
+
"""Return a new set with elements in either set but not in both"""
|
|
84
|
+
sym_diff = Set()
|
|
85
|
+
|
|
86
|
+
# Add elements from this set that are not in other
|
|
87
|
+
for id, element in self.elements.items():
|
|
88
|
+
if id not in other.elements.keys():
|
|
89
|
+
sym_diff.add_element(element)
|
|
90
|
+
|
|
91
|
+
# Add elements from other set that are not in this
|
|
92
|
+
for id, element in other.elements.items():
|
|
93
|
+
if id not in self.elements.keys():
|
|
94
|
+
sym_diff.add_element(element)
|
|
95
|
+
|
|
96
|
+
return sym_diff
|
|
97
|
+
|
|
98
|
+
def is_subset(self, other: "Set") -> bool:
|
|
99
|
+
"""Return True if all elements in this set are also in other"""
|
|
100
|
+
for id in self.elements.keys():
|
|
101
|
+
if id not in other.elements.keys():
|
|
102
|
+
return False
|
|
103
|
+
return True
|
|
104
|
+
|
|
105
|
+
def is_superset(self, other: "Set") -> bool:
|
|
106
|
+
"""Return True if all elements in other set are also in this set"""
|
|
107
|
+
return other.is_subset(self)
|
|
108
|
+
|
|
109
|
+
def is_disjoint(self, other: "Set") -> bool:
|
|
110
|
+
"""Return True if this set and other have no elements in common"""
|
|
111
|
+
for id in self.elements.keys():
|
|
112
|
+
if id in other.elements.keys():
|
|
113
|
+
return False
|
|
114
|
+
return True
|
|
115
|
+
|
|
116
|
+
def is_empty(self) -> bool:
|
|
117
|
+
"""Return True if the set is empty"""
|
|
118
|
+
return len(self.elements) == 0
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def size(self) -> int:
|
|
123
|
+
return len(self.elements)
|
|
124
|
+
|
|
125
|
+
def get_element(self, id: str) -> Element:
|
|
126
|
+
if not id in self.elements.keys():
|
|
127
|
+
raise RuntimeError(f"Element with id {id} not found in the set")
|
|
128
|
+
return self.elements.get(id)
|
|
129
|
+
|
|
130
|
+
def set_id(self, set_id: str) -> "Set":
|
|
131
|
+
self.set_id = set_id
|
|
132
|
+
return self
|
|
133
|
+
|
|
134
|
+
def get_id(self):
|
|
135
|
+
if self.set_id is not None:
|
|
136
|
+
return self.set_id
|
|
137
|
+
|
|
138
|
+
set_id = ""
|
|
139
|
+
for id in self.elements.keys():
|
|
140
|
+
set_id = set_id + "_" + str(id)
|
|
141
|
+
return set_id
|
|
142
|
+
|
|
143
|
+
def flatten_set(self) -> "Set":
|
|
144
|
+
flattened = Set()
|
|
145
|
+
for element in self.get_elements():
|
|
146
|
+
if isinstance(element, Set):
|
|
147
|
+
# Recursively flatten nested Sets
|
|
148
|
+
flattened.union(element.flatten_set())
|
|
149
|
+
else:
|
|
150
|
+
# Add non-Set, non-list elements directly
|
|
151
|
+
flattened.add_element(element)
|
|
152
|
+
return flattened
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def get_random_subset(self, subset_size: int, seed: int) -> "Set":
|
|
156
|
+
elements_list = self.get_elements()
|
|
157
|
+
if subset_size > len(elements_list):
|
|
158
|
+
print(
|
|
159
|
+
f"Caution, subset size is larger than the size of the original set, subset size: {subset_size}, current set size: {len(elements_list)}"
|
|
160
|
+
)
|
|
161
|
+
subset_size = len(elements_list)
|
|
162
|
+
|
|
163
|
+
random.seed(seed)
|
|
164
|
+
random_indices = random.sample(range(len(elements_list)), subset_size)
|
|
165
|
+
original_array = list(elements_list)
|
|
166
|
+
|
|
167
|
+
result = Set()
|
|
168
|
+
for index in random_indices:
|
|
169
|
+
result.add_element(original_array[index])
|
|
170
|
+
|
|
171
|
+
return result
|
|
172
|
+
|
|
173
|
+
def get_elements(self) -> list[Element]:
|
|
174
|
+
return list(self.elements.values())
|
|
175
|
+
|
|
176
|
+
def clone(self) -> "Set":
|
|
177
|
+
cloned_set = Set()
|
|
178
|
+
for element in self.get_elements():
|
|
179
|
+
cloned_set.add_element(element)
|
|
180
|
+
return cloned_set
|
|
181
|
+
|
|
182
|
+
def __str__(self) -> str:
|
|
183
|
+
return self.to_string(0)
|
|
184
|
+
|
|
185
|
+
def to_string(self, level: int = 0) -> str:
|
|
186
|
+
truncate_after = 10
|
|
187
|
+
indent = " " * level
|
|
188
|
+
|
|
189
|
+
result = f"{indent}(size={len(self.elements)})["
|
|
190
|
+
elements_list = self.get_elements()
|
|
191
|
+
element_to_print = min(truncate_after, len(self.elements))
|
|
192
|
+
|
|
193
|
+
for i in range(element_to_print):
|
|
194
|
+
next_element = elements_list[i]
|
|
195
|
+
|
|
196
|
+
if isinstance(next_element, Set):
|
|
197
|
+
# Recursively call to_string for nested Sets
|
|
198
|
+
result += f"\n{next_element.to_string(level + 4)}"
|
|
199
|
+
else:
|
|
200
|
+
result += str(next_element)
|
|
201
|
+
|
|
202
|
+
if i != element_to_print - 1:
|
|
203
|
+
result += ","
|
|
204
|
+
|
|
205
|
+
if len(self.elements) > truncate_after:
|
|
206
|
+
result += "...]"
|
|
207
|
+
else:
|
|
208
|
+
result += "]"
|
|
209
|
+
|
|
210
|
+
return result
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import csv
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import TYPE_CHECKING, Any
|
|
4
|
+
import logging
|
|
5
|
+
from sampling_mining_workflows_dsl.element.Loader import Loader
|
|
6
|
+
from sampling_mining_workflows_dsl.element.Repository import Repository
|
|
7
|
+
from sampling_mining_workflows_dsl.element.Set import Set
|
|
8
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
9
|
+
|
|
10
|
+
if TYPE_CHECKING:
|
|
11
|
+
from sampling_mining_workflows_dsl.metadata.MetadataValue import MetadataValue
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class CsvLoader(Loader):
|
|
15
|
+
def __init__(self, set_path: Path, *metadatas: Metadata):
|
|
16
|
+
super().__init__(*metadatas)
|
|
17
|
+
self.set_path = set_path
|
|
18
|
+
self.set = Set()
|
|
19
|
+
|
|
20
|
+
def load_set(self) -> Set:
|
|
21
|
+
try:
|
|
22
|
+
if self.set_path.is_dir():
|
|
23
|
+
csv_files = sorted(self.set_path.glob("*.csv"))
|
|
24
|
+
if not csv_files:
|
|
25
|
+
raise RuntimeError(
|
|
26
|
+
f"No CSV files found in directory: {self.set_path}"
|
|
27
|
+
)
|
|
28
|
+
else:
|
|
29
|
+
if not self.set_path.exists():
|
|
30
|
+
raise RuntimeError(f"File not found: {self.set_path}")
|
|
31
|
+
csv_files = [self.set_path]
|
|
32
|
+
|
|
33
|
+
for csv_file in csv_files:
|
|
34
|
+
print(f"Loading CSV file: {csv_file}")
|
|
35
|
+
with csv_file.open("r", newline="") as csvfile:
|
|
36
|
+
reader = csv.DictReader(csvfile)
|
|
37
|
+
for row in reader:
|
|
38
|
+
try:
|
|
39
|
+
repository = self.create_repository_from_map(row)
|
|
40
|
+
if repository is not None:
|
|
41
|
+
self.set.add_element(repository)
|
|
42
|
+
except Exception as e:
|
|
43
|
+
logging.info(f"Row skipped due to {e} : {row}")
|
|
44
|
+
return self.set
|
|
45
|
+
except OSError as e:
|
|
46
|
+
raise RuntimeError("Error reading the CSV file", e) from e
|
|
47
|
+
|
|
48
|
+
def create_repository_from_map(self, csv_row: dict[str, Any]) -> Repository:
|
|
49
|
+
id_metadata_value = csv_row.get(self.metadata_id_name)
|
|
50
|
+
|
|
51
|
+
if id_metadata_value is None or id_metadata_value == "":
|
|
52
|
+
raise ValueError(f"Invalid ID {self.metadata_id_name}")
|
|
53
|
+
|
|
54
|
+
repo = Repository(self.metadatas.get(self.metadata_id_name))
|
|
55
|
+
|
|
56
|
+
metadata_values: list[MetadataValue] = []
|
|
57
|
+
for metadata in self.metadatas.values():
|
|
58
|
+
try:
|
|
59
|
+
metadata_value = metadata.create_metadata_value(csv_row.get(metadata.name))
|
|
60
|
+
metadata_values.append(metadata_value)
|
|
61
|
+
except Exception as e:
|
|
62
|
+
raise ValueError(
|
|
63
|
+
f"Error creating metadata value for '{metadata.name}' with value '{csv_row.get(metadata.name)}'",
|
|
64
|
+
) from e
|
|
65
|
+
|
|
66
|
+
repo.add_metadata_values(metadata_values)
|
|
67
|
+
return repo
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import TYPE_CHECKING, Any
|
|
4
|
+
import logging
|
|
5
|
+
from sampling_mining_workflows_dsl.element.Loader import Loader
|
|
6
|
+
from sampling_mining_workflows_dsl.element.Repository import Repository
|
|
7
|
+
from sampling_mining_workflows_dsl.element.Set import Set
|
|
8
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
9
|
+
|
|
10
|
+
if TYPE_CHECKING:
|
|
11
|
+
from sampling_mining_workflows_dsl.metadata.MetadataValue import MetadataValue
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class JsonLoader(Loader):
|
|
15
|
+
def __init__(self, set_path: Path, *metadatas: Metadata):
|
|
16
|
+
super().__init__(*metadatas)
|
|
17
|
+
self.set_path = set_path
|
|
18
|
+
self.set = Set()
|
|
19
|
+
|
|
20
|
+
def load_set(self) -> Set:
|
|
21
|
+
try:
|
|
22
|
+
if self.set_path.is_dir():
|
|
23
|
+
json_files = sorted(self.set_path.glob("*.json"))
|
|
24
|
+
if not json_files:
|
|
25
|
+
raise RuntimeError(
|
|
26
|
+
f"No JSON files found in directory: {self.set_path}"
|
|
27
|
+
)
|
|
28
|
+
else:
|
|
29
|
+
if not self.set_path.exists():
|
|
30
|
+
raise RuntimeError(f"File not found: {self.set_path}")
|
|
31
|
+
json_files = [self.set_path]
|
|
32
|
+
|
|
33
|
+
for json_file in json_files:
|
|
34
|
+
print(f"Loading JSON file: {json_file}")
|
|
35
|
+
with json_file.open("r") as reader:
|
|
36
|
+
json_list: list[dict[str, Any]] = json.load(reader)
|
|
37
|
+
for json_obj in json_list:
|
|
38
|
+
try:
|
|
39
|
+
repository = self.create_repository_from_map(json_obj)
|
|
40
|
+
if repository is not None:
|
|
41
|
+
self.set.add_element(repository)
|
|
42
|
+
except Exception as e:
|
|
43
|
+
logging.info(f"Row skipped due to {e} : {json_obj}")
|
|
44
|
+
return self.set
|
|
45
|
+
except OSError as e:
|
|
46
|
+
raise RuntimeError("Error reading the JSON file", e) from e
|
|
47
|
+
|
|
48
|
+
def create_repository_from_map(self, json_object: dict[str, Any]) -> Repository:
|
|
49
|
+
id_metadata_value = json_object.get(self.metadata_id_name)
|
|
50
|
+
|
|
51
|
+
if id_metadata_value is None or id_metadata_value == "":
|
|
52
|
+
raise ValueError(f"Invalid ID {self.metadata_id_name}")
|
|
53
|
+
|
|
54
|
+
repo = Repository(self.metadatas.get(self.metadata_id_name))
|
|
55
|
+
|
|
56
|
+
metadata_values: list[MetadataValue] = []
|
|
57
|
+
for metadata in self.metadatas.values():
|
|
58
|
+
try:
|
|
59
|
+
metadata_value = metadata.create_metadata_value(json_object.get(metadata.name))
|
|
60
|
+
metadata_values.append(metadata_value)
|
|
61
|
+
except Exception as e:
|
|
62
|
+
raise ValueError(
|
|
63
|
+
f"Error creating metadata value for '{metadata.name}' with value '{json_object.get(metadata.name)}'",
|
|
64
|
+
) from e
|
|
65
|
+
|
|
66
|
+
repo.add_metadata_values(metadata_values)
|
|
67
|
+
return repo
|
|
68
|
+
|
|
69
|
+
@staticmethod
|
|
70
|
+
def parse_args(args: list[str]) -> dict[str, str]:
|
|
71
|
+
import argparse
|
|
72
|
+
|
|
73
|
+
default_input_path = Path(__file__).parent / "input.json"
|
|
74
|
+
|
|
75
|
+
parser = argparse.ArgumentParser(description="Sampling Workflow")
|
|
76
|
+
parser.add_argument(
|
|
77
|
+
"-i",
|
|
78
|
+
"--inputPath",
|
|
79
|
+
type=str,
|
|
80
|
+
default=str(default_input_path),
|
|
81
|
+
help="Input path file",
|
|
82
|
+
)
|
|
83
|
+
parsed_args = parser.parse_args(args)
|
|
84
|
+
return vars(parsed_args)
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
|
|
3
|
+
from sampling_mining_workflows_dsl.element.loader.JsonLoader import JsonLoader
|
|
4
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class LoaderFactory:
|
|
8
|
+
@staticmethod
|
|
9
|
+
def json_loader(set_path: str, *metadatas: Metadata):
|
|
10
|
+
return JsonLoader(Path(set_path), *metadatas)
|
|
11
|
+
|
|
12
|
+
@staticmethod
|
|
13
|
+
def json_loader_from_path(set_path: Path, *metadatas: Metadata):
|
|
14
|
+
return JsonLoader(set_path, *metadatas)
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import csv
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
from sampling_mining_workflows_dsl.element.Repository import Repository
|
|
5
|
+
from sampling_mining_workflows_dsl.element.Set import Set
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
# Note that the depth of the set should be 1
|
|
9
|
+
class CsvWriter:
|
|
10
|
+
def __init__(self, set_path: str):
|
|
11
|
+
self.set_path = Path(set_path)
|
|
12
|
+
|
|
13
|
+
def write_set(self, set_obj: Set):
|
|
14
|
+
if not isinstance(set_obj, Set):
|
|
15
|
+
raise TypeError("Expected a Set object")
|
|
16
|
+
|
|
17
|
+
# Depth-1: elements are Repository instances
|
|
18
|
+
repositories = set_obj.elements.values()
|
|
19
|
+
|
|
20
|
+
if not repositories:
|
|
21
|
+
raise ValueError("The set is empty. Nothing to write.")
|
|
22
|
+
|
|
23
|
+
rows = [self._serialize_repository(repo) for repo in repositories]
|
|
24
|
+
|
|
25
|
+
headers = sorted(rows[0].keys())
|
|
26
|
+
|
|
27
|
+
try:
|
|
28
|
+
with self.set_path.open("w", encoding="utf-8", newline="") as f:
|
|
29
|
+
writer = csv.DictWriter(f, fieldnames=headers)
|
|
30
|
+
writer.writeheader()
|
|
31
|
+
writer.writerows(rows)
|
|
32
|
+
print(f"CSV has been written to {self.set_path}")
|
|
33
|
+
except OSError as e:
|
|
34
|
+
raise RuntimeError("Error while saving file") from e
|
|
35
|
+
|
|
36
|
+
def _serialize_repository(self, repo: Repository):
|
|
37
|
+
if not isinstance(repo, Repository):
|
|
38
|
+
raise TypeError("Expected Repository elements in the set")
|
|
39
|
+
|
|
40
|
+
result = {}
|
|
41
|
+
for meta, val in repo.get_all_metadata_values().items():
|
|
42
|
+
value = val.get_value()
|
|
43
|
+
if isinstance(value, list):
|
|
44
|
+
value = ";".join(map(str, value))
|
|
45
|
+
result[meta.name] = value
|
|
46
|
+
return result
|