sampling-mining-workflows-dsl 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. sampling_mining_workflows_dsl/ __init__.py +0 -0
  2. sampling_mining_workflows_dsl/CompleteWorkflow.py +23 -0
  3. sampling_mining_workflows_dsl/Workflow.py +242 -0
  4. sampling_mining_workflows_dsl/WorkflowBuilder.py +11 -0
  5. sampling_mining_workflows_dsl/analysis/ChiSquareAnalysis.py +37 -0
  6. sampling_mining_workflows_dsl/analysis/CochranTest.py +44 -0
  7. sampling_mining_workflows_dsl/analysis/CochranWorkflowAnalysis.py +37 -0
  8. sampling_mining_workflows_dsl/analysis/CoverageTest.py +60 -0
  9. sampling_mining_workflows_dsl/analysis/DistributionWorkflowAnalysis.py +57 -0
  10. sampling_mining_workflows_dsl/analysis/HistAnalysis.py +201 -0
  11. sampling_mining_workflows_dsl/analysis/HistWorkflowAnalysis.py +72 -0
  12. sampling_mining_workflows_dsl/analysis/KSWorkflowAnalysis.py +217 -0
  13. sampling_mining_workflows_dsl/analysis/WorkflowAnalysis.py +7 -0
  14. sampling_mining_workflows_dsl/analysis/YamaneTest.py +16 -0
  15. sampling_mining_workflows_dsl/analysis/YamaneWorkflowAnalysis.py +27 -0
  16. sampling_mining_workflows_dsl/analysis/kolmogorov_smirnov.py +31 -0
  17. sampling_mining_workflows_dsl/constraint/BoolComparator.py +23 -0
  18. sampling_mining_workflows_dsl/constraint/BoolConstraint.py +50 -0
  19. sampling_mining_workflows_dsl/constraint/BoolConstraintString.py +66 -0
  20. sampling_mining_workflows_dsl/constraint/Comparator.py +16 -0
  21. sampling_mining_workflows_dsl/constraint/Constraint.py +23 -0
  22. sampling_mining_workflows_dsl/constraint/NaturalComparator.py +17 -0
  23. sampling_mining_workflows_dsl/element/Element.py +45 -0
  24. sampling_mining_workflows_dsl/element/Loader.py +17 -0
  25. sampling_mining_workflows_dsl/element/Repository.py +28 -0
  26. sampling_mining_workflows_dsl/element/Set.py +210 -0
  27. sampling_mining_workflows_dsl/element/Writer.py +9 -0
  28. sampling_mining_workflows_dsl/element/loader/CsvLoader.py +67 -0
  29. sampling_mining_workflows_dsl/element/loader/JsonLoader.py +84 -0
  30. sampling_mining_workflows_dsl/element/loader/LoaderFactory.py +14 -0
  31. sampling_mining_workflows_dsl/element/writer/CsvWriter.py +46 -0
  32. sampling_mining_workflows_dsl/element/writer/JsonWriter.py +58 -0
  33. sampling_mining_workflows_dsl/element/writer/WriterFactory.py +7 -0
  34. sampling_mining_workflows_dsl/exec_visualizer/WorkflowVisualizer.py +189 -0
  35. sampling_mining_workflows_dsl/github_seart/loader.py +17 -0
  36. sampling_mining_workflows_dsl/github_seart/metadata.py +106 -0
  37. sampling_mining_workflows_dsl/metadata/Metadata.py +117 -0
  38. sampling_mining_workflows_dsl/metadata/MetadataBoolean.py +14 -0
  39. sampling_mining_workflows_dsl/metadata/MetadataDate.py +75 -0
  40. sampling_mining_workflows_dsl/metadata/MetadataDict.py +33 -0
  41. sampling_mining_workflows_dsl/metadata/MetadataList.py +32 -0
  42. sampling_mining_workflows_dsl/metadata/MetadataNumber.py +29 -0
  43. sampling_mining_workflows_dsl/metadata/MetadataString.py +13 -0
  44. sampling_mining_workflows_dsl/metadata/MetadataValue.py +24 -0
  45. sampling_mining_workflows_dsl/operator/Operator.py +165 -0
  46. sampling_mining_workflows_dsl/operator/OperatorBuilder.py +191 -0
  47. sampling_mining_workflows_dsl/operator/OperatorFactory.py +76 -0
  48. sampling_mining_workflows_dsl/operator/clustering/GroupingOperator.py +36 -0
  49. sampling_mining_workflows_dsl/operator/clustering/SubWorkflowOperatorBuilder.py +76 -0
  50. sampling_mining_workflows_dsl/operator/selection/SelectionOperator.py +5 -0
  51. sampling_mining_workflows_dsl/operator/selection/filter/FilterOperator.py +30 -0
  52. sampling_mining_workflows_dsl/operator/selection/sampling/SamplingOperator.py +11 -0
  53. sampling_mining_workflows_dsl/operator/selection/sampling/automatic/AutomaticSamplingOperator.py +9 -0
  54. sampling_mining_workflows_dsl/operator/selection/sampling/automatic/RandomSelectionOperator.py +24 -0
  55. sampling_mining_workflows_dsl/operator/selection/sampling/automatic/RandomSelectionPartitionOperator.py +62 -0
  56. sampling_mining_workflows_dsl/operator/selection/sampling/automatic/SystematicRandomSelectionOperator.py +14 -0
  57. sampling_mining_workflows_dsl/operator/selection/sampling/automatic/SystematicSelectionOperator.py +31 -0
  58. sampling_mining_workflows_dsl/operator/selection/sampling/manual/InteractiveManualSamplingOperator.py +40 -0
  59. sampling_mining_workflows_dsl/operator/selection/sampling/manual/ManualSamplingOperator.py +28 -0
  60. sampling_mining_workflows_dsl/operator/set_algebra/ExternalSetOperator.py +33 -0
  61. sampling_mining_workflows_dsl/operator/set_algebra/InternalSetOperator.py +47 -0
  62. sampling_mining_workflows_dsl/operator/set_algebra/SetOperator.py +30 -0
  63. sampling_mining_workflows_dsl/operator/set_algebra/external_set_operator/DifferenceOperator.py +17 -0
  64. sampling_mining_workflows_dsl/operator/set_algebra/external_set_operator/IntersectionOperator.py +19 -0
  65. sampling_mining_workflows_dsl/operator/set_algebra/external_set_operator/UnionOperator.py +17 -0
  66. sampling_mining_workflows_dsl/operator/set_algebra/internal_set_operator/DifferenceOperator.py +17 -0
  67. sampling_mining_workflows_dsl/operator/set_algebra/internal_set_operator/IntersectionOperator.py +18 -0
  68. sampling_mining_workflows_dsl/operator/set_algebra/internal_set_operator/UnionOperator.py +17 -0
  69. sampling_mining_workflows_dsl/operator/set_algebra/set_operator/DifferenceOperator.py +15 -0
  70. sampling_mining_workflows_dsl/operator/set_algebra/set_operator/IntersectionOperator.py +18 -0
  71. sampling_mining_workflows_dsl/operator/set_algebra/set_operator/UnionOperator.py +17 -0
  72. sampling_mining_workflows_dsl/test/ __init__.py +0 -0
  73. sampling_mining_workflows_dsl/test/Workflow_simple.py +38 -0
  74. sampling_mining_workflows_dsl/test/input.json +401 -0
  75. sampling_mining_workflows_dsl/toolbox.py +42 -0
  76. sampling_mining_workflows_dsl-0.0.1.dist-info/METADATA +236 -0
  77. sampling_mining_workflows_dsl-0.0.1.dist-info/RECORD +79 -0
  78. sampling_mining_workflows_dsl-0.0.1.dist-info/WHEEL +4 -0
  79. sampling_mining_workflows_dsl-0.0.1.dist-info/licenses/LICENSE.txt +674 -0
@@ -0,0 +1,66 @@
1
+ from typing import TYPE_CHECKING, TypeVar
2
+
3
+ from sampling_mining_workflows_dsl.constraint.Constraint import Constraint
4
+ from sampling_mining_workflows_dsl.element.Element import Element
5
+ from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
6
+ from sampling_mining_workflows_dsl.Workflow import Workflow
7
+
8
+ if TYPE_CHECKING:
9
+ from sampling_mining_workflows_dsl.constraint.BoolConstraint import BoolConstraint
10
+
11
+ T = TypeVar("T")
12
+
13
+
14
+ class BoolConstraintString(Constraint[T]):
15
+ def __init__(
16
+ self,
17
+ workflow: Workflow,
18
+ string_constraint: str,
19
+ *targeted_metadatas: Metadata[T],
20
+ ):
21
+ super().__init__(workflow, *targeted_metadatas)
22
+
23
+ self.string_constraint = string_constraint
24
+ self.or_constraint: Constraint | None = None
25
+ self.and_constraint: Constraint | None = None
26
+
27
+ def set_workflow(self, workflow: Workflow):
28
+ self.workflow = workflow
29
+ worflow_metadatas = workflow.get_all_Metadata()
30
+ self.targeted_metadatas = tuple(worflow_metadatas)
31
+ return self
32
+
33
+ def is_satisfied(self, element: Element) -> bool:
34
+ all_metadata = self.workflow.get_all_Metadata()
35
+
36
+ # Retrieve matching Metadata objects by checking if their names exist in the string_constraint
37
+ matching_metadata = [
38
+ metadata
39
+ for metadata in all_metadata
40
+ if metadata.name in self.string_constraint
41
+ ]
42
+
43
+ metadata_values = {}
44
+ for metadata in matching_metadata:
45
+ metadata_value = element.get_metadata_value(metadata)
46
+ if metadata_value:
47
+ metadata_values[metadata.name] = metadata_value.get_value()
48
+
49
+ # Evaluate the string_constraint
50
+ constraint_result = False
51
+ try:
52
+ constraint_result = eval(self.string_constraint, {}, metadata_values)
53
+ except Exception as e:
54
+ print(f"Error evaluating string_constraint: {e}")
55
+ return constraint_result
56
+
57
+ def or_(self, other: "BoolConstraint") -> "BoolConstraint":
58
+ self.or_constraint = other
59
+ return other
60
+
61
+ def and_(self, other: "BoolConstraint") -> "BoolConstraint":
62
+ self.and_constraint = other
63
+ return other
64
+
65
+ def get_string_constraint(self) -> str:
66
+ return self.string_constraint
@@ -0,0 +1,16 @@
1
+ from abc import ABC, abstractmethod
2
+ from typing import TypeVar
3
+
4
+ from sampling_mining_workflows_dsl.element.Element import Element
5
+ from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
6
+
7
+ T = TypeVar("T")
8
+
9
+
10
+ class Comparator[T](ABC):
11
+ def __init__(self, targeted_metadata: Metadata[T]):
12
+ self.targeted_metadata = targeted_metadata
13
+
14
+ @abstractmethod
15
+ def compare(self, a: Element, b: Element) -> int:
16
+ pass
@@ -0,0 +1,23 @@
1
+ import abc
2
+ from typing import TYPE_CHECKING, TypeVar
3
+
4
+ from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
5
+
6
+ if TYPE_CHECKING:
7
+ from sampling_mining_workflows_dsl.Workflow import Workflow
8
+
9
+ T = TypeVar("T")
10
+
11
+
12
+ class Constraint[T]:
13
+ def __init__(self, workflow: "Workflow", *targeted_metadatas: Metadata[T]):
14
+ self.targeted_metadatas = targeted_metadatas
15
+ self.workflow = workflow
16
+
17
+ @abc.abstractmethod
18
+ def is_satisfied(self, element):
19
+ pass
20
+
21
+ def set_workflow(self, workflow: "Workflow"):
22
+ self.workflow = workflow
23
+ return self
@@ -0,0 +1,17 @@
1
+ from typing import TypeVar
2
+
3
+ from sampling_mining_workflows_dsl.constraint.Comparator import Comparator
4
+ from sampling_mining_workflows_dsl.element.Element import Element
5
+
6
+ T = TypeVar("T")
7
+
8
+
9
+ class NaturalComparator[T](Comparator[T]):
10
+ """
11
+ Generic comparator for any type T that supports < and >.
12
+ """
13
+
14
+ def compare(self, a: Element, b: Element) -> int:
15
+ va = self.targeted_metadata.extract(a)
16
+ vb = self.targeted_metadata.extract(b)
17
+ return (va > vb) - (va < vb)
@@ -0,0 +1,45 @@
1
+ from abc import ABC, abstractmethod
2
+ from typing import TypeVar
3
+
4
+ from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
5
+ from sampling_mining_workflows_dsl.metadata.MetadataValue import MetadataValue
6
+
7
+ T = TypeVar("T")
8
+
9
+
10
+ class Element(ABC):
11
+ def __init__(self):
12
+ self.metadata: dict[Metadata, MetadataValue] = {}
13
+
14
+ def __hash__(self) -> int:
15
+ return hash(frozenset(self.metadata.items()))
16
+
17
+ def __eq__(self, other) -> bool:
18
+ if not isinstance(other, Element):
19
+ return False
20
+ return self.metadata == other.metadata
21
+
22
+ def get_metadata_value(self, metadata: Metadata[T]) -> MetadataValue[T]:
23
+ if metadata not in self.metadata:
24
+ raise RuntimeError(f"Missing metadata {metadata.name}")
25
+ return self.metadata[metadata]
26
+
27
+ def add_metadata_values(self, metadata_values: list[MetadataValue]):
28
+ for metadata_value in metadata_values:
29
+ self.metadata[metadata_value.get_metadata()] = metadata_value
30
+
31
+ def add_metadata_value(self, metadata_value: MetadataValue):
32
+ self.metadata[metadata_value.get_metadata()] = metadata_value
33
+
34
+ def get_all_metadata_values(self) -> dict[Metadata, MetadataValue]:
35
+ return self.metadata.copy()
36
+
37
+ def get_raw_metadata_values(self) -> dict[str, T]:
38
+ self
39
+
40
+ def get_id(self) -> str:
41
+ raise NotImplementedError("Subclasses must implement get_id method")
42
+
43
+ @abstractmethod
44
+ def to_string(self, level: int) -> str:
45
+ pass
@@ -0,0 +1,17 @@
1
+ from abc import ABC, abstractmethod
2
+
3
+ from sampling_mining_workflows_dsl.element import Set
4
+ from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
5
+
6
+
7
+ class Loader(ABC):
8
+ def __init__(self, *metadatas: Metadata | None):
9
+ # Assuming the first metadata is the ID, store its name for further use
10
+ self.metadata_id_name = metadatas[0].name
11
+ self.metadatas: dict[str, Metadata] = {}
12
+ for metadata in metadatas:
13
+ self.metadatas[metadata.name] = metadata
14
+
15
+ @abstractmethod
16
+ def load_set(self) -> Set:
17
+ pass
@@ -0,0 +1,28 @@
1
+ from sampling_mining_workflows_dsl.element.Element import Element
2
+ from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
3
+
4
+
5
+ class Repository(Element):
6
+ def __init__(self, id_metadata: Metadata[str]):
7
+ super().__init__()
8
+ self.id = id_metadata
9
+
10
+ def get_id(self) -> str:
11
+ return self.get_metadata_value(self.id).get_value()
12
+
13
+ def __hash__(self) -> int:
14
+ return hash(self.get_id())
15
+
16
+ def __eq__(self, other):
17
+ if not super().__eq__(other):
18
+ return False
19
+ if not isinstance(other, Repository):
20
+ return False
21
+ return self.get_id() == other.get_id()
22
+
23
+ def __str__(self) -> str:
24
+ return self.to_string(0)
25
+
26
+ def to_string(self, level: int = 0) -> str:
27
+ indent = " " * level
28
+ return f"{indent}{self.get_metadata_value(self.id)}"
@@ -0,0 +1,210 @@
1
+ import random
2
+ from functools import cmp_to_key
3
+ from collections import OrderedDict
4
+
5
+ from sampling_mining_workflows_dsl.constraint.Comparator import Comparator
6
+ from sampling_mining_workflows_dsl.element.Element import Element
7
+
8
+
9
+ class Set(Element):
10
+ def __init__(self):
11
+ super().__init__()
12
+ self.elements = OrderedDict()
13
+ self.ids= set()
14
+ self.set_id = None
15
+
16
+ def __hash__(self):
17
+ return hash(tuple(self.elements.items)) + super().__hash__()
18
+
19
+ def __eq__(self, other):
20
+ if not super().__eq__(other):
21
+ return False
22
+ if not isinstance(other, Set):
23
+ return False
24
+ return self.elements == other.elements
25
+
26
+ def get_element_by_index(self, index: int) -> Element:
27
+ if index < 0 or index >= len(self.elements):
28
+ raise IndexError("Index out of range")
29
+ return list(self.elements.values())[index]
30
+ def remove_all_elements(self) -> "Set":
31
+ self.elements.clear()
32
+ self.ids.clear()
33
+ return self
34
+ def add_element(self, element: Element) -> "Set":
35
+ if not element.get_id() in self.ids:
36
+ self.elements[element.get_id()]=element
37
+ self.ids.add(element.get_id())
38
+ else :
39
+ print(f"{element.get_id()} Already present")
40
+
41
+ return self
42
+
43
+ def sort_by_metadata(self, metadata_name: str, comparator: Comparator, reverse=False) -> "Set":
44
+ # Sort the items of the OrderedDict
45
+ sorted_items = sorted(
46
+ self.elements.items(),
47
+ key=cmp_to_key(lambda x, y: comparator.compare(x[1], y[1])),
48
+ reverse=reverse
49
+ )
50
+
51
+ # Rebuild as OrderedDict
52
+ self.elements = OrderedDict(sorted_items)
53
+ return self
54
+
55
+ def get_depth(self) -> int:
56
+ max_depth = 1
57
+ for element in self.elements.values():
58
+ if isinstance(element, Set):
59
+ max_depth = max(max_depth, 1 + element.get_depth())
60
+ return max_depth
61
+
62
+ def union(self, other: "Set") -> "Set":
63
+ for element in other.elements.values():
64
+ self.add_element(element)
65
+ return self
66
+
67
+ def intersection(self, other: "Set") -> "Set":
68
+ common_elements = Set()
69
+ for id, element in self.elements.items():
70
+ if id in other.elements.keys():
71
+ common_elements.add_element(element)
72
+ return common_elements
73
+
74
+ def difference(self, other: "Set") -> "Set":
75
+ """Return a new set with elements in this set but not in other"""
76
+ diff_elements = Set()
77
+ for id, element in self.elements.items():
78
+ if id not in other.elements.keys():
79
+ diff_elements.add_element(element)
80
+ return diff_elements
81
+
82
+ def symmetric_difference(self, other: "Set") -> "Set":
83
+ """Return a new set with elements in either set but not in both"""
84
+ sym_diff = Set()
85
+
86
+ # Add elements from this set that are not in other
87
+ for id, element in self.elements.items():
88
+ if id not in other.elements.keys():
89
+ sym_diff.add_element(element)
90
+
91
+ # Add elements from other set that are not in this
92
+ for id, element in other.elements.items():
93
+ if id not in self.elements.keys():
94
+ sym_diff.add_element(element)
95
+
96
+ return sym_diff
97
+
98
+ def is_subset(self, other: "Set") -> bool:
99
+ """Return True if all elements in this set are also in other"""
100
+ for id in self.elements.keys():
101
+ if id not in other.elements.keys():
102
+ return False
103
+ return True
104
+
105
+ def is_superset(self, other: "Set") -> bool:
106
+ """Return True if all elements in other set are also in this set"""
107
+ return other.is_subset(self)
108
+
109
+ def is_disjoint(self, other: "Set") -> bool:
110
+ """Return True if this set and other have no elements in common"""
111
+ for id in self.elements.keys():
112
+ if id in other.elements.keys():
113
+ return False
114
+ return True
115
+
116
+ def is_empty(self) -> bool:
117
+ """Return True if the set is empty"""
118
+ return len(self.elements) == 0
119
+
120
+
121
+
122
+ def size(self) -> int:
123
+ return len(self.elements)
124
+
125
+ def get_element(self, id: str) -> Element:
126
+ if not id in self.elements.keys():
127
+ raise RuntimeError(f"Element with id {id} not found in the set")
128
+ return self.elements.get(id)
129
+
130
+ def set_id(self, set_id: str) -> "Set":
131
+ self.set_id = set_id
132
+ return self
133
+
134
+ def get_id(self):
135
+ if self.set_id is not None:
136
+ return self.set_id
137
+
138
+ set_id = ""
139
+ for id in self.elements.keys():
140
+ set_id = set_id + "_" + str(id)
141
+ return set_id
142
+
143
+ def flatten_set(self) -> "Set":
144
+ flattened = Set()
145
+ for element in self.get_elements():
146
+ if isinstance(element, Set):
147
+ # Recursively flatten nested Sets
148
+ flattened.union(element.flatten_set())
149
+ else:
150
+ # Add non-Set, non-list elements directly
151
+ flattened.add_element(element)
152
+ return flattened
153
+
154
+
155
+ def get_random_subset(self, subset_size: int, seed: int) -> "Set":
156
+ elements_list = self.get_elements()
157
+ if subset_size > len(elements_list):
158
+ print(
159
+ f"Caution, subset size is larger than the size of the original set, subset size: {subset_size}, current set size: {len(elements_list)}"
160
+ )
161
+ subset_size = len(elements_list)
162
+
163
+ random.seed(seed)
164
+ random_indices = random.sample(range(len(elements_list)), subset_size)
165
+ original_array = list(elements_list)
166
+
167
+ result = Set()
168
+ for index in random_indices:
169
+ result.add_element(original_array[index])
170
+
171
+ return result
172
+
173
+ def get_elements(self) -> list[Element]:
174
+ return list(self.elements.values())
175
+
176
+ def clone(self) -> "Set":
177
+ cloned_set = Set()
178
+ for element in self.get_elements():
179
+ cloned_set.add_element(element)
180
+ return cloned_set
181
+
182
+ def __str__(self) -> str:
183
+ return self.to_string(0)
184
+
185
+ def to_string(self, level: int = 0) -> str:
186
+ truncate_after = 10
187
+ indent = " " * level
188
+
189
+ result = f"{indent}(size={len(self.elements)})["
190
+ elements_list = self.get_elements()
191
+ element_to_print = min(truncate_after, len(self.elements))
192
+
193
+ for i in range(element_to_print):
194
+ next_element = elements_list[i]
195
+
196
+ if isinstance(next_element, Set):
197
+ # Recursively call to_string for nested Sets
198
+ result += f"\n{next_element.to_string(level + 4)}"
199
+ else:
200
+ result += str(next_element)
201
+
202
+ if i != element_to_print - 1:
203
+ result += ","
204
+
205
+ if len(self.elements) > truncate_after:
206
+ result += "...]"
207
+ else:
208
+ result += "]"
209
+
210
+ return result
@@ -0,0 +1,9 @@
1
+ from abc import ABC, abstractmethod
2
+
3
+ from sampling_mining_workflows_dsl.element.Set import Set
4
+
5
+
6
+ class Writer(ABC):
7
+ @abstractmethod
8
+ def write_set(self, set: Set):
9
+ pass
@@ -0,0 +1,67 @@
1
+ import csv
2
+ from pathlib import Path
3
+ from typing import TYPE_CHECKING, Any
4
+ import logging
5
+ from sampling_mining_workflows_dsl.element.Loader import Loader
6
+ from sampling_mining_workflows_dsl.element.Repository import Repository
7
+ from sampling_mining_workflows_dsl.element.Set import Set
8
+ from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
9
+
10
+ if TYPE_CHECKING:
11
+ from sampling_mining_workflows_dsl.metadata.MetadataValue import MetadataValue
12
+
13
+
14
+ class CsvLoader(Loader):
15
+ def __init__(self, set_path: Path, *metadatas: Metadata):
16
+ super().__init__(*metadatas)
17
+ self.set_path = set_path
18
+ self.set = Set()
19
+
20
+ def load_set(self) -> Set:
21
+ try:
22
+ if self.set_path.is_dir():
23
+ csv_files = sorted(self.set_path.glob("*.csv"))
24
+ if not csv_files:
25
+ raise RuntimeError(
26
+ f"No CSV files found in directory: {self.set_path}"
27
+ )
28
+ else:
29
+ if not self.set_path.exists():
30
+ raise RuntimeError(f"File not found: {self.set_path}")
31
+ csv_files = [self.set_path]
32
+
33
+ for csv_file in csv_files:
34
+ print(f"Loading CSV file: {csv_file}")
35
+ with csv_file.open("r", newline="") as csvfile:
36
+ reader = csv.DictReader(csvfile)
37
+ for row in reader:
38
+ try:
39
+ repository = self.create_repository_from_map(row)
40
+ if repository is not None:
41
+ self.set.add_element(repository)
42
+ except Exception as e:
43
+ logging.info(f"Row skipped due to {e} : {row}")
44
+ return self.set
45
+ except OSError as e:
46
+ raise RuntimeError("Error reading the CSV file", e) from e
47
+
48
+ def create_repository_from_map(self, csv_row: dict[str, Any]) -> Repository:
49
+ id_metadata_value = csv_row.get(self.metadata_id_name)
50
+
51
+ if id_metadata_value is None or id_metadata_value == "":
52
+ raise ValueError(f"Invalid ID {self.metadata_id_name}")
53
+
54
+ repo = Repository(self.metadatas.get(self.metadata_id_name))
55
+
56
+ metadata_values: list[MetadataValue] = []
57
+ for metadata in self.metadatas.values():
58
+ try:
59
+ metadata_value = metadata.create_metadata_value(csv_row.get(metadata.name))
60
+ metadata_values.append(metadata_value)
61
+ except Exception as e:
62
+ raise ValueError(
63
+ f"Error creating metadata value for '{metadata.name}' with value '{csv_row.get(metadata.name)}'",
64
+ ) from e
65
+
66
+ repo.add_metadata_values(metadata_values)
67
+ return repo
@@ -0,0 +1,84 @@
1
+ import json
2
+ from pathlib import Path
3
+ from typing import TYPE_CHECKING, Any
4
+ import logging
5
+ from sampling_mining_workflows_dsl.element.Loader import Loader
6
+ from sampling_mining_workflows_dsl.element.Repository import Repository
7
+ from sampling_mining_workflows_dsl.element.Set import Set
8
+ from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
9
+
10
+ if TYPE_CHECKING:
11
+ from sampling_mining_workflows_dsl.metadata.MetadataValue import MetadataValue
12
+
13
+
14
+ class JsonLoader(Loader):
15
+ def __init__(self, set_path: Path, *metadatas: Metadata):
16
+ super().__init__(*metadatas)
17
+ self.set_path = set_path
18
+ self.set = Set()
19
+
20
+ def load_set(self) -> Set:
21
+ try:
22
+ if self.set_path.is_dir():
23
+ json_files = sorted(self.set_path.glob("*.json"))
24
+ if not json_files:
25
+ raise RuntimeError(
26
+ f"No JSON files found in directory: {self.set_path}"
27
+ )
28
+ else:
29
+ if not self.set_path.exists():
30
+ raise RuntimeError(f"File not found: {self.set_path}")
31
+ json_files = [self.set_path]
32
+
33
+ for json_file in json_files:
34
+ print(f"Loading JSON file: {json_file}")
35
+ with json_file.open("r") as reader:
36
+ json_list: list[dict[str, Any]] = json.load(reader)
37
+ for json_obj in json_list:
38
+ try:
39
+ repository = self.create_repository_from_map(json_obj)
40
+ if repository is not None:
41
+ self.set.add_element(repository)
42
+ except Exception as e:
43
+ logging.info(f"Row skipped due to {e} : {json_obj}")
44
+ return self.set
45
+ except OSError as e:
46
+ raise RuntimeError("Error reading the JSON file", e) from e
47
+
48
+ def create_repository_from_map(self, json_object: dict[str, Any]) -> Repository:
49
+ id_metadata_value = json_object.get(self.metadata_id_name)
50
+
51
+ if id_metadata_value is None or id_metadata_value == "":
52
+ raise ValueError(f"Invalid ID {self.metadata_id_name}")
53
+
54
+ repo = Repository(self.metadatas.get(self.metadata_id_name))
55
+
56
+ metadata_values: list[MetadataValue] = []
57
+ for metadata in self.metadatas.values():
58
+ try:
59
+ metadata_value = metadata.create_metadata_value(json_object.get(metadata.name))
60
+ metadata_values.append(metadata_value)
61
+ except Exception as e:
62
+ raise ValueError(
63
+ f"Error creating metadata value for '{metadata.name}' with value '{json_object.get(metadata.name)}'",
64
+ ) from e
65
+
66
+ repo.add_metadata_values(metadata_values)
67
+ return repo
68
+
69
+ @staticmethod
70
+ def parse_args(args: list[str]) -> dict[str, str]:
71
+ import argparse
72
+
73
+ default_input_path = Path(__file__).parent / "input.json"
74
+
75
+ parser = argparse.ArgumentParser(description="Sampling Workflow")
76
+ parser.add_argument(
77
+ "-i",
78
+ "--inputPath",
79
+ type=str,
80
+ default=str(default_input_path),
81
+ help="Input path file",
82
+ )
83
+ parsed_args = parser.parse_args(args)
84
+ return vars(parsed_args)
@@ -0,0 +1,14 @@
1
+ from pathlib import Path
2
+
3
+ from sampling_mining_workflows_dsl.element.loader.JsonLoader import JsonLoader
4
+ from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
5
+
6
+
7
+ class LoaderFactory:
8
+ @staticmethod
9
+ def json_loader(set_path: str, *metadatas: Metadata):
10
+ return JsonLoader(Path(set_path), *metadatas)
11
+
12
+ @staticmethod
13
+ def json_loader_from_path(set_path: Path, *metadatas: Metadata):
14
+ return JsonLoader(set_path, *metadatas)
@@ -0,0 +1,46 @@
1
+ import csv
2
+ from pathlib import Path
3
+
4
+ from sampling_mining_workflows_dsl.element.Repository import Repository
5
+ from sampling_mining_workflows_dsl.element.Set import Set
6
+
7
+
8
+ # Note that the depth of the set should be 1
9
+ class CsvWriter:
10
+ def __init__(self, set_path: str):
11
+ self.set_path = Path(set_path)
12
+
13
+ def write_set(self, set_obj: Set):
14
+ if not isinstance(set_obj, Set):
15
+ raise TypeError("Expected a Set object")
16
+
17
+ # Depth-1: elements are Repository instances
18
+ repositories = set_obj.elements.values()
19
+
20
+ if not repositories:
21
+ raise ValueError("The set is empty. Nothing to write.")
22
+
23
+ rows = [self._serialize_repository(repo) for repo in repositories]
24
+
25
+ headers = sorted(rows[0].keys())
26
+
27
+ try:
28
+ with self.set_path.open("w", encoding="utf-8", newline="") as f:
29
+ writer = csv.DictWriter(f, fieldnames=headers)
30
+ writer.writeheader()
31
+ writer.writerows(rows)
32
+ print(f"CSV has been written to {self.set_path}")
33
+ except OSError as e:
34
+ raise RuntimeError("Error while saving file") from e
35
+
36
+ def _serialize_repository(self, repo: Repository):
37
+ if not isinstance(repo, Repository):
38
+ raise TypeError("Expected Repository elements in the set")
39
+
40
+ result = {}
41
+ for meta, val in repo.get_all_metadata_values().items():
42
+ value = val.get_value()
43
+ if isinstance(value, list):
44
+ value = ";".join(map(str, value))
45
+ result[meta.name] = value
46
+ return result