sampling-mining-workflows-dsl 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sampling_mining_workflows_dsl/ __init__.py +0 -0
- sampling_mining_workflows_dsl/CompleteWorkflow.py +23 -0
- sampling_mining_workflows_dsl/Workflow.py +242 -0
- sampling_mining_workflows_dsl/WorkflowBuilder.py +11 -0
- sampling_mining_workflows_dsl/analysis/ChiSquareAnalysis.py +37 -0
- sampling_mining_workflows_dsl/analysis/CochranTest.py +44 -0
- sampling_mining_workflows_dsl/analysis/CochranWorkflowAnalysis.py +37 -0
- sampling_mining_workflows_dsl/analysis/CoverageTest.py +60 -0
- sampling_mining_workflows_dsl/analysis/DistributionWorkflowAnalysis.py +57 -0
- sampling_mining_workflows_dsl/analysis/HistAnalysis.py +201 -0
- sampling_mining_workflows_dsl/analysis/HistWorkflowAnalysis.py +72 -0
- sampling_mining_workflows_dsl/analysis/KSWorkflowAnalysis.py +217 -0
- sampling_mining_workflows_dsl/analysis/WorkflowAnalysis.py +7 -0
- sampling_mining_workflows_dsl/analysis/YamaneTest.py +16 -0
- sampling_mining_workflows_dsl/analysis/YamaneWorkflowAnalysis.py +27 -0
- sampling_mining_workflows_dsl/analysis/kolmogorov_smirnov.py +31 -0
- sampling_mining_workflows_dsl/constraint/BoolComparator.py +23 -0
- sampling_mining_workflows_dsl/constraint/BoolConstraint.py +50 -0
- sampling_mining_workflows_dsl/constraint/BoolConstraintString.py +66 -0
- sampling_mining_workflows_dsl/constraint/Comparator.py +16 -0
- sampling_mining_workflows_dsl/constraint/Constraint.py +23 -0
- sampling_mining_workflows_dsl/constraint/NaturalComparator.py +17 -0
- sampling_mining_workflows_dsl/element/Element.py +45 -0
- sampling_mining_workflows_dsl/element/Loader.py +17 -0
- sampling_mining_workflows_dsl/element/Repository.py +28 -0
- sampling_mining_workflows_dsl/element/Set.py +210 -0
- sampling_mining_workflows_dsl/element/Writer.py +9 -0
- sampling_mining_workflows_dsl/element/loader/CsvLoader.py +67 -0
- sampling_mining_workflows_dsl/element/loader/JsonLoader.py +84 -0
- sampling_mining_workflows_dsl/element/loader/LoaderFactory.py +14 -0
- sampling_mining_workflows_dsl/element/writer/CsvWriter.py +46 -0
- sampling_mining_workflows_dsl/element/writer/JsonWriter.py +58 -0
- sampling_mining_workflows_dsl/element/writer/WriterFactory.py +7 -0
- sampling_mining_workflows_dsl/exec_visualizer/WorkflowVisualizer.py +189 -0
- sampling_mining_workflows_dsl/github_seart/loader.py +17 -0
- sampling_mining_workflows_dsl/github_seart/metadata.py +106 -0
- sampling_mining_workflows_dsl/metadata/Metadata.py +117 -0
- sampling_mining_workflows_dsl/metadata/MetadataBoolean.py +14 -0
- sampling_mining_workflows_dsl/metadata/MetadataDate.py +75 -0
- sampling_mining_workflows_dsl/metadata/MetadataDict.py +33 -0
- sampling_mining_workflows_dsl/metadata/MetadataList.py +32 -0
- sampling_mining_workflows_dsl/metadata/MetadataNumber.py +29 -0
- sampling_mining_workflows_dsl/metadata/MetadataString.py +13 -0
- sampling_mining_workflows_dsl/metadata/MetadataValue.py +24 -0
- sampling_mining_workflows_dsl/operator/Operator.py +165 -0
- sampling_mining_workflows_dsl/operator/OperatorBuilder.py +191 -0
- sampling_mining_workflows_dsl/operator/OperatorFactory.py +76 -0
- sampling_mining_workflows_dsl/operator/clustering/GroupingOperator.py +36 -0
- sampling_mining_workflows_dsl/operator/clustering/SubWorkflowOperatorBuilder.py +76 -0
- sampling_mining_workflows_dsl/operator/selection/SelectionOperator.py +5 -0
- sampling_mining_workflows_dsl/operator/selection/filter/FilterOperator.py +30 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/SamplingOperator.py +11 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/automatic/AutomaticSamplingOperator.py +9 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/automatic/RandomSelectionOperator.py +24 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/automatic/RandomSelectionPartitionOperator.py +62 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/automatic/SystematicRandomSelectionOperator.py +14 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/automatic/SystematicSelectionOperator.py +31 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/manual/InteractiveManualSamplingOperator.py +40 -0
- sampling_mining_workflows_dsl/operator/selection/sampling/manual/ManualSamplingOperator.py +28 -0
- sampling_mining_workflows_dsl/operator/set_algebra/ExternalSetOperator.py +33 -0
- sampling_mining_workflows_dsl/operator/set_algebra/InternalSetOperator.py +47 -0
- sampling_mining_workflows_dsl/operator/set_algebra/SetOperator.py +30 -0
- sampling_mining_workflows_dsl/operator/set_algebra/external_set_operator/DifferenceOperator.py +17 -0
- sampling_mining_workflows_dsl/operator/set_algebra/external_set_operator/IntersectionOperator.py +19 -0
- sampling_mining_workflows_dsl/operator/set_algebra/external_set_operator/UnionOperator.py +17 -0
- sampling_mining_workflows_dsl/operator/set_algebra/internal_set_operator/DifferenceOperator.py +17 -0
- sampling_mining_workflows_dsl/operator/set_algebra/internal_set_operator/IntersectionOperator.py +18 -0
- sampling_mining_workflows_dsl/operator/set_algebra/internal_set_operator/UnionOperator.py +17 -0
- sampling_mining_workflows_dsl/operator/set_algebra/set_operator/DifferenceOperator.py +15 -0
- sampling_mining_workflows_dsl/operator/set_algebra/set_operator/IntersectionOperator.py +18 -0
- sampling_mining_workflows_dsl/operator/set_algebra/set_operator/UnionOperator.py +17 -0
- sampling_mining_workflows_dsl/test/ __init__.py +0 -0
- sampling_mining_workflows_dsl/test/Workflow_simple.py +38 -0
- sampling_mining_workflows_dsl/test/input.json +401 -0
- sampling_mining_workflows_dsl/toolbox.py +42 -0
- sampling_mining_workflows_dsl-0.0.1.dist-info/METADATA +236 -0
- sampling_mining_workflows_dsl-0.0.1.dist-info/RECORD +79 -0
- sampling_mining_workflows_dsl-0.0.1.dist-info/WHEEL +4 -0
- sampling_mining_workflows_dsl-0.0.1.dist-info/licenses/LICENSE.txt +674 -0
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from datetime import datetime, date
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from sampling_mining_workflows_dsl.element.Repository import Repository
|
|
6
|
+
from sampling_mining_workflows_dsl.element.Set import Set
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
# Note: Can handle both flat sets (depth 1) and nested sets (depth > 1)
|
|
10
|
+
class JsonWriter:
|
|
11
|
+
def __init__(self, set_path: str):
|
|
12
|
+
self.set_path = Path(set_path)
|
|
13
|
+
|
|
14
|
+
def write_set(self, set_obj: Set):
|
|
15
|
+
print(f"Writing set to JSON at: {self.set_path}")
|
|
16
|
+
if not isinstance(set_obj, Set):
|
|
17
|
+
raise TypeError("Expected a Set object")
|
|
18
|
+
|
|
19
|
+
elements = set_obj.elements.values()
|
|
20
|
+
|
|
21
|
+
if not elements:
|
|
22
|
+
raise ValueError("The set is empty. Nothing to write.")
|
|
23
|
+
|
|
24
|
+
rows = [self._serialize_element(elem) for elem in elements]
|
|
25
|
+
|
|
26
|
+
try:
|
|
27
|
+
with self.set_path.open("w", encoding="utf-8") as f:
|
|
28
|
+
json.dump(rows, f, indent=4, ensure_ascii=False)
|
|
29
|
+
print(f"JSON has been written to {self.set_path}")
|
|
30
|
+
except OSError as e:
|
|
31
|
+
raise RuntimeError("Error while saving file") from e
|
|
32
|
+
|
|
33
|
+
def _serialize_element(self, element):
|
|
34
|
+
"""Serialize an element, which can be a Repository or a Set."""
|
|
35
|
+
if isinstance(element, Set):
|
|
36
|
+
# Nested Set: serialize as an array of elements
|
|
37
|
+
return [self._serialize_element(elem) for elem in element.elements.values()]
|
|
38
|
+
elif isinstance(element, Repository):
|
|
39
|
+
return self._serialize_repository(element)
|
|
40
|
+
else:
|
|
41
|
+
raise TypeError(f"Unexpected element type: {type(element)}")
|
|
42
|
+
|
|
43
|
+
def _serialize_repository(self, repo: Repository):
|
|
44
|
+
if not isinstance(repo, Repository):
|
|
45
|
+
raise TypeError("Expected Repository elements in the set")
|
|
46
|
+
|
|
47
|
+
result = {}
|
|
48
|
+
for meta, val in repo.get_all_metadata_values().items():
|
|
49
|
+
value = val.get_value()
|
|
50
|
+
# Convert datetime/date to timestamp
|
|
51
|
+
if isinstance(value, (datetime, date)):
|
|
52
|
+
if isinstance(value, datetime):
|
|
53
|
+
value = int(value.timestamp())
|
|
54
|
+
else: # date object
|
|
55
|
+
value = int(datetime.combine(value, datetime.min.time()).timestamp())
|
|
56
|
+
# Keep lists as lists in JSON (no need to join like CSV)
|
|
57
|
+
result[meta.name] = value
|
|
58
|
+
return result
|
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from typing import cast
|
|
3
|
+
|
|
4
|
+
from graphviz import Digraph
|
|
5
|
+
|
|
6
|
+
from sampling_mining_workflows_dsl.constraint.BoolConstraintString import BoolConstraintString
|
|
7
|
+
from sampling_mining_workflows_dsl.constraint.BoolConstraint import BoolConstraint
|
|
8
|
+
|
|
9
|
+
from sampling_mining_workflows_dsl.operator.clustering.GroupingOperator import GroupingOperator
|
|
10
|
+
from sampling_mining_workflows_dsl.operator.selection.filter.FilterOperator import FilterOperator
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class WorkflowVisualizer:
|
|
14
|
+
def __init__(self, workflow,output_dir: str = "output_workflow_graph"):
|
|
15
|
+
self.workflow = workflow
|
|
16
|
+
self.global_set_counter = 0 # Global counter shared across all levels
|
|
17
|
+
if output_dir:
|
|
18
|
+
self.output_dir = output_dir
|
|
19
|
+
os.makedirs(self.output_dir, exist_ok=True)
|
|
20
|
+
else:
|
|
21
|
+
self.output_dir = os.path.dirname(__file__)
|
|
22
|
+
|
|
23
|
+
def generate_graph(self, output_file: str = "workflow_graph"):
|
|
24
|
+
svg_path = os.path.join(self.output_dir, output_file)
|
|
25
|
+
dot = Digraph(format="svg")
|
|
26
|
+
dot.attr(rankdir="LR")
|
|
27
|
+
|
|
28
|
+
# Add input node
|
|
29
|
+
input_size = self.workflow.get_workflow_input().flatten_set().size()
|
|
30
|
+
formatted_input_size = f"{input_size:,}".replace(',', ' ')
|
|
31
|
+
dot.node(
|
|
32
|
+
"InputSet",
|
|
33
|
+
label=f"INITIAL\nSAMPLING FRAME\nSize : {formatted_input_size}",
|
|
34
|
+
shape="box",
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
# Traverse and draw the workflow
|
|
38
|
+
self.global_set_counter = 0 # Reset counter for each graph generation
|
|
39
|
+
last_nodes = self._add_nodes_and_edges(
|
|
40
|
+
dot, self.workflow, parent_names=["InputSet"]
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
# Add output node
|
|
44
|
+
output_set = self.workflow.get_workflow_output().flatten_set()
|
|
45
|
+
output_size = output_set.size()
|
|
46
|
+
formatted_output_size = f"{output_size:,}".replace(',', ' ')
|
|
47
|
+
dot.node("OutputSet", label=f"SAMPLE\nSize : {formatted_output_size}", shape="box")
|
|
48
|
+
|
|
49
|
+
# Link last operator(s) to output
|
|
50
|
+
for node in last_nodes:
|
|
51
|
+
dot.edge(node, "OutputSet")
|
|
52
|
+
|
|
53
|
+
# Render
|
|
54
|
+
dot.render(svg_path, format="svg", cleanup=True)
|
|
55
|
+
print(f"Workflow graph saved to {svg_path}")
|
|
56
|
+
|
|
57
|
+
# HTML wrapper
|
|
58
|
+
self.create_html_page(svg_path)
|
|
59
|
+
|
|
60
|
+
def _add_nodes_and_edges(
|
|
61
|
+
self,
|
|
62
|
+
dot,
|
|
63
|
+
workflow,
|
|
64
|
+
level: int = 0,
|
|
65
|
+
workflow_number: int = 0,
|
|
66
|
+
op_number: int = 0,
|
|
67
|
+
parent_names=None,
|
|
68
|
+
) -> list[str]:
|
|
69
|
+
if parent_names is None:
|
|
70
|
+
parent_names = []
|
|
71
|
+
|
|
72
|
+
op = workflow.get_root()
|
|
73
|
+
last_nodes = []
|
|
74
|
+
|
|
75
|
+
while op is not None:
|
|
76
|
+
output_set = op.get_output()
|
|
77
|
+
node_name = f"{op.__class__.__name__}_{level}_{workflow_number}_{op_number}"
|
|
78
|
+
|
|
79
|
+
# Only increment counter for non-grouping operators
|
|
80
|
+
if not isinstance(op, GroupingOperator):
|
|
81
|
+
self.global_set_counter += 1
|
|
82
|
+
|
|
83
|
+
# Label formatting with global set number
|
|
84
|
+
if isinstance(op, FilterOperator) :
|
|
85
|
+
if isinstance(op.get_constraint(), BoolConstraintString):
|
|
86
|
+
constraint = cast(
|
|
87
|
+
"BoolConstraintString", op.get_constraint()
|
|
88
|
+
).get_string_constraint()
|
|
89
|
+
formatted_size = f"{output_set.flatten_set().size():,}".replace(',', ' ')
|
|
90
|
+
label = f"Filter Operator\n{constraint}\n\nSET#{self.global_set_counter}\n Size : {formatted_size}"
|
|
91
|
+
elif isinstance(op.get_constraint(), BoolConstraint):
|
|
92
|
+
constraint = cast(
|
|
93
|
+
"BoolConstraint", op.get_constraint()
|
|
94
|
+
)
|
|
95
|
+
function=constraint.constraint
|
|
96
|
+
|
|
97
|
+
metadata=constraint.targeted_metadatas[0].name
|
|
98
|
+
# Get method name from qualname
|
|
99
|
+
method_name = "unknown"
|
|
100
|
+
if hasattr(function, '__qualname__'):
|
|
101
|
+
qualname = function.__qualname__
|
|
102
|
+
# Extract method name (e.g., "MetadataDate.is_after.<locals>.<lambda>" -> "is_after")
|
|
103
|
+
parts = qualname.split('.')
|
|
104
|
+
for part in parts:
|
|
105
|
+
if part.startswith('is_') or part in ['greater_than', 'less_than', 'equal']:
|
|
106
|
+
method_name = part
|
|
107
|
+
break
|
|
108
|
+
|
|
109
|
+
# Get argument from closure
|
|
110
|
+
argument_value = "unknown"
|
|
111
|
+
if hasattr(function, '__closure__') and function.__closure__:
|
|
112
|
+
for cell in function.__closure__:
|
|
113
|
+
try:
|
|
114
|
+
cell_content = cell.cell_contents
|
|
115
|
+
# Skip metadata objects, look for actual values
|
|
116
|
+
if not hasattr(cell_content, 'name') and not hasattr(cell_content, 'type'):
|
|
117
|
+
# This might be the argument value
|
|
118
|
+
if isinstance(cell_content, (str, int, float, bool)):
|
|
119
|
+
argument_value = str(cell_content)
|
|
120
|
+
elif hasattr(cell_content, '__class__'):
|
|
121
|
+
# For dates, objects, etc.
|
|
122
|
+
argument_value = str(cell_content)
|
|
123
|
+
break
|
|
124
|
+
except (ValueError, AttributeError):
|
|
125
|
+
continue
|
|
126
|
+
formatted_size = f"{output_set.flatten_set().size():,}".replace(',', ' ')
|
|
127
|
+
|
|
128
|
+
label = f"Filter Operator\n{metadata}.{method_name}({argument_value})\n\nSET#{self.global_set_counter}\n Size : {formatted_size}"
|
|
129
|
+
elif isinstance(op, GroupingOperator):
|
|
130
|
+
label = f"Grouping\nOperator" # No number for grouping
|
|
131
|
+
else:
|
|
132
|
+
formatted_size = f"{output_set.flatten_set().size():,}".replace(',', ' ')
|
|
133
|
+
label = f"{op.__class__.__name__.replace('Operator', '')}\nOperator\n\nSET#{self.global_set_counter}\n Size : {formatted_size}"
|
|
134
|
+
|
|
135
|
+
dot.node(node_name, label=label, shape="box")
|
|
136
|
+
|
|
137
|
+
# Link all incoming parent nodes
|
|
138
|
+
for parent in parent_names:
|
|
139
|
+
dot.edge(parent, node_name)
|
|
140
|
+
|
|
141
|
+
# Handle grouping
|
|
142
|
+
if isinstance(op, GroupingOperator):
|
|
143
|
+
sub_last_nodes = []
|
|
144
|
+
for i, sub_workflow in enumerate(op.get_workflows()):
|
|
145
|
+
sub_nodes = self._add_nodes_and_edges(
|
|
146
|
+
dot,
|
|
147
|
+
sub_workflow,
|
|
148
|
+
level=level + 1,
|
|
149
|
+
workflow_number=i,
|
|
150
|
+
op_number=0,
|
|
151
|
+
parent_names=[node_name],
|
|
152
|
+
)
|
|
153
|
+
sub_last_nodes.extend(sub_nodes)
|
|
154
|
+
|
|
155
|
+
parent_names = (
|
|
156
|
+
sub_last_nodes # They become the parents of the next operator
|
|
157
|
+
)
|
|
158
|
+
last_nodes = sub_last_nodes
|
|
159
|
+
else:
|
|
160
|
+
parent_names = [node_name]
|
|
161
|
+
last_nodes = [node_name]
|
|
162
|
+
|
|
163
|
+
op = op.get_next_operator()
|
|
164
|
+
op_number += 1
|
|
165
|
+
|
|
166
|
+
return last_nodes
|
|
167
|
+
|
|
168
|
+
def create_html_page(self, svg_file: str, output_html: str = "workflow.html"):
|
|
169
|
+
html_path = os.path.join(self.output_dir, output_html)
|
|
170
|
+
|
|
171
|
+
html_content = f"""
|
|
172
|
+
<!DOCTYPE html>
|
|
173
|
+
<html lang="en">
|
|
174
|
+
<head>
|
|
175
|
+
<meta charset="UTF-8">
|
|
176
|
+
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
|
177
|
+
<title>Workflow Visualization</title>
|
|
178
|
+
</head>
|
|
179
|
+
<body>
|
|
180
|
+
<h1>Workflow Visualization</h1>
|
|
181
|
+
<div>
|
|
182
|
+
<embed src="{os.path.basename(svg_file)}.svg" type="image/svg+xml" style="width:100%; height:90vh;"></embed>
|
|
183
|
+
</div>
|
|
184
|
+
</body>
|
|
185
|
+
</html>
|
|
186
|
+
"""
|
|
187
|
+
with open(html_path, "w") as f:
|
|
188
|
+
f.write(html_content)
|
|
189
|
+
print(f"HTML page saved to {html_path}")
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
from typing import TYPE_CHECKING, Any
|
|
3
|
+
|
|
4
|
+
from sampling_mining_workflows_dsl.element.loader.CsvLoader import CsvLoader
|
|
5
|
+
from sampling_mining_workflows_dsl.element.Repository import Repository
|
|
6
|
+
from sampling_mining_workflows_dsl.element.Set import Set
|
|
7
|
+
from sampling_mining_workflows_dsl.github_seart.metadata import all_metadatas
|
|
8
|
+
|
|
9
|
+
if TYPE_CHECKING:
|
|
10
|
+
from sampling_mining_workflows_dsl.metadata.MetadataValue import MetadataValue
|
|
11
|
+
|
|
12
|
+
class SEARTGithubLoader(CsvLoader):
|
|
13
|
+
def __init__(self, set_path: Path):
|
|
14
|
+
super().__init__(*all_metadatas)
|
|
15
|
+
self.set_path = set_path
|
|
16
|
+
self.set = Set()
|
|
17
|
+
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
|
|
2
|
+
import json
|
|
3
|
+
from typing import TypedDict
|
|
4
|
+
|
|
5
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
6
|
+
from sampling_mining_workflows_dsl.metadata.MetadataValue import MetadataValue
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class LanguageMetrics(TypedDict):
|
|
10
|
+
commentLines: int
|
|
11
|
+
codeLines: int
|
|
12
|
+
blankLines: int
|
|
13
|
+
|
|
14
|
+
class MetadataMetrics(Metadata[dict[str, LanguageMetrics]]):
|
|
15
|
+
def __init__(self, name: str):
|
|
16
|
+
super().__init__(name, dict[str, LanguageMetrics])
|
|
17
|
+
|
|
18
|
+
def create_metadata_value(self, value):
|
|
19
|
+
# the raw value is list[dict[str, str|int]], convert it to dict[str, LanguageMetrics]
|
|
20
|
+
typed_value = {}
|
|
21
|
+
for item in json.loads(value):
|
|
22
|
+
lang = item["language"]
|
|
23
|
+
metrics = LanguageMetrics(
|
|
24
|
+
commentLines=item.get("commentLines", 0),
|
|
25
|
+
codeLines=item.get("codeLines", 0),
|
|
26
|
+
blankLines=item.get("blankLines", 0),
|
|
27
|
+
)
|
|
28
|
+
typed_value[lang] = metrics
|
|
29
|
+
|
|
30
|
+
return MetadataValue(self, typed_value)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
id = Metadata.of_string("id")
|
|
35
|
+
name = Metadata.of_string("name")
|
|
36
|
+
isFork = Metadata.of_boolean("isFork")
|
|
37
|
+
commits = Metadata.of_integer("commits")
|
|
38
|
+
branches = Metadata.of_integer("branches")
|
|
39
|
+
releases = Metadata.of_integer("releases")
|
|
40
|
+
forks = Metadata.of_integer("forks")
|
|
41
|
+
mainLanguage = Metadata.of_string("mainLanguage")
|
|
42
|
+
defaultBranch = Metadata.of_string("defaultBranch")
|
|
43
|
+
license = Metadata.of_string("license")
|
|
44
|
+
homepage = Metadata.of_string("homepage")
|
|
45
|
+
watchers = Metadata.of_integer("watchers")
|
|
46
|
+
stargazers = Metadata.of_integer("stargazers")
|
|
47
|
+
contributors = Metadata.of_integer("contributors")
|
|
48
|
+
size = Metadata.of_integer("size")
|
|
49
|
+
createdAt = Metadata.of_date("createdAt")
|
|
50
|
+
pushedAt = Metadata.of_date("pushedAt")
|
|
51
|
+
updatedAt = Metadata.of_date("updatedAt")
|
|
52
|
+
totalIssues = Metadata.of_integer("totalIssues")
|
|
53
|
+
openIssues = Metadata.of_integer("openIssues")
|
|
54
|
+
totalPullRequests = Metadata.of_integer("totalPullRequests")
|
|
55
|
+
openPullRequests = Metadata.of_integer("openPullRequests")
|
|
56
|
+
blankLines = Metadata.of_integer("blankLines")
|
|
57
|
+
codeLines = Metadata.of_integer("codeLines")
|
|
58
|
+
commentLines = Metadata.of_integer("commentLines")
|
|
59
|
+
metrics = MetadataMetrics("metrics")
|
|
60
|
+
lastCommit = Metadata.of_date("lastCommit")
|
|
61
|
+
lastCommitSHA = Metadata.of_string("lastCommitSHA")
|
|
62
|
+
hasWiki = Metadata.of_boolean("hasWiki")
|
|
63
|
+
isArchived = Metadata.of_boolean("isArchived")
|
|
64
|
+
isDisabled = Metadata.of_boolean("isDisabled")
|
|
65
|
+
isLocked = Metadata.of_boolean("isLocked")
|
|
66
|
+
languages = Metadata.of_dict("languages", str, int, lambda s: json.loads(s) if s else {})
|
|
67
|
+
labels = Metadata.of_list("labels", list[str], lambda s: s.split(";") if s else [])
|
|
68
|
+
topics = Metadata.of_list("topics", list[str], lambda s: s.split(";") if s else [])
|
|
69
|
+
|
|
70
|
+
all_metadatas = [
|
|
71
|
+
id,
|
|
72
|
+
name,
|
|
73
|
+
isFork,
|
|
74
|
+
commits,
|
|
75
|
+
branches,
|
|
76
|
+
releases,
|
|
77
|
+
forks,
|
|
78
|
+
mainLanguage,
|
|
79
|
+
defaultBranch,
|
|
80
|
+
license,
|
|
81
|
+
homepage,
|
|
82
|
+
watchers,
|
|
83
|
+
stargazers,
|
|
84
|
+
contributors,
|
|
85
|
+
size,
|
|
86
|
+
createdAt,
|
|
87
|
+
pushedAt,
|
|
88
|
+
updatedAt,
|
|
89
|
+
totalIssues,
|
|
90
|
+
openIssues,
|
|
91
|
+
totalPullRequests,
|
|
92
|
+
openPullRequests,
|
|
93
|
+
blankLines,
|
|
94
|
+
codeLines,
|
|
95
|
+
commentLines,
|
|
96
|
+
metrics,
|
|
97
|
+
lastCommit,
|
|
98
|
+
lastCommitSHA,
|
|
99
|
+
hasWiki,
|
|
100
|
+
isArchived,
|
|
101
|
+
isDisabled,
|
|
102
|
+
isLocked,
|
|
103
|
+
languages,
|
|
104
|
+
labels,
|
|
105
|
+
topics
|
|
106
|
+
]
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
from collections.abc import Callable
|
|
2
|
+
from typing import TypeVar
|
|
3
|
+
|
|
4
|
+
T = TypeVar("T")
|
|
5
|
+
V = TypeVar("V")
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class Metadata[T]:
|
|
9
|
+
def __init__(self, name: str, type_: type[T]):
|
|
10
|
+
self.name = name
|
|
11
|
+
self.type = type_
|
|
12
|
+
|
|
13
|
+
def __hash__(self):
|
|
14
|
+
return hash((self.name, self.type))
|
|
15
|
+
|
|
16
|
+
def __eq__(self, other):
|
|
17
|
+
if not isinstance(other, Metadata):
|
|
18
|
+
return False
|
|
19
|
+
return self.name == other.name and self.type == other.type
|
|
20
|
+
|
|
21
|
+
def is_of_type(self, obj):
|
|
22
|
+
return isinstance(obj, self.type)
|
|
23
|
+
|
|
24
|
+
def bool_constraint(self, constraint: Callable[[T], bool]):
|
|
25
|
+
from sampling_mining_workflows_dsl.constraint.BoolConstraint import BoolConstraint
|
|
26
|
+
|
|
27
|
+
return BoolConstraint(constraint, self)
|
|
28
|
+
|
|
29
|
+
def bool_comparator(self, comparator: Callable[[T, T], T]):
|
|
30
|
+
from sampling_mining_workflows_dsl.constraint.BoolComparator import BoolComparator
|
|
31
|
+
|
|
32
|
+
return BoolComparator(self, comparator)
|
|
33
|
+
|
|
34
|
+
def create_metadata_value(self, value):
|
|
35
|
+
typed_value=value
|
|
36
|
+
if not isinstance(value, self.type):
|
|
37
|
+
try:
|
|
38
|
+
typed_value=self.type(value) #instantiate the type to convert it
|
|
39
|
+
except Exception as e:
|
|
40
|
+
raise TypeError(f"Value {value} is not of type {self.type} and cannot be converted to it.",e) from e
|
|
41
|
+
from sampling_mining_workflows_dsl.metadata.MetadataValue import MetadataValue
|
|
42
|
+
|
|
43
|
+
return MetadataValue(self, typed_value)
|
|
44
|
+
|
|
45
|
+
@staticmethod
|
|
46
|
+
def of(name: str, type_: type[T]):
|
|
47
|
+
return Metadata(name, type_)
|
|
48
|
+
|
|
49
|
+
@staticmethod
|
|
50
|
+
def of_date(name: str):
|
|
51
|
+
from sampling_mining_workflows_dsl.metadata.MetadataDate import MetadataDate
|
|
52
|
+
|
|
53
|
+
return MetadataDate(name)
|
|
54
|
+
|
|
55
|
+
@staticmethod
|
|
56
|
+
def of_string(name: str):
|
|
57
|
+
from sampling_mining_workflows_dsl.metadata.MetadataString import MetadataString
|
|
58
|
+
|
|
59
|
+
return MetadataString(name)
|
|
60
|
+
|
|
61
|
+
@staticmethod
|
|
62
|
+
def of_integer(name: str):
|
|
63
|
+
from sampling_mining_workflows_dsl.metadata.MetadataNumber import MetadataNumber
|
|
64
|
+
|
|
65
|
+
return MetadataNumber(name, int)
|
|
66
|
+
|
|
67
|
+
@staticmethod
|
|
68
|
+
def of_double(name: str):
|
|
69
|
+
from sampling_mining_workflows_dsl.metadata.MetadataNumber import MetadataNumber
|
|
70
|
+
|
|
71
|
+
return MetadataNumber(name, float)
|
|
72
|
+
|
|
73
|
+
@staticmethod
|
|
74
|
+
def of_float(name: str):
|
|
75
|
+
from sampling_mining_workflows_dsl.metadata.MetadataNumber import MetadataNumber
|
|
76
|
+
|
|
77
|
+
return MetadataNumber(name, float)
|
|
78
|
+
|
|
79
|
+
@staticmethod
|
|
80
|
+
def of_long(name: str):
|
|
81
|
+
from sampling_mining_workflows_dsl.metadata.MetadataNumber import MetadataNumber
|
|
82
|
+
|
|
83
|
+
return MetadataNumber(name, int)
|
|
84
|
+
|
|
85
|
+
@staticmethod
|
|
86
|
+
def of_character(name: str):
|
|
87
|
+
from sampling_mining_workflows_dsl.metadata.MetadataString import MetadataString
|
|
88
|
+
|
|
89
|
+
return MetadataString(name)
|
|
90
|
+
|
|
91
|
+
@staticmethod
|
|
92
|
+
def of_short(name: str):
|
|
93
|
+
from sampling_mining_workflows_dsl.metadata.MetadataNumber import MetadataNumber
|
|
94
|
+
|
|
95
|
+
return MetadataNumber(name, int)
|
|
96
|
+
|
|
97
|
+
@staticmethod
|
|
98
|
+
def of_boolean(name: str):
|
|
99
|
+
from sampling_mining_workflows_dsl.metadata.MetadataBoolean import MetadataBoolean
|
|
100
|
+
|
|
101
|
+
return MetadataBoolean(name)
|
|
102
|
+
|
|
103
|
+
@staticmethod
|
|
104
|
+
def of_byte(name: str):
|
|
105
|
+
return Metadata(name, bytes)
|
|
106
|
+
|
|
107
|
+
@staticmethod
|
|
108
|
+
def of_list(name: str, type, transfo:Callable[[str],list[T]]=None):
|
|
109
|
+
from sampling_mining_workflows_dsl.metadata.MetadataList import MetadataList
|
|
110
|
+
|
|
111
|
+
return MetadataList(name,type,transfo)
|
|
112
|
+
|
|
113
|
+
@staticmethod
|
|
114
|
+
def of_dict(name: str, key_type, value_type, transfo:Callable[[str],dict[T, V]]=None):
|
|
115
|
+
from sampling_mining_workflows_dsl.metadata.MetadataDict import MetadataDict
|
|
116
|
+
|
|
117
|
+
return MetadataDict(name,key_type,value_type,transfo)
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
from sampling_mining_workflows_dsl.constraint.BoolConstraint import BoolConstraint
|
|
2
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class MetadataBoolean(Metadata[bool]):
|
|
6
|
+
def __init__(self, name: str):
|
|
7
|
+
super().__init__(name, bool)
|
|
8
|
+
|
|
9
|
+
def is_true(self) -> BoolConstraint:
|
|
10
|
+
return BoolConstraint(lambda x: x, self)
|
|
11
|
+
|
|
12
|
+
def is_false(self) -> BoolConstraint:
|
|
13
|
+
return BoolConstraint(lambda x: not x, self)
|
|
14
|
+
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
from datetime import datetime, date
|
|
2
|
+
from typing import Union
|
|
3
|
+
from dateutil import parser
|
|
4
|
+
from sampling_mining_workflows_dsl.constraint.BoolConstraint import BoolConstraint
|
|
5
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class MetadataDate(Metadata[Union[datetime, date]]):
|
|
9
|
+
|
|
10
|
+
def __init__(self, name: str):
|
|
11
|
+
super().__init__(name, Union[datetime, date])
|
|
12
|
+
|
|
13
|
+
def create_metadata_value(self, value):
|
|
14
|
+
if not isinstance(value, self.type):
|
|
15
|
+
try:
|
|
16
|
+
# Check if the string represents a timestamp (long)
|
|
17
|
+
str_value = str(value)
|
|
18
|
+
if str_value.isdigit():
|
|
19
|
+
# Try to parse as timestamp (seconds)
|
|
20
|
+
timestamp = int(str_value)
|
|
21
|
+
# Handle both seconds and milliseconds timestamps
|
|
22
|
+
if timestamp > 1e10: # Likely milliseconds
|
|
23
|
+
typed_value = datetime.fromtimestamp(timestamp / 1000)
|
|
24
|
+
else: # Likely seconds
|
|
25
|
+
typed_value = datetime.fromtimestamp(timestamp)
|
|
26
|
+
else:
|
|
27
|
+
# Parse as date string
|
|
28
|
+
typed_value = parser.parse(str_value)
|
|
29
|
+
except Exception as e:
|
|
30
|
+
raise TypeError(f"Value {value} date is not parsable" ) from e
|
|
31
|
+
else:
|
|
32
|
+
typed_value = value
|
|
33
|
+
|
|
34
|
+
from sampling_mining_workflows_dsl.metadata.MetadataValue import MetadataValue
|
|
35
|
+
|
|
36
|
+
return MetadataValue(self, typed_value)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def is_equal(self, value: Union[datetime, date]) -> BoolConstraint:
|
|
40
|
+
return BoolConstraint(None, lambda x: x == value, self)
|
|
41
|
+
|
|
42
|
+
def is_not_equal(self, value: Union[datetime, date]) -> BoolConstraint:
|
|
43
|
+
return BoolConstraint(None, lambda x: x != value, self)
|
|
44
|
+
|
|
45
|
+
def is_before(self, value: Union[datetime, date]) -> BoolConstraint:
|
|
46
|
+
return BoolConstraint(None, lambda x: x < value, self)
|
|
47
|
+
|
|
48
|
+
def is_after(self, value: Union[datetime, date]) -> BoolConstraint:
|
|
49
|
+
return BoolConstraint(None, lambda x: x > value, self)
|
|
50
|
+
|
|
51
|
+
def is_before_or_equal(self, value: Union[datetime, date]) -> BoolConstraint:
|
|
52
|
+
return BoolConstraint(None, lambda x: x <= value, self)
|
|
53
|
+
|
|
54
|
+
def is_after_or_equal(self, value: Union[datetime, date]) -> BoolConstraint:
|
|
55
|
+
return BoolConstraint(None, lambda x: x >= value, self)
|
|
56
|
+
|
|
57
|
+
def is_between(self, start: Union[datetime, date], end: Union[datetime, date]) -> BoolConstraint:
|
|
58
|
+
return BoolConstraint(None, lambda x: start <= x <= end, self)
|
|
59
|
+
|
|
60
|
+
def is_in_year(self, year: int) -> BoolConstraint:
|
|
61
|
+
return BoolConstraint(None, lambda x: x.year == year, self)
|
|
62
|
+
|
|
63
|
+
def is_in_month(self, month: int) -> BoolConstraint:
|
|
64
|
+
return BoolConstraint(None, lambda x: x.month == month, self)
|
|
65
|
+
|
|
66
|
+
def is_in_day(self, day: int) -> BoolConstraint:
|
|
67
|
+
return BoolConstraint(None, lambda x: x.day == day, self)
|
|
68
|
+
|
|
69
|
+
def is_weekday(self) -> BoolConstraint:
|
|
70
|
+
"""Check if the date is a weekday (Monday=0 to Friday=4)"""
|
|
71
|
+
return BoolConstraint(None, lambda x: x.weekday() < 5, self)
|
|
72
|
+
|
|
73
|
+
def is_weekend(self) -> BoolConstraint:
|
|
74
|
+
"""Check if the date is a weekend (Saturday=5 or Sunday=6)"""
|
|
75
|
+
return BoolConstraint(None, lambda x: x.weekday() >= 5, self)
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
from collections.abc import Callable
|
|
2
|
+
from typing import TypeVar
|
|
3
|
+
|
|
4
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
5
|
+
|
|
6
|
+
K, V = TypeVar("K"), TypeVar("V")
|
|
7
|
+
|
|
8
|
+
class MetadataDict(Metadata[dict[K, V]]):
|
|
9
|
+
transfo : Callable[[str],dict[K, V]]
|
|
10
|
+
|
|
11
|
+
def __init__(self, name: str, key_type: type[K], value_type: type[V], transfo:Callable[[str],dict[K, V]]=None):
|
|
12
|
+
self.full_type=dict[key_type, value_type]
|
|
13
|
+
self.transfo=transfo
|
|
14
|
+
super().__init__(name,dict)
|
|
15
|
+
|
|
16
|
+
def create_metadata_value(self, value):
|
|
17
|
+
#Check if value is a dict of K,V
|
|
18
|
+
result=value
|
|
19
|
+
if not self.checkType(result):
|
|
20
|
+
result=self.transfo(result)
|
|
21
|
+
|
|
22
|
+
if not self.checkType(result):
|
|
23
|
+
raise TypeError(f"Value {value} is not of type {self.full_type}")
|
|
24
|
+
return super().create_metadata_value(result)
|
|
25
|
+
|
|
26
|
+
def addTransformation(self,transfo:Callable[[str],dict[K, V]]):
|
|
27
|
+
self.transfo=transfo
|
|
28
|
+
return self
|
|
29
|
+
|
|
30
|
+
def checkType(self,obj):
|
|
31
|
+
if not isinstance(obj,dict):
|
|
32
|
+
return False
|
|
33
|
+
return all(isinstance(key, self.full_type.__args__[0]) and isinstance(value, self.full_type.__args__[1]) for key, value in obj.items())
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
from collections.abc import Callable
|
|
2
|
+
from typing import TypeVar
|
|
3
|
+
|
|
4
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
5
|
+
|
|
6
|
+
T = TypeVar("T")
|
|
7
|
+
class MetadataList(Metadata[list[T]]):
|
|
8
|
+
transfo : Callable[[str],list[T]]
|
|
9
|
+
|
|
10
|
+
def __init__(self, name: str,type,transfo:Callable[[str],list[T]]=None):
|
|
11
|
+
self.full_type=type
|
|
12
|
+
self.transfo=transfo
|
|
13
|
+
super().__init__(name,list)
|
|
14
|
+
|
|
15
|
+
def create_metadata_value(self, value):
|
|
16
|
+
#Check if value is a List of string of T
|
|
17
|
+
result=value
|
|
18
|
+
if not self.checkType(result):
|
|
19
|
+
result=self.transfo(result)
|
|
20
|
+
|
|
21
|
+
if not self.checkType(result):
|
|
22
|
+
raise TypeError(f"Value {value} is not of type {self.full_type}")
|
|
23
|
+
return super().create_metadata_value(result)
|
|
24
|
+
|
|
25
|
+
def addTransformation(self,transfo:Callable[[str],list[T]]):
|
|
26
|
+
self.transfo=transfo
|
|
27
|
+
return self
|
|
28
|
+
|
|
29
|
+
def checkType(self,obj):
|
|
30
|
+
if not isinstance(obj,list):
|
|
31
|
+
return False
|
|
32
|
+
return all(isinstance(item, self.full_type.__args__[0]) for item in obj)
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
from typing import TypeVar
|
|
2
|
+
|
|
3
|
+
from sampling_mining_workflows_dsl.constraint.BoolConstraint import BoolConstraint
|
|
4
|
+
from sampling_mining_workflows_dsl.metadata.Metadata import Metadata
|
|
5
|
+
|
|
6
|
+
T = TypeVar("T")
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class MetadataNumber(Metadata[T]):
|
|
10
|
+
def __init__(self, name: str, type_: type[T]):
|
|
11
|
+
super().__init__(name, type_)
|
|
12
|
+
|
|
13
|
+
def is_greater_than(self, value: T) -> BoolConstraint:
|
|
14
|
+
return BoolConstraint(None, lambda x: x > value, self)
|
|
15
|
+
|
|
16
|
+
def is_greater_or_equal_than(self, value: T) -> BoolConstraint:
|
|
17
|
+
return BoolConstraint(None, lambda x: x >= value, self)
|
|
18
|
+
|
|
19
|
+
def is_less_than(self, value: T) -> BoolConstraint:
|
|
20
|
+
return BoolConstraint(None, lambda x: x < value, self)
|
|
21
|
+
|
|
22
|
+
def is_less_or_equal_than(self, value: T) -> BoolConstraint:
|
|
23
|
+
return BoolConstraint(None, lambda x: x <= value, self)
|
|
24
|
+
|
|
25
|
+
def is_between(self, lower: T, upper: T) -> BoolConstraint:
|
|
26
|
+
return BoolConstraint(None, lambda x: lower <= x <= upper, self)
|
|
27
|
+
|
|
28
|
+
def is_equal(self, value: T) -> "BoolConstraint":
|
|
29
|
+
return BoolConstraint(None, lambda x: x == value, self)
|