mlrl-testbed-arff 0.12.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
mlrl/__init__.py ADDED
File without changes
File without changes
File without changes
File without changes
@@ -0,0 +1,6 @@
1
+ """
2
+ Author Michael Rapp (michael.rapp.ml@gmail.com)
3
+
4
+ Provides classes that allow to read input data from ARFF files.
5
+ """
6
+ from mlrl.testbed_arff.experiments.input.sources.source_arff import ArffFileSource
@@ -0,0 +1,239 @@
1
+ """
2
+ Author Michael Rapp (michael.rapp.ml@gmail.com)
3
+
4
+ Provides classes that allow reading datasets from ARFF files.
5
+ """
6
+ import logging as log
7
+
8
+ from functools import cached_property
9
+ from os import path
10
+ from typing import Any, List, Optional, Set
11
+ from xml.dom import minidom
12
+
13
+ import arff
14
+ import numpy as np
15
+
16
+ from scipy.sparse import coo_array, csc_array, sparray
17
+
18
+ from mlrl.testbed_arff.experiments.output.sinks.sink_arff import ArffFileSink
19
+
20
+ from mlrl.testbed_sklearn.experiments.dataset import Attribute, AttributeType, TabularDataset
21
+
22
+ from mlrl.testbed.experiments.dataset import Dataset
23
+ from mlrl.testbed.experiments.input.data import DatasetInputData
24
+ from mlrl.testbed.experiments.input.sources.source import DatasetFileSource
25
+ from mlrl.testbed.experiments.state import ExperimentState
26
+ from mlrl.testbed.util.io import open_readable_file
27
+
28
+
29
+ def normalize_attribute_name(name: str) -> str:
30
+ """
31
+ Normalizes the name of an attribute by removing forbidden characters.
32
+
33
+ :param name: The name to be normalized
34
+ :return: The normalized name
35
+ """
36
+ name = name.strip()
37
+ if name.startswith('\'') or name.startswith('"'):
38
+ name = name[1:]
39
+ if name.endswith('\'') or name.endswith('"'):
40
+ name = name[:(len(name) - 1)]
41
+ return name.replace('\\\'', '\'').replace('\\"', '"')
42
+
43
+
44
+ class ArffFileSource(DatasetFileSource):
45
+ """
46
+ Allows to read a dataset from an ARFF file.
47
+ """
48
+
49
+ class ArffFile:
50
+ """
51
+ Provides access to the content of an ARFF file.
52
+ """
53
+
54
+ def __init__(self, matrix: sparray, arff_attributes: List[Any], relation: str):
55
+ """
56
+ param matrix: The data matrix that is stored in the file
57
+ param arff_attributes: The attributes defined in the file
58
+ param relation: The @relation declaration contained in the file
59
+ """
60
+ self.matrix = matrix
61
+ self.arff_attributes = arff_attributes
62
+ self.relation = relation
63
+
64
+ @staticmethod
65
+ def from_file(file_path: str, sparse: bool, dtype: np.dtype) -> 'ArffFileSource.ArffFile':
66
+ """
67
+ Loads the content of an ARFF file.
68
+
69
+ :param file_path: The path to the ARFF file
70
+ :param sparse: True, if the ARFF file is given in sparse format, False otherwise. If the given format
71
+ is incorrect, an `arff.BadLayout` is raised
72
+ :param dtype: The type of the data matrix to be read from the file
73
+ :return: A dictionary that stores the content of the ARFF file
74
+ """
75
+ with open_readable_file(file_path) as arff_file:
76
+ arff_dict = arff.load(arff_file, encode_nominal=True, return_type=arff.COO if sparse else arff.DENSE)
77
+
78
+ if sparse:
79
+ data = arff_dict['data']
80
+ values = data[0]
81
+ row_indices = data[1]
82
+ col_indices = data[2]
83
+ shape = (max(row_indices) + 1, max(col_indices) + 1)
84
+ matrix = coo_array((values, (row_indices, col_indices)), shape=shape, dtype=dtype)
85
+ matrix = matrix.tocsc()
86
+ else:
87
+ data = arff_dict['data']
88
+ matrix = csc_array(data, dtype=dtype)
89
+
90
+ attributes = arff_dict['attributes']
91
+ relation = arff_dict['relation']
92
+ return ArffFileSource.ArffFile(matrix, arff_attributes=attributes, relation=relation)
93
+
94
+ @cached_property
95
+ def attributes(self) -> List[Attribute]:
96
+ """
97
+ A list that contains all attributes defined in the ARFF file.
98
+ """
99
+ attributes = []
100
+
101
+ for attribute in self.arff_attributes:
102
+ attribute_name = normalize_attribute_name(attribute[0])
103
+ type_definition = attribute[1]
104
+
105
+ if isinstance(type_definition, list):
106
+ attribute_type = AttributeType.NOMINAL
107
+ nominal_values = type_definition
108
+ else:
109
+ type_definition = str(type_definition).lower()
110
+ nominal_values = None
111
+
112
+ if type_definition == 'integer':
113
+ attribute_type = AttributeType.ORDINAL
114
+ elif type_definition in ('real', 'numeric'):
115
+ attribute_type = AttributeType.NUMERICAL
116
+ else:
117
+ raise ValueError('Encountered unsupported attribute type: ' + type_definition)
118
+
119
+ attribute = Attribute(name=attribute_name, attribute_type=attribute_type, nominal_values=nominal_values)
120
+ attributes.append(attribute)
121
+
122
+ return attributes
123
+
124
+ class ArffDataset:
125
+ """
126
+ Provides access to the content of an ARFF file and the corresponding Mulan XML file, if available.
127
+ """
128
+
129
+ def __parse_output_names_from_relation(self) -> Set[str]:
130
+ parameter_name = '-C '
131
+ arff_file = self.arff_file
132
+ relation = arff_file.relation
133
+ index = relation.index(parameter_name)
134
+ parameter_value = relation[index + len(parameter_name):]
135
+ index = parameter_value.find(' ')
136
+
137
+ if index >= 0:
138
+ parameter_value = parameter_value[:index]
139
+
140
+ num_outputs = int(parameter_value)
141
+ attributes = arff_file.attributes
142
+ return {normalize_attribute_name(attributes[i].name) for i in range(num_outputs)}
143
+
144
+ def __init__(self, arff_file: 'ArffFileSource.ArffFile', output_names: Optional[Set[str]]):
145
+ """
146
+ :param arff_file: The content of the ARFF file
147
+ :param output_names: The names of all outputs contained in the dataset
148
+ """
149
+ self.arff_file = arff_file
150
+ self.output_names = output_names if output_names else self.__parse_output_names_from_relation()
151
+
152
+ @staticmethod
153
+ def from_file(arff_file: 'ArffFileSource.ArffFile', file_path: str) -> 'ArffFileSource.ArffDataset':
154
+ """
155
+ Creates and returns an ARFF dataset from given ARFF file and a corresponding Mulan XML file, if available.
156
+
157
+
158
+ :param arff_file: The content of the ARFF file
159
+ :param file_path: The path to the XML file
160
+ :return: The ARFF dataset that has been created
161
+ """
162
+ if path.isfile(file_path):
163
+ log.debug('Parsing meta-data from file \"%s\"...', file_path)
164
+ xml_doc = minidom.parse(file_path)
165
+ tags = xml_doc.getElementsByTagName('label')
166
+ output_names = {normalize_attribute_name(tag.getAttribute('name')) for tag in tags}
167
+ else:
168
+ output_names = None
169
+ log.debug(
170
+ 'Mulan XML file \"%s\" does not exist. If possible, information about the dataset\'s outputs is '
171
+ + 'parsed from the ARFF file\'s @relation declaration as intended by the MEKA dataset format...',
172
+ file_path)
173
+
174
+ return ArffFileSource.ArffDataset(arff_file=arff_file, output_names=output_names)
175
+
176
+ @cached_property
177
+ def features(self) -> List[Attribute]:
178
+ """
179
+ A list that stores all features contained in the dataset.
180
+ """
181
+ return [attribute for attribute in self.arff_file.attributes if attribute.name not in self.output_names]
182
+
183
+ @cached_property
184
+ def outputs(self) -> List[Attribute]:
185
+ """
186
+ A list that stores all outputs contained in the dataset.
187
+ """
188
+ return [attribute for attribute in self.arff_file.attributes if attribute.name in self.output_names]
189
+
190
+ @property
191
+ def outputs_at_start(self) -> bool:
192
+ """
193
+ True, if the outputs are defined before the features, False otherwise.
194
+ """
195
+ attributes = self.arff_file.attributes
196
+ return attributes and attributes[0].name in self.output_names
197
+
198
+ @property
199
+ def feature_matrix(self) -> sparray:
200
+ """
201
+ The feature matrix contained in the dataset.
202
+ """
203
+ num_outputs = len(self.outputs)
204
+ matrix = self.arff_file.matrix
205
+ return matrix[:, num_outputs:] if self.outputs_at_start else matrix[:, :-num_outputs]
206
+
207
+ @property
208
+ def output_matrix(self) -> sparray:
209
+ """
210
+ The output matrix contained in the dataset.
211
+ """
212
+ num_outputs = len(self.outputs)
213
+ matrix = self.arff_file.matrix
214
+ return matrix[:, :num_outputs] if self.outputs_at_start else matrix[:, -num_outputs:]
215
+
216
+ @staticmethod
217
+ def __read_arff_file(file_path: str, dtype: np.dtype) -> ArffFile:
218
+ try:
219
+ return ArffFileSource.ArffFile.from_file(file_path, sparse=True, dtype=dtype)
220
+ except arff.BadLayout:
221
+ return ArffFileSource.ArffFile.from_file(file_path, sparse=False, dtype=dtype)
222
+
223
+ def __init__(self, directory: str):
224
+ """
225
+ :param directory: The path to the directory of the file
226
+ """
227
+ super().__init__(directory=directory, suffix=ArffFileSink.SUFFIX_ARFF)
228
+
229
+ def _read_dataset_from_file(self, state: ExperimentState, file_path: str,
230
+ input_data: DatasetInputData) -> Optional[Dataset]:
231
+ properties = input_data.properties
232
+ problem_domain = state.problem_domain
233
+ arff_file = self.__read_arff_file(file_path=file_path, dtype=problem_domain.feature_dtype)
234
+ xml_file_path = path.join(path.dirname(file_path), properties.file_name + '.' + ArffFileSink.SUFFIX_XML)
235
+ arff_dataset = ArffFileSource.ArffDataset.from_file(arff_file=arff_file, file_path=xml_file_path)
236
+ return TabularDataset(x=arff_dataset.feature_matrix.tolil(),
237
+ y=arff_dataset.output_matrix.astype(problem_domain.output_dtype).tolil(),
238
+ features=arff_dataset.features,
239
+ outputs=arff_dataset.outputs)
File without changes
@@ -0,0 +1,6 @@
1
+ """
2
+ Author Michael Rapp (michael.rapp.ml@gmail.com)
3
+
4
+ Provides classes that allow to write output data to ARFF files.
5
+ """
6
+ from mlrl.testbed_arff.experiments.output.sinks.sink_arff import ArffFileSink
@@ -0,0 +1,97 @@
1
+ """
2
+ Author Michael Rapp (michael.rapp.ml@gmail.com)
3
+
4
+ Provides classes that allow writing datasets to ARFF files.
5
+ """
6
+ import xml.etree.ElementTree as XmlTree
7
+
8
+ from xml.dom import minidom
9
+
10
+ import arff
11
+
12
+ from scipy.sparse import dok_array
13
+
14
+ from mlrl.testbed_sklearn.experiments.dataset import AttributeType, TabularDataset
15
+
16
+ from mlrl.testbed.experiments.dataset import Dataset
17
+ from mlrl.testbed.experiments.output.sinks.sink import DatasetFileSink
18
+ from mlrl.testbed.experiments.state import ExperimentState
19
+ from mlrl.testbed.util.io import ENCODING_UTF8, open_writable_file
20
+
21
+ from mlrl.util.options import Options
22
+
23
+
24
+ class ArffFileSink(DatasetFileSink):
25
+ """
26
+ Allows to write a dataset to an ARFF file.
27
+ """
28
+
29
+ SUFFIX_ARFF = 'arff'
30
+
31
+ SUFFIX_XML = 'xml'
32
+
33
+ @staticmethod
34
+ def __write_arff_file(file_path: str, dataset: TabularDataset):
35
+ sparse = dataset.has_sparse_features and dataset.has_sparse_outputs
36
+ num_examples = dataset.num_examples
37
+ num_features = dataset.num_features
38
+ num_outputs = dataset.num_outputs
39
+
40
+ features = dataset.features
41
+ x_features = [(features[i].name, 'NUMERIC' if features[i].attribute_type == AttributeType.NUMERICAL
42
+ or features[i].nominal_values is None else features[i].nominal_values)
43
+ for i in range(num_features)]
44
+
45
+ outputs = dataset.outputs
46
+ y_features = [(outputs[i].name, 'NUMERIC' if outputs[i].attribute_type == AttributeType.NUMERICAL
47
+ or outputs[i].nominal_values is None else outputs[i].nominal_values) for i in range(num_outputs)]
48
+
49
+ if sparse:
50
+ data = [{} for _ in range(num_examples)]
51
+ else:
52
+ data = [[0 for _ in range(num_features + num_outputs)] for _ in range(num_examples)]
53
+
54
+ for keys, value in dok_array(dataset.x).items():
55
+ data[keys[0]][keys[1]] = value
56
+
57
+ for keys, value in dok_array(dataset.y).items():
58
+ data[keys[0]][num_features + keys[1]] = value
59
+
60
+ with open_writable_file(file_path) as arff_file:
61
+ arff_file.write(
62
+ arff.dumps({
63
+ 'description': 'traindata',
64
+ 'relation': 'traindata: -C ' + str(-num_outputs),
65
+ 'attributes': x_features + y_features,
66
+ 'data': data
67
+ }))
68
+
69
+ @staticmethod
70
+ def __write_xml_file(file_path: str, dataset: TabularDataset):
71
+ root_element = XmlTree.Element('labels')
72
+ root_element.set('xmlns', 'http://mulan.sourceforge.net/labels')
73
+
74
+ for output in dataset.outputs:
75
+ label_element = XmlTree.SubElement(root_element, 'label')
76
+ label_element.set('name', output.name)
77
+
78
+ with open_writable_file(file_path) as xml_file:
79
+ xml_string = minidom.parseString(XmlTree.tostring(root_element)).toprettyxml(encoding=ENCODING_UTF8)
80
+ xml_file.write(xml_string.decode(ENCODING_UTF8))
81
+
82
+ def __init__(self, directory: str, options: Options = Options(), create_directory: bool = False):
83
+ """
84
+ :param directory: The path to the directory, where the ARFF file should be located
85
+ :param options: Options to be taken into account
86
+ :param create_directory: True, if the given directory should be created, if it does not exist, False
87
+ otherwise
88
+ """
89
+ super().__init__(directory=directory,
90
+ suffix=self.SUFFIX_ARFF,
91
+ options=options,
92
+ create_directory=create_directory)
93
+
94
+ # pylint: disable=unused-argument
95
+ def _write_dataset_to_file(self, file_path: str, state: ExperimentState, dataset: Dataset, **_):
96
+ self.__write_arff_file(file_path=file_path, dataset=dataset)
97
+ self.__write_xml_file(file_path=file_path.rsplit('.', 1)[0] + '.' + self.SUFFIX_XML, dataset=dataset)
@@ -0,0 +1,37 @@
1
+ Metadata-Version: 2.4
2
+ Name: mlrl-testbed-arff
3
+ Version: 0.12.0
4
+ Summary: Adds support for ARFF files to the package "mlrl-testbed"
5
+ Author-email: Michael Rapp <michael.rapp.ml@gmail.com>
6
+ License-Expression: MIT
7
+ Project-URL: homepage, https://github.com/mrapp-ke/MLRL-Boomer
8
+ Project-URL: source, https://github.com/mrapp-ke/MLRL-Boomer.git
9
+ Project-URL: download, https://github.com/mrapp-ke/MLRL-Boomer/releases
10
+ Project-URL: changelog, https://raw.githubusercontent.com/mrapp-ke/MLRL-Boomer/refs/heads/main/CHANGELOG.md
11
+ Project-URL: documentation, https://mlrl-boomer.readthedocs.io/en/latest
12
+ Project-URL: issues, https://github.com/mrapp-ke/MLRL-Boomer/issues
13
+ Keywords: arff,machine learning
14
+ Classifier: Development Status :: 5 - Production/Stable
15
+ Classifier: Intended Audience :: Science/Research
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
19
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
20
+ Requires-Python: >=3.11
21
+ Description-Content-Type: text/markdown
22
+ Requires-Dist: liac-arff<2.6,>=2.5
23
+ Requires-Dist: mlrl-testbed-sklearn==0.12.0
24
+
25
+ # "MLRL-Testbed-Arff": Adds support for ARFF files to the package "mlrl-testbed"
26
+
27
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT) [![PyPI version](https://badge.fury.io/py/mlrl-testbed-arff.svg)](https://badge.fury.io/py/mlrl-testbed-arff) [![Documentation Status](https://readthedocs.org/projects/mlrl-boomer/badge/?version=latest)](https://mlrl-boomer.readthedocs.io/en/latest/?badge=latest)
28
+
29
+ **:link: Important links:** [API Reference](https://mlrl-boomer.readthedocs.io/en/latest/developer_guide/api/python/testbed-arff/mlrl.testbed_arff.html) | [Issue Tracker](https://github.com/mrapp-ke/MLRL-Boomer/issues) | [Changelog](https://mlrl-boomer.readthedocs.io/en/latest/misc/CHANGELOG.html) | [Contributors](https://mlrl-boomer.readthedocs.io/en/latest/misc/CONTRIBUTORS.html) | [Code of Conduct](https://mlrl-boomer.readthedocs.io/en/latest/misc/CODE_OF_CONDUCT.html) | [License](https://mlrl-boomer.readthedocs.io/en/latest/misc/LICENSE.html)
30
+
31
+ This software package is an extension to the package [mlrl-testbed](https://pypi.org/project/mlrl-testbed/) which provides a command line utility for running machine learning experiments. It adds support for reading and writing tabular datasets in the [ARFF](https://waikato.github.io/weka-wiki/formats_and_processing/arff_stable/) format. For further information on the command line utility, refer to its [documentation](https://mlrl-boomer.readthedocs.io/en/latest/user_guide/testbed/index.html).
32
+
33
+ ## :scroll: License
34
+
35
+ This project is open source software licensed under the terms of the [MIT license](https://mlrl-boomer.readthedocs.io/en/latest/misc/LICENSE.html). We welcome contributions to the project to enhance its functionality and make it more accessible to a broader audience. A frequently updated list of contributors is available [here](https://mlrl-boomer.readthedocs.io/en/latest/misc/CONTRIBUTORS.html).
36
+
37
+ All contributions to the project and discussions on the [issue tracker](https://github.com/mrapp-ke/MLRL-Boomer/issues) are expected to follow the [code of conduct](https://mlrl-boomer.readthedocs.io/en/latest/misc/CODE_OF_CONDUCT.html).
@@ -0,0 +1,13 @@
1
+ mlrl/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
2
+ mlrl/testbed_arff/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
3
+ mlrl/testbed_arff/experiments/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
4
+ mlrl/testbed_arff/experiments/input/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
5
+ mlrl/testbed_arff/experiments/input/sources/__init__.py,sha256=CJvMNffltblBYgIsWzlCYfEdMxzVv52fHIOSG-SdyZ4,204
6
+ mlrl/testbed_arff/experiments/input/sources/source_arff.py,sha256=LB9x2vuVQwFAD1OTeflBXP5Pq9U9HFD069eL3cn5TzQ,10198
7
+ mlrl/testbed_arff/experiments/output/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
8
+ mlrl/testbed_arff/experiments/output/sinks/__init__.py,sha256=hIL0pOCMZeCbG0yx-IV7jPbwGgfcKYqR3FziQZBqwB4,199
9
+ mlrl/testbed_arff/experiments/output/sinks/sink_arff.py,sha256=NC7krP1srJIrMna5F2v54A2XFzTGw29r-1yB6SCeQYg,3926
10
+ mlrl_testbed_arff-0.12.0.dist-info/METADATA,sha256=RwC81wo9WJlqLYTf5Jl27Mo7AHh5VkSIYb09R0_sE_A,3347
11
+ mlrl_testbed_arff-0.12.0.dist-info/WHEEL,sha256=DnLRTWE75wApRYVsjgc6wsVswC54sMSJhAEd4xhDpBk,91
12
+ mlrl_testbed_arff-0.12.0.dist-info/top_level.txt,sha256=o9x5YHV4Syk8twE0D3Or7SqyG69Ryp7e-HEYe8aB4j4,5
13
+ mlrl_testbed_arff-0.12.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (80.4.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1 @@
1
+ mlrl