testomaton 0.2.2__py2.py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- testomaton/__init__.py +0 -0
- testomaton/beaver.py +180 -0
- testomaton/constraint.py +669 -0
- testomaton/exit_codes.py +20 -0
- testomaton/generator.py +300 -0
- testomaton/jigsaw.py +347 -0
- testomaton/model.py +605 -0
- testomaton/solver.py +293 -0
- testomaton/tomato.py +473 -0
- testomaton/validator.py +238 -0
- testomaton-0.2.2.dist-info/LICENSE +661 -0
- testomaton-0.2.2.dist-info/METADATA +1193 -0
- testomaton-0.2.2.dist-info/RECORD +16 -0
- testomaton-0.2.2.dist-info/WHEEL +6 -0
- testomaton-0.2.2.dist-info/entry_points.txt +4 -0
- testomaton-0.2.2.dist-info/top_level.txt +7 -0
testomaton/generator.py
ADDED
|
@@ -0,0 +1,300 @@
|
|
|
1
|
+
#
|
|
2
|
+
# Copyright Testify AS
|
|
3
|
+
#
|
|
4
|
+
# This file is part of testomaton suite
|
|
5
|
+
#
|
|
6
|
+
# testomaton is free software: you can redistribute it and/or modify
|
|
7
|
+
# it under the terms of the GNU Affero General Public License as published by
|
|
8
|
+
# the Free Software Foundation, either version 3 of the License, or
|
|
9
|
+
# (at your option) any later version.
|
|
10
|
+
#
|
|
11
|
+
# For the commercial license, please contact Testify AS.
|
|
12
|
+
#
|
|
13
|
+
# testomaton is distributed in the hope that it will be useful,
|
|
14
|
+
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
15
|
+
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
16
|
+
# GNU Affero General Public License for more details.
|
|
17
|
+
#
|
|
18
|
+
# You should have received a copy of the GNU General Public License
|
|
19
|
+
# along with testomaton. If not, see <http://www.gnu.org/licenses/>.
|
|
20
|
+
#
|
|
21
|
+
# See LICENSE file for the complete license text.
|
|
22
|
+
#
|
|
23
|
+
|
|
24
|
+
from itertools import combinations, product
|
|
25
|
+
from math import ceil
|
|
26
|
+
import random as rand
|
|
27
|
+
from copy import copy, deepcopy
|
|
28
|
+
import sys, time
|
|
29
|
+
|
|
30
|
+
show_progress = False
|
|
31
|
+
step_delay = 0
|
|
32
|
+
|
|
33
|
+
def print_csv_line(header, line, output=sys.stderr):
|
|
34
|
+
""" Prints a line in CSV format to the output."""
|
|
35
|
+
cleaned_line = [str(element) if element is not None else ' ' for element in line]
|
|
36
|
+
if header is not None:
|
|
37
|
+
output.write('(' + header + ')\t')
|
|
38
|
+
output.write(','.join(cleaned_line) + '\n')
|
|
39
|
+
output.flush()
|
|
40
|
+
if step_delay > 0:
|
|
41
|
+
time.sleep(step_delay)
|
|
42
|
+
# pass
|
|
43
|
+
|
|
44
|
+
def clean_line(output=sys.stderr):
|
|
45
|
+
output.write('\033[1A')
|
|
46
|
+
output.write('\033[K')
|
|
47
|
+
output.flush()
|
|
48
|
+
# output.write('\n')
|
|
49
|
+
# pass
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def random(function, solver, length=0, duplicates=False, adaptive=False):
|
|
53
|
+
"""
|
|
54
|
+
A function which generates random test cases for a given function and solver.
|
|
55
|
+
It can work in either adaptive or random mode: in adaptive mode it will make
|
|
56
|
+
use of previous test cases to influence future ones, while in random mode it
|
|
57
|
+
will not use any previous test cases and generate purely new random test cases.
|
|
58
|
+
|
|
59
|
+
Args:
|
|
60
|
+
function: The function to generate test cases for.
|
|
61
|
+
solver: The solver to use for generating the test cases.
|
|
62
|
+
length: The number of test cases to generate. If set to 0, it will generate
|
|
63
|
+
test cases indefinitely.
|
|
64
|
+
duplicates: Whether to allow duplicates in the generated test cases.
|
|
65
|
+
adaptive: Whether to use adaptive mode or not.
|
|
66
|
+
|
|
67
|
+
Yields:
|
|
68
|
+
test_case: A test case generated by the function.
|
|
69
|
+
"""
|
|
70
|
+
ADAPTIVE_HISTORY_SIZE = 100
|
|
71
|
+
params, input_list = function.get_generator_input()
|
|
72
|
+
|
|
73
|
+
assigner = solver
|
|
74
|
+
generated_tests = 0
|
|
75
|
+
recent_tests = []
|
|
76
|
+
|
|
77
|
+
def best_choice(test_case, solver, i, candidate_choices, generated_tests):
|
|
78
|
+
"""
|
|
79
|
+
Used for adaptive random generation, where it takes info from previously
|
|
80
|
+
generated test cases in order to generate the ideal/best test case based on a
|
|
81
|
+
candidate score
|
|
82
|
+
"""
|
|
83
|
+
def score(choice, index, generated_tests):
|
|
84
|
+
return sum([1 if test[index] != choice else 0 for test in generated_tests])
|
|
85
|
+
|
|
86
|
+
best_score = -1
|
|
87
|
+
for choice in candidate_choices:
|
|
88
|
+
candidate_score = score(choice, i, generated_tests)
|
|
89
|
+
if candidate_score > best_score:
|
|
90
|
+
best_score = candidate_score
|
|
91
|
+
best_choice = choice
|
|
92
|
+
return best_choice
|
|
93
|
+
|
|
94
|
+
def random_choice(test_case, solver, i, candidate_choices):
|
|
95
|
+
for choice in candidate_choices:
|
|
96
|
+
candidate = list(copy(test_case))
|
|
97
|
+
candidate[i] = choice
|
|
98
|
+
if show_progress:
|
|
99
|
+
clean_line()
|
|
100
|
+
print_csv_line("", candidate)
|
|
101
|
+
|
|
102
|
+
if solver.test(candidate):
|
|
103
|
+
return choice
|
|
104
|
+
return None
|
|
105
|
+
|
|
106
|
+
while ((generated_tests < length) or (length == 0)):
|
|
107
|
+
indices = list(range(len(input_list)))
|
|
108
|
+
rand.shuffle(indices)
|
|
109
|
+
for i in indices:
|
|
110
|
+
rand.shuffle(input_list[i])
|
|
111
|
+
|
|
112
|
+
#Call the solver to initialize a new test case in the solver
|
|
113
|
+
solver.new_test_case()
|
|
114
|
+
test_case = [None] * len(input_list)
|
|
115
|
+
if show_progress:
|
|
116
|
+
print_csv_line("",test_case)
|
|
117
|
+
#Iterates over each index in 'indeces' and calls the respective function for random/adaptive random generation
|
|
118
|
+
for i in indices:
|
|
119
|
+
if adaptive:
|
|
120
|
+
test_case[i] = best_choice(test_case, solver, i, input_list[i], recent_tests)
|
|
121
|
+
else:
|
|
122
|
+
test_case[i] = random_choice(test_case, solver, i, input_list[i])
|
|
123
|
+
|
|
124
|
+
if duplicates == False:
|
|
125
|
+
solver.restrict_test_case(test_case)
|
|
126
|
+
|
|
127
|
+
#If any choice in 'test_case' is None, break the loop
|
|
128
|
+
if any([choice is None for choice in test_case]):
|
|
129
|
+
if show_progress:
|
|
130
|
+
clean_line()
|
|
131
|
+
break
|
|
132
|
+
|
|
133
|
+
#adapt the state of the solver to the test case, increment 'generated_tests' and yield it
|
|
134
|
+
assigner.adapt(test_case)
|
|
135
|
+
generated_tests += 1
|
|
136
|
+
recent_tests.append(test_case)
|
|
137
|
+
if len(recent_tests) > ADAPTIVE_HISTORY_SIZE:
|
|
138
|
+
recent_tests = recent_tests[1:]
|
|
139
|
+
|
|
140
|
+
if show_progress:
|
|
141
|
+
clean_line()
|
|
142
|
+
yield test_case
|
|
143
|
+
|
|
144
|
+
def cartesian(function, solver):
|
|
145
|
+
"""
|
|
146
|
+
A function which generates all valid test cases for a given function and solver.
|
|
147
|
+
|
|
148
|
+
Args:
|
|
149
|
+
function: The function to generate test cases for.
|
|
150
|
+
solver: The solver to use for generating the test cases.
|
|
151
|
+
|
|
152
|
+
yields:
|
|
153
|
+
test_case: A test case generated by the function.
|
|
154
|
+
"""
|
|
155
|
+
assigner = solver
|
|
156
|
+
params, input_list = function.get_generator_input()
|
|
157
|
+
|
|
158
|
+
#set combinations to the Cartesian product of the input list
|
|
159
|
+
combinations = product(*input_list)
|
|
160
|
+
#Iterates over each combination in the combinations and yields the valid test cases
|
|
161
|
+
for combination in combinations:
|
|
162
|
+
if solver.test(combination):
|
|
163
|
+
test_case = assigner.adapt(list(combination))
|
|
164
|
+
yield test_case
|
|
165
|
+
|
|
166
|
+
def nwise(function, solver, n, coverage=100):
|
|
167
|
+
"""
|
|
168
|
+
A function which generates test cases that covers all n-wise combinations of
|
|
169
|
+
input parameters provided by a given function. It ensures that the test cases
|
|
170
|
+
achieve the desired covereage percentage which is specified in the input parameter.
|
|
171
|
+
|
|
172
|
+
Args:
|
|
173
|
+
function: The function to generate test cases for.
|
|
174
|
+
solver: The solver to use for generating the test cases.
|
|
175
|
+
n: The number of parameters to consider in each combination.
|
|
176
|
+
coverage: The desired coverage percentage to achieve.
|
|
177
|
+
|
|
178
|
+
Yields:
|
|
179
|
+
test_case: A test case generated by the function.
|
|
180
|
+
"""
|
|
181
|
+
assigner = solver
|
|
182
|
+
params, input_list = function.get_generator_input()
|
|
183
|
+
|
|
184
|
+
#If n is greater than the length of the input list, raise a ValueError
|
|
185
|
+
if n >= len(input_list):
|
|
186
|
+
raise ValueError("n must be lower than the length of the input list")
|
|
187
|
+
|
|
188
|
+
def tuples_covered(test_case, n):
|
|
189
|
+
if n > len(test_case):
|
|
190
|
+
raise ValueError("n cannot be greater than the size of the test case")
|
|
191
|
+
|
|
192
|
+
indices = [i for i, _ in enumerate(test_case) if test_case[i] is not None]
|
|
193
|
+
|
|
194
|
+
combinations_ = combinations(indices, n)
|
|
195
|
+
def uncompress(indices):
|
|
196
|
+
return tuple([test_case[i] if i in indices else None for i in range(len(test_case))])
|
|
197
|
+
yield from [uncompress(test) for test in combinations_]
|
|
198
|
+
|
|
199
|
+
def tuples(input_list, n):
|
|
200
|
+
""" Returns all possible tuples of length n from a given 'input_list'."""
|
|
201
|
+
def tuple_template(n, input_size):
|
|
202
|
+
""" Generates templates that indicate whch positions in 'input_list' should be selected to form tuples of length 'n'."""
|
|
203
|
+
if n > input_size:
|
|
204
|
+
raise ValueError("n cannot be greater than m")
|
|
205
|
+
|
|
206
|
+
#Iterates over all compinations from the range of 'n' and yields the template.
|
|
207
|
+
#First a template list of 'False' values, then sets the position of current combination to true.
|
|
208
|
+
for combination in combinations(range(input_size), n):
|
|
209
|
+
template = [False] * input_size
|
|
210
|
+
for index in combination:
|
|
211
|
+
template[index] = True
|
|
212
|
+
yield template
|
|
213
|
+
|
|
214
|
+
def tuples_from_template(input_list, template):
|
|
215
|
+
""" Generates tuples from a given 'input_list' and 'template'."""
|
|
216
|
+
len_input_list = len(input_list)
|
|
217
|
+
if len_input_list != len(template):
|
|
218
|
+
raise ValueError("Input list and template must have the same length")
|
|
219
|
+
|
|
220
|
+
#Creates a list of selected indices from the template where it has the value 'True'
|
|
221
|
+
selected_indices = [i for i, is_selected in enumerate(template) if is_selected]
|
|
222
|
+
|
|
223
|
+
#Iterates over the product of the input list at the selected indices and yields the tuple from the current combination
|
|
224
|
+
for combination in product(*[input_list[i] for i in selected_indices]):
|
|
225
|
+
result = [None] * len_input_list
|
|
226
|
+
for i, index in enumerate(selected_indices):
|
|
227
|
+
result[index] = combination[i]
|
|
228
|
+
yield tuple(result) # tuple is hashable, list is not
|
|
229
|
+
|
|
230
|
+
#If n is greater than the length of the input list, raise a ValueError
|
|
231
|
+
if n > len(input_list):
|
|
232
|
+
raise ValueError("n cannot be greater than the length of the input list")
|
|
233
|
+
|
|
234
|
+
#Iterates over each template generated and yields the tuples from the template one by one
|
|
235
|
+
for template in tuple_template(n, len(input_list)):
|
|
236
|
+
yield from tuples_from_template(input_list, template)
|
|
237
|
+
|
|
238
|
+
#Generates set of tuples length 'n' that meet the solver.test
|
|
239
|
+
tuples_to_cover = {t for t in tuples(input_list, n) if solver.test(t)}
|
|
240
|
+
#Calculate how many tuples can be left uncovered to meet the coverage criteria
|
|
241
|
+
end_tuples_count = ceil(len(tuples_to_cover) * ((100 - coverage) / 100))
|
|
242
|
+
#Define the end condition to see if criteria has been met
|
|
243
|
+
end_condition = lambda: len(tuples_to_cover) <= end_tuples_count
|
|
244
|
+
|
|
245
|
+
def tuple_score(candidate):
|
|
246
|
+
""" Evaluates how well a candidate test case covers the required tuples."""
|
|
247
|
+
#If the candidate test case does not meet the solver.test, return -1
|
|
248
|
+
if not solver.test(candidate):
|
|
249
|
+
return -1
|
|
250
|
+
#Otherwise, get the tuples covered by the candidate test case and return the score
|
|
251
|
+
else:
|
|
252
|
+
covered_tuples = tuples_covered(candidate, n)
|
|
253
|
+
score = len(tuples_to_cover.intersection(covered_tuples))
|
|
254
|
+
return score
|
|
255
|
+
|
|
256
|
+
#Iteratively generates test cases as long as the end condition is not met
|
|
257
|
+
while not end_condition():
|
|
258
|
+
solver.new_test_case()
|
|
259
|
+
tuple_to_cover = rand.choice(list(tuples_to_cover))
|
|
260
|
+
solver.tuple_selected(tuple_to_cover)
|
|
261
|
+
#Get the indices of the tuple to cover that are None and randomize them to ensure different orerings in each iteration
|
|
262
|
+
indices = [i for i, x in enumerate(tuple_to_cover) if x is None]
|
|
263
|
+
rand.shuffle(indices)
|
|
264
|
+
|
|
265
|
+
#Create a copy of the tuple to begin constructing the test case
|
|
266
|
+
constructed_test_case = copy(tuple_to_cover)
|
|
267
|
+
if show_progress:
|
|
268
|
+
# print_csv_line("",constructed_test_case)
|
|
269
|
+
print_csv_line(str(len(tuples_to_cover)) + ' tuples to go', constructed_test_case)
|
|
270
|
+
#Iterates over each position in in the tuple that need values, evaluate all possible values
|
|
271
|
+
#and select the one with the highest score from tuple_score
|
|
272
|
+
for i in indices:
|
|
273
|
+
best_choice, best_candidate, best_score = None, None, -1
|
|
274
|
+
choices = input_list[i]
|
|
275
|
+
rand.shuffle(choices)
|
|
276
|
+
for choice in choices:
|
|
277
|
+
candidate = list(copy(constructed_test_case))
|
|
278
|
+
candidate[i] = choice
|
|
279
|
+
if show_progress:
|
|
280
|
+
clean_line()
|
|
281
|
+
# print_csv_line("", candidate)
|
|
282
|
+
print_csv_line(str(len(tuples_to_cover)) + ' tuples to go', candidate)
|
|
283
|
+
score = tuple_score(candidate)
|
|
284
|
+
if score > best_score:
|
|
285
|
+
best_candidate = candidate
|
|
286
|
+
best_score = tuple_score(candidate)
|
|
287
|
+
best_choice = choice
|
|
288
|
+
|
|
289
|
+
#Updates the constructed test case to the best candidate
|
|
290
|
+
constructed_test_case = best_candidate
|
|
291
|
+
solver.choice_selected(i, best_choice)
|
|
292
|
+
|
|
293
|
+
#Removes the tuples covered by the constructed test case from the tuples to cover
|
|
294
|
+
for t in tuples_covered(constructed_test_case, n):
|
|
295
|
+
tuples_to_cover.discard(t)
|
|
296
|
+
#Adapts the state of the solver to the constructed test case and yields it
|
|
297
|
+
test_case = assigner.adapt(constructed_test_case)
|
|
298
|
+
if show_progress:
|
|
299
|
+
clean_line()
|
|
300
|
+
yield test_case
|
testomaton/jigsaw.py
ADDED
|
@@ -0,0 +1,347 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
|
|
3
|
+
#
|
|
4
|
+
# Copyright Testify AS
|
|
5
|
+
#
|
|
6
|
+
# This file is part of testomaton suite
|
|
7
|
+
#
|
|
8
|
+
# testomaton is free software: you can redistribute it and/or modify
|
|
9
|
+
# it under the terms of the GNU Affero General Public License as published by
|
|
10
|
+
# the Free Software Foundation, either version 3 of the License, or
|
|
11
|
+
# (at your option) any later version.
|
|
12
|
+
#
|
|
13
|
+
# For the commercial license, please contact Testify AS.
|
|
14
|
+
#
|
|
15
|
+
# testomaton is distributed in the hope that it will be useful,
|
|
16
|
+
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
17
|
+
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
18
|
+
# GNU Affero General Public License for more details.
|
|
19
|
+
#
|
|
20
|
+
# You should have received a copy of the GNU General Public License
|
|
21
|
+
# along with testomaton. If not, see <http://www.gnu.org/licenses/>.
|
|
22
|
+
#
|
|
23
|
+
# See LICENSE file for the complete license text.
|
|
24
|
+
#
|
|
25
|
+
|
|
26
|
+
import sys
|
|
27
|
+
import argparse
|
|
28
|
+
import importlib
|
|
29
|
+
import importlib.metadata as metadata
|
|
30
|
+
import csv
|
|
31
|
+
import regex as re
|
|
32
|
+
|
|
33
|
+
global_imports = {}
|
|
34
|
+
appended_columns = prepended_columns = replaced_columns = swapped_columns = []
|
|
35
|
+
columns_whitelist = []
|
|
36
|
+
columns_blacklist = []
|
|
37
|
+
|
|
38
|
+
def read_csv(file, separator=','):
|
|
39
|
+
"""
|
|
40
|
+
reads a csv file and yields rows
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
file: file to read. If None, reads from stdin
|
|
44
|
+
separator: separator used in the csv file
|
|
45
|
+
|
|
46
|
+
Yields:
|
|
47
|
+
a row from the csv file
|
|
48
|
+
|
|
49
|
+
Raises:
|
|
50
|
+
OSError: if the file can't be opened
|
|
51
|
+
"""
|
|
52
|
+
csv_reader = None
|
|
53
|
+
try:
|
|
54
|
+
# if file is one, read from standard input
|
|
55
|
+
if file is None:
|
|
56
|
+
csv_reader = csv.reader(sys.stdin, delimiter=separator)
|
|
57
|
+
for row in csv_reader:
|
|
58
|
+
yield row
|
|
59
|
+
else:
|
|
60
|
+
with open(file) as f:
|
|
61
|
+
csv_reader = csv.reader(f, delimiter=separator)
|
|
62
|
+
for row in csv_reader:
|
|
63
|
+
yield row
|
|
64
|
+
# if there is a file, but it can't be opened, raise an exception
|
|
65
|
+
except OSError:
|
|
66
|
+
print(f"Could not open file {file}")
|
|
67
|
+
sys.exit(1)
|
|
68
|
+
|
|
69
|
+
def parse_column_refactoring_args(args):
|
|
70
|
+
"""
|
|
71
|
+
Parses the arguments for column refactoring
|
|
72
|
+
|
|
73
|
+
Args:
|
|
74
|
+
args: the arguments to parse
|
|
75
|
+
|
|
76
|
+
Returns:
|
|
77
|
+
a list of tuples with the column refactoring information
|
|
78
|
+
|
|
79
|
+
Raises:
|
|
80
|
+
Exception: if the arguments are not in the correct format
|
|
81
|
+
"""
|
|
82
|
+
result = []
|
|
83
|
+
# loops through the args and returns a list of tuples with the column refactoring information
|
|
84
|
+
for col in args if args is not None else []:
|
|
85
|
+
ref = name = expr = None
|
|
86
|
+
if len(col) == 2:
|
|
87
|
+
ref, name, expr = None, col[0], col[1]
|
|
88
|
+
elif len(col) == 3:
|
|
89
|
+
ref, name, expr = col[0], col[1], col[2]
|
|
90
|
+
else:
|
|
91
|
+
raise Exception(f"Invalid format: {col}. Should be '[<column_name|index>] <new_name> <expression>''")
|
|
92
|
+
if name and expr:
|
|
93
|
+
result.append((ref, name, expr))
|
|
94
|
+
return result
|
|
95
|
+
|
|
96
|
+
def parse_columns_swap_args(args):
|
|
97
|
+
"""
|
|
98
|
+
Parses the arguments for column swapping
|
|
99
|
+
|
|
100
|
+
Args:
|
|
101
|
+
args: the arguments to parse
|
|
102
|
+
|
|
103
|
+
Returns:
|
|
104
|
+
a list of tuples with the columns to swap
|
|
105
|
+
|
|
106
|
+
Raises:
|
|
107
|
+
Exception: if the arguments are not in the correct format
|
|
108
|
+
"""
|
|
109
|
+
result = []
|
|
110
|
+
for col in args if args is not None else []:
|
|
111
|
+
name1 = name2 = None
|
|
112
|
+
if len(col) == 2:
|
|
113
|
+
name1, name2 = col
|
|
114
|
+
else:
|
|
115
|
+
raise Exception(f"Invalid format: {col}. Should be '<name1|index1|-1> <name2|index2|-1>'")
|
|
116
|
+
if name1 and name2:
|
|
117
|
+
result.append((name1, name2))
|
|
118
|
+
return result
|
|
119
|
+
|
|
120
|
+
def parse_args():
|
|
121
|
+
"""
|
|
122
|
+
Parses the command line arguments using argparse
|
|
123
|
+
|
|
124
|
+
Returns:
|
|
125
|
+
the parsed arguments
|
|
126
|
+
|
|
127
|
+
Raises:
|
|
128
|
+
Exception: if there is an error parsing the arguments
|
|
129
|
+
"""
|
|
130
|
+
|
|
131
|
+
global global_imports
|
|
132
|
+
global appended_columns, prepended_columns, replaced_columns, swapped_columns
|
|
133
|
+
global columns_whitelist, columns_blacklist
|
|
134
|
+
|
|
135
|
+
# create the parser
|
|
136
|
+
parser = argparse.ArgumentParser(description='Evaluate expressions in a csv file. Expressions are evaluated in the order they appear in the input list. Use @python <expression> to evaluate python expressions. In the expressions, use {column_name} to refer to a column value. Use {row id} to refer to the row number.')
|
|
137
|
+
parser.add_argument('input_file', type=str, nargs='?', default=None, help='Input file. If not provided, stdin is used')
|
|
138
|
+
parser.add_argument('-v', '--version', action='store_true', help='Print version')
|
|
139
|
+
|
|
140
|
+
# general arguments
|
|
141
|
+
parser.add_argument('-n', '--add-linenumber', nargs='?', type=str, default='', help='Add a column with the line number. Optional argument is the name of the column. Default name is empty. The column may be referenced in the expressions by its name.')
|
|
142
|
+
parser.add_argument('-H', '--remove-headrow', action='store_true', default=False, help='Do not output the first line that was read')
|
|
143
|
+
parser.add_argument("-s", "--input-separator", type=str, default=',', help="Separator used in the input")
|
|
144
|
+
parser.add_argument("-S", "--output-separator", type=str, default=None, help="Separator used in the output. If not provided, the input separator is used")
|
|
145
|
+
|
|
146
|
+
# column manipulation arguments
|
|
147
|
+
column_manipulation_group = parser.add_argument_group('Column manipulation. All arguments except white/blacklists are applied in the following order ADD -> REPLACE -> SWAP -> WHITELIST/BLACKLIST.')
|
|
148
|
+
column_manipulation_group.add_argument("-B", "--columns-blacklist", type=str, default=None, help="Comma separated list of columns to remove. A column may be indicated with its index number, range of indices X..Y, or a regex pattern for the column name")
|
|
149
|
+
column_manipulation_group.add_argument("-W", "--columns-whitelist", type=str, default=None, help="Comma separated list of columns to show. A column may be indicated with its index number, range of indices X..Y, or a regex pattern for the column name")
|
|
150
|
+
column_manipulation_group.add_argument("-A", "--add-after", metavar='<column_name|index> <new_name> <expression>', nargs='+', action='append', type=str, default=None, help="Add column[s] after the specified column. If column name is empty, the column is added at the end. Columns are indexed starting at 1.")
|
|
151
|
+
column_manipulation_group.add_argument("-F", "--add-before", metavar='<column_name|index>::<new_name>::<expression>', nargs='+', action='append', type=str, default=None, help="Add column[s] before the specified column. If column name is empty, the column is added at the beginning. Columns are indexed starting at 1.")
|
|
152
|
+
column_manipulation_group.add_argument("-R", "--replace-columns", metavar='<column_name|index>::<new_name>::<expression>', nargs='+', action='append', type=str, default=None, help="Replace column[s] with new header and value. '-1' for the colummn index means the last column. Columns are indexed starting at 1.")
|
|
153
|
+
column_manipulation_group.add_argument("-X", "--swap-columns", metavar='<name1|index1|-1>::<name2|index2|-1>', nargs='+', action='append', type=str, default=None, help="Swaps two columns. '-1' for the colummn index means the last column. Columns are indexed starting at 1.")
|
|
154
|
+
|
|
155
|
+
args = parser.parse_args()
|
|
156
|
+
|
|
157
|
+
# checks if whitelist and blacklist are used
|
|
158
|
+
columns_whitelist = args.columns_whitelist.split(',') if args.columns_whitelist is not None else []
|
|
159
|
+
columns_blacklist = args.columns_blacklist.split(',') if args.columns_blacklist is not None else []
|
|
160
|
+
if len(columns_whitelist) != 0 and len(columns_blacklist) != 0:
|
|
161
|
+
print("You can't use both columns whitelist and blacklist")
|
|
162
|
+
sys.exit(1)
|
|
163
|
+
|
|
164
|
+
if args.output_separator is None:
|
|
165
|
+
args.output_separator = args.input_separator
|
|
166
|
+
|
|
167
|
+
# prints the version of tomato
|
|
168
|
+
if args.version:
|
|
169
|
+
print(f"tomato {version()}")
|
|
170
|
+
sys.exit(0)
|
|
171
|
+
|
|
172
|
+
# use the refactoring methods on the specified inputs
|
|
173
|
+
try:
|
|
174
|
+
appended_columns = parse_column_refactoring_args(args.add_after)
|
|
175
|
+
except Exception as e:
|
|
176
|
+
print(f"Error parsing -A|add_after: {e}")
|
|
177
|
+
sys.exit(1)
|
|
178
|
+
try:
|
|
179
|
+
prepended_columns = parse_column_refactoring_args(args.add_before)
|
|
180
|
+
except Exception as e:
|
|
181
|
+
print(f"Error parsing -B|add_before: {e}")
|
|
182
|
+
sys.exit(1)
|
|
183
|
+
try:
|
|
184
|
+
replaced_columns = parse_column_refactoring_args(args.replace_columns)
|
|
185
|
+
except Exception as e:
|
|
186
|
+
print(f"Error parsing -R|--replace: {e}")
|
|
187
|
+
sys.exit(1)
|
|
188
|
+
|
|
189
|
+
try:
|
|
190
|
+
swapped_columns = parse_columns_swap_args(args.swap_columns)
|
|
191
|
+
except Exception as e:
|
|
192
|
+
print(f"Error parsing -X|--swap_columns: {e}")
|
|
193
|
+
sys.exit(1)
|
|
194
|
+
|
|
195
|
+
return args
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def version():
|
|
199
|
+
return metadata.version('testomaton')
|
|
200
|
+
|
|
201
|
+
def define_output_columns(input_columns):
|
|
202
|
+
"""
|
|
203
|
+
Defines the output columns based on the input columns and the column refactoring arguments
|
|
204
|
+
|
|
205
|
+
Args:
|
|
206
|
+
input_columns: the input columns
|
|
207
|
+
|
|
208
|
+
Returns:
|
|
209
|
+
a tuple with the output columns and the output columns mapping
|
|
210
|
+
|
|
211
|
+
Raises:
|
|
212
|
+
Exception: if the column index is 0
|
|
213
|
+
"""
|
|
214
|
+
output_columns = input_columns.copy()
|
|
215
|
+
output_columns_mapping = {}
|
|
216
|
+
|
|
217
|
+
# loop through the appended columns, adding to the end
|
|
218
|
+
for after, name, expr in appended_columns:
|
|
219
|
+
if after == '-1' or after == None:
|
|
220
|
+
output_columns.append(name)
|
|
221
|
+
elif after.isdigit():
|
|
222
|
+
if(int(before) == 0):
|
|
223
|
+
raise Exception("Column index should start at 1")
|
|
224
|
+
index = int(after) - 1
|
|
225
|
+
output_columns.insert(index + 1, name)
|
|
226
|
+
else:
|
|
227
|
+
for i, col in enumerate(output_columns):
|
|
228
|
+
if re.match(after, col):
|
|
229
|
+
output_columns.insert(i + 1, name)
|
|
230
|
+
break
|
|
231
|
+
output_columns_mapping[name] = expr
|
|
232
|
+
# loop through the prepended columns, adding to the beginning
|
|
233
|
+
for before, name, expr in prepended_columns:
|
|
234
|
+
if before == None:
|
|
235
|
+
output_columns.insert(0, name)
|
|
236
|
+
elif before == '-1':
|
|
237
|
+
output_columns.insert(len(output_columns) - 1, name)
|
|
238
|
+
elif before.isdigit():
|
|
239
|
+
if(int(before) == 0):
|
|
240
|
+
raise Exception("Column index should start at 1")
|
|
241
|
+
output_columns.insert(int(before) - 1, name)
|
|
242
|
+
else:
|
|
243
|
+
for i, col in enumerate(output_columns):
|
|
244
|
+
if re.match(before, col):
|
|
245
|
+
output_columns.insert(i, name)
|
|
246
|
+
break
|
|
247
|
+
output_columns_mapping[name] = expr
|
|
248
|
+
# loop through the replaced columns
|
|
249
|
+
for index, name, expr in replaced_columns:
|
|
250
|
+
if index == None or index == '-1':
|
|
251
|
+
last_index = len(output_columns) - 1
|
|
252
|
+
output_columns[last_index] = name
|
|
253
|
+
elif index.isdigit():
|
|
254
|
+
output_columns[int(index) - 1] = name
|
|
255
|
+
else:
|
|
256
|
+
for i, col in enumerate(output_columns):
|
|
257
|
+
if re.match(index, col):
|
|
258
|
+
output_columns[i] = name
|
|
259
|
+
break
|
|
260
|
+
output_columns_mapping[name] = expr
|
|
261
|
+
# loop through the swapped columns
|
|
262
|
+
for name1, name2 in swapped_columns:
|
|
263
|
+
if name1 == '-1':
|
|
264
|
+
name1 = output_columns[-1]
|
|
265
|
+
if name2 == '-1':
|
|
266
|
+
name2 = output_columns[-2]
|
|
267
|
+
index1 = output_columns.index(name1)
|
|
268
|
+
index2 = output_columns.index(name2)
|
|
269
|
+
output_columns[index1], output_columns[index2] = output_columns[index2], output_columns[index1]
|
|
270
|
+
|
|
271
|
+
# return the output columns and the output columns mapping
|
|
272
|
+
return output_columns, output_columns_mapping
|
|
273
|
+
|
|
274
|
+
def is_column_on_list(list, index, column_name):
|
|
275
|
+
"""
|
|
276
|
+
Checks if the column is on the list
|
|
277
|
+
|
|
278
|
+
Args:
|
|
279
|
+
list: the list of columns
|
|
280
|
+
index: the index of the column
|
|
281
|
+
column_name: the name of the column
|
|
282
|
+
|
|
283
|
+
Returns:
|
|
284
|
+
True if the column is on the list, False otherwise
|
|
285
|
+
"""
|
|
286
|
+
for token in list:
|
|
287
|
+
range_tokens = token.split('..')
|
|
288
|
+
# if the token is a range, check if the index is within the range
|
|
289
|
+
if len(range_tokens) == 2:
|
|
290
|
+
if index >= int(range_tokens[0]) - 1 and index <= int(range_tokens[1]) - 1:
|
|
291
|
+
return True
|
|
292
|
+
# if the token is a digit, check if it matches the index + 1
|
|
293
|
+
elif token.isdigit():
|
|
294
|
+
if int(token) == index + 1:
|
|
295
|
+
return True
|
|
296
|
+
# if the column name matches the token, return True
|
|
297
|
+
elif re.match(token, column_name):
|
|
298
|
+
return True
|
|
299
|
+
|
|
300
|
+
def main():
|
|
301
|
+
"""
|
|
302
|
+
The main method, which calls on the other methods to read and print the processed
|
|
303
|
+
rows
|
|
304
|
+
"""
|
|
305
|
+
args = parse_args()
|
|
306
|
+
|
|
307
|
+
input_columns = next(read_csv(args.input_file, separator=args.input_separator))
|
|
308
|
+
output_columns, output_mapping = define_output_columns(input_columns)
|
|
309
|
+
|
|
310
|
+
# checks if there is a whitelist or blacklist and if there is, filters the columns
|
|
311
|
+
if len(columns_whitelist) == 0 and len(columns_blacklist) == 0:
|
|
312
|
+
printed_columns = output_columns
|
|
313
|
+
elif len(columns_whitelist) != 0:
|
|
314
|
+
printed_columns = [str(output_columns[i]) for i in range(len(output_columns)) if is_column_on_list(columns_whitelist, i, output_columns[i])]
|
|
315
|
+
elif len(columns_blacklist) != 0:
|
|
316
|
+
printed_columns = [str(output_columns[i]) for i in range(len(output_columns)) if not is_column_on_list(columns_blacklist, i, output_columns[i])]
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
if len(printed_columns) == 0:
|
|
320
|
+
print("No columns to print")
|
|
321
|
+
sys.exit(0)
|
|
322
|
+
|
|
323
|
+
# print the first row if the remove headrow argument is not provided
|
|
324
|
+
if not args.remove_headrow:
|
|
325
|
+
first_row = printed_columns.copy()
|
|
326
|
+
if args.add_linenumber:
|
|
327
|
+
first_row.insert(0, args.add_linenumber)
|
|
328
|
+
print(args.output_separator.join(first_row))
|
|
329
|
+
|
|
330
|
+
csv_writer = csv.writer(sys.stdout, delimiter=args.output_separator)
|
|
331
|
+
row_id = 0
|
|
332
|
+
# loop through the rows and write them to the output
|
|
333
|
+
for row in read_csv(args.input_file, separator=args.input_separator):
|
|
334
|
+
row_id += 1
|
|
335
|
+
values = {}
|
|
336
|
+
for index, token in enumerate(row):
|
|
337
|
+
values[input_columns[index]] = token
|
|
338
|
+
for name, expr in output_mapping.items():
|
|
339
|
+
values[name] = expr
|
|
340
|
+
output = [str(values[presented_column]) for presented_column in printed_columns]
|
|
341
|
+
if args.add_linenumber:
|
|
342
|
+
output.insert(0, str(row_id))
|
|
343
|
+
|
|
344
|
+
csv_writer.writerow(output)
|
|
345
|
+
|
|
346
|
+
if __name__ == '__main__':
|
|
347
|
+
main()
|