mainframe-migration-toolkit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mainframe_migration_toolkit-0.2.0.dist-info/METADATA +16 -0
- mainframe_migration_toolkit-0.2.0.dist-info/RECORD +52 -0
- mainframe_migration_toolkit-0.2.0.dist-info/WHEEL +4 -0
- mainframe_migration_toolkit-0.2.0.dist-info/entry_points.txt +3 -0
- mainframe_toolkit/__init__.py +92 -0
- mainframe_toolkit/__main__.py +5 -0
- mainframe_toolkit/_workspace/.claude/skills/analyze-mainframe-similarity/SKILL.md +30 -0
- mainframe_toolkit/_workspace/.claude/skills/migrate-mainframe-job/SKILL.md +63 -0
- mainframe_toolkit/_workspace/.claude/skills/validate-golden-dataset/SKILL.md +12 -0
- mainframe_toolkit/_workspace/AGENTS.md +10 -0
- mainframe_toolkit/_workspace/CLAUDE.md +2 -0
- mainframe_toolkit/_workspace/validator-java/.mvn/wrapper/maven-wrapper.properties +3 -0
- mainframe_toolkit/_workspace/validator-java/mvnw +295 -0
- mainframe_toolkit/_workspace/validator-java/mvnw.cmd +189 -0
- mainframe_toolkit/_workspace/validator-java/pom.xml +116 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/AvroValueFormatter.java +152 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/CsvTabularReader.java +111 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/DataFormat.java +62 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/Difference.java +34 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/InputFileSet.java +45 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/InputOptions.java +23 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/JsonReportWriter.java +40 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/MultiFileTabularReader.java +80 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/Normalization.java +16 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ParquetTabularReader.java +61 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/TabularReader.java +13 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/TabularReaderFactory.java +29 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValidationOptions.java +40 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValidationReport.java +44 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValidationService.java +395 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValidatorCli.java +190 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValueNormalizer.java +27 -0
- mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/DirectoryValidationTest.java +70 -0
- mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/JsonReportWriterTest.java +43 -0
- mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/KeyedValidationTest.java +89 -0
- mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/ParquetValidationTest.java +146 -0
- mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/ValidationServiceTest.java +137 -0
- mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/ValidatorCliTest.java +102 -0
- mainframe_toolkit/cli.py +361 -0
- mainframe_toolkit/cobol.py +106 -0
- mainframe_toolkit/copybook.py +558 -0
- mainframe_toolkit/errors.py +15 -0
- mainframe_toolkit/external.py +48 -0
- mainframe_toolkit/io.py +202 -0
- mainframe_toolkit/jcl.py +126 -0
- mainframe_toolkit/pipeline.py +322 -0
- mainframe_toolkit/sequential.py +259 -0
- mainframe_toolkit/similarity.py +1171 -0
- mainframe_toolkit/sorting.py +60 -0
- mainframe_toolkit/specs.py +312 -0
- mainframe_toolkit/synthetic.py +108 -0
- mainframe_toolkit/workspace.py +133 -0
|
@@ -0,0 +1,395 @@
|
|
|
1
|
+
package io.mainframe.migration.validator;
|
|
2
|
+
|
|
3
|
+
import java.io.IOException;
|
|
4
|
+
import java.nio.file.Files;
|
|
5
|
+
import java.util.ArrayList;
|
|
6
|
+
import java.util.Arrays;
|
|
7
|
+
import java.util.Collections;
|
|
8
|
+
import java.util.HashMap;
|
|
9
|
+
import java.util.LinkedHashMap;
|
|
10
|
+
import java.util.List;
|
|
11
|
+
import java.util.Locale;
|
|
12
|
+
import java.util.Map;
|
|
13
|
+
import java.util.Objects;
|
|
14
|
+
import java.util.TreeSet;
|
|
15
|
+
|
|
16
|
+
public final class ValidationService {
|
|
17
|
+
public ValidationReport validate(ValidationOptions options) throws IOException {
|
|
18
|
+
validateInput(options.expected());
|
|
19
|
+
validateInput(options.actual());
|
|
20
|
+
|
|
21
|
+
DataFormat expectedFormat = options.expected().resolvedFormat();
|
|
22
|
+
DataFormat actualFormat = options.actual().resolvedFormat();
|
|
23
|
+
Accumulator accumulator = new Accumulator(options.maxDifferences());
|
|
24
|
+
|
|
25
|
+
try (TabularReader expected = TabularReaderFactory.open(options.expected());
|
|
26
|
+
TabularReader actual = TabularReaderFactory.open(options.actual())) {
|
|
27
|
+
List<String> expectedColumns = expected.columns();
|
|
28
|
+
List<String> actualColumns = actual.columns();
|
|
29
|
+
if (expected.hasNamedColumns() && actual.hasNamedColumns()) {
|
|
30
|
+
compareColumns(expectedColumns, actualColumns, accumulator);
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
ComparisonStats stats;
|
|
34
|
+
if (options.keys().isEmpty()) {
|
|
35
|
+
stats = comparePositionally(
|
|
36
|
+
expected, actual, expectedColumns, actualColumns, options, accumulator);
|
|
37
|
+
} else {
|
|
38
|
+
if (!expected.hasNamedColumns() || !actual.hasNamedColumns()) {
|
|
39
|
+
throw new IllegalArgumentException("--key exige cabeçalhos ou colunas Parquet nomeadas");
|
|
40
|
+
}
|
|
41
|
+
stats = compareByKey(
|
|
42
|
+
expected, actual, expectedColumns, actualColumns, options, accumulator);
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
List<String> normalizations = Arrays.stream(Normalization.values())
|
|
46
|
+
.filter(options.normalizations()::contains)
|
|
47
|
+
.map(value -> value.name().toLowerCase(Locale.ROOT))
|
|
48
|
+
.toList();
|
|
49
|
+
return new ValidationReport(
|
|
50
|
+
accumulator.count == 0 ? "MATCH" : "DIFFERENT",
|
|
51
|
+
options.expected().path().toString(),
|
|
52
|
+
options.actual().path().toString(),
|
|
53
|
+
expectedFormat.name(),
|
|
54
|
+
actualFormat.name(),
|
|
55
|
+
List.copyOf(options.keys()),
|
|
56
|
+
normalizations,
|
|
57
|
+
stats.expectedRows,
|
|
58
|
+
stats.actualRows,
|
|
59
|
+
stats.expectedColumns,
|
|
60
|
+
stats.actualColumns,
|
|
61
|
+
accumulator.count,
|
|
62
|
+
accumulator.differences.size(),
|
|
63
|
+
accumulator.count > accumulator.differences.size(),
|
|
64
|
+
List.copyOf(accumulator.differences));
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
private static void validateInput(InputOptions input) throws IOException {
|
|
69
|
+
if (!Files.exists(input.path())) {
|
|
70
|
+
throw new IllegalArgumentException("Arquivo ou diretório não encontrado: " + input.path());
|
|
71
|
+
}
|
|
72
|
+
if (!Files.isReadable(input.path())) {
|
|
73
|
+
throw new IllegalArgumentException("Entrada local não legível: " + input.path());
|
|
74
|
+
}
|
|
75
|
+
if (!Files.isRegularFile(input.path()) && !Files.isDirectory(input.path())) {
|
|
76
|
+
throw new IllegalArgumentException("Entrada deve ser arquivo ou diretório local: " + input.path());
|
|
77
|
+
}
|
|
78
|
+
input.resolvedFormat();
|
|
79
|
+
InputFileSet.files(input);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
private static ComparisonStats comparePositionally(
|
|
83
|
+
TabularReader expected,
|
|
84
|
+
TabularReader actual,
|
|
85
|
+
List<String> expectedColumns,
|
|
86
|
+
List<String> actualColumns,
|
|
87
|
+
ValidationOptions options,
|
|
88
|
+
Accumulator accumulator) throws IOException {
|
|
89
|
+
ComparisonStats stats = new ComparisonStats(expectedColumns.size(), actualColumns.size());
|
|
90
|
+
long line = 0;
|
|
91
|
+
while (true) {
|
|
92
|
+
List<String> expectedRow = expected.readRow();
|
|
93
|
+
List<String> actualRow = actual.readRow();
|
|
94
|
+
if (expectedRow == null && actualRow == null) {
|
|
95
|
+
return stats;
|
|
96
|
+
}
|
|
97
|
+
line++;
|
|
98
|
+
stats.observeExpected(expectedRow);
|
|
99
|
+
stats.observeActual(actualRow);
|
|
100
|
+
compareRow(
|
|
101
|
+
line,
|
|
102
|
+
Map.of(),
|
|
103
|
+
expectedRow,
|
|
104
|
+
actualRow,
|
|
105
|
+
expectedColumns,
|
|
106
|
+
actualColumns,
|
|
107
|
+
options,
|
|
108
|
+
accumulator);
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
private static ComparisonStats compareByKey(
|
|
113
|
+
TabularReader expected,
|
|
114
|
+
TabularReader actual,
|
|
115
|
+
List<String> expectedColumns,
|
|
116
|
+
List<String> actualColumns,
|
|
117
|
+
ValidationOptions options,
|
|
118
|
+
Accumulator accumulator) throws IOException {
|
|
119
|
+
int[] expectedKeyIndexes = keyIndexes(options.keys(), expectedColumns, "referência");
|
|
120
|
+
int[] actualKeyIndexes = keyIndexes(options.keys(), actualColumns, "arquivo gerado");
|
|
121
|
+
ComparisonStats stats = new ComparisonStats(expectedColumns.size(), actualColumns.size());
|
|
122
|
+
Map<CompositeKey, List<RowEntry>> expectedRows = indexRows(
|
|
123
|
+
expected, expectedKeyIndexes, options, stats, true);
|
|
124
|
+
Map<CompositeKey, List<RowEntry>> actualRows = indexRows(
|
|
125
|
+
actual, actualKeyIndexes, options, stats, false);
|
|
126
|
+
|
|
127
|
+
TreeSet<CompositeKey> orderedKeys = new TreeSet<>();
|
|
128
|
+
orderedKeys.addAll(expectedRows.keySet());
|
|
129
|
+
orderedKeys.addAll(actualRows.keySet());
|
|
130
|
+
for (CompositeKey key : orderedKeys) {
|
|
131
|
+
List<RowEntry> expectedEntries = expectedRows.getOrDefault(key, List.of());
|
|
132
|
+
List<RowEntry> actualEntries = actualRows.getOrDefault(key, List.of());
|
|
133
|
+
RowEntry expectedRow = expectedEntries.isEmpty() ? null : expectedEntries.get(0);
|
|
134
|
+
RowEntry actualRow = actualEntries.isEmpty() ? null : actualEntries.get(0);
|
|
135
|
+
Map<String, String> displayKey = displayKey(
|
|
136
|
+
options.keys(),
|
|
137
|
+
expectedRow != null ? expectedRow : actualRow,
|
|
138
|
+
expectedRow != null ? expectedKeyIndexes : actualKeyIndexes);
|
|
139
|
+
|
|
140
|
+
reportDuplicates(
|
|
141
|
+
expectedEntries, true, options.keys(), expectedKeyIndexes, accumulator);
|
|
142
|
+
reportDuplicates(
|
|
143
|
+
actualEntries, false, options.keys(), actualKeyIndexes, accumulator);
|
|
144
|
+
|
|
145
|
+
long line = expectedRow != null ? expectedRow.line : actualRow.line;
|
|
146
|
+
compareRow(
|
|
147
|
+
line,
|
|
148
|
+
displayKey,
|
|
149
|
+
expectedRow == null ? null : expectedRow.values,
|
|
150
|
+
actualRow == null ? null : actualRow.values,
|
|
151
|
+
expectedColumns,
|
|
152
|
+
actualColumns,
|
|
153
|
+
options,
|
|
154
|
+
accumulator);
|
|
155
|
+
}
|
|
156
|
+
return stats;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
private static Map<CompositeKey, List<RowEntry>> indexRows(
|
|
160
|
+
TabularReader reader,
|
|
161
|
+
int[] keyIndexes,
|
|
162
|
+
ValidationOptions options,
|
|
163
|
+
ComparisonStats stats,
|
|
164
|
+
boolean expected) throws IOException {
|
|
165
|
+
Map<CompositeKey, List<RowEntry>> rows = new HashMap<>();
|
|
166
|
+
long line = 0;
|
|
167
|
+
while (true) {
|
|
168
|
+
List<String> values = reader.readRow();
|
|
169
|
+
if (values == null) {
|
|
170
|
+
return rows;
|
|
171
|
+
}
|
|
172
|
+
line++;
|
|
173
|
+
if (expected) {
|
|
174
|
+
stats.observeExpected(values);
|
|
175
|
+
} else {
|
|
176
|
+
stats.observeActual(values);
|
|
177
|
+
}
|
|
178
|
+
List<String> keyValues = new ArrayList<>(keyIndexes.length);
|
|
179
|
+
for (int index : keyIndexes) {
|
|
180
|
+
String raw = index < values.size() ? values.get(index) : null;
|
|
181
|
+
keyValues.add(ValueNormalizer.normalize(raw, options.normalizations()));
|
|
182
|
+
}
|
|
183
|
+
CompositeKey key = new CompositeKey(keyValues);
|
|
184
|
+
rows.computeIfAbsent(key, ignored -> new ArrayList<>()).add(new RowEntry(line, values));
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
private static void reportDuplicates(
|
|
189
|
+
List<RowEntry> entries,
|
|
190
|
+
boolean expected,
|
|
191
|
+
List<String> keys,
|
|
192
|
+
int[] keyIndexes,
|
|
193
|
+
Accumulator accumulator) {
|
|
194
|
+
for (int duplicateIndex = 1; duplicateIndex < entries.size(); duplicateIndex++) {
|
|
195
|
+
RowEntry duplicate = entries.get(duplicateIndex);
|
|
196
|
+
Map<String, String> displayKey = displayKey(keys, duplicate, keyIndexes);
|
|
197
|
+
int firstIndex = keyIndexes[0];
|
|
198
|
+
String firstValue = firstIndex < duplicate.values.size() ? duplicate.values.get(firstIndex) : null;
|
|
199
|
+
accumulator.add(new Difference(
|
|
200
|
+
duplicate.line,
|
|
201
|
+
displayKey,
|
|
202
|
+
firstIndex + 1,
|
|
203
|
+
keys.get(0),
|
|
204
|
+
expected ? firstValue : null,
|
|
205
|
+
expected ? null : firstValue,
|
|
206
|
+
expected ? "DUPLICATE_EXPECTED_KEY" : "DUPLICATE_ACTUAL_KEY"));
|
|
207
|
+
}
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
private static Map<String, String> displayKey(List<String> names, RowEntry row, int[] indexes) {
|
|
211
|
+
Map<String, String> result = new LinkedHashMap<>();
|
|
212
|
+
for (int index = 0; index < names.size(); index++) {
|
|
213
|
+
int column = indexes[index];
|
|
214
|
+
result.put(names.get(index), column < row.values.size() ? row.values.get(column) : null);
|
|
215
|
+
}
|
|
216
|
+
return result;
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
private static int[] keyIndexes(List<String> keys, List<String> columns, String side) {
|
|
220
|
+
int[] result = new int[keys.size()];
|
|
221
|
+
for (int keyIndex = 0; keyIndex < keys.size(); keyIndex++) {
|
|
222
|
+
String key = keys.get(keyIndex);
|
|
223
|
+
int found = -1;
|
|
224
|
+
for (int columnIndex = 0; columnIndex < columns.size(); columnIndex++) {
|
|
225
|
+
if (key.equals(columns.get(columnIndex))) {
|
|
226
|
+
if (found >= 0) {
|
|
227
|
+
throw new IllegalArgumentException("Coluna de chave ambígua em " + side + ": " + key);
|
|
228
|
+
}
|
|
229
|
+
found = columnIndex;
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
if (found < 0) {
|
|
233
|
+
throw new IllegalArgumentException("Coluna de chave ausente em " + side + ": " + key);
|
|
234
|
+
}
|
|
235
|
+
result[keyIndex] = found;
|
|
236
|
+
}
|
|
237
|
+
return result;
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
private static void compareColumns(
|
|
241
|
+
List<String> expected,
|
|
242
|
+
List<String> actual,
|
|
243
|
+
Accumulator accumulator) {
|
|
244
|
+
int width = Math.max(expected.size(), actual.size());
|
|
245
|
+
for (int index = 0; index < width; index++) {
|
|
246
|
+
boolean hasExpected = index < expected.size();
|
|
247
|
+
boolean hasActual = index < actual.size();
|
|
248
|
+
String expectedName = hasExpected ? expected.get(index) : null;
|
|
249
|
+
String actualName = hasActual ? actual.get(index) : null;
|
|
250
|
+
if (hasExpected && hasActual && Objects.equals(expectedName, actualName)) {
|
|
251
|
+
continue;
|
|
252
|
+
}
|
|
253
|
+
String reason;
|
|
254
|
+
if (!hasExpected) {
|
|
255
|
+
reason = "UNEXPECTED_COLUMN";
|
|
256
|
+
} else if (!hasActual) {
|
|
257
|
+
reason = "MISSING_COLUMN";
|
|
258
|
+
} else {
|
|
259
|
+
reason = "COLUMN_NAME_MISMATCH";
|
|
260
|
+
}
|
|
261
|
+
accumulator.add(new Difference(
|
|
262
|
+
0,
|
|
263
|
+
index + 1,
|
|
264
|
+
expectedName != null ? expectedName : actualName,
|
|
265
|
+
expectedName,
|
|
266
|
+
actualName,
|
|
267
|
+
reason));
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
private static void compareRow(
|
|
272
|
+
long line,
|
|
273
|
+
Map<String, String> key,
|
|
274
|
+
List<String> expected,
|
|
275
|
+
List<String> actual,
|
|
276
|
+
List<String> expectedColumns,
|
|
277
|
+
List<String> actualColumns,
|
|
278
|
+
ValidationOptions options,
|
|
279
|
+
Accumulator accumulator) {
|
|
280
|
+
int expectedSize = expected == null ? 0 : expected.size();
|
|
281
|
+
int actualSize = actual == null ? 0 : actual.size();
|
|
282
|
+
int width = Math.max(expectedSize, actualSize);
|
|
283
|
+
for (int index = 0; index < width; index++) {
|
|
284
|
+
boolean hasExpected = index < expectedSize;
|
|
285
|
+
boolean hasActual = index < actualSize;
|
|
286
|
+
String expectedValue = hasExpected ? expected.get(index) : null;
|
|
287
|
+
String actualValue = hasActual ? actual.get(index) : null;
|
|
288
|
+
|
|
289
|
+
String reason = null;
|
|
290
|
+
if (!hasExpected) {
|
|
291
|
+
reason = "UNEXPECTED_ACTUAL_CELL";
|
|
292
|
+
} else if (!hasActual) {
|
|
293
|
+
reason = "MISSING_ACTUAL_CELL";
|
|
294
|
+
} else if (!Objects.equals(
|
|
295
|
+
ValueNormalizer.normalize(expectedValue, options.normalizations()),
|
|
296
|
+
ValueNormalizer.normalize(actualValue, options.normalizations()))) {
|
|
297
|
+
reason = "VALUE_MISMATCH";
|
|
298
|
+
}
|
|
299
|
+
if (reason != null) {
|
|
300
|
+
accumulator.add(new Difference(
|
|
301
|
+
line,
|
|
302
|
+
key,
|
|
303
|
+
index + 1,
|
|
304
|
+
columnName(index, expectedColumns, actualColumns),
|
|
305
|
+
expectedValue,
|
|
306
|
+
actualValue,
|
|
307
|
+
reason));
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
private static String columnName(int index, List<String> expected, List<String> actual) {
|
|
313
|
+
if (index < expected.size()) {
|
|
314
|
+
return expected.get(index);
|
|
315
|
+
}
|
|
316
|
+
if (index < actual.size()) {
|
|
317
|
+
return actual.get(index);
|
|
318
|
+
}
|
|
319
|
+
return "COLUMN_" + (index + 1);
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
private record RowEntry(long line, List<String> values) {
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
private record CompositeKey(List<String> values) implements Comparable<CompositeKey> {
|
|
326
|
+
private CompositeKey {
|
|
327
|
+
values = Collections.unmodifiableList(new ArrayList<>(values));
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
@Override
|
|
331
|
+
public int compareTo(CompositeKey other) {
|
|
332
|
+
for (int index = 0; index < values.size(); index++) {
|
|
333
|
+
String left = values.get(index);
|
|
334
|
+
String right = other.values.get(index);
|
|
335
|
+
if (Objects.equals(left, right)) {
|
|
336
|
+
continue;
|
|
337
|
+
}
|
|
338
|
+
if (left == null) {
|
|
339
|
+
return -1;
|
|
340
|
+
}
|
|
341
|
+
if (right == null) {
|
|
342
|
+
return 1;
|
|
343
|
+
}
|
|
344
|
+
int comparison = left.compareTo(right);
|
|
345
|
+
if (comparison != 0) {
|
|
346
|
+
return comparison;
|
|
347
|
+
}
|
|
348
|
+
}
|
|
349
|
+
return Integer.compare(values.size(), other.values.size());
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
private static final class ComparisonStats {
|
|
354
|
+
private long expectedRows;
|
|
355
|
+
private long actualRows;
|
|
356
|
+
private int expectedColumns;
|
|
357
|
+
private int actualColumns;
|
|
358
|
+
|
|
359
|
+
private ComparisonStats(int expectedColumns, int actualColumns) {
|
|
360
|
+
this.expectedColumns = expectedColumns;
|
|
361
|
+
this.actualColumns = actualColumns;
|
|
362
|
+
}
|
|
363
|
+
|
|
364
|
+
private void observeExpected(List<String> row) {
|
|
365
|
+
if (row != null) {
|
|
366
|
+
expectedRows++;
|
|
367
|
+
expectedColumns = Math.max(expectedColumns, row.size());
|
|
368
|
+
}
|
|
369
|
+
}
|
|
370
|
+
|
|
371
|
+
private void observeActual(List<String> row) {
|
|
372
|
+
if (row != null) {
|
|
373
|
+
actualRows++;
|
|
374
|
+
actualColumns = Math.max(actualColumns, row.size());
|
|
375
|
+
}
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
private static final class Accumulator {
|
|
380
|
+
private final int maxDifferences;
|
|
381
|
+
private final List<Difference> differences = new ArrayList<>();
|
|
382
|
+
private long count;
|
|
383
|
+
|
|
384
|
+
private Accumulator(int maxDifferences) {
|
|
385
|
+
this.maxDifferences = maxDifferences;
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
private void add(Difference difference) {
|
|
389
|
+
count++;
|
|
390
|
+
if (differences.size() < maxDifferences) {
|
|
391
|
+
differences.add(difference);
|
|
392
|
+
}
|
|
393
|
+
}
|
|
394
|
+
}
|
|
395
|
+
}
|
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
package io.mainframe.migration.validator;
|
|
2
|
+
|
|
3
|
+
import picocli.CommandLine;
|
|
4
|
+
import picocli.CommandLine.Command;
|
|
5
|
+
import picocli.CommandLine.Option;
|
|
6
|
+
import picocli.CommandLine.Parameters;
|
|
7
|
+
import picocli.CommandLine.Spec;
|
|
8
|
+
import picocli.CommandLine.Model.CommandSpec;
|
|
9
|
+
|
|
10
|
+
import java.io.IOException;
|
|
11
|
+
import java.nio.charset.Charset;
|
|
12
|
+
import java.nio.file.Path;
|
|
13
|
+
import java.util.ArrayList;
|
|
14
|
+
import java.util.EnumSet;
|
|
15
|
+
import java.util.List;
|
|
16
|
+
import java.util.Locale;
|
|
17
|
+
import java.util.concurrent.Callable;
|
|
18
|
+
|
|
19
|
+
@Command(
|
|
20
|
+
name = "dataset-validator",
|
|
21
|
+
description = "Compara arquivos CSV e Parquet célula a célula.",
|
|
22
|
+
mixinStandardHelpOptions = true,
|
|
23
|
+
version = "dataset-validator 0.1.0",
|
|
24
|
+
sortOptions = false)
|
|
25
|
+
public final class ValidatorCli implements Callable<Integer> {
|
|
26
|
+
public static final int EXIT_MATCH = 0;
|
|
27
|
+
public static final int EXIT_DIFFERENT = 1;
|
|
28
|
+
public static final int EXIT_INPUT_ERROR = 2;
|
|
29
|
+
public static final int EXIT_INTERNAL_ERROR = 3;
|
|
30
|
+
|
|
31
|
+
@Parameters(index = "0", paramLabel = "EXPECTED", description = "Arquivo de referência.")
|
|
32
|
+
private Path expected;
|
|
33
|
+
|
|
34
|
+
@Parameters(index = "1", paramLabel = "ACTUAL", description = "Arquivo gerado.")
|
|
35
|
+
private Path actual;
|
|
36
|
+
|
|
37
|
+
@Option(names = "--format", defaultValue = "auto", paramLabel = "auto|csv|parquet",
|
|
38
|
+
description = "Formato dos dois arquivos; padrão: ${DEFAULT-VALUE}.")
|
|
39
|
+
private String format;
|
|
40
|
+
|
|
41
|
+
@Option(names = "--expected-format", paramLabel = "auto|csv|parquet",
|
|
42
|
+
description = "Sobrescreve o formato do arquivo de referência.")
|
|
43
|
+
private String expectedFormat;
|
|
44
|
+
|
|
45
|
+
@Option(names = "--actual-format", paramLabel = "auto|csv|parquet",
|
|
46
|
+
description = "Sobrescreve o formato do arquivo gerado.")
|
|
47
|
+
private String actualFormat;
|
|
48
|
+
|
|
49
|
+
@Option(names = {"-d", "--delimiter"}, defaultValue = ",", paramLabel = "CHAR",
|
|
50
|
+
description = "Separador CSV; aceita caractere, tab, comma, semicolon ou pipe.")
|
|
51
|
+
private String delimiter;
|
|
52
|
+
|
|
53
|
+
@Option(names = "--expected-delimiter", paramLabel = "CHAR",
|
|
54
|
+
description = "Sobrescreve o separador CSV da referência.")
|
|
55
|
+
private String expectedDelimiter;
|
|
56
|
+
|
|
57
|
+
@Option(names = "--actual-delimiter", paramLabel = "CHAR",
|
|
58
|
+
description = "Sobrescreve o separador CSV do arquivo gerado.")
|
|
59
|
+
private String actualDelimiter;
|
|
60
|
+
|
|
61
|
+
@Option(names = "--header", arity = "1", defaultValue = "true", paramLabel = "true|false",
|
|
62
|
+
description = "CSV contém cabeçalho; padrão: ${DEFAULT-VALUE}.")
|
|
63
|
+
private boolean header;
|
|
64
|
+
|
|
65
|
+
@Option(names = "--no-header", description = "CSV não contém cabeçalho.")
|
|
66
|
+
private boolean noHeader;
|
|
67
|
+
|
|
68
|
+
@Option(names = "--expected-header", arity = "1", paramLabel = "true|false",
|
|
69
|
+
description = "Sobrescreve o uso de cabeçalho da referência.")
|
|
70
|
+
private Boolean expectedHeader;
|
|
71
|
+
|
|
72
|
+
@Option(names = "--actual-header", arity = "1", paramLabel = "true|false",
|
|
73
|
+
description = "Sobrescreve o uso de cabeçalho do arquivo gerado.")
|
|
74
|
+
private Boolean actualHeader;
|
|
75
|
+
|
|
76
|
+
@Option(names = "--charset", defaultValue = "UTF-8", paramLabel = "CHARSET",
|
|
77
|
+
description = "Codificação dos CSVs; padrão: ${DEFAULT-VALUE}.")
|
|
78
|
+
private String charset;
|
|
79
|
+
|
|
80
|
+
@Option(names = "--normalize", split = ",", paramLabel = "trim,decimal",
|
|
81
|
+
description = "Normalizações opcionais; repetível.")
|
|
82
|
+
private List<String> normalizationNames = new ArrayList<>();
|
|
83
|
+
|
|
84
|
+
@Option(names = "--trim", description = "Atalho para --normalize trim.")
|
|
85
|
+
private boolean trim;
|
|
86
|
+
|
|
87
|
+
@Option(names = "--normalize-decimal", description = "Atalho para --normalize decimal.")
|
|
88
|
+
private boolean normalizeDecimal;
|
|
89
|
+
|
|
90
|
+
@Option(names = "--max-differences", defaultValue = "100", paramLabel = "N",
|
|
91
|
+
description = "Máximo de diferenças detalhadas no JSON; padrão: ${DEFAULT-VALUE}.")
|
|
92
|
+
private int maxDifferences;
|
|
93
|
+
|
|
94
|
+
@Option(names = "--key", split = ",", paramLabel = "COL[,COL...]",
|
|
95
|
+
description = "Compara linhas por chave; repetível e aceita lista separada por vírgulas.")
|
|
96
|
+
private List<String> keys = new ArrayList<>();
|
|
97
|
+
|
|
98
|
+
@Option(names = {"-r", "--report"}, defaultValue = "-", paramLabel = "PATH",
|
|
99
|
+
description = "Destino do relatório JSON; '-' escreve no stdout.")
|
|
100
|
+
private String report;
|
|
101
|
+
|
|
102
|
+
@Spec
|
|
103
|
+
private CommandSpec spec;
|
|
104
|
+
|
|
105
|
+
@Override
|
|
106
|
+
public Integer call() {
|
|
107
|
+
try {
|
|
108
|
+
DataFormat commonFormat = DataFormat.parse(format);
|
|
109
|
+
DataFormat resolvedExpectedFormat = expectedFormat == null
|
|
110
|
+
? commonFormat
|
|
111
|
+
: DataFormat.parse(expectedFormat);
|
|
112
|
+
DataFormat resolvedActualFormat = actualFormat == null
|
|
113
|
+
? commonFormat
|
|
114
|
+
: DataFormat.parse(actualFormat);
|
|
115
|
+
char commonDelimiter = parseDelimiter(delimiter);
|
|
116
|
+
Charset csvCharset = Charset.forName(charset);
|
|
117
|
+
|
|
118
|
+
EnumSet<Normalization> normalizations = EnumSet.noneOf(Normalization.class);
|
|
119
|
+
for (String name : normalizationNames) {
|
|
120
|
+
normalizations.add(Normalization.parse(name));
|
|
121
|
+
}
|
|
122
|
+
if (trim) {
|
|
123
|
+
normalizations.add(Normalization.TRIM);
|
|
124
|
+
}
|
|
125
|
+
if (normalizeDecimal) {
|
|
126
|
+
normalizations.add(Normalization.DECIMAL);
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
boolean commonHeader = noHeader ? false : header;
|
|
130
|
+
ValidationOptions options = new ValidationOptions(
|
|
131
|
+
new InputOptions(
|
|
132
|
+
expected,
|
|
133
|
+
resolvedExpectedFormat,
|
|
134
|
+
expectedDelimiter == null ? commonDelimiter : parseDelimiter(expectedDelimiter),
|
|
135
|
+
expectedHeader == null ? commonHeader : expectedHeader,
|
|
136
|
+
csvCharset),
|
|
137
|
+
new InputOptions(
|
|
138
|
+
actual,
|
|
139
|
+
resolvedActualFormat,
|
|
140
|
+
actualDelimiter == null ? commonDelimiter : parseDelimiter(actualDelimiter),
|
|
141
|
+
actualHeader == null ? commonHeader : actualHeader,
|
|
142
|
+
csvCharset),
|
|
143
|
+
normalizations,
|
|
144
|
+
maxDifferences,
|
|
145
|
+
keys);
|
|
146
|
+
|
|
147
|
+
ValidationReport result = new ValidationService().validate(options);
|
|
148
|
+
JsonReportWriter writer = new JsonReportWriter();
|
|
149
|
+
if ("-".equals(report)) {
|
|
150
|
+
writer.write(result, spec.commandLine().getOut());
|
|
151
|
+
} else {
|
|
152
|
+
writer.write(result, Path.of(report));
|
|
153
|
+
}
|
|
154
|
+
return result.matches() ? EXIT_MATCH : EXIT_DIFFERENT;
|
|
155
|
+
} catch (IllegalArgumentException | IOException exception) {
|
|
156
|
+
spec.commandLine().getErr().println("Erro: " + exception.getMessage());
|
|
157
|
+
return EXIT_INPUT_ERROR;
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
static char parseDelimiter(String value) {
|
|
162
|
+
if (value == null) {
|
|
163
|
+
throw new IllegalArgumentException("Separador CSV ausente");
|
|
164
|
+
}
|
|
165
|
+
String decoded = switch (value.toLowerCase(Locale.ROOT)) {
|
|
166
|
+
case "tab", "\\t" -> "\t";
|
|
167
|
+
case "comma" -> ",";
|
|
168
|
+
case "semicolon" -> ";";
|
|
169
|
+
case "pipe" -> "|";
|
|
170
|
+
default -> value;
|
|
171
|
+
};
|
|
172
|
+
if (decoded.length() != 1) {
|
|
173
|
+
throw new IllegalArgumentException("Separador CSV deve conter exatamente um caractere: " + value);
|
|
174
|
+
}
|
|
175
|
+
return decoded.charAt(0);
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
public static CommandLine commandLine() {
|
|
179
|
+
CommandLine commandLine = new CommandLine(new ValidatorCli());
|
|
180
|
+
commandLine.setExecutionExceptionHandler((exception, current, parseResult) -> {
|
|
181
|
+
current.getErr().println("Erro interno: " + exception.getMessage());
|
|
182
|
+
return EXIT_INTERNAL_ERROR;
|
|
183
|
+
});
|
|
184
|
+
return commandLine;
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
public static void main(String[] args) {
|
|
188
|
+
System.exit(commandLine().execute(args));
|
|
189
|
+
}
|
|
190
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
package io.mainframe.migration.validator;
|
|
2
|
+
|
|
3
|
+
import java.math.BigDecimal;
|
|
4
|
+
import java.util.Set;
|
|
5
|
+
|
|
6
|
+
final class ValueNormalizer {
|
|
7
|
+
private ValueNormalizer() {
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
static String normalize(String value, Set<Normalization> normalizations) {
|
|
11
|
+
if (value == null) {
|
|
12
|
+
return null;
|
|
13
|
+
}
|
|
14
|
+
String normalized = normalizations.contains(Normalization.TRIM) ? value.strip() : value;
|
|
15
|
+
if (normalizations.contains(Normalization.DECIMAL)) {
|
|
16
|
+
try {
|
|
17
|
+
BigDecimal decimal = new BigDecimal(normalized);
|
|
18
|
+
normalized = decimal.compareTo(BigDecimal.ZERO) == 0
|
|
19
|
+
? "0"
|
|
20
|
+
: decimal.stripTrailingZeros().toPlainString();
|
|
21
|
+
} catch (NumberFormatException ignored) {
|
|
22
|
+
// Non-numeric cells remain unchanged.
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
return normalized;
|
|
26
|
+
}
|
|
27
|
+
}
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
package io.mainframe.migration.validator;
|
|
2
|
+
|
|
3
|
+
import org.junit.jupiter.api.Test;
|
|
4
|
+
import org.junit.jupiter.api.io.TempDir;
|
|
5
|
+
|
|
6
|
+
import java.nio.charset.StandardCharsets;
|
|
7
|
+
import java.nio.file.Files;
|
|
8
|
+
import java.nio.file.Path;
|
|
9
|
+
import java.util.Set;
|
|
10
|
+
|
|
11
|
+
import static org.junit.jupiter.api.Assertions.assertEquals;
|
|
12
|
+
import static org.junit.jupiter.api.Assertions.assertThrows;
|
|
13
|
+
import static org.junit.jupiter.api.Assertions.assertTrue;
|
|
14
|
+
|
|
15
|
+
class DirectoryValidationTest {
|
|
16
|
+
@TempDir
|
|
17
|
+
Path temporaryDirectory;
|
|
18
|
+
|
|
19
|
+
@Test
|
|
20
|
+
void readsSparkCsvPartsInLexicalOrderAndSkipsMarkers() throws Exception {
|
|
21
|
+
Path expected = Files.createDirectory(temporaryDirectory.resolve("expected"));
|
|
22
|
+
Files.writeString(expected.resolve("part-00001.csv"), "id,name\n2,Bia\n");
|
|
23
|
+
Files.writeString(expected.resolve("part-00000.csv"), "id,name\n1,Ana\n");
|
|
24
|
+
Files.writeString(expected.resolve("_SUCCESS"), "");
|
|
25
|
+
Files.writeString(expected.resolve(".part-hidden.csv"), "id,name\n9,Hidden\n");
|
|
26
|
+
Files.writeString(expected.resolve("notes.csv"), "id,name\n8,Ignored\n");
|
|
27
|
+
|
|
28
|
+
Path actual = Files.createDirectory(temporaryDirectory.resolve("actual"));
|
|
29
|
+
Files.writeString(actual.resolve("part-00000.csv"), "id,name\n1,Ana\n2,Bia\n");
|
|
30
|
+
|
|
31
|
+
ValidationReport report = validate(expected, actual, DataFormat.AUTO);
|
|
32
|
+
|
|
33
|
+
assertTrue(report.matches());
|
|
34
|
+
assertEquals(2, report.expectedRows());
|
|
35
|
+
assertEquals(2, report.actualRows());
|
|
36
|
+
assertEquals("CSV", report.expectedFormat());
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
@Test
|
|
40
|
+
void rejectsMixedSparkFormatsWhenFormatIsAuto() throws Exception {
|
|
41
|
+
Path mixed = Files.createDirectory(temporaryDirectory.resolve("mixed"));
|
|
42
|
+
Files.writeString(mixed.resolve("part-00000.csv"), "id\n1\n");
|
|
43
|
+
Files.writeString(mixed.resolve("part-00001.parquet"), "not parquet");
|
|
44
|
+
|
|
45
|
+
IllegalArgumentException exception = assertThrows(
|
|
46
|
+
IllegalArgumentException.class,
|
|
47
|
+
() -> DataFormat.AUTO.resolve(mixed));
|
|
48
|
+
|
|
49
|
+
assertTrue(exception.getMessage().contains("CSV e Parquet"));
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
@Test
|
|
53
|
+
void explicitFormatSelectsOnlyMatchingParts() throws Exception {
|
|
54
|
+
Path expected = Files.createDirectory(temporaryDirectory.resolve("expected"));
|
|
55
|
+
Files.writeString(expected.resolve("part-00000.csv"), "id\n1\n");
|
|
56
|
+
Files.writeString(expected.resolve("part-00001.parquet"), "ignored");
|
|
57
|
+
Path actual = Files.createDirectory(temporaryDirectory.resolve("actual"));
|
|
58
|
+
Files.writeString(actual.resolve("part-00000.csv"), "id\n1\n");
|
|
59
|
+
|
|
60
|
+
ValidationReport report = validate(expected, actual, DataFormat.CSV);
|
|
61
|
+
|
|
62
|
+
assertTrue(report.matches());
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
private ValidationReport validate(Path expected, Path actual, DataFormat format) throws Exception {
|
|
66
|
+
InputOptions left = new InputOptions(expected, format, ',', true, StandardCharsets.UTF_8);
|
|
67
|
+
InputOptions right = new InputOptions(actual, format, ',', true, StandardCharsets.UTF_8);
|
|
68
|
+
return new ValidationService().validate(new ValidationOptions(left, right, Set.of(), 100));
|
|
69
|
+
}
|
|
70
|
+
}
|