symtest-cli 1.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- symtest/__init__.py +45 -0
- symtest/cli.py +549 -0
- symtest/commands/__init__.py +9 -0
- symtest/commands/compare.py +221 -0
- symtest/config/__init__.py +7 -0
- symtest/config/config_io.py +346 -0
- symtest/config/config_schema.py +330 -0
- symtest/config/import_expander.py +149 -0
- symtest/config/inheritance_expander.py +197 -0
- symtest/core/__init__.py +15 -0
- symtest/core/assertions.py +253 -0
- symtest/core/base_runner.py +299 -0
- symtest/core/config_loader.py +536 -0
- symtest/core/execution.py +498 -0
- symtest/core/history_store.py +96 -0
- symtest/core/last_run_store.py +109 -0
- symtest/core/parallel_runner.py +251 -0
- symtest/core/process_worker.py +93 -0
- symtest/core/sequence_state.py +143 -0
- symtest/core/setup.py +137 -0
- symtest/core/test_case.py +76 -0
- symtest/core/types.py +92 -0
- symtest/file_comparator/__init__.py +10 -0
- symtest/file_comparator/base_comparator.py +109 -0
- symtest/file_comparator/binary_comparator.py +399 -0
- symtest/file_comparator/csv_comparator.py +241 -0
- symtest/file_comparator/factory.py +191 -0
- symtest/file_comparator/h5_comparator.py +777 -0
- symtest/file_comparator/json_comparator.py +323 -0
- symtest/file_comparator/result.py +213 -0
- symtest/file_comparator/script_comparator.py +182 -0
- symtest/file_comparator/text_comparator.py +182 -0
- symtest/file_comparator/xml_comparator.py +150 -0
- symtest/logging_config.py +66 -0
- symtest/runners/__init__.py +15 -0
- symtest/runners/config_runner.py +96 -0
- symtest/runners/json_runner.py +21 -0
- symtest/runners/parallel_config_runner.py +278 -0
- symtest/runners/parallel_json_runner.py +26 -0
- symtest/runners/parallel_yaml_runner.py +31 -0
- symtest/runners/yaml_runner.py +26 -0
- symtest/tui/__init__.py +11 -0
- symtest/tui/app.py +90 -0
- symtest/tui/controllers/__init__.py +0 -0
- symtest/tui/controllers/case_controller.py +322 -0
- symtest/tui/screens/__init__.py +0 -0
- symtest/tui/screens/case_editor.py +244 -0
- symtest/tui/screens/case_list.py +255 -0
- symtest/tui/widgets/__init__.py +0 -0
- symtest/tui/widgets/case_table.py +113 -0
- symtest/tui/widgets/expected_editor.py +159 -0
- symtest/tui/widgets/search_bar.py +160 -0
- symtest/tui/widgets/steps_editor.py +243 -0
- symtest/utils/__init__.py +21 -0
- symtest/utils/junit_xml_writer.py +137 -0
- symtest/utils/path_resolver.py +124 -0
- symtest/utils/report_generator.py +208 -0
- symtest_cli-1.3.0.dist-info/METADATA +316 -0
- symtest_cli-1.3.0.dist-info/RECORD +63 -0
- symtest_cli-1.3.0.dist-info/WHEEL +5 -0
- symtest_cli-1.3.0.dist-info/entry_points.txt +4 -0
- symtest_cli-1.3.0.dist-info/licenses/LICENSE +21 -0
- symtest_cli-1.3.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,399 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
|
|
4
|
+
"""
|
|
5
|
+
@file binary_comparator.py
|
|
6
|
+
@brief Binary file comparator implementation with efficient byte-level comparison
|
|
7
|
+
@author Xiaotong Wang
|
|
8
|
+
@date 2025
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import difflib
|
|
12
|
+
import hashlib
|
|
13
|
+
from .base_comparator import BaseComparator
|
|
14
|
+
from .result import Difference
|
|
15
|
+
|
|
16
|
+
class BinaryComparator(BaseComparator):
|
|
17
|
+
"""
|
|
18
|
+
@brief Comparator for binary files with efficient byte-level comparison
|
|
19
|
+
@details This class implements binary file comparison with support for:
|
|
20
|
+
- Byte-level difference detection
|
|
21
|
+
- Similarity index calculation using LCS
|
|
22
|
+
- Parallel processing for large files
|
|
23
|
+
- File hash calculation
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
def __init__(self, encoding="utf-8", chunk_size=8192, verbose=False, similarity=False, num_threads=4, **kwargs):
|
|
27
|
+
"""
|
|
28
|
+
@brief Initialize the binary comparator
|
|
29
|
+
@param encoding str: File encoding (not used for binary files)
|
|
30
|
+
@param chunk_size int: Size of chunks for reading large files
|
|
31
|
+
@param verbose bool: Enable verbose logging
|
|
32
|
+
@param similarity bool: Enable similarity index calculation
|
|
33
|
+
@param num_threads int: Number of threads for parallel processing
|
|
34
|
+
@param **kwargs: Additional parameters (ignored)
|
|
35
|
+
"""
|
|
36
|
+
super().__init__(encoding=encoding, chunk_size=chunk_size, verbose=verbose, **kwargs)
|
|
37
|
+
self.similarity = similarity
|
|
38
|
+
self.num_threads = num_threads
|
|
39
|
+
|
|
40
|
+
def read_content(self, file_path, start_line=0, end_line=None, start_column=0, end_column=None):
|
|
41
|
+
"""
|
|
42
|
+
@brief Read binary content with specified range
|
|
43
|
+
@param file_path Path: Path to the binary file to read
|
|
44
|
+
@param start_line int: Starting byte offset (interpreted as bytes for binary files)
|
|
45
|
+
@param end_line int: Ending byte offset (interpreted as bytes for binary files)
|
|
46
|
+
@param start_column int: Ignored for binary files
|
|
47
|
+
@param end_column int: Ignored for binary files
|
|
48
|
+
@return bytes: Binary content within the specified range
|
|
49
|
+
@throws ValueError: If byte offsets are invalid
|
|
50
|
+
@throws FileNotFoundError: If file doesn't exist
|
|
51
|
+
@throws IOError: If there are other file reading errors
|
|
52
|
+
"""
|
|
53
|
+
try:
|
|
54
|
+
self.logger.debug(f"Reading binary file: {file_path}")
|
|
55
|
+
|
|
56
|
+
# For binary files, interpret start_line as byte offset
|
|
57
|
+
start_offset = start_line
|
|
58
|
+
end_offset = end_line
|
|
59
|
+
|
|
60
|
+
with open(file_path, 'rb') as f:
|
|
61
|
+
if start_offset > 0:
|
|
62
|
+
f.seek(start_offset)
|
|
63
|
+
|
|
64
|
+
if end_offset is not None:
|
|
65
|
+
if end_offset <= start_offset:
|
|
66
|
+
raise ValueError("End offset must be greater than start offset")
|
|
67
|
+
bytes_to_read = end_offset - start_offset
|
|
68
|
+
content = f.read(bytes_to_read)
|
|
69
|
+
else:
|
|
70
|
+
content = f.read()
|
|
71
|
+
|
|
72
|
+
return content
|
|
73
|
+
|
|
74
|
+
except FileNotFoundError:
|
|
75
|
+
raise ValueError(f"File not found: {file_path}")
|
|
76
|
+
except IOError as e:
|
|
77
|
+
raise ValueError(f"Error reading file {file_path}: {str(e)}")
|
|
78
|
+
|
|
79
|
+
def compare_content(self, content1, content2):
|
|
80
|
+
"""
|
|
81
|
+
@brief Compare binary content efficiently
|
|
82
|
+
@param content1 bytes: First binary content to compare
|
|
83
|
+
@param content2 bytes: Second binary content to compare
|
|
84
|
+
@return tuple: (bool, list, bool) - (identical, differences, truncated)
|
|
85
|
+
@details Performs efficient byte-level comparison of binary content.
|
|
86
|
+
Reports differences with hex context and limits the number
|
|
87
|
+
of differences to avoid overwhelming output.
|
|
88
|
+
"""
|
|
89
|
+
self.logger.debug(f"Comparing binary content")
|
|
90
|
+
|
|
91
|
+
if len(content1) != len(content2):
|
|
92
|
+
differences = [Difference(
|
|
93
|
+
position="file size",
|
|
94
|
+
expected=f"{len(content1)} bytes",
|
|
95
|
+
actual=f"{len(content2)} bytes",
|
|
96
|
+
diff_type="size"
|
|
97
|
+
)]
|
|
98
|
+
identical = False
|
|
99
|
+
truncated = False
|
|
100
|
+
elif content1 == content2:
|
|
101
|
+
differences = []
|
|
102
|
+
identical = True
|
|
103
|
+
truncated = False
|
|
104
|
+
else:
|
|
105
|
+
identical = False
|
|
106
|
+
differences = []
|
|
107
|
+
offset = 0
|
|
108
|
+
max_differences = 10 # Limit number of differences reported
|
|
109
|
+
truncated = False
|
|
110
|
+
|
|
111
|
+
for i in range(0, len(content1), self.chunk_size):
|
|
112
|
+
chunk1 = content1[i:i+self.chunk_size]
|
|
113
|
+
chunk2 = content2[i:i+self.chunk_size]
|
|
114
|
+
|
|
115
|
+
if chunk1 != chunk2:
|
|
116
|
+
# Find the exact byte position where the difference starts
|
|
117
|
+
for j in range(len(chunk1)):
|
|
118
|
+
if j >= len(chunk2) or chunk1[j] != chunk2[j]:
|
|
119
|
+
diff_pos = i + j
|
|
120
|
+
# Show a few bytes before and after the difference for context
|
|
121
|
+
context_size = 8
|
|
122
|
+
start_ctx = max(0, diff_pos - context_size)
|
|
123
|
+
end_ctx = min(len(content1), diff_pos + context_size)
|
|
124
|
+
|
|
125
|
+
# Create hex representations of the differing sections
|
|
126
|
+
expected_bytes = content1[start_ctx:end_ctx]
|
|
127
|
+
actual_bytes = content2[start_ctx:min(len(content2), end_ctx)]
|
|
128
|
+
|
|
129
|
+
expected_hex = ' '.join(f"{b:02x}" for b in expected_bytes)
|
|
130
|
+
actual_hex = ' '.join(f"{b:02x}" for b in actual_bytes)
|
|
131
|
+
|
|
132
|
+
differences.append(Difference(
|
|
133
|
+
position=f"byte {diff_pos}",
|
|
134
|
+
expected=expected_hex,
|
|
135
|
+
actual=actual_hex,
|
|
136
|
+
diff_type="content"
|
|
137
|
+
))
|
|
138
|
+
break
|
|
139
|
+
|
|
140
|
+
if len(differences) >= max_differences:
|
|
141
|
+
truncated = True
|
|
142
|
+
break
|
|
143
|
+
|
|
144
|
+
return identical, differences, truncated
|
|
145
|
+
|
|
146
|
+
def _compute_similarity(self, a: bytes, b: bytes) -> float:
|
|
147
|
+
"""
|
|
148
|
+
@brief Compute similarity ratio between two binary sequences.
|
|
149
|
+
@param a bytes: First binary sequence
|
|
150
|
+
@param b bytes: Second binary sequence
|
|
151
|
+
@return float: Similarity ratio in [0.0, 1.0]
|
|
152
|
+
@details Uses difflib.SequenceMatcher for accurate comparison on small/medium
|
|
153
|
+
files, and hash-based chunk comparison for large files to avoid
|
|
154
|
+
O(n*m) complexity that would be infeasible on large binaries.
|
|
155
|
+
"""
|
|
156
|
+
if not a and not b:
|
|
157
|
+
return 1.0
|
|
158
|
+
|
|
159
|
+
total_bytes = len(a) + len(b)
|
|
160
|
+
if total_bytes == 0:
|
|
161
|
+
return 1.0
|
|
162
|
+
|
|
163
|
+
# difflib.SequenceMatcher uses a heuristic matching algorithm that works well
|
|
164
|
+
# for files under ~1 MB. Beyond that, fall back to chunk-hash approximation.
|
|
165
|
+
if total_bytes <= 1024 * 1024:
|
|
166
|
+
matcher = difflib.SequenceMatcher(None, a, b, autojunk=False)
|
|
167
|
+
return matcher.ratio()
|
|
168
|
+
|
|
169
|
+
return self._hash_chunk_similarity(a, b)
|
|
170
|
+
|
|
171
|
+
def _hash_chunk_similarity(self, a: bytes, b: bytes) -> float:
|
|
172
|
+
"""
|
|
173
|
+
@brief Approximate similarity via chunk-level hash matching for large files.
|
|
174
|
+
@param a bytes: First binary sequence
|
|
175
|
+
@param b bytes: Second binary sequence
|
|
176
|
+
@return float: Approximate similarity ratio in [0.0, 1.0]
|
|
177
|
+
@details Splits both inputs into fixed-size chunks (4 KB), hashes each chunk,
|
|
178
|
+
and computes Jaccard similarity on the chunk hash sets. This avoids
|
|
179
|
+
O(n*m) DP while giving a reasonable estimate of binary similarity.
|
|
180
|
+
"""
|
|
181
|
+
chunk_size = 4096
|
|
182
|
+
|
|
183
|
+
hashes_a = set()
|
|
184
|
+
for i in range(0, len(a), chunk_size):
|
|
185
|
+
hashes_a.add(hash(a[i:i + chunk_size]))
|
|
186
|
+
|
|
187
|
+
hashes_b = set()
|
|
188
|
+
for i in range(0, len(b), chunk_size):
|
|
189
|
+
hashes_b.add(hash(b[i:i + chunk_size]))
|
|
190
|
+
|
|
191
|
+
if not hashes_a and not hashes_b:
|
|
192
|
+
return 1.0
|
|
193
|
+
|
|
194
|
+
intersection = len(hashes_a & hashes_b)
|
|
195
|
+
union = len(hashes_a | hashes_b)
|
|
196
|
+
|
|
197
|
+
if union == 0:
|
|
198
|
+
return 1.0
|
|
199
|
+
|
|
200
|
+
return intersection / union
|
|
201
|
+
|
|
202
|
+
def compare_files(self, file1, file2, start_line=0, end_line=None, start_column=0, end_column=None):
|
|
203
|
+
"""
|
|
204
|
+
@brief Compare two binary files with optional similarity calculation using chunk-based streaming
|
|
205
|
+
@param file1 Path: Path to the first binary file
|
|
206
|
+
@param file2 Path: Path to the second binary file
|
|
207
|
+
@param start_line int: Starting byte offset
|
|
208
|
+
@param end_line int: Ending byte offset
|
|
209
|
+
@param start_column int: Ignored for binary files
|
|
210
|
+
@param end_column int: Ignored for binary files
|
|
211
|
+
@return ComparisonResult: Result object containing comparison details
|
|
212
|
+
@details This method implements chunk-based streaming comparison to avoid loading
|
|
213
|
+
entire files into memory, making it suitable for large files with O(1) memory usage.
|
|
214
|
+
"""
|
|
215
|
+
from pathlib import Path
|
|
216
|
+
from .result import ComparisonResult
|
|
217
|
+
result = ComparisonResult(
|
|
218
|
+
file1=str(file1),
|
|
219
|
+
file2=str(file2),
|
|
220
|
+
start_line=start_line,
|
|
221
|
+
end_line=end_line,
|
|
222
|
+
start_column=start_column,
|
|
223
|
+
end_column=end_column
|
|
224
|
+
)
|
|
225
|
+
try:
|
|
226
|
+
self.logger.info(f"Comparing files: {file1} and {file2}")
|
|
227
|
+
file1_path = Path(file1)
|
|
228
|
+
file2_path = Path(file2)
|
|
229
|
+
result.file1_size = file1_path.stat().st_size
|
|
230
|
+
result.file2_size = file2_path.stat().st_size
|
|
231
|
+
|
|
232
|
+
# Quick size check: if file sizes differ and similarity is not requested,
|
|
233
|
+
# we can return early without streaming
|
|
234
|
+
if result.file1_size != result.file2_size and not self.similarity:
|
|
235
|
+
# Adjust sizes based on offset if specified
|
|
236
|
+
adjusted_size1 = result.file1_size - start_line
|
|
237
|
+
adjusted_size2 = result.file2_size - start_line
|
|
238
|
+
if end_line is not None:
|
|
239
|
+
adjusted_size1 = min(adjusted_size1, end_line - start_line)
|
|
240
|
+
adjusted_size2 = min(adjusted_size2, end_line - start_line)
|
|
241
|
+
|
|
242
|
+
if adjusted_size1 != adjusted_size2:
|
|
243
|
+
result.identical = False
|
|
244
|
+
result.differences.append(Difference(
|
|
245
|
+
position="file size",
|
|
246
|
+
expected=f"{result.file1_size} bytes",
|
|
247
|
+
actual=f"{result.file2_size} bytes",
|
|
248
|
+
diff_type="size"
|
|
249
|
+
))
|
|
250
|
+
return result
|
|
251
|
+
|
|
252
|
+
# If similarity calculation is needed, we still need to read full content
|
|
253
|
+
# but for regular comparison, use chunk-based streaming
|
|
254
|
+
if self.similarity:
|
|
255
|
+
# Similarity calculation requires full content for SequenceMatcher
|
|
256
|
+
# or chunk-hash comparison
|
|
257
|
+
self.logger.debug("Reading full content for similarity calculation")
|
|
258
|
+
content1 = self.read_content(file1, start_line, end_line, start_column, end_column)
|
|
259
|
+
content2 = self.read_content(file2, start_line, end_line, start_column, end_column)
|
|
260
|
+
identical, differences, truncated = self.compare_content(content1, content2)
|
|
261
|
+
result.similarity = self._compute_similarity(content1, content2)
|
|
262
|
+
else:
|
|
263
|
+
# Chunk-based streaming comparison for O(1) memory usage
|
|
264
|
+
self.logger.debug("Using chunk-based streaming comparison")
|
|
265
|
+
identical, differences, truncated = self._compare_files_streaming(
|
|
266
|
+
file1_path, file2_path, start_line, end_line
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
result.identical = identical
|
|
270
|
+
result.differences = differences
|
|
271
|
+
result.truncated = truncated
|
|
272
|
+
return result
|
|
273
|
+
except Exception as e:
|
|
274
|
+
self.logger.error(f"Error during comparison: {str(e)}")
|
|
275
|
+
result.error = str(e)
|
|
276
|
+
result.identical = False
|
|
277
|
+
return result
|
|
278
|
+
|
|
279
|
+
def _compare_files_streaming(self, file1_path, file2_path, start_offset=0, end_offset=None):
|
|
280
|
+
"""
|
|
281
|
+
@brief Compare two binary files using chunk-based streaming
|
|
282
|
+
@param file1_path Path: Path to the first binary file
|
|
283
|
+
@param file2_path Path: Path to the second binary file
|
|
284
|
+
@param start_offset int: Starting byte offset
|
|
285
|
+
@param end_offset int: Ending byte offset (None for end of file)
|
|
286
|
+
@return tuple: (bool, list, bool) - (identical, differences, truncated)
|
|
287
|
+
@details This method compares files chunk by chunk without loading entire files
|
|
288
|
+
into memory, achieving O(1) memory complexity.
|
|
289
|
+
"""
|
|
290
|
+
differences = []
|
|
291
|
+
max_differences = 10 # Limit number of differences reported
|
|
292
|
+
truncated = False
|
|
293
|
+
|
|
294
|
+
try:
|
|
295
|
+
with open(file1_path, 'rb') as f1, open(file2_path, 'rb') as f2:
|
|
296
|
+
# Seek to start offset if specified
|
|
297
|
+
if start_offset > 0:
|
|
298
|
+
f1.seek(start_offset)
|
|
299
|
+
f2.seek(start_offset)
|
|
300
|
+
|
|
301
|
+
# Calculate bytes to read if end_offset is specified
|
|
302
|
+
bytes_to_read = None
|
|
303
|
+
if end_offset is not None:
|
|
304
|
+
if end_offset <= start_offset:
|
|
305
|
+
raise ValueError("End offset must be greater than start offset")
|
|
306
|
+
bytes_to_read = end_offset - start_offset
|
|
307
|
+
|
|
308
|
+
chunk_size = self.chunk_size
|
|
309
|
+
current_offset = start_offset
|
|
310
|
+
bytes_read_total = 0
|
|
311
|
+
|
|
312
|
+
while True:
|
|
313
|
+
# Determine how many bytes to read in this chunk
|
|
314
|
+
if bytes_to_read is not None:
|
|
315
|
+
remaining = bytes_to_read - bytes_read_total
|
|
316
|
+
if remaining <= 0:
|
|
317
|
+
break
|
|
318
|
+
read_size = min(chunk_size, remaining)
|
|
319
|
+
else:
|
|
320
|
+
read_size = chunk_size
|
|
321
|
+
|
|
322
|
+
# Read chunks from both files
|
|
323
|
+
chunk1 = f1.read(read_size)
|
|
324
|
+
chunk2 = f2.read(read_size)
|
|
325
|
+
|
|
326
|
+
# If both files are exhausted, we're done
|
|
327
|
+
if not chunk1 and not chunk2:
|
|
328
|
+
break
|
|
329
|
+
|
|
330
|
+
# If one file ends before the other, that's a difference
|
|
331
|
+
if len(chunk1) != len(chunk2):
|
|
332
|
+
differences.append(Difference(
|
|
333
|
+
position=f"byte {current_offset}",
|
|
334
|
+
expected=f"{len(chunk1)} bytes in chunk",
|
|
335
|
+
actual=f"{len(chunk2)} bytes in chunk",
|
|
336
|
+
diff_type="content"
|
|
337
|
+
))
|
|
338
|
+
break
|
|
339
|
+
|
|
340
|
+
# Compare chunks byte by byte
|
|
341
|
+
if chunk1 != chunk2:
|
|
342
|
+
# Find the exact byte position where the difference starts
|
|
343
|
+
for i in range(len(chunk1)):
|
|
344
|
+
if chunk1[i] != chunk2[i]:
|
|
345
|
+
abs_pos = current_offset + i
|
|
346
|
+
|
|
347
|
+
# Show a few bytes before and after the difference for context
|
|
348
|
+
context_size = 8
|
|
349
|
+
context_start = max(0, i - context_size)
|
|
350
|
+
context_end = min(len(chunk1), i + context_size)
|
|
351
|
+
|
|
352
|
+
# Get context bytes (may need to read previous chunk)
|
|
353
|
+
context1 = chunk1[context_start:context_end]
|
|
354
|
+
context2 = chunk2[context_start:context_end]
|
|
355
|
+
|
|
356
|
+
expected_hex = ' '.join(f"{b:02x}" for b in context1)
|
|
357
|
+
actual_hex = ' '.join(f"{b:02x}" for b in context2)
|
|
358
|
+
|
|
359
|
+
differences.append(Difference(
|
|
360
|
+
position=f"byte {abs_pos}",
|
|
361
|
+
expected=expected_hex,
|
|
362
|
+
actual=actual_hex,
|
|
363
|
+
diff_type="content"
|
|
364
|
+
))
|
|
365
|
+
|
|
366
|
+
# Stop after finding first difference in chunk
|
|
367
|
+
# or if we've reached max differences
|
|
368
|
+
if len(differences) >= max_differences:
|
|
369
|
+
truncated = True
|
|
370
|
+
return False, differences, truncated
|
|
371
|
+
break
|
|
372
|
+
|
|
373
|
+
current_offset += len(chunk1)
|
|
374
|
+
bytes_read_total += len(chunk1)
|
|
375
|
+
|
|
376
|
+
# If we didn't read a full chunk, we've reached EOF
|
|
377
|
+
if len(chunk1) < read_size:
|
|
378
|
+
break
|
|
379
|
+
|
|
380
|
+
identical = len(differences) == 0
|
|
381
|
+
return identical, differences, truncated
|
|
382
|
+
|
|
383
|
+
except FileNotFoundError as e:
|
|
384
|
+
raise ValueError(f"File not found: {e}")
|
|
385
|
+
except IOError as e:
|
|
386
|
+
raise ValueError(f"Error reading file: {str(e)}")
|
|
387
|
+
|
|
388
|
+
def get_file_hash(self, file_path, chunk_size=8192):
|
|
389
|
+
"""
|
|
390
|
+
@brief Calculate SHA-256 hash of a file efficiently
|
|
391
|
+
@param file_path Path: Path to the file to hash
|
|
392
|
+
@param chunk_size int: Size of chunks for reading large files
|
|
393
|
+
@return str: Hexadecimal representation of the file's SHA-256 hash
|
|
394
|
+
"""
|
|
395
|
+
h = hashlib.sha256()
|
|
396
|
+
with open(file_path, 'rb') as f:
|
|
397
|
+
for chunk in iter(lambda: f.read(chunk_size), b''):
|
|
398
|
+
h.update(chunk)
|
|
399
|
+
return h.hexdigest()
|
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
|
|
4
|
+
"""
|
|
5
|
+
@file csv_comparator.py
|
|
6
|
+
@brief CSV file comparator implementation with row and column comparison
|
|
7
|
+
@author Xiaotong Wang
|
|
8
|
+
@date 2025
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import csv
|
|
12
|
+
import io
|
|
13
|
+
import math
|
|
14
|
+
import re
|
|
15
|
+
from .text_comparator import TextComparator
|
|
16
|
+
from .result import Difference
|
|
17
|
+
|
|
18
|
+
class CsvComparator(TextComparator):
|
|
19
|
+
"""
|
|
20
|
+
@brief Comparator for CSV files with row and column comparison
|
|
21
|
+
@details This class extends TextComparator to provide specialized CSV comparison
|
|
22
|
+
capabilities, including:
|
|
23
|
+
- Row count comparison
|
|
24
|
+
- Column count comparison
|
|
25
|
+
- Cell value comparison
|
|
26
|
+
- Configurable delimiter and quote character
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
def __init__(self, encoding="utf-8", delimiter=",", quotechar='"', chunk_size=8192, verbose=False, rtol=1e-5, atol=1e-8, data_filter=None, error_analysis=False, **kwargs):
|
|
30
|
+
"""
|
|
31
|
+
@brief Initialize CSV comparator with configuration
|
|
32
|
+
@param encoding str: File encoding (default: utf-8)
|
|
33
|
+
@param delimiter str: CSV field delimiter (default: comma)
|
|
34
|
+
@param quotechar str: Character used for quoting fields (default: double quote)
|
|
35
|
+
@param chunk_size int: Size of chunks for reading large files
|
|
36
|
+
@param verbose bool: Enable verbose output
|
|
37
|
+
@param rtol float: Relative tolerance for numerical comparison (default: 1e-5)
|
|
38
|
+
@param atol float: Absolute tolerance for numerical comparison (default: 1e-8)
|
|
39
|
+
@param data_filter str: Data filter expression applied before numeric comparison
|
|
40
|
+
@param error_analysis bool: Enable streaming error statistics over ALL numeric cells
|
|
41
|
+
@param **kwargs: Additional parameters (ignored)
|
|
42
|
+
"""
|
|
43
|
+
super().__init__(encoding=encoding, chunk_size=chunk_size, verbose=verbose, **kwargs)
|
|
44
|
+
self.delimiter = delimiter
|
|
45
|
+
self.quotechar = quotechar
|
|
46
|
+
self.rtol = rtol
|
|
47
|
+
self.atol = atol
|
|
48
|
+
self.data_filter = data_filter
|
|
49
|
+
self.filter_func = self._parse_filter()
|
|
50
|
+
self.error_analysis = error_analysis
|
|
51
|
+
self._error_stats = None
|
|
52
|
+
|
|
53
|
+
def read_content(self, file_path, start_line=0, end_line=None, start_column=0, end_column=None):
|
|
54
|
+
"""
|
|
55
|
+
@brief Read and parse CSV content from file
|
|
56
|
+
@param file_path Path: Path to the CSV file
|
|
57
|
+
@param start_line int: Starting line number
|
|
58
|
+
@param end_line int: Ending line number
|
|
59
|
+
@param start_column int: Starting column number
|
|
60
|
+
@param end_column int: Ending column number
|
|
61
|
+
@return list: List of rows, where each row is a list of cell values
|
|
62
|
+
@details Reads CSV content and parses it into a structured format,
|
|
63
|
+
supporting line and column range selection
|
|
64
|
+
"""
|
|
65
|
+
# First read the file as text
|
|
66
|
+
text_content = super().read_content(file_path, start_line, end_line, start_column, end_column)
|
|
67
|
+
|
|
68
|
+
# Join the lines to create a CSV string
|
|
69
|
+
csv_text = ''.join(text_content)
|
|
70
|
+
|
|
71
|
+
# Parse the CSV
|
|
72
|
+
csv_data = []
|
|
73
|
+
csv_reader = csv.reader(
|
|
74
|
+
io.StringIO(csv_text),
|
|
75
|
+
delimiter=self.delimiter,
|
|
76
|
+
quotechar=self.quotechar
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
for row in csv_reader:
|
|
80
|
+
# Apply column range if specified
|
|
81
|
+
if start_column > 0 or end_column is not None:
|
|
82
|
+
col_start = start_column
|
|
83
|
+
col_end = end_column if end_column is not None else len(row)
|
|
84
|
+
row = row[col_start:col_end+1]
|
|
85
|
+
csv_data.append(row)
|
|
86
|
+
|
|
87
|
+
return csv_data
|
|
88
|
+
|
|
89
|
+
def compare_content(self, content1, content2):
|
|
90
|
+
"""
|
|
91
|
+
@brief Compare CSV content structurally
|
|
92
|
+
@param content1 list: First CSV data to compare (list of rows)
|
|
93
|
+
@param content2 list: Second CSV data to compare (list of rows)
|
|
94
|
+
@return tuple: (bool, list, bool) - (identical, differences, truncated)
|
|
95
|
+
@details Performs structural comparison of CSV data, including:
|
|
96
|
+
- Row count comparison
|
|
97
|
+
- Column count comparison per row
|
|
98
|
+
- Cell value comparison
|
|
99
|
+
- Limits the number of reported differences
|
|
100
|
+
- When error_analysis=True, accumulates streaming stats over ALL numeric cells
|
|
101
|
+
"""
|
|
102
|
+
self._error_stats = None # Reset per comparison
|
|
103
|
+
|
|
104
|
+
if content1 == content2:
|
|
105
|
+
return True, [], False
|
|
106
|
+
|
|
107
|
+
differences = []
|
|
108
|
+
max_diffs = 10
|
|
109
|
+
|
|
110
|
+
# ── Error analysis accumulators ──
|
|
111
|
+
total_numeric_cells = 0
|
|
112
|
+
mismatched_cells = 0
|
|
113
|
+
sum_abs_error = 0.0
|
|
114
|
+
sum_sq_abs_error = 0.0
|
|
115
|
+
max_abs_error = None
|
|
116
|
+
max_abs_error_at = None
|
|
117
|
+
max_rel_error = None
|
|
118
|
+
max_rel_error_at = None
|
|
119
|
+
|
|
120
|
+
# Check row count
|
|
121
|
+
if len(content1) != len(content2):
|
|
122
|
+
differences.append(Difference(
|
|
123
|
+
position="row count",
|
|
124
|
+
expected=f"{len(content1)} rows",
|
|
125
|
+
actual=f"{len(content2)} rows",
|
|
126
|
+
diff_type="row_count_mismatch"
|
|
127
|
+
))
|
|
128
|
+
|
|
129
|
+
# Compare rows
|
|
130
|
+
for i, (row1, row2) in enumerate(zip(content1, content2)):
|
|
131
|
+
# Check column count in this row
|
|
132
|
+
if len(row1) != len(row2):
|
|
133
|
+
if len(differences) < max_diffs:
|
|
134
|
+
differences.append(Difference(
|
|
135
|
+
position=f"row {i+1}",
|
|
136
|
+
expected=f"{len(row1)} columns",
|
|
137
|
+
actual=f"{len(row2)} columns",
|
|
138
|
+
diff_type="column_count_mismatch"
|
|
139
|
+
))
|
|
140
|
+
if not self.error_analysis and len(differences) >= max_diffs:
|
|
141
|
+
break
|
|
142
|
+
|
|
143
|
+
# Compare column values
|
|
144
|
+
for j, (cell1, cell2) in enumerate(zip(row1, row2)):
|
|
145
|
+
if cell1 != cell2:
|
|
146
|
+
# Try numeric tolerance comparison
|
|
147
|
+
try:
|
|
148
|
+
num1 = float(cell1)
|
|
149
|
+
num2 = float(cell2)
|
|
150
|
+
# Apply data filter
|
|
151
|
+
if self.filter_func:
|
|
152
|
+
if not (self.filter_func(num1) and self.filter_func(num2)):
|
|
153
|
+
continue
|
|
154
|
+
total_numeric_cells += 1
|
|
155
|
+
abs_err = abs(num2 - num1)
|
|
156
|
+
if math.isclose(num1, num2, rel_tol=self.rtol, abs_tol=self.atol):
|
|
157
|
+
continue # Within tolerance
|
|
158
|
+
|
|
159
|
+
# Mismatched numeric cell
|
|
160
|
+
mismatched_cells += 1
|
|
161
|
+
sum_abs_error += abs_err
|
|
162
|
+
sum_sq_abs_error += abs_err * abs_err
|
|
163
|
+
rel_err = abs_err / max(abs(num1), 1e-300) if abs(num1) > 0 else float("inf")
|
|
164
|
+
|
|
165
|
+
if max_abs_error is None or abs_err > max_abs_error:
|
|
166
|
+
max_abs_error = abs_err
|
|
167
|
+
max_abs_error_at = f"row {i+1}, column {j+1}"
|
|
168
|
+
if max_rel_error is None or rel_err > max_rel_error:
|
|
169
|
+
max_rel_error = rel_err
|
|
170
|
+
max_rel_error_at = f"row {i+1}, column {j+1}"
|
|
171
|
+
except (ValueError, TypeError):
|
|
172
|
+
pass # Non-numeric, fall through
|
|
173
|
+
|
|
174
|
+
if len(differences) < max_diffs:
|
|
175
|
+
differences.append(Difference(
|
|
176
|
+
position=f"row {i+1}, column {j+1}",
|
|
177
|
+
expected=cell1,
|
|
178
|
+
actual=cell2,
|
|
179
|
+
diff_type="cell_mismatch"
|
|
180
|
+
))
|
|
181
|
+
|
|
182
|
+
if not self.error_analysis and len(differences) >= max_diffs:
|
|
183
|
+
break
|
|
184
|
+
|
|
185
|
+
truncated = len(differences) >= max_diffs
|
|
186
|
+
|
|
187
|
+
# ── Store error stats ──
|
|
188
|
+
if self.error_analysis and total_numeric_cells > 0:
|
|
189
|
+
self._error_stats = {
|
|
190
|
+
"total_numeric_cells": total_numeric_cells,
|
|
191
|
+
"mismatched_cells": mismatched_cells,
|
|
192
|
+
"max_abs_error": max_abs_error,
|
|
193
|
+
"max_abs_error_at": max_abs_error_at,
|
|
194
|
+
"max_rel_error": max_rel_error,
|
|
195
|
+
"max_rel_error_at": max_rel_error_at,
|
|
196
|
+
"mean_abs_error": sum_abs_error / mismatched_cells if mismatched_cells > 0 else 0.0,
|
|
197
|
+
"rms_abs_error": math.sqrt(sum_sq_abs_error / mismatched_cells) if mismatched_cells > 0 else 0.0,
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
if not differences:
|
|
201
|
+
return True, [], False
|
|
202
|
+
return False, differences, truncated
|
|
203
|
+
|
|
204
|
+
def _parse_filter(self):
|
|
205
|
+
"""Parse data filter string and return a scalar filter function"""
|
|
206
|
+
if not self.data_filter:
|
|
207
|
+
return None
|
|
208
|
+
self.logger.debug(f"Parsing data filter: {self.data_filter}")
|
|
209
|
+
try:
|
|
210
|
+
match = re.match(
|
|
211
|
+
r"^(abs)?([><]=?|==)([-+]?\d*\.?\d+(?:[eE][-+]?\d+)?)$",
|
|
212
|
+
self.data_filter.replace(" ", ""),
|
|
213
|
+
)
|
|
214
|
+
if not match:
|
|
215
|
+
self.logger.warning(
|
|
216
|
+
f"Invalid data filter format: {self.data_filter}. Ignoring filter."
|
|
217
|
+
)
|
|
218
|
+
return None
|
|
219
|
+
use_abs, op, value_str = match.groups()
|
|
220
|
+
value = float(value_str)
|
|
221
|
+
op_map = {
|
|
222
|
+
">": lambda x: x > value,
|
|
223
|
+
">=": lambda x: x >= value,
|
|
224
|
+
"<": lambda x: x < value,
|
|
225
|
+
"<=": lambda x: x <= value,
|
|
226
|
+
"==": lambda x: x == value,
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
def filter_func(x):
|
|
230
|
+
target = abs(x) if use_abs else x
|
|
231
|
+
return op_map[op](target)
|
|
232
|
+
|
|
233
|
+
self.logger.debug(
|
|
234
|
+
f"Created filter function for pattern: {use_abs or ''}{op}{value}"
|
|
235
|
+
)
|
|
236
|
+
return filter_func
|
|
237
|
+
except Exception as e:
|
|
238
|
+
self.logger.error(
|
|
239
|
+
f"Failed to parse data filter '{self.data_filter}': {e}. Ignoring filter."
|
|
240
|
+
)
|
|
241
|
+
return None
|