symtest-cli 1.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- symtest/__init__.py +45 -0
- symtest/cli.py +549 -0
- symtest/commands/__init__.py +9 -0
- symtest/commands/compare.py +221 -0
- symtest/config/__init__.py +7 -0
- symtest/config/config_io.py +346 -0
- symtest/config/config_schema.py +330 -0
- symtest/config/import_expander.py +149 -0
- symtest/config/inheritance_expander.py +197 -0
- symtest/core/__init__.py +15 -0
- symtest/core/assertions.py +253 -0
- symtest/core/base_runner.py +299 -0
- symtest/core/config_loader.py +536 -0
- symtest/core/execution.py +498 -0
- symtest/core/history_store.py +96 -0
- symtest/core/last_run_store.py +109 -0
- symtest/core/parallel_runner.py +251 -0
- symtest/core/process_worker.py +93 -0
- symtest/core/sequence_state.py +143 -0
- symtest/core/setup.py +137 -0
- symtest/core/test_case.py +76 -0
- symtest/core/types.py +92 -0
- symtest/file_comparator/__init__.py +10 -0
- symtest/file_comparator/base_comparator.py +109 -0
- symtest/file_comparator/binary_comparator.py +399 -0
- symtest/file_comparator/csv_comparator.py +241 -0
- symtest/file_comparator/factory.py +191 -0
- symtest/file_comparator/h5_comparator.py +777 -0
- symtest/file_comparator/json_comparator.py +323 -0
- symtest/file_comparator/result.py +213 -0
- symtest/file_comparator/script_comparator.py +182 -0
- symtest/file_comparator/text_comparator.py +182 -0
- symtest/file_comparator/xml_comparator.py +150 -0
- symtest/logging_config.py +66 -0
- symtest/runners/__init__.py +15 -0
- symtest/runners/config_runner.py +96 -0
- symtest/runners/json_runner.py +21 -0
- symtest/runners/parallel_config_runner.py +278 -0
- symtest/runners/parallel_json_runner.py +26 -0
- symtest/runners/parallel_yaml_runner.py +31 -0
- symtest/runners/yaml_runner.py +26 -0
- symtest/tui/__init__.py +11 -0
- symtest/tui/app.py +90 -0
- symtest/tui/controllers/__init__.py +0 -0
- symtest/tui/controllers/case_controller.py +322 -0
- symtest/tui/screens/__init__.py +0 -0
- symtest/tui/screens/case_editor.py +244 -0
- symtest/tui/screens/case_list.py +255 -0
- symtest/tui/widgets/__init__.py +0 -0
- symtest/tui/widgets/case_table.py +113 -0
- symtest/tui/widgets/expected_editor.py +159 -0
- symtest/tui/widgets/search_bar.py +160 -0
- symtest/tui/widgets/steps_editor.py +243 -0
- symtest/utils/__init__.py +21 -0
- symtest/utils/junit_xml_writer.py +137 -0
- symtest/utils/path_resolver.py +124 -0
- symtest/utils/report_generator.py +208 -0
- symtest_cli-1.3.0.dist-info/METADATA +316 -0
- symtest_cli-1.3.0.dist-info/RECORD +63 -0
- symtest_cli-1.3.0.dist-info/WHEEL +5 -0
- symtest_cli-1.3.0.dist-info/entry_points.txt +4 -0
- symtest_cli-1.3.0.dist-info/licenses/LICENSE +21 -0
- symtest_cli-1.3.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,498 @@
|
|
|
1
|
+
import subprocess
|
|
2
|
+
import signal
|
|
3
|
+
import time
|
|
4
|
+
import os
|
|
5
|
+
import shlex
|
|
6
|
+
import logging
|
|
7
|
+
from typing import Any, List, Optional, Dict
|
|
8
|
+
|
|
9
|
+
from .assertions import Assertions, ValidationError
|
|
10
|
+
from .types import ExpectedResult, TestCaseData, TestResultData
|
|
11
|
+
|
|
12
|
+
logger = logging.getLogger("symtest.core.execution")
|
|
13
|
+
|
|
14
|
+
# Default maximum chars for command output in reports.
|
|
15
|
+
# Full output is still written to disk when output_dir is set.
|
|
16
|
+
DEFAULT_OUTPUT_MAX_CHARS = 20000
|
|
17
|
+
|
|
18
|
+
# Maximum number of differences retained per compare_failures entry in JSON output.
|
|
19
|
+
# The text report already limits display to 5; this prevents large CSV/H5 diffs
|
|
20
|
+
# from blowing up AI context windows.
|
|
21
|
+
DEFAULT_MAX_DIFFERENCES = 50
|
|
22
|
+
|
|
23
|
+
# Commands that are shell builtins (not real executables).
|
|
24
|
+
# With shell=False, these must be wrapped via the platform shell.
|
|
25
|
+
if os.name == 'nt':
|
|
26
|
+
_SHELL_BUILTINS = frozenset(['echo', 'dir', 'type', 'copy', 'del', 'ren',
|
|
27
|
+
'cd', 'md', 'rd', 'set', 'cls', 'move'])
|
|
28
|
+
else:
|
|
29
|
+
_SHELL_BUILTINS = frozenset(['echo', 'cd', 'pwd', 'export', 'source'])
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _normalize_cmd_list(command: str, args: List[str]) -> List[str]:
|
|
33
|
+
"""If command is a shell builtin, wrap with the platform shell interpreter."""
|
|
34
|
+
if command.lower() in _SHELL_BUILTINS:
|
|
35
|
+
if os.name == 'nt':
|
|
36
|
+
return ['cmd', '/d', '/c', command, *args]
|
|
37
|
+
else:
|
|
38
|
+
return ['/bin/sh', '-c', shlex.join([command, *args])]
|
|
39
|
+
return [command, *args]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _trim_output(output: str, max_chars: int = DEFAULT_OUTPUT_MAX_CHARS) -> str:
|
|
43
|
+
"""Trim long output: keep head 1/3 + tail 2/3 of max_chars."""
|
|
44
|
+
if len(output) <= max_chars:
|
|
45
|
+
return output
|
|
46
|
+
head_size = max_chars // 3
|
|
47
|
+
tail_size = max_chars - head_size
|
|
48
|
+
trimmed = len(output) - max_chars
|
|
49
|
+
return (
|
|
50
|
+
output[:head_size]
|
|
51
|
+
+ f"\n\n[... {trimmed} chars truncated ...]\n\n"
|
|
52
|
+
+ output[-tail_size:]
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _trim_compare_failures(
|
|
57
|
+
compare_failures: List[Dict[str, Any]],
|
|
58
|
+
max_diffs: int = DEFAULT_MAX_DIFFERENCES,
|
|
59
|
+
) -> List[Dict[str, Any]]:
|
|
60
|
+
"""Truncate the ``differences`` list inside each compare_failure entry.
|
|
61
|
+
|
|
62
|
+
The text report already only shows the first 5 differences, but the JSON
|
|
63
|
+
output previously carried the full list — which could reach megabytes for
|
|
64
|
+
large CSV/H5 comparisons. This function caps the retained differences so
|
|
65
|
+
that AI consumers get a representative sample without context-window blowup.
|
|
66
|
+
"""
|
|
67
|
+
if not compare_failures:
|
|
68
|
+
return compare_failures
|
|
69
|
+
trimmed: List[Dict[str, Any]] = []
|
|
70
|
+
for cf in compare_failures:
|
|
71
|
+
diffs = cf.get("differences", [])
|
|
72
|
+
if len(diffs) > max_diffs:
|
|
73
|
+
cf_copy = dict(cf)
|
|
74
|
+
cf_copy["differences"] = diffs[:max_diffs]
|
|
75
|
+
cf_copy["differences_truncated"] = True
|
|
76
|
+
cf_copy["differences_total"] = len(diffs)
|
|
77
|
+
trimmed.append(cf_copy)
|
|
78
|
+
else:
|
|
79
|
+
trimmed.append(cf)
|
|
80
|
+
return trimmed
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _build_next_action_hint(
|
|
84
|
+
failure_kind: Optional[str],
|
|
85
|
+
*,
|
|
86
|
+
update_baseline: bool = False,
|
|
87
|
+
) -> Optional[Dict[str, Any]]:
|
|
88
|
+
"""Build a structured remediation hint for a failed test case.
|
|
89
|
+
|
|
90
|
+
The ``command`` field is left as ``None`` here because the execution layer
|
|
91
|
+
does not know the config file path; runners fill it in (see
|
|
92
|
+
``BaseRunner._fill_hint_command``).
|
|
93
|
+
|
|
94
|
+
``action`` vocabulary (stable, for AI consumers to branch on):
|
|
95
|
+
- ``update_baseline`` – file comparison failed; accept new output as baseline
|
|
96
|
+
- ``update_expected`` – an expectation assertion failed; fix program or config
|
|
97
|
+
- ``increase_timeout`` – the command timed out
|
|
98
|
+
- ``investigate`` – execution errors and everything else
|
|
99
|
+
"""
|
|
100
|
+
if not failure_kind:
|
|
101
|
+
return None
|
|
102
|
+
if failure_kind == "file_compare":
|
|
103
|
+
if update_baseline:
|
|
104
|
+
return {
|
|
105
|
+
"action": "investigate",
|
|
106
|
+
"command": None,
|
|
107
|
+
"reason": (
|
|
108
|
+
"File comparison failed even though --update-baseline was "
|
|
109
|
+
"enabled; the actual output file may be missing or "
|
|
110
|
+
"unreadable. Inspect compare_failures for details."
|
|
111
|
+
),
|
|
112
|
+
}
|
|
113
|
+
return {
|
|
114
|
+
"action": "update_baseline",
|
|
115
|
+
"command": None,
|
|
116
|
+
"reason": (
|
|
117
|
+
"File comparison failed. If the new output is the intended "
|
|
118
|
+
"behavior, re-run with --update-baseline to accept it as the "
|
|
119
|
+
"new baseline; otherwise inspect compare_failures/diff_summary "
|
|
120
|
+
"and fix the program under test."
|
|
121
|
+
),
|
|
122
|
+
}
|
|
123
|
+
if failure_kind in ("return_code", "output_contains", "output_matches"):
|
|
124
|
+
return {
|
|
125
|
+
"action": "update_expected",
|
|
126
|
+
"command": None,
|
|
127
|
+
"reason": (
|
|
128
|
+
f"Assertion '{failure_kind}' failed. If the new behavior is "
|
|
129
|
+
"intended, update the 'expected' block in the test config; "
|
|
130
|
+
"otherwise fix the program under test. Compare 'expected' "
|
|
131
|
+
"against 'stdout'/'stderr' and 'assertion_results' in this "
|
|
132
|
+
"result to locate the discrepancy."
|
|
133
|
+
),
|
|
134
|
+
}
|
|
135
|
+
if failure_kind == "timeout":
|
|
136
|
+
return {
|
|
137
|
+
"action": "increase_timeout",
|
|
138
|
+
"command": None,
|
|
139
|
+
"reason": (
|
|
140
|
+
"The command timed out. Increase 'timeout' in the test "
|
|
141
|
+
"config or investigate why the program did not finish."
|
|
142
|
+
),
|
|
143
|
+
}
|
|
144
|
+
if failure_kind == "execution_error":
|
|
145
|
+
return {
|
|
146
|
+
"action": "investigate",
|
|
147
|
+
"command": None,
|
|
148
|
+
"reason": (
|
|
149
|
+
"The command failed to execute. Check that it exists, that "
|
|
150
|
+
"arguments are valid, and that the environment is set up "
|
|
151
|
+
"correctly."
|
|
152
|
+
),
|
|
153
|
+
}
|
|
154
|
+
return {
|
|
155
|
+
"action": "investigate",
|
|
156
|
+
"command": None,
|
|
157
|
+
"reason": (
|
|
158
|
+
"Investigate the failure using 'message', 'stdout'/'stderr' and "
|
|
159
|
+
"'expected' in this result."
|
|
160
|
+
),
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def validate_result(
|
|
165
|
+
expected: ExpectedResult,
|
|
166
|
+
actual: TestResultData,
|
|
167
|
+
workspace: Optional[str] = None,
|
|
168
|
+
*,
|
|
169
|
+
update_baseline: bool = False,
|
|
170
|
+
error_analysis: bool = False,
|
|
171
|
+
) -> List[Dict[str, Any]]:
|
|
172
|
+
"""
|
|
173
|
+
Pure validation logic. Collects all assertion failures and raises
|
|
174
|
+
``ValidationError`` with structured data when any fail.
|
|
175
|
+
|
|
176
|
+
:param expected: Expected result specification from the test case.
|
|
177
|
+
:param actual: Actual test result data produced by command execution.
|
|
178
|
+
:param workspace: Working directory; used to resolve relative file paths in
|
|
179
|
+
``compare_files`` assertions.
|
|
180
|
+
:param update_baseline: If True, overwrite baseline files on comparison failure.
|
|
181
|
+
:returns: Per-assertion pass/fail detail (``assertion_results``) when all
|
|
182
|
+
assertions pass. On failure the same data is carried by the
|
|
183
|
+
raised ``ValidationError.assertion_results``.
|
|
184
|
+
"""
|
|
185
|
+
assertions = Assertions()
|
|
186
|
+
failure_messages: List[str] = []
|
|
187
|
+
failure_kind: Optional[str] = None
|
|
188
|
+
compare_failures: List[Dict[str, Any]] = []
|
|
189
|
+
baseline_updated: List[str] = []
|
|
190
|
+
assertion_results: List[Dict[str, Any]] = []
|
|
191
|
+
|
|
192
|
+
# ── return_code ──
|
|
193
|
+
if "return_code" in expected:
|
|
194
|
+
exc_msg = None
|
|
195
|
+
try:
|
|
196
|
+
assertions.return_code_equals(actual["return_code"], expected["return_code"])
|
|
197
|
+
except AssertionError as e:
|
|
198
|
+
exc_msg = str(e)
|
|
199
|
+
if exc_msg:
|
|
200
|
+
failure_messages.append(exc_msg)
|
|
201
|
+
if failure_kind is None:
|
|
202
|
+
failure_kind = "return_code"
|
|
203
|
+
assertion_results.append({"assertion": "return_code", "passed": False, "message": exc_msg})
|
|
204
|
+
else:
|
|
205
|
+
assertion_results.append({"assertion": "return_code", "passed": True})
|
|
206
|
+
|
|
207
|
+
# ── output_contains ──
|
|
208
|
+
if "output_contains" in expected:
|
|
209
|
+
for text in expected["output_contains"]:
|
|
210
|
+
exc_msg = None
|
|
211
|
+
try:
|
|
212
|
+
assertions.contains(actual["output"], text)
|
|
213
|
+
except AssertionError as e:
|
|
214
|
+
exc_msg = str(e)
|
|
215
|
+
if exc_msg:
|
|
216
|
+
failure_messages.append(exc_msg)
|
|
217
|
+
if failure_kind is None:
|
|
218
|
+
failure_kind = "output_contains"
|
|
219
|
+
assertion_results.append({"assertion": "output_contains", "passed": False, "message": exc_msg, "text": text})
|
|
220
|
+
else:
|
|
221
|
+
assertion_results.append({"assertion": "output_contains", "passed": True, "text": text})
|
|
222
|
+
|
|
223
|
+
# ── output_matches ──
|
|
224
|
+
if "output_matches" in expected and expected["output_matches"]:
|
|
225
|
+
exc_msg = None
|
|
226
|
+
try:
|
|
227
|
+
assertions.matches(actual["output"], expected["output_matches"])
|
|
228
|
+
except AssertionError as e:
|
|
229
|
+
exc_msg = str(e)
|
|
230
|
+
if exc_msg:
|
|
231
|
+
failure_messages.append(exc_msg)
|
|
232
|
+
if failure_kind is None:
|
|
233
|
+
failure_kind = "output_matches"
|
|
234
|
+
assertion_results.append({"assertion": "output_matches", "passed": False, "message": exc_msg})
|
|
235
|
+
else:
|
|
236
|
+
assertion_results.append({"assertion": "output_matches", "passed": True})
|
|
237
|
+
|
|
238
|
+
# ── compare_files ──
|
|
239
|
+
if "compare_files" in expected:
|
|
240
|
+
for spec in expected["compare_files"]:
|
|
241
|
+
cf_result = _dispatch_file_compare(spec, workspace, assertions, update_baseline=update_baseline, error_analysis=error_analysis)
|
|
242
|
+
assertion_results.append(cf_result)
|
|
243
|
+
if cf_result.get("passed") is False:
|
|
244
|
+
if failure_kind is None:
|
|
245
|
+
failure_kind = "file_compare"
|
|
246
|
+
compare_failures.extend(cf_result.get("compare_failures", []))
|
|
247
|
+
failure_messages.append(cf_result.get("message", ""))
|
|
248
|
+
if cf_result.get("baseline_updated"):
|
|
249
|
+
baseline_updated.extend(cf_result.get("baseline_updated", []))
|
|
250
|
+
|
|
251
|
+
if failure_messages:
|
|
252
|
+
raise ValidationError(
|
|
253
|
+
message="; ".join(failure_messages),
|
|
254
|
+
failure_kind=failure_kind or "unknown",
|
|
255
|
+
compare_failures=compare_failures,
|
|
256
|
+
baseline_updated=baseline_updated,
|
|
257
|
+
assertion_results=assertion_results,
|
|
258
|
+
)
|
|
259
|
+
|
|
260
|
+
return assertion_results
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _dispatch_file_compare(
|
|
264
|
+
spec: Dict[str, Any],
|
|
265
|
+
workspace: Optional[str],
|
|
266
|
+
assertions: Assertions,
|
|
267
|
+
*,
|
|
268
|
+
update_baseline: bool = False,
|
|
269
|
+
error_analysis: bool = False,
|
|
270
|
+
) -> Dict[str, Any]:
|
|
271
|
+
"""Extract fields from a compare_files spec dict and delegate to Assertions.compare_files.
|
|
272
|
+
|
|
273
|
+
Returns a structured dict with ``passed``, ``compare_failures``, ``baseline_updated``, ``message``.
|
|
274
|
+
"""
|
|
275
|
+
actual_path = spec.get("actual", "")
|
|
276
|
+
baseline_path = spec.get("baseline", "")
|
|
277
|
+
file_type = spec.get("type", None)
|
|
278
|
+
|
|
279
|
+
# All remaining keys are forwarded as comparator kwargs
|
|
280
|
+
known_keys = {"actual", "baseline", "type"}
|
|
281
|
+
comparator_kwargs = {k: v for k, v in spec.items() if k not in known_keys}
|
|
282
|
+
|
|
283
|
+
try:
|
|
284
|
+
cf_result = assertions.compare_files(
|
|
285
|
+
actual_path=actual_path,
|
|
286
|
+
baseline_path=baseline_path,
|
|
287
|
+
file_type=file_type,
|
|
288
|
+
workspace=workspace,
|
|
289
|
+
update_baseline=update_baseline,
|
|
290
|
+
error_analysis=error_analysis,
|
|
291
|
+
**comparator_kwargs,
|
|
292
|
+
)
|
|
293
|
+
if cf_result.get("baseline_updated"):
|
|
294
|
+
return {
|
|
295
|
+
"assertion": "compare_files",
|
|
296
|
+
"passed": True,
|
|
297
|
+
"compare_failures": [],
|
|
298
|
+
"baseline_updated": [baseline_path],
|
|
299
|
+
"message": "",
|
|
300
|
+
}
|
|
301
|
+
return {
|
|
302
|
+
"assertion": "compare_files",
|
|
303
|
+
"passed": True,
|
|
304
|
+
"compare_failures": [],
|
|
305
|
+
"baseline_updated": [],
|
|
306
|
+
"message": "",
|
|
307
|
+
}
|
|
308
|
+
except ValidationError as e:
|
|
309
|
+
return {
|
|
310
|
+
"assertion": "compare_files",
|
|
311
|
+
"passed": False,
|
|
312
|
+
"compare_failures": e.compare_failures,
|
|
313
|
+
"baseline_updated": e.baseline_updated,
|
|
314
|
+
"message": str(e),
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _execute_command_once(
|
|
319
|
+
case: TestCaseData,
|
|
320
|
+
workspace: Optional[str] = None,
|
|
321
|
+
env: Optional[Dict[str, str]] = None,
|
|
322
|
+
*,
|
|
323
|
+
update_baseline: bool = False,
|
|
324
|
+
error_analysis: bool = False,
|
|
325
|
+
output_max_chars: int = DEFAULT_OUTPUT_MAX_CHARS,
|
|
326
|
+
) -> TestResultData:
|
|
327
|
+
"""Execute a single command once (no retry logic)."""
|
|
328
|
+
start_time = time.time()
|
|
329
|
+
cmd_list = _normalize_cmd_list(case["command"], [str(arg) for arg in case["args"]])
|
|
330
|
+
timeout_limit = case.get("timeout", 3600)
|
|
331
|
+
|
|
332
|
+
full_command = " ".join(cmd_list)
|
|
333
|
+
|
|
334
|
+
result: TestResultData = {
|
|
335
|
+
"name": case["name"],
|
|
336
|
+
"status": "failed",
|
|
337
|
+
"message": "",
|
|
338
|
+
"command": full_command,
|
|
339
|
+
"output": "",
|
|
340
|
+
"stdout": "",
|
|
341
|
+
"stderr": "",
|
|
342
|
+
"return_code": None,
|
|
343
|
+
"duration": 0.0,
|
|
344
|
+
"expected": None,
|
|
345
|
+
"description": None,
|
|
346
|
+
"tags": [],
|
|
347
|
+
"failure_kind": None,
|
|
348
|
+
"attempts": 1,
|
|
349
|
+
"flaky": False,
|
|
350
|
+
"attempt_history": [],
|
|
351
|
+
"step_results": [],
|
|
352
|
+
"compare_failures": [],
|
|
353
|
+
"baseline_updated": [],
|
|
354
|
+
"failed_step": None,
|
|
355
|
+
"assertion_results": [],
|
|
356
|
+
"next_action_hint": None,
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
# Prepare environment variables
|
|
360
|
+
# Default to current environment, merge with provided env if any
|
|
361
|
+
current_env = os.environ.copy()
|
|
362
|
+
if env:
|
|
363
|
+
current_env.update(env)
|
|
364
|
+
|
|
365
|
+
try:
|
|
366
|
+
process = subprocess.Popen(
|
|
367
|
+
cmd_list,
|
|
368
|
+
cwd=workspace if workspace else None,
|
|
369
|
+
stdout=subprocess.PIPE,
|
|
370
|
+
stderr=subprocess.PIPE,
|
|
371
|
+
text=True,
|
|
372
|
+
errors="replace",
|
|
373
|
+
start_new_session=True,
|
|
374
|
+
env=current_env,
|
|
375
|
+
)
|
|
376
|
+
|
|
377
|
+
try:
|
|
378
|
+
stdout, stderr = process.communicate(timeout=timeout_limit)
|
|
379
|
+
except subprocess.TimeoutExpired:
|
|
380
|
+
# Kill the entire process group to avoid orphan processes.
|
|
381
|
+
# Never killpg() PID 0 or 1 — they belong to init/system.
|
|
382
|
+
try:
|
|
383
|
+
if os.name == 'posix' and process.pid and process.pid > 1:
|
|
384
|
+
os.killpg(process.pid, signal.SIGKILL)
|
|
385
|
+
else:
|
|
386
|
+
process.kill()
|
|
387
|
+
except (ProcessLookupError, PermissionError, OSError):
|
|
388
|
+
pass # process already exited or cannot be killed
|
|
389
|
+
stdout, stderr = process.communicate() # reap the process
|
|
390
|
+
raw_output = (stdout or "") + (stderr or "")
|
|
391
|
+
result["status"] = "timeout"
|
|
392
|
+
result["failure_kind"] = "timeout"
|
|
393
|
+
result["next_action_hint"] = _build_next_action_hint("timeout")
|
|
394
|
+
result["message"] = f"Timeout reached! Killed after {timeout_limit} seconds."
|
|
395
|
+
result["output"] = _trim_output(raw_output, output_max_chars)
|
|
396
|
+
result["stdout"] = _trim_output(stdout or "", output_max_chars)
|
|
397
|
+
result["stderr"] = _trim_output(stderr or "", output_max_chars)
|
|
398
|
+
result["return_code"] = None
|
|
399
|
+
else:
|
|
400
|
+
raw_output = stdout + stderr
|
|
401
|
+
result["output"] = _trim_output(raw_output, output_max_chars)
|
|
402
|
+
result["stdout"] = _trim_output(stdout, output_max_chars)
|
|
403
|
+
result["stderr"] = _trim_output(stderr, output_max_chars)
|
|
404
|
+
result["return_code"] = process.returncode
|
|
405
|
+
|
|
406
|
+
result["assertion_results"] = validate_result(
|
|
407
|
+
case["expected"], result, workspace,
|
|
408
|
+
update_baseline=update_baseline,
|
|
409
|
+
error_analysis=error_analysis,
|
|
410
|
+
)
|
|
411
|
+
result["status"] = "passed"
|
|
412
|
+
except ValidationError as exc:
|
|
413
|
+
result["message"] = str(exc)
|
|
414
|
+
result["failure_kind"] = exc.failure_kind
|
|
415
|
+
result["compare_failures"] = _trim_compare_failures(exc.compare_failures)
|
|
416
|
+
result["baseline_updated"] = exc.baseline_updated
|
|
417
|
+
result["assertion_results"] = exc.assertion_results
|
|
418
|
+
result["next_action_hint"] = _build_next_action_hint(
|
|
419
|
+
exc.failure_kind, update_baseline=update_baseline,
|
|
420
|
+
)
|
|
421
|
+
except AssertionError as exc:
|
|
422
|
+
# Legacy AssertionError catch for backward compatibility
|
|
423
|
+
result["message"] = str(exc)
|
|
424
|
+
result["failure_kind"] = result["failure_kind"] or "unknown"
|
|
425
|
+
result["next_action_hint"] = _build_next_action_hint(result["failure_kind"])
|
|
426
|
+
except Exception as exc:
|
|
427
|
+
result["message"] = f"Execution error: {str(exc)}"
|
|
428
|
+
result["failure_kind"] = "execution_error"
|
|
429
|
+
result["next_action_hint"] = _build_next_action_hint("execution_error")
|
|
430
|
+
finally:
|
|
431
|
+
result["duration"] = time.time() - start_time
|
|
432
|
+
|
|
433
|
+
return result
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
def execute_single_test_case(
|
|
437
|
+
case: TestCaseData,
|
|
438
|
+
workspace: Optional[str] = None,
|
|
439
|
+
env: Optional[Dict[str, str]] = None,
|
|
440
|
+
*,
|
|
441
|
+
update_baseline: bool = False,
|
|
442
|
+
error_analysis: bool = False,
|
|
443
|
+
output_max_chars: int = DEFAULT_OUTPUT_MAX_CHARS,
|
|
444
|
+
) -> TestResultData:
|
|
445
|
+
"""
|
|
446
|
+
Stateless execution of a single test case with optional retry.
|
|
447
|
+
|
|
448
|
+
Args:
|
|
449
|
+
case: Test case data (may include ``retry_count`` for automatic retries).
|
|
450
|
+
workspace: Working directory for test execution.
|
|
451
|
+
env: Optional environment variables to inject/override (merged with os.environ).
|
|
452
|
+
update_baseline: If True, overwrite baseline files on comparison failure.
|
|
453
|
+
output_max_chars: Max characters for output in result dict.
|
|
454
|
+
|
|
455
|
+
``retry_count`` (defaults to 0) controls how many additional times the
|
|
456
|
+
command is re-run after the first failure. ``retry_count=0`` means no
|
|
457
|
+
retry (behaviour identical to previous versions).
|
|
458
|
+
"""
|
|
459
|
+
retry_count: int = case.get("retry_count", 0)
|
|
460
|
+
max_attempts = retry_count + 1
|
|
461
|
+
total_duration = 0.0
|
|
462
|
+
last_result: Optional[TestResultData] = None
|
|
463
|
+
attempt_history: List[Dict[str, Any]] = []
|
|
464
|
+
|
|
465
|
+
for attempt in range(1, max_attempts + 1):
|
|
466
|
+
result = _execute_command_once(
|
|
467
|
+
case, workspace, env,
|
|
468
|
+
update_baseline=update_baseline,
|
|
469
|
+
error_analysis=error_analysis,
|
|
470
|
+
output_max_chars=output_max_chars,
|
|
471
|
+
)
|
|
472
|
+
total_duration += result["duration"]
|
|
473
|
+
attempt_history.append({
|
|
474
|
+
"attempt": attempt,
|
|
475
|
+
"status": result["status"],
|
|
476
|
+
"message": result.get("message", ""),
|
|
477
|
+
"duration": result["duration"],
|
|
478
|
+
})
|
|
479
|
+
last_result = result
|
|
480
|
+
|
|
481
|
+
if result["status"] == "passed":
|
|
482
|
+
break
|
|
483
|
+
|
|
484
|
+
if attempt < max_attempts:
|
|
485
|
+
logger.info(
|
|
486
|
+
"Retrying '%s' (attempt %d/%d)...",
|
|
487
|
+
case["name"], attempt + 1, max_attempts,
|
|
488
|
+
)
|
|
489
|
+
|
|
490
|
+
if last_result is not None:
|
|
491
|
+
last_result["duration"] = total_duration
|
|
492
|
+
last_result["attempts"] = len(attempt_history)
|
|
493
|
+
last_result["attempt_history"] = attempt_history
|
|
494
|
+
if retry_count > 0 and last_result["status"] == "passed" and len(attempt_history) > 1:
|
|
495
|
+
last_result["flaky"] = True
|
|
496
|
+
|
|
497
|
+
return last_result if last_result is not None else result
|
|
498
|
+
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
|
|
3
|
+
"""
|
|
4
|
+
.symtest Runtime History Store
|
|
5
|
+
|
|
6
|
+
Manages persistent storage of per-case runtime history for smart scheduling
|
|
7
|
+
and regression detection.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import os
|
|
12
|
+
from typing import Dict, Optional
|
|
13
|
+
|
|
14
|
+
SYMTEST_FILENAME = ".symtest"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _empty_history() -> dict:
|
|
18
|
+
"""Return an empty history structure."""
|
|
19
|
+
return {"version": 1, "cases": {}}
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def ensure_symtest(history_dir: str) -> None:
|
|
23
|
+
"""Create .symtest in the directory if it doesn't exist."""
|
|
24
|
+
os.makedirs(history_dir, exist_ok=True)
|
|
25
|
+
path = os.path.join(history_dir, SYMTEST_FILENAME)
|
|
26
|
+
if not os.path.exists(path):
|
|
27
|
+
with open(path, "w", encoding="utf-8") as f:
|
|
28
|
+
json.dump(_empty_history(), f, indent=2)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def load_history(history_dir: str) -> dict:
|
|
32
|
+
"""Load .symtest from the directory. Auto-init if missing."""
|
|
33
|
+
path = os.path.join(history_dir, SYMTEST_FILENAME)
|
|
34
|
+
if not os.path.exists(path):
|
|
35
|
+
ensure_symtest(history_dir)
|
|
36
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
37
|
+
data = json.load(f)
|
|
38
|
+
if "cases" not in data:
|
|
39
|
+
data["cases"] = {}
|
|
40
|
+
return data
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def save_history(history_dir: str, history: dict) -> None:
|
|
44
|
+
"""Write history back to .symtest."""
|
|
45
|
+
os.makedirs(history_dir, exist_ok=True)
|
|
46
|
+
path = os.path.join(history_dir, SYMTEST_FILENAME)
|
|
47
|
+
with open(path, "w", encoding="utf-8") as f:
|
|
48
|
+
json.dump(history, f, indent=2, ensure_ascii=False)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def update_case(history: dict, name: str, duration: float) -> None:
|
|
52
|
+
"""Update a single case's record using cumulative average."""
|
|
53
|
+
cases = history.setdefault("cases", {})
|
|
54
|
+
if name in cases:
|
|
55
|
+
rec = cases[name]
|
|
56
|
+
old_avg = rec["avg_duration"]
|
|
57
|
+
count = rec["run_count"]
|
|
58
|
+
rec["avg_duration"] = (old_avg * count + duration) / (count + 1)
|
|
59
|
+
rec["last_duration"] = duration
|
|
60
|
+
rec["run_count"] = count + 1
|
|
61
|
+
else:
|
|
62
|
+
cases[name] = {
|
|
63
|
+
"avg_duration": duration,
|
|
64
|
+
"last_duration": duration,
|
|
65
|
+
"run_count": 1,
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def reset_cases(history: dict, case_names: set) -> int:
|
|
70
|
+
"""Remove given case names from history. Returns count of removed entries."""
|
|
71
|
+
cases = history.get("cases", {})
|
|
72
|
+
cleared = 0
|
|
73
|
+
for name in case_names:
|
|
74
|
+
if name in cases:
|
|
75
|
+
del cases[name]
|
|
76
|
+
cleared += 1
|
|
77
|
+
return cleared
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def check_regression(
|
|
81
|
+
history: dict, name: str, duration: float, threshold: float = 1.5
|
|
82
|
+
) -> Optional[str]:
|
|
83
|
+
"""Return a warning message if duration exceeds avg * threshold, else None."""
|
|
84
|
+
cases = history.get("cases", {})
|
|
85
|
+
if name not in cases:
|
|
86
|
+
return None
|
|
87
|
+
avg = cases[name]["avg_duration"]
|
|
88
|
+
if avg <= 0:
|
|
89
|
+
return None
|
|
90
|
+
if duration > avg * threshold:
|
|
91
|
+
ratio = duration / avg
|
|
92
|
+
return (
|
|
93
|
+
f"⚠ WARNING: Case '{name}' regressed: "
|
|
94
|
+
f"{duration:.2f}s vs avg {avg:.2f}s ({ratio:.2f}x slower)"
|
|
95
|
+
)
|
|
96
|
+
return None
|