symtest-cli 1.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. symtest/__init__.py +45 -0
  2. symtest/cli.py +549 -0
  3. symtest/commands/__init__.py +9 -0
  4. symtest/commands/compare.py +221 -0
  5. symtest/config/__init__.py +7 -0
  6. symtest/config/config_io.py +346 -0
  7. symtest/config/config_schema.py +330 -0
  8. symtest/config/import_expander.py +149 -0
  9. symtest/config/inheritance_expander.py +197 -0
  10. symtest/core/__init__.py +15 -0
  11. symtest/core/assertions.py +253 -0
  12. symtest/core/base_runner.py +299 -0
  13. symtest/core/config_loader.py +536 -0
  14. symtest/core/execution.py +498 -0
  15. symtest/core/history_store.py +96 -0
  16. symtest/core/last_run_store.py +109 -0
  17. symtest/core/parallel_runner.py +251 -0
  18. symtest/core/process_worker.py +93 -0
  19. symtest/core/sequence_state.py +143 -0
  20. symtest/core/setup.py +137 -0
  21. symtest/core/test_case.py +76 -0
  22. symtest/core/types.py +92 -0
  23. symtest/file_comparator/__init__.py +10 -0
  24. symtest/file_comparator/base_comparator.py +109 -0
  25. symtest/file_comparator/binary_comparator.py +399 -0
  26. symtest/file_comparator/csv_comparator.py +241 -0
  27. symtest/file_comparator/factory.py +191 -0
  28. symtest/file_comparator/h5_comparator.py +777 -0
  29. symtest/file_comparator/json_comparator.py +323 -0
  30. symtest/file_comparator/result.py +213 -0
  31. symtest/file_comparator/script_comparator.py +182 -0
  32. symtest/file_comparator/text_comparator.py +182 -0
  33. symtest/file_comparator/xml_comparator.py +150 -0
  34. symtest/logging_config.py +66 -0
  35. symtest/runners/__init__.py +15 -0
  36. symtest/runners/config_runner.py +96 -0
  37. symtest/runners/json_runner.py +21 -0
  38. symtest/runners/parallel_config_runner.py +278 -0
  39. symtest/runners/parallel_json_runner.py +26 -0
  40. symtest/runners/parallel_yaml_runner.py +31 -0
  41. symtest/runners/yaml_runner.py +26 -0
  42. symtest/tui/__init__.py +11 -0
  43. symtest/tui/app.py +90 -0
  44. symtest/tui/controllers/__init__.py +0 -0
  45. symtest/tui/controllers/case_controller.py +322 -0
  46. symtest/tui/screens/__init__.py +0 -0
  47. symtest/tui/screens/case_editor.py +244 -0
  48. symtest/tui/screens/case_list.py +255 -0
  49. symtest/tui/widgets/__init__.py +0 -0
  50. symtest/tui/widgets/case_table.py +113 -0
  51. symtest/tui/widgets/expected_editor.py +159 -0
  52. symtest/tui/widgets/search_bar.py +160 -0
  53. symtest/tui/widgets/steps_editor.py +243 -0
  54. symtest/utils/__init__.py +21 -0
  55. symtest/utils/junit_xml_writer.py +137 -0
  56. symtest/utils/path_resolver.py +124 -0
  57. symtest/utils/report_generator.py +208 -0
  58. symtest_cli-1.3.0.dist-info/METADATA +316 -0
  59. symtest_cli-1.3.0.dist-info/RECORD +63 -0
  60. symtest_cli-1.3.0.dist-info/WHEEL +5 -0
  61. symtest_cli-1.3.0.dist-info/entry_points.txt +4 -0
  62. symtest_cli-1.3.0.dist-info/licenses/LICENSE +21 -0
  63. symtest_cli-1.3.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,498 @@
1
+ import subprocess
2
+ import signal
3
+ import time
4
+ import os
5
+ import shlex
6
+ import logging
7
+ from typing import Any, List, Optional, Dict
8
+
9
+ from .assertions import Assertions, ValidationError
10
+ from .types import ExpectedResult, TestCaseData, TestResultData
11
+
12
+ logger = logging.getLogger("symtest.core.execution")
13
+
14
+ # Default maximum chars for command output in reports.
15
+ # Full output is still written to disk when output_dir is set.
16
+ DEFAULT_OUTPUT_MAX_CHARS = 20000
17
+
18
+ # Maximum number of differences retained per compare_failures entry in JSON output.
19
+ # The text report already limits display to 5; this prevents large CSV/H5 diffs
20
+ # from blowing up AI context windows.
21
+ DEFAULT_MAX_DIFFERENCES = 50
22
+
23
+ # Commands that are shell builtins (not real executables).
24
+ # With shell=False, these must be wrapped via the platform shell.
25
+ if os.name == 'nt':
26
+ _SHELL_BUILTINS = frozenset(['echo', 'dir', 'type', 'copy', 'del', 'ren',
27
+ 'cd', 'md', 'rd', 'set', 'cls', 'move'])
28
+ else:
29
+ _SHELL_BUILTINS = frozenset(['echo', 'cd', 'pwd', 'export', 'source'])
30
+
31
+
32
+ def _normalize_cmd_list(command: str, args: List[str]) -> List[str]:
33
+ """If command is a shell builtin, wrap with the platform shell interpreter."""
34
+ if command.lower() in _SHELL_BUILTINS:
35
+ if os.name == 'nt':
36
+ return ['cmd', '/d', '/c', command, *args]
37
+ else:
38
+ return ['/bin/sh', '-c', shlex.join([command, *args])]
39
+ return [command, *args]
40
+
41
+
42
+ def _trim_output(output: str, max_chars: int = DEFAULT_OUTPUT_MAX_CHARS) -> str:
43
+ """Trim long output: keep head 1/3 + tail 2/3 of max_chars."""
44
+ if len(output) <= max_chars:
45
+ return output
46
+ head_size = max_chars // 3
47
+ tail_size = max_chars - head_size
48
+ trimmed = len(output) - max_chars
49
+ return (
50
+ output[:head_size]
51
+ + f"\n\n[... {trimmed} chars truncated ...]\n\n"
52
+ + output[-tail_size:]
53
+ )
54
+
55
+
56
+ def _trim_compare_failures(
57
+ compare_failures: List[Dict[str, Any]],
58
+ max_diffs: int = DEFAULT_MAX_DIFFERENCES,
59
+ ) -> List[Dict[str, Any]]:
60
+ """Truncate the ``differences`` list inside each compare_failure entry.
61
+
62
+ The text report already only shows the first 5 differences, but the JSON
63
+ output previously carried the full list — which could reach megabytes for
64
+ large CSV/H5 comparisons. This function caps the retained differences so
65
+ that AI consumers get a representative sample without context-window blowup.
66
+ """
67
+ if not compare_failures:
68
+ return compare_failures
69
+ trimmed: List[Dict[str, Any]] = []
70
+ for cf in compare_failures:
71
+ diffs = cf.get("differences", [])
72
+ if len(diffs) > max_diffs:
73
+ cf_copy = dict(cf)
74
+ cf_copy["differences"] = diffs[:max_diffs]
75
+ cf_copy["differences_truncated"] = True
76
+ cf_copy["differences_total"] = len(diffs)
77
+ trimmed.append(cf_copy)
78
+ else:
79
+ trimmed.append(cf)
80
+ return trimmed
81
+
82
+
83
+ def _build_next_action_hint(
84
+ failure_kind: Optional[str],
85
+ *,
86
+ update_baseline: bool = False,
87
+ ) -> Optional[Dict[str, Any]]:
88
+ """Build a structured remediation hint for a failed test case.
89
+
90
+ The ``command`` field is left as ``None`` here because the execution layer
91
+ does not know the config file path; runners fill it in (see
92
+ ``BaseRunner._fill_hint_command``).
93
+
94
+ ``action`` vocabulary (stable, for AI consumers to branch on):
95
+ - ``update_baseline`` – file comparison failed; accept new output as baseline
96
+ - ``update_expected`` – an expectation assertion failed; fix program or config
97
+ - ``increase_timeout`` – the command timed out
98
+ - ``investigate`` – execution errors and everything else
99
+ """
100
+ if not failure_kind:
101
+ return None
102
+ if failure_kind == "file_compare":
103
+ if update_baseline:
104
+ return {
105
+ "action": "investigate",
106
+ "command": None,
107
+ "reason": (
108
+ "File comparison failed even though --update-baseline was "
109
+ "enabled; the actual output file may be missing or "
110
+ "unreadable. Inspect compare_failures for details."
111
+ ),
112
+ }
113
+ return {
114
+ "action": "update_baseline",
115
+ "command": None,
116
+ "reason": (
117
+ "File comparison failed. If the new output is the intended "
118
+ "behavior, re-run with --update-baseline to accept it as the "
119
+ "new baseline; otherwise inspect compare_failures/diff_summary "
120
+ "and fix the program under test."
121
+ ),
122
+ }
123
+ if failure_kind in ("return_code", "output_contains", "output_matches"):
124
+ return {
125
+ "action": "update_expected",
126
+ "command": None,
127
+ "reason": (
128
+ f"Assertion '{failure_kind}' failed. If the new behavior is "
129
+ "intended, update the 'expected' block in the test config; "
130
+ "otherwise fix the program under test. Compare 'expected' "
131
+ "against 'stdout'/'stderr' and 'assertion_results' in this "
132
+ "result to locate the discrepancy."
133
+ ),
134
+ }
135
+ if failure_kind == "timeout":
136
+ return {
137
+ "action": "increase_timeout",
138
+ "command": None,
139
+ "reason": (
140
+ "The command timed out. Increase 'timeout' in the test "
141
+ "config or investigate why the program did not finish."
142
+ ),
143
+ }
144
+ if failure_kind == "execution_error":
145
+ return {
146
+ "action": "investigate",
147
+ "command": None,
148
+ "reason": (
149
+ "The command failed to execute. Check that it exists, that "
150
+ "arguments are valid, and that the environment is set up "
151
+ "correctly."
152
+ ),
153
+ }
154
+ return {
155
+ "action": "investigate",
156
+ "command": None,
157
+ "reason": (
158
+ "Investigate the failure using 'message', 'stdout'/'stderr' and "
159
+ "'expected' in this result."
160
+ ),
161
+ }
162
+
163
+
164
+ def validate_result(
165
+ expected: ExpectedResult,
166
+ actual: TestResultData,
167
+ workspace: Optional[str] = None,
168
+ *,
169
+ update_baseline: bool = False,
170
+ error_analysis: bool = False,
171
+ ) -> List[Dict[str, Any]]:
172
+ """
173
+ Pure validation logic. Collects all assertion failures and raises
174
+ ``ValidationError`` with structured data when any fail.
175
+
176
+ :param expected: Expected result specification from the test case.
177
+ :param actual: Actual test result data produced by command execution.
178
+ :param workspace: Working directory; used to resolve relative file paths in
179
+ ``compare_files`` assertions.
180
+ :param update_baseline: If True, overwrite baseline files on comparison failure.
181
+ :returns: Per-assertion pass/fail detail (``assertion_results``) when all
182
+ assertions pass. On failure the same data is carried by the
183
+ raised ``ValidationError.assertion_results``.
184
+ """
185
+ assertions = Assertions()
186
+ failure_messages: List[str] = []
187
+ failure_kind: Optional[str] = None
188
+ compare_failures: List[Dict[str, Any]] = []
189
+ baseline_updated: List[str] = []
190
+ assertion_results: List[Dict[str, Any]] = []
191
+
192
+ # ── return_code ──
193
+ if "return_code" in expected:
194
+ exc_msg = None
195
+ try:
196
+ assertions.return_code_equals(actual["return_code"], expected["return_code"])
197
+ except AssertionError as e:
198
+ exc_msg = str(e)
199
+ if exc_msg:
200
+ failure_messages.append(exc_msg)
201
+ if failure_kind is None:
202
+ failure_kind = "return_code"
203
+ assertion_results.append({"assertion": "return_code", "passed": False, "message": exc_msg})
204
+ else:
205
+ assertion_results.append({"assertion": "return_code", "passed": True})
206
+
207
+ # ── output_contains ──
208
+ if "output_contains" in expected:
209
+ for text in expected["output_contains"]:
210
+ exc_msg = None
211
+ try:
212
+ assertions.contains(actual["output"], text)
213
+ except AssertionError as e:
214
+ exc_msg = str(e)
215
+ if exc_msg:
216
+ failure_messages.append(exc_msg)
217
+ if failure_kind is None:
218
+ failure_kind = "output_contains"
219
+ assertion_results.append({"assertion": "output_contains", "passed": False, "message": exc_msg, "text": text})
220
+ else:
221
+ assertion_results.append({"assertion": "output_contains", "passed": True, "text": text})
222
+
223
+ # ── output_matches ──
224
+ if "output_matches" in expected and expected["output_matches"]:
225
+ exc_msg = None
226
+ try:
227
+ assertions.matches(actual["output"], expected["output_matches"])
228
+ except AssertionError as e:
229
+ exc_msg = str(e)
230
+ if exc_msg:
231
+ failure_messages.append(exc_msg)
232
+ if failure_kind is None:
233
+ failure_kind = "output_matches"
234
+ assertion_results.append({"assertion": "output_matches", "passed": False, "message": exc_msg})
235
+ else:
236
+ assertion_results.append({"assertion": "output_matches", "passed": True})
237
+
238
+ # ── compare_files ──
239
+ if "compare_files" in expected:
240
+ for spec in expected["compare_files"]:
241
+ cf_result = _dispatch_file_compare(spec, workspace, assertions, update_baseline=update_baseline, error_analysis=error_analysis)
242
+ assertion_results.append(cf_result)
243
+ if cf_result.get("passed") is False:
244
+ if failure_kind is None:
245
+ failure_kind = "file_compare"
246
+ compare_failures.extend(cf_result.get("compare_failures", []))
247
+ failure_messages.append(cf_result.get("message", ""))
248
+ if cf_result.get("baseline_updated"):
249
+ baseline_updated.extend(cf_result.get("baseline_updated", []))
250
+
251
+ if failure_messages:
252
+ raise ValidationError(
253
+ message="; ".join(failure_messages),
254
+ failure_kind=failure_kind or "unknown",
255
+ compare_failures=compare_failures,
256
+ baseline_updated=baseline_updated,
257
+ assertion_results=assertion_results,
258
+ )
259
+
260
+ return assertion_results
261
+
262
+
263
+ def _dispatch_file_compare(
264
+ spec: Dict[str, Any],
265
+ workspace: Optional[str],
266
+ assertions: Assertions,
267
+ *,
268
+ update_baseline: bool = False,
269
+ error_analysis: bool = False,
270
+ ) -> Dict[str, Any]:
271
+ """Extract fields from a compare_files spec dict and delegate to Assertions.compare_files.
272
+
273
+ Returns a structured dict with ``passed``, ``compare_failures``, ``baseline_updated``, ``message``.
274
+ """
275
+ actual_path = spec.get("actual", "")
276
+ baseline_path = spec.get("baseline", "")
277
+ file_type = spec.get("type", None)
278
+
279
+ # All remaining keys are forwarded as comparator kwargs
280
+ known_keys = {"actual", "baseline", "type"}
281
+ comparator_kwargs = {k: v for k, v in spec.items() if k not in known_keys}
282
+
283
+ try:
284
+ cf_result = assertions.compare_files(
285
+ actual_path=actual_path,
286
+ baseline_path=baseline_path,
287
+ file_type=file_type,
288
+ workspace=workspace,
289
+ update_baseline=update_baseline,
290
+ error_analysis=error_analysis,
291
+ **comparator_kwargs,
292
+ )
293
+ if cf_result.get("baseline_updated"):
294
+ return {
295
+ "assertion": "compare_files",
296
+ "passed": True,
297
+ "compare_failures": [],
298
+ "baseline_updated": [baseline_path],
299
+ "message": "",
300
+ }
301
+ return {
302
+ "assertion": "compare_files",
303
+ "passed": True,
304
+ "compare_failures": [],
305
+ "baseline_updated": [],
306
+ "message": "",
307
+ }
308
+ except ValidationError as e:
309
+ return {
310
+ "assertion": "compare_files",
311
+ "passed": False,
312
+ "compare_failures": e.compare_failures,
313
+ "baseline_updated": e.baseline_updated,
314
+ "message": str(e),
315
+ }
316
+
317
+
318
+ def _execute_command_once(
319
+ case: TestCaseData,
320
+ workspace: Optional[str] = None,
321
+ env: Optional[Dict[str, str]] = None,
322
+ *,
323
+ update_baseline: bool = False,
324
+ error_analysis: bool = False,
325
+ output_max_chars: int = DEFAULT_OUTPUT_MAX_CHARS,
326
+ ) -> TestResultData:
327
+ """Execute a single command once (no retry logic)."""
328
+ start_time = time.time()
329
+ cmd_list = _normalize_cmd_list(case["command"], [str(arg) for arg in case["args"]])
330
+ timeout_limit = case.get("timeout", 3600)
331
+
332
+ full_command = " ".join(cmd_list)
333
+
334
+ result: TestResultData = {
335
+ "name": case["name"],
336
+ "status": "failed",
337
+ "message": "",
338
+ "command": full_command,
339
+ "output": "",
340
+ "stdout": "",
341
+ "stderr": "",
342
+ "return_code": None,
343
+ "duration": 0.0,
344
+ "expected": None,
345
+ "description": None,
346
+ "tags": [],
347
+ "failure_kind": None,
348
+ "attempts": 1,
349
+ "flaky": False,
350
+ "attempt_history": [],
351
+ "step_results": [],
352
+ "compare_failures": [],
353
+ "baseline_updated": [],
354
+ "failed_step": None,
355
+ "assertion_results": [],
356
+ "next_action_hint": None,
357
+ }
358
+
359
+ # Prepare environment variables
360
+ # Default to current environment, merge with provided env if any
361
+ current_env = os.environ.copy()
362
+ if env:
363
+ current_env.update(env)
364
+
365
+ try:
366
+ process = subprocess.Popen(
367
+ cmd_list,
368
+ cwd=workspace if workspace else None,
369
+ stdout=subprocess.PIPE,
370
+ stderr=subprocess.PIPE,
371
+ text=True,
372
+ errors="replace",
373
+ start_new_session=True,
374
+ env=current_env,
375
+ )
376
+
377
+ try:
378
+ stdout, stderr = process.communicate(timeout=timeout_limit)
379
+ except subprocess.TimeoutExpired:
380
+ # Kill the entire process group to avoid orphan processes.
381
+ # Never killpg() PID 0 or 1 — they belong to init/system.
382
+ try:
383
+ if os.name == 'posix' and process.pid and process.pid > 1:
384
+ os.killpg(process.pid, signal.SIGKILL)
385
+ else:
386
+ process.kill()
387
+ except (ProcessLookupError, PermissionError, OSError):
388
+ pass # process already exited or cannot be killed
389
+ stdout, stderr = process.communicate() # reap the process
390
+ raw_output = (stdout or "") + (stderr or "")
391
+ result["status"] = "timeout"
392
+ result["failure_kind"] = "timeout"
393
+ result["next_action_hint"] = _build_next_action_hint("timeout")
394
+ result["message"] = f"Timeout reached! Killed after {timeout_limit} seconds."
395
+ result["output"] = _trim_output(raw_output, output_max_chars)
396
+ result["stdout"] = _trim_output(stdout or "", output_max_chars)
397
+ result["stderr"] = _trim_output(stderr or "", output_max_chars)
398
+ result["return_code"] = None
399
+ else:
400
+ raw_output = stdout + stderr
401
+ result["output"] = _trim_output(raw_output, output_max_chars)
402
+ result["stdout"] = _trim_output(stdout, output_max_chars)
403
+ result["stderr"] = _trim_output(stderr, output_max_chars)
404
+ result["return_code"] = process.returncode
405
+
406
+ result["assertion_results"] = validate_result(
407
+ case["expected"], result, workspace,
408
+ update_baseline=update_baseline,
409
+ error_analysis=error_analysis,
410
+ )
411
+ result["status"] = "passed"
412
+ except ValidationError as exc:
413
+ result["message"] = str(exc)
414
+ result["failure_kind"] = exc.failure_kind
415
+ result["compare_failures"] = _trim_compare_failures(exc.compare_failures)
416
+ result["baseline_updated"] = exc.baseline_updated
417
+ result["assertion_results"] = exc.assertion_results
418
+ result["next_action_hint"] = _build_next_action_hint(
419
+ exc.failure_kind, update_baseline=update_baseline,
420
+ )
421
+ except AssertionError as exc:
422
+ # Legacy AssertionError catch for backward compatibility
423
+ result["message"] = str(exc)
424
+ result["failure_kind"] = result["failure_kind"] or "unknown"
425
+ result["next_action_hint"] = _build_next_action_hint(result["failure_kind"])
426
+ except Exception as exc:
427
+ result["message"] = f"Execution error: {str(exc)}"
428
+ result["failure_kind"] = "execution_error"
429
+ result["next_action_hint"] = _build_next_action_hint("execution_error")
430
+ finally:
431
+ result["duration"] = time.time() - start_time
432
+
433
+ return result
434
+
435
+
436
+ def execute_single_test_case(
437
+ case: TestCaseData,
438
+ workspace: Optional[str] = None,
439
+ env: Optional[Dict[str, str]] = None,
440
+ *,
441
+ update_baseline: bool = False,
442
+ error_analysis: bool = False,
443
+ output_max_chars: int = DEFAULT_OUTPUT_MAX_CHARS,
444
+ ) -> TestResultData:
445
+ """
446
+ Stateless execution of a single test case with optional retry.
447
+
448
+ Args:
449
+ case: Test case data (may include ``retry_count`` for automatic retries).
450
+ workspace: Working directory for test execution.
451
+ env: Optional environment variables to inject/override (merged with os.environ).
452
+ update_baseline: If True, overwrite baseline files on comparison failure.
453
+ output_max_chars: Max characters for output in result dict.
454
+
455
+ ``retry_count`` (defaults to 0) controls how many additional times the
456
+ command is re-run after the first failure. ``retry_count=0`` means no
457
+ retry (behaviour identical to previous versions).
458
+ """
459
+ retry_count: int = case.get("retry_count", 0)
460
+ max_attempts = retry_count + 1
461
+ total_duration = 0.0
462
+ last_result: Optional[TestResultData] = None
463
+ attempt_history: List[Dict[str, Any]] = []
464
+
465
+ for attempt in range(1, max_attempts + 1):
466
+ result = _execute_command_once(
467
+ case, workspace, env,
468
+ update_baseline=update_baseline,
469
+ error_analysis=error_analysis,
470
+ output_max_chars=output_max_chars,
471
+ )
472
+ total_duration += result["duration"]
473
+ attempt_history.append({
474
+ "attempt": attempt,
475
+ "status": result["status"],
476
+ "message": result.get("message", ""),
477
+ "duration": result["duration"],
478
+ })
479
+ last_result = result
480
+
481
+ if result["status"] == "passed":
482
+ break
483
+
484
+ if attempt < max_attempts:
485
+ logger.info(
486
+ "Retrying '%s' (attempt %d/%d)...",
487
+ case["name"], attempt + 1, max_attempts,
488
+ )
489
+
490
+ if last_result is not None:
491
+ last_result["duration"] = total_duration
492
+ last_result["attempts"] = len(attempt_history)
493
+ last_result["attempt_history"] = attempt_history
494
+ if retry_count > 0 and last_result["status"] == "passed" and len(attempt_history) > 1:
495
+ last_result["flaky"] = True
496
+
497
+ return last_result if last_result is not None else result
498
+
@@ -0,0 +1,96 @@
1
+ # -*- coding: utf-8 -*-
2
+
3
+ """
4
+ .symtest Runtime History Store
5
+
6
+ Manages persistent storage of per-case runtime history for smart scheduling
7
+ and regression detection.
8
+ """
9
+
10
+ import json
11
+ import os
12
+ from typing import Dict, Optional
13
+
14
+ SYMTEST_FILENAME = ".symtest"
15
+
16
+
17
+ def _empty_history() -> dict:
18
+ """Return an empty history structure."""
19
+ return {"version": 1, "cases": {}}
20
+
21
+
22
+ def ensure_symtest(history_dir: str) -> None:
23
+ """Create .symtest in the directory if it doesn't exist."""
24
+ os.makedirs(history_dir, exist_ok=True)
25
+ path = os.path.join(history_dir, SYMTEST_FILENAME)
26
+ if not os.path.exists(path):
27
+ with open(path, "w", encoding="utf-8") as f:
28
+ json.dump(_empty_history(), f, indent=2)
29
+
30
+
31
+ def load_history(history_dir: str) -> dict:
32
+ """Load .symtest from the directory. Auto-init if missing."""
33
+ path = os.path.join(history_dir, SYMTEST_FILENAME)
34
+ if not os.path.exists(path):
35
+ ensure_symtest(history_dir)
36
+ with open(path, "r", encoding="utf-8") as f:
37
+ data = json.load(f)
38
+ if "cases" not in data:
39
+ data["cases"] = {}
40
+ return data
41
+
42
+
43
+ def save_history(history_dir: str, history: dict) -> None:
44
+ """Write history back to .symtest."""
45
+ os.makedirs(history_dir, exist_ok=True)
46
+ path = os.path.join(history_dir, SYMTEST_FILENAME)
47
+ with open(path, "w", encoding="utf-8") as f:
48
+ json.dump(history, f, indent=2, ensure_ascii=False)
49
+
50
+
51
+ def update_case(history: dict, name: str, duration: float) -> None:
52
+ """Update a single case's record using cumulative average."""
53
+ cases = history.setdefault("cases", {})
54
+ if name in cases:
55
+ rec = cases[name]
56
+ old_avg = rec["avg_duration"]
57
+ count = rec["run_count"]
58
+ rec["avg_duration"] = (old_avg * count + duration) / (count + 1)
59
+ rec["last_duration"] = duration
60
+ rec["run_count"] = count + 1
61
+ else:
62
+ cases[name] = {
63
+ "avg_duration": duration,
64
+ "last_duration": duration,
65
+ "run_count": 1,
66
+ }
67
+
68
+
69
+ def reset_cases(history: dict, case_names: set) -> int:
70
+ """Remove given case names from history. Returns count of removed entries."""
71
+ cases = history.get("cases", {})
72
+ cleared = 0
73
+ for name in case_names:
74
+ if name in cases:
75
+ del cases[name]
76
+ cleared += 1
77
+ return cleared
78
+
79
+
80
+ def check_regression(
81
+ history: dict, name: str, duration: float, threshold: float = 1.5
82
+ ) -> Optional[str]:
83
+ """Return a warning message if duration exceeds avg * threshold, else None."""
84
+ cases = history.get("cases", {})
85
+ if name not in cases:
86
+ return None
87
+ avg = cases[name]["avg_duration"]
88
+ if avg <= 0:
89
+ return None
90
+ if duration > avg * threshold:
91
+ ratio = duration / avg
92
+ return (
93
+ f"⚠ WARNING: Case '{name}' regressed: "
94
+ f"{duration:.2f}s vs avg {avg:.2f}s ({ratio:.2f}x slower)"
95
+ )
96
+ return None