symtest-cli 1.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. symtest/__init__.py +45 -0
  2. symtest/cli.py +549 -0
  3. symtest/commands/__init__.py +9 -0
  4. symtest/commands/compare.py +221 -0
  5. symtest/config/__init__.py +7 -0
  6. symtest/config/config_io.py +346 -0
  7. symtest/config/config_schema.py +330 -0
  8. symtest/config/import_expander.py +149 -0
  9. symtest/config/inheritance_expander.py +197 -0
  10. symtest/core/__init__.py +15 -0
  11. symtest/core/assertions.py +253 -0
  12. symtest/core/base_runner.py +299 -0
  13. symtest/core/config_loader.py +536 -0
  14. symtest/core/execution.py +498 -0
  15. symtest/core/history_store.py +96 -0
  16. symtest/core/last_run_store.py +109 -0
  17. symtest/core/parallel_runner.py +251 -0
  18. symtest/core/process_worker.py +93 -0
  19. symtest/core/sequence_state.py +143 -0
  20. symtest/core/setup.py +137 -0
  21. symtest/core/test_case.py +76 -0
  22. symtest/core/types.py +92 -0
  23. symtest/file_comparator/__init__.py +10 -0
  24. symtest/file_comparator/base_comparator.py +109 -0
  25. symtest/file_comparator/binary_comparator.py +399 -0
  26. symtest/file_comparator/csv_comparator.py +241 -0
  27. symtest/file_comparator/factory.py +191 -0
  28. symtest/file_comparator/h5_comparator.py +777 -0
  29. symtest/file_comparator/json_comparator.py +323 -0
  30. symtest/file_comparator/result.py +213 -0
  31. symtest/file_comparator/script_comparator.py +182 -0
  32. symtest/file_comparator/text_comparator.py +182 -0
  33. symtest/file_comparator/xml_comparator.py +150 -0
  34. symtest/logging_config.py +66 -0
  35. symtest/runners/__init__.py +15 -0
  36. symtest/runners/config_runner.py +96 -0
  37. symtest/runners/json_runner.py +21 -0
  38. symtest/runners/parallel_config_runner.py +278 -0
  39. symtest/runners/parallel_json_runner.py +26 -0
  40. symtest/runners/parallel_yaml_runner.py +31 -0
  41. symtest/runners/yaml_runner.py +26 -0
  42. symtest/tui/__init__.py +11 -0
  43. symtest/tui/app.py +90 -0
  44. symtest/tui/controllers/__init__.py +0 -0
  45. symtest/tui/controllers/case_controller.py +322 -0
  46. symtest/tui/screens/__init__.py +0 -0
  47. symtest/tui/screens/case_editor.py +244 -0
  48. symtest/tui/screens/case_list.py +255 -0
  49. symtest/tui/widgets/__init__.py +0 -0
  50. symtest/tui/widgets/case_table.py +113 -0
  51. symtest/tui/widgets/expected_editor.py +159 -0
  52. symtest/tui/widgets/search_bar.py +160 -0
  53. symtest/tui/widgets/steps_editor.py +243 -0
  54. symtest/utils/__init__.py +21 -0
  55. symtest/utils/junit_xml_writer.py +137 -0
  56. symtest/utils/path_resolver.py +124 -0
  57. symtest/utils/report_generator.py +208 -0
  58. symtest_cli-1.3.0.dist-info/METADATA +316 -0
  59. symtest_cli-1.3.0.dist-info/RECORD +63 -0
  60. symtest_cli-1.3.0.dist-info/WHEEL +5 -0
  61. symtest_cli-1.3.0.dist-info/entry_points.txt +4 -0
  62. symtest_cli-1.3.0.dist-info/licenses/LICENSE +21 -0
  63. symtest_cli-1.3.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,109 @@
1
+ # -*- coding: utf-8 -*-
2
+
3
+ """
4
+ Last-run state store for ``--last-failed`` support.
5
+
6
+ Stores per-case status in ``<workspace>/.cli-test/last_run.json``.
7
+ Each run **overwrites** the status of every case that was executed,
8
+ so "previously failed but now fixed" cases are immediately removed
9
+ from the failed set. Cases that were not executed in this run
10
+ retain their previous status.
11
+ """
12
+
13
+ import json
14
+ import os
15
+ from typing import Dict, List, Optional
16
+
17
+ LAST_RUN_DIR = ".cli-test"
18
+ LAST_RUN_FILENAME = "last_run.json"
19
+
20
+
21
+ def _last_run_path(workspace: str) -> str:
22
+ """Return the full path to ``last_run.json`` for a workspace."""
23
+ return os.path.join(workspace, LAST_RUN_DIR, LAST_RUN_FILENAME)
24
+
25
+
26
+ def load_last_run(workspace: str) -> Dict:
27
+ """Load last-run state. Returns empty dict when file doesn't exist."""
28
+ path = _last_run_path(workspace)
29
+ if not os.path.exists(path):
30
+ return {}
31
+ with open(path, "r", encoding="utf-8") as f:
32
+ try:
33
+ return json.load(f)
34
+ except (json.JSONDecodeError, ValueError):
35
+ return {}
36
+
37
+
38
+ def save_last_run(workspace: str, data: Dict) -> None:
39
+ """Write last-run state, creating the directory if needed."""
40
+ dir_path = os.path.join(workspace, LAST_RUN_DIR)
41
+ os.makedirs(dir_path, exist_ok=True)
42
+ path = _last_run_path(workspace)
43
+ with open(path, "w", encoding="utf-8") as f:
44
+ json.dump(data, f, indent=2, ensure_ascii=False)
45
+
46
+
47
+ def update_last_run(
48
+ workspace: str,
49
+ results: List[Dict],
50
+ ) -> None:
51
+ """Update last-run state with the results from the current execution.
52
+
53
+ For each result dict in *results*:
54
+ - The case's entry is **overwritten** with the current status.
55
+ - Cases in the existing file that were NOT part of this run
56
+ retain their previous status (e.g. when running a subset via ``-t``).
57
+
58
+ This ensures ``--last-failed`` never includes a case that passed
59
+ in the most recent run, since passing results overwrite any prior
60
+ failure record.
61
+ """
62
+ data = load_last_run(workspace)
63
+
64
+ for result in results:
65
+ name = result.get("name", "")
66
+ if not name:
67
+ continue
68
+ # Only store status + the case name
69
+ data[name] = {
70
+ "status": result.get("status", "unknown"),
71
+ }
72
+
73
+ save_last_run(workspace, data)
74
+
75
+
76
+ def get_last_failed_names(workspace: str) -> List[str]:
77
+ """Return the names of cases that failed in the last run.
78
+
79
+ Returns an empty list when no last-run state exists (first run).
80
+ """
81
+ data = load_last_run(workspace)
82
+ if not data:
83
+ return []
84
+ return sorted(
85
+ name for name, info in data.items()
86
+ if info.get("status") in ("failed", "timeout", "xpassed")
87
+ )
88
+
89
+
90
+ def get_last_run_summary(workspace: str) -> Dict[str, int]:
91
+ """Return a summary counts dict: total, passed, failed, xfailed, xpassed, timeout."""
92
+ data = load_last_run(workspace)
93
+ summary = {"total": 0, "passed": 0, "failed": 0, "xfailed": 0, "xpassed": 0, "timeout": 0}
94
+ for info in data.values():
95
+ st = info.get("status", "unknown")
96
+ summary["total"] += 1
97
+ if st == "passed":
98
+ summary["passed"] += 1
99
+ elif st == "xfailed":
100
+ summary["xfailed"] += 1
101
+ elif st == "xpassed":
102
+ summary["xpassed"] += 1
103
+ summary["failed"] += 1 # xpassed IS a suite failure
104
+ elif st == "failed":
105
+ summary["failed"] += 1
106
+ elif st == "timeout":
107
+ summary["timeout"] += 1
108
+ summary["failed"] += 1 # timeout counts in failure aggregate
109
+ return summary
@@ -0,0 +1,251 @@
1
+ from abc import ABC
2
+ from concurrent.futures import ThreadPoolExecutor, ProcessPoolExecutor, as_completed
3
+ from typing import List, Dict, Any, Optional, Union
4
+ import time
5
+ import threading
6
+ import logging
7
+ from .base_runner import BaseRunner
8
+ from .test_case import TestCase
9
+ from .process_worker import run_test_in_process
10
+
11
+ logger = logging.getLogger("symtest.core.parallel_runner")
12
+
13
+
14
+ class AtomicSemaphore:
15
+ """
16
+ 支持原子级多令牌获取的信号量,消除逐个 acquire 导致的部分占有死锁。
17
+
18
+ 与 threading.Semaphore 不同:acquire(n) 在所有 n 个令牌可用时一次性获取,
19
+ 否则等待(支持超时),不会出现"拿了3个等1个,另一个线程也拿了3个等1个"的死锁。
20
+
21
+ 唤醒策略:按请求令牌数降序优先唤醒,避免大核数任务被小任务持续抢占导致饥饿。
22
+ """
23
+
24
+ def __init__(self, value: int):
25
+ self._value = value
26
+ self._lock = threading.Lock()
27
+ self._waiters: list = [] # list of (required_n, threading.Event)
28
+
29
+ def _grant_tokens(self) -> None:
30
+ """Grant tokens to eligible waiters, largest request first (anti-starvation)."""
31
+ if not self._waiters:
32
+ return
33
+ # Sort by required_n descending so large-core requests get priority
34
+ self._waiters.sort(key=lambda x: -x[0])
35
+ granted: list = []
36
+ remaining: list = []
37
+ for n, event in self._waiters:
38
+ if self._value >= n:
39
+ self._value -= n
40
+ granted.append(event)
41
+ else:
42
+ remaining.append((n, event))
43
+ self._waiters = remaining
44
+ for event in granted:
45
+ event.set()
46
+
47
+ def acquire(self, n: int = 1, timeout: Optional[float] = None) -> bool:
48
+ """Atomically acquire n tokens. Returns True on success, False on timeout."""
49
+ event = threading.Event()
50
+ with self._lock:
51
+ # Fast path: enough tokens and no pending waiters
52
+ if self._value >= n and not self._waiters:
53
+ self._value -= n
54
+ return True
55
+ self._waiters.append((n, event))
56
+ self._grant_tokens()
57
+
58
+ if not event.wait(timeout=timeout):
59
+ # Timeout cleanup – guard against race where release granted
60
+ # tokens between event.wait() returning and lock acquisition
61
+ with self._lock:
62
+ if event.is_set():
63
+ return True
64
+ for i, (_, e) in enumerate(self._waiters):
65
+ if e is event:
66
+ del self._waiters[i]
67
+ self._grant_tokens()
68
+ break
69
+ return False
70
+ return True
71
+
72
+ def release(self, n: int = 1) -> None:
73
+ """Release n tokens, waking eligible waiters with anti-starvation priority."""
74
+ with self._lock:
75
+ self._value += n
76
+ self._grant_tokens()
77
+
78
+ class ParallelRunner(BaseRunner):
79
+ """并行测试运行器基类,支持多线程和多进程执行"""
80
+
81
+ def __init__(self, config_file: str, workspace: Optional[str] = None,
82
+ max_workers: Optional[int] = None,
83
+ execution_mode: str = "thread",
84
+ error_analysis: bool = False,
85
+ **kwargs):
86
+ """
87
+ 初始化并行运行器
88
+
89
+ Args:
90
+ config_file: 配置文件路径
91
+ workspace: 工作目录
92
+ max_workers: 最大并发数,默认为CPU核心数
93
+ execution_mode: 执行模式,'thread'(线程) 或 'process'(进程)
94
+ **kwargs: 透传给 BaseRunner 的额外参数
95
+ (test_case_filter, test_case_tag_filter, history_dir, regression_threshold)
96
+ """
97
+ super().__init__(config_file, workspace, **kwargs)
98
+ self.max_workers = max_workers
99
+ self.execution_mode = execution_mode
100
+ self.error_analysis = error_analysis
101
+ self.lock = threading.Lock() # 用于线程安全的结果更新
102
+
103
+ def run_tests(self) -> bool:
104
+ """并行运行所有测试用例"""
105
+ try:
106
+ self.load_test_cases()
107
+ self._apply_test_case_filter()
108
+ self.results["total"] = len(self.test_cases)
109
+
110
+ if self.results["total"] == 0:
111
+ logger.warning("No test cases to run.")
112
+ return False
113
+
114
+ # 执行setup任务
115
+ self.setup_manager.setup_all()
116
+
117
+ logger.info("Starting parallel test execution... Total tests: %d", self.results["total"])
118
+ logger.info("Execution mode: %s, Max workers: %s", self.execution_mode, self.max_workers or "auto")
119
+ logger.info("=" * 50)
120
+
121
+ start_time = time.time()
122
+
123
+ if self.execution_mode == "process":
124
+ executor_class = ProcessPoolExecutor
125
+ else:
126
+ executor_class = ThreadPoolExecutor
127
+
128
+ with executor_class(max_workers=self.max_workers) as executor:
129
+ # 提交所有测试任务
130
+ if self.execution_mode == "process":
131
+ # 进程模式:使用独立的工作器函数
132
+ future_to_case = {
133
+ executor.submit(
134
+ run_test_in_process,
135
+ i,
136
+ {
137
+ "name": case.name,
138
+ "command": case.command,
139
+ "args": case.args,
140
+ "expected": case.expected,
141
+ "timeout": case.timeout,
142
+ "resources": case.resources,
143
+ "retry_count": case.retry_count,
144
+ "steps": [
145
+ {
146
+ "command": s.command,
147
+ "args": s.args,
148
+ "expected": s.expected,
149
+ "timeout": s.timeout,
150
+ "retry_count": s.retry_count,
151
+ }
152
+ for s in case.steps
153
+ ] if case.steps else None,
154
+ },
155
+ str(self.workspace) if self.workspace else None,
156
+ update_baseline=self.update_baseline,
157
+ error_analysis=self.error_analysis,
158
+ resume=self.resume,
159
+ ): (i, case)
160
+ for i, case in enumerate(self.test_cases, 1)
161
+ }
162
+ else:
163
+ # 线程模式:使用实例方法
164
+ future_to_case = {
165
+ executor.submit(self._run_test_with_index, i, case): (i, case)
166
+ for i, case in enumerate(self.test_cases, 1)
167
+ }
168
+
169
+ # 收集结果
170
+ for future in as_completed(future_to_case):
171
+ test_index, case = future_to_case[future]
172
+ try:
173
+ result = future.result()
174
+ self._update_results(result, test_index, case)
175
+ except Exception as exc:
176
+ error_result = {
177
+ "name": case.name,
178
+ "status": "failed",
179
+ "message": f"Test execution failed: {str(exc)}",
180
+ "output": "",
181
+ "command": "",
182
+ "return_code": None
183
+ }
184
+ self._update_results(error_result, test_index, case)
185
+
186
+ end_time = time.time()
187
+ execution_time = end_time - start_time
188
+
189
+ logger.info("=" * 50)
190
+ logger.info(
191
+ "Parallel test execution completed in %.2f seconds", execution_time,
192
+ )
193
+ logger.info(
194
+ "Passed: %d, Failed: %d, XFailed: %d, XPassed: %d",
195
+ self.results["passed"], self.results["failed"],
196
+ self.results["xfailed"], self.results["xpassed"],
197
+ )
198
+
199
+ # Update history & regression detection
200
+ self._update_history()
201
+
202
+ # Save last-run state for --last-failed
203
+ self._save_last_run()
204
+
205
+ # Exit-code rule: failed + xpassed > 0 → non-zero
206
+ return self.results["failed"] == 0 and self.results["xpassed"] == 0
207
+ finally:
208
+ # 确保teardown总是被执行
209
+ self.setup_manager.teardown_all()
210
+
211
+ def _run_test_with_index(self, test_index: int, case: TestCase) -> Dict[str, Any]:
212
+ """运行单个测试并返回结果(包含索引信息)"""
213
+ logger.info("[Worker] Running test %d: %s", test_index, case.name)
214
+ result = self.run_single_test(case)
215
+ return result
216
+
217
+ def _update_results(self, result: Dict[str, Any], test_index: int, case: TestCase) -> None:
218
+ """线程安全地更新测试结果"""
219
+ with self.lock:
220
+ # Apply xfail status mapping before counting
221
+ self._apply_xfail_status(result, case)
222
+
223
+ self._fill_hint_command(result, case.name)
224
+ self.results["details"].append(result)
225
+ duration = result.get("duration", 0)
226
+ status = result["status"]
227
+ if status == "passed":
228
+ self.results["passed"] += 1
229
+ logger.info("✓ Test %d passed: %s (%.2fs)", test_index, case.name, duration)
230
+ elif status == "xfailed":
231
+ self.results["xfailed"] += 1
232
+ logger.info("✓ Test %d xfailed (expected): %s (%.2fs)", test_index, case.name, duration)
233
+ if result.get("message"):
234
+ logger.info(" Detail: %s", result["message"])
235
+ elif status == "xpassed":
236
+ self.results["xpassed"] += 1
237
+ self.results["failed"] += 1
238
+ logger.error("✗ Test %d xpassed (unexpected!): %s (%.2fs)", test_index, case.name, duration)
239
+ if result.get("message"):
240
+ logger.error(" Error: %s", result["message"])
241
+ logger.warning(" [XPass] Marked as expected_failure but passed — remove the xfail marker.")
242
+ else:
243
+ self.results["failed"] += 1
244
+ logger.error("✗ Test %d failed: %s (%.2fs)", test_index, case.name, duration)
245
+ if result["message"]:
246
+ logger.error(" Error: %s", result["message"])
247
+
248
+ def run_tests_sequential(self) -> bool:
249
+ """回退到顺序执行模式"""
250
+ logger.info("Falling back to sequential execution...")
251
+ return super().run_tests()
@@ -0,0 +1,93 @@
1
+ """
2
+ 进程工作器模块
3
+ 用于多进程并行测试执行,避免序列化问题
4
+ """
5
+
6
+ import logging
7
+ from typing import Dict, Any, List
8
+ from .config_loader import execute_sequence
9
+ from .execution import execute_single_test_case
10
+ from .types import TestCaseData
11
+
12
+ logger = logging.getLogger("symtest.core.process_worker")
13
+
14
+ def _run_sequence_in_process(
15
+ test_index: int,
16
+ case_data: Dict[str, Any],
17
+ workspace: str = None,
18
+ *,
19
+ update_baseline: bool = False,
20
+ error_analysis: bool = False,
21
+ resume: bool = False,
22
+ ) -> Dict[str, Any]:
23
+ """Run a sequence test case with multiple steps (fail-fast) in a process worker."""
24
+ return execute_sequence(
25
+ case_name=case_data["name"],
26
+ steps=case_data["steps"],
27
+ workspace=workspace,
28
+ print_prefix=f"[Process Worker {test_index}]",
29
+ executor=execute_single_test_case,
30
+ case_expected=case_data.get("expected"),
31
+ update_baseline=update_baseline,
32
+ error_analysis=error_analysis,
33
+ resume=resume,
34
+ )
35
+
36
+
37
+ def run_test_in_process(
38
+ test_index: int,
39
+ case_data: Dict[str, Any],
40
+ workspace: str = None,
41
+ *,
42
+ update_baseline: bool = False,
43
+ error_analysis: bool = False,
44
+ resume: bool = False,
45
+ ) -> Dict[str, Any]:
46
+ """
47
+ 在独立进程中运行单个测试用例
48
+
49
+ Args:
50
+ test_index: 测试索引
51
+ case_data: 测试用例数据字典
52
+ workspace: 工作目录
53
+ update_baseline: 是否更新基线文件
54
+ resume: 是否启用断点续跑
55
+
56
+ Returns:
57
+ 测试结果字典
58
+ """
59
+ # Sequence mode
60
+ if case_data.get("steps"):
61
+ return _run_sequence_in_process(
62
+ test_index, case_data, workspace,
63
+ update_baseline=update_baseline,
64
+ error_analysis=error_analysis,
65
+ resume=resume,
66
+ )
67
+
68
+ # Single command mode
69
+ case: TestCaseData = {
70
+ "name": case_data["name"],
71
+ "command": case_data["command"],
72
+ "args": case_data["args"],
73
+ "expected": case_data["expected"],
74
+ "description": case_data.get("description"),
75
+ "timeout": case_data.get("timeout"),
76
+ "resources": case_data.get("resources"),
77
+ "retry_count": case_data.get("retry_count", 0),
78
+ }
79
+
80
+ command_preview = f"{case['command']} {' '.join(case['args'])}".strip()
81
+ logger.info(" [Process Worker %d] Executing command: %s", test_index, command_preview)
82
+
83
+ result = execute_single_test_case(case, workspace, update_baseline=update_baseline, error_analysis=error_analysis)
84
+
85
+ if result["output"].strip():
86
+ logger.debug(" [Process Worker %d] Command output for %s:", test_index, case["name"])
87
+ for line in result["output"].splitlines():
88
+ logger.debug(" %s", line)
89
+
90
+ if result["status"] != "passed" and result.get("message"):
91
+ logger.error(" [Process Worker %d] Error for %s: %s", test_index, case["name"], result["message"])
92
+
93
+ return result
@@ -0,0 +1,143 @@
1
+ """
2
+ Sequence state store for step-level resume (``--resume``).
3
+
4
+ Stores intermediate step results so that when a sequence test case fails,
5
+ re-running with ``--resume`` skips already-passed steps and reconstructs
6
+ ``combined_output`` from cached step outputs.
7
+
8
+ State file: ``<workspace>/.cli-test/sequence_state/<case_name>.json``
9
+ Output cache: ``<workspace>/.cli-test/sequence_state/cache/<case_name>.step<N>.log``
10
+
11
+ **Trust model**: ``--resume`` trusts that workspace artifacts (input files,
12
+ pre-step outputs) have not been modified between runs. No artifact
13
+ validation is performed — the user asserts correctness by opting into resume.
14
+ """
15
+
16
+ import hashlib
17
+ import json
18
+ import logging
19
+ import os
20
+ from typing import Any, Dict, List, Optional
21
+
22
+ logger = logging.getLogger("symtest.core.sequence_state")
23
+
24
+ SEQUENCE_STATE_DIR = ".cli-test/sequence_state"
25
+ CACHE_SUBDIR = "cache"
26
+
27
+
28
+ def _state_dir_path(workspace: str) -> str:
29
+ return os.path.join(workspace, SEQUENCE_STATE_DIR)
30
+
31
+
32
+ def _cache_dir_path(workspace: str) -> str:
33
+ return os.path.join(_state_dir_path(workspace), CACHE_SUBDIR)
34
+
35
+
36
+ def _case_state_path(workspace: str, case_name: str) -> str:
37
+ return os.path.join(_state_dir_path(workspace), f"{case_name}.json")
38
+
39
+
40
+ def _step_output_path(workspace: str, case_name: str, step_idx: int) -> str:
41
+ dir_path = _cache_dir_path(workspace)
42
+ os.makedirs(dir_path, exist_ok=True)
43
+ return os.path.join(dir_path, f"{case_name}.step{step_idx}.log")
44
+
45
+
46
+ def compute_config_hash(
47
+ steps: List[Any],
48
+ case_expected: Optional[Dict[str, Any]] = None,
49
+ ) -> str:
50
+ """Compute a deterministic SHA-256 hash of the step configuration.
51
+
52
+ If any step's command/args/expected or the case-level expected block
53
+ changes, the hash will differ and ``--resume`` will fall back to a full
54
+ re-run.
55
+ """
56
+ def field(step: Any, name: str, default: Any) -> Any:
57
+ if hasattr(step, name):
58
+ return getattr(step, name)
59
+ return step.get(name, default)
60
+
61
+ payload = {
62
+ "steps": [
63
+ {
64
+ "command": field(step, "command", ""),
65
+ # Argument order is significant for command-line programs.
66
+ "args": field(step, "args", []),
67
+ "expected": field(step, "expected", {}),
68
+ "timeout": field(step, "timeout", None),
69
+ "retry_count": field(step, "retry_count", 0),
70
+ }
71
+ for step in steps
72
+ ],
73
+ "case_expected": case_expected or {},
74
+ }
75
+ canonical = json.dumps(
76
+ payload,
77
+ ensure_ascii=False,
78
+ sort_keys=True,
79
+ separators=(",", ":"),
80
+ )
81
+ return hashlib.sha256(canonical.encode("utf-8")).hexdigest()
82
+
83
+
84
+ def load_sequence_state(
85
+ workspace: str, case_name: str
86
+ ) -> Optional[Dict[str, Any]]:
87
+ """Load persisted step state for a case. Returns ``None`` if not found."""
88
+ path = _case_state_path(workspace, case_name)
89
+ if not os.path.exists(path):
90
+ return None
91
+ try:
92
+ with open(path, "r", encoding="utf-8") as f:
93
+ return json.load(f)
94
+ except (json.JSONDecodeError, ValueError):
95
+ logger.warning("Corrupted sequence state for %s, ignoring.", case_name)
96
+ return None
97
+
98
+
99
+ def save_sequence_state(
100
+ workspace: str, case_name: str, state: Dict[str, Any]
101
+ ) -> None:
102
+ """Persist step state for a case."""
103
+ dir_path = _state_dir_path(workspace)
104
+ os.makedirs(dir_path, exist_ok=True)
105
+ path = _case_state_path(workspace, case_name)
106
+ with open(path, "w", encoding="utf-8") as f:
107
+ json.dump(state, f, indent=2, ensure_ascii=False)
108
+
109
+
110
+ def delete_sequence_state(workspace: str, case_name: str) -> None:
111
+ """Remove state file and cached step outputs (case fully passed)."""
112
+ # Remove state file
113
+ path = _case_state_path(workspace, case_name)
114
+ if os.path.exists(path):
115
+ os.remove(path)
116
+ # Remove cached step outputs
117
+ for step_idx in range(1, 100): # reasonable upper bound
118
+ out_path = _step_output_path(workspace, case_name, step_idx)
119
+ if os.path.exists(out_path):
120
+ os.remove(out_path)
121
+ else:
122
+ break
123
+
124
+
125
+ def save_step_output(
126
+ workspace: str, case_name: str, step_idx: int, output: str
127
+ ) -> str:
128
+ """Save step output to cache and return the cache file path."""
129
+ path = _step_output_path(workspace, case_name, step_idx)
130
+ with open(path, "w", encoding="utf-8") as f:
131
+ f.write(output)
132
+ return path
133
+
134
+
135
+ def load_step_output(
136
+ workspace: str, case_name: str, step_idx: int
137
+ ) -> Optional[str]:
138
+ """Load cached step output. Returns ``None`` if not found."""
139
+ path = _step_output_path(workspace, case_name, step_idx)
140
+ if not os.path.exists(path):
141
+ return None
142
+ with open(path, "r", encoding="utf-8") as f:
143
+ return f.read()