symtest-cli 1.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- symtest/__init__.py +45 -0
- symtest/cli.py +549 -0
- symtest/commands/__init__.py +9 -0
- symtest/commands/compare.py +221 -0
- symtest/config/__init__.py +7 -0
- symtest/config/config_io.py +346 -0
- symtest/config/config_schema.py +330 -0
- symtest/config/import_expander.py +149 -0
- symtest/config/inheritance_expander.py +197 -0
- symtest/core/__init__.py +15 -0
- symtest/core/assertions.py +253 -0
- symtest/core/base_runner.py +299 -0
- symtest/core/config_loader.py +536 -0
- symtest/core/execution.py +498 -0
- symtest/core/history_store.py +96 -0
- symtest/core/last_run_store.py +109 -0
- symtest/core/parallel_runner.py +251 -0
- symtest/core/process_worker.py +93 -0
- symtest/core/sequence_state.py +143 -0
- symtest/core/setup.py +137 -0
- symtest/core/test_case.py +76 -0
- symtest/core/types.py +92 -0
- symtest/file_comparator/__init__.py +10 -0
- symtest/file_comparator/base_comparator.py +109 -0
- symtest/file_comparator/binary_comparator.py +399 -0
- symtest/file_comparator/csv_comparator.py +241 -0
- symtest/file_comparator/factory.py +191 -0
- symtest/file_comparator/h5_comparator.py +777 -0
- symtest/file_comparator/json_comparator.py +323 -0
- symtest/file_comparator/result.py +213 -0
- symtest/file_comparator/script_comparator.py +182 -0
- symtest/file_comparator/text_comparator.py +182 -0
- symtest/file_comparator/xml_comparator.py +150 -0
- symtest/logging_config.py +66 -0
- symtest/runners/__init__.py +15 -0
- symtest/runners/config_runner.py +96 -0
- symtest/runners/json_runner.py +21 -0
- symtest/runners/parallel_config_runner.py +278 -0
- symtest/runners/parallel_json_runner.py +26 -0
- symtest/runners/parallel_yaml_runner.py +31 -0
- symtest/runners/yaml_runner.py +26 -0
- symtest/tui/__init__.py +11 -0
- symtest/tui/app.py +90 -0
- symtest/tui/controllers/__init__.py +0 -0
- symtest/tui/controllers/case_controller.py +322 -0
- symtest/tui/screens/__init__.py +0 -0
- symtest/tui/screens/case_editor.py +244 -0
- symtest/tui/screens/case_list.py +255 -0
- symtest/tui/widgets/__init__.py +0 -0
- symtest/tui/widgets/case_table.py +113 -0
- symtest/tui/widgets/expected_editor.py +159 -0
- symtest/tui/widgets/search_bar.py +160 -0
- symtest/tui/widgets/steps_editor.py +243 -0
- symtest/utils/__init__.py +21 -0
- symtest/utils/junit_xml_writer.py +137 -0
- symtest/utils/path_resolver.py +124 -0
- symtest/utils/report_generator.py +208 -0
- symtest_cli-1.3.0.dist-info/METADATA +316 -0
- symtest_cli-1.3.0.dist-info/RECORD +63 -0
- symtest_cli-1.3.0.dist-info/WHEEL +5 -0
- symtest_cli-1.3.0.dist-info/entry_points.txt +4 -0
- symtest_cli-1.3.0.dist-info/licenses/LICENSE +21 -0
- symtest_cli-1.3.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,536 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Unified configuration parsing layer.
|
|
3
|
+
|
|
4
|
+
Shared logic for loading test cases from a config dict (already parsed from
|
|
5
|
+
JSON/YAML) into TestCase objects, and for executing sequence test cases.
|
|
6
|
+
|
|
7
|
+
Backward-compatible: the runner classes still expose ``load_test_cases()`` and
|
|
8
|
+
``_run_sequence()`` as before; they merely delegate to the functions here.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import logging
|
|
14
|
+
import re
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Any, Dict, List, Optional, Tuple
|
|
17
|
+
|
|
18
|
+
from .test_case import TestCase, TestCaseStep
|
|
19
|
+
from .execution import execute_single_test_case
|
|
20
|
+
from ..utils.path_resolver import resolve_paths
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger("symtest.core.config_loader")
|
|
23
|
+
|
|
24
|
+
# ---------------------------------------------------------------------------
|
|
25
|
+
# Placeholder substitution
|
|
26
|
+
# ---------------------------------------------------------------------------
|
|
27
|
+
|
|
28
|
+
_PLACEHOLDER_RE = re.compile(r'\{(\w+)\}')
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def substitute_placeholders(
|
|
32
|
+
config: Dict[str, Any],
|
|
33
|
+
variables: Optional[Dict[str, Any]] = None,
|
|
34
|
+
) -> Dict[str, Any]:
|
|
35
|
+
"""递归替换 config 中字符串值的 ``{placeholder}`` 占位符。
|
|
36
|
+
|
|
37
|
+
只替换 ``variables`` 中存在的 key,未匹配的 ``{xxx}`` 原样保留,
|
|
38
|
+
不会影响 ``expected.matches`` 等字段中的正则模式(如 ``{2}``)。
|
|
39
|
+
"""
|
|
40
|
+
if not variables:
|
|
41
|
+
return config
|
|
42
|
+
|
|
43
|
+
def _sub(value: Any) -> Any:
|
|
44
|
+
if isinstance(value, str):
|
|
45
|
+
return _PLACEHOLDER_RE.sub(
|
|
46
|
+
lambda m: str(variables[m.group(1)])
|
|
47
|
+
if m.group(1) in variables else m.group(0),
|
|
48
|
+
value,
|
|
49
|
+
)
|
|
50
|
+
if isinstance(value, list):
|
|
51
|
+
return [_sub(item) for item in value]
|
|
52
|
+
if isinstance(value, dict):
|
|
53
|
+
return {k: _sub(v) for k, v in value.items()}
|
|
54
|
+
return value
|
|
55
|
+
|
|
56
|
+
return _sub(config)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
# ---------------------------------------------------------------------------
|
|
60
|
+
# Test-case parsing (loaded dict → list[TestCase])
|
|
61
|
+
# ---------------------------------------------------------------------------
|
|
62
|
+
|
|
63
|
+
def _split_and_resolve(
|
|
64
|
+
command_string: str,
|
|
65
|
+
args: List[str],
|
|
66
|
+
workspace: Path,
|
|
67
|
+
path_resolver: Any,
|
|
68
|
+
) -> Tuple[str, List[str]]:
|
|
69
|
+
"""Split a command string into executable + leading args, then resolve paths.
|
|
70
|
+
|
|
71
|
+
``path_resolver`` must be a ``PathResolver`` instance (or duck-typed
|
|
72
|
+
equivalent with ``split_command`` / ``resolve_paths`` methods).
|
|
73
|
+
"""
|
|
74
|
+
executable, leading_args = path_resolver.split_command(command_string)
|
|
75
|
+
return executable, (
|
|
76
|
+
resolve_paths(leading_args, str(workspace))
|
|
77
|
+
+ path_resolver.resolve_paths(args)
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def parse_test_cases(
|
|
82
|
+
config: Dict[str, Any],
|
|
83
|
+
workspace: Optional[Path] = None,
|
|
84
|
+
path_resolver: Any = None,
|
|
85
|
+
) -> List[TestCase]:
|
|
86
|
+
"""Parse ``config['test_cases']`` into a list of ``TestCase`` objects.
|
|
87
|
+
|
|
88
|
+
Supports both single-command mode and sequence (``steps``) mode.
|
|
89
|
+
|
|
90
|
+
When *workspace* and *path_resolver* are provided (Runner mode),
|
|
91
|
+
required fields are validated and command/args paths are resolved.
|
|
92
|
+
When omitted (TUI mode), missing fields get sensible defaults and
|
|
93
|
+
raw values are kept as-is for display purposes.
|
|
94
|
+
"""
|
|
95
|
+
cases: List[TestCase] = []
|
|
96
|
+
resolve = workspace is not None and path_resolver is not None
|
|
97
|
+
|
|
98
|
+
for case in config.get("test_cases", []):
|
|
99
|
+
if "steps" in case:
|
|
100
|
+
# ── Sequence mode ──
|
|
101
|
+
steps: List[TestCaseStep] = []
|
|
102
|
+
for step in case.get("steps", []):
|
|
103
|
+
if resolve:
|
|
104
|
+
step_required = ["command", "args", "expected"]
|
|
105
|
+
if not all(field in step for field in step_required):
|
|
106
|
+
raise ValueError(
|
|
107
|
+
f"Step in test case '{case.get('name', 'unnamed')}' "
|
|
108
|
+
f"is missing required fields"
|
|
109
|
+
)
|
|
110
|
+
executable, resolved_args = _split_and_resolve(
|
|
111
|
+
step["command"], step["args"], workspace, path_resolver
|
|
112
|
+
)
|
|
113
|
+
steps.append(TestCaseStep(
|
|
114
|
+
command=executable,
|
|
115
|
+
args=resolved_args,
|
|
116
|
+
expected=step["expected"],
|
|
117
|
+
timeout=step.get("timeout"),
|
|
118
|
+
retry_count=step.get("retry_count", 0),
|
|
119
|
+
))
|
|
120
|
+
else:
|
|
121
|
+
steps.append(TestCaseStep(
|
|
122
|
+
command=step.get("command", ""),
|
|
123
|
+
args=step.get("args", []),
|
|
124
|
+
expected=step.get("expected", {}),
|
|
125
|
+
timeout=step.get("timeout"),
|
|
126
|
+
retry_count=step.get("retry_count", 0),
|
|
127
|
+
))
|
|
128
|
+
cases.append(TestCase(
|
|
129
|
+
name=case.get("name", ""),
|
|
130
|
+
steps=steps,
|
|
131
|
+
expected=case.get("expected", {}),
|
|
132
|
+
description=case.get("description", ""),
|
|
133
|
+
resources=case.get("resources"),
|
|
134
|
+
tags=case.get("tags", []),
|
|
135
|
+
expected_failure=case.get("expected_failure", False),
|
|
136
|
+
xfail_reason=case.get("xfail_reason", ""),
|
|
137
|
+
xfail_quiet=case.get("xfail_quiet", False),
|
|
138
|
+
))
|
|
139
|
+
else:
|
|
140
|
+
# ── Single-command mode (backward-compatible) ──
|
|
141
|
+
if resolve:
|
|
142
|
+
required_fields = ["name", "command", "args", "expected"]
|
|
143
|
+
if not all(field in case for field in required_fields):
|
|
144
|
+
raise ValueError(
|
|
145
|
+
f"Test case {case.get('name', 'unnamed')} "
|
|
146
|
+
f"is missing required fields"
|
|
147
|
+
)
|
|
148
|
+
executable, resolved_args = _split_and_resolve(
|
|
149
|
+
case["command"], case["args"], workspace, path_resolver
|
|
150
|
+
)
|
|
151
|
+
cases.append(TestCase(
|
|
152
|
+
name=case["name"],
|
|
153
|
+
command=executable,
|
|
154
|
+
args=resolved_args,
|
|
155
|
+
expected=case["expected"],
|
|
156
|
+
description=case.get("description", ""),
|
|
157
|
+
timeout=case.get("timeout"),
|
|
158
|
+
resources=case.get("resources"),
|
|
159
|
+
tags=case.get("tags", []),
|
|
160
|
+
retry_count=case.get("retry_count", 0),
|
|
161
|
+
expected_failure=case.get("expected_failure", False),
|
|
162
|
+
xfail_reason=case.get("xfail_reason", ""),
|
|
163
|
+
xfail_quiet=case.get("xfail_quiet", False),
|
|
164
|
+
))
|
|
165
|
+
else:
|
|
166
|
+
cases.append(TestCase(
|
|
167
|
+
name=case.get("name", ""),
|
|
168
|
+
command=case.get("command", ""),
|
|
169
|
+
args=case.get("args", []),
|
|
170
|
+
expected=case.get("expected", {}),
|
|
171
|
+
description=case.get("description", ""),
|
|
172
|
+
timeout=case.get("timeout"),
|
|
173
|
+
resources=case.get("resources"),
|
|
174
|
+
tags=case.get("tags", []),
|
|
175
|
+
retry_count=case.get("retry_count", 0),
|
|
176
|
+
expected_failure=case.get("expected_failure", False),
|
|
177
|
+
xfail_reason=case.get("xfail_reason", ""),
|
|
178
|
+
xfail_quiet=case.get("xfail_quiet", False),
|
|
179
|
+
))
|
|
180
|
+
|
|
181
|
+
return cases
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
# ---------------------------------------------------------------------------
|
|
185
|
+
# Step helper (duck-typed access for TestCaseStep / dict)
|
|
186
|
+
# ---------------------------------------------------------------------------
|
|
187
|
+
|
|
188
|
+
def _step_attr(step: Any, key: str, default: Any = None) -> Any:
|
|
189
|
+
"""Get attribute from a ``TestCaseStep`` or key from a ``dict``."""
|
|
190
|
+
if isinstance(step, dict):
|
|
191
|
+
return step.get(key, default)
|
|
192
|
+
return getattr(step, key, default)
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
# ---------------------------------------------------------------------------
|
|
196
|
+
# Shared sequence execution (TestCaseStep list → result dict)
|
|
197
|
+
# ---------------------------------------------------------------------------
|
|
198
|
+
|
|
199
|
+
def execute_sequence(
|
|
200
|
+
case_name: str,
|
|
201
|
+
steps: List[Any],
|
|
202
|
+
workspace: Optional[str] = None,
|
|
203
|
+
*,
|
|
204
|
+
print_prefix: str = "",
|
|
205
|
+
lock: Any = None,
|
|
206
|
+
executor: Any = None,
|
|
207
|
+
case_expected: Optional[Dict[str, Any]] = None,
|
|
208
|
+
update_baseline: bool = False,
|
|
209
|
+
error_analysis: bool = False,
|
|
210
|
+
resume: bool = False,
|
|
211
|
+
) -> Dict[str, Any]:
|
|
212
|
+
"""Execute a sequence test case (fail-fast).
|
|
213
|
+
|
|
214
|
+
``steps`` may be a list of ``TestCaseStep`` objects or plain dicts
|
|
215
|
+
containing ``command`` / ``args`` / ``expected`` / (optional) ``timeout``.
|
|
216
|
+
|
|
217
|
+
Parameters
|
|
218
|
+
----------
|
|
219
|
+
case_name:
|
|
220
|
+
Name of the test case (used in step-names and the result).
|
|
221
|
+
steps:
|
|
222
|
+
Ordered list of steps to execute.
|
|
223
|
+
workspace:
|
|
224
|
+
Working directory for command execution.
|
|
225
|
+
print_prefix:
|
|
226
|
+
Optional prefix printed before every message (e.g. ``"[Worker]"``).
|
|
227
|
+
lock:
|
|
228
|
+
Deprecated; retained for backward compatibility with callers that
|
|
229
|
+
still pass a ``threading.Lock``. Logging is natively thread-safe,
|
|
230
|
+
so the lock is no longer used.
|
|
231
|
+
executor:
|
|
232
|
+
Optional override for ``execute_single_test_case``.
|
|
233
|
+
Defaults to the canonical import; callers that need monkeypatch
|
|
234
|
+
support (e.g. process_worker) should pass their own reference.
|
|
235
|
+
case_expected:
|
|
236
|
+
Optional case-level ``expected`` assertions (return_code,
|
|
237
|
+
output_contains, output_matches, compare_files) that are
|
|
238
|
+
evaluated after *all* steps pass. Supports the same fields
|
|
239
|
+
as step-level ``expected``.
|
|
240
|
+
update_baseline:
|
|
241
|
+
If True, overwrite baseline files on comparison failure in steps.
|
|
242
|
+
Forwarded to ``execute_single_test_case`` for each step.
|
|
243
|
+
resume:
|
|
244
|
+
If True, attempt to skip already-passed steps by loading persisted
|
|
245
|
+
state from ``.cli-test/sequence_state/<case_name>.json``. When the
|
|
246
|
+
config hash matches, previously passed steps are skipped and their
|
|
247
|
+
cached outputs are spliced into ``combined_output``. When the full
|
|
248
|
+
case passes, the state file is deleted. Uses a pure-trust model:
|
|
249
|
+
no artifact validation is performed.
|
|
250
|
+
"""
|
|
251
|
+
if executor is None:
|
|
252
|
+
executor = execute_single_test_case
|
|
253
|
+
|
|
254
|
+
combined_output = ""
|
|
255
|
+
total_duration = 0.0
|
|
256
|
+
all_passed = True
|
|
257
|
+
last_result = None
|
|
258
|
+
failed_step = None
|
|
259
|
+
step_results: List[Dict[str, Any]] = []
|
|
260
|
+
case_assertion_results: List[Dict[str, Any]] = []
|
|
261
|
+
case_hint: Optional[Dict[str, Any]] = None
|
|
262
|
+
|
|
263
|
+
prefix = f"{print_prefix} " if print_prefix else ""
|
|
264
|
+
|
|
265
|
+
# ── Resume: skip previously passed steps ──
|
|
266
|
+
start_step = 0
|
|
267
|
+
state = None
|
|
268
|
+
if resume and workspace:
|
|
269
|
+
from .sequence_state import (
|
|
270
|
+
compute_config_hash,
|
|
271
|
+
load_sequence_state,
|
|
272
|
+
load_step_output,
|
|
273
|
+
save_sequence_state,
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
config_hash = compute_config_hash(steps, case_expected)
|
|
277
|
+
state = load_sequence_state(workspace, case_name)
|
|
278
|
+
|
|
279
|
+
if state and state.get("config_hash") == config_hash:
|
|
280
|
+
saved_steps = state.get("steps", {})
|
|
281
|
+
for idx in sorted(int(k) for k in saved_steps.keys()):
|
|
282
|
+
if idx - 1 >= len(steps):
|
|
283
|
+
break
|
|
284
|
+
sinfo = saved_steps[str(idx)]
|
|
285
|
+
if sinfo.get("status") != "passed":
|
|
286
|
+
break # stop at first unpassed step
|
|
287
|
+
|
|
288
|
+
# Reconstruct this step as "resumed"
|
|
289
|
+
cached = load_step_output(workspace, case_name, idx)
|
|
290
|
+
if cached is None:
|
|
291
|
+
logger.warning(
|
|
292
|
+
" %sResume: cached output for step %d missing; "
|
|
293
|
+
"restarting from step 1.",
|
|
294
|
+
prefix, idx,
|
|
295
|
+
)
|
|
296
|
+
start_step = 0
|
|
297
|
+
combined_output = ""
|
|
298
|
+
total_duration = 0.0
|
|
299
|
+
step_results.clear()
|
|
300
|
+
break
|
|
301
|
+
|
|
302
|
+
step = steps[idx - 1]
|
|
303
|
+
command_str = (
|
|
304
|
+
f"{_step_attr(step, 'command')} "
|
|
305
|
+
f"{' '.join(_step_attr(step, 'args'))}".strip()
|
|
306
|
+
)
|
|
307
|
+
step_results.append({
|
|
308
|
+
"step": idx,
|
|
309
|
+
"name": f"{case_name} [step {idx}/{len(steps)}]",
|
|
310
|
+
"status": "passed",
|
|
311
|
+
"message": "",
|
|
312
|
+
"duration": sinfo.get("duration", 0),
|
|
313
|
+
"command": command_str,
|
|
314
|
+
"resumed": True,
|
|
315
|
+
})
|
|
316
|
+
combined_output += cached
|
|
317
|
+
total_duration += sinfo.get("duration", 0)
|
|
318
|
+
start_step = idx # next step to run
|
|
319
|
+
|
|
320
|
+
if start_step > 0:
|
|
321
|
+
logger.info(
|
|
322
|
+
" %sResume: skipping %d already-passed step(s), "
|
|
323
|
+
"starting at step %d.",
|
|
324
|
+
prefix, start_step, start_step + 1,
|
|
325
|
+
)
|
|
326
|
+
elif state and state.get("config_hash") != config_hash:
|
|
327
|
+
logger.info(
|
|
328
|
+
" %sResume: config changed; discarding stale state "
|
|
329
|
+
"and running full sequence.",
|
|
330
|
+
prefix,
|
|
331
|
+
)
|
|
332
|
+
else:
|
|
333
|
+
logger.info(
|
|
334
|
+
" %sResume: no saved state for '%s'; running full sequence.",
|
|
335
|
+
prefix, case_name,
|
|
336
|
+
)
|
|
337
|
+
elif resume and not workspace:
|
|
338
|
+
logger.warning(
|
|
339
|
+
" %sResume: no workspace set; cannot load sequence state.",
|
|
340
|
+
prefix,
|
|
341
|
+
)
|
|
342
|
+
|
|
343
|
+
for i, step in enumerate(steps):
|
|
344
|
+
if i < start_step:
|
|
345
|
+
continue # already resumed
|
|
346
|
+
|
|
347
|
+
step_idx = i + 1
|
|
348
|
+
step_name = f"{case_name} [step {step_idx}/{len(steps)}]"
|
|
349
|
+
step_case: Dict[str, Any] = {
|
|
350
|
+
"name": step_name,
|
|
351
|
+
"command": _step_attr(step, "command"),
|
|
352
|
+
"args": _step_attr(step, "args"),
|
|
353
|
+
"expected": _step_attr(step, "expected"),
|
|
354
|
+
"description": None,
|
|
355
|
+
"timeout": _step_attr(step, "timeout"),
|
|
356
|
+
"resources": None,
|
|
357
|
+
"retry_count": _step_attr(step, "retry_count", 0),
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
command_preview = (
|
|
361
|
+
f"{step_case['command']} {' '.join(step_case['args'])}".strip()
|
|
362
|
+
)
|
|
363
|
+
logger.info(" %sExecuting step %d/%d: %s", prefix, step_idx, len(steps), command_preview)
|
|
364
|
+
|
|
365
|
+
result = executor(step_case, workspace, update_baseline=update_baseline, error_analysis=error_analysis)
|
|
366
|
+
|
|
367
|
+
if result["output"].strip():
|
|
368
|
+
logger.debug(" %sCommand output for %s:", prefix, step_name)
|
|
369
|
+
for line in result["output"].splitlines():
|
|
370
|
+
logger.debug(" %s", line)
|
|
371
|
+
|
|
372
|
+
step_result = {
|
|
373
|
+
"step": step_idx,
|
|
374
|
+
"name": step_name,
|
|
375
|
+
"status": result["status"],
|
|
376
|
+
"message": result.get("message", ""),
|
|
377
|
+
"duration": result.get("duration", 0),
|
|
378
|
+
"command": f"{step_case['command']} {' '.join(step_case['args'])}".strip(),
|
|
379
|
+
}
|
|
380
|
+
step_results.append(step_result)
|
|
381
|
+
|
|
382
|
+
combined_output += result["output"]
|
|
383
|
+
total_duration += result["duration"]
|
|
384
|
+
last_result = result
|
|
385
|
+
|
|
386
|
+
if result["status"] != "passed":
|
|
387
|
+
all_passed = False
|
|
388
|
+
failed_step = step_idx
|
|
389
|
+
if result.get("message"):
|
|
390
|
+
logger.error(" %sError at step %d: %s", prefix, step_idx, result["message"])
|
|
391
|
+
break
|
|
392
|
+
|
|
393
|
+
# ── Persist step progress for resume ──
|
|
394
|
+
if resume and workspace:
|
|
395
|
+
from .sequence_state import (
|
|
396
|
+
compute_config_hash,
|
|
397
|
+
save_sequence_state,
|
|
398
|
+
save_step_output,
|
|
399
|
+
)
|
|
400
|
+
|
|
401
|
+
if state is None:
|
|
402
|
+
config_hash = compute_config_hash(steps, case_expected)
|
|
403
|
+
state = {
|
|
404
|
+
"case": case_name,
|
|
405
|
+
"config_hash": config_hash,
|
|
406
|
+
"steps": {},
|
|
407
|
+
}
|
|
408
|
+
|
|
409
|
+
state["steps"][str(step_idx)] = {
|
|
410
|
+
"status": "passed",
|
|
411
|
+
"duration": result.get("duration", 0),
|
|
412
|
+
}
|
|
413
|
+
save_sequence_state(workspace, case_name, state)
|
|
414
|
+
save_step_output(
|
|
415
|
+
workspace, case_name, step_idx, result["output"],
|
|
416
|
+
)
|
|
417
|
+
|
|
418
|
+
# ── Case-level assertions ──
|
|
419
|
+
if all_passed and case_expected:
|
|
420
|
+
try:
|
|
421
|
+
from .execution import validate_result
|
|
422
|
+
|
|
423
|
+
case_result: Dict[str, Any] = {
|
|
424
|
+
"name": case_name,
|
|
425
|
+
"status": "passed",
|
|
426
|
+
"message": "",
|
|
427
|
+
"command": "",
|
|
428
|
+
# Validate against full combined output; only trim when reporting
|
|
429
|
+
"output": combined_output,
|
|
430
|
+
"return_code": last_result["return_code"] if last_result else None,
|
|
431
|
+
"duration": total_duration,
|
|
432
|
+
}
|
|
433
|
+
case_assertion_results = validate_result(case_expected, case_result, workspace, update_baseline=update_baseline, error_analysis=error_analysis)
|
|
434
|
+
except AssertionError as exc:
|
|
435
|
+
from .execution import _build_next_action_hint
|
|
436
|
+
|
|
437
|
+
all_passed = False
|
|
438
|
+
failed_step = len(steps) + 1 # synthetic step number
|
|
439
|
+
case_assertion_results = getattr(exc, "assertion_results", [])
|
|
440
|
+
case_hint = _build_next_action_hint(
|
|
441
|
+
getattr(exc, "failure_kind", None), update_baseline=update_baseline,
|
|
442
|
+
)
|
|
443
|
+
last_result = {
|
|
444
|
+
"name": case_name,
|
|
445
|
+
"status": "failed",
|
|
446
|
+
"message": f"Case-level assertion failed: {exc}",
|
|
447
|
+
"command": "",
|
|
448
|
+
"output": "",
|
|
449
|
+
"return_code": None,
|
|
450
|
+
"duration": 0.0,
|
|
451
|
+
"failure_kind": getattr(exc, "failure_kind", None),
|
|
452
|
+
"compare_failures": getattr(exc, "compare_failures", []),
|
|
453
|
+
}
|
|
454
|
+
step_result = {
|
|
455
|
+
"step": "case_assertion",
|
|
456
|
+
"name": case_name,
|
|
457
|
+
"status": "failed",
|
|
458
|
+
"message": f"Case-level assertion failed: {exc}",
|
|
459
|
+
"duration": 0,
|
|
460
|
+
"command": "case-level expected check",
|
|
461
|
+
}
|
|
462
|
+
step_results.append(step_result)
|
|
463
|
+
logger.error(" %sCase-level assertion failed: %s", prefix, exc)
|
|
464
|
+
|
|
465
|
+
# ── Resume cleanup: delete state on full pass ──
|
|
466
|
+
if resume and workspace and all_passed:
|
|
467
|
+
from .sequence_state import delete_sequence_state
|
|
468
|
+
|
|
469
|
+
delete_sequence_state(workspace, case_name)
|
|
470
|
+
logger.info(" %sResume: full pass; sequence state cleaned up.", prefix)
|
|
471
|
+
|
|
472
|
+
status = "passed" if all_passed else (last_result["status"] if last_result else "failed")
|
|
473
|
+
message = ""
|
|
474
|
+
if not all_passed:
|
|
475
|
+
total_steps = len(steps) + (
|
|
476
|
+
1 if case_expected and failed_step == len(steps) + 1 else 0
|
|
477
|
+
)
|
|
478
|
+
message = (
|
|
479
|
+
f"Failed at step {failed_step}/{total_steps}: "
|
|
480
|
+
f"{last_result['message']}" if last_result else "Unknown error"
|
|
481
|
+
)
|
|
482
|
+
|
|
483
|
+
# ── Output sliming: only keep the failed step's output ──
|
|
484
|
+
# When all steps pass, keep the full combined output.
|
|
485
|
+
# When a step or case-level assertion fails, only expose the failed step's
|
|
486
|
+
# output (or empty for case-level failures).
|
|
487
|
+
if all_passed:
|
|
488
|
+
slim_output = combined_output
|
|
489
|
+
elif last_result is not None:
|
|
490
|
+
if failed_step == len(steps) + 1:
|
|
491
|
+
slim_output = ""
|
|
492
|
+
else:
|
|
493
|
+
slim_output = last_result.get("output", "")
|
|
494
|
+
else:
|
|
495
|
+
slim_output = ""
|
|
496
|
+
|
|
497
|
+
# ── assertion_results / next_action_hint resolution ──
|
|
498
|
+
# Case-level assertion data takes precedence; otherwise propagate the
|
|
499
|
+
# failed step's data so sequence results honor the same contract as
|
|
500
|
+
# single-command results.
|
|
501
|
+
if case_assertion_results:
|
|
502
|
+
assertion_results = case_assertion_results
|
|
503
|
+
elif not all_passed and last_result is not None:
|
|
504
|
+
assertion_results = last_result.get("assertion_results", [])
|
|
505
|
+
else:
|
|
506
|
+
assertion_results = []
|
|
507
|
+
|
|
508
|
+
if case_hint is not None:
|
|
509
|
+
next_action_hint = case_hint
|
|
510
|
+
elif not all_passed and last_result is not None:
|
|
511
|
+
next_action_hint = last_result.get("next_action_hint")
|
|
512
|
+
else:
|
|
513
|
+
next_action_hint = None
|
|
514
|
+
|
|
515
|
+
command_summary = " -> ".join(
|
|
516
|
+
f"{_step_attr(s, 'command')} {' '.join(_step_attr(s, 'args'))}".strip()
|
|
517
|
+
for s in steps
|
|
518
|
+
)
|
|
519
|
+
|
|
520
|
+
return {
|
|
521
|
+
"name": case_name,
|
|
522
|
+
"status": status,
|
|
523
|
+
"message": message,
|
|
524
|
+
"command": command_summary,
|
|
525
|
+
"output": slim_output,
|
|
526
|
+
"return_code": last_result["return_code"] if last_result else None,
|
|
527
|
+
"duration": total_duration,
|
|
528
|
+
"step_results": step_results,
|
|
529
|
+
"failed_step": failed_step,
|
|
530
|
+
"failure_kind": last_result.get("failure_kind") if last_result else None,
|
|
531
|
+
"compare_failures": last_result.get("compare_failures", []) if last_result else [],
|
|
532
|
+
"attempts": last_result.get("attempts", 1) if last_result else 1,
|
|
533
|
+
"flaky": last_result.get("flaky", False) if last_result else False,
|
|
534
|
+
"assertion_results": assertion_results,
|
|
535
|
+
"next_action_hint": next_action_hint,
|
|
536
|
+
}
|