stepfork 0.1.0a2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stepfork/__init__.py +32 -0
- stepfork/__main__.py +6 -0
- stepfork/cli/__init__.py +1 -0
- stepfork/cli/diff.py +156 -0
- stepfork/cli/export.py +210 -0
- stepfork/cli/inspect.py +155 -0
- stepfork/cli/main.py +54 -0
- stepfork/cli/replay.py +221 -0
- stepfork/cli/validate.py +160 -0
- stepfork/diff/__init__.py +25 -0
- stepfork/diff/compare.py +144 -0
- stepfork/diff/engine.py +267 -0
- stepfork/diff/models.py +63 -0
- stepfork/export/__init__.py +20 -0
- stepfork/export/entrypoint.py +58 -0
- stepfork/export/generator.py +178 -0
- stepfork/export/runtime.py +101 -0
- stepfork/fork/__init__.py +4 -0
- stepfork/inspect/__init__.py +11 -0
- stepfork/inspect/inspector.py +143 -0
- stepfork/inspect/models.py +42 -0
- stepfork/minimize/__init__.py +4 -0
- stepfork/py.typed +0 -0
- stepfork/recorder/__init__.py +21 -0
- stepfork/recorder/llm.py +68 -0
- stepfork/recorder/session.py +500 -0
- stepfork/recorder/tooling.py +208 -0
- stepfork/replay/__init__.py +32 -0
- stepfork/replay/exceptions.py +37 -0
- stepfork/replay/plan.py +282 -0
- stepfork/replay/session.py +338 -0
- stepfork/trace/__init__.py +115 -0
- stepfork/trace/canonical.py +55 -0
- stepfork/trace/hashing.py +75 -0
- stepfork/trace/integrity.py +196 -0
- stepfork/trace/jsonable.py +97 -0
- stepfork/trace/manifest.py +85 -0
- stepfork/trace/models.py +271 -0
- stepfork/trace/redaction.py +294 -0
- stepfork/trace/replay_policy.py +18 -0
- stepfork/trace/schema.py +25 -0
- stepfork/trace/storage.py +250 -0
- stepfork/trace/validation.py +285 -0
- stepfork/version.py +5 -0
- stepfork-0.1.0a2.dist-info/METADATA +377 -0
- stepfork-0.1.0a2.dist-info/RECORD +49 -0
- stepfork-0.1.0a2.dist-info/WHEEL +4 -0
- stepfork-0.1.0a2.dist-info/entry_points.txt +2 -0
- stepfork-0.1.0a2.dist-info/licenses/LICENSE +184 -0
stepfork/__init__.py
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""Stepfork: behavioral regression testing for AI agents."""
|
|
2
|
+
|
|
3
|
+
from stepfork.diff import diff_traces
|
|
4
|
+
from stepfork.recorder import RecordingSession, llm_request, record, trace_tool
|
|
5
|
+
from stepfork.replay import (
|
|
6
|
+
RecordedDependencyError,
|
|
7
|
+
ReplayError,
|
|
8
|
+
ReplayExhaustedError,
|
|
9
|
+
ReplayMismatchError,
|
|
10
|
+
ReplayPolicyError,
|
|
11
|
+
ReplaySession,
|
|
12
|
+
)
|
|
13
|
+
from stepfork.trace import FailureInfo, ToolCall, Trace
|
|
14
|
+
from stepfork.version import __version__
|
|
15
|
+
|
|
16
|
+
__all__ = [
|
|
17
|
+
"FailureInfo",
|
|
18
|
+
"RecordedDependencyError",
|
|
19
|
+
"RecordingSession",
|
|
20
|
+
"ReplayError",
|
|
21
|
+
"ReplayExhaustedError",
|
|
22
|
+
"ReplayMismatchError",
|
|
23
|
+
"ReplayPolicyError",
|
|
24
|
+
"ReplaySession",
|
|
25
|
+
"ToolCall",
|
|
26
|
+
"Trace",
|
|
27
|
+
"__version__",
|
|
28
|
+
"diff_traces",
|
|
29
|
+
"llm_request",
|
|
30
|
+
"record",
|
|
31
|
+
"trace_tool",
|
|
32
|
+
]
|
stepfork/__main__.py
ADDED
stepfork/cli/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Command-line interface for Stepfork."""
|
stepfork/cli/diff.py
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"""`stepfork diff` command."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Annotated, Any
|
|
9
|
+
|
|
10
|
+
import typer
|
|
11
|
+
from rich.console import Console
|
|
12
|
+
|
|
13
|
+
from stepfork.diff import DiffResult, FieldChange, StepDiff, diff_traces
|
|
14
|
+
from stepfork.trace import TraceStorageError
|
|
15
|
+
from stepfork.trace.redaction import redact_json
|
|
16
|
+
|
|
17
|
+
console = Console()
|
|
18
|
+
VALUE_LIMIT = 120
|
|
19
|
+
|
|
20
|
+
RESULT_EQUIVALENT = "BEHAVIOR EQUIVALENT"
|
|
21
|
+
RESULT_CHANGED = "BEHAVIOR CHANGED"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def diff_command(
|
|
25
|
+
baseline: Annotated[
|
|
26
|
+
Path,
|
|
27
|
+
typer.Argument(
|
|
28
|
+
help="Baseline .sftrace directory bundle.",
|
|
29
|
+
exists=False,
|
|
30
|
+
file_okay=False,
|
|
31
|
+
dir_okay=True,
|
|
32
|
+
),
|
|
33
|
+
],
|
|
34
|
+
candidate: Annotated[
|
|
35
|
+
Path,
|
|
36
|
+
typer.Argument(
|
|
37
|
+
help="Candidate .sftrace directory bundle.",
|
|
38
|
+
exists=False,
|
|
39
|
+
file_okay=False,
|
|
40
|
+
dir_okay=True,
|
|
41
|
+
),
|
|
42
|
+
],
|
|
43
|
+
json_output: Annotated[
|
|
44
|
+
bool,
|
|
45
|
+
typer.Option("--json", help="Emit machine-readable JSON (no Rich markup)."),
|
|
46
|
+
] = False,
|
|
47
|
+
) -> None:
|
|
48
|
+
"""Compare the behavior of two recorded agent runs.
|
|
49
|
+
|
|
50
|
+
Exit codes: 0 behavior equivalent, 1 meaningful behavioral difference,
|
|
51
|
+
2 invalid or unreadable input.
|
|
52
|
+
"""
|
|
53
|
+
try:
|
|
54
|
+
result = diff_traces(baseline, candidate)
|
|
55
|
+
except (TraceStorageError, OSError) as exc:
|
|
56
|
+
if json_output:
|
|
57
|
+
_write_json(
|
|
58
|
+
{
|
|
59
|
+
"error": "unreadable_trace",
|
|
60
|
+
"message": _sanitize_text(str(exc)),
|
|
61
|
+
}
|
|
62
|
+
)
|
|
63
|
+
else:
|
|
64
|
+
console.print("[red]Unable to compare traces.[/red]")
|
|
65
|
+
console.print(_sanitize_text(str(exc)))
|
|
66
|
+
raise typer.Exit(2) from exc
|
|
67
|
+
|
|
68
|
+
if json_output:
|
|
69
|
+
sys.stdout.write(result.model_dump_json())
|
|
70
|
+
sys.stdout.write("\n")
|
|
71
|
+
raise typer.Exit(0 if result.equivalent else 1)
|
|
72
|
+
|
|
73
|
+
_print_diff(result)
|
|
74
|
+
raise typer.Exit(0 if result.equivalent else 1)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _print_diff(result: DiffResult) -> None:
|
|
78
|
+
console.print("[bold]Stepfork Behavioral Diff[/bold]")
|
|
79
|
+
console.print()
|
|
80
|
+
console.print(f"Baseline: {result.baseline.path or '(in-memory trace)'}")
|
|
81
|
+
console.print(f"Candidate: {result.candidate.path or '(in-memory trace)'}")
|
|
82
|
+
console.print()
|
|
83
|
+
|
|
84
|
+
for step in result.steps:
|
|
85
|
+
_print_step(step)
|
|
86
|
+
|
|
87
|
+
if result.changed or result.added or result.removed:
|
|
88
|
+
console.print(
|
|
89
|
+
f"Steps: {len(result.steps)} total "
|
|
90
|
+
f"({result.changed} changed, {result.added} added, "
|
|
91
|
+
f"{result.removed} removed, {result.unchanged} unchanged)"
|
|
92
|
+
)
|
|
93
|
+
console.print()
|
|
94
|
+
console.print(f"[red]Result: {RESULT_CHANGED}[/red]")
|
|
95
|
+
else:
|
|
96
|
+
console.print(f"Steps: {len(result.steps)} total (all unchanged)")
|
|
97
|
+
console.print()
|
|
98
|
+
console.print(f"[green]Result: {RESULT_EQUIVALENT}[/green]")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _print_step(step: StepDiff) -> None:
|
|
102
|
+
header = f"{step.step_type}: {step.label}"
|
|
103
|
+
if step.kind == "added":
|
|
104
|
+
console.print(f"[green]+[/green] {header} [green](added)[/green]")
|
|
105
|
+
return
|
|
106
|
+
if step.kind == "removed":
|
|
107
|
+
console.print(f"[red]-[/red] {header} [red](removed)[/red]")
|
|
108
|
+
return
|
|
109
|
+
|
|
110
|
+
console.print(header)
|
|
111
|
+
if not step.changes:
|
|
112
|
+
console.print(" unchanged")
|
|
113
|
+
return
|
|
114
|
+
|
|
115
|
+
groups: dict[str, list[FieldChange]] = {}
|
|
116
|
+
for change in step.changes:
|
|
117
|
+
key = change.path.split(".", 1)[0].split("[", 1)[0]
|
|
118
|
+
groups.setdefault(key, []).append(change)
|
|
119
|
+
for field in step.fields:
|
|
120
|
+
groups.setdefault(field, [])
|
|
121
|
+
|
|
122
|
+
for key, changes in groups.items():
|
|
123
|
+
if not changes:
|
|
124
|
+
console.print(f" {key}: unchanged")
|
|
125
|
+
continue
|
|
126
|
+
console.print(f" {key}:")
|
|
127
|
+
for change in changes:
|
|
128
|
+
console.print(f" expected: {_short(change.expected, absent=True)}")
|
|
129
|
+
console.print(f" actual: {_short(change.actual, absent=True)}")
|
|
130
|
+
if change.path != key:
|
|
131
|
+
console.print(f" at: {change.path}")
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _short(value: Any, *, absent: bool = False) -> str:
|
|
135
|
+
if value is None:
|
|
136
|
+
return "(absent)" if absent else "None"
|
|
137
|
+
sanitized = redact_json(value).value
|
|
138
|
+
try:
|
|
139
|
+
text = json.dumps(sanitized, ensure_ascii=False, sort_keys=True)
|
|
140
|
+
except (TypeError, ValueError):
|
|
141
|
+
return "<unserializable>"
|
|
142
|
+
if len(text) > VALUE_LIMIT:
|
|
143
|
+
return f"{text[: VALUE_LIMIT - 1]}…"
|
|
144
|
+
return text
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _sanitize_text(value: str) -> str:
|
|
148
|
+
sanitized = redact_json(value).value
|
|
149
|
+
if isinstance(sanitized, str):
|
|
150
|
+
return sanitized
|
|
151
|
+
return str(sanitized)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _write_json(payload: dict[str, Any]) -> None:
|
|
155
|
+
sys.stdout.write(json.dumps(payload, sort_keys=True))
|
|
156
|
+
sys.stdout.write("\n")
|
stepfork/cli/export.py
ADDED
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
"""`stepfork export` command."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Annotated
|
|
9
|
+
|
|
10
|
+
import typer
|
|
11
|
+
from rich.console import Console
|
|
12
|
+
|
|
13
|
+
from stepfork.export.entrypoint import EntrypointError, resolve_entrypoint
|
|
14
|
+
from stepfork.export.generator import (
|
|
15
|
+
ExportError,
|
|
16
|
+
ExportExistsError,
|
|
17
|
+
export_pytest_test,
|
|
18
|
+
)
|
|
19
|
+
from stepfork.trace import (
|
|
20
|
+
JsonValue,
|
|
21
|
+
RunEnd,
|
|
22
|
+
RunStatus,
|
|
23
|
+
Trace,
|
|
24
|
+
TraceStorageError,
|
|
25
|
+
validate_bundle,
|
|
26
|
+
)
|
|
27
|
+
from stepfork.trace.redaction import redact_json
|
|
28
|
+
from stepfork.trace.replay_policy import ReplayPolicy
|
|
29
|
+
|
|
30
|
+
console = Console()
|
|
31
|
+
RECORDED_OUTPUT_WARNING = (
|
|
32
|
+
"[yellow]warning:[/yellow] no --expect-output given; the test will "
|
|
33
|
+
"assert the recorded run output. That passes against the code that "
|
|
34
|
+
"produced the recording. Pass --expect-output to assert the corrected "
|
|
35
|
+
"behavior instead."
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def export_command(
|
|
40
|
+
path: Annotated[
|
|
41
|
+
Path,
|
|
42
|
+
typer.Argument(
|
|
43
|
+
help="Path to a .sftrace directory bundle.",
|
|
44
|
+
exists=False,
|
|
45
|
+
file_okay=False,
|
|
46
|
+
dir_okay=True,
|
|
47
|
+
),
|
|
48
|
+
],
|
|
49
|
+
pytest_format: Annotated[
|
|
50
|
+
bool,
|
|
51
|
+
typer.Option(
|
|
52
|
+
"--pytest/--no-pytest",
|
|
53
|
+
help="Export a pytest test (the only v0.1 format).",
|
|
54
|
+
),
|
|
55
|
+
] = True,
|
|
56
|
+
entrypoint: Annotated[
|
|
57
|
+
str | None,
|
|
58
|
+
typer.Option(
|
|
59
|
+
"--entrypoint",
|
|
60
|
+
help="Trusted local MODULE:FUNCTION the generated test executes.",
|
|
61
|
+
),
|
|
62
|
+
] = None,
|
|
63
|
+
expect_output: Annotated[
|
|
64
|
+
Path | None,
|
|
65
|
+
typer.Option(
|
|
66
|
+
"--expect-output",
|
|
67
|
+
help="JSON file defining the expected regression outcome.",
|
|
68
|
+
),
|
|
69
|
+
] = None,
|
|
70
|
+
output: Annotated[
|
|
71
|
+
Path | None,
|
|
72
|
+
typer.Option(
|
|
73
|
+
"--output",
|
|
74
|
+
help="Destination test file (default: ./test_<trace>_regression.py).",
|
|
75
|
+
),
|
|
76
|
+
] = None,
|
|
77
|
+
overwrite: Annotated[
|
|
78
|
+
bool,
|
|
79
|
+
typer.Option(
|
|
80
|
+
"--overwrite",
|
|
81
|
+
help="Replace the output file if it already exists.",
|
|
82
|
+
),
|
|
83
|
+
] = False,
|
|
84
|
+
mode: Annotated[
|
|
85
|
+
str,
|
|
86
|
+
typer.Option(
|
|
87
|
+
"--mode",
|
|
88
|
+
help="Replay mode used by the generated test (frozen recommended).",
|
|
89
|
+
),
|
|
90
|
+
] = "frozen",
|
|
91
|
+
) -> None:
|
|
92
|
+
"""Export an executable pytest regression test from a trace.
|
|
93
|
+
|
|
94
|
+
The generated test loads the frozen trace, executes your trusted
|
|
95
|
+
entrypoint under replay, and fails when behavior diverges from the
|
|
96
|
+
expected outcome.
|
|
97
|
+
"""
|
|
98
|
+
if not pytest_format:
|
|
99
|
+
console.print("[red]v0.1 supports pytest export only (--pytest).[/red]")
|
|
100
|
+
raise typer.Exit(2)
|
|
101
|
+
|
|
102
|
+
try:
|
|
103
|
+
ReplayPolicy(mode)
|
|
104
|
+
except ValueError:
|
|
105
|
+
console.print(f"[red]Invalid replay mode {mode!r}.[/red]")
|
|
106
|
+
raise typer.Exit(2) from None
|
|
107
|
+
|
|
108
|
+
if entrypoint is None:
|
|
109
|
+
console.print(
|
|
110
|
+
"[red]An executable regression test needs --entrypoint "
|
|
111
|
+
"MODULE:FUNCTION.[/red]"
|
|
112
|
+
)
|
|
113
|
+
raise typer.Exit(2)
|
|
114
|
+
|
|
115
|
+
result = validate_bundle(path, strict=True)
|
|
116
|
+
if not result.valid:
|
|
117
|
+
console.print("[red]Trace failed validation.[/red]")
|
|
118
|
+
for issue in result.issues:
|
|
119
|
+
console.print(f" {issue.code}: {issue.message}")
|
|
120
|
+
raise typer.Exit(2)
|
|
121
|
+
|
|
122
|
+
try:
|
|
123
|
+
trace = Trace.load(path)
|
|
124
|
+
except TraceStorageError as exc:
|
|
125
|
+
console.print("[red]Unable to read trace.[/red]")
|
|
126
|
+
console.print(_sanitize_text(str(exc)))
|
|
127
|
+
raise typer.Exit(2) from exc
|
|
128
|
+
|
|
129
|
+
try:
|
|
130
|
+
resolve_entrypoint(entrypoint, import_root=Path.cwd())
|
|
131
|
+
except EntrypointError as exc:
|
|
132
|
+
console.print("[red]Unable to resolve entrypoint.[/red]")
|
|
133
|
+
console.print(_sanitize_text(str(exc)))
|
|
134
|
+
raise typer.Exit(2) from exc
|
|
135
|
+
|
|
136
|
+
warning: str | None = None
|
|
137
|
+
if expect_output is not None:
|
|
138
|
+
expectation = _load_expectation(expect_output)
|
|
139
|
+
has_expectation = True
|
|
140
|
+
else:
|
|
141
|
+
expectation = _recorded_output(trace)
|
|
142
|
+
has_expectation = expectation is not None
|
|
143
|
+
warning = RECORDED_OUTPUT_WARNING
|
|
144
|
+
|
|
145
|
+
destination = output or Path(f"test_{_safe_stem(path)}_regression.py")
|
|
146
|
+
if destination.exists() and not overwrite:
|
|
147
|
+
console.print(
|
|
148
|
+
f"[red]{destination} already exists; pass --overwrite to replace it.[/red]"
|
|
149
|
+
)
|
|
150
|
+
raise typer.Exit(1)
|
|
151
|
+
|
|
152
|
+
try:
|
|
153
|
+
written = export_pytest_test(
|
|
154
|
+
trace_path=path,
|
|
155
|
+
output_path=destination,
|
|
156
|
+
entrypoint=entrypoint,
|
|
157
|
+
expectation=expectation,
|
|
158
|
+
has_expectation=has_expectation,
|
|
159
|
+
mode=mode,
|
|
160
|
+
import_root=Path.cwd(),
|
|
161
|
+
overwrite=overwrite,
|
|
162
|
+
)
|
|
163
|
+
except ExportExistsError as exc:
|
|
164
|
+
console.print(f"[red]{_sanitize_text(str(exc))}[/red]")
|
|
165
|
+
raise typer.Exit(1) from exc
|
|
166
|
+
except ExportError as exc:
|
|
167
|
+
console.print(f"[red]Export failed: {_sanitize_text(str(exc))}[/red]")
|
|
168
|
+
raise typer.Exit(1) from exc
|
|
169
|
+
|
|
170
|
+
if warning:
|
|
171
|
+
console.print(warning)
|
|
172
|
+
console.print(f"[green]Wrote[/green] {written}")
|
|
173
|
+
console.print(f"Entrypoint: {entrypoint}")
|
|
174
|
+
console.print(f"Mode: {mode}")
|
|
175
|
+
if has_expectation:
|
|
176
|
+
console.print(f"Expectation: {_sanitize_text(json.dumps(expectation))}")
|
|
177
|
+
else:
|
|
178
|
+
console.print("Expectation: none (replay fidelity only)")
|
|
179
|
+
raise typer.Exit(0)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _load_expectation(path: Path) -> JsonValue:
|
|
183
|
+
try:
|
|
184
|
+
payload = json.loads(path.read_text(encoding="utf-8"))
|
|
185
|
+
except OSError as exc:
|
|
186
|
+
console.print(f"[red]Cannot read --expect-output:[/red] {exc}")
|
|
187
|
+
raise typer.Exit(2) from exc
|
|
188
|
+
except json.JSONDecodeError as exc:
|
|
189
|
+
console.print(f"[red]--expect-output is not valid JSON:[/red] {exc.msg}")
|
|
190
|
+
raise typer.Exit(2) from exc
|
|
191
|
+
return redact_json(payload).value
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _recorded_output(trace: Trace) -> JsonValue | None:
|
|
195
|
+
for event in reversed(trace.events):
|
|
196
|
+
if isinstance(event, RunEnd) and event.run_status is RunStatus.COMPLETED:
|
|
197
|
+
return event.output
|
|
198
|
+
return None
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _safe_stem(path: Path) -> str:
|
|
202
|
+
stem = re.sub(r"\W+", "_", path.stem).strip("_")
|
|
203
|
+
return stem or "trace"
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _sanitize_text(value: str) -> str:
|
|
207
|
+
sanitized = redact_json(value).value
|
|
208
|
+
if isinstance(sanitized, str):
|
|
209
|
+
return sanitized
|
|
210
|
+
return str(sanitized)
|
stepfork/cli/inspect.py
ADDED
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
"""`stepfork inspect` command."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Annotated, Any
|
|
9
|
+
|
|
10
|
+
import typer
|
|
11
|
+
from rich.console import Console
|
|
12
|
+
from rich.table import Table
|
|
13
|
+
|
|
14
|
+
from stepfork.inspect.inspector import filter_timeline, inspect_bundle
|
|
15
|
+
from stepfork.inspect.models import TraceInspection
|
|
16
|
+
from stepfork.trace import TraceStorageError
|
|
17
|
+
from stepfork.trace.redaction import redact_json
|
|
18
|
+
|
|
19
|
+
console = Console()
|
|
20
|
+
MAX_DETAIL_LENGTH = 180
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _sanitize_text(value: str) -> str:
|
|
24
|
+
sanitized = redact_json(value).value
|
|
25
|
+
if isinstance(sanitized, str):
|
|
26
|
+
return sanitized
|
|
27
|
+
return str(sanitized)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def inspect_command(
|
|
31
|
+
path: Annotated[
|
|
32
|
+
Path,
|
|
33
|
+
typer.Argument(
|
|
34
|
+
help="Path to a .sftrace directory bundle.",
|
|
35
|
+
exists=False,
|
|
36
|
+
file_okay=False,
|
|
37
|
+
dir_okay=True,
|
|
38
|
+
readable=True,
|
|
39
|
+
),
|
|
40
|
+
],
|
|
41
|
+
json_output: Annotated[
|
|
42
|
+
bool,
|
|
43
|
+
typer.Option("--json", help="Emit sanitized machine-readable JSON."),
|
|
44
|
+
] = False,
|
|
45
|
+
events: Annotated[
|
|
46
|
+
bool,
|
|
47
|
+
typer.Option("--events", help="Show sanitized event details."),
|
|
48
|
+
] = False,
|
|
49
|
+
errors_only: Annotated[
|
|
50
|
+
bool,
|
|
51
|
+
typer.Option("--errors-only", help="Show only error events."),
|
|
52
|
+
] = False,
|
|
53
|
+
step: Annotated[
|
|
54
|
+
int | None,
|
|
55
|
+
typer.Option("--step", help="Show events at a specific logical step."),
|
|
56
|
+
] = None,
|
|
57
|
+
) -> None:
|
|
58
|
+
"""Inspect a `.sftrace` bundle."""
|
|
59
|
+
try:
|
|
60
|
+
inspection = filter_timeline(
|
|
61
|
+
inspect_bundle(path),
|
|
62
|
+
errors_only=errors_only,
|
|
63
|
+
step=step,
|
|
64
|
+
)
|
|
65
|
+
except TraceStorageError as exc:
|
|
66
|
+
message = _sanitize_text(str(exc))
|
|
67
|
+
if json_output:
|
|
68
|
+
_write_json(
|
|
69
|
+
{
|
|
70
|
+
"error": "unreadable_trace",
|
|
71
|
+
"message": message,
|
|
72
|
+
}
|
|
73
|
+
)
|
|
74
|
+
else:
|
|
75
|
+
console.print("[red]Unable to inspect trace.[/red]")
|
|
76
|
+
console.print(message)
|
|
77
|
+
raise typer.Exit(2) from exc
|
|
78
|
+
|
|
79
|
+
if step is not None and not inspection.timeline:
|
|
80
|
+
message = f"No events found at step {step}."
|
|
81
|
+
if json_output:
|
|
82
|
+
_write_json({"error": "missing_step", "message": message})
|
|
83
|
+
else:
|
|
84
|
+
console.print(f"[red]{message}[/red]")
|
|
85
|
+
raise typer.Exit(1)
|
|
86
|
+
|
|
87
|
+
if json_output:
|
|
88
|
+
sys.stdout.write(inspection.model_dump_json())
|
|
89
|
+
sys.stdout.write("\n")
|
|
90
|
+
raise typer.Exit(0)
|
|
91
|
+
|
|
92
|
+
_print_summary(inspection)
|
|
93
|
+
_print_timeline(inspection, show_details=events)
|
|
94
|
+
raise typer.Exit(0)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _print_summary(inspection: TraceInspection) -> None:
|
|
98
|
+
console.print("[bold]Stepfork Trace Inspector[/bold]")
|
|
99
|
+
console.print()
|
|
100
|
+
|
|
101
|
+
table = Table.grid(padding=(0, 4))
|
|
102
|
+
table.add_column(style="bold")
|
|
103
|
+
table.add_column()
|
|
104
|
+
table.add_row("Agent:", inspection.agent_name)
|
|
105
|
+
table.add_row("Run:", inspection.run_id)
|
|
106
|
+
table.add_row("Status:", inspection.status.upper())
|
|
107
|
+
table.add_row("Events:", str(inspection.events))
|
|
108
|
+
table.add_row("Schema:", inspection.schema_version)
|
|
109
|
+
table.add_row("Created:", inspection.created_at)
|
|
110
|
+
table.add_row("LLM calls:", str(inspection.llm_calls))
|
|
111
|
+
table.add_row("Tool calls:", str(inspection.tool_calls))
|
|
112
|
+
if inspection.duration_ms is not None:
|
|
113
|
+
table.add_row("Duration:", f"{inspection.duration_ms} ms")
|
|
114
|
+
table.add_row("Integrity:", inspection.integrity.upper())
|
|
115
|
+
if inspection.failure:
|
|
116
|
+
table.add_row("Failure:", _compact(inspection.failure))
|
|
117
|
+
console.print(table)
|
|
118
|
+
console.print()
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _print_timeline(inspection: TraceInspection, *, show_details: bool) -> None:
|
|
122
|
+
console.print("[bold]Event Timeline[/bold]")
|
|
123
|
+
console.print()
|
|
124
|
+
table = Table(show_lines=show_details)
|
|
125
|
+
table.add_column("Step", justify="right")
|
|
126
|
+
table.add_column("Type")
|
|
127
|
+
table.add_column("Name/Model")
|
|
128
|
+
table.add_column("Status")
|
|
129
|
+
if show_details:
|
|
130
|
+
table.add_column("Details")
|
|
131
|
+
|
|
132
|
+
for event in inspection.timeline:
|
|
133
|
+
row = [
|
|
134
|
+
str(event.step),
|
|
135
|
+
event.type,
|
|
136
|
+
event.label,
|
|
137
|
+
event.status,
|
|
138
|
+
]
|
|
139
|
+
if show_details:
|
|
140
|
+
row.append(_compact(event.details))
|
|
141
|
+
table.add_row(*row)
|
|
142
|
+
|
|
143
|
+
console.print(table)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _compact(value: Any) -> str:
|
|
147
|
+
text = json.dumps(value, ensure_ascii=False, sort_keys=True)
|
|
148
|
+
if len(text) <= MAX_DETAIL_LENGTH:
|
|
149
|
+
return text
|
|
150
|
+
return f"{text[: MAX_DETAIL_LENGTH - 1]}…"
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _write_json(value: dict[str, Any]) -> None:
|
|
154
|
+
sys.stdout.write(json.dumps(value, sort_keys=True))
|
|
155
|
+
sys.stdout.write("\n")
|
stepfork/cli/main.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Stepfork command-line interface."""
|
|
2
|
+
|
|
3
|
+
from typing import Annotated
|
|
4
|
+
|
|
5
|
+
import typer
|
|
6
|
+
from rich.console import Console
|
|
7
|
+
|
|
8
|
+
from stepfork import __version__
|
|
9
|
+
from stepfork.cli.diff import diff_command
|
|
10
|
+
from stepfork.cli.export import export_command
|
|
11
|
+
from stepfork.cli.inspect import inspect_command
|
|
12
|
+
from stepfork.cli.replay import replay_command
|
|
13
|
+
from stepfork.cli.validate import validate_command
|
|
14
|
+
|
|
15
|
+
app = typer.Typer(
|
|
16
|
+
name="stepfork",
|
|
17
|
+
help="Behavioral regression testing for AI agents.",
|
|
18
|
+
invoke_without_command=True,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
console = Console()
|
|
22
|
+
|
|
23
|
+
app.command("diff")(diff_command)
|
|
24
|
+
app.command("export")(export_command)
|
|
25
|
+
app.command("inspect")(inspect_command)
|
|
26
|
+
app.command("replay")(replay_command)
|
|
27
|
+
app.command("validate")(validate_command)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def version_callback(value: bool) -> None:
|
|
31
|
+
"""Print the Stepfork version and exit."""
|
|
32
|
+
if value:
|
|
33
|
+
console.print(f"stepfork {__version__}")
|
|
34
|
+
raise typer.Exit()
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@app.callback()
|
|
38
|
+
def main(
|
|
39
|
+
ctx: typer.Context,
|
|
40
|
+
version: Annotated[
|
|
41
|
+
bool | None,
|
|
42
|
+
typer.Option(
|
|
43
|
+
"--version",
|
|
44
|
+
"-V",
|
|
45
|
+
callback=version_callback,
|
|
46
|
+
is_eager=True,
|
|
47
|
+
help="Show the Stepfork version and exit.",
|
|
48
|
+
),
|
|
49
|
+
] = None,
|
|
50
|
+
) -> None:
|
|
51
|
+
"""Turn failed AI-agent runs into reproducible regression tests."""
|
|
52
|
+
if ctx.invoked_subcommand is None:
|
|
53
|
+
console.print(ctx.get_help())
|
|
54
|
+
raise typer.Exit()
|