failstep 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
failstep/cli.py ADDED
@@ -0,0 +1,227 @@
1
+ from __future__ import annotations
2
+
3
+ import sys
4
+ from enum import StrEnum
5
+ from pathlib import Path
6
+
7
+ import typer
8
+
9
+ from failstep import __version__
10
+ from failstep.compare import compare_reports
11
+ from failstep.diagnose import diagnose as diagnose_run
12
+ from failstep.errors import ParseError
13
+ from failstep.models import Report, Severity
14
+ from failstep.parser import load_run
15
+ from failstep.report import (
16
+ format_compare_json,
17
+ format_compare_markdown,
18
+ format_compare_terminal,
19
+ format_diagnose_json,
20
+ format_diagnose_markdown,
21
+ format_diagnose_terminal,
22
+ format_error_json,
23
+ format_error_terminal,
24
+ format_fix_json,
25
+ format_fix_markdown,
26
+ format_fix_terminal,
27
+ format_inspect_json,
28
+ format_inspect_markdown,
29
+ format_inspect_terminal,
30
+ format_internal_json,
31
+ format_internal_terminal,
32
+ )
33
+
34
+ app = typer.Typer(
35
+ add_completion=False,
36
+ no_args_is_help=True,
37
+ pretty_exceptions_enable=False,
38
+ pretty_exceptions_show_locals=False,
39
+ help="Lint one finished AI agent run from a file.",
40
+ )
41
+
42
+
43
+ class OutputFormat(StrEnum):
44
+ terminal = "terminal"
45
+ json = "json"
46
+ markdown = "markdown"
47
+
48
+
49
+ class FailOn(StrEnum):
50
+ error = "error"
51
+ warning = "warning"
52
+
53
+
54
+ def _display_path(path: Path) -> str:
55
+ try:
56
+ relative = path.resolve().relative_to(Path.cwd().resolve())
57
+ return relative.as_posix()
58
+ except ValueError:
59
+ return path.as_posix()
60
+
61
+
62
+ def _emit_error(error: ParseError, output_format: OutputFormat) -> None:
63
+ if output_format is OutputFormat.json:
64
+ sys.stdout.write(format_error_json(error, __version__))
65
+ else:
66
+ sys.stdout.write(format_error_terminal(error))
67
+ raise typer.Exit(2)
68
+
69
+
70
+ def _emit_internal(output_format: OutputFormat) -> None:
71
+ if output_format is OutputFormat.json:
72
+ sys.stdout.write(format_internal_json(__version__))
73
+ else:
74
+ sys.stdout.write(format_internal_terminal())
75
+ raise typer.Exit(3)
76
+
77
+
78
+ def _load(path: Path, output_format: OutputFormat):
79
+ try:
80
+ return load_run(path)
81
+ except ParseError as error:
82
+ _emit_error(error, output_format)
83
+ except OSError as exc:
84
+ _emit_error(ParseError(str(exc), _display_path(path)), output_format)
85
+ raise RuntimeError("unreachable")
86
+
87
+
88
+ def _write_diagnose(report: Report, output_format: OutputFormat) -> None:
89
+ if output_format is OutputFormat.json:
90
+ sys.stdout.write(format_diagnose_json(report, __version__))
91
+ elif output_format is OutputFormat.markdown:
92
+ sys.stdout.write(format_diagnose_markdown(report, __version__))
93
+ else:
94
+ sys.stdout.write(format_diagnose_terminal(report, __version__))
95
+
96
+
97
+ def _exit_for(report: Report, fail_on: FailOn) -> int:
98
+ if not report.findings:
99
+ return 0
100
+ if fail_on is FailOn.warning:
101
+ return 1
102
+ if any(item.severity is Severity.error for item in report.findings):
103
+ return 1
104
+ return 0
105
+
106
+
107
+ @app.command()
108
+ def inspect(
109
+ trace: Path = typer.Argument(..., help="Path to a native JSON or JSONL trace."),
110
+ output_format: OutputFormat = typer.Option(
111
+ OutputFormat.terminal,
112
+ "--format",
113
+ help="terminal, json, or markdown.",
114
+ ),
115
+ ) -> None:
116
+ """Print the run. No verdict."""
117
+ run = _load(trace, output_format)
118
+ display = _display_path(trace)
119
+ if output_format is OutputFormat.json:
120
+ sys.stdout.write(format_inspect_json(run, display, __version__))
121
+ elif output_format is OutputFormat.markdown:
122
+ sys.stdout.write(format_inspect_markdown(run, display, __version__))
123
+ else:
124
+ sys.stdout.write(format_inspect_terminal(run, display, __version__))
125
+
126
+
127
+ @app.command()
128
+ def diagnose(
129
+ trace: Path = typer.Argument(..., help="Path to a native JSON or JSONL trace."),
130
+ output_format: OutputFormat = typer.Option(
131
+ OutputFormat.terminal,
132
+ "--format",
133
+ help="terminal, json, or markdown.",
134
+ ),
135
+ fail_on: FailOn = typer.Option(
136
+ FailOn.error,
137
+ "--fail-on",
138
+ help="Exit 1 on this severity or higher.",
139
+ ),
140
+ no_llm: bool = typer.Option(
141
+ False,
142
+ "--no-llm",
143
+ help="Skip LLM leftover (default path never calls a model).",
144
+ ),
145
+ no_redact: bool = typer.Option(
146
+ False,
147
+ "--no-redact",
148
+ help="Warn; secrets are still redacted before leftover requests.",
149
+ ),
150
+ ) -> None:
151
+ """Print a root cause from deterministic detectors."""
152
+ if no_redact:
153
+ sys.stderr.write(
154
+ "Secrets are still redacted before any leftover request.\n"
155
+ )
156
+ run = _load(trace, output_format)
157
+ try:
158
+ report = diagnose_run(
159
+ run, _display_path(trace), no_llm=no_llm
160
+ )
161
+ except Exception:
162
+ _emit_internal(output_format)
163
+ _write_diagnose(report, output_format)
164
+ code = _exit_for(report, fail_on)
165
+ if code:
166
+ raise typer.Exit(code)
167
+
168
+
169
+ @app.command()
170
+ def compare(
171
+ old: Path = typer.Argument(..., help="Older trace file."),
172
+ new: Path = typer.Argument(..., help="Newer trace file."),
173
+ output_format: OutputFormat = typer.Option(
174
+ OutputFormat.terminal,
175
+ "--format",
176
+ help="terminal, json, or markdown.",
177
+ ),
178
+ ) -> None:
179
+ """Count diffs between two diagnosed runs. Never calls leftover."""
180
+ old_run = _load(old, output_format)
181
+ new_run = _load(new, output_format)
182
+ try:
183
+ result = compare_reports(
184
+ diagnose_run(old_run, _display_path(old), no_llm=True),
185
+ diagnose_run(new_run, _display_path(new), no_llm=True),
186
+ )
187
+ except Exception:
188
+ _emit_internal(output_format)
189
+ if output_format is OutputFormat.json:
190
+ sys.stdout.write(format_compare_json(result, __version__))
191
+ elif output_format is OutputFormat.markdown:
192
+ sys.stdout.write(format_compare_markdown(result, __version__))
193
+ else:
194
+ sys.stdout.write(format_compare_terminal(result, __version__))
195
+ if result.has_diff():
196
+ raise typer.Exit(1)
197
+
198
+
199
+ @app.command()
200
+ def fix(
201
+ trace: Path = typer.Argument(..., help="Path to a native JSON or JSONL trace."),
202
+ output_format: OutputFormat = typer.Option(
203
+ OutputFormat.terminal,
204
+ "--format",
205
+ help="terminal, json, or markdown.",
206
+ ),
207
+ ) -> None:
208
+ """Print a suggested patch. Does not write files. Never calls leftover."""
209
+ run = _load(trace, output_format)
210
+ try:
211
+ report = diagnose_run(run, _display_path(trace), no_llm=True)
212
+ except Exception:
213
+ _emit_internal(output_format)
214
+ if output_format is OutputFormat.json:
215
+ sys.stdout.write(format_fix_json(report, __version__))
216
+ elif output_format is OutputFormat.markdown:
217
+ sys.stdout.write(format_fix_markdown(report, __version__))
218
+ else:
219
+ sys.stdout.write(format_fix_terminal(report, __version__))
220
+ if report.root_cause is not None:
221
+ raise typer.Exit(1)
222
+
223
+
224
+ @app.command()
225
+ def version() -> None:
226
+ """Print the tool version."""
227
+ sys.stdout.write(f"failstep {__version__}\n")
failstep/compare.py ADDED
@@ -0,0 +1,119 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass
4
+ from typing import Any
5
+
6
+ from failstep.diagnose import diagnose
7
+ from failstep.models import Finding, Report, Run
8
+
9
+ _RUN_INT_KEYS = ("steps", "duration_ms", "tokens_in", "tokens_out")
10
+
11
+
12
+ @dataclass(frozen=True)
13
+ class FieldDelta:
14
+ key: str
15
+ old: Any
16
+ new: Any
17
+ delta: int | None = None
18
+
19
+
20
+ @dataclass(frozen=True)
21
+ class CompareResult:
22
+ old_file: str
23
+ new_file: str
24
+ old_run: Run
25
+ new_run: Run
26
+ old_root: Finding | None
27
+ new_root: Finding | None
28
+ old_ids: tuple[str, ...]
29
+ new_ids: tuple[str, ...]
30
+ gone: tuple[str, ...]
31
+ added: tuple[str, ...]
32
+ same: tuple[str, ...]
33
+ run: tuple[FieldDelta, ...]
34
+
35
+ def has_diff(self) -> bool:
36
+ return bool(self.gone or self.added or self.run)
37
+
38
+
39
+ def compare_reports(old: Report, new: Report) -> CompareResult:
40
+ old_ids = _finding_ids(old)
41
+ new_ids = _finding_ids(new)
42
+ old_set = set(old_ids)
43
+ new_set = set(new_ids)
44
+ gone = tuple(item for item in old_ids if item not in new_set)
45
+ added = tuple(item for item in new_ids if item not in old_set)
46
+ same = tuple(item for item in old_ids if item in new_set)
47
+ return CompareResult(
48
+ old_file=old.file,
49
+ new_file=new.file,
50
+ old_run=old.run,
51
+ new_run=new.run,
52
+ old_root=old.root_cause,
53
+ new_root=new.root_cause,
54
+ old_ids=old_ids,
55
+ new_ids=new_ids,
56
+ gone=gone,
57
+ added=added,
58
+ same=same,
59
+ run=_run_deltas(old.run, new.run),
60
+ )
61
+
62
+
63
+ def compare_runs(
64
+ old_run: Run,
65
+ new_run: Run,
66
+ old_file: str,
67
+ new_file: str,
68
+ ) -> CompareResult:
69
+ old = diagnose(old_run, old_file, no_llm=True)
70
+ new = diagnose(new_run, new_file, no_llm=True)
71
+ return compare_reports(old, new)
72
+
73
+
74
+ def _finding_ids(report: Report) -> tuple[str, ...]:
75
+ seen: list[str] = []
76
+ for item in report.findings:
77
+ if item.id not in seen:
78
+ seen.append(item.id)
79
+ return tuple(seen)
80
+
81
+
82
+ def _run_value(run: Run, key: str) -> Any:
83
+ if key == "status":
84
+ return run.status.value
85
+ if key == "steps":
86
+ return len(run.steps)
87
+ if key == "duration_ms":
88
+ return run.duration_ms
89
+ if key == "tokens_in":
90
+ return run.tokens_in
91
+ if key == "tokens_out":
92
+ return run.tokens_out
93
+ raise KeyError(key)
94
+
95
+
96
+ def _run_deltas(old: Run, new: Run) -> tuple[FieldDelta, ...]:
97
+ out: list[FieldDelta] = []
98
+ old_status = old.status.value
99
+ new_status = new.status.value
100
+ if old_status != new_status:
101
+ out.append(FieldDelta(key="status", old=old_status, new=new_status))
102
+ for key in _RUN_INT_KEYS:
103
+ old_value = _run_value(old, key)
104
+ new_value = _run_value(new, key)
105
+ if old_value is None or new_value is None:
106
+ continue
107
+ if old_value == new_value:
108
+ continue
109
+ if not isinstance(old_value, int) or not isinstance(new_value, int):
110
+ continue
111
+ out.append(
112
+ FieldDelta(
113
+ key=key,
114
+ old=old_value,
115
+ new=new_value,
116
+ delta=new_value - old_value,
117
+ )
118
+ )
119
+ return tuple(out)
@@ -0,0 +1,29 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Callable
4
+
5
+ from failstep.detectors.malformed import detect as detect_malformed
6
+ from failstep.detectors.retrieval import detect as detect_retrieval
7
+ from failstep.detectors.retry import detect as detect_retry
8
+ from failstep.detectors.schema import detect as detect_schema
9
+ from failstep.detectors.timeout import detect as detect_timeout
10
+ from failstep.detectors.tool_error import detect as detect_tool_error
11
+ from failstep.models import Finding, Run
12
+
13
+ Detector = Callable[[Run], list[Finding]]
14
+
15
+ DETECTORS: list[Detector] = [
16
+ detect_malformed,
17
+ detect_schema,
18
+ detect_tool_error,
19
+ detect_retry,
20
+ detect_timeout,
21
+ detect_retrieval,
22
+ ]
23
+
24
+
25
+ def run_detectors(run: Run) -> list[Finding]:
26
+ findings: list[Finding] = []
27
+ for detector in DETECTORS:
28
+ findings.extend(detector(run))
29
+ return findings
@@ -0,0 +1,115 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from typing import Any
5
+
6
+ from failstep.evidence import finding
7
+ from failstep.models import Finding, Run, Step
8
+
9
+ _JSON_START = "{["
10
+
11
+
12
+ def detect(run: Run) -> list[Finding]:
13
+ bad: list[Step] = []
14
+ reasons: list[str] = []
15
+ samples: list[str] = []
16
+ for step in run.steps:
17
+ reason = _malformed(step)
18
+ if reason is None:
19
+ continue
20
+ bad.append(step)
21
+ reasons.append(reason)
22
+ samples.append(_sample(step.output))
23
+ if not bad:
24
+ return []
25
+ return [
26
+ finding(
27
+ code="FS001",
28
+ detector="malformed",
29
+ title="malformed output",
30
+ steps=bad,
31
+ evidence=[
32
+ ("steps", len(bad)),
33
+ ("reason", reasons[0]),
34
+ ("output", samples[0]),
35
+ ("tool", bad[0].name),
36
+ ],
37
+ recommendation=(
38
+ "Return complete JSON from the tool. Do not truncate the payload."
39
+ ),
40
+ )
41
+ ]
42
+
43
+
44
+ def _malformed(step: Step) -> str | None:
45
+ output = step.output
46
+ if isinstance(output, str):
47
+ text = output.strip()
48
+ if not text:
49
+ return None
50
+ if text[0] in _JSON_START:
51
+ try:
52
+ json.loads(text)
53
+ except json.JSONDecodeError:
54
+ if text.endswith("...") or not _balanced(text):
55
+ return "truncated json"
56
+ return "invalid json"
57
+ if text.endswith("...") and len(text) > 3:
58
+ return "truncated payload"
59
+ return None
60
+
61
+ schema = _output_schema(step)
62
+ if schema is None or not isinstance(output, dict):
63
+ return None
64
+ required = schema.get("required")
65
+ if not isinstance(required, list):
66
+ return None
67
+ missing = [key for key in required if isinstance(key, str) and key not in output]
68
+ if missing:
69
+ return "missing required output fields"
70
+ return None
71
+
72
+
73
+ def _output_schema(step: Step) -> dict[str, Any] | None:
74
+ raw = step.metadata.get("output_schema")
75
+ if isinstance(raw, dict):
76
+ return raw
77
+ if step.schema_ and isinstance(step.schema_.get("output"), dict):
78
+ return step.schema_["output"]
79
+ return None
80
+
81
+
82
+ def _balanced(text: str) -> bool:
83
+ curly = 0
84
+ square = 0
85
+ in_string = False
86
+ escape = False
87
+ for char in text:
88
+ if in_string:
89
+ if escape:
90
+ escape = False
91
+ elif char == "\\":
92
+ escape = True
93
+ elif char == '"':
94
+ in_string = False
95
+ continue
96
+ if char == '"':
97
+ in_string = True
98
+ elif char == "{":
99
+ curly += 1
100
+ elif char == "}":
101
+ curly -= 1
102
+ elif char == "[":
103
+ square += 1
104
+ elif char == "]":
105
+ square -= 1
106
+ if curly < 0 or square < 0:
107
+ return False
108
+ return curly == 0 and square == 0 and not in_string
109
+
110
+
111
+ def _sample(output: Any) -> str:
112
+ if isinstance(output, str):
113
+ text = output.replace("\n", " ")
114
+ return text if len(text) <= 80 else text[:77] + "..."
115
+ return str(output)