stepfork 0.1.0a2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. stepfork/__init__.py +32 -0
  2. stepfork/__main__.py +6 -0
  3. stepfork/cli/__init__.py +1 -0
  4. stepfork/cli/diff.py +156 -0
  5. stepfork/cli/export.py +210 -0
  6. stepfork/cli/inspect.py +155 -0
  7. stepfork/cli/main.py +54 -0
  8. stepfork/cli/replay.py +221 -0
  9. stepfork/cli/validate.py +160 -0
  10. stepfork/diff/__init__.py +25 -0
  11. stepfork/diff/compare.py +144 -0
  12. stepfork/diff/engine.py +267 -0
  13. stepfork/diff/models.py +63 -0
  14. stepfork/export/__init__.py +20 -0
  15. stepfork/export/entrypoint.py +58 -0
  16. stepfork/export/generator.py +178 -0
  17. stepfork/export/runtime.py +101 -0
  18. stepfork/fork/__init__.py +4 -0
  19. stepfork/inspect/__init__.py +11 -0
  20. stepfork/inspect/inspector.py +143 -0
  21. stepfork/inspect/models.py +42 -0
  22. stepfork/minimize/__init__.py +4 -0
  23. stepfork/py.typed +0 -0
  24. stepfork/recorder/__init__.py +21 -0
  25. stepfork/recorder/llm.py +68 -0
  26. stepfork/recorder/session.py +500 -0
  27. stepfork/recorder/tooling.py +208 -0
  28. stepfork/replay/__init__.py +32 -0
  29. stepfork/replay/exceptions.py +37 -0
  30. stepfork/replay/plan.py +282 -0
  31. stepfork/replay/session.py +338 -0
  32. stepfork/trace/__init__.py +115 -0
  33. stepfork/trace/canonical.py +55 -0
  34. stepfork/trace/hashing.py +75 -0
  35. stepfork/trace/integrity.py +196 -0
  36. stepfork/trace/jsonable.py +97 -0
  37. stepfork/trace/manifest.py +85 -0
  38. stepfork/trace/models.py +271 -0
  39. stepfork/trace/redaction.py +294 -0
  40. stepfork/trace/replay_policy.py +18 -0
  41. stepfork/trace/schema.py +25 -0
  42. stepfork/trace/storage.py +250 -0
  43. stepfork/trace/validation.py +285 -0
  44. stepfork/version.py +5 -0
  45. stepfork-0.1.0a2.dist-info/METADATA +377 -0
  46. stepfork-0.1.0a2.dist-info/RECORD +49 -0
  47. stepfork-0.1.0a2.dist-info/WHEEL +4 -0
  48. stepfork-0.1.0a2.dist-info/entry_points.txt +2 -0
  49. stepfork-0.1.0a2.dist-info/licenses/LICENSE +184 -0
stepfork/cli/replay.py ADDED
@@ -0,0 +1,221 @@
1
+ """`stepfork replay` command."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import Annotated, Any
7
+
8
+ import typer
9
+ from rich.console import Console
10
+ from rich.table import Table
11
+
12
+ from stepfork.export.entrypoint import EntrypointError, resolve_entrypoint
13
+ from stepfork.replay import (
14
+ RecordedDependencyError,
15
+ ReplayError,
16
+ ReplaySession,
17
+ )
18
+ from stepfork.trace import RunStatus, Trace, TraceStorageError
19
+ from stepfork.trace.canonical import canonical_json_bytes
20
+ from stepfork.trace.jsonable import TraceSerializationError, to_json_value
21
+ from stepfork.trace.redaction import redact_json
22
+ from stepfork.trace.replay_policy import ReplayPolicy
23
+
24
+ console = Console()
25
+ PREVIEW_LIMIT = 400
26
+
27
+
28
+ def replay_command(
29
+ path: Annotated[
30
+ Path,
31
+ typer.Argument(
32
+ help="Path to a .sftrace directory bundle.",
33
+ exists=False,
34
+ file_okay=False,
35
+ dir_okay=True,
36
+ readable=True,
37
+ ),
38
+ ],
39
+ entrypoint: Annotated[
40
+ str,
41
+ typer.Option(
42
+ "--entrypoint",
43
+ help="Trusted local MODULE:FUNCTION to execute under replay.",
44
+ ),
45
+ ],
46
+ mode: Annotated[
47
+ str,
48
+ typer.Option(
49
+ "--mode",
50
+ help="Replay mode: frozen, live, forbidden, manual, or derived.",
51
+ ),
52
+ ] = "frozen",
53
+ ) -> None:
54
+ """Execute a trusted entrypoint with dependencies replayed from a trace.
55
+
56
+ Only the entrypoint you name is imported and run. Trace bundles are data:
57
+ Stepfork never executes code embedded in a trace.
58
+ """
59
+ try:
60
+ ReplayPolicy(mode)
61
+ except ValueError:
62
+ console.print(f"[red]Invalid replay mode {mode!r}.[/red]")
63
+ raise typer.Exit(2) from None
64
+
65
+ try:
66
+ trace = Trace.load(path)
67
+ except TraceStorageError as exc:
68
+ console.print("[red]Unable to read trace.[/red]")
69
+ console.print(_sanitize_text(str(exc)))
70
+ raise typer.Exit(2) from exc
71
+
72
+ try:
73
+ entry = resolve_entrypoint(entrypoint, import_root=Path.cwd())
74
+ except EntrypointError as exc:
75
+ console.print("[red]Unable to resolve entrypoint.[/red]")
76
+ console.print(_sanitize_text(str(exc)))
77
+ raise typer.Exit(2) from exc
78
+
79
+ divergence: str | None = None
80
+ result_value: Any = None
81
+ agent_exc: BaseException | None = None
82
+
83
+ try:
84
+ with ReplaySession.from_trace(trace, mode=mode) as replay:
85
+ try:
86
+ result_value = entry()
87
+ except ReplayError as exc:
88
+ divergence = _sanitize_text(str(exc))
89
+ except Exception as exc:
90
+ agent_exc = exc
91
+ if divergence is None:
92
+ try:
93
+ replay.verify_complete()
94
+ except ReplayError as exc:
95
+ divergence = _sanitize_text(str(exc))
96
+ matched = replay.matched
97
+ remaining = replay.pending
98
+ except ReplayError as exc:
99
+ console.print("[red]Replay failed before execution.[/red]")
100
+ console.print(_sanitize_text(str(exc)))
101
+ raise typer.Exit(1) from exc
102
+
103
+ status = "completed"
104
+ if divergence is None and agent_exc is not None:
105
+ failure_type = trace.failure.type if trace.failure else None
106
+ raised_type = type(agent_exc).__name__
107
+ if (trace.status is RunStatus.FAILED and raised_type == failure_type) or (
108
+ isinstance(agent_exc, RecordedDependencyError)
109
+ and trace.status is RunStatus.FAILED
110
+ and agent_exc.error_type == failure_type
111
+ ):
112
+ status = "reproduced_failure"
113
+ else:
114
+ recorded = f"status {trace.status.value}"
115
+ if failure_type:
116
+ recorded += f" (failure {failure_type})"
117
+ divergence = (
118
+ f"entrypoint raised {raised_type}: "
119
+ f"{_sanitize_text(str(agent_exc))} but the trace recorded "
120
+ f"{recorded}"
121
+ )
122
+ elif divergence is None and agent_exc is None:
123
+ if trace.status is RunStatus.FAILED:
124
+ divergence = (
125
+ "trace recorded a failed run but the entrypoint completed "
126
+ "without raising"
127
+ )
128
+
129
+ _print_report(
130
+ trace=trace,
131
+ path=path,
132
+ entrypoint=entrypoint,
133
+ mode=mode,
134
+ matched=matched,
135
+ remaining=remaining,
136
+ status=status,
137
+ divergence=divergence,
138
+ result_value=result_value,
139
+ agent_exc=agent_exc,
140
+ )
141
+
142
+ raise typer.Exit(1 if divergence is not None else 0)
143
+
144
+
145
+ def _print_report(
146
+ *,
147
+ trace: Trace,
148
+ path: Path,
149
+ entrypoint: str,
150
+ mode: str,
151
+ matched: list[Any],
152
+ remaining: int,
153
+ status: str,
154
+ divergence: str | None,
155
+ result_value: Any,
156
+ agent_exc: BaseException | None,
157
+ ) -> None:
158
+ console.print("[bold]Stepfork Replay[/bold]")
159
+ console.print()
160
+
161
+ table = Table.grid(padding=(0, 4))
162
+ table.add_column(style="bold")
163
+ table.add_column()
164
+ table.add_row("Trace", str(path))
165
+ table.add_row("Agent", trace.agent_name)
166
+ table.add_row("Entrypoint", entrypoint)
167
+ table.add_row("Mode", mode)
168
+ console.print(table)
169
+ console.print()
170
+
171
+ console.print("[bold]Dependency Calls[/bold]")
172
+ if not matched:
173
+ console.print(" (no dependency calls matched)")
174
+ else:
175
+ for item in matched:
176
+ console.print(
177
+ f" {item.index + 1}. {item.kind} {item.label} "
178
+ f"[green]({item.action})[/green]"
179
+ )
180
+ if remaining:
181
+ console.print(f" [yellow]{remaining} recorded call(s) unmatched[/yellow]")
182
+ console.print()
183
+
184
+ if divergence is not None:
185
+ console.print("Status: [red]DIVERGED[/red]")
186
+ console.print(_sanitize_text(divergence))
187
+ console.print()
188
+ return
189
+
190
+ if status == "reproduced_failure" and agent_exc is not None:
191
+ console.print(
192
+ f"Status: [yellow]REPRODUCED FAILURE[/yellow] ({type(agent_exc).__name__})"
193
+ )
194
+ else:
195
+ console.print("Status: [green]COMPLETED[/green]")
196
+
197
+ console.print(f"Final result: {_result_preview(result_value)}")
198
+ console.print()
199
+
200
+
201
+ def _result_preview(value: Any) -> str:
202
+ if value is None:
203
+ return "None"
204
+ try:
205
+ payload = to_json_value(value)
206
+ text = canonical_json_bytes(payload).decode("utf-8", errors="replace")
207
+ except TraceSerializationError:
208
+ return f"<non-serializable {type(value).__name__}>"
209
+ except TypeError:
210
+ return "<non-serializable>"
211
+ sanitized = _sanitize_text(text)
212
+ if len(sanitized) > PREVIEW_LIMIT:
213
+ return f"{sanitized[: PREVIEW_LIMIT - 1]}…"
214
+ return sanitized
215
+
216
+
217
+ def _sanitize_text(value: str) -> str:
218
+ sanitized = redact_json(value).value
219
+ if isinstance(sanitized, str):
220
+ return sanitized
221
+ return str(sanitized)
@@ -0,0 +1,160 @@
1
+ """`stepfork validate` command."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import Annotated
7
+
8
+ import typer
9
+ from rich.console import Console
10
+ from rich.table import Table
11
+
12
+ from stepfork.trace import TraceStorageError, ValidationResult, validate_bundle
13
+ from stepfork.trace.integrity import IntegrityStatus, verify_bundle_integrity
14
+ from stepfork.trace.storage import load_trace
15
+
16
+ console = Console()
17
+
18
+ UNREADABLE_CODES = {
19
+ "missing_file",
20
+ "invalid_json",
21
+ "invalid_manifest",
22
+ "invalid_event",
23
+ "unsupported_schema",
24
+ "integrity_metadata_invalid",
25
+ "integrity_algorithm_unsupported",
26
+ }
27
+
28
+
29
+ def validate_command(
30
+ path: Annotated[
31
+ Path,
32
+ typer.Argument(
33
+ help="Path to a .sftrace directory bundle.",
34
+ exists=False,
35
+ file_okay=False,
36
+ dir_okay=True,
37
+ readable=True,
38
+ ),
39
+ ],
40
+ partial: Annotated[
41
+ bool,
42
+ typer.Option(
43
+ "--partial",
44
+ help="Allow incomplete in-progress traces without run_start checks.",
45
+ ),
46
+ ] = False,
47
+ verify_integrity: Annotated[
48
+ bool,
49
+ typer.Option(
50
+ "--verify-integrity",
51
+ help="Verify per-event payload hashes and bundle integrity metadata.",
52
+ ),
53
+ ] = False,
54
+ ) -> None:
55
+ """Validate a `.sftrace` bundle."""
56
+ strict = not partial
57
+ structure_result = validate_bundle(path, strict=strict)
58
+ result = validate_bundle(
59
+ path,
60
+ strict=strict,
61
+ verify_integrity=verify_integrity,
62
+ )
63
+
64
+ _print_header(path, strict=strict)
65
+ _print_metadata(path, strict=strict)
66
+ if verify_integrity and structure_result.valid:
67
+ _print_integrity(path)
68
+
69
+ if result.valid:
70
+ console.print("[green]✓[/green] Manifest valid")
71
+ console.print("[green]✓[/green] Event structure valid")
72
+ console.print("[green]✓[/green] Event IDs unique")
73
+ console.print("[green]✓[/green] Parent references valid")
74
+ console.print("[green]✓[/green] Run IDs consistent")
75
+ console.print("[green]✓[/green] Event count matches")
76
+ console.print()
77
+ console.print("[green]Validation passed.[/green]")
78
+ raise typer.Exit(0)
79
+
80
+ for issue in result.issues:
81
+ console.print(f"[red]✗[/red] {issue.code}")
82
+ console.print(f" {issue.message}")
83
+ if issue.location:
84
+ console.print(f" [dim]{issue.location}[/dim]")
85
+ console.print()
86
+
87
+ console.print(f"[red]Validation failed: {len(result.issues)} issues.[/red]")
88
+ if verify_integrity and _has_legacy_integrity_issue(result):
89
+ raise typer.Exit(3)
90
+ has_unreadable_issue = any(
91
+ issue.code in UNREADABLE_CODES for issue in result.issues
92
+ )
93
+ exit_code = 2 if has_unreadable_issue else 1
94
+ raise typer.Exit(exit_code)
95
+
96
+
97
+ def _print_header(path: Path, *, strict: bool) -> None:
98
+ console.print("[bold]Stepfork Trace Validation[/bold]")
99
+ console.print()
100
+
101
+ table = Table.grid(padding=(0, 4))
102
+ table.add_column(style="bold")
103
+ table.add_column()
104
+ table.add_row("Trace", str(path))
105
+ if not path.exists():
106
+ table.add_row("Mode", "strict" if strict else "partial")
107
+ console.print(table)
108
+ console.print()
109
+
110
+
111
+ def _print_metadata(path: Path, *, strict: bool) -> None:
112
+ try:
113
+ trace = load_trace(path)
114
+ except TraceStorageError:
115
+ return
116
+
117
+ table = Table.grid(padding=(0, 4))
118
+ table.add_column(style="bold")
119
+ table.add_column()
120
+ table.add_row("Schema", "0.1")
121
+ table.add_row("Events", str(len(trace.events)))
122
+ table.add_row("Mode", "strict" if strict else "partial")
123
+ console.print(table)
124
+ console.print()
125
+
126
+
127
+ def _print_integrity(path: Path) -> None:
128
+ result = verify_bundle_integrity(path)
129
+ status_label = (
130
+ "UNVERIFIED (legacy bundle)"
131
+ if result.status is IntegrityStatus.UNVERIFIED_LEGACY
132
+ else result.status.value.upper()
133
+ )
134
+ console.print("Structure: [green]PASS[/green]")
135
+ style = "green" if result.status is IntegrityStatus.VERIFIED else "yellow"
136
+ if result.status is IntegrityStatus.MISMATCH:
137
+ style = "red"
138
+ console.print(f"Integrity: [{style}]{status_label}[/{style}]")
139
+ console.print()
140
+
141
+ table = Table.grid(padding=(0, 4))
142
+ table.add_column()
143
+ table.add_column()
144
+ for filename in ("manifest.json", "events.jsonl", "redactions.json"):
145
+ if filename in result.verified_files:
146
+ table.add_row(filename, "verified")
147
+ continue
148
+ issue = next((item for item in result.issues if item.file == filename), None)
149
+ table.add_row(filename, issue.message if issue else "unverified")
150
+ if result.status is IntegrityStatus.UNVERIFIED_LEGACY:
151
+ table.add_row("integrity.json", "missing legacy metadata")
152
+ console.print(table)
153
+ console.print()
154
+
155
+
156
+ def _has_legacy_integrity_issue(result: ValidationResult) -> bool:
157
+ return any(
158
+ issue.code == "integrity_unverified" and issue.location == "integrity.json"
159
+ for issue in result.issues
160
+ )
@@ -0,0 +1,25 @@
1
+ """Behavioral trace comparison."""
2
+
3
+ from stepfork.diff.compare import collect_field_changes, first_difference, values_equal
4
+ from stepfork.diff.engine import BehaviorStep, diff_traces, extract_behavior
5
+ from stepfork.diff.models import (
6
+ ChangeKind,
7
+ DiffResult,
8
+ FieldChange,
9
+ StepDiff,
10
+ TraceSummary,
11
+ )
12
+
13
+ __all__ = [
14
+ "BehaviorStep",
15
+ "ChangeKind",
16
+ "DiffResult",
17
+ "FieldChange",
18
+ "StepDiff",
19
+ "TraceSummary",
20
+ "collect_field_changes",
21
+ "diff_traces",
22
+ "extract_behavior",
23
+ "first_difference",
24
+ "values_equal",
25
+ ]
@@ -0,0 +1,144 @@
1
+ """Deep comparison helpers for JSON trace payloads.
2
+
3
+ Comparisons use Stepfork's canonical JSON profile so ``1`` and ``1.0`` are
4
+ never treated as equal and dictionary key order never matters.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from stepfork.diff.models import FieldChange
10
+ from stepfork.trace.canonical import canonical_json_bytes
11
+ from stepfork.trace.models import JsonValue
12
+
13
+
14
+ def values_equal(expected: JsonValue, actual: JsonValue) -> bool:
15
+ """Return True when both values are canonically identical."""
16
+ try:
17
+ return canonical_json_bytes(expected) == canonical_json_bytes(actual)
18
+ except TypeError:
19
+ return False
20
+
21
+
22
+ def first_difference(
23
+ expected: JsonValue,
24
+ actual: JsonValue,
25
+ *,
26
+ path: str = "",
27
+ ) -> tuple[str, JsonValue, JsonValue] | None:
28
+ """Return the first differing leaf as ``(path, expected, actual)``."""
29
+ if values_equal(expected, actual):
30
+ return None
31
+
32
+ if isinstance(expected, dict) and isinstance(actual, dict):
33
+ for key in sorted(set(expected) | set(actual)):
34
+ child_path = f"{path}.{key}" if path else key
35
+ if key not in expected:
36
+ return child_path, None, actual[key]
37
+ if key not in actual:
38
+ return child_path, expected[key], None
39
+ nested = first_difference(
40
+ expected[key],
41
+ actual[key],
42
+ path=child_path,
43
+ )
44
+ if nested is not None:
45
+ return nested
46
+ return path, expected, actual
47
+
48
+ if isinstance(expected, list) and isinstance(actual, list):
49
+ shared = min(len(expected), len(actual))
50
+ for index in range(shared):
51
+ nested = first_difference(
52
+ expected[index],
53
+ actual[index],
54
+ path=f"{path}[{index}]",
55
+ )
56
+ if nested is not None:
57
+ return nested
58
+ if len(expected) > len(actual):
59
+ return f"{path}[{shared}]", expected[shared], None
60
+ if len(actual) > len(expected):
61
+ return f"{path}[{shared}]", None, actual[shared]
62
+ return path, expected, actual
63
+
64
+ return path or "$", expected, actual
65
+
66
+
67
+ def collect_field_changes(
68
+ expected: JsonValue,
69
+ actual: JsonValue,
70
+ *,
71
+ path: str,
72
+ output: list[FieldChange],
73
+ ) -> None:
74
+ """Append every differing leaf between two payloads to ``output``."""
75
+ if values_equal(expected, actual):
76
+ return
77
+
78
+ if isinstance(expected, dict) and isinstance(actual, dict):
79
+ for key in sorted(set(expected) | set(actual)):
80
+ child_path = f"{path}.{key}" if path else key
81
+ if key not in expected:
82
+ output.append(
83
+ FieldChange(
84
+ path=child_path,
85
+ kind="added",
86
+ expected=None,
87
+ actual=actual[key],
88
+ )
89
+ )
90
+ elif key not in actual:
91
+ output.append(
92
+ FieldChange(
93
+ path=child_path,
94
+ kind="removed",
95
+ expected=expected[key],
96
+ actual=None,
97
+ )
98
+ )
99
+ else:
100
+ collect_field_changes(
101
+ expected[key],
102
+ actual[key],
103
+ path=child_path,
104
+ output=output,
105
+ )
106
+ return
107
+
108
+ if isinstance(expected, list) and isinstance(actual, list):
109
+ shared = min(len(expected), len(actual))
110
+ for index in range(shared):
111
+ collect_field_changes(
112
+ expected[index],
113
+ actual[index],
114
+ path=f"{path}[{index}]",
115
+ output=output,
116
+ )
117
+ for index in range(shared, len(expected)):
118
+ output.append(
119
+ FieldChange(
120
+ path=f"{path}[{index}]",
121
+ kind="removed",
122
+ expected=expected[index],
123
+ actual=None,
124
+ )
125
+ )
126
+ for index in range(shared, len(actual)):
127
+ output.append(
128
+ FieldChange(
129
+ path=f"{path}[{index}]",
130
+ kind="added",
131
+ expected=None,
132
+ actual=actual[index],
133
+ )
134
+ )
135
+ return
136
+
137
+ output.append(
138
+ FieldChange(
139
+ path=path,
140
+ kind="changed",
141
+ expected=expected,
142
+ actual=actual,
143
+ )
144
+ )