stepfork 0.1.0a2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. stepfork/__init__.py +32 -0
  2. stepfork/__main__.py +6 -0
  3. stepfork/cli/__init__.py +1 -0
  4. stepfork/cli/diff.py +156 -0
  5. stepfork/cli/export.py +210 -0
  6. stepfork/cli/inspect.py +155 -0
  7. stepfork/cli/main.py +54 -0
  8. stepfork/cli/replay.py +221 -0
  9. stepfork/cli/validate.py +160 -0
  10. stepfork/diff/__init__.py +25 -0
  11. stepfork/diff/compare.py +144 -0
  12. stepfork/diff/engine.py +267 -0
  13. stepfork/diff/models.py +63 -0
  14. stepfork/export/__init__.py +20 -0
  15. stepfork/export/entrypoint.py +58 -0
  16. stepfork/export/generator.py +178 -0
  17. stepfork/export/runtime.py +101 -0
  18. stepfork/fork/__init__.py +4 -0
  19. stepfork/inspect/__init__.py +11 -0
  20. stepfork/inspect/inspector.py +143 -0
  21. stepfork/inspect/models.py +42 -0
  22. stepfork/minimize/__init__.py +4 -0
  23. stepfork/py.typed +0 -0
  24. stepfork/recorder/__init__.py +21 -0
  25. stepfork/recorder/llm.py +68 -0
  26. stepfork/recorder/session.py +500 -0
  27. stepfork/recorder/tooling.py +208 -0
  28. stepfork/replay/__init__.py +32 -0
  29. stepfork/replay/exceptions.py +37 -0
  30. stepfork/replay/plan.py +282 -0
  31. stepfork/replay/session.py +338 -0
  32. stepfork/trace/__init__.py +115 -0
  33. stepfork/trace/canonical.py +55 -0
  34. stepfork/trace/hashing.py +75 -0
  35. stepfork/trace/integrity.py +196 -0
  36. stepfork/trace/jsonable.py +97 -0
  37. stepfork/trace/manifest.py +85 -0
  38. stepfork/trace/models.py +271 -0
  39. stepfork/trace/redaction.py +294 -0
  40. stepfork/trace/replay_policy.py +18 -0
  41. stepfork/trace/schema.py +25 -0
  42. stepfork/trace/storage.py +250 -0
  43. stepfork/trace/validation.py +285 -0
  44. stepfork/version.py +5 -0
  45. stepfork-0.1.0a2.dist-info/METADATA +377 -0
  46. stepfork-0.1.0a2.dist-info/RECORD +49 -0
  47. stepfork-0.1.0a2.dist-info/WHEEL +4 -0
  48. stepfork-0.1.0a2.dist-info/entry_points.txt +2 -0
  49. stepfork-0.1.0a2.dist-info/licenses/LICENSE +184 -0
@@ -0,0 +1,143 @@
1
+ """Build sanitized inspection summaries from loaded traces."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+ from stepfork.inspect.models import EventSummary, TraceInspection
9
+ from stepfork.trace import (
10
+ ErrorEvent,
11
+ Event,
12
+ LLMRequest,
13
+ LLMResponse,
14
+ RunEnd,
15
+ RunStart,
16
+ StateChange,
17
+ ToolCall,
18
+ ToolResult,
19
+ Trace,
20
+ TraceStorageError,
21
+ )
22
+ from stepfork.trace.integrity import verify_bundle_integrity
23
+ from stepfork.trace.redaction import redact_json
24
+
25
+
26
+ def inspect_bundle(path: str | Path) -> TraceInspection:
27
+ """Load and inspect a `.sftrace` bundle."""
28
+ bundle_path = Path(path)
29
+ trace = Trace.load(bundle_path)
30
+ integrity = verify_bundle_integrity(bundle_path).status.value
31
+ return inspect_trace(trace, path=str(bundle_path), integrity=integrity)
32
+
33
+
34
+ def inspect_trace(
35
+ trace: Trace,
36
+ *,
37
+ path: str | None = None,
38
+ integrity: str = "not_checked",
39
+ ) -> TraceInspection:
40
+ """Return a sanitized inspection summary for a trace."""
41
+ return TraceInspection(
42
+ path=path,
43
+ agent_name=_sanitize_text(trace.agent_name),
44
+ run_id=trace.run_id,
45
+ schema_version="0.1",
46
+ status=trace.status.value,
47
+ created_at=trace.created_at.isoformat(),
48
+ events=len(trace.events),
49
+ llm_calls=trace.totals.llm_calls,
50
+ tool_calls=trace.totals.tool_calls,
51
+ duration_ms=trace.totals.duration_ms,
52
+ failure=_sanitize_failure(trace.failure.model_dump(mode="json"))
53
+ if trace.failure
54
+ else None,
55
+ integrity=integrity,
56
+ timeline=[_event_summary(event) for event in trace.events],
57
+ )
58
+
59
+
60
+ def filter_timeline(
61
+ inspection: TraceInspection,
62
+ *,
63
+ errors_only: bool = False,
64
+ step: int | None = None,
65
+ ) -> TraceInspection:
66
+ """Return an inspection copy with filtered timeline events."""
67
+ timeline = inspection.timeline
68
+ if errors_only:
69
+ timeline = [
70
+ event
71
+ for event in timeline
72
+ if event.status == "error" or event.type == "error"
73
+ ]
74
+ if step is not None:
75
+ timeline = [event for event in timeline if event.step == step]
76
+ return inspection.model_copy(update={"timeline": timeline})
77
+
78
+
79
+ def _event_summary(event: Event) -> EventSummary:
80
+ payload = _sanitize_json(event.model_dump(mode="json"))
81
+ return EventSummary(
82
+ id=event.id,
83
+ step=event.step,
84
+ type=event.type.value,
85
+ label=_event_label(event),
86
+ status=event.status.value,
87
+ parent_id=event.parent_id,
88
+ duration_ms=event.duration_ms,
89
+ details=_event_details(payload),
90
+ )
91
+
92
+
93
+ def _event_label(event: Event) -> str:
94
+ if isinstance(event, RunStart):
95
+ return "run_start"
96
+ if isinstance(event, LLMRequest):
97
+ return event.model
98
+ if isinstance(event, LLMResponse):
99
+ return event.model or "llm_response"
100
+ if isinstance(event, ToolCall | ToolResult):
101
+ return event.name
102
+ if isinstance(event, StateChange):
103
+ return event.key or "state_change"
104
+ if isinstance(event, ErrorEvent):
105
+ return _sanitize_text(event.error_type)
106
+ if isinstance(event, RunEnd):
107
+ return event.run_status.value
108
+ return event.type.value
109
+
110
+
111
+ def _event_details(payload: dict[str, Any]) -> dict[str, Any]:
112
+ hidden = {
113
+ "id",
114
+ "run_id",
115
+ "parent_id",
116
+ "step",
117
+ "type",
118
+ "timestamp",
119
+ "status",
120
+ "replay_policy",
121
+ "input_hash",
122
+ "output_hash",
123
+ "redactions",
124
+ }
125
+ return {key: value for key, value in payload.items() if key not in hidden}
126
+
127
+
128
+ def _sanitize_failure(value: dict[str, Any]) -> dict[str, Any]:
129
+ sanitized = _sanitize_json(value)
130
+ if not isinstance(sanitized, dict):
131
+ raise TraceStorageError("inspection failure sanitizer returned non-object")
132
+ return sanitized
133
+
134
+
135
+ def _sanitize_json(value: Any) -> Any:
136
+ return redact_json(value).value
137
+
138
+
139
+ def _sanitize_text(value: str) -> str:
140
+ sanitized = redact_json(value).value
141
+ if not isinstance(sanitized, str):
142
+ raise TraceStorageError("inspection text sanitizer returned non-string")
143
+ return sanitized
@@ -0,0 +1,42 @@
1
+ """Structured models for trace inspection output."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from pydantic import BaseModel, ConfigDict
8
+
9
+
10
+ class EventSummary(BaseModel):
11
+ """Concise event data for timeline display."""
12
+
13
+ model_config = ConfigDict(extra="forbid")
14
+
15
+ id: str
16
+ step: int
17
+ type: str
18
+ label: str
19
+ status: str
20
+ parent_id: str | None = None
21
+ duration_ms: int | None = None
22
+ details: dict[str, Any]
23
+
24
+
25
+ class TraceInspection(BaseModel):
26
+ """Sanitized trace inspection result."""
27
+
28
+ model_config = ConfigDict(extra="forbid")
29
+
30
+ path: str | None = None
31
+ agent_name: str
32
+ run_id: str
33
+ schema_version: str
34
+ status: str
35
+ created_at: str
36
+ events: int
37
+ llm_calls: int
38
+ tool_calls: int
39
+ duration_ms: int | None = None
40
+ failure: dict[str, Any] | None = None
41
+ integrity: str
42
+ timeline: list[EventSummary]
@@ -0,0 +1,4 @@
1
+ """Failure minimization.
2
+
3
+ Planned for Stepfork v0.2.
4
+ """
stepfork/py.typed ADDED
File without changes
@@ -0,0 +1,21 @@
1
+ """Agent execution recording primitives."""
2
+
3
+ from stepfork.recorder.llm import llm_request
4
+ from stepfork.recorder.session import (
5
+ DEFAULT_TRACE_DIR,
6
+ LLMCallHandle,
7
+ RecordingSession,
8
+ active_recorder,
9
+ record,
10
+ )
11
+ from stepfork.recorder.tooling import trace_tool
12
+
13
+ __all__ = [
14
+ "DEFAULT_TRACE_DIR",
15
+ "LLMCallHandle",
16
+ "RecordingSession",
17
+ "active_recorder",
18
+ "llm_request",
19
+ "record",
20
+ "trace_tool",
21
+ ]
@@ -0,0 +1,68 @@
1
+ """Provider-independent LLM boundary recording and replay."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable
6
+ from typing import Any
7
+
8
+ from stepfork.recorder.session import active_recorder
9
+ from stepfork.replay.session import active_replay
10
+ from stepfork.trace import ReplayPolicy
11
+ from stepfork.trace.jsonable import Serializer, to_json_value
12
+
13
+
14
+ def llm_request(
15
+ *,
16
+ model: str,
17
+ input: Any,
18
+ call: Callable[[], Any],
19
+ provider: str | None = None,
20
+ replay_policy: ReplayPolicy = ReplayPolicy.FROZEN,
21
+ serializer: Serializer | None = None,
22
+ ) -> Any:
23
+ """Invoke an LLM dependency while recording or replaying its boundary.
24
+
25
+ ``call`` is a zero-argument callable performing the real request. Its
26
+ behavior depends on the active Stepfork context:
27
+
28
+ - recording (``with record(...)``): the request and response are captured
29
+ and ``call()`` executes.
30
+ - frozen replay: ``call()`` is **not** executed; the recorded response is
31
+ returned.
32
+ - live replay: the call matches the recording, then ``call()`` executes.
33
+ - no active context: ``call()`` executes unchanged.
34
+
35
+ This is the replay-safe way for agents to wrap LLM calls.
36
+ """
37
+ replay = active_replay()
38
+ if replay is not None:
39
+ input_json = to_json_value(input, serializer=serializer)
40
+ decision = replay.before_llm(
41
+ provider=provider,
42
+ model=model,
43
+ input=input_json,
44
+ )
45
+ if not decision.execute:
46
+ return decision.value
47
+ return call()
48
+
49
+ session = active_recorder()
50
+ if session is None:
51
+ return call()
52
+
53
+ request = session.record_llm_request(
54
+ model=model,
55
+ input=input,
56
+ provider=provider,
57
+ replay_policy=replay_policy,
58
+ )
59
+ session.push_scope(request.id)
60
+ try:
61
+ output = call()
62
+ except BaseException as exc:
63
+ session.record_llm_failure(request, exc)
64
+ raise
65
+ finally:
66
+ session.pop_scope()
67
+ session.record_llm_response(request, output, model=model)
68
+ return output