stepfork 0.1.0a2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stepfork/__init__.py +32 -0
- stepfork/__main__.py +6 -0
- stepfork/cli/__init__.py +1 -0
- stepfork/cli/diff.py +156 -0
- stepfork/cli/export.py +210 -0
- stepfork/cli/inspect.py +155 -0
- stepfork/cli/main.py +54 -0
- stepfork/cli/replay.py +221 -0
- stepfork/cli/validate.py +160 -0
- stepfork/diff/__init__.py +25 -0
- stepfork/diff/compare.py +144 -0
- stepfork/diff/engine.py +267 -0
- stepfork/diff/models.py +63 -0
- stepfork/export/__init__.py +20 -0
- stepfork/export/entrypoint.py +58 -0
- stepfork/export/generator.py +178 -0
- stepfork/export/runtime.py +101 -0
- stepfork/fork/__init__.py +4 -0
- stepfork/inspect/__init__.py +11 -0
- stepfork/inspect/inspector.py +143 -0
- stepfork/inspect/models.py +42 -0
- stepfork/minimize/__init__.py +4 -0
- stepfork/py.typed +0 -0
- stepfork/recorder/__init__.py +21 -0
- stepfork/recorder/llm.py +68 -0
- stepfork/recorder/session.py +500 -0
- stepfork/recorder/tooling.py +208 -0
- stepfork/replay/__init__.py +32 -0
- stepfork/replay/exceptions.py +37 -0
- stepfork/replay/plan.py +282 -0
- stepfork/replay/session.py +338 -0
- stepfork/trace/__init__.py +115 -0
- stepfork/trace/canonical.py +55 -0
- stepfork/trace/hashing.py +75 -0
- stepfork/trace/integrity.py +196 -0
- stepfork/trace/jsonable.py +97 -0
- stepfork/trace/manifest.py +85 -0
- stepfork/trace/models.py +271 -0
- stepfork/trace/redaction.py +294 -0
- stepfork/trace/replay_policy.py +18 -0
- stepfork/trace/schema.py +25 -0
- stepfork/trace/storage.py +250 -0
- stepfork/trace/validation.py +285 -0
- stepfork/version.py +5 -0
- stepfork-0.1.0a2.dist-info/METADATA +377 -0
- stepfork-0.1.0a2.dist-info/RECORD +49 -0
- stepfork-0.1.0a2.dist-info/WHEEL +4 -0
- stepfork-0.1.0a2.dist-info/entry_points.txt +2 -0
- stepfork-0.1.0a2.dist-info/licenses/LICENSE +184 -0
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"""Build sanitized inspection summaries from loaded traces."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from stepfork.inspect.models import EventSummary, TraceInspection
|
|
9
|
+
from stepfork.trace import (
|
|
10
|
+
ErrorEvent,
|
|
11
|
+
Event,
|
|
12
|
+
LLMRequest,
|
|
13
|
+
LLMResponse,
|
|
14
|
+
RunEnd,
|
|
15
|
+
RunStart,
|
|
16
|
+
StateChange,
|
|
17
|
+
ToolCall,
|
|
18
|
+
ToolResult,
|
|
19
|
+
Trace,
|
|
20
|
+
TraceStorageError,
|
|
21
|
+
)
|
|
22
|
+
from stepfork.trace.integrity import verify_bundle_integrity
|
|
23
|
+
from stepfork.trace.redaction import redact_json
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def inspect_bundle(path: str | Path) -> TraceInspection:
|
|
27
|
+
"""Load and inspect a `.sftrace` bundle."""
|
|
28
|
+
bundle_path = Path(path)
|
|
29
|
+
trace = Trace.load(bundle_path)
|
|
30
|
+
integrity = verify_bundle_integrity(bundle_path).status.value
|
|
31
|
+
return inspect_trace(trace, path=str(bundle_path), integrity=integrity)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def inspect_trace(
|
|
35
|
+
trace: Trace,
|
|
36
|
+
*,
|
|
37
|
+
path: str | None = None,
|
|
38
|
+
integrity: str = "not_checked",
|
|
39
|
+
) -> TraceInspection:
|
|
40
|
+
"""Return a sanitized inspection summary for a trace."""
|
|
41
|
+
return TraceInspection(
|
|
42
|
+
path=path,
|
|
43
|
+
agent_name=_sanitize_text(trace.agent_name),
|
|
44
|
+
run_id=trace.run_id,
|
|
45
|
+
schema_version="0.1",
|
|
46
|
+
status=trace.status.value,
|
|
47
|
+
created_at=trace.created_at.isoformat(),
|
|
48
|
+
events=len(trace.events),
|
|
49
|
+
llm_calls=trace.totals.llm_calls,
|
|
50
|
+
tool_calls=trace.totals.tool_calls,
|
|
51
|
+
duration_ms=trace.totals.duration_ms,
|
|
52
|
+
failure=_sanitize_failure(trace.failure.model_dump(mode="json"))
|
|
53
|
+
if trace.failure
|
|
54
|
+
else None,
|
|
55
|
+
integrity=integrity,
|
|
56
|
+
timeline=[_event_summary(event) for event in trace.events],
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def filter_timeline(
|
|
61
|
+
inspection: TraceInspection,
|
|
62
|
+
*,
|
|
63
|
+
errors_only: bool = False,
|
|
64
|
+
step: int | None = None,
|
|
65
|
+
) -> TraceInspection:
|
|
66
|
+
"""Return an inspection copy with filtered timeline events."""
|
|
67
|
+
timeline = inspection.timeline
|
|
68
|
+
if errors_only:
|
|
69
|
+
timeline = [
|
|
70
|
+
event
|
|
71
|
+
for event in timeline
|
|
72
|
+
if event.status == "error" or event.type == "error"
|
|
73
|
+
]
|
|
74
|
+
if step is not None:
|
|
75
|
+
timeline = [event for event in timeline if event.step == step]
|
|
76
|
+
return inspection.model_copy(update={"timeline": timeline})
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _event_summary(event: Event) -> EventSummary:
|
|
80
|
+
payload = _sanitize_json(event.model_dump(mode="json"))
|
|
81
|
+
return EventSummary(
|
|
82
|
+
id=event.id,
|
|
83
|
+
step=event.step,
|
|
84
|
+
type=event.type.value,
|
|
85
|
+
label=_event_label(event),
|
|
86
|
+
status=event.status.value,
|
|
87
|
+
parent_id=event.parent_id,
|
|
88
|
+
duration_ms=event.duration_ms,
|
|
89
|
+
details=_event_details(payload),
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _event_label(event: Event) -> str:
|
|
94
|
+
if isinstance(event, RunStart):
|
|
95
|
+
return "run_start"
|
|
96
|
+
if isinstance(event, LLMRequest):
|
|
97
|
+
return event.model
|
|
98
|
+
if isinstance(event, LLMResponse):
|
|
99
|
+
return event.model or "llm_response"
|
|
100
|
+
if isinstance(event, ToolCall | ToolResult):
|
|
101
|
+
return event.name
|
|
102
|
+
if isinstance(event, StateChange):
|
|
103
|
+
return event.key or "state_change"
|
|
104
|
+
if isinstance(event, ErrorEvent):
|
|
105
|
+
return _sanitize_text(event.error_type)
|
|
106
|
+
if isinstance(event, RunEnd):
|
|
107
|
+
return event.run_status.value
|
|
108
|
+
return event.type.value
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _event_details(payload: dict[str, Any]) -> dict[str, Any]:
|
|
112
|
+
hidden = {
|
|
113
|
+
"id",
|
|
114
|
+
"run_id",
|
|
115
|
+
"parent_id",
|
|
116
|
+
"step",
|
|
117
|
+
"type",
|
|
118
|
+
"timestamp",
|
|
119
|
+
"status",
|
|
120
|
+
"replay_policy",
|
|
121
|
+
"input_hash",
|
|
122
|
+
"output_hash",
|
|
123
|
+
"redactions",
|
|
124
|
+
}
|
|
125
|
+
return {key: value for key, value in payload.items() if key not in hidden}
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _sanitize_failure(value: dict[str, Any]) -> dict[str, Any]:
|
|
129
|
+
sanitized = _sanitize_json(value)
|
|
130
|
+
if not isinstance(sanitized, dict):
|
|
131
|
+
raise TraceStorageError("inspection failure sanitizer returned non-object")
|
|
132
|
+
return sanitized
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _sanitize_json(value: Any) -> Any:
|
|
136
|
+
return redact_json(value).value
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _sanitize_text(value: str) -> str:
|
|
140
|
+
sanitized = redact_json(value).value
|
|
141
|
+
if not isinstance(sanitized, str):
|
|
142
|
+
raise TraceStorageError("inspection text sanitizer returned non-string")
|
|
143
|
+
return sanitized
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""Structured models for trace inspection output."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from pydantic import BaseModel, ConfigDict
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class EventSummary(BaseModel):
|
|
11
|
+
"""Concise event data for timeline display."""
|
|
12
|
+
|
|
13
|
+
model_config = ConfigDict(extra="forbid")
|
|
14
|
+
|
|
15
|
+
id: str
|
|
16
|
+
step: int
|
|
17
|
+
type: str
|
|
18
|
+
label: str
|
|
19
|
+
status: str
|
|
20
|
+
parent_id: str | None = None
|
|
21
|
+
duration_ms: int | None = None
|
|
22
|
+
details: dict[str, Any]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class TraceInspection(BaseModel):
|
|
26
|
+
"""Sanitized trace inspection result."""
|
|
27
|
+
|
|
28
|
+
model_config = ConfigDict(extra="forbid")
|
|
29
|
+
|
|
30
|
+
path: str | None = None
|
|
31
|
+
agent_name: str
|
|
32
|
+
run_id: str
|
|
33
|
+
schema_version: str
|
|
34
|
+
status: str
|
|
35
|
+
created_at: str
|
|
36
|
+
events: int
|
|
37
|
+
llm_calls: int
|
|
38
|
+
tool_calls: int
|
|
39
|
+
duration_ms: int | None = None
|
|
40
|
+
failure: dict[str, Any] | None = None
|
|
41
|
+
integrity: str
|
|
42
|
+
timeline: list[EventSummary]
|
stepfork/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""Agent execution recording primitives."""
|
|
2
|
+
|
|
3
|
+
from stepfork.recorder.llm import llm_request
|
|
4
|
+
from stepfork.recorder.session import (
|
|
5
|
+
DEFAULT_TRACE_DIR,
|
|
6
|
+
LLMCallHandle,
|
|
7
|
+
RecordingSession,
|
|
8
|
+
active_recorder,
|
|
9
|
+
record,
|
|
10
|
+
)
|
|
11
|
+
from stepfork.recorder.tooling import trace_tool
|
|
12
|
+
|
|
13
|
+
__all__ = [
|
|
14
|
+
"DEFAULT_TRACE_DIR",
|
|
15
|
+
"LLMCallHandle",
|
|
16
|
+
"RecordingSession",
|
|
17
|
+
"active_recorder",
|
|
18
|
+
"llm_request",
|
|
19
|
+
"record",
|
|
20
|
+
"trace_tool",
|
|
21
|
+
]
|
stepfork/recorder/llm.py
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Provider-independent LLM boundary recording and replay."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from stepfork.recorder.session import active_recorder
|
|
9
|
+
from stepfork.replay.session import active_replay
|
|
10
|
+
from stepfork.trace import ReplayPolicy
|
|
11
|
+
from stepfork.trace.jsonable import Serializer, to_json_value
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def llm_request(
|
|
15
|
+
*,
|
|
16
|
+
model: str,
|
|
17
|
+
input: Any,
|
|
18
|
+
call: Callable[[], Any],
|
|
19
|
+
provider: str | None = None,
|
|
20
|
+
replay_policy: ReplayPolicy = ReplayPolicy.FROZEN,
|
|
21
|
+
serializer: Serializer | None = None,
|
|
22
|
+
) -> Any:
|
|
23
|
+
"""Invoke an LLM dependency while recording or replaying its boundary.
|
|
24
|
+
|
|
25
|
+
``call`` is a zero-argument callable performing the real request. Its
|
|
26
|
+
behavior depends on the active Stepfork context:
|
|
27
|
+
|
|
28
|
+
- recording (``with record(...)``): the request and response are captured
|
|
29
|
+
and ``call()`` executes.
|
|
30
|
+
- frozen replay: ``call()`` is **not** executed; the recorded response is
|
|
31
|
+
returned.
|
|
32
|
+
- live replay: the call matches the recording, then ``call()`` executes.
|
|
33
|
+
- no active context: ``call()`` executes unchanged.
|
|
34
|
+
|
|
35
|
+
This is the replay-safe way for agents to wrap LLM calls.
|
|
36
|
+
"""
|
|
37
|
+
replay = active_replay()
|
|
38
|
+
if replay is not None:
|
|
39
|
+
input_json = to_json_value(input, serializer=serializer)
|
|
40
|
+
decision = replay.before_llm(
|
|
41
|
+
provider=provider,
|
|
42
|
+
model=model,
|
|
43
|
+
input=input_json,
|
|
44
|
+
)
|
|
45
|
+
if not decision.execute:
|
|
46
|
+
return decision.value
|
|
47
|
+
return call()
|
|
48
|
+
|
|
49
|
+
session = active_recorder()
|
|
50
|
+
if session is None:
|
|
51
|
+
return call()
|
|
52
|
+
|
|
53
|
+
request = session.record_llm_request(
|
|
54
|
+
model=model,
|
|
55
|
+
input=input,
|
|
56
|
+
provider=provider,
|
|
57
|
+
replay_policy=replay_policy,
|
|
58
|
+
)
|
|
59
|
+
session.push_scope(request.id)
|
|
60
|
+
try:
|
|
61
|
+
output = call()
|
|
62
|
+
except BaseException as exc:
|
|
63
|
+
session.record_llm_failure(request, exc)
|
|
64
|
+
raise
|
|
65
|
+
finally:
|
|
66
|
+
session.pop_scope()
|
|
67
|
+
session.record_llm_response(request, output, model=model)
|
|
68
|
+
return output
|