durable-agents 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. durable_agents/__init__.py +126 -0
  2. durable_agents/api/__init__.py +0 -0
  3. durable_agents/api/app.py +226 -0
  4. durable_agents/cli.py +187 -0
  5. durable_agents/events.py +206 -0
  6. durable_agents/guardrails/__init__.py +0 -0
  7. durable_agents/guardrails/decisions.py +223 -0
  8. durable_agents/guardrails/input_scan.py +24 -0
  9. durable_agents/guardrails/output_validate.py +75 -0
  10. durable_agents/guardrails/patterns.py +199 -0
  11. durable_agents/guardrails/run_level.py +62 -0
  12. durable_agents/guardrails/tool_result_scan.py +37 -0
  13. durable_agents/guardrails/types.py +26 -0
  14. durable_agents/llm/__init__.py +0 -0
  15. durable_agents/llm/openai_compatible.py +193 -0
  16. durable_agents/llm/protocol.py +44 -0
  17. durable_agents/llm/scripted.py +41 -0
  18. durable_agents/orchestrator.py +894 -0
  19. durable_agents/py.typed +0 -0
  20. durable_agents/replay_view.py +309 -0
  21. durable_agents/runtime.py +209 -0
  22. durable_agents/state.py +341 -0
  23. durable_agents/storage/__init__.py +0 -0
  24. durable_agents/storage/memory.py +101 -0
  25. durable_agents/storage/postgres.py +143 -0
  26. durable_agents/storage/protocol.py +91 -0
  27. durable_agents/storage/schema.py +31 -0
  28. durable_agents/storage/schema.sql +27 -0
  29. durable_agents/tools/__init__.py +0 -0
  30. durable_agents/tools/registry.py +151 -0
  31. durable_agents/worker.py +105 -0
  32. durable_agents-0.1.0.dist-info/METADATA +273 -0
  33. durable_agents-0.1.0.dist-info/RECORD +36 -0
  34. durable_agents-0.1.0.dist-info/WHEEL +4 -0
  35. durable_agents-0.1.0.dist-info/entry_points.txt +2 -0
  36. durable_agents-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,126 @@
1
+ """An event-sourced runtime for durable, crash-resumable LLM agents.
2
+
3
+ Everything an agent does — every model call, every tool call, every
4
+ guardrail decision, every human approval — is appended to a log before
5
+ and after it happens. State is a pure fold over that log, so a process
6
+ that dies mid-run can be resumed by a different process that reads the
7
+ log and finishes the job, without repeating side effects that already
8
+ happened.
9
+
10
+ from durable_agents import Runtime, InMemoryEventStore, tool
11
+
12
+ @tool(side_effect=True)
13
+ async def issue_refund(order_id: str, amount: int, idempotency_key: str) -> dict:
14
+ return await payments.refund(order_id, amount, key=idempotency_key)
15
+
16
+ runtime = Runtime(store=InMemoryEventStore(), llm=my_client, tools=[issue_refund])
17
+ state = await runtime.start(goal="Refund order A-8891, item arrived damaged.")
18
+
19
+ Swap InMemoryEventStore for PostgresEventStore when you want the run to
20
+ outlive the process. See README.md for the full walkthrough.
21
+ """
22
+
23
+ from durable_agents.events import (
24
+ ApprovalDenied,
25
+ ApprovalGranted,
26
+ ApprovalRequested,
27
+ Event,
28
+ GuardrailAction,
29
+ GuardrailLayer,
30
+ GuardrailTriggered,
31
+ LLMCallCompleted,
32
+ LLMCallFailed,
33
+ LLMCallRequested,
34
+ RunCompleted,
35
+ RunFailed,
36
+ RunFailureReason,
37
+ RunStarted,
38
+ ToolCallCompleted,
39
+ ToolCallFailed,
40
+ ToolCallInvocation,
41
+ ToolCallRequested,
42
+ hash_system_prompt,
43
+ )
44
+ from durable_agents.guardrails.decisions import PROFILES, GuardrailProfile, decide, get_profile
45
+ from durable_agents.guardrails.types import GuardMatch, ScanResult
46
+ from durable_agents.llm.protocol import LLMClient, LLMResponse
47
+ from durable_agents.llm.scripted import ScriptedLLM
48
+ from durable_agents.orchestrator import Orchestrator
49
+ from durable_agents.runtime import Run, Runtime
50
+ from durable_agents.state import (
51
+ GuardrailHit,
52
+ InFlightOp,
53
+ Message,
54
+ PendingApproval,
55
+ RunState,
56
+ RunStatus,
57
+ rebuild_state,
58
+ )
59
+ from durable_agents.storage.memory import InMemoryEventStore
60
+ from durable_agents.storage.postgres import PostgresEventStore
61
+ from durable_agents.storage.protocol import ConcurrencyConflict, EventStore
62
+ from durable_agents.storage.schema import create_schema, schema_sql
63
+ from durable_agents.tools.registry import Tool, idempotency_key, tool
64
+ from durable_agents.worker import Worker
65
+
66
+ __version__ = "0.1.0"
67
+
68
+ __all__ = [
69
+ # The five names most users need
70
+ "Runtime",
71
+ "Run",
72
+ "Worker",
73
+ "tool",
74
+ "InMemoryEventStore",
75
+ "PostgresEventStore",
76
+ "create_schema",
77
+ # Implement one of these to plug in your model provider
78
+ "LLMClient",
79
+ "LLMResponse",
80
+ "ScriptedLLM",
81
+ # Storage
82
+ "EventStore",
83
+ "ConcurrencyConflict",
84
+ "schema_sql",
85
+ # The loop itself, for anyone who wants it without the Runtime facade
86
+ "Orchestrator",
87
+ # Derived state
88
+ "RunState",
89
+ "RunStatus",
90
+ "Message",
91
+ "InFlightOp",
92
+ "PendingApproval",
93
+ "GuardrailHit",
94
+ "rebuild_state",
95
+ # Tools
96
+ "Tool",
97
+ "idempotency_key",
98
+ # Guardrails
99
+ "GuardrailProfile",
100
+ "PROFILES",
101
+ "decide",
102
+ "get_profile",
103
+ "GuardMatch",
104
+ "ScanResult",
105
+ # Events — the actual source of truth, worth having close to hand
106
+ "Event",
107
+ "RunStarted",
108
+ "RunCompleted",
109
+ "RunFailed",
110
+ "RunFailureReason",
111
+ "LLMCallRequested",
112
+ "LLMCallCompleted",
113
+ "LLMCallFailed",
114
+ "ToolCallInvocation",
115
+ "ToolCallRequested",
116
+ "ToolCallCompleted",
117
+ "ToolCallFailed",
118
+ "ApprovalRequested",
119
+ "ApprovalGranted",
120
+ "ApprovalDenied",
121
+ "GuardrailTriggered",
122
+ "GuardrailLayer",
123
+ "GuardrailAction",
124
+ "hash_system_prompt",
125
+ "__version__",
126
+ ]
File without changes
@@ -0,0 +1,226 @@
1
+ from datetime import datetime, timezone
2
+ from decimal import Decimal
3
+ from typing import Any
4
+ from uuid import UUID, uuid4
5
+
6
+ from fastapi import Depends, FastAPI, HTTPException, Query, Request
7
+ from pydantic import BaseModel, Field, field_validator
8
+
9
+ from durable_agents.events import ApprovalDenied, ApprovalGranted, RunStarted
10
+ from durable_agents.guardrails.decisions import get_profile
11
+ from durable_agents.state import RunState, RunStatus, rebuild_state
12
+ from durable_agents.storage.protocol import ConcurrencyConflict, EventStore
13
+
14
+
15
+ class PendingApprovalResponse(BaseModel):
16
+ tool: str
17
+ arguments: dict[str, Any]
18
+ reason: str
19
+
20
+
21
+ class PendingApprovalListItem(BaseModel):
22
+ run_id: UUID
23
+ tool: str
24
+ arguments: dict[str, Any]
25
+ reason: str
26
+
27
+
28
+ class RunStatusResponse(BaseModel):
29
+ run_id: UUID
30
+ status: RunStatus
31
+ step: int
32
+ total_tokens: int
33
+ total_cost_usd: Decimal
34
+ pending_approval: PendingApprovalResponse | None
35
+ final_answer: str | None
36
+ failure_reason: str | None
37
+
38
+
39
+ class StartRunRequest(BaseModel):
40
+ """Validated here rather than at execution time, because a run is
41
+ recorded into an append-only log: a value accepted now can never be
42
+ corrected, and the run it produces fails on its first execution with
43
+ a reason ("max_steps_exceeded") describing a cap that was never
44
+ reachable. A 422 at submission is far kinder than a 201 followed by
45
+ a corpse.
46
+ """
47
+
48
+ goal: str = Field(min_length=1, max_length=100_000)
49
+ requested_by: str = Field(default="unknown", max_length=500)
50
+ system_prompt: str = Field(default="", max_length=100_000)
51
+ model: str | None = Field(default=None, max_length=200)
52
+ max_steps: int | None = Field(default=None, ge=1, le=10_000)
53
+ max_cost_usd: Decimal | None = Field(default=None, gt=0, le=Decimal("1000000"))
54
+ guardrail_profile: str | None = None
55
+
56
+ @field_validator("guardrail_profile")
57
+ @classmethod
58
+ def _known_profile(cls, value: str | None) -> str | None:
59
+ # get_profile raises on an unknown name, but only when the run is
60
+ # executed — which for an API-created run is inside a worker,
61
+ # where it fails on every poll forever, logging a stack trace
62
+ # each time and never reaching a terminal state.
63
+ if value is not None:
64
+ get_profile(value)
65
+ return value
66
+
67
+
68
+ class ApproveRequest(BaseModel):
69
+ approver: str
70
+
71
+
72
+ class DenyRequest(BaseModel):
73
+ approver: str
74
+ reason: str
75
+
76
+
77
+ def get_store(request: Request) -> EventStore:
78
+ return request.app.state.store # type: ignore[no-any-return]
79
+
80
+
81
+ async def _load_state(run_id: UUID, store: EventStore) -> tuple[RunState, int]:
82
+ """Returns the rebuilt state plus the event count (== next expected seq)."""
83
+ events = await store.read(run_id)
84
+ if not events:
85
+ raise HTTPException(status_code=404, detail="run not found")
86
+ return rebuild_state(events), len(events)
87
+
88
+
89
+ def _status_response(run_id: UUID, state: RunState) -> RunStatusResponse:
90
+ return RunStatusResponse(
91
+ run_id=run_id,
92
+ status=state.status,
93
+ step=state.step,
94
+ total_tokens=state.total_tokens,
95
+ total_cost_usd=state.total_cost_usd,
96
+ pending_approval=(
97
+ PendingApprovalResponse(
98
+ tool=state.pending_approval.tool,
99
+ arguments=state.pending_approval.arguments,
100
+ reason=state.pending_approval.reason,
101
+ )
102
+ if state.pending_approval is not None
103
+ else None
104
+ ),
105
+ final_answer=state.final_answer,
106
+ failure_reason=state.failure_reason,
107
+ )
108
+
109
+
110
+ def create_app(
111
+ store: EventStore,
112
+ *,
113
+ default_model: str = "unspecified",
114
+ default_max_steps: int = 25,
115
+ default_max_cost_usd: Decimal = Decimal("1.00"),
116
+ default_guardrail_profile: str = "validation",
117
+ ) -> FastAPI:
118
+ app = FastAPI(title="durable-agents")
119
+ app.state.store = store
120
+
121
+ @app.post("/runs", status_code=201, response_model=RunStatusResponse)
122
+ async def start_run(
123
+ body: StartRunRequest, store: EventStore = Depends(get_store)
124
+ ) -> RunStatusResponse:
125
+ # Records only — does not execute. An agent run can take minutes;
126
+ # blocking an HTTP request for that is fragile against timeouts,
127
+ # load balancers, and retries, and matches how approve/deny
128
+ # already work here (they record a decision, a separate process
129
+ # does the resuming). This mirrors Runtime.create(), not
130
+ # Runtime.start() — see DECISIONS.md's "API surface" section for
131
+ # the reasoning.
132
+ run_id = uuid4()
133
+ started = RunStarted(
134
+ seq=0,
135
+ created_at=datetime.now(timezone.utc),
136
+ goal=body.goal,
137
+ model=body.model or default_model,
138
+ system_prompt=body.system_prompt,
139
+ max_steps=body.max_steps if body.max_steps is not None else default_max_steps,
140
+ max_cost_usd=(
141
+ body.max_cost_usd if body.max_cost_usd is not None else default_max_cost_usd
142
+ ),
143
+ requested_by=body.requested_by,
144
+ guardrail_profile=body.guardrail_profile or default_guardrail_profile,
145
+ )
146
+ await store.append(run_id, 0, started)
147
+ return _status_response(run_id, rebuild_state([started]))
148
+
149
+ @app.get("/approvals", response_model=list[PendingApprovalListItem])
150
+ async def list_pending_approvals(
151
+ limit: int = Query(default=100, ge=1, le=1000),
152
+ store: EventStore = Depends(get_store),
153
+ ) -> list[PendingApprovalListItem]:
154
+ # An approver's dashboard needs to discover what needs a
155
+ # decision without already knowing a run_id for each one — that
156
+ # is the whole reason this exists. A dedicated resource rather
157
+ # than a filtered /runs, since this queue is not "runs,
158
+ # restricted somehow" — it's its own thing, and /runs?status=...
159
+ # would misleadingly read as a general run-lister that 400s on
160
+ # every value but one.
161
+ pending = await store.find_awaiting_approval(limit=limit)
162
+ return [
163
+ PendingApprovalListItem(
164
+ run_id=run_id,
165
+ tool=event.tool,
166
+ arguments=event.arguments,
167
+ reason=event.reason,
168
+ )
169
+ for run_id, event in pending
170
+ ]
171
+
172
+ @app.get("/runs/{run_id}", response_model=RunStatusResponse)
173
+ async def get_run_status(
174
+ run_id: UUID, store: EventStore = Depends(get_store)
175
+ ) -> RunStatusResponse:
176
+ state, _next_seq = await _load_state(run_id, store)
177
+ return _status_response(run_id, state)
178
+
179
+ @app.post("/runs/{run_id}/approve", status_code=204)
180
+ async def approve_run(
181
+ run_id: UUID, body: ApproveRequest, store: EventStore = Depends(get_store)
182
+ ) -> None:
183
+ state, next_seq = await _load_state(run_id, store)
184
+ if state.status != "awaiting_approval":
185
+ raise HTTPException(
186
+ status_code=409, detail=f"run is not awaiting approval (status={state.status})"
187
+ )
188
+ try:
189
+ await store.append(
190
+ run_id,
191
+ next_seq,
192
+ ApprovalGranted(
193
+ seq=next_seq, created_at=datetime.now(timezone.utc), approver=body.approver
194
+ ),
195
+ )
196
+ except ConcurrencyConflict as exc:
197
+ raise HTTPException(
198
+ status_code=409, detail="run state changed concurrently, retry"
199
+ ) from exc
200
+
201
+ @app.post("/runs/{run_id}/deny", status_code=204)
202
+ async def deny_run(
203
+ run_id: UUID, body: DenyRequest, store: EventStore = Depends(get_store)
204
+ ) -> None:
205
+ state, next_seq = await _load_state(run_id, store)
206
+ if state.status != "awaiting_approval":
207
+ raise HTTPException(
208
+ status_code=409, detail=f"run is not awaiting approval (status={state.status})"
209
+ )
210
+ try:
211
+ await store.append(
212
+ run_id,
213
+ next_seq,
214
+ ApprovalDenied(
215
+ seq=next_seq,
216
+ created_at=datetime.now(timezone.utc),
217
+ approver=body.approver,
218
+ reason=body.reason,
219
+ ),
220
+ )
221
+ except ConcurrencyConflict as exc:
222
+ raise HTTPException(
223
+ status_code=409, detail="run state changed concurrently, retry"
224
+ ) from exc
225
+
226
+ return app
durable_agents/cli.py ADDED
@@ -0,0 +1,187 @@
1
+ import argparse
2
+ import asyncio
3
+ import os
4
+ import sys
5
+ from urllib.parse import urlsplit, urlunsplit
6
+ from uuid import UUID
7
+
8
+ from durable_agents.orchestrator import Orchestrator
9
+ from durable_agents.replay_view import Style, render, should_use_colour
10
+ from durable_agents.state import rebuild_state
11
+ from durable_agents.storage.postgres import PostgresEventStore
12
+ from durable_agents.storage.schema import create_schema
13
+
14
+ # OpenAICompatibleClient is imported inside the one subcommand that uses
15
+ # it, not here. It pulls in httpx, which lives in the optional "openai"
16
+ # extra — importing it at module scope meant `pip install
17
+ # durable-agents` followed by `durable-agents --help` died with
18
+ # ModuleNotFoundError, i.e. the entry point every doc points at was
19
+ # broken on a plain install.
20
+ #
21
+ # Every subcommand here is generic. The fixed refund demo used to live
22
+ # alongside them as `durable-agents demo`; it is now
23
+ # examples/crash_resume_demo.py, since a library's console script has no
24
+ # business shipping a scripted demo of someone else's domain — and it
25
+ # could not have kept working anyway once the refund modules stopped
26
+ # being part of the package.
27
+
28
+ DEFAULT_DSN = "postgresql://durable_agents:durable_agents@localhost:5432/durable_agents"
29
+
30
+
31
+ def redact_dsn(dsn: str) -> str:
32
+ """A connection string with the password starred out, safe to print.
33
+
34
+ Which database was touched is genuinely useful to see; the password
35
+ sitting next to it is not, and stdout is exactly where credentials
36
+ escape — terminal scrollback, CI logs, screen shares. An
37
+ unparseable string is reported as-is minus everything before the
38
+ "@", since guessing at its structure risks leaking the very thing
39
+ this exists to hide.
40
+ """
41
+
42
+ try:
43
+ parts = urlsplit(dsn)
44
+ except ValueError:
45
+ return dsn.rsplit("@", 1)[-1]
46
+
47
+ # urlsplit does not raise on a malformed connection string — given
48
+ # "postgres//user:pw@host/db" (one slash missing) it happily reports
49
+ # no netloc and no password, and returning the input unchanged would
50
+ # then print the password in full. A visible "@" with nothing parsed
51
+ # around it means the structure is not what it looks like, so drop
52
+ # everything before it rather than trusting the parse.
53
+ if not parts.netloc:
54
+ return dsn.rsplit("@", 1)[-1] if "@" in dsn else dsn
55
+
56
+ if parts.password is None:
57
+ return dsn
58
+
59
+ host = parts.hostname or ""
60
+ if parts.port:
61
+ host = f"{host}:{parts.port}"
62
+ userinfo = f"{parts.username}:***@" if parts.username else "***@"
63
+ return urlunsplit((parts.scheme, f"{userinfo}{host}", parts.path, parts.query, parts.fragment))
64
+
65
+
66
+
67
+ async def _replay(run_id: UUID, dsn: str, *, colour: bool | None, show_thinking: bool) -> None:
68
+ store = await PostgresEventStore.connect(dsn)
69
+ events = await store.read(run_id)
70
+
71
+ if not events:
72
+ print(f"No events found for run {run_id}")
73
+ return
74
+
75
+ style = Style.enabled() if should_use_colour(colour) else Style()
76
+ print(render(run_id, events, rebuild_state(events), style, show_thinking))
77
+
78
+
79
+ async def _resume(run_id: UUID, dsn: str) -> None:
80
+ """Resume ANY run — including one created over the API with an
81
+ arbitrary goal — using a real LLM client.
82
+
83
+ Configured entirely through environment variables (LLM_API_KEY /
84
+ LLM_BASE_URL / LLM_MODEL), the same convention tests/live and the
85
+ examples/live_*.py scripts already use, rather than a hardcoded
86
+ provider — see DECISIONS.md's "Provider client" section for why.
87
+
88
+ Runs with NO tools: this command has no way to know what functions a
89
+ specific deployment wants wired up for a given run, so it only
90
+ handles pure conversational/reasoning goals. A run that needs real
91
+ tools needs a real script wired against Runtime/Orchestrator
92
+ directly — see README.md's "Bring your own model" section.
93
+ """
94
+
95
+ try:
96
+ from durable_agents.llm.openai_compatible import OpenAICompatibleClient
97
+ except ImportError as exc:
98
+ print(f"This command needs the 'openai' extra: pip install 'durable-agents[openai]' ({exc})")
99
+ sys.exit(1)
100
+
101
+ api_key = os.environ.get("LLM_API_KEY")
102
+ if not api_key:
103
+ print("Set LLM_API_KEY to resume a run with a real LLM (a free Groq key works).")
104
+ print("This command runs with no tools — only goals that need pure reasoning,")
105
+ print("no tool calls, will complete. For anything else, write a script against")
106
+ print("Runtime/Orchestrator directly (see README.md).")
107
+ sys.exit(1)
108
+
109
+ base_url = os.environ.get("LLM_BASE_URL", "https://api.groq.com/openai/v1")
110
+ model = os.environ.get("LLM_MODEL", "openai/gpt-oss-120b")
111
+
112
+ store = await PostgresEventStore.connect(dsn)
113
+ llm = OpenAICompatibleClient(base_url=base_url, model=model, api_key=api_key)
114
+ try:
115
+ orchestrator = Orchestrator(store=store, llm=llm, tools={})
116
+ final_state = await orchestrator.run(run_id)
117
+ finally:
118
+ await llm.aclose()
119
+
120
+ print(f"Run {run_id}: {final_state.status}")
121
+ print(f"(see the full trace with: durable-agents replay {run_id})")
122
+
123
+
124
+ def main() -> None:
125
+ # Windows consoles still default to a legacy codepage (cp1252),
126
+ # which raises UnicodeEncodeError on any non-ASCII character in a
127
+ # goal, tool result, or final answer — i.e. on most of the world's
128
+ # text. Printing a run's own recorded content must not depend on
129
+ # the operator's locale.
130
+ if hasattr(sys.stdout, "reconfigure"):
131
+ sys.stdout.reconfigure(encoding="utf-8", errors="replace")
132
+
133
+ parser = argparse.ArgumentParser(prog="durable-agents")
134
+ subparsers = parser.add_subparsers(dest="command", required=True)
135
+
136
+ replay_parser = subparsers.add_parser(
137
+ "replay", help="Print the full event trace for a run"
138
+ )
139
+ replay_parser.add_argument("run_id", type=UUID)
140
+ replay_parser.add_argument(
141
+ "--dsn", default=os.environ.get("DATABASE_URL", DEFAULT_DSN)
142
+ )
143
+ replay_parser.add_argument(
144
+ "--thinking",
145
+ action="store_true",
146
+ help="Include LLMCallRequested events (hidden by default as noise)",
147
+ )
148
+ replay_parser.add_argument(
149
+ "--no-color", action="store_true", help="Disable coloured output"
150
+ )
151
+
152
+ init_db_parser = subparsers.add_parser(
153
+ "init-db", help="Create the events table (idempotent, safe to re-run)"
154
+ )
155
+ init_db_parser.add_argument(
156
+ "--dsn", default=os.environ.get("DATABASE_URL", DEFAULT_DSN)
157
+ )
158
+
159
+ resume_parser = subparsers.add_parser(
160
+ "resume",
161
+ help="Resume ANY run with a real LLM (needs LLM_API_KEY; no tools wired up)",
162
+ )
163
+ resume_parser.add_argument("run_id", type=UUID)
164
+ resume_parser.add_argument(
165
+ "--dsn", default=os.environ.get("DATABASE_URL", DEFAULT_DSN)
166
+ )
167
+
168
+ args = parser.parse_args()
169
+
170
+ if args.command == "init-db":
171
+ asyncio.run(create_schema(args.dsn))
172
+ print(f"Schema ready on {redact_dsn(args.dsn)}")
173
+ elif args.command == "replay":
174
+ asyncio.run(
175
+ _replay(
176
+ args.run_id,
177
+ args.dsn,
178
+ colour=False if args.no_color else None,
179
+ show_thinking=args.thinking,
180
+ )
181
+ )
182
+ elif args.command == "resume":
183
+ asyncio.run(_resume(args.run_id, args.dsn))
184
+
185
+
186
+ if __name__ == "__main__":
187
+ main()