durable-agents 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- durable_agents/__init__.py +126 -0
- durable_agents/api/__init__.py +0 -0
- durable_agents/api/app.py +226 -0
- durable_agents/cli.py +187 -0
- durable_agents/events.py +206 -0
- durable_agents/guardrails/__init__.py +0 -0
- durable_agents/guardrails/decisions.py +223 -0
- durable_agents/guardrails/input_scan.py +24 -0
- durable_agents/guardrails/output_validate.py +75 -0
- durable_agents/guardrails/patterns.py +199 -0
- durable_agents/guardrails/run_level.py +62 -0
- durable_agents/guardrails/tool_result_scan.py +37 -0
- durable_agents/guardrails/types.py +26 -0
- durable_agents/llm/__init__.py +0 -0
- durable_agents/llm/openai_compatible.py +193 -0
- durable_agents/llm/protocol.py +44 -0
- durable_agents/llm/scripted.py +41 -0
- durable_agents/orchestrator.py +894 -0
- durable_agents/py.typed +0 -0
- durable_agents/replay_view.py +309 -0
- durable_agents/runtime.py +209 -0
- durable_agents/state.py +341 -0
- durable_agents/storage/__init__.py +0 -0
- durable_agents/storage/memory.py +101 -0
- durable_agents/storage/postgres.py +143 -0
- durable_agents/storage/protocol.py +91 -0
- durable_agents/storage/schema.py +31 -0
- durable_agents/storage/schema.sql +27 -0
- durable_agents/tools/__init__.py +0 -0
- durable_agents/tools/registry.py +151 -0
- durable_agents/worker.py +105 -0
- durable_agents-0.1.0.dist-info/METADATA +273 -0
- durable_agents-0.1.0.dist-info/RECORD +36 -0
- durable_agents-0.1.0.dist-info/WHEEL +4 -0
- durable_agents-0.1.0.dist-info/entry_points.txt +2 -0
- durable_agents-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""An event-sourced runtime for durable, crash-resumable LLM agents.
|
|
2
|
+
|
|
3
|
+
Everything an agent does — every model call, every tool call, every
|
|
4
|
+
guardrail decision, every human approval — is appended to a log before
|
|
5
|
+
and after it happens. State is a pure fold over that log, so a process
|
|
6
|
+
that dies mid-run can be resumed by a different process that reads the
|
|
7
|
+
log and finishes the job, without repeating side effects that already
|
|
8
|
+
happened.
|
|
9
|
+
|
|
10
|
+
from durable_agents import Runtime, InMemoryEventStore, tool
|
|
11
|
+
|
|
12
|
+
@tool(side_effect=True)
|
|
13
|
+
async def issue_refund(order_id: str, amount: int, idempotency_key: str) -> dict:
|
|
14
|
+
return await payments.refund(order_id, amount, key=idempotency_key)
|
|
15
|
+
|
|
16
|
+
runtime = Runtime(store=InMemoryEventStore(), llm=my_client, tools=[issue_refund])
|
|
17
|
+
state = await runtime.start(goal="Refund order A-8891, item arrived damaged.")
|
|
18
|
+
|
|
19
|
+
Swap InMemoryEventStore for PostgresEventStore when you want the run to
|
|
20
|
+
outlive the process. See README.md for the full walkthrough.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from durable_agents.events import (
|
|
24
|
+
ApprovalDenied,
|
|
25
|
+
ApprovalGranted,
|
|
26
|
+
ApprovalRequested,
|
|
27
|
+
Event,
|
|
28
|
+
GuardrailAction,
|
|
29
|
+
GuardrailLayer,
|
|
30
|
+
GuardrailTriggered,
|
|
31
|
+
LLMCallCompleted,
|
|
32
|
+
LLMCallFailed,
|
|
33
|
+
LLMCallRequested,
|
|
34
|
+
RunCompleted,
|
|
35
|
+
RunFailed,
|
|
36
|
+
RunFailureReason,
|
|
37
|
+
RunStarted,
|
|
38
|
+
ToolCallCompleted,
|
|
39
|
+
ToolCallFailed,
|
|
40
|
+
ToolCallInvocation,
|
|
41
|
+
ToolCallRequested,
|
|
42
|
+
hash_system_prompt,
|
|
43
|
+
)
|
|
44
|
+
from durable_agents.guardrails.decisions import PROFILES, GuardrailProfile, decide, get_profile
|
|
45
|
+
from durable_agents.guardrails.types import GuardMatch, ScanResult
|
|
46
|
+
from durable_agents.llm.protocol import LLMClient, LLMResponse
|
|
47
|
+
from durable_agents.llm.scripted import ScriptedLLM
|
|
48
|
+
from durable_agents.orchestrator import Orchestrator
|
|
49
|
+
from durable_agents.runtime import Run, Runtime
|
|
50
|
+
from durable_agents.state import (
|
|
51
|
+
GuardrailHit,
|
|
52
|
+
InFlightOp,
|
|
53
|
+
Message,
|
|
54
|
+
PendingApproval,
|
|
55
|
+
RunState,
|
|
56
|
+
RunStatus,
|
|
57
|
+
rebuild_state,
|
|
58
|
+
)
|
|
59
|
+
from durable_agents.storage.memory import InMemoryEventStore
|
|
60
|
+
from durable_agents.storage.postgres import PostgresEventStore
|
|
61
|
+
from durable_agents.storage.protocol import ConcurrencyConflict, EventStore
|
|
62
|
+
from durable_agents.storage.schema import create_schema, schema_sql
|
|
63
|
+
from durable_agents.tools.registry import Tool, idempotency_key, tool
|
|
64
|
+
from durable_agents.worker import Worker
|
|
65
|
+
|
|
66
|
+
__version__ = "0.1.0"
|
|
67
|
+
|
|
68
|
+
__all__ = [
|
|
69
|
+
# The five names most users need
|
|
70
|
+
"Runtime",
|
|
71
|
+
"Run",
|
|
72
|
+
"Worker",
|
|
73
|
+
"tool",
|
|
74
|
+
"InMemoryEventStore",
|
|
75
|
+
"PostgresEventStore",
|
|
76
|
+
"create_schema",
|
|
77
|
+
# Implement one of these to plug in your model provider
|
|
78
|
+
"LLMClient",
|
|
79
|
+
"LLMResponse",
|
|
80
|
+
"ScriptedLLM",
|
|
81
|
+
# Storage
|
|
82
|
+
"EventStore",
|
|
83
|
+
"ConcurrencyConflict",
|
|
84
|
+
"schema_sql",
|
|
85
|
+
# The loop itself, for anyone who wants it without the Runtime facade
|
|
86
|
+
"Orchestrator",
|
|
87
|
+
# Derived state
|
|
88
|
+
"RunState",
|
|
89
|
+
"RunStatus",
|
|
90
|
+
"Message",
|
|
91
|
+
"InFlightOp",
|
|
92
|
+
"PendingApproval",
|
|
93
|
+
"GuardrailHit",
|
|
94
|
+
"rebuild_state",
|
|
95
|
+
# Tools
|
|
96
|
+
"Tool",
|
|
97
|
+
"idempotency_key",
|
|
98
|
+
# Guardrails
|
|
99
|
+
"GuardrailProfile",
|
|
100
|
+
"PROFILES",
|
|
101
|
+
"decide",
|
|
102
|
+
"get_profile",
|
|
103
|
+
"GuardMatch",
|
|
104
|
+
"ScanResult",
|
|
105
|
+
# Events — the actual source of truth, worth having close to hand
|
|
106
|
+
"Event",
|
|
107
|
+
"RunStarted",
|
|
108
|
+
"RunCompleted",
|
|
109
|
+
"RunFailed",
|
|
110
|
+
"RunFailureReason",
|
|
111
|
+
"LLMCallRequested",
|
|
112
|
+
"LLMCallCompleted",
|
|
113
|
+
"LLMCallFailed",
|
|
114
|
+
"ToolCallInvocation",
|
|
115
|
+
"ToolCallRequested",
|
|
116
|
+
"ToolCallCompleted",
|
|
117
|
+
"ToolCallFailed",
|
|
118
|
+
"ApprovalRequested",
|
|
119
|
+
"ApprovalGranted",
|
|
120
|
+
"ApprovalDenied",
|
|
121
|
+
"GuardrailTriggered",
|
|
122
|
+
"GuardrailLayer",
|
|
123
|
+
"GuardrailAction",
|
|
124
|
+
"hash_system_prompt",
|
|
125
|
+
"__version__",
|
|
126
|
+
]
|
|
File without changes
|
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
from datetime import datetime, timezone
|
|
2
|
+
from decimal import Decimal
|
|
3
|
+
from typing import Any
|
|
4
|
+
from uuid import UUID, uuid4
|
|
5
|
+
|
|
6
|
+
from fastapi import Depends, FastAPI, HTTPException, Query, Request
|
|
7
|
+
from pydantic import BaseModel, Field, field_validator
|
|
8
|
+
|
|
9
|
+
from durable_agents.events import ApprovalDenied, ApprovalGranted, RunStarted
|
|
10
|
+
from durable_agents.guardrails.decisions import get_profile
|
|
11
|
+
from durable_agents.state import RunState, RunStatus, rebuild_state
|
|
12
|
+
from durable_agents.storage.protocol import ConcurrencyConflict, EventStore
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class PendingApprovalResponse(BaseModel):
|
|
16
|
+
tool: str
|
|
17
|
+
arguments: dict[str, Any]
|
|
18
|
+
reason: str
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class PendingApprovalListItem(BaseModel):
|
|
22
|
+
run_id: UUID
|
|
23
|
+
tool: str
|
|
24
|
+
arguments: dict[str, Any]
|
|
25
|
+
reason: str
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class RunStatusResponse(BaseModel):
|
|
29
|
+
run_id: UUID
|
|
30
|
+
status: RunStatus
|
|
31
|
+
step: int
|
|
32
|
+
total_tokens: int
|
|
33
|
+
total_cost_usd: Decimal
|
|
34
|
+
pending_approval: PendingApprovalResponse | None
|
|
35
|
+
final_answer: str | None
|
|
36
|
+
failure_reason: str | None
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class StartRunRequest(BaseModel):
|
|
40
|
+
"""Validated here rather than at execution time, because a run is
|
|
41
|
+
recorded into an append-only log: a value accepted now can never be
|
|
42
|
+
corrected, and the run it produces fails on its first execution with
|
|
43
|
+
a reason ("max_steps_exceeded") describing a cap that was never
|
|
44
|
+
reachable. A 422 at submission is far kinder than a 201 followed by
|
|
45
|
+
a corpse.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
goal: str = Field(min_length=1, max_length=100_000)
|
|
49
|
+
requested_by: str = Field(default="unknown", max_length=500)
|
|
50
|
+
system_prompt: str = Field(default="", max_length=100_000)
|
|
51
|
+
model: str | None = Field(default=None, max_length=200)
|
|
52
|
+
max_steps: int | None = Field(default=None, ge=1, le=10_000)
|
|
53
|
+
max_cost_usd: Decimal | None = Field(default=None, gt=0, le=Decimal("1000000"))
|
|
54
|
+
guardrail_profile: str | None = None
|
|
55
|
+
|
|
56
|
+
@field_validator("guardrail_profile")
|
|
57
|
+
@classmethod
|
|
58
|
+
def _known_profile(cls, value: str | None) -> str | None:
|
|
59
|
+
# get_profile raises on an unknown name, but only when the run is
|
|
60
|
+
# executed — which for an API-created run is inside a worker,
|
|
61
|
+
# where it fails on every poll forever, logging a stack trace
|
|
62
|
+
# each time and never reaching a terminal state.
|
|
63
|
+
if value is not None:
|
|
64
|
+
get_profile(value)
|
|
65
|
+
return value
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class ApproveRequest(BaseModel):
|
|
69
|
+
approver: str
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class DenyRequest(BaseModel):
|
|
73
|
+
approver: str
|
|
74
|
+
reason: str
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def get_store(request: Request) -> EventStore:
|
|
78
|
+
return request.app.state.store # type: ignore[no-any-return]
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
async def _load_state(run_id: UUID, store: EventStore) -> tuple[RunState, int]:
|
|
82
|
+
"""Returns the rebuilt state plus the event count (== next expected seq)."""
|
|
83
|
+
events = await store.read(run_id)
|
|
84
|
+
if not events:
|
|
85
|
+
raise HTTPException(status_code=404, detail="run not found")
|
|
86
|
+
return rebuild_state(events), len(events)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _status_response(run_id: UUID, state: RunState) -> RunStatusResponse:
|
|
90
|
+
return RunStatusResponse(
|
|
91
|
+
run_id=run_id,
|
|
92
|
+
status=state.status,
|
|
93
|
+
step=state.step,
|
|
94
|
+
total_tokens=state.total_tokens,
|
|
95
|
+
total_cost_usd=state.total_cost_usd,
|
|
96
|
+
pending_approval=(
|
|
97
|
+
PendingApprovalResponse(
|
|
98
|
+
tool=state.pending_approval.tool,
|
|
99
|
+
arguments=state.pending_approval.arguments,
|
|
100
|
+
reason=state.pending_approval.reason,
|
|
101
|
+
)
|
|
102
|
+
if state.pending_approval is not None
|
|
103
|
+
else None
|
|
104
|
+
),
|
|
105
|
+
final_answer=state.final_answer,
|
|
106
|
+
failure_reason=state.failure_reason,
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def create_app(
|
|
111
|
+
store: EventStore,
|
|
112
|
+
*,
|
|
113
|
+
default_model: str = "unspecified",
|
|
114
|
+
default_max_steps: int = 25,
|
|
115
|
+
default_max_cost_usd: Decimal = Decimal("1.00"),
|
|
116
|
+
default_guardrail_profile: str = "validation",
|
|
117
|
+
) -> FastAPI:
|
|
118
|
+
app = FastAPI(title="durable-agents")
|
|
119
|
+
app.state.store = store
|
|
120
|
+
|
|
121
|
+
@app.post("/runs", status_code=201, response_model=RunStatusResponse)
|
|
122
|
+
async def start_run(
|
|
123
|
+
body: StartRunRequest, store: EventStore = Depends(get_store)
|
|
124
|
+
) -> RunStatusResponse:
|
|
125
|
+
# Records only — does not execute. An agent run can take minutes;
|
|
126
|
+
# blocking an HTTP request for that is fragile against timeouts,
|
|
127
|
+
# load balancers, and retries, and matches how approve/deny
|
|
128
|
+
# already work here (they record a decision, a separate process
|
|
129
|
+
# does the resuming). This mirrors Runtime.create(), not
|
|
130
|
+
# Runtime.start() — see DECISIONS.md's "API surface" section for
|
|
131
|
+
# the reasoning.
|
|
132
|
+
run_id = uuid4()
|
|
133
|
+
started = RunStarted(
|
|
134
|
+
seq=0,
|
|
135
|
+
created_at=datetime.now(timezone.utc),
|
|
136
|
+
goal=body.goal,
|
|
137
|
+
model=body.model or default_model,
|
|
138
|
+
system_prompt=body.system_prompt,
|
|
139
|
+
max_steps=body.max_steps if body.max_steps is not None else default_max_steps,
|
|
140
|
+
max_cost_usd=(
|
|
141
|
+
body.max_cost_usd if body.max_cost_usd is not None else default_max_cost_usd
|
|
142
|
+
),
|
|
143
|
+
requested_by=body.requested_by,
|
|
144
|
+
guardrail_profile=body.guardrail_profile or default_guardrail_profile,
|
|
145
|
+
)
|
|
146
|
+
await store.append(run_id, 0, started)
|
|
147
|
+
return _status_response(run_id, rebuild_state([started]))
|
|
148
|
+
|
|
149
|
+
@app.get("/approvals", response_model=list[PendingApprovalListItem])
|
|
150
|
+
async def list_pending_approvals(
|
|
151
|
+
limit: int = Query(default=100, ge=1, le=1000),
|
|
152
|
+
store: EventStore = Depends(get_store),
|
|
153
|
+
) -> list[PendingApprovalListItem]:
|
|
154
|
+
# An approver's dashboard needs to discover what needs a
|
|
155
|
+
# decision without already knowing a run_id for each one — that
|
|
156
|
+
# is the whole reason this exists. A dedicated resource rather
|
|
157
|
+
# than a filtered /runs, since this queue is not "runs,
|
|
158
|
+
# restricted somehow" — it's its own thing, and /runs?status=...
|
|
159
|
+
# would misleadingly read as a general run-lister that 400s on
|
|
160
|
+
# every value but one.
|
|
161
|
+
pending = await store.find_awaiting_approval(limit=limit)
|
|
162
|
+
return [
|
|
163
|
+
PendingApprovalListItem(
|
|
164
|
+
run_id=run_id,
|
|
165
|
+
tool=event.tool,
|
|
166
|
+
arguments=event.arguments,
|
|
167
|
+
reason=event.reason,
|
|
168
|
+
)
|
|
169
|
+
for run_id, event in pending
|
|
170
|
+
]
|
|
171
|
+
|
|
172
|
+
@app.get("/runs/{run_id}", response_model=RunStatusResponse)
|
|
173
|
+
async def get_run_status(
|
|
174
|
+
run_id: UUID, store: EventStore = Depends(get_store)
|
|
175
|
+
) -> RunStatusResponse:
|
|
176
|
+
state, _next_seq = await _load_state(run_id, store)
|
|
177
|
+
return _status_response(run_id, state)
|
|
178
|
+
|
|
179
|
+
@app.post("/runs/{run_id}/approve", status_code=204)
|
|
180
|
+
async def approve_run(
|
|
181
|
+
run_id: UUID, body: ApproveRequest, store: EventStore = Depends(get_store)
|
|
182
|
+
) -> None:
|
|
183
|
+
state, next_seq = await _load_state(run_id, store)
|
|
184
|
+
if state.status != "awaiting_approval":
|
|
185
|
+
raise HTTPException(
|
|
186
|
+
status_code=409, detail=f"run is not awaiting approval (status={state.status})"
|
|
187
|
+
)
|
|
188
|
+
try:
|
|
189
|
+
await store.append(
|
|
190
|
+
run_id,
|
|
191
|
+
next_seq,
|
|
192
|
+
ApprovalGranted(
|
|
193
|
+
seq=next_seq, created_at=datetime.now(timezone.utc), approver=body.approver
|
|
194
|
+
),
|
|
195
|
+
)
|
|
196
|
+
except ConcurrencyConflict as exc:
|
|
197
|
+
raise HTTPException(
|
|
198
|
+
status_code=409, detail="run state changed concurrently, retry"
|
|
199
|
+
) from exc
|
|
200
|
+
|
|
201
|
+
@app.post("/runs/{run_id}/deny", status_code=204)
|
|
202
|
+
async def deny_run(
|
|
203
|
+
run_id: UUID, body: DenyRequest, store: EventStore = Depends(get_store)
|
|
204
|
+
) -> None:
|
|
205
|
+
state, next_seq = await _load_state(run_id, store)
|
|
206
|
+
if state.status != "awaiting_approval":
|
|
207
|
+
raise HTTPException(
|
|
208
|
+
status_code=409, detail=f"run is not awaiting approval (status={state.status})"
|
|
209
|
+
)
|
|
210
|
+
try:
|
|
211
|
+
await store.append(
|
|
212
|
+
run_id,
|
|
213
|
+
next_seq,
|
|
214
|
+
ApprovalDenied(
|
|
215
|
+
seq=next_seq,
|
|
216
|
+
created_at=datetime.now(timezone.utc),
|
|
217
|
+
approver=body.approver,
|
|
218
|
+
reason=body.reason,
|
|
219
|
+
),
|
|
220
|
+
)
|
|
221
|
+
except ConcurrencyConflict as exc:
|
|
222
|
+
raise HTTPException(
|
|
223
|
+
status_code=409, detail="run state changed concurrently, retry"
|
|
224
|
+
) from exc
|
|
225
|
+
|
|
226
|
+
return app
|
durable_agents/cli.py
ADDED
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import asyncio
|
|
3
|
+
import os
|
|
4
|
+
import sys
|
|
5
|
+
from urllib.parse import urlsplit, urlunsplit
|
|
6
|
+
from uuid import UUID
|
|
7
|
+
|
|
8
|
+
from durable_agents.orchestrator import Orchestrator
|
|
9
|
+
from durable_agents.replay_view import Style, render, should_use_colour
|
|
10
|
+
from durable_agents.state import rebuild_state
|
|
11
|
+
from durable_agents.storage.postgres import PostgresEventStore
|
|
12
|
+
from durable_agents.storage.schema import create_schema
|
|
13
|
+
|
|
14
|
+
# OpenAICompatibleClient is imported inside the one subcommand that uses
|
|
15
|
+
# it, not here. It pulls in httpx, which lives in the optional "openai"
|
|
16
|
+
# extra — importing it at module scope meant `pip install
|
|
17
|
+
# durable-agents` followed by `durable-agents --help` died with
|
|
18
|
+
# ModuleNotFoundError, i.e. the entry point every doc points at was
|
|
19
|
+
# broken on a plain install.
|
|
20
|
+
#
|
|
21
|
+
# Every subcommand here is generic. The fixed refund demo used to live
|
|
22
|
+
# alongside them as `durable-agents demo`; it is now
|
|
23
|
+
# examples/crash_resume_demo.py, since a library's console script has no
|
|
24
|
+
# business shipping a scripted demo of someone else's domain — and it
|
|
25
|
+
# could not have kept working anyway once the refund modules stopped
|
|
26
|
+
# being part of the package.
|
|
27
|
+
|
|
28
|
+
DEFAULT_DSN = "postgresql://durable_agents:durable_agents@localhost:5432/durable_agents"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def redact_dsn(dsn: str) -> str:
|
|
32
|
+
"""A connection string with the password starred out, safe to print.
|
|
33
|
+
|
|
34
|
+
Which database was touched is genuinely useful to see; the password
|
|
35
|
+
sitting next to it is not, and stdout is exactly where credentials
|
|
36
|
+
escape — terminal scrollback, CI logs, screen shares. An
|
|
37
|
+
unparseable string is reported as-is minus everything before the
|
|
38
|
+
"@", since guessing at its structure risks leaking the very thing
|
|
39
|
+
this exists to hide.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
try:
|
|
43
|
+
parts = urlsplit(dsn)
|
|
44
|
+
except ValueError:
|
|
45
|
+
return dsn.rsplit("@", 1)[-1]
|
|
46
|
+
|
|
47
|
+
# urlsplit does not raise on a malformed connection string — given
|
|
48
|
+
# "postgres//user:pw@host/db" (one slash missing) it happily reports
|
|
49
|
+
# no netloc and no password, and returning the input unchanged would
|
|
50
|
+
# then print the password in full. A visible "@" with nothing parsed
|
|
51
|
+
# around it means the structure is not what it looks like, so drop
|
|
52
|
+
# everything before it rather than trusting the parse.
|
|
53
|
+
if not parts.netloc:
|
|
54
|
+
return dsn.rsplit("@", 1)[-1] if "@" in dsn else dsn
|
|
55
|
+
|
|
56
|
+
if parts.password is None:
|
|
57
|
+
return dsn
|
|
58
|
+
|
|
59
|
+
host = parts.hostname or ""
|
|
60
|
+
if parts.port:
|
|
61
|
+
host = f"{host}:{parts.port}"
|
|
62
|
+
userinfo = f"{parts.username}:***@" if parts.username else "***@"
|
|
63
|
+
return urlunsplit((parts.scheme, f"{userinfo}{host}", parts.path, parts.query, parts.fragment))
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
async def _replay(run_id: UUID, dsn: str, *, colour: bool | None, show_thinking: bool) -> None:
|
|
68
|
+
store = await PostgresEventStore.connect(dsn)
|
|
69
|
+
events = await store.read(run_id)
|
|
70
|
+
|
|
71
|
+
if not events:
|
|
72
|
+
print(f"No events found for run {run_id}")
|
|
73
|
+
return
|
|
74
|
+
|
|
75
|
+
style = Style.enabled() if should_use_colour(colour) else Style()
|
|
76
|
+
print(render(run_id, events, rebuild_state(events), style, show_thinking))
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
async def _resume(run_id: UUID, dsn: str) -> None:
|
|
80
|
+
"""Resume ANY run — including one created over the API with an
|
|
81
|
+
arbitrary goal — using a real LLM client.
|
|
82
|
+
|
|
83
|
+
Configured entirely through environment variables (LLM_API_KEY /
|
|
84
|
+
LLM_BASE_URL / LLM_MODEL), the same convention tests/live and the
|
|
85
|
+
examples/live_*.py scripts already use, rather than a hardcoded
|
|
86
|
+
provider — see DECISIONS.md's "Provider client" section for why.
|
|
87
|
+
|
|
88
|
+
Runs with NO tools: this command has no way to know what functions a
|
|
89
|
+
specific deployment wants wired up for a given run, so it only
|
|
90
|
+
handles pure conversational/reasoning goals. A run that needs real
|
|
91
|
+
tools needs a real script wired against Runtime/Orchestrator
|
|
92
|
+
directly — see README.md's "Bring your own model" section.
|
|
93
|
+
"""
|
|
94
|
+
|
|
95
|
+
try:
|
|
96
|
+
from durable_agents.llm.openai_compatible import OpenAICompatibleClient
|
|
97
|
+
except ImportError as exc:
|
|
98
|
+
print(f"This command needs the 'openai' extra: pip install 'durable-agents[openai]' ({exc})")
|
|
99
|
+
sys.exit(1)
|
|
100
|
+
|
|
101
|
+
api_key = os.environ.get("LLM_API_KEY")
|
|
102
|
+
if not api_key:
|
|
103
|
+
print("Set LLM_API_KEY to resume a run with a real LLM (a free Groq key works).")
|
|
104
|
+
print("This command runs with no tools — only goals that need pure reasoning,")
|
|
105
|
+
print("no tool calls, will complete. For anything else, write a script against")
|
|
106
|
+
print("Runtime/Orchestrator directly (see README.md).")
|
|
107
|
+
sys.exit(1)
|
|
108
|
+
|
|
109
|
+
base_url = os.environ.get("LLM_BASE_URL", "https://api.groq.com/openai/v1")
|
|
110
|
+
model = os.environ.get("LLM_MODEL", "openai/gpt-oss-120b")
|
|
111
|
+
|
|
112
|
+
store = await PostgresEventStore.connect(dsn)
|
|
113
|
+
llm = OpenAICompatibleClient(base_url=base_url, model=model, api_key=api_key)
|
|
114
|
+
try:
|
|
115
|
+
orchestrator = Orchestrator(store=store, llm=llm, tools={})
|
|
116
|
+
final_state = await orchestrator.run(run_id)
|
|
117
|
+
finally:
|
|
118
|
+
await llm.aclose()
|
|
119
|
+
|
|
120
|
+
print(f"Run {run_id}: {final_state.status}")
|
|
121
|
+
print(f"(see the full trace with: durable-agents replay {run_id})")
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def main() -> None:
|
|
125
|
+
# Windows consoles still default to a legacy codepage (cp1252),
|
|
126
|
+
# which raises UnicodeEncodeError on any non-ASCII character in a
|
|
127
|
+
# goal, tool result, or final answer — i.e. on most of the world's
|
|
128
|
+
# text. Printing a run's own recorded content must not depend on
|
|
129
|
+
# the operator's locale.
|
|
130
|
+
if hasattr(sys.stdout, "reconfigure"):
|
|
131
|
+
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
|
132
|
+
|
|
133
|
+
parser = argparse.ArgumentParser(prog="durable-agents")
|
|
134
|
+
subparsers = parser.add_subparsers(dest="command", required=True)
|
|
135
|
+
|
|
136
|
+
replay_parser = subparsers.add_parser(
|
|
137
|
+
"replay", help="Print the full event trace for a run"
|
|
138
|
+
)
|
|
139
|
+
replay_parser.add_argument("run_id", type=UUID)
|
|
140
|
+
replay_parser.add_argument(
|
|
141
|
+
"--dsn", default=os.environ.get("DATABASE_URL", DEFAULT_DSN)
|
|
142
|
+
)
|
|
143
|
+
replay_parser.add_argument(
|
|
144
|
+
"--thinking",
|
|
145
|
+
action="store_true",
|
|
146
|
+
help="Include LLMCallRequested events (hidden by default as noise)",
|
|
147
|
+
)
|
|
148
|
+
replay_parser.add_argument(
|
|
149
|
+
"--no-color", action="store_true", help="Disable coloured output"
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
init_db_parser = subparsers.add_parser(
|
|
153
|
+
"init-db", help="Create the events table (idempotent, safe to re-run)"
|
|
154
|
+
)
|
|
155
|
+
init_db_parser.add_argument(
|
|
156
|
+
"--dsn", default=os.environ.get("DATABASE_URL", DEFAULT_DSN)
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
resume_parser = subparsers.add_parser(
|
|
160
|
+
"resume",
|
|
161
|
+
help="Resume ANY run with a real LLM (needs LLM_API_KEY; no tools wired up)",
|
|
162
|
+
)
|
|
163
|
+
resume_parser.add_argument("run_id", type=UUID)
|
|
164
|
+
resume_parser.add_argument(
|
|
165
|
+
"--dsn", default=os.environ.get("DATABASE_URL", DEFAULT_DSN)
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
args = parser.parse_args()
|
|
169
|
+
|
|
170
|
+
if args.command == "init-db":
|
|
171
|
+
asyncio.run(create_schema(args.dsn))
|
|
172
|
+
print(f"Schema ready on {redact_dsn(args.dsn)}")
|
|
173
|
+
elif args.command == "replay":
|
|
174
|
+
asyncio.run(
|
|
175
|
+
_replay(
|
|
176
|
+
args.run_id,
|
|
177
|
+
args.dsn,
|
|
178
|
+
colour=False if args.no_color else None,
|
|
179
|
+
show_thinking=args.thinking,
|
|
180
|
+
)
|
|
181
|
+
)
|
|
182
|
+
elif args.command == "resume":
|
|
183
|
+
asyncio.run(_resume(args.run_id, args.dsn))
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
if __name__ == "__main__":
|
|
187
|
+
main()
|