agent-workflow-sdk 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_workflow_sdk-0.1.0.dist-info/METADATA +566 -0
- agent_workflow_sdk-0.1.0.dist-info/RECORD +52 -0
- agent_workflow_sdk-0.1.0.dist-info/WHEEL +4 -0
- agent_workflow_sdk-0.1.0.dist-info/licenses/LICENSE +201 -0
- agent_workflow_sdk-0.1.0.dist-info/licenses/NOTICE +10 -0
- agentflow/__init__.py +199 -0
- agentflow/backends/__init__.py +54 -0
- agentflow/backends/_http.py +169 -0
- agentflow/backends/anthropic.py +291 -0
- agentflow/backends/base.py +327 -0
- agentflow/backends/claude_code.py +181 -0
- agentflow/backends/cli_exec.py +237 -0
- agentflow/backends/codex.py +126 -0
- agentflow/backends/kiro.py +523 -0
- agentflow/backends/ollama.py +208 -0
- agentflow/backends/openai.py +249 -0
- agentflow/checkpoint/__init__.py +51 -0
- agentflow/checkpoint/_serde.py +57 -0
- agentflow/checkpoint/base.py +125 -0
- agentflow/checkpoint/file.py +210 -0
- agentflow/checkpoint/memory.py +91 -0
- agentflow/checkpoint/postgres.py +232 -0
- agentflow/checkpoint/redis.py +206 -0
- agentflow/checkpoint/sqlite.py +261 -0
- agentflow/compiled.py +313 -0
- agentflow/controlplane/__init__.py +57 -0
- agentflow/controlplane/memory.py +181 -0
- agentflow/controlplane/postgres.py +297 -0
- agentflow/controlplane/queue.py +89 -0
- agentflow/controlplane/records.py +122 -0
- agentflow/controlplane/registry.py +70 -0
- agentflow/controlplane/worker.py +219 -0
- agentflow/errors.py +186 -0
- agentflow/events.py +226 -0
- agentflow/graph.py +328 -0
- agentflow/observability.py +126 -0
- agentflow/otel.py +124 -0
- agentflow/prebuilt/__init__.py +29 -0
- agentflow/prebuilt/loop.py +133 -0
- agentflow/prebuilt/resilience.py +120 -0
- agentflow/prebuilt/tool_loop.py +185 -0
- agentflow/prometheus.py +175 -0
- agentflow/py.typed +0 -0
- agentflow/redaction.py +99 -0
- agentflow/runtime.py +356 -0
- agentflow/state.py +234 -0
- agentflow/store/__init__.py +40 -0
- agentflow/store/_util.py +54 -0
- agentflow/store/base.py +103 -0
- agentflow/store/memory.py +109 -0
- agentflow/store/postgres.py +210 -0
- agentflow/telemetry.py +197 -0
|
@@ -0,0 +1,566 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: agent-workflow-sdk
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Async-first, LangGraph-style SDK for building agent workflows with pluggable agent and LLM backends.
|
|
5
|
+
Project-URL: Homepage, https://github.com/dankosorkin/agent-workflow-sdk
|
|
6
|
+
Project-URL: Repository, https://github.com/dankosorkin/agent-workflow-sdk
|
|
7
|
+
Project-URL: Changelog, https://github.com/dankosorkin/agent-workflow-sdk/blob/main/CHANGELOG.md
|
|
8
|
+
Project-URL: Issues, https://github.com/dankosorkin/agent-workflow-sdk/issues
|
|
9
|
+
Author: Daniel Sorkin
|
|
10
|
+
License-Expression: Apache-2.0
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
License-File: NOTICE
|
|
13
|
+
Keywords: agents,async,graph,llm,orchestration,workflow
|
|
14
|
+
Classifier: Development Status :: 4 - Beta
|
|
15
|
+
Classifier: Framework :: AsyncIO
|
|
16
|
+
Classifier: Intended Audience :: Developers
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Typing :: Typed
|
|
20
|
+
Requires-Python: >=3.11
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: fakeredis>=2.20; extra == 'dev'
|
|
23
|
+
Requires-Dist: mypy>=1.11; extra == 'dev'
|
|
24
|
+
Requires-Dist: pip-audit>=2.7; extra == 'dev'
|
|
25
|
+
Requires-Dist: prometheus-client>=0.20; extra == 'dev'
|
|
26
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
|
|
27
|
+
Requires-Dist: pytest-cov>=5; extra == 'dev'
|
|
28
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
29
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
30
|
+
Provides-Extra: ollama
|
|
31
|
+
Requires-Dist: httpx>=0.27; extra == 'ollama'
|
|
32
|
+
Provides-Extra: otel
|
|
33
|
+
Requires-Dist: opentelemetry-api>=1.20; extra == 'otel'
|
|
34
|
+
Requires-Dist: opentelemetry-sdk>=1.20; extra == 'otel'
|
|
35
|
+
Provides-Extra: postgres
|
|
36
|
+
Requires-Dist: asyncpg>=0.29; extra == 'postgres'
|
|
37
|
+
Provides-Extra: prometheus
|
|
38
|
+
Requires-Dist: prometheus-client>=0.20; extra == 'prometheus'
|
|
39
|
+
Provides-Extra: redis
|
|
40
|
+
Requires-Dist: redis>=5; extra == 'redis'
|
|
41
|
+
Description-Content-Type: text/markdown
|
|
42
|
+
|
|
43
|
+
# agent-workflow-sdk
|
|
44
|
+
|
|
45
|
+
An async-first, LangGraph-style SDK for building agent workflows from
|
|
46
|
+
composable pieces. You describe a workflow as a graph of nodes over a typed,
|
|
47
|
+
reducer-based state, and run it against a pluggable backend — a coding agent
|
|
48
|
+
(Kiro, Codex, or Claude Code) or a plain LLM (Ollama, any OpenAI-compatible
|
|
49
|
+
endpoint, or Anthropic).
|
|
50
|
+
|
|
51
|
+
The import root is `agentflow`. The distribution name is
|
|
52
|
+
`agent-workflow-sdk`.
|
|
53
|
+
|
|
54
|
+
## Why
|
|
55
|
+
|
|
56
|
+
- Build any workflow from primitives: nodes, edges, conditional routing, and
|
|
57
|
+
a shared state. Not a fixed loop, not a linear chain.
|
|
58
|
+
- Swap the backend without touching workflow code. Agents and LLMs share one
|
|
59
|
+
event vocabulary.
|
|
60
|
+
- Async everywhere: the engine, backends, checkpointing, and streaming.
|
|
61
|
+
- Durable by default: every super-step is checkpointed, runs resume after a
|
|
62
|
+
crash, and human-in-the-loop interrupts suspend and resume a run.
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
## Install
|
|
66
|
+
|
|
67
|
+
The core is dependency-free. Backends that need extra libraries are optional
|
|
68
|
+
extras.
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
pip install -e . # core only
|
|
72
|
+
pip install -e '.[ollama]' # + httpx, for the Ollama LLM backend
|
|
73
|
+
pip install -e '.[dev]' # + pytest, pytest-asyncio
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Python 3.11+ is required. The Kiro backend needs `kiro-cli` on PATH.
|
|
77
|
+
|
|
78
|
+
## Quickstart
|
|
79
|
+
|
|
80
|
+
A graph that loops until a counter reaches a target, then stops.
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
import asyncio
|
|
84
|
+
from typing import Annotated
|
|
85
|
+
from agentflow import Graph, START, END, State, add, append
|
|
86
|
+
|
|
87
|
+
class CountState(State):
|
|
88
|
+
n: Annotated[int, add] # updates are summed
|
|
89
|
+
log: Annotated[list, append] # updates are concatenated
|
|
90
|
+
|
|
91
|
+
async def tick(state, ctx):
|
|
92
|
+
return {"n": 1, "log": f"tick {state.get('n', 0) + 1}"}
|
|
93
|
+
|
|
94
|
+
def route(state):
|
|
95
|
+
return "again" if state["n"] < 5 else "done"
|
|
96
|
+
|
|
97
|
+
g = Graph(CountState)
|
|
98
|
+
g.add_node("tick", tick)
|
|
99
|
+
g.add_edge(START, "tick")
|
|
100
|
+
g.add_conditional_edges("tick", route, {"again": "tick", "done": END})
|
|
101
|
+
app = g.compile()
|
|
102
|
+
|
|
103
|
+
print(asyncio.run(app.invoke({"n": 0, "log": []})))
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
See `examples/hello_graph.py` and `examples/optimize_loop.py` for runnable
|
|
107
|
+
versions.
|
|
108
|
+
|
|
109
|
+
## Core concepts
|
|
110
|
+
|
|
111
|
+
### State and reducers
|
|
112
|
+
|
|
113
|
+
State is a `TypedDict` subclass of `State`. Each field is a channel; annotate
|
|
114
|
+
it with a reducer that folds each node's update into the current value. A
|
|
115
|
+
field without a reducer uses `last` (last-value-wins). Built-in reducers:
|
|
116
|
+
`last`, `append`, `add`, `merge`, `union`. A node returns a partial update; it
|
|
117
|
+
never mutates the state it was given.
|
|
118
|
+
|
|
119
|
+
### Nodes and edges
|
|
120
|
+
|
|
121
|
+
A node is an async callable `(state, ctx) -> update`. Edges are either static
|
|
122
|
+
(`add_edge`) or conditional (`add_conditional_edges` with a `router(state) ->
|
|
123
|
+
key`). `START` and `END` are sentinels. A conditional router may return a list
|
|
124
|
+
of keys to fan out to several nodes at once.
|
|
125
|
+
|
|
126
|
+
### Execution
|
|
127
|
+
|
|
128
|
+
A run proceeds in super-steps. All nodes on the current frontier run
|
|
129
|
+
concurrently against the same immutable state, their updates are folded
|
|
130
|
+
through the reducers in a deterministic order, and the next frontier is
|
|
131
|
+
computed from the edges. A `step_limit` guards against runaway loops.
|
|
132
|
+
|
|
133
|
+
Super-steps are all-or-nothing: if any node in a step raises, that step's
|
|
134
|
+
updates are discarded before they touch the state and no checkpoint is written
|
|
135
|
+
for it, so the run stays at the last committed checkpoint. Input to a run is
|
|
136
|
+
validated at start — an undeclared channel key is rejected with a clear error
|
|
137
|
+
rather than sitting unused in the state.
|
|
138
|
+
|
|
139
|
+
`CompiledGraph` gives you:
|
|
140
|
+
|
|
141
|
+
- `await app.invoke(input, thread=..., timeout=...)` — run to completion,
|
|
142
|
+
return state. `timeout` bounds the whole run; on expiry it raises
|
|
143
|
+
`RunTimeout` and the last checkpoint is preserved for resume.
|
|
144
|
+
- `app.stream(input, thread=...)` — async-iterate `StreamEvent`s as they occur.
|
|
145
|
+
- `await app.resume(thread, value=...)` — continue a suspended run.
|
|
146
|
+
- `await app.get_state(thread)` / `app.history(thread)` — inspect checkpoints.
|
|
147
|
+
|
|
148
|
+
For non-async callers there are blocking wrappers — `app.invoke_sync(...)`,
|
|
149
|
+
`app.resume_sync(...)`, `app.stream_sync(...)` — which run the coroutine via
|
|
150
|
+
`asyncio.run` and refuse to run inside an existing event loop.
|
|
151
|
+
|
|
152
|
+
### Subgraphs
|
|
153
|
+
|
|
154
|
+
A compiled graph composes as a node in another graph:
|
|
155
|
+
|
|
156
|
+
```python
|
|
157
|
+
parent.add_node("sub", compiled_subgraph) # channels shared by name
|
|
158
|
+
parent.add_subgraph("sub", compiled_subgraph, # or map explicitly
|
|
159
|
+
input_map={"query": "q"}, output_map={"answer": "result"})
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
A subgraph runs on an isolated checkpoint sub-thread and merges its result
|
|
163
|
+
back through the parent's reducers. Use `output_map` to route a result into a
|
|
164
|
+
distinct parent channel and avoid double-counting accumulator channels.
|
|
165
|
+
|
|
166
|
+
### Observability
|
|
167
|
+
|
|
168
|
+
Pass a `Hooks` implementation to `compile(hooks=...)` for lifecycle callbacks
|
|
169
|
+
(`on_run_start`, `on_node_start/end/error`, `on_step_end`, `on_run_end`). A
|
|
170
|
+
hook raising never breaks a run. `RunMetrics` is a ready-made `Hooks` that
|
|
171
|
+
records per-node call counts, errors, and durations:
|
|
172
|
+
|
|
173
|
+
```python
|
|
174
|
+
from agentflow import RunMetrics
|
|
175
|
+
m = RunMetrics()
|
|
176
|
+
await g.compile(hooks=m).invoke({...})
|
|
177
|
+
print(m.summary()) # {"steps": 3, "completed": True, "nodes": {...}}
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
For durable telemetry, `JsonlTelemetry` is a `Hooks` that writes one JSON line
|
|
181
|
+
per event (`run_start`, `node_start/end/error`, `event`, `step`, `run_end`) —
|
|
182
|
+
backend events surfaced via `ctx.emit` are logged too. Compose several
|
|
183
|
+
listeners with `MultiHooks`; a failing listener never breaks the run or the
|
|
184
|
+
others:
|
|
185
|
+
|
|
186
|
+
```python
|
|
187
|
+
from agentflow import MultiHooks, RunMetrics, JsonlTelemetry
|
|
188
|
+
metrics = RunMetrics()
|
|
189
|
+
telemetry = JsonlTelemetry(".runs") # dir -> one file per thread
|
|
190
|
+
app = g.compile(hooks=MultiHooks(metrics, telemetry))
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
See `examples/telemetry_demo.py` for a runnable version.
|
|
194
|
+
|
|
195
|
+
For distributed tracing, `agentflow.otel.OtelHooks` (install the `otel` extra)
|
|
196
|
+
is a `Hooks` that emits an OpenTelemetry span per run and per node:
|
|
197
|
+
|
|
198
|
+
```python
|
|
199
|
+
from agentflow.otel import OtelHooks
|
|
200
|
+
app = g.compile(hooks=OtelHooks()) # uses the global tracer/provider
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
For operators who scrape Prometheus instead of running an OTel collector,
|
|
204
|
+
`agentflow.prometheus.PrometheusHooks` (install the `prometheus` extra) is a
|
|
205
|
+
`Hooks` that records run/node/step counters, a node-duration histogram, and
|
|
206
|
+
backend-event counts. It uses a private registry by default; call
|
|
207
|
+
`exposition()` to render the text format for a `/metrics` endpoint (you own the
|
|
208
|
+
HTTP layer).
|
|
209
|
+
|
|
210
|
+
```python
|
|
211
|
+
from agentflow.prometheus import PrometheusHooks
|
|
212
|
+
|
|
213
|
+
metrics = PrometheusHooks()
|
|
214
|
+
app = g.compile(hooks=metrics)
|
|
215
|
+
# ... inside your /metrics handler:
|
|
216
|
+
body, content_type = metrics.exposition()
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
Compose several listeners with `MultiHooks(RunMetrics(), OtelHooks(),
|
|
220
|
+
PrometheusHooks())`.
|
|
221
|
+
|
|
222
|
+
## Backends
|
|
223
|
+
|
|
224
|
+
Two kinds of backend share one event stream, so a node calls either the same
|
|
225
|
+
way.
|
|
226
|
+
|
|
227
|
+
- Agent backends (`AgentBackend`) drive a session that runs its own tools and
|
|
228
|
+
asks permission. Available: `KiroBackend` (persistent ACP session over
|
|
229
|
+
`kiro-cli`), `CodexBackend` (`codex exec --json`), `ClaudeCodeBackend`
|
|
230
|
+
(`claude -p --output-format stream-json`).
|
|
231
|
+
- LLM backends (`LLMBackend`) are stateless: messages in, token stream out.
|
|
232
|
+
They never run tools themselves — a tool call is a request the graph
|
|
233
|
+
fulfils. Available: `OllamaBackend` (local `/api/chat`), `OpenAIBackend`
|
|
234
|
+
(any `/v1/chat/completions` endpoint — OpenAI, Groq, Together, vLLM, LM
|
|
235
|
+
Studio, and Ollama's own `/v1` shim), and `AnthropicBackend` (Messages API).
|
|
236
|
+
|
|
237
|
+
```python
|
|
238
|
+
from agentflow.backends.kiro import KiroBackend
|
|
239
|
+
from agentflow.backends.codex import CodexBackend
|
|
240
|
+
from agentflow.backends.claude_code import ClaudeCodeBackend
|
|
241
|
+
from agentflow.backends.ollama import OllamaBackend
|
|
242
|
+
from agentflow.backends.openai import OpenAIBackend
|
|
243
|
+
from agentflow.backends.anthropic import AnthropicBackend
|
|
244
|
+
|
|
245
|
+
agent = KiroBackend("vibe", engine="v3") # persistent session
|
|
246
|
+
codex = CodexBackend(sandbox="read-only") # one-shot per turn
|
|
247
|
+
claude = ClaudeCodeBackend(model="sonnet") # one-shot per turn
|
|
248
|
+
llm = OllamaBackend("llama3.2") # local HTTP
|
|
249
|
+
openai = OpenAIBackend("gpt-4o-mini", api_key="...") # any OpenAI-compatible API
|
|
250
|
+
anthropic = AnthropicBackend("claude-sonnet-4", api_key="...") # Messages API
|
|
251
|
+
|
|
252
|
+
await agent.start()
|
|
253
|
+
async for event in agent.prompt("summarize the repo"):
|
|
254
|
+
... # TextChunk, ToolCall, ToolResult, ..., TurnEnd
|
|
255
|
+
await agent.close()
|
|
256
|
+
```
|
|
257
|
+
|
|
258
|
+
The three agent backends have different lifecycles under one interface. Kiro
|
|
259
|
+
holds a long-lived JSON-RPC session; Codex and Claude Code run a fresh
|
|
260
|
+
subprocess per turn and carry a resumable session id between turns. Either way
|
|
261
|
+
you call `start()`, `prompt(text)`, `close()` and consume the same event
|
|
262
|
+
stream. Backends are constructed by you and passed into your nodes. The core
|
|
263
|
+
never imports a backend, so importing `agentflow` pulls in no subprocess or
|
|
264
|
+
HTTP dependency.
|
|
265
|
+
|
|
266
|
+
The HTTP LLM backends retry transient failures (connection errors, timeouts,
|
|
267
|
+
429/5xx) on the connect/initial-response phase only — never mid-stream, so a
|
|
268
|
+
partial token stream is never replayed. `Retry-After` is honored. Tune it with
|
|
269
|
+
a `RetryPolicy`:
|
|
270
|
+
|
|
271
|
+
```python
|
|
272
|
+
from agentflow.backends import RetryPolicy
|
|
273
|
+
from agentflow.backends.openai import OpenAIBackend
|
|
274
|
+
|
|
275
|
+
llm = OpenAIBackend("gpt-4o-mini", api_key="...",
|
|
276
|
+
retry=RetryPolicy(max_retries=4, backoff=0.5))
|
|
277
|
+
```
|
|
278
|
+
|
|
279
|
+
Run `python examples/agents_demo.py` to smoke every backend installed on your
|
|
280
|
+
machine (set `KIRO_AGENT` to a valid agent id, e.g. `vibe`, to include Kiro).
|
|
281
|
+
|
|
282
|
+
### Permission policies
|
|
283
|
+
|
|
284
|
+
A `PermissionPolicy` is **required** when constructing an agent backend — there
|
|
285
|
+
is no default, because auto-approving an agent's tool use is a security choice
|
|
286
|
+
the caller must make explicitly. Options: `AllowAll` (trusted local sandbox
|
|
287
|
+
only), `DenyAll`, `Interactive` (escalate to a human via an engine interrupt),
|
|
288
|
+
or `ToolAllowlist({"read_file", ...}, fallback=DenyAll())` for least-privilege
|
|
289
|
+
scoping.
|
|
290
|
+
|
|
291
|
+
```python
|
|
292
|
+
from agentflow.backends import AllowAll, ToolAllowlist, DenyAll
|
|
293
|
+
KiroBackend("vibe", permission=ToolAllowlist({"read_file"}, fallback=DenyAll()))
|
|
294
|
+
```
|
|
295
|
+
|
|
296
|
+
Capability matrix — where the policy actually applies:
|
|
297
|
+
|
|
298
|
+
| Backend | Routes tool requests through `PermissionPolicy`? |
|
|
299
|
+
| --- | --- |
|
|
300
|
+
| `KiroBackend` | Yes — ACP `session/request_permission` is resolved by the policy (and `Interactive` drives a real HITL interrupt). |
|
|
301
|
+
| `CodexBackend` | No — one-shot CLI; permission is governed by its `--sandbox` mode. The policy field is still required but does not intercept prompts. |
|
|
302
|
+
| `ClaudeCodeBackend` | No — one-shot CLI; permission is governed by CLI flags (`--permission-mode`, `--allowed-tools`). |
|
|
303
|
+
|
|
304
|
+
For the one-shot CLI agents, configure their own controls (sandbox, allowed
|
|
305
|
+
tools) — the SDK policy alone does not sandbox them.
|
|
306
|
+
|
|
307
|
+
## Checkpointing and human-in-the-loop
|
|
308
|
+
|
|
309
|
+
Pass a checkpointer to `compile` to make runs durable.
|
|
310
|
+
|
|
311
|
+
```python
|
|
312
|
+
from agentflow import FileCheckpointer
|
|
313
|
+
app = g.compile(checkpointer=FileCheckpointer(".runs"))
|
|
314
|
+
```
|
|
315
|
+
|
|
316
|
+
`MemoryCheckpointer` is for tests; `FileCheckpointer` writes one atomic JSON
|
|
317
|
+
file per super-step under `.runs/<thread>/` (single-writer / single-process);
|
|
318
|
+
`SqliteCheckpointer` stores checkpoints transactionally with atomic per-step
|
|
319
|
+
revisions; `RedisCheckpointer` (install the `redis` extra) persists to a Redis
|
|
320
|
+
server for distributed / multi-process runners; and `PostgresCheckpointer`
|
|
321
|
+
(install the `postgres` extra) persists to a Postgres table. All implement the
|
|
322
|
+
same `Checkpointer` protocol, so they are interchangeable at
|
|
323
|
+
`compile(checkpointer=...)`.
|
|
324
|
+
|
|
325
|
+
```python
|
|
326
|
+
from agentflow.checkpoint import PostgresCheckpointer
|
|
327
|
+
app = g.compile(checkpointer=PostgresCheckpointer("postgresql://localhost/app"))
|
|
328
|
+
```
|
|
329
|
+
|
|
330
|
+
For production durability prefer `PostgresCheckpointer`: a committed row
|
|
331
|
+
survives a crash by design. Redis is only crash-durable when you configure
|
|
332
|
+
AOF/RDB persistence.
|
|
333
|
+
|
|
334
|
+
### Optimistic concurrency and retention
|
|
335
|
+
|
|
336
|
+
Each checkpoint carries a `revision` (the write count for its `(thread, step)`,
|
|
337
|
+
set when you read it back). To guard a read-modify-write against a concurrent
|
|
338
|
+
resume, pass the revision you read as `if_revision`; a stale value raises
|
|
339
|
+
`CheckpointConflict` instead of silently overwriting.
|
|
340
|
+
|
|
341
|
+
```python
|
|
342
|
+
cp = await checkpointer.get(thread, step)
|
|
343
|
+
# ... resume work off cp ...
|
|
344
|
+
await checkpointer.put(new_cp, if_revision=cp.revision) # raises on conflict
|
|
345
|
+
```
|
|
346
|
+
|
|
347
|
+
Two retention methods trim old state: `delete_thread(thread)` drops a whole
|
|
348
|
+
thread, and `prune(thread, before_step=..., older_than=...)` deletes older
|
|
349
|
+
checkpoints and returns how many it removed.
|
|
350
|
+
|
|
351
|
+
A node calls `await ctx.interrupt(payload)` to suspend the run for a human.
|
|
352
|
+
The runtime writes an interrupted checkpoint and stops. Later, `await
|
|
353
|
+
app.resume(thread, value=answer)` continues the run, and the same `interrupt`
|
|
354
|
+
call returns `answer`. This works across process restarts, since the frontier
|
|
355
|
+
is persisted.
|
|
356
|
+
|
|
357
|
+
## Cross-thread memory (Store)
|
|
358
|
+
|
|
359
|
+
A checkpointer persists one thread's execution state so it can resume. A
|
|
360
|
+
`Store` is the other axis: durable key-value data shared across threads such as
|
|
361
|
+
user profiles, learned facts, or long-term agent memory. Items live under a
|
|
362
|
+
`namespace` (a tuple of path segments) and a string `key`; the value is any
|
|
363
|
+
JSON-serializable object.
|
|
364
|
+
|
|
365
|
+
```python
|
|
366
|
+
from agentflow import MemoryStore
|
|
367
|
+
|
|
368
|
+
store = MemoryStore()
|
|
369
|
+
await store.put(("users", "u1", "memories"), "favorite_color", "blue")
|
|
370
|
+
item = await store.get(("users", "u1", "memories"), "favorite_color")
|
|
371
|
+
recent = await store.search(("users", "u1")) # everything under that prefix
|
|
372
|
+
```
|
|
373
|
+
|
|
374
|
+
`MemoryStore` is ephemeral (tests, single process); `PostgresStore` (install
|
|
375
|
+
the `postgres` extra) is durable and shared across processes. Both implement
|
|
376
|
+
the same `Store` protocol. `put(..., ttl=seconds)` expires an item; expired
|
|
377
|
+
items never surface from `get` or `search`.
|
|
378
|
+
|
|
379
|
+
```python
|
|
380
|
+
from agentflow.store import PostgresStore
|
|
381
|
+
|
|
382
|
+
store = PostgresStore("postgresql://localhost/app")
|
|
383
|
+
await store.put(("cache",), "doc-42", payload, ttl=3600)
|
|
384
|
+
```
|
|
385
|
+
|
|
386
|
+
## Control plane (run queue + workers)
|
|
387
|
+
|
|
388
|
+
`invoke`/`stream`/`resume` run a graph inline in your code. The control plane
|
|
389
|
+
lets you manage runs from *outside* that process: enqueue a run, let a pool of
|
|
390
|
+
workers execute it, and list, cancel, or resume it from anywhere sharing the
|
|
391
|
+
same queue. Runs survive a process restart and scale across workers.
|
|
392
|
+
|
|
393
|
+
A run request references a graph by name (graphs are code, not data), so an
|
|
394
|
+
enqueuer and its workers share a `GraphRegistry` mapping names to compiled
|
|
395
|
+
graphs, and the same checkpointer so workers see the run's state.
|
|
396
|
+
|
|
397
|
+
```python
|
|
398
|
+
from agentflow import GraphRegistry, MemoryRunQueue, Worker
|
|
399
|
+
|
|
400
|
+
registry = GraphRegistry()
|
|
401
|
+
registry.register("summarize", lambda: build_graph().compile(checkpointer=cp))
|
|
402
|
+
|
|
403
|
+
queue = MemoryRunQueue()
|
|
404
|
+
run = await queue.enqueue("summarize", {"text": "..."})
|
|
405
|
+
|
|
406
|
+
# A worker (usually a separate long-running process) drains the queue.
|
|
407
|
+
worker = Worker(queue, registry)
|
|
408
|
+
await worker.run_once() # or: await worker.run_forever()
|
|
409
|
+
|
|
410
|
+
record = await queue.get(run.run_id) # status: succeeded / interrupted / ...
|
|
411
|
+
```
|
|
412
|
+
|
|
413
|
+
Run lifecycle: `queued -> running -> succeeded | interrupted | failed |
|
|
414
|
+
cancelled`. An interrupted run (a graph that hit `ctx.interrupt`) is resumed by
|
|
415
|
+
re-queuing it with the human answer:
|
|
416
|
+
|
|
417
|
+
```python
|
|
418
|
+
await queue.enqueue_resume(run.run_id, value="approved") # back to queued
|
|
419
|
+
```
|
|
420
|
+
|
|
421
|
+
Cancellation is cooperative — `await queue.request_cancel(run_id)` stops a
|
|
422
|
+
running graph at the next super-step boundary, leaving a consistent checkpoint.
|
|
423
|
+
|
|
424
|
+
`MemoryRunQueue` is for a single process/tests. `PostgresRunQueue` (install the
|
|
425
|
+
`postgres` extra) is durable and safe for many concurrent workers: `claim`
|
|
426
|
+
hands each run to exactly one worker via `FOR UPDATE SKIP LOCKED`, and a
|
|
427
|
+
time-bounded lease means a crashed worker's run is re-claimed once the lease
|
|
428
|
+
expires.
|
|
429
|
+
|
|
430
|
+
For monitoring, `await queue.stats()` returns a `QueueStats` snapshot (depth by
|
|
431
|
+
status plus `expired_leases` — running runs whose lease is past due, i.e. work
|
|
432
|
+
stranded by a crashed worker awaiting re-claim), and `pool.health()` returns a
|
|
433
|
+
`PoolHealth` snapshot (`workers`/`alive`/`busy`/`idle`, and `healthy` when every
|
|
434
|
+
worker loop is alive). Both are plain values you can expose through your own
|
|
435
|
+
`/healthz` and `/metrics` handlers.
|
|
436
|
+
|
|
437
|
+
```python
|
|
438
|
+
stats = await queue.stats() # stats.queued, stats.running, stats.expired_leases
|
|
439
|
+
if not pool.health().healthy:
|
|
440
|
+
... # a worker loop died — pool is degraded
|
|
441
|
+
```
|
|
442
|
+
|
|
443
|
+
An HTTP layer (each queue method maps 1:1 to an endpoint) is planned and not
|
|
444
|
+
part of this release.
|
|
445
|
+
|
|
446
|
+
## Prebuilt patterns
|
|
447
|
+
|
|
448
|
+
`agentflow.prebuilt.iterate_until_converged(work, ...)` compiles the classic
|
|
449
|
+
baseline-then-iterate optimization loop as a graph. You supply an async
|
|
450
|
+
`work(state) -> Candidate`; the loop owns the best-so-far, the
|
|
451
|
+
no-improvement streak, and the stop decision (`done`, `optimal`, `converged`,
|
|
452
|
+
`exhausted`).
|
|
453
|
+
|
|
454
|
+
```python
|
|
455
|
+
from agentflow.prebuilt import Candidate, iterate_until_converged
|
|
456
|
+
|
|
457
|
+
async def work(state):
|
|
458
|
+
best = state.get("best_score", 0.0)
|
|
459
|
+
return Candidate(score=min(best + 25.0, 100.0))
|
|
460
|
+
|
|
461
|
+
app = iterate_until_converged(work, perfect_score=100.0, patience=4)
|
|
462
|
+
result = await app.invoke({})
|
|
463
|
+
```
|
|
464
|
+
|
|
465
|
+
`agentflow.prebuilt.tool_loop(llm, tools)` compiles the function-calling agent
|
|
466
|
+
loop as a two-node graph: an `agent` node calls the `LLMBackend`, a `tools`
|
|
467
|
+
node runs any tool the model requested and appends the result to the
|
|
468
|
+
conversation, and a conditional edge repeats until the model answers without a
|
|
469
|
+
tool call. The LLM never executes a tool itself — the graph does.
|
|
470
|
+
|
|
471
|
+
```python
|
|
472
|
+
from agentflow.prebuilt import Tool, tool_loop
|
|
473
|
+
from agentflow.events import Message
|
|
474
|
+
from agentflow.backends.ollama import OllamaBackend
|
|
475
|
+
|
|
476
|
+
async def multiply(a: float, b: float) -> str:
|
|
477
|
+
return str(a * b)
|
|
478
|
+
|
|
479
|
+
tools = [Tool("multiply", multiply, description="Multiply two numbers",
|
|
480
|
+
schema={"type": "object",
|
|
481
|
+
"properties": {"a": {"type": "number"}, "b": {"type": "number"}},
|
|
482
|
+
"required": ["a", "b"]})]
|
|
483
|
+
|
|
484
|
+
llm = OllamaBackend("gpt-oss")
|
|
485
|
+
await llm.start()
|
|
486
|
+
app = tool_loop(llm, tools, max_turns=6)
|
|
487
|
+
out = await app.invoke({"messages": [Message("user", "What is 23 * 19?")]})
|
|
488
|
+
print(out["messages"][-1].content) # -> the model's final answer
|
|
489
|
+
await llm.close()
|
|
490
|
+
```
|
|
491
|
+
|
|
492
|
+
See `examples/tool_loop_ollama.py` for a runnable version, and
|
|
493
|
+
`examples/mixed_backends.py` for a single workflow that combines an agent
|
|
494
|
+
backend (Codex/Claude/Kiro) with an LLM tool loop, composing a compiled
|
|
495
|
+
sub-graph inside a node.
|
|
496
|
+
|
|
497
|
+
`agentflow.prebuilt` also ships `with_retry(node, retries=..., on=...)` and
|
|
498
|
+
`with_timeout(node, seconds=...)` — composable wrappers that add retry with
|
|
499
|
+
exponential backoff and per-node timeouts. Both let an `InterruptError`
|
|
500
|
+
through untouched, so human-in-the-loop suspends are never retried or timed
|
|
501
|
+
out.
|
|
502
|
+
|
|
503
|
+
## Project layout
|
|
504
|
+
|
|
505
|
+
```text
|
|
506
|
+
agentflow/
|
|
507
|
+
state.py channels, reducers, update merge
|
|
508
|
+
graph.py Graph builder + validation
|
|
509
|
+
runtime.py super-step scheduler, Context, interrupts
|
|
510
|
+
compiled.py CompiledGraph runnable
|
|
511
|
+
events.py messages, requests, streaming events
|
|
512
|
+
errors.py exception hierarchy
|
|
513
|
+
observability.py Hooks + RunMetrics
|
|
514
|
+
telemetry.py MultiHooks + JsonlTelemetry (durable JSONL)
|
|
515
|
+
backends/ base protocols + kiro/codex/claude_code/ollama/openai/anthropic
|
|
516
|
+
checkpoint/ Checkpointer protocol + memory/file/sqlite/redis/postgres
|
|
517
|
+
store/ Store protocol (cross-thread memory) + memory/postgres
|
|
518
|
+
controlplane/ RunQueue + GraphRegistry + Worker (memory/postgres)
|
|
519
|
+
prebuilt/ iterate_until_converged, tool_loop, with_retry/with_timeout
|
|
520
|
+
examples/ runnable examples
|
|
521
|
+
tests/ pytest suite (async; offline by default, `-m live` for real backends)
|
|
522
|
+
DESIGN.md architecture and contracts
|
|
523
|
+
```
|
|
524
|
+
|
|
525
|
+
## Development
|
|
526
|
+
|
|
527
|
+
```bash
|
|
528
|
+
pip install -e '.[ollama,dev]'
|
|
529
|
+
pytest -q # offline suite only (hermetic, fast)
|
|
530
|
+
pytest --cov # with coverage (source=agentflow, branch)
|
|
531
|
+
```
|
|
532
|
+
|
|
533
|
+
Tests are async and run under `pytest-asyncio` in `auto` mode, so no
|
|
534
|
+
per-test decorator is needed. Coverage is opt-in via `--cov` to keep the
|
|
535
|
+
default run fast.
|
|
536
|
+
|
|
537
|
+
Lint, format, and type checks (run in CI):
|
|
538
|
+
|
|
539
|
+
```bash
|
|
540
|
+
ruff check agentflow tests examples
|
|
541
|
+
ruff format --check agentflow tests examples
|
|
542
|
+
mypy agentflow
|
|
543
|
+
pip-audit
|
|
544
|
+
```
|
|
545
|
+
|
|
546
|
+
See `CHANGELOG.md` for the release history.
|
|
547
|
+
|
|
548
|
+
The suite is split by a `live` marker. The default run skips live tests
|
|
549
|
+
(`addopts = -m 'not live'`) so CI stays hermetic — everything mocks its
|
|
550
|
+
transport. Live smoke tests spawn the real agent CLIs and hit a real Ollama
|
|
551
|
+
server; run them explicitly:
|
|
552
|
+
|
|
553
|
+
```bash
|
|
554
|
+
pytest -m live # runs only backends that are installed/reachable
|
|
555
|
+
KIRO_AGENT=vibe pytest -m live # include the Kiro backend
|
|
556
|
+
```
|
|
557
|
+
|
|
558
|
+
Each live test skips itself when its backend is absent, so `pytest -m live`
|
|
559
|
+
never fails on a missing CLI.
|
|
560
|
+
|
|
561
|
+
## License
|
|
562
|
+
|
|
563
|
+
This project is licensed under the Apache License, Version 2.0 (Apache-2.0).
|
|
564
|
+
See the [LICENSE](LICENSE) and [NOTICE](NOTICE) files for details.
|
|
565
|
+
|
|
566
|
+
Copyright 2026 Daniel Sorkin
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
agentflow/__init__.py,sha256=zUVk_WgcVFCqNYvL9ENoOBijxJ3TL4WI20FJ-G_J_ng,4644
|
|
2
|
+
agentflow/compiled.py,sha256=UaeKCVBmW_8E0dyIsVXA2T6n32wz4PveucczpdR1lEk,11598
|
|
3
|
+
agentflow/errors.py,sha256=BN0ly5LbaFbBV-CZZ6eHKmHx7QOUb0ot7IRifiqK4xw,5991
|
|
4
|
+
agentflow/events.py,sha256=BgD8cRp7ieSz-tR_mU8GEdhXM512UCbw9bbu9lbaKy4,6333
|
|
5
|
+
agentflow/graph.py,sha256=ncRPUTQIP5QBYL4QLYawyGQrhyyxAEb5rcVLjujLljg,12771
|
|
6
|
+
agentflow/observability.py,sha256=8GJtXEEqBp15bPKE8mmXlN5Bh4DCns1if4qNdaKJf1c,4619
|
|
7
|
+
agentflow/otel.py,sha256=U6MHZVWUyjEgJ0ymfNM8wDOYsCSCpudRy4kINp9Ac8w,5139
|
|
8
|
+
agentflow/prometheus.py,sha256=JBiz7VQEweIiVvKX0LNZPBCBug8wJuZVrv55aULREpI,6245
|
|
9
|
+
agentflow/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
10
|
+
agentflow/redaction.py,sha256=ywF2GtwkPjLMR7nMpAnXyBxuiSfyB1-0LBI4xv5zcus,3132
|
|
11
|
+
agentflow/runtime.py,sha256=UUisL6grneAN4_grm8DR124SWHC8vi5lBYSXBYn2lWg,13565
|
|
12
|
+
agentflow/state.py,sha256=U-bCvpJCS4inwTsQti3XCTRVxozgzD01CzNPF7kzDYM,8025
|
|
13
|
+
agentflow/telemetry.py,sha256=WncYHlzTU2_kiYxawOzX1CcBa7Bbei6AnSzTRAaClLk,7193
|
|
14
|
+
agentflow/backends/__init__.py,sha256=vkOZwbiFlO_5wQ1tWnjJxosvwGG9hdcg7F96Jo55kLc,1634
|
|
15
|
+
agentflow/backends/_http.py,sha256=6ObsgaW_5PaByRonwoK7xuL7KwmmLJwfvr4zS9uh6Co,5933
|
|
16
|
+
agentflow/backends/anthropic.py,sha256=zAihLnuRx_MnEhdqh51JC61WMe2cxf0QDzCszx4bp4c,10652
|
|
17
|
+
agentflow/backends/base.py,sha256=qTuUFXpbQ8nj1CA8xBfzPnyrsrv6WN6SgPaV0ZvNp3o,11475
|
|
18
|
+
agentflow/backends/claude_code.py,sha256=96wup5eVtTbJMaTb9bBRSFdu4rs4KvQTBjaUiAMDicA,7397
|
|
19
|
+
agentflow/backends/cli_exec.py,sha256=wNnvigJT-7XcnppbCU8xbKVqS9kjnBO3VF2n_s91myg,8639
|
|
20
|
+
agentflow/backends/codex.py,sha256=l5vAI6RaGvzHrafLsh-YDImr12O8OP7jUdrFJacPo6o,4508
|
|
21
|
+
agentflow/backends/kiro.py,sha256=vayYjr39CqONU4EPLAeul1UZJqtEA2ZqBfxgJ1PReZE,20552
|
|
22
|
+
agentflow/backends/ollama.py,sha256=4IhcHqSSK6c53UsXImo6VLpLhEf2SyUb-vHe1KP4da4,7108
|
|
23
|
+
agentflow/backends/openai.py,sha256=lp3KtmABJvaUUbhF69Qm0AKXCUZCd7ocR0H2g-aw06U,8810
|
|
24
|
+
agentflow/checkpoint/__init__.py,sha256=9iq7PLvq92lZFpwcqpDxcAcH1P_WaRlI7ixn2UkWNR0,1860
|
|
25
|
+
agentflow/checkpoint/_serde.py,sha256=MmiJxWj2Lz4APOVCYor11aE4YeuO5EiJT0ZjOGsmGms,1821
|
|
26
|
+
agentflow/checkpoint/base.py,sha256=9P7UBUmEkCSJj7jQhvbQrDxrjw0FT14-jjCNJB_BceY,4942
|
|
27
|
+
agentflow/checkpoint/file.py,sha256=cUNlBY9yJ2bjihpKr-gviC3zJ9syKjqfAg_9VZPmo8A,8088
|
|
28
|
+
agentflow/checkpoint/memory.py,sha256=bqTRqniP9nc8qzs3wjYAIVAoM384xUoLofvTMKzg6Cw,3262
|
|
29
|
+
agentflow/checkpoint/postgres.py,sha256=no02S4a5TiVimeCr3M-odADipIq0a--YYNdHy3eHTvQ,8972
|
|
30
|
+
agentflow/checkpoint/redis.py,sha256=0z8obfwByZttN42hubeg_EU3jKW6k5_wWZns15g1nWo,8203
|
|
31
|
+
agentflow/checkpoint/sqlite.py,sha256=Rz2uU1usYrddErveyjrYxvFcXaG6_hh5JtlNApXW91A,9956
|
|
32
|
+
agentflow/controlplane/__init__.py,sha256=tacFEYqtGp35KFYNRBOnnoBR5I6n_EWhHVDjQhJ9MVI,2138
|
|
33
|
+
agentflow/controlplane/memory.py,sha256=Hnomsent-9XOsYLWGYvpVO3JMuqL4ZnLgMTtIGvSyx8,6177
|
|
34
|
+
agentflow/controlplane/postgres.py,sha256=dZqLmnKHyEJAhZukHuVbVOpJYysfbqrJpNZxL5rGO0o,11135
|
|
35
|
+
agentflow/controlplane/queue.py,sha256=9bfeFwS2lXeqWRsGIJrui87Y7s4i48enlDbMVcltR50,3520
|
|
36
|
+
agentflow/controlplane/records.py,sha256=94Rx5ZKcAVGETNc9p3iA6VnPB1VOrN7L6CD3d8hIXqw,4208
|
|
37
|
+
agentflow/controlplane/registry.py,sha256=k5IL-VX4hcvN1OHEGkr72tG2N5rEU7PmWh-XH_S8kbA,2486
|
|
38
|
+
agentflow/controlplane/worker.py,sha256=5kztMxSMBFBGU0qfrl7407j4HSGbfxdPSc_HtBE7Elg,8192
|
|
39
|
+
agentflow/prebuilt/__init__.py,sha256=H2RYbmDK0YOVdthWKEGzTkLe8BDxb3aDadafuYYWEj4,1007
|
|
40
|
+
agentflow/prebuilt/loop.py,sha256=zg82FWOREPltU9ZAfwKGPD3CNheDZqYYCbCrm6565RE,4703
|
|
41
|
+
agentflow/prebuilt/resilience.py,sha256=7n34hEZKVIJupaB--Kj-1tDb4HL9YB_-QXJVp8jwBlI,4512
|
|
42
|
+
agentflow/prebuilt/tool_loop.py,sha256=yYVjW_-HhpvE9RSOG_uYgBeP39awvASQeLzn2ZFoIUc,7077
|
|
43
|
+
agentflow/store/__init__.py,sha256=Q2QqhuSpceRnvdF6j2oAiJs2aTeCWK-pPYvtZcFlaP0,1563
|
|
44
|
+
agentflow/store/_util.py,sha256=sqUM5_Fk9LqD3UA0LMAGgsbVwt3Q6cfaK1a5YQmhaM0,2160
|
|
45
|
+
agentflow/store/base.py,sha256=rgqV0PiEnWXcEUYCop3wERPAWrLZKXCcbY7evEakleg,3620
|
|
46
|
+
agentflow/store/memory.py,sha256=6kakPkdRa16zzzW0ob55tDyMEmIdqMwgPHEINyfW_zw,3553
|
|
47
|
+
agentflow/store/postgres.py,sha256=4R4X3ZP66Q6ke_HW4JsrUX0yqhMf-bfifODYgEsT_SM,7253
|
|
48
|
+
agent_workflow_sdk-0.1.0.dist-info/METADATA,sha256=rkpYE7cM9nsWNbBAipO6Xjj1KhDJkq5xAgc4P2wO4tk,22800
|
|
49
|
+
agent_workflow_sdk-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
50
|
+
agent_workflow_sdk-0.1.0.dist-info/licenses/LICENSE,sha256=SaffLZsgtERkFib3uMTqP2pfsENXhB2yzrbVgC405hQ,11343
|
|
51
|
+
agent_workflow_sdk-0.1.0.dist-info/licenses/NOTICE,sha256=2aDBngAt3PzIRSv8Av5wUqwiRRGt9ZYdQAdbZNS_6Ys,326
|
|
52
|
+
agent_workflow_sdk-0.1.0.dist-info/RECORD,,
|