agent-workflow-sdk 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. agent_workflow_sdk-0.1.0.dist-info/METADATA +566 -0
  2. agent_workflow_sdk-0.1.0.dist-info/RECORD +52 -0
  3. agent_workflow_sdk-0.1.0.dist-info/WHEEL +4 -0
  4. agent_workflow_sdk-0.1.0.dist-info/licenses/LICENSE +201 -0
  5. agent_workflow_sdk-0.1.0.dist-info/licenses/NOTICE +10 -0
  6. agentflow/__init__.py +199 -0
  7. agentflow/backends/__init__.py +54 -0
  8. agentflow/backends/_http.py +169 -0
  9. agentflow/backends/anthropic.py +291 -0
  10. agentflow/backends/base.py +327 -0
  11. agentflow/backends/claude_code.py +181 -0
  12. agentflow/backends/cli_exec.py +237 -0
  13. agentflow/backends/codex.py +126 -0
  14. agentflow/backends/kiro.py +523 -0
  15. agentflow/backends/ollama.py +208 -0
  16. agentflow/backends/openai.py +249 -0
  17. agentflow/checkpoint/__init__.py +51 -0
  18. agentflow/checkpoint/_serde.py +57 -0
  19. agentflow/checkpoint/base.py +125 -0
  20. agentflow/checkpoint/file.py +210 -0
  21. agentflow/checkpoint/memory.py +91 -0
  22. agentflow/checkpoint/postgres.py +232 -0
  23. agentflow/checkpoint/redis.py +206 -0
  24. agentflow/checkpoint/sqlite.py +261 -0
  25. agentflow/compiled.py +313 -0
  26. agentflow/controlplane/__init__.py +57 -0
  27. agentflow/controlplane/memory.py +181 -0
  28. agentflow/controlplane/postgres.py +297 -0
  29. agentflow/controlplane/queue.py +89 -0
  30. agentflow/controlplane/records.py +122 -0
  31. agentflow/controlplane/registry.py +70 -0
  32. agentflow/controlplane/worker.py +219 -0
  33. agentflow/errors.py +186 -0
  34. agentflow/events.py +226 -0
  35. agentflow/graph.py +328 -0
  36. agentflow/observability.py +126 -0
  37. agentflow/otel.py +124 -0
  38. agentflow/prebuilt/__init__.py +29 -0
  39. agentflow/prebuilt/loop.py +133 -0
  40. agentflow/prebuilt/resilience.py +120 -0
  41. agentflow/prebuilt/tool_loop.py +185 -0
  42. agentflow/prometheus.py +175 -0
  43. agentflow/py.typed +0 -0
  44. agentflow/redaction.py +99 -0
  45. agentflow/runtime.py +356 -0
  46. agentflow/state.py +234 -0
  47. agentflow/store/__init__.py +40 -0
  48. agentflow/store/_util.py +54 -0
  49. agentflow/store/base.py +103 -0
  50. agentflow/store/memory.py +109 -0
  51. agentflow/store/postgres.py +210 -0
  52. agentflow/telemetry.py +197 -0
@@ -0,0 +1,566 @@
1
+ Metadata-Version: 2.5
2
+ Name: agent-workflow-sdk
3
+ Version: 0.1.0
4
+ Summary: Async-first, LangGraph-style SDK for building agent workflows with pluggable agent and LLM backends.
5
+ Project-URL: Homepage, https://github.com/dankosorkin/agent-workflow-sdk
6
+ Project-URL: Repository, https://github.com/dankosorkin/agent-workflow-sdk
7
+ Project-URL: Changelog, https://github.com/dankosorkin/agent-workflow-sdk/blob/main/CHANGELOG.md
8
+ Project-URL: Issues, https://github.com/dankosorkin/agent-workflow-sdk/issues
9
+ Author: Daniel Sorkin
10
+ License-Expression: Apache-2.0
11
+ License-File: LICENSE
12
+ License-File: NOTICE
13
+ Keywords: agents,async,graph,llm,orchestration,workflow
14
+ Classifier: Development Status :: 4 - Beta
15
+ Classifier: Framework :: AsyncIO
16
+ Classifier: Intended Audience :: Developers
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Typing :: Typed
20
+ Requires-Python: >=3.11
21
+ Provides-Extra: dev
22
+ Requires-Dist: fakeredis>=2.20; extra == 'dev'
23
+ Requires-Dist: mypy>=1.11; extra == 'dev'
24
+ Requires-Dist: pip-audit>=2.7; extra == 'dev'
25
+ Requires-Dist: prometheus-client>=0.20; extra == 'dev'
26
+ Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
27
+ Requires-Dist: pytest-cov>=5; extra == 'dev'
28
+ Requires-Dist: pytest>=8; extra == 'dev'
29
+ Requires-Dist: ruff>=0.6; extra == 'dev'
30
+ Provides-Extra: ollama
31
+ Requires-Dist: httpx>=0.27; extra == 'ollama'
32
+ Provides-Extra: otel
33
+ Requires-Dist: opentelemetry-api>=1.20; extra == 'otel'
34
+ Requires-Dist: opentelemetry-sdk>=1.20; extra == 'otel'
35
+ Provides-Extra: postgres
36
+ Requires-Dist: asyncpg>=0.29; extra == 'postgres'
37
+ Provides-Extra: prometheus
38
+ Requires-Dist: prometheus-client>=0.20; extra == 'prometheus'
39
+ Provides-Extra: redis
40
+ Requires-Dist: redis>=5; extra == 'redis'
41
+ Description-Content-Type: text/markdown
42
+
43
+ # agent-workflow-sdk
44
+
45
+ An async-first, LangGraph-style SDK for building agent workflows from
46
+ composable pieces. You describe a workflow as a graph of nodes over a typed,
47
+ reducer-based state, and run it against a pluggable backend — a coding agent
48
+ (Kiro, Codex, or Claude Code) or a plain LLM (Ollama, any OpenAI-compatible
49
+ endpoint, or Anthropic).
50
+
51
+ The import root is `agentflow`. The distribution name is
52
+ `agent-workflow-sdk`.
53
+
54
+ ## Why
55
+
56
+ - Build any workflow from primitives: nodes, edges, conditional routing, and
57
+ a shared state. Not a fixed loop, not a linear chain.
58
+ - Swap the backend without touching workflow code. Agents and LLMs share one
59
+ event vocabulary.
60
+ - Async everywhere: the engine, backends, checkpointing, and streaming.
61
+ - Durable by default: every super-step is checkpointed, runs resume after a
62
+ crash, and human-in-the-loop interrupts suspend and resume a run.
63
+
64
+
65
+ ## Install
66
+
67
+ The core is dependency-free. Backends that need extra libraries are optional
68
+ extras.
69
+
70
+ ```bash
71
+ pip install -e . # core only
72
+ pip install -e '.[ollama]' # + httpx, for the Ollama LLM backend
73
+ pip install -e '.[dev]' # + pytest, pytest-asyncio
74
+ ```
75
+
76
+ Python 3.11+ is required. The Kiro backend needs `kiro-cli` on PATH.
77
+
78
+ ## Quickstart
79
+
80
+ A graph that loops until a counter reaches a target, then stops.
81
+
82
+ ```python
83
+ import asyncio
84
+ from typing import Annotated
85
+ from agentflow import Graph, START, END, State, add, append
86
+
87
+ class CountState(State):
88
+ n: Annotated[int, add] # updates are summed
89
+ log: Annotated[list, append] # updates are concatenated
90
+
91
+ async def tick(state, ctx):
92
+ return {"n": 1, "log": f"tick {state.get('n', 0) + 1}"}
93
+
94
+ def route(state):
95
+ return "again" if state["n"] < 5 else "done"
96
+
97
+ g = Graph(CountState)
98
+ g.add_node("tick", tick)
99
+ g.add_edge(START, "tick")
100
+ g.add_conditional_edges("tick", route, {"again": "tick", "done": END})
101
+ app = g.compile()
102
+
103
+ print(asyncio.run(app.invoke({"n": 0, "log": []})))
104
+ ```
105
+
106
+ See `examples/hello_graph.py` and `examples/optimize_loop.py` for runnable
107
+ versions.
108
+
109
+ ## Core concepts
110
+
111
+ ### State and reducers
112
+
113
+ State is a `TypedDict` subclass of `State`. Each field is a channel; annotate
114
+ it with a reducer that folds each node's update into the current value. A
115
+ field without a reducer uses `last` (last-value-wins). Built-in reducers:
116
+ `last`, `append`, `add`, `merge`, `union`. A node returns a partial update; it
117
+ never mutates the state it was given.
118
+
119
+ ### Nodes and edges
120
+
121
+ A node is an async callable `(state, ctx) -> update`. Edges are either static
122
+ (`add_edge`) or conditional (`add_conditional_edges` with a `router(state) ->
123
+ key`). `START` and `END` are sentinels. A conditional router may return a list
124
+ of keys to fan out to several nodes at once.
125
+
126
+ ### Execution
127
+
128
+ A run proceeds in super-steps. All nodes on the current frontier run
129
+ concurrently against the same immutable state, their updates are folded
130
+ through the reducers in a deterministic order, and the next frontier is
131
+ computed from the edges. A `step_limit` guards against runaway loops.
132
+
133
+ Super-steps are all-or-nothing: if any node in a step raises, that step's
134
+ updates are discarded before they touch the state and no checkpoint is written
135
+ for it, so the run stays at the last committed checkpoint. Input to a run is
136
+ validated at start — an undeclared channel key is rejected with a clear error
137
+ rather than sitting unused in the state.
138
+
139
+ `CompiledGraph` gives you:
140
+
141
+ - `await app.invoke(input, thread=..., timeout=...)` — run to completion,
142
+ return state. `timeout` bounds the whole run; on expiry it raises
143
+ `RunTimeout` and the last checkpoint is preserved for resume.
144
+ - `app.stream(input, thread=...)` — async-iterate `StreamEvent`s as they occur.
145
+ - `await app.resume(thread, value=...)` — continue a suspended run.
146
+ - `await app.get_state(thread)` / `app.history(thread)` — inspect checkpoints.
147
+
148
+ For non-async callers there are blocking wrappers — `app.invoke_sync(...)`,
149
+ `app.resume_sync(...)`, `app.stream_sync(...)` — which run the coroutine via
150
+ `asyncio.run` and refuse to run inside an existing event loop.
151
+
152
+ ### Subgraphs
153
+
154
+ A compiled graph composes as a node in another graph:
155
+
156
+ ```python
157
+ parent.add_node("sub", compiled_subgraph) # channels shared by name
158
+ parent.add_subgraph("sub", compiled_subgraph, # or map explicitly
159
+ input_map={"query": "q"}, output_map={"answer": "result"})
160
+ ```
161
+
162
+ A subgraph runs on an isolated checkpoint sub-thread and merges its result
163
+ back through the parent's reducers. Use `output_map` to route a result into a
164
+ distinct parent channel and avoid double-counting accumulator channels.
165
+
166
+ ### Observability
167
+
168
+ Pass a `Hooks` implementation to `compile(hooks=...)` for lifecycle callbacks
169
+ (`on_run_start`, `on_node_start/end/error`, `on_step_end`, `on_run_end`). A
170
+ hook raising never breaks a run. `RunMetrics` is a ready-made `Hooks` that
171
+ records per-node call counts, errors, and durations:
172
+
173
+ ```python
174
+ from agentflow import RunMetrics
175
+ m = RunMetrics()
176
+ await g.compile(hooks=m).invoke({...})
177
+ print(m.summary()) # {"steps": 3, "completed": True, "nodes": {...}}
178
+ ```
179
+
180
+ For durable telemetry, `JsonlTelemetry` is a `Hooks` that writes one JSON line
181
+ per event (`run_start`, `node_start/end/error`, `event`, `step`, `run_end`) —
182
+ backend events surfaced via `ctx.emit` are logged too. Compose several
183
+ listeners with `MultiHooks`; a failing listener never breaks the run or the
184
+ others:
185
+
186
+ ```python
187
+ from agentflow import MultiHooks, RunMetrics, JsonlTelemetry
188
+ metrics = RunMetrics()
189
+ telemetry = JsonlTelemetry(".runs") # dir -> one file per thread
190
+ app = g.compile(hooks=MultiHooks(metrics, telemetry))
191
+ ```
192
+
193
+ See `examples/telemetry_demo.py` for a runnable version.
194
+
195
+ For distributed tracing, `agentflow.otel.OtelHooks` (install the `otel` extra)
196
+ is a `Hooks` that emits an OpenTelemetry span per run and per node:
197
+
198
+ ```python
199
+ from agentflow.otel import OtelHooks
200
+ app = g.compile(hooks=OtelHooks()) # uses the global tracer/provider
201
+ ```
202
+
203
+ For operators who scrape Prometheus instead of running an OTel collector,
204
+ `agentflow.prometheus.PrometheusHooks` (install the `prometheus` extra) is a
205
+ `Hooks` that records run/node/step counters, a node-duration histogram, and
206
+ backend-event counts. It uses a private registry by default; call
207
+ `exposition()` to render the text format for a `/metrics` endpoint (you own the
208
+ HTTP layer).
209
+
210
+ ```python
211
+ from agentflow.prometheus import PrometheusHooks
212
+
213
+ metrics = PrometheusHooks()
214
+ app = g.compile(hooks=metrics)
215
+ # ... inside your /metrics handler:
216
+ body, content_type = metrics.exposition()
217
+ ```
218
+
219
+ Compose several listeners with `MultiHooks(RunMetrics(), OtelHooks(),
220
+ PrometheusHooks())`.
221
+
222
+ ## Backends
223
+
224
+ Two kinds of backend share one event stream, so a node calls either the same
225
+ way.
226
+
227
+ - Agent backends (`AgentBackend`) drive a session that runs its own tools and
228
+ asks permission. Available: `KiroBackend` (persistent ACP session over
229
+ `kiro-cli`), `CodexBackend` (`codex exec --json`), `ClaudeCodeBackend`
230
+ (`claude -p --output-format stream-json`).
231
+ - LLM backends (`LLMBackend`) are stateless: messages in, token stream out.
232
+ They never run tools themselves — a tool call is a request the graph
233
+ fulfils. Available: `OllamaBackend` (local `/api/chat`), `OpenAIBackend`
234
+ (any `/v1/chat/completions` endpoint — OpenAI, Groq, Together, vLLM, LM
235
+ Studio, and Ollama's own `/v1` shim), and `AnthropicBackend` (Messages API).
236
+
237
+ ```python
238
+ from agentflow.backends.kiro import KiroBackend
239
+ from agentflow.backends.codex import CodexBackend
240
+ from agentflow.backends.claude_code import ClaudeCodeBackend
241
+ from agentflow.backends.ollama import OllamaBackend
242
+ from agentflow.backends.openai import OpenAIBackend
243
+ from agentflow.backends.anthropic import AnthropicBackend
244
+
245
+ agent = KiroBackend("vibe", engine="v3") # persistent session
246
+ codex = CodexBackend(sandbox="read-only") # one-shot per turn
247
+ claude = ClaudeCodeBackend(model="sonnet") # one-shot per turn
248
+ llm = OllamaBackend("llama3.2") # local HTTP
249
+ openai = OpenAIBackend("gpt-4o-mini", api_key="...") # any OpenAI-compatible API
250
+ anthropic = AnthropicBackend("claude-sonnet-4", api_key="...") # Messages API
251
+
252
+ await agent.start()
253
+ async for event in agent.prompt("summarize the repo"):
254
+ ... # TextChunk, ToolCall, ToolResult, ..., TurnEnd
255
+ await agent.close()
256
+ ```
257
+
258
+ The three agent backends have different lifecycles under one interface. Kiro
259
+ holds a long-lived JSON-RPC session; Codex and Claude Code run a fresh
260
+ subprocess per turn and carry a resumable session id between turns. Either way
261
+ you call `start()`, `prompt(text)`, `close()` and consume the same event
262
+ stream. Backends are constructed by you and passed into your nodes. The core
263
+ never imports a backend, so importing `agentflow` pulls in no subprocess or
264
+ HTTP dependency.
265
+
266
+ The HTTP LLM backends retry transient failures (connection errors, timeouts,
267
+ 429/5xx) on the connect/initial-response phase only — never mid-stream, so a
268
+ partial token stream is never replayed. `Retry-After` is honored. Tune it with
269
+ a `RetryPolicy`:
270
+
271
+ ```python
272
+ from agentflow.backends import RetryPolicy
273
+ from agentflow.backends.openai import OpenAIBackend
274
+
275
+ llm = OpenAIBackend("gpt-4o-mini", api_key="...",
276
+ retry=RetryPolicy(max_retries=4, backoff=0.5))
277
+ ```
278
+
279
+ Run `python examples/agents_demo.py` to smoke every backend installed on your
280
+ machine (set `KIRO_AGENT` to a valid agent id, e.g. `vibe`, to include Kiro).
281
+
282
+ ### Permission policies
283
+
284
+ A `PermissionPolicy` is **required** when constructing an agent backend — there
285
+ is no default, because auto-approving an agent's tool use is a security choice
286
+ the caller must make explicitly. Options: `AllowAll` (trusted local sandbox
287
+ only), `DenyAll`, `Interactive` (escalate to a human via an engine interrupt),
288
+ or `ToolAllowlist({"read_file", ...}, fallback=DenyAll())` for least-privilege
289
+ scoping.
290
+
291
+ ```python
292
+ from agentflow.backends import AllowAll, ToolAllowlist, DenyAll
293
+ KiroBackend("vibe", permission=ToolAllowlist({"read_file"}, fallback=DenyAll()))
294
+ ```
295
+
296
+ Capability matrix — where the policy actually applies:
297
+
298
+ | Backend | Routes tool requests through `PermissionPolicy`? |
299
+ | --- | --- |
300
+ | `KiroBackend` | Yes — ACP `session/request_permission` is resolved by the policy (and `Interactive` drives a real HITL interrupt). |
301
+ | `CodexBackend` | No — one-shot CLI; permission is governed by its `--sandbox` mode. The policy field is still required but does not intercept prompts. |
302
+ | `ClaudeCodeBackend` | No — one-shot CLI; permission is governed by CLI flags (`--permission-mode`, `--allowed-tools`). |
303
+
304
+ For the one-shot CLI agents, configure their own controls (sandbox, allowed
305
+ tools) — the SDK policy alone does not sandbox them.
306
+
307
+ ## Checkpointing and human-in-the-loop
308
+
309
+ Pass a checkpointer to `compile` to make runs durable.
310
+
311
+ ```python
312
+ from agentflow import FileCheckpointer
313
+ app = g.compile(checkpointer=FileCheckpointer(".runs"))
314
+ ```
315
+
316
+ `MemoryCheckpointer` is for tests; `FileCheckpointer` writes one atomic JSON
317
+ file per super-step under `.runs/<thread>/` (single-writer / single-process);
318
+ `SqliteCheckpointer` stores checkpoints transactionally with atomic per-step
319
+ revisions; `RedisCheckpointer` (install the `redis` extra) persists to a Redis
320
+ server for distributed / multi-process runners; and `PostgresCheckpointer`
321
+ (install the `postgres` extra) persists to a Postgres table. All implement the
322
+ same `Checkpointer` protocol, so they are interchangeable at
323
+ `compile(checkpointer=...)`.
324
+
325
+ ```python
326
+ from agentflow.checkpoint import PostgresCheckpointer
327
+ app = g.compile(checkpointer=PostgresCheckpointer("postgresql://localhost/app"))
328
+ ```
329
+
330
+ For production durability prefer `PostgresCheckpointer`: a committed row
331
+ survives a crash by design. Redis is only crash-durable when you configure
332
+ AOF/RDB persistence.
333
+
334
+ ### Optimistic concurrency and retention
335
+
336
+ Each checkpoint carries a `revision` (the write count for its `(thread, step)`,
337
+ set when you read it back). To guard a read-modify-write against a concurrent
338
+ resume, pass the revision you read as `if_revision`; a stale value raises
339
+ `CheckpointConflict` instead of silently overwriting.
340
+
341
+ ```python
342
+ cp = await checkpointer.get(thread, step)
343
+ # ... resume work off cp ...
344
+ await checkpointer.put(new_cp, if_revision=cp.revision) # raises on conflict
345
+ ```
346
+
347
+ Two retention methods trim old state: `delete_thread(thread)` drops a whole
348
+ thread, and `prune(thread, before_step=..., older_than=...)` deletes older
349
+ checkpoints and returns how many it removed.
350
+
351
+ A node calls `await ctx.interrupt(payload)` to suspend the run for a human.
352
+ The runtime writes an interrupted checkpoint and stops. Later, `await
353
+ app.resume(thread, value=answer)` continues the run, and the same `interrupt`
354
+ call returns `answer`. This works across process restarts, since the frontier
355
+ is persisted.
356
+
357
+ ## Cross-thread memory (Store)
358
+
359
+ A checkpointer persists one thread's execution state so it can resume. A
360
+ `Store` is the other axis: durable key-value data shared across threads such as
361
+ user profiles, learned facts, or long-term agent memory. Items live under a
362
+ `namespace` (a tuple of path segments) and a string `key`; the value is any
363
+ JSON-serializable object.
364
+
365
+ ```python
366
+ from agentflow import MemoryStore
367
+
368
+ store = MemoryStore()
369
+ await store.put(("users", "u1", "memories"), "favorite_color", "blue")
370
+ item = await store.get(("users", "u1", "memories"), "favorite_color")
371
+ recent = await store.search(("users", "u1")) # everything under that prefix
372
+ ```
373
+
374
+ `MemoryStore` is ephemeral (tests, single process); `PostgresStore` (install
375
+ the `postgres` extra) is durable and shared across processes. Both implement
376
+ the same `Store` protocol. `put(..., ttl=seconds)` expires an item; expired
377
+ items never surface from `get` or `search`.
378
+
379
+ ```python
380
+ from agentflow.store import PostgresStore
381
+
382
+ store = PostgresStore("postgresql://localhost/app")
383
+ await store.put(("cache",), "doc-42", payload, ttl=3600)
384
+ ```
385
+
386
+ ## Control plane (run queue + workers)
387
+
388
+ `invoke`/`stream`/`resume` run a graph inline in your code. The control plane
389
+ lets you manage runs from *outside* that process: enqueue a run, let a pool of
390
+ workers execute it, and list, cancel, or resume it from anywhere sharing the
391
+ same queue. Runs survive a process restart and scale across workers.
392
+
393
+ A run request references a graph by name (graphs are code, not data), so an
394
+ enqueuer and its workers share a `GraphRegistry` mapping names to compiled
395
+ graphs, and the same checkpointer so workers see the run's state.
396
+
397
+ ```python
398
+ from agentflow import GraphRegistry, MemoryRunQueue, Worker
399
+
400
+ registry = GraphRegistry()
401
+ registry.register("summarize", lambda: build_graph().compile(checkpointer=cp))
402
+
403
+ queue = MemoryRunQueue()
404
+ run = await queue.enqueue("summarize", {"text": "..."})
405
+
406
+ # A worker (usually a separate long-running process) drains the queue.
407
+ worker = Worker(queue, registry)
408
+ await worker.run_once() # or: await worker.run_forever()
409
+
410
+ record = await queue.get(run.run_id) # status: succeeded / interrupted / ...
411
+ ```
412
+
413
+ Run lifecycle: `queued -> running -> succeeded | interrupted | failed |
414
+ cancelled`. An interrupted run (a graph that hit `ctx.interrupt`) is resumed by
415
+ re-queuing it with the human answer:
416
+
417
+ ```python
418
+ await queue.enqueue_resume(run.run_id, value="approved") # back to queued
419
+ ```
420
+
421
+ Cancellation is cooperative — `await queue.request_cancel(run_id)` stops a
422
+ running graph at the next super-step boundary, leaving a consistent checkpoint.
423
+
424
+ `MemoryRunQueue` is for a single process/tests. `PostgresRunQueue` (install the
425
+ `postgres` extra) is durable and safe for many concurrent workers: `claim`
426
+ hands each run to exactly one worker via `FOR UPDATE SKIP LOCKED`, and a
427
+ time-bounded lease means a crashed worker's run is re-claimed once the lease
428
+ expires.
429
+
430
+ For monitoring, `await queue.stats()` returns a `QueueStats` snapshot (depth by
431
+ status plus `expired_leases` — running runs whose lease is past due, i.e. work
432
+ stranded by a crashed worker awaiting re-claim), and `pool.health()` returns a
433
+ `PoolHealth` snapshot (`workers`/`alive`/`busy`/`idle`, and `healthy` when every
434
+ worker loop is alive). Both are plain values you can expose through your own
435
+ `/healthz` and `/metrics` handlers.
436
+
437
+ ```python
438
+ stats = await queue.stats() # stats.queued, stats.running, stats.expired_leases
439
+ if not pool.health().healthy:
440
+ ... # a worker loop died — pool is degraded
441
+ ```
442
+
443
+ An HTTP layer (each queue method maps 1:1 to an endpoint) is planned and not
444
+ part of this release.
445
+
446
+ ## Prebuilt patterns
447
+
448
+ `agentflow.prebuilt.iterate_until_converged(work, ...)` compiles the classic
449
+ baseline-then-iterate optimization loop as a graph. You supply an async
450
+ `work(state) -> Candidate`; the loop owns the best-so-far, the
451
+ no-improvement streak, and the stop decision (`done`, `optimal`, `converged`,
452
+ `exhausted`).
453
+
454
+ ```python
455
+ from agentflow.prebuilt import Candidate, iterate_until_converged
456
+
457
+ async def work(state):
458
+ best = state.get("best_score", 0.0)
459
+ return Candidate(score=min(best + 25.0, 100.0))
460
+
461
+ app = iterate_until_converged(work, perfect_score=100.0, patience=4)
462
+ result = await app.invoke({})
463
+ ```
464
+
465
+ `agentflow.prebuilt.tool_loop(llm, tools)` compiles the function-calling agent
466
+ loop as a two-node graph: an `agent` node calls the `LLMBackend`, a `tools`
467
+ node runs any tool the model requested and appends the result to the
468
+ conversation, and a conditional edge repeats until the model answers without a
469
+ tool call. The LLM never executes a tool itself — the graph does.
470
+
471
+ ```python
472
+ from agentflow.prebuilt import Tool, tool_loop
473
+ from agentflow.events import Message
474
+ from agentflow.backends.ollama import OllamaBackend
475
+
476
+ async def multiply(a: float, b: float) -> str:
477
+ return str(a * b)
478
+
479
+ tools = [Tool("multiply", multiply, description="Multiply two numbers",
480
+ schema={"type": "object",
481
+ "properties": {"a": {"type": "number"}, "b": {"type": "number"}},
482
+ "required": ["a", "b"]})]
483
+
484
+ llm = OllamaBackend("gpt-oss")
485
+ await llm.start()
486
+ app = tool_loop(llm, tools, max_turns=6)
487
+ out = await app.invoke({"messages": [Message("user", "What is 23 * 19?")]})
488
+ print(out["messages"][-1].content) # -> the model's final answer
489
+ await llm.close()
490
+ ```
491
+
492
+ See `examples/tool_loop_ollama.py` for a runnable version, and
493
+ `examples/mixed_backends.py` for a single workflow that combines an agent
494
+ backend (Codex/Claude/Kiro) with an LLM tool loop, composing a compiled
495
+ sub-graph inside a node.
496
+
497
+ `agentflow.prebuilt` also ships `with_retry(node, retries=..., on=...)` and
498
+ `with_timeout(node, seconds=...)` — composable wrappers that add retry with
499
+ exponential backoff and per-node timeouts. Both let an `InterruptError`
500
+ through untouched, so human-in-the-loop suspends are never retried or timed
501
+ out.
502
+
503
+ ## Project layout
504
+
505
+ ```text
506
+ agentflow/
507
+ state.py channels, reducers, update merge
508
+ graph.py Graph builder + validation
509
+ runtime.py super-step scheduler, Context, interrupts
510
+ compiled.py CompiledGraph runnable
511
+ events.py messages, requests, streaming events
512
+ errors.py exception hierarchy
513
+ observability.py Hooks + RunMetrics
514
+ telemetry.py MultiHooks + JsonlTelemetry (durable JSONL)
515
+ backends/ base protocols + kiro/codex/claude_code/ollama/openai/anthropic
516
+ checkpoint/ Checkpointer protocol + memory/file/sqlite/redis/postgres
517
+ store/ Store protocol (cross-thread memory) + memory/postgres
518
+ controlplane/ RunQueue + GraphRegistry + Worker (memory/postgres)
519
+ prebuilt/ iterate_until_converged, tool_loop, with_retry/with_timeout
520
+ examples/ runnable examples
521
+ tests/ pytest suite (async; offline by default, `-m live` for real backends)
522
+ DESIGN.md architecture and contracts
523
+ ```
524
+
525
+ ## Development
526
+
527
+ ```bash
528
+ pip install -e '.[ollama,dev]'
529
+ pytest -q # offline suite only (hermetic, fast)
530
+ pytest --cov # with coverage (source=agentflow, branch)
531
+ ```
532
+
533
+ Tests are async and run under `pytest-asyncio` in `auto` mode, so no
534
+ per-test decorator is needed. Coverage is opt-in via `--cov` to keep the
535
+ default run fast.
536
+
537
+ Lint, format, and type checks (run in CI):
538
+
539
+ ```bash
540
+ ruff check agentflow tests examples
541
+ ruff format --check agentflow tests examples
542
+ mypy agentflow
543
+ pip-audit
544
+ ```
545
+
546
+ See `CHANGELOG.md` for the release history.
547
+
548
+ The suite is split by a `live` marker. The default run skips live tests
549
+ (`addopts = -m 'not live'`) so CI stays hermetic — everything mocks its
550
+ transport. Live smoke tests spawn the real agent CLIs and hit a real Ollama
551
+ server; run them explicitly:
552
+
553
+ ```bash
554
+ pytest -m live # runs only backends that are installed/reachable
555
+ KIRO_AGENT=vibe pytest -m live # include the Kiro backend
556
+ ```
557
+
558
+ Each live test skips itself when its backend is absent, so `pytest -m live`
559
+ never fails on a missing CLI.
560
+
561
+ ## License
562
+
563
+ This project is licensed under the Apache License, Version 2.0 (Apache-2.0).
564
+ See the [LICENSE](LICENSE) and [NOTICE](NOTICE) files for details.
565
+
566
+ Copyright 2026 Daniel Sorkin
@@ -0,0 +1,52 @@
1
+ agentflow/__init__.py,sha256=zUVk_WgcVFCqNYvL9ENoOBijxJ3TL4WI20FJ-G_J_ng,4644
2
+ agentflow/compiled.py,sha256=UaeKCVBmW_8E0dyIsVXA2T6n32wz4PveucczpdR1lEk,11598
3
+ agentflow/errors.py,sha256=BN0ly5LbaFbBV-CZZ6eHKmHx7QOUb0ot7IRifiqK4xw,5991
4
+ agentflow/events.py,sha256=BgD8cRp7ieSz-tR_mU8GEdhXM512UCbw9bbu9lbaKy4,6333
5
+ agentflow/graph.py,sha256=ncRPUTQIP5QBYL4QLYawyGQrhyyxAEb5rcVLjujLljg,12771
6
+ agentflow/observability.py,sha256=8GJtXEEqBp15bPKE8mmXlN5Bh4DCns1if4qNdaKJf1c,4619
7
+ agentflow/otel.py,sha256=U6MHZVWUyjEgJ0ymfNM8wDOYsCSCpudRy4kINp9Ac8w,5139
8
+ agentflow/prometheus.py,sha256=JBiz7VQEweIiVvKX0LNZPBCBug8wJuZVrv55aULREpI,6245
9
+ agentflow/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
10
+ agentflow/redaction.py,sha256=ywF2GtwkPjLMR7nMpAnXyBxuiSfyB1-0LBI4xv5zcus,3132
11
+ agentflow/runtime.py,sha256=UUisL6grneAN4_grm8DR124SWHC8vi5lBYSXBYn2lWg,13565
12
+ agentflow/state.py,sha256=U-bCvpJCS4inwTsQti3XCTRVxozgzD01CzNPF7kzDYM,8025
13
+ agentflow/telemetry.py,sha256=WncYHlzTU2_kiYxawOzX1CcBa7Bbei6AnSzTRAaClLk,7193
14
+ agentflow/backends/__init__.py,sha256=vkOZwbiFlO_5wQ1tWnjJxosvwGG9hdcg7F96Jo55kLc,1634
15
+ agentflow/backends/_http.py,sha256=6ObsgaW_5PaByRonwoK7xuL7KwmmLJwfvr4zS9uh6Co,5933
16
+ agentflow/backends/anthropic.py,sha256=zAihLnuRx_MnEhdqh51JC61WMe2cxf0QDzCszx4bp4c,10652
17
+ agentflow/backends/base.py,sha256=qTuUFXpbQ8nj1CA8xBfzPnyrsrv6WN6SgPaV0ZvNp3o,11475
18
+ agentflow/backends/claude_code.py,sha256=96wup5eVtTbJMaTb9bBRSFdu4rs4KvQTBjaUiAMDicA,7397
19
+ agentflow/backends/cli_exec.py,sha256=wNnvigJT-7XcnppbCU8xbKVqS9kjnBO3VF2n_s91myg,8639
20
+ agentflow/backends/codex.py,sha256=l5vAI6RaGvzHrafLsh-YDImr12O8OP7jUdrFJacPo6o,4508
21
+ agentflow/backends/kiro.py,sha256=vayYjr39CqONU4EPLAeul1UZJqtEA2ZqBfxgJ1PReZE,20552
22
+ agentflow/backends/ollama.py,sha256=4IhcHqSSK6c53UsXImo6VLpLhEf2SyUb-vHe1KP4da4,7108
23
+ agentflow/backends/openai.py,sha256=lp3KtmABJvaUUbhF69Qm0AKXCUZCd7ocR0H2g-aw06U,8810
24
+ agentflow/checkpoint/__init__.py,sha256=9iq7PLvq92lZFpwcqpDxcAcH1P_WaRlI7ixn2UkWNR0,1860
25
+ agentflow/checkpoint/_serde.py,sha256=MmiJxWj2Lz4APOVCYor11aE4YeuO5EiJT0ZjOGsmGms,1821
26
+ agentflow/checkpoint/base.py,sha256=9P7UBUmEkCSJj7jQhvbQrDxrjw0FT14-jjCNJB_BceY,4942
27
+ agentflow/checkpoint/file.py,sha256=cUNlBY9yJ2bjihpKr-gviC3zJ9syKjqfAg_9VZPmo8A,8088
28
+ agentflow/checkpoint/memory.py,sha256=bqTRqniP9nc8qzs3wjYAIVAoM384xUoLofvTMKzg6Cw,3262
29
+ agentflow/checkpoint/postgres.py,sha256=no02S4a5TiVimeCr3M-odADipIq0a--YYNdHy3eHTvQ,8972
30
+ agentflow/checkpoint/redis.py,sha256=0z8obfwByZttN42hubeg_EU3jKW6k5_wWZns15g1nWo,8203
31
+ agentflow/checkpoint/sqlite.py,sha256=Rz2uU1usYrddErveyjrYxvFcXaG6_hh5JtlNApXW91A,9956
32
+ agentflow/controlplane/__init__.py,sha256=tacFEYqtGp35KFYNRBOnnoBR5I6n_EWhHVDjQhJ9MVI,2138
33
+ agentflow/controlplane/memory.py,sha256=Hnomsent-9XOsYLWGYvpVO3JMuqL4ZnLgMTtIGvSyx8,6177
34
+ agentflow/controlplane/postgres.py,sha256=dZqLmnKHyEJAhZukHuVbVOpJYysfbqrJpNZxL5rGO0o,11135
35
+ agentflow/controlplane/queue.py,sha256=9bfeFwS2lXeqWRsGIJrui87Y7s4i48enlDbMVcltR50,3520
36
+ agentflow/controlplane/records.py,sha256=94Rx5ZKcAVGETNc9p3iA6VnPB1VOrN7L6CD3d8hIXqw,4208
37
+ agentflow/controlplane/registry.py,sha256=k5IL-VX4hcvN1OHEGkr72tG2N5rEU7PmWh-XH_S8kbA,2486
38
+ agentflow/controlplane/worker.py,sha256=5kztMxSMBFBGU0qfrl7407j4HSGbfxdPSc_HtBE7Elg,8192
39
+ agentflow/prebuilt/__init__.py,sha256=H2RYbmDK0YOVdthWKEGzTkLe8BDxb3aDadafuYYWEj4,1007
40
+ agentflow/prebuilt/loop.py,sha256=zg82FWOREPltU9ZAfwKGPD3CNheDZqYYCbCrm6565RE,4703
41
+ agentflow/prebuilt/resilience.py,sha256=7n34hEZKVIJupaB--Kj-1tDb4HL9YB_-QXJVp8jwBlI,4512
42
+ agentflow/prebuilt/tool_loop.py,sha256=yYVjW_-HhpvE9RSOG_uYgBeP39awvASQeLzn2ZFoIUc,7077
43
+ agentflow/store/__init__.py,sha256=Q2QqhuSpceRnvdF6j2oAiJs2aTeCWK-pPYvtZcFlaP0,1563
44
+ agentflow/store/_util.py,sha256=sqUM5_Fk9LqD3UA0LMAGgsbVwt3Q6cfaK1a5YQmhaM0,2160
45
+ agentflow/store/base.py,sha256=rgqV0PiEnWXcEUYCop3wERPAWrLZKXCcbY7evEakleg,3620
46
+ agentflow/store/memory.py,sha256=6kakPkdRa16zzzW0ob55tDyMEmIdqMwgPHEINyfW_zw,3553
47
+ agentflow/store/postgres.py,sha256=4R4X3ZP66Q6ke_HW4JsrUX0yqhMf-bfifODYgEsT_SM,7253
48
+ agent_workflow_sdk-0.1.0.dist-info/METADATA,sha256=rkpYE7cM9nsWNbBAipO6Xjj1KhDJkq5xAgc4P2wO4tk,22800
49
+ agent_workflow_sdk-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
50
+ agent_workflow_sdk-0.1.0.dist-info/licenses/LICENSE,sha256=SaffLZsgtERkFib3uMTqP2pfsENXhB2yzrbVgC405hQ,11343
51
+ agent_workflow_sdk-0.1.0.dist-info/licenses/NOTICE,sha256=2aDBngAt3PzIRSv8Av5wUqwiRRGt9ZYdQAdbZNS_6Ys,326
52
+ agent_workflow_sdk-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any