agentpause 0.2.1__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. {agentpause-0.2.1/src/agentpause.egg-info → agentpause-0.3.0}/PKG-INFO +99 -16
  2. {agentpause-0.2.1 → agentpause-0.3.0}/README.md +98 -15
  3. {agentpause-0.2.1 → agentpause-0.3.0}/pyproject.toml +1 -1
  4. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause/__init__.py +9 -1
  5. agentpause-0.3.0/src/agentpause/adapters/anthropic.py +118 -0
  6. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause/adapters/langgraph.py +11 -2
  7. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause/adapters/litellm.py +18 -2
  8. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause/adapters/openai_compat.py +78 -13
  9. agentpause-0.3.0/src/agentpause/coordinator.py +185 -0
  10. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause/estimator.py +65 -7
  11. agentpause-0.3.0/src/agentpause/regression.py +235 -0
  12. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause/risk.py +46 -9
  13. agentpause-0.3.0/src/agentpause/router.py +185 -0
  14. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause/scheduler.py +83 -6
  15. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause/state.py +47 -0
  16. agentpause-0.3.0/src/agentpause/tools.py +77 -0
  17. {agentpause-0.2.1 → agentpause-0.3.0/src/agentpause.egg-info}/PKG-INFO +99 -16
  18. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause.egg-info/SOURCES.txt +14 -0
  19. agentpause-0.3.0/tests/test_anthropic_adapter.py +118 -0
  20. agentpause-0.3.0/tests/test_coordinator.py +121 -0
  21. agentpause-0.3.0/tests/test_estimator_evolution.py +78 -0
  22. agentpause-0.3.0/tests/test_field_adoptions.py +96 -0
  23. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_io_tpm.py +1 -1
  24. agentpause-0.3.0/tests/test_latency_budget.py +77 -0
  25. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_openai_compat.py +75 -1
  26. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_quantile.py +19 -0
  27. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_refill_wait.py +11 -2
  28. agentpause-0.3.0/tests/test_regression.py +109 -0
  29. agentpause-0.3.0/tests/test_router.py +140 -0
  30. agentpause-0.3.0/tests/test_smart_context.py +109 -0
  31. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_state.py +26 -0
  32. agentpause-0.3.0/tests/test_tool_quota.py +74 -0
  33. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_wait_and_rpm.py +3 -3
  34. {agentpause-0.2.1 → agentpause-0.3.0}/LICENSE +0 -0
  35. {agentpause-0.2.1 → agentpause-0.3.0}/setup.cfg +0 -0
  36. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause/adapters/__init__.py +0 -0
  37. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause/breaker.py +0 -0
  38. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause/errors.py +0 -0
  39. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause/fallback.py +0 -0
  40. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause/refill.py +0 -0
  41. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause/retry.py +0 -0
  42. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause.egg-info/dependency_links.txt +0 -0
  43. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause.egg-info/requires.txt +0 -0
  44. {agentpause-0.2.1 → agentpause-0.3.0}/src/agentpause.egg-info/top_level.txt +0 -0
  45. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_async.py +0 -0
  46. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_breaker.py +0 -0
  47. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_estimator.py +0 -0
  48. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_fallback.py +0 -0
  49. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_langgraph_adapter.py +0 -0
  50. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_litellm_adapter.py +0 -0
  51. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_money_and_hooks.py +0 -0
  52. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_regime.py +0 -0
  53. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_risk.py +0 -0
  54. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_robustness.py +0 -0
  55. {agentpause-0.2.1 → agentpause-0.3.0}/tests/test_scheduler.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentpause
3
- Version: 0.2.1
3
+ Version: 0.3.0
4
4
  Summary: Predictive scheduling for autonomous LLM agents: suspend gracefully before rate limits, resume without losing work.
5
5
  Author: Andrea Lelio Ciampolillo
6
6
  License: MIT
@@ -41,25 +41,48 @@ plugin for self-hosted runtimes (llama.cpp, vLLM).
41
41
 
42
42
  Reproduce them yourself with a free key: `python scripts/benchmark_groq.py`.
43
43
  Live run (2026-07-08, Groq free tier, 12 steps, ~1k-token context,
44
- refill-aware waiting enabled):
44
+ refill-aware chunked waiting, leveled windows):
45
45
 
46
46
  | | reactive baseline | agentpause |
47
47
  |---|---|---|
48
48
  | 429 errors suffered | 7 | **0** |
49
49
  | steps redone | 7 | **0** |
50
- | tokens re-sent (waste) | 12,520 | **0** |
51
- | wall-clock | 79 s | 145 s* |
50
+ | tokens re-sent (waste) | 12,840 | **0** |
51
+ | telemetry overhead (pings) | 0 | 259 |
52
+ | wall-clock | 82 s | 86 s |
52
53
 
53
- \* zero waste costs some waiting. Refill-aware math already cuts most waits
54
- from ~50 s to ~10 s; the residue is a benchmark artifact (condition B starts
55
- right after A has drained the shared TPM window) plus the safety cap when a
56
- late-task call needs nearly the whole bucket. Waste is money on paid tiers —
57
- waiting is free.
54
+ Zero errors and zero waste at wall-clock parity: the sensor costs ~2% of
55
+ what crashes waste (259 vs 12,840 tokens) and waste is money on paid tiers.
58
56
 
59
- > Status: **early alpha (0.1)**. Core components, the high-level
60
- > `PredictiveScheduler` API, the LiteLLM adapter (any provider), the LangGraph
61
- > adapter, runnable examples, and a full test suite are in place; the optional
62
- > KV-cache plugin is next.
57
+ **Context slimming vs. answer quality** (`python scripts/quality_ab.py`,
58
+ live 2026-07-10, Groq `llama-3.1-8b-instant`): six facts planted early in a
59
+ verbose conversation, same final quiz under three histories.
60
+
61
+ | condition | prompt chars | facts recalled |
62
+ |---|---|---|
63
+ | A — full history | 8,984 | **6/6** |
64
+ | B — `compact()` (blind truncation) | 4,370 | **0/6** — and the model *invented* plausible replacements |
65
+ | C — `summarize_with()` (one summary call) | 3,608 | **6/6** |
66
+
67
+ A 2.2× larger run (`--big`, 19,145 chars — near the free tier's whole TPM
68
+ window, the physical ceiling for a single call) repeated the pattern:
69
+ A 6/6, B 0/6, C 6/6 at a third of the prompt.
70
+
71
+ The instructive failure is B's, and it is *erratic*: in one run it answered
72
+ confidently with fabricated values (fake codename, fake budget, fake city);
73
+ in the larger run it honestly declined. You cannot know in advance which
74
+ failure you get — and the hallucinating one is the dangerous one. Blind
75
+ truncation is an emergency exit (§8.6 overflow, no LLM available), not a
76
+ strategy; semantic summarization recalled everything at a fraction of the
77
+ prompt, earning its one extra call.
78
+ A 20-step live stress test (2026-07-10) closed the loop: at the §8.6 wall the
79
+ scheduler suspended, compacted the checkpoint offline, resumed slim, and
80
+ completed 20/20 steps with zero 429s.
81
+
82
+ > Status: **0.3.0**. Core scheduler, direct + LiteLLM + Anthropic adapters,
83
+ > LangGraph integration, multi-provider routing, multi-agent shared budgets,
84
+ > feature-based cost/latency estimation, runnable examples, and a full test
85
+ > suite are in place; the optional KV-cache plugin is next.
63
86
 
64
87
  ## Quick example
65
88
 
@@ -182,6 +205,54 @@ Resume the paused thread with `graph.invoke(Command(resume=True), config)`.
182
205
  On resume the guard re-reads telemetry fresh — never from the checkpoint.
183
206
  Install with `pip install -e ".[langgraph]"`; see `examples/langgraph_quickstart.py`.
184
207
 
208
+ ## Scaling up: many providers, many agents, smarter estimates (v0.3)
209
+
210
+ The same predictive idea — read the budget first, act before the error —
211
+ extends in three directions, and they compose:
212
+
213
+ ```python
214
+ from agentpause import BudgetRouter, MultiAgentCoordinator, FeatureEstimator
215
+ from agentpause.adapters.openai_compat import OpenAICompatAdapter
216
+ from agentpause.adapters.anthropic import AnthropicAdapter
217
+
218
+ # 1. Route each call to the provider with the most headroom (predictive
219
+ # fallback: switch BEFORE the 429, not after). Providers in cooldown
220
+ # after a real 429 are skipped until their window resets.
221
+ router = BudgetRouter(
222
+ ("groq", OpenAICompatAdapter.for_model("groq/llama-3.1-8b-instant")),
223
+ ("claude", AnthropicAdapter("claude-haiku-4-5")),
224
+ )
225
+
226
+ # 2. Share ONE rate-limit window across a fleet: every granted call
227
+ # reserves its predicted cost (estimate + k·σ) from the shared pool,
228
+ # so agents can't overcommit the window together. Contention is
229
+ # arbitrated by priority, then longest-waiting.
230
+ coord = MultiAgentCoordinator(telemetry=router.budget)
231
+ coord.register("researcher", priority=1)
232
+ coord.register("summarizer")
233
+
234
+ est = FeatureEstimator() # 3. learn cost from features,
235
+ est.set_context(tool="web_search") # not just context size
236
+
237
+ d = coord.request("researcher", estimated=est.estimate(1200),
238
+ sigma=est.sigma(fallback_estimate=1200))
239
+ if d.action == "continue":
240
+ reply, used = router.backend(messages) # router picks the provider
241
+ coord.complete("researcher", actual_tokens=used)
242
+ est.record(1200, used)
243
+ ```
244
+
245
+ `FeatureEstimator` is a drop-in for the default estimator
246
+ (`PredictiveScheduler(estimator=FeatureEstimator())`): a dependency-free
247
+ ridge regression over features you declare (tool, model, temperature, …)
248
+ that also learns per-step **latency**, feeding the optional time budget
249
+ (`PredictiveScheduler(time_budget_s=...)`) — if the predicted step can't
250
+ finish before the deadline, the answer is checkpoint, never wait (time,
251
+ unlike tokens, does not refill).
252
+
253
+ Runnable demo without any key: `python examples/fleet_quickstart.py`
254
+ (routing switch + predictive WAIT under a shared window, in 30 lines of loop).
255
+
185
256
  ## Why
186
257
 
187
258
  Current agent frameworks persist state *reactively*: they checkpoint after a step
@@ -193,16 +264,23 @@ from the exact step.
193
264
  This library is the engineering counterpart of the research preprint *"A
194
265
  Resource-Aware Predictive Scheduler for Autonomous LLM Agents"*.
195
266
 
196
- ## Components (v0.1)
267
+ ## Components
197
268
 
198
269
  | Module | Role |
199
270
  |--------|------|
200
271
  | `PredictiveScheduler` / `Session` | the high-level API: `session()`, `should_suspend()`, `call()`, `checkpoint()` |
201
272
  | `Estimator` | predicts next-step token cost with a moving-average error correction (ε) and tracks σ |
273
+ | `FeatureEstimator` | drop-in replacement that learns cost *and latency* from declared features (tool, model, …) via dependency-free ridge regression |
202
274
  | `should_checkpoint` / `RiskModel` | the suspension rule (`remaining < estimated + k·σ`) and a diagnostic risk score |
203
- | `Budget` / `decide` | three-dimensional telemetry (TPM, RPM, reset time) and the three-valued rule: continue / wait / checkpoint |
204
- | `StateStore` / `Checkpoint` | atomic logical checkpointing with idempotency keys — works on any provider |
275
+ | `Budget` / `decide` | multi-dimensional telemetry (TPM, RPM, reset time, deadline) and the three-valued rule: continue / wait / checkpoint |
276
+ | `StateStore` / `Checkpoint` | atomic logical checkpointing with idempotency keys — works on any provider; `compact()` / `summarize_with()` shrink a suspended checkpoint offline |
277
+ | `BudgetRouter` | predictive multi-provider routing: reads every provider's budget first, routes to the most headroom, cools down 429'd providers |
278
+ | `MultiAgentCoordinator` | one shared rate-limit window across many agents: granted calls reserve their predicted cost; `arbitrate()` resolves contention by priority + fairness |
279
+ | `ToolQuota` | client-side sliding window for rate-limited tools that expose no headers |
280
+ | `CircuitBreaker` / `FallbackBackend` | reactive safety nets: fail fast on a broken provider, try the next one in order |
205
281
  | `adapters.litellm.LiteLLMAdapter` | backend + telemetry for any LiteLLM-supported provider (headers → budget, stale reading → 1-token ping) |
282
+ | `adapters.openai_compat.OpenAICompatAdapter` | direct HTTP adapter for OpenAI-compatible APIs (Groq, OpenAI, …) — no litellm dependency |
283
+ | `adapters.anthropic.AnthropicAdapter` | direct adapter for the Anthropic Messages API, with `cache_control` prompt caching and measured `cache_read/write_tokens` |
206
284
  | `adapters.langgraph.AgentPauseGuard` | predictive gate for LangGraph nodes: `check()` interrupts the graph before the fatal call, `record()` trains the estimator |
207
285
 
208
286
  ## Install (from source, during development)
@@ -221,6 +299,11 @@ pytest
221
299
  - [x] Runnable quickstart example (no keys)
222
300
  - [x] LiteLLM adapter (works with any provider)
223
301
  - [x] LangGraph adapter (interrupt + checkpointer)
302
+ - [x] Direct adapters (OpenAI-compatible, Anthropic with prompt caching)
303
+ - [x] Predictive multi-provider routing (`BudgetRouter`)
304
+ - [x] Shared budget across agents (`MultiAgentCoordinator`)
305
+ - [x] Feature-based cost & latency estimator (`FeatureEstimator`)
306
+ - [ ] CrewAI / AutoGen / LlamaIndex adapters
224
307
  - [ ] Optional KV-cache plugin for llama.cpp / vLLM (true warm start)
225
308
 
226
309
  ## License
@@ -18,25 +18,48 @@ plugin for self-hosted runtimes (llama.cpp, vLLM).
18
18
 
19
19
  Reproduce them yourself with a free key: `python scripts/benchmark_groq.py`.
20
20
  Live run (2026-07-08, Groq free tier, 12 steps, ~1k-token context,
21
- refill-aware waiting enabled):
21
+ refill-aware chunked waiting, leveled windows):
22
22
 
23
23
  | | reactive baseline | agentpause |
24
24
  |---|---|---|
25
25
  | 429 errors suffered | 7 | **0** |
26
26
  | steps redone | 7 | **0** |
27
- | tokens re-sent (waste) | 12,520 | **0** |
28
- | wall-clock | 79 s | 145 s* |
27
+ | tokens re-sent (waste) | 12,840 | **0** |
28
+ | telemetry overhead (pings) | 0 | 259 |
29
+ | wall-clock | 82 s | 86 s |
29
30
 
30
- \* zero waste costs some waiting. Refill-aware math already cuts most waits
31
- from ~50 s to ~10 s; the residue is a benchmark artifact (condition B starts
32
- right after A has drained the shared TPM window) plus the safety cap when a
33
- late-task call needs nearly the whole bucket. Waste is money on paid tiers —
34
- waiting is free.
31
+ Zero errors and zero waste at wall-clock parity: the sensor costs ~2% of
32
+ what crashes waste (259 vs 12,840 tokens) and waste is money on paid tiers.
35
33
 
36
- > Status: **early alpha (0.1)**. Core components, the high-level
37
- > `PredictiveScheduler` API, the LiteLLM adapter (any provider), the LangGraph
38
- > adapter, runnable examples, and a full test suite are in place; the optional
39
- > KV-cache plugin is next.
34
+ **Context slimming vs. answer quality** (`python scripts/quality_ab.py`,
35
+ live 2026-07-10, Groq `llama-3.1-8b-instant`): six facts planted early in a
36
+ verbose conversation, same final quiz under three histories.
37
+
38
+ | condition | prompt chars | facts recalled |
39
+ |---|---|---|
40
+ | A — full history | 8,984 | **6/6** |
41
+ | B — `compact()` (blind truncation) | 4,370 | **0/6** — and the model *invented* plausible replacements |
42
+ | C — `summarize_with()` (one summary call) | 3,608 | **6/6** |
43
+
44
+ A 2.2× larger run (`--big`, 19,145 chars — near the free tier's whole TPM
45
+ window, the physical ceiling for a single call) repeated the pattern:
46
+ A 6/6, B 0/6, C 6/6 at a third of the prompt.
47
+
48
+ The instructive failure is B's, and it is *erratic*: in one run it answered
49
+ confidently with fabricated values (fake codename, fake budget, fake city);
50
+ in the larger run it honestly declined. You cannot know in advance which
51
+ failure you get — and the hallucinating one is the dangerous one. Blind
52
+ truncation is an emergency exit (§8.6 overflow, no LLM available), not a
53
+ strategy; semantic summarization recalled everything at a fraction of the
54
+ prompt, earning its one extra call.
55
+ A 20-step live stress test (2026-07-10) closed the loop: at the §8.6 wall the
56
+ scheduler suspended, compacted the checkpoint offline, resumed slim, and
57
+ completed 20/20 steps with zero 429s.
58
+
59
+ > Status: **0.3.0**. Core scheduler, direct + LiteLLM + Anthropic adapters,
60
+ > LangGraph integration, multi-provider routing, multi-agent shared budgets,
61
+ > feature-based cost/latency estimation, runnable examples, and a full test
62
+ > suite are in place; the optional KV-cache plugin is next.
40
63
 
41
64
  ## Quick example
42
65
 
@@ -159,6 +182,54 @@ Resume the paused thread with `graph.invoke(Command(resume=True), config)`.
159
182
  On resume the guard re-reads telemetry fresh — never from the checkpoint.
160
183
  Install with `pip install -e ".[langgraph]"`; see `examples/langgraph_quickstart.py`.
161
184
 
185
+ ## Scaling up: many providers, many agents, smarter estimates (v0.3)
186
+
187
+ The same predictive idea — read the budget first, act before the error —
188
+ extends in three directions, and they compose:
189
+
190
+ ```python
191
+ from agentpause import BudgetRouter, MultiAgentCoordinator, FeatureEstimator
192
+ from agentpause.adapters.openai_compat import OpenAICompatAdapter
193
+ from agentpause.adapters.anthropic import AnthropicAdapter
194
+
195
+ # 1. Route each call to the provider with the most headroom (predictive
196
+ # fallback: switch BEFORE the 429, not after). Providers in cooldown
197
+ # after a real 429 are skipped until their window resets.
198
+ router = BudgetRouter(
199
+ ("groq", OpenAICompatAdapter.for_model("groq/llama-3.1-8b-instant")),
200
+ ("claude", AnthropicAdapter("claude-haiku-4-5")),
201
+ )
202
+
203
+ # 2. Share ONE rate-limit window across a fleet: every granted call
204
+ # reserves its predicted cost (estimate + k·σ) from the shared pool,
205
+ # so agents can't overcommit the window together. Contention is
206
+ # arbitrated by priority, then longest-waiting.
207
+ coord = MultiAgentCoordinator(telemetry=router.budget)
208
+ coord.register("researcher", priority=1)
209
+ coord.register("summarizer")
210
+
211
+ est = FeatureEstimator() # 3. learn cost from features,
212
+ est.set_context(tool="web_search") # not just context size
213
+
214
+ d = coord.request("researcher", estimated=est.estimate(1200),
215
+ sigma=est.sigma(fallback_estimate=1200))
216
+ if d.action == "continue":
217
+ reply, used = router.backend(messages) # router picks the provider
218
+ coord.complete("researcher", actual_tokens=used)
219
+ est.record(1200, used)
220
+ ```
221
+
222
+ `FeatureEstimator` is a drop-in for the default estimator
223
+ (`PredictiveScheduler(estimator=FeatureEstimator())`): a dependency-free
224
+ ridge regression over features you declare (tool, model, temperature, …)
225
+ that also learns per-step **latency**, feeding the optional time budget
226
+ (`PredictiveScheduler(time_budget_s=...)`) — if the predicted step can't
227
+ finish before the deadline, the answer is checkpoint, never wait (time,
228
+ unlike tokens, does not refill).
229
+
230
+ Runnable demo without any key: `python examples/fleet_quickstart.py`
231
+ (routing switch + predictive WAIT under a shared window, in 30 lines of loop).
232
+
162
233
  ## Why
163
234
 
164
235
  Current agent frameworks persist state *reactively*: they checkpoint after a step
@@ -170,16 +241,23 @@ from the exact step.
170
241
  This library is the engineering counterpart of the research preprint *"A
171
242
  Resource-Aware Predictive Scheduler for Autonomous LLM Agents"*.
172
243
 
173
- ## Components (v0.1)
244
+ ## Components
174
245
 
175
246
  | Module | Role |
176
247
  |--------|------|
177
248
  | `PredictiveScheduler` / `Session` | the high-level API: `session()`, `should_suspend()`, `call()`, `checkpoint()` |
178
249
  | `Estimator` | predicts next-step token cost with a moving-average error correction (ε) and tracks σ |
250
+ | `FeatureEstimator` | drop-in replacement that learns cost *and latency* from declared features (tool, model, …) via dependency-free ridge regression |
179
251
  | `should_checkpoint` / `RiskModel` | the suspension rule (`remaining < estimated + k·σ`) and a diagnostic risk score |
180
- | `Budget` / `decide` | three-dimensional telemetry (TPM, RPM, reset time) and the three-valued rule: continue / wait / checkpoint |
181
- | `StateStore` / `Checkpoint` | atomic logical checkpointing with idempotency keys — works on any provider |
252
+ | `Budget` / `decide` | multi-dimensional telemetry (TPM, RPM, reset time, deadline) and the three-valued rule: continue / wait / checkpoint |
253
+ | `StateStore` / `Checkpoint` | atomic logical checkpointing with idempotency keys — works on any provider; `compact()` / `summarize_with()` shrink a suspended checkpoint offline |
254
+ | `BudgetRouter` | predictive multi-provider routing: reads every provider's budget first, routes to the most headroom, cools down 429'd providers |
255
+ | `MultiAgentCoordinator` | one shared rate-limit window across many agents: granted calls reserve their predicted cost; `arbitrate()` resolves contention by priority + fairness |
256
+ | `ToolQuota` | client-side sliding window for rate-limited tools that expose no headers |
257
+ | `CircuitBreaker` / `FallbackBackend` | reactive safety nets: fail fast on a broken provider, try the next one in order |
182
258
  | `adapters.litellm.LiteLLMAdapter` | backend + telemetry for any LiteLLM-supported provider (headers → budget, stale reading → 1-token ping) |
259
+ | `adapters.openai_compat.OpenAICompatAdapter` | direct HTTP adapter for OpenAI-compatible APIs (Groq, OpenAI, …) — no litellm dependency |
260
+ | `adapters.anthropic.AnthropicAdapter` | direct adapter for the Anthropic Messages API, with `cache_control` prompt caching and measured `cache_read/write_tokens` |
183
261
  | `adapters.langgraph.AgentPauseGuard` | predictive gate for LangGraph nodes: `check()` interrupts the graph before the fatal call, `record()` trains the estimator |
184
262
 
185
263
  ## Install (from source, during development)
@@ -198,6 +276,11 @@ pytest
198
276
  - [x] Runnable quickstart example (no keys)
199
277
  - [x] LiteLLM adapter (works with any provider)
200
278
  - [x] LangGraph adapter (interrupt + checkpointer)
279
+ - [x] Direct adapters (OpenAI-compatible, Anthropic with prompt caching)
280
+ - [x] Predictive multi-provider routing (`BudgetRouter`)
281
+ - [x] Shared budget across agents (`MultiAgentCoordinator`)
282
+ - [x] Feature-based cost & latency estimator (`FeatureEstimator`)
283
+ - [ ] CrewAI / AutoGen / LlamaIndex adapters
201
284
  - [ ] Optional KV-cache plugin for llama.cpp / vLLM (true warm start)
202
285
 
203
286
  ## License
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "agentpause"
7
- version = "0.2.1"
7
+ version = "0.3.0"
8
8
  description = "Predictive scheduling for autonomous LLM agents: suspend gracefully before rate limits, resume without losing work."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.9"
@@ -14,19 +14,24 @@ from .errors import (
14
14
  )
15
15
  from .breaker import CircuitBreaker, CircuitOpenError
16
16
  from .estimator import Estimator
17
+ from .regression import FeatureEstimator
17
18
  from .fallback import FallbackBackend
18
19
  from .refill import RegimeDetector
19
20
  from .retry import RetryPolicy
20
21
  from .risk import Budget, Decision, RiskModel, decide, should_checkpoint
22
+ from .router import BudgetRouter
23
+ from .coordinator import MultiAgentCoordinator
21
24
  from .state import Checkpoint, StateStore
25
+ from .tools import ToolQuota
22
26
  from .scheduler import PredictiveScheduler, Session
23
27
 
24
- __version__ = "0.2.1"
28
+ __version__ = "0.3.0"
25
29
 
26
30
  __all__ = [
27
31
  "PredictiveScheduler",
28
32
  "Session",
29
33
  "Estimator",
34
+ "FeatureEstimator",
30
35
  "RiskModel",
31
36
  "Budget",
32
37
  "Decision",
@@ -36,9 +41,12 @@ __all__ = [
36
41
  "StateStore",
37
42
  "RetryPolicy",
38
43
  "FallbackBackend",
44
+ "BudgetRouter",
45
+ "MultiAgentCoordinator",
39
46
  "CircuitBreaker",
40
47
  "CircuitOpenError",
41
48
  "RegimeDetector",
49
+ "ToolQuota",
42
50
  "AgentPauseError",
43
51
  "RateLimitHit",
44
52
  "TelemetryError",
@@ -0,0 +1,118 @@
1
+ """Direct adapter for Anthropic's Messages API (F9.6), cache-aware (F9.5).
2
+
3
+ Anthropic is not OpenAI-compatible — different endpoint (``/v1/messages``),
4
+ different auth (``x-api-key``), ``system`` as a top-level field, usage split
5
+ into input/output tokens — so it gets its own thin adapter, built on the
6
+ same telemetry machinery as :class:`OpenAICompatAdapter` (whose header
7
+ parser already understands ``anthropic-ratelimit-*`` names and RFC 3339
8
+ resets).
9
+
10
+ Prompt caching — the cloud analog of the KV warm start
11
+ ------------------------------------------------------
12
+ True KV-cache export is impossible on cloud APIs, but Anthropic caches
13
+ byte-identical prompt prefixes server-side (cache reads cost ~10% of the
14
+ input price). agentpause's resume already re-sends the prefix verbatim by
15
+ design, so with ``cache_prompt=True`` (default) the adapter plants
16
+ ``cache_control`` breakpoints on the stable prefix — the system message and
17
+ the last message before the fresh question — and MEASURES what came back
18
+ from cache (``cache_read_tokens``). Two honest caveats: the cache TTL is
19
+ minutes (helps waits and short suspensions, not tomorrow's resume), and
20
+ cache reads discount price/latency, not the rate-limit token count.
21
+
22
+ from agentpause.adapters.anthropic import AnthropicAdapter
23
+
24
+ adapter = AnthropicAdapter("claude-haiku-4-5")
25
+ sched = PredictiveScheduler(backend=adapter.backend, telemetry=adapter.budget)
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ from typing import Any, Dict, List, Optional, Tuple
31
+
32
+ from .openai_compat import OpenAICompatAdapter
33
+
34
+ __all__ = ["AnthropicAdapter"]
35
+
36
+ ANTHROPIC_VERSION = "2023-06-01"
37
+
38
+
39
+ class AnthropicAdapter(OpenAICompatAdapter):
40
+ """Backend + telemetry for Anthropic models, with prompt-cache breakpoints.
41
+
42
+ Args:
43
+ model: e.g. ``"claude-haiku-4-5"``.
44
+ api_key: literal key, or None to read ``ANTHROPIC_API_KEY``.
45
+ cache_prompt: plant ``cache_control`` breakpoints on the stable
46
+ prefix (system + last message before the fresh question).
47
+ default_max_tokens: the Messages API requires ``max_tokens``.
48
+ Everything else as in :class:`OpenAICompatAdapter`.
49
+ """
50
+
51
+ def __init__(
52
+ self,
53
+ model: str,
54
+ api_key: Optional[str] = None,
55
+ cache_prompt: bool = True,
56
+ default_max_tokens: int = 1024,
57
+ **kwargs: Any,
58
+ ) -> None:
59
+ super().__init__(
60
+ model,
61
+ base_url="https://api.anthropic.com/v1",
62
+ api_key=api_key,
63
+ api_key_env="ANTHROPIC_API_KEY",
64
+ **kwargs,
65
+ )
66
+ self._auth = {
67
+ "x-api-key": self._api_key,
68
+ "anthropic-version": ANTHROPIC_VERSION,
69
+ "content-type": "application/json",
70
+ }
71
+ self.cache_prompt = cache_prompt
72
+ self.default_max_tokens = default_max_tokens
73
+ # measured cache effect (F9.5): never assumed, always counted
74
+ self.cache_read_tokens = 0
75
+ self.cache_write_tokens = 0
76
+
77
+ # -- protocol translation ---------------------------------------------------
78
+
79
+ def _call(self, messages: List[Dict[str, str]], **extra: Any) -> Dict[str, Any]:
80
+ system: Optional[str] = None
81
+ msgs = messages
82
+ if msgs and msgs[0].get("role") == "system":
83
+ system = msgs[0].get("content") or ""
84
+ msgs = msgs[1:]
85
+ payload: Dict[str, Any] = {
86
+ "model": self.model,
87
+ "max_tokens": extra.pop("max_tokens", self.default_max_tokens),
88
+ "messages": [dict(m) for m in msgs],
89
+ **extra,
90
+ }
91
+ if system is not None:
92
+ block: Dict[str, Any] = {"type": "text", "text": system}
93
+ if self.cache_prompt:
94
+ block["cache_control"] = {"type": "ephemeral"}
95
+ payload["system"] = [block]
96
+ if self.cache_prompt and len(payload["messages"]) >= 2:
97
+ # the stable prefix ends just before the fresh question: closing
98
+ # the breakpoint there lets a resumed (byte-identical) history
99
+ # hit the server-side cache
100
+ closer = payload["messages"][-2]
101
+ content = closer.get("content")
102
+ if isinstance(content, str):
103
+ closer["content"] = [{"type": "text", "text": content,
104
+ "cache_control": {"type": "ephemeral"}}]
105
+ body = self._post_and_check(f"{self.base_url}/messages", payload)
106
+ usage = body.get("usage") or {}
107
+ self.cache_read_tokens += int(usage.get("cache_read_input_tokens") or 0)
108
+ self.cache_write_tokens += int(usage.get("cache_creation_input_tokens") or 0)
109
+ return body
110
+
111
+ def backend(self, messages: List[Dict[str, str]]) -> Tuple[str, int]:
112
+ body = self._call(messages, **self.request_kwargs)
113
+ self._prev_ping = self._sample_point() # real call starts a sampling chain
114
+ text = "".join(b.get("text", "") for b in (body.get("content") or [])
115
+ if b.get("type") == "text")
116
+ usage = body.get("usage") or {}
117
+ used = int((usage.get("input_tokens") or 0) + (usage.get("output_tokens") or 0))
118
+ return text, used
@@ -98,6 +98,7 @@ class AgentPauseGuard:
98
98
  sleep_fn: Callable[[float], None] = time.sleep,
99
99
  async_sleep_fn: Optional[Callable[[float], Any]] = None,
100
100
  chunk_s: float = 10.0,
101
+ while_waiting: Optional[Callable[[float], None]] = None,
101
102
  max_resumes: int = 100,
102
103
  ) -> None:
103
104
  self.telemetry = telemetry
@@ -109,6 +110,7 @@ class AgentPauseGuard:
109
110
  self.sleep_fn = sleep_fn
110
111
  self.async_sleep_fn = async_sleep_fn
111
112
  self.chunk_s = chunk_s
113
+ self.while_waiting = while_waiting
112
114
  self.max_resumes = max_resumes
113
115
 
114
116
  # -- internals -----------------------------------------------------------
@@ -149,8 +151,15 @@ class AgentPauseGuard:
149
151
  # reset is imminent: sleeping beats a suspend/resume cycle.
150
152
  # Sleep in chunks and re-read: if the bucket refills sooner,
151
153
  # resume sooner (and the samples feed regime detection).
152
- wait_s = d.wait_seconds or budget.reset_seconds or 1.0
153
- self.sleep_fn(min(wait_s, self.chunk_s) + 0.5)
154
+ # Useful waits (F9.2): if the app registered while_waiting,
155
+ # hand it the time — indexing, memory compression, prompt
156
+ # prep are all work that needs no LLM.
157
+ wait_s = min(d.wait_seconds or budget.reset_seconds or 1.0,
158
+ self.chunk_s) + 0.5
159
+ if self.while_waiting is not None:
160
+ self.while_waiting(wait_s)
161
+ else:
162
+ self.sleep_fn(wait_s)
154
163
  continue
155
164
  # interrupt() raises on the first pass (the node aborts here);
156
165
  # on a resume pass it RETURNS, and the loop re-reads telemetry.
@@ -220,11 +220,23 @@ class LiteLLMAdapter:
220
220
  return Budget(
221
221
  remaining_tokens=await self.atelemetry(),
222
222
  remaining_requests=self._remaining_requests,
223
- reset_seconds=self._reset_seconds,
223
+ reset_seconds=self._aged_reset(),
224
224
  remaining_input_tokens=self._remaining_input,
225
225
  remaining_output_tokens=self._remaining_output,
226
226
  )
227
227
 
228
+ def _aged_reset(self) -> Optional[float]:
229
+ """Reset countdown advanced by the age of the cached reading.
230
+
231
+ The header value was true at capture time; served from cache up to
232
+ ``max_age_s`` later, it overstates the wait. Subtract the elapsed
233
+ time (floor 0) so callers never wait on already-elapsed seconds.
234
+ """
235
+ if self._reset_seconds is None or self._read_at is None:
236
+ return self._reset_seconds
237
+ age = max(0.0, self._clock() - self._read_at)
238
+ return max(0.0, self._reset_seconds - age)
239
+
228
240
  def count_tokens(self, text: str) -> int:
229
241
  """Per-model token count via litellm, with a safe heuristic fallback.
230
242
 
@@ -252,13 +264,17 @@ class LiteLLMAdapter:
252
264
  return Budget(
253
265
  remaining_tokens=self.telemetry(),
254
266
  remaining_requests=self._remaining_requests,
255
- reset_seconds=self._reset_seconds,
267
+ reset_seconds=self._aged_reset(),
256
268
  remaining_input_tokens=self._remaining_input,
257
269
  remaining_output_tokens=self._remaining_output,
258
270
  )
259
271
 
260
272
  # -- internals -------------------------------------------------------------
261
273
 
274
+ def invalidate(self) -> None:
275
+ """Mark the cached telemetry stale; the next read will ping."""
276
+ self._read_at = None
277
+
262
278
  def ping(self) -> None:
263
279
  """Issue a deliberately tiny call whose only purpose is fresh headers."""
264
280
  response = self._call(