retryhop 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- retryhop-0.1.0/.github/workflows/ci.yml +19 -0
- retryhop-0.1.0/.github/workflows/release.yml +34 -0
- retryhop-0.1.0/.gitignore +7 -0
- retryhop-0.1.0/CHANGELOG.md +17 -0
- retryhop-0.1.0/LICENSE +21 -0
- retryhop-0.1.0/PKG-INFO +259 -0
- retryhop-0.1.0/README.md +230 -0
- retryhop-0.1.0/README.zh-CN.md +187 -0
- retryhop-0.1.0/examples/batch_classify.py +59 -0
- retryhop-0.1.0/pyproject.toml +42 -0
- retryhop-0.1.0/src/retryhop/__init__.py +36 -0
- retryhop-0.1.0/src/retryhop/backoff.py +57 -0
- retryhop-0.1.0/src/retryhop/circuit.py +87 -0
- retryhop-0.1.0/src/retryhop/core.py +263 -0
- retryhop-0.1.0/src/retryhop/llm.py +188 -0
- retryhop-0.1.0/src/retryhop/py.typed +0 -0
- retryhop-0.1.0/src/retryhop/ratelimit.py +136 -0
- retryhop-0.1.0/tests/test_llm.py +233 -0
- retryhop-0.1.0/tests/test_ratelimit.py +78 -0
- retryhop-0.1.0/tests/test_retry.py +137 -0
- retryhop-0.1.0/tests/test_sdk_exceptions.py +82 -0
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
name: ci
|
|
2
|
+
on:
|
|
3
|
+
push:
|
|
4
|
+
branches: [main]
|
|
5
|
+
pull_request:
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
test:
|
|
9
|
+
runs-on: ubuntu-latest
|
|
10
|
+
strategy:
|
|
11
|
+
matrix:
|
|
12
|
+
python-version: ["3.9", "3.10", "3.11", "3.12", "3.13"]
|
|
13
|
+
steps:
|
|
14
|
+
- uses: actions/checkout@v4
|
|
15
|
+
- uses: actions/setup-python@v5
|
|
16
|
+
with:
|
|
17
|
+
python-version: ${{ matrix.python-version }}
|
|
18
|
+
- run: pip install -e ".[sdk-test]"
|
|
19
|
+
- run: pytest -q
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
name: release
|
|
2
|
+
on:
|
|
3
|
+
push:
|
|
4
|
+
tags: ["v*"]
|
|
5
|
+
|
|
6
|
+
jobs:
|
|
7
|
+
build:
|
|
8
|
+
runs-on: ubuntu-latest
|
|
9
|
+
steps:
|
|
10
|
+
- uses: actions/checkout@v4
|
|
11
|
+
- uses: actions/setup-python@v5
|
|
12
|
+
with:
|
|
13
|
+
python-version: "3.12"
|
|
14
|
+
- run: pip install build twine pytest
|
|
15
|
+
- run: pip install -e . && pytest -q
|
|
16
|
+
- run: python -m build
|
|
17
|
+
- run: twine check dist/*
|
|
18
|
+
- uses: actions/upload-artifact@v4
|
|
19
|
+
with:
|
|
20
|
+
name: dist
|
|
21
|
+
path: dist/
|
|
22
|
+
|
|
23
|
+
publish:
|
|
24
|
+
needs: build
|
|
25
|
+
runs-on: ubuntu-latest
|
|
26
|
+
environment: pypi
|
|
27
|
+
permissions:
|
|
28
|
+
id-token: write # required for PyPI Trusted Publishing
|
|
29
|
+
steps:
|
|
30
|
+
- uses: actions/download-artifact@v4
|
|
31
|
+
with:
|
|
32
|
+
name: dist
|
|
33
|
+
path: dist/
|
|
34
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.1.0
|
|
4
|
+
|
|
5
|
+
First release.
|
|
6
|
+
|
|
7
|
+
- `llm_retry`: retry decorator for LLM and HTTP API calls. Retries only
|
|
8
|
+
transient errors, uses `Retry-After` / `retry-after-ms` when present, and
|
|
9
|
+
re-raises the original exception on give-up.
|
|
10
|
+
- `is_transient`, `retry_after_of`, `status_code_of`, `describe`. Exceptions
|
|
11
|
+
are matched by attribute and class name. Tested against openai, anthropic,
|
|
12
|
+
httpx and requests; aiohttp is not tested yet.
|
|
13
|
+
- `RateLimiter`: requests-per-minute and tokens-per-minute buckets, usable
|
|
14
|
+
from threads and asyncio, with `consume()` to adjust token estimates.
|
|
15
|
+
- `CircuitBreaker` with closed / open / half-open states.
|
|
16
|
+
- `retry` decorator (sync and async), `retry_call`, and `constant` / `linear` /
|
|
17
|
+
`exponential` backoff.
|
retryhop-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Liang Yan
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
retryhop-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,259 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: retryhop
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Retries, rate limiting and circuit breaking for LLM and HTTP API calls (OpenAI, Anthropic, httpx, requests).
|
|
5
|
+
Project-URL: Homepage, https://github.com/liayan/retryhop
|
|
6
|
+
Project-URL: Source, https://github.com/liayan/retryhop
|
|
7
|
+
Project-URL: Issues, https://github.com/liayan/retryhop/issues
|
|
8
|
+
Author-email: Liang Yan <26383278+liayan@users.noreply.github.com>
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: anthropic,asyncio,backoff,circuit-breaker,llm,openai,rate-limit,retry,retry-after
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Framework :: AsyncIO
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
18
|
+
Classifier: Typing :: Typed
|
|
19
|
+
Requires-Python: >=3.9
|
|
20
|
+
Provides-Extra: sdk-test
|
|
21
|
+
Requires-Dist: anthropic; extra == 'sdk-test'
|
|
22
|
+
Requires-Dist: httpx; extra == 'sdk-test'
|
|
23
|
+
Requires-Dist: openai; extra == 'sdk-test'
|
|
24
|
+
Requires-Dist: pytest>=7; extra == 'sdk-test'
|
|
25
|
+
Requires-Dist: requests; extra == 'sdk-test'
|
|
26
|
+
Provides-Extra: test
|
|
27
|
+
Requires-Dist: pytest>=7; extra == 'test'
|
|
28
|
+
Description-Content-Type: text/markdown
|
|
29
|
+
|
|
30
|
+
# retryhop
|
|
31
|
+
|
|
32
|
+
English | [简体中文](https://github.com/liayan/retryhop/blob/main/README.zh-CN.md)
|
|
33
|
+
|
|
34
|
+
Retries, client-side rate limiting and a circuit breaker for LLM and HTTP API
|
|
35
|
+
calls. No runtime dependencies. Works on sync and async functions.
|
|
36
|
+
|
|
37
|
+
Errors are classified by attribute (`status_code`, `status`,
|
|
38
|
+
`response.headers`) and by class name, so retryhop does not import any SDK.
|
|
39
|
+
The tests use real exception objects from `openai`, `anthropic`, `httpx` and
|
|
40
|
+
`requests`. `aiohttp` errors go through the same checks but are not covered by
|
|
41
|
+
tests yet.
|
|
42
|
+
|
|
43
|
+
Status: alpha. The API may change before 1.0.
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
pip install retryhop
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Python 3.9+.
|
|
50
|
+
|
|
51
|
+
## Usage
|
|
52
|
+
|
|
53
|
+
```python
|
|
54
|
+
from openai import OpenAI
|
|
55
|
+
from retryhop import llm_retry
|
|
56
|
+
|
|
57
|
+
client = OpenAI(max_retries=0)
|
|
58
|
+
|
|
59
|
+
@llm_retry() # defaults: 6 attempts, 300 s deadline
|
|
60
|
+
def ask(prompt: str) -> str:
|
|
61
|
+
r = client.chat.completions.create(
|
|
62
|
+
model=MODEL,
|
|
63
|
+
messages=[{"role": "user", "content": prompt}],
|
|
64
|
+
)
|
|
65
|
+
return r.choices[0].message.content
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Async functions work the same way:
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
from anthropic import AsyncAnthropic
|
|
72
|
+
from retryhop import llm_retry
|
|
73
|
+
|
|
74
|
+
client = AsyncAnthropic(max_retries=0)
|
|
75
|
+
|
|
76
|
+
@llm_retry(attempts=8, deadline=600)
|
|
77
|
+
async def ask(prompt: str) -> str:
|
|
78
|
+
msg = await client.messages.create(
|
|
79
|
+
model=MODEL, max_tokens=1024,
|
|
80
|
+
messages=[{"role": "user", "content": prompt}],
|
|
81
|
+
)
|
|
82
|
+
return msg.content[0].text
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
Create the SDK client with `max_retries=0`. The openai and anthropic clients
|
|
86
|
+
retry twice by default, so otherwise each retryhop attempt can be up to three
|
|
87
|
+
HTTP requests.
|
|
88
|
+
|
|
89
|
+
When retryhop gives up it re-raises the last exception, so existing handlers
|
|
90
|
+
such as `except openai.RateLimitError:` still work. Pass `reraise=False` to
|
|
91
|
+
get a `RetryError` instead.
|
|
92
|
+
|
|
93
|
+
## What is retried
|
|
94
|
+
|
|
95
|
+
`is_transient(exc)` makes the decision. It checks, in order:
|
|
96
|
+
|
|
97
|
+
1. The `x-should-retry` response header, if the server sent one.
|
|
98
|
+
2. The HTTP status. 408, 409, 429, 500, 502, 503, 504 and 529 are retried.
|
|
99
|
+
Any other status is raised immediately.
|
|
100
|
+
3. If there is no status: connection errors and timeouts are retried. This
|
|
101
|
+
covers the built-in `ConnectionError` and `TimeoutError` and the matching
|
|
102
|
+
classes in openai, anthropic, httpx, requests/urllib3 and aiohttp.
|
|
103
|
+
|
|
104
|
+
Anything else is raised immediately.
|
|
105
|
+
|
|
106
|
+
409 is on the list because the openai and anthropic SDKs retry it (they treat
|
|
107
|
+
it as a lock timeout). 529 is Anthropic's "overloaded". If a 409 from your API
|
|
108
|
+
means a real conflict, use `retry(...)` with your own `retry_on`.
|
|
109
|
+
|
|
110
|
+
## Wait time
|
|
111
|
+
|
|
112
|
+
- If the error response has `retry-after-ms` or `Retry-After` (seconds or an
|
|
113
|
+
HTTP date), retryhop waits that long plus 0-0.25 s of jitter, capped at
|
|
114
|
+
`max_retry_after`.
|
|
115
|
+
- Otherwise it uses exponential backoff with full jitter: a random wait in
|
|
116
|
+
[0, 1 s], then [0, 2 s], [0, 4 s], and so on, with the upper bound capped
|
|
117
|
+
at 60 s.
|
|
118
|
+
- `deadline` is the total budget across attempts. If the server asks for a
|
|
119
|
+
wait longer than what is left, retryhop gives up without sleeping. A
|
|
120
|
+
backoff wait is cut short to fit.
|
|
121
|
+
|
|
122
|
+
The deadline is checked between attempts. It does not cancel a request that
|
|
123
|
+
is already running, so also set a request timeout on the client.
|
|
124
|
+
|
|
125
|
+
## `llm_retry` options
|
|
126
|
+
|
|
127
|
+
| Option | Default | Description |
|
|
128
|
+
|---|---|---|
|
|
129
|
+
| `attempts` | `6` | Total calls, including the first. |
|
|
130
|
+
| `deadline` | `300` | Seconds across all attempts. `None` disables it. |
|
|
131
|
+
| `max_retry_after` | `120` | Upper bound, in seconds, on a server-requested wait. |
|
|
132
|
+
| `backoff` | exponential, see above | Used when there is no retry-after header. Any `f(attempt) -> seconds`. |
|
|
133
|
+
| `circuit` | `None` | A shared `CircuitBreaker`. |
|
|
134
|
+
| `on_retry` | `None` | Called with a `RetryState` before each wait. |
|
|
135
|
+
| `reraise` | `True` | `False` raises `RetryError` instead of the last exception. |
|
|
136
|
+
| `sleep` | `None` | Replacement sleep function, for tests. |
|
|
137
|
+
|
|
138
|
+
## Logging
|
|
139
|
+
|
|
140
|
+
```python
|
|
141
|
+
import logging
|
|
142
|
+
from retryhop import llm_retry, describe
|
|
143
|
+
|
|
144
|
+
log = logging.getLogger("llm")
|
|
145
|
+
|
|
146
|
+
@llm_retry(on_retry=lambda s: log.warning(describe(s)))
|
|
147
|
+
def ask(prompt): ...
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
Output looks like:
|
|
151
|
+
|
|
152
|
+
```
|
|
153
|
+
attempt 1 failed: RateLimitError (HTTP 429); waiting 1.10s (server Retry-After), elapsed 0.0s
|
|
154
|
+
attempt 2 failed: InternalServerError (HTTP 503); waiting 1.64s (backoff), elapsed 1.1s
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
## Rate limiting
|
|
158
|
+
|
|
159
|
+
`RateLimiter` keeps one token bucket for requests per minute and one for
|
|
160
|
+
tokens per minute. A call that would go over either limit sleeps until there
|
|
161
|
+
is room.
|
|
162
|
+
|
|
163
|
+
```python
|
|
164
|
+
from retryhop import RateLimiter, llm_retry
|
|
165
|
+
|
|
166
|
+
limiter = RateLimiter(requests_per_minute=500, tokens_per_minute=200_000)
|
|
167
|
+
|
|
168
|
+
def estimate(prompt, **_):
|
|
169
|
+
return len(prompt) // 4 + 1024 # rough input estimate + max output
|
|
170
|
+
|
|
171
|
+
@llm_retry()
|
|
172
|
+
@limiter.limit(tokens=estimate)
|
|
173
|
+
def ask(prompt): ...
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
Put `@limiter.limit` below `@llm_retry` so retries are limited too.
|
|
177
|
+
|
|
178
|
+
The token count is your own estimate, and the provider counts tokens its own
|
|
179
|
+
way. If the response reports actual usage, call
|
|
180
|
+
`limiter.consume(actual - estimated)`. A negative value gives tokens back.
|
|
181
|
+
|
|
182
|
+
Without the decorator: `limiter.acquire(tokens=n)` or
|
|
183
|
+
`await limiter.acquire_async(tokens=n)`.
|
|
184
|
+
|
|
185
|
+
Things to know:
|
|
186
|
+
|
|
187
|
+
- The limiter is thread-safe and can be used from asyncio tasks, but its
|
|
188
|
+
state is per process. If several processes or hosts share one API key,
|
|
189
|
+
split the limits between them.
|
|
190
|
+
- The buckets start full, so a full minute's quota can go out at once right
|
|
191
|
+
after startup.
|
|
192
|
+
- A single call that needs more tokens than `tokens_per_minute` raises
|
|
193
|
+
`ValueError`.
|
|
194
|
+
|
|
195
|
+
## Circuit breaker
|
|
196
|
+
|
|
197
|
+
```python
|
|
198
|
+
from retryhop import CircuitBreaker, CircuitOpenError, llm_retry
|
|
199
|
+
|
|
200
|
+
breaker = CircuitBreaker(failure_threshold=5, recovery_time=30)
|
|
201
|
+
|
|
202
|
+
@llm_retry(circuit=breaker)
|
|
203
|
+
def ask(prompt): ...
|
|
204
|
+
|
|
205
|
+
try:
|
|
206
|
+
ask("hi")
|
|
207
|
+
except CircuitOpenError as e:
|
|
208
|
+
... # e.g. fall back to another model; e.retry_in = seconds until the next trial
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
- After `failure_threshold` consecutive failed attempts the circuit opens.
|
|
212
|
+
Calls then raise `CircuitOpenError` without hitting the API.
|
|
213
|
+
- After `recovery_time` seconds one trial call is let through. Success closes
|
|
214
|
+
the circuit; failure opens it again. Other calls made during the trial get
|
|
215
|
+
`CircuitOpenError`.
|
|
216
|
+
- Failures are counted per attempt, not per call. With the defaults
|
|
217
|
+
(6 attempts, threshold 5) a single call that keeps getting 503 opens the
|
|
218
|
+
circuit by itself, and that call ends with `CircuitOpenError` rather than
|
|
219
|
+
the SDK exception.
|
|
220
|
+
- A non-retryable error such as a 400 counts as a success, because the API
|
|
221
|
+
did respond. It resets the failure count.
|
|
222
|
+
- `CircuitOpenError` is never retried.
|
|
223
|
+
- State is per process.
|
|
224
|
+
|
|
225
|
+
## Lower-level API
|
|
226
|
+
|
|
227
|
+
- `retry(...)`: the decorator `llm_retry` is built on. It takes `attempts`,
|
|
228
|
+
`exceptions`, `retry_on`, `retry_if_result`, `wait_hint`, `backoff`,
|
|
229
|
+
`deadline`, `on_retry`, `reraise`, `circuit` and `sleep`. Its defaults are
|
|
230
|
+
different: 3 attempts, any `Exception` is retried, no deadline, and
|
|
231
|
+
`RetryError` is raised on give-up.
|
|
232
|
+
|
|
233
|
+
Polling a batch job:
|
|
234
|
+
|
|
235
|
+
```python
|
|
236
|
+
@retry(attempts=60, retry_if_result=lambda job: job.status != "completed",
|
|
237
|
+
backoff=constant(30))
|
|
238
|
+
def wait_for_batch(job_id):
|
|
239
|
+
return client.batches.retrieve(job_id)
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
- `retry_call(func, *args, retry_options={...}, **kwargs)`: same as `retry`
|
|
243
|
+
without decorating.
|
|
244
|
+
- `is_transient(exc)`, `status_code_of(exc)`, `retry_after_of(exc)`: the
|
|
245
|
+
checks described above.
|
|
246
|
+
- `describe(state)`: formats a `RetryState` for logging.
|
|
247
|
+
- `constant`, `linear`, `exponential`: backoff functions.
|
|
248
|
+
|
|
249
|
+
## Development
|
|
250
|
+
|
|
251
|
+
```bash
|
|
252
|
+
pip install -e ".[sdk-test]" # ".[test]" skips the tests that need the SDKs
|
|
253
|
+
pytest
|
|
254
|
+
python -m build && twine check dist/*
|
|
255
|
+
```
|
|
256
|
+
|
|
257
|
+
## License
|
|
258
|
+
|
|
259
|
+
MIT
|
retryhop-0.1.0/README.md
ADDED
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
# retryhop
|
|
2
|
+
|
|
3
|
+
English | [简体中文](https://github.com/liayan/retryhop/blob/main/README.zh-CN.md)
|
|
4
|
+
|
|
5
|
+
Retries, client-side rate limiting and a circuit breaker for LLM and HTTP API
|
|
6
|
+
calls. No runtime dependencies. Works on sync and async functions.
|
|
7
|
+
|
|
8
|
+
Errors are classified by attribute (`status_code`, `status`,
|
|
9
|
+
`response.headers`) and by class name, so retryhop does not import any SDK.
|
|
10
|
+
The tests use real exception objects from `openai`, `anthropic`, `httpx` and
|
|
11
|
+
`requests`. `aiohttp` errors go through the same checks but are not covered by
|
|
12
|
+
tests yet.
|
|
13
|
+
|
|
14
|
+
Status: alpha. The API may change before 1.0.
|
|
15
|
+
|
|
16
|
+
```bash
|
|
17
|
+
pip install retryhop
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
Python 3.9+.
|
|
21
|
+
|
|
22
|
+
## Usage
|
|
23
|
+
|
|
24
|
+
```python
|
|
25
|
+
from openai import OpenAI
|
|
26
|
+
from retryhop import llm_retry
|
|
27
|
+
|
|
28
|
+
client = OpenAI(max_retries=0)
|
|
29
|
+
|
|
30
|
+
@llm_retry() # defaults: 6 attempts, 300 s deadline
|
|
31
|
+
def ask(prompt: str) -> str:
|
|
32
|
+
r = client.chat.completions.create(
|
|
33
|
+
model=MODEL,
|
|
34
|
+
messages=[{"role": "user", "content": prompt}],
|
|
35
|
+
)
|
|
36
|
+
return r.choices[0].message.content
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
Async functions work the same way:
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
from anthropic import AsyncAnthropic
|
|
43
|
+
from retryhop import llm_retry
|
|
44
|
+
|
|
45
|
+
client = AsyncAnthropic(max_retries=0)
|
|
46
|
+
|
|
47
|
+
@llm_retry(attempts=8, deadline=600)
|
|
48
|
+
async def ask(prompt: str) -> str:
|
|
49
|
+
msg = await client.messages.create(
|
|
50
|
+
model=MODEL, max_tokens=1024,
|
|
51
|
+
messages=[{"role": "user", "content": prompt}],
|
|
52
|
+
)
|
|
53
|
+
return msg.content[0].text
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Create the SDK client with `max_retries=0`. The openai and anthropic clients
|
|
57
|
+
retry twice by default, so otherwise each retryhop attempt can be up to three
|
|
58
|
+
HTTP requests.
|
|
59
|
+
|
|
60
|
+
When retryhop gives up it re-raises the last exception, so existing handlers
|
|
61
|
+
such as `except openai.RateLimitError:` still work. Pass `reraise=False` to
|
|
62
|
+
get a `RetryError` instead.
|
|
63
|
+
|
|
64
|
+
## What is retried
|
|
65
|
+
|
|
66
|
+
`is_transient(exc)` makes the decision. It checks, in order:
|
|
67
|
+
|
|
68
|
+
1. The `x-should-retry` response header, if the server sent one.
|
|
69
|
+
2. The HTTP status. 408, 409, 429, 500, 502, 503, 504 and 529 are retried.
|
|
70
|
+
Any other status is raised immediately.
|
|
71
|
+
3. If there is no status: connection errors and timeouts are retried. This
|
|
72
|
+
covers the built-in `ConnectionError` and `TimeoutError` and the matching
|
|
73
|
+
classes in openai, anthropic, httpx, requests/urllib3 and aiohttp.
|
|
74
|
+
|
|
75
|
+
Anything else is raised immediately.
|
|
76
|
+
|
|
77
|
+
409 is on the list because the openai and anthropic SDKs retry it (they treat
|
|
78
|
+
it as a lock timeout). 529 is Anthropic's "overloaded". If a 409 from your API
|
|
79
|
+
means a real conflict, use `retry(...)` with your own `retry_on`.
|
|
80
|
+
|
|
81
|
+
## Wait time
|
|
82
|
+
|
|
83
|
+
- If the error response has `retry-after-ms` or `Retry-After` (seconds or an
|
|
84
|
+
HTTP date), retryhop waits that long plus 0-0.25 s of jitter, capped at
|
|
85
|
+
`max_retry_after`.
|
|
86
|
+
- Otherwise it uses exponential backoff with full jitter: a random wait in
|
|
87
|
+
[0, 1 s], then [0, 2 s], [0, 4 s], and so on, with the upper bound capped
|
|
88
|
+
at 60 s.
|
|
89
|
+
- `deadline` is the total budget across attempts. If the server asks for a
|
|
90
|
+
wait longer than what is left, retryhop gives up without sleeping. A
|
|
91
|
+
backoff wait is cut short to fit.
|
|
92
|
+
|
|
93
|
+
The deadline is checked between attempts. It does not cancel a request that
|
|
94
|
+
is already running, so also set a request timeout on the client.
|
|
95
|
+
|
|
96
|
+
## `llm_retry` options
|
|
97
|
+
|
|
98
|
+
| Option | Default | Description |
|
|
99
|
+
|---|---|---|
|
|
100
|
+
| `attempts` | `6` | Total calls, including the first. |
|
|
101
|
+
| `deadline` | `300` | Seconds across all attempts. `None` disables it. |
|
|
102
|
+
| `max_retry_after` | `120` | Upper bound, in seconds, on a server-requested wait. |
|
|
103
|
+
| `backoff` | exponential, see above | Used when there is no retry-after header. Any `f(attempt) -> seconds`. |
|
|
104
|
+
| `circuit` | `None` | A shared `CircuitBreaker`. |
|
|
105
|
+
| `on_retry` | `None` | Called with a `RetryState` before each wait. |
|
|
106
|
+
| `reraise` | `True` | `False` raises `RetryError` instead of the last exception. |
|
|
107
|
+
| `sleep` | `None` | Replacement sleep function, for tests. |
|
|
108
|
+
|
|
109
|
+
## Logging
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
import logging
|
|
113
|
+
from retryhop import llm_retry, describe
|
|
114
|
+
|
|
115
|
+
log = logging.getLogger("llm")
|
|
116
|
+
|
|
117
|
+
@llm_retry(on_retry=lambda s: log.warning(describe(s)))
|
|
118
|
+
def ask(prompt): ...
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
Output looks like:
|
|
122
|
+
|
|
123
|
+
```
|
|
124
|
+
attempt 1 failed: RateLimitError (HTTP 429); waiting 1.10s (server Retry-After), elapsed 0.0s
|
|
125
|
+
attempt 2 failed: InternalServerError (HTTP 503); waiting 1.64s (backoff), elapsed 1.1s
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
## Rate limiting
|
|
129
|
+
|
|
130
|
+
`RateLimiter` keeps one token bucket for requests per minute and one for
|
|
131
|
+
tokens per minute. A call that would go over either limit sleeps until there
|
|
132
|
+
is room.
|
|
133
|
+
|
|
134
|
+
```python
|
|
135
|
+
from retryhop import RateLimiter, llm_retry
|
|
136
|
+
|
|
137
|
+
limiter = RateLimiter(requests_per_minute=500, tokens_per_minute=200_000)
|
|
138
|
+
|
|
139
|
+
def estimate(prompt, **_):
|
|
140
|
+
return len(prompt) // 4 + 1024 # rough input estimate + max output
|
|
141
|
+
|
|
142
|
+
@llm_retry()
|
|
143
|
+
@limiter.limit(tokens=estimate)
|
|
144
|
+
def ask(prompt): ...
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
Put `@limiter.limit` below `@llm_retry` so retries are limited too.
|
|
148
|
+
|
|
149
|
+
The token count is your own estimate, and the provider counts tokens its own
|
|
150
|
+
way. If the response reports actual usage, call
|
|
151
|
+
`limiter.consume(actual - estimated)`. A negative value gives tokens back.
|
|
152
|
+
|
|
153
|
+
Without the decorator: `limiter.acquire(tokens=n)` or
|
|
154
|
+
`await limiter.acquire_async(tokens=n)`.
|
|
155
|
+
|
|
156
|
+
Things to know:
|
|
157
|
+
|
|
158
|
+
- The limiter is thread-safe and can be used from asyncio tasks, but its
|
|
159
|
+
state is per process. If several processes or hosts share one API key,
|
|
160
|
+
split the limits between them.
|
|
161
|
+
- The buckets start full, so a full minute's quota can go out at once right
|
|
162
|
+
after startup.
|
|
163
|
+
- A single call that needs more tokens than `tokens_per_minute` raises
|
|
164
|
+
`ValueError`.
|
|
165
|
+
|
|
166
|
+
## Circuit breaker
|
|
167
|
+
|
|
168
|
+
```python
|
|
169
|
+
from retryhop import CircuitBreaker, CircuitOpenError, llm_retry
|
|
170
|
+
|
|
171
|
+
breaker = CircuitBreaker(failure_threshold=5, recovery_time=30)
|
|
172
|
+
|
|
173
|
+
@llm_retry(circuit=breaker)
|
|
174
|
+
def ask(prompt): ...
|
|
175
|
+
|
|
176
|
+
try:
|
|
177
|
+
ask("hi")
|
|
178
|
+
except CircuitOpenError as e:
|
|
179
|
+
... # e.g. fall back to another model; e.retry_in = seconds until the next trial
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
- After `failure_threshold` consecutive failed attempts the circuit opens.
|
|
183
|
+
Calls then raise `CircuitOpenError` without hitting the API.
|
|
184
|
+
- After `recovery_time` seconds one trial call is let through. Success closes
|
|
185
|
+
the circuit; failure opens it again. Other calls made during the trial get
|
|
186
|
+
`CircuitOpenError`.
|
|
187
|
+
- Failures are counted per attempt, not per call. With the defaults
|
|
188
|
+
(6 attempts, threshold 5) a single call that keeps getting 503 opens the
|
|
189
|
+
circuit by itself, and that call ends with `CircuitOpenError` rather than
|
|
190
|
+
the SDK exception.
|
|
191
|
+
- A non-retryable error such as a 400 counts as a success, because the API
|
|
192
|
+
did respond. It resets the failure count.
|
|
193
|
+
- `CircuitOpenError` is never retried.
|
|
194
|
+
- State is per process.
|
|
195
|
+
|
|
196
|
+
## Lower-level API
|
|
197
|
+
|
|
198
|
+
- `retry(...)`: the decorator `llm_retry` is built on. It takes `attempts`,
|
|
199
|
+
`exceptions`, `retry_on`, `retry_if_result`, `wait_hint`, `backoff`,
|
|
200
|
+
`deadline`, `on_retry`, `reraise`, `circuit` and `sleep`. Its defaults are
|
|
201
|
+
different: 3 attempts, any `Exception` is retried, no deadline, and
|
|
202
|
+
`RetryError` is raised on give-up.
|
|
203
|
+
|
|
204
|
+
Polling a batch job:
|
|
205
|
+
|
|
206
|
+
```python
|
|
207
|
+
@retry(attempts=60, retry_if_result=lambda job: job.status != "completed",
|
|
208
|
+
backoff=constant(30))
|
|
209
|
+
def wait_for_batch(job_id):
|
|
210
|
+
return client.batches.retrieve(job_id)
|
|
211
|
+
```
|
|
212
|
+
|
|
213
|
+
- `retry_call(func, *args, retry_options={...}, **kwargs)`: same as `retry`
|
|
214
|
+
without decorating.
|
|
215
|
+
- `is_transient(exc)`, `status_code_of(exc)`, `retry_after_of(exc)`: the
|
|
216
|
+
checks described above.
|
|
217
|
+
- `describe(state)`: formats a `RetryState` for logging.
|
|
218
|
+
- `constant`, `linear`, `exponential`: backoff functions.
|
|
219
|
+
|
|
220
|
+
## Development
|
|
221
|
+
|
|
222
|
+
```bash
|
|
223
|
+
pip install -e ".[sdk-test]" # ".[test]" skips the tests that need the SDKs
|
|
224
|
+
pytest
|
|
225
|
+
python -m build && twine check dist/*
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
## License
|
|
229
|
+
|
|
230
|
+
MIT
|