cachellm-proxy 0.1.0__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/PKG-INFO +109 -33
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/README.md +99 -25
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/pyproject.toml +14 -9
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/pyproject.toml.orig +20 -12
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/__init__.py +1 -1
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/api/app.py +9 -0
- cachellm_proxy-0.2.0/src/cachellm/api/deps.py +180 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/api/routes_admin.py +70 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/api/routes_chat.py +59 -4
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/cache/analytics.py +50 -4
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/cache/exact_store.py +4 -1
- cachellm_proxy-0.2.0/src/cachellm/cache/memory.py +454 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/cache/redis_client.py +10 -2
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/cache/service.py +7 -4
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/cache/vector_store.py +8 -0
- cachellm_proxy-0.2.0/src/cachellm/cli.py +206 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/providers/base.py +8 -2
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/providers/bedrock.py +12 -2
- cachellm_proxy-0.2.0/src/cachellm/providers/catalog.py +350 -0
- cachellm_proxy-0.2.0/src/cachellm/providers/detect.py +164 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/providers/registry.py +23 -32
- cachellm_proxy-0.2.0/src/cachellm/report.py +217 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/settings.py +30 -1
- cachellm_proxy-0.1.0/src/cachellm/api/deps.py +0 -106
- cachellm_proxy-0.1.0/src/cachellm/cli.py +0 -122
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/__main__.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/api/__init__.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/api/auth.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/api/sse.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/cache/__init__.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/cache/coalesce.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/cache/entry.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/cache/keys.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/cache/policy.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/embeddings/__init__.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/embeddings/base.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/embeddings/fastembed_backend.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/embeddings/hash_backend.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/errors.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/logging_setup.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/models.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/observability/__init__.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/observability/metrics.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/observability/tracing.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/pricing.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/providers/__init__.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/providers/fake.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/providers/openai_compat.py +0 -0
- {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.0}/src/cachellm/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cachellm-proxy
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: A drop-in semantic cache for OpenAI-compatible LLM APIs: match requests by meaning, serve repeats instantly, cut spend.
|
|
5
5
|
Keywords: llm,cache,semantic-cache,openai,bedrock,proxy,vector-search,redis,fastapi,llmops,cost-optimization
|
|
6
6
|
Author: Adarsh Dwivedi
|
|
@@ -14,19 +14,17 @@ Classifier: Programming Language :: Python :: 3.12
|
|
|
14
14
|
Classifier: Programming Language :: Python :: 3.13
|
|
15
15
|
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
16
16
|
Classifier: Typing :: Typed
|
|
17
|
-
Requires-Dist: boto3>=1.43.91
|
|
18
17
|
Requires-Dist: fastapi>=0.141.1
|
|
18
|
+
Requires-Dist: uvicorn[standard]>=0.52.4
|
|
19
19
|
Requires-Dist: fastembed>=0.8.0
|
|
20
|
-
Requires-Dist: httpx>=0.28.1
|
|
21
20
|
Requires-Dist: numpy>=2.4.6
|
|
22
|
-
Requires-Dist: openai>=3.11.0
|
|
23
|
-
Requires-Dist: prometheus-client>=0.26.0
|
|
24
21
|
Requires-Dist: pydantic-settings>=2.15.0
|
|
25
|
-
Requires-Dist:
|
|
26
|
-
Requires-Dist: redisvl>=0.27.1
|
|
22
|
+
Requires-Dist: prometheus-client>=0.26.0
|
|
27
23
|
Requires-Dist: structlog>=26.1.0
|
|
28
24
|
Requires-Dist: typer>=0.27.2
|
|
29
|
-
Requires-Dist:
|
|
25
|
+
Requires-Dist: httpx>=0.28.1
|
|
26
|
+
Requires-Dist: cachellm-proxy[aws,redis,observability,datasets] ; extra == 'all'
|
|
27
|
+
Requires-Dist: boto3>=1.43.91 ; extra == 'aws'
|
|
30
28
|
Requires-Dist: botocore[crt]>=1.43.91 ; extra == 'aws'
|
|
31
29
|
Requires-Dist: datasets>=5.0.1 ; extra == 'datasets'
|
|
32
30
|
Requires-Dist: pandas>=3.0.5 ; extra == 'datasets'
|
|
@@ -37,13 +35,17 @@ Requires-Dist: opentelemetry-instrumentation-fastapi>=0.65b0 ; extra == 'observa
|
|
|
37
35
|
Requires-Dist: opentelemetry-instrumentation-httpx>=0.65b0 ; extra == 'observability'
|
|
38
36
|
Requires-Dist: opentelemetry-instrumentation-redis>=0.65b0 ; extra == 'observability'
|
|
39
37
|
Requires-Dist: opentelemetry-sdk>=1.44.0 ; extra == 'observability'
|
|
38
|
+
Requires-Dist: redis>=8.1.0 ; extra == 'redis'
|
|
39
|
+
Requires-Dist: redisvl>=0.27.1 ; extra == 'redis'
|
|
40
40
|
Requires-Python: >=3.11
|
|
41
41
|
Project-URL: Homepage, https://github.com/adarshcod30/CacheLLM
|
|
42
42
|
Project-URL: Repository, https://github.com/adarshcod30/CacheLLM
|
|
43
43
|
Project-URL: Issues, https://github.com/adarshcod30/CacheLLM/issues
|
|
44
|
+
Provides-Extra: all
|
|
44
45
|
Provides-Extra: aws
|
|
45
46
|
Provides-Extra: datasets
|
|
46
47
|
Provides-Extra: observability
|
|
48
|
+
Provides-Extra: redis
|
|
47
49
|
Description-Content-Type: text/markdown
|
|
48
50
|
|
|
49
51
|
# CacheLLM
|
|
@@ -51,7 +53,7 @@ Description-Content-Type: text/markdown
|
|
|
51
53
|
[](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml)
|
|
52
54
|
[](https://www.python.org/)
|
|
53
55
|
[](LICENSE)
|
|
54
|
-
[](tests/)
|
|
55
57
|
[](https://pypi.org/project/cachellm-proxy/)
|
|
56
58
|
[](docs/evaluation.md)
|
|
57
59
|
|
|
@@ -111,6 +113,7 @@ The same benchmark runs without any cloud credentials against the built-in fake
|
|
|
111
113
|
| Feature | What it does | Why it exists |
|
|
112
114
|
| --- | --- | --- |
|
|
113
115
|
| **Drop-in OpenAI API** | Same request and response shape, streaming included, verified against the official `openai` Python SDK in CI | Adoption has to cost one line, or nobody adopts it |
|
|
116
|
+
| **Runs with nothing installed** | Defaults to an in-process numpy store; uses Redis automatically when it can reach one | A cache you have to provision a server for does not get tried. Redis takes over when you actually need shared, durable state |
|
|
114
117
|
| **Two-tier cache** | Exact-match tier answers literal repeats in about a millisecond without embedding; semantic tier handles rewording | 82% of hits came from the exact tier: free, fast and impossible to get semantically wrong |
|
|
115
118
|
| **Per-model threshold calibration** | Ships measured safe thresholds for six embedding models and picks the right one automatically | Measured safe thresholds span 0.89 to 0.98. A threshold copied between models is a guess |
|
|
116
119
|
| **Cacheability policy** | Classifies every prompt and decides cacheable, category, TTL | Creative writing, live data and personal questions must not be cached like a fact |
|
|
@@ -134,7 +137,8 @@ The same benchmark runs without any cloud credentials against the built-in fake
|
|
|
134
137
|
| API | FastAPI + Uvicorn | Async, native streaming, OpenAPI docs for free |
|
|
135
138
|
| Validation and config | Pydantic v2, pydantic-settings | Typed request shapes and typed configuration from the environment |
|
|
136
139
|
| Embeddings | fastembed, `all-MiniLM-L6-v2` (ONNX, CPU) | In-process and about 6 ms. A hosted embedding API would put 100 ms in front of every cache hit and defeat the point |
|
|
137
|
-
| Vector store |
|
|
140
|
+
| Vector store, default | numpy matrix in the proxy's own memory | Nothing to install. Scanning 20,000 cached prompts takes 0.85 ms, where a Redis round trip alone costs 2 to 3 ms, so below roughly 100k entries this is not a compromise, it is faster |
|
|
141
|
+
| Vector store, at scale | Redis 8 + RedisVL, HNSW over cosine | One process is the exact-match store, the vector index and the stampede lock. Shared across workers, survives restarts, and an approximate index starts paying above ~100k entries |
|
|
138
142
|
| Providers | AWS Bedrock (Converse), any OpenAI-compatible endpoint, deterministic fake | Converse reaches Nova, Claude, Llama and Mistral with one request shape and no extra vendor keys |
|
|
139
143
|
| Metrics | prometheus-client, Prometheus, Grafana | Dashboard ships provisioned, so a reviewer sees data on first boot |
|
|
140
144
|
| Tracing | OpenTelemetry, optional, GenAI semantic conventions | Instrument once, export to Langfuse, Tempo or Jaeger |
|
|
@@ -270,32 +274,47 @@ MiniLM gives four times the safe recall of bge-small at a third of the download
|
|
|
270
274
|
|
|
271
275
|
## Quick start
|
|
272
276
|
|
|
273
|
-
|
|
277
|
+
Python 3.11 or newer, and nothing else.
|
|
274
278
|
|
|
275
279
|
```bash
|
|
276
280
|
pip install cachellm-proxy
|
|
281
|
+
cachellm serve
|
|
277
282
|
```
|
|
278
283
|
|
|
279
|
-
|
|
284
|
+
That is the whole install. No Redis, no Docker, no config file. On startup it looks for a provider you already have: an API key in your environment, Ollama running locally, or AWS credentials. It logs which one it picked and how to override it.
|
|
280
285
|
|
|
281
|
-
|
|
286
|
+
To see what it found before starting anything:
|
|
282
287
|
|
|
283
288
|
```bash
|
|
284
|
-
|
|
285
|
-
cd CacheLLM
|
|
286
|
-
uv sync
|
|
289
|
+
cachellm providers
|
|
287
290
|
```
|
|
288
291
|
|
|
289
|
-
|
|
292
|
+
To try it with no account at all, the built-in test double stands in for a model:
|
|
290
293
|
|
|
291
294
|
```bash
|
|
292
|
-
|
|
295
|
+
CACHELLM_DEFAULT_PROVIDER=fake CACHELLM_FAKE_LATENCY_MS=600 cachellm serve
|
|
293
296
|
```
|
|
294
297
|
|
|
295
|
-
|
|
298
|
+
You only install what you actually route to. The base package is the proxy plus the local embedding model; AWS and Redis are extras, and nothing here installs Grafana or a Prometheus server, which are separate programs rather than Python packages.
|
|
299
|
+
|
|
300
|
+
| You want | Install |
|
|
301
|
+
| --- | --- |
|
|
302
|
+
| Any OpenAI-compatible endpoint: OpenAI, Groq, Gemini, Ollama, vLLM | `pip install cachellm-proxy` |
|
|
303
|
+
| AWS Bedrock | `pip install "cachellm-proxy[aws]"` |
|
|
304
|
+
| Redis instead of the in-process cache | `pip install "cachellm-proxy[redis]"` |
|
|
305
|
+
| Tracing to Langfuse or Tempo | `pip install "cachellm-proxy[observability]"` |
|
|
306
|
+
| Everything | `pip install "cachellm-proxy[all]"` |
|
|
307
|
+
|
|
308
|
+
Ask for a provider or backend whose extra is missing and the proxy names the exact command to fix it rather than raising an import error.
|
|
309
|
+
|
|
310
|
+
The distribution is `cachellm-proxy` because PyPI blocks `cachellm` as too close to an existing `cachelm`. The import name and the CLI are both still `cachellm`.
|
|
311
|
+
|
|
312
|
+
From source, if you plan to change anything:
|
|
296
313
|
|
|
297
314
|
```bash
|
|
298
|
-
|
|
315
|
+
git clone https://github.com/adarshcod30/CacheLLM.git
|
|
316
|
+
cd CacheLLM
|
|
317
|
+
uv sync && uv run cachellm serve
|
|
299
318
|
```
|
|
300
319
|
|
|
301
320
|
In another terminal, ask the same thing twice and watch the second one come back instantly:
|
|
@@ -342,28 +361,53 @@ make tune && make compare-models && make bench
|
|
|
342
361
|
|
|
343
362
|
### Using a real provider
|
|
344
363
|
|
|
345
|
-
|
|
364
|
+
One environment variable per host. Everything except Bedrock speaks the OpenAI protocol, so a single adapter covers all of it and needs no extra.
|
|
365
|
+
|
|
366
|
+
| Host | `CACHELLM_OPENAI_BASE_URL` | Example model |
|
|
367
|
+
| --- | --- | --- |
|
|
368
|
+
| OpenAI | `https://api.openai.com/v1` | `gpt-4o-mini` |
|
|
369
|
+
| Groq | `https://api.groq.com/openai/v1` | `llama-3.3-70b-versatile` |
|
|
370
|
+
| Google Gemini | `https://generativelanguage.googleapis.com/v1beta/openai/` | `gemini-2.5-flash` |
|
|
371
|
+
| Anthropic | `https://api.anthropic.com/v1` | `claude-haiku-4-5` |
|
|
372
|
+
| OpenRouter | `https://openrouter.ai/api/v1` | `anthropic/claude-3.5-sonnet` |
|
|
373
|
+
| Together | `https://api.together.xyz/v1` | `meta-llama/Llama-3.3-70B-Instruct-Turbo` |
|
|
374
|
+
| Ollama, local | `http://localhost:11434/v1` | `llama3.2` |
|
|
375
|
+
| vLLM or LM Studio | `http://localhost:8000/v1` | whatever you served |
|
|
346
376
|
|
|
347
377
|
```bash
|
|
348
|
-
|
|
378
|
+
CACHELLM_OPENAI_BASE_URL=https://api.groq.com/openai/v1 \
|
|
379
|
+
CACHELLM_OPENAI_API_KEY=gsk_your_key \
|
|
380
|
+
cachellm serve
|
|
349
381
|
```
|
|
350
382
|
|
|
351
|
-
|
|
383
|
+
**AWS Bedrock** is the one exception, because it does not speak the OpenAI protocol. It needs the `aws` extra, and then uses your existing AWS credentials with no vendor API key at all:
|
|
352
384
|
|
|
353
385
|
```bash
|
|
354
|
-
|
|
386
|
+
pip install "cachellm-proxy[aws]"
|
|
387
|
+
CACHELLM_DEFAULT_PROVIDER=bedrock AWS_REGION=us-east-1 cachellm serve
|
|
355
388
|
```
|
|
356
389
|
|
|
357
|
-
The proxy detects that case and says so in the error rather than passing along boto's version of the message.
|
|
358
|
-
|
|
359
390
|
```bash
|
|
360
391
|
curl -s http://localhost:8080/v1/chat/completions -H 'Content-Type: application/json' -d '{"model":"bedrock/us.amazon.nova-micro-v1:0","temperature":0,"messages":[{"role":"user","content":"What is Redis used for?"}]}'
|
|
361
392
|
```
|
|
362
393
|
|
|
363
|
-
|
|
394
|
+
If your credentials come from `aws login` rather than static keys or an SSO profile, that provider needs the CRT extra, which the `aws` extra already includes. The proxy detects that case and says so in the error rather than passing along boto's version of the message.
|
|
395
|
+
|
|
396
|
+
### How a model name gets routed
|
|
397
|
+
|
|
398
|
+
Three rules, in order. You never configure a model list.
|
|
399
|
+
|
|
400
|
+
1. **An explicit prefix** this proxy owns: `bedrock/…`, `openai/…`, `fake/…`.
|
|
401
|
+
2. **A recognisable vendor convention.** Bedrock ids are always `vendor.model`, so `amazon.nova-lite-v1:0` and `us.anthropic.claude-3-haiku-20240307-v1:0` are identified with no prefix. OpenAI's own families (`gpt-`, `o1`, `o3`, `text-embedding-`) are identified the same way, whatever the default is.
|
|
402
|
+
3. **Otherwise the configured default**, which is the OpenAI-compatible adapter. Names like `llama3.2`, `mixtral-8x7b-32768` and `qwen2.5-coder:7b` are served by Groq, Ollama, Together and OpenRouter alike, so the endpoint you configured is the only sensible answer.
|
|
403
|
+
|
|
404
|
+
Model ids that legitimately contain a slash, which OpenRouter and Together both use, are forwarded whole. Only this proxy's own prefixes are stripped. Note that `anthropic.claude-…` with a dot is a Bedrock id while `anthropic/claude-…` with a slash is an OpenRouter id, and the router tells them apart.
|
|
405
|
+
|
|
406
|
+
Ask it directly if you are unsure:
|
|
364
407
|
|
|
365
408
|
```bash
|
|
366
|
-
|
|
409
|
+
curl -s localhost:8080/admin/route/meta-llama/Llama-3.3-70B-Instruct-Turbo
|
|
410
|
+
curl -s localhost:8080/admin/providers
|
|
367
411
|
```
|
|
368
412
|
|
|
369
413
|
## API reference
|
|
@@ -400,6 +444,8 @@ Clients can steer per request with `X-Cache-Control`:
|
|
|
400
444
|
| --- | --- |
|
|
401
445
|
| `GET /admin/stats` | Hit rate, tier split, money saved, latency percentiles, entry count |
|
|
402
446
|
| `GET /admin/config` | Effective thresholds, TTLs and rules |
|
|
447
|
+
| `GET /admin/providers` | Every host, its base URL, its pip extra, example model ids |
|
|
448
|
+
| `GET /admin/route/{model}` | Where one model name would go, and why |
|
|
403
449
|
| `POST /admin/invalidate` | Drop by `namespace`, by `model`, or `all` |
|
|
404
450
|
| `GET /admin/near-misses` | Recent lookups that landed just below threshold |
|
|
405
451
|
| `GET /admin/near-miss-histogram` | Cumulative view of what a lower threshold would buy |
|
|
@@ -418,20 +464,50 @@ curl -s http://localhost:8080/admin/threshold-sweep -H 'Content-Type: applicatio
|
|
|
418
464
|
### Command line
|
|
419
465
|
|
|
420
466
|
```bash
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
467
|
+
cachellm serve # run the proxy, auto-detecting a provider
|
|
468
|
+
cachellm providers # every host, and what this machine can reach
|
|
469
|
+
cachellm stats # hit rate, savings, latency, recent requests
|
|
470
|
+
cachellm watch # follow requests live, like tail -f
|
|
471
|
+
cachellm config # effective configuration
|
|
472
|
+
cachellm invalidate --all
|
|
473
|
+
cachellm tune pairs.jsonl
|
|
426
474
|
```
|
|
427
475
|
|
|
476
|
+
`cachellm stats` is the dashboard, in your terminal:
|
|
477
|
+
|
|
478
|
+
```
|
|
479
|
+
CacheLLM · memory backend · all-MiniLM-L6-v2
|
|
480
|
+
|
|
481
|
+
HIT RATE 77.0% ████████████████░░░░░░ 1,540 of 2,000 requests
|
|
482
|
+
1,265 exact · 275 semantic · 460 missed · 0 bypassed
|
|
483
|
+
|
|
484
|
+
SAVED $0.0135 spent $0.0038 78% lower
|
|
485
|
+
116,029 tokens never generated
|
|
486
|
+
|
|
487
|
+
LATENCY cached 2.6 ms p50 · 5.7 ms p95
|
|
488
|
+
uncached 797.0 ms p50 · 1022.8 ms p95 179x faster
|
|
489
|
+
|
|
490
|
+
CACHE 451 entries · 9 coalesced · 17 near misses
|
|
491
|
+
|
|
492
|
+
time result tier latency score saved prompt
|
|
493
|
+
02:30:55 HIT exact 1.2ms 1.000 +$0.000010 What is Redis used for?
|
|
494
|
+
02:30:54 BYPASS 451.5ms · · What is my order 1234… (pii:long_digits)
|
|
495
|
+
02:30:53 HIT semantic 7.4ms 0.952 +$0.000011 What is Redis typically used for?
|
|
496
|
+
02:30:53 MISS 454.8ms · · How do I configure nginx for TLS?
|
|
497
|
+
```
|
|
498
|
+
|
|
499
|
+
Prometheus and Grafana are still there if you want history and alerting, but nothing needs them.
|
|
500
|
+
|
|
428
501
|
## Configuration
|
|
429
502
|
|
|
430
503
|
Every setting is an environment variable prefixed `CACHELLM_`, or a line in `.env`. Copy `.env.example` to start. The ones that matter most:
|
|
431
504
|
|
|
432
505
|
| Variable | Default | Notes |
|
|
433
506
|
| --- | --- | --- |
|
|
507
|
+
| `CACHELLM_BACKEND` | `auto` | `memory` needs nothing, `redis` shares one cache across workers, `auto` uses Redis when reachable and memory when not |
|
|
434
508
|
| `CACHELLM_REDIS_URL` | `redis://localhost:6379/0` | Must be database 0. Redis Search cannot index any other |
|
|
509
|
+
| `CACHELLM_MEMORY_MAX_ENTRIES` | `50000` | Cap for the in-memory store. 50k of 384-dim vectors is about 73 MB |
|
|
510
|
+
| `CACHELLM_MEMORY_SNAPSHOT_PATH` | unset | Persist the in-memory cache to this file so a restart does not start cold |
|
|
435
511
|
| `CACHELLM_API_KEYS` | empty | Comma-separated client keys. Empty disables auth, which is local development only |
|
|
436
512
|
| `CACHELLM_EMBEDDING_MODEL` | `all-MiniLM-L6-v2` | Changing this changes the safe threshold. See the calibration table |
|
|
437
513
|
| `CACHELLM_THRESHOLD_*` | `0` | Zero means use the calibrated value for your model |
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
[](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml)
|
|
4
4
|
[](https://www.python.org/)
|
|
5
5
|
[](LICENSE)
|
|
6
|
-
[](tests/)
|
|
7
7
|
[](https://pypi.org/project/cachellm-proxy/)
|
|
8
8
|
[](docs/evaluation.md)
|
|
9
9
|
|
|
@@ -63,6 +63,7 @@ The same benchmark runs without any cloud credentials against the built-in fake
|
|
|
63
63
|
| Feature | What it does | Why it exists |
|
|
64
64
|
| --- | --- | --- |
|
|
65
65
|
| **Drop-in OpenAI API** | Same request and response shape, streaming included, verified against the official `openai` Python SDK in CI | Adoption has to cost one line, or nobody adopts it |
|
|
66
|
+
| **Runs with nothing installed** | Defaults to an in-process numpy store; uses Redis automatically when it can reach one | A cache you have to provision a server for does not get tried. Redis takes over when you actually need shared, durable state |
|
|
66
67
|
| **Two-tier cache** | Exact-match tier answers literal repeats in about a millisecond without embedding; semantic tier handles rewording | 82% of hits came from the exact tier: free, fast and impossible to get semantically wrong |
|
|
67
68
|
| **Per-model threshold calibration** | Ships measured safe thresholds for six embedding models and picks the right one automatically | Measured safe thresholds span 0.89 to 0.98. A threshold copied between models is a guess |
|
|
68
69
|
| **Cacheability policy** | Classifies every prompt and decides cacheable, category, TTL | Creative writing, live data and personal questions must not be cached like a fact |
|
|
@@ -86,7 +87,8 @@ The same benchmark runs without any cloud credentials against the built-in fake
|
|
|
86
87
|
| API | FastAPI + Uvicorn | Async, native streaming, OpenAPI docs for free |
|
|
87
88
|
| Validation and config | Pydantic v2, pydantic-settings | Typed request shapes and typed configuration from the environment |
|
|
88
89
|
| Embeddings | fastembed, `all-MiniLM-L6-v2` (ONNX, CPU) | In-process and about 6 ms. A hosted embedding API would put 100 ms in front of every cache hit and defeat the point |
|
|
89
|
-
| Vector store |
|
|
90
|
+
| Vector store, default | numpy matrix in the proxy's own memory | Nothing to install. Scanning 20,000 cached prompts takes 0.85 ms, where a Redis round trip alone costs 2 to 3 ms, so below roughly 100k entries this is not a compromise, it is faster |
|
|
91
|
+
| Vector store, at scale | Redis 8 + RedisVL, HNSW over cosine | One process is the exact-match store, the vector index and the stampede lock. Shared across workers, survives restarts, and an approximate index starts paying above ~100k entries |
|
|
90
92
|
| Providers | AWS Bedrock (Converse), any OpenAI-compatible endpoint, deterministic fake | Converse reaches Nova, Claude, Llama and Mistral with one request shape and no extra vendor keys |
|
|
91
93
|
| Metrics | prometheus-client, Prometheus, Grafana | Dashboard ships provisioned, so a reviewer sees data on first boot |
|
|
92
94
|
| Tracing | OpenTelemetry, optional, GenAI semantic conventions | Instrument once, export to Langfuse, Tempo or Jaeger |
|
|
@@ -222,32 +224,47 @@ MiniLM gives four times the safe recall of bge-small at a third of the download
|
|
|
222
224
|
|
|
223
225
|
## Quick start
|
|
224
226
|
|
|
225
|
-
|
|
227
|
+
Python 3.11 or newer, and nothing else.
|
|
226
228
|
|
|
227
229
|
```bash
|
|
228
230
|
pip install cachellm-proxy
|
|
231
|
+
cachellm serve
|
|
229
232
|
```
|
|
230
233
|
|
|
231
|
-
|
|
234
|
+
That is the whole install. No Redis, no Docker, no config file. On startup it looks for a provider you already have: an API key in your environment, Ollama running locally, or AWS credentials. It logs which one it picked and how to override it.
|
|
232
235
|
|
|
233
|
-
|
|
236
|
+
To see what it found before starting anything:
|
|
234
237
|
|
|
235
238
|
```bash
|
|
236
|
-
|
|
237
|
-
cd CacheLLM
|
|
238
|
-
uv sync
|
|
239
|
+
cachellm providers
|
|
239
240
|
```
|
|
240
241
|
|
|
241
|
-
|
|
242
|
+
To try it with no account at all, the built-in test double stands in for a model:
|
|
242
243
|
|
|
243
244
|
```bash
|
|
244
|
-
|
|
245
|
+
CACHELLM_DEFAULT_PROVIDER=fake CACHELLM_FAKE_LATENCY_MS=600 cachellm serve
|
|
245
246
|
```
|
|
246
247
|
|
|
247
|
-
|
|
248
|
+
You only install what you actually route to. The base package is the proxy plus the local embedding model; AWS and Redis are extras, and nothing here installs Grafana or a Prometheus server, which are separate programs rather than Python packages.
|
|
249
|
+
|
|
250
|
+
| You want | Install |
|
|
251
|
+
| --- | --- |
|
|
252
|
+
| Any OpenAI-compatible endpoint: OpenAI, Groq, Gemini, Ollama, vLLM | `pip install cachellm-proxy` |
|
|
253
|
+
| AWS Bedrock | `pip install "cachellm-proxy[aws]"` |
|
|
254
|
+
| Redis instead of the in-process cache | `pip install "cachellm-proxy[redis]"` |
|
|
255
|
+
| Tracing to Langfuse or Tempo | `pip install "cachellm-proxy[observability]"` |
|
|
256
|
+
| Everything | `pip install "cachellm-proxy[all]"` |
|
|
257
|
+
|
|
258
|
+
Ask for a provider or backend whose extra is missing and the proxy names the exact command to fix it rather than raising an import error.
|
|
259
|
+
|
|
260
|
+
The distribution is `cachellm-proxy` because PyPI blocks `cachellm` as too close to an existing `cachelm`. The import name and the CLI are both still `cachellm`.
|
|
261
|
+
|
|
262
|
+
From source, if you plan to change anything:
|
|
248
263
|
|
|
249
264
|
```bash
|
|
250
|
-
|
|
265
|
+
git clone https://github.com/adarshcod30/CacheLLM.git
|
|
266
|
+
cd CacheLLM
|
|
267
|
+
uv sync && uv run cachellm serve
|
|
251
268
|
```
|
|
252
269
|
|
|
253
270
|
In another terminal, ask the same thing twice and watch the second one come back instantly:
|
|
@@ -294,28 +311,53 @@ make tune && make compare-models && make bench
|
|
|
294
311
|
|
|
295
312
|
### Using a real provider
|
|
296
313
|
|
|
297
|
-
|
|
314
|
+
One environment variable per host. Everything except Bedrock speaks the OpenAI protocol, so a single adapter covers all of it and needs no extra.
|
|
315
|
+
|
|
316
|
+
| Host | `CACHELLM_OPENAI_BASE_URL` | Example model |
|
|
317
|
+
| --- | --- | --- |
|
|
318
|
+
| OpenAI | `https://api.openai.com/v1` | `gpt-4o-mini` |
|
|
319
|
+
| Groq | `https://api.groq.com/openai/v1` | `llama-3.3-70b-versatile` |
|
|
320
|
+
| Google Gemini | `https://generativelanguage.googleapis.com/v1beta/openai/` | `gemini-2.5-flash` |
|
|
321
|
+
| Anthropic | `https://api.anthropic.com/v1` | `claude-haiku-4-5` |
|
|
322
|
+
| OpenRouter | `https://openrouter.ai/api/v1` | `anthropic/claude-3.5-sonnet` |
|
|
323
|
+
| Together | `https://api.together.xyz/v1` | `meta-llama/Llama-3.3-70B-Instruct-Turbo` |
|
|
324
|
+
| Ollama, local | `http://localhost:11434/v1` | `llama3.2` |
|
|
325
|
+
| vLLM or LM Studio | `http://localhost:8000/v1` | whatever you served |
|
|
298
326
|
|
|
299
327
|
```bash
|
|
300
|
-
|
|
328
|
+
CACHELLM_OPENAI_BASE_URL=https://api.groq.com/openai/v1 \
|
|
329
|
+
CACHELLM_OPENAI_API_KEY=gsk_your_key \
|
|
330
|
+
cachellm serve
|
|
301
331
|
```
|
|
302
332
|
|
|
303
|
-
|
|
333
|
+
**AWS Bedrock** is the one exception, because it does not speak the OpenAI protocol. It needs the `aws` extra, and then uses your existing AWS credentials with no vendor API key at all:
|
|
304
334
|
|
|
305
335
|
```bash
|
|
306
|
-
|
|
336
|
+
pip install "cachellm-proxy[aws]"
|
|
337
|
+
CACHELLM_DEFAULT_PROVIDER=bedrock AWS_REGION=us-east-1 cachellm serve
|
|
307
338
|
```
|
|
308
339
|
|
|
309
|
-
The proxy detects that case and says so in the error rather than passing along boto's version of the message.
|
|
310
|
-
|
|
311
340
|
```bash
|
|
312
341
|
curl -s http://localhost:8080/v1/chat/completions -H 'Content-Type: application/json' -d '{"model":"bedrock/us.amazon.nova-micro-v1:0","temperature":0,"messages":[{"role":"user","content":"What is Redis used for?"}]}'
|
|
313
342
|
```
|
|
314
343
|
|
|
315
|
-
|
|
344
|
+
If your credentials come from `aws login` rather than static keys or an SSO profile, that provider needs the CRT extra, which the `aws` extra already includes. The proxy detects that case and says so in the error rather than passing along boto's version of the message.
|
|
345
|
+
|
|
346
|
+
### How a model name gets routed
|
|
347
|
+
|
|
348
|
+
Three rules, in order. You never configure a model list.
|
|
349
|
+
|
|
350
|
+
1. **An explicit prefix** this proxy owns: `bedrock/…`, `openai/…`, `fake/…`.
|
|
351
|
+
2. **A recognisable vendor convention.** Bedrock ids are always `vendor.model`, so `amazon.nova-lite-v1:0` and `us.anthropic.claude-3-haiku-20240307-v1:0` are identified with no prefix. OpenAI's own families (`gpt-`, `o1`, `o3`, `text-embedding-`) are identified the same way, whatever the default is.
|
|
352
|
+
3. **Otherwise the configured default**, which is the OpenAI-compatible adapter. Names like `llama3.2`, `mixtral-8x7b-32768` and `qwen2.5-coder:7b` are served by Groq, Ollama, Together and OpenRouter alike, so the endpoint you configured is the only sensible answer.
|
|
353
|
+
|
|
354
|
+
Model ids that legitimately contain a slash, which OpenRouter and Together both use, are forwarded whole. Only this proxy's own prefixes are stripped. Note that `anthropic.claude-…` with a dot is a Bedrock id while `anthropic/claude-…` with a slash is an OpenRouter id, and the router tells them apart.
|
|
355
|
+
|
|
356
|
+
Ask it directly if you are unsure:
|
|
316
357
|
|
|
317
358
|
```bash
|
|
318
|
-
|
|
359
|
+
curl -s localhost:8080/admin/route/meta-llama/Llama-3.3-70B-Instruct-Turbo
|
|
360
|
+
curl -s localhost:8080/admin/providers
|
|
319
361
|
```
|
|
320
362
|
|
|
321
363
|
## API reference
|
|
@@ -352,6 +394,8 @@ Clients can steer per request with `X-Cache-Control`:
|
|
|
352
394
|
| --- | --- |
|
|
353
395
|
| `GET /admin/stats` | Hit rate, tier split, money saved, latency percentiles, entry count |
|
|
354
396
|
| `GET /admin/config` | Effective thresholds, TTLs and rules |
|
|
397
|
+
| `GET /admin/providers` | Every host, its base URL, its pip extra, example model ids |
|
|
398
|
+
| `GET /admin/route/{model}` | Where one model name would go, and why |
|
|
355
399
|
| `POST /admin/invalidate` | Drop by `namespace`, by `model`, or `all` |
|
|
356
400
|
| `GET /admin/near-misses` | Recent lookups that landed just below threshold |
|
|
357
401
|
| `GET /admin/near-miss-histogram` | Cumulative view of what a lower threshold would buy |
|
|
@@ -370,12 +414,39 @@ curl -s http://localhost:8080/admin/threshold-sweep -H 'Content-Type: applicatio
|
|
|
370
414
|
### Command line
|
|
371
415
|
|
|
372
416
|
```bash
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
417
|
+
cachellm serve # run the proxy, auto-detecting a provider
|
|
418
|
+
cachellm providers # every host, and what this machine can reach
|
|
419
|
+
cachellm stats # hit rate, savings, latency, recent requests
|
|
420
|
+
cachellm watch # follow requests live, like tail -f
|
|
421
|
+
cachellm config # effective configuration
|
|
422
|
+
cachellm invalidate --all
|
|
423
|
+
cachellm tune pairs.jsonl
|
|
424
|
+
```
|
|
425
|
+
|
|
426
|
+
`cachellm stats` is the dashboard, in your terminal:
|
|
427
|
+
|
|
378
428
|
```
|
|
429
|
+
CacheLLM · memory backend · all-MiniLM-L6-v2
|
|
430
|
+
|
|
431
|
+
HIT RATE 77.0% ████████████████░░░░░░ 1,540 of 2,000 requests
|
|
432
|
+
1,265 exact · 275 semantic · 460 missed · 0 bypassed
|
|
433
|
+
|
|
434
|
+
SAVED $0.0135 spent $0.0038 78% lower
|
|
435
|
+
116,029 tokens never generated
|
|
436
|
+
|
|
437
|
+
LATENCY cached 2.6 ms p50 · 5.7 ms p95
|
|
438
|
+
uncached 797.0 ms p50 · 1022.8 ms p95 179x faster
|
|
439
|
+
|
|
440
|
+
CACHE 451 entries · 9 coalesced · 17 near misses
|
|
441
|
+
|
|
442
|
+
time result tier latency score saved prompt
|
|
443
|
+
02:30:55 HIT exact 1.2ms 1.000 +$0.000010 What is Redis used for?
|
|
444
|
+
02:30:54 BYPASS 451.5ms · · What is my order 1234… (pii:long_digits)
|
|
445
|
+
02:30:53 HIT semantic 7.4ms 0.952 +$0.000011 What is Redis typically used for?
|
|
446
|
+
02:30:53 MISS 454.8ms · · How do I configure nginx for TLS?
|
|
447
|
+
```
|
|
448
|
+
|
|
449
|
+
Prometheus and Grafana are still there if you want history and alerting, but nothing needs them.
|
|
379
450
|
|
|
380
451
|
## Configuration
|
|
381
452
|
|
|
@@ -383,7 +454,10 @@ Every setting is an environment variable prefixed `CACHELLM_`, or a line in `.en
|
|
|
383
454
|
|
|
384
455
|
| Variable | Default | Notes |
|
|
385
456
|
| --- | --- | --- |
|
|
457
|
+
| `CACHELLM_BACKEND` | `auto` | `memory` needs nothing, `redis` shares one cache across workers, `auto` uses Redis when reachable and memory when not |
|
|
386
458
|
| `CACHELLM_REDIS_URL` | `redis://localhost:6379/0` | Must be database 0. Redis Search cannot index any other |
|
|
459
|
+
| `CACHELLM_MEMORY_MAX_ENTRIES` | `50000` | Cap for the in-memory store. 50k of 384-dim vectors is about 73 MB |
|
|
460
|
+
| `CACHELLM_MEMORY_SNAPSHOT_PATH` | unset | Persist the in-memory cache to this file so a restart does not start cold |
|
|
387
461
|
| `CACHELLM_API_KEYS` | empty | Comma-separated client keys. Empty disables auth, which is local development only |
|
|
388
462
|
| `CACHELLM_EMBEDDING_MODEL` | `all-MiniLM-L6-v2` | Changing this changes the safe threshold. See the calibration table |
|
|
389
463
|
| `CACHELLM_THRESHOLD_*` | `0` | Zero means use the calibrated value for your model |
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "cachellm-proxy"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.2.0"
|
|
4
4
|
description = "A drop-in semantic cache for OpenAI-compatible LLM APIs: match requests by meaning, serve repeats instantly, cut spend."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.11"
|
|
@@ -29,19 +29,15 @@ classifiers = [
|
|
|
29
29
|
"Typing :: Typed",
|
|
30
30
|
]
|
|
31
31
|
dependencies = [
|
|
32
|
-
"boto3>=1.43.91",
|
|
33
32
|
"fastapi>=0.141.1",
|
|
33
|
+
"uvicorn[standard]>=0.52.4",
|
|
34
34
|
"fastembed>=0.8.0",
|
|
35
|
-
"httpx>=0.28.1",
|
|
36
35
|
"numpy>=2.4.6",
|
|
37
|
-
"openai>=3.11.0",
|
|
38
|
-
"prometheus-client>=0.26.0",
|
|
39
36
|
"pydantic-settings>=2.15.0",
|
|
40
|
-
"
|
|
41
|
-
"redisvl>=0.27.1",
|
|
37
|
+
"prometheus-client>=0.26.0",
|
|
42
38
|
"structlog>=26.1.0",
|
|
43
39
|
"typer>=0.27.2",
|
|
44
|
-
"
|
|
40
|
+
"httpx>=0.28.1",
|
|
45
41
|
]
|
|
46
42
|
|
|
47
43
|
[[project.authors]]
|
|
@@ -57,6 +53,14 @@ Issues = "https://github.com/adarshcod30/CacheLLM/issues"
|
|
|
57
53
|
cachellm = "cachellm:main"
|
|
58
54
|
|
|
59
55
|
[project.optional-dependencies]
|
|
56
|
+
aws = [
|
|
57
|
+
"boto3>=1.43.91",
|
|
58
|
+
"botocore[crt]>=1.43.91",
|
|
59
|
+
]
|
|
60
|
+
redis = [
|
|
61
|
+
"redis>=8.1.0",
|
|
62
|
+
"redisvl>=0.27.1",
|
|
63
|
+
]
|
|
60
64
|
observability = [
|
|
61
65
|
"langfuse>=4.15.2",
|
|
62
66
|
"opentelemetry-exporter-otlp-proto-http>=1.44.0",
|
|
@@ -70,7 +74,7 @@ datasets = [
|
|
|
70
74
|
"datasets>=5.0.1",
|
|
71
75
|
"pandas>=3.0.5",
|
|
72
76
|
]
|
|
73
|
-
|
|
77
|
+
all = ["cachellm-proxy[aws,redis,observability,datasets]"]
|
|
74
78
|
|
|
75
79
|
[build-system]
|
|
76
80
|
requires = ["uv_build>=0.11.23,<0.12.0"]
|
|
@@ -166,6 +170,7 @@ exclude_lines = [
|
|
|
166
170
|
[dependency-groups]
|
|
167
171
|
dev = [
|
|
168
172
|
"mypy>=2.3.1",
|
|
173
|
+
"openai>=3.11.0",
|
|
169
174
|
"pytest>=9.1.1",
|
|
170
175
|
"pytest-asyncio>=1.4.0",
|
|
171
176
|
"pytest-cov>=7.1.0",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "cachellm-proxy"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.2.0"
|
|
4
4
|
description = "A drop-in semantic cache for OpenAI-compatible LLM APIs: match requests by meaning, serve repeats instantly, cut spend."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
authors = [
|
|
@@ -23,19 +23,17 @@ classifiers = [
|
|
|
23
23
|
"Typing :: Typed",
|
|
24
24
|
]
|
|
25
25
|
dependencies = [
|
|
26
|
-
|
|
26
|
+
# The proxy itself and the semantic matching. Nothing here is optional:
|
|
27
|
+
# this is the smallest set that can cache by meaning and serve HTTP.
|
|
27
28
|
"fastapi>=0.141.1",
|
|
28
|
-
"
|
|
29
|
-
"
|
|
30
|
-
"numpy>=2.4.6",
|
|
31
|
-
"openai>=3.11.0",
|
|
32
|
-
"prometheus-client>=0.26.0",
|
|
29
|
+
"uvicorn[standard]>=0.52.4",
|
|
30
|
+
"fastembed>=0.8.0", # local embeddings, CPU, no API call
|
|
31
|
+
"numpy>=2.4.6", # the default vector search is one matrix multiply
|
|
33
32
|
"pydantic-settings>=2.15.0",
|
|
34
|
-
"
|
|
35
|
-
"redisvl>=0.27.1",
|
|
33
|
+
"prometheus-client>=0.26.0", # ~50 KB: writes a text page, not a server
|
|
36
34
|
"structlog>=26.1.0",
|
|
37
35
|
"typer>=0.27.2",
|
|
38
|
-
"
|
|
36
|
+
"httpx>=0.28.1",
|
|
39
37
|
]
|
|
40
38
|
|
|
41
39
|
[project.urls]
|
|
@@ -47,6 +45,15 @@ Issues = "https://github.com/adarshcod30/CacheLLM/issues"
|
|
|
47
45
|
cachellm = "cachellm:main"
|
|
48
46
|
|
|
49
47
|
[project.optional-dependencies]
|
|
48
|
+
# Only install what you actually route to.
|
|
49
|
+
aws = [
|
|
50
|
+
"boto3>=1.43.91",
|
|
51
|
+
"botocore[crt]>=1.43.91", # `aws login` credentials need the CRT extra
|
|
52
|
+
]
|
|
53
|
+
redis = [
|
|
54
|
+
"redis>=8.1.0",
|
|
55
|
+
"redisvl>=0.27.1",
|
|
56
|
+
]
|
|
50
57
|
observability = [
|
|
51
58
|
"langfuse>=4.15.2",
|
|
52
59
|
"opentelemetry-exporter-otlp-proto-http>=1.44.0",
|
|
@@ -60,8 +67,8 @@ datasets = [
|
|
|
60
67
|
"datasets>=5.0.1",
|
|
61
68
|
"pandas>=3.0.5",
|
|
62
69
|
]
|
|
63
|
-
|
|
64
|
-
"
|
|
70
|
+
all = [
|
|
71
|
+
"cachellm-proxy[aws,redis,observability,datasets]",
|
|
65
72
|
]
|
|
66
73
|
|
|
67
74
|
[build-system]
|
|
@@ -77,6 +84,7 @@ module-name = "cachellm"
|
|
|
77
84
|
[dependency-groups]
|
|
78
85
|
dev = [
|
|
79
86
|
"mypy>=2.3.1",
|
|
87
|
+
"openai>=3.11.0", # only the SDK-compatibility test needs this
|
|
80
88
|
"pytest>=9.1.1",
|
|
81
89
|
"pytest-asyncio>=1.4.0",
|
|
82
90
|
"pytest-cov>=7.1.0",
|