cachellm-proxy 0.1.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/PKG-INFO +110 -33
  2. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/README.md +100 -25
  3. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/pyproject.toml +14 -9
  4. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/pyproject.toml.orig +20 -12
  5. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/__init__.py +1 -1
  6. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/api/app.py +9 -0
  7. cachellm_proxy-0.2.1/src/cachellm/api/deps.py +180 -0
  8. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/api/routes_admin.py +85 -0
  9. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/api/routes_chat.py +59 -4
  10. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/analytics.py +50 -4
  11. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/exact_store.py +4 -1
  12. cachellm_proxy-0.2.1/src/cachellm/cache/memory.py +454 -0
  13. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/redis_client.py +10 -2
  14. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/service.py +7 -4
  15. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/vector_store.py +8 -0
  16. cachellm_proxy-0.2.1/src/cachellm/cli.py +228 -0
  17. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/base.py +8 -2
  18. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/bedrock.py +12 -2
  19. cachellm_proxy-0.2.1/src/cachellm/providers/catalog.py +350 -0
  20. cachellm_proxy-0.2.1/src/cachellm/providers/detect.py +164 -0
  21. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/registry.py +23 -32
  22. cachellm_proxy-0.2.1/src/cachellm/report.py +217 -0
  23. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/settings.py +30 -1
  24. cachellm_proxy-0.1.0/src/cachellm/api/deps.py +0 -106
  25. cachellm_proxy-0.1.0/src/cachellm/cli.py +0 -122
  26. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/__main__.py +0 -0
  27. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/api/__init__.py +0 -0
  28. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/api/auth.py +0 -0
  29. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/api/sse.py +0 -0
  30. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/__init__.py +0 -0
  31. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/coalesce.py +0 -0
  32. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/entry.py +0 -0
  33. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/keys.py +0 -0
  34. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/policy.py +0 -0
  35. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/embeddings/__init__.py +0 -0
  36. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/embeddings/base.py +0 -0
  37. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/embeddings/fastembed_backend.py +0 -0
  38. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/embeddings/hash_backend.py +0 -0
  39. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/errors.py +0 -0
  40. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/logging_setup.py +0 -0
  41. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/models.py +0 -0
  42. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/observability/__init__.py +0 -0
  43. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/observability/metrics.py +0 -0
  44. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/observability/tracing.py +0 -0
  45. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/pricing.py +0 -0
  46. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/__init__.py +0 -0
  47. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/fake.py +0 -0
  48. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/openai_compat.py +0 -0
  49. {cachellm_proxy-0.1.0 → cachellm_proxy-0.2.1}/src/cachellm/py.typed +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cachellm-proxy
3
- Version: 0.1.0
3
+ Version: 0.2.1
4
4
  Summary: A drop-in semantic cache for OpenAI-compatible LLM APIs: match requests by meaning, serve repeats instantly, cut spend.
5
5
  Keywords: llm,cache,semantic-cache,openai,bedrock,proxy,vector-search,redis,fastapi,llmops,cost-optimization
6
6
  Author: Adarsh Dwivedi
@@ -14,19 +14,17 @@ Classifier: Programming Language :: Python :: 3.12
14
14
  Classifier: Programming Language :: Python :: 3.13
15
15
  Classifier: Topic :: Software Development :: Libraries :: Python Modules
16
16
  Classifier: Typing :: Typed
17
- Requires-Dist: boto3>=1.43.91
18
17
  Requires-Dist: fastapi>=0.141.1
18
+ Requires-Dist: uvicorn[standard]>=0.52.4
19
19
  Requires-Dist: fastembed>=0.8.0
20
- Requires-Dist: httpx>=0.28.1
21
20
  Requires-Dist: numpy>=2.4.6
22
- Requires-Dist: openai>=3.11.0
23
- Requires-Dist: prometheus-client>=0.26.0
24
21
  Requires-Dist: pydantic-settings>=2.15.0
25
- Requires-Dist: redis>=8.1.0
26
- Requires-Dist: redisvl>=0.27.1
22
+ Requires-Dist: prometheus-client>=0.26.0
27
23
  Requires-Dist: structlog>=26.1.0
28
24
  Requires-Dist: typer>=0.27.2
29
- Requires-Dist: uvicorn[standard]>=0.52.4
25
+ Requires-Dist: httpx>=0.28.1
26
+ Requires-Dist: cachellm-proxy[aws,redis,observability,datasets] ; extra == 'all'
27
+ Requires-Dist: boto3>=1.43.91 ; extra == 'aws'
30
28
  Requires-Dist: botocore[crt]>=1.43.91 ; extra == 'aws'
31
29
  Requires-Dist: datasets>=5.0.1 ; extra == 'datasets'
32
30
  Requires-Dist: pandas>=3.0.5 ; extra == 'datasets'
@@ -37,13 +35,17 @@ Requires-Dist: opentelemetry-instrumentation-fastapi>=0.65b0 ; extra == 'observa
37
35
  Requires-Dist: opentelemetry-instrumentation-httpx>=0.65b0 ; extra == 'observability'
38
36
  Requires-Dist: opentelemetry-instrumentation-redis>=0.65b0 ; extra == 'observability'
39
37
  Requires-Dist: opentelemetry-sdk>=1.44.0 ; extra == 'observability'
38
+ Requires-Dist: redis>=8.1.0 ; extra == 'redis'
39
+ Requires-Dist: redisvl>=0.27.1 ; extra == 'redis'
40
40
  Requires-Python: >=3.11
41
41
  Project-URL: Homepage, https://github.com/adarshcod30/CacheLLM
42
42
  Project-URL: Repository, https://github.com/adarshcod30/CacheLLM
43
43
  Project-URL: Issues, https://github.com/adarshcod30/CacheLLM/issues
44
+ Provides-Extra: all
44
45
  Provides-Extra: aws
45
46
  Provides-Extra: datasets
46
47
  Provides-Extra: observability
48
+ Provides-Extra: redis
47
49
  Description-Content-Type: text/markdown
48
50
 
49
51
  # CacheLLM
@@ -51,7 +53,7 @@ Description-Content-Type: text/markdown
51
53
  [![CI](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml/badge.svg)](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml)
52
54
  [![Python 3.11+](https://img.shields.io/badge/python-3.11%20%7C%203.12%20%7C%203.13-blue)](https://www.python.org/)
53
55
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
54
- [![Tests](https://img.shields.io/badge/tests-128%20passing-brightgreen)](tests/)
56
+ [![Tests](https://img.shields.io/badge/tests-265%20passing-brightgreen)](tests/)
55
57
  [![PyPI](https://img.shields.io/pypi/v/cachellm-proxy)](https://pypi.org/project/cachellm-proxy/)
56
58
  [![Hit rate](https://img.shields.io/badge/hit%20rate-77%25%20on%20Bedrock-orange)](docs/evaluation.md)
57
59
 
@@ -111,6 +113,7 @@ The same benchmark runs without any cloud credentials against the built-in fake
111
113
  | Feature | What it does | Why it exists |
112
114
  | --- | --- | --- |
113
115
  | **Drop-in OpenAI API** | Same request and response shape, streaming included, verified against the official `openai` Python SDK in CI | Adoption has to cost one line, or nobody adopts it |
116
+ | **Runs with nothing installed** | Defaults to an in-process numpy store; uses Redis automatically when it can reach one | A cache you have to provision a server for does not get tried. Redis takes over when you actually need shared, durable state |
114
117
  | **Two-tier cache** | Exact-match tier answers literal repeats in about a millisecond without embedding; semantic tier handles rewording | 82% of hits came from the exact tier: free, fast and impossible to get semantically wrong |
115
118
  | **Per-model threshold calibration** | Ships measured safe thresholds for six embedding models and picks the right one automatically | Measured safe thresholds span 0.89 to 0.98. A threshold copied between models is a guess |
116
119
  | **Cacheability policy** | Classifies every prompt and decides cacheable, category, TTL | Creative writing, live data and personal questions must not be cached like a fact |
@@ -134,7 +137,8 @@ The same benchmark runs without any cloud credentials against the built-in fake
134
137
  | API | FastAPI + Uvicorn | Async, native streaming, OpenAPI docs for free |
135
138
  | Validation and config | Pydantic v2, pydantic-settings | Typed request shapes and typed configuration from the environment |
136
139
  | Embeddings | fastembed, `all-MiniLM-L6-v2` (ONNX, CPU) | In-process and about 6 ms. A hosted embedding API would put 100 ms in front of every cache hit and defeat the point |
137
- | Vector store | Redis 8 + RedisVL, HNSW over cosine | One process is the exact-match store, the vector index and the stampede lock. Sub-millisecond, and TTL expiry removes entries from the index for free |
140
+ | Vector store, default | numpy matrix in the proxy's own memory | Nothing to install. Scanning 20,000 cached prompts takes 0.85 ms, where a Redis round trip alone costs 2 to 3 ms, so below roughly 100k entries this is not a compromise, it is faster |
141
+ | Vector store, at scale | Redis 8 + RedisVL, HNSW over cosine | One process is the exact-match store, the vector index and the stampede lock. Shared across workers, survives restarts, and an approximate index starts paying above ~100k entries |
138
142
  | Providers | AWS Bedrock (Converse), any OpenAI-compatible endpoint, deterministic fake | Converse reaches Nova, Claude, Llama and Mistral with one request shape and no extra vendor keys |
139
143
  | Metrics | prometheus-client, Prometheus, Grafana | Dashboard ships provisioned, so a reviewer sees data on first boot |
140
144
  | Tracing | OpenTelemetry, optional, GenAI semantic conventions | Instrument once, export to Langfuse, Tempo or Jaeger |
@@ -270,32 +274,47 @@ MiniLM gives four times the safe recall of bge-small at a third of the download
270
274
 
271
275
  ## Quick start
272
276
 
273
- You need Python 3.11 or newer and a Redis 8 instance. Redis 8 is required because the vector index needs the query engine, and it must be database 0 because Redis Search only indexes that one.
277
+ Python 3.11 or newer, and nothing else.
274
278
 
275
279
  ```bash
276
280
  pip install cachellm-proxy
281
+ cachellm serve
277
282
  ```
278
283
 
279
- The distribution is `cachellm-proxy` because PyPI blocks `cachellm` as too close to an existing `cachelm`. The import name and the CLI are both still `cachellm`.
284
+ That is the whole install. No Redis, no Docker, no config file. On startup it looks for a provider you already have: an API key in your environment, Ollama running locally, or AWS credentials. It logs which one it picked and how to override it.
280
285
 
281
- Or from source, which is what you want if you plan to change anything:
286
+ To see what it found before starting anything:
282
287
 
283
288
  ```bash
284
- git clone https://github.com/adarshcod30/CacheLLM.git
285
- cd CacheLLM
286
- uv sync
289
+ cachellm providers
287
290
  ```
288
291
 
289
- Start Redis if you do not have one running:
292
+ To try it with no account at all, the built-in test double stands in for a model:
290
293
 
291
294
  ```bash
292
- docker run -d --name redis -p 6379:6379 redis:8-alpine
295
+ CACHELLM_DEFAULT_PROVIDER=fake CACHELLM_FAKE_LATENCY_MS=600 cachellm serve
293
296
  ```
294
297
 
295
- Run the proxy. With no configuration it uses the built-in fake provider, so you can see it working before wiring up a real model or spending anything:
298
+ You only install what you actually route to. The base package is the proxy plus the local embedding model; AWS and Redis are extras, and nothing here installs Grafana or a Prometheus server, which are separate programs rather than Python packages.
299
+
300
+ | You want | Install |
301
+ | --- | --- |
302
+ | Any OpenAI-compatible endpoint: OpenAI, Groq, Gemini, Ollama, vLLM | `pip install cachellm-proxy` |
303
+ | AWS Bedrock | `pip install "cachellm-proxy[aws]"` |
304
+ | Redis instead of the in-process cache | `pip install "cachellm-proxy[redis]"` |
305
+ | Tracing to Langfuse or Tempo | `pip install "cachellm-proxy[observability]"` |
306
+ | Everything | `pip install "cachellm-proxy[all]"` |
307
+
308
+ Ask for a provider or backend whose extra is missing and the proxy names the exact command to fix it rather than raising an import error.
309
+
310
+ The distribution is `cachellm-proxy` because PyPI blocks `cachellm` as too close to an existing `cachelm`. The import name and the CLI are both still `cachellm`.
311
+
312
+ From source, if you plan to change anything:
296
313
 
297
314
  ```bash
298
- CACHELLM_DEFAULT_PROVIDER=fake CACHELLM_FAKE_LATENCY_MS=600 uv run cachellm serve
315
+ git clone https://github.com/adarshcod30/CacheLLM.git
316
+ cd CacheLLM
317
+ uv sync && uv run cachellm serve
299
318
  ```
300
319
 
301
320
  In another terminal, ask the same thing twice and watch the second one come back instantly:
@@ -342,28 +361,53 @@ make tune && make compare-models && make bench
342
361
 
343
362
  ### Using a real provider
344
363
 
345
- AWS Bedrock needs no extra vendor keys and reaches Nova, Claude, Llama and Mistral through one API:
364
+ One environment variable per host. Everything except Bedrock speaks the OpenAI protocol, so a single adapter covers all of it and needs no extra.
365
+
366
+ | Host | `CACHELLM_OPENAI_BASE_URL` | Example model |
367
+ | --- | --- | --- |
368
+ | OpenAI | `https://api.openai.com/v1` | `gpt-4o-mini` |
369
+ | Groq | `https://api.groq.com/openai/v1` | `llama-3.3-70b-versatile` |
370
+ | Google Gemini | `https://generativelanguage.googleapis.com/v1beta/openai/` | `gemini-2.5-flash` |
371
+ | Anthropic | `https://api.anthropic.com/v1` | `claude-haiku-4-5` |
372
+ | OpenRouter | `https://openrouter.ai/api/v1` | `anthropic/claude-3.5-sonnet` |
373
+ | Together | `https://api.together.xyz/v1` | `meta-llama/Llama-3.3-70B-Instruct-Turbo` |
374
+ | Ollama, local | `http://localhost:11434/v1` | `llama3.2` |
375
+ | vLLM or LM Studio | `http://localhost:8000/v1` | whatever you served |
346
376
 
347
377
  ```bash
348
- CACHELLM_DEFAULT_PROVIDER=bedrock AWS_REGION=us-east-1 uv run cachellm serve
378
+ CACHELLM_OPENAI_BASE_URL=https://api.groq.com/openai/v1 \
379
+ CACHELLM_OPENAI_API_KEY=gsk_your_key \
380
+ cachellm serve
349
381
  ```
350
382
 
351
- If your credentials come from `aws login` rather than static keys or an SSO profile, install the CRT extra once, because that credential provider needs it:
383
+ **AWS Bedrock** is the one exception, because it does not speak the OpenAI protocol. It needs the `aws` extra, and then uses your existing AWS credentials with no vendor API key at all:
352
384
 
353
385
  ```bash
354
- uv sync --extra aws
386
+ pip install "cachellm-proxy[aws]"
387
+ CACHELLM_DEFAULT_PROVIDER=bedrock AWS_REGION=us-east-1 cachellm serve
355
388
  ```
356
389
 
357
- The proxy detects that case and says so in the error rather than passing along boto's version of the message.
358
-
359
390
  ```bash
360
391
  curl -s http://localhost:8080/v1/chat/completions -H 'Content-Type: application/json' -d '{"model":"bedrock/us.amazon.nova-micro-v1:0","temperature":0,"messages":[{"role":"user","content":"What is Redis used for?"}]}'
361
392
  ```
362
393
 
363
- Any OpenAI-compatible endpoint works too, which covers OpenAI, Groq, Together, OpenRouter and a local vLLM:
394
+ If your credentials come from `aws login` rather than static keys or an SSO profile, that provider needs the CRT extra, which the `aws` extra already includes. The proxy detects that case and says so in the error rather than passing along boto's version of the message.
395
+
396
+ ### How a model name gets routed
397
+
398
+ Three rules, in order. You never configure a model list.
399
+
400
+ 1. **An explicit prefix** this proxy owns: `bedrock/…`, `openai/…`, `fake/…`.
401
+ 2. **A recognisable vendor convention.** Bedrock ids are always `vendor.model`, so `amazon.nova-lite-v1:0` and `us.anthropic.claude-3-haiku-20240307-v1:0` are identified with no prefix. OpenAI's own families (`gpt-`, `o1`, `o3`, `text-embedding-`) are identified the same way, whatever the default is.
402
+ 3. **Otherwise the configured default**, which is the OpenAI-compatible adapter. Names like `llama3.2`, `mixtral-8x7b-32768` and `qwen2.5-coder:7b` are served by Groq, Ollama, Together and OpenRouter alike, so the endpoint you configured is the only sensible answer.
403
+
404
+ Model ids that legitimately contain a slash, which OpenRouter and Together both use, are forwarded whole. Only this proxy's own prefixes are stripped. Note that `anthropic.claude-…` with a dot is a Bedrock id while `anthropic/claude-…` with a slash is an OpenRouter id, and the router tells them apart.
405
+
406
+ Ask it directly if you are unsure:
364
407
 
365
408
  ```bash
366
- CACHELLM_DEFAULT_PROVIDER=openai CACHELLM_OPENAI_BASE_URL=https://api.groq.com/openai/v1 CACHELLM_OPENAI_API_KEY=... uv run cachellm serve
409
+ curl -s localhost:8080/admin/route/meta-llama/Llama-3.3-70B-Instruct-Turbo
410
+ curl -s localhost:8080/admin/providers
367
411
  ```
368
412
 
369
413
  ## API reference
@@ -400,7 +444,10 @@ Clients can steer per request with `X-Cache-Control`:
400
444
  | --- | --- |
401
445
  | `GET /admin/stats` | Hit rate, tier split, money saved, latency percentiles, entry count |
402
446
  | `GET /admin/config` | Effective thresholds, TTLs and rules |
447
+ | `GET /admin/providers` | Every host, its base URL, its pip extra, example model ids |
448
+ | `GET /admin/route/{model}` | Where one model name would go, and why |
403
449
  | `POST /admin/invalidate` | Drop by `namespace`, by `model`, or `all` |
450
+ | `GET /admin/requests` | Recent request log: what the cache did with each one |
404
451
  | `GET /admin/near-misses` | Recent lookups that landed just below threshold |
405
452
  | `GET /admin/near-miss-histogram` | Cumulative view of what a lower threshold would buy |
406
453
  | `GET /admin/entries` | Inspect what is stored |
@@ -418,20 +465,50 @@ curl -s http://localhost:8080/admin/threshold-sweep -H 'Content-Type: applicatio
418
465
  ### Command line
419
466
 
420
467
  ```bash
421
- uv run cachellm serve # run the proxy
422
- uv run cachellm stats # cache statistics
423
- uv run cachellm config # effective configuration
424
- uv run cachellm invalidate --all
425
- uv run cachellm tune pairs.jsonl
468
+ cachellm serve # run the proxy, auto-detecting a provider
469
+ cachellm providers # every host, and what this machine can reach
470
+ cachellm stats # hit rate, savings, latency, recent requests
471
+ cachellm watch # follow requests live, like tail -f
472
+ cachellm config # effective configuration
473
+ cachellm invalidate --all
474
+ cachellm tune pairs.jsonl
426
475
  ```
427
476
 
477
+ `cachellm stats` is the dashboard, in your terminal:
478
+
479
+ ```
480
+ CacheLLM · memory backend · all-MiniLM-L6-v2
481
+
482
+ HIT RATE 77.0% ████████████████░░░░░░ 1,540 of 2,000 requests
483
+ 1,265 exact · 275 semantic · 460 missed · 0 bypassed
484
+
485
+ SAVED $0.0135 spent $0.0038 78% lower
486
+ 116,029 tokens never generated
487
+
488
+ LATENCY cached 2.6 ms p50 · 5.7 ms p95
489
+ uncached 797.0 ms p50 · 1022.8 ms p95 179x faster
490
+
491
+ CACHE 451 entries · 9 coalesced · 17 near misses
492
+
493
+ time result tier latency score saved prompt
494
+ 02:30:55 HIT exact 1.2ms 1.000 +$0.000010 What is Redis used for?
495
+ 02:30:54 BYPASS 451.5ms · · What is my order 1234… (pii:long_digits)
496
+ 02:30:53 HIT semantic 7.4ms 0.952 +$0.000011 What is Redis typically used for?
497
+ 02:30:53 MISS 454.8ms · · How do I configure nginx for TLS?
498
+ ```
499
+
500
+ Prometheus and Grafana are still there if you want history and alerting, but nothing needs them.
501
+
428
502
  ## Configuration
429
503
 
430
504
  Every setting is an environment variable prefixed `CACHELLM_`, or a line in `.env`. Copy `.env.example` to start. The ones that matter most:
431
505
 
432
506
  | Variable | Default | Notes |
433
507
  | --- | --- | --- |
508
+ | `CACHELLM_BACKEND` | `auto` | `memory` needs nothing, `redis` shares one cache across workers, `auto` uses Redis when reachable and memory when not |
434
509
  | `CACHELLM_REDIS_URL` | `redis://localhost:6379/0` | Must be database 0. Redis Search cannot index any other |
510
+ | `CACHELLM_MEMORY_MAX_ENTRIES` | `50000` | Cap for the in-memory store. 50k of 384-dim vectors is about 73 MB |
511
+ | `CACHELLM_MEMORY_SNAPSHOT_PATH` | unset | Persist the in-memory cache to this file so a restart does not start cold |
435
512
  | `CACHELLM_API_KEYS` | empty | Comma-separated client keys. Empty disables auth, which is local development only |
436
513
  | `CACHELLM_EMBEDDING_MODEL` | `all-MiniLM-L6-v2` | Changing this changes the safe threshold. See the calibration table |
437
514
  | `CACHELLM_THRESHOLD_*` | `0` | Zero means use the calibrated value for your model |
@@ -3,7 +3,7 @@
3
3
  [![CI](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml/badge.svg)](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml)
4
4
  [![Python 3.11+](https://img.shields.io/badge/python-3.11%20%7C%203.12%20%7C%203.13-blue)](https://www.python.org/)
5
5
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
6
- [![Tests](https://img.shields.io/badge/tests-128%20passing-brightgreen)](tests/)
6
+ [![Tests](https://img.shields.io/badge/tests-265%20passing-brightgreen)](tests/)
7
7
  [![PyPI](https://img.shields.io/pypi/v/cachellm-proxy)](https://pypi.org/project/cachellm-proxy/)
8
8
  [![Hit rate](https://img.shields.io/badge/hit%20rate-77%25%20on%20Bedrock-orange)](docs/evaluation.md)
9
9
 
@@ -63,6 +63,7 @@ The same benchmark runs without any cloud credentials against the built-in fake
63
63
  | Feature | What it does | Why it exists |
64
64
  | --- | --- | --- |
65
65
  | **Drop-in OpenAI API** | Same request and response shape, streaming included, verified against the official `openai` Python SDK in CI | Adoption has to cost one line, or nobody adopts it |
66
+ | **Runs with nothing installed** | Defaults to an in-process numpy store; uses Redis automatically when it can reach one | A cache you have to provision a server for does not get tried. Redis takes over when you actually need shared, durable state |
66
67
  | **Two-tier cache** | Exact-match tier answers literal repeats in about a millisecond without embedding; semantic tier handles rewording | 82% of hits came from the exact tier: free, fast and impossible to get semantically wrong |
67
68
  | **Per-model threshold calibration** | Ships measured safe thresholds for six embedding models and picks the right one automatically | Measured safe thresholds span 0.89 to 0.98. A threshold copied between models is a guess |
68
69
  | **Cacheability policy** | Classifies every prompt and decides cacheable, category, TTL | Creative writing, live data and personal questions must not be cached like a fact |
@@ -86,7 +87,8 @@ The same benchmark runs without any cloud credentials against the built-in fake
86
87
  | API | FastAPI + Uvicorn | Async, native streaming, OpenAPI docs for free |
87
88
  | Validation and config | Pydantic v2, pydantic-settings | Typed request shapes and typed configuration from the environment |
88
89
  | Embeddings | fastembed, `all-MiniLM-L6-v2` (ONNX, CPU) | In-process and about 6 ms. A hosted embedding API would put 100 ms in front of every cache hit and defeat the point |
89
- | Vector store | Redis 8 + RedisVL, HNSW over cosine | One process is the exact-match store, the vector index and the stampede lock. Sub-millisecond, and TTL expiry removes entries from the index for free |
90
+ | Vector store, default | numpy matrix in the proxy's own memory | Nothing to install. Scanning 20,000 cached prompts takes 0.85 ms, where a Redis round trip alone costs 2 to 3 ms, so below roughly 100k entries this is not a compromise, it is faster |
91
+ | Vector store, at scale | Redis 8 + RedisVL, HNSW over cosine | One process is the exact-match store, the vector index and the stampede lock. Shared across workers, survives restarts, and an approximate index starts paying above ~100k entries |
90
92
  | Providers | AWS Bedrock (Converse), any OpenAI-compatible endpoint, deterministic fake | Converse reaches Nova, Claude, Llama and Mistral with one request shape and no extra vendor keys |
91
93
  | Metrics | prometheus-client, Prometheus, Grafana | Dashboard ships provisioned, so a reviewer sees data on first boot |
92
94
  | Tracing | OpenTelemetry, optional, GenAI semantic conventions | Instrument once, export to Langfuse, Tempo or Jaeger |
@@ -222,32 +224,47 @@ MiniLM gives four times the safe recall of bge-small at a third of the download
222
224
 
223
225
  ## Quick start
224
226
 
225
- You need Python 3.11 or newer and a Redis 8 instance. Redis 8 is required because the vector index needs the query engine, and it must be database 0 because Redis Search only indexes that one.
227
+ Python 3.11 or newer, and nothing else.
226
228
 
227
229
  ```bash
228
230
  pip install cachellm-proxy
231
+ cachellm serve
229
232
  ```
230
233
 
231
- The distribution is `cachellm-proxy` because PyPI blocks `cachellm` as too close to an existing `cachelm`. The import name and the CLI are both still `cachellm`.
234
+ That is the whole install. No Redis, no Docker, no config file. On startup it looks for a provider you already have: an API key in your environment, Ollama running locally, or AWS credentials. It logs which one it picked and how to override it.
232
235
 
233
- Or from source, which is what you want if you plan to change anything:
236
+ To see what it found before starting anything:
234
237
 
235
238
  ```bash
236
- git clone https://github.com/adarshcod30/CacheLLM.git
237
- cd CacheLLM
238
- uv sync
239
+ cachellm providers
239
240
  ```
240
241
 
241
- Start Redis if you do not have one running:
242
+ To try it with no account at all, the built-in test double stands in for a model:
242
243
 
243
244
  ```bash
244
- docker run -d --name redis -p 6379:6379 redis:8-alpine
245
+ CACHELLM_DEFAULT_PROVIDER=fake CACHELLM_FAKE_LATENCY_MS=600 cachellm serve
245
246
  ```
246
247
 
247
- Run the proxy. With no configuration it uses the built-in fake provider, so you can see it working before wiring up a real model or spending anything:
248
+ You only install what you actually route to. The base package is the proxy plus the local embedding model; AWS and Redis are extras, and nothing here installs Grafana or a Prometheus server, which are separate programs rather than Python packages.
249
+
250
+ | You want | Install |
251
+ | --- | --- |
252
+ | Any OpenAI-compatible endpoint: OpenAI, Groq, Gemini, Ollama, vLLM | `pip install cachellm-proxy` |
253
+ | AWS Bedrock | `pip install "cachellm-proxy[aws]"` |
254
+ | Redis instead of the in-process cache | `pip install "cachellm-proxy[redis]"` |
255
+ | Tracing to Langfuse or Tempo | `pip install "cachellm-proxy[observability]"` |
256
+ | Everything | `pip install "cachellm-proxy[all]"` |
257
+
258
+ Ask for a provider or backend whose extra is missing and the proxy names the exact command to fix it rather than raising an import error.
259
+
260
+ The distribution is `cachellm-proxy` because PyPI blocks `cachellm` as too close to an existing `cachelm`. The import name and the CLI are both still `cachellm`.
261
+
262
+ From source, if you plan to change anything:
248
263
 
249
264
  ```bash
250
- CACHELLM_DEFAULT_PROVIDER=fake CACHELLM_FAKE_LATENCY_MS=600 uv run cachellm serve
265
+ git clone https://github.com/adarshcod30/CacheLLM.git
266
+ cd CacheLLM
267
+ uv sync && uv run cachellm serve
251
268
  ```
252
269
 
253
270
  In another terminal, ask the same thing twice and watch the second one come back instantly:
@@ -294,28 +311,53 @@ make tune && make compare-models && make bench
294
311
 
295
312
  ### Using a real provider
296
313
 
297
- AWS Bedrock needs no extra vendor keys and reaches Nova, Claude, Llama and Mistral through one API:
314
+ One environment variable per host. Everything except Bedrock speaks the OpenAI protocol, so a single adapter covers all of it and needs no extra.
315
+
316
+ | Host | `CACHELLM_OPENAI_BASE_URL` | Example model |
317
+ | --- | --- | --- |
318
+ | OpenAI | `https://api.openai.com/v1` | `gpt-4o-mini` |
319
+ | Groq | `https://api.groq.com/openai/v1` | `llama-3.3-70b-versatile` |
320
+ | Google Gemini | `https://generativelanguage.googleapis.com/v1beta/openai/` | `gemini-2.5-flash` |
321
+ | Anthropic | `https://api.anthropic.com/v1` | `claude-haiku-4-5` |
322
+ | OpenRouter | `https://openrouter.ai/api/v1` | `anthropic/claude-3.5-sonnet` |
323
+ | Together | `https://api.together.xyz/v1` | `meta-llama/Llama-3.3-70B-Instruct-Turbo` |
324
+ | Ollama, local | `http://localhost:11434/v1` | `llama3.2` |
325
+ | vLLM or LM Studio | `http://localhost:8000/v1` | whatever you served |
298
326
 
299
327
  ```bash
300
- CACHELLM_DEFAULT_PROVIDER=bedrock AWS_REGION=us-east-1 uv run cachellm serve
328
+ CACHELLM_OPENAI_BASE_URL=https://api.groq.com/openai/v1 \
329
+ CACHELLM_OPENAI_API_KEY=gsk_your_key \
330
+ cachellm serve
301
331
  ```
302
332
 
303
- If your credentials come from `aws login` rather than static keys or an SSO profile, install the CRT extra once, because that credential provider needs it:
333
+ **AWS Bedrock** is the one exception, because it does not speak the OpenAI protocol. It needs the `aws` extra, and then uses your existing AWS credentials with no vendor API key at all:
304
334
 
305
335
  ```bash
306
- uv sync --extra aws
336
+ pip install "cachellm-proxy[aws]"
337
+ CACHELLM_DEFAULT_PROVIDER=bedrock AWS_REGION=us-east-1 cachellm serve
307
338
  ```
308
339
 
309
- The proxy detects that case and says so in the error rather than passing along boto's version of the message.
310
-
311
340
  ```bash
312
341
  curl -s http://localhost:8080/v1/chat/completions -H 'Content-Type: application/json' -d '{"model":"bedrock/us.amazon.nova-micro-v1:0","temperature":0,"messages":[{"role":"user","content":"What is Redis used for?"}]}'
313
342
  ```
314
343
 
315
- Any OpenAI-compatible endpoint works too, which covers OpenAI, Groq, Together, OpenRouter and a local vLLM:
344
+ If your credentials come from `aws login` rather than static keys or an SSO profile, that provider needs the CRT extra, which the `aws` extra already includes. The proxy detects that case and says so in the error rather than passing along boto's version of the message.
345
+
346
+ ### How a model name gets routed
347
+
348
+ Three rules, in order. You never configure a model list.
349
+
350
+ 1. **An explicit prefix** this proxy owns: `bedrock/…`, `openai/…`, `fake/…`.
351
+ 2. **A recognisable vendor convention.** Bedrock ids are always `vendor.model`, so `amazon.nova-lite-v1:0` and `us.anthropic.claude-3-haiku-20240307-v1:0` are identified with no prefix. OpenAI's own families (`gpt-`, `o1`, `o3`, `text-embedding-`) are identified the same way, whatever the default is.
352
+ 3. **Otherwise the configured default**, which is the OpenAI-compatible adapter. Names like `llama3.2`, `mixtral-8x7b-32768` and `qwen2.5-coder:7b` are served by Groq, Ollama, Together and OpenRouter alike, so the endpoint you configured is the only sensible answer.
353
+
354
+ Model ids that legitimately contain a slash, which OpenRouter and Together both use, are forwarded whole. Only this proxy's own prefixes are stripped. Note that `anthropic.claude-…` with a dot is a Bedrock id while `anthropic/claude-…` with a slash is an OpenRouter id, and the router tells them apart.
355
+
356
+ Ask it directly if you are unsure:
316
357
 
317
358
  ```bash
318
- CACHELLM_DEFAULT_PROVIDER=openai CACHELLM_OPENAI_BASE_URL=https://api.groq.com/openai/v1 CACHELLM_OPENAI_API_KEY=... uv run cachellm serve
359
+ curl -s localhost:8080/admin/route/meta-llama/Llama-3.3-70B-Instruct-Turbo
360
+ curl -s localhost:8080/admin/providers
319
361
  ```
320
362
 
321
363
  ## API reference
@@ -352,7 +394,10 @@ Clients can steer per request with `X-Cache-Control`:
352
394
  | --- | --- |
353
395
  | `GET /admin/stats` | Hit rate, tier split, money saved, latency percentiles, entry count |
354
396
  | `GET /admin/config` | Effective thresholds, TTLs and rules |
397
+ | `GET /admin/providers` | Every host, its base URL, its pip extra, example model ids |
398
+ | `GET /admin/route/{model}` | Where one model name would go, and why |
355
399
  | `POST /admin/invalidate` | Drop by `namespace`, by `model`, or `all` |
400
+ | `GET /admin/requests` | Recent request log: what the cache did with each one |
356
401
  | `GET /admin/near-misses` | Recent lookups that landed just below threshold |
357
402
  | `GET /admin/near-miss-histogram` | Cumulative view of what a lower threshold would buy |
358
403
  | `GET /admin/entries` | Inspect what is stored |
@@ -370,12 +415,39 @@ curl -s http://localhost:8080/admin/threshold-sweep -H 'Content-Type: applicatio
370
415
  ### Command line
371
416
 
372
417
  ```bash
373
- uv run cachellm serve # run the proxy
374
- uv run cachellm stats # cache statistics
375
- uv run cachellm config # effective configuration
376
- uv run cachellm invalidate --all
377
- uv run cachellm tune pairs.jsonl
418
+ cachellm serve # run the proxy, auto-detecting a provider
419
+ cachellm providers # every host, and what this machine can reach
420
+ cachellm stats # hit rate, savings, latency, recent requests
421
+ cachellm watch # follow requests live, like tail -f
422
+ cachellm config # effective configuration
423
+ cachellm invalidate --all
424
+ cachellm tune pairs.jsonl
425
+ ```
426
+
427
+ `cachellm stats` is the dashboard, in your terminal:
428
+
378
429
  ```
430
+ CacheLLM · memory backend · all-MiniLM-L6-v2
431
+
432
+ HIT RATE 77.0% ████████████████░░░░░░ 1,540 of 2,000 requests
433
+ 1,265 exact · 275 semantic · 460 missed · 0 bypassed
434
+
435
+ SAVED $0.0135 spent $0.0038 78% lower
436
+ 116,029 tokens never generated
437
+
438
+ LATENCY cached 2.6 ms p50 · 5.7 ms p95
439
+ uncached 797.0 ms p50 · 1022.8 ms p95 179x faster
440
+
441
+ CACHE 451 entries · 9 coalesced · 17 near misses
442
+
443
+ time result tier latency score saved prompt
444
+ 02:30:55 HIT exact 1.2ms 1.000 +$0.000010 What is Redis used for?
445
+ 02:30:54 BYPASS 451.5ms · · What is my order 1234… (pii:long_digits)
446
+ 02:30:53 HIT semantic 7.4ms 0.952 +$0.000011 What is Redis typically used for?
447
+ 02:30:53 MISS 454.8ms · · How do I configure nginx for TLS?
448
+ ```
449
+
450
+ Prometheus and Grafana are still there if you want history and alerting, but nothing needs them.
379
451
 
380
452
  ## Configuration
381
453
 
@@ -383,7 +455,10 @@ Every setting is an environment variable prefixed `CACHELLM_`, or a line in `.en
383
455
 
384
456
  | Variable | Default | Notes |
385
457
  | --- | --- | --- |
458
+ | `CACHELLM_BACKEND` | `auto` | `memory` needs nothing, `redis` shares one cache across workers, `auto` uses Redis when reachable and memory when not |
386
459
  | `CACHELLM_REDIS_URL` | `redis://localhost:6379/0` | Must be database 0. Redis Search cannot index any other |
460
+ | `CACHELLM_MEMORY_MAX_ENTRIES` | `50000` | Cap for the in-memory store. 50k of 384-dim vectors is about 73 MB |
461
+ | `CACHELLM_MEMORY_SNAPSHOT_PATH` | unset | Persist the in-memory cache to this file so a restart does not start cold |
387
462
  | `CACHELLM_API_KEYS` | empty | Comma-separated client keys. Empty disables auth, which is local development only |
388
463
  | `CACHELLM_EMBEDDING_MODEL` | `all-MiniLM-L6-v2` | Changing this changes the safe threshold. See the calibration table |
389
464
  | `CACHELLM_THRESHOLD_*` | `0` | Zero means use the calibrated value for your model |
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cachellm-proxy"
3
- version = "0.1.0"
3
+ version = "0.2.1"
4
4
  description = "A drop-in semantic cache for OpenAI-compatible LLM APIs: match requests by meaning, serve repeats instantly, cut spend."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
@@ -29,19 +29,15 @@ classifiers = [
29
29
  "Typing :: Typed",
30
30
  ]
31
31
  dependencies = [
32
- "boto3>=1.43.91",
33
32
  "fastapi>=0.141.1",
33
+ "uvicorn[standard]>=0.52.4",
34
34
  "fastembed>=0.8.0",
35
- "httpx>=0.28.1",
36
35
  "numpy>=2.4.6",
37
- "openai>=3.11.0",
38
- "prometheus-client>=0.26.0",
39
36
  "pydantic-settings>=2.15.0",
40
- "redis>=8.1.0",
41
- "redisvl>=0.27.1",
37
+ "prometheus-client>=0.26.0",
42
38
  "structlog>=26.1.0",
43
39
  "typer>=0.27.2",
44
- "uvicorn[standard]>=0.52.4",
40
+ "httpx>=0.28.1",
45
41
  ]
46
42
 
47
43
  [[project.authors]]
@@ -57,6 +53,14 @@ Issues = "https://github.com/adarshcod30/CacheLLM/issues"
57
53
  cachellm = "cachellm:main"
58
54
 
59
55
  [project.optional-dependencies]
56
+ aws = [
57
+ "boto3>=1.43.91",
58
+ "botocore[crt]>=1.43.91",
59
+ ]
60
+ redis = [
61
+ "redis>=8.1.0",
62
+ "redisvl>=0.27.1",
63
+ ]
60
64
  observability = [
61
65
  "langfuse>=4.15.2",
62
66
  "opentelemetry-exporter-otlp-proto-http>=1.44.0",
@@ -70,7 +74,7 @@ datasets = [
70
74
  "datasets>=5.0.1",
71
75
  "pandas>=3.0.5",
72
76
  ]
73
- aws = ["botocore[crt]>=1.43.91"]
77
+ all = ["cachellm-proxy[aws,redis,observability,datasets]"]
74
78
 
75
79
  [build-system]
76
80
  requires = ["uv_build>=0.11.23,<0.12.0"]
@@ -166,6 +170,7 @@ exclude_lines = [
166
170
  [dependency-groups]
167
171
  dev = [
168
172
  "mypy>=2.3.1",
173
+ "openai>=3.11.0",
169
174
  "pytest>=9.1.1",
170
175
  "pytest-asyncio>=1.4.0",
171
176
  "pytest-cov>=7.1.0",
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cachellm-proxy"
3
- version = "0.1.0"
3
+ version = "0.2.1"
4
4
  description = "A drop-in semantic cache for OpenAI-compatible LLM APIs: match requests by meaning, serve repeats instantly, cut spend."
5
5
  readme = "README.md"
6
6
  authors = [
@@ -23,19 +23,17 @@ classifiers = [
23
23
  "Typing :: Typed",
24
24
  ]
25
25
  dependencies = [
26
- "boto3>=1.43.91",
26
+ # The proxy itself and the semantic matching. Nothing here is optional:
27
+ # this is the smallest set that can cache by meaning and serve HTTP.
27
28
  "fastapi>=0.141.1",
28
- "fastembed>=0.8.0",
29
- "httpx>=0.28.1",
30
- "numpy>=2.4.6",
31
- "openai>=3.11.0",
32
- "prometheus-client>=0.26.0",
29
+ "uvicorn[standard]>=0.52.4",
30
+ "fastembed>=0.8.0", # local embeddings, CPU, no API call
31
+ "numpy>=2.4.6", # the default vector search is one matrix multiply
33
32
  "pydantic-settings>=2.15.0",
34
- "redis>=8.1.0",
35
- "redisvl>=0.27.1",
33
+ "prometheus-client>=0.26.0", # ~50 KB: writes a text page, not a server
36
34
  "structlog>=26.1.0",
37
35
  "typer>=0.27.2",
38
- "uvicorn[standard]>=0.52.4",
36
+ "httpx>=0.28.1",
39
37
  ]
40
38
 
41
39
  [project.urls]
@@ -47,6 +45,15 @@ Issues = "https://github.com/adarshcod30/CacheLLM/issues"
47
45
  cachellm = "cachellm:main"
48
46
 
49
47
  [project.optional-dependencies]
48
+ # Only install what you actually route to.
49
+ aws = [
50
+ "boto3>=1.43.91",
51
+ "botocore[crt]>=1.43.91", # `aws login` credentials need the CRT extra
52
+ ]
53
+ redis = [
54
+ "redis>=8.1.0",
55
+ "redisvl>=0.27.1",
56
+ ]
50
57
  observability = [
51
58
  "langfuse>=4.15.2",
52
59
  "opentelemetry-exporter-otlp-proto-http>=1.44.0",
@@ -60,8 +67,8 @@ datasets = [
60
67
  "datasets>=5.0.1",
61
68
  "pandas>=3.0.5",
62
69
  ]
63
- aws = [
64
- "botocore[crt]>=1.43.91",
70
+ all = [
71
+ "cachellm-proxy[aws,redis,observability,datasets]",
65
72
  ]
66
73
 
67
74
  [build-system]
@@ -77,6 +84,7 @@ module-name = "cachellm"
77
84
  [dependency-groups]
78
85
  dev = [
79
86
  "mypy>=2.3.1",
87
+ "openai>=3.11.0", # only the SDK-compatibility test needs this
80
88
  "pytest>=9.1.1",
81
89
  "pytest-asyncio>=1.4.0",
82
90
  "pytest-cov>=7.1.0",
@@ -2,7 +2,7 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
- __version__ = "0.1.0"
5
+ __version__ = "0.2.1"
6
6
 
7
7
  __all__ = ["__version__", "main"]
8
8