cachellm-proxy 0.2.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/PKG-INFO +50 -11
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/README.md +49 -10
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/pyproject.toml +1 -1
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/pyproject.toml.orig +1 -1
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/__init__.py +1 -1
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/api/deps.py +57 -5
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/api/routes_admin.py +17 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/api/routes_chat.py +31 -10
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/api/sse.py +11 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/cli.py +79 -57
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/embeddings/fastembed_backend.py +15 -3
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/pricing.py +19 -4
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/providers/catalog.py +6 -8
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/providers/detect.py +12 -8
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/providers/openai_compat.py +23 -1
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/report.py +7 -1
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/__main__.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/api/__init__.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/api/app.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/api/auth.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/cache/__init__.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/cache/analytics.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/cache/coalesce.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/cache/entry.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/cache/exact_store.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/cache/keys.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/cache/memory.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/cache/policy.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/cache/redis_client.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/cache/service.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/cache/vector_store.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/embeddings/__init__.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/embeddings/base.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/embeddings/hash_backend.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/errors.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/logging_setup.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/models.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/observability/__init__.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/observability/metrics.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/observability/tracing.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/providers/__init__.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/providers/base.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/providers/bedrock.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/providers/fake.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/providers/registry.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/py.typed +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.2}/src/cachellm/settings.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cachellm-proxy
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: A drop-in semantic cache for OpenAI-compatible LLM APIs: match requests by meaning, serve repeats instantly, cut spend.
|
|
5
5
|
Keywords: llm,cache,semantic-cache,openai,bedrock,proxy,vector-search,redis,fastapi,llmops,cost-optimization
|
|
6
6
|
Author: Adarsh Dwivedi
|
|
@@ -53,13 +53,13 @@ Description-Content-Type: text/markdown
|
|
|
53
53
|
[](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml)
|
|
54
54
|
[](https://www.python.org/)
|
|
55
55
|
[](LICENSE)
|
|
56
|
-
[](tests/)
|
|
57
57
|
[](https://pypi.org/project/cachellm-proxy/)
|
|
58
58
|
[](docs/evaluation.md)
|
|
59
59
|
|
|
60
60
|
**A drop-in semantic cache for OpenAI-compatible LLM APIs. Change one base URL, and questions your model has already answered come back in milliseconds instead of seconds.**
|
|
61
61
|
|
|
62
|
-
On a 2,000-request replay against **AWS Bedrock** it served **77% of traffic from cache** with **zero false positives** on genuinely new questions, cutting spend by **78%** and p95 latency from 1,023 ms to **5.7 ms
|
|
62
|
+
On a 2,000-request replay against **AWS Bedrock** it served **77% of traffic from cache** with **zero false positives** on genuinely new questions, cutting spend by **78%** and p95 latency from 1,023 ms to **5.7 ms** on a cache hit, or 9.7 ms when the question was reworded.
|
|
63
63
|
|
|
64
64
|
[Quick start](#quick-start) · [How it works](#how-it-works) · [Evaluation](docs/evaluation.md) · [API reference](#api-reference) · [Deployment](#deployment-and-infrastructure)
|
|
65
65
|
|
|
@@ -97,7 +97,8 @@ Everything else stays the same: same request shape, same response shape, same er
|
|
|
97
97
|
| False positives on genuinely new questions | **0 of 368** |
|
|
98
98
|
| Reworded repeats served from cache | 95.2% |
|
|
99
99
|
| Exact repeats served from cache | 99.3% |
|
|
100
|
-
| p95 latency, cache hit | **5.7 ms** |
|
|
100
|
+
| p95 latency, any cache hit | **5.7 ms** |
|
|
101
|
+
| p95 latency, reworded hit | 9.7 ms |
|
|
101
102
|
| p95 latency, cache miss | 1,022.8 ms |
|
|
102
103
|
| p95 speedup | **178.8x** |
|
|
103
104
|
| Cost reduction | **78.0%** |
|
|
@@ -257,7 +258,7 @@ This is where a caching project usually hand-waves. The full write-up is in [doc
|
|
|
257
258
|
|
|
258
259
|
**Result 2: thresholds do not transfer between models.** The safe threshold ranges from 0.89 for MiniLM to 0.98 for Arctic-embed. Every model tested had negative separation, meaning the mean duplicate score sat below the worst hard negative. Bigger and slower did not fix it.
|
|
259
260
|
|
|
260
|
-
| Model | Safe threshold | Recall there | Embed ms |
|
|
261
|
+
| Model | Safe threshold | Recall there | Embed ms, short probe |
|
|
261
262
|
| --- | ---: | ---: | ---: |
|
|
262
263
|
| `all-MiniLM-L6-v2` | **0.89** | **35.0%** | 5.7 |
|
|
263
264
|
| `gte-base` | 0.96 | 26.0% | 20.6 |
|
|
@@ -266,7 +267,7 @@ This is where a caching project usually hand-waves. The full write-up is in [doc
|
|
|
266
267
|
| `snowflake-arctic-embed-s` | 0.98 | 11.4% | 3.2 |
|
|
267
268
|
| `bge-small-en-v1.5` | 0.96 | 8.1% | 3.8 |
|
|
268
269
|
|
|
269
|
-
MiniLM gives four times the safe recall of bge-small
|
|
270
|
+
MiniLM gives four times the safe recall of bge-small, for a slightly larger download of 86 MB against 63 MB, so it is the default. The embed column is a ranking from short synthetic strings, not what a real question costs. The load test measures that directly. The whole table ships in code as `CALIBRATED_THRESHOLDS`, and the proxy warns at startup if you configure a model it has never measured.
|
|
270
271
|
|
|
271
272
|
**Result 3, a negative one: a lexical guard does not rescue it.** The obvious fix is to require matched prompts to share content words. Measured, hard negatives have *higher* token overlap (0.42 mean) than genuine paraphrases share vocabulary, because they differ by exactly one decisive word. The guard rejects good matches and keeps dangerous ones. It was measured and dropped rather than shipped.
|
|
272
273
|
|
|
@@ -366,7 +367,7 @@ One environment variable per host. Everything except Bedrock speaks the OpenAI p
|
|
|
366
367
|
| Host | `CACHELLM_OPENAI_BASE_URL` | Example model |
|
|
367
368
|
| --- | --- | --- |
|
|
368
369
|
| OpenAI | `https://api.openai.com/v1` | `gpt-4o-mini` |
|
|
369
|
-
| Groq | `https://api.groq.com/openai/v1` | `
|
|
370
|
+
| Groq | `https://api.groq.com/openai/v1` | `openai/gpt-oss-20b` |
|
|
370
371
|
| Google Gemini | `https://generativelanguage.googleapis.com/v1beta/openai/` | `gemini-2.5-flash` |
|
|
371
372
|
| Anthropic | `https://api.anthropic.com/v1` | `claude-haiku-4-5` |
|
|
372
373
|
| OpenRouter | `https://openrouter.ai/api/v1` | `anthropic/claude-3.5-sonnet` |
|
|
@@ -380,6 +381,10 @@ CACHELLM_OPENAI_API_KEY=gsk_your_key \
|
|
|
380
381
|
cachellm serve
|
|
381
382
|
```
|
|
382
383
|
|
|
384
|
+
If the key is already in its usual variable, such as `GROQ_API_KEY` or `GEMINI_API_KEY`, you can skip both lines: the proxy finds it at startup and says which host it picked.
|
|
385
|
+
|
|
386
|
+
**Checked live, not just in tests.** On 2026-09-11 the proxy was run against real **Google Gemini** (`gemini-2.5-flash`, `gemini-3.5-flash`) and **Groq** (`openai/gpt-oss-20b`, `openai/gpt-oss-120b`), driven by the official OpenAI SDK, alongside the AWS Bedrock benchmark. Each run covers a miss, an exact hit, a reworded hit, a similar question that must not hit, a stream and its replay, a hot-temperature bypass and an unknown model's error, and checks that the API key never reaches the log. Results are in [docs/evaluation.md](docs/evaluation.md#live-provider-checks), and `bench/live_check.py` reruns them with your own key. The other hosts share the same adapter and its tests, but have not yet been run against the real service.
|
|
387
|
+
|
|
383
388
|
**AWS Bedrock** is the one exception, because it does not speak the OpenAI protocol. It needs the `aws` extra, and then uses your existing AWS credentials with no vendor API key at all:
|
|
384
389
|
|
|
385
390
|
```bash
|
|
@@ -401,7 +406,7 @@ Three rules, in order. You never configure a model list.
|
|
|
401
406
|
2. **A recognisable vendor convention.** Bedrock ids are always `vendor.model`, so `amazon.nova-lite-v1:0` and `us.anthropic.claude-3-haiku-20240307-v1:0` are identified with no prefix. OpenAI's own families (`gpt-`, `o1`, `o3`, `text-embedding-`) are identified the same way, whatever the default is.
|
|
402
407
|
3. **Otherwise the configured default**, which is the OpenAI-compatible adapter. Names like `llama3.2`, `mixtral-8x7b-32768` and `qwen2.5-coder:7b` are served by Groq, Ollama, Together and OpenRouter alike, so the endpoint you configured is the only sensible answer.
|
|
403
408
|
|
|
404
|
-
Model ids that legitimately contain a slash, which OpenRouter and Together both use, are forwarded whole. Only this proxy's own prefixes are stripped. Note that `anthropic.claude-…` with a dot is a Bedrock id while `anthropic/claude-…` with a slash is an OpenRouter id, and the router tells them apart.
|
|
409
|
+
Model ids that legitimately contain a slash, which OpenRouter and Together both use, are forwarded whole. Only this proxy's own prefixes are stripped, and `openai/` only when the upstream is OpenAI itself: Groq, OpenRouter and Together name OpenAI's models `openai/gpt-oss-120b`, so to them the prefix is part of the id. Note that `anthropic.claude-…` with a dot is a Bedrock id while `anthropic/claude-…` with a slash is an OpenRouter id, and the router tells them apart.
|
|
405
410
|
|
|
406
411
|
Ask it directly if you are unsure:
|
|
407
412
|
|
|
@@ -410,6 +415,39 @@ curl -s localhost:8080/admin/route/meta-llama/Llama-3.3-70B-Instruct-Turbo
|
|
|
410
415
|
curl -s localhost:8080/admin/providers
|
|
411
416
|
```
|
|
412
417
|
|
|
418
|
+
### Where the cache lives
|
|
419
|
+
|
|
420
|
+
One setting, `CACHELLM_BACKEND`, with three values. Most people never touch it.
|
|
421
|
+
|
|
422
|
+
| Value | What happens | Choose it when |
|
|
423
|
+
| --- | --- | --- |
|
|
424
|
+
| `auto`, the default | Redis if the extra is installed and a usable server answers, the in-process cache otherwise | You have no opinion |
|
|
425
|
+
| `memory` | A numpy matrix inside the proxy. Nothing to install | One process: a laptop, a side project, a single server |
|
|
426
|
+
| `redis` | One Redis server shared by every copy of the proxy | Several processes must share one cache |
|
|
427
|
+
|
|
428
|
+
Memory is not the slow option. Below roughly 100,000 entries it is the faster one: scanning 20,000 cached prompts takes 0.85 ms, while a Redis round trip alone costs 2 to 3 ms. Redis earns its place when several processes need one shared cache. Run four copies of the proxy on memory and you have four separate caches, each a quarter as warm.
|
|
429
|
+
|
|
430
|
+
Memory can survive a restart too. Give it a file, and it saves there on a clean shutdown and loads it back on start:
|
|
431
|
+
|
|
432
|
+
```bash
|
|
433
|
+
CACHELLM_MEMORY_SNAPSHOT_PATH=$HOME/.cachellm/cache.npz cachellm serve
|
|
434
|
+
```
|
|
435
|
+
|
|
436
|
+
**Setting up Redis.** Install the extra, then run a Redis 8 server:
|
|
437
|
+
|
|
438
|
+
```bash
|
|
439
|
+
pip install "cachellm-proxy[redis]"
|
|
440
|
+
brew install redis && brew services start redis # macOS
|
|
441
|
+
docker run -d -p 6379:6379 redis:8-alpine # anywhere with Docker
|
|
442
|
+
```
|
|
443
|
+
|
|
444
|
+
The server has two requirements, and both are checked at startup:
|
|
445
|
+
|
|
446
|
+
- **It needs the search module**, which is what stores and searches vectors. Redis 8 from Homebrew or the official Docker image includes it. Many Linux distribution packages ship an older Redis without it, so prefer the Docker image there. A hosted Redis works if its plan includes search.
|
|
447
|
+
- **It has to be database 0**, because Redis search cannot index any other. Keep several apps apart with `CACHELLM_INDEX_NAME` instead.
|
|
448
|
+
|
|
449
|
+
If either is missing, `auto` uses memory and says exactly why, both in a startup warning and at the top of `cachellm stats`. With `CACHELLM_BACKEND=redis` the proxy never switches storage behind your back. It keeps answering without a cache and reports itself degraded until Redis is fixed.
|
|
450
|
+
|
|
413
451
|
## API reference
|
|
414
452
|
|
|
415
453
|
### Chat completions
|
|
@@ -447,6 +485,7 @@ Clients can steer per request with `X-Cache-Control`:
|
|
|
447
485
|
| `GET /admin/providers` | Every host, its base URL, its pip extra, example model ids |
|
|
448
486
|
| `GET /admin/route/{model}` | Where one model name would go, and why |
|
|
449
487
|
| `POST /admin/invalidate` | Drop by `namespace`, by `model`, or `all` |
|
|
488
|
+
| `GET /admin/requests` | Recent request log: what the cache did with each one |
|
|
450
489
|
| `GET /admin/near-misses` | Recent lookups that landed just below threshold |
|
|
451
490
|
| `GET /admin/near-miss-histogram` | Cumulative view of what a lower threshold would buy |
|
|
452
491
|
| `GET /admin/entries` | Inspect what is stored |
|
|
@@ -505,7 +544,7 @@ Every setting is an environment variable prefixed `CACHELLM_`, or a line in `.en
|
|
|
505
544
|
| Variable | Default | Notes |
|
|
506
545
|
| --- | --- | --- |
|
|
507
546
|
| `CACHELLM_BACKEND` | `auto` | `memory` needs nothing, `redis` shares one cache across workers, `auto` uses Redis when reachable and memory when not |
|
|
508
|
-
| `CACHELLM_REDIS_URL` | `redis://localhost:6379/0` | Must be database 0.
|
|
547
|
+
| `CACHELLM_REDIS_URL` | `redis://localhost:6379/0` | Must be database 0, and the server needs the search module. See [Where the cache lives](#where-the-cache-lives) |
|
|
509
548
|
| `CACHELLM_MEMORY_MAX_ENTRIES` | `50000` | Cap for the in-memory store. 50k of 384-dim vectors is about 73 MB |
|
|
510
549
|
| `CACHELLM_MEMORY_SNAPSHOT_PATH` | unset | Persist the in-memory cache to this file so a restart does not start cold |
|
|
511
550
|
| `CACHELLM_API_KEYS` | empty | Comma-separated client keys. Empty disables auth, which is local development only |
|
|
@@ -516,7 +555,7 @@ Every setting is an environment variable prefixed `CACHELLM_`, or a line in `.en
|
|
|
516
555
|
| `CACHELLM_TTL_VOLATILE` | `900` | Fifteen minutes for anything about now |
|
|
517
556
|
| `CACHELLM_MAX_CACHEABLE_TEMPERATURE` | `0.3` | Above this, nothing is cached |
|
|
518
557
|
| `CACHELLM_PII_GUARD` | `true` | Refuse to store prompts that look personal |
|
|
519
|
-
| `CACHELLM_DEFAULT_PROVIDER` | `
|
|
558
|
+
| `CACHELLM_DEFAULT_PROVIDER` | `openai` | `openai` covers every OpenAI-compatible host. Also `bedrock` or `fake`. Left unset, the proxy picks from the keys it finds |
|
|
520
559
|
| `CACHELLM_LOG_PROMPTS` | `false` | Prompt text stays out of logs unless you opt in |
|
|
521
560
|
|
|
522
561
|
## Deployment and infrastructure
|
|
@@ -585,7 +624,7 @@ cachellm/
|
|
|
585
624
|
## Testing
|
|
586
625
|
|
|
587
626
|
```bash
|
|
588
|
-
make test #
|
|
627
|
+
make test # 327 tests
|
|
589
628
|
make test-cov # with coverage
|
|
590
629
|
make lint # ruff and mypy
|
|
591
630
|
```
|
|
@@ -3,13 +3,13 @@
|
|
|
3
3
|
[](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml)
|
|
4
4
|
[](https://www.python.org/)
|
|
5
5
|
[](LICENSE)
|
|
6
|
-
[](tests/)
|
|
7
7
|
[](https://pypi.org/project/cachellm-proxy/)
|
|
8
8
|
[](docs/evaluation.md)
|
|
9
9
|
|
|
10
10
|
**A drop-in semantic cache for OpenAI-compatible LLM APIs. Change one base URL, and questions your model has already answered come back in milliseconds instead of seconds.**
|
|
11
11
|
|
|
12
|
-
On a 2,000-request replay against **AWS Bedrock** it served **77% of traffic from cache** with **zero false positives** on genuinely new questions, cutting spend by **78%** and p95 latency from 1,023 ms to **5.7 ms
|
|
12
|
+
On a 2,000-request replay against **AWS Bedrock** it served **77% of traffic from cache** with **zero false positives** on genuinely new questions, cutting spend by **78%** and p95 latency from 1,023 ms to **5.7 ms** on a cache hit, or 9.7 ms when the question was reworded.
|
|
13
13
|
|
|
14
14
|
[Quick start](#quick-start) · [How it works](#how-it-works) · [Evaluation](docs/evaluation.md) · [API reference](#api-reference) · [Deployment](#deployment-and-infrastructure)
|
|
15
15
|
|
|
@@ -47,7 +47,8 @@ Everything else stays the same: same request shape, same response shape, same er
|
|
|
47
47
|
| False positives on genuinely new questions | **0 of 368** |
|
|
48
48
|
| Reworded repeats served from cache | 95.2% |
|
|
49
49
|
| Exact repeats served from cache | 99.3% |
|
|
50
|
-
| p95 latency, cache hit | **5.7 ms** |
|
|
50
|
+
| p95 latency, any cache hit | **5.7 ms** |
|
|
51
|
+
| p95 latency, reworded hit | 9.7 ms |
|
|
51
52
|
| p95 latency, cache miss | 1,022.8 ms |
|
|
52
53
|
| p95 speedup | **178.8x** |
|
|
53
54
|
| Cost reduction | **78.0%** |
|
|
@@ -207,7 +208,7 @@ This is where a caching project usually hand-waves. The full write-up is in [doc
|
|
|
207
208
|
|
|
208
209
|
**Result 2: thresholds do not transfer between models.** The safe threshold ranges from 0.89 for MiniLM to 0.98 for Arctic-embed. Every model tested had negative separation, meaning the mean duplicate score sat below the worst hard negative. Bigger and slower did not fix it.
|
|
209
210
|
|
|
210
|
-
| Model | Safe threshold | Recall there | Embed ms |
|
|
211
|
+
| Model | Safe threshold | Recall there | Embed ms, short probe |
|
|
211
212
|
| --- | ---: | ---: | ---: |
|
|
212
213
|
| `all-MiniLM-L6-v2` | **0.89** | **35.0%** | 5.7 |
|
|
213
214
|
| `gte-base` | 0.96 | 26.0% | 20.6 |
|
|
@@ -216,7 +217,7 @@ This is where a caching project usually hand-waves. The full write-up is in [doc
|
|
|
216
217
|
| `snowflake-arctic-embed-s` | 0.98 | 11.4% | 3.2 |
|
|
217
218
|
| `bge-small-en-v1.5` | 0.96 | 8.1% | 3.8 |
|
|
218
219
|
|
|
219
|
-
MiniLM gives four times the safe recall of bge-small
|
|
220
|
+
MiniLM gives four times the safe recall of bge-small, for a slightly larger download of 86 MB against 63 MB, so it is the default. The embed column is a ranking from short synthetic strings, not what a real question costs. The load test measures that directly. The whole table ships in code as `CALIBRATED_THRESHOLDS`, and the proxy warns at startup if you configure a model it has never measured.
|
|
220
221
|
|
|
221
222
|
**Result 3, a negative one: a lexical guard does not rescue it.** The obvious fix is to require matched prompts to share content words. Measured, hard negatives have *higher* token overlap (0.42 mean) than genuine paraphrases share vocabulary, because they differ by exactly one decisive word. The guard rejects good matches and keeps dangerous ones. It was measured and dropped rather than shipped.
|
|
222
223
|
|
|
@@ -316,7 +317,7 @@ One environment variable per host. Everything except Bedrock speaks the OpenAI p
|
|
|
316
317
|
| Host | `CACHELLM_OPENAI_BASE_URL` | Example model |
|
|
317
318
|
| --- | --- | --- |
|
|
318
319
|
| OpenAI | `https://api.openai.com/v1` | `gpt-4o-mini` |
|
|
319
|
-
| Groq | `https://api.groq.com/openai/v1` | `
|
|
320
|
+
| Groq | `https://api.groq.com/openai/v1` | `openai/gpt-oss-20b` |
|
|
320
321
|
| Google Gemini | `https://generativelanguage.googleapis.com/v1beta/openai/` | `gemini-2.5-flash` |
|
|
321
322
|
| Anthropic | `https://api.anthropic.com/v1` | `claude-haiku-4-5` |
|
|
322
323
|
| OpenRouter | `https://openrouter.ai/api/v1` | `anthropic/claude-3.5-sonnet` |
|
|
@@ -330,6 +331,10 @@ CACHELLM_OPENAI_API_KEY=gsk_your_key \
|
|
|
330
331
|
cachellm serve
|
|
331
332
|
```
|
|
332
333
|
|
|
334
|
+
If the key is already in its usual variable, such as `GROQ_API_KEY` or `GEMINI_API_KEY`, you can skip both lines: the proxy finds it at startup and says which host it picked.
|
|
335
|
+
|
|
336
|
+
**Checked live, not just in tests.** On 2026-09-11 the proxy was run against real **Google Gemini** (`gemini-2.5-flash`, `gemini-3.5-flash`) and **Groq** (`openai/gpt-oss-20b`, `openai/gpt-oss-120b`), driven by the official OpenAI SDK, alongside the AWS Bedrock benchmark. Each run covers a miss, an exact hit, a reworded hit, a similar question that must not hit, a stream and its replay, a hot-temperature bypass and an unknown model's error, and checks that the API key never reaches the log. Results are in [docs/evaluation.md](docs/evaluation.md#live-provider-checks), and `bench/live_check.py` reruns them with your own key. The other hosts share the same adapter and its tests, but have not yet been run against the real service.
|
|
337
|
+
|
|
333
338
|
**AWS Bedrock** is the one exception, because it does not speak the OpenAI protocol. It needs the `aws` extra, and then uses your existing AWS credentials with no vendor API key at all:
|
|
334
339
|
|
|
335
340
|
```bash
|
|
@@ -351,7 +356,7 @@ Three rules, in order. You never configure a model list.
|
|
|
351
356
|
2. **A recognisable vendor convention.** Bedrock ids are always `vendor.model`, so `amazon.nova-lite-v1:0` and `us.anthropic.claude-3-haiku-20240307-v1:0` are identified with no prefix. OpenAI's own families (`gpt-`, `o1`, `o3`, `text-embedding-`) are identified the same way, whatever the default is.
|
|
352
357
|
3. **Otherwise the configured default**, which is the OpenAI-compatible adapter. Names like `llama3.2`, `mixtral-8x7b-32768` and `qwen2.5-coder:7b` are served by Groq, Ollama, Together and OpenRouter alike, so the endpoint you configured is the only sensible answer.
|
|
353
358
|
|
|
354
|
-
Model ids that legitimately contain a slash, which OpenRouter and Together both use, are forwarded whole. Only this proxy's own prefixes are stripped. Note that `anthropic.claude-…` with a dot is a Bedrock id while `anthropic/claude-…` with a slash is an OpenRouter id, and the router tells them apart.
|
|
359
|
+
Model ids that legitimately contain a slash, which OpenRouter and Together both use, are forwarded whole. Only this proxy's own prefixes are stripped, and `openai/` only when the upstream is OpenAI itself: Groq, OpenRouter and Together name OpenAI's models `openai/gpt-oss-120b`, so to them the prefix is part of the id. Note that `anthropic.claude-…` with a dot is a Bedrock id while `anthropic/claude-…` with a slash is an OpenRouter id, and the router tells them apart.
|
|
355
360
|
|
|
356
361
|
Ask it directly if you are unsure:
|
|
357
362
|
|
|
@@ -360,6 +365,39 @@ curl -s localhost:8080/admin/route/meta-llama/Llama-3.3-70B-Instruct-Turbo
|
|
|
360
365
|
curl -s localhost:8080/admin/providers
|
|
361
366
|
```
|
|
362
367
|
|
|
368
|
+
### Where the cache lives
|
|
369
|
+
|
|
370
|
+
One setting, `CACHELLM_BACKEND`, with three values. Most people never touch it.
|
|
371
|
+
|
|
372
|
+
| Value | What happens | Choose it when |
|
|
373
|
+
| --- | --- | --- |
|
|
374
|
+
| `auto`, the default | Redis if the extra is installed and a usable server answers, the in-process cache otherwise | You have no opinion |
|
|
375
|
+
| `memory` | A numpy matrix inside the proxy. Nothing to install | One process: a laptop, a side project, a single server |
|
|
376
|
+
| `redis` | One Redis server shared by every copy of the proxy | Several processes must share one cache |
|
|
377
|
+
|
|
378
|
+
Memory is not the slow option. Below roughly 100,000 entries it is the faster one: scanning 20,000 cached prompts takes 0.85 ms, while a Redis round trip alone costs 2 to 3 ms. Redis earns its place when several processes need one shared cache. Run four copies of the proxy on memory and you have four separate caches, each a quarter as warm.
|
|
379
|
+
|
|
380
|
+
Memory can survive a restart too. Give it a file, and it saves there on a clean shutdown and loads it back on start:
|
|
381
|
+
|
|
382
|
+
```bash
|
|
383
|
+
CACHELLM_MEMORY_SNAPSHOT_PATH=$HOME/.cachellm/cache.npz cachellm serve
|
|
384
|
+
```
|
|
385
|
+
|
|
386
|
+
**Setting up Redis.** Install the extra, then run a Redis 8 server:
|
|
387
|
+
|
|
388
|
+
```bash
|
|
389
|
+
pip install "cachellm-proxy[redis]"
|
|
390
|
+
brew install redis && brew services start redis # macOS
|
|
391
|
+
docker run -d -p 6379:6379 redis:8-alpine # anywhere with Docker
|
|
392
|
+
```
|
|
393
|
+
|
|
394
|
+
The server has two requirements, and both are checked at startup:
|
|
395
|
+
|
|
396
|
+
- **It needs the search module**, which is what stores and searches vectors. Redis 8 from Homebrew or the official Docker image includes it. Many Linux distribution packages ship an older Redis without it, so prefer the Docker image there. A hosted Redis works if its plan includes search.
|
|
397
|
+
- **It has to be database 0**, because Redis search cannot index any other. Keep several apps apart with `CACHELLM_INDEX_NAME` instead.
|
|
398
|
+
|
|
399
|
+
If either is missing, `auto` uses memory and says exactly why, both in a startup warning and at the top of `cachellm stats`. With `CACHELLM_BACKEND=redis` the proxy never switches storage behind your back. It keeps answering without a cache and reports itself degraded until Redis is fixed.
|
|
400
|
+
|
|
363
401
|
## API reference
|
|
364
402
|
|
|
365
403
|
### Chat completions
|
|
@@ -397,6 +435,7 @@ Clients can steer per request with `X-Cache-Control`:
|
|
|
397
435
|
| `GET /admin/providers` | Every host, its base URL, its pip extra, example model ids |
|
|
398
436
|
| `GET /admin/route/{model}` | Where one model name would go, and why |
|
|
399
437
|
| `POST /admin/invalidate` | Drop by `namespace`, by `model`, or `all` |
|
|
438
|
+
| `GET /admin/requests` | Recent request log: what the cache did with each one |
|
|
400
439
|
| `GET /admin/near-misses` | Recent lookups that landed just below threshold |
|
|
401
440
|
| `GET /admin/near-miss-histogram` | Cumulative view of what a lower threshold would buy |
|
|
402
441
|
| `GET /admin/entries` | Inspect what is stored |
|
|
@@ -455,7 +494,7 @@ Every setting is an environment variable prefixed `CACHELLM_`, or a line in `.en
|
|
|
455
494
|
| Variable | Default | Notes |
|
|
456
495
|
| --- | --- | --- |
|
|
457
496
|
| `CACHELLM_BACKEND` | `auto` | `memory` needs nothing, `redis` shares one cache across workers, `auto` uses Redis when reachable and memory when not |
|
|
458
|
-
| `CACHELLM_REDIS_URL` | `redis://localhost:6379/0` | Must be database 0.
|
|
497
|
+
| `CACHELLM_REDIS_URL` | `redis://localhost:6379/0` | Must be database 0, and the server needs the search module. See [Where the cache lives](#where-the-cache-lives) |
|
|
459
498
|
| `CACHELLM_MEMORY_MAX_ENTRIES` | `50000` | Cap for the in-memory store. 50k of 384-dim vectors is about 73 MB |
|
|
460
499
|
| `CACHELLM_MEMORY_SNAPSHOT_PATH` | unset | Persist the in-memory cache to this file so a restart does not start cold |
|
|
461
500
|
| `CACHELLM_API_KEYS` | empty | Comma-separated client keys. Empty disables auth, which is local development only |
|
|
@@ -466,7 +505,7 @@ Every setting is an environment variable prefixed `CACHELLM_`, or a line in `.en
|
|
|
466
505
|
| `CACHELLM_TTL_VOLATILE` | `900` | Fifteen minutes for anything about now |
|
|
467
506
|
| `CACHELLM_MAX_CACHEABLE_TEMPERATURE` | `0.3` | Above this, nothing is cached |
|
|
468
507
|
| `CACHELLM_PII_GUARD` | `true` | Refuse to store prompts that look personal |
|
|
469
|
-
| `CACHELLM_DEFAULT_PROVIDER` | `
|
|
508
|
+
| `CACHELLM_DEFAULT_PROVIDER` | `openai` | `openai` covers every OpenAI-compatible host. Also `bedrock` or `fake`. Left unset, the proxy picks from the keys it finds |
|
|
470
509
|
| `CACHELLM_LOG_PROMPTS` | `false` | Prompt text stays out of logs unless you opt in |
|
|
471
510
|
|
|
472
511
|
## Deployment and infrastructure
|
|
@@ -535,7 +574,7 @@ cachellm/
|
|
|
535
574
|
## Testing
|
|
536
575
|
|
|
537
576
|
```bash
|
|
538
|
-
make test #
|
|
577
|
+
make test # 327 tests
|
|
539
578
|
make test-cov # with coverage
|
|
540
579
|
make lint # ruff and mypy
|
|
541
580
|
```
|
|
@@ -16,8 +16,10 @@ cache that takes an application down when it breaks is worse than no cache.
|
|
|
16
16
|
|
|
17
17
|
from __future__ import annotations
|
|
18
18
|
|
|
19
|
+
import contextlib
|
|
19
20
|
from dataclasses import dataclass, field
|
|
20
21
|
from typing import TYPE_CHECKING, Any
|
|
22
|
+
from urllib.parse import urlparse
|
|
21
23
|
|
|
22
24
|
import structlog
|
|
23
25
|
from fastapi import Request
|
|
@@ -46,6 +48,44 @@ REDIS_MISSING = (
|
|
|
46
48
|
)
|
|
47
49
|
|
|
48
50
|
|
|
51
|
+
class RedisUnusableError(RuntimeError):
|
|
52
|
+
"""Redis answered, but it cannot hold this cache.
|
|
53
|
+
|
|
54
|
+
Different in kind from "no Redis here". Nobody runs a Redis server by
|
|
55
|
+
accident, so when one answers and still cannot be used, the operator almost
|
|
56
|
+
certainly meant to use it and deserves to hear why it was passed over.
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _where(url: str) -> str:
|
|
61
|
+
"""Host and port only. The URL may carry a password, and this gets logged."""
|
|
62
|
+
parsed = urlparse(url)
|
|
63
|
+
if parsed.hostname:
|
|
64
|
+
return f"{parsed.hostname}:{parsed.port or 6379}"
|
|
65
|
+
return "the configured address"
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
async def _require_search(redis: Any, settings: Settings) -> None:
|
|
69
|
+
"""Fail with a plain explanation before RedisVL fails with a cryptic one."""
|
|
70
|
+
where = _where(settings.redis_url)
|
|
71
|
+
db = int(redis.connection_pool.connection_kwargs.get("db", 0) or 0)
|
|
72
|
+
if db != 0:
|
|
73
|
+
raise RedisUnusableError(
|
|
74
|
+
f"Redis at {where} answered, but CACHELLM_REDIS_URL selects database {db}. "
|
|
75
|
+
"Redis search can only index database 0, so end the URL with /0."
|
|
76
|
+
)
|
|
77
|
+
try:
|
|
78
|
+
await redis.execute_command("FT._LIST")
|
|
79
|
+
except Exception as exc:
|
|
80
|
+
if "unknown command" not in str(exc).lower():
|
|
81
|
+
raise
|
|
82
|
+
raise RedisUnusableError(
|
|
83
|
+
f"Redis at {where} answered, but it has no search module, so it cannot "
|
|
84
|
+
"store vectors. Redis 8 from Homebrew or the official Docker image includes "
|
|
85
|
+
"it. Many Linux distribution packages do not."
|
|
86
|
+
) from exc
|
|
87
|
+
|
|
88
|
+
|
|
49
89
|
@dataclass
|
|
50
90
|
class AppState:
|
|
51
91
|
settings: Settings
|
|
@@ -62,6 +102,8 @@ class AppState:
|
|
|
62
102
|
degraded_reason: str = ""
|
|
63
103
|
#: Which backend actually started: "redis" or "memory".
|
|
64
104
|
backend: str = "none"
|
|
105
|
+
#: Why a reachable Redis was passed over, shown by `cachellm stats`.
|
|
106
|
+
backend_note: str = ""
|
|
65
107
|
|
|
66
108
|
@property
|
|
67
109
|
def caching_on(self) -> bool:
|
|
@@ -96,8 +138,12 @@ async def build_state(
|
|
|
96
138
|
except Exception as exc:
|
|
97
139
|
redis_error = f"{type(exc).__name__}: {exc}"
|
|
98
140
|
if settings.backend == "redis":
|
|
99
|
-
|
|
100
|
-
|
|
141
|
+
unusable = isinstance(exc, RedisUnusableError)
|
|
142
|
+
state.degraded_reason = str(exc) if unusable else redis_error
|
|
143
|
+
log.error("redis_unavailable_failing_open", error=state.degraded_reason)
|
|
144
|
+
elif isinstance(exc, RedisUnusableError):
|
|
145
|
+
state.backend_note = f"{exc} Using the in-process cache instead."
|
|
146
|
+
log.warning("redis_unusable_using_memory", reason=str(exc))
|
|
101
147
|
else:
|
|
102
148
|
log.info(
|
|
103
149
|
"redis_unavailable_using_memory",
|
|
@@ -135,9 +181,15 @@ async def _attach_redis(state: AppState, settings: Settings) -> None:
|
|
|
135
181
|
raise ImportError(REDIS_MISSING) from exc
|
|
136
182
|
|
|
137
183
|
redis = build_redis(settings)
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
184
|
+
try:
|
|
185
|
+
await redis.ping()
|
|
186
|
+
await _require_search(redis, settings)
|
|
187
|
+
vectors = VectorStore(redis, settings)
|
|
188
|
+
await vectors.connect()
|
|
189
|
+
except Exception:
|
|
190
|
+
with contextlib.suppress(Exception):
|
|
191
|
+
await redis.aclose()
|
|
192
|
+
raise
|
|
141
193
|
state.redis = redis
|
|
142
194
|
_wire(
|
|
143
195
|
state, settings, vectors, ExactStore(redis, settings), Analytics(redis, settings), "redis"
|
|
@@ -76,6 +76,8 @@ async def stats(request: Request) -> dict[str, Any]:
|
|
|
76
76
|
data = await state.cache.stats()
|
|
77
77
|
data["cache_available"] = True
|
|
78
78
|
data["backend"] = state.backend
|
|
79
|
+
if state.backend_note:
|
|
80
|
+
data["backend_note"] = state.backend_note
|
|
79
81
|
data["caching_enabled"] = state.settings.enabled
|
|
80
82
|
data["shadow_mode"] = state.settings.shadow_mode
|
|
81
83
|
data["in_flight"] = state.singleflight.in_flight
|
|
@@ -212,6 +214,21 @@ async def invalidate(request: Request, body: InvalidateRequest) -> dict[str, Any
|
|
|
212
214
|
}
|
|
213
215
|
|
|
214
216
|
|
|
217
|
+
@router.get("/requests")
|
|
218
|
+
async def requests_log(request: Request, limit: int = Query(50, ge=1, le=500)) -> dict[str, Any]:
|
|
219
|
+
"""The recent request log: what the cache did with each one.
|
|
220
|
+
|
|
221
|
+
This is what `cachellm stats` reads. It has to come over HTTP because with
|
|
222
|
+
the in-memory backend the cache lives inside the serving process, and a CLI
|
|
223
|
+
building its own state would report on an empty cache of its own.
|
|
224
|
+
"""
|
|
225
|
+
state = _require_cache(request)
|
|
226
|
+
if state.settings.require_auth_for_admin:
|
|
227
|
+
verify(request, state.settings.client_keys)
|
|
228
|
+
rows = await state.analytics.recent_requests(limit=limit)
|
|
229
|
+
return {"count": len(rows), "requests": rows}
|
|
230
|
+
|
|
231
|
+
|
|
215
232
|
@router.get("/near-misses")
|
|
216
233
|
async def near_misses(request: Request, limit: int = Query(50, ge=1, le=500)) -> dict[str, Any]:
|
|
217
234
|
state = _require_cache(request)
|
|
@@ -32,7 +32,7 @@ from cachellm.models import (
|
|
|
32
32
|
)
|
|
33
33
|
from cachellm.observability import span
|
|
34
34
|
from cachellm.pricing import estimate_cost
|
|
35
|
-
from cachellm.providers.base import Provider
|
|
35
|
+
from cachellm.providers.base import Provider, StreamEvent
|
|
36
36
|
|
|
37
37
|
log = structlog.get_logger(__name__)
|
|
38
38
|
router = APIRouter()
|
|
@@ -240,6 +240,12 @@ async def chat_completions(request: Request, body: ChatCompletionRequest) -> Any
|
|
|
240
240
|
return await _proxy_once(state, body, lookup, provider, provider_name, started)
|
|
241
241
|
|
|
242
242
|
|
|
243
|
+
async def _count_provider_error(state: AppState, provider_name: str) -> None:
|
|
244
|
+
state.metrics.provider_errors.labels(provider=provider_name).inc()
|
|
245
|
+
if state.analytics is not None:
|
|
246
|
+
await state.analytics.incr("provider_errors")
|
|
247
|
+
|
|
248
|
+
|
|
243
249
|
async def _proxy_once(
|
|
244
250
|
state: AppState,
|
|
245
251
|
body: ChatCompletionRequest,
|
|
@@ -264,9 +270,7 @@ async def _proxy_once(
|
|
|
264
270
|
else:
|
|
265
271
|
result, coalesced = await call(), False
|
|
266
272
|
except UpstreamError:
|
|
267
|
-
state
|
|
268
|
-
if state.analytics is not None:
|
|
269
|
-
await state.analytics.incr("provider_errors")
|
|
273
|
+
await _count_provider_error(state, provider_name)
|
|
270
274
|
raise
|
|
271
275
|
|
|
272
276
|
if coalesced:
|
|
@@ -361,6 +365,25 @@ async def _proxy_stream(
|
|
|
361
365
|
"""
|
|
362
366
|
stream_id = sse.new_stream_id()
|
|
363
367
|
created = int(time.time())
|
|
368
|
+
upstream = aiter(provider.stream(body))
|
|
369
|
+
|
|
370
|
+
# Hold the response until the upstream sends its first event. A wrong model
|
|
371
|
+
# name or a bad key fails right here, so the caller gets the upstream's real
|
|
372
|
+
# status code. Answering first used to turn every such failure into a 200
|
|
373
|
+
# stream with an empty answer and nothing to say why.
|
|
374
|
+
try:
|
|
375
|
+
first: StreamEvent | None = await anext(upstream)
|
|
376
|
+
except StopAsyncIteration:
|
|
377
|
+
first = None
|
|
378
|
+
except UpstreamError:
|
|
379
|
+
await _count_provider_error(state, provider_name)
|
|
380
|
+
raise
|
|
381
|
+
|
|
382
|
+
async def events() -> AsyncIterator[StreamEvent]:
|
|
383
|
+
if first is not None:
|
|
384
|
+
yield first
|
|
385
|
+
async for event in upstream:
|
|
386
|
+
yield event
|
|
364
387
|
|
|
365
388
|
async def generate() -> AsyncIterator[str]:
|
|
366
389
|
buffer: list[str] = []
|
|
@@ -369,7 +392,7 @@ async def _proxy_stream(
|
|
|
369
392
|
completion_tokens = 0
|
|
370
393
|
yield sse.role_chunk(stream_id, body.model, created)
|
|
371
394
|
try:
|
|
372
|
-
async for event in
|
|
395
|
+
async for event in events():
|
|
373
396
|
if event.delta:
|
|
374
397
|
buffer.append(event.delta)
|
|
375
398
|
yield sse.text_chunk(stream_id, body.model, created, event.delta)
|
|
@@ -379,11 +402,9 @@ async def _proxy_stream(
|
|
|
379
402
|
prompt_tokens = event.prompt_tokens or prompt_tokens
|
|
380
403
|
completion_tokens = event.completion_tokens or completion_tokens
|
|
381
404
|
except UpstreamError as exc:
|
|
382
|
-
state
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
log.warning("stream_failed", error=str(exc)[:200])
|
|
386
|
-
yield sse.final_chunk(stream_id, body.model, created, "error", None)
|
|
405
|
+
await _count_provider_error(state, provider_name)
|
|
406
|
+
log.warning("stream_failed", error=exc.message[:200])
|
|
407
|
+
yield sse.error_event(exc.message, exc.err_type, exc.code)
|
|
387
408
|
yield sse.DONE
|
|
388
409
|
return
|
|
389
410
|
|
|
@@ -71,6 +71,17 @@ def final_chunk(
|
|
|
71
71
|
)
|
|
72
72
|
|
|
73
73
|
|
|
74
|
+
def error_event(message: str, err_type: str = "api_error", code: str | None = None) -> str:
|
|
75
|
+
"""A failure after the stream has started, in OpenAI's shape.
|
|
76
|
+
|
|
77
|
+
The status line has already gone out as 200 by then, so this event is the
|
|
78
|
+
only way left to say something broke. The official SDKs raise an error when
|
|
79
|
+
they read it, instead of ending quietly with a half-written answer.
|
|
80
|
+
"""
|
|
81
|
+
payload = {"error": {"message": message, "type": err_type, "param": None, "code": code}}
|
|
82
|
+
return f"data: {json.dumps(payload, separators=(',', ':'))}\n\n"
|
|
83
|
+
|
|
84
|
+
|
|
74
85
|
def split_for_replay(text: str, max_chars: int = 24) -> list[str]:
|
|
75
86
|
"""Chop a cached answer into believable stream chunks.
|
|
76
87
|
|
|
@@ -3,10 +3,10 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import asyncio
|
|
6
|
-
import contextlib
|
|
7
6
|
import json
|
|
7
|
+
import time
|
|
8
8
|
from pathlib import Path
|
|
9
|
-
from typing import Annotated
|
|
9
|
+
from typing import Annotated, Any
|
|
10
10
|
|
|
11
11
|
import typer
|
|
12
12
|
|
|
@@ -57,75 +57,97 @@ def config() -> None:
|
|
|
57
57
|
typer.echo(json.dumps(data, indent=2, default=str))
|
|
58
58
|
|
|
59
59
|
|
|
60
|
+
def _proxy_url(url: str) -> str:
|
|
61
|
+
settings = get_settings()
|
|
62
|
+
return (url or f"http://127.0.0.1:{settings.port}").rstrip("/")
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _fetch(url: str, path: str, params: dict[str, Any] | None = None) -> Any:
|
|
66
|
+
"""Read from a running proxy, or None when nothing is listening.
|
|
67
|
+
|
|
68
|
+
Talking to the server rather than building our own state is not an
|
|
69
|
+
optimisation: with the in-memory backend the cache lives inside the serving
|
|
70
|
+
process, so a CLI that built its own would report on an empty cache of its
|
|
71
|
+
own and always say "no requests yet".
|
|
72
|
+
"""
|
|
73
|
+
import httpx
|
|
74
|
+
|
|
75
|
+
settings = get_settings()
|
|
76
|
+
headers = {}
|
|
77
|
+
if keys := sorted(settings.client_keys):
|
|
78
|
+
headers["Authorization"] = f"Bearer {keys[0]}"
|
|
79
|
+
try:
|
|
80
|
+
response = httpx.get(f"{url}{path}", params=params, headers=headers, timeout=5.0)
|
|
81
|
+
response.raise_for_status()
|
|
82
|
+
return response.json()
|
|
83
|
+
except Exception:
|
|
84
|
+
return None
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _not_running(url: str) -> None:
|
|
88
|
+
typer.secho(f"No proxy answering at {url}.", fg="yellow")
|
|
89
|
+
typer.echo(" Start one with `cachellm serve`, or pass --url if it is elsewhere.")
|
|
90
|
+
typer.echo(
|
|
91
|
+
f" {report.DIM}The cache lives inside the serving process unless you "
|
|
92
|
+
f"use the redis backend, so there is nothing to read without it."
|
|
93
|
+
f"{report.RESET}"
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
|
|
60
97
|
@app.command()
|
|
61
98
|
def stats(
|
|
99
|
+
url: Annotated[str, typer.Option(help="Proxy to read from.")] = "",
|
|
62
100
|
limit: Annotated[int, typer.Option(help="How many recent requests to show.")] = 15,
|
|
63
101
|
json_out: Annotated[bool, typer.Option("--json", help="Raw JSON instead.")] = False,
|
|
64
102
|
) -> None:
|
|
65
103
|
"""Show what the cache has been doing: hit rate, savings, latency, recent requests."""
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
data["cache_available"] = True
|
|
79
|
-
data["caching_enabled"] = settings.enabled
|
|
80
|
-
recent = await state.analytics.recent_requests(limit=limit)
|
|
81
|
-
if json_out:
|
|
82
|
-
typer.echo(json.dumps({"stats": data, "recent": recent}, indent=2))
|
|
83
|
-
return
|
|
84
|
-
typer.echo(report.summary(data, settings.embedding_model))
|
|
85
|
-
typer.echo(report.request_table(recent, limit=limit))
|
|
86
|
-
finally:
|
|
87
|
-
await shutdown_state(state)
|
|
88
|
-
|
|
89
|
-
asyncio.run(run())
|
|
104
|
+
base = _proxy_url(url)
|
|
105
|
+
data = _fetch(base, "/admin/stats")
|
|
106
|
+
if data is None:
|
|
107
|
+
_not_running(base)
|
|
108
|
+
raise typer.Exit(1)
|
|
109
|
+
log = _fetch(base, "/admin/requests", {"limit": limit}) or {}
|
|
110
|
+
recent = log.get("requests", [])
|
|
111
|
+
if json_out:
|
|
112
|
+
typer.echo(json.dumps({"stats": data, "recent": recent}, indent=2))
|
|
113
|
+
return
|
|
114
|
+
typer.echo(report.summary(data, get_settings().embedding_model))
|
|
115
|
+
typer.echo(report.request_table(recent, limit=limit))
|
|
90
116
|
|
|
91
117
|
|
|
92
118
|
@app.command()
|
|
93
119
|
def watch(
|
|
120
|
+
url: Annotated[str, typer.Option(help="Proxy to follow.")] = "",
|
|
94
121
|
interval: Annotated[float, typer.Option(help="Seconds between refreshes.")] = 2.0,
|
|
95
122
|
) -> None:
|
|
96
123
|
"""Follow requests as they happen, like `tail -f` for the cache."""
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
seen.add(marker)
|
|
124
|
+
base = _proxy_url(url)
|
|
125
|
+
if _fetch(base, "/admin/stats") is None:
|
|
126
|
+
_not_running(base)
|
|
127
|
+
raise typer.Exit(1)
|
|
128
|
+
|
|
129
|
+
seen: set[tuple[float, str]] = set()
|
|
130
|
+
first = True
|
|
131
|
+
typer.echo(f"following {base}, ctrl-c to stop\n")
|
|
132
|
+
try:
|
|
133
|
+
while True:
|
|
134
|
+
payload = _fetch(base, "/admin/requests", {"limit": 50}) or {}
|
|
135
|
+
for record in reversed(payload.get("requests", [])):
|
|
136
|
+
marker = (record.get("at", 0.0), record.get("prompt", ""))
|
|
137
|
+
if marker in seen:
|
|
138
|
+
continue
|
|
139
|
+
seen.add(marker)
|
|
140
|
+
if not first:
|
|
115
141
|
typer.echo(report.request_line(record))
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
typer.echo(
|
|
124
|
-
|
|
125
|
-
await shutdown_state(state)
|
|
126
|
-
|
|
127
|
-
with contextlib.suppress(KeyboardInterrupt):
|
|
128
|
-
asyncio.run(run())
|
|
142
|
+
first = False
|
|
143
|
+
if len(seen) > 5_000:
|
|
144
|
+
seen.clear()
|
|
145
|
+
time.sleep(interval)
|
|
146
|
+
except KeyboardInterrupt:
|
|
147
|
+
data = _fetch(base, "/admin/stats")
|
|
148
|
+
if data:
|
|
149
|
+
typer.echo("")
|
|
150
|
+
typer.echo(report.summary(data, get_settings().embedding_model))
|
|
129
151
|
|
|
130
152
|
|
|
131
153
|
@app.command()
|
|
@@ -64,12 +64,24 @@ class FastEmbedEmbedder(Embedder):
|
|
|
64
64
|
return vectors[0]
|
|
65
65
|
|
|
66
66
|
async def embed_batch(self, texts: list[str]) -> list[np.ndarray]:
|
|
67
|
+
"""Embed many texts, running the model once per distinct uncached text.
|
|
68
|
+
|
|
69
|
+
Results are assembled from what was just computed, never read back out
|
|
70
|
+
of the cache. The cache is bounded, so reading back used to fail two
|
|
71
|
+
ways: a batch larger than the cache evicted its own first results, and
|
|
72
|
+
a cache size of 0 made every call fail, which quietly turned caching off.
|
|
73
|
+
"""
|
|
67
74
|
if not texts:
|
|
68
75
|
return []
|
|
69
76
|
model = await self._ensure_model()
|
|
70
|
-
|
|
77
|
+
found: dict[str, np.ndarray] = {}
|
|
78
|
+
for text in texts:
|
|
79
|
+
if (hit := self._cache_get(text)) is not None:
|
|
80
|
+
found[text] = hit
|
|
81
|
+
pending = list(dict.fromkeys(t for t in texts if t not in found))
|
|
71
82
|
if pending:
|
|
72
83
|
raw = await asyncio.to_thread(lambda: list(model.embed(pending)))
|
|
73
84
|
for text, vector in zip(pending, raw, strict=True):
|
|
74
|
-
|
|
75
|
-
|
|
85
|
+
found[text] = self.normalise(np.asarray(vector, dtype=np.float32))
|
|
86
|
+
self._cache_put(text, found[text])
|
|
87
|
+
return [found[t] for t in texts]
|
|
@@ -50,6 +50,18 @@ PRICES: dict[str, ModelPrice] = {
|
|
|
50
50
|
"gpt-4o-mini": ModelPrice(0.15, 0.60),
|
|
51
51
|
"gpt-4o": ModelPrice(2.50, 10.00),
|
|
52
52
|
"gpt-4.1-mini": ModelPrice(0.40, 1.60),
|
|
53
|
+
# --- Google Gemini, paid tier, output includes thinking (checked 2026-09-11) ---
|
|
54
|
+
"gemini-2.5-flash": ModelPrice(0.30, 2.50),
|
|
55
|
+
"gemini-2.5-flash-lite": ModelPrice(0.10, 0.40),
|
|
56
|
+
"gemini-2.5-pro": ModelPrice(1.25, 10.00),
|
|
57
|
+
"gemini-3.5-flash": ModelPrice(1.50, 9.00),
|
|
58
|
+
"gemini-3.5-flash-lite": ModelPrice(0.30, 2.50),
|
|
59
|
+
# --- Groq, keyed without the vendor segment (checked 2026-09-11) ---
|
|
60
|
+
"gpt-oss-20b": ModelPrice(0.075, 0.30),
|
|
61
|
+
"gpt-oss-120b": ModelPrice(0.15, 0.60),
|
|
62
|
+
"gpt-oss-safeguard-20b": ModelPrice(0.075, 0.30),
|
|
63
|
+
"qwen3.6-27b": ModelPrice(0.60, 3.00),
|
|
64
|
+
"qwen3.8-27b": ModelPrice(0.80, 4.00),
|
|
53
65
|
# --- embeddings (input only) ---
|
|
54
66
|
"amazon.titan-embed-text-v2:0": ModelPrice(0.02, 0.0),
|
|
55
67
|
"text-embedding-3-small": ModelPrice(0.02, 0.0),
|
|
@@ -97,10 +109,13 @@ def price_for(model: str) -> ModelPrice:
|
|
|
97
109
|
key = normalise_model_id(model)
|
|
98
110
|
if key in PRICES:
|
|
99
111
|
return PRICES[key]
|
|
100
|
-
# Prefix match so
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
112
|
+
# Prefix match so dated or preview revisions still price sensibly. The
|
|
113
|
+
# longest match wins: `gemini-2.5-flash-lite-preview` is a Flash-Lite, and
|
|
114
|
+
# taking the first match in table order priced it as the dearer Flash.
|
|
115
|
+
stems = {known.split("-2024")[0].split("-v1:")[0]: known for known in PRICES}
|
|
116
|
+
matches = [stem for stem in stems if key.startswith(stem)]
|
|
117
|
+
if matches:
|
|
118
|
+
return PRICES[stems[max(matches, key=len)]]
|
|
104
119
|
return _FALLBACK
|
|
105
120
|
|
|
106
121
|
|
|
@@ -77,12 +77,10 @@ OPENAI_FAMILIES: tuple[str, ...] = (
|
|
|
77
77
|
#: Together both use `vendor/model` ids, and stripping their vendor segment
|
|
78
78
|
#: makes the upstream reject the request as an unknown model.
|
|
79
79
|
#:
|
|
80
|
-
#:
|
|
81
|
-
#:
|
|
82
|
-
#:
|
|
83
|
-
#:
|
|
84
|
-
#: is to point CACHELLM_OPENAI_BASE_URL at Groq and send the bare id if the host
|
|
85
|
-
#: accepts it.
|
|
80
|
+
#: A model id starting with one of these always routes to that adapter. What
|
|
81
|
+
#: reaches the upstream is the adapter's call: `openai/` is removed only when the
|
|
82
|
+
#: upstream is OpenAI's own API, because Groq, OpenRouter and Together name
|
|
83
|
+
#: OpenAI's models `openai/gpt-oss-120b` and would reject the bare id.
|
|
86
84
|
ROUTING_PREFIXES: tuple[str, ...] = ("bedrock", "openai", "fake")
|
|
87
85
|
|
|
88
86
|
|
|
@@ -181,7 +179,7 @@ HOSTS: tuple[Host, ...] = (
|
|
|
181
179
|
"https://generativelanguage.googleapis.com/v1beta/openai/",
|
|
182
180
|
"GEMINI_API_KEY",
|
|
183
181
|
None,
|
|
184
|
-
("gemini-2.5-flash", "gemini-2.
|
|
182
|
+
("gemini-2.5-flash", "gemini-2.5-flash-lite", "gemini-3.5-flash"),
|
|
185
183
|
"Also reads GOOGLE_API_KEY. Generous free tier.",
|
|
186
184
|
),
|
|
187
185
|
Host(
|
|
@@ -201,7 +199,7 @@ HOSTS: tuple[Host, ...] = (
|
|
|
201
199
|
"https://api.groq.com/openai/v1",
|
|
202
200
|
"GROQ_API_KEY",
|
|
203
201
|
None,
|
|
204
|
-
("
|
|
202
|
+
("openai/gpt-oss-20b", "openai/gpt-oss-120b", "qwen/qwen3.6-27b"),
|
|
205
203
|
"Very fast, useful free tier.",
|
|
206
204
|
),
|
|
207
205
|
Host(
|
|
@@ -131,17 +131,21 @@ def apply(settings: Any) -> str | None:
|
|
|
131
131
|
CACHELLM_OPENAI_API_KEY being set means the decision is already made, and
|
|
132
132
|
detection must not second-guess it.
|
|
133
133
|
|
|
134
|
+
"Set" means set anywhere settings are read from, not just exported. A base
|
|
135
|
+
URL written in a .env file used to be replaced by whichever vendor key
|
|
136
|
+
happened to be in the shell, because only the process environment was
|
|
137
|
+
checked. Settings already records which fields came from any source.
|
|
138
|
+
|
|
134
139
|
Returns a one-line description of what it configured, or None.
|
|
135
140
|
"""
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
)
|
|
141
|
+
fields = ("default_provider", "openai_base_url", "openai_api_key")
|
|
142
|
+
exported = any(os.environ.get(f"CACHELLM_{name.upper()}", "").strip() for name in fields)
|
|
143
|
+
# An empty `CACHELLM_OPENAI_API_KEY=` line is a placeholder, not a decision.
|
|
144
|
+
written = set(getattr(settings, "model_fields_set", ()))
|
|
145
|
+
configured = any(
|
|
146
|
+
name in written and str(getattr(settings, name, "") or "").strip() for name in fields
|
|
143
147
|
)
|
|
144
|
-
if
|
|
148
|
+
if exported or configured:
|
|
145
149
|
return None
|
|
146
150
|
|
|
147
151
|
options = available(include_fake=False)
|
|
@@ -11,6 +11,7 @@ from __future__ import annotations
|
|
|
11
11
|
import json
|
|
12
12
|
from collections.abc import AsyncIterator
|
|
13
13
|
from typing import Any
|
|
14
|
+
from urllib.parse import urlparse
|
|
14
15
|
|
|
15
16
|
import httpx
|
|
16
17
|
import structlog
|
|
@@ -32,12 +33,32 @@ class OpenAICompatProvider(Provider):
|
|
|
32
33
|
name = "openai"
|
|
33
34
|
|
|
34
35
|
def __init__(
|
|
35
|
-
self,
|
|
36
|
+
self,
|
|
37
|
+
settings: Settings,
|
|
38
|
+
base_url: str | None = None,
|
|
39
|
+
api_key: str | None = None,
|
|
40
|
+
transport: httpx.AsyncBaseTransport | None = None,
|
|
36
41
|
) -> None:
|
|
37
42
|
self._settings = settings
|
|
38
43
|
self._base_url = (base_url or settings.openai_base_url).rstrip("/")
|
|
39
44
|
self._api_key = api_key if api_key is not None else settings.openai_api_key
|
|
45
|
+
# Injectable so tests exercise the real client construction, including
|
|
46
|
+
# the auth header, against a fake upstream rather than the network.
|
|
47
|
+
self._transport = transport
|
|
40
48
|
self._client: httpx.AsyncClient | None = None
|
|
49
|
+
self._is_openai_itself = urlparse(self._base_url).hostname == "api.openai.com"
|
|
50
|
+
|
|
51
|
+
def resolve_model(self, model: str) -> str:
|
|
52
|
+
"""Drop our `openai/` routing prefix only when the upstream is OpenAI.
|
|
53
|
+
|
|
54
|
+
OpenAI's own API wants `gpt-4o-mini`. Groq, OpenRouter and Together all
|
|
55
|
+
name OpenAI's models `openai/gpt-oss-120b`, so to them the prefix is part
|
|
56
|
+
of the model id. Stripping it there turned Groq's two main chat models
|
|
57
|
+
into 404s, which only showed up against the live API.
|
|
58
|
+
"""
|
|
59
|
+
if model.startswith("openai/") and not self._is_openai_itself:
|
|
60
|
+
return model
|
|
61
|
+
return super().resolve_model(model)
|
|
41
62
|
|
|
42
63
|
def _http(self) -> httpx.AsyncClient:
|
|
43
64
|
if self._client is None:
|
|
@@ -48,6 +69,7 @@ class OpenAICompatProvider(Provider):
|
|
|
48
69
|
base_url=self._base_url,
|
|
49
70
|
headers=headers,
|
|
50
71
|
timeout=httpx.Timeout(self._settings.request_timeout, connect=10.0),
|
|
72
|
+
transport=self._transport,
|
|
51
73
|
)
|
|
52
74
|
return self._client
|
|
53
75
|
|
|
@@ -12,6 +12,7 @@ from __future__ import annotations
|
|
|
12
12
|
|
|
13
13
|
import os
|
|
14
14
|
import sys
|
|
15
|
+
import textwrap
|
|
15
16
|
import time
|
|
16
17
|
from typing import Any
|
|
17
18
|
|
|
@@ -73,7 +74,12 @@ def header(stats: dict[str, Any], embedding_model: str = "") -> list[str]:
|
|
|
73
74
|
bits.append(f"{MAGENTA}shadow mode{RESET}")
|
|
74
75
|
if not stats.get("caching_enabled", True):
|
|
75
76
|
bits.append(f"{AMBER}caching disabled{RESET}")
|
|
76
|
-
|
|
77
|
+
lines = ["", " " + f" {DIM}·{RESET} ".join(bits)]
|
|
78
|
+
# A reachable Redis that was passed over is worth a line of its own:
|
|
79
|
+
# whoever started it expected it to be used.
|
|
80
|
+
if note := stats.get("backend_note"):
|
|
81
|
+
lines += [f" {AMBER}{row}{RESET}" for row in textwrap.wrap(str(note), 74)]
|
|
82
|
+
return [*lines, ""]
|
|
77
83
|
|
|
78
84
|
|
|
79
85
|
def summary(stats: dict[str, Any], embedding_model: str = "") -> str:
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|