cachellm-proxy 0.2.1__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/PKG-INFO +49 -11
  2. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/README.md +48 -10
  3. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/pyproject.toml +1 -1
  4. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/pyproject.toml.orig +1 -1
  5. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/__init__.py +1 -1
  6. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/api/deps.py +57 -5
  7. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/api/routes_admin.py +2 -0
  8. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/api/routes_chat.py +31 -10
  9. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/api/sse.py +11 -0
  10. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/embeddings/fastembed_backend.py +15 -3
  11. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/pricing.py +19 -4
  12. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/providers/catalog.py +6 -8
  13. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/providers/detect.py +12 -8
  14. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/providers/openai_compat.py +23 -1
  15. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/report.py +7 -1
  16. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/__main__.py +0 -0
  17. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/api/__init__.py +0 -0
  18. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/api/app.py +0 -0
  19. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/api/auth.py +0 -0
  20. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/cache/__init__.py +0 -0
  21. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/cache/analytics.py +0 -0
  22. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/cache/coalesce.py +0 -0
  23. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/cache/entry.py +0 -0
  24. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/cache/exact_store.py +0 -0
  25. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/cache/keys.py +0 -0
  26. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/cache/memory.py +0 -0
  27. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/cache/policy.py +0 -0
  28. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/cache/redis_client.py +0 -0
  29. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/cache/service.py +0 -0
  30. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/cache/vector_store.py +0 -0
  31. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/cli.py +0 -0
  32. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/embeddings/__init__.py +0 -0
  33. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/embeddings/base.py +0 -0
  34. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/embeddings/hash_backend.py +0 -0
  35. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/errors.py +0 -0
  36. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/logging_setup.py +0 -0
  37. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/models.py +0 -0
  38. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/observability/__init__.py +0 -0
  39. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/observability/metrics.py +0 -0
  40. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/observability/tracing.py +0 -0
  41. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/providers/__init__.py +0 -0
  42. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/providers/base.py +0 -0
  43. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/providers/bedrock.py +0 -0
  44. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/providers/fake.py +0 -0
  45. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/providers/registry.py +0 -0
  46. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/py.typed +0 -0
  47. {cachellm_proxy-0.2.1 → cachellm_proxy-0.2.2}/src/cachellm/settings.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cachellm-proxy
3
- Version: 0.2.1
3
+ Version: 0.2.2
4
4
  Summary: A drop-in semantic cache for OpenAI-compatible LLM APIs: match requests by meaning, serve repeats instantly, cut spend.
5
5
  Keywords: llm,cache,semantic-cache,openai,bedrock,proxy,vector-search,redis,fastapi,llmops,cost-optimization
6
6
  Author: Adarsh Dwivedi
@@ -53,13 +53,13 @@ Description-Content-Type: text/markdown
53
53
  [![CI](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml/badge.svg)](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml)
54
54
  [![Python 3.11+](https://img.shields.io/badge/python-3.11%20%7C%203.12%20%7C%203.13-blue)](https://www.python.org/)
55
55
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
56
- [![Tests](https://img.shields.io/badge/tests-265%20passing-brightgreen)](tests/)
56
+ [![Tests](https://img.shields.io/badge/tests-327%20passing-brightgreen)](tests/)
57
57
  [![PyPI](https://img.shields.io/pypi/v/cachellm-proxy)](https://pypi.org/project/cachellm-proxy/)
58
58
  [![Hit rate](https://img.shields.io/badge/hit%20rate-77%25%20on%20Bedrock-orange)](docs/evaluation.md)
59
59
 
60
60
  **A drop-in semantic cache for OpenAI-compatible LLM APIs. Change one base URL, and questions your model has already answered come back in milliseconds instead of seconds.**
61
61
 
62
- On a 2,000-request replay against **AWS Bedrock** it served **77% of traffic from cache** with **zero false positives** on genuinely new questions, cutting spend by **78%** and p95 latency from 1,023 ms to **5.7 ms**.
62
+ On a 2,000-request replay against **AWS Bedrock** it served **77% of traffic from cache** with **zero false positives** on genuinely new questions, cutting spend by **78%** and p95 latency from 1,023 ms to **5.7 ms** on a cache hit, or 9.7 ms when the question was reworded.
63
63
 
64
64
  [Quick start](#quick-start) · [How it works](#how-it-works) · [Evaluation](docs/evaluation.md) · [API reference](#api-reference) · [Deployment](#deployment-and-infrastructure)
65
65
 
@@ -97,7 +97,8 @@ Everything else stays the same: same request shape, same response shape, same er
97
97
  | False positives on genuinely new questions | **0 of 368** |
98
98
  | Reworded repeats served from cache | 95.2% |
99
99
  | Exact repeats served from cache | 99.3% |
100
- | p95 latency, cache hit | **5.7 ms** |
100
+ | p95 latency, any cache hit | **5.7 ms** |
101
+ | p95 latency, reworded hit | 9.7 ms |
101
102
  | p95 latency, cache miss | 1,022.8 ms |
102
103
  | p95 speedup | **178.8x** |
103
104
  | Cost reduction | **78.0%** |
@@ -257,7 +258,7 @@ This is where a caching project usually hand-waves. The full write-up is in [doc
257
258
 
258
259
  **Result 2: thresholds do not transfer between models.** The safe threshold ranges from 0.89 for MiniLM to 0.98 for Arctic-embed. Every model tested had negative separation, meaning the mean duplicate score sat below the worst hard negative. Bigger and slower did not fix it.
259
260
 
260
- | Model | Safe threshold | Recall there | Embed ms |
261
+ | Model | Safe threshold | Recall there | Embed ms, short probe |
261
262
  | --- | ---: | ---: | ---: |
262
263
  | `all-MiniLM-L6-v2` | **0.89** | **35.0%** | 5.7 |
263
264
  | `gte-base` | 0.96 | 26.0% | 20.6 |
@@ -266,7 +267,7 @@ This is where a caching project usually hand-waves. The full write-up is in [doc
266
267
  | `snowflake-arctic-embed-s` | 0.98 | 11.4% | 3.2 |
267
268
  | `bge-small-en-v1.5` | 0.96 | 8.1% | 3.8 |
268
269
 
269
- MiniLM gives four times the safe recall of bge-small at a third of the download size, so it is the default. The whole table ships in code as `CALIBRATED_THRESHOLDS`, and the proxy warns at startup if you configure a model it has never measured.
270
+ MiniLM gives four times the safe recall of bge-small, for a slightly larger download of 86 MB against 63 MB, so it is the default. The embed column is a ranking from short synthetic strings, not what a real question costs. The load test measures that directly. The whole table ships in code as `CALIBRATED_THRESHOLDS`, and the proxy warns at startup if you configure a model it has never measured.
270
271
 
271
272
  **Result 3, a negative one: a lexical guard does not rescue it.** The obvious fix is to require matched prompts to share content words. Measured, hard negatives have *higher* token overlap (0.42 mean) than genuine paraphrases share vocabulary, because they differ by exactly one decisive word. The guard rejects good matches and keeps dangerous ones. It was measured and dropped rather than shipped.
272
273
 
@@ -366,7 +367,7 @@ One environment variable per host. Everything except Bedrock speaks the OpenAI p
366
367
  | Host | `CACHELLM_OPENAI_BASE_URL` | Example model |
367
368
  | --- | --- | --- |
368
369
  | OpenAI | `https://api.openai.com/v1` | `gpt-4o-mini` |
369
- | Groq | `https://api.groq.com/openai/v1` | `llama-3.3-70b-versatile` |
370
+ | Groq | `https://api.groq.com/openai/v1` | `openai/gpt-oss-20b` |
370
371
  | Google Gemini | `https://generativelanguage.googleapis.com/v1beta/openai/` | `gemini-2.5-flash` |
371
372
  | Anthropic | `https://api.anthropic.com/v1` | `claude-haiku-4-5` |
372
373
  | OpenRouter | `https://openrouter.ai/api/v1` | `anthropic/claude-3.5-sonnet` |
@@ -380,6 +381,10 @@ CACHELLM_OPENAI_API_KEY=gsk_your_key \
380
381
  cachellm serve
381
382
  ```
382
383
 
384
+ If the key is already in its usual variable, such as `GROQ_API_KEY` or `GEMINI_API_KEY`, you can skip both lines: the proxy finds it at startup and says which host it picked.
385
+
386
+ **Checked live, not just in tests.** On 2026-09-11 the proxy was run against real **Google Gemini** (`gemini-2.5-flash`, `gemini-3.5-flash`) and **Groq** (`openai/gpt-oss-20b`, `openai/gpt-oss-120b`), driven by the official OpenAI SDK, alongside the AWS Bedrock benchmark. Each run covers a miss, an exact hit, a reworded hit, a similar question that must not hit, a stream and its replay, a hot-temperature bypass and an unknown model's error, and checks that the API key never reaches the log. Results are in [docs/evaluation.md](docs/evaluation.md#live-provider-checks), and `bench/live_check.py` reruns them with your own key. The other hosts share the same adapter and its tests, but have not yet been run against the real service.
387
+
383
388
  **AWS Bedrock** is the one exception, because it does not speak the OpenAI protocol. It needs the `aws` extra, and then uses your existing AWS credentials with no vendor API key at all:
384
389
 
385
390
  ```bash
@@ -401,7 +406,7 @@ Three rules, in order. You never configure a model list.
401
406
  2. **A recognisable vendor convention.** Bedrock ids are always `vendor.model`, so `amazon.nova-lite-v1:0` and `us.anthropic.claude-3-haiku-20240307-v1:0` are identified with no prefix. OpenAI's own families (`gpt-`, `o1`, `o3`, `text-embedding-`) are identified the same way, whatever the default is.
402
407
  3. **Otherwise the configured default**, which is the OpenAI-compatible adapter. Names like `llama3.2`, `mixtral-8x7b-32768` and `qwen2.5-coder:7b` are served by Groq, Ollama, Together and OpenRouter alike, so the endpoint you configured is the only sensible answer.
403
408
 
404
- Model ids that legitimately contain a slash, which OpenRouter and Together both use, are forwarded whole. Only this proxy's own prefixes are stripped. Note that `anthropic.claude-…` with a dot is a Bedrock id while `anthropic/claude-…` with a slash is an OpenRouter id, and the router tells them apart.
409
+ Model ids that legitimately contain a slash, which OpenRouter and Together both use, are forwarded whole. Only this proxy's own prefixes are stripped, and `openai/` only when the upstream is OpenAI itself: Groq, OpenRouter and Together name OpenAI's models `openai/gpt-oss-120b`, so to them the prefix is part of the id. Note that `anthropic.claude-…` with a dot is a Bedrock id while `anthropic/claude-…` with a slash is an OpenRouter id, and the router tells them apart.
405
410
 
406
411
  Ask it directly if you are unsure:
407
412
 
@@ -410,6 +415,39 @@ curl -s localhost:8080/admin/route/meta-llama/Llama-3.3-70B-Instruct-Turbo
410
415
  curl -s localhost:8080/admin/providers
411
416
  ```
412
417
 
418
+ ### Where the cache lives
419
+
420
+ One setting, `CACHELLM_BACKEND`, with three values. Most people never touch it.
421
+
422
+ | Value | What happens | Choose it when |
423
+ | --- | --- | --- |
424
+ | `auto`, the default | Redis if the extra is installed and a usable server answers, the in-process cache otherwise | You have no opinion |
425
+ | `memory` | A numpy matrix inside the proxy. Nothing to install | One process: a laptop, a side project, a single server |
426
+ | `redis` | One Redis server shared by every copy of the proxy | Several processes must share one cache |
427
+
428
+ Memory is not the slow option. Below roughly 100,000 entries it is the faster one: scanning 20,000 cached prompts takes 0.85 ms, while a Redis round trip alone costs 2 to 3 ms. Redis earns its place when several processes need one shared cache. Run four copies of the proxy on memory and you have four separate caches, each a quarter as warm.
429
+
430
+ Memory can survive a restart too. Give it a file, and it saves there on a clean shutdown and loads it back on start:
431
+
432
+ ```bash
433
+ CACHELLM_MEMORY_SNAPSHOT_PATH=$HOME/.cachellm/cache.npz cachellm serve
434
+ ```
435
+
436
+ **Setting up Redis.** Install the extra, then run a Redis 8 server:
437
+
438
+ ```bash
439
+ pip install "cachellm-proxy[redis]"
440
+ brew install redis && brew services start redis # macOS
441
+ docker run -d -p 6379:6379 redis:8-alpine # anywhere with Docker
442
+ ```
443
+
444
+ The server has two requirements, and both are checked at startup:
445
+
446
+ - **It needs the search module**, which is what stores and searches vectors. Redis 8 from Homebrew or the official Docker image includes it. Many Linux distribution packages ship an older Redis without it, so prefer the Docker image there. A hosted Redis works if its plan includes search.
447
+ - **It has to be database 0**, because Redis search cannot index any other. Keep several apps apart with `CACHELLM_INDEX_NAME` instead.
448
+
449
+ If either is missing, `auto` uses memory and says exactly why, both in a startup warning and at the top of `cachellm stats`. With `CACHELLM_BACKEND=redis` the proxy never switches storage behind your back. It keeps answering without a cache and reports itself degraded until Redis is fixed.
450
+
413
451
  ## API reference
414
452
 
415
453
  ### Chat completions
@@ -506,7 +544,7 @@ Every setting is an environment variable prefixed `CACHELLM_`, or a line in `.en
506
544
  | Variable | Default | Notes |
507
545
  | --- | --- | --- |
508
546
  | `CACHELLM_BACKEND` | `auto` | `memory` needs nothing, `redis` shares one cache across workers, `auto` uses Redis when reachable and memory when not |
509
- | `CACHELLM_REDIS_URL` | `redis://localhost:6379/0` | Must be database 0. Redis Search cannot index any other |
547
+ | `CACHELLM_REDIS_URL` | `redis://localhost:6379/0` | Must be database 0, and the server needs the search module. See [Where the cache lives](#where-the-cache-lives) |
510
548
  | `CACHELLM_MEMORY_MAX_ENTRIES` | `50000` | Cap for the in-memory store. 50k of 384-dim vectors is about 73 MB |
511
549
  | `CACHELLM_MEMORY_SNAPSHOT_PATH` | unset | Persist the in-memory cache to this file so a restart does not start cold |
512
550
  | `CACHELLM_API_KEYS` | empty | Comma-separated client keys. Empty disables auth, which is local development only |
@@ -517,7 +555,7 @@ Every setting is an environment variable prefixed `CACHELLM_`, or a line in `.en
517
555
  | `CACHELLM_TTL_VOLATILE` | `900` | Fifteen minutes for anything about now |
518
556
  | `CACHELLM_MAX_CACHEABLE_TEMPERATURE` | `0.3` | Above this, nothing is cached |
519
557
  | `CACHELLM_PII_GUARD` | `true` | Refuse to store prompts that look personal |
520
- | `CACHELLM_DEFAULT_PROVIDER` | `bedrock` | `bedrock`, `openai` or `fake` |
558
+ | `CACHELLM_DEFAULT_PROVIDER` | `openai` | `openai` covers every OpenAI-compatible host. Also `bedrock` or `fake`. Left unset, the proxy picks from the keys it finds |
521
559
  | `CACHELLM_LOG_PROMPTS` | `false` | Prompt text stays out of logs unless you opt in |
522
560
 
523
561
  ## Deployment and infrastructure
@@ -586,7 +624,7 @@ cachellm/
586
624
  ## Testing
587
625
 
588
626
  ```bash
589
- make test # 126 tests
627
+ make test # 327 tests
590
628
  make test-cov # with coverage
591
629
  make lint # ruff and mypy
592
630
  ```
@@ -3,13 +3,13 @@
3
3
  [![CI](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml/badge.svg)](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml)
4
4
  [![Python 3.11+](https://img.shields.io/badge/python-3.11%20%7C%203.12%20%7C%203.13-blue)](https://www.python.org/)
5
5
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
6
- [![Tests](https://img.shields.io/badge/tests-265%20passing-brightgreen)](tests/)
6
+ [![Tests](https://img.shields.io/badge/tests-327%20passing-brightgreen)](tests/)
7
7
  [![PyPI](https://img.shields.io/pypi/v/cachellm-proxy)](https://pypi.org/project/cachellm-proxy/)
8
8
  [![Hit rate](https://img.shields.io/badge/hit%20rate-77%25%20on%20Bedrock-orange)](docs/evaluation.md)
9
9
 
10
10
  **A drop-in semantic cache for OpenAI-compatible LLM APIs. Change one base URL, and questions your model has already answered come back in milliseconds instead of seconds.**
11
11
 
12
- On a 2,000-request replay against **AWS Bedrock** it served **77% of traffic from cache** with **zero false positives** on genuinely new questions, cutting spend by **78%** and p95 latency from 1,023 ms to **5.7 ms**.
12
+ On a 2,000-request replay against **AWS Bedrock** it served **77% of traffic from cache** with **zero false positives** on genuinely new questions, cutting spend by **78%** and p95 latency from 1,023 ms to **5.7 ms** on a cache hit, or 9.7 ms when the question was reworded.
13
13
 
14
14
  [Quick start](#quick-start) · [How it works](#how-it-works) · [Evaluation](docs/evaluation.md) · [API reference](#api-reference) · [Deployment](#deployment-and-infrastructure)
15
15
 
@@ -47,7 +47,8 @@ Everything else stays the same: same request shape, same response shape, same er
47
47
  | False positives on genuinely new questions | **0 of 368** |
48
48
  | Reworded repeats served from cache | 95.2% |
49
49
  | Exact repeats served from cache | 99.3% |
50
- | p95 latency, cache hit | **5.7 ms** |
50
+ | p95 latency, any cache hit | **5.7 ms** |
51
+ | p95 latency, reworded hit | 9.7 ms |
51
52
  | p95 latency, cache miss | 1,022.8 ms |
52
53
  | p95 speedup | **178.8x** |
53
54
  | Cost reduction | **78.0%** |
@@ -207,7 +208,7 @@ This is where a caching project usually hand-waves. The full write-up is in [doc
207
208
 
208
209
  **Result 2: thresholds do not transfer between models.** The safe threshold ranges from 0.89 for MiniLM to 0.98 for Arctic-embed. Every model tested had negative separation, meaning the mean duplicate score sat below the worst hard negative. Bigger and slower did not fix it.
209
210
 
210
- | Model | Safe threshold | Recall there | Embed ms |
211
+ | Model | Safe threshold | Recall there | Embed ms, short probe |
211
212
  | --- | ---: | ---: | ---: |
212
213
  | `all-MiniLM-L6-v2` | **0.89** | **35.0%** | 5.7 |
213
214
  | `gte-base` | 0.96 | 26.0% | 20.6 |
@@ -216,7 +217,7 @@ This is where a caching project usually hand-waves. The full write-up is in [doc
216
217
  | `snowflake-arctic-embed-s` | 0.98 | 11.4% | 3.2 |
217
218
  | `bge-small-en-v1.5` | 0.96 | 8.1% | 3.8 |
218
219
 
219
- MiniLM gives four times the safe recall of bge-small at a third of the download size, so it is the default. The whole table ships in code as `CALIBRATED_THRESHOLDS`, and the proxy warns at startup if you configure a model it has never measured.
220
+ MiniLM gives four times the safe recall of bge-small, for a slightly larger download of 86 MB against 63 MB, so it is the default. The embed column is a ranking from short synthetic strings, not what a real question costs. The load test measures that directly. The whole table ships in code as `CALIBRATED_THRESHOLDS`, and the proxy warns at startup if you configure a model it has never measured.
220
221
 
221
222
  **Result 3, a negative one: a lexical guard does not rescue it.** The obvious fix is to require matched prompts to share content words. Measured, hard negatives have *higher* token overlap (0.42 mean) than genuine paraphrases share vocabulary, because they differ by exactly one decisive word. The guard rejects good matches and keeps dangerous ones. It was measured and dropped rather than shipped.
222
223
 
@@ -316,7 +317,7 @@ One environment variable per host. Everything except Bedrock speaks the OpenAI p
316
317
  | Host | `CACHELLM_OPENAI_BASE_URL` | Example model |
317
318
  | --- | --- | --- |
318
319
  | OpenAI | `https://api.openai.com/v1` | `gpt-4o-mini` |
319
- | Groq | `https://api.groq.com/openai/v1` | `llama-3.3-70b-versatile` |
320
+ | Groq | `https://api.groq.com/openai/v1` | `openai/gpt-oss-20b` |
320
321
  | Google Gemini | `https://generativelanguage.googleapis.com/v1beta/openai/` | `gemini-2.5-flash` |
321
322
  | Anthropic | `https://api.anthropic.com/v1` | `claude-haiku-4-5` |
322
323
  | OpenRouter | `https://openrouter.ai/api/v1` | `anthropic/claude-3.5-sonnet` |
@@ -330,6 +331,10 @@ CACHELLM_OPENAI_API_KEY=gsk_your_key \
330
331
  cachellm serve
331
332
  ```
332
333
 
334
+ If the key is already in its usual variable, such as `GROQ_API_KEY` or `GEMINI_API_KEY`, you can skip both lines: the proxy finds it at startup and says which host it picked.
335
+
336
+ **Checked live, not just in tests.** On 2026-09-11 the proxy was run against real **Google Gemini** (`gemini-2.5-flash`, `gemini-3.5-flash`) and **Groq** (`openai/gpt-oss-20b`, `openai/gpt-oss-120b`), driven by the official OpenAI SDK, alongside the AWS Bedrock benchmark. Each run covers a miss, an exact hit, a reworded hit, a similar question that must not hit, a stream and its replay, a hot-temperature bypass and an unknown model's error, and checks that the API key never reaches the log. Results are in [docs/evaluation.md](docs/evaluation.md#live-provider-checks), and `bench/live_check.py` reruns them with your own key. The other hosts share the same adapter and its tests, but have not yet been run against the real service.
337
+
333
338
  **AWS Bedrock** is the one exception, because it does not speak the OpenAI protocol. It needs the `aws` extra, and then uses your existing AWS credentials with no vendor API key at all:
334
339
 
335
340
  ```bash
@@ -351,7 +356,7 @@ Three rules, in order. You never configure a model list.
351
356
  2. **A recognisable vendor convention.** Bedrock ids are always `vendor.model`, so `amazon.nova-lite-v1:0` and `us.anthropic.claude-3-haiku-20240307-v1:0` are identified with no prefix. OpenAI's own families (`gpt-`, `o1`, `o3`, `text-embedding-`) are identified the same way, whatever the default is.
352
357
  3. **Otherwise the configured default**, which is the OpenAI-compatible adapter. Names like `llama3.2`, `mixtral-8x7b-32768` and `qwen2.5-coder:7b` are served by Groq, Ollama, Together and OpenRouter alike, so the endpoint you configured is the only sensible answer.
353
358
 
354
- Model ids that legitimately contain a slash, which OpenRouter and Together both use, are forwarded whole. Only this proxy's own prefixes are stripped. Note that `anthropic.claude-…` with a dot is a Bedrock id while `anthropic/claude-…` with a slash is an OpenRouter id, and the router tells them apart.
359
+ Model ids that legitimately contain a slash, which OpenRouter and Together both use, are forwarded whole. Only this proxy's own prefixes are stripped, and `openai/` only when the upstream is OpenAI itself: Groq, OpenRouter and Together name OpenAI's models `openai/gpt-oss-120b`, so to them the prefix is part of the id. Note that `anthropic.claude-…` with a dot is a Bedrock id while `anthropic/claude-…` with a slash is an OpenRouter id, and the router tells them apart.
355
360
 
356
361
  Ask it directly if you are unsure:
357
362
 
@@ -360,6 +365,39 @@ curl -s localhost:8080/admin/route/meta-llama/Llama-3.3-70B-Instruct-Turbo
360
365
  curl -s localhost:8080/admin/providers
361
366
  ```
362
367
 
368
+ ### Where the cache lives
369
+
370
+ One setting, `CACHELLM_BACKEND`, with three values. Most people never touch it.
371
+
372
+ | Value | What happens | Choose it when |
373
+ | --- | --- | --- |
374
+ | `auto`, the default | Redis if the extra is installed and a usable server answers, the in-process cache otherwise | You have no opinion |
375
+ | `memory` | A numpy matrix inside the proxy. Nothing to install | One process: a laptop, a side project, a single server |
376
+ | `redis` | One Redis server shared by every copy of the proxy | Several processes must share one cache |
377
+
378
+ Memory is not the slow option. Below roughly 100,000 entries it is the faster one: scanning 20,000 cached prompts takes 0.85 ms, while a Redis round trip alone costs 2 to 3 ms. Redis earns its place when several processes need one shared cache. Run four copies of the proxy on memory and you have four separate caches, each a quarter as warm.
379
+
380
+ Memory can survive a restart too. Give it a file, and it saves there on a clean shutdown and loads it back on start:
381
+
382
+ ```bash
383
+ CACHELLM_MEMORY_SNAPSHOT_PATH=$HOME/.cachellm/cache.npz cachellm serve
384
+ ```
385
+
386
+ **Setting up Redis.** Install the extra, then run a Redis 8 server:
387
+
388
+ ```bash
389
+ pip install "cachellm-proxy[redis]"
390
+ brew install redis && brew services start redis # macOS
391
+ docker run -d -p 6379:6379 redis:8-alpine # anywhere with Docker
392
+ ```
393
+
394
+ The server has two requirements, and both are checked at startup:
395
+
396
+ - **It needs the search module**, which is what stores and searches vectors. Redis 8 from Homebrew or the official Docker image includes it. Many Linux distribution packages ship an older Redis without it, so prefer the Docker image there. A hosted Redis works if its plan includes search.
397
+ - **It has to be database 0**, because Redis search cannot index any other. Keep several apps apart with `CACHELLM_INDEX_NAME` instead.
398
+
399
+ If either is missing, `auto` uses memory and says exactly why, both in a startup warning and at the top of `cachellm stats`. With `CACHELLM_BACKEND=redis` the proxy never switches storage behind your back. It keeps answering without a cache and reports itself degraded until Redis is fixed.
400
+
363
401
  ## API reference
364
402
 
365
403
  ### Chat completions
@@ -456,7 +494,7 @@ Every setting is an environment variable prefixed `CACHELLM_`, or a line in `.en
456
494
  | Variable | Default | Notes |
457
495
  | --- | --- | --- |
458
496
  | `CACHELLM_BACKEND` | `auto` | `memory` needs nothing, `redis` shares one cache across workers, `auto` uses Redis when reachable and memory when not |
459
- | `CACHELLM_REDIS_URL` | `redis://localhost:6379/0` | Must be database 0. Redis Search cannot index any other |
497
+ | `CACHELLM_REDIS_URL` | `redis://localhost:6379/0` | Must be database 0, and the server needs the search module. See [Where the cache lives](#where-the-cache-lives) |
460
498
  | `CACHELLM_MEMORY_MAX_ENTRIES` | `50000` | Cap for the in-memory store. 50k of 384-dim vectors is about 73 MB |
461
499
  | `CACHELLM_MEMORY_SNAPSHOT_PATH` | unset | Persist the in-memory cache to this file so a restart does not start cold |
462
500
  | `CACHELLM_API_KEYS` | empty | Comma-separated client keys. Empty disables auth, which is local development only |
@@ -467,7 +505,7 @@ Every setting is an environment variable prefixed `CACHELLM_`, or a line in `.en
467
505
  | `CACHELLM_TTL_VOLATILE` | `900` | Fifteen minutes for anything about now |
468
506
  | `CACHELLM_MAX_CACHEABLE_TEMPERATURE` | `0.3` | Above this, nothing is cached |
469
507
  | `CACHELLM_PII_GUARD` | `true` | Refuse to store prompts that look personal |
470
- | `CACHELLM_DEFAULT_PROVIDER` | `bedrock` | `bedrock`, `openai` or `fake` |
508
+ | `CACHELLM_DEFAULT_PROVIDER` | `openai` | `openai` covers every OpenAI-compatible host. Also `bedrock` or `fake`. Left unset, the proxy picks from the keys it finds |
471
509
  | `CACHELLM_LOG_PROMPTS` | `false` | Prompt text stays out of logs unless you opt in |
472
510
 
473
511
  ## Deployment and infrastructure
@@ -536,7 +574,7 @@ cachellm/
536
574
  ## Testing
537
575
 
538
576
  ```bash
539
- make test # 126 tests
577
+ make test # 327 tests
540
578
  make test-cov # with coverage
541
579
  make lint # ruff and mypy
542
580
  ```
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cachellm-proxy"
3
- version = "0.2.1"
3
+ version = "0.2.2"
4
4
  description = "A drop-in semantic cache for OpenAI-compatible LLM APIs: match requests by meaning, serve repeats instantly, cut spend."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cachellm-proxy"
3
- version = "0.2.1"
3
+ version = "0.2.2"
4
4
  description = "A drop-in semantic cache for OpenAI-compatible LLM APIs: match requests by meaning, serve repeats instantly, cut spend."
5
5
  readme = "README.md"
6
6
  authors = [
@@ -2,7 +2,7 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
- __version__ = "0.2.1"
5
+ __version__ = "0.2.2"
6
6
 
7
7
  __all__ = ["__version__", "main"]
8
8
 
@@ -16,8 +16,10 @@ cache that takes an application down when it breaks is worse than no cache.
16
16
 
17
17
  from __future__ import annotations
18
18
 
19
+ import contextlib
19
20
  from dataclasses import dataclass, field
20
21
  from typing import TYPE_CHECKING, Any
22
+ from urllib.parse import urlparse
21
23
 
22
24
  import structlog
23
25
  from fastapi import Request
@@ -46,6 +48,44 @@ REDIS_MISSING = (
46
48
  )
47
49
 
48
50
 
51
+ class RedisUnusableError(RuntimeError):
52
+ """Redis answered, but it cannot hold this cache.
53
+
54
+ Different in kind from "no Redis here". Nobody runs a Redis server by
55
+ accident, so when one answers and still cannot be used, the operator almost
56
+ certainly meant to use it and deserves to hear why it was passed over.
57
+ """
58
+
59
+
60
+ def _where(url: str) -> str:
61
+ """Host and port only. The URL may carry a password, and this gets logged."""
62
+ parsed = urlparse(url)
63
+ if parsed.hostname:
64
+ return f"{parsed.hostname}:{parsed.port or 6379}"
65
+ return "the configured address"
66
+
67
+
68
+ async def _require_search(redis: Any, settings: Settings) -> None:
69
+ """Fail with a plain explanation before RedisVL fails with a cryptic one."""
70
+ where = _where(settings.redis_url)
71
+ db = int(redis.connection_pool.connection_kwargs.get("db", 0) or 0)
72
+ if db != 0:
73
+ raise RedisUnusableError(
74
+ f"Redis at {where} answered, but CACHELLM_REDIS_URL selects database {db}. "
75
+ "Redis search can only index database 0, so end the URL with /0."
76
+ )
77
+ try:
78
+ await redis.execute_command("FT._LIST")
79
+ except Exception as exc:
80
+ if "unknown command" not in str(exc).lower():
81
+ raise
82
+ raise RedisUnusableError(
83
+ f"Redis at {where} answered, but it has no search module, so it cannot "
84
+ "store vectors. Redis 8 from Homebrew or the official Docker image includes "
85
+ "it. Many Linux distribution packages do not."
86
+ ) from exc
87
+
88
+
49
89
  @dataclass
50
90
  class AppState:
51
91
  settings: Settings
@@ -62,6 +102,8 @@ class AppState:
62
102
  degraded_reason: str = ""
63
103
  #: Which backend actually started: "redis" or "memory".
64
104
  backend: str = "none"
105
+ #: Why a reachable Redis was passed over, shown by `cachellm stats`.
106
+ backend_note: str = ""
65
107
 
66
108
  @property
67
109
  def caching_on(self) -> bool:
@@ -96,8 +138,12 @@ async def build_state(
96
138
  except Exception as exc:
97
139
  redis_error = f"{type(exc).__name__}: {exc}"
98
140
  if settings.backend == "redis":
99
- state.degraded_reason = redis_error
100
- log.error("redis_unavailable_failing_open", error=redis_error)
141
+ unusable = isinstance(exc, RedisUnusableError)
142
+ state.degraded_reason = str(exc) if unusable else redis_error
143
+ log.error("redis_unavailable_failing_open", error=state.degraded_reason)
144
+ elif isinstance(exc, RedisUnusableError):
145
+ state.backend_note = f"{exc} Using the in-process cache instead."
146
+ log.warning("redis_unusable_using_memory", reason=str(exc))
101
147
  else:
102
148
  log.info(
103
149
  "redis_unavailable_using_memory",
@@ -135,9 +181,15 @@ async def _attach_redis(state: AppState, settings: Settings) -> None:
135
181
  raise ImportError(REDIS_MISSING) from exc
136
182
 
137
183
  redis = build_redis(settings)
138
- await redis.ping()
139
- vectors = VectorStore(redis, settings)
140
- await vectors.connect()
184
+ try:
185
+ await redis.ping()
186
+ await _require_search(redis, settings)
187
+ vectors = VectorStore(redis, settings)
188
+ await vectors.connect()
189
+ except Exception:
190
+ with contextlib.suppress(Exception):
191
+ await redis.aclose()
192
+ raise
141
193
  state.redis = redis
142
194
  _wire(
143
195
  state, settings, vectors, ExactStore(redis, settings), Analytics(redis, settings), "redis"
@@ -76,6 +76,8 @@ async def stats(request: Request) -> dict[str, Any]:
76
76
  data = await state.cache.stats()
77
77
  data["cache_available"] = True
78
78
  data["backend"] = state.backend
79
+ if state.backend_note:
80
+ data["backend_note"] = state.backend_note
79
81
  data["caching_enabled"] = state.settings.enabled
80
82
  data["shadow_mode"] = state.settings.shadow_mode
81
83
  data["in_flight"] = state.singleflight.in_flight
@@ -32,7 +32,7 @@ from cachellm.models import (
32
32
  )
33
33
  from cachellm.observability import span
34
34
  from cachellm.pricing import estimate_cost
35
- from cachellm.providers.base import Provider
35
+ from cachellm.providers.base import Provider, StreamEvent
36
36
 
37
37
  log = structlog.get_logger(__name__)
38
38
  router = APIRouter()
@@ -240,6 +240,12 @@ async def chat_completions(request: Request, body: ChatCompletionRequest) -> Any
240
240
  return await _proxy_once(state, body, lookup, provider, provider_name, started)
241
241
 
242
242
 
243
+ async def _count_provider_error(state: AppState, provider_name: str) -> None:
244
+ state.metrics.provider_errors.labels(provider=provider_name).inc()
245
+ if state.analytics is not None:
246
+ await state.analytics.incr("provider_errors")
247
+
248
+
243
249
  async def _proxy_once(
244
250
  state: AppState,
245
251
  body: ChatCompletionRequest,
@@ -264,9 +270,7 @@ async def _proxy_once(
264
270
  else:
265
271
  result, coalesced = await call(), False
266
272
  except UpstreamError:
267
- state.metrics.provider_errors.labels(provider=provider_name).inc()
268
- if state.analytics is not None:
269
- await state.analytics.incr("provider_errors")
273
+ await _count_provider_error(state, provider_name)
270
274
  raise
271
275
 
272
276
  if coalesced:
@@ -361,6 +365,25 @@ async def _proxy_stream(
361
365
  """
362
366
  stream_id = sse.new_stream_id()
363
367
  created = int(time.time())
368
+ upstream = aiter(provider.stream(body))
369
+
370
+ # Hold the response until the upstream sends its first event. A wrong model
371
+ # name or a bad key fails right here, so the caller gets the upstream's real
372
+ # status code. Answering first used to turn every such failure into a 200
373
+ # stream with an empty answer and nothing to say why.
374
+ try:
375
+ first: StreamEvent | None = await anext(upstream)
376
+ except StopAsyncIteration:
377
+ first = None
378
+ except UpstreamError:
379
+ await _count_provider_error(state, provider_name)
380
+ raise
381
+
382
+ async def events() -> AsyncIterator[StreamEvent]:
383
+ if first is not None:
384
+ yield first
385
+ async for event in upstream:
386
+ yield event
364
387
 
365
388
  async def generate() -> AsyncIterator[str]:
366
389
  buffer: list[str] = []
@@ -369,7 +392,7 @@ async def _proxy_stream(
369
392
  completion_tokens = 0
370
393
  yield sse.role_chunk(stream_id, body.model, created)
371
394
  try:
372
- async for event in provider.stream(body):
395
+ async for event in events():
373
396
  if event.delta:
374
397
  buffer.append(event.delta)
375
398
  yield sse.text_chunk(stream_id, body.model, created, event.delta)
@@ -379,11 +402,9 @@ async def _proxy_stream(
379
402
  prompt_tokens = event.prompt_tokens or prompt_tokens
380
403
  completion_tokens = event.completion_tokens or completion_tokens
381
404
  except UpstreamError as exc:
382
- state.metrics.provider_errors.labels(provider=provider_name).inc()
383
- if state.analytics is not None:
384
- await state.analytics.incr("provider_errors")
385
- log.warning("stream_failed", error=str(exc)[:200])
386
- yield sse.final_chunk(stream_id, body.model, created, "error", None)
405
+ await _count_provider_error(state, provider_name)
406
+ log.warning("stream_failed", error=exc.message[:200])
407
+ yield sse.error_event(exc.message, exc.err_type, exc.code)
387
408
  yield sse.DONE
388
409
  return
389
410
 
@@ -71,6 +71,17 @@ def final_chunk(
71
71
  )
72
72
 
73
73
 
74
+ def error_event(message: str, err_type: str = "api_error", code: str | None = None) -> str:
75
+ """A failure after the stream has started, in OpenAI's shape.
76
+
77
+ The status line has already gone out as 200 by then, so this event is the
78
+ only way left to say something broke. The official SDKs raise an error when
79
+ they read it, instead of ending quietly with a half-written answer.
80
+ """
81
+ payload = {"error": {"message": message, "type": err_type, "param": None, "code": code}}
82
+ return f"data: {json.dumps(payload, separators=(',', ':'))}\n\n"
83
+
84
+
74
85
  def split_for_replay(text: str, max_chars: int = 24) -> list[str]:
75
86
  """Chop a cached answer into believable stream chunks.
76
87
 
@@ -64,12 +64,24 @@ class FastEmbedEmbedder(Embedder):
64
64
  return vectors[0]
65
65
 
66
66
  async def embed_batch(self, texts: list[str]) -> list[np.ndarray]:
67
+ """Embed many texts, running the model once per distinct uncached text.
68
+
69
+ Results are assembled from what was just computed, never read back out
70
+ of the cache. The cache is bounded, so reading back used to fail two
71
+ ways: a batch larger than the cache evicted its own first results, and
72
+ a cache size of 0 made every call fail, which quietly turned caching off.
73
+ """
67
74
  if not texts:
68
75
  return []
69
76
  model = await self._ensure_model()
70
- pending = [t for t in texts if self._cache_get(t) is None]
77
+ found: dict[str, np.ndarray] = {}
78
+ for text in texts:
79
+ if (hit := self._cache_get(text)) is not None:
80
+ found[text] = hit
81
+ pending = list(dict.fromkeys(t for t in texts if t not in found))
71
82
  if pending:
72
83
  raw = await asyncio.to_thread(lambda: list(model.embed(pending)))
73
84
  for text, vector in zip(pending, raw, strict=True):
74
- self._cache_put(text, self.normalise(np.asarray(vector, dtype=np.float32)))
75
- return [self._cache[t] for t in texts]
85
+ found[text] = self.normalise(np.asarray(vector, dtype=np.float32))
86
+ self._cache_put(text, found[text])
87
+ return [found[t] for t in texts]
@@ -50,6 +50,18 @@ PRICES: dict[str, ModelPrice] = {
50
50
  "gpt-4o-mini": ModelPrice(0.15, 0.60),
51
51
  "gpt-4o": ModelPrice(2.50, 10.00),
52
52
  "gpt-4.1-mini": ModelPrice(0.40, 1.60),
53
+ # --- Google Gemini, paid tier, output includes thinking (checked 2026-09-11) ---
54
+ "gemini-2.5-flash": ModelPrice(0.30, 2.50),
55
+ "gemini-2.5-flash-lite": ModelPrice(0.10, 0.40),
56
+ "gemini-2.5-pro": ModelPrice(1.25, 10.00),
57
+ "gemini-3.5-flash": ModelPrice(1.50, 9.00),
58
+ "gemini-3.5-flash-lite": ModelPrice(0.30, 2.50),
59
+ # --- Groq, keyed without the vendor segment (checked 2026-09-11) ---
60
+ "gpt-oss-20b": ModelPrice(0.075, 0.30),
61
+ "gpt-oss-120b": ModelPrice(0.15, 0.60),
62
+ "gpt-oss-safeguard-20b": ModelPrice(0.075, 0.30),
63
+ "qwen3.6-27b": ModelPrice(0.60, 3.00),
64
+ "qwen3.8-27b": ModelPrice(0.80, 4.00),
53
65
  # --- embeddings (input only) ---
54
66
  "amazon.titan-embed-text-v2:0": ModelPrice(0.02, 0.0),
55
67
  "text-embedding-3-small": ModelPrice(0.02, 0.0),
@@ -97,10 +109,13 @@ def price_for(model: str) -> ModelPrice:
97
109
  key = normalise_model_id(model)
98
110
  if key in PRICES:
99
111
  return PRICES[key]
100
- # Prefix match so undated model revisions still price sensibly.
101
- for known, price in PRICES.items():
102
- if key.startswith(known.split("-2024")[0].split("-v1:")[0]):
103
- return price
112
+ # Prefix match so dated or preview revisions still price sensibly. The
113
+ # longest match wins: `gemini-2.5-flash-lite-preview` is a Flash-Lite, and
114
+ # taking the first match in table order priced it as the dearer Flash.
115
+ stems = {known.split("-2024")[0].split("-v1:")[0]: known for known in PRICES}
116
+ matches = [stem for stem in stems if key.startswith(stem)]
117
+ if matches:
118
+ return PRICES[stems[max(matches, key=len)]]
104
119
  return _FALLBACK
105
120
 
106
121
 
@@ -77,12 +77,10 @@ OPENAI_FAMILIES: tuple[str, ...] = (
77
77
  #: Together both use `vendor/model` ids, and stripping their vendor segment
78
78
  #: makes the upstream reject the request as an unknown model.
79
79
  #:
80
- #: These three names are therefore reserved. A model id that genuinely begins
81
- #: `openai/`, `bedrock/` or `fake/` is read as a routing instruction: Groq's
82
- #: `openai/gpt-oss-120b` reaches the upstream as `gpt-oss-120b`. That is the one
83
- #: known collision, it affects one model family on one host, and the workaround
84
- #: is to point CACHELLM_OPENAI_BASE_URL at Groq and send the bare id if the host
85
- #: accepts it.
80
+ #: A model id starting with one of these always routes to that adapter. What
81
+ #: reaches the upstream is the adapter's call: `openai/` is removed only when the
82
+ #: upstream is OpenAI's own API, because Groq, OpenRouter and Together name
83
+ #: OpenAI's models `openai/gpt-oss-120b` and would reject the bare id.
86
84
  ROUTING_PREFIXES: tuple[str, ...] = ("bedrock", "openai", "fake")
87
85
 
88
86
 
@@ -181,7 +179,7 @@ HOSTS: tuple[Host, ...] = (
181
179
  "https://generativelanguage.googleapis.com/v1beta/openai/",
182
180
  "GEMINI_API_KEY",
183
181
  None,
184
- ("gemini-2.5-flash", "gemini-2.0-flash", "gemini-1.5-pro"),
182
+ ("gemini-2.5-flash", "gemini-2.5-flash-lite", "gemini-3.5-flash"),
185
183
  "Also reads GOOGLE_API_KEY. Generous free tier.",
186
184
  ),
187
185
  Host(
@@ -201,7 +199,7 @@ HOSTS: tuple[Host, ...] = (
201
199
  "https://api.groq.com/openai/v1",
202
200
  "GROQ_API_KEY",
203
201
  None,
204
- ("llama-3.3-70b-versatile", "llama-3.1-8b-instant", "gemma2-9b-it"),
202
+ ("openai/gpt-oss-20b", "openai/gpt-oss-120b", "qwen/qwen3.6-27b"),
205
203
  "Very fast, useful free tier.",
206
204
  ),
207
205
  Host(
@@ -131,17 +131,21 @@ def apply(settings: Any) -> str | None:
131
131
  CACHELLM_OPENAI_API_KEY being set means the decision is already made, and
132
132
  detection must not second-guess it.
133
133
 
134
+ "Set" means set anywhere settings are read from, not just exported. A base
135
+ URL written in a .env file used to be replaced by whichever vendor key
136
+ happened to be in the shell, because only the process environment was
137
+ checked. Settings already records which fields came from any source.
138
+
134
139
  Returns a one-line description of what it configured, or None.
135
140
  """
136
- explicit = any(
137
- os.environ.get(name, "").strip()
138
- for name in (
139
- "CACHELLM_DEFAULT_PROVIDER",
140
- "CACHELLM_OPENAI_BASE_URL",
141
- "CACHELLM_OPENAI_API_KEY",
142
- )
141
+ fields = ("default_provider", "openai_base_url", "openai_api_key")
142
+ exported = any(os.environ.get(f"CACHELLM_{name.upper()}", "").strip() for name in fields)
143
+ # An empty `CACHELLM_OPENAI_API_KEY=` line is a placeholder, not a decision.
144
+ written = set(getattr(settings, "model_fields_set", ()))
145
+ configured = any(
146
+ name in written and str(getattr(settings, name, "") or "").strip() for name in fields
143
147
  )
144
- if explicit:
148
+ if exported or configured:
145
149
  return None
146
150
 
147
151
  options = available(include_fake=False)
@@ -11,6 +11,7 @@ from __future__ import annotations
11
11
  import json
12
12
  from collections.abc import AsyncIterator
13
13
  from typing import Any
14
+ from urllib.parse import urlparse
14
15
 
15
16
  import httpx
16
17
  import structlog
@@ -32,12 +33,32 @@ class OpenAICompatProvider(Provider):
32
33
  name = "openai"
33
34
 
34
35
  def __init__(
35
- self, settings: Settings, base_url: str | None = None, api_key: str | None = None
36
+ self,
37
+ settings: Settings,
38
+ base_url: str | None = None,
39
+ api_key: str | None = None,
40
+ transport: httpx.AsyncBaseTransport | None = None,
36
41
  ) -> None:
37
42
  self._settings = settings
38
43
  self._base_url = (base_url or settings.openai_base_url).rstrip("/")
39
44
  self._api_key = api_key if api_key is not None else settings.openai_api_key
45
+ # Injectable so tests exercise the real client construction, including
46
+ # the auth header, against a fake upstream rather than the network.
47
+ self._transport = transport
40
48
  self._client: httpx.AsyncClient | None = None
49
+ self._is_openai_itself = urlparse(self._base_url).hostname == "api.openai.com"
50
+
51
+ def resolve_model(self, model: str) -> str:
52
+ """Drop our `openai/` routing prefix only when the upstream is OpenAI.
53
+
54
+ OpenAI's own API wants `gpt-4o-mini`. Groq, OpenRouter and Together all
55
+ name OpenAI's models `openai/gpt-oss-120b`, so to them the prefix is part
56
+ of the model id. Stripping it there turned Groq's two main chat models
57
+ into 404s, which only showed up against the live API.
58
+ """
59
+ if model.startswith("openai/") and not self._is_openai_itself:
60
+ return model
61
+ return super().resolve_model(model)
41
62
 
42
63
  def _http(self) -> httpx.AsyncClient:
43
64
  if self._client is None:
@@ -48,6 +69,7 @@ class OpenAICompatProvider(Provider):
48
69
  base_url=self._base_url,
49
70
  headers=headers,
50
71
  timeout=httpx.Timeout(self._settings.request_timeout, connect=10.0),
72
+ transport=self._transport,
51
73
  )
52
74
  return self._client
53
75
 
@@ -12,6 +12,7 @@ from __future__ import annotations
12
12
 
13
13
  import os
14
14
  import sys
15
+ import textwrap
15
16
  import time
16
17
  from typing import Any
17
18
 
@@ -73,7 +74,12 @@ def header(stats: dict[str, Any], embedding_model: str = "") -> list[str]:
73
74
  bits.append(f"{MAGENTA}shadow mode{RESET}")
74
75
  if not stats.get("caching_enabled", True):
75
76
  bits.append(f"{AMBER}caching disabled{RESET}")
76
- return ["", " " + f" {DIM}·{RESET} ".join(bits), ""]
77
+ lines = ["", " " + f" {DIM}·{RESET} ".join(bits)]
78
+ # A reachable Redis that was passed over is worth a line of its own:
79
+ # whoever started it expected it to be used.
80
+ if note := stats.get("backend_note"):
81
+ lines += [f" {AMBER}{row}{RESET}" for row in textwrap.wrap(str(note), 74)]
82
+ return [*lines, ""]
77
83
 
78
84
 
79
85
  def summary(stats: dict[str, Any], embedding_model: str = "") -> str: