cachellm-proxy 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/PKG-INFO +3 -2
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/README.md +2 -1
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/pyproject.toml +1 -1
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/pyproject.toml.orig +1 -1
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/__init__.py +1 -1
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/api/routes_admin.py +15 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cli.py +79 -57
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/__main__.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/api/__init__.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/api/app.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/api/auth.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/api/deps.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/api/routes_chat.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/api/sse.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/__init__.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/analytics.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/coalesce.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/entry.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/exact_store.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/keys.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/memory.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/policy.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/redis_client.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/service.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/vector_store.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/embeddings/__init__.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/embeddings/base.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/embeddings/fastembed_backend.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/embeddings/hash_backend.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/errors.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/logging_setup.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/models.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/observability/__init__.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/observability/metrics.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/observability/tracing.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/pricing.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/__init__.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/base.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/bedrock.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/catalog.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/detect.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/fake.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/openai_compat.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/registry.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/py.typed +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/report.py +0 -0
- {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/settings.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cachellm-proxy
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: A drop-in semantic cache for OpenAI-compatible LLM APIs: match requests by meaning, serve repeats instantly, cut spend.
|
|
5
5
|
Keywords: llm,cache,semantic-cache,openai,bedrock,proxy,vector-search,redis,fastapi,llmops,cost-optimization
|
|
6
6
|
Author: Adarsh Dwivedi
|
|
@@ -53,7 +53,7 @@ Description-Content-Type: text/markdown
|
|
|
53
53
|
[](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml)
|
|
54
54
|
[](https://www.python.org/)
|
|
55
55
|
[](LICENSE)
|
|
56
|
-
[](tests/)
|
|
57
57
|
[](https://pypi.org/project/cachellm-proxy/)
|
|
58
58
|
[](docs/evaluation.md)
|
|
59
59
|
|
|
@@ -447,6 +447,7 @@ Clients can steer per request with `X-Cache-Control`:
|
|
|
447
447
|
| `GET /admin/providers` | Every host, its base URL, its pip extra, example model ids |
|
|
448
448
|
| `GET /admin/route/{model}` | Where one model name would go, and why |
|
|
449
449
|
| `POST /admin/invalidate` | Drop by `namespace`, by `model`, or `all` |
|
|
450
|
+
| `GET /admin/requests` | Recent request log: what the cache did with each one |
|
|
450
451
|
| `GET /admin/near-misses` | Recent lookups that landed just below threshold |
|
|
451
452
|
| `GET /admin/near-miss-histogram` | Cumulative view of what a lower threshold would buy |
|
|
452
453
|
| `GET /admin/entries` | Inspect what is stored |
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
[](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml)
|
|
4
4
|
[](https://www.python.org/)
|
|
5
5
|
[](LICENSE)
|
|
6
|
-
[](tests/)
|
|
7
7
|
[](https://pypi.org/project/cachellm-proxy/)
|
|
8
8
|
[](docs/evaluation.md)
|
|
9
9
|
|
|
@@ -397,6 +397,7 @@ Clients can steer per request with `X-Cache-Control`:
|
|
|
397
397
|
| `GET /admin/providers` | Every host, its base URL, its pip extra, example model ids |
|
|
398
398
|
| `GET /admin/route/{model}` | Where one model name would go, and why |
|
|
399
399
|
| `POST /admin/invalidate` | Drop by `namespace`, by `model`, or `all` |
|
|
400
|
+
| `GET /admin/requests` | Recent request log: what the cache did with each one |
|
|
400
401
|
| `GET /admin/near-misses` | Recent lookups that landed just below threshold |
|
|
401
402
|
| `GET /admin/near-miss-histogram` | Cumulative view of what a lower threshold would buy |
|
|
402
403
|
| `GET /admin/entries` | Inspect what is stored |
|
|
@@ -212,6 +212,21 @@ async def invalidate(request: Request, body: InvalidateRequest) -> dict[str, Any
|
|
|
212
212
|
}
|
|
213
213
|
|
|
214
214
|
|
|
215
|
+
@router.get("/requests")
|
|
216
|
+
async def requests_log(request: Request, limit: int = Query(50, ge=1, le=500)) -> dict[str, Any]:
|
|
217
|
+
"""The recent request log: what the cache did with each one.
|
|
218
|
+
|
|
219
|
+
This is what `cachellm stats` reads. It has to come over HTTP because with
|
|
220
|
+
the in-memory backend the cache lives inside the serving process, and a CLI
|
|
221
|
+
building its own state would report on an empty cache of its own.
|
|
222
|
+
"""
|
|
223
|
+
state = _require_cache(request)
|
|
224
|
+
if state.settings.require_auth_for_admin:
|
|
225
|
+
verify(request, state.settings.client_keys)
|
|
226
|
+
rows = await state.analytics.recent_requests(limit=limit)
|
|
227
|
+
return {"count": len(rows), "requests": rows}
|
|
228
|
+
|
|
229
|
+
|
|
215
230
|
@router.get("/near-misses")
|
|
216
231
|
async def near_misses(request: Request, limit: int = Query(50, ge=1, le=500)) -> dict[str, Any]:
|
|
217
232
|
state = _require_cache(request)
|
|
@@ -3,10 +3,10 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import asyncio
|
|
6
|
-
import contextlib
|
|
7
6
|
import json
|
|
7
|
+
import time
|
|
8
8
|
from pathlib import Path
|
|
9
|
-
from typing import Annotated
|
|
9
|
+
from typing import Annotated, Any
|
|
10
10
|
|
|
11
11
|
import typer
|
|
12
12
|
|
|
@@ -57,75 +57,97 @@ def config() -> None:
|
|
|
57
57
|
typer.echo(json.dumps(data, indent=2, default=str))
|
|
58
58
|
|
|
59
59
|
|
|
60
|
+
def _proxy_url(url: str) -> str:
|
|
61
|
+
settings = get_settings()
|
|
62
|
+
return (url or f"http://127.0.0.1:{settings.port}").rstrip("/")
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _fetch(url: str, path: str, params: dict[str, Any] | None = None) -> Any:
|
|
66
|
+
"""Read from a running proxy, or None when nothing is listening.
|
|
67
|
+
|
|
68
|
+
Talking to the server rather than building our own state is not an
|
|
69
|
+
optimisation: with the in-memory backend the cache lives inside the serving
|
|
70
|
+
process, so a CLI that built its own would report on an empty cache of its
|
|
71
|
+
own and always say "no requests yet".
|
|
72
|
+
"""
|
|
73
|
+
import httpx
|
|
74
|
+
|
|
75
|
+
settings = get_settings()
|
|
76
|
+
headers = {}
|
|
77
|
+
if keys := sorted(settings.client_keys):
|
|
78
|
+
headers["Authorization"] = f"Bearer {keys[0]}"
|
|
79
|
+
try:
|
|
80
|
+
response = httpx.get(f"{url}{path}", params=params, headers=headers, timeout=5.0)
|
|
81
|
+
response.raise_for_status()
|
|
82
|
+
return response.json()
|
|
83
|
+
except Exception:
|
|
84
|
+
return None
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _not_running(url: str) -> None:
|
|
88
|
+
typer.secho(f"No proxy answering at {url}.", fg="yellow")
|
|
89
|
+
typer.echo(" Start one with `cachellm serve`, or pass --url if it is elsewhere.")
|
|
90
|
+
typer.echo(
|
|
91
|
+
f" {report.DIM}The cache lives inside the serving process unless you "
|
|
92
|
+
f"use the redis backend, so there is nothing to read without it."
|
|
93
|
+
f"{report.RESET}"
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
|
|
60
97
|
@app.command()
|
|
61
98
|
def stats(
|
|
99
|
+
url: Annotated[str, typer.Option(help="Proxy to read from.")] = "",
|
|
62
100
|
limit: Annotated[int, typer.Option(help="How many recent requests to show.")] = 15,
|
|
63
101
|
json_out: Annotated[bool, typer.Option("--json", help="Raw JSON instead.")] = False,
|
|
64
102
|
) -> None:
|
|
65
103
|
"""Show what the cache has been doing: hit rate, savings, latency, recent requests."""
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
data["cache_available"] = True
|
|
79
|
-
data["caching_enabled"] = settings.enabled
|
|
80
|
-
recent = await state.analytics.recent_requests(limit=limit)
|
|
81
|
-
if json_out:
|
|
82
|
-
typer.echo(json.dumps({"stats": data, "recent": recent}, indent=2))
|
|
83
|
-
return
|
|
84
|
-
typer.echo(report.summary(data, settings.embedding_model))
|
|
85
|
-
typer.echo(report.request_table(recent, limit=limit))
|
|
86
|
-
finally:
|
|
87
|
-
await shutdown_state(state)
|
|
88
|
-
|
|
89
|
-
asyncio.run(run())
|
|
104
|
+
base = _proxy_url(url)
|
|
105
|
+
data = _fetch(base, "/admin/stats")
|
|
106
|
+
if data is None:
|
|
107
|
+
_not_running(base)
|
|
108
|
+
raise typer.Exit(1)
|
|
109
|
+
log = _fetch(base, "/admin/requests", {"limit": limit}) or {}
|
|
110
|
+
recent = log.get("requests", [])
|
|
111
|
+
if json_out:
|
|
112
|
+
typer.echo(json.dumps({"stats": data, "recent": recent}, indent=2))
|
|
113
|
+
return
|
|
114
|
+
typer.echo(report.summary(data, get_settings().embedding_model))
|
|
115
|
+
typer.echo(report.request_table(recent, limit=limit))
|
|
90
116
|
|
|
91
117
|
|
|
92
118
|
@app.command()
|
|
93
119
|
def watch(
|
|
120
|
+
url: Annotated[str, typer.Option(help="Proxy to follow.")] = "",
|
|
94
121
|
interval: Annotated[float, typer.Option(help="Seconds between refreshes.")] = 2.0,
|
|
95
122
|
) -> None:
|
|
96
123
|
"""Follow requests as they happen, like `tail -f` for the cache."""
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
seen.add(marker)
|
|
124
|
+
base = _proxy_url(url)
|
|
125
|
+
if _fetch(base, "/admin/stats") is None:
|
|
126
|
+
_not_running(base)
|
|
127
|
+
raise typer.Exit(1)
|
|
128
|
+
|
|
129
|
+
seen: set[tuple[float, str]] = set()
|
|
130
|
+
first = True
|
|
131
|
+
typer.echo(f"following {base}, ctrl-c to stop\n")
|
|
132
|
+
try:
|
|
133
|
+
while True:
|
|
134
|
+
payload = _fetch(base, "/admin/requests", {"limit": 50}) or {}
|
|
135
|
+
for record in reversed(payload.get("requests", [])):
|
|
136
|
+
marker = (record.get("at", 0.0), record.get("prompt", ""))
|
|
137
|
+
if marker in seen:
|
|
138
|
+
continue
|
|
139
|
+
seen.add(marker)
|
|
140
|
+
if not first:
|
|
115
141
|
typer.echo(report.request_line(record))
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
typer.echo(
|
|
124
|
-
|
|
125
|
-
await shutdown_state(state)
|
|
126
|
-
|
|
127
|
-
with contextlib.suppress(KeyboardInterrupt):
|
|
128
|
-
asyncio.run(run())
|
|
142
|
+
first = False
|
|
143
|
+
if len(seen) > 5_000:
|
|
144
|
+
seen.clear()
|
|
145
|
+
time.sleep(interval)
|
|
146
|
+
except KeyboardInterrupt:
|
|
147
|
+
data = _fetch(base, "/admin/stats")
|
|
148
|
+
if data:
|
|
149
|
+
typer.echo("")
|
|
150
|
+
typer.echo(report.summary(data, get_settings().embedding_model))
|
|
129
151
|
|
|
130
152
|
|
|
131
153
|
@app.command()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|