cachellm-proxy 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/PKG-INFO +3 -2
  2. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/README.md +2 -1
  3. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/pyproject.toml +1 -1
  4. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/pyproject.toml.orig +1 -1
  5. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/__init__.py +1 -1
  6. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/api/routes_admin.py +15 -0
  7. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cli.py +79 -57
  8. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/__main__.py +0 -0
  9. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/api/__init__.py +0 -0
  10. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/api/app.py +0 -0
  11. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/api/auth.py +0 -0
  12. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/api/deps.py +0 -0
  13. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/api/routes_chat.py +0 -0
  14. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/api/sse.py +0 -0
  15. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/__init__.py +0 -0
  16. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/analytics.py +0 -0
  17. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/coalesce.py +0 -0
  18. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/entry.py +0 -0
  19. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/exact_store.py +0 -0
  20. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/keys.py +0 -0
  21. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/memory.py +0 -0
  22. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/policy.py +0 -0
  23. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/redis_client.py +0 -0
  24. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/service.py +0 -0
  25. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/cache/vector_store.py +0 -0
  26. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/embeddings/__init__.py +0 -0
  27. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/embeddings/base.py +0 -0
  28. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/embeddings/fastembed_backend.py +0 -0
  29. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/embeddings/hash_backend.py +0 -0
  30. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/errors.py +0 -0
  31. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/logging_setup.py +0 -0
  32. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/models.py +0 -0
  33. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/observability/__init__.py +0 -0
  34. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/observability/metrics.py +0 -0
  35. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/observability/tracing.py +0 -0
  36. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/pricing.py +0 -0
  37. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/__init__.py +0 -0
  38. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/base.py +0 -0
  39. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/bedrock.py +0 -0
  40. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/catalog.py +0 -0
  41. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/detect.py +0 -0
  42. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/fake.py +0 -0
  43. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/openai_compat.py +0 -0
  44. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/providers/registry.py +0 -0
  45. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/py.typed +0 -0
  46. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/report.py +0 -0
  47. {cachellm_proxy-0.2.0 → cachellm_proxy-0.2.1}/src/cachellm/settings.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cachellm-proxy
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: A drop-in semantic cache for OpenAI-compatible LLM APIs: match requests by meaning, serve repeats instantly, cut spend.
5
5
  Keywords: llm,cache,semantic-cache,openai,bedrock,proxy,vector-search,redis,fastapi,llmops,cost-optimization
6
6
  Author: Adarsh Dwivedi
@@ -53,7 +53,7 @@ Description-Content-Type: text/markdown
53
53
  [![CI](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml/badge.svg)](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml)
54
54
  [![Python 3.11+](https://img.shields.io/badge/python-3.11%20%7C%203.12%20%7C%203.13-blue)](https://www.python.org/)
55
55
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
56
- [![Tests](https://img.shields.io/badge/tests-261%20passing-brightgreen)](tests/)
56
+ [![Tests](https://img.shields.io/badge/tests-265%20passing-brightgreen)](tests/)
57
57
  [![PyPI](https://img.shields.io/pypi/v/cachellm-proxy)](https://pypi.org/project/cachellm-proxy/)
58
58
  [![Hit rate](https://img.shields.io/badge/hit%20rate-77%25%20on%20Bedrock-orange)](docs/evaluation.md)
59
59
 
@@ -447,6 +447,7 @@ Clients can steer per request with `X-Cache-Control`:
447
447
  | `GET /admin/providers` | Every host, its base URL, its pip extra, example model ids |
448
448
  | `GET /admin/route/{model}` | Where one model name would go, and why |
449
449
  | `POST /admin/invalidate` | Drop by `namespace`, by `model`, or `all` |
450
+ | `GET /admin/requests` | Recent request log: what the cache did with each one |
450
451
  | `GET /admin/near-misses` | Recent lookups that landed just below threshold |
451
452
  | `GET /admin/near-miss-histogram` | Cumulative view of what a lower threshold would buy |
452
453
  | `GET /admin/entries` | Inspect what is stored |
@@ -3,7 +3,7 @@
3
3
  [![CI](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml/badge.svg)](https://github.com/adarshcod30/CacheLLM/actions/workflows/ci.yml)
4
4
  [![Python 3.11+](https://img.shields.io/badge/python-3.11%20%7C%203.12%20%7C%203.13-blue)](https://www.python.org/)
5
5
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
6
- [![Tests](https://img.shields.io/badge/tests-261%20passing-brightgreen)](tests/)
6
+ [![Tests](https://img.shields.io/badge/tests-265%20passing-brightgreen)](tests/)
7
7
  [![PyPI](https://img.shields.io/pypi/v/cachellm-proxy)](https://pypi.org/project/cachellm-proxy/)
8
8
  [![Hit rate](https://img.shields.io/badge/hit%20rate-77%25%20on%20Bedrock-orange)](docs/evaluation.md)
9
9
 
@@ -397,6 +397,7 @@ Clients can steer per request with `X-Cache-Control`:
397
397
  | `GET /admin/providers` | Every host, its base URL, its pip extra, example model ids |
398
398
  | `GET /admin/route/{model}` | Where one model name would go, and why |
399
399
  | `POST /admin/invalidate` | Drop by `namespace`, by `model`, or `all` |
400
+ | `GET /admin/requests` | Recent request log: what the cache did with each one |
400
401
  | `GET /admin/near-misses` | Recent lookups that landed just below threshold |
401
402
  | `GET /admin/near-miss-histogram` | Cumulative view of what a lower threshold would buy |
402
403
  | `GET /admin/entries` | Inspect what is stored |
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cachellm-proxy"
3
- version = "0.2.0"
3
+ version = "0.2.1"
4
4
  description = "A drop-in semantic cache for OpenAI-compatible LLM APIs: match requests by meaning, serve repeats instantly, cut spend."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cachellm-proxy"
3
- version = "0.2.0"
3
+ version = "0.2.1"
4
4
  description = "A drop-in semantic cache for OpenAI-compatible LLM APIs: match requests by meaning, serve repeats instantly, cut spend."
5
5
  readme = "README.md"
6
6
  authors = [
@@ -2,7 +2,7 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
- __version__ = "0.2.0"
5
+ __version__ = "0.2.1"
6
6
 
7
7
  __all__ = ["__version__", "main"]
8
8
 
@@ -212,6 +212,21 @@ async def invalidate(request: Request, body: InvalidateRequest) -> dict[str, Any
212
212
  }
213
213
 
214
214
 
215
+ @router.get("/requests")
216
+ async def requests_log(request: Request, limit: int = Query(50, ge=1, le=500)) -> dict[str, Any]:
217
+ """The recent request log: what the cache did with each one.
218
+
219
+ This is what `cachellm stats` reads. It has to come over HTTP because with
220
+ the in-memory backend the cache lives inside the serving process, and a CLI
221
+ building its own state would report on an empty cache of its own.
222
+ """
223
+ state = _require_cache(request)
224
+ if state.settings.require_auth_for_admin:
225
+ verify(request, state.settings.client_keys)
226
+ rows = await state.analytics.recent_requests(limit=limit)
227
+ return {"count": len(rows), "requests": rows}
228
+
229
+
215
230
  @router.get("/near-misses")
216
231
  async def near_misses(request: Request, limit: int = Query(50, ge=1, le=500)) -> dict[str, Any]:
217
232
  state = _require_cache(request)
@@ -3,10 +3,10 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import asyncio
6
- import contextlib
7
6
  import json
7
+ import time
8
8
  from pathlib import Path
9
- from typing import Annotated
9
+ from typing import Annotated, Any
10
10
 
11
11
  import typer
12
12
 
@@ -57,75 +57,97 @@ def config() -> None:
57
57
  typer.echo(json.dumps(data, indent=2, default=str))
58
58
 
59
59
 
60
+ def _proxy_url(url: str) -> str:
61
+ settings = get_settings()
62
+ return (url or f"http://127.0.0.1:{settings.port}").rstrip("/")
63
+
64
+
65
+ def _fetch(url: str, path: str, params: dict[str, Any] | None = None) -> Any:
66
+ """Read from a running proxy, or None when nothing is listening.
67
+
68
+ Talking to the server rather than building our own state is not an
69
+ optimisation: with the in-memory backend the cache lives inside the serving
70
+ process, so a CLI that built its own would report on an empty cache of its
71
+ own and always say "no requests yet".
72
+ """
73
+ import httpx
74
+
75
+ settings = get_settings()
76
+ headers = {}
77
+ if keys := sorted(settings.client_keys):
78
+ headers["Authorization"] = f"Bearer {keys[0]}"
79
+ try:
80
+ response = httpx.get(f"{url}{path}", params=params, headers=headers, timeout=5.0)
81
+ response.raise_for_status()
82
+ return response.json()
83
+ except Exception:
84
+ return None
85
+
86
+
87
+ def _not_running(url: str) -> None:
88
+ typer.secho(f"No proxy answering at {url}.", fg="yellow")
89
+ typer.echo(" Start one with `cachellm serve`, or pass --url if it is elsewhere.")
90
+ typer.echo(
91
+ f" {report.DIM}The cache lives inside the serving process unless you "
92
+ f"use the redis backend, so there is nothing to read without it."
93
+ f"{report.RESET}"
94
+ )
95
+
96
+
60
97
  @app.command()
61
98
  def stats(
99
+ url: Annotated[str, typer.Option(help="Proxy to read from.")] = "",
62
100
  limit: Annotated[int, typer.Option(help="How many recent requests to show.")] = 15,
63
101
  json_out: Annotated[bool, typer.Option("--json", help="Raw JSON instead.")] = False,
64
102
  ) -> None:
65
103
  """Show what the cache has been doing: hit rate, savings, latency, recent requests."""
66
-
67
- async def run() -> None:
68
- from cachellm.api.deps import build_state, shutdown_state
69
-
70
- settings = get_settings()
71
- state = await build_state(settings)
72
- try:
73
- if state.cache is None or state.analytics is None:
74
- typer.secho(f"cache unavailable: {state.degraded_reason}", fg="red")
75
- raise typer.Exit(1)
76
- data = await state.cache.stats()
77
- data["backend"] = state.backend
78
- data["cache_available"] = True
79
- data["caching_enabled"] = settings.enabled
80
- recent = await state.analytics.recent_requests(limit=limit)
81
- if json_out:
82
- typer.echo(json.dumps({"stats": data, "recent": recent}, indent=2))
83
- return
84
- typer.echo(report.summary(data, settings.embedding_model))
85
- typer.echo(report.request_table(recent, limit=limit))
86
- finally:
87
- await shutdown_state(state)
88
-
89
- asyncio.run(run())
104
+ base = _proxy_url(url)
105
+ data = _fetch(base, "/admin/stats")
106
+ if data is None:
107
+ _not_running(base)
108
+ raise typer.Exit(1)
109
+ log = _fetch(base, "/admin/requests", {"limit": limit}) or {}
110
+ recent = log.get("requests", [])
111
+ if json_out:
112
+ typer.echo(json.dumps({"stats": data, "recent": recent}, indent=2))
113
+ return
114
+ typer.echo(report.summary(data, get_settings().embedding_model))
115
+ typer.echo(report.request_table(recent, limit=limit))
90
116
 
91
117
 
92
118
  @app.command()
93
119
  def watch(
120
+ url: Annotated[str, typer.Option(help="Proxy to follow.")] = "",
94
121
  interval: Annotated[float, typer.Option(help="Seconds between refreshes.")] = 2.0,
95
122
  ) -> None:
96
123
  """Follow requests as they happen, like `tail -f` for the cache."""
97
-
98
- async def run() -> None:
99
- from cachellm.api.deps import build_state, shutdown_state
100
-
101
- settings = get_settings()
102
- state = await build_state(settings)
103
- if state.cache is None or state.analytics is None:
104
- typer.secho(f"cache unavailable: {state.degraded_reason}", fg="red")
105
- raise typer.Exit(1)
106
- seen: set[tuple[float, str]] = set()
107
- typer.echo(f"watching {state.backend} backend, ctrl-c to stop\n")
108
- try:
109
- while True:
110
- for record in reversed(await state.analytics.recent_requests(limit=50)):
111
- marker = (record.get("at", 0.0), record.get("prompt", ""))
112
- if marker in seen:
113
- continue
114
- seen.add(marker)
124
+ base = _proxy_url(url)
125
+ if _fetch(base, "/admin/stats") is None:
126
+ _not_running(base)
127
+ raise typer.Exit(1)
128
+
129
+ seen: set[tuple[float, str]] = set()
130
+ first = True
131
+ typer.echo(f"following {base}, ctrl-c to stop\n")
132
+ try:
133
+ while True:
134
+ payload = _fetch(base, "/admin/requests", {"limit": 50}) or {}
135
+ for record in reversed(payload.get("requests", [])):
136
+ marker = (record.get("at", 0.0), record.get("prompt", ""))
137
+ if marker in seen:
138
+ continue
139
+ seen.add(marker)
140
+ if not first:
115
141
  typer.echo(report.request_line(record))
116
- if len(seen) > 5_000:
117
- seen.clear()
118
- await asyncio.sleep(interval)
119
- except (KeyboardInterrupt, asyncio.CancelledError):
120
- data = await state.cache.stats()
121
- data["backend"] = state.backend
122
- data["cache_available"] = True
123
- typer.echo(report.summary(data, settings.embedding_model))
124
- finally:
125
- await shutdown_state(state)
126
-
127
- with contextlib.suppress(KeyboardInterrupt):
128
- asyncio.run(run())
142
+ first = False
143
+ if len(seen) > 5_000:
144
+ seen.clear()
145
+ time.sleep(interval)
146
+ except KeyboardInterrupt:
147
+ data = _fetch(base, "/admin/stats")
148
+ if data:
149
+ typer.echo("")
150
+ typer.echo(report.summary(data, get_settings().embedding_model))
129
151
 
130
152
 
131
153
  @app.command()