reqly 0.4.0__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {reqly-0.4.0 → reqly-0.5.0}/PKG-INFO +38 -2
- {reqly-0.4.0 → reqly-0.5.0}/README.md +37 -1
- {reqly-0.4.0 → reqly-0.5.0}/pyproject.toml +1 -1
- {reqly-0.4.0 → reqly-0.5.0}/reqly/__init__.py +79 -2
- {reqly-0.4.0 → reqly-0.5.0}/reqly/core/capture.py +11 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly/core/client.py +26 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly/core/config.py +17 -0
- reqly-0.5.0/reqly/core/request_context.py +167 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly/integrations/django.py +22 -6
- {reqly-0.4.0 → reqly-0.5.0}/reqly/integrations/fastapi.py +13 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly/integrations/flask.py +11 -0
- reqly-0.5.0/reqly/integrations/wsgi.py +157 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly.egg-info/PKG-INFO +38 -2
- {reqly-0.4.0 → reqly-0.5.0}/reqly.egg-info/SOURCES.txt +3 -0
- reqly-0.5.0/tests/test_consumers_llm_wsgi.py +239 -0
- {reqly-0.4.0 → reqly-0.5.0}/tests/test_more_frameworks.py +20 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly/core/__init__.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly/core/buffer.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly/core/openapi_push.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly/core/sampling.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly/core/shipper.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly/integrations/__init__.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly/integrations/litestar.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly/integrations/starlette.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly.egg-info/dependency_links.txt +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly.egg-info/requires.txt +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/reqly.egg-info/top_level.txt +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/setup.cfg +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/tests/test_buffer.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/tests/test_capture.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/tests/test_fastapi_integration.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/tests/test_flask_integration.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/tests/test_instrument.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/tests/test_openapi_push.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/tests/test_sampling.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/tests/test_shipper.py +0 -0
- {reqly-0.4.0 → reqly-0.5.0}/tests/test_v2_fields.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: reqly
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: Self-hosted API monitoring for FastAPI, Flask, Django, Starlette and Litestar in two lines: latency percentiles, error rates and release tracking per route, with deploy-aware alerts and weekly AI anomaly reports.
|
|
5
5
|
Author: Tanish Poddar
|
|
6
6
|
License-Expression: GPL-3.0-or-later
|
|
@@ -110,6 +110,34 @@ MIDDLEWARE = [
|
|
|
110
110
|
REQLY = {"service_name": "checkout-api", "api_key": "your-ingest-key"} # optional
|
|
111
111
|
```
|
|
112
112
|
|
|
113
|
+
**Any other WSGI or ASGI app** (Bottle, Pyramid, Falcon, CherryPy, a bare ASGI app) — wrap it
|
|
114
|
+
and serve the result. A `route_resolver` returns the route template, because only the framework
|
|
115
|
+
knows it; without one every request is recorded as `__unmatched__`, never as a raw path:
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
app = reqly.instrument_wsgi(
|
|
119
|
+
app, service_name="checkout-api",
|
|
120
|
+
route_resolver=lambda environ: environ["bottle.route"].rule, # Bottle
|
|
121
|
+
)
|
|
122
|
+
# Pyramid: environ["bfg.routes.route"].pattern; ASGI: reqly.instrument_asgi(app, route_resolver=...)
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
**Who is calling** — tag each request with its API consumer. The id is hashed (HMAC-SHA256
|
|
126
|
+
with your secret salt) before it leaves the app:
|
|
127
|
+
|
|
128
|
+
```python
|
|
129
|
+
reqly.instrument(app, consumer_header="X-API-Key", consumer_salt=os.environ["REQLY_CONSUMER_SALT"])
|
|
130
|
+
# or any logic: consumer=lambda info: info.headers.get("x-tenant-id")
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
**LLM cost per route** — record token usage where you call a model; the collector prices it:
|
|
134
|
+
|
|
135
|
+
```python
|
|
136
|
+
completion = client.chat.completions.create(model="gpt-4o-mini", messages=messages)
|
|
137
|
+
reqly.record_llm_response(completion) # OpenAI / Anthropic responses, or:
|
|
138
|
+
reqly.record_llm_usage("gpt-4o-mini", input_tokens=1200, output_tokens=240)
|
|
139
|
+
```
|
|
140
|
+
|
|
113
141
|
`instrument()` detects the framework by itself — no decorators, no middleware to wire up.
|
|
114
142
|
Routes are recorded as templates in one style across frameworks: Django's
|
|
115
143
|
`users/<int:pk>/` and DRF's `^users/(?P<pk>[^/.]+)/$` both become `/users/{pk}/`.
|
|
@@ -131,6 +159,9 @@ In the Reqly dashboard and collector:
|
|
|
131
159
|
- **Deploy markers and per-release health** — each release's error rate and p95
|
|
132
160
|
- **Hourly alerts** to Slack, Discord or a webhook when a route breaks from its usual
|
|
133
161
|
weekday-hour pattern, with **root-cause hints**
|
|
162
|
+
- **Top consumers** — requests and error rate per API client, and which clients an incident
|
|
163
|
+
hit (in the alert itself)
|
|
164
|
+
- **LLM cost per route** — tokens and estimated spend per route and model
|
|
134
165
|
- **API surface vs your OpenAPI spec** — undocumented endpoints that get traffic, documented
|
|
135
166
|
ones nobody calls, and deprecated ones still in use (`push_openapi=True`)
|
|
136
167
|
- **Weekly AI report** — statistics find the anomalies, Groq (gpt-oss-120b) writes the
|
|
@@ -166,6 +197,10 @@ Resolution order: **argument → environment variable → default**.
|
|
|
166
197
|
| `max_batch_size` | `REQLY_MAX_BATCH_SIZE` | `200` |
|
|
167
198
|
| `max_queue_size` | `REQLY_MAX_QUEUE_SIZE` | `2000` |
|
|
168
199
|
| `ignore_routes` | `REQLY_IGNORE_ROUTES` (comma-separated) | `/health,/metrics` |
|
|
200
|
+
| `consumer_header` | `REQLY_CONSUMER_HEADER` | `None` — header that identifies the caller, e.g. `X-API-Key` |
|
|
201
|
+
| `consumer` | — | `None` — `callable(RequestInfo) -> str \| None`, instead of a header |
|
|
202
|
+
| `consumer_salt` | `REQLY_CONSUMER_SALT` | `None` — secret for hashing consumer ids (set it) |
|
|
203
|
+
| `hash_consumer` | `REQLY_HASH_CONSUMER` | `True` — `False` sends ids unhashed (only for non-secret ids) |
|
|
169
204
|
| `push_openapi` | `REQLY_PUSH_OPENAPI` | `False` — upload the app's OpenAPI spec (FastAPI, Litestar) on the first request |
|
|
170
205
|
| `capture_request_body` | `REQLY_CAPTURE_REQUEST_BODY` | `False` (not implemented yet) |
|
|
171
206
|
|
|
@@ -204,7 +239,8 @@ are shipped instead of silently queuing forever.
|
|
|
204
239
|
| Litestar | 2.0+ |
|
|
205
240
|
| Flask | 2.3+ |
|
|
206
241
|
| Django | 4.2+, sync and async views; DRF and Django Ninja |
|
|
207
|
-
|
|
|
242
|
+
| Other WSGI / ASGI | any, with `instrument_wsgi()` / `instrument_asgi()` and a `route_resolver` |
|
|
243
|
+
| Collector | any version; `release`, `environment` and body sizes are stored by collector 0.3.0+ and ignored by older ones; `push_openapi` needs 0.7.0+; consumer and LLM views need 0.8.0+ |
|
|
208
244
|
|
|
209
245
|
## Self-hosting the collector
|
|
210
246
|
|
|
@@ -63,6 +63,34 @@ MIDDLEWARE = [
|
|
|
63
63
|
REQLY = {"service_name": "checkout-api", "api_key": "your-ingest-key"} # optional
|
|
64
64
|
```
|
|
65
65
|
|
|
66
|
+
**Any other WSGI or ASGI app** (Bottle, Pyramid, Falcon, CherryPy, a bare ASGI app) — wrap it
|
|
67
|
+
and serve the result. A `route_resolver` returns the route template, because only the framework
|
|
68
|
+
knows it; without one every request is recorded as `__unmatched__`, never as a raw path:
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
app = reqly.instrument_wsgi(
|
|
72
|
+
app, service_name="checkout-api",
|
|
73
|
+
route_resolver=lambda environ: environ["bottle.route"].rule, # Bottle
|
|
74
|
+
)
|
|
75
|
+
# Pyramid: environ["bfg.routes.route"].pattern; ASGI: reqly.instrument_asgi(app, route_resolver=...)
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
**Who is calling** — tag each request with its API consumer. The id is hashed (HMAC-SHA256
|
|
79
|
+
with your secret salt) before it leaves the app:
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
reqly.instrument(app, consumer_header="X-API-Key", consumer_salt=os.environ["REQLY_CONSUMER_SALT"])
|
|
83
|
+
# or any logic: consumer=lambda info: info.headers.get("x-tenant-id")
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
**LLM cost per route** — record token usage where you call a model; the collector prices it:
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
completion = client.chat.completions.create(model="gpt-4o-mini", messages=messages)
|
|
90
|
+
reqly.record_llm_response(completion) # OpenAI / Anthropic responses, or:
|
|
91
|
+
reqly.record_llm_usage("gpt-4o-mini", input_tokens=1200, output_tokens=240)
|
|
92
|
+
```
|
|
93
|
+
|
|
66
94
|
`instrument()` detects the framework by itself — no decorators, no middleware to wire up.
|
|
67
95
|
Routes are recorded as templates in one style across frameworks: Django's
|
|
68
96
|
`users/<int:pk>/` and DRF's `^users/(?P<pk>[^/.]+)/$` both become `/users/{pk}/`.
|
|
@@ -84,6 +112,9 @@ In the Reqly dashboard and collector:
|
|
|
84
112
|
- **Deploy markers and per-release health** — each release's error rate and p95
|
|
85
113
|
- **Hourly alerts** to Slack, Discord or a webhook when a route breaks from its usual
|
|
86
114
|
weekday-hour pattern, with **root-cause hints**
|
|
115
|
+
- **Top consumers** — requests and error rate per API client, and which clients an incident
|
|
116
|
+
hit (in the alert itself)
|
|
117
|
+
- **LLM cost per route** — tokens and estimated spend per route and model
|
|
87
118
|
- **API surface vs your OpenAPI spec** — undocumented endpoints that get traffic, documented
|
|
88
119
|
ones nobody calls, and deprecated ones still in use (`push_openapi=True`)
|
|
89
120
|
- **Weekly AI report** — statistics find the anomalies, Groq (gpt-oss-120b) writes the
|
|
@@ -119,6 +150,10 @@ Resolution order: **argument → environment variable → default**.
|
|
|
119
150
|
| `max_batch_size` | `REQLY_MAX_BATCH_SIZE` | `200` |
|
|
120
151
|
| `max_queue_size` | `REQLY_MAX_QUEUE_SIZE` | `2000` |
|
|
121
152
|
| `ignore_routes` | `REQLY_IGNORE_ROUTES` (comma-separated) | `/health,/metrics` |
|
|
153
|
+
| `consumer_header` | `REQLY_CONSUMER_HEADER` | `None` — header that identifies the caller, e.g. `X-API-Key` |
|
|
154
|
+
| `consumer` | — | `None` — `callable(RequestInfo) -> str \| None`, instead of a header |
|
|
155
|
+
| `consumer_salt` | `REQLY_CONSUMER_SALT` | `None` — secret for hashing consumer ids (set it) |
|
|
156
|
+
| `hash_consumer` | `REQLY_HASH_CONSUMER` | `True` — `False` sends ids unhashed (only for non-secret ids) |
|
|
122
157
|
| `push_openapi` | `REQLY_PUSH_OPENAPI` | `False` — upload the app's OpenAPI spec (FastAPI, Litestar) on the first request |
|
|
123
158
|
| `capture_request_body` | `REQLY_CAPTURE_REQUEST_BODY` | `False` (not implemented yet) |
|
|
124
159
|
|
|
@@ -157,7 +192,8 @@ are shipped instead of silently queuing forever.
|
|
|
157
192
|
| Litestar | 2.0+ |
|
|
158
193
|
| Flask | 2.3+ |
|
|
159
194
|
| Django | 4.2+, sync and async views; DRF and Django Ninja |
|
|
160
|
-
|
|
|
195
|
+
| Other WSGI / ASGI | any, with `instrument_wsgi()` / `instrument_asgi()` and a `route_resolver` |
|
|
196
|
+
| Collector | any version; `release`, `environment` and body sizes are stored by collector 0.3.0+ and ignored by older ones; `push_openapi` needs 0.7.0+; consumer and LLM views need 0.8.0+ |
|
|
161
197
|
|
|
162
198
|
## Self-hosting the collector
|
|
163
199
|
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "reqly"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.5.0"
|
|
8
8
|
description = "Self-hosted API monitoring for FastAPI, Flask, Django, Starlette and Litestar in two lines: latency percentiles, error rates and release tracking per route, with deploy-aware alerts and weekly AI anomaly reports."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.9"
|
|
@@ -4,10 +4,18 @@ import logging
|
|
|
4
4
|
|
|
5
5
|
from .core.client import ReqlyClient
|
|
6
6
|
from .core.config import Config, _get_sdk_version
|
|
7
|
+
from .core.request_context import RequestInfo, record_llm_response, record_llm_usage
|
|
7
8
|
|
|
8
9
|
__version__ = _get_sdk_version()
|
|
9
10
|
|
|
10
|
-
__all__ = [
|
|
11
|
+
__all__ = [
|
|
12
|
+
"instrument",
|
|
13
|
+
"instrument_asgi",
|
|
14
|
+
"instrument_wsgi",
|
|
15
|
+
"record_llm_usage",
|
|
16
|
+
"record_llm_response",
|
|
17
|
+
"RequestInfo",
|
|
18
|
+
]
|
|
11
19
|
|
|
12
20
|
logger = logging.getLogger("reqly")
|
|
13
21
|
|
|
@@ -34,7 +42,8 @@ def _detect_framework(app) -> str:
|
|
|
34
42
|
raise TypeError(
|
|
35
43
|
"reqly.instrument(): could not detect framework for app of type "
|
|
36
44
|
f"{type(app)!r}. Supported: FastAPI, Starlette, Litestar, Flask "
|
|
37
|
-
"(Django: add reqly.integrations.django.ReqlyMiddleware to MIDDLEWARE
|
|
45
|
+
"(Django: add reqly.integrations.django.ReqlyMiddleware to MIDDLEWARE; any other "
|
|
46
|
+
"WSGI/ASGI app: reqly.instrument_wsgi() / reqly.instrument_asgi())."
|
|
38
47
|
)
|
|
39
48
|
|
|
40
49
|
|
|
@@ -66,6 +75,10 @@ def instrument(
|
|
|
66
75
|
release: str | None = None,
|
|
67
76
|
environment: str | None = None,
|
|
68
77
|
push_openapi: bool | None = None,
|
|
78
|
+
consumer_header: str | None = None,
|
|
79
|
+
consumer=None,
|
|
80
|
+
consumer_salt: str | None = None,
|
|
81
|
+
hash_consumer: bool | None = None,
|
|
69
82
|
) -> ReqlyClient | None:
|
|
70
83
|
"""Instrument a FastAPI, Starlette, Litestar or Flask app with one line.
|
|
71
84
|
(Django: add ``reqly.integrations.django.ReqlyMiddleware`` to MIDDLEWARE.)
|
|
@@ -74,6 +87,11 @@ def instrument(
|
|
|
74
87
|
environment variable (REQLY_*) > default. See core.config.Config
|
|
75
88
|
for the full list of environment variables.
|
|
76
89
|
|
|
90
|
+
Consumers (who is calling): ``consumer_header="X-API-Key"`` or
|
|
91
|
+
``consumer=lambda info: ...`` (gets a RequestInfo, returns an id or
|
|
92
|
+
None). Ids are HMAC-SHA256-hashed with ``consumer_salt``
|
|
93
|
+
(REQLY_CONSUMER_SALT) before leaving the app unless ``hash_consumer=False``.
|
|
94
|
+
|
|
77
95
|
``push_openapi=True`` (or REQLY_PUSH_OPENAPI=true) uploads the app's
|
|
78
96
|
OpenAPI spec (FastAPI, Litestar) to the collector on the first request,
|
|
79
97
|
so the dashboard can show undocumented, unused and deprecated-but-used
|
|
@@ -107,6 +125,10 @@ def instrument(
|
|
|
107
125
|
release=release,
|
|
108
126
|
environment=environment,
|
|
109
127
|
push_openapi=push_openapi,
|
|
128
|
+
consumer_header=consumer_header,
|
|
129
|
+
consumer=consumer,
|
|
130
|
+
consumer_salt=consumer_salt,
|
|
131
|
+
hash_consumer=hash_consumer,
|
|
110
132
|
)
|
|
111
133
|
client = ReqlyClient(config)
|
|
112
134
|
if config.push_openapi:
|
|
@@ -143,3 +165,58 @@ def instrument(
|
|
|
143
165
|
exc_info=True,
|
|
144
166
|
)
|
|
145
167
|
return None
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
_GENERIC_OPTIONS = (
|
|
171
|
+
"service_name", "collector_url", "api_key", "sample_rate", "flush_interval_seconds",
|
|
172
|
+
"max_batch_size", "max_queue_size", "ignore_routes", "capture_request_body",
|
|
173
|
+
"release", "environment", "consumer_header", "consumer", "consumer_salt", "hash_consumer",
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _generic_client(kind: str, route_resolver, options: dict) -> ReqlyClient:
|
|
178
|
+
if route_resolver is None:
|
|
179
|
+
logger.warning(
|
|
180
|
+
"reqly: instrument_%s() without route_resolver records every request as "
|
|
181
|
+
"__unmatched__; pass a function that returns the route template", kind,
|
|
182
|
+
)
|
|
183
|
+
if options.pop("push_openapi", None):
|
|
184
|
+
logger.warning("reqly: push_openapi is not available for generic %s apps", kind.upper())
|
|
185
|
+
unknown = set(options) - set(_GENERIC_OPTIONS)
|
|
186
|
+
if unknown:
|
|
187
|
+
logger.warning("reqly: instrument_%s() ignoring unknown options %s", kind, sorted(unknown))
|
|
188
|
+
return ReqlyClient(Config.resolve(**{key: options.get(key) for key in _GENERIC_OPTIONS}))
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def instrument_wsgi(app, *, route_resolver=None, **options):
|
|
192
|
+
"""Wrap any WSGI app (Bottle, Pyramid, Falcon, CherryPy...) and return
|
|
193
|
+
the wrapped app -- serve that one::
|
|
194
|
+
|
|
195
|
+
app = reqly.instrument_wsgi(app, service_name="api",
|
|
196
|
+
route_resolver=lambda environ: environ["bottle.route"].rule)
|
|
197
|
+
|
|
198
|
+
``route_resolver(environ)`` runs after the app handled the request and
|
|
199
|
+
returns the route template or None. ``options`` are those of
|
|
200
|
+
``instrument()``. Like ``instrument()``, a failure leaves the app
|
|
201
|
+
unwrapped instead of raising."""
|
|
202
|
+
try:
|
|
203
|
+
from .integrations.wsgi import ReqlyWSGIMiddleware
|
|
204
|
+
|
|
205
|
+
return ReqlyWSGIMiddleware(app, _generic_client("wsgi", route_resolver, options), route_resolver)
|
|
206
|
+
except Exception:
|
|
207
|
+
logger.warning("reqly: instrument_wsgi() failed, app will run uninstrumented", exc_info=True)
|
|
208
|
+
return app
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def instrument_asgi(app, *, route_resolver=None, **options):
|
|
212
|
+
"""Wrap any ASGI app and return the wrapped app. ``route_resolver(scope)``
|
|
213
|
+
gets a copy of the scope as it arrived; a template the framework puts in
|
|
214
|
+
``scope["route"]`` or ``scope["path_template"]`` is used first."""
|
|
215
|
+
try:
|
|
216
|
+
from .integrations.fastapi import ReqlyASGIMiddleware
|
|
217
|
+
|
|
218
|
+
client = _generic_client("asgi", route_resolver, options)
|
|
219
|
+
return ReqlyASGIMiddleware(app, client, route_resolver)
|
|
220
|
+
except Exception:
|
|
221
|
+
logger.warning("reqly: instrument_asgi() failed, app will run uninstrumented", exc_info=True)
|
|
222
|
+
return app
|
|
@@ -31,6 +31,10 @@ class RequestEvent:
|
|
|
31
31
|
host: str = _HOSTNAME
|
|
32
32
|
request_bytes: int | None = None
|
|
33
33
|
response_bytes: int | None = None
|
|
34
|
+
consumer_id: str | None = None
|
|
35
|
+
llm_model: str | None = None
|
|
36
|
+
llm_input_tokens: int | None = None
|
|
37
|
+
llm_output_tokens: int | None = None
|
|
34
38
|
sdk_version: str = field(default_factory=_get_sdk_version)
|
|
35
39
|
|
|
36
40
|
def to_dict(self) -> dict:
|
|
@@ -59,7 +63,10 @@ def build_event(
|
|
|
59
63
|
sdk_version: str,
|
|
60
64
|
request_bytes: int | None = None,
|
|
61
65
|
response_bytes: int | None = None,
|
|
66
|
+
consumer_id: str | None = None,
|
|
67
|
+
llm: tuple[str, int, int] | None = None,
|
|
62
68
|
) -> RequestEvent:
|
|
69
|
+
llm_model, llm_input_tokens, llm_output_tokens = llm if llm else (None, None, None)
|
|
63
70
|
return RequestEvent(
|
|
64
71
|
service_name=service_name,
|
|
65
72
|
method=method,
|
|
@@ -71,4 +78,8 @@ def build_event(
|
|
|
71
78
|
sdk_version=sdk_version,
|
|
72
79
|
request_bytes=request_bytes,
|
|
73
80
|
response_bytes=response_bytes,
|
|
81
|
+
consumer_id=consumer_id,
|
|
82
|
+
llm_model=llm_model,
|
|
83
|
+
llm_input_tokens=llm_input_tokens,
|
|
84
|
+
llm_output_tokens=llm_output_tokens,
|
|
74
85
|
)
|
|
@@ -6,6 +6,7 @@ from .buffer import EventBuffer
|
|
|
6
6
|
from .capture import build_event
|
|
7
7
|
from .config import Config
|
|
8
8
|
from .openapi_push import OpenAPIPusher
|
|
9
|
+
from .request_context import ConsumerResolver
|
|
9
10
|
from .sampling import Sampler
|
|
10
11
|
from .shipper import Shipper
|
|
11
12
|
|
|
@@ -28,8 +29,16 @@ class ReqlyClient:
|
|
|
28
29
|
self._disabled = False
|
|
29
30
|
self._ignore_routes = set(config.ignore_routes)
|
|
30
31
|
self._openapi_pusher: OpenAPIPusher | None = None
|
|
32
|
+
self._consumers: ConsumerResolver | None = None
|
|
31
33
|
|
|
32
34
|
try:
|
|
35
|
+
if config.consumer_header or config.consumer is not None:
|
|
36
|
+
self._consumers = ConsumerResolver(
|
|
37
|
+
header=config.consumer_header,
|
|
38
|
+
func=config.consumer,
|
|
39
|
+
salt=config.consumer_salt,
|
|
40
|
+
hash_ids=config.hash_consumer,
|
|
41
|
+
)
|
|
33
42
|
self._sampler = Sampler(config.sample_rate)
|
|
34
43
|
shipper = Shipper(
|
|
35
44
|
collector_url=config.collector_url,
|
|
@@ -63,7 +72,12 @@ class ReqlyClient:
|
|
|
63
72
|
error_type: str | None,
|
|
64
73
|
request_bytes: int | None = None,
|
|
65
74
|
response_bytes: int | None = None,
|
|
75
|
+
request_info=None,
|
|
76
|
+
llm: tuple[str, int, int] | None = None,
|
|
66
77
|
) -> None:
|
|
78
|
+
"""``request_info``: zero-argument callable returning a RequestInfo,
|
|
79
|
+
only called when consumer tracking is on. ``llm``: (model,
|
|
80
|
+
input_tokens, output_tokens) recorded during the request."""
|
|
67
81
|
if self._disabled:
|
|
68
82
|
return
|
|
69
83
|
try:
|
|
@@ -74,6 +88,8 @@ class ReqlyClient:
|
|
|
74
88
|
if not self._sampler.should_sample():
|
|
75
89
|
return
|
|
76
90
|
event = build_event(
|
|
91
|
+
consumer_id=self._consumer_id(request_info),
|
|
92
|
+
llm=llm,
|
|
77
93
|
service_name=self.config.service_name,
|
|
78
94
|
method=method,
|
|
79
95
|
route=route,
|
|
@@ -93,6 +109,16 @@ class ReqlyClient:
|
|
|
93
109
|
)
|
|
94
110
|
self._disabled = True
|
|
95
111
|
|
|
112
|
+
def _consumer_id(self, request_info) -> str | None:
|
|
113
|
+
if self._consumers is None or request_info is None:
|
|
114
|
+
return None
|
|
115
|
+
try:
|
|
116
|
+
return self._consumers.resolve(request_info)
|
|
117
|
+
except Exception:
|
|
118
|
+
# A failing consumer= callable costs the consumer id, not the event.
|
|
119
|
+
logger.warning("reqly: consumer lookup failed", exc_info=True)
|
|
120
|
+
return None
|
|
121
|
+
|
|
96
122
|
def enable_openapi_push(self, spec_factory) -> None:
|
|
97
123
|
"""Upload ``spec_factory()`` (the app's OpenAPI document) to the
|
|
98
124
|
collector once, when the first request is recorded."""
|
|
@@ -2,6 +2,7 @@ from __future__ import annotations
|
|
|
2
2
|
|
|
3
3
|
import os
|
|
4
4
|
from dataclasses import dataclass, field
|
|
5
|
+
from typing import Any
|
|
5
6
|
from importlib.metadata import PackageNotFoundError, version as _pkg_version
|
|
6
7
|
|
|
7
8
|
|
|
@@ -87,6 +88,10 @@ class Config:
|
|
|
87
88
|
release: str | None = None
|
|
88
89
|
environment: str | None = None
|
|
89
90
|
push_openapi: bool = False
|
|
91
|
+
consumer_header: str | None = None
|
|
92
|
+
consumer: Any = None # callable(RequestInfo) -> str | None
|
|
93
|
+
consumer_salt: str | None = None
|
|
94
|
+
hash_consumer: bool = True
|
|
90
95
|
sdk_version: str = field(default_factory=_get_sdk_version)
|
|
91
96
|
|
|
92
97
|
@classmethod
|
|
@@ -104,6 +109,10 @@ class Config:
|
|
|
104
109
|
release: str | None = None,
|
|
105
110
|
environment: str | None = None,
|
|
106
111
|
push_openapi: bool | None = None,
|
|
112
|
+
consumer_header: str | None = None,
|
|
113
|
+
consumer: Any = None,
|
|
114
|
+
consumer_salt: str | None = None,
|
|
115
|
+
hash_consumer: bool | None = None,
|
|
107
116
|
) -> "Config":
|
|
108
117
|
import sys as _sys
|
|
109
118
|
|
|
@@ -168,4 +177,12 @@ class Config:
|
|
|
168
177
|
if push_openapi is not None
|
|
169
178
|
else _env_bool("REQLY_PUSH_OPENAPI", False)
|
|
170
179
|
),
|
|
180
|
+
consumer_header=consumer_header or os.environ.get("REQLY_CONSUMER_HEADER") or None,
|
|
181
|
+
consumer=consumer,
|
|
182
|
+
consumer_salt=consumer_salt or os.environ.get("REQLY_CONSUMER_SALT") or None,
|
|
183
|
+
hash_consumer=(
|
|
184
|
+
hash_consumer
|
|
185
|
+
if hash_consumer is not None
|
|
186
|
+
else _env_bool("REQLY_HASH_CONSUMER", True)
|
|
187
|
+
),
|
|
171
188
|
)
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
"""Per-request state the integrations collect besides timing: who the
|
|
2
|
+
caller is (consumer) and what LLM usage the request caused.
|
|
3
|
+
|
|
4
|
+
LLM usage is recorded from inside the request handler with
|
|
5
|
+
``reqly.record_llm_usage()``, so it lives in a ContextVar the middleware
|
|
6
|
+
sets when the request starts. The variable holds a mutable accumulator, so
|
|
7
|
+
usage recorded in a thread-pool copy of the context (Starlette runs sync
|
|
8
|
+
endpoints that way) still lands on the request's accumulator.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import contextvars
|
|
14
|
+
import hashlib
|
|
15
|
+
import hmac
|
|
16
|
+
import logging
|
|
17
|
+
from dataclasses import dataclass, field
|
|
18
|
+
from typing import Any, Callable, Mapping, Optional
|
|
19
|
+
|
|
20
|
+
logger = logging.getLogger("reqly")
|
|
21
|
+
|
|
22
|
+
_CONSUMER_HASH_HEX = 16
|
|
23
|
+
_MAX_CONSUMER_LEN = 128
|
|
24
|
+
_MAX_MODEL_LEN = 255
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class RequestInfo:
|
|
29
|
+
"""What a ``consumer=`` callable receives: the method, path, headers
|
|
30
|
+
(lower-case names) and the framework's own request object as ``raw``
|
|
31
|
+
(an ASGI scope, a Flask/Django request or a WSGI environ)."""
|
|
32
|
+
|
|
33
|
+
method: str
|
|
34
|
+
path: str
|
|
35
|
+
headers: Mapping[str, str]
|
|
36
|
+
raw: Any = None
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
# --- LLM usage -----------------------------------------------------------------
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class LLMUsage:
|
|
44
|
+
tokens_by_model: dict = field(default_factory=dict) # model -> [input, output]
|
|
45
|
+
|
|
46
|
+
def add(self, model: str, input_tokens: int, output_tokens: int) -> None:
|
|
47
|
+
totals = self.tokens_by_model.setdefault(model, [0, 0])
|
|
48
|
+
totals[0] += input_tokens
|
|
49
|
+
totals[1] += output_tokens
|
|
50
|
+
|
|
51
|
+
def summary(self) -> tuple[str, int, int] | None:
|
|
52
|
+
"""(model, input_tokens, output_tokens) for the event. A request
|
|
53
|
+
that used several models is attributed to the one with the most
|
|
54
|
+
tokens and carries the token totals of all of them."""
|
|
55
|
+
if not self.tokens_by_model:
|
|
56
|
+
return None
|
|
57
|
+
model = max(self.tokens_by_model, key=lambda m: sum(self.tokens_by_model[m]))
|
|
58
|
+
input_total = sum(t[0] for t in self.tokens_by_model.values())
|
|
59
|
+
output_total = sum(t[1] for t in self.tokens_by_model.values())
|
|
60
|
+
return model, input_total, output_total
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
_current_usage: contextvars.ContextVar[Optional[LLMUsage]] = contextvars.ContextVar(
|
|
64
|
+
"reqly_llm_usage", default=None
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def begin_request() -> contextvars.Token:
|
|
69
|
+
return _current_usage.set(LLMUsage())
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def end_request(token: contextvars.Token | None) -> tuple[str, int, int] | None:
|
|
73
|
+
usage = _current_usage.get()
|
|
74
|
+
if token is not None:
|
|
75
|
+
try:
|
|
76
|
+
_current_usage.reset(token)
|
|
77
|
+
except ValueError: # token from another context; nothing to restore
|
|
78
|
+
_current_usage.set(None)
|
|
79
|
+
return usage.summary() if usage is not None else None
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _as_count(value) -> int:
|
|
83
|
+
count = int(value or 0)
|
|
84
|
+
if count < 0:
|
|
85
|
+
raise ValueError("token counts can't be negative")
|
|
86
|
+
return count
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def record_llm_usage(model: str, input_tokens: int = 0, output_tokens: int = 0) -> None:
|
|
90
|
+
"""Attribute LLM token usage to the request being served. Call it after
|
|
91
|
+
each model call inside a request handler; calls add up. Outside an
|
|
92
|
+
instrumented request it does nothing. Never raises."""
|
|
93
|
+
try:
|
|
94
|
+
usage = _current_usage.get()
|
|
95
|
+
if usage is None:
|
|
96
|
+
logger.debug("reqly: record_llm_usage() outside an instrumented request, ignored")
|
|
97
|
+
return
|
|
98
|
+
usage.add(str(model)[:_MAX_MODEL_LEN], _as_count(input_tokens), _as_count(output_tokens))
|
|
99
|
+
except Exception:
|
|
100
|
+
logger.warning("reqly: record_llm_usage() failed", exc_info=True)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _field(obj, name):
|
|
104
|
+
return obj.get(name) if isinstance(obj, Mapping) else getattr(obj, name, None)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def record_llm_response(response) -> None:
|
|
108
|
+
"""``record_llm_usage()`` from a provider response object or dict:
|
|
109
|
+
OpenAI-style (``usage.prompt_tokens`` / ``completion_tokens``, also
|
|
110
|
+
Groq, Mistral, vLLM, LiteLLM...), OpenAI Responses API and Anthropic
|
|
111
|
+
(``usage.input_tokens`` / ``output_tokens``). Never raises."""
|
|
112
|
+
try:
|
|
113
|
+
usage = _field(response, "usage")
|
|
114
|
+
if usage is None:
|
|
115
|
+
return
|
|
116
|
+
input_tokens = _field(usage, "prompt_tokens")
|
|
117
|
+
output_tokens = _field(usage, "completion_tokens")
|
|
118
|
+
if input_tokens is None and output_tokens is None:
|
|
119
|
+
input_tokens = _field(usage, "input_tokens")
|
|
120
|
+
output_tokens = _field(usage, "output_tokens")
|
|
121
|
+
record_llm_usage(_field(response, "model") or "unknown", input_tokens or 0, output_tokens or 0)
|
|
122
|
+
except Exception:
|
|
123
|
+
logger.warning("reqly: record_llm_response() failed", exc_info=True)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
# --- consumers -----------------------------------------------------------------
|
|
127
|
+
|
|
128
|
+
ConsumerCallable = Callable[[RequestInfo], Optional[str]]
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
class ConsumerResolver:
|
|
132
|
+
"""Turns a request into a consumer id: from a header or a callable,
|
|
133
|
+
then (by default) HMAC-SHA256 with the app's salt, truncated -- the raw
|
|
134
|
+
API key or user id never leaves the app."""
|
|
135
|
+
|
|
136
|
+
def __init__(
|
|
137
|
+
self,
|
|
138
|
+
*,
|
|
139
|
+
header: str | None,
|
|
140
|
+
func: ConsumerCallable | None,
|
|
141
|
+
salt: str | None,
|
|
142
|
+
hash_ids: bool,
|
|
143
|
+
) -> None:
|
|
144
|
+
self._header = header.lower() if header else None
|
|
145
|
+
self._func = func
|
|
146
|
+
self._salt = (salt or "").encode()
|
|
147
|
+
self._hash = hash_ids
|
|
148
|
+
if hash_ids and not salt:
|
|
149
|
+
logger.warning(
|
|
150
|
+
"reqly: consumer tracking without REQLY_CONSUMER_SALT -- hashed ids of guessable "
|
|
151
|
+
"values (user ids, emails) can be reversed by trying candidates; set a secret salt"
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
@property
|
|
155
|
+
def enabled(self) -> bool:
|
|
156
|
+
return bool(self._header or self._func)
|
|
157
|
+
|
|
158
|
+
def resolve(self, info_factory: Callable[[], RequestInfo]) -> str | None:
|
|
159
|
+
info = info_factory()
|
|
160
|
+
value = self._func(info) if self._func is not None else info.headers.get(self._header)
|
|
161
|
+
if value is None or value == "":
|
|
162
|
+
return None
|
|
163
|
+
value = str(value)
|
|
164
|
+
if not self._hash:
|
|
165
|
+
return value[:_MAX_CONSUMER_LEN]
|
|
166
|
+
digest = hmac.new(self._salt, value.encode(), hashlib.sha256).hexdigest()
|
|
167
|
+
return digest[:_CONSUMER_HASH_HEX]
|
|
@@ -31,6 +31,7 @@ from django.core.exceptions import MiddlewareNotUsed
|
|
|
31
31
|
from ..core.capture import normalize_route
|
|
32
32
|
from ..core.client import ReqlyClient
|
|
33
33
|
from ..core.config import Config
|
|
34
|
+
from ..core.request_context import RequestInfo, begin_request, end_request
|
|
34
35
|
|
|
35
36
|
logger = logging.getLogger("reqly")
|
|
36
37
|
|
|
@@ -42,7 +43,7 @@ _client_lock = threading.Lock()
|
|
|
42
43
|
_CONFIG_KEYS = (
|
|
43
44
|
"service_name", "collector_url", "api_key", "sample_rate", "flush_interval_seconds",
|
|
44
45
|
"max_batch_size", "max_queue_size", "ignore_routes", "capture_request_body",
|
|
45
|
-
"release", "environment",
|
|
46
|
+
"release", "environment", "consumer_header", "consumer", "consumer_salt", "hash_consumer",
|
|
46
47
|
)
|
|
47
48
|
|
|
48
49
|
# <int:pk>, <slug:slug>, <pk> -> {pk}
|
|
@@ -102,14 +103,22 @@ class ReqlyMiddleware:
|
|
|
102
103
|
if self._is_async:
|
|
103
104
|
return self.__acall__(request)
|
|
104
105
|
start = time.perf_counter()
|
|
105
|
-
|
|
106
|
-
|
|
106
|
+
token = begin_request()
|
|
107
|
+
try:
|
|
108
|
+
response = self.get_response(request)
|
|
109
|
+
finally:
|
|
110
|
+
llm = end_request(token)
|
|
111
|
+
self._record(request, response, start, llm)
|
|
107
112
|
return response
|
|
108
113
|
|
|
109
114
|
async def __acall__(self, request):
|
|
110
115
|
start = time.perf_counter()
|
|
111
|
-
|
|
112
|
-
|
|
116
|
+
token = begin_request()
|
|
117
|
+
try:
|
|
118
|
+
response = await self.get_response(request)
|
|
119
|
+
finally:
|
|
120
|
+
llm = end_request(token)
|
|
121
|
+
self._record(request, response, start, llm)
|
|
113
122
|
return response
|
|
114
123
|
|
|
115
124
|
def process_exception(self, request, exception):
|
|
@@ -118,7 +127,7 @@ class ReqlyMiddleware:
|
|
|
118
127
|
setattr(request, _EXCEPTION_ATTR, exception)
|
|
119
128
|
return None
|
|
120
129
|
|
|
121
|
-
def _record(self, request, response, start: float) -> None:
|
|
130
|
+
def _record(self, request, response, start: float, llm=None) -> None:
|
|
122
131
|
try:
|
|
123
132
|
duration_ms = (time.perf_counter() - start) * 1000
|
|
124
133
|
match = getattr(request, "resolver_match", None)
|
|
@@ -135,6 +144,13 @@ class ReqlyMiddleware:
|
|
|
135
144
|
error_type=type(exception).__name__ if exception is not None else None,
|
|
136
145
|
request_bytes=int(content_length) if content_length and content_length.isdigit() else None,
|
|
137
146
|
response_bytes=None if getattr(response, "streaming", False) else len(response.content),
|
|
147
|
+
request_info=lambda: RequestInfo(
|
|
148
|
+
method=request.method,
|
|
149
|
+
path=request.path,
|
|
150
|
+
headers={k.lower(): v for k, v in request.headers.items()},
|
|
151
|
+
raw=request,
|
|
152
|
+
),
|
|
153
|
+
llm=llm,
|
|
138
154
|
)
|
|
139
155
|
except Exception:
|
|
140
156
|
# record_request is already fail-open; this guards the attribute
|