llmshim 0.3.4__tar.gz → 0.3.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {llmshim-0.3.4 → llmshim-0.3.6}/CLAUDE.md +36 -1
- {llmshim-0.3.4 → llmshim-0.3.6}/Cargo.lock +1 -1
- {llmshim-0.3.4 → llmshim-0.3.6}/Cargo.toml +11 -2
- {llmshim-0.3.4 → llmshim-0.3.6}/PKG-INFO +1 -1
- {llmshim-0.3.4 → llmshim-0.3.6}/README.md +15 -0
- llmshim-0.3.6/benchmarks/gateway_loadtest.rs +203 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/reference/errors.md +12 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/client.rs +46 -0
- llmshim-0.3.6/src/gateway/auth.rs +199 -0
- llmshim-0.3.6/src/gateway/distributed.rs +958 -0
- llmshim-0.3.6/src/gateway/http.rs +571 -0
- llmshim-0.3.6/src/gateway/idempotency.rs +75 -0
- llmshim-0.3.6/src/gateway/metrics.rs +291 -0
- llmshim-0.3.6/src/gateway/mod.rs +1333 -0
- llmshim-0.3.6/src/gateway/quota.rs +147 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/lib.rs +3 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/main.rs +123 -1
- {llmshim-0.3.4 → llmshim-0.3.6}/src/models.rs +9 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/providers/anthropic.rs +15 -14
- {llmshim-0.3.4 → llmshim-0.3.6}/src/providers/gemini.rs +19 -17
- {llmshim-0.3.4 → llmshim-0.3.6}/src/providers/openai.rs +9 -8
- {llmshim-0.3.4 → llmshim-0.3.6}/src/providers/xai.rs +9 -8
- {llmshim-0.3.4 → llmshim-0.3.6}/src/proxy/error.rs +12 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/proxy/mod.rs +2 -2
- {llmshim-0.3.4 → llmshim-0.3.6}/src/proxy/ratelimit.rs +3 -0
- llmshim-0.3.6/tests/support/completion_status.rs +124 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_gemini.rs +13 -6
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_models.rs +1 -1
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_openai.rs +3 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/.github/workflows/pages.yml +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/.github/workflows/release.yml +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/.gitignore +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/CODE_OF_CONDUCT.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/CONTRIBUTING.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/LICENSE-APACHE +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/LICENSE-MIT +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/NOTICE +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/SECURITY.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/benchmarks/bench.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/benchmarks/bench_python.py +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/benchmarks/loadtest.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/.gitignore +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/book.toml +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/mermaid-init.js +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/mermaid.min.js +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/SUMMARY.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/concepts/contracts.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/concepts/conversations.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/concepts/portability.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/concepts/routing.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/concepts/translation-flow.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/guides/fallbacks.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/guides/images.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/guides/native-controls.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/guides/reasoning.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/guides/streaming.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/guides/tools.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/introduction.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/proxy/deployment.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/proxy/http-api.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/proxy/scaling.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/reference/api.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/reference/cli.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/reference/configuration.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/reference/models.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/reference/providers.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/reference/request-fields.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/reference/surfaces.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/start/choose.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/start/cli.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/start/clients.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/start/configure.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/start/proxy.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/docs/src/start/rust.md +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/examples/chat.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/examples/stream.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/llmshim/__init__.py +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/llmshim/_client.py +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/llmshim/_server.py +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/llmshim/types.py +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/pyproject.toml +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/config.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/env.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/error.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/fallback.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/log.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/provider.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/providers/mod.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/providers/openai_compat.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/providers/openrouter.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/proxy/convert.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/proxy/handlers.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/proxy/types.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/router.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/src/vision.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/integration.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/integration_fallback.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/integration_gemini.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/integration_gemini_tools.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/integration_long_context.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/integration_multimodel.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/integration_openrouter.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/integration_proxy.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/integration_sglang.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/integration_thinking.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/integration_tool_roundtrip.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/integration_vision.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/integration_xai.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_anthropic.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_fallback.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_fast_mode.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_log.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_multimodel.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_openai_compat.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_openrouter.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_proxy.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_proxy_convert.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_router.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_sse.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_tools.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_vision.rs +0 -0
- {llmshim-0.3.4 → llmshim-0.3.6}/tests/unit_xai.rs +0 -0
|
@@ -14,7 +14,7 @@ This is a public crate on crates.io. Do NOT make breaking changes to `pub` items
|
|
|
14
14
|
|
|
15
15
|
- **OpenAI:** `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-5.5`, `gpt-5.5-pro`, `gpt-5.4`, `gpt-5.4-pro`, `gpt-5.4-mini`, `gpt-5.4-nano`
|
|
16
16
|
- **Anthropic:** `claude-opus-5`, `claude-opus-4-8`, `claude-sonnet-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-sonnet-4-6`, `claude-haiku-4-5-20251001`
|
|
17
|
-
- **Gemini:** `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite`, `gemini-3.1-flash-lite`
|
|
17
|
+
- **Gemini:** `gemini-3.8-flash`, `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite`, `gemini-3.1-flash-lite`
|
|
18
18
|
- **xAI:** `grok-4.6`, `grok-4.5`, `grok-4.3`, `grok-4.20-multi-agent-beta-0309`, `grok-4.20-beta-0309-reasoning`, `grok-4.20-beta-0309-non-reasoning`
|
|
19
19
|
- **OpenRouter:** not enumerated (huge/dynamic catalog) — any `openrouter/<vendor>/<model>` slug routes through, e.g. `openrouter/anthropic/claude-sonnet-4.5`.
|
|
20
20
|
- **vLLM / SGLang:** not enumerated (self-hosted) — any `vllm/<served-model>` or `sglang/<served-model>` routes through to the configured server, e.g. `sglang/Qwen/Qwen3.6-35B-A3B-FP8`.
|
|
@@ -58,12 +58,23 @@ llmshim::completion(router, request)
|
|
|
58
58
|
|
|
59
59
|
Every provider implements: `transform_request`, `transform_response`, `transform_stream_chunk`.
|
|
60
60
|
|
|
61
|
+
Non-streaming OpenAI/xAI Responses, Gemini and Anthropic transformations require
|
|
62
|
+
supported terminal metadata. Do not turn absent or unrecognized status into
|
|
63
|
+
`finish_reason: "stop"`. Rejected metadata returns a fixed 502 provider error
|
|
64
|
+
without copying provider-supplied status/content into the diagnostic. Tests
|
|
65
|
+
live in `tests/support/completion_status.rs`, included by `unit_openai`.
|
|
66
|
+
Streaming status handling is separate and is not changed by this policy.
|
|
67
|
+
|
|
61
68
|
### Router (`src/router.rs`)
|
|
62
69
|
|
|
63
70
|
Parses `"provider/model"` strings by splitting on the **first** `/` only, so an OpenRouter slug's internal slash survives (`openrouter/anthropic/claude-sonnet-4.5` → provider `openrouter`, model `anthropic/claude-sonnet-4.5`). Auto-infers provider from prefix (`gpt*`/`o*` → openai, `claude*` → anthropic, `gemini*` → gemini, `grok*` → xai); **OpenRouter, vLLM, and SGLang have no prefix inference** — their slugs collide with everyone's, so address them explicitly (`openrouter/…`, `vllm/…`, `sglang/…`); the first-slash split also preserves HF-style served-model slugs (`vllm/meta-llama/Llama-3.1-8B-Instruct`). Supports aliases. `Router::from_env()` reads API-key env vars, plus `VLLM_BASE_URL` / `SGLANG_BASE_URL` (+ optional `*_API_KEY`) for the self-hosted providers.
|
|
64
71
|
|
|
65
72
|
### HTTP Client (`src/client.rs`)
|
|
66
73
|
|
|
74
|
+
Automatic redirects are disabled on the shared client. Keep prompts and
|
|
75
|
+
provider-specific credential headers at the configured endpoint; 3xx responses
|
|
76
|
+
remain provider errors. Callers must configure the final URL directly.
|
|
77
|
+
|
|
67
78
|
`ShimClient` with shared connection pool (`LazyLock`), HTTP/2, gzip/brotli/zstd compression, TCP keepalive + nodelay. Automatic retry (3 attempts by default) on transport errors and 429/500/502/503/504/529 status codes. This is the **reactive** layer: on a retryable *response* it honors the server's `Retry-After` header (integer seconds or HTTP-date) and provider reset hints (OpenAI `x-ratelimit-reset-*`, Anthropic `anthropic-ratelimit-*-reset`), clamped to a cap and nudged with a little jitter; when there's no server hint (or a transport error) it falls back to full-jitter exponential backoff (uniform in `[0, min(cap, base·2^attempt)]`) to avoid a thundering herd. Tunable via `LLMSHIM_MAX_RETRIES` and `LLMSHIM_MAX_BACKOFF_SECS`. `warmup()` pre-establishes TCP+TLS connections. `SseStream` buffers bytes, extracts `data:` lines, routes through provider's `transform_stream_chunk`.
|
|
68
79
|
|
|
69
80
|
### Fallback chains (`src/fallback.rs`)
|
|
@@ -147,6 +158,30 @@ Env config (all optional, safe defaults; when no RPM/TPM limits are set the limi
|
|
|
147
158
|
|
|
148
159
|
Topologies: **sidecar / zero-infra** (default in-memory limiter — set limits to `global / N` when running N replicas) vs. **Redis-coordinated fleet** (one shared global limit across all replicas).
|
|
149
160
|
|
|
161
|
+
### Experimental: priority-queue gateway (`src/gateway/`, feature `gateway`)
|
|
162
|
+
|
|
163
|
+
An experimental scheduler that inverts the proxy's admission model: instead of *rejecting* when a provider's token bucket is empty, it **enqueues** each request into a per-provider priority queue and dispatches when capacity frees — ordered by **priority tier** (paying customer > free) then **FIFO** within a tier. Built for a fleet doing thousands of req/s that can't fire every call the instant it arrives. `gateway = ["proxy"]` (reuses the `RateLimiter` token buckets, `Backpressure`, and the proxy's request/response converters — `convert`/`error` are `pub(crate)` for this).
|
|
164
|
+
|
|
165
|
+
- **`Scheduler`** (`src/gateway/mod.rs`) owns one lane (queue + dispatcher task) per provider, so a rate-limited OpenAI queue never blocks a ready Anthropic one. `submit()` (unary) / `submit_stream()` (SSE) enqueue + await; a client disconnect / `max_wait` timeout cancels the queued job so it never burns a token.
|
|
166
|
+
- **Dispatcher** is event/timer-driven: pops the highest-priority job, acquires a concurrency slot *then* a rate token (so a saturated semaphore never wastes a token), and on a rate-limit miss **requeues** (preserving priority) and sleeps for exactly the `RetryAfter` the limiter reports — waking early on new work. No busy-wait, no `RateLimiter` changes. `max_wait` bounds **queue residence only** (a `started` signal releases it at dispatch), never the upstream call.
|
|
167
|
+
- **Fairness/aging** (`InMemoryQueue`): per-tier FIFO deques; dequeue picks the front with the highest *effective* priority = `tier + min(max_boost, wait/aging_step)`, so a starved low tier eventually overtakes a high-tier flood. Defaults (`aging_step` 5s, `max_boost` 16) keep normal load strictly priority-then-FIFO. `O(#tiers)` dequeue.
|
|
168
|
+
- **Streaming** (`submit_stream`): a `Delivery::Stream` job hands back an `mpsc` receiver once the upstream opens; the dispatcher forwards chunks and holds the concurrency permit for the whole stream (freed when it ends / the client disconnects). HTTP `POST /v1/chat/stream` (and `stream:true`) bridge it to SSE.
|
|
169
|
+
- **`RequestQueue`** trait is the in-process pluggable backend (in-memory default). Cross-process is a *separate* seam, not this trait (a `Job` holds a `oneshot`).
|
|
170
|
+
- **Distributed** (`src/gateway/distributed.rs`, feature `gateway-redis` = `gateway` + `redis-coordination`): a fleet shares a Redis **ZSET priority queue** per provider and a Redis **pub/sub response bus** keyed by request-id, so any instance dispatches any job and routes the result back to the origin's HTTP connection. Reuses the shared `RedisRateLimiter` for fleet-wide rate coordination. `llmshim gateway` auto-selects distributed mode when `LLMSHIM_REDIS_URL` is set and the binary has `gateway-redis`. Supports **unary + streaming** (typed `BusMessage`s over the channel). Full-parity with the in-memory lane:
|
|
171
|
+
- **Aging** via a *virtual-deadline* score `enqueue_ms − tier·aging_step` popped with `ZPOPMIN` — one static score gives priority, FIFO, and anti-starvation aging with no re-scoring.
|
|
172
|
+
- **At-least-once** via an atomic **lease** (Lua: `ZPOPMIN` queue → `processing` ZSET with a visibility deadline + recorded score) + ack/release + a background **reaper** (`reap_once`, also public for external cron) that requeues expired leases; streams refresh their lease as they run. A redelivered job may run twice (idempotent upstream calls). Env: `LLMSHIM_GATEWAY_LEASE_TIMEOUT_MS`, `LLMSHIM_GATEWAY_REQUEST_TIMEOUT_MS`.
|
|
173
|
+
- **HTTP** (`src/gateway/http.rs`): serves the proxy's `POST /v1/chat` + `/v1/chat/stream` contract routed through the scheduler; tier from the **`x-llmshim-priority`** header (uint, default 0). `RealDispatch` calls `completion_with_logger`/`stream`, mapping an upstream 429 → bucket penalty. Run: `llmshim gateway` (needs `--features gateway`, or `gateway-redis` for a fleet). Env: `LLMSHIM_GATEWAY_MAX_WAIT_MS`, `LLMSHIM_GATEWAY_QUEUE_DEPTH`, `LLMSHIM_GATEWAY_MAX_CONCURRENCY`, `LLMSHIM_GATEWAY_AGING_STEP_MS`, `LLMSHIM_GATEWAY_MAX_BOOST`, `LLMSHIM_GATEWAY_REQUEST_TIMEOUT_MS`, `LLMSHIM_GATEWAY_LEASE_TIMEOUT_MS`, `LLMSHIM_GATEWAY_MAX_ATTEMPTS`, `LLMSHIM_GATEWAY_IDEMPOTENCY_TTL_SECS`, plus all the proxy rate-limit vars.
|
|
174
|
+
|
|
175
|
+
#### Production hardening (gateway)
|
|
176
|
+
|
|
177
|
+
- **Auth + tier from identity** (`src/gateway/auth.rs`): set `LLMSHIM_GATEWAY_KEYS_FILE` to a JSON map `{ "<api-key>": {"tenant","tier","rpm","tpm"} }`. Then `Authorization: Bearer <key>` is **required** and tier/tenant come from the key — the client `x-llmshim-priority` header is ignored (closes the queue-jump exploit); missing/invalid key → 401. Unset ⇒ open dev mode (header trusted, tenant `anonymous`).
|
|
178
|
+
- **Per-tenant quotas** (`src/gateway/quota.rs`): per-`(tenant,provider)` RPM/TPM token buckets from the caller's identity, enforced in the HTTP layer (proactive 429 + `Retry-After`) on top of the global provider limits — one tenant can't monopolize shared capacity.
|
|
179
|
+
- **Idempotency** (`src/gateway/idempotency.rs`): `Idempotency-Key` header → cache-after-completion so a client retry returns the first result (no second billed call); in-memory for local, Redis for the fleet. Distributed at-least-once also has a **done-marker** (a completed job that gets redelivered is skipped) and a **dead-letter queue** after `max_attempts` (poison-job guard).
|
|
180
|
+
- **Observability**: `GET /metrics` (Prometheus, `src/gateway/metrics.rs`, dependency-free) — requests/dispatched/rejected counters, in-flight + queue-depth gauges, queue-wait + upstream-latency histograms; `GET /ready` (503 if Redis is down in distributed mode); `GET /v1/gateway/stats` (queue depths, dead-letter counts); every response carries `x-request-id`.
|
|
181
|
+
- **Graceful shutdown**: `gateway`/`proxy` drain in-flight on SIGTERM/Ctrl-C.
|
|
182
|
+
- **Load test**: `cargo run --release --features gateway --example gateway_loadtest` (throughput + zero-loss + priority-under-load asserts). **Deploy**: `Dockerfile` builds `--features gateway` by default (`--build-arg FEATURES=gateway-redis` for a fleet); `docker run … llmshim gateway`.
|
|
183
|
+
- Intentionally **not** done: per-tier (as opposed to per-tenant) queue-depth caps — the global depth cap + per-tenant quotas cover it; single-flight idempotency for *concurrent* same-key requests (only cache-after-completion).
|
|
184
|
+
|
|
150
185
|
## Client libraries (`clients/`)
|
|
151
186
|
|
|
152
187
|
Thin clients that speak the proxy's HTTP API. They are faithful to the OpenAPI contract in `api/openapi.yaml` — when you change the proxy's request/response shapes, update that spec and keep the clients in sync. All publish in lockstep with the crate version on every release.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "llmshim"
|
|
3
|
-
version = "0.3.
|
|
3
|
+
version = "0.3.6"
|
|
4
4
|
edition = "2021"
|
|
5
5
|
description = "Blazing fast LLM API translation layer in pure Rust"
|
|
6
6
|
license = "MIT OR Apache-2.0"
|
|
@@ -8,12 +8,16 @@ repository = "https://github.com/sanjay920/llmshim"
|
|
|
8
8
|
homepage = "https://github.com/sanjay920/llmshim"
|
|
9
9
|
keywords = ["llm", "openai", "anthropic", "gemini", "ai"]
|
|
10
10
|
categories = ["api-bindings", "web-programming"]
|
|
11
|
-
exclude = ["clients/", "api/", ".claude/", "Dockerfile", ".dockerignore"]
|
|
11
|
+
exclude = ["clients/", "api/", ".claude/", "Dockerfile", ".dockerignore", "gateway-test-ui.html"]
|
|
12
12
|
readme = "README.md"
|
|
13
13
|
|
|
14
14
|
[features]
|
|
15
15
|
default = []
|
|
16
16
|
proxy = ["dep:axum", "dep:tower", "dep:tower-http", "dep:async-stream", "dep:async-trait"]
|
|
17
|
+
# Experimental priority-queue gateway. Reuses the proxy's RateLimiter/Backpressure.
|
|
18
|
+
gateway = ["proxy"]
|
|
19
|
+
# Gateway + Redis-backed distributed queue & response bus (cross-instance fleet).
|
|
20
|
+
gateway-redis = ["gateway", "redis-coordination"]
|
|
17
21
|
# Opt-in distributed rate-limit coordination via Redis. Keeps the default proxy
|
|
18
22
|
# binary lean — redis is only pulled in when this feature is explicitly enabled.
|
|
19
23
|
redis-coordination = ["proxy", "dep:redis"]
|
|
@@ -67,3 +71,8 @@ path = "benchmarks/bench.rs"
|
|
|
67
71
|
name = "loadtest"
|
|
68
72
|
path = "benchmarks/loadtest.rs"
|
|
69
73
|
required-features = ["proxy"]
|
|
74
|
+
|
|
75
|
+
[[example]]
|
|
76
|
+
name = "gateway_loadtest"
|
|
77
|
+
path = "benchmarks/gateway_loadtest.rs"
|
|
78
|
+
required-features = ["gateway"]
|
|
@@ -76,8 +76,23 @@ llmshim configure # interactive prompt
|
|
|
76
76
|
|
|
77
77
|
---
|
|
78
78
|
|
|
79
|
+
## Endpoint redirects
|
|
80
|
+
|
|
81
|
+
The shared HTTP client does not follow redirects. Configure the final API URL
|
|
82
|
+
directly: a 3xx response is returned as a provider error instead of forwarding
|
|
83
|
+
the prompt and provider-specific credential headers to another endpoint.
|
|
84
|
+
This applies to both streaming and non-streaming requests.
|
|
85
|
+
|
|
79
86
|
## Use it from Rust
|
|
80
87
|
|
|
88
|
+
Non-streaming OpenAI Responses, xAI Responses, Gemini, and Anthropic results
|
|
89
|
+
require supported terminal metadata. Missing, malformed, or unsupported
|
|
90
|
+
statuses return a `ProviderError` (502) with a fixed diagnostic instead of
|
|
91
|
+
being interpreted as successful completion. Known completion, output-limit,
|
|
92
|
+
filtering and Anthropic tool-call mappings remain available. Additional native
|
|
93
|
+
reasons require an explicit mapping; streaming transforms are unchanged.
|
|
94
|
+
Provider mocks should include the native terminal status/stop reason.
|
|
95
|
+
|
|
81
96
|
```bash
|
|
82
97
|
cargo add llmshim tokio serde_json
|
|
83
98
|
```
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
//! Gateway load test — drives the in-process priority [`Scheduler`] with tens of
|
|
2
|
+
//! thousands of concurrent mixed-priority requests against a fake (no-network)
|
|
3
|
+
//! dispatcher, so it exercises the scheduler itself, not any provider.
|
|
4
|
+
//!
|
|
5
|
+
//! Run: `cargo run --release --features gateway --example gateway_loadtest`
|
|
6
|
+
//!
|
|
7
|
+
//! Reports throughput + latency percentiles and asserts two invariants:
|
|
8
|
+
//! 1. **Zero loss** under a burst far larger than the concurrency limit.
|
|
9
|
+
//! 2. **Priority holds under load** — high-tier p50 latency ≪ low-tier p50
|
|
10
|
+
//! when the concurrency slots are saturated.
|
|
11
|
+
//! Exits non-zero if either invariant fails.
|
|
12
|
+
|
|
13
|
+
use std::sync::atomic::{AtomicU64, Ordering};
|
|
14
|
+
use std::sync::Arc;
|
|
15
|
+
use std::time::{Duration, Instant};
|
|
16
|
+
|
|
17
|
+
use async_trait::async_trait;
|
|
18
|
+
use llmshim::gateway::{Dispatch, DispatchError, GatewayConfig, GatewayRequest, Scheduler};
|
|
19
|
+
use llmshim::proxy::ratelimit::{InMemoryRateLimiter, RateLimitConfig};
|
|
20
|
+
use serde_json::{json, Value};
|
|
21
|
+
|
|
22
|
+
/// A no-network dispatcher with a fixed simulated latency.
|
|
23
|
+
struct FakeDispatch {
|
|
24
|
+
latency: Duration,
|
|
25
|
+
served: Arc<AtomicU64>,
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
#[async_trait]
|
|
29
|
+
impl Dispatch for FakeDispatch {
|
|
30
|
+
async fn dispatch(&self, _provider: &str, payload: Value) -> Result<Value, DispatchError> {
|
|
31
|
+
if !self.latency.is_zero() {
|
|
32
|
+
tokio::time::sleep(self.latency).await;
|
|
33
|
+
}
|
|
34
|
+
self.served.fetch_add(1, Ordering::Relaxed);
|
|
35
|
+
Ok(payload)
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
fn unlimited() -> Arc<InMemoryRateLimiter> {
|
|
40
|
+
Arc::new(InMemoryRateLimiter::new(RateLimitConfig::default()))
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
fn percentile(sorted_us: &[u128], p: f64) -> f64 {
|
|
44
|
+
if sorted_us.is_empty() {
|
|
45
|
+
return 0.0;
|
|
46
|
+
}
|
|
47
|
+
let idx = ((sorted_us.len() as f64 - 1.0) * p).round() as usize;
|
|
48
|
+
sorted_us[idx] as f64 / 1000.0 // → ms
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
fn report(name: &str, mut latencies_us: Vec<u128>, wall: Duration) {
|
|
52
|
+
latencies_us.sort_unstable();
|
|
53
|
+
let n = latencies_us.len();
|
|
54
|
+
let rps = n as f64 / wall.as_secs_f64();
|
|
55
|
+
println!(
|
|
56
|
+
"{name}: {n} reqs in {:.2}s = {rps:.0} req/s | p50={:.1}ms p95={:.1}ms p99={:.1}ms max={:.1}ms",
|
|
57
|
+
wall.as_secs_f64(),
|
|
58
|
+
percentile(&latencies_us, 0.50),
|
|
59
|
+
percentile(&latencies_us, 0.95),
|
|
60
|
+
percentile(&latencies_us, 0.99),
|
|
61
|
+
percentile(&latencies_us, 1.00),
|
|
62
|
+
);
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
#[tokio::main(flavor = "multi_thread")]
|
|
66
|
+
async fn main() {
|
|
67
|
+
let mut failures = 0;
|
|
68
|
+
|
|
69
|
+
// ---- Scenario A: throughput + zero loss under a large burst ------------
|
|
70
|
+
{
|
|
71
|
+
let served = Arc::new(AtomicU64::new(0));
|
|
72
|
+
let config = GatewayConfig {
|
|
73
|
+
max_queue_depth: 500_000,
|
|
74
|
+
max_concurrency_per_provider: 512,
|
|
75
|
+
max_wait: Duration::from_secs(60),
|
|
76
|
+
..Default::default()
|
|
77
|
+
};
|
|
78
|
+
let sched = Scheduler::new(
|
|
79
|
+
config,
|
|
80
|
+
unlimited(),
|
|
81
|
+
Arc::new(FakeDispatch {
|
|
82
|
+
latency: Duration::from_millis(2),
|
|
83
|
+
served: served.clone(),
|
|
84
|
+
}),
|
|
85
|
+
);
|
|
86
|
+
|
|
87
|
+
let n = 20_000u64;
|
|
88
|
+
let ok = Arc::new(AtomicU64::new(0));
|
|
89
|
+
let errs = Arc::new(AtomicU64::new(0));
|
|
90
|
+
let start = Instant::now();
|
|
91
|
+
let mut handles = Vec::with_capacity(n as usize);
|
|
92
|
+
for i in 0..n {
|
|
93
|
+
let s = sched.clone();
|
|
94
|
+
let ok = ok.clone();
|
|
95
|
+
let errs = errs.clone();
|
|
96
|
+
handles.push(tokio::spawn(async move {
|
|
97
|
+
let t = Instant::now();
|
|
98
|
+
let r = s
|
|
99
|
+
.submit(GatewayRequest {
|
|
100
|
+
provider: "bench".into(),
|
|
101
|
+
tier: (i % 3) as u8,
|
|
102
|
+
permits: 1,
|
|
103
|
+
payload: json!({ "id": i }),
|
|
104
|
+
})
|
|
105
|
+
.await;
|
|
106
|
+
match r {
|
|
107
|
+
Ok(_) => ok.fetch_add(1, Ordering::Relaxed),
|
|
108
|
+
Err(_) => errs.fetch_add(1, Ordering::Relaxed),
|
|
109
|
+
};
|
|
110
|
+
t.elapsed().as_micros()
|
|
111
|
+
}));
|
|
112
|
+
}
|
|
113
|
+
let mut lat = Vec::with_capacity(n as usize);
|
|
114
|
+
for h in handles {
|
|
115
|
+
lat.push(h.await.unwrap());
|
|
116
|
+
}
|
|
117
|
+
let wall = start.elapsed();
|
|
118
|
+
report("throughput", lat, wall);
|
|
119
|
+
let (ok, errs) = (ok.load(Ordering::Relaxed), errs.load(Ordering::Relaxed));
|
|
120
|
+
println!(
|
|
121
|
+
" ok={ok} errors={errs} served={}",
|
|
122
|
+
served.load(Ordering::Relaxed)
|
|
123
|
+
);
|
|
124
|
+
if errs != 0 || ok != n {
|
|
125
|
+
eprintln!(" FAIL: expected {n} successful, zero lost — got ok={ok} errs={errs}");
|
|
126
|
+
failures += 1;
|
|
127
|
+
} else {
|
|
128
|
+
println!(" PASS: zero loss under burst");
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
// ---- Scenario B: priority holds when concurrency is saturated ----------
|
|
133
|
+
{
|
|
134
|
+
let served = Arc::new(AtomicU64::new(0));
|
|
135
|
+
let config = GatewayConfig {
|
|
136
|
+
max_queue_depth: 100_000,
|
|
137
|
+
max_concurrency_per_provider: 8, // the bottleneck
|
|
138
|
+
max_wait: Duration::from_secs(120),
|
|
139
|
+
..Default::default()
|
|
140
|
+
};
|
|
141
|
+
let sched = Scheduler::new(
|
|
142
|
+
config,
|
|
143
|
+
unlimited(),
|
|
144
|
+
Arc::new(FakeDispatch {
|
|
145
|
+
latency: Duration::from_millis(25),
|
|
146
|
+
served: served.clone(),
|
|
147
|
+
}),
|
|
148
|
+
);
|
|
149
|
+
|
|
150
|
+
let per_tier = 800u64;
|
|
151
|
+
let low_lat = Arc::new(std::sync::Mutex::new(Vec::<u128>::new()));
|
|
152
|
+
let high_lat = Arc::new(std::sync::Mutex::new(Vec::<u128>::new()));
|
|
153
|
+
let start = Instant::now();
|
|
154
|
+
let mut handles = Vec::new();
|
|
155
|
+
// Interleave low- and high-tier submissions.
|
|
156
|
+
for i in 0..per_tier {
|
|
157
|
+
for (tier, bucket) in [(0u8, &low_lat), (5u8, &high_lat)] {
|
|
158
|
+
let s = sched.clone();
|
|
159
|
+
let bucket = bucket.clone();
|
|
160
|
+
handles.push(tokio::spawn(async move {
|
|
161
|
+
let t = Instant::now();
|
|
162
|
+
let _ = s
|
|
163
|
+
.submit(GatewayRequest {
|
|
164
|
+
provider: "bench".into(),
|
|
165
|
+
tier,
|
|
166
|
+
permits: 1,
|
|
167
|
+
payload: json!({ "id": i }),
|
|
168
|
+
})
|
|
169
|
+
.await;
|
|
170
|
+
bucket.lock().unwrap().push(t.elapsed().as_micros());
|
|
171
|
+
}));
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
for h in handles {
|
|
175
|
+
h.await.unwrap();
|
|
176
|
+
}
|
|
177
|
+
let wall = start.elapsed();
|
|
178
|
+
let low = Arc::try_unwrap(low_lat).unwrap().into_inner().unwrap();
|
|
179
|
+
let high = Arc::try_unwrap(high_lat).unwrap().into_inner().unwrap();
|
|
180
|
+
report("priority(low tier)", low.clone(), wall);
|
|
181
|
+
report("priority(high tier)", high.clone(), wall);
|
|
182
|
+
let mut lo = low;
|
|
183
|
+
let mut hi = high;
|
|
184
|
+
lo.sort_unstable();
|
|
185
|
+
hi.sort_unstable();
|
|
186
|
+
let low_p50 = percentile(&lo, 0.50);
|
|
187
|
+
let high_p50 = percentile(&hi, 0.50);
|
|
188
|
+
if high_p50 < low_p50 {
|
|
189
|
+
println!(" PASS: high-tier p50 {high_p50:.1}ms < low-tier p50 {low_p50:.1}ms");
|
|
190
|
+
} else {
|
|
191
|
+
eprintln!(
|
|
192
|
+
" FAIL: priority not honored — high p50 {high_p50:.1}ms >= low p50 {low_p50:.1}ms"
|
|
193
|
+
);
|
|
194
|
+
failures += 1;
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
if failures > 0 {
|
|
199
|
+
eprintln!("\n{failures} scenario(s) FAILED");
|
|
200
|
+
std::process::exit(1);
|
|
201
|
+
}
|
|
202
|
+
println!("\nAll load-test invariants held.");
|
|
203
|
+
}
|
|
@@ -23,6 +23,18 @@ failure can occur after streaming begins.
|
|
|
23
23
|
|
|
24
24
|
## What the shared client retries
|
|
25
25
|
|
|
26
|
+
Provider requests do not follow HTTP redirects. A `3xx` response returns a
|
|
27
|
+
`ProviderError` with that status; configure the final endpoint URL directly.
|
|
28
|
+
This keeps prompts and provider authentication headers at the configured
|
|
29
|
+
destination for both completion and streaming requests.
|
|
30
|
+
|
|
31
|
+
Non-streaming native normalizers require supported terminal metadata. Missing,
|
|
32
|
+
malformed, nonterminal or unrecognized status/finish-reason values produce a
|
|
33
|
+
`ProviderError` with status `502` and a fixed diagnostic, rather than labeling
|
|
34
|
+
partial text as a successful `stop`. The diagnostic excludes response text
|
|
35
|
+
and the provider-supplied reason. Known completion, length, filtering and tool
|
|
36
|
+
mappings remain supported. Streaming transformations are unchanged.
|
|
37
|
+
|
|
26
38
|
The shared provider client automatically retries:
|
|
27
39
|
|
|
28
40
|
- transport connect, timeout, request, and body failures;
|
|
@@ -65,6 +65,9 @@ impl ShimClient {
|
|
|
65
65
|
pub fn new() -> Self {
|
|
66
66
|
Self {
|
|
67
67
|
http: Client::builder()
|
|
68
|
+
// Prompts and custom provider credentials belong only at the
|
|
69
|
+
// configured endpoint, never an HTTP Location target.
|
|
70
|
+
.redirect(reqwest::redirect::Policy::none())
|
|
68
71
|
.pool_idle_timeout(Duration::from_secs(90))
|
|
69
72
|
.pool_max_idle_per_host(4)
|
|
70
73
|
.tcp_keepalive(Duration::from_secs(30))
|
|
@@ -670,6 +673,49 @@ mod tests {
|
|
|
670
673
|
|
|
671
674
|
// --- Integration (mockito, local only — no provider API calls) ----------
|
|
672
675
|
|
|
676
|
+
#[tokio::test]
|
|
677
|
+
async fn redirects_do_not_forward_prompts_or_credentials() {
|
|
678
|
+
for status in [301, 302, 303, 307, 308] {
|
|
679
|
+
for same_origin in [false, true] {
|
|
680
|
+
let mut origin = mockito::Server::new_async().await;
|
|
681
|
+
let mut other = mockito::Server::new_async().await;
|
|
682
|
+
let destination = if same_origin { &mut origin } else { &mut other };
|
|
683
|
+
let location = format!("{}/moved", destination.url());
|
|
684
|
+
let forwarded = destination
|
|
685
|
+
.mock(if status <= 303 { "GET" } else { "POST" }, "/moved")
|
|
686
|
+
.with_status(200)
|
|
687
|
+
.with_body("unexpected forwarding")
|
|
688
|
+
.expect(0)
|
|
689
|
+
.create_async()
|
|
690
|
+
.await;
|
|
691
|
+
let redirect = origin
|
|
692
|
+
.mock("POST", "/v1/chat/completions")
|
|
693
|
+
.match_header("x-api-key", "test-secret")
|
|
694
|
+
.match_body(mockito::Matcher::Json(serde_json::json!({
|
|
695
|
+
"messages": [{"role": "user", "content": "private transcript"}]
|
|
696
|
+
})))
|
|
697
|
+
.with_status(status)
|
|
698
|
+
.with_header("location", &location)
|
|
699
|
+
.expect(1)
|
|
700
|
+
.create_async()
|
|
701
|
+
.await;
|
|
702
|
+
let request = ProviderRequest {
|
|
703
|
+
url: format!("{}/v1/chat/completions", origin.url()),
|
|
704
|
+
headers: vec![("x-api-key".into(), "test-secret".into())],
|
|
705
|
+
body: serde_json::json!({
|
|
706
|
+
"messages": [{"role": "user", "content": "private transcript"}]
|
|
707
|
+
}),
|
|
708
|
+
};
|
|
709
|
+
let result = ShimClient::new().send(&request).await;
|
|
710
|
+
redirect.assert_async().await;
|
|
711
|
+
forwarded.assert_async().await;
|
|
712
|
+
assert!(
|
|
713
|
+
matches!(result, Err(ShimError::ProviderError { status: actual, .. }) if actual == status as u16)
|
|
714
|
+
);
|
|
715
|
+
}
|
|
716
|
+
}
|
|
717
|
+
}
|
|
718
|
+
|
|
673
719
|
#[tokio::test]
|
|
674
720
|
async fn honors_retry_after_then_succeeds() {
|
|
675
721
|
let mut server = mockito::Server::new_async().await;
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
//! Gateway authentication and per-key identity (tenant, tier, per-tenant quotas).
|
|
2
|
+
//!
|
|
3
|
+
//! The priority tier must come from an **authenticated** identity, not a
|
|
4
|
+
//! client-supplied header — otherwise any caller sets `x-llmshim-priority: 255`
|
|
5
|
+
//! and jumps the queue. When an API-key file is configured, the gateway requires
|
|
6
|
+
//! `Authorization: Bearer <key>` and derives tier + tenant from the key (the
|
|
7
|
+
//! header is ignored). When no keys are configured it stays **open** for local
|
|
8
|
+
//! dev and honors the header, so nothing breaks out of the box.
|
|
9
|
+
|
|
10
|
+
use std::collections::HashMap;
|
|
11
|
+
|
|
12
|
+
use axum::http::HeaderMap;
|
|
13
|
+
use serde::Deserialize;
|
|
14
|
+
|
|
15
|
+
fn anonymous() -> String {
|
|
16
|
+
"anonymous".to_string()
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/// Who a request belongs to, resolved from its API key (or the header in open
|
|
20
|
+
/// mode). `rpm`/`tpm` are optional per-tenant limits enforced by the gateway.
|
|
21
|
+
#[derive(Clone, Debug, Deserialize)]
|
|
22
|
+
pub struct Identity {
|
|
23
|
+
#[serde(default = "anonymous")]
|
|
24
|
+
pub tenant: String,
|
|
25
|
+
#[serde(default)]
|
|
26
|
+
pub tier: u8,
|
|
27
|
+
#[serde(default)]
|
|
28
|
+
pub rpm: Option<u32>,
|
|
29
|
+
#[serde(default)]
|
|
30
|
+
pub tpm: Option<u32>,
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/// Authentication failure — both map to HTTP 401.
|
|
34
|
+
#[derive(Debug)]
|
|
35
|
+
pub enum AuthError {
|
|
36
|
+
MissingKey,
|
|
37
|
+
InvalidKey,
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/// API-key registry. `Open` = no auth (dev); `Enforced` = Bearer required.
|
|
41
|
+
pub enum KeyStore {
|
|
42
|
+
Open,
|
|
43
|
+
Enforced(HashMap<String, Identity>),
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
impl KeyStore {
|
|
47
|
+
/// Load from `LLMSHIM_GATEWAY_KEYS_FILE` (a JSON object mapping API key →
|
|
48
|
+
/// identity). Unset → open mode. A set-but-unreadable/invalid file is a boot
|
|
49
|
+
/// error and **exits** rather than silently running unauthenticated.
|
|
50
|
+
pub fn from_env() -> Self {
|
|
51
|
+
match std::env::var("LLMSHIM_GATEWAY_KEYS_FILE") {
|
|
52
|
+
Ok(path) if !path.trim().is_empty() => {
|
|
53
|
+
let data = std::fs::read_to_string(&path).unwrap_or_else(|e| {
|
|
54
|
+
eprintln!("gateway: cannot read LLMSHIM_GATEWAY_KEYS_FILE {path}: {e}");
|
|
55
|
+
std::process::exit(1);
|
|
56
|
+
});
|
|
57
|
+
let map: HashMap<String, Identity> =
|
|
58
|
+
serde_json::from_str(&data).unwrap_or_else(|e| {
|
|
59
|
+
eprintln!("gateway: invalid keys file {path}: {e}");
|
|
60
|
+
std::process::exit(1);
|
|
61
|
+
});
|
|
62
|
+
eprintln!(
|
|
63
|
+
" Auth: ENFORCED ({} API key(s); tier from key, x-llmshim-priority ignored)",
|
|
64
|
+
map.len()
|
|
65
|
+
);
|
|
66
|
+
KeyStore::Enforced(map)
|
|
67
|
+
}
|
|
68
|
+
_ => {
|
|
69
|
+
eprintln!(
|
|
70
|
+
" Auth: OPEN (no LLMSHIM_GATEWAY_KEYS_FILE; x-llmshim-priority trusted)"
|
|
71
|
+
);
|
|
72
|
+
KeyStore::Open
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/// Build an enforced store directly (tests / embedding).
|
|
78
|
+
pub fn enforced(keys: HashMap<String, Identity>) -> Self {
|
|
79
|
+
KeyStore::Enforced(keys)
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/// Whether Bearer auth is required.
|
|
83
|
+
pub fn is_enforced(&self) -> bool {
|
|
84
|
+
matches!(self, KeyStore::Enforced(_))
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/// Resolve the caller's identity. Open mode reads the tier from the client
|
|
88
|
+
/// header; enforced mode requires a valid Bearer key and takes tier + tenant
|
|
89
|
+
/// from it (header ignored, so callers can't self-escalate).
|
|
90
|
+
pub fn identify(&self, headers: &HeaderMap) -> Result<Identity, AuthError> {
|
|
91
|
+
match self {
|
|
92
|
+
KeyStore::Open => Ok(Identity {
|
|
93
|
+
tenant: anonymous(),
|
|
94
|
+
tier: header_tier(headers),
|
|
95
|
+
rpm: None,
|
|
96
|
+
tpm: None,
|
|
97
|
+
}),
|
|
98
|
+
KeyStore::Enforced(map) => {
|
|
99
|
+
let key = bearer_token(headers).ok_or(AuthError::MissingKey)?;
|
|
100
|
+
map.get(key).cloned().ok_or(AuthError::InvalidKey)
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/// Parse the client `x-llmshim-priority` header into a tier (default `0`).
|
|
107
|
+
pub fn header_tier(headers: &HeaderMap) -> u8 {
|
|
108
|
+
headers
|
|
109
|
+
.get("x-llmshim-priority")
|
|
110
|
+
.and_then(|v| v.to_str().ok())
|
|
111
|
+
.and_then(|s| s.trim().parse::<u8>().ok())
|
|
112
|
+
.unwrap_or(0)
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/// Extract the `Authorization: Bearer <token>` value.
|
|
116
|
+
fn bearer_token(headers: &HeaderMap) -> Option<&str> {
|
|
117
|
+
let h = headers
|
|
118
|
+
.get(axum::http::header::AUTHORIZATION)?
|
|
119
|
+
.to_str()
|
|
120
|
+
.ok()?;
|
|
121
|
+
h.strip_prefix("Bearer ")
|
|
122
|
+
.or_else(|| h.strip_prefix("bearer "))
|
|
123
|
+
.map(str::trim)
|
|
124
|
+
.filter(|s| !s.is_empty())
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
#[cfg(test)]
|
|
128
|
+
mod tests {
|
|
129
|
+
use super::*;
|
|
130
|
+
use axum::http::{HeaderMap, HeaderName, HeaderValue};
|
|
131
|
+
|
|
132
|
+
fn hdrs(pairs: &[(&str, &str)]) -> HeaderMap {
|
|
133
|
+
let mut h = HeaderMap::new();
|
|
134
|
+
for (k, v) in pairs {
|
|
135
|
+
h.insert(
|
|
136
|
+
HeaderName::from_bytes(k.as_bytes()).unwrap(),
|
|
137
|
+
HeaderValue::from_str(v).unwrap(),
|
|
138
|
+
);
|
|
139
|
+
}
|
|
140
|
+
h
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
#[test]
|
|
144
|
+
fn open_mode_trusts_the_priority_header() {
|
|
145
|
+
let store = KeyStore::Open;
|
|
146
|
+
let id = store
|
|
147
|
+
.identify(&hdrs(&[("x-llmshim-priority", "7")]))
|
|
148
|
+
.unwrap();
|
|
149
|
+
assert_eq!(id.tier, 7);
|
|
150
|
+
assert_eq!(id.tenant, "anonymous");
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
#[test]
|
|
154
|
+
fn enforced_mode_ignores_the_header_and_uses_the_key() {
|
|
155
|
+
let mut keys = HashMap::new();
|
|
156
|
+
keys.insert(
|
|
157
|
+
"sk-paid".to_string(),
|
|
158
|
+
Identity {
|
|
159
|
+
tenant: "acme".into(),
|
|
160
|
+
tier: 5,
|
|
161
|
+
rpm: Some(100),
|
|
162
|
+
tpm: None,
|
|
163
|
+
},
|
|
164
|
+
);
|
|
165
|
+
let store = KeyStore::enforced(keys);
|
|
166
|
+
|
|
167
|
+
// A valid key → its tier, regardless of the (spoofed) header.
|
|
168
|
+
let id = store
|
|
169
|
+
.identify(&hdrs(&[
|
|
170
|
+
("authorization", "Bearer sk-paid"),
|
|
171
|
+
("x-llmshim-priority", "255"),
|
|
172
|
+
]))
|
|
173
|
+
.unwrap();
|
|
174
|
+
assert_eq!(id.tier, 5, "tier must come from the key, not the header");
|
|
175
|
+
assert_eq!(id.tenant, "acme");
|
|
176
|
+
|
|
177
|
+
// Missing / bad key → rejected.
|
|
178
|
+
assert!(matches!(
|
|
179
|
+
store.identify(&hdrs(&[("x-llmshim-priority", "255")])),
|
|
180
|
+
Err(AuthError::MissingKey)
|
|
181
|
+
));
|
|
182
|
+
assert!(matches!(
|
|
183
|
+
store.identify(&hdrs(&[("authorization", "Bearer nope")])),
|
|
184
|
+
Err(AuthError::InvalidKey)
|
|
185
|
+
));
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
#[test]
|
|
189
|
+
fn identity_json_defaults() {
|
|
190
|
+
let id: Identity = serde_json::from_str(r#"{"tenant":"t"}"#).unwrap();
|
|
191
|
+
assert_eq!(id.tier, 0);
|
|
192
|
+
assert!(id.rpm.is_none());
|
|
193
|
+
let full: Identity =
|
|
194
|
+
serde_json::from_str(r#"{"tenant":"t","tier":3,"rpm":50,"tpm":9000}"#).unwrap();
|
|
195
|
+
assert_eq!(full.tier, 3);
|
|
196
|
+
assert_eq!(full.rpm, Some(50));
|
|
197
|
+
assert_eq!(full.tpm, Some(9000));
|
|
198
|
+
}
|
|
199
|
+
}
|