llmshim 0.3.3__tar.gz → 0.3.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {llmshim-0.3.3 → llmshim-0.3.5}/CLAUDE.md +25 -1
- {llmshim-0.3.3 → llmshim-0.3.5}/Cargo.lock +1 -1
- {llmshim-0.3.3 → llmshim-0.3.5}/Cargo.toml +11 -2
- {llmshim-0.3.3 → llmshim-0.3.5}/PKG-INFO +81 -5
- {llmshim-0.3.3 → llmshim-0.3.5}/README.md +1 -1
- llmshim-0.3.5/benchmarks/gateway_loadtest.rs +203 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/guides/reasoning.md +1 -1
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/models.md +2 -1
- {llmshim-0.3.3 → llmshim-0.3.5}/llmshim/__init__.py +6 -1
- {llmshim-0.3.3 → llmshim-0.3.5}/llmshim/_client.py +18 -0
- llmshim-0.3.5/src/gateway/auth.rs +199 -0
- llmshim-0.3.5/src/gateway/distributed.rs +958 -0
- llmshim-0.3.5/src/gateway/http.rs +571 -0
- llmshim-0.3.5/src/gateway/idempotency.rs +75 -0
- llmshim-0.3.5/src/gateway/metrics.rs +291 -0
- llmshim-0.3.5/src/gateway/mod.rs +1333 -0
- llmshim-0.3.5/src/gateway/quota.rs +147 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/lib.rs +3 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/main.rs +124 -1
- {llmshim-0.3.3 → llmshim-0.3.5}/src/models.rs +18 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/providers/gemini.rs +8 -7
- {llmshim-0.3.3 → llmshim-0.3.5}/src/proxy/error.rs +12 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/proxy/mod.rs +2 -2
- {llmshim-0.3.3 → llmshim-0.3.5}/src/proxy/ratelimit.rs +3 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_gemini.rs +20 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_gemini.rs +14 -6
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_models.rs +1 -1
- {llmshim-0.3.3 → llmshim-0.3.5}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/.github/workflows/pages.yml +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/.github/workflows/release.yml +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/.gitignore +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/CODE_OF_CONDUCT.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/CONTRIBUTING.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/LICENSE-APACHE +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/LICENSE-MIT +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/NOTICE +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/SECURITY.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/benchmarks/bench.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/benchmarks/bench_python.py +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/benchmarks/loadtest.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/.gitignore +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/book.toml +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/mermaid-init.js +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/mermaid.min.js +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/SUMMARY.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/concepts/contracts.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/concepts/conversations.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/concepts/portability.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/concepts/routing.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/concepts/translation-flow.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/guides/fallbacks.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/guides/images.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/guides/native-controls.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/guides/streaming.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/guides/tools.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/introduction.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/proxy/deployment.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/proxy/http-api.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/proxy/scaling.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/api.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/cli.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/configuration.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/errors.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/providers.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/request-fields.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/surfaces.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/start/choose.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/start/cli.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/start/clients.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/start/configure.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/start/proxy.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/start/rust.md +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/examples/chat.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/examples/stream.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/llmshim/_server.py +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/llmshim/types.py +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/pyproject.toml +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/client.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/config.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/env.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/error.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/fallback.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/log.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/provider.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/providers/anthropic.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/providers/mod.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/providers/openai.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/providers/openai_compat.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/providers/openrouter.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/providers/xai.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/proxy/convert.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/proxy/handlers.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/proxy/types.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/router.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/src/vision.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_fallback.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_gemini_tools.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_long_context.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_multimodel.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_openrouter.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_proxy.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_sglang.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_thinking.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_tool_roundtrip.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_vision.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_xai.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_anthropic.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_fallback.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_fast_mode.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_log.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_multimodel.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_openai.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_openai_compat.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_openrouter.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_proxy.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_proxy_convert.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_router.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_sse.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_tools.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_vision.rs +0 -0
- {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_xai.rs +0 -0
|
@@ -14,7 +14,7 @@ This is a public crate on crates.io. Do NOT make breaking changes to `pub` items
|
|
|
14
14
|
|
|
15
15
|
- **OpenAI:** `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-5.5`, `gpt-5.5-pro`, `gpt-5.4`, `gpt-5.4-pro`, `gpt-5.4-mini`, `gpt-5.4-nano`
|
|
16
16
|
- **Anthropic:** `claude-opus-5`, `claude-opus-4-8`, `claude-sonnet-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-sonnet-4-6`, `claude-haiku-4-5-20251001`
|
|
17
|
-
- **Gemini:** `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite`
|
|
17
|
+
- **Gemini:** `gemini-3.8-flash`, `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite`, `gemini-3.1-flash-lite`
|
|
18
18
|
- **xAI:** `grok-4.6`, `grok-4.5`, `grok-4.3`, `grok-4.20-multi-agent-beta-0309`, `grok-4.20-beta-0309-reasoning`, `grok-4.20-beta-0309-non-reasoning`
|
|
19
19
|
- **OpenRouter:** not enumerated (huge/dynamic catalog) — any `openrouter/<vendor>/<model>` slug routes through, e.g. `openrouter/anthropic/claude-sonnet-4.5`.
|
|
20
20
|
- **vLLM / SGLang:** not enumerated (self-hosted) — any `vllm/<served-model>` or `sglang/<served-model>` routes through to the configured server, e.g. `sglang/Qwen/Qwen3.6-35B-A3B-FP8`.
|
|
@@ -147,6 +147,30 @@ Env config (all optional, safe defaults; when no RPM/TPM limits are set the limi
|
|
|
147
147
|
|
|
148
148
|
Topologies: **sidecar / zero-infra** (default in-memory limiter — set limits to `global / N` when running N replicas) vs. **Redis-coordinated fleet** (one shared global limit across all replicas).
|
|
149
149
|
|
|
150
|
+
### Experimental: priority-queue gateway (`src/gateway/`, feature `gateway`)
|
|
151
|
+
|
|
152
|
+
An experimental scheduler that inverts the proxy's admission model: instead of *rejecting* when a provider's token bucket is empty, it **enqueues** each request into a per-provider priority queue and dispatches when capacity frees — ordered by **priority tier** (paying customer > free) then **FIFO** within a tier. Built for a fleet doing thousands of req/s that can't fire every call the instant it arrives. `gateway = ["proxy"]` (reuses the `RateLimiter` token buckets, `Backpressure`, and the proxy's request/response converters — `convert`/`error` are `pub(crate)` for this).
|
|
153
|
+
|
|
154
|
+
- **`Scheduler`** (`src/gateway/mod.rs`) owns one lane (queue + dispatcher task) per provider, so a rate-limited OpenAI queue never blocks a ready Anthropic one. `submit()` (unary) / `submit_stream()` (SSE) enqueue + await; a client disconnect / `max_wait` timeout cancels the queued job so it never burns a token.
|
|
155
|
+
- **Dispatcher** is event/timer-driven: pops the highest-priority job, acquires a concurrency slot *then* a rate token (so a saturated semaphore never wastes a token), and on a rate-limit miss **requeues** (preserving priority) and sleeps for exactly the `RetryAfter` the limiter reports — waking early on new work. No busy-wait, no `RateLimiter` changes. `max_wait` bounds **queue residence only** (a `started` signal releases it at dispatch), never the upstream call.
|
|
156
|
+
- **Fairness/aging** (`InMemoryQueue`): per-tier FIFO deques; dequeue picks the front with the highest *effective* priority = `tier + min(max_boost, wait/aging_step)`, so a starved low tier eventually overtakes a high-tier flood. Defaults (`aging_step` 5s, `max_boost` 16) keep normal load strictly priority-then-FIFO. `O(#tiers)` dequeue.
|
|
157
|
+
- **Streaming** (`submit_stream`): a `Delivery::Stream` job hands back an `mpsc` receiver once the upstream opens; the dispatcher forwards chunks and holds the concurrency permit for the whole stream (freed when it ends / the client disconnects). HTTP `POST /v1/chat/stream` (and `stream:true`) bridge it to SSE.
|
|
158
|
+
- **`RequestQueue`** trait is the in-process pluggable backend (in-memory default). Cross-process is a *separate* seam, not this trait (a `Job` holds a `oneshot`).
|
|
159
|
+
- **Distributed** (`src/gateway/distributed.rs`, feature `gateway-redis` = `gateway` + `redis-coordination`): a fleet shares a Redis **ZSET priority queue** per provider and a Redis **pub/sub response bus** keyed by request-id, so any instance dispatches any job and routes the result back to the origin's HTTP connection. Reuses the shared `RedisRateLimiter` for fleet-wide rate coordination. `llmshim gateway` auto-selects distributed mode when `LLMSHIM_REDIS_URL` is set and the binary has `gateway-redis`. Supports **unary + streaming** (typed `BusMessage`s over the channel). Full-parity with the in-memory lane:
|
|
160
|
+
- **Aging** via a *virtual-deadline* score `enqueue_ms − tier·aging_step` popped with `ZPOPMIN` — one static score gives priority, FIFO, and anti-starvation aging with no re-scoring.
|
|
161
|
+
- **At-least-once** via an atomic **lease** (Lua: `ZPOPMIN` queue → `processing` ZSET with a visibility deadline + recorded score) + ack/release + a background **reaper** (`reap_once`, also public for external cron) that requeues expired leases; streams refresh their lease as they run. A redelivered job may run twice (idempotent upstream calls). Env: `LLMSHIM_GATEWAY_LEASE_TIMEOUT_MS`, `LLMSHIM_GATEWAY_REQUEST_TIMEOUT_MS`.
|
|
162
|
+
- **HTTP** (`src/gateway/http.rs`): serves the proxy's `POST /v1/chat` + `/v1/chat/stream` contract routed through the scheduler; tier from the **`x-llmshim-priority`** header (uint, default 0). `RealDispatch` calls `completion_with_logger`/`stream`, mapping an upstream 429 → bucket penalty. Run: `llmshim gateway` (needs `--features gateway`, or `gateway-redis` for a fleet). Env: `LLMSHIM_GATEWAY_MAX_WAIT_MS`, `LLMSHIM_GATEWAY_QUEUE_DEPTH`, `LLMSHIM_GATEWAY_MAX_CONCURRENCY`, `LLMSHIM_GATEWAY_AGING_STEP_MS`, `LLMSHIM_GATEWAY_MAX_BOOST`, `LLMSHIM_GATEWAY_REQUEST_TIMEOUT_MS`, `LLMSHIM_GATEWAY_LEASE_TIMEOUT_MS`, `LLMSHIM_GATEWAY_MAX_ATTEMPTS`, `LLMSHIM_GATEWAY_IDEMPOTENCY_TTL_SECS`, plus all the proxy rate-limit vars.
|
|
163
|
+
|
|
164
|
+
#### Production hardening (gateway)
|
|
165
|
+
|
|
166
|
+
- **Auth + tier from identity** (`src/gateway/auth.rs`): set `LLMSHIM_GATEWAY_KEYS_FILE` to a JSON map `{ "<api-key>": {"tenant","tier","rpm","tpm"} }`. Then `Authorization: Bearer <key>` is **required** and tier/tenant come from the key — the client `x-llmshim-priority` header is ignored (closes the queue-jump exploit); missing/invalid key → 401. Unset ⇒ open dev mode (header trusted, tenant `anonymous`).
|
|
167
|
+
- **Per-tenant quotas** (`src/gateway/quota.rs`): per-`(tenant,provider)` RPM/TPM token buckets from the caller's identity, enforced in the HTTP layer (proactive 429 + `Retry-After`) on top of the global provider limits — one tenant can't monopolize shared capacity.
|
|
168
|
+
- **Idempotency** (`src/gateway/idempotency.rs`): `Idempotency-Key` header → cache-after-completion so a client retry returns the first result (no second billed call); in-memory for local, Redis for the fleet. Distributed at-least-once also has a **done-marker** (a completed job that gets redelivered is skipped) and a **dead-letter queue** after `max_attempts` (poison-job guard).
|
|
169
|
+
- **Observability**: `GET /metrics` (Prometheus, `src/gateway/metrics.rs`, dependency-free) — requests/dispatched/rejected counters, in-flight + queue-depth gauges, queue-wait + upstream-latency histograms; `GET /ready` (503 if Redis is down in distributed mode); `GET /v1/gateway/stats` (queue depths, dead-letter counts); every response carries `x-request-id`.
|
|
170
|
+
- **Graceful shutdown**: `gateway`/`proxy` drain in-flight on SIGTERM/Ctrl-C.
|
|
171
|
+
- **Load test**: `cargo run --release --features gateway --example gateway_loadtest` (throughput + zero-loss + priority-under-load asserts). **Deploy**: `Dockerfile` builds `--features gateway` by default (`--build-arg FEATURES=gateway-redis` for a fleet); `docker run … llmshim gateway`.
|
|
172
|
+
- Intentionally **not** done: per-tier (as opposed to per-tenant) queue-depth caps — the global depth cap + per-tenant quotas cover it; single-flight idempotency for *concurrent* same-key requests (only cache-after-completion).
|
|
173
|
+
|
|
150
174
|
## Client libraries (`clients/`)
|
|
151
175
|
|
|
152
176
|
Thin clients that speak the proxy's HTTP API. They are faithful to the OpenAPI contract in `api/openapi.yaml` — when you change the proxy's request/response shapes, update that spec and keep the clients in sync. All publish in lockstep with the crate version on every release.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "llmshim"
|
|
3
|
-
version = "0.3.
|
|
3
|
+
version = "0.3.5"
|
|
4
4
|
edition = "2021"
|
|
5
5
|
description = "Blazing fast LLM API translation layer in pure Rust"
|
|
6
6
|
license = "MIT OR Apache-2.0"
|
|
@@ -8,12 +8,16 @@ repository = "https://github.com/sanjay920/llmshim"
|
|
|
8
8
|
homepage = "https://github.com/sanjay920/llmshim"
|
|
9
9
|
keywords = ["llm", "openai", "anthropic", "gemini", "ai"]
|
|
10
10
|
categories = ["api-bindings", "web-programming"]
|
|
11
|
-
exclude = ["clients/", "api/", ".claude/", "Dockerfile", ".dockerignore"]
|
|
11
|
+
exclude = ["clients/", "api/", ".claude/", "Dockerfile", ".dockerignore", "gateway-test-ui.html"]
|
|
12
12
|
readme = "README.md"
|
|
13
13
|
|
|
14
14
|
[features]
|
|
15
15
|
default = []
|
|
16
16
|
proxy = ["dep:axum", "dep:tower", "dep:tower-http", "dep:async-stream", "dep:async-trait"]
|
|
17
|
+
# Experimental priority-queue gateway. Reuses the proxy's RateLimiter/Backpressure.
|
|
18
|
+
gateway = ["proxy"]
|
|
19
|
+
# Gateway + Redis-backed distributed queue & response bus (cross-instance fleet).
|
|
20
|
+
gateway-redis = ["gateway", "redis-coordination"]
|
|
17
21
|
# Opt-in distributed rate-limit coordination via Redis. Keeps the default proxy
|
|
18
22
|
# binary lean — redis is only pulled in when this feature is explicitly enabled.
|
|
19
23
|
redis-coordination = ["proxy", "dep:redis"]
|
|
@@ -67,3 +71,8 @@ path = "benchmarks/bench.rs"
|
|
|
67
71
|
name = "loadtest"
|
|
68
72
|
path = "benchmarks/loadtest.rs"
|
|
69
73
|
required-features = ["proxy"]
|
|
74
|
+
|
|
75
|
+
[[example]]
|
|
76
|
+
name = "gateway_loadtest"
|
|
77
|
+
path = "benchmarks/gateway_loadtest.rs"
|
|
78
|
+
required-features = ["gateway"]
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: llmshim
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.5
|
|
4
4
|
Classifier: Development Status :: 4 - Beta
|
|
5
5
|
Classifier: Intended Audience :: Developers
|
|
6
6
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -37,11 +37,30 @@ llmshim.configure(
|
|
|
37
37
|
openai="sk-...",
|
|
38
38
|
gemini="AIza...",
|
|
39
39
|
xai="xai-...",
|
|
40
|
+
openrouter="sk-or-...",
|
|
40
41
|
)
|
|
41
42
|
```
|
|
42
43
|
|
|
43
44
|
Or from the CLI: `llmshim configure`
|
|
44
45
|
|
|
46
|
+
### Self-hosted servers (vLLM / SGLang)
|
|
47
|
+
|
|
48
|
+
vLLM and SGLang are configured via **environment variables** (not
|
|
49
|
+
`config.toml`) — the auto-spawned proxy inherits them from your Python
|
|
50
|
+
process. Set them before your first call:
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
import os
|
|
54
|
+
|
|
55
|
+
os.environ["VLLM_BASE_URL"] = "http://localhost:8000/v1"
|
|
56
|
+
os.environ["VLLM_API_KEY"] = "..." # optional
|
|
57
|
+
os.environ["SGLANG_BASE_URL"] = "http://localhost:30000/v1"
|
|
58
|
+
os.environ["SGLANG_API_KEY"] = "..." # optional
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Then address them via the model string — `vllm/<served-model>` or
|
|
62
|
+
`sglang/<served-model>` (see the model table below).
|
|
63
|
+
|
|
45
64
|
## Chat
|
|
46
65
|
|
|
47
66
|
```python
|
|
@@ -106,17 +125,63 @@ print(f"GPT: {r2['message']['content']}")
|
|
|
106
125
|
|
|
107
126
|
## Reasoning / Thinking
|
|
108
127
|
|
|
128
|
+
Two provider-agnostic knobs control reasoning; both are clamped to the
|
|
129
|
+
nearest tier the target model supports:
|
|
130
|
+
|
|
131
|
+
- `reasoning_effort` — `"none"`, `"low"`, `"medium"`, `"high"`, `"xhigh"`, or `"max"`
|
|
132
|
+
- `reasoning_mode` — `"standard"` (default) or `"pro"` (requests substantially
|
|
133
|
+
more model work; native on OpenAI gpt-5.6/-pro, emulated as an effort bump
|
|
134
|
+
elsewhere)
|
|
135
|
+
|
|
109
136
|
```python
|
|
110
137
|
resp = llmshim.chat(
|
|
111
|
-
"claude-sonnet-
|
|
138
|
+
"claude-sonnet-5",
|
|
112
139
|
"Solve: x^2 - 5x + 6 = 0",
|
|
113
140
|
max_tokens=4000,
|
|
114
141
|
reasoning_effort="high",
|
|
142
|
+
reasoning_mode="pro",
|
|
115
143
|
)
|
|
116
144
|
print(resp["reasoning"]) # thinking content
|
|
117
145
|
print(resp["message"]["content"]) # answer
|
|
118
146
|
```
|
|
119
147
|
|
|
148
|
+
For full native control, bypass the unified mapping with a namespaced
|
|
149
|
+
`provider_config` (see below), e.g.
|
|
150
|
+
`provider_config={"x-anthropic": {"thinking": {"type": "enabled", "budget_tokens": 4000}}}`.
|
|
151
|
+
|
|
152
|
+
## Provider-Specific Controls (`provider_config`)
|
|
153
|
+
|
|
154
|
+
`provider_config` merges into the request **root** and carries anything the
|
|
155
|
+
unified `config` doesn't cover. Native provider controls MUST be **namespaced**
|
|
156
|
+
per provider (`x-anthropic`, `x-openai`, `x-gemini`, `x-openrouter`, `x-vllm`,
|
|
157
|
+
`x-sglang`); it also carries the top-level `tools`, `response_format`, and
|
|
158
|
+
`reasoning_summary` keys.
|
|
159
|
+
|
|
160
|
+
```python
|
|
161
|
+
resp = llmshim.chat(
|
|
162
|
+
"anthropic/claude-sonnet-5",
|
|
163
|
+
"Solve this step by step: 17 * 23",
|
|
164
|
+
max_tokens=4000,
|
|
165
|
+
provider_config={
|
|
166
|
+
# native Anthropic extended-thinking control
|
|
167
|
+
"x-anthropic": {"thinking": {"type": "enabled", "budget_tokens": 4000}},
|
|
168
|
+
# structured output
|
|
169
|
+
"response_format": {"type": "json_object"},
|
|
170
|
+
},
|
|
171
|
+
)
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
OpenRouter routing preferences use the `x-openrouter` namespace:
|
|
175
|
+
|
|
176
|
+
```python
|
|
177
|
+
resp = llmshim.chat(
|
|
178
|
+
"openrouter/anthropic/claude-sonnet-4.5",
|
|
179
|
+
"Hello",
|
|
180
|
+
max_tokens=200,
|
|
181
|
+
provider_config={"x-openrouter": {"provider": {"sort": "throughput"}}},
|
|
182
|
+
)
|
|
183
|
+
```
|
|
184
|
+
|
|
120
185
|
## Tool Use / Function Calling
|
|
121
186
|
|
|
122
187
|
```python
|
|
@@ -147,7 +212,7 @@ resp = llmshim.chat(
|
|
|
147
212
|
"anthropic/claude-sonnet-4-6",
|
|
148
213
|
"Hello",
|
|
149
214
|
max_tokens=100,
|
|
150
|
-
fallback=["openai/gpt-5.
|
|
215
|
+
fallback=["openai/gpt-5.6-sol", "gemini/gemini-3.5-flash"],
|
|
151
216
|
)
|
|
152
217
|
```
|
|
153
218
|
|
|
@@ -220,10 +285,21 @@ billed provider calls; run it only when you deliberately want to hit real APIs.
|
|
|
220
285
|
|
|
221
286
|
| Provider | Models |
|
|
222
287
|
|----------|--------|
|
|
223
|
-
| OpenAI | `gpt-5.5`, `gpt-5.5-pro`, `gpt-5.4`, `gpt-5.4-pro`, `gpt-5.4-mini`, `gpt-5.4-nano` |
|
|
288
|
+
| OpenAI | `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-5.5`, `gpt-5.5-pro`, `gpt-5.4`, `gpt-5.4-pro`, `gpt-5.4-mini`, `gpt-5.4-nano` |
|
|
224
289
|
| Anthropic | `claude-opus-5`, `claude-opus-4-8`, `claude-sonnet-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-sonnet-4-6`, `claude-haiku-4-5-20251001` |
|
|
225
|
-
| Gemini | `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite` |
|
|
290
|
+
| Gemini | `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite`, `gemini-3.1-flash-lite` |
|
|
226
291
|
| xAI | `grok-4.6`, `grok-4.5`, `grok-4.3`, `grok-4.20-multi-agent-beta-0309`, `grok-4.20-beta-0309-reasoning`, `grok-4.20-beta-0309-non-reasoning` |
|
|
227
292
|
|
|
228
293
|
Call `llmshim.models()` for the live list filtered to your configured providers.
|
|
229
294
|
|
|
295
|
+
### OpenRouter & self-hosted (vLLM / SGLang)
|
|
296
|
+
|
|
297
|
+
These providers are addressed by the model string plus environment variables —
|
|
298
|
+
any model the upstream serves is reachable, so they aren't in the table above.
|
|
299
|
+
|
|
300
|
+
| Provider | Address as | Env vars | Native controls |
|
|
301
|
+
|----------|-----------|----------|-----------------|
|
|
302
|
+
| OpenRouter | `openrouter/<vendor>/<model>` (e.g. `openrouter/anthropic/claude-sonnet-4.5`) | `OPENROUTER_API_KEY` (or `llmshim.configure(openrouter=...)`) | `provider_config={"x-openrouter": {...}}` (`provider`, `models`, `transforms`) |
|
|
303
|
+
| vLLM | `vllm/<served-model>` | `VLLM_BASE_URL` (+ optional `VLLM_API_KEY`) | `provider_config={"x-vllm": {...}}` |
|
|
304
|
+
| SGLang | `sglang/<served-model>` | `SGLANG_BASE_URL` (+ optional `SGLANG_API_KEY`) | `provider_config={"x-sglang": {...}}` |
|
|
305
|
+
|
|
@@ -364,7 +364,7 @@ Standard library only. Full docs: [`clients/ruby/README.md`](clients/ruby/README
|
|
|
364
364
|
|----------|--------|-------------------|
|
|
365
365
|
| **OpenAI** | `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-5.5`, `gpt-5.5-pro`, `gpt-5.4`, `gpt-5.4-pro`, `gpt-5.4-mini`, `gpt-5.4-nano` | Yes (summaries) |
|
|
366
366
|
| **Anthropic** | `claude-opus-5`, `claude-opus-4-8`, `claude-sonnet-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-sonnet-4-6`, `claude-haiku-4-5-20251001` | Yes (full thinking) |
|
|
367
|
-
| **Google Gemini** | `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite` | Yes (thought summaries) |
|
|
367
|
+
| **Google Gemini** | `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite`, `gemini-3.1-flash-lite` | Yes (thought summaries) |
|
|
368
368
|
| **xAI** | `grok-4.6`, `grok-4.5`, `grok-4.3`, `grok-4.20-multi-agent-beta-0309`, `grok-4.20-beta-0309-reasoning`, `grok-4.20-beta-0309-non-reasoning` | No (hidden) |
|
|
369
369
|
|
|
370
370
|
Use a bare model name (auto-detected by prefix) or an explicit `provider/model` string.
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
//! Gateway load test — drives the in-process priority [`Scheduler`] with tens of
|
|
2
|
+
//! thousands of concurrent mixed-priority requests against a fake (no-network)
|
|
3
|
+
//! dispatcher, so it exercises the scheduler itself, not any provider.
|
|
4
|
+
//!
|
|
5
|
+
//! Run: `cargo run --release --features gateway --example gateway_loadtest`
|
|
6
|
+
//!
|
|
7
|
+
//! Reports throughput + latency percentiles and asserts two invariants:
|
|
8
|
+
//! 1. **Zero loss** under a burst far larger than the concurrency limit.
|
|
9
|
+
//! 2. **Priority holds under load** — high-tier p50 latency ≪ low-tier p50
|
|
10
|
+
//! when the concurrency slots are saturated.
|
|
11
|
+
//! Exits non-zero if either invariant fails.
|
|
12
|
+
|
|
13
|
+
use std::sync::atomic::{AtomicU64, Ordering};
|
|
14
|
+
use std::sync::Arc;
|
|
15
|
+
use std::time::{Duration, Instant};
|
|
16
|
+
|
|
17
|
+
use async_trait::async_trait;
|
|
18
|
+
use llmshim::gateway::{Dispatch, DispatchError, GatewayConfig, GatewayRequest, Scheduler};
|
|
19
|
+
use llmshim::proxy::ratelimit::{InMemoryRateLimiter, RateLimitConfig};
|
|
20
|
+
use serde_json::{json, Value};
|
|
21
|
+
|
|
22
|
+
/// A no-network dispatcher with a fixed simulated latency.
|
|
23
|
+
struct FakeDispatch {
|
|
24
|
+
latency: Duration,
|
|
25
|
+
served: Arc<AtomicU64>,
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
#[async_trait]
|
|
29
|
+
impl Dispatch for FakeDispatch {
|
|
30
|
+
async fn dispatch(&self, _provider: &str, payload: Value) -> Result<Value, DispatchError> {
|
|
31
|
+
if !self.latency.is_zero() {
|
|
32
|
+
tokio::time::sleep(self.latency).await;
|
|
33
|
+
}
|
|
34
|
+
self.served.fetch_add(1, Ordering::Relaxed);
|
|
35
|
+
Ok(payload)
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
fn unlimited() -> Arc<InMemoryRateLimiter> {
|
|
40
|
+
Arc::new(InMemoryRateLimiter::new(RateLimitConfig::default()))
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
fn percentile(sorted_us: &[u128], p: f64) -> f64 {
|
|
44
|
+
if sorted_us.is_empty() {
|
|
45
|
+
return 0.0;
|
|
46
|
+
}
|
|
47
|
+
let idx = ((sorted_us.len() as f64 - 1.0) * p).round() as usize;
|
|
48
|
+
sorted_us[idx] as f64 / 1000.0 // → ms
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
fn report(name: &str, mut latencies_us: Vec<u128>, wall: Duration) {
|
|
52
|
+
latencies_us.sort_unstable();
|
|
53
|
+
let n = latencies_us.len();
|
|
54
|
+
let rps = n as f64 / wall.as_secs_f64();
|
|
55
|
+
println!(
|
|
56
|
+
"{name}: {n} reqs in {:.2}s = {rps:.0} req/s | p50={:.1}ms p95={:.1}ms p99={:.1}ms max={:.1}ms",
|
|
57
|
+
wall.as_secs_f64(),
|
|
58
|
+
percentile(&latencies_us, 0.50),
|
|
59
|
+
percentile(&latencies_us, 0.95),
|
|
60
|
+
percentile(&latencies_us, 0.99),
|
|
61
|
+
percentile(&latencies_us, 1.00),
|
|
62
|
+
);
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
#[tokio::main(flavor = "multi_thread")]
|
|
66
|
+
async fn main() {
|
|
67
|
+
let mut failures = 0;
|
|
68
|
+
|
|
69
|
+
// ---- Scenario A: throughput + zero loss under a large burst ------------
|
|
70
|
+
{
|
|
71
|
+
let served = Arc::new(AtomicU64::new(0));
|
|
72
|
+
let config = GatewayConfig {
|
|
73
|
+
max_queue_depth: 500_000,
|
|
74
|
+
max_concurrency_per_provider: 512,
|
|
75
|
+
max_wait: Duration::from_secs(60),
|
|
76
|
+
..Default::default()
|
|
77
|
+
};
|
|
78
|
+
let sched = Scheduler::new(
|
|
79
|
+
config,
|
|
80
|
+
unlimited(),
|
|
81
|
+
Arc::new(FakeDispatch {
|
|
82
|
+
latency: Duration::from_millis(2),
|
|
83
|
+
served: served.clone(),
|
|
84
|
+
}),
|
|
85
|
+
);
|
|
86
|
+
|
|
87
|
+
let n = 20_000u64;
|
|
88
|
+
let ok = Arc::new(AtomicU64::new(0));
|
|
89
|
+
let errs = Arc::new(AtomicU64::new(0));
|
|
90
|
+
let start = Instant::now();
|
|
91
|
+
let mut handles = Vec::with_capacity(n as usize);
|
|
92
|
+
for i in 0..n {
|
|
93
|
+
let s = sched.clone();
|
|
94
|
+
let ok = ok.clone();
|
|
95
|
+
let errs = errs.clone();
|
|
96
|
+
handles.push(tokio::spawn(async move {
|
|
97
|
+
let t = Instant::now();
|
|
98
|
+
let r = s
|
|
99
|
+
.submit(GatewayRequest {
|
|
100
|
+
provider: "bench".into(),
|
|
101
|
+
tier: (i % 3) as u8,
|
|
102
|
+
permits: 1,
|
|
103
|
+
payload: json!({ "id": i }),
|
|
104
|
+
})
|
|
105
|
+
.await;
|
|
106
|
+
match r {
|
|
107
|
+
Ok(_) => ok.fetch_add(1, Ordering::Relaxed),
|
|
108
|
+
Err(_) => errs.fetch_add(1, Ordering::Relaxed),
|
|
109
|
+
};
|
|
110
|
+
t.elapsed().as_micros()
|
|
111
|
+
}));
|
|
112
|
+
}
|
|
113
|
+
let mut lat = Vec::with_capacity(n as usize);
|
|
114
|
+
for h in handles {
|
|
115
|
+
lat.push(h.await.unwrap());
|
|
116
|
+
}
|
|
117
|
+
let wall = start.elapsed();
|
|
118
|
+
report("throughput", lat, wall);
|
|
119
|
+
let (ok, errs) = (ok.load(Ordering::Relaxed), errs.load(Ordering::Relaxed));
|
|
120
|
+
println!(
|
|
121
|
+
" ok={ok} errors={errs} served={}",
|
|
122
|
+
served.load(Ordering::Relaxed)
|
|
123
|
+
);
|
|
124
|
+
if errs != 0 || ok != n {
|
|
125
|
+
eprintln!(" FAIL: expected {n} successful, zero lost — got ok={ok} errs={errs}");
|
|
126
|
+
failures += 1;
|
|
127
|
+
} else {
|
|
128
|
+
println!(" PASS: zero loss under burst");
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
// ---- Scenario B: priority holds when concurrency is saturated ----------
|
|
133
|
+
{
|
|
134
|
+
let served = Arc::new(AtomicU64::new(0));
|
|
135
|
+
let config = GatewayConfig {
|
|
136
|
+
max_queue_depth: 100_000,
|
|
137
|
+
max_concurrency_per_provider: 8, // the bottleneck
|
|
138
|
+
max_wait: Duration::from_secs(120),
|
|
139
|
+
..Default::default()
|
|
140
|
+
};
|
|
141
|
+
let sched = Scheduler::new(
|
|
142
|
+
config,
|
|
143
|
+
unlimited(),
|
|
144
|
+
Arc::new(FakeDispatch {
|
|
145
|
+
latency: Duration::from_millis(25),
|
|
146
|
+
served: served.clone(),
|
|
147
|
+
}),
|
|
148
|
+
);
|
|
149
|
+
|
|
150
|
+
let per_tier = 800u64;
|
|
151
|
+
let low_lat = Arc::new(std::sync::Mutex::new(Vec::<u128>::new()));
|
|
152
|
+
let high_lat = Arc::new(std::sync::Mutex::new(Vec::<u128>::new()));
|
|
153
|
+
let start = Instant::now();
|
|
154
|
+
let mut handles = Vec::new();
|
|
155
|
+
// Interleave low- and high-tier submissions.
|
|
156
|
+
for i in 0..per_tier {
|
|
157
|
+
for (tier, bucket) in [(0u8, &low_lat), (5u8, &high_lat)] {
|
|
158
|
+
let s = sched.clone();
|
|
159
|
+
let bucket = bucket.clone();
|
|
160
|
+
handles.push(tokio::spawn(async move {
|
|
161
|
+
let t = Instant::now();
|
|
162
|
+
let _ = s
|
|
163
|
+
.submit(GatewayRequest {
|
|
164
|
+
provider: "bench".into(),
|
|
165
|
+
tier,
|
|
166
|
+
permits: 1,
|
|
167
|
+
payload: json!({ "id": i }),
|
|
168
|
+
})
|
|
169
|
+
.await;
|
|
170
|
+
bucket.lock().unwrap().push(t.elapsed().as_micros());
|
|
171
|
+
}));
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
for h in handles {
|
|
175
|
+
h.await.unwrap();
|
|
176
|
+
}
|
|
177
|
+
let wall = start.elapsed();
|
|
178
|
+
let low = Arc::try_unwrap(low_lat).unwrap().into_inner().unwrap();
|
|
179
|
+
let high = Arc::try_unwrap(high_lat).unwrap().into_inner().unwrap();
|
|
180
|
+
report("priority(low tier)", low.clone(), wall);
|
|
181
|
+
report("priority(high tier)", high.clone(), wall);
|
|
182
|
+
let mut lo = low;
|
|
183
|
+
let mut hi = high;
|
|
184
|
+
lo.sort_unstable();
|
|
185
|
+
hi.sort_unstable();
|
|
186
|
+
let low_p50 = percentile(&lo, 0.50);
|
|
187
|
+
let high_p50 = percentile(&hi, 0.50);
|
|
188
|
+
if high_p50 < low_p50 {
|
|
189
|
+
println!(" PASS: high-tier p50 {high_p50:.1}ms < low-tier p50 {low_p50:.1}ms");
|
|
190
|
+
} else {
|
|
191
|
+
eprintln!(
|
|
192
|
+
" FAIL: priority not honored — high p50 {high_p50:.1}ms >= low p50 {low_p50:.1}ms"
|
|
193
|
+
);
|
|
194
|
+
failures += 1;
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
if failures > 0 {
|
|
199
|
+
eprintln!("\n{failures} scenario(s) FAILED");
|
|
200
|
+
std::process::exit(1);
|
|
201
|
+
}
|
|
202
|
+
println!("\nAll load-test invariants held.");
|
|
203
|
+
}
|
|
@@ -117,7 +117,7 @@ token budget scaled from `max_tokens` and floored at 1024:
|
|
|
117
117
|
Gemini uses the four-rung
|
|
118
118
|
`generationConfig.thinkingConfig.thinkingLevel` enum:
|
|
119
119
|
|
|
120
|
-
| unified | gemini-3.5-flash / 3.6-flash / 3.5-flash-lite | gemini-3.7-flash (and gemini-3.1-pro) |
|
|
120
|
+
| unified | gemini-3.5-flash / 3.6-flash / 3.5-flash-lite / 3.1-flash-lite | gemini-3.7-flash (and gemini-3.1-pro) |
|
|
121
121
|
|---|---|---|
|
|
122
122
|
| `none` | `minimal` (zero thinking tokens) | **`low`** (this model cannot disable thinking) |
|
|
123
123
|
| `low` | `low` | `low` |
|
|
@@ -14,7 +14,7 @@ the ID and display label.
|
|
|
14
14
|
|
|
15
15
|
## Registered catalog
|
|
16
16
|
|
|
17
|
-
The current registry contains
|
|
17
|
+
The current registry contains 27 entries, newest first within each provider.
|
|
18
18
|
This page mirrors `src/models.rs`; use runtime discovery rather than parsing
|
|
19
19
|
this table in applications.
|
|
20
20
|
|
|
@@ -52,6 +52,7 @@ this table in applications.
|
|
|
52
52
|
| `gemini/gemini-3.6-flash` | Gemini 3.6 Flash |
|
|
53
53
|
| `gemini/gemini-3.5-flash` | Gemini 3.5 Flash |
|
|
54
54
|
| `gemini/gemini-3.5-flash-lite` | Gemini 3.5 Flash Lite |
|
|
55
|
+
| `gemini/gemini-3.1-flash-lite` | Gemini 3.1 Flash Lite |
|
|
55
56
|
|
|
56
57
|
### xAI
|
|
57
58
|
|
|
@@ -14,6 +14,8 @@ Spec-faithful TypedDicts live in ``llmshim.types`` (also re-exported here) for
|
|
|
14
14
|
static type-checking of requests, responses, and stream events.
|
|
15
15
|
"""
|
|
16
16
|
|
|
17
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
18
|
+
|
|
17
19
|
from llmshim import types
|
|
18
20
|
from llmshim._client import LlmShimError, chat, configure, health, models, stream
|
|
19
21
|
from llmshim.types import (
|
|
@@ -55,4 +57,7 @@ __all__ = [
|
|
|
55
57
|
"ToolCall",
|
|
56
58
|
"Usage",
|
|
57
59
|
]
|
|
58
|
-
|
|
60
|
+
try:
|
|
61
|
+
__version__ = version("llmshim")
|
|
62
|
+
except PackageNotFoundError: # not installed (e.g. running from source tree)
|
|
63
|
+
__version__ = "0.0.0"
|
|
@@ -89,6 +89,7 @@ def configure(
|
|
|
89
89
|
anthropic: Optional[str] = None,
|
|
90
90
|
gemini: Optional[str] = None,
|
|
91
91
|
xai: Optional[str] = None,
|
|
92
|
+
openrouter: Optional[str] = None,
|
|
92
93
|
) -> None:
|
|
93
94
|
"""Configure API keys. Writes to ~/.llmshim/config.toml.
|
|
94
95
|
|
|
@@ -98,6 +99,19 @@ def configure(
|
|
|
98
99
|
Usage:
|
|
99
100
|
import llmshim
|
|
100
101
|
llmshim.configure(anthropic="sk-ant-...", openai="sk-...")
|
|
102
|
+
|
|
103
|
+
Self-hosted servers (vLLM, SGLang) are configured via environment
|
|
104
|
+
variables — NOT config.toml — which the auto-spawned proxy inherits from
|
|
105
|
+
the Python process. Set them before your first call, e.g.::
|
|
106
|
+
|
|
107
|
+
import os
|
|
108
|
+
os.environ["VLLM_BASE_URL"] = "http://localhost:8000/v1"
|
|
109
|
+
os.environ["VLLM_API_KEY"] = "..." # optional
|
|
110
|
+
os.environ["SGLANG_BASE_URL"] = "http://localhost:30000/v1"
|
|
111
|
+
os.environ["SGLANG_API_KEY"] = "..." # optional
|
|
112
|
+
|
|
113
|
+
Then address them via the model string, e.g. ``vllm/<served-model>`` or
|
|
114
|
+
``sglang/<served-model>``.
|
|
101
115
|
"""
|
|
102
116
|
config_dir = Path.home() / ".llmshim"
|
|
103
117
|
config_path = config_dir / "config.toml"
|
|
@@ -125,6 +139,8 @@ def configure(
|
|
|
125
139
|
keys["gemini"] = gemini
|
|
126
140
|
if xai is not None:
|
|
127
141
|
keys["xai"] = xai
|
|
142
|
+
if openrouter is not None:
|
|
143
|
+
keys["openrouter"] = openrouter
|
|
128
144
|
|
|
129
145
|
# Write back
|
|
130
146
|
config_dir.mkdir(parents=True, exist_ok=True)
|
|
@@ -152,6 +168,8 @@ def configure(
|
|
|
152
168
|
os.environ["GEMINI_API_KEY"] = gemini
|
|
153
169
|
if xai:
|
|
154
170
|
os.environ["XAI_API_KEY"] = xai
|
|
171
|
+
if openrouter:
|
|
172
|
+
os.environ["OPENROUTER_API_KEY"] = openrouter
|
|
155
173
|
|
|
156
174
|
# If server is already running, it won't pick up new keys until restart.
|
|
157
175
|
# Force restart on next call.
|