llmshim 0.3.3__tar.gz → 0.3.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (124) hide show
  1. {llmshim-0.3.3 → llmshim-0.3.5}/CLAUDE.md +25 -1
  2. {llmshim-0.3.3 → llmshim-0.3.5}/Cargo.lock +1 -1
  3. {llmshim-0.3.3 → llmshim-0.3.5}/Cargo.toml +11 -2
  4. {llmshim-0.3.3 → llmshim-0.3.5}/PKG-INFO +81 -5
  5. {llmshim-0.3.3 → llmshim-0.3.5}/README.md +1 -1
  6. llmshim-0.3.5/benchmarks/gateway_loadtest.rs +203 -0
  7. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/guides/reasoning.md +1 -1
  8. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/models.md +2 -1
  9. {llmshim-0.3.3 → llmshim-0.3.5}/llmshim/__init__.py +6 -1
  10. {llmshim-0.3.3 → llmshim-0.3.5}/llmshim/_client.py +18 -0
  11. llmshim-0.3.5/src/gateway/auth.rs +199 -0
  12. llmshim-0.3.5/src/gateway/distributed.rs +958 -0
  13. llmshim-0.3.5/src/gateway/http.rs +571 -0
  14. llmshim-0.3.5/src/gateway/idempotency.rs +75 -0
  15. llmshim-0.3.5/src/gateway/metrics.rs +291 -0
  16. llmshim-0.3.5/src/gateway/mod.rs +1333 -0
  17. llmshim-0.3.5/src/gateway/quota.rs +147 -0
  18. {llmshim-0.3.3 → llmshim-0.3.5}/src/lib.rs +3 -0
  19. {llmshim-0.3.3 → llmshim-0.3.5}/src/main.rs +124 -1
  20. {llmshim-0.3.3 → llmshim-0.3.5}/src/models.rs +18 -0
  21. {llmshim-0.3.3 → llmshim-0.3.5}/src/providers/gemini.rs +8 -7
  22. {llmshim-0.3.3 → llmshim-0.3.5}/src/proxy/error.rs +12 -0
  23. {llmshim-0.3.3 → llmshim-0.3.5}/src/proxy/mod.rs +2 -2
  24. {llmshim-0.3.3 → llmshim-0.3.5}/src/proxy/ratelimit.rs +3 -0
  25. {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_gemini.rs +20 -0
  26. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_gemini.rs +14 -6
  27. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_models.rs +1 -1
  28. {llmshim-0.3.3 → llmshim-0.3.5}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
  29. {llmshim-0.3.3 → llmshim-0.3.5}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
  30. {llmshim-0.3.3 → llmshim-0.3.5}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  31. {llmshim-0.3.3 → llmshim-0.3.5}/.github/workflows/pages.yml +0 -0
  32. {llmshim-0.3.3 → llmshim-0.3.5}/.github/workflows/release.yml +0 -0
  33. {llmshim-0.3.3 → llmshim-0.3.5}/.gitignore +0 -0
  34. {llmshim-0.3.3 → llmshim-0.3.5}/CODE_OF_CONDUCT.md +0 -0
  35. {llmshim-0.3.3 → llmshim-0.3.5}/CONTRIBUTING.md +0 -0
  36. {llmshim-0.3.3 → llmshim-0.3.5}/LICENSE-APACHE +0 -0
  37. {llmshim-0.3.3 → llmshim-0.3.5}/LICENSE-MIT +0 -0
  38. {llmshim-0.3.3 → llmshim-0.3.5}/NOTICE +0 -0
  39. {llmshim-0.3.3 → llmshim-0.3.5}/SECURITY.md +0 -0
  40. {llmshim-0.3.3 → llmshim-0.3.5}/benchmarks/bench.rs +0 -0
  41. {llmshim-0.3.3 → llmshim-0.3.5}/benchmarks/bench_python.py +0 -0
  42. {llmshim-0.3.3 → llmshim-0.3.5}/benchmarks/loadtest.rs +0 -0
  43. {llmshim-0.3.3 → llmshim-0.3.5}/docs/.gitignore +0 -0
  44. {llmshim-0.3.3 → llmshim-0.3.5}/docs/book.toml +0 -0
  45. {llmshim-0.3.3 → llmshim-0.3.5}/docs/mermaid-init.js +0 -0
  46. {llmshim-0.3.3 → llmshim-0.3.5}/docs/mermaid.min.js +0 -0
  47. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/SUMMARY.md +0 -0
  48. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/concepts/contracts.md +0 -0
  49. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/concepts/conversations.md +0 -0
  50. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/concepts/portability.md +0 -0
  51. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/concepts/routing.md +0 -0
  52. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/concepts/translation-flow.md +0 -0
  53. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/guides/fallbacks.md +0 -0
  54. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/guides/images.md +0 -0
  55. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/guides/native-controls.md +0 -0
  56. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/guides/streaming.md +0 -0
  57. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/guides/tools.md +0 -0
  58. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/introduction.md +0 -0
  59. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/proxy/deployment.md +0 -0
  60. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/proxy/http-api.md +0 -0
  61. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/proxy/scaling.md +0 -0
  62. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/api.md +0 -0
  63. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/cli.md +0 -0
  64. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/configuration.md +0 -0
  65. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/errors.md +0 -0
  66. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/providers.md +0 -0
  67. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/request-fields.md +0 -0
  68. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/reference/surfaces.md +0 -0
  69. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/start/choose.md +0 -0
  70. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/start/cli.md +0 -0
  71. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/start/clients.md +0 -0
  72. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/start/configure.md +0 -0
  73. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/start/proxy.md +0 -0
  74. {llmshim-0.3.3 → llmshim-0.3.5}/docs/src/start/rust.md +0 -0
  75. {llmshim-0.3.3 → llmshim-0.3.5}/examples/chat.rs +0 -0
  76. {llmshim-0.3.3 → llmshim-0.3.5}/examples/stream.rs +0 -0
  77. {llmshim-0.3.3 → llmshim-0.3.5}/llmshim/_server.py +0 -0
  78. {llmshim-0.3.3 → llmshim-0.3.5}/llmshim/types.py +0 -0
  79. {llmshim-0.3.3 → llmshim-0.3.5}/pyproject.toml +0 -0
  80. {llmshim-0.3.3 → llmshim-0.3.5}/src/client.rs +0 -0
  81. {llmshim-0.3.3 → llmshim-0.3.5}/src/config.rs +0 -0
  82. {llmshim-0.3.3 → llmshim-0.3.5}/src/env.rs +0 -0
  83. {llmshim-0.3.3 → llmshim-0.3.5}/src/error.rs +0 -0
  84. {llmshim-0.3.3 → llmshim-0.3.5}/src/fallback.rs +0 -0
  85. {llmshim-0.3.3 → llmshim-0.3.5}/src/log.rs +0 -0
  86. {llmshim-0.3.3 → llmshim-0.3.5}/src/provider.rs +0 -0
  87. {llmshim-0.3.3 → llmshim-0.3.5}/src/providers/anthropic.rs +0 -0
  88. {llmshim-0.3.3 → llmshim-0.3.5}/src/providers/mod.rs +0 -0
  89. {llmshim-0.3.3 → llmshim-0.3.5}/src/providers/openai.rs +0 -0
  90. {llmshim-0.3.3 → llmshim-0.3.5}/src/providers/openai_compat.rs +0 -0
  91. {llmshim-0.3.3 → llmshim-0.3.5}/src/providers/openrouter.rs +0 -0
  92. {llmshim-0.3.3 → llmshim-0.3.5}/src/providers/xai.rs +0 -0
  93. {llmshim-0.3.3 → llmshim-0.3.5}/src/proxy/convert.rs +0 -0
  94. {llmshim-0.3.3 → llmshim-0.3.5}/src/proxy/handlers.rs +0 -0
  95. {llmshim-0.3.3 → llmshim-0.3.5}/src/proxy/types.rs +0 -0
  96. {llmshim-0.3.3 → llmshim-0.3.5}/src/router.rs +0 -0
  97. {llmshim-0.3.3 → llmshim-0.3.5}/src/vision.rs +0 -0
  98. {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration.rs +0 -0
  99. {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_fallback.rs +0 -0
  100. {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_gemini_tools.rs +0 -0
  101. {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_long_context.rs +0 -0
  102. {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_multimodel.rs +0 -0
  103. {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_openrouter.rs +0 -0
  104. {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_proxy.rs +0 -0
  105. {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_sglang.rs +0 -0
  106. {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_thinking.rs +0 -0
  107. {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_tool_roundtrip.rs +0 -0
  108. {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_vision.rs +0 -0
  109. {llmshim-0.3.3 → llmshim-0.3.5}/tests/integration_xai.rs +0 -0
  110. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_anthropic.rs +0 -0
  111. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_fallback.rs +0 -0
  112. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_fast_mode.rs +0 -0
  113. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_log.rs +0 -0
  114. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_multimodel.rs +0 -0
  115. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_openai.rs +0 -0
  116. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_openai_compat.rs +0 -0
  117. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_openrouter.rs +0 -0
  118. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_proxy.rs +0 -0
  119. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_proxy_convert.rs +0 -0
  120. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_router.rs +0 -0
  121. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_sse.rs +0 -0
  122. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_tools.rs +0 -0
  123. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_vision.rs +0 -0
  124. {llmshim-0.3.3 → llmshim-0.3.5}/tests/unit_xai.rs +0 -0
@@ -14,7 +14,7 @@ This is a public crate on crates.io. Do NOT make breaking changes to `pub` items
14
14
 
15
15
  - **OpenAI:** `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-5.5`, `gpt-5.5-pro`, `gpt-5.4`, `gpt-5.4-pro`, `gpt-5.4-mini`, `gpt-5.4-nano`
16
16
  - **Anthropic:** `claude-opus-5`, `claude-opus-4-8`, `claude-sonnet-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-sonnet-4-6`, `claude-haiku-4-5-20251001`
17
- - **Gemini:** `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite`
17
+ - **Gemini:** `gemini-3.8-flash`, `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite`, `gemini-3.1-flash-lite`
18
18
  - **xAI:** `grok-4.6`, `grok-4.5`, `grok-4.3`, `grok-4.20-multi-agent-beta-0309`, `grok-4.20-beta-0309-reasoning`, `grok-4.20-beta-0309-non-reasoning`
19
19
  - **OpenRouter:** not enumerated (huge/dynamic catalog) — any `openrouter/<vendor>/<model>` slug routes through, e.g. `openrouter/anthropic/claude-sonnet-4.5`.
20
20
  - **vLLM / SGLang:** not enumerated (self-hosted) — any `vllm/<served-model>` or `sglang/<served-model>` routes through to the configured server, e.g. `sglang/Qwen/Qwen3.6-35B-A3B-FP8`.
@@ -147,6 +147,30 @@ Env config (all optional, safe defaults; when no RPM/TPM limits are set the limi
147
147
 
148
148
  Topologies: **sidecar / zero-infra** (default in-memory limiter — set limits to `global / N` when running N replicas) vs. **Redis-coordinated fleet** (one shared global limit across all replicas).
149
149
 
150
+ ### Experimental: priority-queue gateway (`src/gateway/`, feature `gateway`)
151
+
152
+ An experimental scheduler that inverts the proxy's admission model: instead of *rejecting* when a provider's token bucket is empty, it **enqueues** each request into a per-provider priority queue and dispatches when capacity frees — ordered by **priority tier** (paying customer > free) then **FIFO** within a tier. Built for a fleet doing thousands of req/s that can't fire every call the instant it arrives. `gateway = ["proxy"]` (reuses the `RateLimiter` token buckets, `Backpressure`, and the proxy's request/response converters — `convert`/`error` are `pub(crate)` for this).
153
+
154
+ - **`Scheduler`** (`src/gateway/mod.rs`) owns one lane (queue + dispatcher task) per provider, so a rate-limited OpenAI queue never blocks a ready Anthropic one. `submit()` (unary) / `submit_stream()` (SSE) enqueue + await; a client disconnect / `max_wait` timeout cancels the queued job so it never burns a token.
155
+ - **Dispatcher** is event/timer-driven: pops the highest-priority job, acquires a concurrency slot *then* a rate token (so a saturated semaphore never wastes a token), and on a rate-limit miss **requeues** (preserving priority) and sleeps for exactly the `RetryAfter` the limiter reports — waking early on new work. No busy-wait, no `RateLimiter` changes. `max_wait` bounds **queue residence only** (a `started` signal releases it at dispatch), never the upstream call.
156
+ - **Fairness/aging** (`InMemoryQueue`): per-tier FIFO deques; dequeue picks the front with the highest *effective* priority = `tier + min(max_boost, wait/aging_step)`, so a starved low tier eventually overtakes a high-tier flood. Defaults (`aging_step` 5s, `max_boost` 16) keep normal load strictly priority-then-FIFO. `O(#tiers)` dequeue.
157
+ - **Streaming** (`submit_stream`): a `Delivery::Stream` job hands back an `mpsc` receiver once the upstream opens; the dispatcher forwards chunks and holds the concurrency permit for the whole stream (freed when it ends / the client disconnects). HTTP `POST /v1/chat/stream` (and `stream:true`) bridge it to SSE.
158
+ - **`RequestQueue`** trait is the in-process pluggable backend (in-memory default). Cross-process is a *separate* seam, not this trait (a `Job` holds a `oneshot`).
159
+ - **Distributed** (`src/gateway/distributed.rs`, feature `gateway-redis` = `gateway` + `redis-coordination`): a fleet shares a Redis **ZSET priority queue** per provider and a Redis **pub/sub response bus** keyed by request-id, so any instance dispatches any job and routes the result back to the origin's HTTP connection. Reuses the shared `RedisRateLimiter` for fleet-wide rate coordination. `llmshim gateway` auto-selects distributed mode when `LLMSHIM_REDIS_URL` is set and the binary has `gateway-redis`. Supports **unary + streaming** (typed `BusMessage`s over the channel). Full-parity with the in-memory lane:
160
+ - **Aging** via a *virtual-deadline* score `enqueue_ms − tier·aging_step` popped with `ZPOPMIN` — one static score gives priority, FIFO, and anti-starvation aging with no re-scoring.
161
+ - **At-least-once** via an atomic **lease** (Lua: `ZPOPMIN` queue → `processing` ZSET with a visibility deadline + recorded score) + ack/release + a background **reaper** (`reap_once`, also public for external cron) that requeues expired leases; streams refresh their lease as they run. A redelivered job may run twice (idempotent upstream calls). Env: `LLMSHIM_GATEWAY_LEASE_TIMEOUT_MS`, `LLMSHIM_GATEWAY_REQUEST_TIMEOUT_MS`.
162
+ - **HTTP** (`src/gateway/http.rs`): serves the proxy's `POST /v1/chat` + `/v1/chat/stream` contract routed through the scheduler; tier from the **`x-llmshim-priority`** header (uint, default 0). `RealDispatch` calls `completion_with_logger`/`stream`, mapping an upstream 429 → bucket penalty. Run: `llmshim gateway` (needs `--features gateway`, or `gateway-redis` for a fleet). Env: `LLMSHIM_GATEWAY_MAX_WAIT_MS`, `LLMSHIM_GATEWAY_QUEUE_DEPTH`, `LLMSHIM_GATEWAY_MAX_CONCURRENCY`, `LLMSHIM_GATEWAY_AGING_STEP_MS`, `LLMSHIM_GATEWAY_MAX_BOOST`, `LLMSHIM_GATEWAY_REQUEST_TIMEOUT_MS`, `LLMSHIM_GATEWAY_LEASE_TIMEOUT_MS`, `LLMSHIM_GATEWAY_MAX_ATTEMPTS`, `LLMSHIM_GATEWAY_IDEMPOTENCY_TTL_SECS`, plus all the proxy rate-limit vars.
163
+
164
+ #### Production hardening (gateway)
165
+
166
+ - **Auth + tier from identity** (`src/gateway/auth.rs`): set `LLMSHIM_GATEWAY_KEYS_FILE` to a JSON map `{ "<api-key>": {"tenant","tier","rpm","tpm"} }`. Then `Authorization: Bearer <key>` is **required** and tier/tenant come from the key — the client `x-llmshim-priority` header is ignored (closes the queue-jump exploit); missing/invalid key → 401. Unset ⇒ open dev mode (header trusted, tenant `anonymous`).
167
+ - **Per-tenant quotas** (`src/gateway/quota.rs`): per-`(tenant,provider)` RPM/TPM token buckets from the caller's identity, enforced in the HTTP layer (proactive 429 + `Retry-After`) on top of the global provider limits — one tenant can't monopolize shared capacity.
168
+ - **Idempotency** (`src/gateway/idempotency.rs`): `Idempotency-Key` header → cache-after-completion so a client retry returns the first result (no second billed call); in-memory for local, Redis for the fleet. Distributed at-least-once also has a **done-marker** (a completed job that gets redelivered is skipped) and a **dead-letter queue** after `max_attempts` (poison-job guard).
169
+ - **Observability**: `GET /metrics` (Prometheus, `src/gateway/metrics.rs`, dependency-free) — requests/dispatched/rejected counters, in-flight + queue-depth gauges, queue-wait + upstream-latency histograms; `GET /ready` (503 if Redis is down in distributed mode); `GET /v1/gateway/stats` (queue depths, dead-letter counts); every response carries `x-request-id`.
170
+ - **Graceful shutdown**: `gateway`/`proxy` drain in-flight on SIGTERM/Ctrl-C.
171
+ - **Load test**: `cargo run --release --features gateway --example gateway_loadtest` (throughput + zero-loss + priority-under-load asserts). **Deploy**: `Dockerfile` builds `--features gateway` by default (`--build-arg FEATURES=gateway-redis` for a fleet); `docker run … llmshim gateway`.
172
+ - Intentionally **not** done: per-tier (as opposed to per-tenant) queue-depth caps — the global depth cap + per-tenant quotas cover it; single-flight idempotency for *concurrent* same-key requests (only cache-after-completion).
173
+
150
174
  ## Client libraries (`clients/`)
151
175
 
152
176
  Thin clients that speak the proxy's HTTP API. They are faithful to the OpenAPI contract in `api/openapi.yaml` — when you change the proxy's request/response shapes, update that spec and keep the clients in sync. All publish in lockstep with the crate version on every release.
@@ -958,7 +958,7 @@ checksum = "6373607a59f0be73a39b6fe456b8192fcc3585f602af20751600e974dd455e77"
958
958
 
959
959
  [[package]]
960
960
  name = "llmshim"
961
- version = "0.3.3"
961
+ version = "0.3.5"
962
962
  dependencies = [
963
963
  "async-stream",
964
964
  "async-trait",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "llmshim"
3
- version = "0.3.3"
3
+ version = "0.3.5"
4
4
  edition = "2021"
5
5
  description = "Blazing fast LLM API translation layer in pure Rust"
6
6
  license = "MIT OR Apache-2.0"
@@ -8,12 +8,16 @@ repository = "https://github.com/sanjay920/llmshim"
8
8
  homepage = "https://github.com/sanjay920/llmshim"
9
9
  keywords = ["llm", "openai", "anthropic", "gemini", "ai"]
10
10
  categories = ["api-bindings", "web-programming"]
11
- exclude = ["clients/", "api/", ".claude/", "Dockerfile", ".dockerignore"]
11
+ exclude = ["clients/", "api/", ".claude/", "Dockerfile", ".dockerignore", "gateway-test-ui.html"]
12
12
  readme = "README.md"
13
13
 
14
14
  [features]
15
15
  default = []
16
16
  proxy = ["dep:axum", "dep:tower", "dep:tower-http", "dep:async-stream", "dep:async-trait"]
17
+ # Experimental priority-queue gateway. Reuses the proxy's RateLimiter/Backpressure.
18
+ gateway = ["proxy"]
19
+ # Gateway + Redis-backed distributed queue & response bus (cross-instance fleet).
20
+ gateway-redis = ["gateway", "redis-coordination"]
17
21
  # Opt-in distributed rate-limit coordination via Redis. Keeps the default proxy
18
22
  # binary lean — redis is only pulled in when this feature is explicitly enabled.
19
23
  redis-coordination = ["proxy", "dep:redis"]
@@ -67,3 +71,8 @@ path = "benchmarks/bench.rs"
67
71
  name = "loadtest"
68
72
  path = "benchmarks/loadtest.rs"
69
73
  required-features = ["proxy"]
74
+
75
+ [[example]]
76
+ name = "gateway_loadtest"
77
+ path = "benchmarks/gateway_loadtest.rs"
78
+ required-features = ["gateway"]
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llmshim
3
- Version: 0.3.3
3
+ Version: 0.3.5
4
4
  Classifier: Development Status :: 4 - Beta
5
5
  Classifier: Intended Audience :: Developers
6
6
  Classifier: License :: OSI Approved :: MIT License
@@ -37,11 +37,30 @@ llmshim.configure(
37
37
  openai="sk-...",
38
38
  gemini="AIza...",
39
39
  xai="xai-...",
40
+ openrouter="sk-or-...",
40
41
  )
41
42
  ```
42
43
 
43
44
  Or from the CLI: `llmshim configure`
44
45
 
46
+ ### Self-hosted servers (vLLM / SGLang)
47
+
48
+ vLLM and SGLang are configured via **environment variables** (not
49
+ `config.toml`) — the auto-spawned proxy inherits them from your Python
50
+ process. Set them before your first call:
51
+
52
+ ```python
53
+ import os
54
+
55
+ os.environ["VLLM_BASE_URL"] = "http://localhost:8000/v1"
56
+ os.environ["VLLM_API_KEY"] = "..." # optional
57
+ os.environ["SGLANG_BASE_URL"] = "http://localhost:30000/v1"
58
+ os.environ["SGLANG_API_KEY"] = "..." # optional
59
+ ```
60
+
61
+ Then address them via the model string — `vllm/<served-model>` or
62
+ `sglang/<served-model>` (see the model table below).
63
+
45
64
  ## Chat
46
65
 
47
66
  ```python
@@ -106,17 +125,63 @@ print(f"GPT: {r2['message']['content']}")
106
125
 
107
126
  ## Reasoning / Thinking
108
127
 
128
+ Two provider-agnostic knobs control reasoning; both are clamped to the
129
+ nearest tier the target model supports:
130
+
131
+ - `reasoning_effort` — `"none"`, `"low"`, `"medium"`, `"high"`, `"xhigh"`, or `"max"`
132
+ - `reasoning_mode` — `"standard"` (default) or `"pro"` (requests substantially
133
+ more model work; native on OpenAI gpt-5.6/-pro, emulated as an effort bump
134
+ elsewhere)
135
+
109
136
  ```python
110
137
  resp = llmshim.chat(
111
- "claude-sonnet-4-6",
138
+ "claude-sonnet-5",
112
139
  "Solve: x^2 - 5x + 6 = 0",
113
140
  max_tokens=4000,
114
141
  reasoning_effort="high",
142
+ reasoning_mode="pro",
115
143
  )
116
144
  print(resp["reasoning"]) # thinking content
117
145
  print(resp["message"]["content"]) # answer
118
146
  ```
119
147
 
148
+ For full native control, bypass the unified mapping with a namespaced
149
+ `provider_config` (see below), e.g.
150
+ `provider_config={"x-anthropic": {"thinking": {"type": "enabled", "budget_tokens": 4000}}}`.
151
+
152
+ ## Provider-Specific Controls (`provider_config`)
153
+
154
+ `provider_config` merges into the request **root** and carries anything the
155
+ unified `config` doesn't cover. Native provider controls MUST be **namespaced**
156
+ per provider (`x-anthropic`, `x-openai`, `x-gemini`, `x-openrouter`, `x-vllm`,
157
+ `x-sglang`); it also carries the top-level `tools`, `response_format`, and
158
+ `reasoning_summary` keys.
159
+
160
+ ```python
161
+ resp = llmshim.chat(
162
+ "anthropic/claude-sonnet-5",
163
+ "Solve this step by step: 17 * 23",
164
+ max_tokens=4000,
165
+ provider_config={
166
+ # native Anthropic extended-thinking control
167
+ "x-anthropic": {"thinking": {"type": "enabled", "budget_tokens": 4000}},
168
+ # structured output
169
+ "response_format": {"type": "json_object"},
170
+ },
171
+ )
172
+ ```
173
+
174
+ OpenRouter routing preferences use the `x-openrouter` namespace:
175
+
176
+ ```python
177
+ resp = llmshim.chat(
178
+ "openrouter/anthropic/claude-sonnet-4.5",
179
+ "Hello",
180
+ max_tokens=200,
181
+ provider_config={"x-openrouter": {"provider": {"sort": "throughput"}}},
182
+ )
183
+ ```
184
+
120
185
  ## Tool Use / Function Calling
121
186
 
122
187
  ```python
@@ -147,7 +212,7 @@ resp = llmshim.chat(
147
212
  "anthropic/claude-sonnet-4-6",
148
213
  "Hello",
149
214
  max_tokens=100,
150
- fallback=["openai/gpt-5.5", "gemini/gemini-3-flash-preview"],
215
+ fallback=["openai/gpt-5.6-sol", "gemini/gemini-3.5-flash"],
151
216
  )
152
217
  ```
153
218
 
@@ -220,10 +285,21 @@ billed provider calls; run it only when you deliberately want to hit real APIs.
220
285
 
221
286
  | Provider | Models |
222
287
  |----------|--------|
223
- | OpenAI | `gpt-5.5`, `gpt-5.5-pro`, `gpt-5.4`, `gpt-5.4-pro`, `gpt-5.4-mini`, `gpt-5.4-nano` |
288
+ | OpenAI | `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-5.5`, `gpt-5.5-pro`, `gpt-5.4`, `gpt-5.4-pro`, `gpt-5.4-mini`, `gpt-5.4-nano` |
224
289
  | Anthropic | `claude-opus-5`, `claude-opus-4-8`, `claude-sonnet-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-sonnet-4-6`, `claude-haiku-4-5-20251001` |
225
- | Gemini | `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite` |
290
+ | Gemini | `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite`, `gemini-3.1-flash-lite` |
226
291
  | xAI | `grok-4.6`, `grok-4.5`, `grok-4.3`, `grok-4.20-multi-agent-beta-0309`, `grok-4.20-beta-0309-reasoning`, `grok-4.20-beta-0309-non-reasoning` |
227
292
 
228
293
  Call `llmshim.models()` for the live list filtered to your configured providers.
229
294
 
295
+ ### OpenRouter & self-hosted (vLLM / SGLang)
296
+
297
+ These providers are addressed by the model string plus environment variables —
298
+ any model the upstream serves is reachable, so they aren't in the table above.
299
+
300
+ | Provider | Address as | Env vars | Native controls |
301
+ |----------|-----------|----------|-----------------|
302
+ | OpenRouter | `openrouter/<vendor>/<model>` (e.g. `openrouter/anthropic/claude-sonnet-4.5`) | `OPENROUTER_API_KEY` (or `llmshim.configure(openrouter=...)`) | `provider_config={"x-openrouter": {...}}` (`provider`, `models`, `transforms`) |
303
+ | vLLM | `vllm/<served-model>` | `VLLM_BASE_URL` (+ optional `VLLM_API_KEY`) | `provider_config={"x-vllm": {...}}` |
304
+ | SGLang | `sglang/<served-model>` | `SGLANG_BASE_URL` (+ optional `SGLANG_API_KEY`) | `provider_config={"x-sglang": {...}}` |
305
+
@@ -364,7 +364,7 @@ Standard library only. Full docs: [`clients/ruby/README.md`](clients/ruby/README
364
364
  |----------|--------|-------------------|
365
365
  | **OpenAI** | `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-5.5`, `gpt-5.5-pro`, `gpt-5.4`, `gpt-5.4-pro`, `gpt-5.4-mini`, `gpt-5.4-nano` | Yes (summaries) |
366
366
  | **Anthropic** | `claude-opus-5`, `claude-opus-4-8`, `claude-sonnet-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-sonnet-4-6`, `claude-haiku-4-5-20251001` | Yes (full thinking) |
367
- | **Google Gemini** | `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite` | Yes (thought summaries) |
367
+ | **Google Gemini** | `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite`, `gemini-3.1-flash-lite` | Yes (thought summaries) |
368
368
  | **xAI** | `grok-4.6`, `grok-4.5`, `grok-4.3`, `grok-4.20-multi-agent-beta-0309`, `grok-4.20-beta-0309-reasoning`, `grok-4.20-beta-0309-non-reasoning` | No (hidden) |
369
369
 
370
370
  Use a bare model name (auto-detected by prefix) or an explicit `provider/model` string.
@@ -0,0 +1,203 @@
1
+ //! Gateway load test — drives the in-process priority [`Scheduler`] with tens of
2
+ //! thousands of concurrent mixed-priority requests against a fake (no-network)
3
+ //! dispatcher, so it exercises the scheduler itself, not any provider.
4
+ //!
5
+ //! Run: `cargo run --release --features gateway --example gateway_loadtest`
6
+ //!
7
+ //! Reports throughput + latency percentiles and asserts two invariants:
8
+ //! 1. **Zero loss** under a burst far larger than the concurrency limit.
9
+ //! 2. **Priority holds under load** — high-tier p50 latency ≪ low-tier p50
10
+ //! when the concurrency slots are saturated.
11
+ //! Exits non-zero if either invariant fails.
12
+
13
+ use std::sync::atomic::{AtomicU64, Ordering};
14
+ use std::sync::Arc;
15
+ use std::time::{Duration, Instant};
16
+
17
+ use async_trait::async_trait;
18
+ use llmshim::gateway::{Dispatch, DispatchError, GatewayConfig, GatewayRequest, Scheduler};
19
+ use llmshim::proxy::ratelimit::{InMemoryRateLimiter, RateLimitConfig};
20
+ use serde_json::{json, Value};
21
+
22
+ /// A no-network dispatcher with a fixed simulated latency.
23
+ struct FakeDispatch {
24
+ latency: Duration,
25
+ served: Arc<AtomicU64>,
26
+ }
27
+
28
+ #[async_trait]
29
+ impl Dispatch for FakeDispatch {
30
+ async fn dispatch(&self, _provider: &str, payload: Value) -> Result<Value, DispatchError> {
31
+ if !self.latency.is_zero() {
32
+ tokio::time::sleep(self.latency).await;
33
+ }
34
+ self.served.fetch_add(1, Ordering::Relaxed);
35
+ Ok(payload)
36
+ }
37
+ }
38
+
39
+ fn unlimited() -> Arc<InMemoryRateLimiter> {
40
+ Arc::new(InMemoryRateLimiter::new(RateLimitConfig::default()))
41
+ }
42
+
43
+ fn percentile(sorted_us: &[u128], p: f64) -> f64 {
44
+ if sorted_us.is_empty() {
45
+ return 0.0;
46
+ }
47
+ let idx = ((sorted_us.len() as f64 - 1.0) * p).round() as usize;
48
+ sorted_us[idx] as f64 / 1000.0 // → ms
49
+ }
50
+
51
+ fn report(name: &str, mut latencies_us: Vec<u128>, wall: Duration) {
52
+ latencies_us.sort_unstable();
53
+ let n = latencies_us.len();
54
+ let rps = n as f64 / wall.as_secs_f64();
55
+ println!(
56
+ "{name}: {n} reqs in {:.2}s = {rps:.0} req/s | p50={:.1}ms p95={:.1}ms p99={:.1}ms max={:.1}ms",
57
+ wall.as_secs_f64(),
58
+ percentile(&latencies_us, 0.50),
59
+ percentile(&latencies_us, 0.95),
60
+ percentile(&latencies_us, 0.99),
61
+ percentile(&latencies_us, 1.00),
62
+ );
63
+ }
64
+
65
+ #[tokio::main(flavor = "multi_thread")]
66
+ async fn main() {
67
+ let mut failures = 0;
68
+
69
+ // ---- Scenario A: throughput + zero loss under a large burst ------------
70
+ {
71
+ let served = Arc::new(AtomicU64::new(0));
72
+ let config = GatewayConfig {
73
+ max_queue_depth: 500_000,
74
+ max_concurrency_per_provider: 512,
75
+ max_wait: Duration::from_secs(60),
76
+ ..Default::default()
77
+ };
78
+ let sched = Scheduler::new(
79
+ config,
80
+ unlimited(),
81
+ Arc::new(FakeDispatch {
82
+ latency: Duration::from_millis(2),
83
+ served: served.clone(),
84
+ }),
85
+ );
86
+
87
+ let n = 20_000u64;
88
+ let ok = Arc::new(AtomicU64::new(0));
89
+ let errs = Arc::new(AtomicU64::new(0));
90
+ let start = Instant::now();
91
+ let mut handles = Vec::with_capacity(n as usize);
92
+ for i in 0..n {
93
+ let s = sched.clone();
94
+ let ok = ok.clone();
95
+ let errs = errs.clone();
96
+ handles.push(tokio::spawn(async move {
97
+ let t = Instant::now();
98
+ let r = s
99
+ .submit(GatewayRequest {
100
+ provider: "bench".into(),
101
+ tier: (i % 3) as u8,
102
+ permits: 1,
103
+ payload: json!({ "id": i }),
104
+ })
105
+ .await;
106
+ match r {
107
+ Ok(_) => ok.fetch_add(1, Ordering::Relaxed),
108
+ Err(_) => errs.fetch_add(1, Ordering::Relaxed),
109
+ };
110
+ t.elapsed().as_micros()
111
+ }));
112
+ }
113
+ let mut lat = Vec::with_capacity(n as usize);
114
+ for h in handles {
115
+ lat.push(h.await.unwrap());
116
+ }
117
+ let wall = start.elapsed();
118
+ report("throughput", lat, wall);
119
+ let (ok, errs) = (ok.load(Ordering::Relaxed), errs.load(Ordering::Relaxed));
120
+ println!(
121
+ " ok={ok} errors={errs} served={}",
122
+ served.load(Ordering::Relaxed)
123
+ );
124
+ if errs != 0 || ok != n {
125
+ eprintln!(" FAIL: expected {n} successful, zero lost — got ok={ok} errs={errs}");
126
+ failures += 1;
127
+ } else {
128
+ println!(" PASS: zero loss under burst");
129
+ }
130
+ }
131
+
132
+ // ---- Scenario B: priority holds when concurrency is saturated ----------
133
+ {
134
+ let served = Arc::new(AtomicU64::new(0));
135
+ let config = GatewayConfig {
136
+ max_queue_depth: 100_000,
137
+ max_concurrency_per_provider: 8, // the bottleneck
138
+ max_wait: Duration::from_secs(120),
139
+ ..Default::default()
140
+ };
141
+ let sched = Scheduler::new(
142
+ config,
143
+ unlimited(),
144
+ Arc::new(FakeDispatch {
145
+ latency: Duration::from_millis(25),
146
+ served: served.clone(),
147
+ }),
148
+ );
149
+
150
+ let per_tier = 800u64;
151
+ let low_lat = Arc::new(std::sync::Mutex::new(Vec::<u128>::new()));
152
+ let high_lat = Arc::new(std::sync::Mutex::new(Vec::<u128>::new()));
153
+ let start = Instant::now();
154
+ let mut handles = Vec::new();
155
+ // Interleave low- and high-tier submissions.
156
+ for i in 0..per_tier {
157
+ for (tier, bucket) in [(0u8, &low_lat), (5u8, &high_lat)] {
158
+ let s = sched.clone();
159
+ let bucket = bucket.clone();
160
+ handles.push(tokio::spawn(async move {
161
+ let t = Instant::now();
162
+ let _ = s
163
+ .submit(GatewayRequest {
164
+ provider: "bench".into(),
165
+ tier,
166
+ permits: 1,
167
+ payload: json!({ "id": i }),
168
+ })
169
+ .await;
170
+ bucket.lock().unwrap().push(t.elapsed().as_micros());
171
+ }));
172
+ }
173
+ }
174
+ for h in handles {
175
+ h.await.unwrap();
176
+ }
177
+ let wall = start.elapsed();
178
+ let low = Arc::try_unwrap(low_lat).unwrap().into_inner().unwrap();
179
+ let high = Arc::try_unwrap(high_lat).unwrap().into_inner().unwrap();
180
+ report("priority(low tier)", low.clone(), wall);
181
+ report("priority(high tier)", high.clone(), wall);
182
+ let mut lo = low;
183
+ let mut hi = high;
184
+ lo.sort_unstable();
185
+ hi.sort_unstable();
186
+ let low_p50 = percentile(&lo, 0.50);
187
+ let high_p50 = percentile(&hi, 0.50);
188
+ if high_p50 < low_p50 {
189
+ println!(" PASS: high-tier p50 {high_p50:.1}ms < low-tier p50 {low_p50:.1}ms");
190
+ } else {
191
+ eprintln!(
192
+ " FAIL: priority not honored — high p50 {high_p50:.1}ms >= low p50 {low_p50:.1}ms"
193
+ );
194
+ failures += 1;
195
+ }
196
+ }
197
+
198
+ if failures > 0 {
199
+ eprintln!("\n{failures} scenario(s) FAILED");
200
+ std::process::exit(1);
201
+ }
202
+ println!("\nAll load-test invariants held.");
203
+ }
@@ -117,7 +117,7 @@ token budget scaled from `max_tokens` and floored at 1024:
117
117
  Gemini uses the four-rung
118
118
  `generationConfig.thinkingConfig.thinkingLevel` enum:
119
119
 
120
- | unified | gemini-3.5-flash / 3.6-flash / 3.5-flash-lite | gemini-3.7-flash (and gemini-3.1-pro) |
120
+ | unified | gemini-3.5-flash / 3.6-flash / 3.5-flash-lite / 3.1-flash-lite | gemini-3.7-flash (and gemini-3.1-pro) |
121
121
  |---|---|---|
122
122
  | `none` | `minimal` (zero thinking tokens) | **`low`** (this model cannot disable thinking) |
123
123
  | `low` | `low` | `low` |
@@ -14,7 +14,7 @@ the ID and display label.
14
14
 
15
15
  ## Registered catalog
16
16
 
17
- The current registry contains 26 entries, newest first within each provider.
17
+ The current registry contains 27 entries, newest first within each provider.
18
18
  This page mirrors `src/models.rs`; use runtime discovery rather than parsing
19
19
  this table in applications.
20
20
 
@@ -52,6 +52,7 @@ this table in applications.
52
52
  | `gemini/gemini-3.6-flash` | Gemini 3.6 Flash |
53
53
  | `gemini/gemini-3.5-flash` | Gemini 3.5 Flash |
54
54
  | `gemini/gemini-3.5-flash-lite` | Gemini 3.5 Flash Lite |
55
+ | `gemini/gemini-3.1-flash-lite` | Gemini 3.1 Flash Lite |
55
56
 
56
57
  ### xAI
57
58
 
@@ -14,6 +14,8 @@ Spec-faithful TypedDicts live in ``llmshim.types`` (also re-exported here) for
14
14
  static type-checking of requests, responses, and stream events.
15
15
  """
16
16
 
17
+ from importlib.metadata import PackageNotFoundError, version
18
+
17
19
  from llmshim import types
18
20
  from llmshim._client import LlmShimError, chat, configure, health, models, stream
19
21
  from llmshim.types import (
@@ -55,4 +57,7 @@ __all__ = [
55
57
  "ToolCall",
56
58
  "Usage",
57
59
  ]
58
- __version__ = "0.1.22"
60
+ try:
61
+ __version__ = version("llmshim")
62
+ except PackageNotFoundError: # not installed (e.g. running from source tree)
63
+ __version__ = "0.0.0"
@@ -89,6 +89,7 @@ def configure(
89
89
  anthropic: Optional[str] = None,
90
90
  gemini: Optional[str] = None,
91
91
  xai: Optional[str] = None,
92
+ openrouter: Optional[str] = None,
92
93
  ) -> None:
93
94
  """Configure API keys. Writes to ~/.llmshim/config.toml.
94
95
 
@@ -98,6 +99,19 @@ def configure(
98
99
  Usage:
99
100
  import llmshim
100
101
  llmshim.configure(anthropic="sk-ant-...", openai="sk-...")
102
+
103
+ Self-hosted servers (vLLM, SGLang) are configured via environment
104
+ variables — NOT config.toml — which the auto-spawned proxy inherits from
105
+ the Python process. Set them before your first call, e.g.::
106
+
107
+ import os
108
+ os.environ["VLLM_BASE_URL"] = "http://localhost:8000/v1"
109
+ os.environ["VLLM_API_KEY"] = "..." # optional
110
+ os.environ["SGLANG_BASE_URL"] = "http://localhost:30000/v1"
111
+ os.environ["SGLANG_API_KEY"] = "..." # optional
112
+
113
+ Then address them via the model string, e.g. ``vllm/<served-model>`` or
114
+ ``sglang/<served-model>``.
101
115
  """
102
116
  config_dir = Path.home() / ".llmshim"
103
117
  config_path = config_dir / "config.toml"
@@ -125,6 +139,8 @@ def configure(
125
139
  keys["gemini"] = gemini
126
140
  if xai is not None:
127
141
  keys["xai"] = xai
142
+ if openrouter is not None:
143
+ keys["openrouter"] = openrouter
128
144
 
129
145
  # Write back
130
146
  config_dir.mkdir(parents=True, exist_ok=True)
@@ -152,6 +168,8 @@ def configure(
152
168
  os.environ["GEMINI_API_KEY"] = gemini
153
169
  if xai:
154
170
  os.environ["XAI_API_KEY"] = xai
171
+ if openrouter:
172
+ os.environ["OPENROUTER_API_KEY"] = openrouter
155
173
 
156
174
  # If server is already running, it won't pick up new keys until restart.
157
175
  # Force restart on next call.