llmshim 0.2.3__tar.gz → 0.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (115) hide show
  1. {llmshim-0.2.3 → llmshim-0.3.1}/CLAUDE.md +12 -4
  2. {llmshim-0.2.3 → llmshim-0.3.1}/Cargo.lock +1 -1
  3. {llmshim-0.2.3 → llmshim-0.3.1}/Cargo.toml +1 -1
  4. {llmshim-0.2.3 → llmshim-0.3.1}/PKG-INFO +1 -1
  5. {llmshim-0.2.3 → llmshim-0.3.1}/README.md +20 -1
  6. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/reference/configuration.md +3 -0
  7. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/reference/models.md +9 -0
  8. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/reference/providers.md +5 -0
  9. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/reference/request-fields.md +2 -0
  10. {llmshim-0.2.3 → llmshim-0.3.1}/src/config.rs +3 -0
  11. {llmshim-0.2.3 → llmshim-0.3.1}/src/main.rs +9 -1
  12. {llmshim-0.2.3 → llmshim-0.3.1}/src/providers/mod.rs +2 -0
  13. llmshim-0.3.1/src/providers/openai_compat.rs +196 -0
  14. llmshim-0.3.1/src/providers/openrouter.rs +283 -0
  15. {llmshim-0.2.3 → llmshim-0.3.1}/src/router.rs +21 -0
  16. {llmshim-0.2.3 → llmshim-0.3.1}/src/vision.rs +68 -0
  17. llmshim-0.3.1/tests/integration_openrouter.rs +118 -0
  18. llmshim-0.3.1/tests/integration_sglang.rs +99 -0
  19. llmshim-0.3.1/tests/unit_openai_compat.rs +204 -0
  20. llmshim-0.3.1/tests/unit_openrouter.rs +316 -0
  21. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_router.rs +34 -0
  22. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_vision.rs +28 -0
  23. {llmshim-0.2.3 → llmshim-0.3.1}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
  24. {llmshim-0.2.3 → llmshim-0.3.1}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
  25. {llmshim-0.2.3 → llmshim-0.3.1}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  26. {llmshim-0.2.3 → llmshim-0.3.1}/.github/workflows/pages.yml +0 -0
  27. {llmshim-0.2.3 → llmshim-0.3.1}/.github/workflows/release.yml +0 -0
  28. {llmshim-0.2.3 → llmshim-0.3.1}/.gitignore +0 -0
  29. {llmshim-0.2.3 → llmshim-0.3.1}/CODE_OF_CONDUCT.md +0 -0
  30. {llmshim-0.2.3 → llmshim-0.3.1}/CONTRIBUTING.md +0 -0
  31. {llmshim-0.2.3 → llmshim-0.3.1}/LICENSE-APACHE +0 -0
  32. {llmshim-0.2.3 → llmshim-0.3.1}/LICENSE-MIT +0 -0
  33. {llmshim-0.2.3 → llmshim-0.3.1}/NOTICE +0 -0
  34. {llmshim-0.2.3 → llmshim-0.3.1}/SECURITY.md +0 -0
  35. {llmshim-0.2.3 → llmshim-0.3.1}/benchmarks/bench.rs +0 -0
  36. {llmshim-0.2.3 → llmshim-0.3.1}/benchmarks/bench_python.py +0 -0
  37. {llmshim-0.2.3 → llmshim-0.3.1}/benchmarks/loadtest.rs +0 -0
  38. {llmshim-0.2.3 → llmshim-0.3.1}/docs/.gitignore +0 -0
  39. {llmshim-0.2.3 → llmshim-0.3.1}/docs/book.toml +0 -0
  40. {llmshim-0.2.3 → llmshim-0.3.1}/docs/mermaid-init.js +0 -0
  41. {llmshim-0.2.3 → llmshim-0.3.1}/docs/mermaid.min.js +0 -0
  42. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/SUMMARY.md +0 -0
  43. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/concepts/contracts.md +0 -0
  44. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/concepts/conversations.md +0 -0
  45. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/concepts/portability.md +0 -0
  46. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/concepts/routing.md +0 -0
  47. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/concepts/translation-flow.md +0 -0
  48. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/guides/fallbacks.md +0 -0
  49. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/guides/images.md +0 -0
  50. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/guides/native-controls.md +0 -0
  51. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/guides/reasoning.md +0 -0
  52. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/guides/streaming.md +0 -0
  53. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/guides/tools.md +0 -0
  54. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/introduction.md +0 -0
  55. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/proxy/deployment.md +0 -0
  56. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/proxy/http-api.md +0 -0
  57. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/proxy/scaling.md +0 -0
  58. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/reference/api.md +0 -0
  59. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/reference/cli.md +0 -0
  60. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/reference/errors.md +0 -0
  61. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/reference/surfaces.md +0 -0
  62. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/start/choose.md +0 -0
  63. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/start/cli.md +0 -0
  64. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/start/clients.md +0 -0
  65. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/start/configure.md +0 -0
  66. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/start/proxy.md +0 -0
  67. {llmshim-0.2.3 → llmshim-0.3.1}/docs/src/start/rust.md +0 -0
  68. {llmshim-0.2.3 → llmshim-0.3.1}/examples/chat.rs +0 -0
  69. {llmshim-0.2.3 → llmshim-0.3.1}/examples/stream.rs +0 -0
  70. {llmshim-0.2.3 → llmshim-0.3.1}/llmshim/__init__.py +0 -0
  71. {llmshim-0.2.3 → llmshim-0.3.1}/llmshim/_client.py +0 -0
  72. {llmshim-0.2.3 → llmshim-0.3.1}/llmshim/_server.py +0 -0
  73. {llmshim-0.2.3 → llmshim-0.3.1}/llmshim/types.py +0 -0
  74. {llmshim-0.2.3 → llmshim-0.3.1}/pyproject.toml +0 -0
  75. {llmshim-0.2.3 → llmshim-0.3.1}/src/client.rs +0 -0
  76. {llmshim-0.2.3 → llmshim-0.3.1}/src/env.rs +0 -0
  77. {llmshim-0.2.3 → llmshim-0.3.1}/src/error.rs +0 -0
  78. {llmshim-0.2.3 → llmshim-0.3.1}/src/fallback.rs +0 -0
  79. {llmshim-0.2.3 → llmshim-0.3.1}/src/lib.rs +0 -0
  80. {llmshim-0.2.3 → llmshim-0.3.1}/src/log.rs +0 -0
  81. {llmshim-0.2.3 → llmshim-0.3.1}/src/models.rs +0 -0
  82. {llmshim-0.2.3 → llmshim-0.3.1}/src/provider.rs +0 -0
  83. {llmshim-0.2.3 → llmshim-0.3.1}/src/providers/anthropic.rs +0 -0
  84. {llmshim-0.2.3 → llmshim-0.3.1}/src/providers/gemini.rs +0 -0
  85. {llmshim-0.2.3 → llmshim-0.3.1}/src/providers/openai.rs +0 -0
  86. {llmshim-0.2.3 → llmshim-0.3.1}/src/providers/xai.rs +0 -0
  87. {llmshim-0.2.3 → llmshim-0.3.1}/src/proxy/convert.rs +0 -0
  88. {llmshim-0.2.3 → llmshim-0.3.1}/src/proxy/error.rs +0 -0
  89. {llmshim-0.2.3 → llmshim-0.3.1}/src/proxy/handlers.rs +0 -0
  90. {llmshim-0.2.3 → llmshim-0.3.1}/src/proxy/mod.rs +0 -0
  91. {llmshim-0.2.3 → llmshim-0.3.1}/src/proxy/ratelimit.rs +0 -0
  92. {llmshim-0.2.3 → llmshim-0.3.1}/src/proxy/types.rs +0 -0
  93. {llmshim-0.2.3 → llmshim-0.3.1}/tests/integration.rs +0 -0
  94. {llmshim-0.2.3 → llmshim-0.3.1}/tests/integration_fallback.rs +0 -0
  95. {llmshim-0.2.3 → llmshim-0.3.1}/tests/integration_gemini.rs +0 -0
  96. {llmshim-0.2.3 → llmshim-0.3.1}/tests/integration_gemini_tools.rs +0 -0
  97. {llmshim-0.2.3 → llmshim-0.3.1}/tests/integration_long_context.rs +0 -0
  98. {llmshim-0.2.3 → llmshim-0.3.1}/tests/integration_multimodel.rs +0 -0
  99. {llmshim-0.2.3 → llmshim-0.3.1}/tests/integration_proxy.rs +0 -0
  100. {llmshim-0.2.3 → llmshim-0.3.1}/tests/integration_thinking.rs +0 -0
  101. {llmshim-0.2.3 → llmshim-0.3.1}/tests/integration_tool_roundtrip.rs +0 -0
  102. {llmshim-0.2.3 → llmshim-0.3.1}/tests/integration_vision.rs +0 -0
  103. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_anthropic.rs +0 -0
  104. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_fallback.rs +0 -0
  105. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_fast_mode.rs +0 -0
  106. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_gemini.rs +0 -0
  107. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_log.rs +0 -0
  108. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_models.rs +0 -0
  109. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_multimodel.rs +0 -0
  110. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_openai.rs +0 -0
  111. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_proxy.rs +0 -0
  112. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_proxy_convert.rs +0 -0
  113. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_sse.rs +0 -0
  114. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_tools.rs +0 -0
  115. {llmshim-0.2.3 → llmshim-0.3.1}/tests/unit_xai.rs +0 -0
@@ -4,7 +4,7 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
4
4
 
5
5
  ## What is llmshim
6
6
 
7
- A pure Rust LLM API translation layer. Takes OpenAI-format JSON requests, translates them to provider-native formats (and back), with zero infrastructure requirements. Supports OpenAI (Responses API), Anthropic, Google Gemini, and xAI. Includes an interactive CLI chat with streaming, reasoning, and mid-conversation model switching.
7
+ A pure Rust LLM API translation layer. Takes OpenAI-format JSON requests, translates them to provider-native formats (and back), with zero infrastructure requirements. Supports OpenAI (Responses API), Anthropic, Google Gemini, xAI, OpenRouter (an OpenAI Chat Completions-compatible aggregator), and self-hosted **vLLM** / **SGLang** servers (OpenAI Chat Completions-compatible, local or remote). Includes an interactive CLI chat with streaming, reasoning, and mid-conversation model switching.
8
8
 
9
9
  **Published on crates.io as `llmshim`** — https://crates.io/crates/llmshim
10
10
 
@@ -16,6 +16,8 @@ This is a public crate on crates.io. Do NOT make breaking changes to `pub` items
16
16
  - **Anthropic:** `claude-opus-5`, `claude-opus-4-8`, `claude-sonnet-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-sonnet-4-6`, `claude-haiku-4-5-20251001`
17
17
  - **Gemini:** `gemini-3.5-flash`, `gemini-3.1-pro-preview`, `gemini-3-flash-preview`
18
18
  - **xAI:** `grok-4.5`, `grok-4.3`, `grok-4.20-multi-agent-beta-0309`, `grok-4.20-beta-0309-reasoning`, `grok-4.20-beta-0309-non-reasoning`
19
+ - **OpenRouter:** not enumerated (huge/dynamic catalog) — any `openrouter/<vendor>/<model>` slug routes through, e.g. `openrouter/anthropic/claude-sonnet-4.5`.
20
+ - **vLLM / SGLang:** not enumerated (self-hosted) — any `vllm/<served-model>` or `sglang/<served-model>` routes through to the configured server, e.g. `sglang/Qwen/Qwen3.6-35B-A3B-FP8`.
19
21
 
20
22
  ## Build & Test
21
23
 
@@ -30,13 +32,17 @@ cargo run # interactive CLI chat
30
32
  cargo run --features proxy -- proxy # proxy server on :3000
31
33
  ```
32
34
 
33
- API keys: `~/.llmshim/config.toml` (via `llmshim configure`) or env vars `OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `GEMINI_API_KEY`, `XAI_API_KEY`. Precedence: env vars > config file.
35
+ API keys: `~/.llmshim/config.toml` (via `llmshim configure`) or env vars `OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `GEMINI_API_KEY`, `XAI_API_KEY`, `OPENROUTER_API_KEY`. Precedence: env vars > config file. Self-hosted servers are configured by **base URL** instead of a key: `VLLM_BASE_URL` / `SGLANG_BASE_URL` (each with an optional `VLLM_API_KEY` / `SGLANG_API_KEY`); the provider registers only when its base URL is set. Local vs remote is just the URL value.
34
36
 
35
37
  ## Architecture
36
38
 
37
39
  ### Value-based transforms, no canonical struct
38
40
 
39
- Requests flow as `serde_json::Value`. Each provider's transform takes raw JSON and maps only what it understands. Provider-specific features use `x-anthropic`, `x-gemini` namespaces.
41
+ Requests flow as `serde_json::Value`. Each provider's transform takes raw JSON and maps only what it understands. Provider-specific features use `x-anthropic`, `x-gemini`, `x-openrouter`, `x-vllm`, `x-sglang` namespaces.
42
+
43
+ **Self-hosted passthrough providers (vLLM / SGLang).** `src/providers/openai_compat.rs` is one generic OpenAI Chat Completions passthrough backing both `vllm` and `sglang` (`OpenAiCompatible::new(name, base_url, api_key: Option)`, registered per env base URL). Two things differ from the hosted providers: the **base URL is configuration** (local `http://localhost:8000/v1` vs remote `https://host/v1`), and **auth is optional** (self-hosted servers are unauthenticated unless launched with `--api-key`, so the `Authorization` header is sent only when a key is set). Passthrough transforms; `reasoning`/`reasoning_content` normalized to `reasoning_content` (vLLM is migrating the field name); `reasoning_effort` forwarded as-is (honored per-model, not clamped); server-specific params go under `x-<name>` (`chat_template_kwargs`, `separate_reasoning`, `guided_json`, `top_k`, …). Note: reasoning/tool parsing are **launch-time server flags** (`--reasoning-parser`, `--tool-call-parser`), so a request only gets that behavior if the server was started for it — llmshim can't enable it per request.
44
+
45
+ **OpenRouter is the one passthrough provider.** Every other provider translates the OpenAI-format input *away* to a native dialect; OpenRouter (`src/providers/openrouter.rs`) *is* OpenAI Chat Completions, so its transforms are near-identity — messages, tools, vision (`image_url`), and `response_format` are forwarded unchanged; `reasoning_effort` maps 1:1 to OpenRouter's `reasoning:{effort}` (its effort vocabulary is a superset, so no clamping); `message.reasoning` is normalized to `reasoning_content` on responses. OpenRouter models are **not enumerated** in `src/models.rs` (the catalog is huge and dynamic) — any `openrouter/<vendor>/<model>` slug routes through. `x-openrouter` carries OpenRouter-only controls (`provider`, `models`, `transforms`, `route`, native `reasoning`; plus `http_referer`/`x_title` which become headers). The `middle-out` transform is disabled by default for faithful passthrough. Uses `image_url` (Chat Completions) vision via `vision::to_openai_chat`.
40
46
 
41
47
  ### Request flow
42
48
 
@@ -54,7 +60,7 @@ Every provider implements: `transform_request`, `transform_response`, `transform
54
60
 
55
61
  ### Router (`src/router.rs`)
56
62
 
57
- Parses `"provider/model"` strings. Auto-infers provider from prefix (`gpt*`/`o*` → openai, `claude*` → anthropic, `gemini*` → gemini, `grok*` → xai). Supports aliases. `Router::from_env()` reads API key env vars.
63
+ Parses `"provider/model"` strings by splitting on the **first** `/` only, so an OpenRouter slug's internal slash survives (`openrouter/anthropic/claude-sonnet-4.5` → provider `openrouter`, model `anthropic/claude-sonnet-4.5`). Auto-infers provider from prefix (`gpt*`/`o*` → openai, `claude*` → anthropic, `gemini*` → gemini, `grok*` → xai); **OpenRouter, vLLM, and SGLang have no prefix inference** — their slugs collide with everyone's, so address them explicitly (`openrouter/…`, `vllm/…`, `sglang/…`); the first-slash split also preserves HF-style served-model slugs (`vllm/meta-llama/Llama-3.1-8B-Instruct`). Supports aliases. `Router::from_env()` reads API-key env vars, plus `VLLM_BASE_URL` / `SGLANG_BASE_URL` (+ optional `*_API_KEY`) for the self-hosted providers.
58
64
 
59
65
  ### HTTP Client (`src/client.rs`)
60
66
 
@@ -90,6 +96,8 @@ llmshim accepts tools in OpenAI Chat Completions format (nested `function` objec
90
96
  - **OpenAI (Responses API):** Tool definitions flattened from `{"type": "function", "function": {"name": ..., "parameters": ...}}` to `{"type": "function", "name": ..., "parameters": ...}`. Assistant messages with `tool_calls` → `function_call` items. `role: "tool"` messages → `function_call_output` items. Streaming function call events (`response.output_item.added`, `response.function_call_arguments.delta`) translated to Chat Completions chunk format.
91
97
  - **Anthropic:** Tools translated to `{"name": ..., "description": ..., "input_schema": ...}` format. Tool results translated to Anthropic's `tool_result` content blocks.
92
98
  - **xAI:** Same flat format as OpenAI Responses API — `translate_tools()` flattens nested format.
99
+ - **OpenRouter:** No translation — it accepts the Chat Completions nested `{"type":"function","function":{…}}` format directly, so `tools`/`tool_choice`/`tool_calls` pass through unchanged.
100
+ - **vLLM / SGLang:** Same as OpenRouter — Chat Completions nested tool format passes through unchanged (the server must be launched with `--tool-call-parser` / `--enable-auto-tool-choice` for tool calls to be parsed).
93
101
  - **Gemini:** Tools wrapped in `functionDeclarations`. Tool results translated to `functionResponse` format.
94
102
 
95
103
  ### CLI (`src/main.rs`)
@@ -958,7 +958,7 @@ checksum = "6373607a59f0be73a39b6fe456b8192fcc3585f602af20751600e974dd455e77"
958
958
 
959
959
  [[package]]
960
960
  name = "llmshim"
961
- version = "0.2.3"
961
+ version = "0.3.1"
962
962
  dependencies = [
963
963
  "async-stream",
964
964
  "async-trait",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "llmshim"
3
- version = "0.2.3"
3
+ version = "0.3.1"
4
4
  edition = "2021"
5
5
  description = "Blazing fast LLM API translation layer in pure Rust"
6
6
  license = "MIT OR Apache-2.0"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llmshim
3
- Version: 0.2.3
3
+ Version: 0.3.1
4
4
  Classifier: Development Status :: 4 - Beta
5
5
  Classifier: Intended Audience :: Developers
6
6
  Classifier: License :: OSI Approved :: MIT License
@@ -1,6 +1,6 @@
1
1
  # llmshim
2
2
 
3
- A blazing-fast LLM API translation layer written in **pure Rust**. One request format, every provider — OpenAI, Anthropic, Google Gemini, and xAI.
3
+ A blazing-fast LLM API translation layer written in **pure Rust**. One request format, every provider — OpenAI, Anthropic, Google Gemini, xAI, OpenRouter, and self-hosted vLLM / SGLang.
4
4
 
5
5
  Send an OpenAI-style request, pick any model, and llmshim translates it to that provider's native API (and translates the response back). Switch providers by changing one string.
6
6
 
@@ -47,8 +47,27 @@ export OPENAI_API_KEY=sk-...
47
47
  export ANTHROPIC_API_KEY=sk-ant-...
48
48
  export GEMINI_API_KEY=AIza...
49
49
  export XAI_API_KEY=xai-...
50
+ export OPENROUTER_API_KEY=sk-or-...
50
51
  ```
51
52
 
53
+ Reach any model through [OpenRouter](https://openrouter.ai) by addressing it as
54
+ `openrouter/<vendor>/<model>` (e.g. `openrouter/anthropic/claude-sonnet-4.5`).
55
+ OpenRouter is OpenAI Chat Completions-compatible, so tools, vision, streaming,
56
+ and `reasoning_effort` all pass through; OpenRouter-only controls (provider
57
+ routing, model fallbacks, transforms) go under an `x-openrouter` key.
58
+
59
+ Point at a **self-hosted vLLM or SGLang** server (local or remote) by setting its
60
+ base URL — no key needed unless the server was launched with one:
61
+
62
+ ```bash
63
+ export SGLANG_BASE_URL=http://localhost:30000/v1 # or https://your-host/v1
64
+ export VLLM_BASE_URL=http://localhost:8000/v1
65
+ ```
66
+
67
+ Then address the served model as `sglang/<served-model>` or `vllm/<served-model>`
68
+ (e.g. `sglang/Qwen/Qwen3.6-35B-A3B-FP8`). Server-specific knobs go under
69
+ `x-vllm` / `x-sglang`.
70
+
52
71
  Or persist them to the config file (used by all three surfaces):
53
72
 
54
73
  ```bash
@@ -12,6 +12,9 @@ set. Therefore **environment variables take precedence over the file**.
12
12
  | Anthropic | `ANTHROPIC_API_KEY` | `keys.anthropic` |
13
13
  | Google Gemini | `GEMINI_API_KEY` | `keys.gemini` |
14
14
  | xAI | `XAI_API_KEY` | `keys.xai` |
15
+ | OpenRouter | `OPENROUTER_API_KEY` | `keys.openrouter` |
16
+ | vLLM (self-hosted) | `VLLM_BASE_URL` (+ optional `VLLM_API_KEY`) | — (env only) |
17
+ | SGLang (self-hosted) | `SGLANG_BASE_URL` (+ optional `SGLANG_API_KEY`) | — (env only) |
15
18
 
16
19
  The config file shape is:
17
20
 
@@ -119,6 +119,15 @@ provider with `arbitrary-model-id` unchanged. A bare, unregistered model name
119
119
  works only when its prefix identifies a provider (`gpt`, `o1`, `o3`, `o4`,
120
120
  `claude`, `gemini`, or `grok`).
121
121
 
122
+ **OpenRouter** is intentionally not enumerated above — its catalog is large and
123
+ dynamic. Any `openrouter/<vendor>/<model>` slug routes through (e.g.
124
+ `openrouter/anthropic/claude-sonnet-4.5`, `openrouter/meta-llama/llama-3.1-70b-instruct:nitro`);
125
+ the slug's internal slash and `:variant` suffix are preserved. Because its
126
+ slugs collide with other providers' prefixes, OpenRouter has no bare-model
127
+ inference — always address it explicitly as `openrouter/…`.
128
+
129
+ **Self-hosted vLLM / SGLang** are likewise not enumerated. Set `VLLM_BASE_URL` or `SGLANG_BASE_URL` (with an optional `*_API_KEY`) and address the served model as `vllm/<served-model>` or `sglang/<served-model>` — e.g. `sglang/Qwen/Qwen3.6-35B-A3B-FP8`. Local vs remote is just the base-URL value.
130
+
122
131
  Rust applications can also define one-level Router aliases with
123
132
  `Router::alias`. Those aliases are not part of the static registry and are not
124
133
  configured by the stock CLI or proxy. See [Models and the Router](../concepts/routing.md).
@@ -9,6 +9,9 @@ different native API and translates only the fields that API understands.
9
9
  | Anthropic | Messages API | `claude*` | `x-anthropic` |
10
10
  | Google Gemini | `generateContent` / `streamGenerateContent` | `gemini*` | `x-gemini` |
11
11
  | xAI | Responses API | `grok*` | none |
12
+ | OpenRouter | Chat Completions (aggregator) | none — address as `openrouter/<vendor>/<model>` | `x-openrouter` |
13
+ | vLLM | Chat Completions (self-hosted, `VLLM_BASE_URL`) | none — address as `vllm/<served-model>` | `x-vllm` |
14
+ | SGLang | Chat Completions (self-hosted, `SGLANG_BASE_URL`) | none — address as `sglang/<served-model>` | `x-sglang` |
12
15
 
13
16
  An explicit address such as `anthropic/claude-sonnet-5` avoids inference.
14
17
  The named provider must be registered in the Router—that normally means its
@@ -22,6 +25,8 @@ API key is configured.
22
25
  | Anthropic | Messages become Anthropic content blocks; tools use `input_schema`, `tool_use`, and `tool_result` | `max_tokens` defaults to 8192 when absent; supported models receive the 1M-context beta by default; `x-anthropic.extra_betas` appends beta headers and `disable_1m_context` suppresses that automatic header |
23
26
  | Gemini | Messages become `contents`; tools use `functionDeclarations`, `functionCall`, and `functionResponse` | Base64 images become `inline_data`, but a remote image URL becomes a text placeholder because Gemini cannot consume it directly; `x-gemini.thinkingConfig` replaces mapped thinking configuration |
24
27
  | xAI | System/developer text becomes Responses `instructions`; tools are flattened like OpenAI Responses | Unified reasoning becomes `reasoning: {effort}` where the model accepts it; grok-4.20 reasoning is encoded in the model name; there is no `x-xai` namespace |
28
+ | OpenRouter | Passthrough — messages, tools, `image_url` vision, and `response_format` are already Chat Completions and forwarded unchanged | `reasoning_effort` maps 1:1 to OpenRouter's `reasoning:{effort}` (superset vocabulary, no clamping); `message.reasoning` is normalized to `reasoning_content`; the `middle-out` transform is disabled by default; `x-openrouter` carries `provider`/`models`/`transforms`/`route`/native `reasoning` (and `http_referer`/`x_title` headers) |
29
+ | vLLM / SGLang | Passthrough to a self-hosted server — configured by base URL (local or remote), auth optional | `reasoning`/`reasoning_content` normalized to `reasoning_content`; `reasoning_effort` forwarded (honored per-model); server-specific params (`chat_template_kwargs`, `guided_json`, `top_k`, `separate_reasoning`, …) go under `x-vllm`/`x-sglang`. Reasoning/tool parsing depend on the server's launch flags (`--reasoning-parser`, `--tool-call-parser`) |
25
30
 
26
31
  OpenAI and xAI do not receive `temperature`, `top_p`, `top_k`, or `stop` from
27
32
  the portable top-level request. Anthropic and Gemini do. This is the
@@ -29,6 +29,8 @@ renamed or reshaped for the selected provider; unsupported controls are omitted.
29
29
  | `x-openai` | object | Native passthrough | Fields copied to an OpenAI Responses request |
30
30
  | `x-anthropic` | object | Native passthrough/control | Anthropic body fields plus `extra_betas` and `disable_1m_context` controls |
31
31
  | `x-gemini` | object | Native passthrough | Gemini body fields; `thinkingConfig` goes under `generationConfig` |
32
+ | `x-openrouter` | object | Native passthrough/control | OpenRouter body fields (`provider`, `models`, `transforms`, `route`, native `reasoning`); `http_referer`/`x_title` become request headers |
33
+ | `x-vllm` / `x-sglang` | object | Native passthrough | Server-specific body params for a self-hosted vLLM/SGLang server (`chat_template_kwargs`, `separate_reasoning`, `guided_json`, `top_k`, `min_p`, …) |
32
34
 
33
35
  There is no `x-xai` namespace. OpenAI additionally recognizes `store`,
34
36
  `prompt_cache_key`, `prompt_cache_retention`, `safety_identifier`, and `speed`.
@@ -25,6 +25,8 @@ pub struct Keys {
25
25
  pub anthropic: Option<String>,
26
26
  pub gemini: Option<String>,
27
27
  pub xai: Option<String>,
28
+ #[serde(default)]
29
+ pub openrouter: Option<String>,
28
30
  }
29
31
 
30
32
  #[derive(Debug, Serialize, Deserialize)]
@@ -93,6 +95,7 @@ pub fn apply_to_env(config: &Config) {
93
95
  ("ANTHROPIC_API_KEY", &config.keys.anthropic),
94
96
  ("GEMINI_API_KEY", &config.keys.gemini),
95
97
  ("XAI_API_KEY", &config.keys.xai),
98
+ ("OPENROUTER_API_KEY", &config.keys.openrouter),
96
99
  ];
97
100
  for (env_key, value) in mappings {
98
101
  if std::env::var(env_key).is_err() {
@@ -504,6 +504,11 @@ fn cmd_configure() {
504
504
  cfg.keys.xai = Some(xai);
505
505
  }
506
506
 
507
+ let openrouter = config_prompt("OpenRouter API Key", cfg.keys.openrouter.as_deref());
508
+ if !openrouter.is_empty() {
509
+ cfg.keys.openrouter = Some(openrouter);
510
+ }
511
+
507
512
  let host = config_prompt_plain("Proxy host", &cfg.proxy.host);
508
513
  if !host.is_empty() {
509
514
  cfg.proxy.host = host;
@@ -535,6 +540,7 @@ fn cmd_set(key: &str, value: &str) {
535
540
  "anthropic" => cfg.keys.anthropic = Some(value.to_string()),
536
541
  "gemini" => cfg.keys.gemini = Some(value.to_string()),
537
542
  "xai" => cfg.keys.xai = Some(value.to_string()),
543
+ "openrouter" => cfg.keys.openrouter = Some(value.to_string()),
538
544
  "proxy.host" => cfg.proxy.host = value.to_string(),
539
545
  "proxy.port" => {
540
546
  cfg.proxy.port = value.parse().unwrap_or_else(|_| {
@@ -544,7 +550,7 @@ fn cmd_set(key: &str, value: &str) {
544
550
  }
545
551
  _ => {
546
552
  eprintln!(
547
- "Unknown key: {}. Valid: openai, anthropic, gemini, xai, proxy.host, proxy.port",
553
+ "Unknown key: {}. Valid: openai, anthropic, gemini, xai, openrouter, proxy.host, proxy.port",
548
554
  key
549
555
  );
550
556
  std::process::exit(1);
@@ -574,6 +580,7 @@ fn cmd_get(key: &str) {
574
580
  "anthropic" => cfg.keys.anthropic.as_deref().map(mask_key),
575
581
  "gemini" => cfg.keys.gemini.as_deref().map(mask_key),
576
582
  "xai" => cfg.keys.xai.as_deref().map(mask_key),
583
+ "openrouter" => cfg.keys.openrouter.as_deref().map(mask_key),
577
584
  "proxy.host" => Some(cfg.proxy.host.clone()),
578
585
  "proxy.port" => Some(cfg.proxy.port.to_string()),
579
586
  _ => {
@@ -593,6 +600,7 @@ fn cmd_list() {
593
600
  ("anthropic", &cfg.keys.anthropic),
594
601
  ("gemini", &cfg.keys.gemini),
595
602
  ("xai", &cfg.keys.xai),
603
+ ("openrouter", &cfg.keys.openrouter),
596
604
  ] {
597
605
  let display = match value {
598
606
  Some(v) if !v.is_empty() => mask_key(v),
@@ -1,4 +1,6 @@
1
1
  pub mod anthropic;
2
2
  pub mod gemini;
3
3
  pub mod openai;
4
+ pub mod openai_compat;
5
+ pub mod openrouter;
4
6
  pub mod xai;
@@ -0,0 +1,196 @@
1
+ use crate::error::{Result, ShimError};
2
+ use crate::provider::{Provider, ProviderRequest};
3
+ use crate::vision;
4
+ use serde_json::{json, Value};
5
+
6
+ /// A generic OpenAI Chat Completions-compatible provider for **self-hosted**
7
+ /// inference servers — vLLM and SGLang. Like OpenRouter it's a passthrough
8
+ /// (messages, tools, `image_url` vision, and `response_format` are already in
9
+ /// the target shape), but two things differ from a hosted aggregator:
10
+ ///
11
+ /// - **The base URL is configuration**, not a constant — that's what "local vs
12
+ /// remote" means (`http://localhost:8000/v1` vs `https://host/v1`).
13
+ /// - **Auth is optional** — these servers accept unauthenticated requests unless
14
+ /// launched with `--api-key`, so the `Authorization` header is sent only when
15
+ /// a key is configured.
16
+ ///
17
+ /// `name` (e.g. `"vllm"` / `"sglang"`) is both the provider key and the
18
+ /// extension namespace: server-specific params (`chat_template_kwargs`,
19
+ /// `separate_reasoning`, `guided_json`, `top_k`, …) go under `x-<name>`.
20
+ pub struct OpenAiCompatible {
21
+ pub name: String,
22
+ pub base_url: String,
23
+ pub api_key: Option<String>,
24
+ }
25
+
26
+ impl OpenAiCompatible {
27
+ pub fn new(
28
+ name: impl Into<String>,
29
+ base_url: impl Into<String>,
30
+ api_key: Option<String>,
31
+ ) -> Self {
32
+ Self {
33
+ name: name.into(),
34
+ base_url: base_url.into(),
35
+ api_key,
36
+ }
37
+ }
38
+ }
39
+
40
+ /// Strip llmshim-normalized / foreign-provider fields and normalize content
41
+ /// blocks to Chat Completions form. Messages, `tool_calls`, and `role: "tool"`
42
+ /// stay in Chat Completions shape (the target format).
43
+ fn sanitize_messages(messages: &[Value]) -> Vec<Value> {
44
+ messages
45
+ .iter()
46
+ .map(|msg| {
47
+ let mut out = msg.clone();
48
+ if let Some(obj) = out.as_object_mut() {
49
+ obj.remove("reasoning_content"); // regenerated server-side; don't echo back
50
+ obj.remove("reasoning_signature"); // opaque Anthropic token — never forward
51
+ obj.remove("redacted_reasoning_content"); // opaque Anthropic token — never forward
52
+ obj.remove("annotations");
53
+ obj.remove("refusal");
54
+ }
55
+ if let Some(content) = out.get("content").cloned() {
56
+ if content.is_array() {
57
+ let translated =
58
+ vision::translate_content_blocks(&content, vision::to_openai_chat);
59
+ out["content"] = vision::text_blocks_to_chat(&translated);
60
+ }
61
+ }
62
+ out
63
+ })
64
+ .collect()
65
+ }
66
+
67
+ /// Copy OpenRouter/vLLM/SGLang's `reasoning` field into llmshim's
68
+ /// `reasoning_content` convention if the latter isn't already present. (vLLM is
69
+ /// migrating `reasoning_content` → `reasoning`; SGLang uses `reasoning_content`.)
70
+ fn normalize_reasoning(obj: &mut serde_json::Map<String, Value>) {
71
+ if obj.contains_key("reasoning_content") {
72
+ return;
73
+ }
74
+ if let Some(r) = obj
75
+ .get("reasoning")
76
+ .and_then(|r| r.as_str())
77
+ .filter(|s| !s.is_empty())
78
+ .map(str::to_string)
79
+ {
80
+ obj.insert("reasoning_content".to_string(), json!(r));
81
+ }
82
+ }
83
+
84
+ impl Provider for OpenAiCompatible {
85
+ fn name(&self) -> &str {
86
+ &self.name
87
+ }
88
+
89
+ fn transform_request(&self, model: &str, request: &Value) -> Result<ProviderRequest> {
90
+ let obj = request.as_object().ok_or(ShimError::MissingModel)?;
91
+ let messages = obj
92
+ .get("messages")
93
+ .and_then(|m| m.as_array())
94
+ .ok_or(ShimError::MissingModel)?;
95
+
96
+ let mut body = json!({
97
+ "model": model,
98
+ "messages": sanitize_messages(messages),
99
+ });
100
+ let body_obj = body.as_object_mut().unwrap();
101
+
102
+ // Standard Chat Completions params (plus reasoning_effort, which vLLM and
103
+ // some SGLang models honor natively) — forwarded unchanged.
104
+ for key in [
105
+ "max_tokens",
106
+ "max_completion_tokens",
107
+ "temperature",
108
+ "top_p",
109
+ "frequency_penalty",
110
+ "presence_penalty",
111
+ "stop",
112
+ "seed",
113
+ "stream",
114
+ "stream_options",
115
+ "tools",
116
+ "tool_choice",
117
+ "parallel_tool_calls",
118
+ "response_format",
119
+ "logprobs",
120
+ "top_logprobs",
121
+ "n",
122
+ "reasoning_effort",
123
+ ] {
124
+ if let Some(v) = obj.get(key) {
125
+ body_obj.insert(key.to_string(), v.clone());
126
+ }
127
+ }
128
+
129
+ // x-<name> namespace: server-specific params (sampling knobs like top_k /
130
+ // min_p, guided_json / regex / ebnf, chat_template_kwargs,
131
+ // separate_reasoning, …) are copied straight into the body.
132
+ let ns = format!("x-{}", self.name);
133
+ if let Some(ext) = obj.get(&ns).and_then(|e| e.as_object()) {
134
+ for (k, v) in ext {
135
+ body_obj.insert(k.clone(), v.clone());
136
+ }
137
+ }
138
+
139
+ let mut headers = vec![("Content-Type".to_string(), "application/json".to_string())];
140
+ // Auth is optional — self-hosted servers are unauthenticated unless
141
+ // launched with --api-key.
142
+ if let Some(key) = &self.api_key {
143
+ if !key.is_empty() {
144
+ headers.push(("Authorization".to_string(), format!("Bearer {key}")));
145
+ }
146
+ }
147
+
148
+ let url = format!("{}/chat/completions", self.base_url.trim_end_matches('/'));
149
+ Ok(ProviderRequest { url, headers, body })
150
+ }
151
+
152
+ fn transform_response(&self, _model: &str, mut response: Value) -> Result<Value> {
153
+ if let Some(err) = response.get("error") {
154
+ if !err.is_null() {
155
+ let message = err
156
+ .get("message")
157
+ .and_then(|m| m.as_str())
158
+ .unwrap_or("unknown error")
159
+ .to_string();
160
+ let status = err.get("code").and_then(|c| c.as_u64()).unwrap_or(400) as u16;
161
+ return Err(ShimError::ProviderError {
162
+ status,
163
+ body: message,
164
+ });
165
+ }
166
+ }
167
+
168
+ // Already Chat Completions-shaped. Normalize the reasoning field name.
169
+ if let Some(choices) = response.get_mut("choices").and_then(|c| c.as_array_mut()) {
170
+ for choice in choices {
171
+ if let Some(msg) = choice.get_mut("message").and_then(|m| m.as_object_mut()) {
172
+ normalize_reasoning(msg);
173
+ }
174
+ }
175
+ }
176
+
177
+ Ok(response)
178
+ }
179
+
180
+ fn transform_stream_chunk(&self, _model: &str, chunk: &str) -> Result<Option<String>> {
181
+ let mut parsed: Value = match serde_json::from_str(chunk) {
182
+ Ok(v) => v,
183
+ Err(_) => return Ok(None),
184
+ };
185
+
186
+ if let Some(choices) = parsed.get_mut("choices").and_then(|c| c.as_array_mut()) {
187
+ for choice in choices {
188
+ if let Some(delta) = choice.get_mut("delta").and_then(|d| d.as_object_mut()) {
189
+ normalize_reasoning(delta);
190
+ }
191
+ }
192
+ }
193
+
194
+ Ok(Some(serde_json::to_string(&parsed)?))
195
+ }
196
+ }