llmshim 0.3.5__tar.gz → 0.3.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {llmshim-0.3.5 → llmshim-0.3.7}/CLAUDE.md +102 -11
- {llmshim-0.3.5 → llmshim-0.3.7}/Cargo.lock +48 -3
- {llmshim-0.3.5 → llmshim-0.3.7}/Cargo.toml +4 -1
- {llmshim-0.3.5 → llmshim-0.3.7}/PKG-INFO +21 -23
- {llmshim-0.3.5 → llmshim-0.3.7}/README.md +80 -10
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/concepts/conversations.md +1 -1
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/concepts/routing.md +14 -9
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/guides/fallbacks.md +5 -5
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/guides/native-controls.md +16 -5
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/guides/reasoning.md +56 -43
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/introduction.md +1 -1
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/proxy/http-api.md +2 -2
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/cli.md +11 -1
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/configuration.md +32 -1
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/errors.md +12 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/models.md +39 -33
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/providers.md +49 -1
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/start/configure.md +18 -2
- {llmshim-0.3.5 → llmshim-0.3.7}/src/client.rs +54 -2
- {llmshim-0.3.5 → llmshim-0.3.7}/src/main.rs +75 -52
- {llmshim-0.3.5 → llmshim-0.3.7}/src/models.rs +200 -91
- {llmshim-0.3.5 → llmshim-0.3.7}/src/provider.rs +11 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/providers/anthropic.rs +78 -16
- llmshim-0.3.7/src/providers/chatgpt/auth.rs +466 -0
- llmshim-0.3.7/src/providers/chatgpt/mod.rs +218 -0
- llmshim-0.3.7/src/providers/chatgpt/streaming.rs +205 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/providers/gemini.rs +11 -10
- {llmshim-0.3.5 → llmshim-0.3.7}/src/providers/mod.rs +1 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/providers/openai.rs +20 -11
- {llmshim-0.3.5 → llmshim-0.3.7}/src/providers/xai.rs +9 -8
- {llmshim-0.3.5 → llmshim-0.3.7}/src/router.rs +7 -1
- llmshim-0.3.7/tests/fixtures/chatgpt-red.png +0 -0
- llmshim-0.3.7/tests/integration_chatgpt.rs +35 -0
- llmshim-0.3.7/tests/integration_chatgpt_proxy.rs +265 -0
- llmshim-0.3.7/tests/integration_current_models.rs +177 -0
- llmshim-0.3.7/tests/support/completion_status.rs +124 -0
- llmshim-0.3.7/tests/unit_advertised_models.rs +107 -0
- llmshim-0.3.7/tests/unit_chatgpt.rs +833 -0
- llmshim-0.3.7/tests/unit_fable.rs +213 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_models.rs +15 -4
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_openai.rs +3 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/.github/workflows/pages.yml +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/.github/workflows/release.yml +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/.gitignore +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/CODE_OF_CONDUCT.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/CONTRIBUTING.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/LICENSE-APACHE +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/LICENSE-MIT +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/NOTICE +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/SECURITY.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/benchmarks/bench.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/benchmarks/bench_python.py +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/benchmarks/gateway_loadtest.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/benchmarks/loadtest.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/.gitignore +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/book.toml +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/mermaid-init.js +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/mermaid.min.js +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/SUMMARY.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/concepts/contracts.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/concepts/portability.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/concepts/translation-flow.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/guides/images.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/guides/streaming.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/guides/tools.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/proxy/deployment.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/proxy/scaling.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/api.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/request-fields.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/surfaces.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/start/choose.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/start/cli.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/start/clients.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/start/proxy.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/start/rust.md +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/examples/chat.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/examples/stream.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/llmshim/__init__.py +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/llmshim/_client.py +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/llmshim/_server.py +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/llmshim/types.py +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/pyproject.toml +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/config.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/env.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/error.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/fallback.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/gateway/auth.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/gateway/distributed.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/gateway/http.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/gateway/idempotency.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/gateway/metrics.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/gateway/mod.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/gateway/quota.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/lib.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/log.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/providers/openai_compat.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/providers/openrouter.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/proxy/convert.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/proxy/error.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/proxy/handlers.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/proxy/mod.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/proxy/ratelimit.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/proxy/types.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/src/vision.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_fallback.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_gemini.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_gemini_tools.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_long_context.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_multimodel.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_openrouter.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_proxy.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_sglang.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_thinking.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_tool_roundtrip.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_vision.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_xai.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_anthropic.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_fallback.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_fast_mode.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_gemini.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_log.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_multimodel.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_openai_compat.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_openrouter.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_proxy.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_proxy_convert.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_router.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_sse.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_tools.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_vision.rs +0 -0
- {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_xai.rs +0 -0
|
@@ -10,13 +10,14 @@ A pure Rust LLM API translation layer. Takes OpenAI-format JSON requests, transl
|
|
|
10
10
|
|
|
11
11
|
This is a public crate on crates.io. Do NOT make breaking changes to `pub` items in `src/lib.rs`, `src/router.rs`, `src/provider.rs`, `src/error.rs`, `src/fallback.rs`, `src/log.rs`, `src/config.rs`, `src/models.rs`, or `src/vision.rs` without a semver bump.
|
|
12
12
|
|
|
13
|
-
##
|
|
14
|
-
|
|
15
|
-
- **OpenAI:** `gpt-
|
|
16
|
-
- **
|
|
17
|
-
- **
|
|
18
|
-
- **
|
|
19
|
-
- **
|
|
13
|
+
## Advertised models
|
|
14
|
+
|
|
15
|
+
- **OpenAI:** `gpt-6-astra`, `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`
|
|
16
|
+
- **ChatGPT subscription (OAuth):** only `chatgpt/gpt-6-astra`, `chatgpt/gpt-5.6-sol`, `chatgpt/gpt-5.6-terra`, and `chatgpt/gpt-5.6-luna`. `CHATGPT_MODELS` in `src/models.rs` is shared by discovery, CLI selection, and validation; older/unlisted models fail before authentication or network calls.
|
|
17
|
+
- **Anthropic:** `claude-fable-5-1`, `claude-opus-5`, `claude-sonnet-5`, `claude-haiku-4-5-20251001`
|
|
18
|
+
- **Gemini:** `gemini-3.8-flash`, `gemini-3.5-flash-lite`
|
|
19
|
+
- **xAI:** `grok-4.6`
|
|
20
|
+
- **OpenRouter:** not enumerated (huge/dynamic catalog) — any `openrouter/<vendor>/<model>` slug routes through, e.g. `openrouter/anthropic/claude-sonnet-5`.
|
|
20
21
|
- **vLLM / SGLang:** not enumerated (self-hosted) — any `vllm/<served-model>` or `sglang/<served-model>` routes through to the configured server, e.g. `sglang/Qwen/Qwen3.6-35B-A3B-FP8`.
|
|
21
22
|
|
|
22
23
|
## Build & Test
|
|
@@ -36,10 +37,60 @@ API keys: `~/.llmshim/config.toml` (via `llmshim configure`) or env vars `OPENAI
|
|
|
36
37
|
|
|
37
38
|
## Architecture
|
|
38
39
|
|
|
40
|
+
### Curated discovery
|
|
41
|
+
|
|
42
|
+
`src/models.rs::MODELS` is the single advertised list, imported directly by
|
|
43
|
+
`src/main.rs` and used by `available_models()` for proxy discovery. Keep the
|
|
44
|
+
current model in each retained tier: four OpenAI, four Anthropic, two stable
|
|
45
|
+
Gemini, one xAI, and four ChatGPT routes. Do not add previews or bring back
|
|
46
|
+
superseded generations without an explicit catalog decision.
|
|
47
|
+
|
|
48
|
+
Historical metadata lives in private `LEGACY_MODELS` and remains queryable via
|
|
49
|
+
`spec()`. Pruning discovery does not delete legacy transforms, tests, or
|
|
50
|
+
explicit-ID routing. ChatGPT keeps its separately authorized four-ID request
|
|
51
|
+
allowlist. Keep reader-facing model tables and examples current; retain the
|
|
52
|
+
actual model IDs in historical benchmark results and regression fixtures.
|
|
53
|
+
|
|
39
54
|
### Value-based transforms, no canonical struct
|
|
40
55
|
|
|
41
56
|
Requests flow as `serde_json::Value`. Each provider's transform takes raw JSON and maps only what it understands. Provider-specific features use `x-anthropic`, `x-gemini`, `x-openrouter`, `x-vllm`, `x-sglang` namespaces.
|
|
42
57
|
|
|
58
|
+
**ChatGPT OAuth (`src/providers/chatgpt/`).** Run `llmshim login chatgpt` for
|
|
59
|
+
device-code authentication, `login chatgpt --status` for a local check, and
|
|
60
|
+
`logout chatgpt` to remove the selected cache. The independent default cache
|
|
61
|
+
is `~/.llmshim/chatgpt/auth.json`; `CHATGPT_TOKEN_DIR`/`CHATGPT_AUTH_FILE` can
|
|
62
|
+
override it. Never read or overwrite Codex credentials implicitly. The
|
|
63
|
+
object-safe `Provider::prepare_request` hook defaults to `transform_request`;
|
|
64
|
+
ChatGPT uses it to refresh asynchronously with cross-process file locking and
|
|
65
|
+
atomic owner-only token writes. Requests never initiate interactive login.
|
|
66
|
+
|
|
67
|
+
The subscription backend requires SSE, `stream: true`, and `store: false`.
|
|
68
|
+
ChatGPT reuses the Responses translator, enforces the backend field allowlist
|
|
69
|
+
after `x-chatgpt` overrides, and aggregates a validated terminal event for
|
|
70
|
+
non-streaming callers. EOF or `[DONE]` without a terminal event is an error.
|
|
71
|
+
Preserve `chatgpt/<model>` in normalized responses and chunks: a bare GPT name
|
|
72
|
+
is otherwise misattributed to API-key OpenAI by the proxy/gateway when both
|
|
73
|
+
providers are registered. Astra preserves reasoning effort `max`; `none` and
|
|
74
|
+
`minimal` clamp to `low`.
|
|
75
|
+
For ChatGPT streaming, emit function calls from `response.output_item.done`
|
|
76
|
+
with complete arguments. Forwarding `response.output_item.added` followed by
|
|
77
|
+
argument-only deltas loses arguments at the proxy's `tool_call` boundary.
|
|
78
|
+
Text and reasoning remain incremental.
|
|
79
|
+
|
|
80
|
+
Offline coverage lives in `tests/unit_chatgpt.rs`. The live server check starts
|
|
81
|
+
its own loopback CLI process and stops it on completion/failure:
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
cargo test --features proxy --test integration_chatgpt_proxy -- --ignored --nocapture
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
It checks all four models through `/v1/chat` and `/v1/chat/stream`, provider
|
|
88
|
+
identity with API-key OpenAI also registered, model discovery, old-model
|
|
89
|
+
rejection, Astra tool calls (normal and streaming), a tool-result round trip,
|
|
90
|
+
and image input. It uses the saved ChatGPT login and consumes
|
|
91
|
+
subscription usage; it is ignored during offline CI. Mount the whole token
|
|
92
|
+
directory writable for container use so refresh locks and atomic saves work.
|
|
93
|
+
|
|
43
94
|
**Self-hosted passthrough providers (vLLM / SGLang).** `src/providers/openai_compat.rs` is one generic OpenAI Chat Completions passthrough backing both `vllm` and `sglang` (`OpenAiCompatible::new(name, base_url, api_key: Option)`, registered per env base URL). Two things differ from the hosted providers: the **base URL is configuration** (local `http://localhost:8000/v1` vs remote `https://host/v1`), and **auth is optional** (self-hosted servers are unauthenticated unless launched with `--api-key`, so the `Authorization` header is sent only when a key is set). Passthrough transforms; `reasoning`/`reasoning_content` normalized to `reasoning_content` (vLLM is migrating the field name); `reasoning_effort` forwarded as-is (honored per-model, not clamped); server-specific params go under `x-<name>` (`chat_template_kwargs`, `separate_reasoning`, `guided_json`, `top_k`, …). Note: reasoning/tool parsing are **launch-time server flags** (`--reasoning-parser`, `--tool-call-parser`), so a request only gets that behavior if the server was started for it — llmshim can't enable it per request.
|
|
44
95
|
|
|
45
96
|
**OpenRouter is the one passthrough provider.** Every other provider translates the OpenAI-format input *away* to a native dialect; OpenRouter (`src/providers/openrouter.rs`) *is* OpenAI Chat Completions, so its transforms are near-identity — messages, tools, vision (`image_url`), and `response_format` are forwarded unchanged; `reasoning_effort` maps 1:1 to OpenRouter's `reasoning:{effort}` (its effort vocabulary is a superset, so no clamping); `message.reasoning` is normalized to `reasoning_content` on responses. OpenRouter models are **not enumerated** in `src/models.rs` (the catalog is huge and dynamic) — any `openrouter/<vendor>/<model>` slug routes through. `x-openrouter` carries OpenRouter-only controls (`provider`, `models`, `transforms`, `route`, native `reasoning`; plus `http_referer`/`x_title` which become headers). The `middle-out` transform is disabled by default for faithful passthrough. Uses `image_url` (Chat Completions) vision via `vision::to_openai_chat`.
|
|
@@ -48,8 +99,8 @@ Requests flow as `serde_json::Value`. Each provider's transform takes raw JSON a
|
|
|
48
99
|
|
|
49
100
|
```
|
|
50
101
|
llmshim::completion(router, request)
|
|
51
|
-
→ router.resolve("anthropic/claude-sonnet-
|
|
52
|
-
→ provider.
|
|
102
|
+
→ router.resolve("anthropic/claude-sonnet-5") // parse "provider/model"
|
|
103
|
+
→ provider.prepare_request(model, &value).await // refresh OAuth if needed, then transform
|
|
53
104
|
→ client.send(provider_request) // HTTP
|
|
54
105
|
→ provider.transform_response(model, body) // provider-native → OpenAI JSON
|
|
55
106
|
```
|
|
@@ -58,12 +109,23 @@ llmshim::completion(router, request)
|
|
|
58
109
|
|
|
59
110
|
Every provider implements: `transform_request`, `transform_response`, `transform_stream_chunk`.
|
|
60
111
|
|
|
112
|
+
Non-streaming OpenAI/xAI Responses, Gemini and Anthropic transformations require
|
|
113
|
+
supported terminal metadata. Do not turn absent or unrecognized status into
|
|
114
|
+
`finish_reason: "stop"`. Rejected metadata returns a fixed 502 provider error
|
|
115
|
+
without copying provider-supplied status/content into the diagnostic. Tests
|
|
116
|
+
live in `tests/support/completion_status.rs`, included by `unit_openai`.
|
|
117
|
+
Streaming status handling is separate and is not changed by this policy.
|
|
118
|
+
|
|
61
119
|
### Router (`src/router.rs`)
|
|
62
120
|
|
|
63
|
-
Parses `"provider/model"` strings by splitting on the **first** `/` only, so an OpenRouter slug's internal slash survives (`openrouter/anthropic/claude-sonnet-
|
|
121
|
+
Parses `"provider/model"` strings by splitting on the **first** `/` only, so an OpenRouter slug's internal slash survives (`openrouter/anthropic/claude-sonnet-5` → provider `openrouter`, model `anthropic/claude-sonnet-5`). Auto-infers provider from prefix (`gpt*`/`o*` → openai, `claude*` → anthropic, `gemini*` → gemini, `grok*` → xai); **OpenRouter, vLLM, and SGLang have no prefix inference** — their slugs collide with everyone's, so address them explicitly (`openrouter/…`, `vllm/…`, `sglang/…`); the first-slash split also preserves HF-style served-model slugs (`vllm/meta-llama/Llama-3.1-8B-Instruct`). Supports aliases. `Router::from_env()` reads API-key env vars, plus `VLLM_BASE_URL` / `SGLANG_BASE_URL` (+ optional `*_API_KEY`) for the self-hosted providers.
|
|
64
122
|
|
|
65
123
|
### HTTP Client (`src/client.rs`)
|
|
66
124
|
|
|
125
|
+
Automatic redirects are disabled on the shared client. Keep prompts and
|
|
126
|
+
provider-specific credential headers at the configured endpoint; 3xx responses
|
|
127
|
+
remain provider errors. Callers must configure the final URL directly.
|
|
128
|
+
|
|
67
129
|
`ShimClient` with shared connection pool (`LazyLock`), HTTP/2, gzip/brotli/zstd compression, TCP keepalive + nodelay. Automatic retry (3 attempts by default) on transport errors and 429/500/502/503/504/529 status codes. This is the **reactive** layer: on a retryable *response* it honors the server's `Retry-After` header (integer seconds or HTTP-date) and provider reset hints (OpenAI `x-ratelimit-reset-*`, Anthropic `anthropic-ratelimit-*-reset`), clamped to a cap and nudged with a little jitter; when there's no server hint (or a transport error) it falls back to full-jitter exponential backoff (uniform in `[0, min(cap, base·2^attempt)]`) to avoid a thundering herd. Tunable via `LLMSHIM_MAX_RETRIES` and `LLMSHIM_MAX_BACKOFF_SECS`. `warmup()` pre-establishes TCP+TLS connections. `SseStream` buffers bytes, extracts `data:` lines, routes through provider's `transform_stream_chunk`.
|
|
68
130
|
|
|
69
131
|
### Fallback chains (`src/fallback.rs`)
|
|
@@ -87,7 +149,36 @@ Callers pass provider-specific controls under these keys. Each provider copies w
|
|
|
87
149
|
|
|
88
150
|
### Unified reasoning controls
|
|
89
151
|
|
|
90
|
-
|
|
152
|
+
**Fable compatibility (verified September 2026).** Advertise Fable 5.1;
|
|
153
|
+
retain Fable 5 behavior and metadata for explicit requests. Both use always-on adaptive thinking; unified `none` clamps to `low`,
|
|
154
|
+
and native disabled/manual thinking fails locally. Strip `temperature`,
|
|
155
|
+
`top_p`, and `top_k` for Fable and Opus 5 even without an explicit thinking
|
|
156
|
+
object. Fable 5.1 rejects forced tool selection (`required` / `any` / `tool`);
|
|
157
|
+
do not silently turn it into `auto`. Fable 5 still accepts forced tools.
|
|
158
|
+
Both versions reject assistant prefill. Refusal stop reasons on Fable/Opus 5
|
|
159
|
+
map to `content_filter` in normal and streaming engine responses.
|
|
160
|
+
|
|
161
|
+
Fable 5.1's thinking signatures are conversation-bound. Preserve appended
|
|
162
|
+
system turns instead of hoisting them into the initial prompt. Keep the
|
|
163
|
+
initial system/tools/message prefix stable during a signed-thinking round
|
|
164
|
+
trip; errors must not become success. Test with the
|
|
165
|
+
`thinking-binding-controls-2026-08-01` beta and
|
|
166
|
+
`thinking.block_binding.prefix_mismatch_behavior: "error"` so older API
|
|
167
|
+
accounts exercise enforcement too. The provider handles incompatible
|
|
168
|
+
thinking on a switch to older models; do not infer signatures from text.
|
|
169
|
+
|
|
170
|
+
`tests/unit_fable.rs` pins these rules. Live checks for Fable 5, Fable 5.1,
|
|
171
|
+
Opus 5, Gemini 3.8 Flash, and Grok 4.6:
|
|
172
|
+
|
|
173
|
+
```bash
|
|
174
|
+
cargo test --features proxy --test integration_current_models -- --ignored --nocapture
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
The live tests consume API usage and are ignored during offline preflight.
|
|
178
|
+
Gemini 3.8 Flash and Opus 5 already had catalog/adapter support; Grok 4.6 is
|
|
179
|
+
the verified xAI model ID. Keep the ChatGPT four-model allowlist independent.
|
|
180
|
+
|
|
181
|
+
Two knobs work across every provider: `reasoning_effort` (`none|low|medium|high|xhigh|max`) and `reasoning_mode` (`standard|pro`). A third, `reasoning_summary` (`auto|none`), controls reasoning-text visibility → Anthropic `thinking.display` (`auto`→`summarized`, the default when `reasoning_effort` is present so newer models like Sonnet 5 / Opus 4.7-4.8 return reasoning text instead of the API-default `omitted`; `none`→`omitted` for lower latency). Applies to both the adaptive and pre-4.6 enabled thinking builders; a caller-supplied `thinking` block bypasses it. Each provider transform maps them to its native dialect, **clamping to the nearest tier the target model accepts** (all boundaries verified live — e.g. `max` is native on OpenAI gpt-5.6 and GPT-6 Astra; Anthropic 4.6 rejects `xhigh` but has `max`; Gemini's enum tops out at `high`; xAI grok-4.20 models reject any reasoning param). `mode: "pro"` is native on OpenAI gpt-5.6/-pro models (`reasoning.mode`), emulated as a one-tier effort bump elsewhere; explicit `none` always wins. Native passthrough (`x-openai.reasoning`, `x-anthropic.thinking`, `x-gemini.thinkingConfig`) bypasses the mapping entirely and always takes precedence. **Full per-provider mapping tables: `docs/src/guides/reasoning.md`** — update it and the pinning tests in `tests/unit_*.rs` together whenever a mapping changes.
|
|
91
182
|
|
|
92
183
|
### Tool format translation
|
|
93
184
|
|
|
@@ -371,7 +371,7 @@ dependencies = [
|
|
|
371
371
|
"crossterm_winapi",
|
|
372
372
|
"mio",
|
|
373
373
|
"parking_lot",
|
|
374
|
-
"rustix",
|
|
374
|
+
"rustix 0.38.44",
|
|
375
375
|
"signal-hook",
|
|
376
376
|
"signal-hook-mio",
|
|
377
377
|
"winapi",
|
|
@@ -503,6 +503,16 @@ dependencies = [
|
|
|
503
503
|
"percent-encoding",
|
|
504
504
|
]
|
|
505
505
|
|
|
506
|
+
[[package]]
|
|
507
|
+
name = "fs2"
|
|
508
|
+
version = "0.4.3"
|
|
509
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
510
|
+
checksum = "9564fc758e15025b46aa6643b1b77d047d1a56a1aea6e01002ac0c7026876213"
|
|
511
|
+
dependencies = [
|
|
512
|
+
"libc",
|
|
513
|
+
"winapi",
|
|
514
|
+
]
|
|
515
|
+
|
|
506
516
|
[[package]]
|
|
507
517
|
name = "futures"
|
|
508
518
|
version = "0.3.32"
|
|
@@ -950,6 +960,12 @@ version = "0.4.15"
|
|
|
950
960
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
951
961
|
checksum = "d26c52dbd32dccf2d10cac7725f8eae5296885fb5703b261f7d0a0739ec807ab"
|
|
952
962
|
|
|
963
|
+
[[package]]
|
|
964
|
+
name = "linux-raw-sys"
|
|
965
|
+
version = "0.12.1"
|
|
966
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
967
|
+
checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53"
|
|
968
|
+
|
|
953
969
|
[[package]]
|
|
954
970
|
name = "litemap"
|
|
955
971
|
version = "0.8.1"
|
|
@@ -958,22 +974,25 @@ checksum = "6373607a59f0be73a39b6fe456b8192fcc3585f602af20751600e974dd455e77"
|
|
|
958
974
|
|
|
959
975
|
[[package]]
|
|
960
976
|
name = "llmshim"
|
|
961
|
-
version = "0.3.
|
|
977
|
+
version = "0.3.7"
|
|
962
978
|
dependencies = [
|
|
963
979
|
"async-stream",
|
|
964
980
|
"async-trait",
|
|
965
981
|
"axum",
|
|
982
|
+
"base64",
|
|
966
983
|
"bytes",
|
|
967
984
|
"chrono",
|
|
968
985
|
"crossterm",
|
|
969
986
|
"dirs",
|
|
970
987
|
"eventsource-stream",
|
|
988
|
+
"fs2",
|
|
971
989
|
"futures",
|
|
972
990
|
"mockito",
|
|
973
991
|
"redis",
|
|
974
992
|
"reqwest",
|
|
975
993
|
"serde",
|
|
976
994
|
"serde_json",
|
|
995
|
+
"tempfile",
|
|
977
996
|
"thiserror",
|
|
978
997
|
"tokio",
|
|
979
998
|
"tokio-stream",
|
|
@@ -1441,10 +1460,23 @@ dependencies = [
|
|
|
1441
1460
|
"bitflags",
|
|
1442
1461
|
"errno",
|
|
1443
1462
|
"libc",
|
|
1444
|
-
"linux-raw-sys",
|
|
1463
|
+
"linux-raw-sys 0.4.15",
|
|
1445
1464
|
"windows-sys 0.52.0",
|
|
1446
1465
|
]
|
|
1447
1466
|
|
|
1467
|
+
[[package]]
|
|
1468
|
+
name = "rustix"
|
|
1469
|
+
version = "1.1.4"
|
|
1470
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
1471
|
+
checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190"
|
|
1472
|
+
dependencies = [
|
|
1473
|
+
"bitflags",
|
|
1474
|
+
"errno",
|
|
1475
|
+
"libc",
|
|
1476
|
+
"linux-raw-sys 0.12.1",
|
|
1477
|
+
"windows-sys 0.61.2",
|
|
1478
|
+
]
|
|
1479
|
+
|
|
1448
1480
|
[[package]]
|
|
1449
1481
|
name = "rustls"
|
|
1450
1482
|
version = "0.23.37"
|
|
@@ -1737,6 +1769,19 @@ dependencies = [
|
|
|
1737
1769
|
"syn",
|
|
1738
1770
|
]
|
|
1739
1771
|
|
|
1772
|
+
[[package]]
|
|
1773
|
+
name = "tempfile"
|
|
1774
|
+
version = "3.27.0"
|
|
1775
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
1776
|
+
checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd"
|
|
1777
|
+
dependencies = [
|
|
1778
|
+
"fastrand",
|
|
1779
|
+
"getrandom 0.3.4",
|
|
1780
|
+
"once_cell",
|
|
1781
|
+
"rustix 1.1.4",
|
|
1782
|
+
"windows-sys 0.61.2",
|
|
1783
|
+
]
|
|
1784
|
+
|
|
1740
1785
|
[[package]]
|
|
1741
1786
|
name = "thiserror"
|
|
1742
1787
|
version = "2.0.18"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "llmshim"
|
|
3
|
-
version = "0.3.
|
|
3
|
+
version = "0.3.7"
|
|
4
4
|
edition = "2021"
|
|
5
5
|
description = "Blazing fast LLM API translation layer in pure Rust"
|
|
6
6
|
license = "MIT OR Apache-2.0"
|
|
@@ -36,6 +36,9 @@ chrono = { version = "0.4", features = ["serde"] }
|
|
|
36
36
|
crossterm = "0.28"
|
|
37
37
|
toml = "0.8"
|
|
38
38
|
dirs = "6"
|
|
39
|
+
base64 = "0.22"
|
|
40
|
+
tempfile = "3"
|
|
41
|
+
fs2 = "0.4"
|
|
39
42
|
|
|
40
43
|
# Proxy dependencies (feature-gated)
|
|
41
44
|
axum = { version = "0.8", features = ["json"], optional = true }
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: llmshim
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.7
|
|
4
4
|
Classifier: Development Status :: 4 - Beta
|
|
5
5
|
Classifier: Intended Audience :: Developers
|
|
6
6
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -66,7 +66,7 @@ Then address them via the model string — `vllm/<served-model>` or
|
|
|
66
66
|
```python
|
|
67
67
|
import llmshim
|
|
68
68
|
|
|
69
|
-
resp = llmshim.chat("claude-sonnet-
|
|
69
|
+
resp = llmshim.chat("claude-sonnet-5", "What is Rust?")
|
|
70
70
|
print(resp["message"]["content"])
|
|
71
71
|
```
|
|
72
72
|
|
|
@@ -74,7 +74,7 @@ With options (all map to the API's provider-agnostic `config`):
|
|
|
74
74
|
|
|
75
75
|
```python
|
|
76
76
|
resp = llmshim.chat(
|
|
77
|
-
"openai/gpt-5.
|
|
77
|
+
"openai/gpt-5.6-sol",
|
|
78
78
|
"Explain quicksort",
|
|
79
79
|
max_tokens=500,
|
|
80
80
|
temperature=0.7,
|
|
@@ -88,7 +88,7 @@ resp = llmshim.chat(
|
|
|
88
88
|
With message history:
|
|
89
89
|
|
|
90
90
|
```python
|
|
91
|
-
resp = llmshim.chat("claude-sonnet-
|
|
91
|
+
resp = llmshim.chat("claude-sonnet-5", [
|
|
92
92
|
{"role": "system", "content": "You are a pirate."},
|
|
93
93
|
{"role": "user", "content": "Hello!"},
|
|
94
94
|
], max_tokens=500)
|
|
@@ -97,7 +97,7 @@ resp = llmshim.chat("claude-sonnet-4-6", [
|
|
|
97
97
|
## Streaming
|
|
98
98
|
|
|
99
99
|
```python
|
|
100
|
-
for event in llmshim.stream("claude-sonnet-
|
|
100
|
+
for event in llmshim.stream("claude-sonnet-5", "Write a poem"):
|
|
101
101
|
if event["type"] == "content":
|
|
102
102
|
print(event["text"], end="", flush=True)
|
|
103
103
|
elif event["type"] == "reasoning":
|
|
@@ -113,13 +113,13 @@ Switch models mid-conversation. History carries over.
|
|
|
113
113
|
```python
|
|
114
114
|
messages = [{"role": "user", "content": "What is a closure?"}]
|
|
115
115
|
|
|
116
|
-
r1 = llmshim.chat("claude-sonnet-
|
|
116
|
+
r1 = llmshim.chat("claude-sonnet-5", messages, max_tokens=500)
|
|
117
117
|
print(f"Claude: {r1['message']['content']}")
|
|
118
118
|
|
|
119
119
|
messages.append({"role": "assistant", "content": r1["message"]["content"]})
|
|
120
120
|
messages.append({"role": "user", "content": "Now explain differently."})
|
|
121
121
|
|
|
122
|
-
r2 = llmshim.chat("gpt-5.
|
|
122
|
+
r2 = llmshim.chat("gpt-5.6-sol", messages, max_tokens=500)
|
|
123
123
|
print(f"GPT: {r2['message']['content']}")
|
|
124
124
|
```
|
|
125
125
|
|
|
@@ -147,7 +147,7 @@ print(resp["message"]["content"]) # answer
|
|
|
147
147
|
|
|
148
148
|
For full native control, bypass the unified mapping with a namespaced
|
|
149
149
|
`provider_config` (see below), e.g.
|
|
150
|
-
`provider_config={"x-anthropic": {"thinking": {"type": "
|
|
150
|
+
`provider_config={"x-anthropic": {"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}}}`.
|
|
151
151
|
|
|
152
152
|
## Provider-Specific Controls (`provider_config`)
|
|
153
153
|
|
|
@@ -163,10 +163,8 @@ resp = llmshim.chat(
|
|
|
163
163
|
"Solve this step by step: 17 * 23",
|
|
164
164
|
max_tokens=4000,
|
|
165
165
|
provider_config={
|
|
166
|
-
# native Anthropic
|
|
167
|
-
"x-anthropic": {"thinking": {"type": "
|
|
168
|
-
# structured output
|
|
169
|
-
"response_format": {"type": "json_object"},
|
|
166
|
+
# native Anthropic adaptive-thinking control
|
|
167
|
+
"x-anthropic": {"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}},
|
|
170
168
|
},
|
|
171
169
|
)
|
|
172
170
|
```
|
|
@@ -175,7 +173,7 @@ OpenRouter routing preferences use the `x-openrouter` namespace:
|
|
|
175
173
|
|
|
176
174
|
```python
|
|
177
175
|
resp = llmshim.chat(
|
|
178
|
-
"openrouter/anthropic/claude-sonnet-
|
|
176
|
+
"openrouter/anthropic/claude-sonnet-5",
|
|
179
177
|
"Hello",
|
|
180
178
|
max_tokens=200,
|
|
181
179
|
provider_config={"x-openrouter": {"provider": {"sort": "throughput"}}},
|
|
@@ -198,7 +196,7 @@ tools = [{
|
|
|
198
196
|
},
|
|
199
197
|
}]
|
|
200
198
|
|
|
201
|
-
resp = llmshim.chat("claude-sonnet-
|
|
199
|
+
resp = llmshim.chat("claude-sonnet-5", "Weather in Tokyo?", max_tokens=500, tools=tools)
|
|
202
200
|
for tc in resp["message"].get("tool_calls", []):
|
|
203
201
|
print(f"{tc['function']['name']}({tc['function']['arguments']})")
|
|
204
202
|
```
|
|
@@ -209,10 +207,10 @@ Tools are accepted in OpenAI Chat Completions format and auto-translated to each
|
|
|
209
207
|
|
|
210
208
|
```python
|
|
211
209
|
resp = llmshim.chat(
|
|
212
|
-
"anthropic/claude-sonnet-
|
|
210
|
+
"anthropic/claude-sonnet-5",
|
|
213
211
|
"Hello",
|
|
214
212
|
max_tokens=100,
|
|
215
|
-
fallback=["openai/gpt-5.6-sol", "gemini/gemini-3.
|
|
213
|
+
fallback=["openai/gpt-5.6-sol", "gemini/gemini-3.8-flash"],
|
|
216
214
|
)
|
|
217
215
|
```
|
|
218
216
|
|
|
@@ -242,7 +240,7 @@ common ones are re-exported at the top level) for static type-checking:
|
|
|
242
240
|
```python
|
|
243
241
|
from llmshim.types import ChatResponse, StreamEvent, Message, Config
|
|
244
242
|
|
|
245
|
-
resp: ChatResponse = llmshim.chat("claude-sonnet-
|
|
243
|
+
resp: ChatResponse = llmshim.chat("claude-sonnet-5", "hi")
|
|
246
244
|
```
|
|
247
245
|
|
|
248
246
|
Available: `ChatRequest`, `ChatResponse`, `Config`, `Message`, `ToolCall`,
|
|
@@ -281,14 +279,14 @@ pytest tests/
|
|
|
281
279
|
`test_e2e.py` is a separate LIVE suite that spawns the real binary and makes
|
|
282
280
|
billed provider calls; run it only when you deliberately want to hit real APIs.
|
|
283
281
|
|
|
284
|
-
##
|
|
282
|
+
## Advertised Models
|
|
285
283
|
|
|
286
284
|
| Provider | Models |
|
|
287
285
|
|----------|--------|
|
|
288
|
-
| OpenAI | `gpt-
|
|
289
|
-
| Anthropic | `claude-
|
|
290
|
-
| Gemini | `gemini-3.
|
|
291
|
-
| xAI | `grok-4.6
|
|
286
|
+
| OpenAI | `gpt-6-astra`, `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna` |
|
|
287
|
+
| Anthropic | `claude-fable-5-1`, `claude-opus-5`, `claude-sonnet-5`, `claude-haiku-4-5-20251001` |
|
|
288
|
+
| Gemini | `gemini-3.8-flash`, `gemini-3.5-flash-lite` |
|
|
289
|
+
| xAI | `grok-4.6` |
|
|
292
290
|
|
|
293
291
|
Call `llmshim.models()` for the live list filtered to your configured providers.
|
|
294
292
|
|
|
@@ -299,7 +297,7 @@ any model the upstream serves is reachable, so they aren't in the table above.
|
|
|
299
297
|
|
|
300
298
|
| Provider | Address as | Env vars | Native controls |
|
|
301
299
|
|----------|-----------|----------|-----------------|
|
|
302
|
-
| OpenRouter | `openrouter/<vendor>/<model>` (e.g. `openrouter/anthropic/claude-sonnet-
|
|
300
|
+
| OpenRouter | `openrouter/<vendor>/<model>` (e.g. `openrouter/anthropic/claude-sonnet-5`) | `OPENROUTER_API_KEY` (or `llmshim.configure(openrouter=...)`) | `provider_config={"x-openrouter": {...}}` (`provider`, `models`, `transforms`) |
|
|
303
301
|
| vLLM | `vllm/<served-model>` | `VLLM_BASE_URL` (+ optional `VLLM_API_KEY`) | `provider_config={"x-vllm": {...}}` |
|
|
304
302
|
| SGLang | `sglang/<served-model>` | `SGLANG_BASE_URL` (+ optional `SGLANG_API_KEY`) | `provider_config={"x-sglang": {...}}` |
|
|
305
303
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# llmshim
|
|
2
2
|
|
|
3
|
-
A blazing-fast LLM API translation layer written in **pure Rust**. One request format, every provider — OpenAI, Anthropic, Google Gemini, xAI, OpenRouter, and self-hosted vLLM / SGLang.
|
|
3
|
+
A blazing-fast LLM API translation layer written in **pure Rust**. One request format, every provider — OpenAI, ChatGPT subscriptions, Anthropic, Google Gemini, xAI, OpenRouter, and self-hosted vLLM / SGLang.
|
|
4
4
|
|
|
5
5
|
Send an OpenAI-style request, pick any model, and llmshim translates it to that provider's native API (and translates the response back). Switch providers by changing one string.
|
|
6
6
|
|
|
@@ -51,7 +51,7 @@ export OPENROUTER_API_KEY=sk-or-...
|
|
|
51
51
|
```
|
|
52
52
|
|
|
53
53
|
Reach any model through [OpenRouter](https://openrouter.ai) by addressing it as
|
|
54
|
-
`openrouter/<vendor>/<model>` (e.g. `openrouter/anthropic/claude-sonnet-
|
|
54
|
+
`openrouter/<vendor>/<model>` (e.g. `openrouter/anthropic/claude-sonnet-5`).
|
|
55
55
|
OpenRouter is OpenAI Chat Completions-compatible, so tools, vision, streaming,
|
|
56
56
|
and `reasoning_effort` all pass through; OpenRouter-only controls (provider
|
|
57
57
|
routing, model fallbacks, transforms) go under an `x-openrouter` key.
|
|
@@ -76,8 +76,71 @@ llmshim configure # interactive prompt
|
|
|
76
76
|
|
|
77
77
|
---
|
|
78
78
|
|
|
79
|
+
## ChatGPT subscription (OAuth)
|
|
80
|
+
|
|
81
|
+
Sign in once with a ChatGPT account, then use `chatgpt/<model>` from Rust,
|
|
82
|
+
the CLI, or any proxy client. No `OPENAI_API_KEY` is needed for this route.
|
|
83
|
+
The device-code flow follows [LiteLLM's ChatGPT provider](https://docs.litellm.ai/docs/providers/chatgpt).
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
llmshim login chatgpt # open the printed URL and enter the code
|
|
87
|
+
llmshim login chatgpt --status # inspect the saved login without network access
|
|
88
|
+
llmshim chat # select a ChatGPT model
|
|
89
|
+
# or: llmshim proxy
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
If needed, enable device-code login in your ChatGPT security settings or
|
|
93
|
+
workspace permissions ([OpenAI authentication docs](https://learn.chatgpt.com/docs/auth#preferred-device-code-authentication-beta)).
|
|
94
|
+
|
|
95
|
+
```json
|
|
96
|
+
{
|
|
97
|
+
"model": "chatgpt/gpt-6-astra",
|
|
98
|
+
"messages": [{"role": "user", "content": "Hello!"}]
|
|
99
|
+
}
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Tokens live in `~/.llmshim/chatgpt/auth.json`; expired access tokens refresh
|
|
103
|
+
automatically, with file locking and atomic saves. `CHATGPT_TOKEN_DIR` and
|
|
104
|
+
`CHATGPT_AUTH_FILE` select a different cache (LiteLLM's flat auth-file format
|
|
105
|
+
is supported). This cache is independent of Codex's login. Run
|
|
106
|
+
`llmshim logout chatgpt` to remove the selected local cache; this does not
|
|
107
|
+
revoke the session at OpenAI. Login is explicit, never started inside a proxy
|
|
108
|
+
request. Restart an existing proxy after the first login so it registers the
|
|
109
|
+
provider. For containers, mount the token directory writable and set
|
|
110
|
+
`CHATGPT_TOKEN_DIR` to its container path.
|
|
111
|
+
|
|
112
|
+
Both completion and streaming calls use the subscription Responses backend.
|
|
113
|
+
Non-streaming calls collect the upstream stream into one normal response.
|
|
114
|
+
Tools, images, and reasoning use the existing Responses translation;
|
|
115
|
+
`x-chatgpt` supplies supported native fields (under `provider_config` in proxy
|
|
116
|
+
requests). The backend requires `store: false` and `stream: true` and rejects
|
|
117
|
+
token limits, sampling fields, and metadata, so these constraints also apply
|
|
118
|
+
to native overrides. Bare `gpt-*` names still route to API-key OpenAI.
|
|
119
|
+
The ChatGPT route supports only `chatgpt/gpt-6-astra`, `chatgpt/gpt-5.6-sol`,
|
|
120
|
+
`chatgpt/gpt-5.6-terra`, and `chatgpt/gpt-5.6-luna`. Older and unlisted model
|
|
121
|
+
IDs return a local error before authentication or an upstream request.
|
|
122
|
+
Access to these models and usage limits depend on the ChatGPT account.
|
|
123
|
+
|
|
124
|
+
See [provider configuration](docs/src/reference/configuration.md#chatgpt-oauth)
|
|
125
|
+
for endpoint and header overrides.
|
|
126
|
+
|
|
127
|
+
## Endpoint redirects
|
|
128
|
+
|
|
129
|
+
The shared HTTP client does not follow redirects. Configure the final API URL
|
|
130
|
+
directly: a 3xx response is returned as a provider error instead of forwarding
|
|
131
|
+
the prompt and provider-specific credential headers to another endpoint.
|
|
132
|
+
This applies to both streaming and non-streaming requests.
|
|
133
|
+
|
|
79
134
|
## Use it from Rust
|
|
80
135
|
|
|
136
|
+
Non-streaming OpenAI Responses, xAI Responses, Gemini, and Anthropic results
|
|
137
|
+
require supported terminal metadata. Missing, malformed, or unsupported
|
|
138
|
+
statuses return a `ProviderError` (502) with a fixed diagnostic instead of
|
|
139
|
+
being interpreted as successful completion. Known completion, output-limit,
|
|
140
|
+
filtering and Anthropic tool-call mappings remain available. Additional native
|
|
141
|
+
reasons require an explicit mapping; streaming transforms are unchanged.
|
|
142
|
+
Provider mocks should include the native terminal status/stop reason.
|
|
143
|
+
|
|
81
144
|
```bash
|
|
82
145
|
cargo add llmshim tokio serde_json
|
|
83
146
|
```
|
|
@@ -141,6 +204,7 @@ cargo install llmshim --features proxy # from source (any platform)
|
|
|
141
204
|
llmshim # show help
|
|
142
205
|
llmshim chat # interactive multi-model chat (streaming, /model to switch)
|
|
143
206
|
llmshim configure # set API keys
|
|
207
|
+
llmshim login chatgpt # sign in with a ChatGPT subscription
|
|
144
208
|
llmshim set <key> <value> # set a config value
|
|
145
209
|
llmshim list # show configured keys
|
|
146
210
|
llmshim models # list available models
|
|
@@ -298,7 +362,7 @@ resp = llmshim.chat(
|
|
|
298
362
|
"anthropic/claude-sonnet-5",
|
|
299
363
|
"Hello",
|
|
300
364
|
max_tokens=100,
|
|
301
|
-
fallback=["openai/gpt-5.6-sol", "gemini/gemini-3.
|
|
365
|
+
fallback=["openai/gpt-5.6-sol", "gemini/gemini-3.8-flash"],
|
|
302
366
|
)
|
|
303
367
|
```
|
|
304
368
|
|
|
@@ -358,16 +422,22 @@ Standard library only. Full docs: [`clients/ruby/README.md`](clients/ruby/README
|
|
|
358
422
|
|
|
359
423
|
---
|
|
360
424
|
|
|
361
|
-
##
|
|
425
|
+
## Advertised models
|
|
362
426
|
|
|
363
427
|
| Provider | Models | Reasoning visible |
|
|
364
428
|
|----------|--------|-------------------|
|
|
365
|
-
| **OpenAI** | `gpt-
|
|
366
|
-
| **Anthropic** | `claude-
|
|
367
|
-
| **Google Gemini** | `gemini-3.
|
|
368
|
-
| **xAI** | `grok-4.6
|
|
369
|
-
|
|
370
|
-
|
|
429
|
+
| **OpenAI** | `gpt-6-astra`, `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna` | Yes (summaries) |
|
|
430
|
+
| **Anthropic** | `claude-fable-5-1`, `claude-opus-5`, `claude-sonnet-5`, `claude-haiku-4-5-20251001` | Yes (thinking summaries) |
|
|
431
|
+
| **Google Gemini** | `gemini-3.8-flash`, `gemini-3.5-flash-lite` | Yes (thought summaries) |
|
|
432
|
+
| **xAI** | `grok-4.6` | No (hidden) |
|
|
433
|
+
|
|
434
|
+
The CLI and server advertise these current tiers. ChatGPT subscription access
|
|
435
|
+
uses the same four OpenAI models under `chatgpt/`. OpenRouter and self-hosted
|
|
436
|
+
providers accept caller-selected IDs without a fixed advertised list.
|
|
437
|
+
|
|
438
|
+
Use a bare model name (auto-detected by prefix) or an explicit `provider/model`
|
|
439
|
+
string. Older explicit IDs retain their provider routing and metadata; the
|
|
440
|
+
ChatGPT route continues to enforce its four-model allowlist.
|
|
371
441
|
|
|
372
442
|
## Docker
|
|
373
443
|
|
|
@@ -10,9 +10,9 @@ The most explicit form is `provider/model`:
|
|
|
10
10
|
|
|
11
11
|
```text
|
|
12
12
|
openai/gpt-5.6-sol
|
|
13
|
-
anthropic/claude-opus-
|
|
14
|
-
gemini/gemini-3.
|
|
15
|
-
xai/grok-4.
|
|
13
|
+
anthropic/claude-opus-5
|
|
14
|
+
gemini/gemini-3.8-flash
|
|
15
|
+
xai/grok-4.6
|
|
16
16
|
```
|
|
17
17
|
|
|
18
18
|
The part before the first slash is the Router registration key. The remainder
|
|
@@ -39,11 +39,16 @@ explicit address when inference cannot identify the provider.
|
|
|
39
39
|
Resolution succeeds only when the selected provider key is registered on the
|
|
40
40
|
Router. The built-in `Router::from_env()` registers OpenAI, Anthropic, Gemini,
|
|
41
41
|
and xAI only when their corresponding environment variables are present.
|
|
42
|
+
ChatGPT is registered when its OAuth cache exists; sign in with
|
|
43
|
+
`llmshim login chatgpt` before starting the proxy. Use `chatgpt/<model>` to
|
|
44
|
+
select subscription access. Bare GPT names continue to use OpenAI API keys.
|
|
42
45
|
|
|
43
46
|
The static model registry powers `llmshim models` and `GET /v1/models`. Those
|
|
44
47
|
commands are discovery aids, filtered to configured providers. The registry is
|
|
45
|
-
not an allowlist:
|
|
46
|
-
|
|
48
|
+
not generally an allowlist: most providers accept models absent from that list.
|
|
49
|
+
ChatGPT accepts only `chatgpt/gpt-6-astra` and the three
|
|
50
|
+
`chatgpt/gpt-5.6-{sol,terra,luna}` models. The provider rejects other IDs
|
|
51
|
+
before authentication or network calls, including through aliases.
|
|
47
52
|
|
|
48
53
|
For that reason, this documentation does not maintain another static model
|
|
49
54
|
table. Use runtime discovery for the current curated list.
|
|
@@ -54,7 +59,7 @@ Rust applications can attach a one-level alias while building a Router:
|
|
|
54
59
|
|
|
55
60
|
```rust
|
|
56
61
|
let router = llmshim::router::Router::from_env()
|
|
57
|
-
.alias("smart", "anthropic/claude-opus-
|
|
62
|
+
.alias("smart", "anthropic/claude-opus-5");
|
|
58
63
|
```
|
|
59
64
|
|
|
60
65
|
The Router checks an alias before parsing the provider address. An alias target
|
|
@@ -67,9 +72,10 @@ API, or language clients.
|
|
|
67
72
|
|
|
68
73
|
## Environment variables versus `config.toml`
|
|
69
74
|
|
|
70
|
-
`Router::from_env()`
|
|
75
|
+
`Router::from_env()` reads provider environment variables such as
|
|
71
76
|
`OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `GEMINI_API_KEY`, and `XAI_API_KEY`. It
|
|
72
|
-
does not read `~/.llmshim/config.toml` by itself.
|
|
77
|
+
does not read `~/.llmshim/config.toml` by itself. It also discovers the selected
|
|
78
|
+
ChatGPT OAuth cache, whose default location is `~/.llmshim/chatgpt/auth.json`.
|
|
73
79
|
|
|
74
80
|
The CLI and proxy call `llmshim::env::load_all()` before constructing their
|
|
75
81
|
Router. That function loads the config file and fills only environment
|
|
@@ -84,4 +90,3 @@ let router = llmshim::router::Router::from_env();
|
|
|
84
90
|
|
|
85
91
|
Applications that manage secrets themselves can call `Router::from_env()`
|
|
86
92
|
directly or construct a Router by registering provider implementations.
|
|
87
|
-
|