llmshim 0.3.5__tar.gz → 0.3.7__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. {llmshim-0.3.5 → llmshim-0.3.7}/CLAUDE.md +102 -11
  2. {llmshim-0.3.5 → llmshim-0.3.7}/Cargo.lock +48 -3
  3. {llmshim-0.3.5 → llmshim-0.3.7}/Cargo.toml +4 -1
  4. {llmshim-0.3.5 → llmshim-0.3.7}/PKG-INFO +21 -23
  5. {llmshim-0.3.5 → llmshim-0.3.7}/README.md +80 -10
  6. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/concepts/conversations.md +1 -1
  7. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/concepts/routing.md +14 -9
  8. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/guides/fallbacks.md +5 -5
  9. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/guides/native-controls.md +16 -5
  10. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/guides/reasoning.md +56 -43
  11. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/introduction.md +1 -1
  12. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/proxy/http-api.md +2 -2
  13. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/cli.md +11 -1
  14. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/configuration.md +32 -1
  15. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/errors.md +12 -0
  16. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/models.md +39 -33
  17. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/providers.md +49 -1
  18. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/start/configure.md +18 -2
  19. {llmshim-0.3.5 → llmshim-0.3.7}/src/client.rs +54 -2
  20. {llmshim-0.3.5 → llmshim-0.3.7}/src/main.rs +75 -52
  21. {llmshim-0.3.5 → llmshim-0.3.7}/src/models.rs +200 -91
  22. {llmshim-0.3.5 → llmshim-0.3.7}/src/provider.rs +11 -0
  23. {llmshim-0.3.5 → llmshim-0.3.7}/src/providers/anthropic.rs +78 -16
  24. llmshim-0.3.7/src/providers/chatgpt/auth.rs +466 -0
  25. llmshim-0.3.7/src/providers/chatgpt/mod.rs +218 -0
  26. llmshim-0.3.7/src/providers/chatgpt/streaming.rs +205 -0
  27. {llmshim-0.3.5 → llmshim-0.3.7}/src/providers/gemini.rs +11 -10
  28. {llmshim-0.3.5 → llmshim-0.3.7}/src/providers/mod.rs +1 -0
  29. {llmshim-0.3.5 → llmshim-0.3.7}/src/providers/openai.rs +20 -11
  30. {llmshim-0.3.5 → llmshim-0.3.7}/src/providers/xai.rs +9 -8
  31. {llmshim-0.3.5 → llmshim-0.3.7}/src/router.rs +7 -1
  32. llmshim-0.3.7/tests/fixtures/chatgpt-red.png +0 -0
  33. llmshim-0.3.7/tests/integration_chatgpt.rs +35 -0
  34. llmshim-0.3.7/tests/integration_chatgpt_proxy.rs +265 -0
  35. llmshim-0.3.7/tests/integration_current_models.rs +177 -0
  36. llmshim-0.3.7/tests/support/completion_status.rs +124 -0
  37. llmshim-0.3.7/tests/unit_advertised_models.rs +107 -0
  38. llmshim-0.3.7/tests/unit_chatgpt.rs +833 -0
  39. llmshim-0.3.7/tests/unit_fable.rs +213 -0
  40. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_models.rs +15 -4
  41. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_openai.rs +3 -0
  42. {llmshim-0.3.5 → llmshim-0.3.7}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
  43. {llmshim-0.3.5 → llmshim-0.3.7}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
  44. {llmshim-0.3.5 → llmshim-0.3.7}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  45. {llmshim-0.3.5 → llmshim-0.3.7}/.github/workflows/pages.yml +0 -0
  46. {llmshim-0.3.5 → llmshim-0.3.7}/.github/workflows/release.yml +0 -0
  47. {llmshim-0.3.5 → llmshim-0.3.7}/.gitignore +0 -0
  48. {llmshim-0.3.5 → llmshim-0.3.7}/CODE_OF_CONDUCT.md +0 -0
  49. {llmshim-0.3.5 → llmshim-0.3.7}/CONTRIBUTING.md +0 -0
  50. {llmshim-0.3.5 → llmshim-0.3.7}/LICENSE-APACHE +0 -0
  51. {llmshim-0.3.5 → llmshim-0.3.7}/LICENSE-MIT +0 -0
  52. {llmshim-0.3.5 → llmshim-0.3.7}/NOTICE +0 -0
  53. {llmshim-0.3.5 → llmshim-0.3.7}/SECURITY.md +0 -0
  54. {llmshim-0.3.5 → llmshim-0.3.7}/benchmarks/bench.rs +0 -0
  55. {llmshim-0.3.5 → llmshim-0.3.7}/benchmarks/bench_python.py +0 -0
  56. {llmshim-0.3.5 → llmshim-0.3.7}/benchmarks/gateway_loadtest.rs +0 -0
  57. {llmshim-0.3.5 → llmshim-0.3.7}/benchmarks/loadtest.rs +0 -0
  58. {llmshim-0.3.5 → llmshim-0.3.7}/docs/.gitignore +0 -0
  59. {llmshim-0.3.5 → llmshim-0.3.7}/docs/book.toml +0 -0
  60. {llmshim-0.3.5 → llmshim-0.3.7}/docs/mermaid-init.js +0 -0
  61. {llmshim-0.3.5 → llmshim-0.3.7}/docs/mermaid.min.js +0 -0
  62. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/SUMMARY.md +0 -0
  63. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/concepts/contracts.md +0 -0
  64. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/concepts/portability.md +0 -0
  65. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/concepts/translation-flow.md +0 -0
  66. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/guides/images.md +0 -0
  67. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/guides/streaming.md +0 -0
  68. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/guides/tools.md +0 -0
  69. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/proxy/deployment.md +0 -0
  70. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/proxy/scaling.md +0 -0
  71. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/api.md +0 -0
  72. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/request-fields.md +0 -0
  73. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/reference/surfaces.md +0 -0
  74. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/start/choose.md +0 -0
  75. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/start/cli.md +0 -0
  76. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/start/clients.md +0 -0
  77. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/start/proxy.md +0 -0
  78. {llmshim-0.3.5 → llmshim-0.3.7}/docs/src/start/rust.md +0 -0
  79. {llmshim-0.3.5 → llmshim-0.3.7}/examples/chat.rs +0 -0
  80. {llmshim-0.3.5 → llmshim-0.3.7}/examples/stream.rs +0 -0
  81. {llmshim-0.3.5 → llmshim-0.3.7}/llmshim/__init__.py +0 -0
  82. {llmshim-0.3.5 → llmshim-0.3.7}/llmshim/_client.py +0 -0
  83. {llmshim-0.3.5 → llmshim-0.3.7}/llmshim/_server.py +0 -0
  84. {llmshim-0.3.5 → llmshim-0.3.7}/llmshim/types.py +0 -0
  85. {llmshim-0.3.5 → llmshim-0.3.7}/pyproject.toml +0 -0
  86. {llmshim-0.3.5 → llmshim-0.3.7}/src/config.rs +0 -0
  87. {llmshim-0.3.5 → llmshim-0.3.7}/src/env.rs +0 -0
  88. {llmshim-0.3.5 → llmshim-0.3.7}/src/error.rs +0 -0
  89. {llmshim-0.3.5 → llmshim-0.3.7}/src/fallback.rs +0 -0
  90. {llmshim-0.3.5 → llmshim-0.3.7}/src/gateway/auth.rs +0 -0
  91. {llmshim-0.3.5 → llmshim-0.3.7}/src/gateway/distributed.rs +0 -0
  92. {llmshim-0.3.5 → llmshim-0.3.7}/src/gateway/http.rs +0 -0
  93. {llmshim-0.3.5 → llmshim-0.3.7}/src/gateway/idempotency.rs +0 -0
  94. {llmshim-0.3.5 → llmshim-0.3.7}/src/gateway/metrics.rs +0 -0
  95. {llmshim-0.3.5 → llmshim-0.3.7}/src/gateway/mod.rs +0 -0
  96. {llmshim-0.3.5 → llmshim-0.3.7}/src/gateway/quota.rs +0 -0
  97. {llmshim-0.3.5 → llmshim-0.3.7}/src/lib.rs +0 -0
  98. {llmshim-0.3.5 → llmshim-0.3.7}/src/log.rs +0 -0
  99. {llmshim-0.3.5 → llmshim-0.3.7}/src/providers/openai_compat.rs +0 -0
  100. {llmshim-0.3.5 → llmshim-0.3.7}/src/providers/openrouter.rs +0 -0
  101. {llmshim-0.3.5 → llmshim-0.3.7}/src/proxy/convert.rs +0 -0
  102. {llmshim-0.3.5 → llmshim-0.3.7}/src/proxy/error.rs +0 -0
  103. {llmshim-0.3.5 → llmshim-0.3.7}/src/proxy/handlers.rs +0 -0
  104. {llmshim-0.3.5 → llmshim-0.3.7}/src/proxy/mod.rs +0 -0
  105. {llmshim-0.3.5 → llmshim-0.3.7}/src/proxy/ratelimit.rs +0 -0
  106. {llmshim-0.3.5 → llmshim-0.3.7}/src/proxy/types.rs +0 -0
  107. {llmshim-0.3.5 → llmshim-0.3.7}/src/vision.rs +0 -0
  108. {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration.rs +0 -0
  109. {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_fallback.rs +0 -0
  110. {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_gemini.rs +0 -0
  111. {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_gemini_tools.rs +0 -0
  112. {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_long_context.rs +0 -0
  113. {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_multimodel.rs +0 -0
  114. {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_openrouter.rs +0 -0
  115. {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_proxy.rs +0 -0
  116. {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_sglang.rs +0 -0
  117. {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_thinking.rs +0 -0
  118. {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_tool_roundtrip.rs +0 -0
  119. {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_vision.rs +0 -0
  120. {llmshim-0.3.5 → llmshim-0.3.7}/tests/integration_xai.rs +0 -0
  121. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_anthropic.rs +0 -0
  122. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_fallback.rs +0 -0
  123. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_fast_mode.rs +0 -0
  124. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_gemini.rs +0 -0
  125. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_log.rs +0 -0
  126. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_multimodel.rs +0 -0
  127. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_openai_compat.rs +0 -0
  128. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_openrouter.rs +0 -0
  129. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_proxy.rs +0 -0
  130. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_proxy_convert.rs +0 -0
  131. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_router.rs +0 -0
  132. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_sse.rs +0 -0
  133. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_tools.rs +0 -0
  134. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_vision.rs +0 -0
  135. {llmshim-0.3.5 → llmshim-0.3.7}/tests/unit_xai.rs +0 -0
@@ -10,13 +10,14 @@ A pure Rust LLM API translation layer. Takes OpenAI-format JSON requests, transl
10
10
 
11
11
  This is a public crate on crates.io. Do NOT make breaking changes to `pub` items in `src/lib.rs`, `src/router.rs`, `src/provider.rs`, `src/error.rs`, `src/fallback.rs`, `src/log.rs`, `src/config.rs`, `src/models.rs`, or `src/vision.rs` without a semver bump.
12
12
 
13
- ## Supported models
14
-
15
- - **OpenAI:** `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-5.5`, `gpt-5.5-pro`, `gpt-5.4`, `gpt-5.4-pro`, `gpt-5.4-mini`, `gpt-5.4-nano`
16
- - **Anthropic:** `claude-opus-5`, `claude-opus-4-8`, `claude-sonnet-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-sonnet-4-6`, `claude-haiku-4-5-20251001`
17
- - **Gemini:** `gemini-3.8-flash`, `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite`, `gemini-3.1-flash-lite`
18
- - **xAI:** `grok-4.6`, `grok-4.5`, `grok-4.3`, `grok-4.20-multi-agent-beta-0309`, `grok-4.20-beta-0309-reasoning`, `grok-4.20-beta-0309-non-reasoning`
19
- - **OpenRouter:** not enumerated (huge/dynamic catalog) — any `openrouter/<vendor>/<model>` slug routes through, e.g. `openrouter/anthropic/claude-sonnet-4.5`.
13
+ ## Advertised models
14
+
15
+ - **OpenAI:** `gpt-6-astra`, `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`
16
+ - **ChatGPT subscription (OAuth):** only `chatgpt/gpt-6-astra`, `chatgpt/gpt-5.6-sol`, `chatgpt/gpt-5.6-terra`, and `chatgpt/gpt-5.6-luna`. `CHATGPT_MODELS` in `src/models.rs` is shared by discovery, CLI selection, and validation; older/unlisted models fail before authentication or network calls.
17
+ - **Anthropic:** `claude-fable-5-1`, `claude-opus-5`, `claude-sonnet-5`, `claude-haiku-4-5-20251001`
18
+ - **Gemini:** `gemini-3.8-flash`, `gemini-3.5-flash-lite`
19
+ - **xAI:** `grok-4.6`
20
+ - **OpenRouter:** not enumerated (huge/dynamic catalog) — any `openrouter/<vendor>/<model>` slug routes through, e.g. `openrouter/anthropic/claude-sonnet-5`.
20
21
  - **vLLM / SGLang:** not enumerated (self-hosted) — any `vllm/<served-model>` or `sglang/<served-model>` routes through to the configured server, e.g. `sglang/Qwen/Qwen3.6-35B-A3B-FP8`.
21
22
 
22
23
  ## Build & Test
@@ -36,10 +37,60 @@ API keys: `~/.llmshim/config.toml` (via `llmshim configure`) or env vars `OPENAI
36
37
 
37
38
  ## Architecture
38
39
 
40
+ ### Curated discovery
41
+
42
+ `src/models.rs::MODELS` is the single advertised list, imported directly by
43
+ `src/main.rs` and used by `available_models()` for proxy discovery. Keep the
44
+ current model in each retained tier: four OpenAI, four Anthropic, two stable
45
+ Gemini, one xAI, and four ChatGPT routes. Do not add previews or bring back
46
+ superseded generations without an explicit catalog decision.
47
+
48
+ Historical metadata lives in private `LEGACY_MODELS` and remains queryable via
49
+ `spec()`. Pruning discovery does not delete legacy transforms, tests, or
50
+ explicit-ID routing. ChatGPT keeps its separately authorized four-ID request
51
+ allowlist. Keep reader-facing model tables and examples current; retain the
52
+ actual model IDs in historical benchmark results and regression fixtures.
53
+
39
54
  ### Value-based transforms, no canonical struct
40
55
 
41
56
  Requests flow as `serde_json::Value`. Each provider's transform takes raw JSON and maps only what it understands. Provider-specific features use `x-anthropic`, `x-gemini`, `x-openrouter`, `x-vllm`, `x-sglang` namespaces.
42
57
 
58
+ **ChatGPT OAuth (`src/providers/chatgpt/`).** Run `llmshim login chatgpt` for
59
+ device-code authentication, `login chatgpt --status` for a local check, and
60
+ `logout chatgpt` to remove the selected cache. The independent default cache
61
+ is `~/.llmshim/chatgpt/auth.json`; `CHATGPT_TOKEN_DIR`/`CHATGPT_AUTH_FILE` can
62
+ override it. Never read or overwrite Codex credentials implicitly. The
63
+ object-safe `Provider::prepare_request` hook defaults to `transform_request`;
64
+ ChatGPT uses it to refresh asynchronously with cross-process file locking and
65
+ atomic owner-only token writes. Requests never initiate interactive login.
66
+
67
+ The subscription backend requires SSE, `stream: true`, and `store: false`.
68
+ ChatGPT reuses the Responses translator, enforces the backend field allowlist
69
+ after `x-chatgpt` overrides, and aggregates a validated terminal event for
70
+ non-streaming callers. EOF or `[DONE]` without a terminal event is an error.
71
+ Preserve `chatgpt/<model>` in normalized responses and chunks: a bare GPT name
72
+ is otherwise misattributed to API-key OpenAI by the proxy/gateway when both
73
+ providers are registered. Astra preserves reasoning effort `max`; `none` and
74
+ `minimal` clamp to `low`.
75
+ For ChatGPT streaming, emit function calls from `response.output_item.done`
76
+ with complete arguments. Forwarding `response.output_item.added` followed by
77
+ argument-only deltas loses arguments at the proxy's `tool_call` boundary.
78
+ Text and reasoning remain incremental.
79
+
80
+ Offline coverage lives in `tests/unit_chatgpt.rs`. The live server check starts
81
+ its own loopback CLI process and stops it on completion/failure:
82
+
83
+ ```bash
84
+ cargo test --features proxy --test integration_chatgpt_proxy -- --ignored --nocapture
85
+ ```
86
+
87
+ It checks all four models through `/v1/chat` and `/v1/chat/stream`, provider
88
+ identity with API-key OpenAI also registered, model discovery, old-model
89
+ rejection, Astra tool calls (normal and streaming), a tool-result round trip,
90
+ and image input. It uses the saved ChatGPT login and consumes
91
+ subscription usage; it is ignored during offline CI. Mount the whole token
92
+ directory writable for container use so refresh locks and atomic saves work.
93
+
43
94
  **Self-hosted passthrough providers (vLLM / SGLang).** `src/providers/openai_compat.rs` is one generic OpenAI Chat Completions passthrough backing both `vllm` and `sglang` (`OpenAiCompatible::new(name, base_url, api_key: Option)`, registered per env base URL). Two things differ from the hosted providers: the **base URL is configuration** (local `http://localhost:8000/v1` vs remote `https://host/v1`), and **auth is optional** (self-hosted servers are unauthenticated unless launched with `--api-key`, so the `Authorization` header is sent only when a key is set). Passthrough transforms; `reasoning`/`reasoning_content` normalized to `reasoning_content` (vLLM is migrating the field name); `reasoning_effort` forwarded as-is (honored per-model, not clamped); server-specific params go under `x-<name>` (`chat_template_kwargs`, `separate_reasoning`, `guided_json`, `top_k`, …). Note: reasoning/tool parsing are **launch-time server flags** (`--reasoning-parser`, `--tool-call-parser`), so a request only gets that behavior if the server was started for it — llmshim can't enable it per request.
44
95
 
45
96
  **OpenRouter is the one passthrough provider.** Every other provider translates the OpenAI-format input *away* to a native dialect; OpenRouter (`src/providers/openrouter.rs`) *is* OpenAI Chat Completions, so its transforms are near-identity — messages, tools, vision (`image_url`), and `response_format` are forwarded unchanged; `reasoning_effort` maps 1:1 to OpenRouter's `reasoning:{effort}` (its effort vocabulary is a superset, so no clamping); `message.reasoning` is normalized to `reasoning_content` on responses. OpenRouter models are **not enumerated** in `src/models.rs` (the catalog is huge and dynamic) — any `openrouter/<vendor>/<model>` slug routes through. `x-openrouter` carries OpenRouter-only controls (`provider`, `models`, `transforms`, `route`, native `reasoning`; plus `http_referer`/`x_title` which become headers). The `middle-out` transform is disabled by default for faithful passthrough. Uses `image_url` (Chat Completions) vision via `vision::to_openai_chat`.
@@ -48,8 +99,8 @@ Requests flow as `serde_json::Value`. Each provider's transform takes raw JSON a
48
99
 
49
100
  ```
50
101
  llmshim::completion(router, request)
51
- → router.resolve("anthropic/claude-sonnet-4-6") // parse "provider/model"
52
- → provider.transform_request(model, &value) // OpenAI JSON → provider-native
102
+ → router.resolve("anthropic/claude-sonnet-5") // parse "provider/model"
103
+ → provider.prepare_request(model, &value).await // refresh OAuth if needed, then transform
53
104
  → client.send(provider_request) // HTTP
54
105
  → provider.transform_response(model, body) // provider-native → OpenAI JSON
55
106
  ```
@@ -58,12 +109,23 @@ llmshim::completion(router, request)
58
109
 
59
110
  Every provider implements: `transform_request`, `transform_response`, `transform_stream_chunk`.
60
111
 
112
+ Non-streaming OpenAI/xAI Responses, Gemini and Anthropic transformations require
113
+ supported terminal metadata. Do not turn absent or unrecognized status into
114
+ `finish_reason: "stop"`. Rejected metadata returns a fixed 502 provider error
115
+ without copying provider-supplied status/content into the diagnostic. Tests
116
+ live in `tests/support/completion_status.rs`, included by `unit_openai`.
117
+ Streaming status handling is separate and is not changed by this policy.
118
+
61
119
  ### Router (`src/router.rs`)
62
120
 
63
- Parses `"provider/model"` strings by splitting on the **first** `/` only, so an OpenRouter slug's internal slash survives (`openrouter/anthropic/claude-sonnet-4.5` → provider `openrouter`, model `anthropic/claude-sonnet-4.5`). Auto-infers provider from prefix (`gpt*`/`o*` → openai, `claude*` → anthropic, `gemini*` → gemini, `grok*` → xai); **OpenRouter, vLLM, and SGLang have no prefix inference** — their slugs collide with everyone's, so address them explicitly (`openrouter/…`, `vllm/…`, `sglang/…`); the first-slash split also preserves HF-style served-model slugs (`vllm/meta-llama/Llama-3.1-8B-Instruct`). Supports aliases. `Router::from_env()` reads API-key env vars, plus `VLLM_BASE_URL` / `SGLANG_BASE_URL` (+ optional `*_API_KEY`) for the self-hosted providers.
121
+ Parses `"provider/model"` strings by splitting on the **first** `/` only, so an OpenRouter slug's internal slash survives (`openrouter/anthropic/claude-sonnet-5` → provider `openrouter`, model `anthropic/claude-sonnet-5`). Auto-infers provider from prefix (`gpt*`/`o*` → openai, `claude*` → anthropic, `gemini*` → gemini, `grok*` → xai); **OpenRouter, vLLM, and SGLang have no prefix inference** — their slugs collide with everyone's, so address them explicitly (`openrouter/…`, `vllm/…`, `sglang/…`); the first-slash split also preserves HF-style served-model slugs (`vllm/meta-llama/Llama-3.1-8B-Instruct`). Supports aliases. `Router::from_env()` reads API-key env vars, plus `VLLM_BASE_URL` / `SGLANG_BASE_URL` (+ optional `*_API_KEY`) for the self-hosted providers.
64
122
 
65
123
  ### HTTP Client (`src/client.rs`)
66
124
 
125
+ Automatic redirects are disabled on the shared client. Keep prompts and
126
+ provider-specific credential headers at the configured endpoint; 3xx responses
127
+ remain provider errors. Callers must configure the final URL directly.
128
+
67
129
  `ShimClient` with shared connection pool (`LazyLock`), HTTP/2, gzip/brotli/zstd compression, TCP keepalive + nodelay. Automatic retry (3 attempts by default) on transport errors and 429/500/502/503/504/529 status codes. This is the **reactive** layer: on a retryable *response* it honors the server's `Retry-After` header (integer seconds or HTTP-date) and provider reset hints (OpenAI `x-ratelimit-reset-*`, Anthropic `anthropic-ratelimit-*-reset`), clamped to a cap and nudged with a little jitter; when there's no server hint (or a transport error) it falls back to full-jitter exponential backoff (uniform in `[0, min(cap, base·2^attempt)]`) to avoid a thundering herd. Tunable via `LLMSHIM_MAX_RETRIES` and `LLMSHIM_MAX_BACKOFF_SECS`. `warmup()` pre-establishes TCP+TLS connections. `SseStream` buffers bytes, extracts `data:` lines, routes through provider's `transform_stream_chunk`.
68
130
 
69
131
  ### Fallback chains (`src/fallback.rs`)
@@ -87,7 +149,36 @@ Callers pass provider-specific controls under these keys. Each provider copies w
87
149
 
88
150
  ### Unified reasoning controls
89
151
 
90
- Two knobs work across every provider: `reasoning_effort` (`none|low|medium|high|xhigh|max`) and `reasoning_mode` (`standard|pro`). A third, `reasoning_summary` (`auto|none`), controls reasoning-text visibility → Anthropic `thinking.display` (`auto`→`summarized`, the default when `reasoning_effort` is present so newer models like Sonnet 5 / Opus 4.7-4.8 return reasoning text instead of the API-default `omitted`; `none`→`omitted` for lower latency). Applies to both the adaptive and pre-4.6 enabled thinking builders; a caller-supplied `thinking` block bypasses it. Each provider transform maps them to its native dialect, **clamping to the nearest tier the target model accepts** (all boundaries verified live — e.g. `max` is native only on OpenAI gpt-5.6; Anthropic 4.6 rejects `xhigh` but has `max`; Gemini's enum tops out at `high`; xAI grok-4.20 models reject any reasoning param). `mode: "pro"` is native on OpenAI gpt-5.6/-pro models (`reasoning.mode`), emulated as a one-tier effort bump elsewhere; explicit `none` always wins. Native passthrough (`x-openai.reasoning`, `x-anthropic.thinking`, `x-gemini.thinkingConfig`) bypasses the mapping entirely and always takes precedence. **Full per-provider mapping tables: `docs/src/guides/reasoning.md`** — update it and the pinning tests in `tests/unit_*.rs` together whenever a mapping changes.
152
+ **Fable compatibility (verified September 2026).** Advertise Fable 5.1;
153
+ retain Fable 5 behavior and metadata for explicit requests. Both use always-on adaptive thinking; unified `none` clamps to `low`,
154
+ and native disabled/manual thinking fails locally. Strip `temperature`,
155
+ `top_p`, and `top_k` for Fable and Opus 5 even without an explicit thinking
156
+ object. Fable 5.1 rejects forced tool selection (`required` / `any` / `tool`);
157
+ do not silently turn it into `auto`. Fable 5 still accepts forced tools.
158
+ Both versions reject assistant prefill. Refusal stop reasons on Fable/Opus 5
159
+ map to `content_filter` in normal and streaming engine responses.
160
+
161
+ Fable 5.1's thinking signatures are conversation-bound. Preserve appended
162
+ system turns instead of hoisting them into the initial prompt. Keep the
163
+ initial system/tools/message prefix stable during a signed-thinking round
164
+ trip; errors must not become success. Test with the
165
+ `thinking-binding-controls-2026-08-01` beta and
166
+ `thinking.block_binding.prefix_mismatch_behavior: "error"` so older API
167
+ accounts exercise enforcement too. The provider handles incompatible
168
+ thinking on a switch to older models; do not infer signatures from text.
169
+
170
+ `tests/unit_fable.rs` pins these rules. Live checks for Fable 5, Fable 5.1,
171
+ Opus 5, Gemini 3.8 Flash, and Grok 4.6:
172
+
173
+ ```bash
174
+ cargo test --features proxy --test integration_current_models -- --ignored --nocapture
175
+ ```
176
+
177
+ The live tests consume API usage and are ignored during offline preflight.
178
+ Gemini 3.8 Flash and Opus 5 already had catalog/adapter support; Grok 4.6 is
179
+ the verified xAI model ID. Keep the ChatGPT four-model allowlist independent.
180
+
181
+ Two knobs work across every provider: `reasoning_effort` (`none|low|medium|high|xhigh|max`) and `reasoning_mode` (`standard|pro`). A third, `reasoning_summary` (`auto|none`), controls reasoning-text visibility → Anthropic `thinking.display` (`auto`→`summarized`, the default when `reasoning_effort` is present so newer models like Sonnet 5 / Opus 4.7-4.8 return reasoning text instead of the API-default `omitted`; `none`→`omitted` for lower latency). Applies to both the adaptive and pre-4.6 enabled thinking builders; a caller-supplied `thinking` block bypasses it. Each provider transform maps them to its native dialect, **clamping to the nearest tier the target model accepts** (all boundaries verified live — e.g. `max` is native on OpenAI gpt-5.6 and GPT-6 Astra; Anthropic 4.6 rejects `xhigh` but has `max`; Gemini's enum tops out at `high`; xAI grok-4.20 models reject any reasoning param). `mode: "pro"` is native on OpenAI gpt-5.6/-pro models (`reasoning.mode`), emulated as a one-tier effort bump elsewhere; explicit `none` always wins. Native passthrough (`x-openai.reasoning`, `x-anthropic.thinking`, `x-gemini.thinkingConfig`) bypasses the mapping entirely and always takes precedence. **Full per-provider mapping tables: `docs/src/guides/reasoning.md`** — update it and the pinning tests in `tests/unit_*.rs` together whenever a mapping changes.
91
182
 
92
183
  ### Tool format translation
93
184
 
@@ -371,7 +371,7 @@ dependencies = [
371
371
  "crossterm_winapi",
372
372
  "mio",
373
373
  "parking_lot",
374
- "rustix",
374
+ "rustix 0.38.44",
375
375
  "signal-hook",
376
376
  "signal-hook-mio",
377
377
  "winapi",
@@ -503,6 +503,16 @@ dependencies = [
503
503
  "percent-encoding",
504
504
  ]
505
505
 
506
+ [[package]]
507
+ name = "fs2"
508
+ version = "0.4.3"
509
+ source = "registry+https://github.com/rust-lang/crates.io-index"
510
+ checksum = "9564fc758e15025b46aa6643b1b77d047d1a56a1aea6e01002ac0c7026876213"
511
+ dependencies = [
512
+ "libc",
513
+ "winapi",
514
+ ]
515
+
506
516
  [[package]]
507
517
  name = "futures"
508
518
  version = "0.3.32"
@@ -950,6 +960,12 @@ version = "0.4.15"
950
960
  source = "registry+https://github.com/rust-lang/crates.io-index"
951
961
  checksum = "d26c52dbd32dccf2d10cac7725f8eae5296885fb5703b261f7d0a0739ec807ab"
952
962
 
963
+ [[package]]
964
+ name = "linux-raw-sys"
965
+ version = "0.12.1"
966
+ source = "registry+https://github.com/rust-lang/crates.io-index"
967
+ checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53"
968
+
953
969
  [[package]]
954
970
  name = "litemap"
955
971
  version = "0.8.1"
@@ -958,22 +974,25 @@ checksum = "6373607a59f0be73a39b6fe456b8192fcc3585f602af20751600e974dd455e77"
958
974
 
959
975
  [[package]]
960
976
  name = "llmshim"
961
- version = "0.3.5"
977
+ version = "0.3.7"
962
978
  dependencies = [
963
979
  "async-stream",
964
980
  "async-trait",
965
981
  "axum",
982
+ "base64",
966
983
  "bytes",
967
984
  "chrono",
968
985
  "crossterm",
969
986
  "dirs",
970
987
  "eventsource-stream",
988
+ "fs2",
971
989
  "futures",
972
990
  "mockito",
973
991
  "redis",
974
992
  "reqwest",
975
993
  "serde",
976
994
  "serde_json",
995
+ "tempfile",
977
996
  "thiserror",
978
997
  "tokio",
979
998
  "tokio-stream",
@@ -1441,10 +1460,23 @@ dependencies = [
1441
1460
  "bitflags",
1442
1461
  "errno",
1443
1462
  "libc",
1444
- "linux-raw-sys",
1463
+ "linux-raw-sys 0.4.15",
1445
1464
  "windows-sys 0.52.0",
1446
1465
  ]
1447
1466
 
1467
+ [[package]]
1468
+ name = "rustix"
1469
+ version = "1.1.4"
1470
+ source = "registry+https://github.com/rust-lang/crates.io-index"
1471
+ checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190"
1472
+ dependencies = [
1473
+ "bitflags",
1474
+ "errno",
1475
+ "libc",
1476
+ "linux-raw-sys 0.12.1",
1477
+ "windows-sys 0.61.2",
1478
+ ]
1479
+
1448
1480
  [[package]]
1449
1481
  name = "rustls"
1450
1482
  version = "0.23.37"
@@ -1737,6 +1769,19 @@ dependencies = [
1737
1769
  "syn",
1738
1770
  ]
1739
1771
 
1772
+ [[package]]
1773
+ name = "tempfile"
1774
+ version = "3.27.0"
1775
+ source = "registry+https://github.com/rust-lang/crates.io-index"
1776
+ checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd"
1777
+ dependencies = [
1778
+ "fastrand",
1779
+ "getrandom 0.3.4",
1780
+ "once_cell",
1781
+ "rustix 1.1.4",
1782
+ "windows-sys 0.61.2",
1783
+ ]
1784
+
1740
1785
  [[package]]
1741
1786
  name = "thiserror"
1742
1787
  version = "2.0.18"
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "llmshim"
3
- version = "0.3.5"
3
+ version = "0.3.7"
4
4
  edition = "2021"
5
5
  description = "Blazing fast LLM API translation layer in pure Rust"
6
6
  license = "MIT OR Apache-2.0"
@@ -36,6 +36,9 @@ chrono = { version = "0.4", features = ["serde"] }
36
36
  crossterm = "0.28"
37
37
  toml = "0.8"
38
38
  dirs = "6"
39
+ base64 = "0.22"
40
+ tempfile = "3"
41
+ fs2 = "0.4"
39
42
 
40
43
  # Proxy dependencies (feature-gated)
41
44
  axum = { version = "0.8", features = ["json"], optional = true }
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llmshim
3
- Version: 0.3.5
3
+ Version: 0.3.7
4
4
  Classifier: Development Status :: 4 - Beta
5
5
  Classifier: Intended Audience :: Developers
6
6
  Classifier: License :: OSI Approved :: MIT License
@@ -66,7 +66,7 @@ Then address them via the model string — `vllm/<served-model>` or
66
66
  ```python
67
67
  import llmshim
68
68
 
69
- resp = llmshim.chat("claude-sonnet-4-6", "What is Rust?")
69
+ resp = llmshim.chat("claude-sonnet-5", "What is Rust?")
70
70
  print(resp["message"]["content"])
71
71
  ```
72
72
 
@@ -74,7 +74,7 @@ With options (all map to the API's provider-agnostic `config`):
74
74
 
75
75
  ```python
76
76
  resp = llmshim.chat(
77
- "openai/gpt-5.5",
77
+ "openai/gpt-5.6-sol",
78
78
  "Explain quicksort",
79
79
  max_tokens=500,
80
80
  temperature=0.7,
@@ -88,7 +88,7 @@ resp = llmshim.chat(
88
88
  With message history:
89
89
 
90
90
  ```python
91
- resp = llmshim.chat("claude-sonnet-4-6", [
91
+ resp = llmshim.chat("claude-sonnet-5", [
92
92
  {"role": "system", "content": "You are a pirate."},
93
93
  {"role": "user", "content": "Hello!"},
94
94
  ], max_tokens=500)
@@ -97,7 +97,7 @@ resp = llmshim.chat("claude-sonnet-4-6", [
97
97
  ## Streaming
98
98
 
99
99
  ```python
100
- for event in llmshim.stream("claude-sonnet-4-6", "Write a poem"):
100
+ for event in llmshim.stream("claude-sonnet-5", "Write a poem"):
101
101
  if event["type"] == "content":
102
102
  print(event["text"], end="", flush=True)
103
103
  elif event["type"] == "reasoning":
@@ -113,13 +113,13 @@ Switch models mid-conversation. History carries over.
113
113
  ```python
114
114
  messages = [{"role": "user", "content": "What is a closure?"}]
115
115
 
116
- r1 = llmshim.chat("claude-sonnet-4-6", messages, max_tokens=500)
116
+ r1 = llmshim.chat("claude-sonnet-5", messages, max_tokens=500)
117
117
  print(f"Claude: {r1['message']['content']}")
118
118
 
119
119
  messages.append({"role": "assistant", "content": r1["message"]["content"]})
120
120
  messages.append({"role": "user", "content": "Now explain differently."})
121
121
 
122
- r2 = llmshim.chat("gpt-5.5", messages, max_tokens=500)
122
+ r2 = llmshim.chat("gpt-5.6-sol", messages, max_tokens=500)
123
123
  print(f"GPT: {r2['message']['content']}")
124
124
  ```
125
125
 
@@ -147,7 +147,7 @@ print(resp["message"]["content"]) # answer
147
147
 
148
148
  For full native control, bypass the unified mapping with a namespaced
149
149
  `provider_config` (see below), e.g.
150
- `provider_config={"x-anthropic": {"thinking": {"type": "enabled", "budget_tokens": 4000}}}`.
150
+ `provider_config={"x-anthropic": {"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}}}`.
151
151
 
152
152
  ## Provider-Specific Controls (`provider_config`)
153
153
 
@@ -163,10 +163,8 @@ resp = llmshim.chat(
163
163
  "Solve this step by step: 17 * 23",
164
164
  max_tokens=4000,
165
165
  provider_config={
166
- # native Anthropic extended-thinking control
167
- "x-anthropic": {"thinking": {"type": "enabled", "budget_tokens": 4000}},
168
- # structured output
169
- "response_format": {"type": "json_object"},
166
+ # native Anthropic adaptive-thinking control
167
+ "x-anthropic": {"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}},
170
168
  },
171
169
  )
172
170
  ```
@@ -175,7 +173,7 @@ OpenRouter routing preferences use the `x-openrouter` namespace:
175
173
 
176
174
  ```python
177
175
  resp = llmshim.chat(
178
- "openrouter/anthropic/claude-sonnet-4.5",
176
+ "openrouter/anthropic/claude-sonnet-5",
179
177
  "Hello",
180
178
  max_tokens=200,
181
179
  provider_config={"x-openrouter": {"provider": {"sort": "throughput"}}},
@@ -198,7 +196,7 @@ tools = [{
198
196
  },
199
197
  }]
200
198
 
201
- resp = llmshim.chat("claude-sonnet-4-6", "Weather in Tokyo?", max_tokens=500, tools=tools)
199
+ resp = llmshim.chat("claude-sonnet-5", "Weather in Tokyo?", max_tokens=500, tools=tools)
202
200
  for tc in resp["message"].get("tool_calls", []):
203
201
  print(f"{tc['function']['name']}({tc['function']['arguments']})")
204
202
  ```
@@ -209,10 +207,10 @@ Tools are accepted in OpenAI Chat Completions format and auto-translated to each
209
207
 
210
208
  ```python
211
209
  resp = llmshim.chat(
212
- "anthropic/claude-sonnet-4-6",
210
+ "anthropic/claude-sonnet-5",
213
211
  "Hello",
214
212
  max_tokens=100,
215
- fallback=["openai/gpt-5.6-sol", "gemini/gemini-3.5-flash"],
213
+ fallback=["openai/gpt-5.6-sol", "gemini/gemini-3.8-flash"],
216
214
  )
217
215
  ```
218
216
 
@@ -242,7 +240,7 @@ common ones are re-exported at the top level) for static type-checking:
242
240
  ```python
243
241
  from llmshim.types import ChatResponse, StreamEvent, Message, Config
244
242
 
245
- resp: ChatResponse = llmshim.chat("claude-sonnet-4-6", "hi")
243
+ resp: ChatResponse = llmshim.chat("claude-sonnet-5", "hi")
246
244
  ```
247
245
 
248
246
  Available: `ChatRequest`, `ChatResponse`, `Config`, `Message`, `ToolCall`,
@@ -281,14 +279,14 @@ pytest tests/
281
279
  `test_e2e.py` is a separate LIVE suite that spawns the real binary and makes
282
280
  billed provider calls; run it only when you deliberately want to hit real APIs.
283
281
 
284
- ## Supported Models
282
+ ## Advertised Models
285
283
 
286
284
  | Provider | Models |
287
285
  |----------|--------|
288
- | OpenAI | `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-5.5`, `gpt-5.5-pro`, `gpt-5.4`, `gpt-5.4-pro`, `gpt-5.4-mini`, `gpt-5.4-nano` |
289
- | Anthropic | `claude-opus-5`, `claude-opus-4-8`, `claude-sonnet-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-sonnet-4-6`, `claude-haiku-4-5-20251001` |
290
- | Gemini | `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite`, `gemini-3.1-flash-lite` |
291
- | xAI | `grok-4.6`, `grok-4.5`, `grok-4.3`, `grok-4.20-multi-agent-beta-0309`, `grok-4.20-beta-0309-reasoning`, `grok-4.20-beta-0309-non-reasoning` |
286
+ | OpenAI | `gpt-6-astra`, `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna` |
287
+ | Anthropic | `claude-fable-5-1`, `claude-opus-5`, `claude-sonnet-5`, `claude-haiku-4-5-20251001` |
288
+ | Gemini | `gemini-3.8-flash`, `gemini-3.5-flash-lite` |
289
+ | xAI | `grok-4.6` |
292
290
 
293
291
  Call `llmshim.models()` for the live list filtered to your configured providers.
294
292
 
@@ -299,7 +297,7 @@ any model the upstream serves is reachable, so they aren't in the table above.
299
297
 
300
298
  | Provider | Address as | Env vars | Native controls |
301
299
  |----------|-----------|----------|-----------------|
302
- | OpenRouter | `openrouter/<vendor>/<model>` (e.g. `openrouter/anthropic/claude-sonnet-4.5`) | `OPENROUTER_API_KEY` (or `llmshim.configure(openrouter=...)`) | `provider_config={"x-openrouter": {...}}` (`provider`, `models`, `transforms`) |
300
+ | OpenRouter | `openrouter/<vendor>/<model>` (e.g. `openrouter/anthropic/claude-sonnet-5`) | `OPENROUTER_API_KEY` (or `llmshim.configure(openrouter=...)`) | `provider_config={"x-openrouter": {...}}` (`provider`, `models`, `transforms`) |
303
301
  | vLLM | `vllm/<served-model>` | `VLLM_BASE_URL` (+ optional `VLLM_API_KEY`) | `provider_config={"x-vllm": {...}}` |
304
302
  | SGLang | `sglang/<served-model>` | `SGLANG_BASE_URL` (+ optional `SGLANG_API_KEY`) | `provider_config={"x-sglang": {...}}` |
305
303
 
@@ -1,6 +1,6 @@
1
1
  # llmshim
2
2
 
3
- A blazing-fast LLM API translation layer written in **pure Rust**. One request format, every provider — OpenAI, Anthropic, Google Gemini, xAI, OpenRouter, and self-hosted vLLM / SGLang.
3
+ A blazing-fast LLM API translation layer written in **pure Rust**. One request format, every provider — OpenAI, ChatGPT subscriptions, Anthropic, Google Gemini, xAI, OpenRouter, and self-hosted vLLM / SGLang.
4
4
 
5
5
  Send an OpenAI-style request, pick any model, and llmshim translates it to that provider's native API (and translates the response back). Switch providers by changing one string.
6
6
 
@@ -51,7 +51,7 @@ export OPENROUTER_API_KEY=sk-or-...
51
51
  ```
52
52
 
53
53
  Reach any model through [OpenRouter](https://openrouter.ai) by addressing it as
54
- `openrouter/<vendor>/<model>` (e.g. `openrouter/anthropic/claude-sonnet-4.5`).
54
+ `openrouter/<vendor>/<model>` (e.g. `openrouter/anthropic/claude-sonnet-5`).
55
55
  OpenRouter is OpenAI Chat Completions-compatible, so tools, vision, streaming,
56
56
  and `reasoning_effort` all pass through; OpenRouter-only controls (provider
57
57
  routing, model fallbacks, transforms) go under an `x-openrouter` key.
@@ -76,8 +76,71 @@ llmshim configure # interactive prompt
76
76
 
77
77
  ---
78
78
 
79
+ ## ChatGPT subscription (OAuth)
80
+
81
+ Sign in once with a ChatGPT account, then use `chatgpt/<model>` from Rust,
82
+ the CLI, or any proxy client. No `OPENAI_API_KEY` is needed for this route.
83
+ The device-code flow follows [LiteLLM's ChatGPT provider](https://docs.litellm.ai/docs/providers/chatgpt).
84
+
85
+ ```bash
86
+ llmshim login chatgpt # open the printed URL and enter the code
87
+ llmshim login chatgpt --status # inspect the saved login without network access
88
+ llmshim chat # select a ChatGPT model
89
+ # or: llmshim proxy
90
+ ```
91
+
92
+ If needed, enable device-code login in your ChatGPT security settings or
93
+ workspace permissions ([OpenAI authentication docs](https://learn.chatgpt.com/docs/auth#preferred-device-code-authentication-beta)).
94
+
95
+ ```json
96
+ {
97
+ "model": "chatgpt/gpt-6-astra",
98
+ "messages": [{"role": "user", "content": "Hello!"}]
99
+ }
100
+ ```
101
+
102
+ Tokens live in `~/.llmshim/chatgpt/auth.json`; expired access tokens refresh
103
+ automatically, with file locking and atomic saves. `CHATGPT_TOKEN_DIR` and
104
+ `CHATGPT_AUTH_FILE` select a different cache (LiteLLM's flat auth-file format
105
+ is supported). This cache is independent of Codex's login. Run
106
+ `llmshim logout chatgpt` to remove the selected local cache; this does not
107
+ revoke the session at OpenAI. Login is explicit, never started inside a proxy
108
+ request. Restart an existing proxy after the first login so it registers the
109
+ provider. For containers, mount the token directory writable and set
110
+ `CHATGPT_TOKEN_DIR` to its container path.
111
+
112
+ Both completion and streaming calls use the subscription Responses backend.
113
+ Non-streaming calls collect the upstream stream into one normal response.
114
+ Tools, images, and reasoning use the existing Responses translation;
115
+ `x-chatgpt` supplies supported native fields (under `provider_config` in proxy
116
+ requests). The backend requires `store: false` and `stream: true` and rejects
117
+ token limits, sampling fields, and metadata, so these constraints also apply
118
+ to native overrides. Bare `gpt-*` names still route to API-key OpenAI.
119
+ The ChatGPT route supports only `chatgpt/gpt-6-astra`, `chatgpt/gpt-5.6-sol`,
120
+ `chatgpt/gpt-5.6-terra`, and `chatgpt/gpt-5.6-luna`. Older and unlisted model
121
+ IDs return a local error before authentication or an upstream request.
122
+ Access to these models and usage limits depend on the ChatGPT account.
123
+
124
+ See [provider configuration](docs/src/reference/configuration.md#chatgpt-oauth)
125
+ for endpoint and header overrides.
126
+
127
+ ## Endpoint redirects
128
+
129
+ The shared HTTP client does not follow redirects. Configure the final API URL
130
+ directly: a 3xx response is returned as a provider error instead of forwarding
131
+ the prompt and provider-specific credential headers to another endpoint.
132
+ This applies to both streaming and non-streaming requests.
133
+
79
134
  ## Use it from Rust
80
135
 
136
+ Non-streaming OpenAI Responses, xAI Responses, Gemini, and Anthropic results
137
+ require supported terminal metadata. Missing, malformed, or unsupported
138
+ statuses return a `ProviderError` (502) with a fixed diagnostic instead of
139
+ being interpreted as successful completion. Known completion, output-limit,
140
+ filtering and Anthropic tool-call mappings remain available. Additional native
141
+ reasons require an explicit mapping; streaming transforms are unchanged.
142
+ Provider mocks should include the native terminal status/stop reason.
143
+
81
144
  ```bash
82
145
  cargo add llmshim tokio serde_json
83
146
  ```
@@ -141,6 +204,7 @@ cargo install llmshim --features proxy # from source (any platform)
141
204
  llmshim # show help
142
205
  llmshim chat # interactive multi-model chat (streaming, /model to switch)
143
206
  llmshim configure # set API keys
207
+ llmshim login chatgpt # sign in with a ChatGPT subscription
144
208
  llmshim set <key> <value> # set a config value
145
209
  llmshim list # show configured keys
146
210
  llmshim models # list available models
@@ -298,7 +362,7 @@ resp = llmshim.chat(
298
362
  "anthropic/claude-sonnet-5",
299
363
  "Hello",
300
364
  max_tokens=100,
301
- fallback=["openai/gpt-5.6-sol", "gemini/gemini-3.5-flash"],
365
+ fallback=["openai/gpt-5.6-sol", "gemini/gemini-3.8-flash"],
302
366
  )
303
367
  ```
304
368
 
@@ -358,16 +422,22 @@ Standard library only. Full docs: [`clients/ruby/README.md`](clients/ruby/README
358
422
 
359
423
  ---
360
424
 
361
- ## Supported models
425
+ ## Advertised models
362
426
 
363
427
  | Provider | Models | Reasoning visible |
364
428
  |----------|--------|-------------------|
365
- | **OpenAI** | `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-5.5`, `gpt-5.5-pro`, `gpt-5.4`, `gpt-5.4-pro`, `gpt-5.4-mini`, `gpt-5.4-nano` | Yes (summaries) |
366
- | **Anthropic** | `claude-opus-5`, `claude-opus-4-8`, `claude-sonnet-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-sonnet-4-6`, `claude-haiku-4-5-20251001` | Yes (full thinking) |
367
- | **Google Gemini** | `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite`, `gemini-3.1-flash-lite` | Yes (thought summaries) |
368
- | **xAI** | `grok-4.6`, `grok-4.5`, `grok-4.3`, `grok-4.20-multi-agent-beta-0309`, `grok-4.20-beta-0309-reasoning`, `grok-4.20-beta-0309-non-reasoning` | No (hidden) |
369
-
370
- Use a bare model name (auto-detected by prefix) or an explicit `provider/model` string.
429
+ | **OpenAI** | `gpt-6-astra`, `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna` | Yes (summaries) |
430
+ | **Anthropic** | `claude-fable-5-1`, `claude-opus-5`, `claude-sonnet-5`, `claude-haiku-4-5-20251001` | Yes (thinking summaries) |
431
+ | **Google Gemini** | `gemini-3.8-flash`, `gemini-3.5-flash-lite` | Yes (thought summaries) |
432
+ | **xAI** | `grok-4.6` | No (hidden) |
433
+
434
+ The CLI and server advertise these current tiers. ChatGPT subscription access
435
+ uses the same four OpenAI models under `chatgpt/`. OpenRouter and self-hosted
436
+ providers accept caller-selected IDs without a fixed advertised list.
437
+
438
+ Use a bare model name (auto-detected by prefix) or an explicit `provider/model`
439
+ string. Older explicit IDs retain their provider routing and metadata; the
440
+ ChatGPT route continues to enforce its four-model allowlist.
371
441
 
372
442
  ## Docker
373
443
 
@@ -16,7 +16,7 @@ Start with a history:
16
16
  Send it to another provider by keeping `messages` and changing the address:
17
17
 
18
18
  ```diff
19
- - "model": "anthropic/claude-opus-4-8"
19
+ - "model": "anthropic/claude-opus-5"
20
20
  + "model": "openai/gpt-5.6-sol"
21
21
  ```
22
22
 
@@ -10,9 +10,9 @@ The most explicit form is `provider/model`:
10
10
 
11
11
  ```text
12
12
  openai/gpt-5.6-sol
13
- anthropic/claude-opus-4-8
14
- gemini/gemini-3.5-flash
15
- xai/grok-4.5
13
+ anthropic/claude-opus-5
14
+ gemini/gemini-3.8-flash
15
+ xai/grok-4.6
16
16
  ```
17
17
 
18
18
  The part before the first slash is the Router registration key. The remainder
@@ -39,11 +39,16 @@ explicit address when inference cannot identify the provider.
39
39
  Resolution succeeds only when the selected provider key is registered on the
40
40
  Router. The built-in `Router::from_env()` registers OpenAI, Anthropic, Gemini,
41
41
  and xAI only when their corresponding environment variables are present.
42
+ ChatGPT is registered when its OAuth cache exists; sign in with
43
+ `llmshim login chatgpt` before starting the proxy. Use `chatgpt/<model>` to
44
+ select subscription access. Bare GPT names continue to use OpenAI API keys.
42
45
 
43
46
  The static model registry powers `llmshim models` and `GET /v1/models`. Those
44
47
  commands are discovery aids, filtered to configured providers. The registry is
45
- not an allowlist: routing does not reject a model merely because it is absent
46
- from that list.
48
+ not generally an allowlist: most providers accept models absent from that list.
49
+ ChatGPT accepts only `chatgpt/gpt-6-astra` and the three
50
+ `chatgpt/gpt-5.6-{sol,terra,luna}` models. The provider rejects other IDs
51
+ before authentication or network calls, including through aliases.
47
52
 
48
53
  For that reason, this documentation does not maintain another static model
49
54
  table. Use runtime discovery for the current curated list.
@@ -54,7 +59,7 @@ Rust applications can attach a one-level alias while building a Router:
54
59
 
55
60
  ```rust
56
61
  let router = llmshim::router::Router::from_env()
57
- .alias("smart", "anthropic/claude-opus-4-8");
62
+ .alias("smart", "anthropic/claude-opus-5");
58
63
  ```
59
64
 
60
65
  The Router checks an alias before parsing the provider address. An alias target
@@ -67,9 +72,10 @@ API, or language clients.
67
72
 
68
73
  ## Environment variables versus `config.toml`
69
74
 
70
- `Router::from_env()` means exactly what its name says: it reads
75
+ `Router::from_env()` reads provider environment variables such as
71
76
  `OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `GEMINI_API_KEY`, and `XAI_API_KEY`. It
72
- does not read `~/.llmshim/config.toml` by itself.
77
+ does not read `~/.llmshim/config.toml` by itself. It also discovers the selected
78
+ ChatGPT OAuth cache, whose default location is `~/.llmshim/chatgpt/auth.json`.
73
79
 
74
80
  The CLI and proxy call `llmshim::env::load_all()` before constructing their
75
81
  Router. That function loads the config file and fills only environment
@@ -84,4 +90,3 @@ let router = llmshim::router::Router::from_env();
84
90
 
85
91
  Applications that manage secrets themselves can call `Router::from_env()`
86
92
  directly or construct a Router by registering provider implementations.
87
-