llmshim 0.4.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {llmshim-0.4.0 → llmshim-0.6.0}/CLAUDE.md +84 -1
- {llmshim-0.4.0 → llmshim-0.6.0}/Cargo.lock +1 -1
- {llmshim-0.4.0 → llmshim-0.6.0}/Cargo.toml +1 -1
- {llmshim-0.4.0 → llmshim-0.6.0}/PKG-INFO +1 -1
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/concepts/routing.md +47 -1
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/guides/caching.md +13 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/guides/fallbacks.md +13 -1
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/proxy/http-api.md +13 -1
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/proxy/native-apis.md +12 -1
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/proxy/scaling.md +61 -2
- {llmshim-0.4.0 → llmshim-0.6.0}/llmshim/types.py +6 -1
- llmshim-0.6.0/src/breaker.rs +454 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/client.rs +16 -7
- {llmshim-0.4.0 → llmshim-0.6.0}/src/config.rs +40 -1
- llmshim-0.6.0/src/cost.rs +280 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/fallback.rs +28 -2
- {llmshim-0.4.0 → llmshim-0.6.0}/src/gateway/auth.rs +28 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/gateway/distributed.rs +29 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/gateway/http.rs +103 -14
- {llmshim-0.4.0 → llmshim-0.6.0}/src/gateway/metrics.rs +4 -0
- llmshim-0.6.0/src/gateway/quota.rs +500 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/lib.rs +27 -2
- {llmshim-0.4.0 → llmshim-0.6.0}/src/log.rs +9 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/proxy/convert.rs +37 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/proxy/handlers.rs +4 -15
- llmshim-0.6.0/src/proxy/health.rs +208 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/proxy/mod.rs +4 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/proxy/types.rs +29 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/proxy/wire/mod.rs +25 -3
- {llmshim-0.4.0 → llmshim-0.6.0}/src/router.rs +107 -1
- {llmshim-0.4.0 → llmshim-0.6.0}/src/shim.rs +3 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/usage.rs +79 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_fallback.rs +80 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_proxy.rs +60 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_proxy_convert.rs +3 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_router.rs +115 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_wire.rs +51 -0
- llmshim-0.4.0/src/gateway/quota.rs +0 -147
- {llmshim-0.4.0 → llmshim-0.6.0}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/.github/workflows/catalog-refresh.yml +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/.github/workflows/pages.yml +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/.github/workflows/release.yml +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/.gitignore +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/CODE_OF_CONDUCT.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/CONTRIBUTING.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/LICENSE-APACHE +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/LICENSE-MIT +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/NOTICE +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/README.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/SECURITY.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/benchmarks/bench.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/benchmarks/bench_python.py +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/benchmarks/gateway_loadtest.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/benchmarks/loadtest.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/Cargo.toml +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/LICENSE-APACHE +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/LICENSE-MIT +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/README.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/data/LICENSE.models.dev +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/data/README.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/data/models.dev.json +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/src/aliases.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/src/builtin.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/src/capabilities.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/src/lib.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/src/merge.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/src/parse.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/src/refresh.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/src/types.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/tests/catalog.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/crates/llmshim-catalog/tests/refresh.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/.gitignore +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/book.toml +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/mermaid-init.js +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/mermaid.min.js +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/SUMMARY.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/concepts/contracts.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/concepts/conversations.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/concepts/portability.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/concepts/translation-flow.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/guides/capabilities.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/guides/images.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/guides/native-controls.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/guides/reasoning.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/guides/schemas.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/guides/streaming.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/guides/tools.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/introduction.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/proxy/deployment.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/reference/api.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/reference/cli.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/reference/configuration.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/reference/errors.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/reference/models.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/reference/providers.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/reference/request-fields.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/reference/surfaces.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/start/choose.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/start/cli.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/start/clients.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/start/configure.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/start/proxy.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/docs/src/start/rust.md +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/examples/chat.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/examples/stream.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/llmshim/__init__.py +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/llmshim/_client.py +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/llmshim/_server.py +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/pyproject.toml +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/cache.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/cli.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/env.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/error/normalize.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/error.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/gateway/idempotency.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/gateway/mod.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/main.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/models.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/provider.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/providers/anthropic.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/providers/anthropic_reasoning.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/providers/anthropic_signature.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/providers/chatgpt/auth.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/providers/chatgpt/mod.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/providers/chatgpt/streaming.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/providers/gemini.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/providers/mod.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/providers/openai.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/providers/openai_compat.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/providers/openrouter.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/providers/xai.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/proxy/error.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/proxy/ratelimit.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/proxy/wire/receipts.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/reasoning/normalize.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/reasoning.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/schema/memo.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/schema/mod.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/schema/validate.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/schema/walk.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/streaming.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/toolcall/streaming.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/toolcall.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/src/vision.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/fixtures/chatgpt-red.png +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_chatgpt.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_chatgpt_proxy.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_current_models.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_fallback.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_gemini.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_gemini_tools.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_long_context.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_multimodel.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_openrouter.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_proxy.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_sglang.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_thinking.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_tool_roundtrip.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_vision.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/integration_xai.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/support/completion_status.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_advertised_models.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_anthropic.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_cache.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_chatgpt.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_cli.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_fable.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_fast_mode.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_gemini.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_log.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_models.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_multimodel.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_openai.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_openai_compat.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_openrouter.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_provider_contracts.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_reasoning.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_reasoning_profile.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_schema.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_shim.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_signature.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_sse.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_toolcall.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_tools.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_usage.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_vision.rs +0 -0
- {llmshim-0.4.0 → llmshim-0.6.0}/tests/unit_xai.rs +0 -0
|
@@ -53,6 +53,35 @@ refresh in a completion; hold a snapshot for decisions that must agree.
|
|
|
53
53
|
normalized responses and usage chunks, logs, and proxy usage. Native token
|
|
54
54
|
fields remain readable. `src/usage.rs` owns extraction; the streaming client
|
|
55
55
|
merges Anthropic's start/delta usage before normalizing terminal counts.
|
|
56
|
+
`usage.uncached_input_tokens` joins them: the providers disagree on whether
|
|
57
|
+
their prompt total already includes the cache read (Anthropic's excludes it,
|
|
58
|
+
OpenAI/Chat Completions/Gemini include it), and only the transport boundary
|
|
59
|
+
still knows which convention the body used. The convention is read off the
|
|
60
|
+
cache-read field that actually matched, never a provider-name table.
|
|
61
|
+
|
|
62
|
+
### USD cost accounting
|
|
63
|
+
|
|
64
|
+
`src/cost.rs` is the only thing that multiplies a catalog `Cost` (USD per
|
|
65
|
+
million tokens) by those counters. Input is charged on `uncached_input_tokens`
|
|
66
|
+
so a cached prompt is never billed twice; cache reads and writes are charged at
|
|
67
|
+
their own rates; reasoning tokens are already inside `completion_tokens`.
|
|
68
|
+
**An absent price is `None`, never `0.0`** — a positive count in a bucket with
|
|
69
|
+
no rate poisons the whole total rather than producing a partial sum that reads
|
|
70
|
+
as a complete one. `client.rs` stamps `usage.cost_usd` at the transport
|
|
71
|
+
boundary (the last place that knows the dispatch target) for completions and
|
|
72
|
+
for whichever stream chunk carries usage; `log.rs`, `proxy::types::Usage`, both
|
|
73
|
+
native facades and the four bundled clients carry it through as a nullable
|
|
74
|
+
field. `null` means unknown, not free.
|
|
75
|
+
|
|
76
|
+
`src/gateway/quota.rs` adds a per-identity dollar cap beside the RPM/TPM
|
|
77
|
+
buckets: `budget_usd` + `budget_window_secs` on an `Identity`, checked before
|
|
78
|
+
dispatch and charged after (cost is only knowable once a response exists, so
|
|
79
|
+
one in-flight request can overshoot). Windows tumble rather than slide, because
|
|
80
|
+
the fleet-wide store is one counter per window. `SpendCap::with_store` takes the
|
|
81
|
+
Redis-backed `DistributedGateway` in distributed mode so `$100/day` means one
|
|
82
|
+
hundred dollars fleet-wide, not per replica. **A response the catalog cannot
|
|
83
|
+
price is not charged** — recording zero would let an unpriced model run forever
|
|
84
|
+
under a budget; `cost_usd: null` is the signal that a price is missing.
|
|
56
85
|
The native Chat Completions streams must use their own parser in the client;
|
|
57
86
|
passing them to the Responses parser silently drops all events.
|
|
58
87
|
|
|
@@ -189,6 +218,49 @@ remain provider errors. Callers must configure the final URL directly.
|
|
|
189
218
|
|
|
190
219
|
`FallbackConfig` defines an ordered list of models to try. On retryable errors (429, 500, 502, 503, 529), retries with exponential backoff then falls through to the next model. `completion_with_fallback()` is the top-level API. The proxy supports this via `"fallback": ["model1", "model2"]` in the request body.
|
|
191
220
|
|
|
221
|
+
### Provider health (`src/breaker.rs`, `src/proxy/health.rs`)
|
|
222
|
+
|
|
223
|
+
**Health is not rate-limit backoff.** The token buckets already slow a provider
|
|
224
|
+
down after a 429 — a 429 means the provider is alive and asking for less. The
|
|
225
|
+
breaker counts what retrying cannot fix: 5xx (500/502/503/504/529) and
|
|
226
|
+
transport failures. Adapted from `rcode-provider`'s `ProviderBreaker`, which we
|
|
227
|
+
own: sliding failure window, open state, and a single half-open probe admitted
|
|
228
|
+
after the cooldown. Config: `LLMSHIM_BREAKER_WINDOW_SECS` (60),
|
|
229
|
+
`LLMSHIM_BREAKER_TRIP_THRESHOLD` (3; `0` disables), `LLMSHIM_BREAKER_COOLDOWN_SECS` (30).
|
|
230
|
+
|
|
231
|
+
The breaker hangs on the `Router` (`Router::breaker()` / `with_breaker`). Every
|
|
232
|
+
dispatch path *observes* outcomes so health accrues from ordinary traffic; only
|
|
233
|
+
`fallback.rs` *refuses*, and it checks before every attempt rather than once per
|
|
234
|
+
chain entry — the attempt that opens a circuit is usually the chain's own, so a
|
|
235
|
+
per-entry check would still retry into a target it just watched die. A single-target call is still dispatched:
|
|
236
|
+
with no alternative, refusing would only convert an upstream failure into a
|
|
237
|
+
local one. `proxy::health::build_breaker()` attaches a Redis-coordinated
|
|
238
|
+
`SharedHealth` (failure ZSET + open marker + `SET NX` probe) when
|
|
239
|
+
`LLMSHIM_REDIS_URL` is set and `redis-coordination` is compiled in, mirroring
|
|
240
|
+
`build_limiter`. Every shared operation **fails open**.
|
|
241
|
+
|
|
242
|
+
### Named routes (`src/config.rs`, `src/router.rs`)
|
|
243
|
+
|
|
244
|
+
A caller-defined name maps to a model plus request settings:
|
|
245
|
+
|
|
246
|
+
```toml
|
|
247
|
+
[routes.compaction]
|
|
248
|
+
model = "anthropic/claude-haiku-4-5-20251001"
|
|
249
|
+
reasoning_effort = "low"
|
|
250
|
+
max_tokens = 4096
|
|
251
|
+
```
|
|
252
|
+
|
|
253
|
+
Addressed as `"model": "route/compaction"`, so a route travels through the
|
|
254
|
+
existing `provider/model` grammar — an OpenAI SDK, the CLI and the proxy's
|
|
255
|
+
admission control all handle it without learning a new field. `resolve_key`
|
|
256
|
+
resolves the indirection, so rate limiting never sees an unrecognized string.
|
|
257
|
+
|
|
258
|
+
**llmshim must not learn harness vocabulary.** The name is opaque: a harness may
|
|
259
|
+
call a route `compaction`, `advisor` or `webSearch`, and llmshim only knows it
|
|
260
|
+
maps to a model. Route settings are defaults — a per-request key always wins —
|
|
261
|
+
and an unknown name is a 400, never a silent fall back to the default model.
|
|
262
|
+
Routes do not chain.
|
|
263
|
+
|
|
192
264
|
### Vision (`src/vision.rs`)
|
|
193
265
|
|
|
194
266
|
Image content blocks are translated between providers automatically. Users can send images in any format (OpenAI `image_url`, Anthropic `image`, Gemini `inline_data`) and the correct provider sees its native format. Base64 data URIs and plain URLs are both handled. Gemini falls back to a text placeholder for URL images (only supports `inline_data`).
|
|
@@ -303,7 +375,13 @@ HTTP proxy with our own API spec (not OpenAI-compatible). Built on axum.
|
|
|
303
375
|
Endpoints:
|
|
304
376
|
- `POST /v1/chat` — non-streaming (or streaming if `stream: true`)
|
|
305
377
|
- `POST /v1/chat/stream` — always SSE streaming with typed events (`content`, `reasoning`, `tool_call`, `usage`, `done`, `error`)
|
|
306
|
-
- `GET /v1/models` — list available models (filtered to configured providers)
|
|
378
|
+
- `GET /v1/models` — list available models (filtered to configured providers).
|
|
379
|
+
Serves two audiences from one body: an OpenAI SDK reads the `object: "list"` /
|
|
380
|
+
`data[]` envelope (so `client.models.list()` works unmodified), llmshim's own
|
|
381
|
+
clients read `models[]`. Both issue the same request, so there is no path to
|
|
382
|
+
split on — the union *is* the split. `data[].id` is the routing id, requestable
|
|
383
|
+
back as `model`. Built once in `proxy::convert::models_response`, shared with
|
|
384
|
+
the gateway.
|
|
307
385
|
- `GET /health` — health check with provider list
|
|
308
386
|
|
|
309
387
|
Request format uses `config` for provider-agnostic settings and `provider_config` for raw passthrough. OpenAPI 3.1 spec at `api/openapi.yaml`.
|
|
@@ -444,6 +522,11 @@ and native endpoints. Keep display messages readable and source type/code/param
|
|
|
444
522
|
metadata separate; do not move this logic back into a wire-only formatter.
|
|
445
523
|
JSON responses carry native metadata in response extensions, and SSE errors carry
|
|
446
524
|
an optional structured error object so native rendering remains lossless.
|
|
525
|
+
`n > 1` is refused rather than emulated: the OpenAI backend is the Responses
|
|
526
|
+
API (no `n`), Anthropic Messages and Gemini have no `n` either, and the
|
|
527
|
+
single-message proxy projects choice zero. The refusal is a correctly shaped
|
|
528
|
+
`{"error":{type,message,param:"n",code:"unsupported_parameter"}}`, carried
|
|
529
|
+
through `normalize_error` so both facades render it natively.
|
|
447
530
|
Tests: `unit_wire`, gateway `http::native_tests`. Use a temporary receipt directory
|
|
448
531
|
in tests; never put real signatures, credentials, or conversations in fixtures.
|
|
449
532
|
|
|
@@ -70,11 +70,57 @@ perform the second lookup.
|
|
|
70
70
|
Aliases are not currently configurable through the CLI, config file, proxy
|
|
71
71
|
API, or language clients.
|
|
72
72
|
|
|
73
|
+
## Named routes
|
|
74
|
+
|
|
75
|
+
An alias renames a model. A **named route** goes further: it maps a
|
|
76
|
+
caller-defined name to a model *plus* request settings, configured in
|
|
77
|
+
`~/.llmshim/config.toml`.
|
|
78
|
+
|
|
79
|
+
```toml
|
|
80
|
+
[routes.compaction]
|
|
81
|
+
model = "anthropic/claude-haiku-4-5-20251001"
|
|
82
|
+
reasoning_effort = "low"
|
|
83
|
+
max_tokens = 4096
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
Address it as a model:
|
|
87
|
+
|
|
88
|
+
```json
|
|
89
|
+
{"model": "route/compaction", "messages": [{"role": "user", "content": "…"}]}
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
Because it reuses the `provider/model` grammar, a route works everywhere a
|
|
93
|
+
model address does — the Rust API, the CLI, the proxy, the native endpoints and
|
|
94
|
+
an unmodified OpenAI SDK.
|
|
95
|
+
|
|
96
|
+
The name is **opaque to llmshim**. A harness may call a route `compaction`,
|
|
97
|
+
`advisor` or `webSearch`; llmshim never interprets it and has no built-in role
|
|
98
|
+
vocabulary. The harness decides what a name means; llmshim provides only the
|
|
99
|
+
mechanism.
|
|
100
|
+
|
|
101
|
+
Three rules:
|
|
102
|
+
|
|
103
|
+
- **Settings are defaults.** A key the request already carries wins, so a
|
|
104
|
+
caller can pick a route and still raise `reasoning_effort` for one call.
|
|
105
|
+
- **An unknown name is an error** (HTTP 400), never a silent fall back to a
|
|
106
|
+
default model.
|
|
107
|
+
- **Routes do not chain.** A route's `model` may not be another `route/…`.
|
|
108
|
+
|
|
109
|
+
A Rust application can register routes directly:
|
|
110
|
+
|
|
111
|
+
```rust
|
|
112
|
+
use llmshim::config::Route;
|
|
113
|
+
|
|
114
|
+
let router = llmshim::router::Router::new()
|
|
115
|
+
.route("compaction", Route { model: "anthropic/claude-haiku-4-5-20251001".into(), ..Default::default() });
|
|
116
|
+
```
|
|
117
|
+
|
|
73
118
|
## Environment variables versus `config.toml`
|
|
74
119
|
|
|
75
120
|
`Router::from_env()` reads provider environment variables such as
|
|
76
121
|
`OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `GEMINI_API_KEY`, and `XAI_API_KEY`. It
|
|
77
|
-
does not read `~/.llmshim/config.toml` by itself
|
|
122
|
+
does not read API keys from `~/.llmshim/config.toml` by itself; it does read
|
|
123
|
+
that file's `[routes]` table, which has no environment equivalent. It also discovers the selected
|
|
78
124
|
ChatGPT OAuth cache, whose default location is `~/.llmshim/chatgpt/auth.json`.
|
|
79
125
|
|
|
80
126
|
The CLI and proxy call `llmshim::env::load_all()` before constructing their
|
|
@@ -46,6 +46,19 @@ Every response and usage event reports `cache_read_tokens` and
|
|
|
46
46
|
add repeated streaming snapshots together. Zero means no cache tokens were
|
|
47
47
|
reported, not that the request was free.
|
|
48
48
|
|
|
49
|
+
They are joined by `uncached_input_tokens` — the prompt minus whatever cache
|
|
50
|
+
read the provider already counted inside it. Providers disagree on that:
|
|
51
|
+
Anthropic's `input_tokens` excludes the cache read, while the OpenAI Responses,
|
|
52
|
+
Chat Completions and Gemini prompt totals include it. llmshim resolves the
|
|
53
|
+
disagreement at the transport boundary so one counter means one thing.
|
|
54
|
+
|
|
55
|
+
`cost_usd` is the USD charged for the response: uncached input, output, cache
|
|
56
|
+
reads and cache writes each at their own catalog rate, so a cached prompt is
|
|
57
|
+
never billed twice. **`null` means the catalog carries no price for the model —
|
|
58
|
+
it never means free.** A model priced for input but not for the cache reads a
|
|
59
|
+
response actually used also yields `null`, rather than a partial sum that would
|
|
60
|
+
read as a complete one.
|
|
61
|
+
|
|
49
62
|
`ProviderRequest::can_continue_from` compares endpoint, credential headers,
|
|
50
63
|
settings and the full prior input prefix. `include`, `store`, `reasoning`, tool
|
|
51
64
|
schemas and other settings participate. The library sends full stateless
|
|
@@ -89,7 +89,19 @@ address. A success returns immediately. If every address fails, Rust returns
|
|
|
89
89
|
`all_failed` error response.
|
|
90
90
|
|
|
91
91
|
Each fallback model must resolve to a provider registered on the Router. A
|
|
92
|
-
chain can cross providers, but only when their keys are configured.
|
|
92
|
+
chain can cross providers, but only when their keys are configured. A chain
|
|
93
|
+
entry may also be a [named route](../concepts/routing.md#named-routes).
|
|
94
|
+
|
|
95
|
+
## Open circuits are skipped, not retried
|
|
96
|
+
|
|
97
|
+
A chain also consults provider health, before every attempt rather than once per
|
|
98
|
+
entry — the attempt that opens a circuit is usually the chain's own. When a
|
|
99
|
+
provider's circuit is open, because of too many `5xx`/transport failures in the
|
|
100
|
+
window, the chain abandons that entry, records `circuit open for provider …`
|
|
101
|
+
among the collected errors, and moves to the next address instead of spending
|
|
102
|
+
the rest of its retry budget on a target it already knows is dead. A `429` never
|
|
103
|
+
opens a circuit; that is rate limiting, handled by backoff. See
|
|
104
|
+
[Scaling and rate limits](../proxy/scaling.md#provider-health) for the knobs.
|
|
93
105
|
|
|
94
106
|
## Fallback is non-streaming only
|
|
95
107
|
|
|
@@ -146,16 +146,28 @@ consumption patterns, see [Streaming](../guides/streaming.md).
|
|
|
146
146
|
|
|
147
147
|
## Models and health
|
|
148
148
|
|
|
149
|
-
`GET /v1/models` returns the registry entries for configured providers
|
|
149
|
+
`GET /v1/models` returns the registry entries for configured providers, in two
|
|
150
|
+
shapes from one body:
|
|
150
151
|
|
|
151
152
|
```json
|
|
152
153
|
{
|
|
154
|
+
"object": "list",
|
|
155
|
+
"data": [
|
|
156
|
+
{"id": "openai/gpt-5.6-terra", "object": "model", "created": 0, "owned_by": "openai"}
|
|
157
|
+
],
|
|
153
158
|
"models": [
|
|
154
159
|
{"id": "openai/gpt-5.6-terra", "provider": "openai", "name": "gpt-5.6-terra"}
|
|
155
160
|
]
|
|
156
161
|
}
|
|
157
162
|
```
|
|
158
163
|
|
|
164
|
+
`object` + `data` is the OpenAI list envelope, so an OpenAI SDK's
|
|
165
|
+
`client.models.list()` works against llmshim unmodified. `models` is llmshim's
|
|
166
|
+
own shape and is unchanged. Both audiences issue the same `GET /v1/models`, so
|
|
167
|
+
there is no second path to split on. A `data[].id` is the routing id and can be
|
|
168
|
+
sent straight back as `model`; `created` is the catalog release date as a Unix
|
|
169
|
+
timestamp, or `0` when unknown.
|
|
170
|
+
|
|
159
171
|
This is discovery, not an allowlist: an arbitrary provider model ID can still
|
|
160
172
|
be routed explicitly. See [Model discovery](../reference/models.md).
|
|
161
173
|
|
|
@@ -33,7 +33,18 @@ its standard message/function shapes and `response_format`; JSON-object mode als
|
|
|
33
33
|
gets object validation. Generation fields supported by the destination retain
|
|
34
34
|
their ordinary adapter behavior. The facade supports one completion (`n:1`) and
|
|
35
35
|
custom function tools; it rejects provider-hosted tool definitions rather than
|
|
36
|
-
turning them into client-executed functions.
|
|
36
|
+
turning them into client-executed functions. `n > 1` is refused rather than
|
|
37
|
+
emulated — the OpenAI backend is the Responses API, which has no `n`, and
|
|
38
|
+
neither do Anthropic Messages or Gemini; fanning out N requests would change
|
|
39
|
+
the cost, rate-limit footprint and cache behavior of what was asked for. The
|
|
40
|
+
refusal is a properly shaped OpenAI error naming the parameter:
|
|
41
|
+
|
|
42
|
+
```json
|
|
43
|
+
{"error": {"message": "Unsupported value: 'n' must be 1. …",
|
|
44
|
+
"type": "invalid_request_error", "param": "n",
|
|
45
|
+
"code": "unsupported_parameter"}}
|
|
46
|
+
```
|
|
47
|
+
Provider-specific endpoints such as
|
|
37
48
|
batches, token counting, uploads and Responses are not served by these aliases.
|
|
38
49
|
|
|
39
50
|
With `stream:true`, text arrives incrementally. Chat Completions emits
|
|
@@ -75,6 +75,63 @@ When neither RPM nor TPM is set, proactive rate limiting is disabled;
|
|
|
75
75
|
concurrency backpressure still applies. Token permits are estimates based on
|
|
76
76
|
request size and requested output, not provider billing measurements.
|
|
77
77
|
|
|
78
|
+
## Provider health
|
|
79
|
+
|
|
80
|
+
Rate limiting and health are different questions. A `429` means the provider is
|
|
81
|
+
alive and asking for less, and the token buckets already slow it down. A
|
|
82
|
+
circuit breaker counts what retrying cannot fix — `500`, `502`, `503`, `504`,
|
|
83
|
+
`529` and transport failures — over a sliding window, opens the circuit at the
|
|
84
|
+
threshold, and admits a single probe after the cooldown.
|
|
85
|
+
|
|
86
|
+
| Variable | Default | Meaning |
|
|
87
|
+
|---|---:|---|
|
|
88
|
+
| `LLMSHIM_BREAKER_WINDOW_SECS` | `60` | Sliding window over which failures are counted |
|
|
89
|
+
| `LLMSHIM_BREAKER_TRIP_THRESHOLD` | `3` | Failures that open a circuit; `0` disables the breaker |
|
|
90
|
+
| `LLMSHIM_BREAKER_COOLDOWN_SECS` | `30` | Time an open circuit waits before admitting a probe |
|
|
91
|
+
|
|
92
|
+
Every dispatch path *observes* outcomes, so health accrues from ordinary
|
|
93
|
+
traffic. Only a [fallback chain](../guides/fallbacks.md) *refuses*: it skips a
|
|
94
|
+
provider with an open circuit instead of spending its retry budget on a target
|
|
95
|
+
it already knows is dead. A single-target request is still dispatched — with no
|
|
96
|
+
alternative, refusing would only convert an upstream failure into a local one.
|
|
97
|
+
|
|
98
|
+
## Spend caps
|
|
99
|
+
|
|
100
|
+
The experimental gateway enforces a per-identity USD cap beside the RPM/TPM
|
|
101
|
+
buckets. A gateway key's identity may carry `budget_usd` and an optional
|
|
102
|
+
`budget_window_secs` (default one day); over budget is a `429` with
|
|
103
|
+
`Retry-After` set to the window reset.
|
|
104
|
+
|
|
105
|
+
```json
|
|
106
|
+
{"sk-example": {"tenant": "acme", "tier": 1, "budget_usd": 100, "budget_window_secs": 86400}}
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
Cost is only knowable after a response, so the cap is checked before dispatch
|
|
110
|
+
and charged after: one in-flight request can overshoot.
|
|
111
|
+
|
|
112
|
+
A response the catalog cannot price is **not** charged — recording zero would let
|
|
113
|
+
an unpriced model run forever under a budget. So that a cap cannot silently stop
|
|
114
|
+
binding, a request whose target has **no catalog price is refused before it runs**
|
|
115
|
+
when a budget is set:
|
|
116
|
+
|
|
117
|
+
```
|
|
118
|
+
400 {"error":{"code":"unpriceable_under_budget","param":"model", …}}
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
It is deliberately not a `429`: retrying never clears it. Three ways forward —
|
|
122
|
+
use a priced model, add a local price override in the catalog, or accept the risk
|
|
123
|
+
explicitly per key:
|
|
124
|
+
|
|
125
|
+
```json
|
|
126
|
+
{"sk-example": {"tenant": "acme", "budget_usd": 100, "budget_allow_unpriced": true}}
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
`budget_allow_unpriced` defaults to `false`. With it set, those requests run and
|
|
130
|
+
are not charged, and each one logs a warning and increments
|
|
131
|
+
`llmshim_gateway_unpriced_under_cap_total{provider,model}` — a non-zero counter
|
|
132
|
+
means the budget is not binding for that target. An accepted risk should stay
|
|
133
|
+
measurable rather than become an assumption.
|
|
134
|
+
|
|
78
135
|
## One replica or a coordinated fleet
|
|
79
136
|
|
|
80
137
|
The default buckets are in memory. With `N` replicas, each replica enforces
|
|
@@ -88,8 +145,10 @@ cargo install llmshim --features redis-coordination
|
|
|
88
145
|
LLMSHIM_REDIS_URL=redis://redis.internal:6379 llmshim proxy
|
|
89
146
|
```
|
|
90
147
|
|
|
91
|
-
`redis-coordination` includes the `proxy` feature. Redis
|
|
92
|
-
|
|
148
|
+
`redis-coordination` includes the `proxy` feature. Redis coordinates rate-limit
|
|
149
|
+
buckets, provider health and — on the gateway — spend, so a shared limit, a
|
|
150
|
+
dead provider and a dollar cap all mean the same thing on every replica;
|
|
151
|
+
connection pools and concurrency limits remain per process. If
|
|
93
152
|
Redis becomes unavailable at runtime, limiting fails open so requests continue.
|
|
94
153
|
If the Redis client cannot be initialized—or the binary lacks the feature—the
|
|
95
154
|
proxy warns and falls back to in-memory buckets.
|
|
@@ -11,7 +11,7 @@ the module stays compatible with Python 3.9 (no ``typing.NotRequired``).
|
|
|
11
11
|
|
|
12
12
|
from __future__ import annotations
|
|
13
13
|
|
|
14
|
-
from typing import Any, List, Literal, TypedDict, Union
|
|
14
|
+
from typing import Any, List, Literal, Optional, TypedDict, Union
|
|
15
15
|
|
|
16
16
|
__all__ = [
|
|
17
17
|
"Role",
|
|
@@ -190,6 +190,9 @@ class Usage(TypedDict, total=False):
|
|
|
190
190
|
total_tokens: int
|
|
191
191
|
cache_read_tokens: int
|
|
192
192
|
cache_write_tokens: int
|
|
193
|
+
#: USD charged for this response. ``None`` means the server could not price
|
|
194
|
+
#: the model — it never means free.
|
|
195
|
+
cost_usd: Optional[float]
|
|
193
196
|
|
|
194
197
|
|
|
195
198
|
class _ResponseMessageBase(TypedDict):
|
|
@@ -285,6 +288,8 @@ class UsageEvent(TypedDict, total=False):
|
|
|
285
288
|
total_tokens: int
|
|
286
289
|
cache_read_tokens: int
|
|
287
290
|
cache_write_tokens: int
|
|
291
|
+
#: ``None`` when the server could not price the model, never ``0.0``.
|
|
292
|
+
cost_usd: Optional[float]
|
|
288
293
|
|
|
289
294
|
|
|
290
295
|
class _DoneBase(TypedDict):
|