llmshim 0.6.0__tar.gz → 0.7.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {llmshim-0.6.0 → llmshim-0.7.1}/CLAUDE.md +10 -2
- {llmshim-0.6.0 → llmshim-0.7.1}/Cargo.lock +1 -1
- {llmshim-0.6.0 → llmshim-0.7.1}/Cargo.toml +1 -1
- {llmshim-0.6.0 → llmshim-0.7.1}/PKG-INFO +1 -1
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/caching.md +8 -1
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/proxy/scaling.md +12 -5
- {llmshim-0.6.0 → llmshim-0.7.1}/src/cache.rs +55 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/client.rs +76 -1
- {llmshim-0.6.0 → llmshim-0.7.1}/src/cost.rs +57 -11
- {llmshim-0.6.0 → llmshim-0.7.1}/src/fallback.rs +3 -6
- {llmshim-0.6.0 → llmshim-0.7.1}/src/lib.rs +11 -14
- {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/anthropic.rs +11 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/gemini.rs +6 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/openai.rs +7 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/openai_compat.rs +10 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/openrouter.rs +8 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/xai.rs +7 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/types.rs +6 -1
- llmshim-0.7.1/tests/unit_client_breaker.rs +118 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_provider_contracts.rs +53 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/.github/workflows/catalog-refresh.yml +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/.github/workflows/pages.yml +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/.github/workflows/release.yml +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/.gitignore +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/CODE_OF_CONDUCT.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/CONTRIBUTING.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/LICENSE-APACHE +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/LICENSE-MIT +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/NOTICE +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/README.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/SECURITY.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/benchmarks/bench.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/benchmarks/bench_python.py +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/benchmarks/gateway_loadtest.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/benchmarks/loadtest.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/Cargo.toml +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/LICENSE-APACHE +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/LICENSE-MIT +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/README.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/data/LICENSE.models.dev +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/data/README.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/data/models.dev.json +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/aliases.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/builtin.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/capabilities.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/lib.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/merge.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/parse.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/refresh.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/types.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/tests/catalog.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/tests/refresh.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/.gitignore +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/book.toml +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/mermaid-init.js +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/mermaid.min.js +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/SUMMARY.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/concepts/contracts.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/concepts/conversations.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/concepts/portability.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/concepts/routing.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/concepts/translation-flow.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/capabilities.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/fallbacks.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/images.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/native-controls.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/reasoning.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/schemas.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/streaming.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/tools.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/introduction.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/proxy/deployment.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/proxy/http-api.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/proxy/native-apis.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/api.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/cli.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/configuration.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/errors.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/models.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/providers.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/request-fields.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/surfaces.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/start/choose.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/start/cli.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/start/clients.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/start/configure.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/start/proxy.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/start/rust.md +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/examples/chat.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/examples/stream.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/llmshim/__init__.py +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/llmshim/_client.py +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/llmshim/_server.py +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/llmshim/types.py +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/pyproject.toml +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/breaker.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/cli.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/config.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/env.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/error/normalize.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/error.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/gateway/auth.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/gateway/distributed.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/gateway/http.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/gateway/idempotency.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/gateway/metrics.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/gateway/mod.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/gateway/quota.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/log.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/main.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/models.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/provider.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/anthropic_reasoning.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/anthropic_signature.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/chatgpt/auth.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/chatgpt/mod.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/chatgpt/streaming.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/mod.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/convert.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/error.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/handlers.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/health.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/mod.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/ratelimit.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/wire/mod.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/wire/receipts.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/reasoning/normalize.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/reasoning.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/router.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/schema/memo.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/schema/mod.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/schema/validate.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/schema/walk.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/shim.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/streaming.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/toolcall/streaming.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/toolcall.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/usage.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/src/vision.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/fixtures/chatgpt-red.png +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_chatgpt.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_chatgpt_proxy.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_current_models.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_fallback.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_gemini.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_gemini_tools.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_long_context.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_multimodel.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_openrouter.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_proxy.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_sglang.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_thinking.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_tool_roundtrip.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_vision.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_xai.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/support/completion_status.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_advertised_models.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_anthropic.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_cache.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_chatgpt.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_cli.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_fable.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_fallback.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_fast_mode.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_gemini.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_log.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_models.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_multimodel.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_openai.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_openai_compat.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_openrouter.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_proxy.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_proxy_convert.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_reasoning.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_reasoning_profile.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_router.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_schema.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_shim.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_signature.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_sse.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_toolcall.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_tools.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_usage.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_vision.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_wire.rs +0 -0
- {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_xai.rs +0 -0
|
@@ -228,8 +228,16 @@ own: sliding failure window, open state, and a single half-open probe admitted
|
|
|
228
228
|
after the cooldown. Config: `LLMSHIM_BREAKER_WINDOW_SECS` (60),
|
|
229
229
|
`LLMSHIM_BREAKER_TRIP_THRESHOLD` (3; `0` disables), `LLMSHIM_BREAKER_COOLDOWN_SECS` (30).
|
|
230
230
|
|
|
231
|
-
The breaker hangs on the `Router` (`Router::breaker()` / `with_breaker`)
|
|
232
|
-
|
|
231
|
+
The breaker hangs on the `Router` (`Router::breaker()` / `with_breaker`), but
|
|
232
|
+
the *counting* happens in `ShimClient`: `ShimClient::with_breaker` attaches one,
|
|
233
|
+
and `completion` / `stream` / `stream_owned` observe their final result exactly
|
|
234
|
+
once. The top-level entry points bind the router's breaker to the shared client
|
|
235
|
+
per call (`lib.rs::bound_client`), so `llmshim::completion`, `stream`,
|
|
236
|
+
`completion_with_fallback` and a caller that resolves its own provider and dials
|
|
237
|
+
`ShimClient` directly all feed the same breaker — the last one only if it opted
|
|
238
|
+
in with `ShimClient::new().with_breaker(router.breaker().clone())`; a bare
|
|
239
|
+
`ShimClient::new()` reports to nobody. Do not add a second `.observe` around a
|
|
240
|
+
client call: one call, one observation (`tests/unit_client_breaker.rs`). Only
|
|
233
241
|
`fallback.rs` *refuses*, and it checks before every attempt rather than once per
|
|
234
242
|
chain entry — the attempt that opens a circuit is usually the chain's own, so a
|
|
235
243
|
per-entry check would still retry into a target it just watched die. A single-target call is still dispatched:
|
|
@@ -22,7 +22,14 @@ llmshim translates the declaration into the provider's caching mechanism.
|
|
|
22
22
|
}
|
|
23
23
|
```
|
|
24
24
|
|
|
25
|
-
`upto_message` is a zero-based index into the
|
|
25
|
+
`upto_message` is a zero-based index into the `messages` array exactly as you
|
|
26
|
+
sent it — system and developer messages count, and each `role: "tool"` result
|
|
27
|
+
counts as its own entry — not into the provider-native array (Anthropic hoists
|
|
28
|
+
the system prompt out, so native numbering differs). A caller whose own message
|
|
29
|
+
model expands into more wire messages than it holds must remap before copying
|
|
30
|
+
an index across. On Anthropic an index past the end is rejected with a 400,
|
|
31
|
+
and one that lands too early silently caches less than intended; on every
|
|
32
|
+
other provider segments are parsed but neither checked nor placed. Labels are
|
|
26
33
|
informational and are never used to guess prompt semantics. Keep stable sections
|
|
27
34
|
before volatile sections. The proxy accepts the same top-level `x-cache` field;
|
|
28
35
|
Python and Ruby provide a `cache=`/`cache:` keyword, Go has `ChatRequest.Cache`,
|
|
@@ -106,11 +106,18 @@ buckets. A gateway key's identity may carry `budget_usd` and an optional
|
|
|
106
106
|
{"sk-example": {"tenant": "acme", "tier": 1, "budget_usd": 100, "budget_window_secs": 86400}}
|
|
107
107
|
```
|
|
108
108
|
|
|
109
|
-
Cost is only knowable after a response, so the cap is checked before dispatch
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
109
|
+
Cost is only knowable after a response, so the cap is checked before dispatch and
|
|
110
|
+
charged after. Everything admitted between the last charge and the next check
|
|
111
|
+
passes, so the overshoot bound is **admitted concurrency × the most expensive
|
|
112
|
+
request**, multiplied again across replicas that have not yet shared their
|
|
113
|
+
ledger. Size a cap with that headroom in mind rather than as a hard ceiling.
|
|
114
|
+
|
|
115
|
+
A response the catalog cannot price at all is **not** charged — recording zero
|
|
116
|
+
would let an unpriced model run forever under a budget. A model that prices only
|
|
117
|
+
*some* token classes is charged at its highest published rate for the rest, so a
|
|
118
|
+
partial price bounds the charge from above instead of voiding it: 2,537 of the
|
|
119
|
+
7,461 priced models in the catalog publish no `cache_read` rate, and voiding
|
|
120
|
+
those would have reopened this same hole one layer down. So that a cap cannot silently stop
|
|
114
121
|
binding, a request whose target has **no catalog price is refused before it runs**
|
|
115
122
|
when a budget is set:
|
|
116
123
|
|
|
@@ -10,20 +10,75 @@ use std::collections::BTreeMap;
|
|
|
10
10
|
|
|
11
11
|
pub const ANTHROPIC_BREAKPOINT_LIMIT: usize = 4;
|
|
12
12
|
|
|
13
|
+
/// How long the caller expects a prefix to stay byte-stable. Only the caller
|
|
14
|
+
/// knows; llmshim never infers it.
|
|
13
15
|
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
|
|
14
16
|
#[serde(rename_all = "lowercase")]
|
|
15
17
|
pub enum Stability {
|
|
18
|
+
/// Stable across sessions — a one-hour breakpoint on Anthropic.
|
|
16
19
|
Static,
|
|
20
|
+
/// Stable for this session — a five-minute breakpoint on Anthropic.
|
|
17
21
|
Session,
|
|
22
|
+
/// The volatile tail. Places no marker; it exists so a caller can name
|
|
23
|
+
/// where stability ends.
|
|
18
24
|
Turn,
|
|
19
25
|
}
|
|
26
|
+
|
|
27
|
+
/// One caller-declared stability boundary: every message up to and including
|
|
28
|
+
/// `upto_message` is expected to stay byte-stable for as long as `stability`
|
|
29
|
+
/// says.
|
|
30
|
+
///
|
|
31
|
+
/// **`upto_message` indexes the request's own `messages` array, exactly as the
|
|
32
|
+
/// caller sent it.** Zero-based, and every entry counts — `system` and
|
|
33
|
+
/// `developer` messages included, and each `role: "tool"` result as its own
|
|
34
|
+
/// entry — because the annotation is applied before any adapter hoists the
|
|
35
|
+
/// system prompt out or reshapes tool results into native turns. It is *not*
|
|
36
|
+
/// an index into the provider-native array Anthropic receives, where the
|
|
37
|
+
/// system message is gone and the numbering has shifted.
|
|
38
|
+
///
|
|
39
|
+
/// A caller whose own message model is richer than the wire's — one entry that
|
|
40
|
+
/// expands into a leading system message plus one wire message per tool
|
|
41
|
+
/// result, say — must remap to the index of the wire message it actually sent.
|
|
42
|
+
/// Copied across unchanged, the boundary lands on the wrong message: too far
|
|
43
|
+
/// and the request is rejected (`400 invalid x-cache: segment message index is
|
|
44
|
+
/// out of range`); too near and less of the prefix is cached than was hashed,
|
|
45
|
+
/// with nothing to say so.
|
|
46
|
+
///
|
|
47
|
+
/// llmshim's own insertions never move this index. A managed-output
|
|
48
|
+
/// instruction merges into an existing system message, and when it has to
|
|
49
|
+
/// prepend one it renumbers every segment itself (`shim.rs`,
|
|
50
|
+
/// `prepend_instruction`).
|
|
51
|
+
///
|
|
52
|
+
/// Honoured on the Anthropic Messages wire only, where it becomes a
|
|
53
|
+
/// `cache_control` breakpoint on the last block of that message (or the last
|
|
54
|
+
/// tool call, or the tool result itself) — unless that last block is a
|
|
55
|
+
/// thinking block, in which case the segment is skipped without a marker.
|
|
56
|
+
/// The bounds check is Anthropic-only too: on every other wire the policy is
|
|
57
|
+
/// still parsed (malformed `x-cache` fails everywhere), but a segment's index
|
|
58
|
+
/// is neither checked nor placed — an out-of-range index there is silently
|
|
59
|
+
/// ignored, not rejected. `label` is the caller's own tag; llmshim never
|
|
60
|
+
/// reads it.
|
|
20
61
|
#[derive(Debug, Clone, Serialize, Deserialize)]
|
|
21
62
|
pub struct CacheSegment {
|
|
63
|
+
/// Index into the request's `messages` as sent — see the type-level note.
|
|
22
64
|
pub upto_message: usize,
|
|
23
65
|
#[serde(default)]
|
|
24
66
|
pub label: String,
|
|
25
67
|
pub stability: Stability,
|
|
26
68
|
}
|
|
69
|
+
|
|
70
|
+
/// The caller's `x-cache` annotation: stability boundaries plus an optional
|
|
71
|
+
/// cache identity. Read once per request and never forwarded to a provider.
|
|
72
|
+
///
|
|
73
|
+
/// The two halves land on different wires. `segments` are Anthropic-only —
|
|
74
|
+
/// see [`CacheSegment`] for the index convention and the no-op rule elsewhere.
|
|
75
|
+
/// `key` is not: on the OpenAI Responses wire it becomes `prompt_cache_key`,
|
|
76
|
+
/// and is ignored on the others. A request may carry both; each wire takes the
|
|
77
|
+
/// half it can use.
|
|
78
|
+
///
|
|
79
|
+
/// Explicit segments supersede any request-level `cache_control` the caller
|
|
80
|
+
/// placed, and Anthropic's four-breakpoint budget is spent from the end of the
|
|
81
|
+
/// request backwards, so the boundaries nearest the tail are the ones kept.
|
|
27
82
|
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
|
|
28
83
|
pub struct CachePolicy {
|
|
29
84
|
#[serde(default)]
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
use crate::breaker::ProviderBreaker;
|
|
1
2
|
use crate::error::{Result, ShimError};
|
|
2
3
|
use crate::provider::{Provider, ProviderRequest};
|
|
3
4
|
use bytes::Bytes;
|
|
@@ -7,6 +8,7 @@ use futures::{Stream, StreamExt};
|
|
|
7
8
|
use reqwest::header::HeaderMap;
|
|
8
9
|
use reqwest::Client;
|
|
9
10
|
use std::pin::Pin;
|
|
11
|
+
use std::sync::Arc;
|
|
10
12
|
use std::time::Duration;
|
|
11
13
|
|
|
12
14
|
/// Retry bounds, resolved once from the environment (with defaults) at
|
|
@@ -55,6 +57,9 @@ fn env_parse<T: std::str::FromStr>(key: &str) -> Option<T> {
|
|
|
55
57
|
pub struct ShimClient {
|
|
56
58
|
http: Client,
|
|
57
59
|
retry: RetryConfig,
|
|
60
|
+
/// Provider health, fed by every dispatch this client makes. `None` means
|
|
61
|
+
/// this client reports to nobody — see [`ShimClient::with_breaker`].
|
|
62
|
+
breaker: Option<Arc<ProviderBreaker>>,
|
|
58
63
|
}
|
|
59
64
|
|
|
60
65
|
impl Default for ShimClient {
|
|
@@ -77,6 +82,36 @@ impl ShimClient {
|
|
|
77
82
|
.build()
|
|
78
83
|
.expect("failed to build HTTP client"),
|
|
79
84
|
retry: RetryConfig::from_env(),
|
|
85
|
+
breaker: None,
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/// Report every dispatch's outcome to `breaker`.
|
|
90
|
+
///
|
|
91
|
+
/// The breaker lives on the [`Router`](crate::router::Router), so a caller
|
|
92
|
+
/// that resolves a provider itself and comes straight here has to hand it
|
|
93
|
+
/// over: `ShimClient::new().with_breaker(router.breaker().clone())`. The
|
|
94
|
+
/// crate's own entry points (`llmshim::completion`, `stream`,
|
|
95
|
+
/// `completion_with_fallback`) bind the router's breaker this way, so this
|
|
96
|
+
/// is the one place a dispatch is counted — whichever door it came in by.
|
|
97
|
+
///
|
|
98
|
+
/// Cheap: the HTTP connection pool is shared by clone, so binding a breaker
|
|
99
|
+
/// per call costs an `Arc` clone, not a new pool.
|
|
100
|
+
pub fn with_breaker(mut self, breaker: Arc<ProviderBreaker>) -> Self {
|
|
101
|
+
self.breaker = Some(breaker);
|
|
102
|
+
self
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/// Record one dispatch against the attached breaker, if any. Called once
|
|
106
|
+
/// per public entry point on its *final* result — after transport retries
|
|
107
|
+
/// and any output-contract repair — so one caller-visible call is one
|
|
108
|
+
/// observation, never one per attempt.
|
|
109
|
+
///
|
|
110
|
+
/// Takes the projected outcome rather than the result itself so a stream's
|
|
111
|
+
/// non-`Sync` body is never borrowed across the await.
|
|
112
|
+
async fn observe(&self, provider: &dyn Provider, outcome: std::result::Result<(), &ShimError>) {
|
|
113
|
+
if let Some(breaker) = &self.breaker {
|
|
114
|
+
breaker.observe(provider.name(), outcome).await;
|
|
80
115
|
}
|
|
81
116
|
}
|
|
82
117
|
|
|
@@ -167,6 +202,17 @@ impl ShimClient {
|
|
|
167
202
|
provider: &dyn Provider,
|
|
168
203
|
model: &str,
|
|
169
204
|
request: &serde_json::Value,
|
|
205
|
+
) -> Result<serde_json::Value> {
|
|
206
|
+
let result = self.completion_unobserved(provider, model, request).await;
|
|
207
|
+
self.observe(provider, result.as_ref().map(|_| ())).await;
|
|
208
|
+
result
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
async fn completion_unobserved(
|
|
212
|
+
&self,
|
|
213
|
+
provider: &dyn Provider,
|
|
214
|
+
model: &str,
|
|
215
|
+
request: &serde_json::Value,
|
|
170
216
|
) -> Result<serde_json::Value> {
|
|
171
217
|
let plan = crate::shim::Plan::new(
|
|
172
218
|
provider.name(),
|
|
@@ -232,6 +278,19 @@ impl ShimClient {
|
|
|
232
278
|
provider: &dyn Provider,
|
|
233
279
|
model: &str,
|
|
234
280
|
request: &serde_json::Value,
|
|
281
|
+
) -> Result<Pin<Box<dyn Stream<Item = Result<String>> + Send>>> {
|
|
282
|
+
// A stream's health verdict is whether it opened; per-chunk failures
|
|
283
|
+
// are the transport's business, not the breaker's.
|
|
284
|
+
let opened = self.stream_unobserved(provider, model, request).await;
|
|
285
|
+
self.observe(provider, opened.as_ref().map(|_| ())).await;
|
|
286
|
+
opened
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
async fn stream_unobserved(
|
|
290
|
+
&self,
|
|
291
|
+
provider: &dyn Provider,
|
|
292
|
+
model: &str,
|
|
293
|
+
request: &serde_json::Value,
|
|
235
294
|
) -> Result<Pin<Box<dyn Stream<Item = Result<String>> + Send>>> {
|
|
236
295
|
let plan = crate::shim::Plan::new(
|
|
237
296
|
provider.name(),
|
|
@@ -278,7 +337,23 @@ impl ShimClient {
|
|
|
278
337
|
/// and keepalives while validation and a possible repair are in progress.
|
|
279
338
|
pub async fn stream_owned(
|
|
280
339
|
&self,
|
|
281
|
-
provider:
|
|
340
|
+
provider: Arc<dyn Provider>,
|
|
341
|
+
model: &str,
|
|
342
|
+
request: &serde_json::Value,
|
|
343
|
+
) -> Result<Pin<Box<dyn Stream<Item = Result<String>> + Send>>> {
|
|
344
|
+
// Observed on the open only. A buffered plan's repair re-opens inside
|
|
345
|
+
// the returned stream; that second dial is not a separate verdict.
|
|
346
|
+
let opened = self
|
|
347
|
+
.stream_owned_unobserved(provider.clone(), model, request)
|
|
348
|
+
.await;
|
|
349
|
+
self.observe(provider.as_ref(), opened.as_ref().map(|_| ()))
|
|
350
|
+
.await;
|
|
351
|
+
opened
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
async fn stream_owned_unobserved(
|
|
355
|
+
&self,
|
|
356
|
+
provider: Arc<dyn Provider>,
|
|
282
357
|
model: &str,
|
|
283
358
|
request: &serde_json::Value,
|
|
284
359
|
) -> Result<Pin<Box<dyn Stream<Item = Result<String>> + Send>>> {
|
|
@@ -29,6 +29,27 @@ fn charge(tokens: u64, rate: Option<f64>) -> Option<f64> {
|
|
|
29
29
|
Some(tokens as f64 * rate? / PER_MILLION)
|
|
30
30
|
}
|
|
31
31
|
|
|
32
|
+
/// The highest rate this model publishes, used where a class has none of its own.
|
|
33
|
+
///
|
|
34
|
+
/// Catalogs price partially and often: **2,537 of 7,461 priced models in the
|
|
35
|
+
/// vendored snapshot carry no `cache_read` rate and 5,892 no `cache_write`**,
|
|
36
|
+
/// including `gpt-5-pro` and `o3-pro`, priced `{input, output}` only. Returning
|
|
37
|
+
/// `None` for those made the whole response unpriceable the moment it reported a
|
|
38
|
+
/// cached token — and a discarded charge is a spend cap that silently stops
|
|
39
|
+
/// binding, which is the failure this crate exists to prevent.
|
|
40
|
+
///
|
|
41
|
+
/// Substituting the highest published rate can only over-estimate, never under.
|
|
42
|
+
/// For a cap that is the safe direction: it spends a budget slightly early, where
|
|
43
|
+
/// under-estimating spends it forever.
|
|
44
|
+
fn highest_published_rate(cost: &Cost) -> Option<f64> {
|
|
45
|
+
[cost.input, cost.output, cost.cache_read, cost.cache_write]
|
|
46
|
+
.into_iter()
|
|
47
|
+
.flatten()
|
|
48
|
+
.fold(None, |best: Option<f64>, rate| {
|
|
49
|
+
Some(best.map_or(rate, |b| b.max(rate)))
|
|
50
|
+
})
|
|
51
|
+
}
|
|
52
|
+
|
|
32
53
|
/// Price one normalized usage object.
|
|
33
54
|
///
|
|
34
55
|
/// Input is charged on `uncached_input_tokens` — the prompt minus whatever
|
|
@@ -53,11 +74,17 @@ pub fn price(usage: &Value, cost: &Cost) -> Option<f64> {
|
|
|
53
74
|
return None;
|
|
54
75
|
}
|
|
55
76
|
|
|
77
|
+
// A class the model does not price falls back to its highest published rate
|
|
78
|
+
// rather than voiding the total. See `highest_published_rate`: the result is
|
|
79
|
+
// an upper bound, which is the only safe direction for a budget.
|
|
80
|
+
let ceiling = highest_published_rate(cost);
|
|
81
|
+
let rate_for = |own: Option<f64>| own.or(ceiling);
|
|
82
|
+
|
|
56
83
|
Some(
|
|
57
|
-
charge(input, cost.input)?
|
|
58
|
-
+ charge(count(usage, "completion_tokens"), cost.output)?
|
|
59
|
-
+ charge(cache_read, cost.cache_read)?
|
|
60
|
-
+ charge(cache_write, cost.cache_write)?,
|
|
84
|
+
charge(input, rate_for(cost.input))?
|
|
85
|
+
+ charge(count(usage, "completion_tokens"), rate_for(cost.output))?
|
|
86
|
+
+ charge(cache_read, rate_for(cost.cache_read))?
|
|
87
|
+
+ charge(cache_write, rate_for(cost.cache_write))?,
|
|
61
88
|
)
|
|
62
89
|
}
|
|
63
90
|
|
|
@@ -221,14 +248,21 @@ mod tests {
|
|
|
221
248
|
}
|
|
222
249
|
|
|
223
250
|
#[test]
|
|
224
|
-
fn
|
|
251
|
+
fn a_missing_cache_rate_is_charged_at_the_models_highest_rate() {
|
|
252
|
+
// This replaced an assertion that a used-but-unpriced class voids the
|
|
253
|
+
// total. That instinct was right — never under-report — but the remedy
|
|
254
|
+
// was wrong: `None` is discarded by `SpendCap::record`, so a spend cap
|
|
255
|
+
// silently stopped binding for the 2,537 catalogued models that publish
|
|
256
|
+
// no `cache_read` rate. An upper bound honours the same principle
|
|
257
|
+
// without the hole.
|
|
225
258
|
let partial = Cost {
|
|
226
259
|
input: Some(3.0),
|
|
227
260
|
output: Some(15.0),
|
|
228
261
|
cache_read: None,
|
|
229
262
|
cache_write: None,
|
|
230
263
|
};
|
|
231
|
-
|
|
264
|
+
|
|
265
|
+
// No cache tokens: the missing rates never come into play.
|
|
232
266
|
close(
|
|
233
267
|
price(
|
|
234
268
|
&json!({"uncached_input_tokens": 1_000_000, "completion_tokens": 0, "cache_read_tokens": 0}),
|
|
@@ -236,14 +270,26 @@ mod tests {
|
|
|
236
270
|
),
|
|
237
271
|
3.0,
|
|
238
272
|
);
|
|
239
|
-
|
|
240
|
-
|
|
273
|
+
|
|
274
|
+
// Cache tokens used with no rate of their own: charged at 15.0, the
|
|
275
|
+
// highest rate this model publishes — an over-estimate, never an under.
|
|
276
|
+
close(
|
|
241
277
|
price(
|
|
242
|
-
&json!({"uncached_input_tokens": 1_000_000, "cache_read_tokens":
|
|
278
|
+
&json!({"uncached_input_tokens": 1_000_000, "cache_read_tokens": 1_000_000}),
|
|
243
279
|
&partial,
|
|
244
280
|
),
|
|
245
|
-
|
|
246
|
-
|
|
281
|
+
18.0,
|
|
282
|
+
);
|
|
283
|
+
|
|
284
|
+
// The bound must never fall below what a complete price would charge.
|
|
285
|
+
let complete = Cost {
|
|
286
|
+
cache_read: Some(0.3),
|
|
287
|
+
..partial
|
|
288
|
+
};
|
|
289
|
+
let usage = json!({"uncached_input_tokens": 1_000_000, "cache_read_tokens": 1_000_000});
|
|
290
|
+
assert!(
|
|
291
|
+
price(&usage, &partial).unwrap() >= price(&usage, &complete).unwrap(),
|
|
292
|
+
"a fallback rate must bound the real one from above, or a cap under-charges"
|
|
247
293
|
);
|
|
248
294
|
}
|
|
249
295
|
|
|
@@ -1,7 +1,6 @@
|
|
|
1
1
|
use crate::error::{Result, ShimError};
|
|
2
2
|
use crate::log::{LogEntry, Logger, RequestTimer};
|
|
3
3
|
use crate::router::Router;
|
|
4
|
-
use crate::SHARED_CLIENT;
|
|
5
4
|
use serde_json::Value;
|
|
6
5
|
use std::time::Duration;
|
|
7
6
|
|
|
@@ -75,7 +74,9 @@ pub async fn completion_with_fallback(
|
|
|
75
74
|
};
|
|
76
75
|
|
|
77
76
|
let mut errors: Vec<String> = Vec::new();
|
|
78
|
-
|
|
77
|
+
// Every attempt below is counted by the client against this router's
|
|
78
|
+
// breaker; the loop only asks `admit` before dialling.
|
|
79
|
+
let client = crate::bound_client(router);
|
|
79
80
|
|
|
80
81
|
for model_str in &models {
|
|
81
82
|
// Build request with this model. A named route expands to its model and
|
|
@@ -118,10 +119,6 @@ pub async fn completion_with_fallback(
|
|
|
118
119
|
// Keep OAuth preparation, SSE-only providers, reasoning provenance,
|
|
119
120
|
// and tool normalization identical to an ordinary completion.
|
|
120
121
|
let outcome = client.completion(provider, &model, &req).await;
|
|
121
|
-
router
|
|
122
|
-
.breaker()
|
|
123
|
-
.observe(provider.name(), outcome.as_ref().map(|_| ()))
|
|
124
|
-
.await;
|
|
125
122
|
match outcome {
|
|
126
123
|
Ok(result) => {
|
|
127
124
|
if let Some(logger) = logger {
|
|
@@ -79,16 +79,13 @@ pub async fn completion_with_logger(
|
|
|
79
79
|
.ok_or(error::ShimError::MissingModel)?;
|
|
80
80
|
|
|
81
81
|
let (provider, model) = router.resolve(model_str)?;
|
|
82
|
-
let client =
|
|
82
|
+
let client = bound_client(router);
|
|
83
83
|
let timer = RequestTimer::start();
|
|
84
84
|
|
|
85
85
|
// Ordinary traffic feeds provider health too, so a chain's first fallback
|
|
86
86
|
// decision is not the first thing that ever noticed a provider is down.
|
|
87
|
+
// The client does the counting; see `ShimClient::with_breaker`.
|
|
87
88
|
let result = client.completion(provider, &model, request).await;
|
|
88
|
-
router
|
|
89
|
-
.breaker()
|
|
90
|
-
.observe(provider.name(), result.as_ref().map(|_| ()))
|
|
91
|
-
.await;
|
|
92
89
|
|
|
93
90
|
match result {
|
|
94
91
|
Ok(resp) => {
|
|
@@ -132,13 +129,13 @@ pub async fn stream(
|
|
|
132
129
|
// Observed but not gated: a single-target call has no alternative, so
|
|
133
130
|
// refusing here would only convert an upstream failure into a local one.
|
|
134
131
|
// The breaker refuses where there is somewhere else to go — `fallback.rs`.
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
132
|
+
bound_client(router)
|
|
133
|
+
.stream_owned(provider, &model, request)
|
|
134
|
+
.await
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
/// The shared HTTP client, reporting to this router's breaker. The pool is
|
|
138
|
+
/// shared by clone; only the breaker handle is per call.
|
|
139
|
+
pub(crate) fn bound_client(router: &Router) -> ShimClient {
|
|
140
|
+
SHARED_CLIENT.clone().with_breaker(router.breaker().clone())
|
|
144
141
|
}
|
|
@@ -116,6 +116,17 @@ fn extract_system_message(
|
|
|
116
116
|
(system, rest)
|
|
117
117
|
}
|
|
118
118
|
|
|
119
|
+
/// One Chat Completions message in, one Anthropic message out — a `role:
|
|
120
|
+
/// "tool"` result included, which becomes its own `user` message even when it
|
|
121
|
+
/// sits beside another.
|
|
122
|
+
///
|
|
123
|
+
/// Same-role adjacency is deliberately left alone. The Messages API accepts
|
|
124
|
+
/// it: its reference states that consecutive `user` or `assistant` turns in a
|
|
125
|
+
/// request are combined into a single turn server-side (platform.claude.com,
|
|
126
|
+
/// Messages API, `messages` parameter), and parallel tool results already
|
|
127
|
+
/// reach it here as back-to-back `user` messages. Merging locally would only
|
|
128
|
+
/// destroy message boundaries a caller may key on. Gemini is the one wire in
|
|
129
|
+
/// this crate that rejects adjacency, and it merges in its own adapter.
|
|
119
130
|
fn transform_messages(messages: &[Value]) -> Vec<Value> {
|
|
120
131
|
messages
|
|
121
132
|
.iter()
|
|
@@ -114,6 +114,12 @@ fn transform_messages(messages: &[Value]) -> (Option<Value>, Vec<Value>) {
|
|
|
114
114
|
(system_instruction, contents)
|
|
115
115
|
}
|
|
116
116
|
|
|
117
|
+
/// Gemini's own repair, not a shared one. This is the only wire in the crate
|
|
118
|
+
/// known to reject adjacent same-role turns, so it is the only adapter that
|
|
119
|
+
/// folds them together. Every other adapter passes adjacency through — each
|
|
120
|
+
/// says why on its own message pass — because a merge destroys message
|
|
121
|
+
/// boundaries a caller may depend on, and only a wire that would otherwise
|
|
122
|
+
/// fail the request earns that.
|
|
117
123
|
fn merge_same_role(turns: Vec<Value>) -> Vec<Value> {
|
|
118
124
|
let mut merged: Vec<Value> = Vec::new();
|
|
119
125
|
for turn in turns {
|
|
@@ -46,6 +46,13 @@ fn strip_cache_control(value: &mut Value) {
|
|
|
46
46
|
/// - Assistant messages with tool_calls → split into the assistant message +
|
|
47
47
|
/// separate `function_call` items
|
|
48
48
|
/// - `role: "tool"` messages → `function_call_output` items
|
|
49
|
+
///
|
|
50
|
+
/// Same-role adjacency passes through. `input` is a flat item list, and the
|
|
51
|
+
/// Responses reference documents roles and their precedence but no ordering or
|
|
52
|
+
/// alternation rule (developers.openai.com, Responses API, `input`); this
|
|
53
|
+
/// adapter already emits several `function_call_output` items in a row for
|
|
54
|
+
/// parallel tool calls. Two adjacent assistant messages stay two items. The
|
|
55
|
+
/// ChatGPT adapter goes through this same translator and inherits the stance.
|
|
49
56
|
fn sanitize_messages(messages: &[Value]) -> Vec<Value> {
|
|
50
57
|
let mut result = Vec::new();
|
|
51
58
|
for msg in messages {
|
|
@@ -40,6 +40,16 @@ impl OpenAiCompatible {
|
|
|
40
40
|
/// Strip llmshim-normalized / foreign-provider fields and normalize content
|
|
41
41
|
/// blocks to Chat Completions form. Messages, `tool_calls`, and `role: "tool"`
|
|
42
42
|
/// stay in Chat Completions shape (the target format).
|
|
43
|
+
///
|
|
44
|
+
/// Same-role adjacency passes through unchanged, and here the answer is
|
|
45
|
+
/// genuinely the served model's. vLLM and SGLang render `messages` through the
|
|
46
|
+
/// tokenizer's Jinja chat template (or `--chat-template`), so acceptance is a
|
|
47
|
+
/// property of that template: most current ones accept adjacent turns, some
|
|
48
|
+
/// older ones raise — Mistral-7B-Instruct-v0.1's template errors with
|
|
49
|
+
/// "conversation roles must alternate user/assistant/user/assistant/...".
|
|
50
|
+
/// llmshim cannot see the template, so it does not merge; a strict template's
|
|
51
|
+
/// rejection surfaces as the server's own 400. No `x-vllm` / `x-sglang`
|
|
52
|
+
/// parameter changes this.
|
|
43
53
|
fn sanitize_messages(messages: &[Value]) -> Vec<Value> {
|
|
44
54
|
messages
|
|
45
55
|
.iter()
|
|
@@ -61,6 +61,14 @@ fn normalize_openrouter_effort(effort: &str, pro: bool) -> &'static str {
|
|
|
61
61
|
/// llmshim-normalized / foreign-provider fields so multi-model conversations
|
|
62
62
|
/// don't leak them, and normalize vision blocks to OpenAI form. Messages,
|
|
63
63
|
/// `tool_calls`, and `role: "tool"` all stay in Chat Completions shape.
|
|
64
|
+
///
|
|
65
|
+
/// Same-role adjacency passes through unchanged. OpenRouter is an aggregator:
|
|
66
|
+
/// its own API is Chat Completions and documents no alternation rule
|
|
67
|
+
/// (openrouter.ai/docs, API reference and parameters). Whether the vendor
|
|
68
|
+
/// behind a given slug rejects adjacent turns is unknown from here and is
|
|
69
|
+
/// OpenRouter's to reconcile; a faithful passthrough does not pre-empt it by
|
|
70
|
+
/// merging, which would destroy message boundaries for the vendors that
|
|
71
|
+
/// accept them.
|
|
64
72
|
fn sanitize_messages(messages: &[Value]) -> Vec<Value> {
|
|
65
73
|
messages
|
|
66
74
|
.iter()
|
|
@@ -29,6 +29,13 @@ impl Xai {
|
|
|
29
29
|
/// - Assistant messages with tool_calls → split into the assistant message +
|
|
30
30
|
/// separate `function_call` items
|
|
31
31
|
/// - `role: "tool"` messages → `function_call_output` items
|
|
32
|
+
///
|
|
33
|
+
/// Same-role adjacency passes through unchanged. xAI's API is OpenAI
|
|
34
|
+
/// Responses-shaped and its documentation (docs.x.ai, chat guide and API
|
|
35
|
+
/// reference) states no ordering or alternation rule — but silence is not a
|
|
36
|
+
/// verified acceptance, and this has not been checked live. Nothing is merged
|
|
37
|
+
/// here because merging would destroy message boundaries a caller may depend
|
|
38
|
+
/// on; if xAI ever rejects adjacency the rejection arrives as its own 400.
|
|
32
39
|
fn sanitize_messages(messages: &[Value]) -> Vec<Value> {
|
|
33
40
|
let mut result = Vec::new();
|
|
34
41
|
for msg in messages {
|
|
@@ -119,7 +119,12 @@ pub struct Usage {
|
|
|
119
119
|
pub cache_read_tokens: u64,
|
|
120
120
|
pub cache_write_tokens: u64,
|
|
121
121
|
/// USD charged for this response. `null` means the catalog carries no price
|
|
122
|
-
/// for the model — it never means free.
|
|
122
|
+
/// for the model at all — it never means free.
|
|
123
|
+
///
|
|
124
|
+
/// Where a model prices some token classes and not others, the unpriced ones
|
|
125
|
+
/// are charged at its highest published rate, so this is an **upper bound**
|
|
126
|
+
/// rather than `null`. Under-reporting would let a spend cap stop binding;
|
|
127
|
+
/// over-reporting merely spends a budget slightly early. See `llmshim::cost`.
|
|
123
128
|
pub cost_usd: Option<f64>,
|
|
124
129
|
}
|
|
125
130
|
|