llmshim 0.7.1__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llmshim-0.8.0/.github/scripts/npm-verify-served.sh +37 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/.github/workflows/release.yml +18 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/Cargo.lock +1 -1
- {llmshim-0.7.1 → llmshim-0.8.0}/Cargo.toml +1 -1
- {llmshim-0.7.1 → llmshim-0.8.0}/PKG-INFO +1 -1
- {llmshim-0.7.1 → llmshim-0.8.0}/README.md +12 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/concepts/routing.md +22 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/reference/models.md +4 -1
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/reference/providers.md +2 -2
- {llmshim-0.7.1 → llmshim-0.8.0}/src/breaker.rs +4 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/cache.rs +1 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/client.rs +5 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/error.rs +9 -1
- {llmshim-0.7.1 → llmshim-0.8.0}/src/gateway/http.rs +6 -1
- {llmshim-0.7.1 → llmshim-0.8.0}/src/providers/anthropic.rs +5 -3
- {llmshim-0.7.1 → llmshim-0.8.0}/src/providers/anthropic_reasoning.rs +1 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/providers/chatgpt/auth.rs +1 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/providers/gemini.rs +3 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/providers/openai.rs +9 -2
- {llmshim-0.7.1 → llmshim-0.8.0}/src/providers/openai_compat.rs +92 -17
- {llmshim-0.7.1 → llmshim-0.8.0}/src/providers/openrouter.rs +2 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/providers/xai.rs +3 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/proxy/error.rs +1 -1
- {llmshim-0.7.1 → llmshim-0.8.0}/src/reasoning/normalize.rs +38 -12
- {llmshim-0.7.1 → llmshim-0.8.0}/src/reasoning.rs +1 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/router.rs +67 -19
- {llmshim-0.7.1 → llmshim-0.8.0}/src/schema/validate.rs +2 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/shim.rs +20 -6
- {llmshim-0.7.1 → llmshim-0.8.0}/src/toolcall.rs +2 -0
- llmshim-0.8.0/tests/fixtures/sglang-models.toml +6 -0
- llmshim-0.8.0/tests/fixtures/sglang-responses-stream.sse +75 -0
- llmshim-0.8.0/tests/fixtures/sglang-responses-turn1.json +94 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_long_context.rs +6 -2
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_proxy.rs +3 -4
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_sglang.rs +69 -0
- llmshim-0.8.0/tests/unit_client_retry_after.rs +97 -0
- llmshim-0.8.0/tests/unit_openai_compat_responses.rs +290 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_reasoning.rs +62 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_router.rs +63 -1
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_shim.rs +3 -3
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_sse.rs +3 -3
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_toolcall.rs +1 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_wire.rs +7 -7
- {llmshim-0.7.1 → llmshim-0.8.0}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/.github/workflows/catalog-refresh.yml +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/.github/workflows/pages.yml +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/.gitignore +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/CLAUDE.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/CODE_OF_CONDUCT.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/CONTRIBUTING.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/LICENSE-APACHE +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/LICENSE-MIT +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/NOTICE +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/SECURITY.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/benchmarks/bench.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/benchmarks/bench_python.py +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/benchmarks/gateway_loadtest.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/benchmarks/loadtest.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/Cargo.toml +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/LICENSE-APACHE +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/LICENSE-MIT +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/README.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/data/LICENSE.models.dev +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/data/README.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/data/models.dev.json +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/src/aliases.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/src/builtin.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/src/capabilities.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/src/lib.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/src/merge.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/src/parse.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/src/refresh.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/src/types.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/tests/catalog.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/crates/llmshim-catalog/tests/refresh.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/.gitignore +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/book.toml +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/mermaid-init.js +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/mermaid.min.js +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/SUMMARY.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/concepts/contracts.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/concepts/conversations.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/concepts/portability.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/concepts/translation-flow.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/guides/caching.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/guides/capabilities.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/guides/fallbacks.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/guides/images.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/guides/native-controls.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/guides/reasoning.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/guides/schemas.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/guides/streaming.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/guides/tools.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/introduction.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/proxy/deployment.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/proxy/http-api.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/proxy/native-apis.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/proxy/scaling.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/reference/api.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/reference/cli.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/reference/configuration.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/reference/errors.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/reference/request-fields.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/reference/surfaces.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/start/choose.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/start/cli.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/start/clients.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/start/configure.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/start/proxy.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/docs/src/start/rust.md +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/examples/chat.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/examples/stream.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/llmshim/__init__.py +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/llmshim/_client.py +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/llmshim/_server.py +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/llmshim/types.py +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/pyproject.toml +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/cli.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/config.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/cost.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/env.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/error/normalize.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/fallback.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/gateway/auth.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/gateway/distributed.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/gateway/idempotency.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/gateway/metrics.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/gateway/mod.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/gateway/quota.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/lib.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/log.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/main.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/models.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/provider.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/providers/anthropic_signature.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/providers/chatgpt/mod.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/providers/chatgpt/streaming.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/providers/mod.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/proxy/convert.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/proxy/handlers.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/proxy/health.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/proxy/mod.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/proxy/ratelimit.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/proxy/types.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/proxy/wire/mod.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/proxy/wire/receipts.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/schema/memo.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/schema/mod.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/schema/walk.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/streaming.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/toolcall/streaming.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/usage.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/src/vision.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/fixtures/chatgpt-red.png +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_chatgpt.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_chatgpt_proxy.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_current_models.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_fallback.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_gemini.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_gemini_tools.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_multimodel.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_openrouter.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_thinking.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_tool_roundtrip.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_vision.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/integration_xai.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/support/completion_status.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_advertised_models.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_anthropic.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_cache.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_chatgpt.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_cli.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_client_breaker.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_fable.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_fallback.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_fast_mode.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_gemini.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_log.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_models.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_multimodel.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_openai.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_openai_compat.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_openrouter.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_provider_contracts.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_proxy.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_proxy_convert.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_reasoning_profile.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_schema.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_signature.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_tools.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_usage.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_vision.rs +0 -0
- {llmshim-0.7.1 → llmshim-0.8.0}/tests/unit_xai.rs +0 -0
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# `npm publish` exiting 0 means the registry *accepted* the tarball, not that it
|
|
3
|
+
# serves it. 0.7.1 was accepted, logged to Sigstore, answered "may take a few
|
|
4
|
+
# minutes to become available" — and half an hour later still served nothing,
|
|
5
|
+
# behind a green job. So a publish is not done until `npm view` returns the
|
|
6
|
+
# version. Run from the package's directory, after its publish step.
|
|
7
|
+
#
|
|
8
|
+
# Bound: 20 polls, 30 s apart — about ten minutes. npm's "few minutes" has been
|
|
9
|
+
# minutes, never seconds, and a version still unserved after ten is stuck in
|
|
10
|
+
# npm's processing, which more waiting here will not fix.
|
|
11
|
+
#
|
|
12
|
+
# Compares `npm view`'s output to the version rather than its exit status: for a
|
|
13
|
+
# package that exists with a version that does not, npm's exit status has
|
|
14
|
+
# differed across releases. `--prefer-online` revalidates on every poll, so the
|
|
15
|
+
# first poll's packument — cached without the new version — cannot answer for
|
|
16
|
+
# the rest of the window.
|
|
17
|
+
set -euo pipefail
|
|
18
|
+
|
|
19
|
+
ATTEMPTS=20
|
|
20
|
+
INTERVAL=30
|
|
21
|
+
NAME=$(node -p "require('./package.json').name")
|
|
22
|
+
VER=$(node -p "require('./package.json').version")
|
|
23
|
+
|
|
24
|
+
for attempt in $(seq 1 "$ATTEMPTS"); do
|
|
25
|
+
served=$(npm view --prefer-online "$NAME@$VER" version 2>/dev/null || true)
|
|
26
|
+
if [ "$served" = "$VER" ]; then
|
|
27
|
+
echo "$NAME@$VER is served by the registry."
|
|
28
|
+
exit 0
|
|
29
|
+
fi
|
|
30
|
+
echo "poll $attempt/$ATTEMPTS: $NAME@$VER not served yet"
|
|
31
|
+
if [ "$attempt" -lt "$ATTEMPTS" ]; then
|
|
32
|
+
sleep "$INTERVAL"
|
|
33
|
+
fi
|
|
34
|
+
done
|
|
35
|
+
|
|
36
|
+
echo "::error::$NAME@$VER was accepted by npm but is still not served after $((ATTEMPTS * INTERVAL / 60)) minutes. The publish step exited 0 and its output carries the Sigstore provenance line, so the tarball is stuck in npm's processing, not missing from this run. Look at https://www.npmjs.com/package/$NAME?activeTab=versions and https://status.npmjs.org; once it appears, re-run this job — the publish step sees it and skips. If it never appears, npm will refuse a second publish of the same version and support has to release it."
|
|
37
|
+
exit 1
|
|
@@ -223,6 +223,9 @@ jobs:
|
|
|
223
223
|
else
|
|
224
224
|
npm publish --access public --provenance
|
|
225
225
|
fi
|
|
226
|
+
- name: Verify npm serves llmshim-darwin-arm64
|
|
227
|
+
working-directory: clients/typescript/packages/llmshim-darwin-arm64
|
|
228
|
+
run: bash "$GITHUB_WORKSPACE/.github/scripts/npm-verify-served.sh"
|
|
226
229
|
- name: Publish llmshim-darwin-x64
|
|
227
230
|
working-directory: clients/typescript/packages/llmshim-darwin-x64
|
|
228
231
|
run: |
|
|
@@ -234,6 +237,9 @@ jobs:
|
|
|
234
237
|
else
|
|
235
238
|
npm publish --access public --provenance
|
|
236
239
|
fi
|
|
240
|
+
- name: Verify npm serves llmshim-darwin-x64
|
|
241
|
+
working-directory: clients/typescript/packages/llmshim-darwin-x64
|
|
242
|
+
run: bash "$GITHUB_WORKSPACE/.github/scripts/npm-verify-served.sh"
|
|
237
243
|
|
|
238
244
|
npm-linux-x64-binary:
|
|
239
245
|
needs: test
|
|
@@ -274,6 +280,9 @@ jobs:
|
|
|
274
280
|
else
|
|
275
281
|
npm publish --access public --provenance
|
|
276
282
|
fi
|
|
283
|
+
- name: Verify npm serves llmshim-linux-x64
|
|
284
|
+
working-directory: clients/typescript/packages/llmshim-linux-x64
|
|
285
|
+
run: bash "$GITHUB_WORKSPACE/.github/scripts/npm-verify-served.sh"
|
|
277
286
|
|
|
278
287
|
# Native arm64 runner (GA, free on public repos) — no cross-compilation needed.
|
|
279
288
|
npm-linux-arm64-binary:
|
|
@@ -315,6 +324,9 @@ jobs:
|
|
|
315
324
|
else
|
|
316
325
|
npm publish --access public --provenance
|
|
317
326
|
fi
|
|
327
|
+
- name: Verify npm serves llmshim-linux-arm64
|
|
328
|
+
working-directory: clients/typescript/packages/llmshim-linux-arm64
|
|
329
|
+
run: bash "$GITHUB_WORKSPACE/.github/scripts/npm-verify-served.sh"
|
|
318
330
|
|
|
319
331
|
npm-windows-binary:
|
|
320
332
|
needs: test
|
|
@@ -358,6 +370,10 @@ jobs:
|
|
|
358
370
|
else
|
|
359
371
|
npm publish --access public --provenance
|
|
360
372
|
fi
|
|
373
|
+
- name: Verify npm serves llmshim-win32-x64
|
|
374
|
+
shell: bash
|
|
375
|
+
working-directory: clients/typescript/packages/llmshim-win32-x64
|
|
376
|
+
run: bash "$GITHUB_WORKSPACE/.github/scripts/npm-verify-served.sh"
|
|
361
377
|
|
|
362
378
|
# --- Publish TypeScript client to npm ---
|
|
363
379
|
# First release must be published manually once (npm's trusted-publisher UI
|
|
@@ -410,6 +426,8 @@ jobs:
|
|
|
410
426
|
else
|
|
411
427
|
npm publish --access public --provenance
|
|
412
428
|
fi
|
|
429
|
+
- name: Verify npm serves the TypeScript client
|
|
430
|
+
run: bash "$GITHUB_WORKSPACE/.github/scripts/npm-verify-served.sh"
|
|
413
431
|
|
|
414
432
|
# --- Publish Ruby client to RubyGems ---
|
|
415
433
|
# Uses a "pending trusted publisher" — configurable on rubygems.org before
|
|
@@ -100,6 +100,18 @@ Then address the served model as `sglang/<served-model>` or `vllm/<served-model>
|
|
|
100
100
|
(e.g. `sglang/Qwen/Qwen3.6-35B-A3B-FP8`). Server-specific knobs go under
|
|
101
101
|
`x-vllm` / `x-sglang`.
|
|
102
102
|
|
|
103
|
+
A server that also serves `/v1/responses` (SGLang does) can be spoken to on
|
|
104
|
+
that wire with `SGLANG_WIRE=responses` (or `VLLM_WIRE=responses`). Reasoning
|
|
105
|
+
then comes back as an item with its own id rather than bare `reasoning_content`,
|
|
106
|
+
and is replayed as that item. Replay is gated on a known model family on every
|
|
107
|
+
wire, and the public catalog does not know a served model: declare it once in
|
|
108
|
+
`.llmshim/models.toml` —
|
|
109
|
+
|
|
110
|
+
```toml
|
|
111
|
+
[models."sglang/<served-model>"]
|
|
112
|
+
family = "qwen"
|
|
113
|
+
```
|
|
114
|
+
|
|
103
115
|
Or persist them to the config file (used by all three surfaces):
|
|
104
116
|
|
|
105
117
|
```bash
|
|
@@ -136,3 +136,25 @@ let router = llmshim::router::Router::from_env();
|
|
|
136
136
|
|
|
137
137
|
Applications that manage secrets themselves can call `Router::from_env()`
|
|
138
138
|
directly or construct a Router by registering provider implementations.
|
|
139
|
+
|
|
140
|
+
## Catalog refresh is the daemon's default, not the embedder's
|
|
141
|
+
|
|
142
|
+
`Router::from_env()` also schedules one background fetch of the model catalog
|
|
143
|
+
(prices, context windows, capabilities). That is right for the proxy, which
|
|
144
|
+
starts once and runs for days. A program that starts many times a day, or runs
|
|
145
|
+
air-gapped, should build its router with
|
|
146
|
+
`Router::from_env_without_catalog_refresh()` — identical, except that
|
|
147
|
+
constructing it makes no network call — and refresh only when it decides to:
|
|
148
|
+
|
|
149
|
+
```rust
|
|
150
|
+
let router = llmshim::router::Router::from_env_without_catalog_refresh();
|
|
151
|
+
if user_asked_for_it {
|
|
152
|
+
// Rides the caller's Tokio runtime; `None` when there is none, when
|
|
153
|
+
// LLMSHIM_CATALOG_OFFLINE=1 is set, or when the local catalog is invalid.
|
|
154
|
+
router.refresh_catalog_in_background();
|
|
155
|
+
}
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
The offline router still resolves and prices every model in the vendored
|
|
159
|
+
snapshot, any earlier cached download, and the local override files; its data
|
|
160
|
+
is simply never newer than the disk.
|
|
@@ -28,7 +28,10 @@ compile-time API for curated and historical facts.
|
|
|
28
28
|
|
|
29
29
|
Startup reads the vendored floor, cached data, and local overrides without
|
|
30
30
|
waiting for network. Refresh uses ETags and a 24-hour TTL; errors keep the old
|
|
31
|
-
snapshot available.
|
|
31
|
+
snapshot available. `Router::from_env()` schedules one such refresh;
|
|
32
|
+
`Router::from_env_without_catalog_refresh()` schedules none and leaves it to
|
|
33
|
+
`Router::refresh_catalog_in_background()`. Offline mode uses only vendored and
|
|
34
|
+
local layers. User
|
|
32
35
|
overrides live in `~/.config/llmshim/models.toml`, project overrides in
|
|
33
36
|
`.llmshim/models.toml`, and cached data in `~/.cache/llmshim/models.dev.json`.
|
|
34
37
|
Local overrides win; verified builtin assertions win over models.dev data.
|
|
@@ -11,8 +11,8 @@ different native API and translates only the fields that API understands.
|
|
|
11
11
|
| Google Gemini | `generateContent` / `streamGenerateContent` | `gemini*` | `x-gemini` |
|
|
12
12
|
| xAI | Responses API | `grok*` | none |
|
|
13
13
|
| OpenRouter | Chat Completions (aggregator) | none — address as `openrouter/<vendor>/<model>` | `x-openrouter` |
|
|
14
|
-
| vLLM | Chat Completions (self-hosted, `VLLM_BASE_URL`) | none — address as `vllm/<served-model>` | `x-vllm` |
|
|
15
|
-
| SGLang | Chat Completions (self-hosted, `SGLANG_BASE_URL`) | none — address as `sglang/<served-model>` | `x-sglang` |
|
|
14
|
+
| vLLM | Chat Completions (self-hosted, `VLLM_BASE_URL`); Responses API with `VLLM_WIRE=responses` | none — address as `vllm/<served-model>` | `x-vllm` |
|
|
15
|
+
| SGLang | Chat Completions (self-hosted, `SGLANG_BASE_URL`); Responses API with `SGLANG_WIRE=responses` | none — address as `sglang/<served-model>` | `x-sglang` |
|
|
16
16
|
|
|
17
17
|
An explicit address such as `anthropic/claude-sonnet-5` avoids inference.
|
|
18
18
|
The named provider must be registered in the Router—that normally means its
|
|
@@ -304,6 +304,7 @@ mod tests {
|
|
|
304
304
|
ShimError::ProviderError {
|
|
305
305
|
status: 503,
|
|
306
306
|
body: "down".into(),
|
|
307
|
+
retry_after: None,
|
|
307
308
|
}
|
|
308
309
|
}
|
|
309
310
|
|
|
@@ -327,17 +328,20 @@ mod tests {
|
|
|
327
328
|
assert!(!counts_toward_health(&ShimError::ProviderError {
|
|
328
329
|
status: 429,
|
|
329
330
|
body: "slow down".into(),
|
|
331
|
+
retry_after: None,
|
|
330
332
|
}));
|
|
331
333
|
for status in [500, 502, 503, 504, 529] {
|
|
332
334
|
assert!(counts_toward_health(&ShimError::ProviderError {
|
|
333
335
|
status,
|
|
334
336
|
body: String::new(),
|
|
337
|
+
retry_after: None,
|
|
335
338
|
}));
|
|
336
339
|
}
|
|
337
340
|
for status in [400, 401, 403, 404, 422] {
|
|
338
341
|
assert!(!counts_toward_health(&ShimError::ProviderError {
|
|
339
342
|
status,
|
|
340
343
|
body: String::new(),
|
|
344
|
+
retry_after: None,
|
|
341
345
|
}));
|
|
342
346
|
}
|
|
343
347
|
assert!(!counts_toward_health(&ShimError::MissingModel));
|
|
@@ -169,10 +169,15 @@ impl ShimClient {
|
|
|
169
169
|
tokio::time::sleep(wait).await;
|
|
170
170
|
continue;
|
|
171
171
|
}
|
|
172
|
+
// Read before the body is consumed: the header is the
|
|
173
|
+
// server's own wait, and a caller with its own backoff
|
|
174
|
+
// above this client gets to honour it too.
|
|
175
|
+
let retry_after = parse_retry_after(resp.headers());
|
|
172
176
|
let body = resp.text().await.unwrap_or_default();
|
|
173
177
|
return Err(ShimError::ProviderError {
|
|
174
178
|
status: status_code,
|
|
175
179
|
body,
|
|
180
|
+
retry_after,
|
|
176
181
|
});
|
|
177
182
|
}
|
|
178
183
|
// Transport errors carry no headers: always jittered backoff.
|
|
@@ -19,8 +19,16 @@ pub enum ShimError {
|
|
|
19
19
|
#[error("JSON error: {0}")]
|
|
20
20
|
Json(#[from] serde_json::Error),
|
|
21
21
|
|
|
22
|
+
/// A non-success response from the provider. `retry_after` is the
|
|
23
|
+
/// server's own `Retry-After` (delay-seconds or HTTP-date, parsed at
|
|
24
|
+
/// receipt), so a caller with its own backoff can wait what was asked
|
|
25
|
+
/// instead of guessing; `None` when the header was absent or unparseable.
|
|
22
26
|
#[error("provider error ({status}): {body}")]
|
|
23
|
-
ProviderError {
|
|
27
|
+
ProviderError {
|
|
28
|
+
status: u16,
|
|
29
|
+
body: String,
|
|
30
|
+
retry_after: Option<std::time::Duration>,
|
|
31
|
+
},
|
|
24
32
|
|
|
25
33
|
#[error("stream error: {0}")]
|
|
26
34
|
Stream(String),
|
|
@@ -55,7 +55,9 @@ pub struct RealDispatch {
|
|
|
55
55
|
impl RealDispatch {
|
|
56
56
|
fn map_err(err: ShimError) -> DispatchError {
|
|
57
57
|
match err {
|
|
58
|
-
ShimError::ProviderError {
|
|
58
|
+
ShimError::ProviderError {
|
|
59
|
+
status: 429, body, ..
|
|
60
|
+
} => DispatchError {
|
|
59
61
|
message: body,
|
|
60
62
|
retry_after: Some(penalty_duration()),
|
|
61
63
|
},
|
|
@@ -373,6 +375,7 @@ impl GatewayState {
|
|
|
373
375
|
key to run it uncharged.\",\"type\":\"invalid_request_error\",\
|
|
374
376
|
\"param\":\"model\",\"code\":\"unpriceable_under_budget\"}}}}"
|
|
375
377
|
),
|
|
378
|
+
retry_after: None,
|
|
376
379
|
}))
|
|
377
380
|
}
|
|
378
381
|
}
|
|
@@ -425,10 +428,12 @@ fn gateway_err_to_api(state: &GatewayState, err: GatewayError) -> ApiError {
|
|
|
425
428
|
GatewayError::Shutdown => ApiError::from(ShimError::ProviderError {
|
|
426
429
|
status: 503,
|
|
427
430
|
body: "gateway shutting down".to_string(),
|
|
431
|
+
retry_after: None,
|
|
428
432
|
}),
|
|
429
433
|
GatewayError::Upstream(message) => ApiError::from(ShimError::ProviderError {
|
|
430
434
|
status: 502,
|
|
431
435
|
body: message,
|
|
436
|
+
retry_after: None,
|
|
432
437
|
}),
|
|
433
438
|
}
|
|
434
439
|
}
|
|
@@ -382,6 +382,7 @@ fn transform_response_to_openai(model: &str, resp: &Value) -> Result<Value> {
|
|
|
382
382
|
return Err(ShimError::ProviderError {
|
|
383
383
|
status: 502,
|
|
384
384
|
body: "Anthropic response has no supported terminal stop reason".into(),
|
|
385
|
+
retry_after: None,
|
|
385
386
|
})
|
|
386
387
|
}
|
|
387
388
|
};
|
|
@@ -643,7 +644,7 @@ impl Provider for Anthropic {
|
|
|
643
644
|
if let Some(thinking) = body_obj.get("thinking") {
|
|
644
645
|
if thinking["type"] != "adaptive" {
|
|
645
646
|
return Err(ShimError::ProviderError { status: 400, body:
|
|
646
|
-
"Claude Fable requires adaptive thinking; use reasoning_effort to control depth".into() });
|
|
647
|
+
"Claude Fable requires adaptive thinking; use reasoning_effort to control depth".into(), retry_after: None });
|
|
647
648
|
}
|
|
648
649
|
}
|
|
649
650
|
if body_obj
|
|
@@ -653,7 +654,7 @@ impl Provider for Anthropic {
|
|
|
653
654
|
.is_some_and(|message| message["role"] == "assistant")
|
|
654
655
|
{
|
|
655
656
|
return Err(ShimError::ProviderError { status: 400, body:
|
|
656
|
-
"Claude Fable does not support assistant prefill; end the request with a user turn".into() });
|
|
657
|
+
"Claude Fable does not support assistant prefill; end the request with a user turn".into(), retry_after: None });
|
|
657
658
|
}
|
|
658
659
|
}
|
|
659
660
|
if model == "claude-fable-5-1"
|
|
@@ -663,7 +664,7 @@ impl Provider for Anthropic {
|
|
|
663
664
|
.is_some_and(|kind| matches!(kind, "any" | "tool"))
|
|
664
665
|
{
|
|
665
666
|
return Err(ShimError::ProviderError { status: 400, body:
|
|
666
|
-
"Claude Fable 5.1 supports only auto or none tool choice; request the desired tool in the prompt".into() });
|
|
667
|
+
"Claude Fable 5.1 supports only auto or none tool choice; request the desired tool in the prompt".into(), retry_after: None });
|
|
667
668
|
}
|
|
668
669
|
|
|
669
670
|
// Fast mode support: extract "speed" from the request and apply
|
|
@@ -768,6 +769,7 @@ impl Anthropic {
|
|
|
768
769
|
return Err(ShimError::ProviderError {
|
|
769
770
|
status: 400,
|
|
770
771
|
body: msg.to_string(),
|
|
772
|
+
retry_after: None,
|
|
771
773
|
});
|
|
772
774
|
}
|
|
773
775
|
|
|
@@ -288,6 +288,7 @@ fn transform_response_to_openai(model: &str, resp: &Value) -> Result<Value> {
|
|
|
288
288
|
.ok_or_else(|| ShimError::ProviderError {
|
|
289
289
|
status: 500,
|
|
290
290
|
body: format!("no candidates in response: {}", resp),
|
|
291
|
+
retry_after: None,
|
|
291
292
|
})?;
|
|
292
293
|
|
|
293
294
|
let parts = candidate
|
|
@@ -354,6 +355,7 @@ fn transform_response_to_openai(model: &str, resp: &Value) -> Result<Value> {
|
|
|
354
355
|
return Err(ShimError::ProviderError {
|
|
355
356
|
status: 502,
|
|
356
357
|
body: "Gemini response has no supported terminal finish reason".into(),
|
|
358
|
+
retry_after: None,
|
|
357
359
|
})
|
|
358
360
|
}
|
|
359
361
|
};
|
|
@@ -606,6 +608,7 @@ impl Gemini {
|
|
|
606
608
|
return Err(ShimError::ProviderError {
|
|
607
609
|
status: code,
|
|
608
610
|
body: msg.to_string(),
|
|
611
|
+
retry_after: None,
|
|
609
612
|
});
|
|
610
613
|
}
|
|
611
614
|
transform_response_to_openai(model, &response)
|
|
@@ -363,7 +363,7 @@ impl Provider for OpenAi {
|
|
|
363
363
|
}
|
|
364
364
|
|
|
365
365
|
impl OpenAi {
|
|
366
|
-
fn transform_response_native(&self, model: &str, response: Value) -> Result<Value> {
|
|
366
|
+
pub(crate) fn transform_response_native(&self, model: &str, response: Value) -> Result<Value> {
|
|
367
367
|
// Check for error (Responses API returns "error": null on success)
|
|
368
368
|
if let Some(err) = response.get("error") {
|
|
369
369
|
if !err.is_null() {
|
|
@@ -374,6 +374,7 @@ impl OpenAi {
|
|
|
374
374
|
return Err(ShimError::ProviderError {
|
|
375
375
|
status: 400,
|
|
376
376
|
body: msg.to_string(),
|
|
377
|
+
retry_after: None,
|
|
377
378
|
});
|
|
378
379
|
}
|
|
379
380
|
}
|
|
@@ -384,6 +385,7 @@ impl OpenAi {
|
|
|
384
385
|
.ok_or_else(|| ShimError::ProviderError {
|
|
385
386
|
status: 500,
|
|
386
387
|
body: "no output in response".to_string(),
|
|
388
|
+
retry_after: None,
|
|
387
389
|
})?;
|
|
388
390
|
|
|
389
391
|
// Extract reasoning summary
|
|
@@ -449,6 +451,7 @@ impl OpenAi {
|
|
|
449
451
|
return Err(ShimError::ProviderError {
|
|
450
452
|
status: 502,
|
|
451
453
|
body: "OpenAI response has no supported terminal status".into(),
|
|
454
|
+
retry_after: None,
|
|
452
455
|
})
|
|
453
456
|
}
|
|
454
457
|
};
|
|
@@ -471,7 +474,11 @@ impl OpenAi {
|
|
|
471
474
|
}
|
|
472
475
|
|
|
473
476
|
impl OpenAi {
|
|
474
|
-
fn transform_stream_chunk_native(
|
|
477
|
+
pub(crate) fn transform_stream_chunk_native(
|
|
478
|
+
&self,
|
|
479
|
+
model: &str,
|
|
480
|
+
chunk: &str,
|
|
481
|
+
) -> Result<Option<String>> {
|
|
475
482
|
let trimmed = chunk.trim();
|
|
476
483
|
if trimmed.is_empty() || trimmed == "[DONE]" {
|
|
477
484
|
return Ok(None);
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
use crate::error::{Result, ShimError};
|
|
2
2
|
use crate::provider::{Provider, ProviderRequest};
|
|
3
|
+
use crate::providers::openai::OpenAi;
|
|
4
|
+
use crate::reasoning::{ReplayTarget, WireFormat};
|
|
3
5
|
use crate::vision;
|
|
4
6
|
use serde_json::{json, Value};
|
|
5
7
|
|
|
@@ -17,10 +19,18 @@ use serde_json::{json, Value};
|
|
|
17
19
|
/// `name` (e.g. `"vllm"` / `"sglang"`) is both the provider key and the
|
|
18
20
|
/// extension namespace: server-specific params (`chat_template_kwargs`,
|
|
19
21
|
/// `separate_reasoning`, `guided_json`, `top_k`, …) go under `x-<name>`.
|
|
22
|
+
///
|
|
23
|
+
/// The wire is Chat Completions unless [`OpenAiCompatible::with_wire`] selects
|
|
24
|
+
/// the Responses API, which SGLang also serves at `<base>/responses`. On that
|
|
25
|
+
/// wire reasoning comes back as an item with its own id, so it can be replayed
|
|
26
|
+
/// as a keyed block instead of bare `reasoning_content`; the request and
|
|
27
|
+
/// response translation is the OpenAI adapter's, with this server's URL, its
|
|
28
|
+
/// optional auth, and its `x-<name>` namespace.
|
|
20
29
|
pub struct OpenAiCompatible {
|
|
21
30
|
pub name: String,
|
|
22
31
|
pub base_url: String,
|
|
23
32
|
pub api_key: Option<String>,
|
|
33
|
+
wire: WireFormat,
|
|
24
34
|
}
|
|
25
35
|
|
|
26
36
|
impl OpenAiCompatible {
|
|
@@ -33,7 +43,66 @@ impl OpenAiCompatible {
|
|
|
33
43
|
name: name.into(),
|
|
34
44
|
base_url: base_url.into(),
|
|
35
45
|
api_key,
|
|
46
|
+
wire: WireFormat::OpenAiChat,
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/// Select the wire this server is spoken to on: `OpenAiChat` (the
|
|
51
|
+
/// default, `<base>/chat/completions`) or `OpenAiResponses`
|
|
52
|
+
/// (`<base>/responses`). The other wires are not OpenAI-compatible and
|
|
53
|
+
/// are a caller error.
|
|
54
|
+
pub fn with_wire(mut self, wire: WireFormat) -> Self {
|
|
55
|
+
assert!(
|
|
56
|
+
matches!(wire, WireFormat::OpenAiChat | WireFormat::OpenAiResponses),
|
|
57
|
+
"an OpenAI-compatible server speaks Chat Completions or Responses, not {wire:?}"
|
|
58
|
+
);
|
|
59
|
+
self.wire = wire;
|
|
60
|
+
self
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
pub fn wire(&self) -> WireFormat {
|
|
64
|
+
self.wire
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/// Auth is optional — self-hosted servers are unauthenticated unless
|
|
68
|
+
/// launched with --api-key.
|
|
69
|
+
fn headers(&self) -> Vec<(String, String)> {
|
|
70
|
+
let mut headers = vec![("Content-Type".to_string(), "application/json".to_string())];
|
|
71
|
+
if let Some(key) = self.api_key.as_deref().filter(|k| !k.is_empty()) {
|
|
72
|
+
headers.push(("Authorization".to_string(), format!("Bearer {key}")));
|
|
73
|
+
}
|
|
74
|
+
headers
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/// The OpenAI adapter pointed at this server, for the Responses wire. Its
|
|
78
|
+
/// translation is reused whole; only the URL, the auth and the extension
|
|
79
|
+
/// namespace are this server's.
|
|
80
|
+
fn responses_adapter(&self) -> OpenAi {
|
|
81
|
+
OpenAi::new(self.api_key.clone().unwrap_or_default())
|
|
82
|
+
.with_base_url(self.base_url.trim_end_matches('/').to_string())
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
fn transform_request_responses(&self, model: &str, request: &Value) -> Result<ProviderRequest> {
|
|
86
|
+
// The OpenAI adapter reads its overrides from `x-openai`, and applies
|
|
87
|
+
// them before the stateless and native-tool passes. Moving `x-<name>`
|
|
88
|
+
// there keeps that order, so an override cannot re-enable storage.
|
|
89
|
+
let mut request = request.clone();
|
|
90
|
+
let namespace = format!("x-{}", self.name);
|
|
91
|
+
if let Some(ext) = request
|
|
92
|
+
.as_object_mut()
|
|
93
|
+
.and_then(|obj| obj.remove(&namespace))
|
|
94
|
+
.and_then(|ext| ext.as_object().cloned())
|
|
95
|
+
{
|
|
96
|
+
let target = request["x-openai"].as_object().cloned().unwrap_or_default();
|
|
97
|
+
request["x-openai"] = Value::Object(target.into_iter().chain(ext).collect());
|
|
36
98
|
}
|
|
99
|
+
let mut sent = self.responses_adapter().transform_request_for_target(
|
|
100
|
+
model,
|
|
101
|
+
&request,
|
|
102
|
+
&self.replay_target(model),
|
|
103
|
+
)?;
|
|
104
|
+
sent.headers = self.headers();
|
|
105
|
+
Ok(sent)
|
|
37
106
|
}
|
|
38
107
|
}
|
|
39
108
|
|
|
@@ -76,16 +145,15 @@ impl Provider for OpenAiCompatible {
|
|
|
76
145
|
&self.name
|
|
77
146
|
}
|
|
78
147
|
|
|
79
|
-
fn replay_target(&self, model: &str) ->
|
|
80
|
-
|
|
81
|
-
self.
|
|
82
|
-
model,
|
|
83
|
-
crate::reasoning::WireFormat::OpenAiChat,
|
|
84
|
-
)
|
|
85
|
-
.bind_account(&self.base_url, self.api_key.as_deref())
|
|
148
|
+
fn replay_target(&self, model: &str) -> ReplayTarget {
|
|
149
|
+
ReplayTarget::new(self.name(), model, self.wire)
|
|
150
|
+
.bind_account(&self.base_url, self.api_key.as_deref())
|
|
86
151
|
}
|
|
87
152
|
|
|
88
153
|
fn transform_request(&self, model: &str, request: &Value) -> Result<ProviderRequest> {
|
|
154
|
+
if self.wire == WireFormat::OpenAiResponses {
|
|
155
|
+
return self.transform_request_responses(model, request);
|
|
156
|
+
}
|
|
89
157
|
let request = crate::schema::prepare_request(request);
|
|
90
158
|
let request =
|
|
91
159
|
crate::cache::prepare_request(&request, crate::reasoning::WireFormat::OpenAiChat)?;
|
|
@@ -140,14 +208,7 @@ impl Provider for OpenAiCompatible {
|
|
|
140
208
|
}
|
|
141
209
|
}
|
|
142
210
|
|
|
143
|
-
let
|
|
144
|
-
// Auth is optional — self-hosted servers are unauthenticated unless
|
|
145
|
-
// launched with --api-key.
|
|
146
|
-
if let Some(key) = &self.api_key {
|
|
147
|
-
if !key.is_empty() {
|
|
148
|
-
headers.push(("Authorization".to_string(), format!("Bearer {key}")));
|
|
149
|
-
}
|
|
150
|
-
}
|
|
211
|
+
let headers = self.headers();
|
|
151
212
|
|
|
152
213
|
let url = format!("{}/chat/completions", self.base_url.trim_end_matches('/'));
|
|
153
214
|
crate::toolcall::validate_native(&body, &self.replay_target(model))?;
|
|
@@ -167,14 +228,26 @@ impl Provider for OpenAiCompatible {
|
|
|
167
228
|
|
|
168
229
|
fn transform_response(&self, model: &str, response: Value) -> Result<Value> {
|
|
169
230
|
let native = response.clone();
|
|
170
|
-
let mut result = self.
|
|
231
|
+
let mut result = match self.wire {
|
|
232
|
+
WireFormat::OpenAiResponses => self
|
|
233
|
+
.responses_adapter()
|
|
234
|
+
.transform_response_native(model, response)?,
|
|
235
|
+
_ => self.transform_response_native(model, response)?,
|
|
236
|
+
};
|
|
237
|
+
// Captured against this server's target, not the OpenAI adapter's, so
|
|
238
|
+
// the block's origin names the provider that actually issued it.
|
|
171
239
|
crate::reasoning::capture_response(&self.replay_target(model), &native, &mut result);
|
|
172
240
|
crate::toolcall::capture_response(&self.replay_target(model), &native, &mut result)?;
|
|
173
241
|
Ok(result)
|
|
174
242
|
}
|
|
175
243
|
|
|
176
244
|
fn transform_stream_chunk(&self, model: &str, chunk: &str) -> Result<Option<String>> {
|
|
177
|
-
let result = self.
|
|
245
|
+
let result = match self.wire {
|
|
246
|
+
WireFormat::OpenAiResponses => self
|
|
247
|
+
.responses_adapter()
|
|
248
|
+
.transform_stream_chunk_native(model, chunk)?,
|
|
249
|
+
_ => self.transform_stream_chunk_native(model, chunk)?,
|
|
250
|
+
};
|
|
178
251
|
let native: Value = match serde_json::from_str(chunk) {
|
|
179
252
|
Ok(v) => v,
|
|
180
253
|
Err(_) => return Ok(result),
|
|
@@ -189,6 +262,7 @@ impl OpenAiCompatible {
|
|
|
189
262
|
return Err(ShimError::ProviderError {
|
|
190
263
|
status: 502,
|
|
191
264
|
body: "invalid upstream response shape".into(),
|
|
265
|
+
retry_after: None,
|
|
192
266
|
});
|
|
193
267
|
}
|
|
194
268
|
if let Some(err) = response.get("error") {
|
|
@@ -202,6 +276,7 @@ impl OpenAiCompatible {
|
|
|
202
276
|
return Err(ShimError::ProviderError {
|
|
203
277
|
status,
|
|
204
278
|
body: message,
|
|
279
|
+
retry_after: None,
|
|
205
280
|
});
|
|
206
281
|
}
|
|
207
282
|
}
|
|
@@ -267,6 +267,7 @@ impl OpenRouter {
|
|
|
267
267
|
return Err(ShimError::ProviderError {
|
|
268
268
|
status: 502,
|
|
269
269
|
body: "invalid upstream response shape".into(),
|
|
270
|
+
retry_after: None,
|
|
270
271
|
});
|
|
271
272
|
}
|
|
272
273
|
// Non-stream errors usually surface via HTTP status, but a body-level
|
|
@@ -282,6 +283,7 @@ impl OpenRouter {
|
|
|
282
283
|
return Err(ShimError::ProviderError {
|
|
283
284
|
status,
|
|
284
285
|
body: message,
|
|
286
|
+
retry_after: None,
|
|
285
287
|
});
|
|
286
288
|
}
|
|
287
289
|
}
|
|
@@ -370,6 +370,7 @@ impl Xai {
|
|
|
370
370
|
return Err(ShimError::ProviderError {
|
|
371
371
|
status: 400,
|
|
372
372
|
body: msg.to_string(),
|
|
373
|
+
retry_after: None,
|
|
373
374
|
});
|
|
374
375
|
}
|
|
375
376
|
}
|
|
@@ -380,6 +381,7 @@ impl Xai {
|
|
|
380
381
|
.ok_or_else(|| ShimError::ProviderError {
|
|
381
382
|
status: 500,
|
|
382
383
|
body: "no output in response".to_string(),
|
|
384
|
+
retry_after: None,
|
|
383
385
|
})?;
|
|
384
386
|
|
|
385
387
|
let mut text_content: Option<String> = None;
|
|
@@ -440,6 +442,7 @@ impl Xai {
|
|
|
440
442
|
return Err(ShimError::ProviderError {
|
|
441
443
|
status: 502,
|
|
442
444
|
body: "xAI response has no supported terminal status".into(),
|
|
445
|
+
retry_after: None,
|
|
443
446
|
});
|
|
444
447
|
}
|
|
445
448
|
};
|
|
@@ -72,7 +72,7 @@ impl IntoResponse for ApiError {
|
|
|
72
72
|
"unknown_provider",
|
|
73
73
|
format!("Unknown provider or model: {}", p),
|
|
74
74
|
),
|
|
75
|
-
crate::error::ShimError::ProviderError { status, body } => {
|
|
75
|
+
crate::error::ShimError::ProviderError { status, body, .. } => {
|
|
76
76
|
let http_status = StatusCode::from_u16(*status).unwrap_or(StatusCode::BAD_GATEWAY);
|
|
77
77
|
let code = if *status == 400 {
|
|
78
78
|
"invalid_request"
|