llmshim 0.3.7__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llmshim-0.5.0/.github/workflows/catalog-refresh.yml +35 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/.github/workflows/release.yml +17 -4
- {llmshim-0.3.7 → llmshim-0.5.0}/CLAUDE.md +282 -9
- {llmshim-0.3.7 → llmshim-0.5.0}/Cargo.lock +438 -18
- {llmshim-0.3.7 → llmshim-0.5.0}/Cargo.toml +13 -2
- {llmshim-0.3.7 → llmshim-0.5.0}/PKG-INFO +1 -1
- {llmshim-0.3.7 → llmshim-0.5.0}/README.md +48 -16
- {llmshim-0.3.7 → llmshim-0.5.0}/SECURITY.md +17 -0
- llmshim-0.5.0/benchmarks/bench.rs +241 -0
- llmshim-0.5.0/crates/llmshim-catalog/Cargo.toml +28 -0
- llmshim-0.5.0/crates/llmshim-catalog/LICENSE-APACHE +202 -0
- llmshim-0.5.0/crates/llmshim-catalog/LICENSE-MIT +21 -0
- llmshim-0.5.0/crates/llmshim-catalog/README.md +118 -0
- llmshim-0.5.0/crates/llmshim-catalog/data/LICENSE.models.dev +21 -0
- llmshim-0.5.0/crates/llmshim-catalog/data/README.md +8 -0
- llmshim-0.5.0/crates/llmshim-catalog/data/models.dev.json +1 -0
- llmshim-0.5.0/crates/llmshim-catalog/src/aliases.rs +35 -0
- llmshim-0.3.7/src/models.rs → llmshim-0.5.0/crates/llmshim-catalog/src/builtin.rs +119 -137
- llmshim-0.5.0/crates/llmshim-catalog/src/capabilities.rs +105 -0
- llmshim-0.5.0/crates/llmshim-catalog/src/lib.rs +274 -0
- llmshim-0.5.0/crates/llmshim-catalog/src/merge.rs +110 -0
- llmshim-0.5.0/crates/llmshim-catalog/src/parse.rs +124 -0
- llmshim-0.5.0/crates/llmshim-catalog/src/refresh.rs +356 -0
- llmshim-0.5.0/crates/llmshim-catalog/src/types.rs +161 -0
- llmshim-0.5.0/crates/llmshim-catalog/tests/catalog.rs +241 -0
- llmshim-0.5.0/crates/llmshim-catalog/tests/refresh.rs +232 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/SUMMARY.md +4 -1
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/concepts/contracts.md +1 -1
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/concepts/routing.md +47 -1
- llmshim-0.5.0/docs/src/guides/caching.md +66 -0
- llmshim-0.5.0/docs/src/guides/capabilities.md +91 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/guides/fallbacks.md +13 -1
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/guides/reasoning.md +77 -4
- llmshim-0.5.0/docs/src/guides/schemas.md +91 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/guides/streaming.md +13 -10
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/guides/tools.md +44 -5
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/proxy/http-api.md +22 -4
- llmshim-0.5.0/docs/src/proxy/native-apis.md +85 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/proxy/scaling.md +40 -2
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/cli.md +13 -5
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/errors.md +16 -2
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/models.md +23 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/providers.md +5 -5
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/request-fields.md +19 -14
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/surfaces.md +8 -6
- {llmshim-0.3.7 → llmshim-0.5.0}/llmshim/_client.py +21 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/llmshim/types.py +108 -7
- llmshim-0.5.0/src/breaker.rs +454 -0
- llmshim-0.5.0/src/cache.rs +269 -0
- llmshim-0.5.0/src/cli.rs +221 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/client.rs +275 -90
- {llmshim-0.3.7 → llmshim-0.5.0}/src/config.rs +40 -1
- llmshim-0.5.0/src/cost.rs +272 -0
- llmshim-0.5.0/src/error/normalize.rs +143 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/error.rs +5 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/fallback.rs +39 -55
- {llmshim-0.3.7 → llmshim-0.5.0}/src/gateway/auth.rs +17 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/gateway/distributed.rs +29 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/gateway/http.rs +208 -15
- llmshim-0.5.0/src/gateway/quota.rs +354 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/lib.rs +37 -3
- {llmshim-0.3.7 → llmshim-0.5.0}/src/log.rs +24 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/main.rs +100 -35
- llmshim-0.5.0/src/models.rs +7 -0
- llmshim-0.5.0/src/provider.rs +72 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/anthropic.rs +83 -248
- llmshim-0.5.0/src/providers/anthropic_reasoning.rs +111 -0
- llmshim-0.5.0/src/providers/anthropic_signature.rs +165 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/chatgpt/auth.rs +4 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/chatgpt/mod.rs +66 -4
- {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/chatgpt/streaming.rs +1 -78
- {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/gemini.rs +100 -299
- {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/mod.rs +2 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/openai.rs +232 -245
- {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/openai_compat.rs +58 -38
- {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/openrouter.rs +58 -39
- {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/xai.rs +80 -87
- llmshim-0.5.0/src/proxy/convert.rs +328 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/proxy/error.rs +55 -6
- {llmshim-0.3.7 → llmshim-0.5.0}/src/proxy/handlers.rs +10 -21
- llmshim-0.5.0/src/proxy/health.rs +208 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/proxy/mod.rs +8 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/proxy/ratelimit.rs +11 -3
- {llmshim-0.3.7 → llmshim-0.5.0}/src/proxy/types.rs +67 -2
- llmshim-0.5.0/src/proxy/wire/mod.rs +686 -0
- llmshim-0.5.0/src/proxy/wire/receipts.rs +71 -0
- llmshim-0.5.0/src/reasoning/normalize.rs +422 -0
- llmshim-0.5.0/src/reasoning.rs +617 -0
- llmshim-0.5.0/src/router.rs +275 -0
- llmshim-0.5.0/src/schema/memo.rs +426 -0
- llmshim-0.5.0/src/schema/mod.rs +252 -0
- llmshim-0.5.0/src/schema/validate.rs +89 -0
- llmshim-0.5.0/src/schema/walk.rs +1000 -0
- llmshim-0.5.0/src/shim.rs +833 -0
- llmshim-0.5.0/src/streaming.rs +190 -0
- llmshim-0.5.0/src/toolcall/streaming.rs +594 -0
- llmshim-0.5.0/src/toolcall.rs +683 -0
- llmshim-0.5.0/src/usage.rs +191 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_current_models.rs +2 -2
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_openrouter.rs +1 -1
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_sglang.rs +1 -1
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_thinking.rs +4 -4
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_anthropic.rs +68 -32
- llmshim-0.5.0/tests/unit_cache.rs +244 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_chatgpt.rs +53 -7
- llmshim-0.5.0/tests/unit_cli.rs +66 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_fable.rs +4 -4
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_fallback.rs +80 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_gemini.rs +75 -199
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_multimodel.rs +10 -4
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_openai.rs +42 -34
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_openai_compat.rs +7 -4
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_openrouter.rs +3 -3
- llmshim-0.5.0/tests/unit_provider_contracts.rs +219 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_proxy.rs +81 -1
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_proxy_convert.rs +24 -1
- llmshim-0.5.0/tests/unit_reasoning.rs +323 -0
- llmshim-0.5.0/tests/unit_reasoning_profile.rs +52 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_router.rs +142 -0
- llmshim-0.5.0/tests/unit_schema.rs +426 -0
- llmshim-0.5.0/tests/unit_shim.rs +547 -0
- llmshim-0.5.0/tests/unit_signature.rs +155 -0
- llmshim-0.5.0/tests/unit_toolcall.rs +536 -0
- llmshim-0.5.0/tests/unit_usage.rs +303 -0
- llmshim-0.5.0/tests/unit_wire.rs +606 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_xai.rs +51 -58
- llmshim-0.3.7/benchmarks/bench.rs +0 -178
- llmshim-0.3.7/src/gateway/quota.rs +0 -147
- llmshim-0.3.7/src/provider.rs +0 -36
- llmshim-0.3.7/src/proxy/convert.rs +0 -177
- llmshim-0.3.7/src/router.rs +0 -133
- {llmshim-0.3.7 → llmshim-0.5.0}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/.github/workflows/pages.yml +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/.gitignore +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/CODE_OF_CONDUCT.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/CONTRIBUTING.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/LICENSE-APACHE +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/LICENSE-MIT +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/NOTICE +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/benchmarks/bench_python.py +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/benchmarks/gateway_loadtest.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/benchmarks/loadtest.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/.gitignore +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/book.toml +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/mermaid-init.js +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/mermaid.min.js +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/concepts/conversations.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/concepts/portability.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/concepts/translation-flow.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/guides/images.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/guides/native-controls.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/introduction.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/proxy/deployment.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/api.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/configuration.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/start/choose.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/start/cli.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/start/clients.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/start/configure.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/start/proxy.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/start/rust.md +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/examples/chat.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/examples/stream.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/llmshim/__init__.py +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/llmshim/_server.py +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/pyproject.toml +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/env.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/gateway/idempotency.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/gateway/metrics.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/gateway/mod.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/src/vision.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/fixtures/chatgpt-red.png +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_chatgpt.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_chatgpt_proxy.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_fallback.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_gemini.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_gemini_tools.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_long_context.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_multimodel.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_proxy.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_tool_roundtrip.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_vision.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_xai.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/support/completion_status.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_advertised_models.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_fast_mode.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_log.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_models.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_sse.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_tools.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_vision.rs +0 -0
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
name: Refresh model catalog snapshot
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
schedule:
|
|
5
|
+
- cron: '19 6 * * 1'
|
|
6
|
+
workflow_dispatch:
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: write
|
|
10
|
+
pull-requests: write
|
|
11
|
+
|
|
12
|
+
concurrency:
|
|
13
|
+
group: model-catalog-refresh
|
|
14
|
+
cancel-in-progress: false
|
|
15
|
+
|
|
16
|
+
jobs:
|
|
17
|
+
snapshot:
|
|
18
|
+
runs-on: ubuntu-latest
|
|
19
|
+
steps:
|
|
20
|
+
- uses: actions/checkout@v4
|
|
21
|
+
- uses: dtolnay/rust-toolchain@stable
|
|
22
|
+
- name: Download public catalog
|
|
23
|
+
run: |
|
|
24
|
+
curl --fail --silent --show-error --max-time 60 \
|
|
25
|
+
https://models.dev/api.json -o crates/llmshim-catalog/data/models.dev.json
|
|
26
|
+
- name: Validate catalog and refresh behavior
|
|
27
|
+
run: cargo test -p llmshim-catalog
|
|
28
|
+
- name: Propose snapshot update
|
|
29
|
+
uses: peter-evans/create-pull-request@v7
|
|
30
|
+
with:
|
|
31
|
+
branch: automation/model-catalog
|
|
32
|
+
add-paths: crates/llmshim-catalog/data/models.dev.json
|
|
33
|
+
commit-message: 'chore: refresh public model catalog snapshot'
|
|
34
|
+
title: 'Refresh public model catalog snapshot'
|
|
35
|
+
body: 'Updates the vendored models.dev snapshot. Verified builtin assertions and local overrides retain precedence. Catalog and refresh tests passed.'
|
|
@@ -16,11 +16,13 @@ jobs:
|
|
|
16
16
|
- uses: actions/checkout@v4
|
|
17
17
|
- uses: dtolnay/rust-toolchain@stable
|
|
18
18
|
- name: Check formatting
|
|
19
|
-
run: cargo fmt --check
|
|
19
|
+
run: cargo fmt --all --check
|
|
20
20
|
- name: Clippy
|
|
21
|
-
run: cargo clippy --features proxy -- -D warnings
|
|
21
|
+
run: cargo clippy --workspace --features proxy -- -D warnings
|
|
22
22
|
- name: Unit tests
|
|
23
|
-
|
|
23
|
+
env:
|
|
24
|
+
LLMSHIM_CATALOG_OFFLINE: '1'
|
|
25
|
+
run: cargo test --workspace --features proxy --tests
|
|
24
26
|
|
|
25
27
|
# --- Publish to crates.io ---
|
|
26
28
|
crates-io:
|
|
@@ -29,6 +31,17 @@ jobs:
|
|
|
29
31
|
steps:
|
|
30
32
|
- uses: actions/checkout@v4
|
|
31
33
|
- uses: dtolnay/rust-toolchain@stable
|
|
34
|
+
- name: Publish catalog dependency to crates.io
|
|
35
|
+
env:
|
|
36
|
+
CARGO_REGISTRY_TOKEN: ${{ secrets.CARGO_REGISTRY_TOKEN }}
|
|
37
|
+
run: |
|
|
38
|
+
set +e
|
|
39
|
+
out=$(cargo publish -p llmshim-catalog --allow-dirty 2>&1); code=$?
|
|
40
|
+
echo "$out"
|
|
41
|
+
[ $code -eq 0 ] && exit 0
|
|
42
|
+
echo "$out" | grep -qiE "already (been )?uploaded|already exists" \
|
|
43
|
+
&& { echo "Catalog version already on crates.io; skipping."; exit 0; }
|
|
44
|
+
exit $code
|
|
32
45
|
- name: Publish to crates.io
|
|
33
46
|
env:
|
|
34
47
|
CARGO_REGISTRY_TOKEN: ${{ secrets.CARGO_REGISTRY_TOKEN }}
|
|
@@ -36,7 +49,7 @@ jobs:
|
|
|
36
49
|
# release (e.g. to fill a failed downstream registry) is safe.
|
|
37
50
|
run: |
|
|
38
51
|
set +e
|
|
39
|
-
out=$(cargo publish --allow-dirty 2>&1); code=$?
|
|
52
|
+
out=$(cargo publish -p llmshim --allow-dirty 2>&1); code=$?
|
|
40
53
|
echo "$out"
|
|
41
54
|
[ $code -eq 0 ] && exit 0
|
|
42
55
|
echo "$out" | grep -qiE "already (been )?uploaded|already exists" \
|
|
@@ -37,6 +37,91 @@ API keys: `~/.llmshim/config.toml` (via `llmshim configure`) or env vars `OPENAI
|
|
|
37
37
|
|
|
38
38
|
## Architecture
|
|
39
39
|
|
|
40
|
+
### Catalog and usage normalization (local 0.4 development)
|
|
41
|
+
|
|
42
|
+
`llmshim-catalog` is a standalone workspace member. Its `builtin` module owns
|
|
43
|
+
the curated/historical constants; `src/models.rs` reexports the legacy borrowed
|
|
44
|
+
API. Owned metadata and snapshots live in `llmshim::catalog`. Keep the curated
|
|
45
|
+
15-route discovery and four-entry ChatGPT allowlist distinct from catalog
|
|
46
|
+
coverage. Local policy > provider capabilities > verified builtin assertions >
|
|
47
|
+
models.dev, per field; unknowns never erase assertions. Provider APIs never
|
|
48
|
+
contribute pricing. See `crates/llmshim-catalog/README.md` for cache and override
|
|
49
|
+
paths, offline behavior, aliases, and publication order. Never await catalog
|
|
50
|
+
refresh in a completion; hold a snapshot for decisions that must agree.
|
|
51
|
+
|
|
52
|
+
`usage.cache_read_tokens` and `usage.cache_write_tokens` are always present on
|
|
53
|
+
normalized responses and usage chunks, logs, and proxy usage. Native token
|
|
54
|
+
fields remain readable. `src/usage.rs` owns extraction; the streaming client
|
|
55
|
+
merges Anthropic's start/delta usage before normalizing terminal counts.
|
|
56
|
+
`usage.uncached_input_tokens` joins them: the providers disagree on whether
|
|
57
|
+
their prompt total already includes the cache read (Anthropic's excludes it,
|
|
58
|
+
OpenAI/Chat Completions/Gemini include it), and only the transport boundary
|
|
59
|
+
still knows which convention the body used. The convention is read off the
|
|
60
|
+
cache-read field that actually matched, never a provider-name table.
|
|
61
|
+
|
|
62
|
+
### USD cost accounting
|
|
63
|
+
|
|
64
|
+
`src/cost.rs` is the only thing that multiplies a catalog `Cost` (USD per
|
|
65
|
+
million tokens) by those counters. Input is charged on `uncached_input_tokens`
|
|
66
|
+
so a cached prompt is never billed twice; cache reads and writes are charged at
|
|
67
|
+
their own rates; reasoning tokens are already inside `completion_tokens`.
|
|
68
|
+
**An absent price is `None`, never `0.0`** — a positive count in a bucket with
|
|
69
|
+
no rate poisons the whole total rather than producing a partial sum that reads
|
|
70
|
+
as a complete one. `client.rs` stamps `usage.cost_usd` at the transport
|
|
71
|
+
boundary (the last place that knows the dispatch target) for completions and
|
|
72
|
+
for whichever stream chunk carries usage; `log.rs`, `proxy::types::Usage`, both
|
|
73
|
+
native facades and the four bundled clients carry it through as a nullable
|
|
74
|
+
field. `null` means unknown, not free.
|
|
75
|
+
|
|
76
|
+
`src/gateway/quota.rs` adds a per-identity dollar cap beside the RPM/TPM
|
|
77
|
+
buckets: `budget_usd` + `budget_window_secs` on an `Identity`, checked before
|
|
78
|
+
dispatch and charged after (cost is only knowable once a response exists, so
|
|
79
|
+
one in-flight request can overshoot). Windows tumble rather than slide, because
|
|
80
|
+
the fleet-wide store is one counter per window. `SpendCap::with_store` takes the
|
|
81
|
+
Redis-backed `DistributedGateway` in distributed mode so `$100/day` means one
|
|
82
|
+
hundred dollars fleet-wide, not per replica. **A response the catalog cannot
|
|
83
|
+
price is not charged** — recording zero would let an unpriced model run forever
|
|
84
|
+
under a budget; `cost_usd: null` is the signal that a price is missing.
|
|
85
|
+
The native Chat Completions streams must use their own parser in the client;
|
|
86
|
+
passing them to the Responses parser silently drops all events.
|
|
87
|
+
|
|
88
|
+
Offline checks for this work:
|
|
89
|
+
|
|
90
|
+
```sh
|
|
91
|
+
LLMSHIM_CATALOG_OFFLINE=1 cargo test --workspace --features proxy --tests
|
|
92
|
+
cargo test -p llmshim-catalog
|
|
93
|
+
cargo clippy --workspace --features proxy -- -D warnings
|
|
94
|
+
cargo package -p llmshim-catalog --allow-dirty
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
The root package is 0.4.0 because log/proxy usage structs gain fields; no
|
|
98
|
+
release has been performed. Future release workflows must publish the catalog
|
|
99
|
+
dependency before llmshim. Public code, fixtures, artifacts, and docs must use
|
|
100
|
+
generic examples and contain no private consumer identities or context.
|
|
101
|
+
|
|
102
|
+
### Cache annotations and shared schema normalization
|
|
103
|
+
|
|
104
|
+
`src/cache.rs` translates caller `x-cache` segments into native Anthropic
|
|
105
|
+
breakpoints and an explicit Responses prompt_cache_key. It never infers
|
|
106
|
+
stability. Last eligible boundaries win within the four-slot budget; existing
|
|
107
|
+
explicit markers consume slots. Managed segments supersede automatic caching.
|
|
108
|
+
One-hour markers must precede five-minute markers. Marker scans inspect actual
|
|
109
|
+
cache locations, not arbitrary schema/default JSON. Without annotations, native
|
|
110
|
+
passthrough remains unchanged. `ProviderRequest::can_continue_from` checks
|
|
111
|
+
endpoint, headers, settings (including include/store/reasoning) and input prefix;
|
|
112
|
+
there is no stored/delta continuation engine. See the caching guide.
|
|
113
|
+
|
|
114
|
+
`src/schema/` owns the single schema walker and local resource resolver. Every
|
|
115
|
+
adapter normalizes native tool schemas after overrides. MCP inputSchema is
|
|
116
|
+
normalized on ingest and raw MCP tool definitions are accepted. Keep literal
|
|
117
|
+
values/property names separate from schema-node traversal, preserve meaningful
|
|
118
|
+
stripped constraints in descriptions, and never fetch an external reference.
|
|
119
|
+
Cycles, unresolved resources, incompatible residues and expansion-budget failures
|
|
120
|
+
fall back per tool. Only successful enforcement permits strict:true. Global
|
|
121
|
+
bypass flags: LLMSHIM_NO_SCHEMA_NORMALIZATION and LLMSHIM_NO_STRICT. OutputSchema
|
|
122
|
+
provides a reversible non-object wrapper; generated-instance validation and
|
|
123
|
+
repair remain a separate concern. See `docs/src/guides/schemas.md`.
|
|
124
|
+
|
|
40
125
|
### Curated discovery
|
|
41
126
|
|
|
42
127
|
`src/models.rs::MODELS` is the single advertised list, imported directly by
|
|
@@ -72,10 +157,11 @@ Preserve `chatgpt/<model>` in normalized responses and chunks: a bare GPT name
|
|
|
72
157
|
is otherwise misattributed to API-key OpenAI by the proxy/gateway when both
|
|
73
158
|
providers are registered. Astra preserves reasoning effort `max`; `none` and
|
|
74
159
|
`minimal` clamp to `low`.
|
|
75
|
-
For ChatGPT streaming,
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
160
|
+
For ChatGPT streaming, use the same `StreamNormalizer`/`ToolStream` as other
|
|
161
|
+
transports. Do not reintroduce a separate completed-call emitter: it used to
|
|
162
|
+
lose or duplicate argument fragments at the proxy boundary. Text and reasoning
|
|
163
|
+
remain incremental; callable tools are complete and emitted once at termination.
|
|
164
|
+
|
|
79
165
|
|
|
80
166
|
Offline coverage lives in `tests/unit_chatgpt.rs`. The live server check starts
|
|
81
167
|
its own loopback CLI process and stops it on completion/failure:
|
|
@@ -91,9 +177,9 @@ and image input. It uses the saved ChatGPT login and consumes
|
|
|
91
177
|
subscription usage; it is ignored during offline CI. Mount the whole token
|
|
92
178
|
directory writable for container use so refresh locks and atomic saves work.
|
|
93
179
|
|
|
94
|
-
**Self-hosted passthrough providers (vLLM / SGLang).** `src/providers/openai_compat.rs` is one generic OpenAI Chat Completions passthrough backing both `vllm` and `sglang` (`OpenAiCompatible::new(name, base_url, api_key: Option)`, registered per env base URL). Two things differ from the hosted providers: the **base URL is configuration** (local `http://localhost:8000/v1` vs remote `https://host/v1`), and **auth is optional** (self-hosted servers are unauthenticated unless launched with `--api-key`, so the `Authorization` header is sent only when a key is set). Passthrough transforms;
|
|
180
|
+
**Self-hosted passthrough providers (vLLM / SGLang).** `src/providers/openai_compat.rs` is one generic OpenAI Chat Completions passthrough backing both `vllm` and `sglang` (`OpenAiCompatible::new(name, base_url, api_key: Option)`, registered per env base URL). Two things differ from the hosted providers: the **base URL is configuration** (local `http://localhost:8000/v1` vs remote `https://host/v1`), and **auth is optional** (self-hosted servers are unauthenticated unless launched with `--api-key`, so the `Authorization` header is sent only when a key is set). Passthrough transforms; reasoning normalized to typed `reasoning[]` with provenance (vLLM is migrating the field name); `reasoning_effort` forwarded as-is (honored per-model, not clamped); server-specific params go under `x-<name>` (`chat_template_kwargs`, `separate_reasoning`, `guided_json`, `top_k`, …). Note: reasoning/tool parsing are **launch-time server flags** (`--reasoning-parser`, `--tool-call-parser`), so a request only gets that behavior if the server was started for it — llmshim can't enable it per request.
|
|
95
181
|
|
|
96
|
-
**OpenRouter is the one passthrough provider.** Every other provider translates the OpenAI-format input *away* to a native dialect; OpenRouter (`src/providers/openrouter.rs`) *is* OpenAI Chat Completions, so its transforms are near-identity — messages, tools, vision (`image_url`), and `response_format` are forwarded unchanged; `reasoning_effort` maps 1:1 to OpenRouter's `reasoning:{effort}` (its effort vocabulary is a superset, so no clamping);
|
|
182
|
+
**OpenRouter is the one passthrough provider.** Every other provider translates the OpenAI-format input *away* to a native dialect; OpenRouter (`src/providers/openrouter.rs`) *is* OpenAI Chat Completions, so its transforms are near-identity — messages, tools, vision (`image_url`), and `response_format` are forwarded unchanged; `reasoning_effort` maps 1:1 to OpenRouter's `reasoning:{effort}` (its effort vocabulary is a superset, so no clamping); reasoning is normalized to typed `reasoning[]` with provenance on responses. OpenRouter models are **not enumerated** in `src/models.rs` (the catalog is huge and dynamic) — any `openrouter/<vendor>/<model>` slug routes through. `x-openrouter` carries OpenRouter-only controls (`provider`, `models`, `transforms`, `route`, native `reasoning`; plus `http_referer`/`x_title` which become headers). The `middle-out` transform is disabled by default for faithful passthrough. Uses `image_url` (Chat Completions) vision via `vision::to_openai_chat`.
|
|
97
183
|
|
|
98
184
|
### Request flow
|
|
99
185
|
|
|
@@ -126,19 +212,78 @@ Automatic redirects are disabled on the shared client. Keep prompts and
|
|
|
126
212
|
provider-specific credential headers at the configured endpoint; 3xx responses
|
|
127
213
|
remain provider errors. Callers must configure the final URL directly.
|
|
128
214
|
|
|
129
|
-
`ShimClient` with shared connection pool (`LazyLock`), HTTP/2, gzip/brotli/zstd compression, TCP keepalive + nodelay. Automatic retry (3 attempts by default) on transport errors and 429/500/502/503/504/529 status codes. This is the **reactive** layer: on a retryable *response* it honors the server's `Retry-After` header (integer seconds or HTTP-date) and provider reset hints (OpenAI `x-ratelimit-reset-*`, Anthropic `anthropic-ratelimit-*-reset`), clamped to a cap and nudged with a little jitter; when there's no server hint (or a transport error) it falls back to full-jitter exponential backoff (uniform in `[0, min(cap, base·2^attempt)]`) to avoid a thundering herd. Tunable via `LLMSHIM_MAX_RETRIES` and `LLMSHIM_MAX_BACKOFF_SECS`. `warmup()` pre-establishes TCP+TLS connections. `SseStream`
|
|
215
|
+
`ShimClient` with shared connection pool (`LazyLock`), HTTP/2, gzip/brotli/zstd compression, TCP keepalive + nodelay. Automatic retry (3 attempts by default) on transport errors and 429/500/502/503/504/529 status codes. This is the **reactive** layer: on a retryable *response* it honors the server's `Retry-After` header (integer seconds or HTTP-date) and provider reset hints (OpenAI `x-ratelimit-reset-*`, Anthropic `anthropic-ratelimit-*-reset`), clamped to a cap and nudged with a little jitter; when there's no server hint (or a transport error) it falls back to full-jitter exponential backoff (uniform in `[0, min(cap, base·2^attempt)]`) to avoid a thundering herd. Tunable via `LLMSHIM_MAX_RETRIES` and `LLMSHIM_MAX_BACKOFF_SECS`. `warmup()` pre-establishes TCP+TLS connections. `SseStream` decodes bytes with eventsource-stream and feeds one per-response `StreamNormalizer`; do not restore lossy UTF-8 line buffering.
|
|
130
216
|
|
|
131
217
|
### Fallback chains (`src/fallback.rs`)
|
|
132
218
|
|
|
133
219
|
`FallbackConfig` defines an ordered list of models to try. On retryable errors (429, 500, 502, 503, 529), retries with exponential backoff then falls through to the next model. `completion_with_fallback()` is the top-level API. The proxy supports this via `"fallback": ["model1", "model2"]` in the request body.
|
|
134
220
|
|
|
221
|
+
### Provider health (`src/breaker.rs`, `src/proxy/health.rs`)
|
|
222
|
+
|
|
223
|
+
**Health is not rate-limit backoff.** The token buckets already slow a provider
|
|
224
|
+
down after a 429 — a 429 means the provider is alive and asking for less. The
|
|
225
|
+
breaker counts what retrying cannot fix: 5xx (500/502/503/504/529) and
|
|
226
|
+
transport failures. Adapted from `rcode-provider`'s `ProviderBreaker`, which we
|
|
227
|
+
own: sliding failure window, open state, and a single half-open probe admitted
|
|
228
|
+
after the cooldown. Config: `LLMSHIM_BREAKER_WINDOW_SECS` (60),
|
|
229
|
+
`LLMSHIM_BREAKER_TRIP_THRESHOLD` (3; `0` disables), `LLMSHIM_BREAKER_COOLDOWN_SECS` (30).
|
|
230
|
+
|
|
231
|
+
The breaker hangs on the `Router` (`Router::breaker()` / `with_breaker`). Every
|
|
232
|
+
dispatch path *observes* outcomes so health accrues from ordinary traffic; only
|
|
233
|
+
`fallback.rs` *refuses*, and it checks before every attempt rather than once per
|
|
234
|
+
chain entry — the attempt that opens a circuit is usually the chain's own, so a
|
|
235
|
+
per-entry check would still retry into a target it just watched die. A single-target call is still dispatched:
|
|
236
|
+
with no alternative, refusing would only convert an upstream failure into a
|
|
237
|
+
local one. `proxy::health::build_breaker()` attaches a Redis-coordinated
|
|
238
|
+
`SharedHealth` (failure ZSET + open marker + `SET NX` probe) when
|
|
239
|
+
`LLMSHIM_REDIS_URL` is set and `redis-coordination` is compiled in, mirroring
|
|
240
|
+
`build_limiter`. Every shared operation **fails open**.
|
|
241
|
+
|
|
242
|
+
### Named routes (`src/config.rs`, `src/router.rs`)
|
|
243
|
+
|
|
244
|
+
A caller-defined name maps to a model plus request settings:
|
|
245
|
+
|
|
246
|
+
```toml
|
|
247
|
+
[routes.compaction]
|
|
248
|
+
model = "anthropic/claude-haiku-4-5-20251001"
|
|
249
|
+
reasoning_effort = "low"
|
|
250
|
+
max_tokens = 4096
|
|
251
|
+
```
|
|
252
|
+
|
|
253
|
+
Addressed as `"model": "route/compaction"`, so a route travels through the
|
|
254
|
+
existing `provider/model` grammar — an OpenAI SDK, the CLI and the proxy's
|
|
255
|
+
admission control all handle it without learning a new field. `resolve_key`
|
|
256
|
+
resolves the indirection, so rate limiting never sees an unrecognized string.
|
|
257
|
+
|
|
258
|
+
**llmshim must not learn harness vocabulary.** The name is opaque: a harness may
|
|
259
|
+
call a route `compaction`, `advisor` or `webSearch`, and llmshim only knows it
|
|
260
|
+
maps to a model. Route settings are defaults — a per-request key always wins —
|
|
261
|
+
and an unknown name is a 400, never a silent fall back to the default model.
|
|
262
|
+
Routes do not chain.
|
|
263
|
+
|
|
135
264
|
### Vision (`src/vision.rs`)
|
|
136
265
|
|
|
137
266
|
Image content blocks are translated between providers automatically. Users can send images in any format (OpenAI `image_url`, Anthropic `image`, Gemini `inline_data`) and the correct provider sees its native format. Base64 data URIs and plain URLs are both handled. Gemini falls back to a text placeholder for URL images (only supports `inline_data`).
|
|
138
267
|
|
|
139
268
|
### Multi-model conversations
|
|
140
269
|
|
|
141
|
-
|
|
270
|
+
`src/reasoning.rs` owns the single replay policy, typed blocks, signature origins,
|
|
271
|
+
issuer bindings, and drop counters. Every adapter filters before serialization
|
|
272
|
+
and captures from the original response to preserve ordered blocks and encrypted
|
|
273
|
+
Responses items. The shared HTTP client binds origins to the actual request
|
|
274
|
+
(including refreshed OAuth account headers). Unknown provenance/families fail
|
|
275
|
+
closed; matching family+wire permits replay, with same-account binding also
|
|
276
|
+
required for encrypted blocks. Never route by inspecting opaque bytes.
|
|
277
|
+
|
|
278
|
+
New outputs use `message.reasoning[]` and `thought_signature:{data,origin}`.
|
|
279
|
+
Legacy sibling readers exist for one migration release and drop untracked data.
|
|
280
|
+
Native overrides cannot bypass this filter or enable provider-side Responses
|
|
281
|
+
storage. The CLI and proxy retain the whole message; streaming consumers use
|
|
282
|
+
`ReasoningAccumulator` and must preserve signatures and completed item snapshots.
|
|
283
|
+
The shared SSE reader handles split UTF-8, CRLF, multiline events, late usage,
|
|
284
|
+
and terminal markers; do not restore the old per-byte lossy string buffer.
|
|
285
|
+
Tests: `tests/unit_reasoning.rs`, the client SSE tests, and provider regressions.
|
|
286
|
+
See `docs/src/guides/reasoning.md` for the full shape and migration contract.
|
|
142
287
|
|
|
143
288
|
### Provider extension namespaces (`x-anthropic`, `x-gemini`)
|
|
144
289
|
|
|
@@ -180,6 +325,30 @@ the verified xAI model ID. Keep the ChatGPT four-model allowlist independent.
|
|
|
180
325
|
|
|
181
326
|
Two knobs work across every provider: `reasoning_effort` (`none|low|medium|high|xhigh|max`) and `reasoning_mode` (`standard|pro`). A third, `reasoning_summary` (`auto|none`), controls reasoning-text visibility → Anthropic `thinking.display` (`auto`→`summarized`, the default when `reasoning_effort` is present so newer models like Sonnet 5 / Opus 4.7-4.8 return reasoning text instead of the API-default `omitted`; `none`→`omitted` for lower latency). Applies to both the adaptive and pre-4.6 enabled thinking builders; a caller-supplied `thinking` block bypasses it. Each provider transform maps them to its native dialect, **clamping to the nearest tier the target model accepts** (all boundaries verified live — e.g. `max` is native on OpenAI gpt-5.6 and GPT-6 Astra; Anthropic 4.6 rejects `xhigh` but has `max`; Gemini's enum tops out at `high`; xAI grok-4.20 models reject any reasoning param). `mode: "pro"` is native on OpenAI gpt-5.6/-pro models (`reasoning.mode`), emulated as a one-tier effort bump elsewhere; explicit `none` always wins. Native passthrough (`x-openai.reasoning`, `x-anthropic.thinking`, `x-gemini.thinkingConfig`) bypasses the mapping entirely and always takes precedence. **Full per-provider mapping tables: `docs/src/guides/reasoning.md`** — update it and the pinning tests in `tests/unit_*.rs` together whenever a mapping changes.
|
|
182
327
|
|
|
328
|
+
### Owned tool identities and stream state
|
|
329
|
+
|
|
330
|
+
`src/toolcall.rs` owns `WireToolId`, the bidirectional map, request projection,
|
|
331
|
+
and central call/result validation. Canonical ids are minted as `call_ls_*` and
|
|
332
|
+
never use a provider id directly. `wire_ids` must remain on persisted calls;
|
|
333
|
+
Responses `item_id` is distinct from correlation `id`. Source wire ids (including
|
|
334
|
+
Gemini's missing id) are restored on both calls/results, preserving signed
|
|
335
|
+
prefixes. Legacy input ids remain readable, but an owned id without its mapping
|
|
336
|
+
is an error. Never drop invalid tool history to make a request succeed.
|
|
337
|
+
|
|
338
|
+
`src/toolcall/streaming.rs` parses native events into `ToolDelta` and assembles
|
|
339
|
+
JSON arguments once. `src/streaming.rs::StreamNormalizer` is the public stateful
|
|
340
|
+
entry point used by HTTP streaming and manual SSE readers. Stateless provider
|
|
341
|
+
chunk methods no longer expose partial callable records. Tool signatures can
|
|
342
|
+
arrive after names/arguments; preserve their exact data/origin and original wire
|
|
343
|
+
container. Parallel choices close independently. The single-message proxy
|
|
344
|
+
projects choice zero; Rust retains choice indices.
|
|
345
|
+
|
|
346
|
+
Google's current Generate Content contract requires a signature on the first
|
|
347
|
+
function call of each current Gemini 3 batch, not every parallel call. Validate
|
|
348
|
+
that rule after filtering, preserve additional signatures exactly where present,
|
|
349
|
+
and retain optional native function ids on both sides. `unit_toolcall` covers
|
|
350
|
+
paired replay, order, signatures, invalid history, and transport delta sequences.
|
|
351
|
+
|
|
183
352
|
### Tool format translation
|
|
184
353
|
|
|
185
354
|
llmshim accepts tools in OpenAI Chat Completions format (nested `function` object) and translates them to each provider's native format:
|
|
@@ -206,7 +375,13 @@ HTTP proxy with our own API spec (not OpenAI-compatible). Built on axum.
|
|
|
206
375
|
Endpoints:
|
|
207
376
|
- `POST /v1/chat` — non-streaming (or streaming if `stream: true`)
|
|
208
377
|
- `POST /v1/chat/stream` — always SSE streaming with typed events (`content`, `reasoning`, `tool_call`, `usage`, `done`, `error`)
|
|
209
|
-
- `GET /v1/models` — list available models (filtered to configured providers)
|
|
378
|
+
- `GET /v1/models` — list available models (filtered to configured providers).
|
|
379
|
+
Serves two audiences from one body: an OpenAI SDK reads the `object: "list"` /
|
|
380
|
+
`data[]` envelope (so `client.models.list()` works unmodified), llmshim's own
|
|
381
|
+
clients read `models[]`. Both issue the same request, so there is no path to
|
|
382
|
+
split on — the union *is* the split. `data[].id` is the routing id, requestable
|
|
383
|
+
back as `model`. Built once in `proxy::convert::models_response`, shared with
|
|
384
|
+
the gateway.
|
|
210
385
|
- `GET /health` — health check with provider list
|
|
211
386
|
|
|
212
387
|
Request format uses `config` for provider-agnostic settings and `provider_config` for raw passthrough. OpenAPI 3.1 spec at `api/openapi.yaml`.
|
|
@@ -292,3 +467,101 @@ Common maintenance workflows are packaged as [skills](https://code.claude.com/do
|
|
|
292
467
|
- `/add-provider key Name` — wire up a brand-new upstream provider.
|
|
293
468
|
- `/preflight` — run the fmt + clippy + test trio CI enforces.
|
|
294
469
|
- `/release 0.1.22` — bump version and tag so CI publishes.
|
|
470
|
+
|
|
471
|
+
### Capability plans and instance validation
|
|
472
|
+
|
|
473
|
+
`src/shim.rs::Plan` owns catalog-driven structured output, prompt tool calling,
|
|
474
|
+
and optional brief-rationale capture. `ShimClient` applies the same plan on
|
|
475
|
+
completion, stream, and fallback paths; direct provider transforms only support
|
|
476
|
+
native response-format translation. Never emit hidden synthetic calls, native
|
|
477
|
+
deliberation about those calls, or invalid attempts to logs/streams. Managed
|
|
478
|
+
streams buffer with a 32 MiB bound; top-level streaming uses an Arc-owned provider
|
|
479
|
+
so HTTP headers/keepalives can proceed while output is validated.
|
|
480
|
+
|
|
481
|
+
Validate generated data against the ORIGINAL schema with the network/filesystem
|
|
482
|
+
retriever disabled (`schema::validate`). Schema compile budgets are 96 levels,
|
|
483
|
+
32,768 JSON values and 8 MiB strings/keys. Only complete invalid answers receive
|
|
484
|
+
one repair; refusals and incomplete responses do not. Preserve both attempts'
|
|
485
|
+
reported usage. Unknown catalog fields remain unknown; `forced_tool_choice` is
|
|
486
|
+
independent of tools, with the verified Fable 5.1 prohibition overriding auto.
|
|
487
|
+
Keep synthetic schemas/instructions deterministic so unchanged requests preserve
|
|
488
|
+
prefix caching. Tests: `unit_shim`, proxy conversion tests, client request mocks.
|
|
489
|
+
|
|
490
|
+
### Signature observations and reasoning profiles
|
|
491
|
+
|
|
492
|
+
`providers/anthropic_signature.rs` has the default-on `signature-introspection`
|
|
493
|
+
feature and Option-only stub. Never use decoded metadata to change provenance,
|
|
494
|
+
messages, replay, routing, fallback, or caches. Synthetic fixtures only. Capture
|
|
495
|
+
one metric observation per response/terminal stream; logs read the observation
|
|
496
|
+
without incrementing it. `x-llmshim-served-model` belongs at response/event level.
|
|
497
|
+
|
|
498
|
+
`providers/anthropic_reasoning.rs::Profile` consumes catalog effort/budget unions
|
|
499
|
+
once per provider transform. Builtin verified options win over community data;
|
|
500
|
+
local/provider overrides retain documented precedence. Keep historical fallback
|
|
501
|
+
entries explicit in `catalog::builtin::anthropic_reasoning_options`, rather than
|
|
502
|
+
guessing support for new model names. Preserve mandatory native model constraints
|
|
503
|
+
(e.g. Fable adaptive-only and forbidden forced choices) separately.
|
|
504
|
+
|
|
505
|
+
### Native inbound facades
|
|
506
|
+
|
|
507
|
+
`proxy::wire` translates `/v1/messages` and `/v1/chat/completions` through the
|
|
508
|
+
existing chat handlers on both proxy and gateway. Do not create another dispatch,
|
|
509
|
+
auth, quota, or queue path. Gateway authenticates before receipt lookup and again
|
|
510
|
+
in its ordinary admission path; x-api-key maps to Bearer only when Authorization
|
|
511
|
+
is absent. Native idempotency keys include credential and protocol scope.
|
|
512
|
+
|
|
513
|
+
The native facade persists issued replay metadata in private atomic local files,
|
|
514
|
+
configured by `LLMSHIM_REPLAY_RECEIPTS_DIR`. Clients must preserve the native
|
|
515
|
+
message/ID and server receipts across restarts. Never stamp unknown native thinking
|
|
516
|
+
with current-target provenance. Receipts restore original blocks, then the common
|
|
517
|
+
replay filter decides eligibility. Missing owned IDs and edited calls error.
|
|
518
|
+
Text remains incremental; complete reasoning/tool blocks follow once metadata is
|
|
519
|
+
ready. Status/Retry-After and gateway request IDs survive error translation.
|
|
520
|
+
`src/error/normalize.rs` owns error unwrapping for compact JSON/SSE, gateway,
|
|
521
|
+
and native endpoints. Keep display messages readable and source type/code/param
|
|
522
|
+
metadata separate; do not move this logic back into a wire-only formatter.
|
|
523
|
+
JSON responses carry native metadata in response extensions, and SSE errors carry
|
|
524
|
+
an optional structured error object so native rendering remains lossless.
|
|
525
|
+
`n > 1` is refused rather than emulated: the OpenAI backend is the Responses
|
|
526
|
+
API (no `n`), Anthropic Messages and Gemini have no `n` either, and the
|
|
527
|
+
single-message proxy projects choice zero. The refusal is a correctly shaped
|
|
528
|
+
`{"error":{type,message,param:"n",code:"unsupported_parameter"}}`, carried
|
|
529
|
+
through `normalize_error` so both facades render it natively.
|
|
530
|
+
Tests: `unit_wire`, gateway `http::native_tests`. Use a temporary receipt directory
|
|
531
|
+
in tests; never put real signatures, credentials, or conversations in fixtures.
|
|
532
|
+
|
|
533
|
+
### Provider and CLI regression gates
|
|
534
|
+
|
|
535
|
+
`tests/unit_provider_contracts.rs` discovers Provider implementations from the
|
|
536
|
+
Rust AST and requires a fixture for each. New adapters must preserve compatible
|
|
537
|
+
reasoning, drop foreign/untracked blocks and signatures, and reject invalid tool
|
|
538
|
+
history before serialization. Tool-call containers accept arrays or null only.
|
|
539
|
+
HTTP handlers validate canonical history before committing to an SSE response.
|
|
540
|
+
|
|
541
|
+
`src/cli.rs` validates arguments before side effects. Server options override env
|
|
542
|
+
and saved host/port; help must not start a service. Bind errors are returned to
|
|
543
|
+
main for a clean exit. Tests use ephemeral loopback ports and never touch an
|
|
544
|
+
existing server. Run unit_cli, unit_provider_contracts, unit_wire and gateway
|
|
545
|
+
HTTP tests when changing these boundaries.
|
|
546
|
+
|
|
547
|
+
### Schema memoization and benchmark receipts
|
|
548
|
+
|
|
549
|
+
`src/schema/memo.rs` caches only schema inputs/results and `Normalization` reports.
|
|
550
|
+
The process-local cache is bounded to 512 entries/16 MiB estimated owned bytes,
|
|
551
|
+
uses LRU eviction, verifies input/options after a hash hit, and performs expensive
|
|
552
|
+
normalization outside the lock. Key all effective Options, including budgets.
|
|
553
|
+
Resolve environment overrides before lookup. Preserve object order and signed
|
|
554
|
+
zero: semantically equal numbers can produce different spilled descriptions.
|
|
555
|
+
No-op hits can retain the input value only after exact input/output comparison.
|
|
556
|
+
|
|
557
|
+
`LLMSHIM_NO_SCHEMA_CACHE=1` bypasses memoization while keeping normalization active.
|
|
558
|
+
Tests cover collisions, eviction, concurrency, returned-value independence,
|
|
559
|
+
report/fallback fidelity, environment changes and fresh tool metadata.
|
|
560
|
+
|
|
561
|
+
`cargo run --release --example bench` loads the two configured native API keys and
|
|
562
|
+
makes 43 logical requests plus client retries. Missing keys fail before network
|
|
563
|
+
calls with setup instructions. Provider errors must never print credentials.
|
|
564
|
+
`--transforms-only` needs no keys or HTTP calls; use it for cache-on/off comparison.
|
|
565
|
+
Keep README measurements dated, name the actual models/host/build/workload, and
|
|
566
|
+
separate API latency, full-transform CPU time, and RSS. Do not mix fresh Rust
|
|
567
|
+
figures with old Python results or claim unmeasured network-only latency shares.
|