llmshim 0.3.7__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llmshim-0.4.0/.github/workflows/catalog-refresh.yml +35 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/.github/workflows/release.yml +17 -4
- {llmshim-0.3.7 → llmshim-0.4.0}/CLAUDE.md +198 -8
- {llmshim-0.3.7 → llmshim-0.4.0}/Cargo.lock +438 -18
- {llmshim-0.3.7 → llmshim-0.4.0}/Cargo.toml +13 -2
- {llmshim-0.3.7 → llmshim-0.4.0}/PKG-INFO +1 -1
- {llmshim-0.3.7 → llmshim-0.4.0}/README.md +48 -16
- {llmshim-0.3.7 → llmshim-0.4.0}/SECURITY.md +17 -0
- llmshim-0.4.0/benchmarks/bench.rs +241 -0
- llmshim-0.4.0/crates/llmshim-catalog/Cargo.toml +28 -0
- llmshim-0.4.0/crates/llmshim-catalog/LICENSE-APACHE +202 -0
- llmshim-0.4.0/crates/llmshim-catalog/LICENSE-MIT +21 -0
- llmshim-0.4.0/crates/llmshim-catalog/README.md +118 -0
- llmshim-0.4.0/crates/llmshim-catalog/data/LICENSE.models.dev +21 -0
- llmshim-0.4.0/crates/llmshim-catalog/data/README.md +8 -0
- llmshim-0.4.0/crates/llmshim-catalog/data/models.dev.json +1 -0
- llmshim-0.4.0/crates/llmshim-catalog/src/aliases.rs +35 -0
- llmshim-0.3.7/src/models.rs → llmshim-0.4.0/crates/llmshim-catalog/src/builtin.rs +119 -137
- llmshim-0.4.0/crates/llmshim-catalog/src/capabilities.rs +105 -0
- llmshim-0.4.0/crates/llmshim-catalog/src/lib.rs +274 -0
- llmshim-0.4.0/crates/llmshim-catalog/src/merge.rs +110 -0
- llmshim-0.4.0/crates/llmshim-catalog/src/parse.rs +124 -0
- llmshim-0.4.0/crates/llmshim-catalog/src/refresh.rs +356 -0
- llmshim-0.4.0/crates/llmshim-catalog/src/types.rs +161 -0
- llmshim-0.4.0/crates/llmshim-catalog/tests/catalog.rs +241 -0
- llmshim-0.4.0/crates/llmshim-catalog/tests/refresh.rs +232 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/SUMMARY.md +4 -1
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/concepts/contracts.md +1 -1
- llmshim-0.4.0/docs/src/guides/caching.md +53 -0
- llmshim-0.4.0/docs/src/guides/capabilities.md +91 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/guides/reasoning.md +77 -4
- llmshim-0.4.0/docs/src/guides/schemas.md +91 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/guides/streaming.md +13 -10
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/guides/tools.md +44 -5
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/proxy/http-api.md +9 -3
- llmshim-0.4.0/docs/src/proxy/native-apis.md +74 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/reference/cli.md +13 -5
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/reference/errors.md +16 -2
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/reference/models.md +23 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/reference/providers.md +5 -5
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/reference/request-fields.md +19 -14
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/reference/surfaces.md +8 -6
- {llmshim-0.3.7 → llmshim-0.4.0}/llmshim/_client.py +21 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/llmshim/types.py +102 -6
- llmshim-0.4.0/src/cache.rs +269 -0
- llmshim-0.4.0/src/cli.rs +221 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/client.rs +267 -91
- llmshim-0.4.0/src/error/normalize.rs +143 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/error.rs +5 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/fallback.rs +12 -54
- {llmshim-0.3.7 → llmshim-0.4.0}/src/gateway/http.rs +156 -1
- {llmshim-0.3.7 → llmshim-0.4.0}/src/lib.rs +11 -2
- {llmshim-0.3.7 → llmshim-0.4.0}/src/log.rs +15 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/main.rs +100 -35
- llmshim-0.4.0/src/models.rs +7 -0
- llmshim-0.4.0/src/provider.rs +72 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/providers/anthropic.rs +83 -248
- llmshim-0.4.0/src/providers/anthropic_reasoning.rs +111 -0
- llmshim-0.4.0/src/providers/anthropic_signature.rs +165 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/providers/chatgpt/auth.rs +4 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/providers/chatgpt/mod.rs +66 -4
- {llmshim-0.3.7 → llmshim-0.4.0}/src/providers/chatgpt/streaming.rs +1 -78
- {llmshim-0.3.7 → llmshim-0.4.0}/src/providers/gemini.rs +100 -299
- {llmshim-0.3.7 → llmshim-0.4.0}/src/providers/mod.rs +2 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/providers/openai.rs +232 -245
- {llmshim-0.3.7 → llmshim-0.4.0}/src/providers/openai_compat.rs +58 -38
- {llmshim-0.3.7 → llmshim-0.4.0}/src/providers/openrouter.rs +58 -39
- {llmshim-0.3.7 → llmshim-0.4.0}/src/providers/xai.rs +80 -87
- llmshim-0.4.0/src/proxy/convert.rs +291 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/proxy/error.rs +55 -6
- {llmshim-0.3.7 → llmshim-0.4.0}/src/proxy/handlers.rs +6 -6
- {llmshim-0.3.7 → llmshim-0.4.0}/src/proxy/mod.rs +4 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/proxy/ratelimit.rs +11 -3
- {llmshim-0.3.7 → llmshim-0.4.0}/src/proxy/types.rs +38 -2
- llmshim-0.4.0/src/proxy/wire/mod.rs +664 -0
- llmshim-0.4.0/src/proxy/wire/receipts.rs +71 -0
- llmshim-0.4.0/src/reasoning/normalize.rs +422 -0
- llmshim-0.4.0/src/reasoning.rs +617 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/router.rs +41 -5
- llmshim-0.4.0/src/schema/memo.rs +426 -0
- llmshim-0.4.0/src/schema/mod.rs +252 -0
- llmshim-0.4.0/src/schema/validate.rs +89 -0
- llmshim-0.4.0/src/schema/walk.rs +1000 -0
- llmshim-0.4.0/src/shim.rs +830 -0
- llmshim-0.4.0/src/streaming.rs +190 -0
- llmshim-0.4.0/src/toolcall/streaming.rs +594 -0
- llmshim-0.4.0/src/toolcall.rs +683 -0
- llmshim-0.4.0/src/usage.rs +112 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_current_models.rs +2 -2
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_openrouter.rs +1 -1
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_sglang.rs +1 -1
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_thinking.rs +4 -4
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_anthropic.rs +68 -32
- llmshim-0.4.0/tests/unit_cache.rs +244 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_chatgpt.rs +53 -7
- llmshim-0.4.0/tests/unit_cli.rs +66 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_fable.rs +4 -4
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_gemini.rs +75 -199
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_multimodel.rs +10 -4
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_openai.rs +42 -34
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_openai_compat.rs +7 -4
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_openrouter.rs +3 -3
- llmshim-0.4.0/tests/unit_provider_contracts.rs +219 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_proxy.rs +21 -1
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_proxy_convert.rs +21 -1
- llmshim-0.4.0/tests/unit_reasoning.rs +323 -0
- llmshim-0.4.0/tests/unit_reasoning_profile.rs +52 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_router.rs +27 -0
- llmshim-0.4.0/tests/unit_schema.rs +426 -0
- llmshim-0.4.0/tests/unit_shim.rs +547 -0
- llmshim-0.4.0/tests/unit_signature.rs +155 -0
- llmshim-0.4.0/tests/unit_toolcall.rs +536 -0
- llmshim-0.4.0/tests/unit_usage.rs +303 -0
- llmshim-0.4.0/tests/unit_wire.rs +555 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_xai.rs +51 -58
- llmshim-0.3.7/benchmarks/bench.rs +0 -178
- llmshim-0.3.7/src/provider.rs +0 -36
- llmshim-0.3.7/src/proxy/convert.rs +0 -177
- {llmshim-0.3.7 → llmshim-0.4.0}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/.github/workflows/pages.yml +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/.gitignore +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/CODE_OF_CONDUCT.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/CONTRIBUTING.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/LICENSE-APACHE +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/LICENSE-MIT +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/NOTICE +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/benchmarks/bench_python.py +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/benchmarks/gateway_loadtest.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/benchmarks/loadtest.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/.gitignore +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/book.toml +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/mermaid-init.js +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/mermaid.min.js +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/concepts/conversations.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/concepts/portability.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/concepts/routing.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/concepts/translation-flow.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/guides/fallbacks.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/guides/images.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/guides/native-controls.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/introduction.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/proxy/deployment.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/proxy/scaling.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/reference/api.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/reference/configuration.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/start/choose.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/start/cli.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/start/clients.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/start/configure.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/start/proxy.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/docs/src/start/rust.md +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/examples/chat.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/examples/stream.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/llmshim/__init__.py +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/llmshim/_server.py +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/pyproject.toml +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/config.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/env.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/gateway/auth.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/gateway/distributed.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/gateway/idempotency.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/gateway/metrics.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/gateway/mod.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/gateway/quota.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/src/vision.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/fixtures/chatgpt-red.png +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_chatgpt.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_chatgpt_proxy.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_fallback.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_gemini.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_gemini_tools.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_long_context.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_multimodel.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_proxy.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_tool_roundtrip.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_vision.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/integration_xai.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/support/completion_status.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_advertised_models.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_fallback.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_fast_mode.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_log.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_models.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_sse.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_tools.rs +0 -0
- {llmshim-0.3.7 → llmshim-0.4.0}/tests/unit_vision.rs +0 -0
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
name: Refresh model catalog snapshot
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
schedule:
|
|
5
|
+
- cron: '19 6 * * 1'
|
|
6
|
+
workflow_dispatch:
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: write
|
|
10
|
+
pull-requests: write
|
|
11
|
+
|
|
12
|
+
concurrency:
|
|
13
|
+
group: model-catalog-refresh
|
|
14
|
+
cancel-in-progress: false
|
|
15
|
+
|
|
16
|
+
jobs:
|
|
17
|
+
snapshot:
|
|
18
|
+
runs-on: ubuntu-latest
|
|
19
|
+
steps:
|
|
20
|
+
- uses: actions/checkout@v4
|
|
21
|
+
- uses: dtolnay/rust-toolchain@stable
|
|
22
|
+
- name: Download public catalog
|
|
23
|
+
run: |
|
|
24
|
+
curl --fail --silent --show-error --max-time 60 \
|
|
25
|
+
https://models.dev/api.json -o crates/llmshim-catalog/data/models.dev.json
|
|
26
|
+
- name: Validate catalog and refresh behavior
|
|
27
|
+
run: cargo test -p llmshim-catalog
|
|
28
|
+
- name: Propose snapshot update
|
|
29
|
+
uses: peter-evans/create-pull-request@v7
|
|
30
|
+
with:
|
|
31
|
+
branch: automation/model-catalog
|
|
32
|
+
add-paths: crates/llmshim-catalog/data/models.dev.json
|
|
33
|
+
commit-message: 'chore: refresh public model catalog snapshot'
|
|
34
|
+
title: 'Refresh public model catalog snapshot'
|
|
35
|
+
body: 'Updates the vendored models.dev snapshot. Verified builtin assertions and local overrides retain precedence. Catalog and refresh tests passed.'
|
|
@@ -16,11 +16,13 @@ jobs:
|
|
|
16
16
|
- uses: actions/checkout@v4
|
|
17
17
|
- uses: dtolnay/rust-toolchain@stable
|
|
18
18
|
- name: Check formatting
|
|
19
|
-
run: cargo fmt --check
|
|
19
|
+
run: cargo fmt --all --check
|
|
20
20
|
- name: Clippy
|
|
21
|
-
run: cargo clippy --features proxy -- -D warnings
|
|
21
|
+
run: cargo clippy --workspace --features proxy -- -D warnings
|
|
22
22
|
- name: Unit tests
|
|
23
|
-
|
|
23
|
+
env:
|
|
24
|
+
LLMSHIM_CATALOG_OFFLINE: '1'
|
|
25
|
+
run: cargo test --workspace --features proxy --tests
|
|
24
26
|
|
|
25
27
|
# --- Publish to crates.io ---
|
|
26
28
|
crates-io:
|
|
@@ -29,6 +31,17 @@ jobs:
|
|
|
29
31
|
steps:
|
|
30
32
|
- uses: actions/checkout@v4
|
|
31
33
|
- uses: dtolnay/rust-toolchain@stable
|
|
34
|
+
- name: Publish catalog dependency to crates.io
|
|
35
|
+
env:
|
|
36
|
+
CARGO_REGISTRY_TOKEN: ${{ secrets.CARGO_REGISTRY_TOKEN }}
|
|
37
|
+
run: |
|
|
38
|
+
set +e
|
|
39
|
+
out=$(cargo publish -p llmshim-catalog --allow-dirty 2>&1); code=$?
|
|
40
|
+
echo "$out"
|
|
41
|
+
[ $code -eq 0 ] && exit 0
|
|
42
|
+
echo "$out" | grep -qiE "already (been )?uploaded|already exists" \
|
|
43
|
+
&& { echo "Catalog version already on crates.io; skipping."; exit 0; }
|
|
44
|
+
exit $code
|
|
32
45
|
- name: Publish to crates.io
|
|
33
46
|
env:
|
|
34
47
|
CARGO_REGISTRY_TOKEN: ${{ secrets.CARGO_REGISTRY_TOKEN }}
|
|
@@ -36,7 +49,7 @@ jobs:
|
|
|
36
49
|
# release (e.g. to fill a failed downstream registry) is safe.
|
|
37
50
|
run: |
|
|
38
51
|
set +e
|
|
39
|
-
out=$(cargo publish --allow-dirty 2>&1); code=$?
|
|
52
|
+
out=$(cargo publish -p llmshim --allow-dirty 2>&1); code=$?
|
|
40
53
|
echo "$out"
|
|
41
54
|
[ $code -eq 0 ] && exit 0
|
|
42
55
|
echo "$out" | grep -qiE "already (been )?uploaded|already exists" \
|
|
@@ -37,6 +37,62 @@ API keys: `~/.llmshim/config.toml` (via `llmshim configure`) or env vars `OPENAI
|
|
|
37
37
|
|
|
38
38
|
## Architecture
|
|
39
39
|
|
|
40
|
+
### Catalog and usage normalization (local 0.4 development)
|
|
41
|
+
|
|
42
|
+
`llmshim-catalog` is a standalone workspace member. Its `builtin` module owns
|
|
43
|
+
the curated/historical constants; `src/models.rs` reexports the legacy borrowed
|
|
44
|
+
API. Owned metadata and snapshots live in `llmshim::catalog`. Keep the curated
|
|
45
|
+
15-route discovery and four-entry ChatGPT allowlist distinct from catalog
|
|
46
|
+
coverage. Local policy > provider capabilities > verified builtin assertions >
|
|
47
|
+
models.dev, per field; unknowns never erase assertions. Provider APIs never
|
|
48
|
+
contribute pricing. See `crates/llmshim-catalog/README.md` for cache and override
|
|
49
|
+
paths, offline behavior, aliases, and publication order. Never await catalog
|
|
50
|
+
refresh in a completion; hold a snapshot for decisions that must agree.
|
|
51
|
+
|
|
52
|
+
`usage.cache_read_tokens` and `usage.cache_write_tokens` are always present on
|
|
53
|
+
normalized responses and usage chunks, logs, and proxy usage. Native token
|
|
54
|
+
fields remain readable. `src/usage.rs` owns extraction; the streaming client
|
|
55
|
+
merges Anthropic's start/delta usage before normalizing terminal counts.
|
|
56
|
+
The native Chat Completions streams must use their own parser in the client;
|
|
57
|
+
passing them to the Responses parser silently drops all events.
|
|
58
|
+
|
|
59
|
+
Offline checks for this work:
|
|
60
|
+
|
|
61
|
+
```sh
|
|
62
|
+
LLMSHIM_CATALOG_OFFLINE=1 cargo test --workspace --features proxy --tests
|
|
63
|
+
cargo test -p llmshim-catalog
|
|
64
|
+
cargo clippy --workspace --features proxy -- -D warnings
|
|
65
|
+
cargo package -p llmshim-catalog --allow-dirty
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
The root package is 0.4.0 because log/proxy usage structs gain fields; no
|
|
69
|
+
release has been performed. Future release workflows must publish the catalog
|
|
70
|
+
dependency before llmshim. Public code, fixtures, artifacts, and docs must use
|
|
71
|
+
generic examples and contain no private consumer identities or context.
|
|
72
|
+
|
|
73
|
+
### Cache annotations and shared schema normalization
|
|
74
|
+
|
|
75
|
+
`src/cache.rs` translates caller `x-cache` segments into native Anthropic
|
|
76
|
+
breakpoints and an explicit Responses prompt_cache_key. It never infers
|
|
77
|
+
stability. Last eligible boundaries win within the four-slot budget; existing
|
|
78
|
+
explicit markers consume slots. Managed segments supersede automatic caching.
|
|
79
|
+
One-hour markers must precede five-minute markers. Marker scans inspect actual
|
|
80
|
+
cache locations, not arbitrary schema/default JSON. Without annotations, native
|
|
81
|
+
passthrough remains unchanged. `ProviderRequest::can_continue_from` checks
|
|
82
|
+
endpoint, headers, settings (including include/store/reasoning) and input prefix;
|
|
83
|
+
there is no stored/delta continuation engine. See the caching guide.
|
|
84
|
+
|
|
85
|
+
`src/schema/` owns the single schema walker and local resource resolver. Every
|
|
86
|
+
adapter normalizes native tool schemas after overrides. MCP inputSchema is
|
|
87
|
+
normalized on ingest and raw MCP tool definitions are accepted. Keep literal
|
|
88
|
+
values/property names separate from schema-node traversal, preserve meaningful
|
|
89
|
+
stripped constraints in descriptions, and never fetch an external reference.
|
|
90
|
+
Cycles, unresolved resources, incompatible residues and expansion-budget failures
|
|
91
|
+
fall back per tool. Only successful enforcement permits strict:true. Global
|
|
92
|
+
bypass flags: LLMSHIM_NO_SCHEMA_NORMALIZATION and LLMSHIM_NO_STRICT. OutputSchema
|
|
93
|
+
provides a reversible non-object wrapper; generated-instance validation and
|
|
94
|
+
repair remain a separate concern. See `docs/src/guides/schemas.md`.
|
|
95
|
+
|
|
40
96
|
### Curated discovery
|
|
41
97
|
|
|
42
98
|
`src/models.rs::MODELS` is the single advertised list, imported directly by
|
|
@@ -72,10 +128,11 @@ Preserve `chatgpt/<model>` in normalized responses and chunks: a bare GPT name
|
|
|
72
128
|
is otherwise misattributed to API-key OpenAI by the proxy/gateway when both
|
|
73
129
|
providers are registered. Astra preserves reasoning effort `max`; `none` and
|
|
74
130
|
`minimal` clamp to `low`.
|
|
75
|
-
For ChatGPT streaming,
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
131
|
+
For ChatGPT streaming, use the same `StreamNormalizer`/`ToolStream` as other
|
|
132
|
+
transports. Do not reintroduce a separate completed-call emitter: it used to
|
|
133
|
+
lose or duplicate argument fragments at the proxy boundary. Text and reasoning
|
|
134
|
+
remain incremental; callable tools are complete and emitted once at termination.
|
|
135
|
+
|
|
79
136
|
|
|
80
137
|
Offline coverage lives in `tests/unit_chatgpt.rs`. The live server check starts
|
|
81
138
|
its own loopback CLI process and stops it on completion/failure:
|
|
@@ -91,9 +148,9 @@ and image input. It uses the saved ChatGPT login and consumes
|
|
|
91
148
|
subscription usage; it is ignored during offline CI. Mount the whole token
|
|
92
149
|
directory writable for container use so refresh locks and atomic saves work.
|
|
93
150
|
|
|
94
|
-
**Self-hosted passthrough providers (vLLM / SGLang).** `src/providers/openai_compat.rs` is one generic OpenAI Chat Completions passthrough backing both `vllm` and `sglang` (`OpenAiCompatible::new(name, base_url, api_key: Option)`, registered per env base URL). Two things differ from the hosted providers: the **base URL is configuration** (local `http://localhost:8000/v1` vs remote `https://host/v1`), and **auth is optional** (self-hosted servers are unauthenticated unless launched with `--api-key`, so the `Authorization` header is sent only when a key is set). Passthrough transforms;
|
|
151
|
+
**Self-hosted passthrough providers (vLLM / SGLang).** `src/providers/openai_compat.rs` is one generic OpenAI Chat Completions passthrough backing both `vllm` and `sglang` (`OpenAiCompatible::new(name, base_url, api_key: Option)`, registered per env base URL). Two things differ from the hosted providers: the **base URL is configuration** (local `http://localhost:8000/v1` vs remote `https://host/v1`), and **auth is optional** (self-hosted servers are unauthenticated unless launched with `--api-key`, so the `Authorization` header is sent only when a key is set). Passthrough transforms; reasoning normalized to typed `reasoning[]` with provenance (vLLM is migrating the field name); `reasoning_effort` forwarded as-is (honored per-model, not clamped); server-specific params go under `x-<name>` (`chat_template_kwargs`, `separate_reasoning`, `guided_json`, `top_k`, …). Note: reasoning/tool parsing are **launch-time server flags** (`--reasoning-parser`, `--tool-call-parser`), so a request only gets that behavior if the server was started for it — llmshim can't enable it per request.
|
|
95
152
|
|
|
96
|
-
**OpenRouter is the one passthrough provider.** Every other provider translates the OpenAI-format input *away* to a native dialect; OpenRouter (`src/providers/openrouter.rs`) *is* OpenAI Chat Completions, so its transforms are near-identity — messages, tools, vision (`image_url`), and `response_format` are forwarded unchanged; `reasoning_effort` maps 1:1 to OpenRouter's `reasoning:{effort}` (its effort vocabulary is a superset, so no clamping);
|
|
153
|
+
**OpenRouter is the one passthrough provider.** Every other provider translates the OpenAI-format input *away* to a native dialect; OpenRouter (`src/providers/openrouter.rs`) *is* OpenAI Chat Completions, so its transforms are near-identity — messages, tools, vision (`image_url`), and `response_format` are forwarded unchanged; `reasoning_effort` maps 1:1 to OpenRouter's `reasoning:{effort}` (its effort vocabulary is a superset, so no clamping); reasoning is normalized to typed `reasoning[]` with provenance on responses. OpenRouter models are **not enumerated** in `src/models.rs` (the catalog is huge and dynamic) — any `openrouter/<vendor>/<model>` slug routes through. `x-openrouter` carries OpenRouter-only controls (`provider`, `models`, `transforms`, `route`, native `reasoning`; plus `http_referer`/`x_title` which become headers). The `middle-out` transform is disabled by default for faithful passthrough. Uses `image_url` (Chat Completions) vision via `vision::to_openai_chat`.
|
|
97
154
|
|
|
98
155
|
### Request flow
|
|
99
156
|
|
|
@@ -126,7 +183,7 @@ Automatic redirects are disabled on the shared client. Keep prompts and
|
|
|
126
183
|
provider-specific credential headers at the configured endpoint; 3xx responses
|
|
127
184
|
remain provider errors. Callers must configure the final URL directly.
|
|
128
185
|
|
|
129
|
-
`ShimClient` with shared connection pool (`LazyLock`), HTTP/2, gzip/brotli/zstd compression, TCP keepalive + nodelay. Automatic retry (3 attempts by default) on transport errors and 429/500/502/503/504/529 status codes. This is the **reactive** layer: on a retryable *response* it honors the server's `Retry-After` header (integer seconds or HTTP-date) and provider reset hints (OpenAI `x-ratelimit-reset-*`, Anthropic `anthropic-ratelimit-*-reset`), clamped to a cap and nudged with a little jitter; when there's no server hint (or a transport error) it falls back to full-jitter exponential backoff (uniform in `[0, min(cap, base·2^attempt)]`) to avoid a thundering herd. Tunable via `LLMSHIM_MAX_RETRIES` and `LLMSHIM_MAX_BACKOFF_SECS`. `warmup()` pre-establishes TCP+TLS connections. `SseStream`
|
|
186
|
+
`ShimClient` with shared connection pool (`LazyLock`), HTTP/2, gzip/brotli/zstd compression, TCP keepalive + nodelay. Automatic retry (3 attempts by default) on transport errors and 429/500/502/503/504/529 status codes. This is the **reactive** layer: on a retryable *response* it honors the server's `Retry-After` header (integer seconds or HTTP-date) and provider reset hints (OpenAI `x-ratelimit-reset-*`, Anthropic `anthropic-ratelimit-*-reset`), clamped to a cap and nudged with a little jitter; when there's no server hint (or a transport error) it falls back to full-jitter exponential backoff (uniform in `[0, min(cap, base·2^attempt)]`) to avoid a thundering herd. Tunable via `LLMSHIM_MAX_RETRIES` and `LLMSHIM_MAX_BACKOFF_SECS`. `warmup()` pre-establishes TCP+TLS connections. `SseStream` decodes bytes with eventsource-stream and feeds one per-response `StreamNormalizer`; do not restore lossy UTF-8 line buffering.
|
|
130
187
|
|
|
131
188
|
### Fallback chains (`src/fallback.rs`)
|
|
132
189
|
|
|
@@ -138,7 +195,23 @@ Image content blocks are translated between providers automatically. Users can s
|
|
|
138
195
|
|
|
139
196
|
### Multi-model conversations
|
|
140
197
|
|
|
141
|
-
|
|
198
|
+
`src/reasoning.rs` owns the single replay policy, typed blocks, signature origins,
|
|
199
|
+
issuer bindings, and drop counters. Every adapter filters before serialization
|
|
200
|
+
and captures from the original response to preserve ordered blocks and encrypted
|
|
201
|
+
Responses items. The shared HTTP client binds origins to the actual request
|
|
202
|
+
(including refreshed OAuth account headers). Unknown provenance/families fail
|
|
203
|
+
closed; matching family+wire permits replay, with same-account binding also
|
|
204
|
+
required for encrypted blocks. Never route by inspecting opaque bytes.
|
|
205
|
+
|
|
206
|
+
New outputs use `message.reasoning[]` and `thought_signature:{data,origin}`.
|
|
207
|
+
Legacy sibling readers exist for one migration release and drop untracked data.
|
|
208
|
+
Native overrides cannot bypass this filter or enable provider-side Responses
|
|
209
|
+
storage. The CLI and proxy retain the whole message; streaming consumers use
|
|
210
|
+
`ReasoningAccumulator` and must preserve signatures and completed item snapshots.
|
|
211
|
+
The shared SSE reader handles split UTF-8, CRLF, multiline events, late usage,
|
|
212
|
+
and terminal markers; do not restore the old per-byte lossy string buffer.
|
|
213
|
+
Tests: `tests/unit_reasoning.rs`, the client SSE tests, and provider regressions.
|
|
214
|
+
See `docs/src/guides/reasoning.md` for the full shape and migration contract.
|
|
142
215
|
|
|
143
216
|
### Provider extension namespaces (`x-anthropic`, `x-gemini`)
|
|
144
217
|
|
|
@@ -180,6 +253,30 @@ the verified xAI model ID. Keep the ChatGPT four-model allowlist independent.
|
|
|
180
253
|
|
|
181
254
|
Two knobs work across every provider: `reasoning_effort` (`none|low|medium|high|xhigh|max`) and `reasoning_mode` (`standard|pro`). A third, `reasoning_summary` (`auto|none`), controls reasoning-text visibility → Anthropic `thinking.display` (`auto`→`summarized`, the default when `reasoning_effort` is present so newer models like Sonnet 5 / Opus 4.7-4.8 return reasoning text instead of the API-default `omitted`; `none`→`omitted` for lower latency). Applies to both the adaptive and pre-4.6 enabled thinking builders; a caller-supplied `thinking` block bypasses it. Each provider transform maps them to its native dialect, **clamping to the nearest tier the target model accepts** (all boundaries verified live — e.g. `max` is native on OpenAI gpt-5.6 and GPT-6 Astra; Anthropic 4.6 rejects `xhigh` but has `max`; Gemini's enum tops out at `high`; xAI grok-4.20 models reject any reasoning param). `mode: "pro"` is native on OpenAI gpt-5.6/-pro models (`reasoning.mode`), emulated as a one-tier effort bump elsewhere; explicit `none` always wins. Native passthrough (`x-openai.reasoning`, `x-anthropic.thinking`, `x-gemini.thinkingConfig`) bypasses the mapping entirely and always takes precedence. **Full per-provider mapping tables: `docs/src/guides/reasoning.md`** — update it and the pinning tests in `tests/unit_*.rs` together whenever a mapping changes.
|
|
182
255
|
|
|
256
|
+
### Owned tool identities and stream state
|
|
257
|
+
|
|
258
|
+
`src/toolcall.rs` owns `WireToolId`, the bidirectional map, request projection,
|
|
259
|
+
and central call/result validation. Canonical ids are minted as `call_ls_*` and
|
|
260
|
+
never use a provider id directly. `wire_ids` must remain on persisted calls;
|
|
261
|
+
Responses `item_id` is distinct from correlation `id`. Source wire ids (including
|
|
262
|
+
Gemini's missing id) are restored on both calls/results, preserving signed
|
|
263
|
+
prefixes. Legacy input ids remain readable, but an owned id without its mapping
|
|
264
|
+
is an error. Never drop invalid tool history to make a request succeed.
|
|
265
|
+
|
|
266
|
+
`src/toolcall/streaming.rs` parses native events into `ToolDelta` and assembles
|
|
267
|
+
JSON arguments once. `src/streaming.rs::StreamNormalizer` is the public stateful
|
|
268
|
+
entry point used by HTTP streaming and manual SSE readers. Stateless provider
|
|
269
|
+
chunk methods no longer expose partial callable records. Tool signatures can
|
|
270
|
+
arrive after names/arguments; preserve their exact data/origin and original wire
|
|
271
|
+
container. Parallel choices close independently. The single-message proxy
|
|
272
|
+
projects choice zero; Rust retains choice indices.
|
|
273
|
+
|
|
274
|
+
Google's current Generate Content contract requires a signature on the first
|
|
275
|
+
function call of each current Gemini 3 batch, not every parallel call. Validate
|
|
276
|
+
that rule after filtering, preserve additional signatures exactly where present,
|
|
277
|
+
and retain optional native function ids on both sides. `unit_toolcall` covers
|
|
278
|
+
paired replay, order, signatures, invalid history, and transport delta sequences.
|
|
279
|
+
|
|
183
280
|
### Tool format translation
|
|
184
281
|
|
|
185
282
|
llmshim accepts tools in OpenAI Chat Completions format (nested `function` object) and translates them to each provider's native format:
|
|
@@ -292,3 +389,96 @@ Common maintenance workflows are packaged as [skills](https://code.claude.com/do
|
|
|
292
389
|
- `/add-provider key Name` — wire up a brand-new upstream provider.
|
|
293
390
|
- `/preflight` — run the fmt + clippy + test trio CI enforces.
|
|
294
391
|
- `/release 0.1.22` — bump version and tag so CI publishes.
|
|
392
|
+
|
|
393
|
+
### Capability plans and instance validation
|
|
394
|
+
|
|
395
|
+
`src/shim.rs::Plan` owns catalog-driven structured output, prompt tool calling,
|
|
396
|
+
and optional brief-rationale capture. `ShimClient` applies the same plan on
|
|
397
|
+
completion, stream, and fallback paths; direct provider transforms only support
|
|
398
|
+
native response-format translation. Never emit hidden synthetic calls, native
|
|
399
|
+
deliberation about those calls, or invalid attempts to logs/streams. Managed
|
|
400
|
+
streams buffer with a 32 MiB bound; top-level streaming uses an Arc-owned provider
|
|
401
|
+
so HTTP headers/keepalives can proceed while output is validated.
|
|
402
|
+
|
|
403
|
+
Validate generated data against the ORIGINAL schema with the network/filesystem
|
|
404
|
+
retriever disabled (`schema::validate`). Schema compile budgets are 96 levels,
|
|
405
|
+
32,768 JSON values and 8 MiB strings/keys. Only complete invalid answers receive
|
|
406
|
+
one repair; refusals and incomplete responses do not. Preserve both attempts'
|
|
407
|
+
reported usage. Unknown catalog fields remain unknown; `forced_tool_choice` is
|
|
408
|
+
independent of tools, with the verified Fable 5.1 prohibition overriding auto.
|
|
409
|
+
Keep synthetic schemas/instructions deterministic so unchanged requests preserve
|
|
410
|
+
prefix caching. Tests: `unit_shim`, proxy conversion tests, client request mocks.
|
|
411
|
+
|
|
412
|
+
### Signature observations and reasoning profiles
|
|
413
|
+
|
|
414
|
+
`providers/anthropic_signature.rs` has the default-on `signature-introspection`
|
|
415
|
+
feature and Option-only stub. Never use decoded metadata to change provenance,
|
|
416
|
+
messages, replay, routing, fallback, or caches. Synthetic fixtures only. Capture
|
|
417
|
+
one metric observation per response/terminal stream; logs read the observation
|
|
418
|
+
without incrementing it. `x-llmshim-served-model` belongs at response/event level.
|
|
419
|
+
|
|
420
|
+
`providers/anthropic_reasoning.rs::Profile` consumes catalog effort/budget unions
|
|
421
|
+
once per provider transform. Builtin verified options win over community data;
|
|
422
|
+
local/provider overrides retain documented precedence. Keep historical fallback
|
|
423
|
+
entries explicit in `catalog::builtin::anthropic_reasoning_options`, rather than
|
|
424
|
+
guessing support for new model names. Preserve mandatory native model constraints
|
|
425
|
+
(e.g. Fable adaptive-only and forbidden forced choices) separately.
|
|
426
|
+
|
|
427
|
+
### Native inbound facades
|
|
428
|
+
|
|
429
|
+
`proxy::wire` translates `/v1/messages` and `/v1/chat/completions` through the
|
|
430
|
+
existing chat handlers on both proxy and gateway. Do not create another dispatch,
|
|
431
|
+
auth, quota, or queue path. Gateway authenticates before receipt lookup and again
|
|
432
|
+
in its ordinary admission path; x-api-key maps to Bearer only when Authorization
|
|
433
|
+
is absent. Native idempotency keys include credential and protocol scope.
|
|
434
|
+
|
|
435
|
+
The native facade persists issued replay metadata in private atomic local files,
|
|
436
|
+
configured by `LLMSHIM_REPLAY_RECEIPTS_DIR`. Clients must preserve the native
|
|
437
|
+
message/ID and server receipts across restarts. Never stamp unknown native thinking
|
|
438
|
+
with current-target provenance. Receipts restore original blocks, then the common
|
|
439
|
+
replay filter decides eligibility. Missing owned IDs and edited calls error.
|
|
440
|
+
Text remains incremental; complete reasoning/tool blocks follow once metadata is
|
|
441
|
+
ready. Status/Retry-After and gateway request IDs survive error translation.
|
|
442
|
+
`src/error/normalize.rs` owns error unwrapping for compact JSON/SSE, gateway,
|
|
443
|
+
and native endpoints. Keep display messages readable and source type/code/param
|
|
444
|
+
metadata separate; do not move this logic back into a wire-only formatter.
|
|
445
|
+
JSON responses carry native metadata in response extensions, and SSE errors carry
|
|
446
|
+
an optional structured error object so native rendering remains lossless.
|
|
447
|
+
Tests: `unit_wire`, gateway `http::native_tests`. Use a temporary receipt directory
|
|
448
|
+
in tests; never put real signatures, credentials, or conversations in fixtures.
|
|
449
|
+
|
|
450
|
+
### Provider and CLI regression gates
|
|
451
|
+
|
|
452
|
+
`tests/unit_provider_contracts.rs` discovers Provider implementations from the
|
|
453
|
+
Rust AST and requires a fixture for each. New adapters must preserve compatible
|
|
454
|
+
reasoning, drop foreign/untracked blocks and signatures, and reject invalid tool
|
|
455
|
+
history before serialization. Tool-call containers accept arrays or null only.
|
|
456
|
+
HTTP handlers validate canonical history before committing to an SSE response.
|
|
457
|
+
|
|
458
|
+
`src/cli.rs` validates arguments before side effects. Server options override env
|
|
459
|
+
and saved host/port; help must not start a service. Bind errors are returned to
|
|
460
|
+
main for a clean exit. Tests use ephemeral loopback ports and never touch an
|
|
461
|
+
existing server. Run unit_cli, unit_provider_contracts, unit_wire and gateway
|
|
462
|
+
HTTP tests when changing these boundaries.
|
|
463
|
+
|
|
464
|
+
### Schema memoization and benchmark receipts
|
|
465
|
+
|
|
466
|
+
`src/schema/memo.rs` caches only schema inputs/results and `Normalization` reports.
|
|
467
|
+
The process-local cache is bounded to 512 entries/16 MiB estimated owned bytes,
|
|
468
|
+
uses LRU eviction, verifies input/options after a hash hit, and performs expensive
|
|
469
|
+
normalization outside the lock. Key all effective Options, including budgets.
|
|
470
|
+
Resolve environment overrides before lookup. Preserve object order and signed
|
|
471
|
+
zero: semantically equal numbers can produce different spilled descriptions.
|
|
472
|
+
No-op hits can retain the input value only after exact input/output comparison.
|
|
473
|
+
|
|
474
|
+
`LLMSHIM_NO_SCHEMA_CACHE=1` bypasses memoization while keeping normalization active.
|
|
475
|
+
Tests cover collisions, eviction, concurrency, returned-value independence,
|
|
476
|
+
report/fallback fidelity, environment changes and fresh tool metadata.
|
|
477
|
+
|
|
478
|
+
`cargo run --release --example bench` loads the two configured native API keys and
|
|
479
|
+
makes 43 logical requests plus client retries. Missing keys fail before network
|
|
480
|
+
calls with setup instructions. Provider errors must never print credentials.
|
|
481
|
+
`--transforms-only` needs no keys or HTTP calls; use it for cache-on/off comparison.
|
|
482
|
+
Keep README measurements dated, name the actual models/host/build/workload, and
|
|
483
|
+
separate API latency, full-transform CPU time, and RSS. Do not mix fresh Rust
|
|
484
|
+
figures with old Python results or claim unmeasured network-only latency shares.
|