llmshim 0.3.7__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (194) hide show
  1. llmshim-0.5.0/.github/workflows/catalog-refresh.yml +35 -0
  2. {llmshim-0.3.7 → llmshim-0.5.0}/.github/workflows/release.yml +17 -4
  3. {llmshim-0.3.7 → llmshim-0.5.0}/CLAUDE.md +282 -9
  4. {llmshim-0.3.7 → llmshim-0.5.0}/Cargo.lock +438 -18
  5. {llmshim-0.3.7 → llmshim-0.5.0}/Cargo.toml +13 -2
  6. {llmshim-0.3.7 → llmshim-0.5.0}/PKG-INFO +1 -1
  7. {llmshim-0.3.7 → llmshim-0.5.0}/README.md +48 -16
  8. {llmshim-0.3.7 → llmshim-0.5.0}/SECURITY.md +17 -0
  9. llmshim-0.5.0/benchmarks/bench.rs +241 -0
  10. llmshim-0.5.0/crates/llmshim-catalog/Cargo.toml +28 -0
  11. llmshim-0.5.0/crates/llmshim-catalog/LICENSE-APACHE +202 -0
  12. llmshim-0.5.0/crates/llmshim-catalog/LICENSE-MIT +21 -0
  13. llmshim-0.5.0/crates/llmshim-catalog/README.md +118 -0
  14. llmshim-0.5.0/crates/llmshim-catalog/data/LICENSE.models.dev +21 -0
  15. llmshim-0.5.0/crates/llmshim-catalog/data/README.md +8 -0
  16. llmshim-0.5.0/crates/llmshim-catalog/data/models.dev.json +1 -0
  17. llmshim-0.5.0/crates/llmshim-catalog/src/aliases.rs +35 -0
  18. llmshim-0.3.7/src/models.rs → llmshim-0.5.0/crates/llmshim-catalog/src/builtin.rs +119 -137
  19. llmshim-0.5.0/crates/llmshim-catalog/src/capabilities.rs +105 -0
  20. llmshim-0.5.0/crates/llmshim-catalog/src/lib.rs +274 -0
  21. llmshim-0.5.0/crates/llmshim-catalog/src/merge.rs +110 -0
  22. llmshim-0.5.0/crates/llmshim-catalog/src/parse.rs +124 -0
  23. llmshim-0.5.0/crates/llmshim-catalog/src/refresh.rs +356 -0
  24. llmshim-0.5.0/crates/llmshim-catalog/src/types.rs +161 -0
  25. llmshim-0.5.0/crates/llmshim-catalog/tests/catalog.rs +241 -0
  26. llmshim-0.5.0/crates/llmshim-catalog/tests/refresh.rs +232 -0
  27. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/SUMMARY.md +4 -1
  28. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/concepts/contracts.md +1 -1
  29. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/concepts/routing.md +47 -1
  30. llmshim-0.5.0/docs/src/guides/caching.md +66 -0
  31. llmshim-0.5.0/docs/src/guides/capabilities.md +91 -0
  32. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/guides/fallbacks.md +13 -1
  33. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/guides/reasoning.md +77 -4
  34. llmshim-0.5.0/docs/src/guides/schemas.md +91 -0
  35. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/guides/streaming.md +13 -10
  36. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/guides/tools.md +44 -5
  37. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/proxy/http-api.md +22 -4
  38. llmshim-0.5.0/docs/src/proxy/native-apis.md +85 -0
  39. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/proxy/scaling.md +40 -2
  40. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/cli.md +13 -5
  41. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/errors.md +16 -2
  42. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/models.md +23 -0
  43. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/providers.md +5 -5
  44. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/request-fields.md +19 -14
  45. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/surfaces.md +8 -6
  46. {llmshim-0.3.7 → llmshim-0.5.0}/llmshim/_client.py +21 -0
  47. {llmshim-0.3.7 → llmshim-0.5.0}/llmshim/types.py +108 -7
  48. llmshim-0.5.0/src/breaker.rs +454 -0
  49. llmshim-0.5.0/src/cache.rs +269 -0
  50. llmshim-0.5.0/src/cli.rs +221 -0
  51. {llmshim-0.3.7 → llmshim-0.5.0}/src/client.rs +275 -90
  52. {llmshim-0.3.7 → llmshim-0.5.0}/src/config.rs +40 -1
  53. llmshim-0.5.0/src/cost.rs +272 -0
  54. llmshim-0.5.0/src/error/normalize.rs +143 -0
  55. {llmshim-0.3.7 → llmshim-0.5.0}/src/error.rs +5 -0
  56. {llmshim-0.3.7 → llmshim-0.5.0}/src/fallback.rs +39 -55
  57. {llmshim-0.3.7 → llmshim-0.5.0}/src/gateway/auth.rs +17 -0
  58. {llmshim-0.3.7 → llmshim-0.5.0}/src/gateway/distributed.rs +29 -0
  59. {llmshim-0.3.7 → llmshim-0.5.0}/src/gateway/http.rs +208 -15
  60. llmshim-0.5.0/src/gateway/quota.rs +354 -0
  61. {llmshim-0.3.7 → llmshim-0.5.0}/src/lib.rs +37 -3
  62. {llmshim-0.3.7 → llmshim-0.5.0}/src/log.rs +24 -0
  63. {llmshim-0.3.7 → llmshim-0.5.0}/src/main.rs +100 -35
  64. llmshim-0.5.0/src/models.rs +7 -0
  65. llmshim-0.5.0/src/provider.rs +72 -0
  66. {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/anthropic.rs +83 -248
  67. llmshim-0.5.0/src/providers/anthropic_reasoning.rs +111 -0
  68. llmshim-0.5.0/src/providers/anthropic_signature.rs +165 -0
  69. {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/chatgpt/auth.rs +4 -0
  70. {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/chatgpt/mod.rs +66 -4
  71. {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/chatgpt/streaming.rs +1 -78
  72. {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/gemini.rs +100 -299
  73. {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/mod.rs +2 -0
  74. {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/openai.rs +232 -245
  75. {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/openai_compat.rs +58 -38
  76. {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/openrouter.rs +58 -39
  77. {llmshim-0.3.7 → llmshim-0.5.0}/src/providers/xai.rs +80 -87
  78. llmshim-0.5.0/src/proxy/convert.rs +328 -0
  79. {llmshim-0.3.7 → llmshim-0.5.0}/src/proxy/error.rs +55 -6
  80. {llmshim-0.3.7 → llmshim-0.5.0}/src/proxy/handlers.rs +10 -21
  81. llmshim-0.5.0/src/proxy/health.rs +208 -0
  82. {llmshim-0.3.7 → llmshim-0.5.0}/src/proxy/mod.rs +8 -0
  83. {llmshim-0.3.7 → llmshim-0.5.0}/src/proxy/ratelimit.rs +11 -3
  84. {llmshim-0.3.7 → llmshim-0.5.0}/src/proxy/types.rs +67 -2
  85. llmshim-0.5.0/src/proxy/wire/mod.rs +686 -0
  86. llmshim-0.5.0/src/proxy/wire/receipts.rs +71 -0
  87. llmshim-0.5.0/src/reasoning/normalize.rs +422 -0
  88. llmshim-0.5.0/src/reasoning.rs +617 -0
  89. llmshim-0.5.0/src/router.rs +275 -0
  90. llmshim-0.5.0/src/schema/memo.rs +426 -0
  91. llmshim-0.5.0/src/schema/mod.rs +252 -0
  92. llmshim-0.5.0/src/schema/validate.rs +89 -0
  93. llmshim-0.5.0/src/schema/walk.rs +1000 -0
  94. llmshim-0.5.0/src/shim.rs +833 -0
  95. llmshim-0.5.0/src/streaming.rs +190 -0
  96. llmshim-0.5.0/src/toolcall/streaming.rs +594 -0
  97. llmshim-0.5.0/src/toolcall.rs +683 -0
  98. llmshim-0.5.0/src/usage.rs +191 -0
  99. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_current_models.rs +2 -2
  100. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_openrouter.rs +1 -1
  101. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_sglang.rs +1 -1
  102. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_thinking.rs +4 -4
  103. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_anthropic.rs +68 -32
  104. llmshim-0.5.0/tests/unit_cache.rs +244 -0
  105. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_chatgpt.rs +53 -7
  106. llmshim-0.5.0/tests/unit_cli.rs +66 -0
  107. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_fable.rs +4 -4
  108. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_fallback.rs +80 -0
  109. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_gemini.rs +75 -199
  110. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_multimodel.rs +10 -4
  111. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_openai.rs +42 -34
  112. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_openai_compat.rs +7 -4
  113. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_openrouter.rs +3 -3
  114. llmshim-0.5.0/tests/unit_provider_contracts.rs +219 -0
  115. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_proxy.rs +81 -1
  116. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_proxy_convert.rs +24 -1
  117. llmshim-0.5.0/tests/unit_reasoning.rs +323 -0
  118. llmshim-0.5.0/tests/unit_reasoning_profile.rs +52 -0
  119. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_router.rs +142 -0
  120. llmshim-0.5.0/tests/unit_schema.rs +426 -0
  121. llmshim-0.5.0/tests/unit_shim.rs +547 -0
  122. llmshim-0.5.0/tests/unit_signature.rs +155 -0
  123. llmshim-0.5.0/tests/unit_toolcall.rs +536 -0
  124. llmshim-0.5.0/tests/unit_usage.rs +303 -0
  125. llmshim-0.5.0/tests/unit_wire.rs +606 -0
  126. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_xai.rs +51 -58
  127. llmshim-0.3.7/benchmarks/bench.rs +0 -178
  128. llmshim-0.3.7/src/gateway/quota.rs +0 -147
  129. llmshim-0.3.7/src/provider.rs +0 -36
  130. llmshim-0.3.7/src/proxy/convert.rs +0 -177
  131. llmshim-0.3.7/src/router.rs +0 -133
  132. {llmshim-0.3.7 → llmshim-0.5.0}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
  133. {llmshim-0.3.7 → llmshim-0.5.0}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
  134. {llmshim-0.3.7 → llmshim-0.5.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  135. {llmshim-0.3.7 → llmshim-0.5.0}/.github/workflows/pages.yml +0 -0
  136. {llmshim-0.3.7 → llmshim-0.5.0}/.gitignore +0 -0
  137. {llmshim-0.3.7 → llmshim-0.5.0}/CODE_OF_CONDUCT.md +0 -0
  138. {llmshim-0.3.7 → llmshim-0.5.0}/CONTRIBUTING.md +0 -0
  139. {llmshim-0.3.7 → llmshim-0.5.0}/LICENSE-APACHE +0 -0
  140. {llmshim-0.3.7 → llmshim-0.5.0}/LICENSE-MIT +0 -0
  141. {llmshim-0.3.7 → llmshim-0.5.0}/NOTICE +0 -0
  142. {llmshim-0.3.7 → llmshim-0.5.0}/benchmarks/bench_python.py +0 -0
  143. {llmshim-0.3.7 → llmshim-0.5.0}/benchmarks/gateway_loadtest.rs +0 -0
  144. {llmshim-0.3.7 → llmshim-0.5.0}/benchmarks/loadtest.rs +0 -0
  145. {llmshim-0.3.7 → llmshim-0.5.0}/docs/.gitignore +0 -0
  146. {llmshim-0.3.7 → llmshim-0.5.0}/docs/book.toml +0 -0
  147. {llmshim-0.3.7 → llmshim-0.5.0}/docs/mermaid-init.js +0 -0
  148. {llmshim-0.3.7 → llmshim-0.5.0}/docs/mermaid.min.js +0 -0
  149. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/concepts/conversations.md +0 -0
  150. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/concepts/portability.md +0 -0
  151. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/concepts/translation-flow.md +0 -0
  152. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/guides/images.md +0 -0
  153. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/guides/native-controls.md +0 -0
  154. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/introduction.md +0 -0
  155. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/proxy/deployment.md +0 -0
  156. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/api.md +0 -0
  157. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/reference/configuration.md +0 -0
  158. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/start/choose.md +0 -0
  159. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/start/cli.md +0 -0
  160. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/start/clients.md +0 -0
  161. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/start/configure.md +0 -0
  162. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/start/proxy.md +0 -0
  163. {llmshim-0.3.7 → llmshim-0.5.0}/docs/src/start/rust.md +0 -0
  164. {llmshim-0.3.7 → llmshim-0.5.0}/examples/chat.rs +0 -0
  165. {llmshim-0.3.7 → llmshim-0.5.0}/examples/stream.rs +0 -0
  166. {llmshim-0.3.7 → llmshim-0.5.0}/llmshim/__init__.py +0 -0
  167. {llmshim-0.3.7 → llmshim-0.5.0}/llmshim/_server.py +0 -0
  168. {llmshim-0.3.7 → llmshim-0.5.0}/pyproject.toml +0 -0
  169. {llmshim-0.3.7 → llmshim-0.5.0}/src/env.rs +0 -0
  170. {llmshim-0.3.7 → llmshim-0.5.0}/src/gateway/idempotency.rs +0 -0
  171. {llmshim-0.3.7 → llmshim-0.5.0}/src/gateway/metrics.rs +0 -0
  172. {llmshim-0.3.7 → llmshim-0.5.0}/src/gateway/mod.rs +0 -0
  173. {llmshim-0.3.7 → llmshim-0.5.0}/src/vision.rs +0 -0
  174. {llmshim-0.3.7 → llmshim-0.5.0}/tests/fixtures/chatgpt-red.png +0 -0
  175. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration.rs +0 -0
  176. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_chatgpt.rs +0 -0
  177. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_chatgpt_proxy.rs +0 -0
  178. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_fallback.rs +0 -0
  179. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_gemini.rs +0 -0
  180. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_gemini_tools.rs +0 -0
  181. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_long_context.rs +0 -0
  182. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_multimodel.rs +0 -0
  183. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_proxy.rs +0 -0
  184. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_tool_roundtrip.rs +0 -0
  185. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_vision.rs +0 -0
  186. {llmshim-0.3.7 → llmshim-0.5.0}/tests/integration_xai.rs +0 -0
  187. {llmshim-0.3.7 → llmshim-0.5.0}/tests/support/completion_status.rs +0 -0
  188. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_advertised_models.rs +0 -0
  189. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_fast_mode.rs +0 -0
  190. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_log.rs +0 -0
  191. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_models.rs +0 -0
  192. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_sse.rs +0 -0
  193. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_tools.rs +0 -0
  194. {llmshim-0.3.7 → llmshim-0.5.0}/tests/unit_vision.rs +0 -0
@@ -0,0 +1,35 @@
1
+ name: Refresh model catalog snapshot
2
+
3
+ on:
4
+ schedule:
5
+ - cron: '19 6 * * 1'
6
+ workflow_dispatch:
7
+
8
+ permissions:
9
+ contents: write
10
+ pull-requests: write
11
+
12
+ concurrency:
13
+ group: model-catalog-refresh
14
+ cancel-in-progress: false
15
+
16
+ jobs:
17
+ snapshot:
18
+ runs-on: ubuntu-latest
19
+ steps:
20
+ - uses: actions/checkout@v4
21
+ - uses: dtolnay/rust-toolchain@stable
22
+ - name: Download public catalog
23
+ run: |
24
+ curl --fail --silent --show-error --max-time 60 \
25
+ https://models.dev/api.json -o crates/llmshim-catalog/data/models.dev.json
26
+ - name: Validate catalog and refresh behavior
27
+ run: cargo test -p llmshim-catalog
28
+ - name: Propose snapshot update
29
+ uses: peter-evans/create-pull-request@v7
30
+ with:
31
+ branch: automation/model-catalog
32
+ add-paths: crates/llmshim-catalog/data/models.dev.json
33
+ commit-message: 'chore: refresh public model catalog snapshot'
34
+ title: 'Refresh public model catalog snapshot'
35
+ body: 'Updates the vendored models.dev snapshot. Verified builtin assertions and local overrides retain precedence. Catalog and refresh tests passed.'
@@ -16,11 +16,13 @@ jobs:
16
16
  - uses: actions/checkout@v4
17
17
  - uses: dtolnay/rust-toolchain@stable
18
18
  - name: Check formatting
19
- run: cargo fmt --check
19
+ run: cargo fmt --all --check
20
20
  - name: Clippy
21
- run: cargo clippy --features proxy -- -D warnings
21
+ run: cargo clippy --workspace --features proxy -- -D warnings
22
22
  - name: Unit tests
23
- run: cargo test --features proxy --tests
23
+ env:
24
+ LLMSHIM_CATALOG_OFFLINE: '1'
25
+ run: cargo test --workspace --features proxy --tests
24
26
 
25
27
  # --- Publish to crates.io ---
26
28
  crates-io:
@@ -29,6 +31,17 @@ jobs:
29
31
  steps:
30
32
  - uses: actions/checkout@v4
31
33
  - uses: dtolnay/rust-toolchain@stable
34
+ - name: Publish catalog dependency to crates.io
35
+ env:
36
+ CARGO_REGISTRY_TOKEN: ${{ secrets.CARGO_REGISTRY_TOKEN }}
37
+ run: |
38
+ set +e
39
+ out=$(cargo publish -p llmshim-catalog --allow-dirty 2>&1); code=$?
40
+ echo "$out"
41
+ [ $code -eq 0 ] && exit 0
42
+ echo "$out" | grep -qiE "already (been )?uploaded|already exists" \
43
+ && { echo "Catalog version already on crates.io; skipping."; exit 0; }
44
+ exit $code
32
45
  - name: Publish to crates.io
33
46
  env:
34
47
  CARGO_REGISTRY_TOKEN: ${{ secrets.CARGO_REGISTRY_TOKEN }}
@@ -36,7 +49,7 @@ jobs:
36
49
  # release (e.g. to fill a failed downstream registry) is safe.
37
50
  run: |
38
51
  set +e
39
- out=$(cargo publish --allow-dirty 2>&1); code=$?
52
+ out=$(cargo publish -p llmshim --allow-dirty 2>&1); code=$?
40
53
  echo "$out"
41
54
  [ $code -eq 0 ] && exit 0
42
55
  echo "$out" | grep -qiE "already (been )?uploaded|already exists" \
@@ -37,6 +37,91 @@ API keys: `~/.llmshim/config.toml` (via `llmshim configure`) or env vars `OPENAI
37
37
 
38
38
  ## Architecture
39
39
 
40
+ ### Catalog and usage normalization (local 0.4 development)
41
+
42
+ `llmshim-catalog` is a standalone workspace member. Its `builtin` module owns
43
+ the curated/historical constants; `src/models.rs` reexports the legacy borrowed
44
+ API. Owned metadata and snapshots live in `llmshim::catalog`. Keep the curated
45
+ 15-route discovery and four-entry ChatGPT allowlist distinct from catalog
46
+ coverage. Local policy > provider capabilities > verified builtin assertions >
47
+ models.dev, per field; unknowns never erase assertions. Provider APIs never
48
+ contribute pricing. See `crates/llmshim-catalog/README.md` for cache and override
49
+ paths, offline behavior, aliases, and publication order. Never await catalog
50
+ refresh in a completion; hold a snapshot for decisions that must agree.
51
+
52
+ `usage.cache_read_tokens` and `usage.cache_write_tokens` are always present on
53
+ normalized responses and usage chunks, logs, and proxy usage. Native token
54
+ fields remain readable. `src/usage.rs` owns extraction; the streaming client
55
+ merges Anthropic's start/delta usage before normalizing terminal counts.
56
+ `usage.uncached_input_tokens` joins them: the providers disagree on whether
57
+ their prompt total already includes the cache read (Anthropic's excludes it,
58
+ OpenAI/Chat Completions/Gemini include it), and only the transport boundary
59
+ still knows which convention the body used. The convention is read off the
60
+ cache-read field that actually matched, never a provider-name table.
61
+
62
+ ### USD cost accounting
63
+
64
+ `src/cost.rs` is the only thing that multiplies a catalog `Cost` (USD per
65
+ million tokens) by those counters. Input is charged on `uncached_input_tokens`
66
+ so a cached prompt is never billed twice; cache reads and writes are charged at
67
+ their own rates; reasoning tokens are already inside `completion_tokens`.
68
+ **An absent price is `None`, never `0.0`** — a positive count in a bucket with
69
+ no rate poisons the whole total rather than producing a partial sum that reads
70
+ as a complete one. `client.rs` stamps `usage.cost_usd` at the transport
71
+ boundary (the last place that knows the dispatch target) for completions and
72
+ for whichever stream chunk carries usage; `log.rs`, `proxy::types::Usage`, both
73
+ native facades and the four bundled clients carry it through as a nullable
74
+ field. `null` means unknown, not free.
75
+
76
+ `src/gateway/quota.rs` adds a per-identity dollar cap beside the RPM/TPM
77
+ buckets: `budget_usd` + `budget_window_secs` on an `Identity`, checked before
78
+ dispatch and charged after (cost is only knowable once a response exists, so
79
+ one in-flight request can overshoot). Windows tumble rather than slide, because
80
+ the fleet-wide store is one counter per window. `SpendCap::with_store` takes the
81
+ Redis-backed `DistributedGateway` in distributed mode so `$100/day` means one
82
+ hundred dollars fleet-wide, not per replica. **A response the catalog cannot
83
+ price is not charged** — recording zero would let an unpriced model run forever
84
+ under a budget; `cost_usd: null` is the signal that a price is missing.
85
+ The native Chat Completions streams must use their own parser in the client;
86
+ passing them to the Responses parser silently drops all events.
87
+
88
+ Offline checks for this work:
89
+
90
+ ```sh
91
+ LLMSHIM_CATALOG_OFFLINE=1 cargo test --workspace --features proxy --tests
92
+ cargo test -p llmshim-catalog
93
+ cargo clippy --workspace --features proxy -- -D warnings
94
+ cargo package -p llmshim-catalog --allow-dirty
95
+ ```
96
+
97
+ The root package is 0.4.0 because log/proxy usage structs gain fields; no
98
+ release has been performed. Future release workflows must publish the catalog
99
+ dependency before llmshim. Public code, fixtures, artifacts, and docs must use
100
+ generic examples and contain no private consumer identities or context.
101
+
102
+ ### Cache annotations and shared schema normalization
103
+
104
+ `src/cache.rs` translates caller `x-cache` segments into native Anthropic
105
+ breakpoints and an explicit Responses prompt_cache_key. It never infers
106
+ stability. Last eligible boundaries win within the four-slot budget; existing
107
+ explicit markers consume slots. Managed segments supersede automatic caching.
108
+ One-hour markers must precede five-minute markers. Marker scans inspect actual
109
+ cache locations, not arbitrary schema/default JSON. Without annotations, native
110
+ passthrough remains unchanged. `ProviderRequest::can_continue_from` checks
111
+ endpoint, headers, settings (including include/store/reasoning) and input prefix;
112
+ there is no stored/delta continuation engine. See the caching guide.
113
+
114
+ `src/schema/` owns the single schema walker and local resource resolver. Every
115
+ adapter normalizes native tool schemas after overrides. MCP inputSchema is
116
+ normalized on ingest and raw MCP tool definitions are accepted. Keep literal
117
+ values/property names separate from schema-node traversal, preserve meaningful
118
+ stripped constraints in descriptions, and never fetch an external reference.
119
+ Cycles, unresolved resources, incompatible residues and expansion-budget failures
120
+ fall back per tool. Only successful enforcement permits strict:true. Global
121
+ bypass flags: LLMSHIM_NO_SCHEMA_NORMALIZATION and LLMSHIM_NO_STRICT. OutputSchema
122
+ provides a reversible non-object wrapper; generated-instance validation and
123
+ repair remain a separate concern. See `docs/src/guides/schemas.md`.
124
+
40
125
  ### Curated discovery
41
126
 
42
127
  `src/models.rs::MODELS` is the single advertised list, imported directly by
@@ -72,10 +157,11 @@ Preserve `chatgpt/<model>` in normalized responses and chunks: a bare GPT name
72
157
  is otherwise misattributed to API-key OpenAI by the proxy/gateway when both
73
158
  providers are registered. Astra preserves reasoning effort `max`; `none` and
74
159
  `minimal` clamp to `low`.
75
- For ChatGPT streaming, emit function calls from `response.output_item.done`
76
- with complete arguments. Forwarding `response.output_item.added` followed by
77
- argument-only deltas loses arguments at the proxy's `tool_call` boundary.
78
- Text and reasoning remain incremental.
160
+ For ChatGPT streaming, use the same `StreamNormalizer`/`ToolStream` as other
161
+ transports. Do not reintroduce a separate completed-call emitter: it used to
162
+ lose or duplicate argument fragments at the proxy boundary. Text and reasoning
163
+ remain incremental; callable tools are complete and emitted once at termination.
164
+
79
165
 
80
166
  Offline coverage lives in `tests/unit_chatgpt.rs`. The live server check starts
81
167
  its own loopback CLI process and stops it on completion/failure:
@@ -91,9 +177,9 @@ and image input. It uses the saved ChatGPT login and consumes
91
177
  subscription usage; it is ignored during offline CI. Mount the whole token
92
178
  directory writable for container use so refresh locks and atomic saves work.
93
179
 
94
- **Self-hosted passthrough providers (vLLM / SGLang).** `src/providers/openai_compat.rs` is one generic OpenAI Chat Completions passthrough backing both `vllm` and `sglang` (`OpenAiCompatible::new(name, base_url, api_key: Option)`, registered per env base URL). Two things differ from the hosted providers: the **base URL is configuration** (local `http://localhost:8000/v1` vs remote `https://host/v1`), and **auth is optional** (self-hosted servers are unauthenticated unless launched with `--api-key`, so the `Authorization` header is sent only when a key is set). Passthrough transforms; `reasoning`/`reasoning_content` normalized to `reasoning_content` (vLLM is migrating the field name); `reasoning_effort` forwarded as-is (honored per-model, not clamped); server-specific params go under `x-<name>` (`chat_template_kwargs`, `separate_reasoning`, `guided_json`, `top_k`, …). Note: reasoning/tool parsing are **launch-time server flags** (`--reasoning-parser`, `--tool-call-parser`), so a request only gets that behavior if the server was started for it — llmshim can't enable it per request.
180
+ **Self-hosted passthrough providers (vLLM / SGLang).** `src/providers/openai_compat.rs` is one generic OpenAI Chat Completions passthrough backing both `vllm` and `sglang` (`OpenAiCompatible::new(name, base_url, api_key: Option)`, registered per env base URL). Two things differ from the hosted providers: the **base URL is configuration** (local `http://localhost:8000/v1` vs remote `https://host/v1`), and **auth is optional** (self-hosted servers are unauthenticated unless launched with `--api-key`, so the `Authorization` header is sent only when a key is set). Passthrough transforms; reasoning normalized to typed `reasoning[]` with provenance (vLLM is migrating the field name); `reasoning_effort` forwarded as-is (honored per-model, not clamped); server-specific params go under `x-<name>` (`chat_template_kwargs`, `separate_reasoning`, `guided_json`, `top_k`, …). Note: reasoning/tool parsing are **launch-time server flags** (`--reasoning-parser`, `--tool-call-parser`), so a request only gets that behavior if the server was started for it — llmshim can't enable it per request.
95
181
 
96
- **OpenRouter is the one passthrough provider.** Every other provider translates the OpenAI-format input *away* to a native dialect; OpenRouter (`src/providers/openrouter.rs`) *is* OpenAI Chat Completions, so its transforms are near-identity — messages, tools, vision (`image_url`), and `response_format` are forwarded unchanged; `reasoning_effort` maps 1:1 to OpenRouter's `reasoning:{effort}` (its effort vocabulary is a superset, so no clamping); `message.reasoning` is normalized to `reasoning_content` on responses. OpenRouter models are **not enumerated** in `src/models.rs` (the catalog is huge and dynamic) — any `openrouter/<vendor>/<model>` slug routes through. `x-openrouter` carries OpenRouter-only controls (`provider`, `models`, `transforms`, `route`, native `reasoning`; plus `http_referer`/`x_title` which become headers). The `middle-out` transform is disabled by default for faithful passthrough. Uses `image_url` (Chat Completions) vision via `vision::to_openai_chat`.
182
+ **OpenRouter is the one passthrough provider.** Every other provider translates the OpenAI-format input *away* to a native dialect; OpenRouter (`src/providers/openrouter.rs`) *is* OpenAI Chat Completions, so its transforms are near-identity — messages, tools, vision (`image_url`), and `response_format` are forwarded unchanged; `reasoning_effort` maps 1:1 to OpenRouter's `reasoning:{effort}` (its effort vocabulary is a superset, so no clamping); reasoning is normalized to typed `reasoning[]` with provenance on responses. OpenRouter models are **not enumerated** in `src/models.rs` (the catalog is huge and dynamic) — any `openrouter/<vendor>/<model>` slug routes through. `x-openrouter` carries OpenRouter-only controls (`provider`, `models`, `transforms`, `route`, native `reasoning`; plus `http_referer`/`x_title` which become headers). The `middle-out` transform is disabled by default for faithful passthrough. Uses `image_url` (Chat Completions) vision via `vision::to_openai_chat`.
97
183
 
98
184
  ### Request flow
99
185
 
@@ -126,19 +212,78 @@ Automatic redirects are disabled on the shared client. Keep prompts and
126
212
  provider-specific credential headers at the configured endpoint; 3xx responses
127
213
  remain provider errors. Callers must configure the final URL directly.
128
214
 
129
- `ShimClient` with shared connection pool (`LazyLock`), HTTP/2, gzip/brotli/zstd compression, TCP keepalive + nodelay. Automatic retry (3 attempts by default) on transport errors and 429/500/502/503/504/529 status codes. This is the **reactive** layer: on a retryable *response* it honors the server's `Retry-After` header (integer seconds or HTTP-date) and provider reset hints (OpenAI `x-ratelimit-reset-*`, Anthropic `anthropic-ratelimit-*-reset`), clamped to a cap and nudged with a little jitter; when there's no server hint (or a transport error) it falls back to full-jitter exponential backoff (uniform in `[0, min(cap, base·2^attempt)]`) to avoid a thundering herd. Tunable via `LLMSHIM_MAX_RETRIES` and `LLMSHIM_MAX_BACKOFF_SECS`. `warmup()` pre-establishes TCP+TLS connections. `SseStream` buffers bytes, extracts `data:` lines, routes through provider's `transform_stream_chunk`.
215
+ `ShimClient` with shared connection pool (`LazyLock`), HTTP/2, gzip/brotli/zstd compression, TCP keepalive + nodelay. Automatic retry (3 attempts by default) on transport errors and 429/500/502/503/504/529 status codes. This is the **reactive** layer: on a retryable *response* it honors the server's `Retry-After` header (integer seconds or HTTP-date) and provider reset hints (OpenAI `x-ratelimit-reset-*`, Anthropic `anthropic-ratelimit-*-reset`), clamped to a cap and nudged with a little jitter; when there's no server hint (or a transport error) it falls back to full-jitter exponential backoff (uniform in `[0, min(cap, base·2^attempt)]`) to avoid a thundering herd. Tunable via `LLMSHIM_MAX_RETRIES` and `LLMSHIM_MAX_BACKOFF_SECS`. `warmup()` pre-establishes TCP+TLS connections. `SseStream` decodes bytes with eventsource-stream and feeds one per-response `StreamNormalizer`; do not restore lossy UTF-8 line buffering.
130
216
 
131
217
  ### Fallback chains (`src/fallback.rs`)
132
218
 
133
219
  `FallbackConfig` defines an ordered list of models to try. On retryable errors (429, 500, 502, 503, 529), retries with exponential backoff then falls through to the next model. `completion_with_fallback()` is the top-level API. The proxy supports this via `"fallback": ["model1", "model2"]` in the request body.
134
220
 
221
+ ### Provider health (`src/breaker.rs`, `src/proxy/health.rs`)
222
+
223
+ **Health is not rate-limit backoff.** The token buckets already slow a provider
224
+ down after a 429 — a 429 means the provider is alive and asking for less. The
225
+ breaker counts what retrying cannot fix: 5xx (500/502/503/504/529) and
226
+ transport failures. Adapted from `rcode-provider`'s `ProviderBreaker`, which we
227
+ own: sliding failure window, open state, and a single half-open probe admitted
228
+ after the cooldown. Config: `LLMSHIM_BREAKER_WINDOW_SECS` (60),
229
+ `LLMSHIM_BREAKER_TRIP_THRESHOLD` (3; `0` disables), `LLMSHIM_BREAKER_COOLDOWN_SECS` (30).
230
+
231
+ The breaker hangs on the `Router` (`Router::breaker()` / `with_breaker`). Every
232
+ dispatch path *observes* outcomes so health accrues from ordinary traffic; only
233
+ `fallback.rs` *refuses*, and it checks before every attempt rather than once per
234
+ chain entry — the attempt that opens a circuit is usually the chain's own, so a
235
+ per-entry check would still retry into a target it just watched die. A single-target call is still dispatched:
236
+ with no alternative, refusing would only convert an upstream failure into a
237
+ local one. `proxy::health::build_breaker()` attaches a Redis-coordinated
238
+ `SharedHealth` (failure ZSET + open marker + `SET NX` probe) when
239
+ `LLMSHIM_REDIS_URL` is set and `redis-coordination` is compiled in, mirroring
240
+ `build_limiter`. Every shared operation **fails open**.
241
+
242
+ ### Named routes (`src/config.rs`, `src/router.rs`)
243
+
244
+ A caller-defined name maps to a model plus request settings:
245
+
246
+ ```toml
247
+ [routes.compaction]
248
+ model = "anthropic/claude-haiku-4-5-20251001"
249
+ reasoning_effort = "low"
250
+ max_tokens = 4096
251
+ ```
252
+
253
+ Addressed as `"model": "route/compaction"`, so a route travels through the
254
+ existing `provider/model` grammar — an OpenAI SDK, the CLI and the proxy's
255
+ admission control all handle it without learning a new field. `resolve_key`
256
+ resolves the indirection, so rate limiting never sees an unrecognized string.
257
+
258
+ **llmshim must not learn harness vocabulary.** The name is opaque: a harness may
259
+ call a route `compaction`, `advisor` or `webSearch`, and llmshim only knows it
260
+ maps to a model. Route settings are defaults — a per-request key always wins —
261
+ and an unknown name is a 400, never a silent fall back to the default model.
262
+ Routes do not chain.
263
+
135
264
  ### Vision (`src/vision.rs`)
136
265
 
137
266
  Image content blocks are translated between providers automatically. Users can send images in any format (OpenAI `image_url`, Anthropic `image`, Gemini `inline_data`) and the correct provider sees its native format. Base64 data URIs and plain URLs are both handled. Gemini falls back to a text placeholder for URL images (only supports `inline_data`).
138
267
 
139
268
  ### Multi-model conversations
140
269
 
141
- Each provider sanitizes messages from other providers in `transform_request`. OpenAI's `annotations`/`refusal` stripped for Anthropic/Gemini. `reasoning_content` is stripped by other providers, but **Anthropic reconstructs a native `thinking` block** from `reasoning_content` + `reasoning_signature` (and `redacted_thinking` from `redacted_reasoning_content`) as the first block of the assistant turn, so extended-thinking + tool-use round-trips losslessly (surfaced on responses incl. streaming; opaque signatures are stripped by other providers so they never leak cross-provider; no signature → still stripped). Symmetric to the tool-call `thought_signature` round-trip. Tool calls normalized to OpenAI format in responses, translated back per-provider on input.
270
+ `src/reasoning.rs` owns the single replay policy, typed blocks, signature origins,
271
+ issuer bindings, and drop counters. Every adapter filters before serialization
272
+ and captures from the original response to preserve ordered blocks and encrypted
273
+ Responses items. The shared HTTP client binds origins to the actual request
274
+ (including refreshed OAuth account headers). Unknown provenance/families fail
275
+ closed; matching family+wire permits replay, with same-account binding also
276
+ required for encrypted blocks. Never route by inspecting opaque bytes.
277
+
278
+ New outputs use `message.reasoning[]` and `thought_signature:{data,origin}`.
279
+ Legacy sibling readers exist for one migration release and drop untracked data.
280
+ Native overrides cannot bypass this filter or enable provider-side Responses
281
+ storage. The CLI and proxy retain the whole message; streaming consumers use
282
+ `ReasoningAccumulator` and must preserve signatures and completed item snapshots.
283
+ The shared SSE reader handles split UTF-8, CRLF, multiline events, late usage,
284
+ and terminal markers; do not restore the old per-byte lossy string buffer.
285
+ Tests: `tests/unit_reasoning.rs`, the client SSE tests, and provider regressions.
286
+ See `docs/src/guides/reasoning.md` for the full shape and migration contract.
142
287
 
143
288
  ### Provider extension namespaces (`x-anthropic`, `x-gemini`)
144
289
 
@@ -180,6 +325,30 @@ the verified xAI model ID. Keep the ChatGPT four-model allowlist independent.
180
325
 
181
326
  Two knobs work across every provider: `reasoning_effort` (`none|low|medium|high|xhigh|max`) and `reasoning_mode` (`standard|pro`). A third, `reasoning_summary` (`auto|none`), controls reasoning-text visibility → Anthropic `thinking.display` (`auto`→`summarized`, the default when `reasoning_effort` is present so newer models like Sonnet 5 / Opus 4.7-4.8 return reasoning text instead of the API-default `omitted`; `none`→`omitted` for lower latency). Applies to both the adaptive and pre-4.6 enabled thinking builders; a caller-supplied `thinking` block bypasses it. Each provider transform maps them to its native dialect, **clamping to the nearest tier the target model accepts** (all boundaries verified live — e.g. `max` is native on OpenAI gpt-5.6 and GPT-6 Astra; Anthropic 4.6 rejects `xhigh` but has `max`; Gemini's enum tops out at `high`; xAI grok-4.20 models reject any reasoning param). `mode: "pro"` is native on OpenAI gpt-5.6/-pro models (`reasoning.mode`), emulated as a one-tier effort bump elsewhere; explicit `none` always wins. Native passthrough (`x-openai.reasoning`, `x-anthropic.thinking`, `x-gemini.thinkingConfig`) bypasses the mapping entirely and always takes precedence. **Full per-provider mapping tables: `docs/src/guides/reasoning.md`** — update it and the pinning tests in `tests/unit_*.rs` together whenever a mapping changes.
182
327
 
328
+ ### Owned tool identities and stream state
329
+
330
+ `src/toolcall.rs` owns `WireToolId`, the bidirectional map, request projection,
331
+ and central call/result validation. Canonical ids are minted as `call_ls_*` and
332
+ never use a provider id directly. `wire_ids` must remain on persisted calls;
333
+ Responses `item_id` is distinct from correlation `id`. Source wire ids (including
334
+ Gemini's missing id) are restored on both calls/results, preserving signed
335
+ prefixes. Legacy input ids remain readable, but an owned id without its mapping
336
+ is an error. Never drop invalid tool history to make a request succeed.
337
+
338
+ `src/toolcall/streaming.rs` parses native events into `ToolDelta` and assembles
339
+ JSON arguments once. `src/streaming.rs::StreamNormalizer` is the public stateful
340
+ entry point used by HTTP streaming and manual SSE readers. Stateless provider
341
+ chunk methods no longer expose partial callable records. Tool signatures can
342
+ arrive after names/arguments; preserve their exact data/origin and original wire
343
+ container. Parallel choices close independently. The single-message proxy
344
+ projects choice zero; Rust retains choice indices.
345
+
346
+ Google's current Generate Content contract requires a signature on the first
347
+ function call of each current Gemini 3 batch, not every parallel call. Validate
348
+ that rule after filtering, preserve additional signatures exactly where present,
349
+ and retain optional native function ids on both sides. `unit_toolcall` covers
350
+ paired replay, order, signatures, invalid history, and transport delta sequences.
351
+
183
352
  ### Tool format translation
184
353
 
185
354
  llmshim accepts tools in OpenAI Chat Completions format (nested `function` object) and translates them to each provider's native format:
@@ -206,7 +375,13 @@ HTTP proxy with our own API spec (not OpenAI-compatible). Built on axum.
206
375
  Endpoints:
207
376
  - `POST /v1/chat` — non-streaming (or streaming if `stream: true`)
208
377
  - `POST /v1/chat/stream` — always SSE streaming with typed events (`content`, `reasoning`, `tool_call`, `usage`, `done`, `error`)
209
- - `GET /v1/models` — list available models (filtered to configured providers)
378
+ - `GET /v1/models` — list available models (filtered to configured providers).
379
+ Serves two audiences from one body: an OpenAI SDK reads the `object: "list"` /
380
+ `data[]` envelope (so `client.models.list()` works unmodified), llmshim's own
381
+ clients read `models[]`. Both issue the same request, so there is no path to
382
+ split on — the union *is* the split. `data[].id` is the routing id, requestable
383
+ back as `model`. Built once in `proxy::convert::models_response`, shared with
384
+ the gateway.
210
385
  - `GET /health` — health check with provider list
211
386
 
212
387
  Request format uses `config` for provider-agnostic settings and `provider_config` for raw passthrough. OpenAPI 3.1 spec at `api/openapi.yaml`.
@@ -292,3 +467,101 @@ Common maintenance workflows are packaged as [skills](https://code.claude.com/do
292
467
  - `/add-provider key Name` — wire up a brand-new upstream provider.
293
468
  - `/preflight` — run the fmt + clippy + test trio CI enforces.
294
469
  - `/release 0.1.22` — bump version and tag so CI publishes.
470
+
471
+ ### Capability plans and instance validation
472
+
473
+ `src/shim.rs::Plan` owns catalog-driven structured output, prompt tool calling,
474
+ and optional brief-rationale capture. `ShimClient` applies the same plan on
475
+ completion, stream, and fallback paths; direct provider transforms only support
476
+ native response-format translation. Never emit hidden synthetic calls, native
477
+ deliberation about those calls, or invalid attempts to logs/streams. Managed
478
+ streams buffer with a 32 MiB bound; top-level streaming uses an Arc-owned provider
479
+ so HTTP headers/keepalives can proceed while output is validated.
480
+
481
+ Validate generated data against the ORIGINAL schema with the network/filesystem
482
+ retriever disabled (`schema::validate`). Schema compile budgets are 96 levels,
483
+ 32,768 JSON values and 8 MiB strings/keys. Only complete invalid answers receive
484
+ one repair; refusals and incomplete responses do not. Preserve both attempts'
485
+ reported usage. Unknown catalog fields remain unknown; `forced_tool_choice` is
486
+ independent of tools, with the verified Fable 5.1 prohibition overriding auto.
487
+ Keep synthetic schemas/instructions deterministic so unchanged requests preserve
488
+ prefix caching. Tests: `unit_shim`, proxy conversion tests, client request mocks.
489
+
490
+ ### Signature observations and reasoning profiles
491
+
492
+ `providers/anthropic_signature.rs` has the default-on `signature-introspection`
493
+ feature and Option-only stub. Never use decoded metadata to change provenance,
494
+ messages, replay, routing, fallback, or caches. Synthetic fixtures only. Capture
495
+ one metric observation per response/terminal stream; logs read the observation
496
+ without incrementing it. `x-llmshim-served-model` belongs at response/event level.
497
+
498
+ `providers/anthropic_reasoning.rs::Profile` consumes catalog effort/budget unions
499
+ once per provider transform. Builtin verified options win over community data;
500
+ local/provider overrides retain documented precedence. Keep historical fallback
501
+ entries explicit in `catalog::builtin::anthropic_reasoning_options`, rather than
502
+ guessing support for new model names. Preserve mandatory native model constraints
503
+ (e.g. Fable adaptive-only and forbidden forced choices) separately.
504
+
505
+ ### Native inbound facades
506
+
507
+ `proxy::wire` translates `/v1/messages` and `/v1/chat/completions` through the
508
+ existing chat handlers on both proxy and gateway. Do not create another dispatch,
509
+ auth, quota, or queue path. Gateway authenticates before receipt lookup and again
510
+ in its ordinary admission path; x-api-key maps to Bearer only when Authorization
511
+ is absent. Native idempotency keys include credential and protocol scope.
512
+
513
+ The native facade persists issued replay metadata in private atomic local files,
514
+ configured by `LLMSHIM_REPLAY_RECEIPTS_DIR`. Clients must preserve the native
515
+ message/ID and server receipts across restarts. Never stamp unknown native thinking
516
+ with current-target provenance. Receipts restore original blocks, then the common
517
+ replay filter decides eligibility. Missing owned IDs and edited calls error.
518
+ Text remains incremental; complete reasoning/tool blocks follow once metadata is
519
+ ready. Status/Retry-After and gateway request IDs survive error translation.
520
+ `src/error/normalize.rs` owns error unwrapping for compact JSON/SSE, gateway,
521
+ and native endpoints. Keep display messages readable and source type/code/param
522
+ metadata separate; do not move this logic back into a wire-only formatter.
523
+ JSON responses carry native metadata in response extensions, and SSE errors carry
524
+ an optional structured error object so native rendering remains lossless.
525
+ `n > 1` is refused rather than emulated: the OpenAI backend is the Responses
526
+ API (no `n`), Anthropic Messages and Gemini have no `n` either, and the
527
+ single-message proxy projects choice zero. The refusal is a correctly shaped
528
+ `{"error":{type,message,param:"n",code:"unsupported_parameter"}}`, carried
529
+ through `normalize_error` so both facades render it natively.
530
+ Tests: `unit_wire`, gateway `http::native_tests`. Use a temporary receipt directory
531
+ in tests; never put real signatures, credentials, or conversations in fixtures.
532
+
533
+ ### Provider and CLI regression gates
534
+
535
+ `tests/unit_provider_contracts.rs` discovers Provider implementations from the
536
+ Rust AST and requires a fixture for each. New adapters must preserve compatible
537
+ reasoning, drop foreign/untracked blocks and signatures, and reject invalid tool
538
+ history before serialization. Tool-call containers accept arrays or null only.
539
+ HTTP handlers validate canonical history before committing to an SSE response.
540
+
541
+ `src/cli.rs` validates arguments before side effects. Server options override env
542
+ and saved host/port; help must not start a service. Bind errors are returned to
543
+ main for a clean exit. Tests use ephemeral loopback ports and never touch an
544
+ existing server. Run unit_cli, unit_provider_contracts, unit_wire and gateway
545
+ HTTP tests when changing these boundaries.
546
+
547
+ ### Schema memoization and benchmark receipts
548
+
549
+ `src/schema/memo.rs` caches only schema inputs/results and `Normalization` reports.
550
+ The process-local cache is bounded to 512 entries/16 MiB estimated owned bytes,
551
+ uses LRU eviction, verifies input/options after a hash hit, and performs expensive
552
+ normalization outside the lock. Key all effective Options, including budgets.
553
+ Resolve environment overrides before lookup. Preserve object order and signed
554
+ zero: semantically equal numbers can produce different spilled descriptions.
555
+ No-op hits can retain the input value only after exact input/output comparison.
556
+
557
+ `LLMSHIM_NO_SCHEMA_CACHE=1` bypasses memoization while keeping normalization active.
558
+ Tests cover collisions, eviction, concurrency, returned-value independence,
559
+ report/fallback fidelity, environment changes and fresh tool metadata.
560
+
561
+ `cargo run --release --example bench` loads the two configured native API keys and
562
+ makes 43 logical requests plus client retries. Missing keys fail before network
563
+ calls with setup instructions. Provider errors must never print credentials.
564
+ `--transforms-only` needs no keys or HTTP calls; use it for cache-on/off comparison.
565
+ Keep README measurements dated, name the actual models/host/build/workload, and
566
+ separate API latency, full-transform CPU time, and RSS. Do not mix fresh Rust
567
+ figures with old Python results or claim unmeasured network-only latency shares.