llmshim 0.3.6__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (191) hide show
  1. llmshim-0.4.0/.github/workflows/catalog-refresh.yml +35 -0
  2. {llmshim-0.3.6 → llmshim-0.4.0}/.github/workflows/release.yml +17 -4
  3. {llmshim-0.3.6 → llmshim-0.4.0}/CLAUDE.md +285 -15
  4. {llmshim-0.3.6 → llmshim-0.4.0}/Cargo.lock +484 -19
  5. {llmshim-0.3.6 → llmshim-0.4.0}/Cargo.toml +16 -2
  6. {llmshim-0.3.6 → llmshim-0.4.0}/PKG-INFO +21 -23
  7. {llmshim-0.3.6 → llmshim-0.4.0}/README.md +113 -26
  8. {llmshim-0.3.6 → llmshim-0.4.0}/SECURITY.md +17 -0
  9. llmshim-0.4.0/benchmarks/bench.rs +241 -0
  10. llmshim-0.4.0/crates/llmshim-catalog/Cargo.toml +28 -0
  11. llmshim-0.4.0/crates/llmshim-catalog/LICENSE-APACHE +202 -0
  12. llmshim-0.4.0/crates/llmshim-catalog/LICENSE-MIT +21 -0
  13. llmshim-0.4.0/crates/llmshim-catalog/README.md +118 -0
  14. llmshim-0.4.0/crates/llmshim-catalog/data/LICENSE.models.dev +21 -0
  15. llmshim-0.4.0/crates/llmshim-catalog/data/README.md +8 -0
  16. llmshim-0.4.0/crates/llmshim-catalog/data/models.dev.json +1 -0
  17. llmshim-0.4.0/crates/llmshim-catalog/src/aliases.rs +35 -0
  18. llmshim-0.3.6/src/models.rs → llmshim-0.4.0/crates/llmshim-catalog/src/builtin.rs +300 -209
  19. llmshim-0.4.0/crates/llmshim-catalog/src/capabilities.rs +105 -0
  20. llmshim-0.4.0/crates/llmshim-catalog/src/lib.rs +274 -0
  21. llmshim-0.4.0/crates/llmshim-catalog/src/merge.rs +110 -0
  22. llmshim-0.4.0/crates/llmshim-catalog/src/parse.rs +124 -0
  23. llmshim-0.4.0/crates/llmshim-catalog/src/refresh.rs +356 -0
  24. llmshim-0.4.0/crates/llmshim-catalog/src/types.rs +161 -0
  25. llmshim-0.4.0/crates/llmshim-catalog/tests/catalog.rs +241 -0
  26. llmshim-0.4.0/crates/llmshim-catalog/tests/refresh.rs +232 -0
  27. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/SUMMARY.md +4 -1
  28. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/concepts/contracts.md +1 -1
  29. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/concepts/conversations.md +1 -1
  30. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/concepts/routing.md +14 -9
  31. llmshim-0.4.0/docs/src/guides/caching.md +53 -0
  32. llmshim-0.4.0/docs/src/guides/capabilities.md +91 -0
  33. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/guides/fallbacks.md +5 -5
  34. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/guides/native-controls.md +16 -5
  35. llmshim-0.4.0/docs/src/guides/reasoning.md +267 -0
  36. llmshim-0.4.0/docs/src/guides/schemas.md +91 -0
  37. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/guides/streaming.md +13 -10
  38. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/guides/tools.md +44 -5
  39. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/introduction.md +1 -1
  40. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/proxy/http-api.md +11 -5
  41. llmshim-0.4.0/docs/src/proxy/native-apis.md +74 -0
  42. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/reference/cli.md +24 -6
  43. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/reference/configuration.md +32 -1
  44. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/reference/errors.md +16 -2
  45. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/reference/models.md +62 -33
  46. llmshim-0.4.0/docs/src/reference/providers.md +107 -0
  47. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/reference/request-fields.md +19 -14
  48. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/reference/surfaces.md +8 -6
  49. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/start/configure.md +18 -2
  50. {llmshim-0.3.6 → llmshim-0.4.0}/llmshim/_client.py +21 -0
  51. {llmshim-0.3.6 → llmshim-0.4.0}/llmshim/types.py +102 -6
  52. llmshim-0.4.0/src/cache.rs +269 -0
  53. llmshim-0.4.0/src/cli.rs +221 -0
  54. {llmshim-0.3.6 → llmshim-0.4.0}/src/client.rs +270 -88
  55. llmshim-0.4.0/src/error/normalize.rs +143 -0
  56. {llmshim-0.3.6 → llmshim-0.4.0}/src/error.rs +5 -0
  57. {llmshim-0.3.6 → llmshim-0.4.0}/src/fallback.rs +12 -54
  58. {llmshim-0.3.6 → llmshim-0.4.0}/src/gateway/http.rs +156 -1
  59. {llmshim-0.3.6 → llmshim-0.4.0}/src/lib.rs +11 -2
  60. {llmshim-0.3.6 → llmshim-0.4.0}/src/log.rs +15 -0
  61. {llmshim-0.3.6 → llmshim-0.4.0}/src/main.rs +175 -87
  62. llmshim-0.4.0/src/models.rs +7 -0
  63. llmshim-0.4.0/src/provider.rs +72 -0
  64. {llmshim-0.3.6 → llmshim-0.4.0}/src/providers/anthropic.rs +143 -247
  65. llmshim-0.4.0/src/providers/anthropic_reasoning.rs +111 -0
  66. llmshim-0.4.0/src/providers/anthropic_signature.rs +165 -0
  67. llmshim-0.4.0/src/providers/chatgpt/auth.rs +470 -0
  68. llmshim-0.4.0/src/providers/chatgpt/mod.rs +280 -0
  69. llmshim-0.4.0/src/providers/chatgpt/streaming.rs +128 -0
  70. {llmshim-0.3.6 → llmshim-0.4.0}/src/providers/gemini.rs +100 -299
  71. {llmshim-0.3.6 → llmshim-0.4.0}/src/providers/mod.rs +3 -0
  72. {llmshim-0.3.6 → llmshim-0.4.0}/src/providers/openai.rs +242 -247
  73. {llmshim-0.3.6 → llmshim-0.4.0}/src/providers/openai_compat.rs +58 -38
  74. {llmshim-0.3.6 → llmshim-0.4.0}/src/providers/openrouter.rs +58 -39
  75. {llmshim-0.3.6 → llmshim-0.4.0}/src/providers/xai.rs +80 -87
  76. llmshim-0.4.0/src/proxy/convert.rs +291 -0
  77. {llmshim-0.3.6 → llmshim-0.4.0}/src/proxy/error.rs +55 -6
  78. {llmshim-0.3.6 → llmshim-0.4.0}/src/proxy/handlers.rs +6 -6
  79. {llmshim-0.3.6 → llmshim-0.4.0}/src/proxy/mod.rs +4 -0
  80. {llmshim-0.3.6 → llmshim-0.4.0}/src/proxy/ratelimit.rs +11 -3
  81. {llmshim-0.3.6 → llmshim-0.4.0}/src/proxy/types.rs +38 -2
  82. llmshim-0.4.0/src/proxy/wire/mod.rs +664 -0
  83. llmshim-0.4.0/src/proxy/wire/receipts.rs +71 -0
  84. llmshim-0.4.0/src/reasoning/normalize.rs +422 -0
  85. llmshim-0.4.0/src/reasoning.rs +617 -0
  86. {llmshim-0.3.6 → llmshim-0.4.0}/src/router.rs +48 -6
  87. llmshim-0.4.0/src/schema/memo.rs +426 -0
  88. llmshim-0.4.0/src/schema/mod.rs +252 -0
  89. llmshim-0.4.0/src/schema/validate.rs +89 -0
  90. llmshim-0.4.0/src/schema/walk.rs +1000 -0
  91. llmshim-0.4.0/src/shim.rs +830 -0
  92. llmshim-0.4.0/src/streaming.rs +190 -0
  93. llmshim-0.4.0/src/toolcall/streaming.rs +594 -0
  94. llmshim-0.4.0/src/toolcall.rs +683 -0
  95. llmshim-0.4.0/src/usage.rs +112 -0
  96. llmshim-0.4.0/tests/fixtures/chatgpt-red.png +0 -0
  97. llmshim-0.4.0/tests/integration_chatgpt.rs +35 -0
  98. llmshim-0.4.0/tests/integration_chatgpt_proxy.rs +265 -0
  99. llmshim-0.4.0/tests/integration_current_models.rs +177 -0
  100. {llmshim-0.3.6 → llmshim-0.4.0}/tests/integration_openrouter.rs +1 -1
  101. {llmshim-0.3.6 → llmshim-0.4.0}/tests/integration_sglang.rs +1 -1
  102. {llmshim-0.3.6 → llmshim-0.4.0}/tests/integration_thinking.rs +4 -4
  103. llmshim-0.4.0/tests/unit_advertised_models.rs +107 -0
  104. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_anthropic.rs +68 -32
  105. llmshim-0.4.0/tests/unit_cache.rs +244 -0
  106. llmshim-0.4.0/tests/unit_chatgpt.rs +879 -0
  107. llmshim-0.4.0/tests/unit_cli.rs +66 -0
  108. llmshim-0.4.0/tests/unit_fable.rs +213 -0
  109. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_gemini.rs +75 -199
  110. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_models.rs +15 -4
  111. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_multimodel.rs +10 -4
  112. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_openai.rs +42 -34
  113. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_openai_compat.rs +7 -4
  114. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_openrouter.rs +3 -3
  115. llmshim-0.4.0/tests/unit_provider_contracts.rs +219 -0
  116. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_proxy.rs +21 -1
  117. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_proxy_convert.rs +21 -1
  118. llmshim-0.4.0/tests/unit_reasoning.rs +323 -0
  119. llmshim-0.4.0/tests/unit_reasoning_profile.rs +52 -0
  120. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_router.rs +27 -0
  121. llmshim-0.4.0/tests/unit_schema.rs +426 -0
  122. llmshim-0.4.0/tests/unit_shim.rs +547 -0
  123. llmshim-0.4.0/tests/unit_signature.rs +155 -0
  124. llmshim-0.4.0/tests/unit_toolcall.rs +536 -0
  125. llmshim-0.4.0/tests/unit_usage.rs +303 -0
  126. llmshim-0.4.0/tests/unit_wire.rs +555 -0
  127. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_xai.rs +51 -58
  128. llmshim-0.3.6/benchmarks/bench.rs +0 -178
  129. llmshim-0.3.6/docs/src/guides/reasoning.md +0 -181
  130. llmshim-0.3.6/docs/src/reference/providers.md +0 -59
  131. llmshim-0.3.6/src/provider.rs +0 -25
  132. llmshim-0.3.6/src/proxy/convert.rs +0 -177
  133. {llmshim-0.3.6 → llmshim-0.4.0}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
  134. {llmshim-0.3.6 → llmshim-0.4.0}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
  135. {llmshim-0.3.6 → llmshim-0.4.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  136. {llmshim-0.3.6 → llmshim-0.4.0}/.github/workflows/pages.yml +0 -0
  137. {llmshim-0.3.6 → llmshim-0.4.0}/.gitignore +0 -0
  138. {llmshim-0.3.6 → llmshim-0.4.0}/CODE_OF_CONDUCT.md +0 -0
  139. {llmshim-0.3.6 → llmshim-0.4.0}/CONTRIBUTING.md +0 -0
  140. {llmshim-0.3.6 → llmshim-0.4.0}/LICENSE-APACHE +0 -0
  141. {llmshim-0.3.6 → llmshim-0.4.0}/LICENSE-MIT +0 -0
  142. {llmshim-0.3.6 → llmshim-0.4.0}/NOTICE +0 -0
  143. {llmshim-0.3.6 → llmshim-0.4.0}/benchmarks/bench_python.py +0 -0
  144. {llmshim-0.3.6 → llmshim-0.4.0}/benchmarks/gateway_loadtest.rs +0 -0
  145. {llmshim-0.3.6 → llmshim-0.4.0}/benchmarks/loadtest.rs +0 -0
  146. {llmshim-0.3.6 → llmshim-0.4.0}/docs/.gitignore +0 -0
  147. {llmshim-0.3.6 → llmshim-0.4.0}/docs/book.toml +0 -0
  148. {llmshim-0.3.6 → llmshim-0.4.0}/docs/mermaid-init.js +0 -0
  149. {llmshim-0.3.6 → llmshim-0.4.0}/docs/mermaid.min.js +0 -0
  150. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/concepts/portability.md +0 -0
  151. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/concepts/translation-flow.md +0 -0
  152. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/guides/images.md +0 -0
  153. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/proxy/deployment.md +0 -0
  154. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/proxy/scaling.md +0 -0
  155. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/reference/api.md +0 -0
  156. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/start/choose.md +0 -0
  157. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/start/cli.md +0 -0
  158. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/start/clients.md +0 -0
  159. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/start/proxy.md +0 -0
  160. {llmshim-0.3.6 → llmshim-0.4.0}/docs/src/start/rust.md +0 -0
  161. {llmshim-0.3.6 → llmshim-0.4.0}/examples/chat.rs +0 -0
  162. {llmshim-0.3.6 → llmshim-0.4.0}/examples/stream.rs +0 -0
  163. {llmshim-0.3.6 → llmshim-0.4.0}/llmshim/__init__.py +0 -0
  164. {llmshim-0.3.6 → llmshim-0.4.0}/llmshim/_server.py +0 -0
  165. {llmshim-0.3.6 → llmshim-0.4.0}/pyproject.toml +0 -0
  166. {llmshim-0.3.6 → llmshim-0.4.0}/src/config.rs +0 -0
  167. {llmshim-0.3.6 → llmshim-0.4.0}/src/env.rs +0 -0
  168. {llmshim-0.3.6 → llmshim-0.4.0}/src/gateway/auth.rs +0 -0
  169. {llmshim-0.3.6 → llmshim-0.4.0}/src/gateway/distributed.rs +0 -0
  170. {llmshim-0.3.6 → llmshim-0.4.0}/src/gateway/idempotency.rs +0 -0
  171. {llmshim-0.3.6 → llmshim-0.4.0}/src/gateway/metrics.rs +0 -0
  172. {llmshim-0.3.6 → llmshim-0.4.0}/src/gateway/mod.rs +0 -0
  173. {llmshim-0.3.6 → llmshim-0.4.0}/src/gateway/quota.rs +0 -0
  174. {llmshim-0.3.6 → llmshim-0.4.0}/src/vision.rs +0 -0
  175. {llmshim-0.3.6 → llmshim-0.4.0}/tests/integration.rs +0 -0
  176. {llmshim-0.3.6 → llmshim-0.4.0}/tests/integration_fallback.rs +0 -0
  177. {llmshim-0.3.6 → llmshim-0.4.0}/tests/integration_gemini.rs +0 -0
  178. {llmshim-0.3.6 → llmshim-0.4.0}/tests/integration_gemini_tools.rs +0 -0
  179. {llmshim-0.3.6 → llmshim-0.4.0}/tests/integration_long_context.rs +0 -0
  180. {llmshim-0.3.6 → llmshim-0.4.0}/tests/integration_multimodel.rs +0 -0
  181. {llmshim-0.3.6 → llmshim-0.4.0}/tests/integration_proxy.rs +0 -0
  182. {llmshim-0.3.6 → llmshim-0.4.0}/tests/integration_tool_roundtrip.rs +0 -0
  183. {llmshim-0.3.6 → llmshim-0.4.0}/tests/integration_vision.rs +0 -0
  184. {llmshim-0.3.6 → llmshim-0.4.0}/tests/integration_xai.rs +0 -0
  185. {llmshim-0.3.6 → llmshim-0.4.0}/tests/support/completion_status.rs +0 -0
  186. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_fallback.rs +0 -0
  187. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_fast_mode.rs +0 -0
  188. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_log.rs +0 -0
  189. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_sse.rs +0 -0
  190. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_tools.rs +0 -0
  191. {llmshim-0.3.6 → llmshim-0.4.0}/tests/unit_vision.rs +0 -0
@@ -0,0 +1,35 @@
1
+ name: Refresh model catalog snapshot
2
+
3
+ on:
4
+ schedule:
5
+ - cron: '19 6 * * 1'
6
+ workflow_dispatch:
7
+
8
+ permissions:
9
+ contents: write
10
+ pull-requests: write
11
+
12
+ concurrency:
13
+ group: model-catalog-refresh
14
+ cancel-in-progress: false
15
+
16
+ jobs:
17
+ snapshot:
18
+ runs-on: ubuntu-latest
19
+ steps:
20
+ - uses: actions/checkout@v4
21
+ - uses: dtolnay/rust-toolchain@stable
22
+ - name: Download public catalog
23
+ run: |
24
+ curl --fail --silent --show-error --max-time 60 \
25
+ https://models.dev/api.json -o crates/llmshim-catalog/data/models.dev.json
26
+ - name: Validate catalog and refresh behavior
27
+ run: cargo test -p llmshim-catalog
28
+ - name: Propose snapshot update
29
+ uses: peter-evans/create-pull-request@v7
30
+ with:
31
+ branch: automation/model-catalog
32
+ add-paths: crates/llmshim-catalog/data/models.dev.json
33
+ commit-message: 'chore: refresh public model catalog snapshot'
34
+ title: 'Refresh public model catalog snapshot'
35
+ body: 'Updates the vendored models.dev snapshot. Verified builtin assertions and local overrides retain precedence. Catalog and refresh tests passed.'
@@ -16,11 +16,13 @@ jobs:
16
16
  - uses: actions/checkout@v4
17
17
  - uses: dtolnay/rust-toolchain@stable
18
18
  - name: Check formatting
19
- run: cargo fmt --check
19
+ run: cargo fmt --all --check
20
20
  - name: Clippy
21
- run: cargo clippy --features proxy -- -D warnings
21
+ run: cargo clippy --workspace --features proxy -- -D warnings
22
22
  - name: Unit tests
23
- run: cargo test --features proxy --tests
23
+ env:
24
+ LLMSHIM_CATALOG_OFFLINE: '1'
25
+ run: cargo test --workspace --features proxy --tests
24
26
 
25
27
  # --- Publish to crates.io ---
26
28
  crates-io:
@@ -29,6 +31,17 @@ jobs:
29
31
  steps:
30
32
  - uses: actions/checkout@v4
31
33
  - uses: dtolnay/rust-toolchain@stable
34
+ - name: Publish catalog dependency to crates.io
35
+ env:
36
+ CARGO_REGISTRY_TOKEN: ${{ secrets.CARGO_REGISTRY_TOKEN }}
37
+ run: |
38
+ set +e
39
+ out=$(cargo publish -p llmshim-catalog --allow-dirty 2>&1); code=$?
40
+ echo "$out"
41
+ [ $code -eq 0 ] && exit 0
42
+ echo "$out" | grep -qiE "already (been )?uploaded|already exists" \
43
+ && { echo "Catalog version already on crates.io; skipping."; exit 0; }
44
+ exit $code
32
45
  - name: Publish to crates.io
33
46
  env:
34
47
  CARGO_REGISTRY_TOKEN: ${{ secrets.CARGO_REGISTRY_TOKEN }}
@@ -36,7 +49,7 @@ jobs:
36
49
  # release (e.g. to fill a failed downstream registry) is safe.
37
50
  run: |
38
51
  set +e
39
- out=$(cargo publish --allow-dirty 2>&1); code=$?
52
+ out=$(cargo publish -p llmshim --allow-dirty 2>&1); code=$?
40
53
  echo "$out"
41
54
  [ $code -eq 0 ] && exit 0
42
55
  echo "$out" | grep -qiE "already (been )?uploaded|already exists" \
@@ -10,13 +10,14 @@ A pure Rust LLM API translation layer. Takes OpenAI-format JSON requests, transl
10
10
 
11
11
  This is a public crate on crates.io. Do NOT make breaking changes to `pub` items in `src/lib.rs`, `src/router.rs`, `src/provider.rs`, `src/error.rs`, `src/fallback.rs`, `src/log.rs`, `src/config.rs`, `src/models.rs`, or `src/vision.rs` without a semver bump.
12
12
 
13
- ## Supported models
14
-
15
- - **OpenAI:** `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-5.5`, `gpt-5.5-pro`, `gpt-5.4`, `gpt-5.4-pro`, `gpt-5.4-mini`, `gpt-5.4-nano`
16
- - **Anthropic:** `claude-opus-5`, `claude-opus-4-8`, `claude-sonnet-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-sonnet-4-6`, `claude-haiku-4-5-20251001`
17
- - **Gemini:** `gemini-3.8-flash`, `gemini-3.7-flash`, `gemini-3.6-flash`, `gemini-3.5-flash`, `gemini-3.5-flash-lite`, `gemini-3.1-flash-lite`
18
- - **xAI:** `grok-4.6`, `grok-4.5`, `grok-4.3`, `grok-4.20-multi-agent-beta-0309`, `grok-4.20-beta-0309-reasoning`, `grok-4.20-beta-0309-non-reasoning`
19
- - **OpenRouter:** not enumerated (huge/dynamic catalog) — any `openrouter/<vendor>/<model>` slug routes through, e.g. `openrouter/anthropic/claude-sonnet-4.5`.
13
+ ## Advertised models
14
+
15
+ - **OpenAI:** `gpt-6-astra`, `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`
16
+ - **ChatGPT subscription (OAuth):** only `chatgpt/gpt-6-astra`, `chatgpt/gpt-5.6-sol`, `chatgpt/gpt-5.6-terra`, and `chatgpt/gpt-5.6-luna`. `CHATGPT_MODELS` in `src/models.rs` is shared by discovery, CLI selection, and validation; older/unlisted models fail before authentication or network calls.
17
+ - **Anthropic:** `claude-fable-5-1`, `claude-opus-5`, `claude-sonnet-5`, `claude-haiku-4-5-20251001`
18
+ - **Gemini:** `gemini-3.8-flash`, `gemini-3.5-flash-lite`
19
+ - **xAI:** `grok-4.6`
20
+ - **OpenRouter:** not enumerated (huge/dynamic catalog) — any `openrouter/<vendor>/<model>` slug routes through, e.g. `openrouter/anthropic/claude-sonnet-5`.
20
21
  - **vLLM / SGLang:** not enumerated (self-hosted) — any `vllm/<served-model>` or `sglang/<served-model>` routes through to the configured server, e.g. `sglang/Qwen/Qwen3.6-35B-A3B-FP8`.
21
22
 
22
23
  ## Build & Test
@@ -36,20 +37,127 @@ API keys: `~/.llmshim/config.toml` (via `llmshim configure`) or env vars `OPENAI
36
37
 
37
38
  ## Architecture
38
39
 
40
+ ### Catalog and usage normalization (local 0.4 development)
41
+
42
+ `llmshim-catalog` is a standalone workspace member. Its `builtin` module owns
43
+ the curated/historical constants; `src/models.rs` reexports the legacy borrowed
44
+ API. Owned metadata and snapshots live in `llmshim::catalog`. Keep the curated
45
+ 15-route discovery and four-entry ChatGPT allowlist distinct from catalog
46
+ coverage. Local policy > provider capabilities > verified builtin assertions >
47
+ models.dev, per field; unknowns never erase assertions. Provider APIs never
48
+ contribute pricing. See `crates/llmshim-catalog/README.md` for cache and override
49
+ paths, offline behavior, aliases, and publication order. Never await catalog
50
+ refresh in a completion; hold a snapshot for decisions that must agree.
51
+
52
+ `usage.cache_read_tokens` and `usage.cache_write_tokens` are always present on
53
+ normalized responses and usage chunks, logs, and proxy usage. Native token
54
+ fields remain readable. `src/usage.rs` owns extraction; the streaming client
55
+ merges Anthropic's start/delta usage before normalizing terminal counts.
56
+ The native Chat Completions streams must use their own parser in the client;
57
+ passing them to the Responses parser silently drops all events.
58
+
59
+ Offline checks for this work:
60
+
61
+ ```sh
62
+ LLMSHIM_CATALOG_OFFLINE=1 cargo test --workspace --features proxy --tests
63
+ cargo test -p llmshim-catalog
64
+ cargo clippy --workspace --features proxy -- -D warnings
65
+ cargo package -p llmshim-catalog --allow-dirty
66
+ ```
67
+
68
+ The root package is 0.4.0 because log/proxy usage structs gain fields; no
69
+ release has been performed. Future release workflows must publish the catalog
70
+ dependency before llmshim. Public code, fixtures, artifacts, and docs must use
71
+ generic examples and contain no private consumer identities or context.
72
+
73
+ ### Cache annotations and shared schema normalization
74
+
75
+ `src/cache.rs` translates caller `x-cache` segments into native Anthropic
76
+ breakpoints and an explicit Responses prompt_cache_key. It never infers
77
+ stability. Last eligible boundaries win within the four-slot budget; existing
78
+ explicit markers consume slots. Managed segments supersede automatic caching.
79
+ One-hour markers must precede five-minute markers. Marker scans inspect actual
80
+ cache locations, not arbitrary schema/default JSON. Without annotations, native
81
+ passthrough remains unchanged. `ProviderRequest::can_continue_from` checks
82
+ endpoint, headers, settings (including include/store/reasoning) and input prefix;
83
+ there is no stored/delta continuation engine. See the caching guide.
84
+
85
+ `src/schema/` owns the single schema walker and local resource resolver. Every
86
+ adapter normalizes native tool schemas after overrides. MCP inputSchema is
87
+ normalized on ingest and raw MCP tool definitions are accepted. Keep literal
88
+ values/property names separate from schema-node traversal, preserve meaningful
89
+ stripped constraints in descriptions, and never fetch an external reference.
90
+ Cycles, unresolved resources, incompatible residues and expansion-budget failures
91
+ fall back per tool. Only successful enforcement permits strict:true. Global
92
+ bypass flags: LLMSHIM_NO_SCHEMA_NORMALIZATION and LLMSHIM_NO_STRICT. OutputSchema
93
+ provides a reversible non-object wrapper; generated-instance validation and
94
+ repair remain a separate concern. See `docs/src/guides/schemas.md`.
95
+
96
+ ### Curated discovery
97
+
98
+ `src/models.rs::MODELS` is the single advertised list, imported directly by
99
+ `src/main.rs` and used by `available_models()` for proxy discovery. Keep the
100
+ current model in each retained tier: four OpenAI, four Anthropic, two stable
101
+ Gemini, one xAI, and four ChatGPT routes. Do not add previews or bring back
102
+ superseded generations without an explicit catalog decision.
103
+
104
+ Historical metadata lives in private `LEGACY_MODELS` and remains queryable via
105
+ `spec()`. Pruning discovery does not delete legacy transforms, tests, or
106
+ explicit-ID routing. ChatGPT keeps its separately authorized four-ID request
107
+ allowlist. Keep reader-facing model tables and examples current; retain the
108
+ actual model IDs in historical benchmark results and regression fixtures.
109
+
39
110
  ### Value-based transforms, no canonical struct
40
111
 
41
112
  Requests flow as `serde_json::Value`. Each provider's transform takes raw JSON and maps only what it understands. Provider-specific features use `x-anthropic`, `x-gemini`, `x-openrouter`, `x-vllm`, `x-sglang` namespaces.
42
113
 
43
- **Self-hosted passthrough providers (vLLM / SGLang).** `src/providers/openai_compat.rs` is one generic OpenAI Chat Completions passthrough backing both `vllm` and `sglang` (`OpenAiCompatible::new(name, base_url, api_key: Option)`, registered per env base URL). Two things differ from the hosted providers: the **base URL is configuration** (local `http://localhost:8000/v1` vs remote `https://host/v1`), and **auth is optional** (self-hosted servers are unauthenticated unless launched with `--api-key`, so the `Authorization` header is sent only when a key is set). Passthrough transforms; `reasoning`/`reasoning_content` normalized to `reasoning_content` (vLLM is migrating the field name); `reasoning_effort` forwarded as-is (honored per-model, not clamped); server-specific params go under `x-<name>` (`chat_template_kwargs`, `separate_reasoning`, `guided_json`, `top_k`, …). Note: reasoning/tool parsing are **launch-time server flags** (`--reasoning-parser`, `--tool-call-parser`), so a request only gets that behavior if the server was started for it — llmshim can't enable it per request.
114
+ **ChatGPT OAuth (`src/providers/chatgpt/`).** Run `llmshim login chatgpt` for
115
+ device-code authentication, `login chatgpt --status` for a local check, and
116
+ `logout chatgpt` to remove the selected cache. The independent default cache
117
+ is `~/.llmshim/chatgpt/auth.json`; `CHATGPT_TOKEN_DIR`/`CHATGPT_AUTH_FILE` can
118
+ override it. Never read or overwrite Codex credentials implicitly. The
119
+ object-safe `Provider::prepare_request` hook defaults to `transform_request`;
120
+ ChatGPT uses it to refresh asynchronously with cross-process file locking and
121
+ atomic owner-only token writes. Requests never initiate interactive login.
122
+
123
+ The subscription backend requires SSE, `stream: true`, and `store: false`.
124
+ ChatGPT reuses the Responses translator, enforces the backend field allowlist
125
+ after `x-chatgpt` overrides, and aggregates a validated terminal event for
126
+ non-streaming callers. EOF or `[DONE]` without a terminal event is an error.
127
+ Preserve `chatgpt/<model>` in normalized responses and chunks: a bare GPT name
128
+ is otherwise misattributed to API-key OpenAI by the proxy/gateway when both
129
+ providers are registered. Astra preserves reasoning effort `max`; `none` and
130
+ `minimal` clamp to `low`.
131
+ For ChatGPT streaming, use the same `StreamNormalizer`/`ToolStream` as other
132
+ transports. Do not reintroduce a separate completed-call emitter: it used to
133
+ lose or duplicate argument fragments at the proxy boundary. Text and reasoning
134
+ remain incremental; callable tools are complete and emitted once at termination.
135
+
136
+
137
+ Offline coverage lives in `tests/unit_chatgpt.rs`. The live server check starts
138
+ its own loopback CLI process and stops it on completion/failure:
139
+
140
+ ```bash
141
+ cargo test --features proxy --test integration_chatgpt_proxy -- --ignored --nocapture
142
+ ```
143
+
144
+ It checks all four models through `/v1/chat` and `/v1/chat/stream`, provider
145
+ identity with API-key OpenAI also registered, model discovery, old-model
146
+ rejection, Astra tool calls (normal and streaming), a tool-result round trip,
147
+ and image input. It uses the saved ChatGPT login and consumes
148
+ subscription usage; it is ignored during offline CI. Mount the whole token
149
+ directory writable for container use so refresh locks and atomic saves work.
44
150
 
45
- **OpenRouter is the one passthrough provider.** Every other provider translates the OpenAI-format input *away* to a native dialect; OpenRouter (`src/providers/openrouter.rs`) *is* OpenAI Chat Completions, so its transforms are near-identity — messages, tools, vision (`image_url`), and `response_format` are forwarded unchanged; `reasoning_effort` maps 1:1 to OpenRouter's `reasoning:{effort}` (its effort vocabulary is a superset, so no clamping); `message.reasoning` is normalized to `reasoning_content` on responses. OpenRouter models are **not enumerated** in `src/models.rs` (the catalog is huge and dynamic) — any `openrouter/<vendor>/<model>` slug routes through. `x-openrouter` carries OpenRouter-only controls (`provider`, `models`, `transforms`, `route`, native `reasoning`; plus `http_referer`/`x_title` which become headers). The `middle-out` transform is disabled by default for faithful passthrough. Uses `image_url` (Chat Completions) vision via `vision::to_openai_chat`.
151
+ **Self-hosted passthrough providers (vLLM / SGLang).** `src/providers/openai_compat.rs` is one generic OpenAI Chat Completions passthrough backing both `vllm` and `sglang` (`OpenAiCompatible::new(name, base_url, api_key: Option)`, registered per env base URL). Two things differ from the hosted providers: the **base URL is configuration** (local `http://localhost:8000/v1` vs remote `https://host/v1`), and **auth is optional** (self-hosted servers are unauthenticated unless launched with `--api-key`, so the `Authorization` header is sent only when a key is set). Passthrough transforms; reasoning normalized to typed `reasoning[]` with provenance (vLLM is migrating the field name); `reasoning_effort` forwarded as-is (honored per-model, not clamped); server-specific params go under `x-<name>` (`chat_template_kwargs`, `separate_reasoning`, `guided_json`, `top_k`, …). Note: reasoning/tool parsing are **launch-time server flags** (`--reasoning-parser`, `--tool-call-parser`), so a request only gets that behavior if the server was started for it — llmshim can't enable it per request.
152
+
153
+ **OpenRouter is the one passthrough provider.** Every other provider translates the OpenAI-format input *away* to a native dialect; OpenRouter (`src/providers/openrouter.rs`) *is* OpenAI Chat Completions, so its transforms are near-identity — messages, tools, vision (`image_url`), and `response_format` are forwarded unchanged; `reasoning_effort` maps 1:1 to OpenRouter's `reasoning:{effort}` (its effort vocabulary is a superset, so no clamping); reasoning is normalized to typed `reasoning[]` with provenance on responses. OpenRouter models are **not enumerated** in `src/models.rs` (the catalog is huge and dynamic) — any `openrouter/<vendor>/<model>` slug routes through. `x-openrouter` carries OpenRouter-only controls (`provider`, `models`, `transforms`, `route`, native `reasoning`; plus `http_referer`/`x_title` which become headers). The `middle-out` transform is disabled by default for faithful passthrough. Uses `image_url` (Chat Completions) vision via `vision::to_openai_chat`.
46
154
 
47
155
  ### Request flow
48
156
 
49
157
  ```
50
158
  llmshim::completion(router, request)
51
- → router.resolve("anthropic/claude-sonnet-4-6") // parse "provider/model"
52
- → provider.transform_request(model, &value) // OpenAI JSON → provider-native
159
+ → router.resolve("anthropic/claude-sonnet-5") // parse "provider/model"
160
+ → provider.prepare_request(model, &value).await // refresh OAuth if needed, then transform
53
161
  → client.send(provider_request) // HTTP
54
162
  → provider.transform_response(model, body) // provider-native → OpenAI JSON
55
163
  ```
@@ -67,7 +175,7 @@ Streaming status handling is separate and is not changed by this policy.
67
175
 
68
176
  ### Router (`src/router.rs`)
69
177
 
70
- Parses `"provider/model"` strings by splitting on the **first** `/` only, so an OpenRouter slug's internal slash survives (`openrouter/anthropic/claude-sonnet-4.5` → provider `openrouter`, model `anthropic/claude-sonnet-4.5`). Auto-infers provider from prefix (`gpt*`/`o*` → openai, `claude*` → anthropic, `gemini*` → gemini, `grok*` → xai); **OpenRouter, vLLM, and SGLang have no prefix inference** — their slugs collide with everyone's, so address them explicitly (`openrouter/…`, `vllm/…`, `sglang/…`); the first-slash split also preserves HF-style served-model slugs (`vllm/meta-llama/Llama-3.1-8B-Instruct`). Supports aliases. `Router::from_env()` reads API-key env vars, plus `VLLM_BASE_URL` / `SGLANG_BASE_URL` (+ optional `*_API_KEY`) for the self-hosted providers.
178
+ Parses `"provider/model"` strings by splitting on the **first** `/` only, so an OpenRouter slug's internal slash survives (`openrouter/anthropic/claude-sonnet-5` → provider `openrouter`, model `anthropic/claude-sonnet-5`). Auto-infers provider from prefix (`gpt*`/`o*` → openai, `claude*` → anthropic, `gemini*` → gemini, `grok*` → xai); **OpenRouter, vLLM, and SGLang have no prefix inference** — their slugs collide with everyone's, so address them explicitly (`openrouter/…`, `vllm/…`, `sglang/…`); the first-slash split also preserves HF-style served-model slugs (`vllm/meta-llama/Llama-3.1-8B-Instruct`). Supports aliases. `Router::from_env()` reads API-key env vars, plus `VLLM_BASE_URL` / `SGLANG_BASE_URL` (+ optional `*_API_KEY`) for the self-hosted providers.
71
179
 
72
180
  ### HTTP Client (`src/client.rs`)
73
181
 
@@ -75,7 +183,7 @@ Automatic redirects are disabled on the shared client. Keep prompts and
75
183
  provider-specific credential headers at the configured endpoint; 3xx responses
76
184
  remain provider errors. Callers must configure the final URL directly.
77
185
 
78
- `ShimClient` with shared connection pool (`LazyLock`), HTTP/2, gzip/brotli/zstd compression, TCP keepalive + nodelay. Automatic retry (3 attempts by default) on transport errors and 429/500/502/503/504/529 status codes. This is the **reactive** layer: on a retryable *response* it honors the server's `Retry-After` header (integer seconds or HTTP-date) and provider reset hints (OpenAI `x-ratelimit-reset-*`, Anthropic `anthropic-ratelimit-*-reset`), clamped to a cap and nudged with a little jitter; when there's no server hint (or a transport error) it falls back to full-jitter exponential backoff (uniform in `[0, min(cap, base·2^attempt)]`) to avoid a thundering herd. Tunable via `LLMSHIM_MAX_RETRIES` and `LLMSHIM_MAX_BACKOFF_SECS`. `warmup()` pre-establishes TCP+TLS connections. `SseStream` buffers bytes, extracts `data:` lines, routes through provider's `transform_stream_chunk`.
186
+ `ShimClient` with shared connection pool (`LazyLock`), HTTP/2, gzip/brotli/zstd compression, TCP keepalive + nodelay. Automatic retry (3 attempts by default) on transport errors and 429/500/502/503/504/529 status codes. This is the **reactive** layer: on a retryable *response* it honors the server's `Retry-After` header (integer seconds or HTTP-date) and provider reset hints (OpenAI `x-ratelimit-reset-*`, Anthropic `anthropic-ratelimit-*-reset`), clamped to a cap and nudged with a little jitter; when there's no server hint (or a transport error) it falls back to full-jitter exponential backoff (uniform in `[0, min(cap, base·2^attempt)]`) to avoid a thundering herd. Tunable via `LLMSHIM_MAX_RETRIES` and `LLMSHIM_MAX_BACKOFF_SECS`. `warmup()` pre-establishes TCP+TLS connections. `SseStream` decodes bytes with eventsource-stream and feeds one per-response `StreamNormalizer`; do not restore lossy UTF-8 line buffering.
79
187
 
80
188
  ### Fallback chains (`src/fallback.rs`)
81
189
 
@@ -87,7 +195,23 @@ Image content blocks are translated between providers automatically. Users can s
87
195
 
88
196
  ### Multi-model conversations
89
197
 
90
- Each provider sanitizes messages from other providers in `transform_request`. OpenAI's `annotations`/`refusal` stripped for Anthropic/Gemini. `reasoning_content` is stripped by other providers, but **Anthropic reconstructs a native `thinking` block** from `reasoning_content` + `reasoning_signature` (and `redacted_thinking` from `redacted_reasoning_content`) as the first block of the assistant turn, so extended-thinking + tool-use round-trips losslessly (surfaced on responses incl. streaming; opaque signatures are stripped by other providers so they never leak cross-provider; no signature → still stripped). Symmetric to the tool-call `thought_signature` round-trip. Tool calls normalized to OpenAI format in responses, translated back per-provider on input.
198
+ `src/reasoning.rs` owns the single replay policy, typed blocks, signature origins,
199
+ issuer bindings, and drop counters. Every adapter filters before serialization
200
+ and captures from the original response to preserve ordered blocks and encrypted
201
+ Responses items. The shared HTTP client binds origins to the actual request
202
+ (including refreshed OAuth account headers). Unknown provenance/families fail
203
+ closed; matching family+wire permits replay, with same-account binding also
204
+ required for encrypted blocks. Never route by inspecting opaque bytes.
205
+
206
+ New outputs use `message.reasoning[]` and `thought_signature:{data,origin}`.
207
+ Legacy sibling readers exist for one migration release and drop untracked data.
208
+ Native overrides cannot bypass this filter or enable provider-side Responses
209
+ storage. The CLI and proxy retain the whole message; streaming consumers use
210
+ `ReasoningAccumulator` and must preserve signatures and completed item snapshots.
211
+ The shared SSE reader handles split UTF-8, CRLF, multiline events, late usage,
212
+ and terminal markers; do not restore the old per-byte lossy string buffer.
213
+ Tests: `tests/unit_reasoning.rs`, the client SSE tests, and provider regressions.
214
+ See `docs/src/guides/reasoning.md` for the full shape and migration contract.
91
215
 
92
216
  ### Provider extension namespaces (`x-anthropic`, `x-gemini`)
93
217
 
@@ -98,7 +222,60 @@ Callers pass provider-specific controls under these keys. Each provider copies w
98
222
 
99
223
  ### Unified reasoning controls
100
224
 
101
- Two knobs work across every provider: `reasoning_effort` (`none|low|medium|high|xhigh|max`) and `reasoning_mode` (`standard|pro`). A third, `reasoning_summary` (`auto|none`), controls reasoning-text visibility → Anthropic `thinking.display` (`auto`→`summarized`, the default when `reasoning_effort` is present so newer models like Sonnet 5 / Opus 4.7-4.8 return reasoning text instead of the API-default `omitted`; `none`→`omitted` for lower latency). Applies to both the adaptive and pre-4.6 enabled thinking builders; a caller-supplied `thinking` block bypasses it. Each provider transform maps them to its native dialect, **clamping to the nearest tier the target model accepts** (all boundaries verified live — e.g. `max` is native only on OpenAI gpt-5.6; Anthropic 4.6 rejects `xhigh` but has `max`; Gemini's enum tops out at `high`; xAI grok-4.20 models reject any reasoning param). `mode: "pro"` is native on OpenAI gpt-5.6/-pro models (`reasoning.mode`), emulated as a one-tier effort bump elsewhere; explicit `none` always wins. Native passthrough (`x-openai.reasoning`, `x-anthropic.thinking`, `x-gemini.thinkingConfig`) bypasses the mapping entirely and always takes precedence. **Full per-provider mapping tables: `docs/src/guides/reasoning.md`** — update it and the pinning tests in `tests/unit_*.rs` together whenever a mapping changes.
225
+ **Fable compatibility (verified September 2026).** Advertise Fable 5.1;
226
+ retain Fable 5 behavior and metadata for explicit requests. Both use always-on adaptive thinking; unified `none` clamps to `low`,
227
+ and native disabled/manual thinking fails locally. Strip `temperature`,
228
+ `top_p`, and `top_k` for Fable and Opus 5 even without an explicit thinking
229
+ object. Fable 5.1 rejects forced tool selection (`required` / `any` / `tool`);
230
+ do not silently turn it into `auto`. Fable 5 still accepts forced tools.
231
+ Both versions reject assistant prefill. Refusal stop reasons on Fable/Opus 5
232
+ map to `content_filter` in normal and streaming engine responses.
233
+
234
+ Fable 5.1's thinking signatures are conversation-bound. Preserve appended
235
+ system turns instead of hoisting them into the initial prompt. Keep the
236
+ initial system/tools/message prefix stable during a signed-thinking round
237
+ trip; errors must not become success. Test with the
238
+ `thinking-binding-controls-2026-08-01` beta and
239
+ `thinking.block_binding.prefix_mismatch_behavior: "error"` so older API
240
+ accounts exercise enforcement too. The provider handles incompatible
241
+ thinking on a switch to older models; do not infer signatures from text.
242
+
243
+ `tests/unit_fable.rs` pins these rules. Live checks for Fable 5, Fable 5.1,
244
+ Opus 5, Gemini 3.8 Flash, and Grok 4.6:
245
+
246
+ ```bash
247
+ cargo test --features proxy --test integration_current_models -- --ignored --nocapture
248
+ ```
249
+
250
+ The live tests consume API usage and are ignored during offline preflight.
251
+ Gemini 3.8 Flash and Opus 5 already had catalog/adapter support; Grok 4.6 is
252
+ the verified xAI model ID. Keep the ChatGPT four-model allowlist independent.
253
+
254
+ Two knobs work across every provider: `reasoning_effort` (`none|low|medium|high|xhigh|max`) and `reasoning_mode` (`standard|pro`). A third, `reasoning_summary` (`auto|none`), controls reasoning-text visibility → Anthropic `thinking.display` (`auto`→`summarized`, the default when `reasoning_effort` is present so newer models like Sonnet 5 / Opus 4.7-4.8 return reasoning text instead of the API-default `omitted`; `none`→`omitted` for lower latency). Applies to both the adaptive and pre-4.6 enabled thinking builders; a caller-supplied `thinking` block bypasses it. Each provider transform maps them to its native dialect, **clamping to the nearest tier the target model accepts** (all boundaries verified live — e.g. `max` is native on OpenAI gpt-5.6 and GPT-6 Astra; Anthropic 4.6 rejects `xhigh` but has `max`; Gemini's enum tops out at `high`; xAI grok-4.20 models reject any reasoning param). `mode: "pro"` is native on OpenAI gpt-5.6/-pro models (`reasoning.mode`), emulated as a one-tier effort bump elsewhere; explicit `none` always wins. Native passthrough (`x-openai.reasoning`, `x-anthropic.thinking`, `x-gemini.thinkingConfig`) bypasses the mapping entirely and always takes precedence. **Full per-provider mapping tables: `docs/src/guides/reasoning.md`** — update it and the pinning tests in `tests/unit_*.rs` together whenever a mapping changes.
255
+
256
+ ### Owned tool identities and stream state
257
+
258
+ `src/toolcall.rs` owns `WireToolId`, the bidirectional map, request projection,
259
+ and central call/result validation. Canonical ids are minted as `call_ls_*` and
260
+ never use a provider id directly. `wire_ids` must remain on persisted calls;
261
+ Responses `item_id` is distinct from correlation `id`. Source wire ids (including
262
+ Gemini's missing id) are restored on both calls/results, preserving signed
263
+ prefixes. Legacy input ids remain readable, but an owned id without its mapping
264
+ is an error. Never drop invalid tool history to make a request succeed.
265
+
266
+ `src/toolcall/streaming.rs` parses native events into `ToolDelta` and assembles
267
+ JSON arguments once. `src/streaming.rs::StreamNormalizer` is the public stateful
268
+ entry point used by HTTP streaming and manual SSE readers. Stateless provider
269
+ chunk methods no longer expose partial callable records. Tool signatures can
270
+ arrive after names/arguments; preserve their exact data/origin and original wire
271
+ container. Parallel choices close independently. The single-message proxy
272
+ projects choice zero; Rust retains choice indices.
273
+
274
+ Google's current Generate Content contract requires a signature on the first
275
+ function call of each current Gemini 3 batch, not every parallel call. Validate
276
+ that rule after filtering, preserve additional signatures exactly where present,
277
+ and retain optional native function ids on both sides. `unit_toolcall` covers
278
+ paired replay, order, signatures, invalid history, and transport delta sequences.
102
279
 
103
280
  ### Tool format translation
104
281
 
@@ -212,3 +389,96 @@ Common maintenance workflows are packaged as [skills](https://code.claude.com/do
212
389
  - `/add-provider key Name` — wire up a brand-new upstream provider.
213
390
  - `/preflight` — run the fmt + clippy + test trio CI enforces.
214
391
  - `/release 0.1.22` — bump version and tag so CI publishes.
392
+
393
+ ### Capability plans and instance validation
394
+
395
+ `src/shim.rs::Plan` owns catalog-driven structured output, prompt tool calling,
396
+ and optional brief-rationale capture. `ShimClient` applies the same plan on
397
+ completion, stream, and fallback paths; direct provider transforms only support
398
+ native response-format translation. Never emit hidden synthetic calls, native
399
+ deliberation about those calls, or invalid attempts to logs/streams. Managed
400
+ streams buffer with a 32 MiB bound; top-level streaming uses an Arc-owned provider
401
+ so HTTP headers/keepalives can proceed while output is validated.
402
+
403
+ Validate generated data against the ORIGINAL schema with the network/filesystem
404
+ retriever disabled (`schema::validate`). Schema compile budgets are 96 levels,
405
+ 32,768 JSON values and 8 MiB strings/keys. Only complete invalid answers receive
406
+ one repair; refusals and incomplete responses do not. Preserve both attempts'
407
+ reported usage. Unknown catalog fields remain unknown; `forced_tool_choice` is
408
+ independent of tools, with the verified Fable 5.1 prohibition overriding auto.
409
+ Keep synthetic schemas/instructions deterministic so unchanged requests preserve
410
+ prefix caching. Tests: `unit_shim`, proxy conversion tests, client request mocks.
411
+
412
+ ### Signature observations and reasoning profiles
413
+
414
+ `providers/anthropic_signature.rs` has the default-on `signature-introspection`
415
+ feature and Option-only stub. Never use decoded metadata to change provenance,
416
+ messages, replay, routing, fallback, or caches. Synthetic fixtures only. Capture
417
+ one metric observation per response/terminal stream; logs read the observation
418
+ without incrementing it. `x-llmshim-served-model` belongs at response/event level.
419
+
420
+ `providers/anthropic_reasoning.rs::Profile` consumes catalog effort/budget unions
421
+ once per provider transform. Builtin verified options win over community data;
422
+ local/provider overrides retain documented precedence. Keep historical fallback
423
+ entries explicit in `catalog::builtin::anthropic_reasoning_options`, rather than
424
+ guessing support for new model names. Preserve mandatory native model constraints
425
+ (e.g. Fable adaptive-only and forbidden forced choices) separately.
426
+
427
+ ### Native inbound facades
428
+
429
+ `proxy::wire` translates `/v1/messages` and `/v1/chat/completions` through the
430
+ existing chat handlers on both proxy and gateway. Do not create another dispatch,
431
+ auth, quota, or queue path. Gateway authenticates before receipt lookup and again
432
+ in its ordinary admission path; x-api-key maps to Bearer only when Authorization
433
+ is absent. Native idempotency keys include credential and protocol scope.
434
+
435
+ The native facade persists issued replay metadata in private atomic local files,
436
+ configured by `LLMSHIM_REPLAY_RECEIPTS_DIR`. Clients must preserve the native
437
+ message/ID and server receipts across restarts. Never stamp unknown native thinking
438
+ with current-target provenance. Receipts restore original blocks, then the common
439
+ replay filter decides eligibility. Missing owned IDs and edited calls error.
440
+ Text remains incremental; complete reasoning/tool blocks follow once metadata is
441
+ ready. Status/Retry-After and gateway request IDs survive error translation.
442
+ `src/error/normalize.rs` owns error unwrapping for compact JSON/SSE, gateway,
443
+ and native endpoints. Keep display messages readable and source type/code/param
444
+ metadata separate; do not move this logic back into a wire-only formatter.
445
+ JSON responses carry native metadata in response extensions, and SSE errors carry
446
+ an optional structured error object so native rendering remains lossless.
447
+ Tests: `unit_wire`, gateway `http::native_tests`. Use a temporary receipt directory
448
+ in tests; never put real signatures, credentials, or conversations in fixtures.
449
+
450
+ ### Provider and CLI regression gates
451
+
452
+ `tests/unit_provider_contracts.rs` discovers Provider implementations from the
453
+ Rust AST and requires a fixture for each. New adapters must preserve compatible
454
+ reasoning, drop foreign/untracked blocks and signatures, and reject invalid tool
455
+ history before serialization. Tool-call containers accept arrays or null only.
456
+ HTTP handlers validate canonical history before committing to an SSE response.
457
+
458
+ `src/cli.rs` validates arguments before side effects. Server options override env
459
+ and saved host/port; help must not start a service. Bind errors are returned to
460
+ main for a clean exit. Tests use ephemeral loopback ports and never touch an
461
+ existing server. Run unit_cli, unit_provider_contracts, unit_wire and gateway
462
+ HTTP tests when changing these boundaries.
463
+
464
+ ### Schema memoization and benchmark receipts
465
+
466
+ `src/schema/memo.rs` caches only schema inputs/results and `Normalization` reports.
467
+ The process-local cache is bounded to 512 entries/16 MiB estimated owned bytes,
468
+ uses LRU eviction, verifies input/options after a hash hit, and performs expensive
469
+ normalization outside the lock. Key all effective Options, including budgets.
470
+ Resolve environment overrides before lookup. Preserve object order and signed
471
+ zero: semantically equal numbers can produce different spilled descriptions.
472
+ No-op hits can retain the input value only after exact input/output comparison.
473
+
474
+ `LLMSHIM_NO_SCHEMA_CACHE=1` bypasses memoization while keeping normalization active.
475
+ Tests cover collisions, eviction, concurrency, returned-value independence,
476
+ report/fallback fidelity, environment changes and fresh tool metadata.
477
+
478
+ `cargo run --release --example bench` loads the two configured native API keys and
479
+ makes 43 logical requests plus client retries. Missing keys fail before network
480
+ calls with setup instructions. Provider errors must never print credentials.
481
+ `--transforms-only` needs no keys or HTTP calls; use it for cache-on/off comparison.
482
+ Keep README measurements dated, name the actual models/host/build/workload, and
483
+ separate API latency, full-transform CPU time, and RSS. Do not mix fresh Rust
484
+ figures with old Python results or claim unmeasured network-only latency shares.