llmshim 0.4.0__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (190) hide show
  1. {llmshim-0.4.0 → llmshim-0.5.0}/CLAUDE.md +84 -1
  2. {llmshim-0.4.0 → llmshim-0.5.0}/Cargo.lock +1 -1
  3. {llmshim-0.4.0 → llmshim-0.5.0}/Cargo.toml +1 -1
  4. {llmshim-0.4.0 → llmshim-0.5.0}/PKG-INFO +1 -1
  5. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/concepts/routing.md +47 -1
  6. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/guides/caching.md +13 -0
  7. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/guides/fallbacks.md +13 -1
  8. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/proxy/http-api.md +13 -1
  9. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/proxy/native-apis.md +12 -1
  10. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/proxy/scaling.md +40 -2
  11. {llmshim-0.4.0 → llmshim-0.5.0}/llmshim/types.py +6 -1
  12. llmshim-0.5.0/src/breaker.rs +454 -0
  13. {llmshim-0.4.0 → llmshim-0.5.0}/src/client.rs +16 -7
  14. {llmshim-0.4.0 → llmshim-0.5.0}/src/config.rs +40 -1
  15. llmshim-0.5.0/src/cost.rs +272 -0
  16. {llmshim-0.4.0 → llmshim-0.5.0}/src/fallback.rs +28 -2
  17. {llmshim-0.4.0 → llmshim-0.5.0}/src/gateway/auth.rs +17 -0
  18. {llmshim-0.4.0 → llmshim-0.5.0}/src/gateway/distributed.rs +29 -0
  19. {llmshim-0.4.0 → llmshim-0.5.0}/src/gateway/http.rs +52 -14
  20. llmshim-0.5.0/src/gateway/quota.rs +354 -0
  21. {llmshim-0.4.0 → llmshim-0.5.0}/src/lib.rs +27 -2
  22. {llmshim-0.4.0 → llmshim-0.5.0}/src/log.rs +9 -0
  23. {llmshim-0.4.0 → llmshim-0.5.0}/src/proxy/convert.rs +37 -0
  24. {llmshim-0.4.0 → llmshim-0.5.0}/src/proxy/handlers.rs +4 -15
  25. llmshim-0.5.0/src/proxy/health.rs +208 -0
  26. {llmshim-0.4.0 → llmshim-0.5.0}/src/proxy/mod.rs +4 -0
  27. {llmshim-0.4.0 → llmshim-0.5.0}/src/proxy/types.rs +29 -0
  28. {llmshim-0.4.0 → llmshim-0.5.0}/src/proxy/wire/mod.rs +25 -3
  29. {llmshim-0.4.0 → llmshim-0.5.0}/src/router.rs +107 -1
  30. {llmshim-0.4.0 → llmshim-0.5.0}/src/shim.rs +3 -0
  31. {llmshim-0.4.0 → llmshim-0.5.0}/src/usage.rs +79 -0
  32. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_fallback.rs +80 -0
  33. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_proxy.rs +60 -0
  34. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_proxy_convert.rs +3 -0
  35. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_router.rs +115 -0
  36. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_wire.rs +51 -0
  37. llmshim-0.4.0/src/gateway/quota.rs +0 -147
  38. {llmshim-0.4.0 → llmshim-0.5.0}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
  39. {llmshim-0.4.0 → llmshim-0.5.0}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
  40. {llmshim-0.4.0 → llmshim-0.5.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  41. {llmshim-0.4.0 → llmshim-0.5.0}/.github/workflows/catalog-refresh.yml +0 -0
  42. {llmshim-0.4.0 → llmshim-0.5.0}/.github/workflows/pages.yml +0 -0
  43. {llmshim-0.4.0 → llmshim-0.5.0}/.github/workflows/release.yml +0 -0
  44. {llmshim-0.4.0 → llmshim-0.5.0}/.gitignore +0 -0
  45. {llmshim-0.4.0 → llmshim-0.5.0}/CODE_OF_CONDUCT.md +0 -0
  46. {llmshim-0.4.0 → llmshim-0.5.0}/CONTRIBUTING.md +0 -0
  47. {llmshim-0.4.0 → llmshim-0.5.0}/LICENSE-APACHE +0 -0
  48. {llmshim-0.4.0 → llmshim-0.5.0}/LICENSE-MIT +0 -0
  49. {llmshim-0.4.0 → llmshim-0.5.0}/NOTICE +0 -0
  50. {llmshim-0.4.0 → llmshim-0.5.0}/README.md +0 -0
  51. {llmshim-0.4.0 → llmshim-0.5.0}/SECURITY.md +0 -0
  52. {llmshim-0.4.0 → llmshim-0.5.0}/benchmarks/bench.rs +0 -0
  53. {llmshim-0.4.0 → llmshim-0.5.0}/benchmarks/bench_python.py +0 -0
  54. {llmshim-0.4.0 → llmshim-0.5.0}/benchmarks/gateway_loadtest.rs +0 -0
  55. {llmshim-0.4.0 → llmshim-0.5.0}/benchmarks/loadtest.rs +0 -0
  56. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/Cargo.toml +0 -0
  57. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/LICENSE-APACHE +0 -0
  58. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/LICENSE-MIT +0 -0
  59. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/README.md +0 -0
  60. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/data/LICENSE.models.dev +0 -0
  61. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/data/README.md +0 -0
  62. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/data/models.dev.json +0 -0
  63. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/src/aliases.rs +0 -0
  64. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/src/builtin.rs +0 -0
  65. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/src/capabilities.rs +0 -0
  66. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/src/lib.rs +0 -0
  67. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/src/merge.rs +0 -0
  68. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/src/parse.rs +0 -0
  69. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/src/refresh.rs +0 -0
  70. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/src/types.rs +0 -0
  71. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/tests/catalog.rs +0 -0
  72. {llmshim-0.4.0 → llmshim-0.5.0}/crates/llmshim-catalog/tests/refresh.rs +0 -0
  73. {llmshim-0.4.0 → llmshim-0.5.0}/docs/.gitignore +0 -0
  74. {llmshim-0.4.0 → llmshim-0.5.0}/docs/book.toml +0 -0
  75. {llmshim-0.4.0 → llmshim-0.5.0}/docs/mermaid-init.js +0 -0
  76. {llmshim-0.4.0 → llmshim-0.5.0}/docs/mermaid.min.js +0 -0
  77. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/SUMMARY.md +0 -0
  78. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/concepts/contracts.md +0 -0
  79. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/concepts/conversations.md +0 -0
  80. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/concepts/portability.md +0 -0
  81. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/concepts/translation-flow.md +0 -0
  82. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/guides/capabilities.md +0 -0
  83. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/guides/images.md +0 -0
  84. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/guides/native-controls.md +0 -0
  85. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/guides/reasoning.md +0 -0
  86. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/guides/schemas.md +0 -0
  87. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/guides/streaming.md +0 -0
  88. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/guides/tools.md +0 -0
  89. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/introduction.md +0 -0
  90. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/proxy/deployment.md +0 -0
  91. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/reference/api.md +0 -0
  92. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/reference/cli.md +0 -0
  93. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/reference/configuration.md +0 -0
  94. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/reference/errors.md +0 -0
  95. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/reference/models.md +0 -0
  96. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/reference/providers.md +0 -0
  97. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/reference/request-fields.md +0 -0
  98. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/reference/surfaces.md +0 -0
  99. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/start/choose.md +0 -0
  100. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/start/cli.md +0 -0
  101. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/start/clients.md +0 -0
  102. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/start/configure.md +0 -0
  103. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/start/proxy.md +0 -0
  104. {llmshim-0.4.0 → llmshim-0.5.0}/docs/src/start/rust.md +0 -0
  105. {llmshim-0.4.0 → llmshim-0.5.0}/examples/chat.rs +0 -0
  106. {llmshim-0.4.0 → llmshim-0.5.0}/examples/stream.rs +0 -0
  107. {llmshim-0.4.0 → llmshim-0.5.0}/llmshim/__init__.py +0 -0
  108. {llmshim-0.4.0 → llmshim-0.5.0}/llmshim/_client.py +0 -0
  109. {llmshim-0.4.0 → llmshim-0.5.0}/llmshim/_server.py +0 -0
  110. {llmshim-0.4.0 → llmshim-0.5.0}/pyproject.toml +0 -0
  111. {llmshim-0.4.0 → llmshim-0.5.0}/src/cache.rs +0 -0
  112. {llmshim-0.4.0 → llmshim-0.5.0}/src/cli.rs +0 -0
  113. {llmshim-0.4.0 → llmshim-0.5.0}/src/env.rs +0 -0
  114. {llmshim-0.4.0 → llmshim-0.5.0}/src/error/normalize.rs +0 -0
  115. {llmshim-0.4.0 → llmshim-0.5.0}/src/error.rs +0 -0
  116. {llmshim-0.4.0 → llmshim-0.5.0}/src/gateway/idempotency.rs +0 -0
  117. {llmshim-0.4.0 → llmshim-0.5.0}/src/gateway/metrics.rs +0 -0
  118. {llmshim-0.4.0 → llmshim-0.5.0}/src/gateway/mod.rs +0 -0
  119. {llmshim-0.4.0 → llmshim-0.5.0}/src/main.rs +0 -0
  120. {llmshim-0.4.0 → llmshim-0.5.0}/src/models.rs +0 -0
  121. {llmshim-0.4.0 → llmshim-0.5.0}/src/provider.rs +0 -0
  122. {llmshim-0.4.0 → llmshim-0.5.0}/src/providers/anthropic.rs +0 -0
  123. {llmshim-0.4.0 → llmshim-0.5.0}/src/providers/anthropic_reasoning.rs +0 -0
  124. {llmshim-0.4.0 → llmshim-0.5.0}/src/providers/anthropic_signature.rs +0 -0
  125. {llmshim-0.4.0 → llmshim-0.5.0}/src/providers/chatgpt/auth.rs +0 -0
  126. {llmshim-0.4.0 → llmshim-0.5.0}/src/providers/chatgpt/mod.rs +0 -0
  127. {llmshim-0.4.0 → llmshim-0.5.0}/src/providers/chatgpt/streaming.rs +0 -0
  128. {llmshim-0.4.0 → llmshim-0.5.0}/src/providers/gemini.rs +0 -0
  129. {llmshim-0.4.0 → llmshim-0.5.0}/src/providers/mod.rs +0 -0
  130. {llmshim-0.4.0 → llmshim-0.5.0}/src/providers/openai.rs +0 -0
  131. {llmshim-0.4.0 → llmshim-0.5.0}/src/providers/openai_compat.rs +0 -0
  132. {llmshim-0.4.0 → llmshim-0.5.0}/src/providers/openrouter.rs +0 -0
  133. {llmshim-0.4.0 → llmshim-0.5.0}/src/providers/xai.rs +0 -0
  134. {llmshim-0.4.0 → llmshim-0.5.0}/src/proxy/error.rs +0 -0
  135. {llmshim-0.4.0 → llmshim-0.5.0}/src/proxy/ratelimit.rs +0 -0
  136. {llmshim-0.4.0 → llmshim-0.5.0}/src/proxy/wire/receipts.rs +0 -0
  137. {llmshim-0.4.0 → llmshim-0.5.0}/src/reasoning/normalize.rs +0 -0
  138. {llmshim-0.4.0 → llmshim-0.5.0}/src/reasoning.rs +0 -0
  139. {llmshim-0.4.0 → llmshim-0.5.0}/src/schema/memo.rs +0 -0
  140. {llmshim-0.4.0 → llmshim-0.5.0}/src/schema/mod.rs +0 -0
  141. {llmshim-0.4.0 → llmshim-0.5.0}/src/schema/validate.rs +0 -0
  142. {llmshim-0.4.0 → llmshim-0.5.0}/src/schema/walk.rs +0 -0
  143. {llmshim-0.4.0 → llmshim-0.5.0}/src/streaming.rs +0 -0
  144. {llmshim-0.4.0 → llmshim-0.5.0}/src/toolcall/streaming.rs +0 -0
  145. {llmshim-0.4.0 → llmshim-0.5.0}/src/toolcall.rs +0 -0
  146. {llmshim-0.4.0 → llmshim-0.5.0}/src/vision.rs +0 -0
  147. {llmshim-0.4.0 → llmshim-0.5.0}/tests/fixtures/chatgpt-red.png +0 -0
  148. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration.rs +0 -0
  149. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_chatgpt.rs +0 -0
  150. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_chatgpt_proxy.rs +0 -0
  151. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_current_models.rs +0 -0
  152. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_fallback.rs +0 -0
  153. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_gemini.rs +0 -0
  154. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_gemini_tools.rs +0 -0
  155. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_long_context.rs +0 -0
  156. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_multimodel.rs +0 -0
  157. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_openrouter.rs +0 -0
  158. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_proxy.rs +0 -0
  159. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_sglang.rs +0 -0
  160. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_thinking.rs +0 -0
  161. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_tool_roundtrip.rs +0 -0
  162. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_vision.rs +0 -0
  163. {llmshim-0.4.0 → llmshim-0.5.0}/tests/integration_xai.rs +0 -0
  164. {llmshim-0.4.0 → llmshim-0.5.0}/tests/support/completion_status.rs +0 -0
  165. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_advertised_models.rs +0 -0
  166. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_anthropic.rs +0 -0
  167. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_cache.rs +0 -0
  168. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_chatgpt.rs +0 -0
  169. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_cli.rs +0 -0
  170. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_fable.rs +0 -0
  171. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_fast_mode.rs +0 -0
  172. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_gemini.rs +0 -0
  173. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_log.rs +0 -0
  174. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_models.rs +0 -0
  175. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_multimodel.rs +0 -0
  176. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_openai.rs +0 -0
  177. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_openai_compat.rs +0 -0
  178. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_openrouter.rs +0 -0
  179. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_provider_contracts.rs +0 -0
  180. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_reasoning.rs +0 -0
  181. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_reasoning_profile.rs +0 -0
  182. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_schema.rs +0 -0
  183. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_shim.rs +0 -0
  184. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_signature.rs +0 -0
  185. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_sse.rs +0 -0
  186. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_toolcall.rs +0 -0
  187. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_tools.rs +0 -0
  188. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_usage.rs +0 -0
  189. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_vision.rs +0 -0
  190. {llmshim-0.4.0 → llmshim-0.5.0}/tests/unit_xai.rs +0 -0
@@ -53,6 +53,35 @@ refresh in a completion; hold a snapshot for decisions that must agree.
53
53
  normalized responses and usage chunks, logs, and proxy usage. Native token
54
54
  fields remain readable. `src/usage.rs` owns extraction; the streaming client
55
55
  merges Anthropic's start/delta usage before normalizing terminal counts.
56
+ `usage.uncached_input_tokens` joins them: the providers disagree on whether
57
+ their prompt total already includes the cache read (Anthropic's excludes it,
58
+ OpenAI/Chat Completions/Gemini include it), and only the transport boundary
59
+ still knows which convention the body used. The convention is read off the
60
+ cache-read field that actually matched, never a provider-name table.
61
+
62
+ ### USD cost accounting
63
+
64
+ `src/cost.rs` is the only thing that multiplies a catalog `Cost` (USD per
65
+ million tokens) by those counters. Input is charged on `uncached_input_tokens`
66
+ so a cached prompt is never billed twice; cache reads and writes are charged at
67
+ their own rates; reasoning tokens are already inside `completion_tokens`.
68
+ **An absent price is `None`, never `0.0`** — a positive count in a bucket with
69
+ no rate poisons the whole total rather than producing a partial sum that reads
70
+ as a complete one. `client.rs` stamps `usage.cost_usd` at the transport
71
+ boundary (the last place that knows the dispatch target) for completions and
72
+ for whichever stream chunk carries usage; `log.rs`, `proxy::types::Usage`, both
73
+ native facades and the four bundled clients carry it through as a nullable
74
+ field. `null` means unknown, not free.
75
+
76
+ `src/gateway/quota.rs` adds a per-identity dollar cap beside the RPM/TPM
77
+ buckets: `budget_usd` + `budget_window_secs` on an `Identity`, checked before
78
+ dispatch and charged after (cost is only knowable once a response exists, so
79
+ one in-flight request can overshoot). Windows tumble rather than slide, because
80
+ the fleet-wide store is one counter per window. `SpendCap::with_store` takes the
81
+ Redis-backed `DistributedGateway` in distributed mode so `$100/day` means one
82
+ hundred dollars fleet-wide, not per replica. **A response the catalog cannot
83
+ price is not charged** — recording zero would let an unpriced model run forever
84
+ under a budget; `cost_usd: null` is the signal that a price is missing.
56
85
  The native Chat Completions streams must use their own parser in the client;
57
86
  passing them to the Responses parser silently drops all events.
58
87
 
@@ -189,6 +218,49 @@ remain provider errors. Callers must configure the final URL directly.
189
218
 
190
219
  `FallbackConfig` defines an ordered list of models to try. On retryable errors (429, 500, 502, 503, 529), retries with exponential backoff then falls through to the next model. `completion_with_fallback()` is the top-level API. The proxy supports this via `"fallback": ["model1", "model2"]` in the request body.
191
220
 
221
+ ### Provider health (`src/breaker.rs`, `src/proxy/health.rs`)
222
+
223
+ **Health is not rate-limit backoff.** The token buckets already slow a provider
224
+ down after a 429 — a 429 means the provider is alive and asking for less. The
225
+ breaker counts what retrying cannot fix: 5xx (500/502/503/504/529) and
226
+ transport failures. Adapted from `rcode-provider`'s `ProviderBreaker`, which we
227
+ own: sliding failure window, open state, and a single half-open probe admitted
228
+ after the cooldown. Config: `LLMSHIM_BREAKER_WINDOW_SECS` (60),
229
+ `LLMSHIM_BREAKER_TRIP_THRESHOLD` (3; `0` disables), `LLMSHIM_BREAKER_COOLDOWN_SECS` (30).
230
+
231
+ The breaker hangs on the `Router` (`Router::breaker()` / `with_breaker`). Every
232
+ dispatch path *observes* outcomes so health accrues from ordinary traffic; only
233
+ `fallback.rs` *refuses*, and it checks before every attempt rather than once per
234
+ chain entry — the attempt that opens a circuit is usually the chain's own, so a
235
+ per-entry check would still retry into a target it just watched die. A single-target call is still dispatched:
236
+ with no alternative, refusing would only convert an upstream failure into a
237
+ local one. `proxy::health::build_breaker()` attaches a Redis-coordinated
238
+ `SharedHealth` (failure ZSET + open marker + `SET NX` probe) when
239
+ `LLMSHIM_REDIS_URL` is set and `redis-coordination` is compiled in, mirroring
240
+ `build_limiter`. Every shared operation **fails open**.
241
+
242
+ ### Named routes (`src/config.rs`, `src/router.rs`)
243
+
244
+ A caller-defined name maps to a model plus request settings:
245
+
246
+ ```toml
247
+ [routes.compaction]
248
+ model = "anthropic/claude-haiku-4-5-20251001"
249
+ reasoning_effort = "low"
250
+ max_tokens = 4096
251
+ ```
252
+
253
+ Addressed as `"model": "route/compaction"`, so a route travels through the
254
+ existing `provider/model` grammar — an OpenAI SDK, the CLI and the proxy's
255
+ admission control all handle it without learning a new field. `resolve_key`
256
+ resolves the indirection, so rate limiting never sees an unrecognized string.
257
+
258
+ **llmshim must not learn harness vocabulary.** The name is opaque: a harness may
259
+ call a route `compaction`, `advisor` or `webSearch`, and llmshim only knows it
260
+ maps to a model. Route settings are defaults — a per-request key always wins —
261
+ and an unknown name is a 400, never a silent fall back to the default model.
262
+ Routes do not chain.
263
+
192
264
  ### Vision (`src/vision.rs`)
193
265
 
194
266
  Image content blocks are translated between providers automatically. Users can send images in any format (OpenAI `image_url`, Anthropic `image`, Gemini `inline_data`) and the correct provider sees its native format. Base64 data URIs and plain URLs are both handled. Gemini falls back to a text placeholder for URL images (only supports `inline_data`).
@@ -303,7 +375,13 @@ HTTP proxy with our own API spec (not OpenAI-compatible). Built on axum.
303
375
  Endpoints:
304
376
  - `POST /v1/chat` — non-streaming (or streaming if `stream: true`)
305
377
  - `POST /v1/chat/stream` — always SSE streaming with typed events (`content`, `reasoning`, `tool_call`, `usage`, `done`, `error`)
306
- - `GET /v1/models` — list available models (filtered to configured providers)
378
+ - `GET /v1/models` — list available models (filtered to configured providers).
379
+ Serves two audiences from one body: an OpenAI SDK reads the `object: "list"` /
380
+ `data[]` envelope (so `client.models.list()` works unmodified), llmshim's own
381
+ clients read `models[]`. Both issue the same request, so there is no path to
382
+ split on — the union *is* the split. `data[].id` is the routing id, requestable
383
+ back as `model`. Built once in `proxy::convert::models_response`, shared with
384
+ the gateway.
307
385
  - `GET /health` — health check with provider list
308
386
 
309
387
  Request format uses `config` for provider-agnostic settings and `provider_config` for raw passthrough. OpenAPI 3.1 spec at `api/openapi.yaml`.
@@ -444,6 +522,11 @@ and native endpoints. Keep display messages readable and source type/code/param
444
522
  metadata separate; do not move this logic back into a wire-only formatter.
445
523
  JSON responses carry native metadata in response extensions, and SSE errors carry
446
524
  an optional structured error object so native rendering remains lossless.
525
+ `n > 1` is refused rather than emulated: the OpenAI backend is the Responses
526
+ API (no `n`), Anthropic Messages and Gemini have no `n` either, and the
527
+ single-message proxy projects choice zero. The refusal is a correctly shaped
528
+ `{"error":{type,message,param:"n",code:"unsupported_parameter"}}`, carried
529
+ through `normalize_error` so both facades render it natively.
447
530
  Tests: `unit_wire`, gateway `http::native_tests`. Use a temporary receipt directory
448
531
  in tests; never put real signatures, credentials, or conversations in fixtures.
449
532
 
@@ -1180,7 +1180,7 @@ checksum = "6373607a59f0be73a39b6fe456b8192fcc3585f602af20751600e974dd455e77"
1180
1180
 
1181
1181
  [[package]]
1182
1182
  name = "llmshim"
1183
- version = "0.4.0"
1183
+ version = "0.5.0"
1184
1184
  dependencies = [
1185
1185
  "async-stream",
1186
1186
  "async-trait",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "llmshim"
3
- version = "0.4.0"
3
+ version = "0.5.0"
4
4
  edition = "2021"
5
5
  description = "Blazing fast LLM API translation layer in pure Rust"
6
6
  license = "MIT OR Apache-2.0"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llmshim
3
- Version: 0.4.0
3
+ Version: 0.5.0
4
4
  Classifier: Development Status :: 4 - Beta
5
5
  Classifier: Intended Audience :: Developers
6
6
  Classifier: License :: OSI Approved :: MIT License
@@ -70,11 +70,57 @@ perform the second lookup.
70
70
  Aliases are not currently configurable through the CLI, config file, proxy
71
71
  API, or language clients.
72
72
 
73
+ ## Named routes
74
+
75
+ An alias renames a model. A **named route** goes further: it maps a
76
+ caller-defined name to a model *plus* request settings, configured in
77
+ `~/.llmshim/config.toml`.
78
+
79
+ ```toml
80
+ [routes.compaction]
81
+ model = "anthropic/claude-haiku-4-5-20251001"
82
+ reasoning_effort = "low"
83
+ max_tokens = 4096
84
+ ```
85
+
86
+ Address it as a model:
87
+
88
+ ```json
89
+ {"model": "route/compaction", "messages": [{"role": "user", "content": "…"}]}
90
+ ```
91
+
92
+ Because it reuses the `provider/model` grammar, a route works everywhere a
93
+ model address does — the Rust API, the CLI, the proxy, the native endpoints and
94
+ an unmodified OpenAI SDK.
95
+
96
+ The name is **opaque to llmshim**. A harness may call a route `compaction`,
97
+ `advisor` or `webSearch`; llmshim never interprets it and has no built-in role
98
+ vocabulary. The harness decides what a name means; llmshim provides only the
99
+ mechanism.
100
+
101
+ Three rules:
102
+
103
+ - **Settings are defaults.** A key the request already carries wins, so a
104
+ caller can pick a route and still raise `reasoning_effort` for one call.
105
+ - **An unknown name is an error** (HTTP 400), never a silent fall back to a
106
+ default model.
107
+ - **Routes do not chain.** A route's `model` may not be another `route/…`.
108
+
109
+ A Rust application can register routes directly:
110
+
111
+ ```rust
112
+ use llmshim::config::Route;
113
+
114
+ let router = llmshim::router::Router::new()
115
+ .route("compaction", Route { model: "anthropic/claude-haiku-4-5-20251001".into(), ..Default::default() });
116
+ ```
117
+
73
118
  ## Environment variables versus `config.toml`
74
119
 
75
120
  `Router::from_env()` reads provider environment variables such as
76
121
  `OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `GEMINI_API_KEY`, and `XAI_API_KEY`. It
77
- does not read `~/.llmshim/config.toml` by itself. It also discovers the selected
122
+ does not read API keys from `~/.llmshim/config.toml` by itself; it does read
123
+ that file's `[routes]` table, which has no environment equivalent. It also discovers the selected
78
124
  ChatGPT OAuth cache, whose default location is `~/.llmshim/chatgpt/auth.json`.
79
125
 
80
126
  The CLI and proxy call `llmshim::env::load_all()` before constructing their
@@ -46,6 +46,19 @@ Every response and usage event reports `cache_read_tokens` and
46
46
  add repeated streaming snapshots together. Zero means no cache tokens were
47
47
  reported, not that the request was free.
48
48
 
49
+ They are joined by `uncached_input_tokens` — the prompt minus whatever cache
50
+ read the provider already counted inside it. Providers disagree on that:
51
+ Anthropic's `input_tokens` excludes the cache read, while the OpenAI Responses,
52
+ Chat Completions and Gemini prompt totals include it. llmshim resolves the
53
+ disagreement at the transport boundary so one counter means one thing.
54
+
55
+ `cost_usd` is the USD charged for the response: uncached input, output, cache
56
+ reads and cache writes each at their own catalog rate, so a cached prompt is
57
+ never billed twice. **`null` means the catalog carries no price for the model —
58
+ it never means free.** A model priced for input but not for the cache reads a
59
+ response actually used also yields `null`, rather than a partial sum that would
60
+ read as a complete one.
61
+
49
62
  `ProviderRequest::can_continue_from` compares endpoint, credential headers,
50
63
  settings and the full prior input prefix. `include`, `store`, `reasoning`, tool
51
64
  schemas and other settings participate. The library sends full stateless
@@ -89,7 +89,19 @@ address. A success returns immediately. If every address fails, Rust returns
89
89
  `all_failed` error response.
90
90
 
91
91
  Each fallback model must resolve to a provider registered on the Router. A
92
- chain can cross providers, but only when their keys are configured.
92
+ chain can cross providers, but only when their keys are configured. A chain
93
+ entry may also be a [named route](../concepts/routing.md#named-routes).
94
+
95
+ ## Open circuits are skipped, not retried
96
+
97
+ A chain also consults provider health, before every attempt rather than once per
98
+ entry — the attempt that opens a circuit is usually the chain's own. When a
99
+ provider's circuit is open, because of too many `5xx`/transport failures in the
100
+ window, the chain abandons that entry, records `circuit open for provider …`
101
+ among the collected errors, and moves to the next address instead of spending
102
+ the rest of its retry budget on a target it already knows is dead. A `429` never
103
+ opens a circuit; that is rate limiting, handled by backoff. See
104
+ [Scaling and rate limits](../proxy/scaling.md#provider-health) for the knobs.
93
105
 
94
106
  ## Fallback is non-streaming only
95
107
 
@@ -146,16 +146,28 @@ consumption patterns, see [Streaming](../guides/streaming.md).
146
146
 
147
147
  ## Models and health
148
148
 
149
- `GET /v1/models` returns the registry entries for configured providers:
149
+ `GET /v1/models` returns the registry entries for configured providers, in two
150
+ shapes from one body:
150
151
 
151
152
  ```json
152
153
  {
154
+ "object": "list",
155
+ "data": [
156
+ {"id": "openai/gpt-5.6-terra", "object": "model", "created": 0, "owned_by": "openai"}
157
+ ],
153
158
  "models": [
154
159
  {"id": "openai/gpt-5.6-terra", "provider": "openai", "name": "gpt-5.6-terra"}
155
160
  ]
156
161
  }
157
162
  ```
158
163
 
164
+ `object` + `data` is the OpenAI list envelope, so an OpenAI SDK's
165
+ `client.models.list()` works against llmshim unmodified. `models` is llmshim's
166
+ own shape and is unchanged. Both audiences issue the same `GET /v1/models`, so
167
+ there is no second path to split on. A `data[].id` is the routing id and can be
168
+ sent straight back as `model`; `created` is the catalog release date as a Unix
169
+ timestamp, or `0` when unknown.
170
+
159
171
  This is discovery, not an allowlist: an arbitrary provider model ID can still
160
172
  be routed explicitly. See [Model discovery](../reference/models.md).
161
173
 
@@ -33,7 +33,18 @@ its standard message/function shapes and `response_format`; JSON-object mode als
33
33
  gets object validation. Generation fields supported by the destination retain
34
34
  their ordinary adapter behavior. The facade supports one completion (`n:1`) and
35
35
  custom function tools; it rejects provider-hosted tool definitions rather than
36
- turning them into client-executed functions. Provider-specific endpoints such as
36
+ turning them into client-executed functions. `n > 1` is refused rather than
37
+ emulated — the OpenAI backend is the Responses API, which has no `n`, and
38
+ neither do Anthropic Messages or Gemini; fanning out N requests would change
39
+ the cost, rate-limit footprint and cache behavior of what was asked for. The
40
+ refusal is a properly shaped OpenAI error naming the parameter:
41
+
42
+ ```json
43
+ {"error": {"message": "Unsupported value: 'n' must be 1. …",
44
+ "type": "invalid_request_error", "param": "n",
45
+ "code": "unsupported_parameter"}}
46
+ ```
47
+ Provider-specific endpoints such as
37
48
  batches, token counting, uploads and Responses are not served by these aliases.
38
49
 
39
50
  With `stream:true`, text arrives incrementally. Chat Completions emits
@@ -75,6 +75,42 @@ When neither RPM nor TPM is set, proactive rate limiting is disabled;
75
75
  concurrency backpressure still applies. Token permits are estimates based on
76
76
  request size and requested output, not provider billing measurements.
77
77
 
78
+ ## Provider health
79
+
80
+ Rate limiting and health are different questions. A `429` means the provider is
81
+ alive and asking for less, and the token buckets already slow it down. A
82
+ circuit breaker counts what retrying cannot fix — `500`, `502`, `503`, `504`,
83
+ `529` and transport failures — over a sliding window, opens the circuit at the
84
+ threshold, and admits a single probe after the cooldown.
85
+
86
+ | Variable | Default | Meaning |
87
+ |---|---:|---|
88
+ | `LLMSHIM_BREAKER_WINDOW_SECS` | `60` | Sliding window over which failures are counted |
89
+ | `LLMSHIM_BREAKER_TRIP_THRESHOLD` | `3` | Failures that open a circuit; `0` disables the breaker |
90
+ | `LLMSHIM_BREAKER_COOLDOWN_SECS` | `30` | Time an open circuit waits before admitting a probe |
91
+
92
+ Every dispatch path *observes* outcomes, so health accrues from ordinary
93
+ traffic. Only a [fallback chain](../guides/fallbacks.md) *refuses*: it skips a
94
+ provider with an open circuit instead of spending its retry budget on a target
95
+ it already knows is dead. A single-target request is still dispatched — with no
96
+ alternative, refusing would only convert an upstream failure into a local one.
97
+
98
+ ## Spend caps
99
+
100
+ The experimental gateway enforces a per-identity USD cap beside the RPM/TPM
101
+ buckets. A gateway key's identity may carry `budget_usd` and an optional
102
+ `budget_window_secs` (default one day); over budget is a `429` with
103
+ `Retry-After` set to the window reset.
104
+
105
+ ```json
106
+ {"sk-example": {"tenant": "acme", "tier": 1, "budget_usd": 100, "budget_window_secs": 86400}}
107
+ ```
108
+
109
+ Cost is only knowable after a response, so the cap is checked before dispatch
110
+ and charged after: one in-flight request can overshoot. A response the catalog
111
+ cannot price (`cost_usd: null`) is **not** charged — recording zero would let an
112
+ unpriced model run forever under a budget, so a hard cap requires priced models.
113
+
78
114
  ## One replica or a coordinated fleet
79
115
 
80
116
  The default buckets are in memory. With `N` replicas, each replica enforces
@@ -88,8 +124,10 @@ cargo install llmshim --features redis-coordination
88
124
  LLMSHIM_REDIS_URL=redis://redis.internal:6379 llmshim proxy
89
125
  ```
90
126
 
91
- `redis-coordination` includes the `proxy` feature. Redis is used for rate-limit
92
- coordination; connection pools and concurrency limits remain per process. If
127
+ `redis-coordination` includes the `proxy` feature. Redis coordinates rate-limit
128
+ buckets, provider health and — on the gateway — spend, so a shared limit, a
129
+ dead provider and a dollar cap all mean the same thing on every replica;
130
+ connection pools and concurrency limits remain per process. If
93
131
  Redis becomes unavailable at runtime, limiting fails open so requests continue.
94
132
  If the Redis client cannot be initialized—or the binary lacks the feature—the
95
133
  proxy warns and falls back to in-memory buckets.
@@ -11,7 +11,7 @@ the module stays compatible with Python 3.9 (no ``typing.NotRequired``).
11
11
 
12
12
  from __future__ import annotations
13
13
 
14
- from typing import Any, List, Literal, TypedDict, Union
14
+ from typing import Any, List, Literal, Optional, TypedDict, Union
15
15
 
16
16
  __all__ = [
17
17
  "Role",
@@ -190,6 +190,9 @@ class Usage(TypedDict, total=False):
190
190
  total_tokens: int
191
191
  cache_read_tokens: int
192
192
  cache_write_tokens: int
193
+ #: USD charged for this response. ``None`` means the server could not price
194
+ #: the model — it never means free.
195
+ cost_usd: Optional[float]
193
196
 
194
197
 
195
198
  class _ResponseMessageBase(TypedDict):
@@ -285,6 +288,8 @@ class UsageEvent(TypedDict, total=False):
285
288
  total_tokens: int
286
289
  cache_read_tokens: int
287
290
  cache_write_tokens: int
291
+ #: ``None`` when the server could not price the model, never ``0.0``.
292
+ cost_usd: Optional[float]
288
293
 
289
294
 
290
295
  class _DoneBase(TypedDict):