llmshim 0.6.0__tar.gz → 0.7.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (190) hide show
  1. {llmshim-0.6.0 → llmshim-0.7.1}/CLAUDE.md +10 -2
  2. {llmshim-0.6.0 → llmshim-0.7.1}/Cargo.lock +1 -1
  3. {llmshim-0.6.0 → llmshim-0.7.1}/Cargo.toml +1 -1
  4. {llmshim-0.6.0 → llmshim-0.7.1}/PKG-INFO +1 -1
  5. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/caching.md +8 -1
  6. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/proxy/scaling.md +12 -5
  7. {llmshim-0.6.0 → llmshim-0.7.1}/src/cache.rs +55 -0
  8. {llmshim-0.6.0 → llmshim-0.7.1}/src/client.rs +76 -1
  9. {llmshim-0.6.0 → llmshim-0.7.1}/src/cost.rs +57 -11
  10. {llmshim-0.6.0 → llmshim-0.7.1}/src/fallback.rs +3 -6
  11. {llmshim-0.6.0 → llmshim-0.7.1}/src/lib.rs +11 -14
  12. {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/anthropic.rs +11 -0
  13. {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/gemini.rs +6 -0
  14. {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/openai.rs +7 -0
  15. {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/openai_compat.rs +10 -0
  16. {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/openrouter.rs +8 -0
  17. {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/xai.rs +7 -0
  18. {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/types.rs +6 -1
  19. llmshim-0.7.1/tests/unit_client_breaker.rs +118 -0
  20. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_provider_contracts.rs +53 -0
  21. {llmshim-0.6.0 → llmshim-0.7.1}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
  22. {llmshim-0.6.0 → llmshim-0.7.1}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
  23. {llmshim-0.6.0 → llmshim-0.7.1}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  24. {llmshim-0.6.0 → llmshim-0.7.1}/.github/workflows/catalog-refresh.yml +0 -0
  25. {llmshim-0.6.0 → llmshim-0.7.1}/.github/workflows/pages.yml +0 -0
  26. {llmshim-0.6.0 → llmshim-0.7.1}/.github/workflows/release.yml +0 -0
  27. {llmshim-0.6.0 → llmshim-0.7.1}/.gitignore +0 -0
  28. {llmshim-0.6.0 → llmshim-0.7.1}/CODE_OF_CONDUCT.md +0 -0
  29. {llmshim-0.6.0 → llmshim-0.7.1}/CONTRIBUTING.md +0 -0
  30. {llmshim-0.6.0 → llmshim-0.7.1}/LICENSE-APACHE +0 -0
  31. {llmshim-0.6.0 → llmshim-0.7.1}/LICENSE-MIT +0 -0
  32. {llmshim-0.6.0 → llmshim-0.7.1}/NOTICE +0 -0
  33. {llmshim-0.6.0 → llmshim-0.7.1}/README.md +0 -0
  34. {llmshim-0.6.0 → llmshim-0.7.1}/SECURITY.md +0 -0
  35. {llmshim-0.6.0 → llmshim-0.7.1}/benchmarks/bench.rs +0 -0
  36. {llmshim-0.6.0 → llmshim-0.7.1}/benchmarks/bench_python.py +0 -0
  37. {llmshim-0.6.0 → llmshim-0.7.1}/benchmarks/gateway_loadtest.rs +0 -0
  38. {llmshim-0.6.0 → llmshim-0.7.1}/benchmarks/loadtest.rs +0 -0
  39. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/Cargo.toml +0 -0
  40. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/LICENSE-APACHE +0 -0
  41. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/LICENSE-MIT +0 -0
  42. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/README.md +0 -0
  43. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/data/LICENSE.models.dev +0 -0
  44. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/data/README.md +0 -0
  45. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/data/models.dev.json +0 -0
  46. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/aliases.rs +0 -0
  47. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/builtin.rs +0 -0
  48. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/capabilities.rs +0 -0
  49. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/lib.rs +0 -0
  50. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/merge.rs +0 -0
  51. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/parse.rs +0 -0
  52. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/refresh.rs +0 -0
  53. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/src/types.rs +0 -0
  54. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/tests/catalog.rs +0 -0
  55. {llmshim-0.6.0 → llmshim-0.7.1}/crates/llmshim-catalog/tests/refresh.rs +0 -0
  56. {llmshim-0.6.0 → llmshim-0.7.1}/docs/.gitignore +0 -0
  57. {llmshim-0.6.0 → llmshim-0.7.1}/docs/book.toml +0 -0
  58. {llmshim-0.6.0 → llmshim-0.7.1}/docs/mermaid-init.js +0 -0
  59. {llmshim-0.6.0 → llmshim-0.7.1}/docs/mermaid.min.js +0 -0
  60. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/SUMMARY.md +0 -0
  61. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/concepts/contracts.md +0 -0
  62. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/concepts/conversations.md +0 -0
  63. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/concepts/portability.md +0 -0
  64. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/concepts/routing.md +0 -0
  65. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/concepts/translation-flow.md +0 -0
  66. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/capabilities.md +0 -0
  67. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/fallbacks.md +0 -0
  68. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/images.md +0 -0
  69. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/native-controls.md +0 -0
  70. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/reasoning.md +0 -0
  71. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/schemas.md +0 -0
  72. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/streaming.md +0 -0
  73. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/guides/tools.md +0 -0
  74. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/introduction.md +0 -0
  75. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/proxy/deployment.md +0 -0
  76. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/proxy/http-api.md +0 -0
  77. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/proxy/native-apis.md +0 -0
  78. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/api.md +0 -0
  79. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/cli.md +0 -0
  80. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/configuration.md +0 -0
  81. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/errors.md +0 -0
  82. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/models.md +0 -0
  83. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/providers.md +0 -0
  84. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/request-fields.md +0 -0
  85. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/reference/surfaces.md +0 -0
  86. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/start/choose.md +0 -0
  87. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/start/cli.md +0 -0
  88. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/start/clients.md +0 -0
  89. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/start/configure.md +0 -0
  90. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/start/proxy.md +0 -0
  91. {llmshim-0.6.0 → llmshim-0.7.1}/docs/src/start/rust.md +0 -0
  92. {llmshim-0.6.0 → llmshim-0.7.1}/examples/chat.rs +0 -0
  93. {llmshim-0.6.0 → llmshim-0.7.1}/examples/stream.rs +0 -0
  94. {llmshim-0.6.0 → llmshim-0.7.1}/llmshim/__init__.py +0 -0
  95. {llmshim-0.6.0 → llmshim-0.7.1}/llmshim/_client.py +0 -0
  96. {llmshim-0.6.0 → llmshim-0.7.1}/llmshim/_server.py +0 -0
  97. {llmshim-0.6.0 → llmshim-0.7.1}/llmshim/types.py +0 -0
  98. {llmshim-0.6.0 → llmshim-0.7.1}/pyproject.toml +0 -0
  99. {llmshim-0.6.0 → llmshim-0.7.1}/src/breaker.rs +0 -0
  100. {llmshim-0.6.0 → llmshim-0.7.1}/src/cli.rs +0 -0
  101. {llmshim-0.6.0 → llmshim-0.7.1}/src/config.rs +0 -0
  102. {llmshim-0.6.0 → llmshim-0.7.1}/src/env.rs +0 -0
  103. {llmshim-0.6.0 → llmshim-0.7.1}/src/error/normalize.rs +0 -0
  104. {llmshim-0.6.0 → llmshim-0.7.1}/src/error.rs +0 -0
  105. {llmshim-0.6.0 → llmshim-0.7.1}/src/gateway/auth.rs +0 -0
  106. {llmshim-0.6.0 → llmshim-0.7.1}/src/gateway/distributed.rs +0 -0
  107. {llmshim-0.6.0 → llmshim-0.7.1}/src/gateway/http.rs +0 -0
  108. {llmshim-0.6.0 → llmshim-0.7.1}/src/gateway/idempotency.rs +0 -0
  109. {llmshim-0.6.0 → llmshim-0.7.1}/src/gateway/metrics.rs +0 -0
  110. {llmshim-0.6.0 → llmshim-0.7.1}/src/gateway/mod.rs +0 -0
  111. {llmshim-0.6.0 → llmshim-0.7.1}/src/gateway/quota.rs +0 -0
  112. {llmshim-0.6.0 → llmshim-0.7.1}/src/log.rs +0 -0
  113. {llmshim-0.6.0 → llmshim-0.7.1}/src/main.rs +0 -0
  114. {llmshim-0.6.0 → llmshim-0.7.1}/src/models.rs +0 -0
  115. {llmshim-0.6.0 → llmshim-0.7.1}/src/provider.rs +0 -0
  116. {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/anthropic_reasoning.rs +0 -0
  117. {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/anthropic_signature.rs +0 -0
  118. {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/chatgpt/auth.rs +0 -0
  119. {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/chatgpt/mod.rs +0 -0
  120. {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/chatgpt/streaming.rs +0 -0
  121. {llmshim-0.6.0 → llmshim-0.7.1}/src/providers/mod.rs +0 -0
  122. {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/convert.rs +0 -0
  123. {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/error.rs +0 -0
  124. {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/handlers.rs +0 -0
  125. {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/health.rs +0 -0
  126. {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/mod.rs +0 -0
  127. {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/ratelimit.rs +0 -0
  128. {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/wire/mod.rs +0 -0
  129. {llmshim-0.6.0 → llmshim-0.7.1}/src/proxy/wire/receipts.rs +0 -0
  130. {llmshim-0.6.0 → llmshim-0.7.1}/src/reasoning/normalize.rs +0 -0
  131. {llmshim-0.6.0 → llmshim-0.7.1}/src/reasoning.rs +0 -0
  132. {llmshim-0.6.0 → llmshim-0.7.1}/src/router.rs +0 -0
  133. {llmshim-0.6.0 → llmshim-0.7.1}/src/schema/memo.rs +0 -0
  134. {llmshim-0.6.0 → llmshim-0.7.1}/src/schema/mod.rs +0 -0
  135. {llmshim-0.6.0 → llmshim-0.7.1}/src/schema/validate.rs +0 -0
  136. {llmshim-0.6.0 → llmshim-0.7.1}/src/schema/walk.rs +0 -0
  137. {llmshim-0.6.0 → llmshim-0.7.1}/src/shim.rs +0 -0
  138. {llmshim-0.6.0 → llmshim-0.7.1}/src/streaming.rs +0 -0
  139. {llmshim-0.6.0 → llmshim-0.7.1}/src/toolcall/streaming.rs +0 -0
  140. {llmshim-0.6.0 → llmshim-0.7.1}/src/toolcall.rs +0 -0
  141. {llmshim-0.6.0 → llmshim-0.7.1}/src/usage.rs +0 -0
  142. {llmshim-0.6.0 → llmshim-0.7.1}/src/vision.rs +0 -0
  143. {llmshim-0.6.0 → llmshim-0.7.1}/tests/fixtures/chatgpt-red.png +0 -0
  144. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration.rs +0 -0
  145. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_chatgpt.rs +0 -0
  146. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_chatgpt_proxy.rs +0 -0
  147. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_current_models.rs +0 -0
  148. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_fallback.rs +0 -0
  149. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_gemini.rs +0 -0
  150. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_gemini_tools.rs +0 -0
  151. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_long_context.rs +0 -0
  152. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_multimodel.rs +0 -0
  153. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_openrouter.rs +0 -0
  154. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_proxy.rs +0 -0
  155. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_sglang.rs +0 -0
  156. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_thinking.rs +0 -0
  157. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_tool_roundtrip.rs +0 -0
  158. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_vision.rs +0 -0
  159. {llmshim-0.6.0 → llmshim-0.7.1}/tests/integration_xai.rs +0 -0
  160. {llmshim-0.6.0 → llmshim-0.7.1}/tests/support/completion_status.rs +0 -0
  161. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_advertised_models.rs +0 -0
  162. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_anthropic.rs +0 -0
  163. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_cache.rs +0 -0
  164. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_chatgpt.rs +0 -0
  165. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_cli.rs +0 -0
  166. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_fable.rs +0 -0
  167. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_fallback.rs +0 -0
  168. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_fast_mode.rs +0 -0
  169. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_gemini.rs +0 -0
  170. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_log.rs +0 -0
  171. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_models.rs +0 -0
  172. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_multimodel.rs +0 -0
  173. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_openai.rs +0 -0
  174. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_openai_compat.rs +0 -0
  175. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_openrouter.rs +0 -0
  176. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_proxy.rs +0 -0
  177. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_proxy_convert.rs +0 -0
  178. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_reasoning.rs +0 -0
  179. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_reasoning_profile.rs +0 -0
  180. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_router.rs +0 -0
  181. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_schema.rs +0 -0
  182. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_shim.rs +0 -0
  183. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_signature.rs +0 -0
  184. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_sse.rs +0 -0
  185. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_toolcall.rs +0 -0
  186. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_tools.rs +0 -0
  187. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_usage.rs +0 -0
  188. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_vision.rs +0 -0
  189. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_wire.rs +0 -0
  190. {llmshim-0.6.0 → llmshim-0.7.1}/tests/unit_xai.rs +0 -0
@@ -228,8 +228,16 @@ own: sliding failure window, open state, and a single half-open probe admitted
228
228
  after the cooldown. Config: `LLMSHIM_BREAKER_WINDOW_SECS` (60),
229
229
  `LLMSHIM_BREAKER_TRIP_THRESHOLD` (3; `0` disables), `LLMSHIM_BREAKER_COOLDOWN_SECS` (30).
230
230
 
231
- The breaker hangs on the `Router` (`Router::breaker()` / `with_breaker`). Every
232
- dispatch path *observes* outcomes so health accrues from ordinary traffic; only
231
+ The breaker hangs on the `Router` (`Router::breaker()` / `with_breaker`), but
232
+ the *counting* happens in `ShimClient`: `ShimClient::with_breaker` attaches one,
233
+ and `completion` / `stream` / `stream_owned` observe their final result exactly
234
+ once. The top-level entry points bind the router's breaker to the shared client
235
+ per call (`lib.rs::bound_client`), so `llmshim::completion`, `stream`,
236
+ `completion_with_fallback` and a caller that resolves its own provider and dials
237
+ `ShimClient` directly all feed the same breaker — the last one only if it opted
238
+ in with `ShimClient::new().with_breaker(router.breaker().clone())`; a bare
239
+ `ShimClient::new()` reports to nobody. Do not add a second `.observe` around a
240
+ client call: one call, one observation (`tests/unit_client_breaker.rs`). Only
233
241
  `fallback.rs` *refuses*, and it checks before every attempt rather than once per
234
242
  chain entry — the attempt that opens a circuit is usually the chain's own, so a
235
243
  per-entry check would still retry into a target it just watched die. A single-target call is still dispatched:
@@ -1180,7 +1180,7 @@ checksum = "6373607a59f0be73a39b6fe456b8192fcc3585f602af20751600e974dd455e77"
1180
1180
 
1181
1181
  [[package]]
1182
1182
  name = "llmshim"
1183
- version = "0.6.0"
1183
+ version = "0.7.1"
1184
1184
  dependencies = [
1185
1185
  "async-stream",
1186
1186
  "async-trait",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "llmshim"
3
- version = "0.6.0"
3
+ version = "0.7.1"
4
4
  edition = "2021"
5
5
  description = "Blazing fast LLM API translation layer in pure Rust"
6
6
  license = "MIT OR Apache-2.0"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llmshim
3
- Version: 0.6.0
3
+ Version: 0.7.1
4
4
  Classifier: Development Status :: 4 - Beta
5
5
  Classifier: Intended Audience :: Developers
6
6
  Classifier: License :: OSI Approved :: MIT License
@@ -22,7 +22,14 @@ llmshim translates the declaration into the provider's caching mechanism.
22
22
  }
23
23
  ```
24
24
 
25
- `upto_message` is a zero-based index into the supplied messages. Labels are
25
+ `upto_message` is a zero-based index into the `messages` array exactly as you
26
+ sent it — system and developer messages count, and each `role: "tool"` result
27
+ counts as its own entry — not into the provider-native array (Anthropic hoists
28
+ the system prompt out, so native numbering differs). A caller whose own message
29
+ model expands into more wire messages than it holds must remap before copying
30
+ an index across. On Anthropic an index past the end is rejected with a 400,
31
+ and one that lands too early silently caches less than intended; on every
32
+ other provider segments are parsed but neither checked nor placed. Labels are
26
33
  informational and are never used to guess prompt semantics. Keep stable sections
27
34
  before volatile sections. The proxy accepts the same top-level `x-cache` field;
28
35
  Python and Ruby provide a `cache=`/`cache:` keyword, Go has `ChatRequest.Cache`,
@@ -106,11 +106,18 @@ buckets. A gateway key's identity may carry `budget_usd` and an optional
106
106
  {"sk-example": {"tenant": "acme", "tier": 1, "budget_usd": 100, "budget_window_secs": 86400}}
107
107
  ```
108
108
 
109
- Cost is only knowable after a response, so the cap is checked before dispatch
110
- and charged after: one in-flight request can overshoot.
111
-
112
- A response the catalog cannot price is **not** charged — recording zero would let
113
- an unpriced model run forever under a budget. So that a cap cannot silently stop
109
+ Cost is only knowable after a response, so the cap is checked before dispatch and
110
+ charged after. Everything admitted between the last charge and the next check
111
+ passes, so the overshoot bound is **admitted concurrency × the most expensive
112
+ request**, multiplied again across replicas that have not yet shared their
113
+ ledger. Size a cap with that headroom in mind rather than as a hard ceiling.
114
+
115
+ A response the catalog cannot price at all is **not** charged — recording zero
116
+ would let an unpriced model run forever under a budget. A model that prices only
117
+ *some* token classes is charged at its highest published rate for the rest, so a
118
+ partial price bounds the charge from above instead of voiding it: 2,537 of the
119
+ 7,461 priced models in the catalog publish no `cache_read` rate, and voiding
120
+ those would have reopened this same hole one layer down. So that a cap cannot silently stop
114
121
  binding, a request whose target has **no catalog price is refused before it runs**
115
122
  when a budget is set:
116
123
 
@@ -10,20 +10,75 @@ use std::collections::BTreeMap;
10
10
 
11
11
  pub const ANTHROPIC_BREAKPOINT_LIMIT: usize = 4;
12
12
 
13
+ /// How long the caller expects a prefix to stay byte-stable. Only the caller
14
+ /// knows; llmshim never infers it.
13
15
  #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
14
16
  #[serde(rename_all = "lowercase")]
15
17
  pub enum Stability {
18
+ /// Stable across sessions — a one-hour breakpoint on Anthropic.
16
19
  Static,
20
+ /// Stable for this session — a five-minute breakpoint on Anthropic.
17
21
  Session,
22
+ /// The volatile tail. Places no marker; it exists so a caller can name
23
+ /// where stability ends.
18
24
  Turn,
19
25
  }
26
+
27
+ /// One caller-declared stability boundary: every message up to and including
28
+ /// `upto_message` is expected to stay byte-stable for as long as `stability`
29
+ /// says.
30
+ ///
31
+ /// **`upto_message` indexes the request's own `messages` array, exactly as the
32
+ /// caller sent it.** Zero-based, and every entry counts — `system` and
33
+ /// `developer` messages included, and each `role: "tool"` result as its own
34
+ /// entry — because the annotation is applied before any adapter hoists the
35
+ /// system prompt out or reshapes tool results into native turns. It is *not*
36
+ /// an index into the provider-native array Anthropic receives, where the
37
+ /// system message is gone and the numbering has shifted.
38
+ ///
39
+ /// A caller whose own message model is richer than the wire's — one entry that
40
+ /// expands into a leading system message plus one wire message per tool
41
+ /// result, say — must remap to the index of the wire message it actually sent.
42
+ /// Copied across unchanged, the boundary lands on the wrong message: too far
43
+ /// and the request is rejected (`400 invalid x-cache: segment message index is
44
+ /// out of range`); too near and less of the prefix is cached than was hashed,
45
+ /// with nothing to say so.
46
+ ///
47
+ /// llmshim's own insertions never move this index. A managed-output
48
+ /// instruction merges into an existing system message, and when it has to
49
+ /// prepend one it renumbers every segment itself (`shim.rs`,
50
+ /// `prepend_instruction`).
51
+ ///
52
+ /// Honoured on the Anthropic Messages wire only, where it becomes a
53
+ /// `cache_control` breakpoint on the last block of that message (or the last
54
+ /// tool call, or the tool result itself) — unless that last block is a
55
+ /// thinking block, in which case the segment is skipped without a marker.
56
+ /// The bounds check is Anthropic-only too: on every other wire the policy is
57
+ /// still parsed (malformed `x-cache` fails everywhere), but a segment's index
58
+ /// is neither checked nor placed — an out-of-range index there is silently
59
+ /// ignored, not rejected. `label` is the caller's own tag; llmshim never
60
+ /// reads it.
20
61
  #[derive(Debug, Clone, Serialize, Deserialize)]
21
62
  pub struct CacheSegment {
63
+ /// Index into the request's `messages` as sent — see the type-level note.
22
64
  pub upto_message: usize,
23
65
  #[serde(default)]
24
66
  pub label: String,
25
67
  pub stability: Stability,
26
68
  }
69
+
70
+ /// The caller's `x-cache` annotation: stability boundaries plus an optional
71
+ /// cache identity. Read once per request and never forwarded to a provider.
72
+ ///
73
+ /// The two halves land on different wires. `segments` are Anthropic-only —
74
+ /// see [`CacheSegment`] for the index convention and the no-op rule elsewhere.
75
+ /// `key` is not: on the OpenAI Responses wire it becomes `prompt_cache_key`,
76
+ /// and is ignored on the others. A request may carry both; each wire takes the
77
+ /// half it can use.
78
+ ///
79
+ /// Explicit segments supersede any request-level `cache_control` the caller
80
+ /// placed, and Anthropic's four-breakpoint budget is spent from the end of the
81
+ /// request backwards, so the boundaries nearest the tail are the ones kept.
27
82
  #[derive(Debug, Clone, Default, Serialize, Deserialize)]
28
83
  pub struct CachePolicy {
29
84
  #[serde(default)]
@@ -1,3 +1,4 @@
1
+ use crate::breaker::ProviderBreaker;
1
2
  use crate::error::{Result, ShimError};
2
3
  use crate::provider::{Provider, ProviderRequest};
3
4
  use bytes::Bytes;
@@ -7,6 +8,7 @@ use futures::{Stream, StreamExt};
7
8
  use reqwest::header::HeaderMap;
8
9
  use reqwest::Client;
9
10
  use std::pin::Pin;
11
+ use std::sync::Arc;
10
12
  use std::time::Duration;
11
13
 
12
14
  /// Retry bounds, resolved once from the environment (with defaults) at
@@ -55,6 +57,9 @@ fn env_parse<T: std::str::FromStr>(key: &str) -> Option<T> {
55
57
  pub struct ShimClient {
56
58
  http: Client,
57
59
  retry: RetryConfig,
60
+ /// Provider health, fed by every dispatch this client makes. `None` means
61
+ /// this client reports to nobody — see [`ShimClient::with_breaker`].
62
+ breaker: Option<Arc<ProviderBreaker>>,
58
63
  }
59
64
 
60
65
  impl Default for ShimClient {
@@ -77,6 +82,36 @@ impl ShimClient {
77
82
  .build()
78
83
  .expect("failed to build HTTP client"),
79
84
  retry: RetryConfig::from_env(),
85
+ breaker: None,
86
+ }
87
+ }
88
+
89
+ /// Report every dispatch's outcome to `breaker`.
90
+ ///
91
+ /// The breaker lives on the [`Router`](crate::router::Router), so a caller
92
+ /// that resolves a provider itself and comes straight here has to hand it
93
+ /// over: `ShimClient::new().with_breaker(router.breaker().clone())`. The
94
+ /// crate's own entry points (`llmshim::completion`, `stream`,
95
+ /// `completion_with_fallback`) bind the router's breaker this way, so this
96
+ /// is the one place a dispatch is counted — whichever door it came in by.
97
+ ///
98
+ /// Cheap: the HTTP connection pool is shared by clone, so binding a breaker
99
+ /// per call costs an `Arc` clone, not a new pool.
100
+ pub fn with_breaker(mut self, breaker: Arc<ProviderBreaker>) -> Self {
101
+ self.breaker = Some(breaker);
102
+ self
103
+ }
104
+
105
+ /// Record one dispatch against the attached breaker, if any. Called once
106
+ /// per public entry point on its *final* result — after transport retries
107
+ /// and any output-contract repair — so one caller-visible call is one
108
+ /// observation, never one per attempt.
109
+ ///
110
+ /// Takes the projected outcome rather than the result itself so a stream's
111
+ /// non-`Sync` body is never borrowed across the await.
112
+ async fn observe(&self, provider: &dyn Provider, outcome: std::result::Result<(), &ShimError>) {
113
+ if let Some(breaker) = &self.breaker {
114
+ breaker.observe(provider.name(), outcome).await;
80
115
  }
81
116
  }
82
117
 
@@ -167,6 +202,17 @@ impl ShimClient {
167
202
  provider: &dyn Provider,
168
203
  model: &str,
169
204
  request: &serde_json::Value,
205
+ ) -> Result<serde_json::Value> {
206
+ let result = self.completion_unobserved(provider, model, request).await;
207
+ self.observe(provider, result.as_ref().map(|_| ())).await;
208
+ result
209
+ }
210
+
211
+ async fn completion_unobserved(
212
+ &self,
213
+ provider: &dyn Provider,
214
+ model: &str,
215
+ request: &serde_json::Value,
170
216
  ) -> Result<serde_json::Value> {
171
217
  let plan = crate::shim::Plan::new(
172
218
  provider.name(),
@@ -232,6 +278,19 @@ impl ShimClient {
232
278
  provider: &dyn Provider,
233
279
  model: &str,
234
280
  request: &serde_json::Value,
281
+ ) -> Result<Pin<Box<dyn Stream<Item = Result<String>> + Send>>> {
282
+ // A stream's health verdict is whether it opened; per-chunk failures
283
+ // are the transport's business, not the breaker's.
284
+ let opened = self.stream_unobserved(provider, model, request).await;
285
+ self.observe(provider, opened.as_ref().map(|_| ())).await;
286
+ opened
287
+ }
288
+
289
+ async fn stream_unobserved(
290
+ &self,
291
+ provider: &dyn Provider,
292
+ model: &str,
293
+ request: &serde_json::Value,
235
294
  ) -> Result<Pin<Box<dyn Stream<Item = Result<String>> + Send>>> {
236
295
  let plan = crate::shim::Plan::new(
237
296
  provider.name(),
@@ -278,7 +337,23 @@ impl ShimClient {
278
337
  /// and keepalives while validation and a possible repair are in progress.
279
338
  pub async fn stream_owned(
280
339
  &self,
281
- provider: std::sync::Arc<dyn Provider>,
340
+ provider: Arc<dyn Provider>,
341
+ model: &str,
342
+ request: &serde_json::Value,
343
+ ) -> Result<Pin<Box<dyn Stream<Item = Result<String>> + Send>>> {
344
+ // Observed on the open only. A buffered plan's repair re-opens inside
345
+ // the returned stream; that second dial is not a separate verdict.
346
+ let opened = self
347
+ .stream_owned_unobserved(provider.clone(), model, request)
348
+ .await;
349
+ self.observe(provider.as_ref(), opened.as_ref().map(|_| ()))
350
+ .await;
351
+ opened
352
+ }
353
+
354
+ async fn stream_owned_unobserved(
355
+ &self,
356
+ provider: Arc<dyn Provider>,
282
357
  model: &str,
283
358
  request: &serde_json::Value,
284
359
  ) -> Result<Pin<Box<dyn Stream<Item = Result<String>> + Send>>> {
@@ -29,6 +29,27 @@ fn charge(tokens: u64, rate: Option<f64>) -> Option<f64> {
29
29
  Some(tokens as f64 * rate? / PER_MILLION)
30
30
  }
31
31
 
32
+ /// The highest rate this model publishes, used where a class has none of its own.
33
+ ///
34
+ /// Catalogs price partially and often: **2,537 of 7,461 priced models in the
35
+ /// vendored snapshot carry no `cache_read` rate and 5,892 no `cache_write`**,
36
+ /// including `gpt-5-pro` and `o3-pro`, priced `{input, output}` only. Returning
37
+ /// `None` for those made the whole response unpriceable the moment it reported a
38
+ /// cached token — and a discarded charge is a spend cap that silently stops
39
+ /// binding, which is the failure this crate exists to prevent.
40
+ ///
41
+ /// Substituting the highest published rate can only over-estimate, never under.
42
+ /// For a cap that is the safe direction: it spends a budget slightly early, where
43
+ /// under-estimating spends it forever.
44
+ fn highest_published_rate(cost: &Cost) -> Option<f64> {
45
+ [cost.input, cost.output, cost.cache_read, cost.cache_write]
46
+ .into_iter()
47
+ .flatten()
48
+ .fold(None, |best: Option<f64>, rate| {
49
+ Some(best.map_or(rate, |b| b.max(rate)))
50
+ })
51
+ }
52
+
32
53
  /// Price one normalized usage object.
33
54
  ///
34
55
  /// Input is charged on `uncached_input_tokens` — the prompt minus whatever
@@ -53,11 +74,17 @@ pub fn price(usage: &Value, cost: &Cost) -> Option<f64> {
53
74
  return None;
54
75
  }
55
76
 
77
+ // A class the model does not price falls back to its highest published rate
78
+ // rather than voiding the total. See `highest_published_rate`: the result is
79
+ // an upper bound, which is the only safe direction for a budget.
80
+ let ceiling = highest_published_rate(cost);
81
+ let rate_for = |own: Option<f64>| own.or(ceiling);
82
+
56
83
  Some(
57
- charge(input, cost.input)?
58
- + charge(count(usage, "completion_tokens"), cost.output)?
59
- + charge(cache_read, cost.cache_read)?
60
- + charge(cache_write, cost.cache_write)?,
84
+ charge(input, rate_for(cost.input))?
85
+ + charge(count(usage, "completion_tokens"), rate_for(cost.output))?
86
+ + charge(cache_read, rate_for(cost.cache_read))?
87
+ + charge(cache_write, rate_for(cost.cache_write))?,
61
88
  )
62
89
  }
63
90
 
@@ -221,14 +248,21 @@ mod tests {
221
248
  }
222
249
 
223
250
  #[test]
224
- fn a_missing_cache_rate_poisons_the_total_only_when_it_is_used() {
251
+ fn a_missing_cache_rate_is_charged_at_the_models_highest_rate() {
252
+ // This replaced an assertion that a used-but-unpriced class voids the
253
+ // total. That instinct was right — never under-report — but the remedy
254
+ // was wrong: `None` is discarded by `SpendCap::record`, so a spend cap
255
+ // silently stopped binding for the 2,537 catalogued models that publish
256
+ // no `cache_read` rate. An upper bound honours the same principle
257
+ // without the hole.
225
258
  let partial = Cost {
226
259
  input: Some(3.0),
227
260
  output: Some(15.0),
228
261
  cache_read: None,
229
262
  cache_write: None,
230
263
  };
231
- // No cache tokens: the missing rates are irrelevant.
264
+
265
+ // No cache tokens: the missing rates never come into play.
232
266
  close(
233
267
  price(
234
268
  &json!({"uncached_input_tokens": 1_000_000, "completion_tokens": 0, "cache_read_tokens": 0}),
@@ -236,14 +270,26 @@ mod tests {
236
270
  ),
237
271
  3.0,
238
272
  );
239
- // Cache tokens actually used, with no rate to charge them at.
240
- assert_eq!(
273
+
274
+ // Cache tokens used with no rate of their own: charged at 15.0, the
275
+ // highest rate this model publishes — an over-estimate, never an under.
276
+ close(
241
277
  price(
242
- &json!({"uncached_input_tokens": 1_000_000, "cache_read_tokens": 1}),
278
+ &json!({"uncached_input_tokens": 1_000_000, "cache_read_tokens": 1_000_000}),
243
279
  &partial,
244
280
  ),
245
- None,
246
- "an unpriced bucket that was used must not round down to a partial sum"
281
+ 18.0,
282
+ );
283
+
284
+ // The bound must never fall below what a complete price would charge.
285
+ let complete = Cost {
286
+ cache_read: Some(0.3),
287
+ ..partial
288
+ };
289
+ let usage = json!({"uncached_input_tokens": 1_000_000, "cache_read_tokens": 1_000_000});
290
+ assert!(
291
+ price(&usage, &partial).unwrap() >= price(&usage, &complete).unwrap(),
292
+ "a fallback rate must bound the real one from above, or a cap under-charges"
247
293
  );
248
294
  }
249
295
 
@@ -1,7 +1,6 @@
1
1
  use crate::error::{Result, ShimError};
2
2
  use crate::log::{LogEntry, Logger, RequestTimer};
3
3
  use crate::router::Router;
4
- use crate::SHARED_CLIENT;
5
4
  use serde_json::Value;
6
5
  use std::time::Duration;
7
6
 
@@ -75,7 +74,9 @@ pub async fn completion_with_fallback(
75
74
  };
76
75
 
77
76
  let mut errors: Vec<String> = Vec::new();
78
- let client = &*SHARED_CLIENT;
77
+ // Every attempt below is counted by the client against this router's
78
+ // breaker; the loop only asks `admit` before dialling.
79
+ let client = crate::bound_client(router);
79
80
 
80
81
  for model_str in &models {
81
82
  // Build request with this model. A named route expands to its model and
@@ -118,10 +119,6 @@ pub async fn completion_with_fallback(
118
119
  // Keep OAuth preparation, SSE-only providers, reasoning provenance,
119
120
  // and tool normalization identical to an ordinary completion.
120
121
  let outcome = client.completion(provider, &model, &req).await;
121
- router
122
- .breaker()
123
- .observe(provider.name(), outcome.as_ref().map(|_| ()))
124
- .await;
125
122
  match outcome {
126
123
  Ok(result) => {
127
124
  if let Some(logger) = logger {
@@ -79,16 +79,13 @@ pub async fn completion_with_logger(
79
79
  .ok_or(error::ShimError::MissingModel)?;
80
80
 
81
81
  let (provider, model) = router.resolve(model_str)?;
82
- let client = &*SHARED_CLIENT;
82
+ let client = bound_client(router);
83
83
  let timer = RequestTimer::start();
84
84
 
85
85
  // Ordinary traffic feeds provider health too, so a chain's first fallback
86
86
  // decision is not the first thing that ever noticed a provider is down.
87
+ // The client does the counting; see `ShimClient::with_breaker`.
87
88
  let result = client.completion(provider, &model, request).await;
88
- router
89
- .breaker()
90
- .observe(provider.name(), result.as_ref().map(|_| ()))
91
- .await;
92
89
 
93
90
  match result {
94
91
  Ok(resp) => {
@@ -132,13 +129,13 @@ pub async fn stream(
132
129
  // Observed but not gated: a single-target call has no alternative, so
133
130
  // refusing here would only convert an upstream failure into a local one.
134
131
  // The breaker refuses where there is somewhere else to go — `fallback.rs`.
135
- let client = &*SHARED_CLIENT;
136
- let opened = client.stream_owned(provider.clone(), &model, request).await;
137
- // A stream's health verdict is whether it opened; per-chunk failures are
138
- // the transport's business, not the breaker's.
139
- router
140
- .breaker()
141
- .observe(provider.name(), opened.as_ref().map(|_| ()))
142
- .await;
143
- opened
132
+ bound_client(router)
133
+ .stream_owned(provider, &model, request)
134
+ .await
135
+ }
136
+
137
+ /// The shared HTTP client, reporting to this router's breaker. The pool is
138
+ /// shared by clone; only the breaker handle is per call.
139
+ pub(crate) fn bound_client(router: &Router) -> ShimClient {
140
+ SHARED_CLIENT.clone().with_breaker(router.breaker().clone())
144
141
  }
@@ -116,6 +116,17 @@ fn extract_system_message(
116
116
  (system, rest)
117
117
  }
118
118
 
119
+ /// One Chat Completions message in, one Anthropic message out — a `role:
120
+ /// "tool"` result included, which becomes its own `user` message even when it
121
+ /// sits beside another.
122
+ ///
123
+ /// Same-role adjacency is deliberately left alone. The Messages API accepts
124
+ /// it: its reference states that consecutive `user` or `assistant` turns in a
125
+ /// request are combined into a single turn server-side (platform.claude.com,
126
+ /// Messages API, `messages` parameter), and parallel tool results already
127
+ /// reach it here as back-to-back `user` messages. Merging locally would only
128
+ /// destroy message boundaries a caller may key on. Gemini is the one wire in
129
+ /// this crate that rejects adjacency, and it merges in its own adapter.
119
130
  fn transform_messages(messages: &[Value]) -> Vec<Value> {
120
131
  messages
121
132
  .iter()
@@ -114,6 +114,12 @@ fn transform_messages(messages: &[Value]) -> (Option<Value>, Vec<Value>) {
114
114
  (system_instruction, contents)
115
115
  }
116
116
 
117
+ /// Gemini's own repair, not a shared one. This is the only wire in the crate
118
+ /// known to reject adjacent same-role turns, so it is the only adapter that
119
+ /// folds them together. Every other adapter passes adjacency through — each
120
+ /// says why on its own message pass — because a merge destroys message
121
+ /// boundaries a caller may depend on, and only a wire that would otherwise
122
+ /// fail the request earns that.
117
123
  fn merge_same_role(turns: Vec<Value>) -> Vec<Value> {
118
124
  let mut merged: Vec<Value> = Vec::new();
119
125
  for turn in turns {
@@ -46,6 +46,13 @@ fn strip_cache_control(value: &mut Value) {
46
46
  /// - Assistant messages with tool_calls → split into the assistant message +
47
47
  /// separate `function_call` items
48
48
  /// - `role: "tool"` messages → `function_call_output` items
49
+ ///
50
+ /// Same-role adjacency passes through. `input` is a flat item list, and the
51
+ /// Responses reference documents roles and their precedence but no ordering or
52
+ /// alternation rule (developers.openai.com, Responses API, `input`); this
53
+ /// adapter already emits several `function_call_output` items in a row for
54
+ /// parallel tool calls. Two adjacent assistant messages stay two items. The
55
+ /// ChatGPT adapter goes through this same translator and inherits the stance.
49
56
  fn sanitize_messages(messages: &[Value]) -> Vec<Value> {
50
57
  let mut result = Vec::new();
51
58
  for msg in messages {
@@ -40,6 +40,16 @@ impl OpenAiCompatible {
40
40
  /// Strip llmshim-normalized / foreign-provider fields and normalize content
41
41
  /// blocks to Chat Completions form. Messages, `tool_calls`, and `role: "tool"`
42
42
  /// stay in Chat Completions shape (the target format).
43
+ ///
44
+ /// Same-role adjacency passes through unchanged, and here the answer is
45
+ /// genuinely the served model's. vLLM and SGLang render `messages` through the
46
+ /// tokenizer's Jinja chat template (or `--chat-template`), so acceptance is a
47
+ /// property of that template: most current ones accept adjacent turns, some
48
+ /// older ones raise — Mistral-7B-Instruct-v0.1's template errors with
49
+ /// "conversation roles must alternate user/assistant/user/assistant/...".
50
+ /// llmshim cannot see the template, so it does not merge; a strict template's
51
+ /// rejection surfaces as the server's own 400. No `x-vllm` / `x-sglang`
52
+ /// parameter changes this.
43
53
  fn sanitize_messages(messages: &[Value]) -> Vec<Value> {
44
54
  messages
45
55
  .iter()
@@ -61,6 +61,14 @@ fn normalize_openrouter_effort(effort: &str, pro: bool) -> &'static str {
61
61
  /// llmshim-normalized / foreign-provider fields so multi-model conversations
62
62
  /// don't leak them, and normalize vision blocks to OpenAI form. Messages,
63
63
  /// `tool_calls`, and `role: "tool"` all stay in Chat Completions shape.
64
+ ///
65
+ /// Same-role adjacency passes through unchanged. OpenRouter is an aggregator:
66
+ /// its own API is Chat Completions and documents no alternation rule
67
+ /// (openrouter.ai/docs, API reference and parameters). Whether the vendor
68
+ /// behind a given slug rejects adjacent turns is unknown from here and is
69
+ /// OpenRouter's to reconcile; a faithful passthrough does not pre-empt it by
70
+ /// merging, which would destroy message boundaries for the vendors that
71
+ /// accept them.
64
72
  fn sanitize_messages(messages: &[Value]) -> Vec<Value> {
65
73
  messages
66
74
  .iter()
@@ -29,6 +29,13 @@ impl Xai {
29
29
  /// - Assistant messages with tool_calls → split into the assistant message +
30
30
  /// separate `function_call` items
31
31
  /// - `role: "tool"` messages → `function_call_output` items
32
+ ///
33
+ /// Same-role adjacency passes through unchanged. xAI's API is OpenAI
34
+ /// Responses-shaped and its documentation (docs.x.ai, chat guide and API
35
+ /// reference) states no ordering or alternation rule — but silence is not a
36
+ /// verified acceptance, and this has not been checked live. Nothing is merged
37
+ /// here because merging would destroy message boundaries a caller may depend
38
+ /// on; if xAI ever rejects adjacency the rejection arrives as its own 400.
32
39
  fn sanitize_messages(messages: &[Value]) -> Vec<Value> {
33
40
  let mut result = Vec::new();
34
41
  for msg in messages {
@@ -119,7 +119,12 @@ pub struct Usage {
119
119
  pub cache_read_tokens: u64,
120
120
  pub cache_write_tokens: u64,
121
121
  /// USD charged for this response. `null` means the catalog carries no price
122
- /// for the model — it never means free. See `llmshim::cost`.
122
+ /// for the model at all — it never means free.
123
+ ///
124
+ /// Where a model prices some token classes and not others, the unpriced ones
125
+ /// are charged at its highest published rate, so this is an **upper bound**
126
+ /// rather than `null`. Under-reporting would let a spend cap stop binding;
127
+ /// over-reporting merely spends a budget slightly early. See `llmshim::cost`.
123
128
  pub cost_usd: Option<f64>,
124
129
  }
125
130