llmshim 0.7.2__tar.gz → 0.8.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (200) hide show
  1. {llmshim-0.7.2 → llmshim-0.8.1}/.github/workflows/release.yml +1 -1
  2. {llmshim-0.7.2 → llmshim-0.8.1}/CLAUDE.md +15 -4
  3. {llmshim-0.7.2 → llmshim-0.8.1}/Cargo.lock +2 -2
  4. {llmshim-0.7.2 → llmshim-0.8.1}/Cargo.toml +2 -2
  5. {llmshim-0.7.2 → llmshim-0.8.1}/PKG-INFO +1 -1
  6. {llmshim-0.7.2 → llmshim-0.8.1}/README.md +13 -1
  7. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/Cargo.toml +1 -1
  8. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/README.md +22 -0
  9. llmshim-0.8.1/crates/llmshim-catalog/data/README.md +26 -0
  10. llmshim-0.8.1/crates/llmshim-catalog/data/verified.json +43 -0
  11. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/src/builtin.rs +13 -3
  12. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/src/lib.rs +17 -0
  13. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/src/merge.rs +2 -1
  14. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/src/parse.rs +20 -1
  15. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/src/types.rs +56 -0
  16. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/tests/catalog.rs +101 -0
  17. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/guides/caching.md +6 -3
  18. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/guides/reasoning.md +14 -3
  19. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/reference/models.md +25 -1
  20. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/reference/providers.md +2 -2
  21. {llmshim-0.7.2 → llmshim-0.8.1}/src/breaker.rs +4 -0
  22. {llmshim-0.7.2 → llmshim-0.8.1}/src/cache.rs +1 -0
  23. {llmshim-0.7.2 → llmshim-0.8.1}/src/client.rs +5 -0
  24. {llmshim-0.7.2 → llmshim-0.8.1}/src/cost.rs +25 -1
  25. {llmshim-0.7.2 → llmshim-0.8.1}/src/error.rs +9 -1
  26. {llmshim-0.7.2 → llmshim-0.8.1}/src/gateway/http.rs +6 -1
  27. {llmshim-0.7.2 → llmshim-0.8.1}/src/providers/anthropic.rs +5 -3
  28. {llmshim-0.7.2 → llmshim-0.8.1}/src/providers/anthropic_reasoning.rs +1 -0
  29. {llmshim-0.7.2 → llmshim-0.8.1}/src/providers/chatgpt/auth.rs +1 -0
  30. {llmshim-0.7.2 → llmshim-0.8.1}/src/providers/gemini.rs +3 -0
  31. {llmshim-0.7.2 → llmshim-0.8.1}/src/providers/openai.rs +9 -2
  32. {llmshim-0.7.2 → llmshim-0.8.1}/src/providers/openai_compat.rs +92 -17
  33. {llmshim-0.7.2 → llmshim-0.8.1}/src/providers/openrouter.rs +2 -0
  34. {llmshim-0.7.2 → llmshim-0.8.1}/src/providers/xai.rs +13 -6
  35. {llmshim-0.7.2 → llmshim-0.8.1}/src/proxy/error.rs +1 -1
  36. {llmshim-0.7.2 → llmshim-0.8.1}/src/reasoning/normalize.rs +38 -12
  37. {llmshim-0.7.2 → llmshim-0.8.1}/src/reasoning.rs +1 -0
  38. {llmshim-0.7.2 → llmshim-0.8.1}/src/router.rs +24 -12
  39. {llmshim-0.7.2 → llmshim-0.8.1}/src/schema/validate.rs +2 -0
  40. {llmshim-0.7.2 → llmshim-0.8.1}/src/shim.rs +20 -6
  41. {llmshim-0.7.2 → llmshim-0.8.1}/src/toolcall.rs +2 -0
  42. llmshim-0.8.1/tests/fixtures/sglang-models.toml +6 -0
  43. llmshim-0.8.1/tests/fixtures/sglang-responses-stream.sse +75 -0
  44. llmshim-0.8.1/tests/fixtures/sglang-responses-turn1.json +94 -0
  45. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_current_models.rs +2 -2
  46. llmshim-0.8.1/tests/integration_grok_4_7.rs +298 -0
  47. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_long_context.rs +6 -2
  48. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_proxy.rs +3 -4
  49. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_sglang.rs +69 -0
  50. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_advertised_models.rs +1 -1
  51. llmshim-0.8.1/tests/unit_client_retry_after.rs +97 -0
  52. llmshim-0.8.1/tests/unit_grok_4_7.rs +78 -0
  53. llmshim-0.8.1/tests/unit_openai_compat_responses.rs +290 -0
  54. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_reasoning.rs +62 -0
  55. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_router.rs +3 -1
  56. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_shim.rs +3 -3
  57. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_sse.rs +3 -3
  58. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_toolcall.rs +1 -0
  59. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_wire.rs +7 -7
  60. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_xai.rs +3 -3
  61. llmshim-0.7.2/crates/llmshim-catalog/data/README.md +0 -8
  62. {llmshim-0.7.2 → llmshim-0.8.1}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
  63. {llmshim-0.7.2 → llmshim-0.8.1}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
  64. {llmshim-0.7.2 → llmshim-0.8.1}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  65. {llmshim-0.7.2 → llmshim-0.8.1}/.github/scripts/npm-verify-served.sh +0 -0
  66. {llmshim-0.7.2 → llmshim-0.8.1}/.github/workflows/catalog-refresh.yml +0 -0
  67. {llmshim-0.7.2 → llmshim-0.8.1}/.github/workflows/pages.yml +0 -0
  68. {llmshim-0.7.2 → llmshim-0.8.1}/.gitignore +0 -0
  69. {llmshim-0.7.2 → llmshim-0.8.1}/CODE_OF_CONDUCT.md +0 -0
  70. {llmshim-0.7.2 → llmshim-0.8.1}/CONTRIBUTING.md +0 -0
  71. {llmshim-0.7.2 → llmshim-0.8.1}/LICENSE-APACHE +0 -0
  72. {llmshim-0.7.2 → llmshim-0.8.1}/LICENSE-MIT +0 -0
  73. {llmshim-0.7.2 → llmshim-0.8.1}/NOTICE +0 -0
  74. {llmshim-0.7.2 → llmshim-0.8.1}/SECURITY.md +0 -0
  75. {llmshim-0.7.2 → llmshim-0.8.1}/benchmarks/bench.rs +0 -0
  76. {llmshim-0.7.2 → llmshim-0.8.1}/benchmarks/bench_python.py +0 -0
  77. {llmshim-0.7.2 → llmshim-0.8.1}/benchmarks/gateway_loadtest.rs +0 -0
  78. {llmshim-0.7.2 → llmshim-0.8.1}/benchmarks/loadtest.rs +0 -0
  79. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/LICENSE-APACHE +0 -0
  80. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/LICENSE-MIT +0 -0
  81. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/data/LICENSE.models.dev +0 -0
  82. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/data/models.dev.json +0 -0
  83. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/src/aliases.rs +0 -0
  84. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/src/capabilities.rs +0 -0
  85. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/src/refresh.rs +0 -0
  86. {llmshim-0.7.2 → llmshim-0.8.1}/crates/llmshim-catalog/tests/refresh.rs +0 -0
  87. {llmshim-0.7.2 → llmshim-0.8.1}/docs/.gitignore +0 -0
  88. {llmshim-0.7.2 → llmshim-0.8.1}/docs/book.toml +0 -0
  89. {llmshim-0.7.2 → llmshim-0.8.1}/docs/mermaid-init.js +0 -0
  90. {llmshim-0.7.2 → llmshim-0.8.1}/docs/mermaid.min.js +0 -0
  91. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/SUMMARY.md +0 -0
  92. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/concepts/contracts.md +0 -0
  93. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/concepts/conversations.md +0 -0
  94. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/concepts/portability.md +0 -0
  95. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/concepts/routing.md +0 -0
  96. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/concepts/translation-flow.md +0 -0
  97. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/guides/capabilities.md +0 -0
  98. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/guides/fallbacks.md +0 -0
  99. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/guides/images.md +0 -0
  100. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/guides/native-controls.md +0 -0
  101. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/guides/schemas.md +0 -0
  102. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/guides/streaming.md +0 -0
  103. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/guides/tools.md +0 -0
  104. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/introduction.md +0 -0
  105. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/proxy/deployment.md +0 -0
  106. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/proxy/http-api.md +0 -0
  107. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/proxy/native-apis.md +0 -0
  108. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/proxy/scaling.md +0 -0
  109. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/reference/api.md +0 -0
  110. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/reference/cli.md +0 -0
  111. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/reference/configuration.md +0 -0
  112. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/reference/errors.md +0 -0
  113. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/reference/request-fields.md +0 -0
  114. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/reference/surfaces.md +0 -0
  115. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/start/choose.md +0 -0
  116. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/start/cli.md +0 -0
  117. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/start/clients.md +0 -0
  118. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/start/configure.md +0 -0
  119. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/start/proxy.md +0 -0
  120. {llmshim-0.7.2 → llmshim-0.8.1}/docs/src/start/rust.md +0 -0
  121. {llmshim-0.7.2 → llmshim-0.8.1}/examples/chat.rs +0 -0
  122. {llmshim-0.7.2 → llmshim-0.8.1}/examples/stream.rs +0 -0
  123. {llmshim-0.7.2 → llmshim-0.8.1}/llmshim/__init__.py +0 -0
  124. {llmshim-0.7.2 → llmshim-0.8.1}/llmshim/_client.py +0 -0
  125. {llmshim-0.7.2 → llmshim-0.8.1}/llmshim/_server.py +0 -0
  126. {llmshim-0.7.2 → llmshim-0.8.1}/llmshim/types.py +0 -0
  127. {llmshim-0.7.2 → llmshim-0.8.1}/pyproject.toml +0 -0
  128. {llmshim-0.7.2 → llmshim-0.8.1}/src/cli.rs +0 -0
  129. {llmshim-0.7.2 → llmshim-0.8.1}/src/config.rs +0 -0
  130. {llmshim-0.7.2 → llmshim-0.8.1}/src/env.rs +0 -0
  131. {llmshim-0.7.2 → llmshim-0.8.1}/src/error/normalize.rs +0 -0
  132. {llmshim-0.7.2 → llmshim-0.8.1}/src/fallback.rs +0 -0
  133. {llmshim-0.7.2 → llmshim-0.8.1}/src/gateway/auth.rs +0 -0
  134. {llmshim-0.7.2 → llmshim-0.8.1}/src/gateway/distributed.rs +0 -0
  135. {llmshim-0.7.2 → llmshim-0.8.1}/src/gateway/idempotency.rs +0 -0
  136. {llmshim-0.7.2 → llmshim-0.8.1}/src/gateway/metrics.rs +0 -0
  137. {llmshim-0.7.2 → llmshim-0.8.1}/src/gateway/mod.rs +0 -0
  138. {llmshim-0.7.2 → llmshim-0.8.1}/src/gateway/quota.rs +0 -0
  139. {llmshim-0.7.2 → llmshim-0.8.1}/src/lib.rs +0 -0
  140. {llmshim-0.7.2 → llmshim-0.8.1}/src/log.rs +0 -0
  141. {llmshim-0.7.2 → llmshim-0.8.1}/src/main.rs +0 -0
  142. {llmshim-0.7.2 → llmshim-0.8.1}/src/models.rs +0 -0
  143. {llmshim-0.7.2 → llmshim-0.8.1}/src/provider.rs +0 -0
  144. {llmshim-0.7.2 → llmshim-0.8.1}/src/providers/anthropic_signature.rs +0 -0
  145. {llmshim-0.7.2 → llmshim-0.8.1}/src/providers/chatgpt/mod.rs +0 -0
  146. {llmshim-0.7.2 → llmshim-0.8.1}/src/providers/chatgpt/streaming.rs +0 -0
  147. {llmshim-0.7.2 → llmshim-0.8.1}/src/providers/mod.rs +0 -0
  148. {llmshim-0.7.2 → llmshim-0.8.1}/src/proxy/convert.rs +0 -0
  149. {llmshim-0.7.2 → llmshim-0.8.1}/src/proxy/handlers.rs +0 -0
  150. {llmshim-0.7.2 → llmshim-0.8.1}/src/proxy/health.rs +0 -0
  151. {llmshim-0.7.2 → llmshim-0.8.1}/src/proxy/mod.rs +0 -0
  152. {llmshim-0.7.2 → llmshim-0.8.1}/src/proxy/ratelimit.rs +0 -0
  153. {llmshim-0.7.2 → llmshim-0.8.1}/src/proxy/types.rs +0 -0
  154. {llmshim-0.7.2 → llmshim-0.8.1}/src/proxy/wire/mod.rs +0 -0
  155. {llmshim-0.7.2 → llmshim-0.8.1}/src/proxy/wire/receipts.rs +0 -0
  156. {llmshim-0.7.2 → llmshim-0.8.1}/src/schema/memo.rs +0 -0
  157. {llmshim-0.7.2 → llmshim-0.8.1}/src/schema/mod.rs +0 -0
  158. {llmshim-0.7.2 → llmshim-0.8.1}/src/schema/walk.rs +0 -0
  159. {llmshim-0.7.2 → llmshim-0.8.1}/src/streaming.rs +0 -0
  160. {llmshim-0.7.2 → llmshim-0.8.1}/src/toolcall/streaming.rs +0 -0
  161. {llmshim-0.7.2 → llmshim-0.8.1}/src/usage.rs +0 -0
  162. {llmshim-0.7.2 → llmshim-0.8.1}/src/vision.rs +0 -0
  163. {llmshim-0.7.2 → llmshim-0.8.1}/tests/fixtures/chatgpt-red.png +0 -0
  164. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration.rs +0 -0
  165. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_chatgpt.rs +0 -0
  166. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_chatgpt_proxy.rs +0 -0
  167. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_fallback.rs +0 -0
  168. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_gemini.rs +0 -0
  169. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_gemini_tools.rs +0 -0
  170. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_multimodel.rs +0 -0
  171. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_openrouter.rs +0 -0
  172. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_thinking.rs +0 -0
  173. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_tool_roundtrip.rs +0 -0
  174. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_vision.rs +0 -0
  175. {llmshim-0.7.2 → llmshim-0.8.1}/tests/integration_xai.rs +0 -0
  176. {llmshim-0.7.2 → llmshim-0.8.1}/tests/support/completion_status.rs +0 -0
  177. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_anthropic.rs +0 -0
  178. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_cache.rs +0 -0
  179. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_chatgpt.rs +0 -0
  180. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_cli.rs +0 -0
  181. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_client_breaker.rs +0 -0
  182. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_fable.rs +0 -0
  183. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_fallback.rs +0 -0
  184. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_fast_mode.rs +0 -0
  185. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_gemini.rs +0 -0
  186. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_log.rs +0 -0
  187. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_models.rs +0 -0
  188. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_multimodel.rs +0 -0
  189. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_openai.rs +0 -0
  190. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_openai_compat.rs +0 -0
  191. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_openrouter.rs +0 -0
  192. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_provider_contracts.rs +0 -0
  193. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_proxy.rs +0 -0
  194. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_proxy_convert.rs +0 -0
  195. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_reasoning_profile.rs +0 -0
  196. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_schema.rs +0 -0
  197. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_signature.rs +0 -0
  198. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_tools.rs +0 -0
  199. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_usage.rs +0 -0
  200. {llmshim-0.7.2 → llmshim-0.8.1}/tests/unit_vision.rs +0 -0
@@ -18,7 +18,7 @@ jobs:
18
18
  - name: Check formatting
19
19
  run: cargo fmt --all --check
20
20
  - name: Clippy
21
- run: cargo clippy --workspace --features proxy -- -D warnings
21
+ run: cargo clippy --workspace --all-targets --features proxy -- -D warnings
22
22
  - name: Unit tests
23
23
  env:
24
24
  LLMSHIM_CATALOG_OFFLINE: '1'
@@ -16,7 +16,7 @@ This is a public crate on crates.io. Do NOT make breaking changes to `pub` items
16
16
  - **ChatGPT subscription (OAuth):** only `chatgpt/gpt-6-astra`, `chatgpt/gpt-5.6-sol`, `chatgpt/gpt-5.6-terra`, and `chatgpt/gpt-5.6-luna`. `CHATGPT_MODELS` in `src/models.rs` is shared by discovery, CLI selection, and validation; older/unlisted models fail before authentication or network calls.
17
17
  - **Anthropic:** `claude-fable-5-1`, `claude-opus-5`, `claude-sonnet-5`, `claude-haiku-4-5-20251001`
18
18
  - **Gemini:** `gemini-3.8-flash`, `gemini-3.5-flash-lite`
19
- - **xAI:** `grok-4.6`
19
+ - **xAI:** `grok-4.7`
20
20
  - **OpenRouter:** not enumerated (huge/dynamic catalog) — any `openrouter/<vendor>/<model>` slug routes through, e.g. `openrouter/anthropic/claude-sonnet-5`.
21
21
  - **vLLM / SGLang:** not enumerated (self-hosted) — any `vllm/<served-model>` or `sglang/<served-model>` routes through to the configured server, e.g. `sglang/Qwen/Qwen3.6-35B-A3B-FP8`.
22
22
 
@@ -73,6 +73,17 @@ for whichever stream chunk carries usage; `log.rs`, `proxy::types::Usage`, both
73
73
  native facades and the four bundled clients carry it through as a nullable
74
74
  field. `null` means unknown, not free.
75
75
 
76
+ `ModelInfo.context_cost_tiers` adds context-dependent standard rates without
77
+ changing the public `Cost` struct. `cost_for_input_tokens` includes cached input
78
+ in tier selection; `src/cost.rs` applies the resulting rates to the whole
79
+ response. Local per-field prices outrank lower-source tier rates. Builtin
80
+ launch metadata lives in `crates/llmshim-catalog/data/verified.json`, separate
81
+ from the unmodified models.dev snapshot. Grok 4.7 live regression tests are in
82
+ `tests/integration_grok_4_7.rs`; they use an ephemeral proxy port, configured
83
+ keys, and billed API calls. `none` clamps to `low`; named tool choice must be
84
+ flat on the Responses wire. Preserve encrypted reasoning on both normal and
85
+ streaming tool round trips.
86
+
76
87
  `src/gateway/quota.rs` adds a per-identity dollar cap beside the RPM/TPM
77
88
  buckets: `budget_usd` + `budget_window_secs` on an `Identity`, checked before
78
89
  dispatch and charged after (cost is only knowable once a response exists, so
@@ -321,15 +332,15 @@ accounts exercise enforcement too. The provider handles incompatible
321
332
  thinking on a switch to older models; do not infer signatures from text.
322
333
 
323
334
  `tests/unit_fable.rs` pins these rules. Live checks for Fable 5, Fable 5.1,
324
- Opus 5, Gemini 3.8 Flash, and Grok 4.6:
335
+ Opus 5, Gemini 3.8 Flash, and Grok 4.7:
325
336
 
326
337
  ```bash
327
338
  cargo test --features proxy --test integration_current_models -- --ignored --nocapture
328
339
  ```
329
340
 
330
341
  The live tests consume API usage and are ignored during offline preflight.
331
- Gemini 3.8 Flash and Opus 5 already had catalog/adapter support; Grok 4.6 is
332
- the verified xAI model ID. Keep the ChatGPT four-model allowlist independent.
342
+ Gemini 3.8 Flash and Opus 5 already had catalog/adapter support; Grok 4.7 is
343
+ the current xAI model ID (verified live 2026-09-21); retain Grok 4.6 for explicit routing. Keep the ChatGPT four-model allowlist independent.
333
344
 
334
345
  Two knobs work across every provider: `reasoning_effort` (`none|low|medium|high|xhigh|max`) and `reasoning_mode` (`standard|pro`). A third, `reasoning_summary` (`auto|none`), controls reasoning-text visibility → Anthropic `thinking.display` (`auto`→`summarized`, the default when `reasoning_effort` is present so newer models like Sonnet 5 / Opus 4.7-4.8 return reasoning text instead of the API-default `omitted`; `none`→`omitted` for lower latency). Applies to both the adaptive and pre-4.6 enabled thinking builders; a caller-supplied `thinking` block bypasses it. Each provider transform maps them to its native dialect, **clamping to the nearest tier the target model accepts** (all boundaries verified live — e.g. `max` is native on OpenAI gpt-5.6 and GPT-6 Astra; Anthropic 4.6 rejects `xhigh` but has `max`; Gemini's enum tops out at `high`; xAI grok-4.20 models reject any reasoning param). `mode: "pro"` is native on OpenAI gpt-5.6/-pro models (`reasoning.mode`), emulated as a one-tier effort bump elsewhere; explicit `none` always wins. Native passthrough (`x-openai.reasoning`, `x-anthropic.thinking`, `x-gemini.thinkingConfig`) bypasses the mapping entirely and always takes precedence. **Full per-provider mapping tables: `docs/src/guides/reasoning.md`** — update it and the pinning tests in `tests/unit_*.rs` together whenever a mapping changes.
335
346
 
@@ -1180,7 +1180,7 @@ checksum = "6373607a59f0be73a39b6fe456b8192fcc3585f602af20751600e974dd455e77"
1180
1180
 
1181
1181
  [[package]]
1182
1182
  name = "llmshim"
1183
- version = "0.7.2"
1183
+ version = "0.8.1"
1184
1184
  dependencies = [
1185
1185
  "async-stream",
1186
1186
  "async-trait",
@@ -1215,7 +1215,7 @@ dependencies = [
1215
1215
 
1216
1216
  [[package]]
1217
1217
  name = "llmshim-catalog"
1218
- version = "0.1.0"
1218
+ version = "0.1.1"
1219
1219
  dependencies = [
1220
1220
  "chrono",
1221
1221
  "dirs",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "llmshim"
3
- version = "0.7.2"
3
+ version = "0.8.1"
4
4
  edition = "2021"
5
5
  description = "Blazing fast LLM API translation layer in pure Rust"
6
6
  license = "MIT OR Apache-2.0"
@@ -24,7 +24,7 @@ gateway-redis = ["gateway", "redis-coordination"]
24
24
  redis-coordination = ["proxy", "dep:redis"]
25
25
 
26
26
  [dependencies]
27
- llmshim-catalog = { version = "0.1.0", path = "crates/llmshim-catalog" }
27
+ llmshim-catalog = { version = "0.1.1", path = "crates/llmshim-catalog" }
28
28
  tokio = { version = "1", features = ["full"] }
29
29
  reqwest = { version = "0.12", default-features = false, features = ["json", "stream", "rustls-tls", "http2", "gzip", "brotli", "zstd", "deflate"] }
30
30
  serde = { version = "1", features = ["derive"] }
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llmshim
3
- Version: 0.7.2
3
+ Version: 0.8.1
4
4
  Classifier: Development Status :: 4 - Beta
5
5
  Classifier: Intended Audience :: Developers
6
6
  Classifier: License :: OSI Approved :: MIT License
@@ -100,6 +100,18 @@ Then address the served model as `sglang/<served-model>` or `vllm/<served-model>
100
100
  (e.g. `sglang/Qwen/Qwen3.6-35B-A3B-FP8`). Server-specific knobs go under
101
101
  `x-vllm` / `x-sglang`.
102
102
 
103
+ A server that also serves `/v1/responses` (SGLang does) can be spoken to on
104
+ that wire with `SGLANG_WIRE=responses` (or `VLLM_WIRE=responses`). Reasoning
105
+ then comes back as an item with its own id rather than bare `reasoning_content`,
106
+ and is replayed as that item. Replay is gated on a known model family on every
107
+ wire, and the public catalog does not know a served model: declare it once in
108
+ `.llmshim/models.toml` —
109
+
110
+ ```toml
111
+ [models."sglang/<served-model>"]
112
+ family = "qwen"
113
+ ```
114
+
103
115
  Or persist them to the config file (used by all three surfaces):
104
116
 
105
117
  ```bash
@@ -461,7 +473,7 @@ Standard library only. Full docs: [`clients/ruby/README.md`](clients/ruby/README
461
473
  | **OpenAI** | `gpt-6-astra`, `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna` | Yes (summaries) |
462
474
  | **Anthropic** | `claude-fable-5-1`, `claude-opus-5`, `claude-sonnet-5`, `claude-haiku-4-5-20251001` | Yes (thinking summaries) |
463
475
  | **Google Gemini** | `gemini-3.8-flash`, `gemini-3.5-flash-lite` | Yes (thought summaries) |
464
- | **xAI** | `grok-4.6` | No (hidden) |
476
+ | **xAI** | `grok-4.7` | No (hidden) |
465
477
 
466
478
  The CLI and server advertise these current tiers. ChatGPT subscription access
467
479
  uses the same four OpenAI models under `chatgpt/`. OpenRouter and self-hosted
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "llmshim-catalog"
3
- version = "0.1.0"
3
+ version = "0.1.1"
4
4
  edition = "2021"
5
5
  description = "Offline-first model metadata with verified defaults, models.dev, and local overrides"
6
6
  license = "MIT OR Apache-2.0"
@@ -18,9 +18,31 @@ input/output modalities, knowledge/release dates, and field-level sources.
18
18
  Missing data remains unknown. A date given only as a month is not turned into
19
19
  an invented day. Unknown family keys remain absent and must not authorize replay.
20
20
 
21
+ `ModelInfo::cost_for_input_tokens(total)` selects context-dependent rates from
22
+ `context_cost_tiers`; `cost` remains the base rates. The total includes cached
23
+ input. A tier applies to the entire request strictly above its threshold, with
24
+ missing rates inherited from the preceding tier. Models.dev `cost.tiers` entries
25
+ whose type is `context` are imported. Local policy can set
26
+ `context_cost_tiers = []` to disable inherited tiers, or use:
27
+
28
+ ```toml
29
+ [[models."xai/grok-4.7".context_cost_tiers]]
30
+ above_input_tokens = 200000
31
+ [models."xai/grok-4.7".context_cost_tiers.cost]
32
+ input = 4.0
33
+ cache_read = 1.0
34
+ output = 12.0
35
+ ```
36
+
37
+ The strongest source replaces the tier list as a whole. Higher-priority base
38
+ price overrides still win per field over lower-priority tier rates. Provider
39
+ discovery cannot supply either base prices or tiers.
40
+
21
41
  The compiled `builtin` module retains the small verified discovery table,
22
42
  historical metadata, and the separate ChatGPT subscription allowlist. Catalog
23
43
  coverage does not authorize or configure an API route or a subscription model.
44
+ `data/verified.json` supplies sourced launch metadata independently of the
45
+ models.dev snapshot. It does not add models to the curated discovery list.
24
46
 
25
47
  ## Layers and policy
26
48
 
@@ -0,0 +1,26 @@
1
+ `models.dev.json` is the unmodified response downloaded from
2
+ https://models.dev/api.json on 2026-09-16. It is third-party data under the MIT
3
+ license in `LICENSE.models.dev`. No credentials, local overrides, provider
4
+ account discovery, or private fixtures belong in this directory.
5
+
6
+ The refresh workflow replaces only this public snapshot. Tests validate its
7
+ structure and minimum coverage before proposing an update. The builtin table
8
+ continues to win conflicts for the fields it asserts.
9
+
10
+ `verified.json` contains separate verified builtin assertions, checked on
11
+ 2026-09-21. It is maintained here, not copied into the models.dev artifact:
12
+
13
+ - Native Grok 4.7 capabilities, reasoning efforts, dates, and standard token
14
+ prices: https://docs.x.ai/developers/grok-4-7 and
15
+ https://docs.x.ai/developers/release-notes (including Grok 4.6 context pricing).
16
+ - Grok 4.7 streaming, image input, structured output, named tool selection,
17
+ and encrypted-reasoning round trips: `tests/integration_grok_4_7.rs` in the
18
+ parent repository, verified against the live xAI Responses API.
19
+ - OpenRouter's Grok 4.7 identity, capabilities, limits and standard endpoint
20
+ prices: https://openrouter.ai/api/v1/models/x-ai/grok-4.7/endpoints . Priority
21
+ endpoint rates are different; the catalog entries describe standard rates.
22
+ The `:nitro` entry supplies replay metadata without a fixed price because
23
+ standard and priority endpoints can compete for that route.
24
+
25
+ No public `grok-4.7-fast` API route is asserted. Its launch availability is
26
+ limited to Cursor and Grok Build.
@@ -0,0 +1,43 @@
1
+ {
2
+ "xai/grok-4.7": {
3
+ "name": "Grok 4.7",
4
+ "family": "grok",
5
+ "limit": {"context": 500000},
6
+ "release_date": "2026-09-21",
7
+ "modalities": {"input": ["text", "image"], "output": ["text"]},
8
+ "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh"]}],
9
+ "streaming": true,
10
+ "forced_tool_choice": true,
11
+ "cost": {"input": 2, "cache_read": 0.5, "output": 6},
12
+ "context_cost_tiers": [{"above_input_tokens": 200000, "cost": {"input": 4, "cache_read": 1, "output": 12}}]
13
+ },
14
+ "xai/grok-4.6": {
15
+ "cost": {"input": 2, "cache_read": 0.5, "output": 6},
16
+ "context_cost_tiers": [{"above_input_tokens": 200000, "cost": {"input": 4, "cache_read": 1, "output": 12}}]
17
+ },
18
+ "openrouter/x-ai/grok-4.7:nitro": {
19
+ "name": "Grok 4.7 (Nitro)",
20
+ "family": "grok",
21
+ "limit": {"context": 500000, "output": 450000},
22
+ "modalities": {"input": ["text", "image", "file"], "output": ["text"]},
23
+ "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh"]}],
24
+ "reasoning": true,
25
+ "tool_call": true,
26
+ "structured_output": true,
27
+ "forced_tool_choice": true
28
+ },
29
+ "openrouter/x-ai/grok-4.7": {
30
+ "name": "Grok 4.7",
31
+ "family": "grok",
32
+ "limit": {"context": 500000, "output": 450000},
33
+ "release_date": "2026-09-21",
34
+ "modalities": {"input": ["text", "image", "file"], "output": ["text"]},
35
+ "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh"]}],
36
+ "reasoning": true,
37
+ "tool_call": true,
38
+ "structured_output": true,
39
+ "forced_tool_choice": true,
40
+ "cost": {"input": 1.6, "cache_read": 0.4, "output": 4.8},
41
+ "context_cost_tiers": [{"above_input_tokens": 200000, "cost": {"input": 3.2, "cache_read": 0.8, "output": 9.6}}]
42
+ }
43
+ }
@@ -261,11 +261,11 @@ pub const MODELS: &[BuiltinModelInfo] = &[
261
261
  capabilities: CAPS_STD,
262
262
  },
263
263
  BuiltinModelInfo {
264
- id: "xai/grok-4.6",
264
+ id: "xai/grok-4.7",
265
265
  provider: "xai",
266
266
  family: Some(ModelFamily::Grok),
267
- name: "grok-4.6",
268
- label: "Grok 4.6",
267
+ name: "grok-4.7",
268
+ label: "Grok 4.7",
269
269
  context_window_tokens: Some(500_000),
270
270
  max_output_tokens: None,
271
271
  capabilities: CAPS_XAI,
@@ -279,6 +279,16 @@ pub const MODELS: &[BuiltinModelInfo] = &[
279
279
  /// Historical metadata remains available to explicit `spec()` lookups without
280
280
  /// appearing in discovery or model pickers. Routing is provider-owned.
281
281
  const LEGACY_MODELS: &[BuiltinModelInfo] = &[
282
+ BuiltinModelInfo {
283
+ id: "xai/grok-4.6",
284
+ provider: "xai",
285
+ family: Some(ModelFamily::Grok),
286
+ name: "grok-4.6",
287
+ label: "Grok 4.6",
288
+ context_window_tokens: Some(500_000),
289
+ max_output_tokens: None,
290
+ capabilities: CAPS_XAI,
291
+ },
282
292
  BuiltinModelInfo {
283
293
  id: "openai/gpt-5.5",
284
294
  provider: "openai",
@@ -165,6 +165,20 @@ impl Catalog {
165
165
  model.source = CatalogSource::Builtin;
166
166
  self.merge_model(model);
167
167
  }
168
+ // Verified launch-day metadata is separate from the unmodified
169
+ // models.dev snapshot and does not expand curated discovery.
170
+ let entries: Value = serde_json::from_str(include_str!("../data/verified.json"))
171
+ .expect("validated builtin metadata");
172
+ for (id, value) in entries.as_object().expect("builtin metadata object") {
173
+ let (provider, name) = id.split_once('/').expect("qualified builtin id");
174
+ self.merge_model(parse::model_from_value(
175
+ provider,
176
+ name,
177
+ value,
178
+ CatalogSource::Builtin,
179
+ None,
180
+ ));
181
+ }
168
182
  }
169
183
 
170
184
  pub fn merge_models_dev(
@@ -235,6 +249,9 @@ impl Catalog {
235
249
  ))?;
236
250
  let mut model =
237
251
  parse::model_from_value(provider, name, v, CatalogSource::Local, None);
252
+ if v.get("context_cost_tiers").is_some() && model.context_cost_tiers.is_none() {
253
+ return Err(CatalogError::Invalid("invalid context_cost_tiers"));
254
+ }
238
255
  if let Some(label) = v["label"].as_str() {
239
256
  model.label = label.into();
240
257
  }
@@ -1,6 +1,6 @@
1
1
  use crate::{CatalogSource, Cost, ModelInfo, Support};
2
2
 
3
- fn rank(source: CatalogSource) -> u8 {
3
+ pub(crate) fn rank(source: CatalogSource) -> u8 {
4
4
  match source {
5
5
  CatalogSource::ModelsDev => 1,
6
6
  CatalogSource::Builtin => 2,
@@ -76,6 +76,7 @@ pub(crate) fn merge(target: &mut ModelInfo, incoming: &ModelInfo) {
76
76
  }
77
77
  }
78
78
  if source != CatalogSource::ProviderApi {
79
+ field!(context_cost_tiers, incoming.context_cost_tiers.is_some());
79
80
  if let Some(cost) = incoming.cost {
80
81
  let mut merged = target.cost.unwrap_or_default();
81
82
  for (key, value, slot) in [
@@ -1,5 +1,6 @@
1
1
  use crate::{
2
- CatalogError, CatalogSource, Cost, ModelCapabilities, ModelFamily, ModelInfo, Support,
2
+ CatalogError, CatalogSource, ContextCostTier, Cost, ModelCapabilities, ModelFamily, ModelInfo,
3
+ Support,
3
4
  };
4
5
  use chrono::{DateTime, NaiveDate, Utc};
5
6
  use serde_json::Value;
@@ -77,6 +78,24 @@ pub(crate) fn model_from_value(
77
78
  };
78
79
  }
79
80
  m.cost = cost(&v["cost"]);
81
+ m.context_cost_tiers = if let Some(tiers) = v.get("context_cost_tiers") {
82
+ serde_json::from_value(tiers.clone()).ok()
83
+ } else {
84
+ v["cost"]["tiers"].as_array().map(|tiers| {
85
+ tiers
86
+ .iter()
87
+ .filter_map(|tier| {
88
+ if tier["tier"]["type"] != "context" {
89
+ return None;
90
+ }
91
+ Some(ContextCostTier {
92
+ above_input_tokens: tier["tier"]["size"].as_u64()?,
93
+ cost: cost(tier)?,
94
+ })
95
+ })
96
+ .collect()
97
+ })
98
+ };
80
99
  m.reasoning_options = v["reasoning_options"]
81
100
  .as_array()
82
101
  .into_iter()
@@ -81,6 +81,14 @@ pub struct Cost {
81
81
  pub cache_write: Option<f64>,
82
82
  }
83
83
 
84
+ /// Rates for the entire request when total input (including cached input)
85
+ /// exceeds `above_input_tokens`. Missing rates inherit the previous tier.
86
+ #[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)]
87
+ pub struct ContextCostTier {
88
+ pub above_input_tokens: u64,
89
+ pub cost: Cost,
90
+ }
91
+
84
92
  /// Media types are strings so new catalog modalities survive without a release.
85
93
  #[derive(Debug, Clone, PartialEq, Eq, Default, Serialize, Deserialize)]
86
94
  #[serde(default)]
@@ -123,6 +131,9 @@ pub struct ModelInfo {
123
131
  pub capabilities: crate::ModelCapabilities,
124
132
  pub family: Option<ModelFamily>,
125
133
  pub cost: Option<Cost>,
134
+ /// None means unspecified; an explicit empty list disables inherited tiers.
135
+ #[serde(default)]
136
+ pub context_cost_tiers: Option<Vec<ContextCostTier>>,
126
137
  pub reasoning_options: Vec<ReasoningOption>,
127
138
  pub modalities: Modalities,
128
139
  pub knowledge_cutoff: Option<NaiveDate>,
@@ -148,6 +159,7 @@ impl ModelInfo {
148
159
  capabilities: crate::ModelCapabilities::unknown(),
149
160
  family: None,
150
161
  cost: None,
162
+ context_cost_tiers: None,
151
163
  reasoning_options: Vec::new(),
152
164
  modalities: Modalities::default(),
153
165
  knowledge_cutoff: None,
@@ -158,4 +170,48 @@ impl ModelInfo {
158
170
  field_sources: BTreeMap::new(),
159
171
  }
160
172
  }
173
+
174
+ /// Resolve context-dependent rates without changing the base `Cost` API.
175
+ /// Higher-priority per-field base overrides also override lower-priority
176
+ /// tier rates. Tier lists are replaced atomically by a stronger source.
177
+ pub fn cost_for_input_tokens(&self, input_tokens: u64) -> Option<Cost> {
178
+ let mut cost = self.cost?;
179
+ let mut tiers: Vec<_> = self
180
+ .context_cost_tiers
181
+ .as_deref()
182
+ .unwrap_or_default()
183
+ .iter()
184
+ .filter(|t| input_tokens > t.above_input_tokens)
185
+ .collect();
186
+ tiers.sort_by_key(|t| t.above_input_tokens);
187
+ let tier_source = self
188
+ .field_sources
189
+ .get("context_cost_tiers")
190
+ .copied()
191
+ .unwrap_or(self.source);
192
+ for tier in tiers {
193
+ for (key, rate, slot) in [
194
+ ("cost.input", tier.cost.input, &mut cost.input),
195
+ ("cost.output", tier.cost.output, &mut cost.output),
196
+ (
197
+ "cost.cache_read",
198
+ tier.cost.cache_read,
199
+ &mut cost.cache_read,
200
+ ),
201
+ (
202
+ "cost.cache_write",
203
+ tier.cost.cache_write,
204
+ &mut cost.cache_write,
205
+ ),
206
+ ] {
207
+ let base_source = self.field_sources.get(key).copied().unwrap_or(self.source);
208
+ if rate.is_some_and(|n| n.is_finite() && n >= 0.0)
209
+ && crate::merge::rank(tier_source) >= crate::merge::rank(base_source)
210
+ {
211
+ *slot = rate;
212
+ }
213
+ }
214
+ }
215
+ Some(cost)
216
+ }
161
217
  }
@@ -8,6 +8,107 @@ fn feed(value: serde_json::Value) -> String {
8
8
  json!({"test": {"models": {"m": value}}}).to_string()
9
9
  }
10
10
 
11
+ #[test]
12
+ fn grok_4_7_is_available_offline_and_4_6_remains_explicit() {
13
+ let c = Catalog::vendored();
14
+ for id in ["xai/grok-4.7", "openrouter/x-ai/grok-4.7"] {
15
+ let m = c.resolve(id).unwrap();
16
+ assert_eq!(m.family, Some(ModelFamily::Grok));
17
+ assert_eq!(m.context_window_tokens, Some(500_000));
18
+ assert_eq!(m.reasoning_options.len(), 1);
19
+ assert_eq!(m.field_sources["family"], CatalogSource::Builtin);
20
+ assert!(
21
+ m.cost_for_input_tokens(200_001).unwrap().input.unwrap()
22
+ > m.cost.unwrap().input.unwrap()
23
+ );
24
+ }
25
+ assert_eq!(c.resolve("grok-4.7").unwrap().id, "xai/grok-4.7");
26
+ assert!(llmshim_catalog::builtin::spec("grok-4.6").is_some());
27
+ assert!(!llmshim_catalog::builtin::MODELS
28
+ .iter()
29
+ .any(|m| m.id == "xai/grok-4.6"));
30
+ assert_eq!(
31
+ llmshim_catalog::builtin::spec("grok-4.7")
32
+ .unwrap()
33
+ .max_output_tokens,
34
+ None
35
+ );
36
+ }
37
+
38
+ #[test]
39
+ fn context_price_tiers_select_by_full_input_and_inherit_missing_rates() {
40
+ let mut c = Catalog::empty();
41
+ c.merge_models_dev(
42
+ &feed(
43
+ json!({"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[
44
+ {"tier":{"type":"context","size":400000},"output":18},
45
+ {"tier":{"type":"context","size":200000},"input":4,"output":12,"cache_read":1}
46
+ ]}}),
47
+ ),
48
+ None,
49
+ )
50
+ .unwrap();
51
+ let m = c.resolve("test/m").unwrap();
52
+ assert_eq!(m.cost_for_input_tokens(200_000), m.cost);
53
+ let long = m.cost_for_input_tokens(200_001).unwrap();
54
+ assert_eq!(
55
+ (long.input, long.output, long.cache_read),
56
+ (Some(4.0), Some(12.0), Some(1.0))
57
+ );
58
+ let longer = m.cost_for_input_tokens(400_001).unwrap();
59
+ assert_eq!((longer.input, longer.output), (Some(4.0), Some(18.0)));
60
+ let restored: ModelInfo = serde_json::from_value(serde_json::to_value(m).unwrap()).unwrap();
61
+ assert_eq!(restored.cost_for_input_tokens(400_001), Some(longer));
62
+ }
63
+
64
+ #[test]
65
+ fn tier_policy_preserves_local_prices_and_ignores_provider_billing() {
66
+ let mut c = Catalog::vendored();
67
+ c.merge_local_toml("[models.\"xai/grok-4.7\".cost]\ninput=0.25")
68
+ .unwrap();
69
+ c.merge_provider_models(
70
+ "xai",
71
+ &json!({"data":[{"id":"grok-4.7","cost":{"input":999},"context_cost_tiers":[]}]}),
72
+ Utc::now(),
73
+ )
74
+ .unwrap();
75
+ let m = c.resolve("xai/grok-4.7").unwrap();
76
+ let long = m.cost_for_input_tokens(300_000).unwrap();
77
+ assert_eq!(long.input, Some(0.25));
78
+ assert_eq!(long.output, Some(12.0));
79
+ c.merge_local_toml("[models.\"xai/grok-4.7\"]\ncontext_cost_tiers=[]")
80
+ .unwrap();
81
+ c.merge_builtins();
82
+ assert_eq!(
83
+ c.resolve("xai/grok-4.7")
84
+ .unwrap()
85
+ .cost_for_input_tokens(300_000)
86
+ .unwrap()
87
+ .output,
88
+ Some(6.0)
89
+ );
90
+ assert!(c
91
+ .merge_local_toml("[models.\"xai/grok-4.7\"]\ncontext_cost_tiers=\"bad\"")
92
+ .is_err());
93
+ }
94
+
95
+ #[test]
96
+ fn launch_metadata_wins_over_stale_community_data_in_either_order() {
97
+ for builtins_first in [true, false] {
98
+ let mut c = Catalog::empty();
99
+ if builtins_first {
100
+ c.merge_builtins();
101
+ }
102
+ c.merge_models_dev(&json!({"xai":{"models":{"grok-4.7":{"family":"gpt","cost":{"input":99},"context_cost_tiers":[]}}}}).to_string(),None).unwrap();
103
+ if !builtins_first {
104
+ c.merge_builtins();
105
+ }
106
+ let m = c.resolve("xai/grok-4.7").unwrap();
107
+ assert_eq!(m.family, Some(ModelFamily::Grok));
108
+ assert_eq!(m.cost_for_input_tokens(300_000).unwrap().input, Some(4.0));
109
+ }
110
+ }
111
+
11
112
  #[test]
12
113
  fn absent_fields_are_unknown_and_zero_is_a_real_price() {
13
114
  let mut catalog = Catalog::empty();
@@ -59,12 +59,15 @@ Anthropic's `input_tokens` excludes the cache read, while the OpenAI Responses,
59
59
  Chat Completions and Gemini prompt totals include it. llmshim resolves the
60
60
  disagreement at the transport boundary so one counter means one thing.
61
61
 
62
- `cost_usd` is the USD charged for the response: uncached input, output, cache
62
+ `cost_usd` estimates standard token charges: uncached input, output, cache
63
63
  reads and cache writes each at their own catalog rate, so a cached prompt is
64
64
  never billed twice. **`null` means the catalog carries no price for the model —
65
65
  it never means free.** A model priced for input but not for the cache reads a
66
- response actually used also yields `null`, rather than a partial sum that would
67
- read as a complete one.
66
+ response actually used charges that class at the model's highest published
67
+ rate, yielding a conservative estimate. Context tiers include cached input
68
+ when choosing the rate and apply to the entire request. For native Grok 4.7,
69
+ rates double above 200,000 input tokens. Provider tool fees, regional premiums,
70
+ and priority endpoint surcharges are not included in these catalog estimates.
68
71
 
69
72
  `ProviderRequest::can_continue_from` compares endpoint, credential headers,
70
73
  settings and the full prior input prefix. `include`, `store`, `reasoning`, tool
@@ -193,7 +193,7 @@ Native `x-chatgpt.reasoning` overrides the unified mapping.
193
193
 
194
194
  xAI receives the nested native shape `reasoning: {effort}`:
195
195
 
196
- | unified | Grok 4.6 |
196
+ | unified | Grok 4.7 |
197
197
  |---|---|
198
198
  | `none` | **`low`** |
199
199
  | `low` | `low` |
@@ -202,7 +202,18 @@ xAI receives the nested native shape `reasoning: {effort}`:
202
202
  | `xhigh` | `xhigh` |
203
203
  | `max` | **`xhigh`** |
204
204
 
205
- Grok 4.6 cannot disable reasoning, so `none` clamps to `low`.
205
+ Grok 4.7 cannot disable reasoning, so `none` clamps to `low`.
206
+
207
+ Grok 4.7 always returns encrypted reasoning on Responses. llmshim preserves
208
+ the opaque payload and its provenance in `message.reasoning`, requests
209
+ `store: false`, and replays compatible payloads unchanged on subsequent turns.
210
+ Persist the full assistant message, including tool wire IDs and reasoning;
211
+ storing only its text loses this context. Decoding does not modify
212
+ `origin.model` or determine routing.
213
+
214
+ The adapter translates named tool selection to the Responses shape
215
+ `{"type":"function","name":"lookup"}`. Callers may continue to send the
216
+ Chat Completions shape with `function.name`.
206
217
 
207
218
  ## Mode mapping: `reasoning_mode: "pro"`
208
219
 
@@ -212,7 +223,7 @@ Grok 4.6 cannot disable reasoning, so `none` clamps to `low`.
212
223
  | OpenAI GPT-6 Astra | One-tier effort bump (`low → medium → high → xhigh`); explicit `max` stays `max` |
213
224
  | Anthropic | One-tier effort bump (`low → medium → high → xhigh → max`) |
214
225
  | Gemini | One-tier bump within its four-rung enum, capped at `high` |
215
- | xAI Grok 4.6 | One-tier bump, capped at `xhigh` |
226
+ | xAI Grok 4.7 | One-tier bump, capped at `xhigh` |
216
227
 
217
228
  Rules that hold across providers:
218
229
 
@@ -73,7 +73,31 @@ are stable releases only. Credentials determine which providers are listed.
73
73
 
74
74
  | ID | Display name |
75
75
  |---|---|
76
- | `xai/grok-4.6` | Grok 4.6 |
76
+ | `xai/grok-4.7` | Grok 4.7 |
77
+
78
+ Grok 4.7 accepts text and images with a 500,000-token context window.
79
+ Reasoning supports `low`, `medium`, `high` (the upstream default), and `xhigh`;
80
+ llmshim maps `none` to `low` and `max` to `xhigh`. Explicit `xai/grok-4.6`
81
+ requests remain supported, but discovery advertises 4.7.
82
+
83
+ Native xAI standard rates, USD per million tokens:
84
+
85
+ | Total input tokens, including cached input | Input | Cached input | Output |
86
+ |---|---:|---:|---:|
87
+ | Up to 200,000 | $2 | $0.50 | $6 |
88
+ | Over 200,000 | $4 | $1 | $12 |
89
+
90
+ The higher tier applies to the whole request. These are native xAI rates;
91
+ OpenRouter has its own prices. See the [release notes](https://docs.x.ai/developers/release-notes).
92
+ Grok 4.7 Fast is available in Cursor/Grok Build, not as a public xAI API model.
93
+
94
+ For OpenRouter, use `openrouter/x-ai/grok-4.7`. You can append `:nitro` and set
95
+ `x-openrouter.provider.zdr = true` together. Nitro prioritizes throughput and
96
+ allows priority endpoints, whose rates can be higher. On `/v1/chat`, put
97
+ `x-openrouter` inside `provider_config`. These routing options do not enforce
98
+ ZDR on an llmshim fallback to a different provider.
99
+ The Nitro route has replay metadata but no fixed catalog price; its
100
+ `cost_usd` remains `null` unless local policy supplies rates.
77
101
 
78
102
  ### ChatGPT subscription
79
103
 
@@ -11,8 +11,8 @@ different native API and translates only the fields that API understands.
11
11
  | Google Gemini | `generateContent` / `streamGenerateContent` | `gemini*` | `x-gemini` |
12
12
  | xAI | Responses API | `grok*` | none |
13
13
  | OpenRouter | Chat Completions (aggregator) | none — address as `openrouter/<vendor>/<model>` | `x-openrouter` |
14
- | vLLM | Chat Completions (self-hosted, `VLLM_BASE_URL`) | none — address as `vllm/<served-model>` | `x-vllm` |
15
- | SGLang | Chat Completions (self-hosted, `SGLANG_BASE_URL`) | none — address as `sglang/<served-model>` | `x-sglang` |
14
+ | vLLM | Chat Completions (self-hosted, `VLLM_BASE_URL`); Responses API with `VLLM_WIRE=responses` | none — address as `vllm/<served-model>` | `x-vllm` |
15
+ | SGLang | Chat Completions (self-hosted, `SGLANG_BASE_URL`); Responses API with `SGLANG_WIRE=responses` | none — address as `sglang/<served-model>` | `x-sglang` |
16
16
 
17
17
  An explicit address such as `anthropic/claude-sonnet-5` avoids inference.
18
18
  The named provider must be registered in the Router—that normally means its