llmshim 0.7.2__tar.gz → 0.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (196) hide show
  1. {llmshim-0.7.2 → llmshim-0.8.0}/Cargo.lock +1 -1
  2. {llmshim-0.7.2 → llmshim-0.8.0}/Cargo.toml +1 -1
  3. {llmshim-0.7.2 → llmshim-0.8.0}/PKG-INFO +1 -1
  4. {llmshim-0.7.2 → llmshim-0.8.0}/README.md +12 -0
  5. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/reference/providers.md +2 -2
  6. {llmshim-0.7.2 → llmshim-0.8.0}/src/breaker.rs +4 -0
  7. {llmshim-0.7.2 → llmshim-0.8.0}/src/cache.rs +1 -0
  8. {llmshim-0.7.2 → llmshim-0.8.0}/src/client.rs +5 -0
  9. {llmshim-0.7.2 → llmshim-0.8.0}/src/error.rs +9 -1
  10. {llmshim-0.7.2 → llmshim-0.8.0}/src/gateway/http.rs +6 -1
  11. {llmshim-0.7.2 → llmshim-0.8.0}/src/providers/anthropic.rs +5 -3
  12. {llmshim-0.7.2 → llmshim-0.8.0}/src/providers/anthropic_reasoning.rs +1 -0
  13. {llmshim-0.7.2 → llmshim-0.8.0}/src/providers/chatgpt/auth.rs +1 -0
  14. {llmshim-0.7.2 → llmshim-0.8.0}/src/providers/gemini.rs +3 -0
  15. {llmshim-0.7.2 → llmshim-0.8.0}/src/providers/openai.rs +9 -2
  16. {llmshim-0.7.2 → llmshim-0.8.0}/src/providers/openai_compat.rs +92 -17
  17. {llmshim-0.7.2 → llmshim-0.8.0}/src/providers/openrouter.rs +2 -0
  18. {llmshim-0.7.2 → llmshim-0.8.0}/src/providers/xai.rs +3 -0
  19. {llmshim-0.7.2 → llmshim-0.8.0}/src/proxy/error.rs +1 -1
  20. {llmshim-0.7.2 → llmshim-0.8.0}/src/reasoning/normalize.rs +38 -12
  21. {llmshim-0.7.2 → llmshim-0.8.0}/src/reasoning.rs +1 -0
  22. {llmshim-0.7.2 → llmshim-0.8.0}/src/router.rs +24 -12
  23. {llmshim-0.7.2 → llmshim-0.8.0}/src/schema/validate.rs +2 -0
  24. {llmshim-0.7.2 → llmshim-0.8.0}/src/shim.rs +20 -6
  25. {llmshim-0.7.2 → llmshim-0.8.0}/src/toolcall.rs +2 -0
  26. llmshim-0.8.0/tests/fixtures/sglang-models.toml +6 -0
  27. llmshim-0.8.0/tests/fixtures/sglang-responses-stream.sse +75 -0
  28. llmshim-0.8.0/tests/fixtures/sglang-responses-turn1.json +94 -0
  29. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_long_context.rs +6 -2
  30. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_proxy.rs +3 -4
  31. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_sglang.rs +69 -0
  32. llmshim-0.8.0/tests/unit_client_retry_after.rs +97 -0
  33. llmshim-0.8.0/tests/unit_openai_compat_responses.rs +290 -0
  34. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_reasoning.rs +62 -0
  35. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_router.rs +3 -1
  36. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_shim.rs +3 -3
  37. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_sse.rs +3 -3
  38. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_toolcall.rs +1 -0
  39. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_wire.rs +7 -7
  40. {llmshim-0.7.2 → llmshim-0.8.0}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
  41. {llmshim-0.7.2 → llmshim-0.8.0}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
  42. {llmshim-0.7.2 → llmshim-0.8.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  43. {llmshim-0.7.2 → llmshim-0.8.0}/.github/scripts/npm-verify-served.sh +0 -0
  44. {llmshim-0.7.2 → llmshim-0.8.0}/.github/workflows/catalog-refresh.yml +0 -0
  45. {llmshim-0.7.2 → llmshim-0.8.0}/.github/workflows/pages.yml +0 -0
  46. {llmshim-0.7.2 → llmshim-0.8.0}/.github/workflows/release.yml +0 -0
  47. {llmshim-0.7.2 → llmshim-0.8.0}/.gitignore +0 -0
  48. {llmshim-0.7.2 → llmshim-0.8.0}/CLAUDE.md +0 -0
  49. {llmshim-0.7.2 → llmshim-0.8.0}/CODE_OF_CONDUCT.md +0 -0
  50. {llmshim-0.7.2 → llmshim-0.8.0}/CONTRIBUTING.md +0 -0
  51. {llmshim-0.7.2 → llmshim-0.8.0}/LICENSE-APACHE +0 -0
  52. {llmshim-0.7.2 → llmshim-0.8.0}/LICENSE-MIT +0 -0
  53. {llmshim-0.7.2 → llmshim-0.8.0}/NOTICE +0 -0
  54. {llmshim-0.7.2 → llmshim-0.8.0}/SECURITY.md +0 -0
  55. {llmshim-0.7.2 → llmshim-0.8.0}/benchmarks/bench.rs +0 -0
  56. {llmshim-0.7.2 → llmshim-0.8.0}/benchmarks/bench_python.py +0 -0
  57. {llmshim-0.7.2 → llmshim-0.8.0}/benchmarks/gateway_loadtest.rs +0 -0
  58. {llmshim-0.7.2 → llmshim-0.8.0}/benchmarks/loadtest.rs +0 -0
  59. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/Cargo.toml +0 -0
  60. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/LICENSE-APACHE +0 -0
  61. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/LICENSE-MIT +0 -0
  62. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/README.md +0 -0
  63. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/data/LICENSE.models.dev +0 -0
  64. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/data/README.md +0 -0
  65. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/data/models.dev.json +0 -0
  66. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/src/aliases.rs +0 -0
  67. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/src/builtin.rs +0 -0
  68. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/src/capabilities.rs +0 -0
  69. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/src/lib.rs +0 -0
  70. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/src/merge.rs +0 -0
  71. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/src/parse.rs +0 -0
  72. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/src/refresh.rs +0 -0
  73. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/src/types.rs +0 -0
  74. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/tests/catalog.rs +0 -0
  75. {llmshim-0.7.2 → llmshim-0.8.0}/crates/llmshim-catalog/tests/refresh.rs +0 -0
  76. {llmshim-0.7.2 → llmshim-0.8.0}/docs/.gitignore +0 -0
  77. {llmshim-0.7.2 → llmshim-0.8.0}/docs/book.toml +0 -0
  78. {llmshim-0.7.2 → llmshim-0.8.0}/docs/mermaid-init.js +0 -0
  79. {llmshim-0.7.2 → llmshim-0.8.0}/docs/mermaid.min.js +0 -0
  80. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/SUMMARY.md +0 -0
  81. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/concepts/contracts.md +0 -0
  82. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/concepts/conversations.md +0 -0
  83. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/concepts/portability.md +0 -0
  84. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/concepts/routing.md +0 -0
  85. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/concepts/translation-flow.md +0 -0
  86. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/guides/caching.md +0 -0
  87. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/guides/capabilities.md +0 -0
  88. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/guides/fallbacks.md +0 -0
  89. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/guides/images.md +0 -0
  90. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/guides/native-controls.md +0 -0
  91. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/guides/reasoning.md +0 -0
  92. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/guides/schemas.md +0 -0
  93. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/guides/streaming.md +0 -0
  94. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/guides/tools.md +0 -0
  95. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/introduction.md +0 -0
  96. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/proxy/deployment.md +0 -0
  97. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/proxy/http-api.md +0 -0
  98. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/proxy/native-apis.md +0 -0
  99. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/proxy/scaling.md +0 -0
  100. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/reference/api.md +0 -0
  101. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/reference/cli.md +0 -0
  102. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/reference/configuration.md +0 -0
  103. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/reference/errors.md +0 -0
  104. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/reference/models.md +0 -0
  105. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/reference/request-fields.md +0 -0
  106. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/reference/surfaces.md +0 -0
  107. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/start/choose.md +0 -0
  108. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/start/cli.md +0 -0
  109. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/start/clients.md +0 -0
  110. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/start/configure.md +0 -0
  111. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/start/proxy.md +0 -0
  112. {llmshim-0.7.2 → llmshim-0.8.0}/docs/src/start/rust.md +0 -0
  113. {llmshim-0.7.2 → llmshim-0.8.0}/examples/chat.rs +0 -0
  114. {llmshim-0.7.2 → llmshim-0.8.0}/examples/stream.rs +0 -0
  115. {llmshim-0.7.2 → llmshim-0.8.0}/llmshim/__init__.py +0 -0
  116. {llmshim-0.7.2 → llmshim-0.8.0}/llmshim/_client.py +0 -0
  117. {llmshim-0.7.2 → llmshim-0.8.0}/llmshim/_server.py +0 -0
  118. {llmshim-0.7.2 → llmshim-0.8.0}/llmshim/types.py +0 -0
  119. {llmshim-0.7.2 → llmshim-0.8.0}/pyproject.toml +0 -0
  120. {llmshim-0.7.2 → llmshim-0.8.0}/src/cli.rs +0 -0
  121. {llmshim-0.7.2 → llmshim-0.8.0}/src/config.rs +0 -0
  122. {llmshim-0.7.2 → llmshim-0.8.0}/src/cost.rs +0 -0
  123. {llmshim-0.7.2 → llmshim-0.8.0}/src/env.rs +0 -0
  124. {llmshim-0.7.2 → llmshim-0.8.0}/src/error/normalize.rs +0 -0
  125. {llmshim-0.7.2 → llmshim-0.8.0}/src/fallback.rs +0 -0
  126. {llmshim-0.7.2 → llmshim-0.8.0}/src/gateway/auth.rs +0 -0
  127. {llmshim-0.7.2 → llmshim-0.8.0}/src/gateway/distributed.rs +0 -0
  128. {llmshim-0.7.2 → llmshim-0.8.0}/src/gateway/idempotency.rs +0 -0
  129. {llmshim-0.7.2 → llmshim-0.8.0}/src/gateway/metrics.rs +0 -0
  130. {llmshim-0.7.2 → llmshim-0.8.0}/src/gateway/mod.rs +0 -0
  131. {llmshim-0.7.2 → llmshim-0.8.0}/src/gateway/quota.rs +0 -0
  132. {llmshim-0.7.2 → llmshim-0.8.0}/src/lib.rs +0 -0
  133. {llmshim-0.7.2 → llmshim-0.8.0}/src/log.rs +0 -0
  134. {llmshim-0.7.2 → llmshim-0.8.0}/src/main.rs +0 -0
  135. {llmshim-0.7.2 → llmshim-0.8.0}/src/models.rs +0 -0
  136. {llmshim-0.7.2 → llmshim-0.8.0}/src/provider.rs +0 -0
  137. {llmshim-0.7.2 → llmshim-0.8.0}/src/providers/anthropic_signature.rs +0 -0
  138. {llmshim-0.7.2 → llmshim-0.8.0}/src/providers/chatgpt/mod.rs +0 -0
  139. {llmshim-0.7.2 → llmshim-0.8.0}/src/providers/chatgpt/streaming.rs +0 -0
  140. {llmshim-0.7.2 → llmshim-0.8.0}/src/providers/mod.rs +0 -0
  141. {llmshim-0.7.2 → llmshim-0.8.0}/src/proxy/convert.rs +0 -0
  142. {llmshim-0.7.2 → llmshim-0.8.0}/src/proxy/handlers.rs +0 -0
  143. {llmshim-0.7.2 → llmshim-0.8.0}/src/proxy/health.rs +0 -0
  144. {llmshim-0.7.2 → llmshim-0.8.0}/src/proxy/mod.rs +0 -0
  145. {llmshim-0.7.2 → llmshim-0.8.0}/src/proxy/ratelimit.rs +0 -0
  146. {llmshim-0.7.2 → llmshim-0.8.0}/src/proxy/types.rs +0 -0
  147. {llmshim-0.7.2 → llmshim-0.8.0}/src/proxy/wire/mod.rs +0 -0
  148. {llmshim-0.7.2 → llmshim-0.8.0}/src/proxy/wire/receipts.rs +0 -0
  149. {llmshim-0.7.2 → llmshim-0.8.0}/src/schema/memo.rs +0 -0
  150. {llmshim-0.7.2 → llmshim-0.8.0}/src/schema/mod.rs +0 -0
  151. {llmshim-0.7.2 → llmshim-0.8.0}/src/schema/walk.rs +0 -0
  152. {llmshim-0.7.2 → llmshim-0.8.0}/src/streaming.rs +0 -0
  153. {llmshim-0.7.2 → llmshim-0.8.0}/src/toolcall/streaming.rs +0 -0
  154. {llmshim-0.7.2 → llmshim-0.8.0}/src/usage.rs +0 -0
  155. {llmshim-0.7.2 → llmshim-0.8.0}/src/vision.rs +0 -0
  156. {llmshim-0.7.2 → llmshim-0.8.0}/tests/fixtures/chatgpt-red.png +0 -0
  157. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration.rs +0 -0
  158. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_chatgpt.rs +0 -0
  159. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_chatgpt_proxy.rs +0 -0
  160. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_current_models.rs +0 -0
  161. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_fallback.rs +0 -0
  162. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_gemini.rs +0 -0
  163. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_gemini_tools.rs +0 -0
  164. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_multimodel.rs +0 -0
  165. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_openrouter.rs +0 -0
  166. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_thinking.rs +0 -0
  167. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_tool_roundtrip.rs +0 -0
  168. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_vision.rs +0 -0
  169. {llmshim-0.7.2 → llmshim-0.8.0}/tests/integration_xai.rs +0 -0
  170. {llmshim-0.7.2 → llmshim-0.8.0}/tests/support/completion_status.rs +0 -0
  171. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_advertised_models.rs +0 -0
  172. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_anthropic.rs +0 -0
  173. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_cache.rs +0 -0
  174. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_chatgpt.rs +0 -0
  175. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_cli.rs +0 -0
  176. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_client_breaker.rs +0 -0
  177. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_fable.rs +0 -0
  178. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_fallback.rs +0 -0
  179. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_fast_mode.rs +0 -0
  180. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_gemini.rs +0 -0
  181. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_log.rs +0 -0
  182. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_models.rs +0 -0
  183. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_multimodel.rs +0 -0
  184. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_openai.rs +0 -0
  185. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_openai_compat.rs +0 -0
  186. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_openrouter.rs +0 -0
  187. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_provider_contracts.rs +0 -0
  188. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_proxy.rs +0 -0
  189. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_proxy_convert.rs +0 -0
  190. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_reasoning_profile.rs +0 -0
  191. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_schema.rs +0 -0
  192. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_signature.rs +0 -0
  193. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_tools.rs +0 -0
  194. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_usage.rs +0 -0
  195. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_vision.rs +0 -0
  196. {llmshim-0.7.2 → llmshim-0.8.0}/tests/unit_xai.rs +0 -0
@@ -1180,7 +1180,7 @@ checksum = "6373607a59f0be73a39b6fe456b8192fcc3585f602af20751600e974dd455e77"
1180
1180
 
1181
1181
  [[package]]
1182
1182
  name = "llmshim"
1183
- version = "0.7.2"
1183
+ version = "0.8.0"
1184
1184
  dependencies = [
1185
1185
  "async-stream",
1186
1186
  "async-trait",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "llmshim"
3
- version = "0.7.2"
3
+ version = "0.8.0"
4
4
  edition = "2021"
5
5
  description = "Blazing fast LLM API translation layer in pure Rust"
6
6
  license = "MIT OR Apache-2.0"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llmshim
3
- Version: 0.7.2
3
+ Version: 0.8.0
4
4
  Classifier: Development Status :: 4 - Beta
5
5
  Classifier: Intended Audience :: Developers
6
6
  Classifier: License :: OSI Approved :: MIT License
@@ -100,6 +100,18 @@ Then address the served model as `sglang/<served-model>` or `vllm/<served-model>
100
100
  (e.g. `sglang/Qwen/Qwen3.6-35B-A3B-FP8`). Server-specific knobs go under
101
101
  `x-vllm` / `x-sglang`.
102
102
 
103
+ A server that also serves `/v1/responses` (SGLang does) can be spoken to on
104
+ that wire with `SGLANG_WIRE=responses` (or `VLLM_WIRE=responses`). Reasoning
105
+ then comes back as an item with its own id rather than bare `reasoning_content`,
106
+ and is replayed as that item. Replay is gated on a known model family on every
107
+ wire, and the public catalog does not know a served model: declare it once in
108
+ `.llmshim/models.toml` —
109
+
110
+ ```toml
111
+ [models."sglang/<served-model>"]
112
+ family = "qwen"
113
+ ```
114
+
103
115
  Or persist them to the config file (used by all three surfaces):
104
116
 
105
117
  ```bash
@@ -11,8 +11,8 @@ different native API and translates only the fields that API understands.
11
11
  | Google Gemini | `generateContent` / `streamGenerateContent` | `gemini*` | `x-gemini` |
12
12
  | xAI | Responses API | `grok*` | none |
13
13
  | OpenRouter | Chat Completions (aggregator) | none — address as `openrouter/<vendor>/<model>` | `x-openrouter` |
14
- | vLLM | Chat Completions (self-hosted, `VLLM_BASE_URL`) | none — address as `vllm/<served-model>` | `x-vllm` |
15
- | SGLang | Chat Completions (self-hosted, `SGLANG_BASE_URL`) | none — address as `sglang/<served-model>` | `x-sglang` |
14
+ | vLLM | Chat Completions (self-hosted, `VLLM_BASE_URL`); Responses API with `VLLM_WIRE=responses` | none — address as `vllm/<served-model>` | `x-vllm` |
15
+ | SGLang | Chat Completions (self-hosted, `SGLANG_BASE_URL`); Responses API with `SGLANG_WIRE=responses` | none — address as `sglang/<served-model>` | `x-sglang` |
16
16
 
17
17
  An explicit address such as `anthropic/claude-sonnet-5` avoids inference.
18
18
  The named provider must be registered in the Router—that normally means its
@@ -304,6 +304,7 @@ mod tests {
304
304
  ShimError::ProviderError {
305
305
  status: 503,
306
306
  body: "down".into(),
307
+ retry_after: None,
307
308
  }
308
309
  }
309
310
 
@@ -327,17 +328,20 @@ mod tests {
327
328
  assert!(!counts_toward_health(&ShimError::ProviderError {
328
329
  status: 429,
329
330
  body: "slow down".into(),
331
+ retry_after: None,
330
332
  }));
331
333
  for status in [500, 502, 503, 504, 529] {
332
334
  assert!(counts_toward_health(&ShimError::ProviderError {
333
335
  status,
334
336
  body: String::new(),
337
+ retry_after: None,
335
338
  }));
336
339
  }
337
340
  for status in [400, 401, 403, 404, 422] {
338
341
  assert!(!counts_toward_health(&ShimError::ProviderError {
339
342
  status,
340
343
  body: String::new(),
344
+ retry_after: None,
341
345
  }));
342
346
  }
343
347
  assert!(!counts_toward_health(&ShimError::MissingModel));
@@ -91,6 +91,7 @@ fn invalid(message: &str) -> ShimError {
91
91
  ShimError::ProviderError {
92
92
  status: 400,
93
93
  body: format!("invalid x-cache: {message}"),
94
+ retry_after: None,
94
95
  }
95
96
  }
96
97
  fn policy(request: &Value) -> Result<Option<CachePolicy>> {
@@ -169,10 +169,15 @@ impl ShimClient {
169
169
  tokio::time::sleep(wait).await;
170
170
  continue;
171
171
  }
172
+ // Read before the body is consumed: the header is the
173
+ // server's own wait, and a caller with its own backoff
174
+ // above this client gets to honour it too.
175
+ let retry_after = parse_retry_after(resp.headers());
172
176
  let body = resp.text().await.unwrap_or_default();
173
177
  return Err(ShimError::ProviderError {
174
178
  status: status_code,
175
179
  body,
180
+ retry_after,
176
181
  });
177
182
  }
178
183
  // Transport errors carry no headers: always jittered backoff.
@@ -19,8 +19,16 @@ pub enum ShimError {
19
19
  #[error("JSON error: {0}")]
20
20
  Json(#[from] serde_json::Error),
21
21
 
22
+ /// A non-success response from the provider. `retry_after` is the
23
+ /// server's own `Retry-After` (delay-seconds or HTTP-date, parsed at
24
+ /// receipt), so a caller with its own backoff can wait what was asked
25
+ /// instead of guessing; `None` when the header was absent or unparseable.
22
26
  #[error("provider error ({status}): {body}")]
23
- ProviderError { status: u16, body: String },
27
+ ProviderError {
28
+ status: u16,
29
+ body: String,
30
+ retry_after: Option<std::time::Duration>,
31
+ },
24
32
 
25
33
  #[error("stream error: {0}")]
26
34
  Stream(String),
@@ -55,7 +55,9 @@ pub struct RealDispatch {
55
55
  impl RealDispatch {
56
56
  fn map_err(err: ShimError) -> DispatchError {
57
57
  match err {
58
- ShimError::ProviderError { status: 429, body } => DispatchError {
58
+ ShimError::ProviderError {
59
+ status: 429, body, ..
60
+ } => DispatchError {
59
61
  message: body,
60
62
  retry_after: Some(penalty_duration()),
61
63
  },
@@ -373,6 +375,7 @@ impl GatewayState {
373
375
  key to run it uncharged.\",\"type\":\"invalid_request_error\",\
374
376
  \"param\":\"model\",\"code\":\"unpriceable_under_budget\"}}}}"
375
377
  ),
378
+ retry_after: None,
376
379
  }))
377
380
  }
378
381
  }
@@ -425,10 +428,12 @@ fn gateway_err_to_api(state: &GatewayState, err: GatewayError) -> ApiError {
425
428
  GatewayError::Shutdown => ApiError::from(ShimError::ProviderError {
426
429
  status: 503,
427
430
  body: "gateway shutting down".to_string(),
431
+ retry_after: None,
428
432
  }),
429
433
  GatewayError::Upstream(message) => ApiError::from(ShimError::ProviderError {
430
434
  status: 502,
431
435
  body: message,
436
+ retry_after: None,
432
437
  }),
433
438
  }
434
439
  }
@@ -382,6 +382,7 @@ fn transform_response_to_openai(model: &str, resp: &Value) -> Result<Value> {
382
382
  return Err(ShimError::ProviderError {
383
383
  status: 502,
384
384
  body: "Anthropic response has no supported terminal stop reason".into(),
385
+ retry_after: None,
385
386
  })
386
387
  }
387
388
  };
@@ -643,7 +644,7 @@ impl Provider for Anthropic {
643
644
  if let Some(thinking) = body_obj.get("thinking") {
644
645
  if thinking["type"] != "adaptive" {
645
646
  return Err(ShimError::ProviderError { status: 400, body:
646
- "Claude Fable requires adaptive thinking; use reasoning_effort to control depth".into() });
647
+ "Claude Fable requires adaptive thinking; use reasoning_effort to control depth".into(), retry_after: None });
647
648
  }
648
649
  }
649
650
  if body_obj
@@ -653,7 +654,7 @@ impl Provider for Anthropic {
653
654
  .is_some_and(|message| message["role"] == "assistant")
654
655
  {
655
656
  return Err(ShimError::ProviderError { status: 400, body:
656
- "Claude Fable does not support assistant prefill; end the request with a user turn".into() });
657
+ "Claude Fable does not support assistant prefill; end the request with a user turn".into(), retry_after: None });
657
658
  }
658
659
  }
659
660
  if model == "claude-fable-5-1"
@@ -663,7 +664,7 @@ impl Provider for Anthropic {
663
664
  .is_some_and(|kind| matches!(kind, "any" | "tool"))
664
665
  {
665
666
  return Err(ShimError::ProviderError { status: 400, body:
666
- "Claude Fable 5.1 supports only auto or none tool choice; request the desired tool in the prompt".into() });
667
+ "Claude Fable 5.1 supports only auto or none tool choice; request the desired tool in the prompt".into(), retry_after: None });
667
668
  }
668
669
 
669
670
  // Fast mode support: extract "speed" from the request and apply
@@ -768,6 +769,7 @@ impl Anthropic {
768
769
  return Err(ShimError::ProviderError {
769
770
  status: 400,
770
771
  body: msg.to_string(),
772
+ retry_after: None,
771
773
  });
772
774
  }
773
775
 
@@ -107,5 +107,6 @@ fn invalid() -> ShimError {
107
107
  ShimError::ProviderError {
108
108
  status: 400,
109
109
  body: "max_tokens cannot accommodate the configured reasoning budget".into(),
110
+ retry_after: None,
110
111
  }
111
112
  }
@@ -23,6 +23,7 @@ pub(super) fn auth_error(status: u16, message: &str) -> ShimError {
23
23
  ShimError::ProviderError {
24
24
  status,
25
25
  body: format!("ChatGPT: {message}"),
26
+ retry_after: None,
26
27
  }
27
28
  }
28
29
 
@@ -288,6 +288,7 @@ fn transform_response_to_openai(model: &str, resp: &Value) -> Result<Value> {
288
288
  .ok_or_else(|| ShimError::ProviderError {
289
289
  status: 500,
290
290
  body: format!("no candidates in response: {}", resp),
291
+ retry_after: None,
291
292
  })?;
292
293
 
293
294
  let parts = candidate
@@ -354,6 +355,7 @@ fn transform_response_to_openai(model: &str, resp: &Value) -> Result<Value> {
354
355
  return Err(ShimError::ProviderError {
355
356
  status: 502,
356
357
  body: "Gemini response has no supported terminal finish reason".into(),
358
+ retry_after: None,
357
359
  })
358
360
  }
359
361
  };
@@ -606,6 +608,7 @@ impl Gemini {
606
608
  return Err(ShimError::ProviderError {
607
609
  status: code,
608
610
  body: msg.to_string(),
611
+ retry_after: None,
609
612
  });
610
613
  }
611
614
  transform_response_to_openai(model, &response)
@@ -363,7 +363,7 @@ impl Provider for OpenAi {
363
363
  }
364
364
 
365
365
  impl OpenAi {
366
- fn transform_response_native(&self, model: &str, response: Value) -> Result<Value> {
366
+ pub(crate) fn transform_response_native(&self, model: &str, response: Value) -> Result<Value> {
367
367
  // Check for error (Responses API returns "error": null on success)
368
368
  if let Some(err) = response.get("error") {
369
369
  if !err.is_null() {
@@ -374,6 +374,7 @@ impl OpenAi {
374
374
  return Err(ShimError::ProviderError {
375
375
  status: 400,
376
376
  body: msg.to_string(),
377
+ retry_after: None,
377
378
  });
378
379
  }
379
380
  }
@@ -384,6 +385,7 @@ impl OpenAi {
384
385
  .ok_or_else(|| ShimError::ProviderError {
385
386
  status: 500,
386
387
  body: "no output in response".to_string(),
388
+ retry_after: None,
387
389
  })?;
388
390
 
389
391
  // Extract reasoning summary
@@ -449,6 +451,7 @@ impl OpenAi {
449
451
  return Err(ShimError::ProviderError {
450
452
  status: 502,
451
453
  body: "OpenAI response has no supported terminal status".into(),
454
+ retry_after: None,
452
455
  })
453
456
  }
454
457
  };
@@ -471,7 +474,11 @@ impl OpenAi {
471
474
  }
472
475
 
473
476
  impl OpenAi {
474
- fn transform_stream_chunk_native(&self, model: &str, chunk: &str) -> Result<Option<String>> {
477
+ pub(crate) fn transform_stream_chunk_native(
478
+ &self,
479
+ model: &str,
480
+ chunk: &str,
481
+ ) -> Result<Option<String>> {
475
482
  let trimmed = chunk.trim();
476
483
  if trimmed.is_empty() || trimmed == "[DONE]" {
477
484
  return Ok(None);
@@ -1,5 +1,7 @@
1
1
  use crate::error::{Result, ShimError};
2
2
  use crate::provider::{Provider, ProviderRequest};
3
+ use crate::providers::openai::OpenAi;
4
+ use crate::reasoning::{ReplayTarget, WireFormat};
3
5
  use crate::vision;
4
6
  use serde_json::{json, Value};
5
7
 
@@ -17,10 +19,18 @@ use serde_json::{json, Value};
17
19
  /// `name` (e.g. `"vllm"` / `"sglang"`) is both the provider key and the
18
20
  /// extension namespace: server-specific params (`chat_template_kwargs`,
19
21
  /// `separate_reasoning`, `guided_json`, `top_k`, …) go under `x-<name>`.
22
+ ///
23
+ /// The wire is Chat Completions unless [`OpenAiCompatible::with_wire`] selects
24
+ /// the Responses API, which SGLang also serves at `<base>/responses`. On that
25
+ /// wire reasoning comes back as an item with its own id, so it can be replayed
26
+ /// as a keyed block instead of bare `reasoning_content`; the request and
27
+ /// response translation is the OpenAI adapter's, with this server's URL, its
28
+ /// optional auth, and its `x-<name>` namespace.
20
29
  pub struct OpenAiCompatible {
21
30
  pub name: String,
22
31
  pub base_url: String,
23
32
  pub api_key: Option<String>,
33
+ wire: WireFormat,
24
34
  }
25
35
 
26
36
  impl OpenAiCompatible {
@@ -33,7 +43,66 @@ impl OpenAiCompatible {
33
43
  name: name.into(),
34
44
  base_url: base_url.into(),
35
45
  api_key,
46
+ wire: WireFormat::OpenAiChat,
47
+ }
48
+ }
49
+
50
+ /// Select the wire this server is spoken to on: `OpenAiChat` (the
51
+ /// default, `<base>/chat/completions`) or `OpenAiResponses`
52
+ /// (`<base>/responses`). The other wires are not OpenAI-compatible and
53
+ /// are a caller error.
54
+ pub fn with_wire(mut self, wire: WireFormat) -> Self {
55
+ assert!(
56
+ matches!(wire, WireFormat::OpenAiChat | WireFormat::OpenAiResponses),
57
+ "an OpenAI-compatible server speaks Chat Completions or Responses, not {wire:?}"
58
+ );
59
+ self.wire = wire;
60
+ self
61
+ }
62
+
63
+ pub fn wire(&self) -> WireFormat {
64
+ self.wire
65
+ }
66
+
67
+ /// Auth is optional — self-hosted servers are unauthenticated unless
68
+ /// launched with --api-key.
69
+ fn headers(&self) -> Vec<(String, String)> {
70
+ let mut headers = vec![("Content-Type".to_string(), "application/json".to_string())];
71
+ if let Some(key) = self.api_key.as_deref().filter(|k| !k.is_empty()) {
72
+ headers.push(("Authorization".to_string(), format!("Bearer {key}")));
73
+ }
74
+ headers
75
+ }
76
+
77
+ /// The OpenAI adapter pointed at this server, for the Responses wire. Its
78
+ /// translation is reused whole; only the URL, the auth and the extension
79
+ /// namespace are this server's.
80
+ fn responses_adapter(&self) -> OpenAi {
81
+ OpenAi::new(self.api_key.clone().unwrap_or_default())
82
+ .with_base_url(self.base_url.trim_end_matches('/').to_string())
83
+ }
84
+
85
+ fn transform_request_responses(&self, model: &str, request: &Value) -> Result<ProviderRequest> {
86
+ // The OpenAI adapter reads its overrides from `x-openai`, and applies
87
+ // them before the stateless and native-tool passes. Moving `x-<name>`
88
+ // there keeps that order, so an override cannot re-enable storage.
89
+ let mut request = request.clone();
90
+ let namespace = format!("x-{}", self.name);
91
+ if let Some(ext) = request
92
+ .as_object_mut()
93
+ .and_then(|obj| obj.remove(&namespace))
94
+ .and_then(|ext| ext.as_object().cloned())
95
+ {
96
+ let target = request["x-openai"].as_object().cloned().unwrap_or_default();
97
+ request["x-openai"] = Value::Object(target.into_iter().chain(ext).collect());
36
98
  }
99
+ let mut sent = self.responses_adapter().transform_request_for_target(
100
+ model,
101
+ &request,
102
+ &self.replay_target(model),
103
+ )?;
104
+ sent.headers = self.headers();
105
+ Ok(sent)
37
106
  }
38
107
  }
39
108
 
@@ -76,16 +145,15 @@ impl Provider for OpenAiCompatible {
76
145
  &self.name
77
146
  }
78
147
 
79
- fn replay_target(&self, model: &str) -> crate::reasoning::ReplayTarget {
80
- crate::reasoning::ReplayTarget::new(
81
- self.name(),
82
- model,
83
- crate::reasoning::WireFormat::OpenAiChat,
84
- )
85
- .bind_account(&self.base_url, self.api_key.as_deref())
148
+ fn replay_target(&self, model: &str) -> ReplayTarget {
149
+ ReplayTarget::new(self.name(), model, self.wire)
150
+ .bind_account(&self.base_url, self.api_key.as_deref())
86
151
  }
87
152
 
88
153
  fn transform_request(&self, model: &str, request: &Value) -> Result<ProviderRequest> {
154
+ if self.wire == WireFormat::OpenAiResponses {
155
+ return self.transform_request_responses(model, request);
156
+ }
89
157
  let request = crate::schema::prepare_request(request);
90
158
  let request =
91
159
  crate::cache::prepare_request(&request, crate::reasoning::WireFormat::OpenAiChat)?;
@@ -140,14 +208,7 @@ impl Provider for OpenAiCompatible {
140
208
  }
141
209
  }
142
210
 
143
- let mut headers = vec![("Content-Type".to_string(), "application/json".to_string())];
144
- // Auth is optional — self-hosted servers are unauthenticated unless
145
- // launched with --api-key.
146
- if let Some(key) = &self.api_key {
147
- if !key.is_empty() {
148
- headers.push(("Authorization".to_string(), format!("Bearer {key}")));
149
- }
150
- }
211
+ let headers = self.headers();
151
212
 
152
213
  let url = format!("{}/chat/completions", self.base_url.trim_end_matches('/'));
153
214
  crate::toolcall::validate_native(&body, &self.replay_target(model))?;
@@ -167,14 +228,26 @@ impl Provider for OpenAiCompatible {
167
228
 
168
229
  fn transform_response(&self, model: &str, response: Value) -> Result<Value> {
169
230
  let native = response.clone();
170
- let mut result = self.transform_response_native(model, response)?;
231
+ let mut result = match self.wire {
232
+ WireFormat::OpenAiResponses => self
233
+ .responses_adapter()
234
+ .transform_response_native(model, response)?,
235
+ _ => self.transform_response_native(model, response)?,
236
+ };
237
+ // Captured against this server's target, not the OpenAI adapter's, so
238
+ // the block's origin names the provider that actually issued it.
171
239
  crate::reasoning::capture_response(&self.replay_target(model), &native, &mut result);
172
240
  crate::toolcall::capture_response(&self.replay_target(model), &native, &mut result)?;
173
241
  Ok(result)
174
242
  }
175
243
 
176
244
  fn transform_stream_chunk(&self, model: &str, chunk: &str) -> Result<Option<String>> {
177
- let result = self.transform_stream_chunk_native(model, chunk)?;
245
+ let result = match self.wire {
246
+ WireFormat::OpenAiResponses => self
247
+ .responses_adapter()
248
+ .transform_stream_chunk_native(model, chunk)?,
249
+ _ => self.transform_stream_chunk_native(model, chunk)?,
250
+ };
178
251
  let native: Value = match serde_json::from_str(chunk) {
179
252
  Ok(v) => v,
180
253
  Err(_) => return Ok(result),
@@ -189,6 +262,7 @@ impl OpenAiCompatible {
189
262
  return Err(ShimError::ProviderError {
190
263
  status: 502,
191
264
  body: "invalid upstream response shape".into(),
265
+ retry_after: None,
192
266
  });
193
267
  }
194
268
  if let Some(err) = response.get("error") {
@@ -202,6 +276,7 @@ impl OpenAiCompatible {
202
276
  return Err(ShimError::ProviderError {
203
277
  status,
204
278
  body: message,
279
+ retry_after: None,
205
280
  });
206
281
  }
207
282
  }
@@ -267,6 +267,7 @@ impl OpenRouter {
267
267
  return Err(ShimError::ProviderError {
268
268
  status: 502,
269
269
  body: "invalid upstream response shape".into(),
270
+ retry_after: None,
270
271
  });
271
272
  }
272
273
  // Non-stream errors usually surface via HTTP status, but a body-level
@@ -282,6 +283,7 @@ impl OpenRouter {
282
283
  return Err(ShimError::ProviderError {
283
284
  status,
284
285
  body: message,
286
+ retry_after: None,
285
287
  });
286
288
  }
287
289
  }
@@ -370,6 +370,7 @@ impl Xai {
370
370
  return Err(ShimError::ProviderError {
371
371
  status: 400,
372
372
  body: msg.to_string(),
373
+ retry_after: None,
373
374
  });
374
375
  }
375
376
  }
@@ -380,6 +381,7 @@ impl Xai {
380
381
  .ok_or_else(|| ShimError::ProviderError {
381
382
  status: 500,
382
383
  body: "no output in response".to_string(),
384
+ retry_after: None,
383
385
  })?;
384
386
 
385
387
  let mut text_content: Option<String> = None;
@@ -440,6 +442,7 @@ impl Xai {
440
442
  return Err(ShimError::ProviderError {
441
443
  status: 502,
442
444
  body: "xAI response has no supported terminal status".into(),
445
+ retry_after: None,
443
446
  });
444
447
  }
445
448
  };
@@ -72,7 +72,7 @@ impl IntoResponse for ApiError {
72
72
  "unknown_provider",
73
73
  format!("Unknown provider or model: {}", p),
74
74
  ),
75
- crate::error::ShimError::ProviderError { status, body } => {
75
+ crate::error::ShimError::ProviderError { status, body, .. } => {
76
76
  let http_status = StatusCode::from_u16(*status).unwrap_or(StatusCode::BAD_GATEWAY);
77
77
  let code = if *status == 400 {
78
78
  "invalid_request"
@@ -50,14 +50,30 @@ fn structured_block(
50
50
  Some(b)
51
51
  }
52
52
 
53
+ /// The readable text of a Responses `reasoning` item. OpenAI's hosted models
54
+ /// put it in `summary[]`; a server returning the model's own reasoning puts it
55
+ /// in `content[]` as `reasoning_text` and leaves the summary empty. The summary
56
+ /// is preferred when both exist; the content is the fallback, or the item's
57
+ /// completed snapshot would replace every streamed delta with nothing.
53
58
  fn summary_text(payload: &Value) -> String {
54
- payload["summary"]
55
- .as_array()
56
- .into_iter()
57
- .flatten()
58
- .filter_map(|p| p["text"].as_str())
59
- .collect::<Vec<_>>()
60
- .join("\n")
59
+ let joined = |field: &str, kind: Option<&str>| {
60
+ payload[field]
61
+ .as_array()
62
+ .into_iter()
63
+ .flatten()
64
+ .filter(|p| kind.is_none_or(|kind| p["type"] == kind))
65
+ .filter_map(|p| p["text"].as_str())
66
+ .collect::<Vec<_>>()
67
+ .join("\n")
68
+ };
69
+ // Every summary part is read, as before; only the content fallback is
70
+ // typed, because a content part can be something other than reasoning.
71
+ let summary = joined("summary", None);
72
+ if summary.is_empty() {
73
+ joined("content", Some("reasoning_text"))
74
+ } else {
75
+ summary
76
+ }
61
77
  }
62
78
 
63
79
  pub(super) fn chat_blocks(message: &Value, origin: &ReasoningOrigin) -> Vec<ReasoningBlock> {
@@ -352,11 +368,21 @@ pub struct ReasoningAccumulator {
352
368
  }
353
369
  impl ReasoningAccumulator {
354
370
  pub fn push(&mut self, message: &Value) {
355
- for fragment in message["reasoning"].as_array().into_iter().flatten() {
356
- let key = fragment["item_id"]
357
- .as_str()
358
- .map(|s| format!("item:{s}"))
359
- .unwrap_or_else(|| format!("index:{}", fragment.get("index").unwrap_or(&json!(0))));
371
+ for (position, fragment) in message["reasoning"]
372
+ .as_array()
373
+ .into_iter()
374
+ .flatten()
375
+ .enumerate()
376
+ {
377
+ // A fragment carrying neither key is a whole block already: a buffered
378
+ // answer's blocks lose their stream index when they are assembled. Its
379
+ // position in this message keys it, or every unkeyed block would merge
380
+ // into one and an encrypted block could swallow a readable one.
381
+ let key = match (fragment["item_id"].as_str(), fragment.get("index")) {
382
+ (Some(id), _) => format!("item:{id}"),
383
+ (None, Some(index)) => format!("index:{index}"),
384
+ (None, None) => format!("index:{position}"),
385
+ };
360
386
  if !self.parts.contains_key(&key) {
361
387
  self.order.push(key.clone());
362
388
  }
@@ -571,6 +571,7 @@ pub(crate) fn enforce_stateless(body: &mut Value) -> crate::error::Result<()> {
571
571
  .ok_or_else(|| crate::error::ShimError::ProviderError {
572
572
  status: 400,
573
573
  body: "include must be an array".into(),
574
+ retry_after: None,
574
575
  })?;
575
576
  if !include.iter().any(|v| v == "reasoning.encrypted_content") {
576
577
  include.push(json!("reasoning.encrypted_content"));