routeplane 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- routeplane/__init__.py +27 -0
- routeplane/_streaming.py +87 -0
- routeplane/_types.py +36 -0
- routeplane/_version.py +1 -0
- routeplane/async_client.py +173 -0
- routeplane/client.py +189 -0
- routeplane/headers.py +131 -0
- routeplane/meta.py +75 -0
- routeplane/py.typed +0 -0
- routeplane/resources/__init__.py +34 -0
- routeplane/resources/_base.py +112 -0
- routeplane/resources/analytics.py +23 -0
- routeplane/resources/cache.py +15 -0
- routeplane/resources/feedback.py +24 -0
- routeplane/resources/finops.py +50 -0
- routeplane/resources/logs.py +19 -0
- routeplane/resources/mcp.py +92 -0
- routeplane/resources/models.py +28 -0
- routeplane/resources/prompts.py +42 -0
- routeplane/resources/providers.py +37 -0
- routeplane/resources/residency.py +23 -0
- routeplane/resources/status.py +25 -0
- routeplane-0.1.0.dist-info/METADATA +278 -0
- routeplane-0.1.0.dist-info/RECORD +26 -0
- routeplane-0.1.0.dist-info/WHEEL +4 -0
- routeplane-0.1.0.dist-info/licenses/LICENSE +201 -0
routeplane/__init__.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Official Python SDK for the Routeplane AI Gateway.
|
|
2
|
+
|
|
3
|
+
Three ways in, smallest to largest:
|
|
4
|
+
|
|
5
|
+
1. Change nothing but ``base_url`` on the stock ``openai`` client.
|
|
6
|
+
2. Splat :func:`headers` into ``extra_headers`` on any OpenAI-compatible client.
|
|
7
|
+
3. Use :class:`Routeplane` / :class:`AsyncRouteplane` for auth + defaults +
|
|
8
|
+
typed response metadata (:class:`RouteplaneMeta`), the ``*_with_meta``
|
|
9
|
+
convenience, and the non-OpenAI resource namespaces.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from ._streaming import AsyncRouteplaneStream, RouteplaneStream
|
|
13
|
+
from ._version import __version__
|
|
14
|
+
from .async_client import AsyncRouteplane
|
|
15
|
+
from .client import Routeplane
|
|
16
|
+
from .headers import headers
|
|
17
|
+
from .meta import RouteplaneMeta
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"Routeplane",
|
|
21
|
+
"AsyncRouteplane",
|
|
22
|
+
"RouteplaneStream",
|
|
23
|
+
"AsyncRouteplaneStream",
|
|
24
|
+
"headers",
|
|
25
|
+
"RouteplaneMeta",
|
|
26
|
+
"__version__",
|
|
27
|
+
]
|
routeplane/_streaming.py
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""Thin wrappers that pair an OpenAI stream with its gateway metadata.
|
|
2
|
+
|
|
3
|
+
The OpenAI SDK's ``Stream`` exposes chunks but not the response headers, and the
|
|
4
|
+
``x-routeplane-*`` headers arrive on the response *before* the first chunk. These
|
|
5
|
+
wrappers capture that header snapshot as a :class:`RouteplaneMeta` while
|
|
6
|
+
delegating iteration straight through to the underlying stream.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import AsyncIterator, Generic, Iterator, Mapping, TypeVar
|
|
12
|
+
|
|
13
|
+
from .meta import RouteplaneMeta
|
|
14
|
+
|
|
15
|
+
__all__ = ["RouteplaneStream", "AsyncRouteplaneStream"]
|
|
16
|
+
|
|
17
|
+
_T = TypeVar("_T")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class RouteplaneStream(Generic[_T]):
|
|
21
|
+
"""Iterable wrapper over a sync OpenAI ``Stream`` that also carries ``meta``.
|
|
22
|
+
|
|
23
|
+
Use it exactly like the underlying stream::
|
|
24
|
+
|
|
25
|
+
stream = client.stream_with_meta(model="gpt-4o", messages=[...])
|
|
26
|
+
print(stream.meta.provider) # available before the first chunk
|
|
27
|
+
for chunk in stream:
|
|
28
|
+
...
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
meta: RouteplaneMeta
|
|
32
|
+
|
|
33
|
+
def __init__(self, stream: Iterator[_T], headers: Mapping[str, str]) -> None:
|
|
34
|
+
self._stream = stream
|
|
35
|
+
self.meta = RouteplaneMeta.from_headers(headers)
|
|
36
|
+
|
|
37
|
+
def __iter__(self) -> Iterator[_T]:
|
|
38
|
+
return self._stream
|
|
39
|
+
|
|
40
|
+
def __next__(self) -> _T:
|
|
41
|
+
return next(self._stream)
|
|
42
|
+
|
|
43
|
+
def __enter__(self) -> "RouteplaneStream[_T]":
|
|
44
|
+
return self
|
|
45
|
+
|
|
46
|
+
def __exit__(self, *exc: object) -> None:
|
|
47
|
+
self.close()
|
|
48
|
+
|
|
49
|
+
def close(self) -> None:
|
|
50
|
+
close = getattr(self._stream, "close", None)
|
|
51
|
+
if callable(close):
|
|
52
|
+
close()
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class AsyncRouteplaneStream(Generic[_T]):
|
|
56
|
+
"""Async twin of :class:`RouteplaneStream` for ``AsyncRouteplane``.
|
|
57
|
+
|
|
58
|
+
::
|
|
59
|
+
|
|
60
|
+
stream = await client.stream_with_meta(model="gpt-4o", messages=[...])
|
|
61
|
+
print(stream.meta.provider)
|
|
62
|
+
async for chunk in stream:
|
|
63
|
+
...
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
meta: RouteplaneMeta
|
|
67
|
+
|
|
68
|
+
def __init__(self, stream: AsyncIterator[_T], headers: Mapping[str, str]) -> None:
|
|
69
|
+
self._stream = stream
|
|
70
|
+
self.meta = RouteplaneMeta.from_headers(headers)
|
|
71
|
+
|
|
72
|
+
def __aiter__(self) -> AsyncIterator[_T]:
|
|
73
|
+
return self._stream
|
|
74
|
+
|
|
75
|
+
async def __anext__(self) -> _T:
|
|
76
|
+
return await self._stream.__anext__()
|
|
77
|
+
|
|
78
|
+
async def __aenter__(self) -> "AsyncRouteplaneStream[_T]":
|
|
79
|
+
return self
|
|
80
|
+
|
|
81
|
+
async def __aexit__(self, *exc: object) -> None:
|
|
82
|
+
await self.aclose()
|
|
83
|
+
|
|
84
|
+
async def aclose(self) -> None:
|
|
85
|
+
close = getattr(self._stream, "close", None)
|
|
86
|
+
if callable(close):
|
|
87
|
+
await close()
|
routeplane/_types.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Typed models for Routeplane's non-OpenAI endpoints.
|
|
2
|
+
|
|
3
|
+
:class:`Status` (the model behind ``GET /status``) is the one endpoint that
|
|
4
|
+
parses into a dataclass. The other resource namespaces return permissive
|
|
5
|
+
``dict`` / ``list[dict]`` payloads straight from the gateway, so a gateway that
|
|
6
|
+
grows a field never breaks an older SDK.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import Any, Optional
|
|
13
|
+
|
|
14
|
+
__all__ = ["Status"]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(frozen=True)
|
|
18
|
+
class Status:
|
|
19
|
+
"""Response of ``GET /status`` — gateway liveness/build metadata.
|
|
20
|
+
|
|
21
|
+
Fields mirror the gateway's status payload but stay permissive: unknown keys
|
|
22
|
+
are preserved in :attr:`raw` so a gateway that grows a field does not break
|
|
23
|
+
older SDKs.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
status: Optional[str] = None
|
|
27
|
+
version: Optional[str] = None
|
|
28
|
+
raw: dict[str, Any] = field(default_factory=dict)
|
|
29
|
+
|
|
30
|
+
@classmethod
|
|
31
|
+
def from_dict(cls, data: dict[str, Any]) -> "Status":
|
|
32
|
+
return cls(
|
|
33
|
+
status=data.get("status"),
|
|
34
|
+
version=data.get("version"),
|
|
35
|
+
raw=dict(data),
|
|
36
|
+
)
|
routeplane/_version.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
"""The :class:`AsyncRouteplane` client — a thin subclass of ``openai.AsyncOpenAI``.
|
|
2
|
+
|
|
3
|
+
Async twin of :class:`routeplane.Routeplane`; same header injection, the same
|
|
4
|
+
resource namespaces, and async-native ``create_with_meta`` / ``stream_with_meta``.
|
|
5
|
+
|
|
6
|
+
The REST resource namespaces (``prompts``, ``finops``, …) are the same synchronous
|
|
7
|
+
``httpx``-backed objects the sync client exposes — they cover low-frequency
|
|
8
|
+
admin/analytics endpoints, so they block briefly rather than dragging a second
|
|
9
|
+
async HTTP stack into the SDK. The chat/embeddings hot path stays fully async via
|
|
10
|
+
the inherited ``openai`` client.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from typing import Any, Mapping, Optional, Tuple, cast
|
|
16
|
+
|
|
17
|
+
import httpx
|
|
18
|
+
import openai
|
|
19
|
+
from openai.types.chat import ChatCompletion, ChatCompletionChunk
|
|
20
|
+
|
|
21
|
+
from ._streaming import AsyncRouteplaneStream
|
|
22
|
+
from .client import DEFAULT_BASE_URL
|
|
23
|
+
from .headers import headers as build_headers
|
|
24
|
+
from .meta import RouteplaneMeta
|
|
25
|
+
from .resources import (
|
|
26
|
+
AnalyticsResource,
|
|
27
|
+
CacheResource,
|
|
28
|
+
FeedbackResource,
|
|
29
|
+
FinopsResource,
|
|
30
|
+
LogsResource,
|
|
31
|
+
McpResource,
|
|
32
|
+
ModelsResource,
|
|
33
|
+
PromptsResource,
|
|
34
|
+
ProvidersResource,
|
|
35
|
+
ResidencyResource,
|
|
36
|
+
StatusResource,
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
__all__ = ["AsyncRouteplane"]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class AsyncRouteplane(openai.AsyncOpenAI):
|
|
43
|
+
"""Drop-in ``openai.AsyncOpenAI`` pointed at the Routeplane gateway.
|
|
44
|
+
|
|
45
|
+
See :class:`routeplane.Routeplane` for the auth/header semantics — this is
|
|
46
|
+
the ``async``/``await`` variant.
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
_rp_defaults: dict[str, Any]
|
|
50
|
+
_rp_http: httpx.Client
|
|
51
|
+
|
|
52
|
+
prompts: PromptsResource
|
|
53
|
+
logs: LogsResource
|
|
54
|
+
finops: FinopsResource
|
|
55
|
+
cache: CacheResource
|
|
56
|
+
feedback: FeedbackResource
|
|
57
|
+
residency: ResidencyResource
|
|
58
|
+
mcp_security: McpResource
|
|
59
|
+
rp_models: ModelsResource
|
|
60
|
+
rp_providers: ProvidersResource
|
|
61
|
+
analytics: AnalyticsResource
|
|
62
|
+
status: StatusResource
|
|
63
|
+
|
|
64
|
+
def __init__(
|
|
65
|
+
self,
|
|
66
|
+
api_key: str,
|
|
67
|
+
base_url: str = DEFAULT_BASE_URL,
|
|
68
|
+
*,
|
|
69
|
+
provider: Optional[str] = None,
|
|
70
|
+
strategy: Optional[str] = None,
|
|
71
|
+
residency: Optional[str] = None,
|
|
72
|
+
use_case: Optional[str] = None,
|
|
73
|
+
timeout_ms: Optional[int] = None,
|
|
74
|
+
**kwargs: Any,
|
|
75
|
+
) -> None:
|
|
76
|
+
self._rp_defaults = {
|
|
77
|
+
"provider": provider,
|
|
78
|
+
"strategy": strategy,
|
|
79
|
+
"residency": residency,
|
|
80
|
+
"use_case": use_case,
|
|
81
|
+
"timeout_ms": timeout_ms,
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
default_hdrs = build_headers(
|
|
85
|
+
provider=provider,
|
|
86
|
+
strategy=strategy, # type: ignore[arg-type]
|
|
87
|
+
residency=residency,
|
|
88
|
+
use_case=use_case,
|
|
89
|
+
timeout_ms=timeout_ms,
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
caller_headers = dict(kwargs.pop("default_headers", None) or {})
|
|
93
|
+
merged = {
|
|
94
|
+
"x-routeplane-api-key": api_key,
|
|
95
|
+
**default_hdrs,
|
|
96
|
+
**caller_headers,
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
super().__init__(
|
|
100
|
+
api_key=api_key,
|
|
101
|
+
base_url=base_url,
|
|
102
|
+
default_headers=merged,
|
|
103
|
+
**kwargs,
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
self._rp_http = httpx.Client()
|
|
107
|
+
self._install_resources(api_key=api_key, base_url=base_url, routing=default_hdrs)
|
|
108
|
+
|
|
109
|
+
def _install_resources(
|
|
110
|
+
self, *, api_key: str, base_url: str, routing: Mapping[str, str]
|
|
111
|
+
) -> None:
|
|
112
|
+
common: dict[str, Any] = {
|
|
113
|
+
"api_key": api_key,
|
|
114
|
+
"base_url": base_url,
|
|
115
|
+
"http_client": self._rp_http,
|
|
116
|
+
"default_headers": routing,
|
|
117
|
+
}
|
|
118
|
+
self.prompts = PromptsResource(**common)
|
|
119
|
+
self.logs = LogsResource(**common)
|
|
120
|
+
self.finops = FinopsResource(**common)
|
|
121
|
+
self.cache = CacheResource(**common)
|
|
122
|
+
self.feedback = FeedbackResource(**common)
|
|
123
|
+
self.residency = ResidencyResource(**common)
|
|
124
|
+
self.mcp_security = McpResource(**common)
|
|
125
|
+
self.rp_models = ModelsResource(**common)
|
|
126
|
+
self.rp_providers = ProvidersResource(**common)
|
|
127
|
+
self.analytics = AnalyticsResource(**common)
|
|
128
|
+
self.status = StatusResource(**common)
|
|
129
|
+
|
|
130
|
+
@staticmethod
|
|
131
|
+
def meta_from_headers(headers: Mapping[str, str]) -> RouteplaneMeta:
|
|
132
|
+
"""Parse ``x-routeplane-*`` response headers into a :class:`RouteplaneMeta`.
|
|
133
|
+
|
|
134
|
+
Pair with the OpenAI SDK's async ``with_raw_response``::
|
|
135
|
+
|
|
136
|
+
raw = await client.chat.completions.with_raw_response.create(...)
|
|
137
|
+
meta = client.meta_from_headers(raw.headers)
|
|
138
|
+
completion = raw.parse()
|
|
139
|
+
"""
|
|
140
|
+
return RouteplaneMeta.from_headers(headers)
|
|
141
|
+
|
|
142
|
+
async def create_with_meta(self, **kwargs: Any) -> Tuple[ChatCompletion, RouteplaneMeta]:
|
|
143
|
+
"""Chat completion that also returns the gateway :class:`RouteplaneMeta`.
|
|
144
|
+
|
|
145
|
+
::
|
|
146
|
+
|
|
147
|
+
completion, meta = await client.create_with_meta(model="gpt-4o", messages=[...])
|
|
148
|
+
"""
|
|
149
|
+
raw = await self.chat.completions.with_raw_response.create(**kwargs)
|
|
150
|
+
completion = cast(ChatCompletion, raw.parse())
|
|
151
|
+
return completion, RouteplaneMeta.from_headers(raw.headers)
|
|
152
|
+
|
|
153
|
+
async def stream_with_meta(self, **kwargs: Any) -> AsyncRouteplaneStream[ChatCompletionChunk]:
|
|
154
|
+
"""Streaming chat completion that also exposes ``meta`` on the stream.
|
|
155
|
+
|
|
156
|
+
::
|
|
157
|
+
|
|
158
|
+
stream = await client.stream_with_meta(model="gpt-4o", messages=[...])
|
|
159
|
+
print(stream.meta.provider)
|
|
160
|
+
async for chunk in stream:
|
|
161
|
+
...
|
|
162
|
+
"""
|
|
163
|
+
kwargs["stream"] = True
|
|
164
|
+
raw = await self.chat.completions.with_raw_response.create(**kwargs)
|
|
165
|
+
stream: Any = raw.parse()
|
|
166
|
+
return AsyncRouteplaneStream(stream, raw.headers)
|
|
167
|
+
|
|
168
|
+
async def close(self) -> None:
|
|
169
|
+
"""Close the OpenAI transport *and* the shared Routeplane httpx client."""
|
|
170
|
+
try:
|
|
171
|
+
self._rp_http.close()
|
|
172
|
+
finally:
|
|
173
|
+
await super().close()
|
routeplane/client.py
ADDED
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
"""The :class:`Routeplane` client — a thin subclass of ``openai.OpenAI``.
|
|
2
|
+
|
|
3
|
+
Everything the OpenAI SDK can do works unchanged (``client.chat.completions``,
|
|
4
|
+
``client.embeddings``, streaming, retries). The subclass adds three things:
|
|
5
|
+
|
|
6
|
+
1. gateway auth + default ``x-routeplane-*`` routing headers, so callers don't
|
|
7
|
+
repeat them on every request;
|
|
8
|
+
2. ``create_with_meta`` / ``stream_with_meta`` convenience that returns the
|
|
9
|
+
parsed completion *and* the :class:`RouteplaneMeta` gateway metadata;
|
|
10
|
+
3. the non-OpenAI resource namespaces (prompts, logs, finops, cache, feedback,
|
|
11
|
+
residency, mcp, models, providers, analytics, status).
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from typing import Any, Mapping, Optional, Tuple, cast
|
|
17
|
+
|
|
18
|
+
import httpx
|
|
19
|
+
import openai
|
|
20
|
+
from openai.types.chat import ChatCompletion, ChatCompletionChunk
|
|
21
|
+
|
|
22
|
+
from ._streaming import RouteplaneStream
|
|
23
|
+
from .headers import headers as build_headers
|
|
24
|
+
from .meta import RouteplaneMeta
|
|
25
|
+
from .resources import (
|
|
26
|
+
AnalyticsResource,
|
|
27
|
+
CacheResource,
|
|
28
|
+
FeedbackResource,
|
|
29
|
+
FinopsResource,
|
|
30
|
+
LogsResource,
|
|
31
|
+
McpResource,
|
|
32
|
+
ModelsResource,
|
|
33
|
+
PromptsResource,
|
|
34
|
+
ProvidersResource,
|
|
35
|
+
ResidencyResource,
|
|
36
|
+
StatusResource,
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
__all__ = ["Routeplane"]
|
|
40
|
+
|
|
41
|
+
DEFAULT_BASE_URL = "https://api.routeplane.ai/v1"
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class Routeplane(openai.OpenAI):
|
|
45
|
+
"""Drop-in ``openai.OpenAI`` pointed at the Routeplane gateway.
|
|
46
|
+
|
|
47
|
+
Auth is sent as the native ``x-routeplane-api-key`` header. The OpenAI SDK
|
|
48
|
+
additionally sets ``Authorization: Bearer <api_key>`` from ``api_key``; the
|
|
49
|
+
gateway accepts both and the native header takes precedence, so passing the
|
|
50
|
+
same value to both is intentional and harmless.
|
|
51
|
+
|
|
52
|
+
Per-request routing headers still win over the client defaults set here — use
|
|
53
|
+
:func:`routeplane.headers` with ``extra_headers=`` on any individual call.
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
_rp_defaults: dict[str, Any]
|
|
57
|
+
_rp_http: httpx.Client
|
|
58
|
+
|
|
59
|
+
# Routeplane-only resource namespaces (the OpenAI-shaped surfaces stay on the
|
|
60
|
+
# inherited attributes; these carry the ``rp_``/``_security`` prefixes only
|
|
61
|
+
# where the plain name is already taken by the OpenAI SDK).
|
|
62
|
+
prompts: PromptsResource
|
|
63
|
+
logs: LogsResource
|
|
64
|
+
finops: FinopsResource
|
|
65
|
+
cache: CacheResource
|
|
66
|
+
feedback: FeedbackResource
|
|
67
|
+
residency: ResidencyResource
|
|
68
|
+
mcp_security: McpResource
|
|
69
|
+
rp_models: ModelsResource
|
|
70
|
+
rp_providers: ProvidersResource
|
|
71
|
+
analytics: AnalyticsResource
|
|
72
|
+
status: StatusResource
|
|
73
|
+
|
|
74
|
+
def __init__(
|
|
75
|
+
self,
|
|
76
|
+
api_key: str,
|
|
77
|
+
base_url: str = DEFAULT_BASE_URL,
|
|
78
|
+
*,
|
|
79
|
+
provider: Optional[str] = None,
|
|
80
|
+
strategy: Optional[str] = None,
|
|
81
|
+
residency: Optional[str] = None,
|
|
82
|
+
use_case: Optional[str] = None,
|
|
83
|
+
timeout_ms: Optional[int] = None,
|
|
84
|
+
**kwargs: Any,
|
|
85
|
+
) -> None:
|
|
86
|
+
self._rp_defaults = {
|
|
87
|
+
"provider": provider,
|
|
88
|
+
"strategy": strategy,
|
|
89
|
+
"residency": residency,
|
|
90
|
+
"use_case": use_case,
|
|
91
|
+
"timeout_ms": timeout_ms,
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
default_hdrs = build_headers(
|
|
95
|
+
provider=provider,
|
|
96
|
+
strategy=strategy, # type: ignore[arg-type]
|
|
97
|
+
residency=residency,
|
|
98
|
+
use_case=use_case,
|
|
99
|
+
timeout_ms=timeout_ms,
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
# Merge, never clobber, any default_headers the caller passed through.
|
|
103
|
+
caller_headers = dict(kwargs.pop("default_headers", None) or {})
|
|
104
|
+
merged = {
|
|
105
|
+
"x-routeplane-api-key": api_key,
|
|
106
|
+
**default_hdrs,
|
|
107
|
+
**caller_headers,
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
super().__init__(
|
|
111
|
+
api_key=api_key,
|
|
112
|
+
base_url=base_url,
|
|
113
|
+
default_headers=merged,
|
|
114
|
+
**kwargs,
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
# One shared httpx client for every REST-shaped namespace; the routing
|
|
118
|
+
# defaults ride along so prompt completions honour them too.
|
|
119
|
+
self._rp_http = httpx.Client()
|
|
120
|
+
self._install_resources(api_key=api_key, base_url=base_url, routing=default_hdrs)
|
|
121
|
+
|
|
122
|
+
def _install_resources(
|
|
123
|
+
self, *, api_key: str, base_url: str, routing: Mapping[str, str]
|
|
124
|
+
) -> None:
|
|
125
|
+
common: dict[str, Any] = {
|
|
126
|
+
"api_key": api_key,
|
|
127
|
+
"base_url": base_url,
|
|
128
|
+
"http_client": self._rp_http,
|
|
129
|
+
"default_headers": routing,
|
|
130
|
+
}
|
|
131
|
+
self.prompts = PromptsResource(**common)
|
|
132
|
+
self.logs = LogsResource(**common)
|
|
133
|
+
self.finops = FinopsResource(**common)
|
|
134
|
+
self.cache = CacheResource(**common)
|
|
135
|
+
self.feedback = FeedbackResource(**common)
|
|
136
|
+
self.residency = ResidencyResource(**common)
|
|
137
|
+
self.mcp_security = McpResource(**common)
|
|
138
|
+
self.rp_models = ModelsResource(**common)
|
|
139
|
+
self.rp_providers = ProvidersResource(**common)
|
|
140
|
+
self.analytics = AnalyticsResource(**common)
|
|
141
|
+
self.status = StatusResource(**common)
|
|
142
|
+
|
|
143
|
+
@staticmethod
|
|
144
|
+
def meta_from_headers(headers: Mapping[str, str]) -> RouteplaneMeta:
|
|
145
|
+
"""Parse ``x-routeplane-*`` response headers into a :class:`RouteplaneMeta`.
|
|
146
|
+
|
|
147
|
+
Pair with the OpenAI SDK's ``with_raw_response`` to inspect what the
|
|
148
|
+
gateway did::
|
|
149
|
+
|
|
150
|
+
raw = client.chat.completions.with_raw_response.create(...)
|
|
151
|
+
meta = client.meta_from_headers(raw.headers)
|
|
152
|
+
completion = raw.parse()
|
|
153
|
+
"""
|
|
154
|
+
return RouteplaneMeta.from_headers(headers)
|
|
155
|
+
|
|
156
|
+
def create_with_meta(self, **kwargs: Any) -> Tuple[ChatCompletion, RouteplaneMeta]:
|
|
157
|
+
"""Chat completion that also returns the gateway :class:`RouteplaneMeta`.
|
|
158
|
+
|
|
159
|
+
::
|
|
160
|
+
|
|
161
|
+
completion, meta = client.create_with_meta(model="gpt-4o", messages=[...])
|
|
162
|
+
print(meta.provider, meta.cache)
|
|
163
|
+
"""
|
|
164
|
+
raw = self.chat.completions.with_raw_response.create(**kwargs)
|
|
165
|
+
completion = cast(ChatCompletion, raw.parse())
|
|
166
|
+
return completion, RouteplaneMeta.from_headers(raw.headers)
|
|
167
|
+
|
|
168
|
+
def stream_with_meta(self, **kwargs: Any) -> RouteplaneStream[ChatCompletionChunk]:
|
|
169
|
+
"""Streaming chat completion that also exposes ``meta`` on the stream.
|
|
170
|
+
|
|
171
|
+
``meta`` is populated from the response headers, which arrive before the
|
|
172
|
+
first chunk, so it can be read immediately::
|
|
173
|
+
|
|
174
|
+
stream = client.stream_with_meta(model="gpt-4o", messages=[...])
|
|
175
|
+
print(stream.meta.provider)
|
|
176
|
+
for chunk in stream:
|
|
177
|
+
...
|
|
178
|
+
"""
|
|
179
|
+
kwargs["stream"] = True
|
|
180
|
+
raw = self.chat.completions.with_raw_response.create(**kwargs)
|
|
181
|
+
stream: Any = raw.parse()
|
|
182
|
+
return RouteplaneStream(stream, raw.headers)
|
|
183
|
+
|
|
184
|
+
def close(self) -> None:
|
|
185
|
+
"""Close the OpenAI transport *and* the shared Routeplane httpx client."""
|
|
186
|
+
try:
|
|
187
|
+
self._rp_http.close()
|
|
188
|
+
finally:
|
|
189
|
+
super().close()
|
routeplane/headers.py
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""Typed builder for the ``x-routeplane-*`` request headers.
|
|
2
|
+
|
|
3
|
+
The gateway is configured entirely through request headers, so this builder is
|
|
4
|
+
the single most reusable piece of the SDK: it works with *any* OpenAI-compatible
|
|
5
|
+
client (stock ``openai``, LangChain, LlamaIndex, a raw ``httpx`` call) via that
|
|
6
|
+
client's ``extra_headers`` / ``default_headers`` escape hatch.
|
|
7
|
+
|
|
8
|
+
Only headers with non-``None`` values are emitted. Dict-valued options
|
|
9
|
+
(``config``, ``metadata``) are JSON-serialized; integers are stringified.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
from typing import Any, Literal
|
|
16
|
+
|
|
17
|
+
__all__ = ["headers", "HEADER_NAMES"]
|
|
18
|
+
|
|
19
|
+
# Maps each keyword argument to its wire header name. Keeping this as data (not
|
|
20
|
+
# branches) means the omit-None / serialize logic below stays uniform.
|
|
21
|
+
HEADER_NAMES: dict[str, str] = {
|
|
22
|
+
"provider": "x-routeplane-provider",
|
|
23
|
+
"residency": "x-routeplane-residency",
|
|
24
|
+
"strategy": "x-routeplane-strategy",
|
|
25
|
+
"config": "x-routeplane-config",
|
|
26
|
+
"timeout_ms": "x-routeplane-timeout-ms",
|
|
27
|
+
"use_case": "x-routeplane-use-case",
|
|
28
|
+
"log_level": "x-routeplane-log-level",
|
|
29
|
+
"conversation_id": "x-routeplane-conversation-id",
|
|
30
|
+
"currency": "x-routeplane-currency",
|
|
31
|
+
"metadata": "x-routeplane-metadata",
|
|
32
|
+
"pii_mode": "x-routeplane-pii-mode",
|
|
33
|
+
"output_mask": "x-routeplane-output-mask",
|
|
34
|
+
"cache_control": "x-routeplane-cache-control",
|
|
35
|
+
"idempotency_key": "x-routeplane-idempotency-key",
|
|
36
|
+
"cohort": "x-routeplane-cohort",
|
|
37
|
+
"batch": "x-routeplane-batch",
|
|
38
|
+
"trace_id": "x-routeplane-trace-id",
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def headers(
|
|
43
|
+
*,
|
|
44
|
+
provider: str | None = None,
|
|
45
|
+
residency: str | None = None,
|
|
46
|
+
strategy: Literal["priority", "weighted", "cost", "latency"] | None = None,
|
|
47
|
+
config: dict[str, Any] | None = None,
|
|
48
|
+
timeout_ms: int | None = None,
|
|
49
|
+
use_case: str | None = None,
|
|
50
|
+
log_level: Literal["metadata", "none", "full"] | None = None,
|
|
51
|
+
conversation_id: str | None = None,
|
|
52
|
+
currency: str | None = None,
|
|
53
|
+
metadata: dict[str, str] | None = None,
|
|
54
|
+
pii_mode: Literal["tokenize"] | None = None,
|
|
55
|
+
output_mask: str | None = None,
|
|
56
|
+
cache_control: Literal["no-store"] | None = None,
|
|
57
|
+
idempotency_key: str | None = None,
|
|
58
|
+
cohort: str | None = None,
|
|
59
|
+
batch: str | None = None,
|
|
60
|
+
trace_id: str | None = None,
|
|
61
|
+
) -> dict[str, str]:
|
|
62
|
+
"""Build a dict of ``x-routeplane-*`` headers for any OpenAI-compatible client.
|
|
63
|
+
|
|
64
|
+
Only includes headers whose value is non-``None``. Dict values (``config``,
|
|
65
|
+
``metadata``) are JSON-serialized; ``timeout_ms`` is stringified.
|
|
66
|
+
|
|
67
|
+
Works with ``extra_headers=headers(...)`` on the stock OpenAI SDK,
|
|
68
|
+
``default_headers=headers(...)`` on LangChain, and any client that forwards
|
|
69
|
+
extra headers to the underlying HTTP request.
|
|
70
|
+
|
|
71
|
+
Args:
|
|
72
|
+
provider: Provider or comma-separated fallback chain (e.g. ``"openai"``
|
|
73
|
+
or ``"openai,anthropic"``). Overridden by sovereign routing when the
|
|
74
|
+
request carries personal data and a residency region.
|
|
75
|
+
residency: Requested data-residency region (e.g. ``"IN"``). Only enforced
|
|
76
|
+
when the request also carries personal data.
|
|
77
|
+
strategy: Provider-ordering strategy.
|
|
78
|
+
config: Inline routing/policy config, JSON-serialized onto the wire.
|
|
79
|
+
timeout_ms: Per-request upstream timeout in milliseconds.
|
|
80
|
+
use_case: Free-form use-case label for analytics/FinOps attribution.
|
|
81
|
+
log_level: Per-request logging verbosity.
|
|
82
|
+
conversation_id: Groups requests into one logical conversation.
|
|
83
|
+
currency: Preferred currency for cost reporting (e.g. ``"INR"``).
|
|
84
|
+
metadata: Arbitrary key/value tags, JSON-serialized onto the wire.
|
|
85
|
+
pii_mode: PII handling mode.
|
|
86
|
+
output_mask: Output masking policy reference.
|
|
87
|
+
cache_control: Response-cache directive.
|
|
88
|
+
idempotency_key: Client-supplied idempotency key for safe retries.
|
|
89
|
+
cohort: Experiment/cohort label.
|
|
90
|
+
batch: Batch identifier.
|
|
91
|
+
trace_id: Client-supplied distributed-trace id (echoed back on the response).
|
|
92
|
+
|
|
93
|
+
Returns:
|
|
94
|
+
A ``dict[str, str]`` of header name to value, suitable to splat into any
|
|
95
|
+
client's extra/default headers.
|
|
96
|
+
"""
|
|
97
|
+
values: dict[str, Any] = {
|
|
98
|
+
"provider": provider,
|
|
99
|
+
"residency": residency,
|
|
100
|
+
"strategy": strategy,
|
|
101
|
+
"config": config,
|
|
102
|
+
"timeout_ms": timeout_ms,
|
|
103
|
+
"use_case": use_case,
|
|
104
|
+
"log_level": log_level,
|
|
105
|
+
"conversation_id": conversation_id,
|
|
106
|
+
"currency": currency,
|
|
107
|
+
"metadata": metadata,
|
|
108
|
+
"pii_mode": pii_mode,
|
|
109
|
+
"output_mask": output_mask,
|
|
110
|
+
"cache_control": cache_control,
|
|
111
|
+
"idempotency_key": idempotency_key,
|
|
112
|
+
"cohort": cohort,
|
|
113
|
+
"batch": batch,
|
|
114
|
+
"trace_id": trace_id,
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
out: dict[str, str] = {}
|
|
118
|
+
for key, value in values.items():
|
|
119
|
+
if value is None:
|
|
120
|
+
continue
|
|
121
|
+
header_name = HEADER_NAMES[key]
|
|
122
|
+
if isinstance(value, dict):
|
|
123
|
+
out[header_name] = json.dumps(value, separators=(",", ":"))
|
|
124
|
+
elif isinstance(value, bool):
|
|
125
|
+
# Guard before int: bool is a subclass of int in Python.
|
|
126
|
+
out[header_name] = "true" if value else "false"
|
|
127
|
+
elif isinstance(value, int):
|
|
128
|
+
out[header_name] = str(value)
|
|
129
|
+
else:
|
|
130
|
+
out[header_name] = str(value)
|
|
131
|
+
return out
|