saia-python 0.8.0__tar.gz → 0.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {saia_python-0.8.0/saia_python.egg-info → saia_python-0.9.0}/PKG-INFO +51 -2
- {saia_python-0.8.0 → saia_python-0.9.0}/README.md +46 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/pyproject.toml +14 -7
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/__init__.py +30 -1
- saia_python-0.9.0/saia_python/_async_http.py +127 -0
- saia_python-0.9.0/saia_python/_async_streaming.py +157 -0
- saia_python-0.9.0/saia_python/_payloads.py +81 -0
- saia_python-0.9.0/saia_python/aio.py +464 -0
- saia_python-0.9.0/saia_python/exceptions.py +124 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/rate_limits.py +36 -1
- {saia_python-0.8.0 → saia_python-0.9.0/saia_python.egg-info}/PKG-INFO +51 -2
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python.egg-info/SOURCES.txt +12 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python.egg-info/requires.txt +4 -1
- saia_python-0.9.0/tests/test_async_arcana.py +84 -0
- saia_python-0.9.0/tests/test_async_chat.py +59 -0
- saia_python-0.9.0/tests/test_async_client.py +64 -0
- saia_python-0.9.0/tests/test_async_httpx_integration.py +139 -0
- saia_python-0.9.0/tests/test_async_streaming.py +142 -0
- saia_python-0.9.0/tests/test_async_transport.py +174 -0
- saia_python-0.9.0/tests/test_payloads.py +62 -0
- saia_python-0.9.0/tests/test_rate_limit_message.py +31 -0
- saia_python-0.8.0/saia_python/exceptions.py +0 -68
- {saia_python-0.8.0 → saia_python-0.9.0}/LICENSE +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/_http.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/_streaming.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/_util.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/arcana.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/arcana_references.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/auth.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/chat.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/client.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/documents.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/models.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/openai_compat.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/py.typed +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/responses.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/tokenizer.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python/voice.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python.egg-info/dependency_links.txt +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/saia_python.egg-info/top_level.txt +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/setup.cfg +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_arcana.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_arcana_references.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_auth.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_chat.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_client.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_documents.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_exceptions.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_health_check.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_models.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_openai_compat.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_rate_limits.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_responses.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_setup_from_directory.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_streaming.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_tokenizer.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_transport_policy.py +0 -0
- {saia_python-0.8.0 → saia_python-0.9.0}/tests/test_voice.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: saia-python
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.9.0
|
|
4
4
|
Summary: Python wrapper for the GWDG SAIA platform REST API
|
|
5
5
|
Author: Friedrich Schwarz
|
|
6
6
|
License-Expression: AGPL-3.0-only
|
|
@@ -22,6 +22,7 @@ Classifier: Programming Language :: Python :: 3.12
|
|
|
22
22
|
Classifier: Programming Language :: Python :: 3.13
|
|
23
23
|
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
24
|
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
25
|
+
Classifier: Framework :: AsyncIO
|
|
25
26
|
Classifier: Typing :: Typed
|
|
26
27
|
Requires-Python: >=3.10
|
|
27
28
|
Description-Content-Type: text/markdown
|
|
@@ -31,6 +32,8 @@ Requires-Dist: tqdm>=4.60
|
|
|
31
32
|
Requires-Dist: tomlkit>=0.12
|
|
32
33
|
Provides-Extra: openai
|
|
33
34
|
Requires-Dist: openai>=1.0; extra == "openai"
|
|
35
|
+
Provides-Extra: async
|
|
36
|
+
Requires-Dist: httpx>=0.27; extra == "async"
|
|
34
37
|
Provides-Extra: tokenizer
|
|
35
38
|
Requires-Dist: transformers>=4.40; extra == "tokenizer"
|
|
36
39
|
Requires-Dist: huggingface-hub>=0.20; extra == "tokenizer"
|
|
@@ -39,7 +42,7 @@ Requires-Dist: sentencepiece>=0.1.99; extra == "tokenizer"
|
|
|
39
42
|
Provides-Extra: test
|
|
40
43
|
Requires-Dist: pytest>=7.0; extra == "test"
|
|
41
44
|
Requires-Dist: pytest-cov>=4.0; extra == "test"
|
|
42
|
-
Requires-Dist: saia-python[openai]; extra == "test"
|
|
45
|
+
Requires-Dist: saia-python[async,openai]; extra == "test"
|
|
43
46
|
Provides-Extra: docs
|
|
44
47
|
Requires-Dist: sphinx>=7.0; extra == "docs"
|
|
45
48
|
Requires-Dist: pydata-sphinx-theme>=0.15; extra == "docs"
|
|
@@ -113,6 +116,52 @@ list_model_ids()
|
|
|
113
116
|
chat_completion(model="meta-llama-3.1-8b-instruct", messages=[...])
|
|
114
117
|
```
|
|
115
118
|
|
|
119
|
+
### Async
|
|
120
|
+
|
|
121
|
+
For concurrent workloads (e.g. an ASGI service), install the `[async]` extra and
|
|
122
|
+
use `AsyncSAIAClient` — the `httpx.AsyncClient` twin of the data plane, carrying
|
|
123
|
+
the **same** `RetryPolicy` and rate-limit handling as the sync client (not the
|
|
124
|
+
`openai_async` shim, which bypasses them):
|
|
125
|
+
|
|
126
|
+
```bash
|
|
127
|
+
pip install saia-python[async]
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
import asyncio
|
|
132
|
+
from saia_python.aio import AsyncSAIAClient
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
async def main():
|
|
136
|
+
async with AsyncSAIAClient() as client:
|
|
137
|
+
# Non-streaming RAG chat
|
|
138
|
+
answer = await client.arcana.chat(
|
|
139
|
+
model="openai-gpt-oss-120b",
|
|
140
|
+
messages=[{"role": "user", "content": "Summarise the DLBCL first line."}],
|
|
141
|
+
arcana_id="owner/kb",
|
|
142
|
+
)
|
|
143
|
+
print(answer["choices"][0]["message"]["content"])
|
|
144
|
+
|
|
145
|
+
# Streaming plain chat — retry=False fails fast with an informative 429
|
|
146
|
+
stream = await client.chat.completions(
|
|
147
|
+
model="meta-llama-3.1-8b-instruct",
|
|
148
|
+
messages=[{"role": "user", "content": "Hello!"}],
|
|
149
|
+
stream=True,
|
|
150
|
+
retry=False,
|
|
151
|
+
)
|
|
152
|
+
async for chunk in stream:
|
|
153
|
+
...
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
asyncio.run(main())
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
Async covers the data plane (chat, ARCANA RAG chat, streaming) plus the
|
|
160
|
+
read-only control-plane calls (`models`, arcana `version`/`heartbeat`/`list`/
|
|
161
|
+
`get`, `health_check`). File upload/index/sync, voice, and document conversion
|
|
162
|
+
remain synchronous on `SAIAClient` — see
|
|
163
|
+
[ADR-0007](docs/adr/0007-native-async-transport.md).
|
|
164
|
+
|
|
116
165
|
## Supported Services
|
|
117
166
|
|
|
118
167
|
| Service | Description | GWDG Docs |
|
|
@@ -56,6 +56,52 @@ list_model_ids()
|
|
|
56
56
|
chat_completion(model="meta-llama-3.1-8b-instruct", messages=[...])
|
|
57
57
|
```
|
|
58
58
|
|
|
59
|
+
### Async
|
|
60
|
+
|
|
61
|
+
For concurrent workloads (e.g. an ASGI service), install the `[async]` extra and
|
|
62
|
+
use `AsyncSAIAClient` — the `httpx.AsyncClient` twin of the data plane, carrying
|
|
63
|
+
the **same** `RetryPolicy` and rate-limit handling as the sync client (not the
|
|
64
|
+
`openai_async` shim, which bypasses them):
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
pip install saia-python[async]
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
import asyncio
|
|
72
|
+
from saia_python.aio import AsyncSAIAClient
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
async def main():
|
|
76
|
+
async with AsyncSAIAClient() as client:
|
|
77
|
+
# Non-streaming RAG chat
|
|
78
|
+
answer = await client.arcana.chat(
|
|
79
|
+
model="openai-gpt-oss-120b",
|
|
80
|
+
messages=[{"role": "user", "content": "Summarise the DLBCL first line."}],
|
|
81
|
+
arcana_id="owner/kb",
|
|
82
|
+
)
|
|
83
|
+
print(answer["choices"][0]["message"]["content"])
|
|
84
|
+
|
|
85
|
+
# Streaming plain chat — retry=False fails fast with an informative 429
|
|
86
|
+
stream = await client.chat.completions(
|
|
87
|
+
model="meta-llama-3.1-8b-instruct",
|
|
88
|
+
messages=[{"role": "user", "content": "Hello!"}],
|
|
89
|
+
stream=True,
|
|
90
|
+
retry=False,
|
|
91
|
+
)
|
|
92
|
+
async for chunk in stream:
|
|
93
|
+
...
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
asyncio.run(main())
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
Async covers the data plane (chat, ARCANA RAG chat, streaming) plus the
|
|
100
|
+
read-only control-plane calls (`models`, arcana `version`/`heartbeat`/`list`/
|
|
101
|
+
`get`, `health_check`). File upload/index/sync, voice, and document conversion
|
|
102
|
+
remain synchronous on `SAIAClient` — see
|
|
103
|
+
[ADR-0007](docs/adr/0007-native-async-transport.md).
|
|
104
|
+
|
|
59
105
|
## Supported Services
|
|
60
106
|
|
|
61
107
|
| Service | Description | GWDG Docs |
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "saia-python"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.9.0"
|
|
8
8
|
description = "Python wrapper for the GWDG SAIA platform REST API"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -37,6 +37,7 @@ classifiers = [
|
|
|
37
37
|
"Programming Language :: Python :: 3.13",
|
|
38
38
|
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
39
39
|
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
40
|
+
"Framework :: AsyncIO",
|
|
40
41
|
"Typing :: Typed",
|
|
41
42
|
]
|
|
42
43
|
dependencies = [
|
|
@@ -62,6 +63,11 @@ saia_python = ["py.typed"]
|
|
|
62
63
|
openai = [
|
|
63
64
|
"openai>=1.0",
|
|
64
65
|
]
|
|
66
|
+
async = [
|
|
67
|
+
# httpx.AsyncClient backs the native async transport (saia_python.aio).
|
|
68
|
+
# requests stays the sync core; httpx is pulled only when you want async.
|
|
69
|
+
"httpx>=0.27",
|
|
70
|
+
]
|
|
65
71
|
tokenizer = [
|
|
66
72
|
# AutoTokenizer + chat-template (apply_chat_template) engine
|
|
67
73
|
"transformers>=4.40",
|
|
@@ -75,7 +81,7 @@ tokenizer = [
|
|
|
75
81
|
test = [
|
|
76
82
|
"pytest>=7.0",
|
|
77
83
|
"pytest-cov>=4.0",
|
|
78
|
-
"saia-python[openai]",
|
|
84
|
+
"saia-python[openai,async]",
|
|
79
85
|
]
|
|
80
86
|
docs = [
|
|
81
87
|
"sphinx>=7.0",
|
|
@@ -119,9 +125,10 @@ module = "saia_python.auth"
|
|
|
119
125
|
disable_error_code = ["index", "operator", "union-attr", "call-arg"]
|
|
120
126
|
|
|
121
127
|
[[tool.mypy.overrides]]
|
|
122
|
-
# `ModelsService.list` / `ArcanaService.list`
|
|
123
|
-
# `list[...]` annotations inside
|
|
124
|
-
#
|
|
125
|
-
# annotations are qualified (e.g.
|
|
126
|
-
|
|
128
|
+
# `ModelsService.list` / `ArcanaService.list` (and their async twins in
|
|
129
|
+
# `saia_python.aio`) shadow the builtin `list`, so `list[...]` annotations inside
|
|
130
|
+
# these classes resolve to the method rather than the type. Renaming would break
|
|
131
|
+
# the public API; scoped out until the annotations are qualified (e.g.
|
|
132
|
+
# `builtins.list`).
|
|
133
|
+
module = ["saia_python.models", "saia_python.arcana", "saia_python.aio"]
|
|
127
134
|
disable_error_code = ["valid-type"]
|
|
@@ -18,6 +18,12 @@ import concurrent.futures
|
|
|
18
18
|
from importlib.metadata import PackageNotFoundError, version
|
|
19
19
|
|
|
20
20
|
from ._http import RetryPolicy
|
|
21
|
+
from ._payloads import (
|
|
22
|
+
INFERENCE_SERVICE,
|
|
23
|
+
apply_arcana_fields,
|
|
24
|
+
arcana_chat_headers,
|
|
25
|
+
build_chat_body,
|
|
26
|
+
)
|
|
21
27
|
from ._streaming import SSEStream
|
|
22
28
|
from .arcana_references import (
|
|
23
29
|
ArcanaReference,
|
|
@@ -40,7 +46,7 @@ from .client import SAIAClient
|
|
|
40
46
|
from .documents import ConversionImage, ConversionResult
|
|
41
47
|
from .exceptions import APIError, AuthenticationError, RateLimitError, SAIAError
|
|
42
48
|
from .openai_compat import create_openai_client
|
|
43
|
-
from .rate_limits import RateLimitInfo, parse_rate_limits
|
|
49
|
+
from .rate_limits import RateLimitInfo, format_rate_limit_error, parse_rate_limits
|
|
44
50
|
from .responses import text_of
|
|
45
51
|
from .tokenizer import (
|
|
46
52
|
DEFAULT_TOKENIZER_DIR,
|
|
@@ -95,6 +101,12 @@ __all__ = [
|
|
|
95
101
|
# Rate limits
|
|
96
102
|
"RateLimitInfo",
|
|
97
103
|
"parse_rate_limits",
|
|
104
|
+
"format_rate_limit_error",
|
|
105
|
+
# Request builders (pure — shared by sync, async, and external gateways)
|
|
106
|
+
"build_chat_body",
|
|
107
|
+
"apply_arcana_fields",
|
|
108
|
+
"arcana_chat_headers",
|
|
109
|
+
"INFERENCE_SERVICE",
|
|
98
110
|
# Response helpers
|
|
99
111
|
"text_of",
|
|
100
112
|
"SSEStream",
|
|
@@ -142,6 +154,23 @@ __all__ = [
|
|
|
142
154
|
"ConversionImage",
|
|
143
155
|
]
|
|
144
156
|
|
|
157
|
+
# Async API (the ``[async]`` extra: ``pip install saia-python[async]``). Exposed
|
|
158
|
+
# lazily so importing ``saia_python`` never pulls ``httpx`` for sync-only users;
|
|
159
|
+
# ``from saia_python import AsyncSAIAClient`` still works when httpx is present.
|
|
160
|
+
# Kept OUT of ``__all__`` so ``from saia_python import *`` stays httpx-free.
|
|
161
|
+
_ASYNC_EXPORTS = frozenset(
|
|
162
|
+
{"AsyncSAIAClient", "AsyncChatService", "AsyncArcanaService", "AsyncModelsService"}
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def __getattr__(name: str):
|
|
167
|
+
"""Lazily resolve the async client from :mod:`saia_python.aio` on access."""
|
|
168
|
+
if name in _ASYNC_EXPORTS:
|
|
169
|
+
from . import aio
|
|
170
|
+
|
|
171
|
+
return getattr(aio, name)
|
|
172
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
173
|
+
|
|
145
174
|
|
|
146
175
|
def _make_client(api_key: str | None = None, base_url: str | None = None) -> SAIAClient:
|
|
147
176
|
kwargs: dict = {}
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
"""Async HTTP plumbing — the ``httpx.AsyncClient`` twin of :mod:`saia_python._http`.
|
|
2
|
+
|
|
3
|
+
Mirrors :func:`~saia_python._http.execute` and
|
|
4
|
+
:func:`~saia_python._http.post_chat_completion` over ``httpx.AsyncClient``,
|
|
5
|
+
**reusing the pure retry brains** (:class:`~saia_python._http.RetryPolicy`,
|
|
6
|
+
``_plan``, ``_jitter``) and :func:`~saia_python.rate_limits.parse_rate_limits`
|
|
7
|
+
unchanged — so the sync and async paths can never drift on rate-limit handling.
|
|
8
|
+
Only the socket-touching parts (``await client.request(...)`` /
|
|
9
|
+
``await resp.aclose()`` / ``await asyncio.sleep(...)``) are re-implemented.
|
|
10
|
+
|
|
11
|
+
This module never imports ``httpx`` at runtime — it only *calls methods* on a
|
|
12
|
+
client object the caller supplies, so it stays import-safe without the
|
|
13
|
+
``[async]`` extra and is trivially testable with a fake client. Construct the
|
|
14
|
+
client via :class:`saia_python.aio.AsyncSAIAClient` (or pass your own
|
|
15
|
+
``httpx.AsyncClient``).
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import asyncio
|
|
21
|
+
import logging
|
|
22
|
+
from collections.abc import Awaitable, Callable
|
|
23
|
+
from typing import TYPE_CHECKING, Any
|
|
24
|
+
|
|
25
|
+
from ._http import RetryPolicy, _jitter, _plan
|
|
26
|
+
from .exceptions import raise_for_status
|
|
27
|
+
from .rate_limits import parse_rate_limits
|
|
28
|
+
|
|
29
|
+
if TYPE_CHECKING:
|
|
30
|
+
import httpx
|
|
31
|
+
|
|
32
|
+
log = logging.getLogger(__name__)
|
|
33
|
+
|
|
34
|
+
#: An awaitable ``sleep(seconds)`` — ``asyncio.sleep`` in production; tests pass
|
|
35
|
+
#: a recorder so they never actually block.
|
|
36
|
+
AsyncSleep = Callable[..., Awaitable[Any]]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
async def aexecute(
|
|
40
|
+
client: httpx.AsyncClient,
|
|
41
|
+
method: str,
|
|
42
|
+
url: str,
|
|
43
|
+
*,
|
|
44
|
+
policy: RetryPolicy,
|
|
45
|
+
idempotent: bool,
|
|
46
|
+
sleep: AsyncSleep = asyncio.sleep,
|
|
47
|
+
**kwargs: Any,
|
|
48
|
+
) -> httpx.Response:
|
|
49
|
+
"""Issue a request under a transport policy and return the response.
|
|
50
|
+
|
|
51
|
+
The async analogue of :func:`saia_python._http.execute`: dispatches
|
|
52
|
+
``getattr(client, method)(url, **kwargs)`` (``method`` is the lowercase verb
|
|
53
|
+
— ``"post"`` / ``"get"`` — matching the sync seam) and, on HTTP 429 that the
|
|
54
|
+
``policy`` permits, waits per :func:`~saia_python._http._plan` and retries.
|
|
55
|
+
|
|
56
|
+
Like the sync version it returns the **raw response** unchanged on success
|
|
57
|
+
*or* on give-up, so the caller's
|
|
58
|
+
:func:`~saia_python.exceptions.raise_for_status` still raises
|
|
59
|
+
:class:`~saia_python.RateLimitError` when retry is off, the budget is spent,
|
|
60
|
+
or the window must not be waited on. Only status + headers are inspected —
|
|
61
|
+
never the body — so a give-up never consumes the response. Streaming is a
|
|
62
|
+
separate seam (:class:`saia_python.aio.AsyncSSEStream`), because ``httpx``
|
|
63
|
+
exposes a streamed body only inside ``client.stream(...)``.
|
|
64
|
+
"""
|
|
65
|
+
attempt = 0
|
|
66
|
+
while True:
|
|
67
|
+
resp = await getattr(client, method)(url, **kwargs)
|
|
68
|
+
if resp.status_code != 429 or not policy.applies(idempotent):
|
|
69
|
+
return resp
|
|
70
|
+
wait = _plan(parse_rate_limits(resp.headers), policy, attempt)
|
|
71
|
+
if wait is None:
|
|
72
|
+
return resp
|
|
73
|
+
await resp.aclose()
|
|
74
|
+
attempt += 1
|
|
75
|
+
wait += _jitter(policy)
|
|
76
|
+
log.info("SAIA rate limit (429) — waiting %.1fs before retry %d", wait, attempt)
|
|
77
|
+
await sleep(wait)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
async def apost_chat_completion(
|
|
81
|
+
client: httpx.AsyncClient,
|
|
82
|
+
url: str,
|
|
83
|
+
body: dict,
|
|
84
|
+
*,
|
|
85
|
+
headers: dict | None = None,
|
|
86
|
+
stream: bool = False,
|
|
87
|
+
policy: RetryPolicy | None = None,
|
|
88
|
+
sleep: AsyncSleep = asyncio.sleep,
|
|
89
|
+
) -> dict | Any:
|
|
90
|
+
"""POST a chat-completion request and normalise the response (async).
|
|
91
|
+
|
|
92
|
+
The async twin of :func:`saia_python._http.post_chat_completion`. For
|
|
93
|
+
``stream=False`` it returns the response dict with an extra
|
|
94
|
+
``"_rate_limits"`` key; for ``stream=True`` it returns a connected
|
|
95
|
+
:class:`~saia_python.aio.AsyncSSEStream` (imported lazily to avoid a cycle),
|
|
96
|
+
with the initial-429 retry already applied *before* the stream is exposed —
|
|
97
|
+
never mid-stream, exactly like the sync path.
|
|
98
|
+
"""
|
|
99
|
+
policy = policy if policy is not None else RetryPolicy()
|
|
100
|
+
if stream:
|
|
101
|
+
from ._async_streaming import AsyncSSEStream
|
|
102
|
+
|
|
103
|
+
stream_body = {**body, "stream": True}
|
|
104
|
+
stream_headers = {**(headers or {}), "Accept": "text/event-stream"}
|
|
105
|
+
return await AsyncSSEStream.open(
|
|
106
|
+
client,
|
|
107
|
+
url,
|
|
108
|
+
json=stream_body,
|
|
109
|
+
headers=stream_headers,
|
|
110
|
+
policy=policy,
|
|
111
|
+
sleep=sleep,
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
resp = await aexecute(
|
|
115
|
+
client,
|
|
116
|
+
"post",
|
|
117
|
+
url,
|
|
118
|
+
policy=policy,
|
|
119
|
+
idempotent=True,
|
|
120
|
+
sleep=sleep,
|
|
121
|
+
json=body,
|
|
122
|
+
headers=headers,
|
|
123
|
+
)
|
|
124
|
+
raise_for_status(resp)
|
|
125
|
+
result = resp.json()
|
|
126
|
+
result["_rate_limits"] = parse_rate_limits(resp.headers).to_dict()
|
|
127
|
+
return result
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
"""Async SSE streaming — the ``httpx.AsyncClient`` twin of :mod:`saia_python._streaming`.
|
|
2
|
+
|
|
3
|
+
``httpx`` only exposes a streamed body inside an ``async with client.stream(...)``
|
|
4
|
+
context, so unlike the sync :class:`~saia_python._streaming.SSEStream` (which
|
|
5
|
+
wraps an already-open ``requests`` response) this class **owns** that context
|
|
6
|
+
manager: :meth:`AsyncSSEStream.open` enters it — retrying an initial 429 *before*
|
|
7
|
+
any body is exposed, exactly like the sync path — and :meth:`aclose` / the
|
|
8
|
+
``async with`` protocol exits it.
|
|
9
|
+
|
|
10
|
+
Like :mod:`saia_python._async_http`, this module never imports ``httpx`` at
|
|
11
|
+
runtime; it only calls methods on the client / response the caller supplies,
|
|
12
|
+
staying import-safe without the ``[async]`` extra and trivially fakeable in tests.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import asyncio
|
|
18
|
+
import json
|
|
19
|
+
import logging
|
|
20
|
+
from collections.abc import AsyncIterator
|
|
21
|
+
from typing import TYPE_CHECKING, Any
|
|
22
|
+
|
|
23
|
+
from ._async_http import AsyncSleep
|
|
24
|
+
from ._http import RetryPolicy, _jitter, _plan
|
|
25
|
+
from .exceptions import raise_for_status
|
|
26
|
+
from .rate_limits import parse_rate_limits
|
|
27
|
+
|
|
28
|
+
if TYPE_CHECKING:
|
|
29
|
+
import httpx
|
|
30
|
+
|
|
31
|
+
log = logging.getLogger(__name__)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class AsyncSSEStream:
|
|
35
|
+
"""Async iterable over the SSE chunks of an ``httpx`` streaming response.
|
|
36
|
+
|
|
37
|
+
The async twin of :class:`~saia_python._streaming.SSEStream`. Two
|
|
38
|
+
consumption modes — pick **one**, because a streamed body can only be read
|
|
39
|
+
once:
|
|
40
|
+
|
|
41
|
+
* ``async for chunk in stream`` — decoded ``dict`` chunks (the high-level
|
|
42
|
+
API). Surfaces a typed error (:class:`~saia_python.RateLimitError` with an
|
|
43
|
+
informative message on 429, etc.) *before* the first chunk if the upstream
|
|
44
|
+
status is an error.
|
|
45
|
+
* ``async for line in stream.aiter_lines()`` — the raw decoded ``str`` SSE
|
|
46
|
+
lines. Does **not** raise, so a gateway can frame upstream errors itself
|
|
47
|
+
(this is what the AVOR adapter uses to keep its verbatim ``[DONE]`` /
|
|
48
|
+
non-``data:`` line passthrough).
|
|
49
|
+
|
|
50
|
+
Either way, use it as an async context manager (``async with stream:``) or
|
|
51
|
+
call :meth:`aclose` so the upstream connection is released.
|
|
52
|
+
|
|
53
|
+
Attributes:
|
|
54
|
+
status_code: The final upstream status code (after any retry).
|
|
55
|
+
rate_limits: A JSON-serializable dict of the response's rate-limit
|
|
56
|
+
headers (available immediately — headers arrive before the body).
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
def __init__(self, cm: Any, response: httpx.Response):
|
|
60
|
+
self._cm: Any = cm
|
|
61
|
+
self._response = response
|
|
62
|
+
self.status_code: int = response.status_code
|
|
63
|
+
self.rate_limits: dict = parse_rate_limits(response.headers).to_dict()
|
|
64
|
+
|
|
65
|
+
@classmethod
|
|
66
|
+
async def open(
|
|
67
|
+
cls,
|
|
68
|
+
client: httpx.AsyncClient,
|
|
69
|
+
url: str,
|
|
70
|
+
*,
|
|
71
|
+
json: dict,
|
|
72
|
+
headers: dict | None = None,
|
|
73
|
+
policy: RetryPolicy | None = None,
|
|
74
|
+
sleep: AsyncSleep = asyncio.sleep,
|
|
75
|
+
) -> AsyncSSEStream:
|
|
76
|
+
"""Open the stream, retrying an initial 429 before exposing the body.
|
|
77
|
+
|
|
78
|
+
Mirrors :func:`saia_python._http.execute`'s retry loop, but over
|
|
79
|
+
``client.stream(...)``: it enters the context, and on a retryable 429
|
|
80
|
+
exits it (releasing the socket) and re-issues after the planned wait.
|
|
81
|
+
Only status + headers are inspected, so the streamed body is never
|
|
82
|
+
consumed by the retry decision. Returns a connected stream whose
|
|
83
|
+
``status_code`` reflects the final attempt (which may still be an error
|
|
84
|
+
the caller inspects).
|
|
85
|
+
"""
|
|
86
|
+
policy = policy if policy is not None else RetryPolicy()
|
|
87
|
+
attempt = 0
|
|
88
|
+
while True:
|
|
89
|
+
cm = client.stream("POST", url, json=json, headers=headers)
|
|
90
|
+
response = await cm.__aenter__()
|
|
91
|
+
if response.status_code != 429 or not policy.applies(True):
|
|
92
|
+
return cls(cm, response)
|
|
93
|
+
wait = _plan(parse_rate_limits(response.headers), policy, attempt)
|
|
94
|
+
if wait is None:
|
|
95
|
+
return cls(cm, response)
|
|
96
|
+
await cm.__aexit__(None, None, None)
|
|
97
|
+
attempt += 1
|
|
98
|
+
wait += _jitter(policy)
|
|
99
|
+
log.info(
|
|
100
|
+
"SAIA rate limit (429) — waiting %.1fs before retry %d", wait, attempt
|
|
101
|
+
)
|
|
102
|
+
await sleep(wait)
|
|
103
|
+
|
|
104
|
+
async def aiter_lines(self) -> AsyncIterator[str]:
|
|
105
|
+
"""Yield the raw decoded SSE lines (no dict parsing, no raise).
|
|
106
|
+
|
|
107
|
+
The low-level surface: the caller sees every line verbatim (``data:``
|
|
108
|
+
payloads, ``data: [DONE]``, blank event terminators) and decides how to
|
|
109
|
+
frame errors and what to forward. Releases the connection when the
|
|
110
|
+
consumer stops, whether it finished, broke early, or was cancelled.
|
|
111
|
+
"""
|
|
112
|
+
try:
|
|
113
|
+
async for raw in self._response.aiter_lines():
|
|
114
|
+
yield raw if isinstance(raw, str) else raw.decode("utf-8")
|
|
115
|
+
finally:
|
|
116
|
+
await self.aclose()
|
|
117
|
+
|
|
118
|
+
async def __aiter__(self) -> AsyncIterator[dict]:
|
|
119
|
+
"""Yield decoded ``dict`` chunks (high-level; raises on an error status).
|
|
120
|
+
|
|
121
|
+
Copies the sync :func:`~saia_python._streaming.iter_sse` decision logic
|
|
122
|
+
verbatim: skip non-``data:`` lines, stop on ``[DONE]``, ``json.loads``
|
|
123
|
+
each payload, silently skip an unparseable one.
|
|
124
|
+
"""
|
|
125
|
+
if self.status_code >= 400:
|
|
126
|
+
await self._response.aread()
|
|
127
|
+
raise_for_status(self._response)
|
|
128
|
+
try:
|
|
129
|
+
async for raw in self._response.aiter_lines():
|
|
130
|
+
line = raw if isinstance(raw, str) else raw.decode("utf-8")
|
|
131
|
+
if not line or not line.startswith("data:"):
|
|
132
|
+
continue
|
|
133
|
+
payload = line[len("data:") :].strip()
|
|
134
|
+
if payload == "[DONE]":
|
|
135
|
+
return
|
|
136
|
+
try:
|
|
137
|
+
yield json.loads(payload)
|
|
138
|
+
except json.JSONDecodeError:
|
|
139
|
+
continue
|
|
140
|
+
finally:
|
|
141
|
+
await self.aclose()
|
|
142
|
+
|
|
143
|
+
async def aread(self) -> bytes:
|
|
144
|
+
"""Read and return the full upstream body (used to frame an error)."""
|
|
145
|
+
return await self._response.aread()
|
|
146
|
+
|
|
147
|
+
async def aclose(self) -> None:
|
|
148
|
+
"""Release the upstream connection. Idempotent."""
|
|
149
|
+
if self._cm is not None:
|
|
150
|
+
cm, self._cm = self._cm, None
|
|
151
|
+
await cm.__aexit__(None, None, None)
|
|
152
|
+
|
|
153
|
+
async def __aenter__(self) -> AsyncSSEStream:
|
|
154
|
+
return self
|
|
155
|
+
|
|
156
|
+
async def __aexit__(self, *exc: object) -> None:
|
|
157
|
+
await self.aclose()
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Pure request-body / header builders shared across transports.
|
|
2
|
+
|
|
3
|
+
Transport-free — no ``Session``, no ``httpx.AsyncClient``, no I/O — so these
|
|
4
|
+
helpers are reused **verbatim** by the sync services (:mod:`saia_python.chat`,
|
|
5
|
+
:mod:`saia_python.arcana`), the async services (:mod:`saia_python.aio`), and
|
|
6
|
+
external gateways that assemble the request themselves (the AVOR adapter builds
|
|
7
|
+
its own body so it can inject a per-conversation system prompt, then reuses
|
|
8
|
+
:func:`apply_arcana_fields` to keep the ARCANA injection identical to ours).
|
|
9
|
+
|
|
10
|
+
Keeping the ARCANA injection in ONE place matters: GWDG only routes a request
|
|
11
|
+
through the retrieval pipeline when **all three** of ``enable-tools`` + the
|
|
12
|
+
``arcana.id`` body field *and* the ``inference-service`` header are present.
|
|
13
|
+
Drop any one and the request still returns 200 — but with no retrieval. That
|
|
14
|
+
invariant now lives in :func:`apply_arcana_fields` + :func:`arcana_chat_headers`
|
|
15
|
+
instead of being retyped at every call site.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
#: The GWDG gateway value that opts a chat request into the ARCANA retrieval
|
|
23
|
+
#: pipeline. Sent as the ``inference-service`` header (see
|
|
24
|
+
#: :func:`arcana_chat_headers`); without it retrieval never fires.
|
|
25
|
+
INFERENCE_SERVICE = "saia-openai-gateway"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def build_chat_body(
|
|
29
|
+
model: str,
|
|
30
|
+
messages: list[dict],
|
|
31
|
+
*,
|
|
32
|
+
temperature: float | None = None,
|
|
33
|
+
top_p: float | None = None,
|
|
34
|
+
max_tokens: int | None = None,
|
|
35
|
+
**kwargs: Any,
|
|
36
|
+
) -> dict:
|
|
37
|
+
"""Assemble an OpenAI-shaped ``/chat/completions`` request body.
|
|
38
|
+
|
|
39
|
+
The sampling knobs are omitted from the body when ``None`` (so the server's
|
|
40
|
+
own defaults apply) rather than being sent as ``null``. Extra ``kwargs`` are
|
|
41
|
+
merged verbatim, letting callers pass ``stop``, ``stream``, ``seed``, etc.
|
|
42
|
+
"""
|
|
43
|
+
body: dict = {"model": model, "messages": messages, **kwargs}
|
|
44
|
+
if temperature is not None:
|
|
45
|
+
body["temperature"] = temperature
|
|
46
|
+
if top_p is not None:
|
|
47
|
+
body["top_p"] = top_p
|
|
48
|
+
if max_tokens is not None:
|
|
49
|
+
body["max_tokens"] = max_tokens
|
|
50
|
+
return body
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def apply_arcana_fields(body: dict, arcana_id: str) -> dict:
|
|
54
|
+
"""Return ``body`` with the ARCANA RAG **body** fields injected.
|
|
55
|
+
|
|
56
|
+
Adds ``enable-tools: true`` and ``arcana: {"id": arcana_id}``. Returns a
|
|
57
|
+
**new** dict (the input is left unmutated) with the ARCANA fields applied
|
|
58
|
+
last, so they cannot be silently clobbered by an earlier body key. This is
|
|
59
|
+
the body half of the three-part retrieval invariant; pair it with
|
|
60
|
+
:func:`arcana_chat_headers` (or set the ``inference-service`` header
|
|
61
|
+
yourself) for the header half.
|
|
62
|
+
"""
|
|
63
|
+
return {**body, "enable-tools": True, "arcana": {"id": arcana_id}}
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def arcana_chat_headers(api_key: str, *, extra: dict | None = None) -> dict:
|
|
67
|
+
"""Build the header set for an ARCANA chat call.
|
|
68
|
+
|
|
69
|
+
``Bearer`` auth + ``Accept: application/json`` + the ``inference-service``
|
|
70
|
+
gateway header (the header half of the retrieval invariant). ``extra`` is
|
|
71
|
+
merged last, so a caller can add a correlation id (e.g. ``X-Request-ID``)
|
|
72
|
+
without losing the required headers.
|
|
73
|
+
"""
|
|
74
|
+
headers = {
|
|
75
|
+
"Authorization": f"Bearer {api_key}",
|
|
76
|
+
"Accept": "application/json",
|
|
77
|
+
"inference-service": INFERENCE_SERVICE,
|
|
78
|
+
}
|
|
79
|
+
if extra:
|
|
80
|
+
headers.update(extra)
|
|
81
|
+
return headers
|