inferhub-client 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- inferhub_client-0.1.0/.gitignore +82 -0
- inferhub_client-0.1.0/LICENSE +21 -0
- inferhub_client-0.1.0/PKG-INFO +158 -0
- inferhub_client-0.1.0/README.md +130 -0
- inferhub_client-0.1.0/pyproject.toml +58 -0
- inferhub_client-0.1.0/src/inferhub_client/__init__.py +45 -0
- inferhub_client-0.1.0/src/inferhub_client/_async_client.py +160 -0
- inferhub_client-0.1.0/src/inferhub_client/_base.py +109 -0
- inferhub_client-0.1.0/src/inferhub_client/_client.py +157 -0
- inferhub_client-0.1.0/src/inferhub_client/_exceptions.py +31 -0
- inferhub_client-0.1.0/src/inferhub_client/_models.py +334 -0
- inferhub_client-0.1.0/src/inferhub_client/_version.py +1 -0
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
# ---------------------------------------------------------------------------
|
|
2
|
+
# Internal build briefs — not part of the public repository. Keep them local.
|
|
3
|
+
#
|
|
4
|
+
# plans/CLAUDE.md is the one exception: it is the *format* rather than a brief,
|
|
5
|
+
# and the root CLAUDE.md points at it, so a fresh clone that lacked it would
|
|
6
|
+
# teach a reader the context does not exist rather than that it is local.
|
|
7
|
+
#
|
|
8
|
+
# The `/plans/*` form is required, not cosmetic — git does not descend into an
|
|
9
|
+
# excluded *directory*, so `plans/` followed by a negation matches nothing.
|
|
10
|
+
# ---------------------------------------------------------------------------
|
|
11
|
+
/plans/*
|
|
12
|
+
!/plans/CLAUDE.md
|
|
13
|
+
|
|
14
|
+
# ---------------------------------------------------------------------------
|
|
15
|
+
# .NET / Visual Studio
|
|
16
|
+
# ---------------------------------------------------------------------------
|
|
17
|
+
bin/
|
|
18
|
+
obj/
|
|
19
|
+
out/
|
|
20
|
+
[Dd]ebug/
|
|
21
|
+
[Rr]elease/
|
|
22
|
+
x64/
|
|
23
|
+
x86/
|
|
24
|
+
[Aa][Rr][Mm]/
|
|
25
|
+
[Aa][Rr][Mm]64/
|
|
26
|
+
[Bb]uild/
|
|
27
|
+
[Bb]in/
|
|
28
|
+
[Oo]bj/
|
|
29
|
+
*.user
|
|
30
|
+
*.userosscache
|
|
31
|
+
*.suo
|
|
32
|
+
*.sln.docstates
|
|
33
|
+
.vs/
|
|
34
|
+
.vscode/
|
|
35
|
+
*.swp
|
|
36
|
+
*~
|
|
37
|
+
|
|
38
|
+
# Build results / packages
|
|
39
|
+
*.dll
|
|
40
|
+
*.exe
|
|
41
|
+
*.pdb
|
|
42
|
+
*.nupkg
|
|
43
|
+
*.snupkg
|
|
44
|
+
project.lock.json
|
|
45
|
+
project.fragment.lock.json
|
|
46
|
+
artifacts/
|
|
47
|
+
nupkgs/
|
|
48
|
+
|
|
49
|
+
# Test results
|
|
50
|
+
[Tt]est[Rr]esult*/
|
|
51
|
+
*.trx
|
|
52
|
+
*.coverage
|
|
53
|
+
*.coveragexml
|
|
54
|
+
coverage*.json
|
|
55
|
+
coverage*.xml
|
|
56
|
+
coverage*.info
|
|
57
|
+
|
|
58
|
+
# Rider / JetBrains
|
|
59
|
+
.idea/
|
|
60
|
+
*.sln.iml
|
|
61
|
+
|
|
62
|
+
# OS
|
|
63
|
+
.DS_Store
|
|
64
|
+
Thumbs.db
|
|
65
|
+
|
|
66
|
+
# Local secrets / env
|
|
67
|
+
*.env
|
|
68
|
+
*.secrets
|
|
69
|
+
|
|
70
|
+
# ---------------------------------------------------------------------------
|
|
71
|
+
# Python
|
|
72
|
+
# ---------------------------------------------------------------------------
|
|
73
|
+
__pycache__/
|
|
74
|
+
*.py[cod]
|
|
75
|
+
*.egg-info/
|
|
76
|
+
.pytest_cache/
|
|
77
|
+
.mypy_cache/
|
|
78
|
+
.ruff_cache/
|
|
79
|
+
python/dist/
|
|
80
|
+
python/build/
|
|
81
|
+
python/.venv/
|
|
82
|
+
.venv/
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Dev Art Solutions
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: inferhub-client
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A small, typed Python client for InferHub — a self-hosted, Ollama-compatible inference mesh.
|
|
5
|
+
Project-URL: Homepage, https://github.com/Dev-Art-Solutions/InferHub.Clients
|
|
6
|
+
Project-URL: Repository, https://github.com/Dev-Art-Solutions/InferHub.Clients
|
|
7
|
+
Project-URL: Issues, https://github.com/Dev-Art-Solutions/InferHub.Clients/issues
|
|
8
|
+
Author: Dev Art Solutions
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Typing :: Typed
|
|
21
|
+
Requires-Python: >=3.9
|
|
22
|
+
Requires-Dist: httpx>=0.24
|
|
23
|
+
Provides-Extra: test
|
|
24
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == 'test'
|
|
25
|
+
Requires-Dist: pytest>=7; extra == 'test'
|
|
26
|
+
Requires-Dist: ruff==0.15.13; extra == 'test'
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
|
|
29
|
+
# inferhub-client — the Python client
|
|
30
|
+
|
|
31
|
+
[](https://pypi.org/project/inferhub-client/)
|
|
32
|
+
[](LICENSE)
|
|
33
|
+
|
|
34
|
+
A small, typed Python client for [InferHub](https://github.com/Dev-Art-Solutions/InferHub) — a
|
|
35
|
+
self-hosted, Ollama-compatible inference mesh. `v0.1.0` is the **core** surface: chat, generate
|
|
36
|
+
(blocking and streaming), embeddings, model listing, status and health. Retrieval (vectors, RAG,
|
|
37
|
+
ingestion, search) lands in `0.2.0`; modalities, admin and the node in `1.0.0` — see
|
|
38
|
+
`plans/roadmap-polyglot-clients.md` for the shape of the rest of the track.
|
|
39
|
+
|
|
40
|
+
**One dependency: `httpx`.** No pydantic — dataclasses do the job and every response type carries an
|
|
41
|
+
`extra` dict for fields this version does not know about yet, the same escape hatch the C# client's
|
|
42
|
+
`[JsonExtensionData]` gives it.
|
|
43
|
+
|
|
44
|
+
## Install
|
|
45
|
+
|
|
46
|
+
```
|
|
47
|
+
pip install inferhub-client
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
## Quick start
|
|
51
|
+
|
|
52
|
+
Sync:
|
|
53
|
+
|
|
54
|
+
```python
|
|
55
|
+
from inferhub_client import InferHubClient, ChatMessage, ChatRequest
|
|
56
|
+
|
|
57
|
+
with InferHubClient("http://localhost:5080/", api_key="sk-client-token-1") as client:
|
|
58
|
+
answer = client.chat(ChatRequest(
|
|
59
|
+
model="llama3",
|
|
60
|
+
messages=[ChatMessage(role="user", content="Say hi in one word.")],
|
|
61
|
+
))
|
|
62
|
+
print(answer.message.content)
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Async — the same shapes, `await`ed:
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
import asyncio
|
|
69
|
+
from inferhub_client import AsyncInferHubClient, ChatMessage, ChatRequest
|
|
70
|
+
|
|
71
|
+
async def main():
|
|
72
|
+
async with AsyncInferHubClient("http://localhost:5080/", api_key="sk-client-token-1") as client:
|
|
73
|
+
answer = await client.chat(ChatRequest(
|
|
74
|
+
model="llama3",
|
|
75
|
+
messages=[ChatMessage(role="user", content="Say hi in one word.")],
|
|
76
|
+
))
|
|
77
|
+
print(answer.message.content)
|
|
78
|
+
|
|
79
|
+
asyncio.run(main())
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
`InferHubClient` and `AsyncInferHubClient` are **two thin façades over the same rules**
|
|
83
|
+
(`_base.py`'s header building, error mapping and NDJSON parsing) rather than one client with a
|
|
84
|
+
sync-over-`asyncio.run` shim — the latter breaks the moment a sync call happens inside code that is
|
|
85
|
+
already running an event loop, which is exactly where a web framework's request handler lives.
|
|
86
|
+
|
|
87
|
+
## API surface (v0.1.0)
|
|
88
|
+
|
|
89
|
+
| Method | Endpoint |
|
|
90
|
+
|---|---|
|
|
91
|
+
| `list_models()` | `GET /api/tags` |
|
|
92
|
+
| `chat(request)` | `POST /api/chat` with `stream:false` |
|
|
93
|
+
| `chat_stream(request)` | `POST /api/chat` with `stream:true` — an iterator/async iterator of `ChatResponse` |
|
|
94
|
+
| `generate(request)` | `POST /api/generate` with `stream:false` |
|
|
95
|
+
| `generate_stream(request)` | `POST /api/generate` with `stream:true` |
|
|
96
|
+
| `embed(request)` | `POST /api/embed` (batch — a string or a list of strings) |
|
|
97
|
+
| `embed_legacy(request)` | `POST /api/embeddings` (legacy single prompt) |
|
|
98
|
+
| `get_status()` | `GET /api/status` |
|
|
99
|
+
| `ping()` | `GET /health` — `True`/`False`, never raises for a non-success status |
|
|
100
|
+
|
|
101
|
+
## Streaming
|
|
102
|
+
|
|
103
|
+
```python
|
|
104
|
+
for chunk in client.chat_stream(ChatRequest(model="llama3", messages=[...])):
|
|
105
|
+
print(chunk.message.content, end="", flush=True)
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
A terminal error chunk (`{"error": "...", "done": true}`) raises `InferHubError` out of the loop
|
|
109
|
+
instead of the iterator hanging or ending quietly with a partial answer nobody was told about.
|
|
110
|
+
|
|
111
|
+
## Errors
|
|
112
|
+
|
|
113
|
+
Every non-success response raises `InferHubError(status_code, message, response_body,
|
|
114
|
+
retry_after=...)`. `retry_after` is populated from `Retry-After` when the hub sends one — the
|
|
115
|
+
refusals that carry it are the ones worth retrying rather than only reporting.
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
from inferhub_client import InferHubError
|
|
119
|
+
|
|
120
|
+
try:
|
|
121
|
+
client.embed(EmbedRequest.from_text("nomic-embed-text", "hello"))
|
|
122
|
+
except InferHubError as e:
|
|
123
|
+
print(e.status_code, e.message, e.retry_after)
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
## `extra`: the fields this version does not know about yet
|
|
127
|
+
|
|
128
|
+
`ChatRequest`/`GenerateRequest.extra` merges straight into the request body (Ollama's `options`,
|
|
129
|
+
`format`, `keep_alive`, tool definitions — anything the hub accepts that this client has not typed);
|
|
130
|
+
every response dataclass keeps unrecognized fields in its own `.extra` dict on the way back. Typing
|
|
131
|
+
every Ollama option was considered and rejected, same as the C# client: the hub owns that schema and
|
|
132
|
+
grows it independently of this package's release cadence.
|
|
133
|
+
|
|
134
|
+
## A node as a target
|
|
135
|
+
|
|
136
|
+
A solo InferHub node serves this same Ollama-dialect surface on its own address — pointing
|
|
137
|
+
`InferHubClient`/`AsyncInferHubClient` at a node's URL instead of a coordinator's is the whole of
|
|
138
|
+
"run it locally." `probe()` and the node-only routes (`/api/version`, the `/api/collections`
|
|
139
|
+
lifecycle) land in `1.0.0`, mirroring the C# client's phase 14 (`14 D7`).
|
|
140
|
+
|
|
141
|
+
## Development
|
|
142
|
+
|
|
143
|
+
```
|
|
144
|
+
pip install -e ".[test]"
|
|
145
|
+
pytest # 31 pass, 9 skipped (cases outside v0.1.0's surface, see below)
|
|
146
|
+
ruff check src tests examples
|
|
147
|
+
ruff format --check src tests examples
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
`tests/test_conformance.py` drives the shared corpus at `../conformance/cases.json` — the same file
|
|
151
|
+
the C# client's `ConformanceCorpusTests.cs` reads. A case whose `kind` this client does not cover
|
|
152
|
+
yet (retrieval, the node, the OpenAI dialect) is skipped with a named reason rather than silently
|
|
153
|
+
omitted; four cases (the mid-stream terminal error, `424` vs `404`, both `X-InferHub-Sources`
|
|
154
|
+
shapes) pass today, unmodified, because the corpus already knew the answer.
|
|
155
|
+
|
|
156
|
+
## License
|
|
157
|
+
|
|
158
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
# inferhub-client — the Python client
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/inferhub-client/)
|
|
4
|
+
[](LICENSE)
|
|
5
|
+
|
|
6
|
+
A small, typed Python client for [InferHub](https://github.com/Dev-Art-Solutions/InferHub) — a
|
|
7
|
+
self-hosted, Ollama-compatible inference mesh. `v0.1.0` is the **core** surface: chat, generate
|
|
8
|
+
(blocking and streaming), embeddings, model listing, status and health. Retrieval (vectors, RAG,
|
|
9
|
+
ingestion, search) lands in `0.2.0`; modalities, admin and the node in `1.0.0` — see
|
|
10
|
+
`plans/roadmap-polyglot-clients.md` for the shape of the rest of the track.
|
|
11
|
+
|
|
12
|
+
**One dependency: `httpx`.** No pydantic — dataclasses do the job and every response type carries an
|
|
13
|
+
`extra` dict for fields this version does not know about yet, the same escape hatch the C# client's
|
|
14
|
+
`[JsonExtensionData]` gives it.
|
|
15
|
+
|
|
16
|
+
## Install
|
|
17
|
+
|
|
18
|
+
```
|
|
19
|
+
pip install inferhub-client
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
## Quick start
|
|
23
|
+
|
|
24
|
+
Sync:
|
|
25
|
+
|
|
26
|
+
```python
|
|
27
|
+
from inferhub_client import InferHubClient, ChatMessage, ChatRequest
|
|
28
|
+
|
|
29
|
+
with InferHubClient("http://localhost:5080/", api_key="sk-client-token-1") as client:
|
|
30
|
+
answer = client.chat(ChatRequest(
|
|
31
|
+
model="llama3",
|
|
32
|
+
messages=[ChatMessage(role="user", content="Say hi in one word.")],
|
|
33
|
+
))
|
|
34
|
+
print(answer.message.content)
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Async — the same shapes, `await`ed:
|
|
38
|
+
|
|
39
|
+
```python
|
|
40
|
+
import asyncio
|
|
41
|
+
from inferhub_client import AsyncInferHubClient, ChatMessage, ChatRequest
|
|
42
|
+
|
|
43
|
+
async def main():
|
|
44
|
+
async with AsyncInferHubClient("http://localhost:5080/", api_key="sk-client-token-1") as client:
|
|
45
|
+
answer = await client.chat(ChatRequest(
|
|
46
|
+
model="llama3",
|
|
47
|
+
messages=[ChatMessage(role="user", content="Say hi in one word.")],
|
|
48
|
+
))
|
|
49
|
+
print(answer.message.content)
|
|
50
|
+
|
|
51
|
+
asyncio.run(main())
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
`InferHubClient` and `AsyncInferHubClient` are **two thin façades over the same rules**
|
|
55
|
+
(`_base.py`'s header building, error mapping and NDJSON parsing) rather than one client with a
|
|
56
|
+
sync-over-`asyncio.run` shim — the latter breaks the moment a sync call happens inside code that is
|
|
57
|
+
already running an event loop, which is exactly where a web framework's request handler lives.
|
|
58
|
+
|
|
59
|
+
## API surface (v0.1.0)
|
|
60
|
+
|
|
61
|
+
| Method | Endpoint |
|
|
62
|
+
|---|---|
|
|
63
|
+
| `list_models()` | `GET /api/tags` |
|
|
64
|
+
| `chat(request)` | `POST /api/chat` with `stream:false` |
|
|
65
|
+
| `chat_stream(request)` | `POST /api/chat` with `stream:true` — an iterator/async iterator of `ChatResponse` |
|
|
66
|
+
| `generate(request)` | `POST /api/generate` with `stream:false` |
|
|
67
|
+
| `generate_stream(request)` | `POST /api/generate` with `stream:true` |
|
|
68
|
+
| `embed(request)` | `POST /api/embed` (batch — a string or a list of strings) |
|
|
69
|
+
| `embed_legacy(request)` | `POST /api/embeddings` (legacy single prompt) |
|
|
70
|
+
| `get_status()` | `GET /api/status` |
|
|
71
|
+
| `ping()` | `GET /health` — `True`/`False`, never raises for a non-success status |
|
|
72
|
+
|
|
73
|
+
## Streaming
|
|
74
|
+
|
|
75
|
+
```python
|
|
76
|
+
for chunk in client.chat_stream(ChatRequest(model="llama3", messages=[...])):
|
|
77
|
+
print(chunk.message.content, end="", flush=True)
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
A terminal error chunk (`{"error": "...", "done": true}`) raises `InferHubError` out of the loop
|
|
81
|
+
instead of the iterator hanging or ending quietly with a partial answer nobody was told about.
|
|
82
|
+
|
|
83
|
+
## Errors
|
|
84
|
+
|
|
85
|
+
Every non-success response raises `InferHubError(status_code, message, response_body,
|
|
86
|
+
retry_after=...)`. `retry_after` is populated from `Retry-After` when the hub sends one — the
|
|
87
|
+
refusals that carry it are the ones worth retrying rather than only reporting.
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
from inferhub_client import InferHubError
|
|
91
|
+
|
|
92
|
+
try:
|
|
93
|
+
client.embed(EmbedRequest.from_text("nomic-embed-text", "hello"))
|
|
94
|
+
except InferHubError as e:
|
|
95
|
+
print(e.status_code, e.message, e.retry_after)
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
## `extra`: the fields this version does not know about yet
|
|
99
|
+
|
|
100
|
+
`ChatRequest`/`GenerateRequest.extra` merges straight into the request body (Ollama's `options`,
|
|
101
|
+
`format`, `keep_alive`, tool definitions — anything the hub accepts that this client has not typed);
|
|
102
|
+
every response dataclass keeps unrecognized fields in its own `.extra` dict on the way back. Typing
|
|
103
|
+
every Ollama option was considered and rejected, same as the C# client: the hub owns that schema and
|
|
104
|
+
grows it independently of this package's release cadence.
|
|
105
|
+
|
|
106
|
+
## A node as a target
|
|
107
|
+
|
|
108
|
+
A solo InferHub node serves this same Ollama-dialect surface on its own address — pointing
|
|
109
|
+
`InferHubClient`/`AsyncInferHubClient` at a node's URL instead of a coordinator's is the whole of
|
|
110
|
+
"run it locally." `probe()` and the node-only routes (`/api/version`, the `/api/collections`
|
|
111
|
+
lifecycle) land in `1.0.0`, mirroring the C# client's phase 14 (`14 D7`).
|
|
112
|
+
|
|
113
|
+
## Development
|
|
114
|
+
|
|
115
|
+
```
|
|
116
|
+
pip install -e ".[test]"
|
|
117
|
+
pytest # 31 pass, 9 skipped (cases outside v0.1.0's surface, see below)
|
|
118
|
+
ruff check src tests examples
|
|
119
|
+
ruff format --check src tests examples
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
`tests/test_conformance.py` drives the shared corpus at `../conformance/cases.json` — the same file
|
|
123
|
+
the C# client's `ConformanceCorpusTests.cs` reads. A case whose `kind` this client does not cover
|
|
124
|
+
yet (retrieval, the node, the OpenAI dialect) is skipped with a named reason rather than silently
|
|
125
|
+
omitted; four cases (the mid-stream terminal error, `424` vs `404`, both `X-InferHub-Sources`
|
|
126
|
+
shapes) pass today, unmodified, because the corpus already knew the answer.
|
|
127
|
+
|
|
128
|
+
## License
|
|
129
|
+
|
|
130
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "inferhub-client"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "A small, typed Python client for InferHub — a self-hosted, Ollama-compatible inference mesh."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
authors = [{ name = "Dev Art Solutions" }]
|
|
12
|
+
requires-python = ">=3.9"
|
|
13
|
+
dependencies = ["httpx>=0.24"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 3 - Alpha",
|
|
16
|
+
"Intended Audience :: Developers",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Programming Language :: Python :: 3.9",
|
|
20
|
+
"Programming Language :: Python :: 3.10",
|
|
21
|
+
"Programming Language :: Python :: 3.11",
|
|
22
|
+
"Programming Language :: Python :: 3.12",
|
|
23
|
+
"Programming Language :: Python :: 3.13",
|
|
24
|
+
"Typing :: Typed",
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
[project.urls]
|
|
28
|
+
Homepage = "https://github.com/Dev-Art-Solutions/InferHub.Clients"
|
|
29
|
+
Repository = "https://github.com/Dev-Art-Solutions/InferHub.Clients"
|
|
30
|
+
Issues = "https://github.com/Dev-Art-Solutions/InferHub.Clients/issues"
|
|
31
|
+
|
|
32
|
+
[project.optional-dependencies]
|
|
33
|
+
# ruff pinned exactly, not >=: two different minor releases (0.15 -> 0.16) changed which lint
|
|
34
|
+
# rules are on by default with nothing in this file changed, which broke CI twice in a row on an
|
|
35
|
+
# unpinned version. A linter whose pass/fail depends on which version happened to install is not
|
|
36
|
+
# a check, so this one is deterministic instead — bump it deliberately, in its own commit, when
|
|
37
|
+
# there is a reason to.
|
|
38
|
+
test = ["pytest>=7", "pytest-asyncio>=0.23", "ruff==0.15.13"]
|
|
39
|
+
|
|
40
|
+
[tool.hatch.build.targets.wheel]
|
|
41
|
+
packages = ["src/inferhub_client"]
|
|
42
|
+
|
|
43
|
+
[tool.hatch.build.targets.sdist]
|
|
44
|
+
include = ["src", "README.md", "LICENSE"]
|
|
45
|
+
|
|
46
|
+
[tool.pytest.ini_options]
|
|
47
|
+
asyncio_mode = "auto"
|
|
48
|
+
testpaths = ["tests"]
|
|
49
|
+
|
|
50
|
+
[tool.ruff]
|
|
51
|
+
target-version = "py39"
|
|
52
|
+
|
|
53
|
+
[tool.ruff.lint]
|
|
54
|
+
# Pinned explicitly rather than left to ruff's default selection: 0.16 started enabling `I`
|
|
55
|
+
# (isort) by default where 0.15 did not, which broke CI on an unpinned `ruff>=0.6` the moment
|
|
56
|
+
# the runner picked up a newer release with nothing in this file changed. Extending the default
|
|
57
|
+
# set (E4, E7, E9, F) rather than replacing it.
|
|
58
|
+
extend-select = ["I"]
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""inferhub-client — a small, typed Python client for InferHub.
|
|
2
|
+
|
|
3
|
+
>>> from inferhub_client import InferHubClient, ChatMessage, ChatRequest
|
|
4
|
+
>>> with InferHubClient("http://localhost:5080/", api_key="sk-...") as client:
|
|
5
|
+
... answer = client.chat(ChatRequest(model="llama3", messages=[ChatMessage("user", "hi")]))
|
|
6
|
+
... print(answer.message.content)
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from ._async_client import AsyncInferHubClient
|
|
10
|
+
from ._client import InferHubClient
|
|
11
|
+
from ._exceptions import InferHubError
|
|
12
|
+
from ._models import (
|
|
13
|
+
ChatMessage,
|
|
14
|
+
ChatRequest,
|
|
15
|
+
ChatResponse,
|
|
16
|
+
EmbeddingsRequest,
|
|
17
|
+
EmbeddingsResponse,
|
|
18
|
+
EmbedRequest,
|
|
19
|
+
EmbedResponse,
|
|
20
|
+
GenerateRequest,
|
|
21
|
+
GenerateResponse,
|
|
22
|
+
ModelInfo,
|
|
23
|
+
StatusResponse,
|
|
24
|
+
TagsResponse,
|
|
25
|
+
)
|
|
26
|
+
from ._version import __version__
|
|
27
|
+
|
|
28
|
+
__all__ = [
|
|
29
|
+
"__version__",
|
|
30
|
+
"InferHubClient",
|
|
31
|
+
"AsyncInferHubClient",
|
|
32
|
+
"InferHubError",
|
|
33
|
+
"ChatMessage",
|
|
34
|
+
"ChatRequest",
|
|
35
|
+
"ChatResponse",
|
|
36
|
+
"GenerateRequest",
|
|
37
|
+
"GenerateResponse",
|
|
38
|
+
"EmbedRequest",
|
|
39
|
+
"EmbedResponse",
|
|
40
|
+
"EmbeddingsRequest",
|
|
41
|
+
"EmbeddingsResponse",
|
|
42
|
+
"ModelInfo",
|
|
43
|
+
"TagsResponse",
|
|
44
|
+
"StatusResponse",
|
|
45
|
+
]
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import AsyncIterator, Optional
|
|
4
|
+
|
|
5
|
+
import httpx
|
|
6
|
+
|
|
7
|
+
from ._base import (
|
|
8
|
+
DEFAULT_BASE_URL,
|
|
9
|
+
build_headers,
|
|
10
|
+
parse_ndjson_line,
|
|
11
|
+
raise_for_status,
|
|
12
|
+
read_served_by,
|
|
13
|
+
read_source_ids,
|
|
14
|
+
)
|
|
15
|
+
from ._exceptions import InferHubError
|
|
16
|
+
from ._models import (
|
|
17
|
+
ChatRequest,
|
|
18
|
+
ChatResponse,
|
|
19
|
+
EmbeddingsRequest,
|
|
20
|
+
EmbeddingsResponse,
|
|
21
|
+
EmbedRequest,
|
|
22
|
+
EmbedResponse,
|
|
23
|
+
GenerateRequest,
|
|
24
|
+
GenerateResponse,
|
|
25
|
+
StatusResponse,
|
|
26
|
+
TagsResponse,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class AsyncInferHubClient:
|
|
31
|
+
"""Async client for an InferHub coordinator (or a solo node — same address, same client, see
|
|
32
|
+
``python/README.md``). Covers the Ollama-dialect core surface: chat, generate (blocking and
|
|
33
|
+
streaming), embeddings, model listing, status and health.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
def __init__(
|
|
37
|
+
self,
|
|
38
|
+
base_url: str = DEFAULT_BASE_URL,
|
|
39
|
+
api_key: Optional[str] = None,
|
|
40
|
+
*,
|
|
41
|
+
timeout: float = 100.0,
|
|
42
|
+
http_client: Optional[httpx.AsyncClient] = None,
|
|
43
|
+
) -> None:
|
|
44
|
+
self._owns_client = http_client is None
|
|
45
|
+
self._http = http_client or httpx.AsyncClient(
|
|
46
|
+
base_url=base_url, headers=build_headers(api_key), timeout=timeout
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
async def __aenter__(self) -> "AsyncInferHubClient":
|
|
50
|
+
return self
|
|
51
|
+
|
|
52
|
+
async def __aexit__(self, *exc_info: object) -> None:
|
|
53
|
+
await self.aclose()
|
|
54
|
+
|
|
55
|
+
async def aclose(self) -> None:
|
|
56
|
+
if self._owns_client:
|
|
57
|
+
await self._http.aclose()
|
|
58
|
+
|
|
59
|
+
async def list_models(self) -> TagsResponse:
|
|
60
|
+
"""``GET /api/tags`` — models advertised by the mesh."""
|
|
61
|
+
response = await self._http.get("api/tags")
|
|
62
|
+
raise_for_status(response)
|
|
63
|
+
return TagsResponse.from_json(response.json())
|
|
64
|
+
|
|
65
|
+
async def chat(self, request: ChatRequest) -> ChatResponse:
|
|
66
|
+
"""Blocking chat — ``POST /api/chat`` with ``stream:false``."""
|
|
67
|
+
request.stream = False
|
|
68
|
+
response = await self._http.post("api/chat", json=request.to_json())
|
|
69
|
+
raise_for_status(response)
|
|
70
|
+
result = ChatResponse.from_json(response.json())
|
|
71
|
+
result.served_by = read_served_by(response)
|
|
72
|
+
result.source_ids = read_source_ids(response)
|
|
73
|
+
return result
|
|
74
|
+
|
|
75
|
+
async def chat_stream(self, request: ChatRequest) -> AsyncIterator[ChatResponse]:
|
|
76
|
+
"""Streaming chat — ``POST /api/chat`` with ``stream:true``. Yields one
|
|
77
|
+
:class:`ChatResponse` per NDJSON line; a terminal error chunk raises
|
|
78
|
+
:class:`InferHubError` instead of the iterator hanging or ending quietly."""
|
|
79
|
+
request.stream = True
|
|
80
|
+
async with self._http.stream(
|
|
81
|
+
"POST", "api/chat", json=request.to_json()
|
|
82
|
+
) as response:
|
|
83
|
+
raise_for_status(response)
|
|
84
|
+
served_by = read_served_by(response)
|
|
85
|
+
source_ids = read_source_ids(response)
|
|
86
|
+
async for line in response.aiter_lines():
|
|
87
|
+
chunk = parse_ndjson_line(line)
|
|
88
|
+
if chunk is None:
|
|
89
|
+
continue
|
|
90
|
+
result = ChatResponse.from_json(chunk)
|
|
91
|
+
result.served_by = served_by
|
|
92
|
+
result.source_ids = source_ids
|
|
93
|
+
yield result
|
|
94
|
+
if result.done:
|
|
95
|
+
return
|
|
96
|
+
|
|
97
|
+
async def generate(self, request: GenerateRequest) -> GenerateResponse:
|
|
98
|
+
"""Blocking generate — ``POST /api/generate`` with ``stream:false``."""
|
|
99
|
+
request.stream = False
|
|
100
|
+
response = await self._http.post("api/generate", json=request.to_json())
|
|
101
|
+
raise_for_status(response)
|
|
102
|
+
result = GenerateResponse.from_json(response.json())
|
|
103
|
+
result.served_by = read_served_by(response)
|
|
104
|
+
result.source_ids = read_source_ids(response)
|
|
105
|
+
return result
|
|
106
|
+
|
|
107
|
+
async def generate_stream(
|
|
108
|
+
self, request: GenerateRequest
|
|
109
|
+
) -> AsyncIterator[GenerateResponse]:
|
|
110
|
+
"""Streaming generate — ``POST /api/generate`` with ``stream:true``."""
|
|
111
|
+
request.stream = True
|
|
112
|
+
async with self._http.stream(
|
|
113
|
+
"POST", "api/generate", json=request.to_json()
|
|
114
|
+
) as response:
|
|
115
|
+
raise_for_status(response)
|
|
116
|
+
served_by = read_served_by(response)
|
|
117
|
+
source_ids = read_source_ids(response)
|
|
118
|
+
async for line in response.aiter_lines():
|
|
119
|
+
chunk = parse_ndjson_line(line)
|
|
120
|
+
if chunk is None:
|
|
121
|
+
continue
|
|
122
|
+
result = GenerateResponse.from_json(chunk)
|
|
123
|
+
result.served_by = served_by
|
|
124
|
+
result.source_ids = source_ids
|
|
125
|
+
yield result
|
|
126
|
+
if result.done:
|
|
127
|
+
return
|
|
128
|
+
|
|
129
|
+
async def embed(self, request: EmbedRequest) -> EmbedResponse:
|
|
130
|
+
"""``POST /api/embed`` — batch embeddings. An empty vector list on a 200 is treated as a
|
|
131
|
+
malformed response and raised, never silently returned."""
|
|
132
|
+
response = await self._http.post("api/embed", json=request.to_json())
|
|
133
|
+
raise_for_status(response)
|
|
134
|
+
result = EmbedResponse.from_json(response.json())
|
|
135
|
+
if not result.embeddings:
|
|
136
|
+
raise InferHubError(response.status_code, "embed response had no vectors")
|
|
137
|
+
return result
|
|
138
|
+
|
|
139
|
+
async def embed_legacy(self, request: EmbeddingsRequest) -> EmbeddingsResponse:
|
|
140
|
+
"""``POST /api/embeddings`` — the legacy single-input endpoint. Prefer :meth:`embed`."""
|
|
141
|
+
response = await self._http.post("api/embeddings", json=request.to_json())
|
|
142
|
+
raise_for_status(response)
|
|
143
|
+
result = EmbeddingsResponse.from_json(response.json())
|
|
144
|
+
if not result.embedding:
|
|
145
|
+
raise InferHubError(
|
|
146
|
+
response.status_code, "embeddings response had no vector"
|
|
147
|
+
)
|
|
148
|
+
return result
|
|
149
|
+
|
|
150
|
+
async def get_status(self) -> StatusResponse:
|
|
151
|
+
"""``GET /api/status`` — coordinator/fleet snapshot."""
|
|
152
|
+
response = await self._http.get("api/status")
|
|
153
|
+
raise_for_status(response)
|
|
154
|
+
return StatusResponse.from_json(response.json())
|
|
155
|
+
|
|
156
|
+
async def ping(self) -> bool:
|
|
157
|
+
"""``GET /health`` — ``True`` on 2xx, ``False`` otherwise. Never raises for a non-success
|
|
158
|
+
status; raises only on a transport error."""
|
|
159
|
+
response = await self._http.get("health")
|
|
160
|
+
return response.is_success
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""Shared, I/O-free plumbing used by both :class:`InferHubClient` (sync) and
|
|
2
|
+
:class:`AsyncInferHubClient` (async): headers, error mapping, and NDJSON chunk parsing. Kept out of
|
|
3
|
+
either client so the two stay thin façades over the same rules rather than two copies that drift —
|
|
4
|
+
see ``python/README.md`` on why this is two classes and not one client with a sync-over-async shim.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
from typing import Any, Dict, Optional
|
|
11
|
+
|
|
12
|
+
import httpx
|
|
13
|
+
|
|
14
|
+
from ._exceptions import InferHubError
|
|
15
|
+
|
|
16
|
+
DEFAULT_BASE_URL = "http://localhost:5080/"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def build_headers(api_key: Optional[str]) -> Dict[str, str]:
|
|
20
|
+
headers = {"Content-Type": "application/json"}
|
|
21
|
+
if api_key:
|
|
22
|
+
headers["Authorization"] = f"Bearer {api_key}"
|
|
23
|
+
return headers
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _extract_error_message(body: str) -> Optional[str]:
|
|
27
|
+
"""The Ollama dialect answers ``{"error": "..."}"``. A non-JSON or differently-shaped body
|
|
28
|
+
falls back to the raw text, same as the C# client's ``TryExtractErrorMessage``."""
|
|
29
|
+
|
|
30
|
+
if not body.strip():
|
|
31
|
+
return None
|
|
32
|
+
try:
|
|
33
|
+
parsed = json.loads(body)
|
|
34
|
+
except (json.JSONDecodeError, ValueError):
|
|
35
|
+
return body
|
|
36
|
+
if isinstance(parsed, dict) and isinstance(parsed.get("error"), str):
|
|
37
|
+
return parsed["error"]
|
|
38
|
+
return body
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _retry_after(response: httpx.Response) -> Optional[float]:
|
|
42
|
+
header = response.headers.get("Retry-After")
|
|
43
|
+
if not header:
|
|
44
|
+
return None
|
|
45
|
+
try:
|
|
46
|
+
return float(header)
|
|
47
|
+
except ValueError:
|
|
48
|
+
return None # An HTTP-date form exists but the hub always writes delta-seconds.
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def raise_for_status(response: httpx.Response) -> None:
|
|
52
|
+
if response.is_success:
|
|
53
|
+
return
|
|
54
|
+
|
|
55
|
+
body = response.text
|
|
56
|
+
message = _extract_error_message(body) or (
|
|
57
|
+
f"InferHub request failed with status {response.status_code}."
|
|
58
|
+
)
|
|
59
|
+
raise InferHubError(
|
|
60
|
+
response.status_code, message, body, retry_after=_retry_after(response)
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def parse_ndjson_line(line: str) -> Optional[Dict[str, Any]]:
|
|
65
|
+
"""One line of an NDJSON stream, or ``None`` for a blank line to skip. Raises
|
|
66
|
+
:class:`InferHubError` on a terminal error chunk (``{"error": ..., "done": true}``) so a caller's
|
|
67
|
+
loop stops with a clear exception instead of hanging or silently finishing early."""
|
|
68
|
+
|
|
69
|
+
if not line.strip():
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
chunk = json.loads(line)
|
|
73
|
+
error = chunk.get("error")
|
|
74
|
+
if error:
|
|
75
|
+
raise InferHubError(200, error, line)
|
|
76
|
+
return chunk
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def read_served_by(response: httpx.Response) -> Optional[str]:
|
|
80
|
+
"""Which node or ``provider:<id>`` answered — surfaced, never interpreted. This client does
|
|
81
|
+
not route, retry elsewhere or prefer on it (root ``CLAUDE.md`` rule 8); reading it here is what
|
|
82
|
+
lets a caller log or display it without this library making a decision on its behalf."""
|
|
83
|
+
|
|
84
|
+
value = response.headers.get("X-InferHub-Served-By")
|
|
85
|
+
return value.strip() or None if value else None
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def read_source_ids(response: httpx.Response) -> Optional[list]:
|
|
89
|
+
"""``X-InferHub-Sources`` arrives as a JSON array, but a real hub has also sent it
|
|
90
|
+
comma-separated — ``spec/README.md`` calls this the conformance corpus's first case, and both
|
|
91
|
+
shapes are parsed here even though ``v0.1.0`` has no way yet to opt into retrieval (phase 17):
|
|
92
|
+
the header is part of the core response contract, and a caller reading it manually should not
|
|
93
|
+
have to wait for this client to grow RAG headers first."""
|
|
94
|
+
|
|
95
|
+
raw = response.headers.get("X-InferHub-Sources")
|
|
96
|
+
if raw is None:
|
|
97
|
+
return None
|
|
98
|
+
raw = raw.strip()
|
|
99
|
+
if not raw:
|
|
100
|
+
return []
|
|
101
|
+
try:
|
|
102
|
+
parsed = json.loads(raw)
|
|
103
|
+
if isinstance(parsed, list):
|
|
104
|
+
return [
|
|
105
|
+
str(item) for item in parsed if item is not None and str(item) != ""
|
|
106
|
+
]
|
|
107
|
+
except (json.JSONDecodeError, ValueError):
|
|
108
|
+
pass
|
|
109
|
+
return [part.strip() for part in raw.split(",") if part.strip()]
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Iterator, Optional
|
|
4
|
+
|
|
5
|
+
import httpx
|
|
6
|
+
|
|
7
|
+
from ._base import (
|
|
8
|
+
DEFAULT_BASE_URL,
|
|
9
|
+
build_headers,
|
|
10
|
+
parse_ndjson_line,
|
|
11
|
+
raise_for_status,
|
|
12
|
+
read_served_by,
|
|
13
|
+
read_source_ids,
|
|
14
|
+
)
|
|
15
|
+
from ._exceptions import InferHubError
|
|
16
|
+
from ._models import (
|
|
17
|
+
ChatRequest,
|
|
18
|
+
ChatResponse,
|
|
19
|
+
EmbeddingsRequest,
|
|
20
|
+
EmbeddingsResponse,
|
|
21
|
+
EmbedRequest,
|
|
22
|
+
EmbedResponse,
|
|
23
|
+
GenerateRequest,
|
|
24
|
+
GenerateResponse,
|
|
25
|
+
StatusResponse,
|
|
26
|
+
TagsResponse,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class InferHubClient:
|
|
31
|
+
"""Sync client for an InferHub coordinator (or a solo node — same address, same client, see
|
|
32
|
+
``python/README.md``). A thin façade over the same rules :class:`AsyncInferHubClient` uses
|
|
33
|
+
(``_base.py``), for callers not already in an event loop. Covers the Ollama-dialect core
|
|
34
|
+
surface: chat, generate (blocking and streaming), embeddings, model listing, status and health.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
def __init__(
|
|
38
|
+
self,
|
|
39
|
+
base_url: str = DEFAULT_BASE_URL,
|
|
40
|
+
api_key: Optional[str] = None,
|
|
41
|
+
*,
|
|
42
|
+
timeout: float = 100.0,
|
|
43
|
+
http_client: Optional[httpx.Client] = None,
|
|
44
|
+
) -> None:
|
|
45
|
+
self._owns_client = http_client is None
|
|
46
|
+
self._http = http_client or httpx.Client(
|
|
47
|
+
base_url=base_url, headers=build_headers(api_key), timeout=timeout
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
def __enter__(self) -> "InferHubClient":
|
|
51
|
+
return self
|
|
52
|
+
|
|
53
|
+
def __exit__(self, *exc_info: object) -> None:
|
|
54
|
+
self.close()
|
|
55
|
+
|
|
56
|
+
def close(self) -> None:
|
|
57
|
+
if self._owns_client:
|
|
58
|
+
self._http.close()
|
|
59
|
+
|
|
60
|
+
def list_models(self) -> TagsResponse:
|
|
61
|
+
"""``GET /api/tags`` — models advertised by the mesh."""
|
|
62
|
+
response = self._http.get("api/tags")
|
|
63
|
+
raise_for_status(response)
|
|
64
|
+
return TagsResponse.from_json(response.json())
|
|
65
|
+
|
|
66
|
+
def chat(self, request: ChatRequest) -> ChatResponse:
|
|
67
|
+
"""Blocking chat — ``POST /api/chat`` with ``stream:false``."""
|
|
68
|
+
request.stream = False
|
|
69
|
+
response = self._http.post("api/chat", json=request.to_json())
|
|
70
|
+
raise_for_status(response)
|
|
71
|
+
result = ChatResponse.from_json(response.json())
|
|
72
|
+
result.served_by = read_served_by(response)
|
|
73
|
+
result.source_ids = read_source_ids(response)
|
|
74
|
+
return result
|
|
75
|
+
|
|
76
|
+
def chat_stream(self, request: ChatRequest) -> Iterator[ChatResponse]:
|
|
77
|
+
"""Streaming chat — ``POST /api/chat`` with ``stream:true``. Yields one
|
|
78
|
+
:class:`ChatResponse` per NDJSON line; a terminal error chunk raises
|
|
79
|
+
:class:`InferHubError` instead of the iterator hanging or ending quietly."""
|
|
80
|
+
request.stream = True
|
|
81
|
+
with self._http.stream("POST", "api/chat", json=request.to_json()) as response:
|
|
82
|
+
raise_for_status(response)
|
|
83
|
+
served_by = read_served_by(response)
|
|
84
|
+
source_ids = read_source_ids(response)
|
|
85
|
+
for line in response.iter_lines():
|
|
86
|
+
chunk = parse_ndjson_line(line)
|
|
87
|
+
if chunk is None:
|
|
88
|
+
continue
|
|
89
|
+
result = ChatResponse.from_json(chunk)
|
|
90
|
+
result.served_by = served_by
|
|
91
|
+
result.source_ids = source_ids
|
|
92
|
+
yield result
|
|
93
|
+
if result.done:
|
|
94
|
+
return
|
|
95
|
+
|
|
96
|
+
def generate(self, request: GenerateRequest) -> GenerateResponse:
|
|
97
|
+
"""Blocking generate — ``POST /api/generate`` with ``stream:false``."""
|
|
98
|
+
request.stream = False
|
|
99
|
+
response = self._http.post("api/generate", json=request.to_json())
|
|
100
|
+
raise_for_status(response)
|
|
101
|
+
result = GenerateResponse.from_json(response.json())
|
|
102
|
+
result.served_by = read_served_by(response)
|
|
103
|
+
result.source_ids = read_source_ids(response)
|
|
104
|
+
return result
|
|
105
|
+
|
|
106
|
+
def generate_stream(self, request: GenerateRequest) -> Iterator[GenerateResponse]:
|
|
107
|
+
"""Streaming generate — ``POST /api/generate`` with ``stream:true``."""
|
|
108
|
+
request.stream = True
|
|
109
|
+
with self._http.stream(
|
|
110
|
+
"POST", "api/generate", json=request.to_json()
|
|
111
|
+
) as response:
|
|
112
|
+
raise_for_status(response)
|
|
113
|
+
served_by = read_served_by(response)
|
|
114
|
+
source_ids = read_source_ids(response)
|
|
115
|
+
for line in response.iter_lines():
|
|
116
|
+
chunk = parse_ndjson_line(line)
|
|
117
|
+
if chunk is None:
|
|
118
|
+
continue
|
|
119
|
+
result = GenerateResponse.from_json(chunk)
|
|
120
|
+
result.served_by = served_by
|
|
121
|
+
result.source_ids = source_ids
|
|
122
|
+
yield result
|
|
123
|
+
if result.done:
|
|
124
|
+
return
|
|
125
|
+
|
|
126
|
+
def embed(self, request: EmbedRequest) -> EmbedResponse:
|
|
127
|
+
"""``POST /api/embed`` — batch embeddings. An empty vector list on a 200 is treated as a
|
|
128
|
+
malformed response and raised, never silently returned."""
|
|
129
|
+
response = self._http.post("api/embed", json=request.to_json())
|
|
130
|
+
raise_for_status(response)
|
|
131
|
+
result = EmbedResponse.from_json(response.json())
|
|
132
|
+
if not result.embeddings:
|
|
133
|
+
raise InferHubError(response.status_code, "embed response had no vectors")
|
|
134
|
+
return result
|
|
135
|
+
|
|
136
|
+
def embed_legacy(self, request: EmbeddingsRequest) -> EmbeddingsResponse:
|
|
137
|
+
"""``POST /api/embeddings`` — the legacy single-input endpoint. Prefer :meth:`embed`."""
|
|
138
|
+
response = self._http.post("api/embeddings", json=request.to_json())
|
|
139
|
+
raise_for_status(response)
|
|
140
|
+
result = EmbeddingsResponse.from_json(response.json())
|
|
141
|
+
if not result.embedding:
|
|
142
|
+
raise InferHubError(
|
|
143
|
+
response.status_code, "embeddings response had no vector"
|
|
144
|
+
)
|
|
145
|
+
return result
|
|
146
|
+
|
|
147
|
+
def get_status(self) -> StatusResponse:
|
|
148
|
+
"""``GET /api/status`` — coordinator/fleet snapshot."""
|
|
149
|
+
response = self._http.get("api/status")
|
|
150
|
+
raise_for_status(response)
|
|
151
|
+
return StatusResponse.from_json(response.json())
|
|
152
|
+
|
|
153
|
+
def ping(self) -> bool:
|
|
154
|
+
"""``GET /health`` — ``True`` on 2xx, ``False`` otherwise. Never raises for a non-success
|
|
155
|
+
status; raises only on a transport error."""
|
|
156
|
+
response = self._http.get("health")
|
|
157
|
+
return response.is_success
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class InferHubError(Exception):
|
|
7
|
+
"""Raised when the coordinator (or a solo node) answers a non-success HTTP status.
|
|
8
|
+
|
|
9
|
+
Carries the raw status code so a caller can distinguish 404 (model or collection missing),
|
|
10
|
+
401/403 (auth), 501 (a backend that structurally cannot serve a capability) from 503 (a
|
|
11
|
+
capability an operator disabled, temporary, and worth the ``retry_after`` seconds).
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
def __init__(
|
|
15
|
+
self,
|
|
16
|
+
status_code: int,
|
|
17
|
+
message: str,
|
|
18
|
+
response_body: str = "",
|
|
19
|
+
*,
|
|
20
|
+
retry_after: Optional[float] = None,
|
|
21
|
+
) -> None:
|
|
22
|
+
super().__init__(message)
|
|
23
|
+
self.status_code = status_code
|
|
24
|
+
self.message = message
|
|
25
|
+
self.response_body = response_body
|
|
26
|
+
self.retry_after = retry_after
|
|
27
|
+
|
|
28
|
+
def __repr__(self) -> str: # pragma: no cover - cosmetic
|
|
29
|
+
return (
|
|
30
|
+
f"InferHubError(status_code={self.status_code!r}, message={self.message!r})"
|
|
31
|
+
)
|
|
@@ -0,0 +1,334 @@
|
|
|
1
|
+
"""Typed request/response shapes for the Ollama-dialect core surface.
|
|
2
|
+
|
|
3
|
+
Dataclasses, not pydantic (root ``CLAUDE.md`` rule 2 / roadmap-polyglot-clients D5): the wire is
|
|
4
|
+
small and stable enough that a validation library is somebody else's dependency war inherited by
|
|
5
|
+
every consumer. Every response type keeps an ``extra`` dict for fields the hub sends that this
|
|
6
|
+
version does not know about yet — the Python equivalent of the C# client's ``[JsonExtensionData]``
|
|
7
|
+
bag — so a caller reaching for a brand-new field is never blocked on a new release of this package.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from typing import Any, Dict, List, Optional, Union
|
|
14
|
+
|
|
15
|
+
JsonDict = Dict[str, Any]
|
|
16
|
+
|
|
17
|
+
_KNOWN_MESSAGE_FIELDS = {"role", "content", "images", "tool_calls"}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass
|
|
21
|
+
class ChatMessage:
|
|
22
|
+
"""One message in a chat request or response."""
|
|
23
|
+
|
|
24
|
+
role: str
|
|
25
|
+
content: str = ""
|
|
26
|
+
images: Optional[List[str]] = None
|
|
27
|
+
tool_calls: Optional[List[JsonDict]] = None
|
|
28
|
+
extra: JsonDict = field(default_factory=dict)
|
|
29
|
+
|
|
30
|
+
def to_json(self) -> JsonDict:
|
|
31
|
+
body: JsonDict = {"role": self.role, "content": self.content}
|
|
32
|
+
if self.images is not None:
|
|
33
|
+
body["images"] = self.images
|
|
34
|
+
if self.tool_calls is not None:
|
|
35
|
+
body["tool_calls"] = self.tool_calls
|
|
36
|
+
body.update(self.extra)
|
|
37
|
+
return body
|
|
38
|
+
|
|
39
|
+
@classmethod
|
|
40
|
+
def from_json(cls, data: JsonDict) -> "ChatMessage":
|
|
41
|
+
return cls(
|
|
42
|
+
role=data.get("role", ""),
|
|
43
|
+
content=data.get("content", ""),
|
|
44
|
+
images=data.get("images"),
|
|
45
|
+
tool_calls=data.get("tool_calls"),
|
|
46
|
+
extra={k: v for k, v in data.items() if k not in _KNOWN_MESSAGE_FIELDS},
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass
|
|
51
|
+
class ChatRequest:
|
|
52
|
+
"""``POST /api/chat``. ``extra`` merges straight into the top-level body, untouched — the
|
|
53
|
+
passthrough that keeps every Ollama option (``options``, ``format``, ``keep_alive``, tool
|
|
54
|
+
definitions) reachable without this client typing each one (roadmap-polyglot-clients,
|
|
55
|
+
"typed request builders for every option" is a rejected non-goal, same as the C# client)."""
|
|
56
|
+
|
|
57
|
+
model: str
|
|
58
|
+
messages: List[ChatMessage] = field(default_factory=list)
|
|
59
|
+
stream: bool = False
|
|
60
|
+
options: Optional[JsonDict] = None
|
|
61
|
+
format: Optional[Union[str, JsonDict]] = None
|
|
62
|
+
keep_alive: Optional[str] = None
|
|
63
|
+
extra: JsonDict = field(default_factory=dict)
|
|
64
|
+
|
|
65
|
+
def to_json(self) -> JsonDict:
|
|
66
|
+
body: JsonDict = {
|
|
67
|
+
"model": self.model,
|
|
68
|
+
"messages": [m.to_json() for m in self.messages],
|
|
69
|
+
"stream": self.stream,
|
|
70
|
+
}
|
|
71
|
+
if self.options is not None:
|
|
72
|
+
body["options"] = self.options
|
|
73
|
+
if self.format is not None:
|
|
74
|
+
body["format"] = self.format
|
|
75
|
+
if self.keep_alive is not None:
|
|
76
|
+
body["keep_alive"] = self.keep_alive
|
|
77
|
+
body.update(self.extra)
|
|
78
|
+
return body
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
_KNOWN_CHAT_RESPONSE_FIELDS = {
|
|
82
|
+
"model",
|
|
83
|
+
"created_at",
|
|
84
|
+
"message",
|
|
85
|
+
"done",
|
|
86
|
+
"done_reason",
|
|
87
|
+
"total_duration",
|
|
88
|
+
"load_duration",
|
|
89
|
+
"prompt_eval_count",
|
|
90
|
+
"prompt_eval_duration",
|
|
91
|
+
"eval_count",
|
|
92
|
+
"eval_duration",
|
|
93
|
+
"error",
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
@dataclass
|
|
98
|
+
class ChatResponse:
|
|
99
|
+
"""A blocking answer, or one NDJSON chunk of a streamed one."""
|
|
100
|
+
|
|
101
|
+
model: str = ""
|
|
102
|
+
created_at: Optional[str] = None
|
|
103
|
+
message: Optional[ChatMessage] = None
|
|
104
|
+
done: Optional[bool] = None
|
|
105
|
+
done_reason: Optional[str] = None
|
|
106
|
+
total_duration: Optional[int] = None
|
|
107
|
+
load_duration: Optional[int] = None
|
|
108
|
+
prompt_eval_count: Optional[int] = None
|
|
109
|
+
prompt_eval_duration: Optional[int] = None
|
|
110
|
+
eval_count: Optional[int] = None
|
|
111
|
+
eval_duration: Optional[int] = None
|
|
112
|
+
error: Optional[str] = None
|
|
113
|
+
extra: JsonDict = field(default_factory=dict)
|
|
114
|
+
# Set from response headers, not the body — never part of round-tripping the JSON.
|
|
115
|
+
served_by: Optional[str] = None
|
|
116
|
+
source_ids: Optional[List[str]] = None
|
|
117
|
+
|
|
118
|
+
@classmethod
|
|
119
|
+
def from_json(cls, data: JsonDict) -> "ChatResponse":
|
|
120
|
+
message = data.get("message")
|
|
121
|
+
return cls(
|
|
122
|
+
model=data.get("model", ""),
|
|
123
|
+
created_at=data.get("created_at"),
|
|
124
|
+
message=ChatMessage.from_json(message) if message is not None else None,
|
|
125
|
+
done=data.get("done"),
|
|
126
|
+
done_reason=data.get("done_reason"),
|
|
127
|
+
total_duration=data.get("total_duration"),
|
|
128
|
+
load_duration=data.get("load_duration"),
|
|
129
|
+
prompt_eval_count=data.get("prompt_eval_count"),
|
|
130
|
+
prompt_eval_duration=data.get("prompt_eval_duration"),
|
|
131
|
+
eval_count=data.get("eval_count"),
|
|
132
|
+
eval_duration=data.get("eval_duration"),
|
|
133
|
+
error=data.get("error"),
|
|
134
|
+
extra={
|
|
135
|
+
k: v for k, v in data.items() if k not in _KNOWN_CHAT_RESPONSE_FIELDS
|
|
136
|
+
},
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
@dataclass
|
|
141
|
+
class GenerateRequest:
|
|
142
|
+
"""``POST /api/generate``. Same extension-bag contract as :class:`ChatRequest`."""
|
|
143
|
+
|
|
144
|
+
model: str
|
|
145
|
+
prompt: str = ""
|
|
146
|
+
stream: bool = False
|
|
147
|
+
options: Optional[JsonDict] = None
|
|
148
|
+
format: Optional[Union[str, JsonDict]] = None
|
|
149
|
+
keep_alive: Optional[str] = None
|
|
150
|
+
extra: JsonDict = field(default_factory=dict)
|
|
151
|
+
|
|
152
|
+
def to_json(self) -> JsonDict:
|
|
153
|
+
body: JsonDict = {
|
|
154
|
+
"model": self.model,
|
|
155
|
+
"prompt": self.prompt,
|
|
156
|
+
"stream": self.stream,
|
|
157
|
+
}
|
|
158
|
+
if self.options is not None:
|
|
159
|
+
body["options"] = self.options
|
|
160
|
+
if self.format is not None:
|
|
161
|
+
body["format"] = self.format
|
|
162
|
+
if self.keep_alive is not None:
|
|
163
|
+
body["keep_alive"] = self.keep_alive
|
|
164
|
+
body.update(self.extra)
|
|
165
|
+
return body
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
_KNOWN_GENERATE_RESPONSE_FIELDS = {
|
|
169
|
+
"model",
|
|
170
|
+
"created_at",
|
|
171
|
+
"response",
|
|
172
|
+
"done",
|
|
173
|
+
"done_reason",
|
|
174
|
+
"context",
|
|
175
|
+
"total_duration",
|
|
176
|
+
"load_duration",
|
|
177
|
+
"prompt_eval_count",
|
|
178
|
+
"prompt_eval_duration",
|
|
179
|
+
"eval_count",
|
|
180
|
+
"eval_duration",
|
|
181
|
+
"error",
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
@dataclass
|
|
186
|
+
class GenerateResponse:
|
|
187
|
+
model: str = ""
|
|
188
|
+
created_at: Optional[str] = None
|
|
189
|
+
response: str = ""
|
|
190
|
+
done: Optional[bool] = None
|
|
191
|
+
done_reason: Optional[str] = None
|
|
192
|
+
context: Optional[List[int]] = None
|
|
193
|
+
total_duration: Optional[int] = None
|
|
194
|
+
load_duration: Optional[int] = None
|
|
195
|
+
prompt_eval_count: Optional[int] = None
|
|
196
|
+
prompt_eval_duration: Optional[int] = None
|
|
197
|
+
eval_count: Optional[int] = None
|
|
198
|
+
eval_duration: Optional[int] = None
|
|
199
|
+
error: Optional[str] = None
|
|
200
|
+
extra: JsonDict = field(default_factory=dict)
|
|
201
|
+
served_by: Optional[str] = None
|
|
202
|
+
source_ids: Optional[List[str]] = None
|
|
203
|
+
|
|
204
|
+
@classmethod
|
|
205
|
+
def from_json(cls, data: JsonDict) -> "GenerateResponse":
|
|
206
|
+
return cls(
|
|
207
|
+
model=data.get("model", ""),
|
|
208
|
+
created_at=data.get("created_at"),
|
|
209
|
+
response=data.get("response", ""),
|
|
210
|
+
done=data.get("done"),
|
|
211
|
+
done_reason=data.get("done_reason"),
|
|
212
|
+
context=data.get("context"),
|
|
213
|
+
total_duration=data.get("total_duration"),
|
|
214
|
+
load_duration=data.get("load_duration"),
|
|
215
|
+
prompt_eval_count=data.get("prompt_eval_count"),
|
|
216
|
+
prompt_eval_duration=data.get("prompt_eval_duration"),
|
|
217
|
+
eval_count=data.get("eval_count"),
|
|
218
|
+
eval_duration=data.get("eval_duration"),
|
|
219
|
+
error=data.get("error"),
|
|
220
|
+
extra={
|
|
221
|
+
k: v
|
|
222
|
+
for k, v in data.items()
|
|
223
|
+
if k not in _KNOWN_GENERATE_RESPONSE_FIELDS
|
|
224
|
+
},
|
|
225
|
+
)
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
@dataclass
|
|
229
|
+
class EmbedRequest:
|
|
230
|
+
"""``POST /api/embed`` — the modern batch endpoint. ``input`` is a single string or a list."""
|
|
231
|
+
|
|
232
|
+
model: str
|
|
233
|
+
input: Union[str, List[str]]
|
|
234
|
+
|
|
235
|
+
def to_json(self) -> JsonDict:
|
|
236
|
+
return {"model": self.model, "input": self.input}
|
|
237
|
+
|
|
238
|
+
@classmethod
|
|
239
|
+
def from_text(cls, model: str, text: str) -> "EmbedRequest":
|
|
240
|
+
return cls(model=model, input=text)
|
|
241
|
+
|
|
242
|
+
@classmethod
|
|
243
|
+
def from_texts(cls, model: str, texts: List[str]) -> "EmbedRequest":
|
|
244
|
+
return cls(model=model, input=list(texts))
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
@dataclass
|
|
248
|
+
class EmbedResponse:
|
|
249
|
+
model: str = ""
|
|
250
|
+
embeddings: List[List[float]] = field(default_factory=list)
|
|
251
|
+
|
|
252
|
+
@classmethod
|
|
253
|
+
def from_json(cls, data: JsonDict) -> "EmbedResponse":
|
|
254
|
+
return cls(model=data.get("model", ""), embeddings=data.get("embeddings") or [])
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
@dataclass
|
|
258
|
+
class EmbeddingsRequest:
|
|
259
|
+
"""``POST /api/embeddings`` — the legacy single-input endpoint. Prefer :class:`EmbedRequest`."""
|
|
260
|
+
|
|
261
|
+
model: str
|
|
262
|
+
prompt: str
|
|
263
|
+
|
|
264
|
+
def to_json(self) -> JsonDict:
|
|
265
|
+
return {"model": self.model, "prompt": self.prompt}
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
@dataclass
|
|
269
|
+
class EmbeddingsResponse:
|
|
270
|
+
embedding: List[float] = field(default_factory=list)
|
|
271
|
+
|
|
272
|
+
@classmethod
|
|
273
|
+
def from_json(cls, data: JsonDict) -> "EmbeddingsResponse":
|
|
274
|
+
return cls(embedding=data.get("embedding") or [])
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
@dataclass
|
|
278
|
+
class ModelInfo:
|
|
279
|
+
name: str
|
|
280
|
+
digest: Optional[str] = None
|
|
281
|
+
size: Optional[int] = None
|
|
282
|
+
|
|
283
|
+
@classmethod
|
|
284
|
+
def from_json(cls, data: JsonDict) -> "ModelInfo":
|
|
285
|
+
return cls(
|
|
286
|
+
name=data.get("name", ""), digest=data.get("digest"), size=data.get("size")
|
|
287
|
+
)
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
@dataclass
|
|
291
|
+
class TagsResponse:
|
|
292
|
+
models: List[ModelInfo] = field(default_factory=list)
|
|
293
|
+
|
|
294
|
+
@classmethod
|
|
295
|
+
def from_json(cls, data: JsonDict) -> "TagsResponse":
|
|
296
|
+
return cls(models=[ModelInfo.from_json(m) for m in data.get("models") or []])
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
_KNOWN_STATUS_FIELDS = {
|
|
300
|
+
"coordinatorVersion",
|
|
301
|
+
"nowUtc",
|
|
302
|
+
"uptimeSeconds",
|
|
303
|
+
"nodes",
|
|
304
|
+
"models",
|
|
305
|
+
"metrics",
|
|
306
|
+
"vector",
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
@dataclass
|
|
311
|
+
class StatusResponse:
|
|
312
|
+
"""``GET /api/status`` on a coordinator. See :mod:`inferhub_client.probe` for the solo-node shape
|
|
313
|
+
(added in a later phase) and how a caller tells the two apart."""
|
|
314
|
+
|
|
315
|
+
coordinator_version: Optional[str] = None
|
|
316
|
+
now_utc: Optional[str] = None
|
|
317
|
+
uptime_seconds: Optional[float] = None
|
|
318
|
+
nodes: Optional[List[JsonDict]] = None
|
|
319
|
+
models: Optional[List[ModelInfo]] = None
|
|
320
|
+
extra: JsonDict = field(default_factory=dict)
|
|
321
|
+
|
|
322
|
+
@classmethod
|
|
323
|
+
def from_json(cls, data: JsonDict) -> "StatusResponse":
|
|
324
|
+
models = data.get("models")
|
|
325
|
+
return cls(
|
|
326
|
+
coordinator_version=data.get("coordinatorVersion"),
|
|
327
|
+
now_utc=data.get("nowUtc"),
|
|
328
|
+
uptime_seconds=data.get("uptimeSeconds"),
|
|
329
|
+
nodes=data.get("nodes"),
|
|
330
|
+
models=[ModelInfo.from_json(m) for m in models]
|
|
331
|
+
if models is not None
|
|
332
|
+
else None,
|
|
333
|
+
extra={k: v for k, v in data.items() if k not in _KNOWN_STATUS_FIELDS},
|
|
334
|
+
)
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|