inferhub-client 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,82 @@
1
+ # ---------------------------------------------------------------------------
2
+ # Internal build briefs — not part of the public repository. Keep them local.
3
+ #
4
+ # plans/CLAUDE.md is the one exception: it is the *format* rather than a brief,
5
+ # and the root CLAUDE.md points at it, so a fresh clone that lacked it would
6
+ # teach a reader the context does not exist rather than that it is local.
7
+ #
8
+ # The `/plans/*` form is required, not cosmetic — git does not descend into an
9
+ # excluded *directory*, so `plans/` followed by a negation matches nothing.
10
+ # ---------------------------------------------------------------------------
11
+ /plans/*
12
+ !/plans/CLAUDE.md
13
+
14
+ # ---------------------------------------------------------------------------
15
+ # .NET / Visual Studio
16
+ # ---------------------------------------------------------------------------
17
+ bin/
18
+ obj/
19
+ out/
20
+ [Dd]ebug/
21
+ [Rr]elease/
22
+ x64/
23
+ x86/
24
+ [Aa][Rr][Mm]/
25
+ [Aa][Rr][Mm]64/
26
+ [Bb]uild/
27
+ [Bb]in/
28
+ [Oo]bj/
29
+ *.user
30
+ *.userosscache
31
+ *.suo
32
+ *.sln.docstates
33
+ .vs/
34
+ .vscode/
35
+ *.swp
36
+ *~
37
+
38
+ # Build results / packages
39
+ *.dll
40
+ *.exe
41
+ *.pdb
42
+ *.nupkg
43
+ *.snupkg
44
+ project.lock.json
45
+ project.fragment.lock.json
46
+ artifacts/
47
+ nupkgs/
48
+
49
+ # Test results
50
+ [Tt]est[Rr]esult*/
51
+ *.trx
52
+ *.coverage
53
+ *.coveragexml
54
+ coverage*.json
55
+ coverage*.xml
56
+ coverage*.info
57
+
58
+ # Rider / JetBrains
59
+ .idea/
60
+ *.sln.iml
61
+
62
+ # OS
63
+ .DS_Store
64
+ Thumbs.db
65
+
66
+ # Local secrets / env
67
+ *.env
68
+ *.secrets
69
+
70
+ # ---------------------------------------------------------------------------
71
+ # Python
72
+ # ---------------------------------------------------------------------------
73
+ __pycache__/
74
+ *.py[cod]
75
+ *.egg-info/
76
+ .pytest_cache/
77
+ .mypy_cache/
78
+ .ruff_cache/
79
+ python/dist/
80
+ python/build/
81
+ python/.venv/
82
+ .venv/
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Dev Art Solutions
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,158 @@
1
+ Metadata-Version: 2.5
2
+ Name: inferhub-client
3
+ Version: 0.1.0
4
+ Summary: A small, typed Python client for InferHub — a self-hosted, Ollama-compatible inference mesh.
5
+ Project-URL: Homepage, https://github.com/Dev-Art-Solutions/InferHub.Clients
6
+ Project-URL: Repository, https://github.com/Dev-Art-Solutions/InferHub.Clients
7
+ Project-URL: Issues, https://github.com/Dev-Art-Solutions/InferHub.Clients/issues
8
+ Author: Dev Art Solutions
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Typing :: Typed
21
+ Requires-Python: >=3.9
22
+ Requires-Dist: httpx>=0.24
23
+ Provides-Extra: test
24
+ Requires-Dist: pytest-asyncio>=0.23; extra == 'test'
25
+ Requires-Dist: pytest>=7; extra == 'test'
26
+ Requires-Dist: ruff==0.15.13; extra == 'test'
27
+ Description-Content-Type: text/markdown
28
+
29
+ # inferhub-client — the Python client
30
+
31
+ [![PyPI](https://img.shields.io/pypi/v/inferhub-client.svg)](https://pypi.org/project/inferhub-client/)
32
+ [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
33
+
34
+ A small, typed Python client for [InferHub](https://github.com/Dev-Art-Solutions/InferHub) — a
35
+ self-hosted, Ollama-compatible inference mesh. `v0.1.0` is the **core** surface: chat, generate
36
+ (blocking and streaming), embeddings, model listing, status and health. Retrieval (vectors, RAG,
37
+ ingestion, search) lands in `0.2.0`; modalities, admin and the node in `1.0.0` — see
38
+ `plans/roadmap-polyglot-clients.md` for the shape of the rest of the track.
39
+
40
+ **One dependency: `httpx`.** No pydantic — dataclasses do the job and every response type carries an
41
+ `extra` dict for fields this version does not know about yet, the same escape hatch the C# client's
42
+ `[JsonExtensionData]` gives it.
43
+
44
+ ## Install
45
+
46
+ ```
47
+ pip install inferhub-client
48
+ ```
49
+
50
+ ## Quick start
51
+
52
+ Sync:
53
+
54
+ ```python
55
+ from inferhub_client import InferHubClient, ChatMessage, ChatRequest
56
+
57
+ with InferHubClient("http://localhost:5080/", api_key="sk-client-token-1") as client:
58
+ answer = client.chat(ChatRequest(
59
+ model="llama3",
60
+ messages=[ChatMessage(role="user", content="Say hi in one word.")],
61
+ ))
62
+ print(answer.message.content)
63
+ ```
64
+
65
+ Async — the same shapes, `await`ed:
66
+
67
+ ```python
68
+ import asyncio
69
+ from inferhub_client import AsyncInferHubClient, ChatMessage, ChatRequest
70
+
71
+ async def main():
72
+ async with AsyncInferHubClient("http://localhost:5080/", api_key="sk-client-token-1") as client:
73
+ answer = await client.chat(ChatRequest(
74
+ model="llama3",
75
+ messages=[ChatMessage(role="user", content="Say hi in one word.")],
76
+ ))
77
+ print(answer.message.content)
78
+
79
+ asyncio.run(main())
80
+ ```
81
+
82
+ `InferHubClient` and `AsyncInferHubClient` are **two thin façades over the same rules**
83
+ (`_base.py`'s header building, error mapping and NDJSON parsing) rather than one client with a
84
+ sync-over-`asyncio.run` shim — the latter breaks the moment a sync call happens inside code that is
85
+ already running an event loop, which is exactly where a web framework's request handler lives.
86
+
87
+ ## API surface (v0.1.0)
88
+
89
+ | Method | Endpoint |
90
+ |---|---|
91
+ | `list_models()` | `GET /api/tags` |
92
+ | `chat(request)` | `POST /api/chat` with `stream:false` |
93
+ | `chat_stream(request)` | `POST /api/chat` with `stream:true` — an iterator/async iterator of `ChatResponse` |
94
+ | `generate(request)` | `POST /api/generate` with `stream:false` |
95
+ | `generate_stream(request)` | `POST /api/generate` with `stream:true` |
96
+ | `embed(request)` | `POST /api/embed` (batch — a string or a list of strings) |
97
+ | `embed_legacy(request)` | `POST /api/embeddings` (legacy single prompt) |
98
+ | `get_status()` | `GET /api/status` |
99
+ | `ping()` | `GET /health` — `True`/`False`, never raises for a non-success status |
100
+
101
+ ## Streaming
102
+
103
+ ```python
104
+ for chunk in client.chat_stream(ChatRequest(model="llama3", messages=[...])):
105
+ print(chunk.message.content, end="", flush=True)
106
+ ```
107
+
108
+ A terminal error chunk (`{"error": "...", "done": true}`) raises `InferHubError` out of the loop
109
+ instead of the iterator hanging or ending quietly with a partial answer nobody was told about.
110
+
111
+ ## Errors
112
+
113
+ Every non-success response raises `InferHubError(status_code, message, response_body,
114
+ retry_after=...)`. `retry_after` is populated from `Retry-After` when the hub sends one — the
115
+ refusals that carry it are the ones worth retrying rather than only reporting.
116
+
117
+ ```python
118
+ from inferhub_client import InferHubError
119
+
120
+ try:
121
+ client.embed(EmbedRequest.from_text("nomic-embed-text", "hello"))
122
+ except InferHubError as e:
123
+ print(e.status_code, e.message, e.retry_after)
124
+ ```
125
+
126
+ ## `extra`: the fields this version does not know about yet
127
+
128
+ `ChatRequest`/`GenerateRequest.extra` merges straight into the request body (Ollama's `options`,
129
+ `format`, `keep_alive`, tool definitions — anything the hub accepts that this client has not typed);
130
+ every response dataclass keeps unrecognized fields in its own `.extra` dict on the way back. Typing
131
+ every Ollama option was considered and rejected, same as the C# client: the hub owns that schema and
132
+ grows it independently of this package's release cadence.
133
+
134
+ ## A node as a target
135
+
136
+ A solo InferHub node serves this same Ollama-dialect surface on its own address — pointing
137
+ `InferHubClient`/`AsyncInferHubClient` at a node's URL instead of a coordinator's is the whole of
138
+ "run it locally." `probe()` and the node-only routes (`/api/version`, the `/api/collections`
139
+ lifecycle) land in `1.0.0`, mirroring the C# client's phase 14 (`14 D7`).
140
+
141
+ ## Development
142
+
143
+ ```
144
+ pip install -e ".[test]"
145
+ pytest # 31 pass, 9 skipped (cases outside v0.1.0's surface, see below)
146
+ ruff check src tests examples
147
+ ruff format --check src tests examples
148
+ ```
149
+
150
+ `tests/test_conformance.py` drives the shared corpus at `../conformance/cases.json` — the same file
151
+ the C# client's `ConformanceCorpusTests.cs` reads. A case whose `kind` this client does not cover
152
+ yet (retrieval, the node, the OpenAI dialect) is skipped with a named reason rather than silently
153
+ omitted; four cases (the mid-stream terminal error, `424` vs `404`, both `X-InferHub-Sources`
154
+ shapes) pass today, unmodified, because the corpus already knew the answer.
155
+
156
+ ## License
157
+
158
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,130 @@
1
+ # inferhub-client — the Python client
2
+
3
+ [![PyPI](https://img.shields.io/pypi/v/inferhub-client.svg)](https://pypi.org/project/inferhub-client/)
4
+ [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
5
+
6
+ A small, typed Python client for [InferHub](https://github.com/Dev-Art-Solutions/InferHub) — a
7
+ self-hosted, Ollama-compatible inference mesh. `v0.1.0` is the **core** surface: chat, generate
8
+ (blocking and streaming), embeddings, model listing, status and health. Retrieval (vectors, RAG,
9
+ ingestion, search) lands in `0.2.0`; modalities, admin and the node in `1.0.0` — see
10
+ `plans/roadmap-polyglot-clients.md` for the shape of the rest of the track.
11
+
12
+ **One dependency: `httpx`.** No pydantic — dataclasses do the job and every response type carries an
13
+ `extra` dict for fields this version does not know about yet, the same escape hatch the C# client's
14
+ `[JsonExtensionData]` gives it.
15
+
16
+ ## Install
17
+
18
+ ```
19
+ pip install inferhub-client
20
+ ```
21
+
22
+ ## Quick start
23
+
24
+ Sync:
25
+
26
+ ```python
27
+ from inferhub_client import InferHubClient, ChatMessage, ChatRequest
28
+
29
+ with InferHubClient("http://localhost:5080/", api_key="sk-client-token-1") as client:
30
+ answer = client.chat(ChatRequest(
31
+ model="llama3",
32
+ messages=[ChatMessage(role="user", content="Say hi in one word.")],
33
+ ))
34
+ print(answer.message.content)
35
+ ```
36
+
37
+ Async — the same shapes, `await`ed:
38
+
39
+ ```python
40
+ import asyncio
41
+ from inferhub_client import AsyncInferHubClient, ChatMessage, ChatRequest
42
+
43
+ async def main():
44
+ async with AsyncInferHubClient("http://localhost:5080/", api_key="sk-client-token-1") as client:
45
+ answer = await client.chat(ChatRequest(
46
+ model="llama3",
47
+ messages=[ChatMessage(role="user", content="Say hi in one word.")],
48
+ ))
49
+ print(answer.message.content)
50
+
51
+ asyncio.run(main())
52
+ ```
53
+
54
+ `InferHubClient` and `AsyncInferHubClient` are **two thin façades over the same rules**
55
+ (`_base.py`'s header building, error mapping and NDJSON parsing) rather than one client with a
56
+ sync-over-`asyncio.run` shim — the latter breaks the moment a sync call happens inside code that is
57
+ already running an event loop, which is exactly where a web framework's request handler lives.
58
+
59
+ ## API surface (v0.1.0)
60
+
61
+ | Method | Endpoint |
62
+ |---|---|
63
+ | `list_models()` | `GET /api/tags` |
64
+ | `chat(request)` | `POST /api/chat` with `stream:false` |
65
+ | `chat_stream(request)` | `POST /api/chat` with `stream:true` — an iterator/async iterator of `ChatResponse` |
66
+ | `generate(request)` | `POST /api/generate` with `stream:false` |
67
+ | `generate_stream(request)` | `POST /api/generate` with `stream:true` |
68
+ | `embed(request)` | `POST /api/embed` (batch — a string or a list of strings) |
69
+ | `embed_legacy(request)` | `POST /api/embeddings` (legacy single prompt) |
70
+ | `get_status()` | `GET /api/status` |
71
+ | `ping()` | `GET /health` — `True`/`False`, never raises for a non-success status |
72
+
73
+ ## Streaming
74
+
75
+ ```python
76
+ for chunk in client.chat_stream(ChatRequest(model="llama3", messages=[...])):
77
+ print(chunk.message.content, end="", flush=True)
78
+ ```
79
+
80
+ A terminal error chunk (`{"error": "...", "done": true}`) raises `InferHubError` out of the loop
81
+ instead of the iterator hanging or ending quietly with a partial answer nobody was told about.
82
+
83
+ ## Errors
84
+
85
+ Every non-success response raises `InferHubError(status_code, message, response_body,
86
+ retry_after=...)`. `retry_after` is populated from `Retry-After` when the hub sends one — the
87
+ refusals that carry it are the ones worth retrying rather than only reporting.
88
+
89
+ ```python
90
+ from inferhub_client import InferHubError
91
+
92
+ try:
93
+ client.embed(EmbedRequest.from_text("nomic-embed-text", "hello"))
94
+ except InferHubError as e:
95
+ print(e.status_code, e.message, e.retry_after)
96
+ ```
97
+
98
+ ## `extra`: the fields this version does not know about yet
99
+
100
+ `ChatRequest`/`GenerateRequest.extra` merges straight into the request body (Ollama's `options`,
101
+ `format`, `keep_alive`, tool definitions — anything the hub accepts that this client has not typed);
102
+ every response dataclass keeps unrecognized fields in its own `.extra` dict on the way back. Typing
103
+ every Ollama option was considered and rejected, same as the C# client: the hub owns that schema and
104
+ grows it independently of this package's release cadence.
105
+
106
+ ## A node as a target
107
+
108
+ A solo InferHub node serves this same Ollama-dialect surface on its own address — pointing
109
+ `InferHubClient`/`AsyncInferHubClient` at a node's URL instead of a coordinator's is the whole of
110
+ "run it locally." `probe()` and the node-only routes (`/api/version`, the `/api/collections`
111
+ lifecycle) land in `1.0.0`, mirroring the C# client's phase 14 (`14 D7`).
112
+
113
+ ## Development
114
+
115
+ ```
116
+ pip install -e ".[test]"
117
+ pytest # 31 pass, 9 skipped (cases outside v0.1.0's surface, see below)
118
+ ruff check src tests examples
119
+ ruff format --check src tests examples
120
+ ```
121
+
122
+ `tests/test_conformance.py` drives the shared corpus at `../conformance/cases.json` — the same file
123
+ the C# client's `ConformanceCorpusTests.cs` reads. A case whose `kind` this client does not cover
124
+ yet (retrieval, the node, the OpenAI dialect) is skipped with a named reason rather than silently
125
+ omitted; four cases (the mid-stream terminal error, `424` vs `404`, both `X-InferHub-Sources`
126
+ shapes) pass today, unmodified, because the corpus already knew the answer.
127
+
128
+ ## License
129
+
130
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,58 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "inferhub-client"
7
+ version = "0.1.0"
8
+ description = "A small, typed Python client for InferHub — a self-hosted, Ollama-compatible inference mesh."
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ authors = [{ name = "Dev Art Solutions" }]
12
+ requires-python = ">=3.9"
13
+ dependencies = ["httpx>=0.24"]
14
+ classifiers = [
15
+ "Development Status :: 3 - Alpha",
16
+ "Intended Audience :: Developers",
17
+ "License :: OSI Approved :: MIT License",
18
+ "Programming Language :: Python :: 3",
19
+ "Programming Language :: Python :: 3.9",
20
+ "Programming Language :: Python :: 3.10",
21
+ "Programming Language :: Python :: 3.11",
22
+ "Programming Language :: Python :: 3.12",
23
+ "Programming Language :: Python :: 3.13",
24
+ "Typing :: Typed",
25
+ ]
26
+
27
+ [project.urls]
28
+ Homepage = "https://github.com/Dev-Art-Solutions/InferHub.Clients"
29
+ Repository = "https://github.com/Dev-Art-Solutions/InferHub.Clients"
30
+ Issues = "https://github.com/Dev-Art-Solutions/InferHub.Clients/issues"
31
+
32
+ [project.optional-dependencies]
33
+ # ruff pinned exactly, not >=: two different minor releases (0.15 -> 0.16) changed which lint
34
+ # rules are on by default with nothing in this file changed, which broke CI twice in a row on an
35
+ # unpinned version. A linter whose pass/fail depends on which version happened to install is not
36
+ # a check, so this one is deterministic instead — bump it deliberately, in its own commit, when
37
+ # there is a reason to.
38
+ test = ["pytest>=7", "pytest-asyncio>=0.23", "ruff==0.15.13"]
39
+
40
+ [tool.hatch.build.targets.wheel]
41
+ packages = ["src/inferhub_client"]
42
+
43
+ [tool.hatch.build.targets.sdist]
44
+ include = ["src", "README.md", "LICENSE"]
45
+
46
+ [tool.pytest.ini_options]
47
+ asyncio_mode = "auto"
48
+ testpaths = ["tests"]
49
+
50
+ [tool.ruff]
51
+ target-version = "py39"
52
+
53
+ [tool.ruff.lint]
54
+ # Pinned explicitly rather than left to ruff's default selection: 0.16 started enabling `I`
55
+ # (isort) by default where 0.15 did not, which broke CI on an unpinned `ruff>=0.6` the moment
56
+ # the runner picked up a newer release with nothing in this file changed. Extending the default
57
+ # set (E4, E7, E9, F) rather than replacing it.
58
+ extend-select = ["I"]
@@ -0,0 +1,45 @@
1
+ """inferhub-client — a small, typed Python client for InferHub.
2
+
3
+ >>> from inferhub_client import InferHubClient, ChatMessage, ChatRequest
4
+ >>> with InferHubClient("http://localhost:5080/", api_key="sk-...") as client:
5
+ ... answer = client.chat(ChatRequest(model="llama3", messages=[ChatMessage("user", "hi")]))
6
+ ... print(answer.message.content)
7
+ """
8
+
9
+ from ._async_client import AsyncInferHubClient
10
+ from ._client import InferHubClient
11
+ from ._exceptions import InferHubError
12
+ from ._models import (
13
+ ChatMessage,
14
+ ChatRequest,
15
+ ChatResponse,
16
+ EmbeddingsRequest,
17
+ EmbeddingsResponse,
18
+ EmbedRequest,
19
+ EmbedResponse,
20
+ GenerateRequest,
21
+ GenerateResponse,
22
+ ModelInfo,
23
+ StatusResponse,
24
+ TagsResponse,
25
+ )
26
+ from ._version import __version__
27
+
28
+ __all__ = [
29
+ "__version__",
30
+ "InferHubClient",
31
+ "AsyncInferHubClient",
32
+ "InferHubError",
33
+ "ChatMessage",
34
+ "ChatRequest",
35
+ "ChatResponse",
36
+ "GenerateRequest",
37
+ "GenerateResponse",
38
+ "EmbedRequest",
39
+ "EmbedResponse",
40
+ "EmbeddingsRequest",
41
+ "EmbeddingsResponse",
42
+ "ModelInfo",
43
+ "TagsResponse",
44
+ "StatusResponse",
45
+ ]
@@ -0,0 +1,160 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import AsyncIterator, Optional
4
+
5
+ import httpx
6
+
7
+ from ._base import (
8
+ DEFAULT_BASE_URL,
9
+ build_headers,
10
+ parse_ndjson_line,
11
+ raise_for_status,
12
+ read_served_by,
13
+ read_source_ids,
14
+ )
15
+ from ._exceptions import InferHubError
16
+ from ._models import (
17
+ ChatRequest,
18
+ ChatResponse,
19
+ EmbeddingsRequest,
20
+ EmbeddingsResponse,
21
+ EmbedRequest,
22
+ EmbedResponse,
23
+ GenerateRequest,
24
+ GenerateResponse,
25
+ StatusResponse,
26
+ TagsResponse,
27
+ )
28
+
29
+
30
+ class AsyncInferHubClient:
31
+ """Async client for an InferHub coordinator (or a solo node — same address, same client, see
32
+ ``python/README.md``). Covers the Ollama-dialect core surface: chat, generate (blocking and
33
+ streaming), embeddings, model listing, status and health.
34
+ """
35
+
36
+ def __init__(
37
+ self,
38
+ base_url: str = DEFAULT_BASE_URL,
39
+ api_key: Optional[str] = None,
40
+ *,
41
+ timeout: float = 100.0,
42
+ http_client: Optional[httpx.AsyncClient] = None,
43
+ ) -> None:
44
+ self._owns_client = http_client is None
45
+ self._http = http_client or httpx.AsyncClient(
46
+ base_url=base_url, headers=build_headers(api_key), timeout=timeout
47
+ )
48
+
49
+ async def __aenter__(self) -> "AsyncInferHubClient":
50
+ return self
51
+
52
+ async def __aexit__(self, *exc_info: object) -> None:
53
+ await self.aclose()
54
+
55
+ async def aclose(self) -> None:
56
+ if self._owns_client:
57
+ await self._http.aclose()
58
+
59
+ async def list_models(self) -> TagsResponse:
60
+ """``GET /api/tags`` — models advertised by the mesh."""
61
+ response = await self._http.get("api/tags")
62
+ raise_for_status(response)
63
+ return TagsResponse.from_json(response.json())
64
+
65
+ async def chat(self, request: ChatRequest) -> ChatResponse:
66
+ """Blocking chat — ``POST /api/chat`` with ``stream:false``."""
67
+ request.stream = False
68
+ response = await self._http.post("api/chat", json=request.to_json())
69
+ raise_for_status(response)
70
+ result = ChatResponse.from_json(response.json())
71
+ result.served_by = read_served_by(response)
72
+ result.source_ids = read_source_ids(response)
73
+ return result
74
+
75
+ async def chat_stream(self, request: ChatRequest) -> AsyncIterator[ChatResponse]:
76
+ """Streaming chat — ``POST /api/chat`` with ``stream:true``. Yields one
77
+ :class:`ChatResponse` per NDJSON line; a terminal error chunk raises
78
+ :class:`InferHubError` instead of the iterator hanging or ending quietly."""
79
+ request.stream = True
80
+ async with self._http.stream(
81
+ "POST", "api/chat", json=request.to_json()
82
+ ) as response:
83
+ raise_for_status(response)
84
+ served_by = read_served_by(response)
85
+ source_ids = read_source_ids(response)
86
+ async for line in response.aiter_lines():
87
+ chunk = parse_ndjson_line(line)
88
+ if chunk is None:
89
+ continue
90
+ result = ChatResponse.from_json(chunk)
91
+ result.served_by = served_by
92
+ result.source_ids = source_ids
93
+ yield result
94
+ if result.done:
95
+ return
96
+
97
+ async def generate(self, request: GenerateRequest) -> GenerateResponse:
98
+ """Blocking generate — ``POST /api/generate`` with ``stream:false``."""
99
+ request.stream = False
100
+ response = await self._http.post("api/generate", json=request.to_json())
101
+ raise_for_status(response)
102
+ result = GenerateResponse.from_json(response.json())
103
+ result.served_by = read_served_by(response)
104
+ result.source_ids = read_source_ids(response)
105
+ return result
106
+
107
+ async def generate_stream(
108
+ self, request: GenerateRequest
109
+ ) -> AsyncIterator[GenerateResponse]:
110
+ """Streaming generate — ``POST /api/generate`` with ``stream:true``."""
111
+ request.stream = True
112
+ async with self._http.stream(
113
+ "POST", "api/generate", json=request.to_json()
114
+ ) as response:
115
+ raise_for_status(response)
116
+ served_by = read_served_by(response)
117
+ source_ids = read_source_ids(response)
118
+ async for line in response.aiter_lines():
119
+ chunk = parse_ndjson_line(line)
120
+ if chunk is None:
121
+ continue
122
+ result = GenerateResponse.from_json(chunk)
123
+ result.served_by = served_by
124
+ result.source_ids = source_ids
125
+ yield result
126
+ if result.done:
127
+ return
128
+
129
+ async def embed(self, request: EmbedRequest) -> EmbedResponse:
130
+ """``POST /api/embed`` — batch embeddings. An empty vector list on a 200 is treated as a
131
+ malformed response and raised, never silently returned."""
132
+ response = await self._http.post("api/embed", json=request.to_json())
133
+ raise_for_status(response)
134
+ result = EmbedResponse.from_json(response.json())
135
+ if not result.embeddings:
136
+ raise InferHubError(response.status_code, "embed response had no vectors")
137
+ return result
138
+
139
+ async def embed_legacy(self, request: EmbeddingsRequest) -> EmbeddingsResponse:
140
+ """``POST /api/embeddings`` — the legacy single-input endpoint. Prefer :meth:`embed`."""
141
+ response = await self._http.post("api/embeddings", json=request.to_json())
142
+ raise_for_status(response)
143
+ result = EmbeddingsResponse.from_json(response.json())
144
+ if not result.embedding:
145
+ raise InferHubError(
146
+ response.status_code, "embeddings response had no vector"
147
+ )
148
+ return result
149
+
150
+ async def get_status(self) -> StatusResponse:
151
+ """``GET /api/status`` — coordinator/fleet snapshot."""
152
+ response = await self._http.get("api/status")
153
+ raise_for_status(response)
154
+ return StatusResponse.from_json(response.json())
155
+
156
+ async def ping(self) -> bool:
157
+ """``GET /health`` — ``True`` on 2xx, ``False`` otherwise. Never raises for a non-success
158
+ status; raises only on a transport error."""
159
+ response = await self._http.get("health")
160
+ return response.is_success
@@ -0,0 +1,109 @@
1
+ """Shared, I/O-free plumbing used by both :class:`InferHubClient` (sync) and
2
+ :class:`AsyncInferHubClient` (async): headers, error mapping, and NDJSON chunk parsing. Kept out of
3
+ either client so the two stay thin façades over the same rules rather than two copies that drift —
4
+ see ``python/README.md`` on why this is two classes and not one client with a sync-over-async shim.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ from typing import Any, Dict, Optional
11
+
12
+ import httpx
13
+
14
+ from ._exceptions import InferHubError
15
+
16
+ DEFAULT_BASE_URL = "http://localhost:5080/"
17
+
18
+
19
+ def build_headers(api_key: Optional[str]) -> Dict[str, str]:
20
+ headers = {"Content-Type": "application/json"}
21
+ if api_key:
22
+ headers["Authorization"] = f"Bearer {api_key}"
23
+ return headers
24
+
25
+
26
+ def _extract_error_message(body: str) -> Optional[str]:
27
+ """The Ollama dialect answers ``{"error": "..."}"``. A non-JSON or differently-shaped body
28
+ falls back to the raw text, same as the C# client's ``TryExtractErrorMessage``."""
29
+
30
+ if not body.strip():
31
+ return None
32
+ try:
33
+ parsed = json.loads(body)
34
+ except (json.JSONDecodeError, ValueError):
35
+ return body
36
+ if isinstance(parsed, dict) and isinstance(parsed.get("error"), str):
37
+ return parsed["error"]
38
+ return body
39
+
40
+
41
+ def _retry_after(response: httpx.Response) -> Optional[float]:
42
+ header = response.headers.get("Retry-After")
43
+ if not header:
44
+ return None
45
+ try:
46
+ return float(header)
47
+ except ValueError:
48
+ return None # An HTTP-date form exists but the hub always writes delta-seconds.
49
+
50
+
51
+ def raise_for_status(response: httpx.Response) -> None:
52
+ if response.is_success:
53
+ return
54
+
55
+ body = response.text
56
+ message = _extract_error_message(body) or (
57
+ f"InferHub request failed with status {response.status_code}."
58
+ )
59
+ raise InferHubError(
60
+ response.status_code, message, body, retry_after=_retry_after(response)
61
+ )
62
+
63
+
64
+ def parse_ndjson_line(line: str) -> Optional[Dict[str, Any]]:
65
+ """One line of an NDJSON stream, or ``None`` for a blank line to skip. Raises
66
+ :class:`InferHubError` on a terminal error chunk (``{"error": ..., "done": true}``) so a caller's
67
+ loop stops with a clear exception instead of hanging or silently finishing early."""
68
+
69
+ if not line.strip():
70
+ return None
71
+
72
+ chunk = json.loads(line)
73
+ error = chunk.get("error")
74
+ if error:
75
+ raise InferHubError(200, error, line)
76
+ return chunk
77
+
78
+
79
+ def read_served_by(response: httpx.Response) -> Optional[str]:
80
+ """Which node or ``provider:<id>`` answered — surfaced, never interpreted. This client does
81
+ not route, retry elsewhere or prefer on it (root ``CLAUDE.md`` rule 8); reading it here is what
82
+ lets a caller log or display it without this library making a decision on its behalf."""
83
+
84
+ value = response.headers.get("X-InferHub-Served-By")
85
+ return value.strip() or None if value else None
86
+
87
+
88
+ def read_source_ids(response: httpx.Response) -> Optional[list]:
89
+ """``X-InferHub-Sources`` arrives as a JSON array, but a real hub has also sent it
90
+ comma-separated — ``spec/README.md`` calls this the conformance corpus's first case, and both
91
+ shapes are parsed here even though ``v0.1.0`` has no way yet to opt into retrieval (phase 17):
92
+ the header is part of the core response contract, and a caller reading it manually should not
93
+ have to wait for this client to grow RAG headers first."""
94
+
95
+ raw = response.headers.get("X-InferHub-Sources")
96
+ if raw is None:
97
+ return None
98
+ raw = raw.strip()
99
+ if not raw:
100
+ return []
101
+ try:
102
+ parsed = json.loads(raw)
103
+ if isinstance(parsed, list):
104
+ return [
105
+ str(item) for item in parsed if item is not None and str(item) != ""
106
+ ]
107
+ except (json.JSONDecodeError, ValueError):
108
+ pass
109
+ return [part.strip() for part in raw.split(",") if part.strip()]
@@ -0,0 +1,157 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import Iterator, Optional
4
+
5
+ import httpx
6
+
7
+ from ._base import (
8
+ DEFAULT_BASE_URL,
9
+ build_headers,
10
+ parse_ndjson_line,
11
+ raise_for_status,
12
+ read_served_by,
13
+ read_source_ids,
14
+ )
15
+ from ._exceptions import InferHubError
16
+ from ._models import (
17
+ ChatRequest,
18
+ ChatResponse,
19
+ EmbeddingsRequest,
20
+ EmbeddingsResponse,
21
+ EmbedRequest,
22
+ EmbedResponse,
23
+ GenerateRequest,
24
+ GenerateResponse,
25
+ StatusResponse,
26
+ TagsResponse,
27
+ )
28
+
29
+
30
+ class InferHubClient:
31
+ """Sync client for an InferHub coordinator (or a solo node — same address, same client, see
32
+ ``python/README.md``). A thin façade over the same rules :class:`AsyncInferHubClient` uses
33
+ (``_base.py``), for callers not already in an event loop. Covers the Ollama-dialect core
34
+ surface: chat, generate (blocking and streaming), embeddings, model listing, status and health.
35
+ """
36
+
37
+ def __init__(
38
+ self,
39
+ base_url: str = DEFAULT_BASE_URL,
40
+ api_key: Optional[str] = None,
41
+ *,
42
+ timeout: float = 100.0,
43
+ http_client: Optional[httpx.Client] = None,
44
+ ) -> None:
45
+ self._owns_client = http_client is None
46
+ self._http = http_client or httpx.Client(
47
+ base_url=base_url, headers=build_headers(api_key), timeout=timeout
48
+ )
49
+
50
+ def __enter__(self) -> "InferHubClient":
51
+ return self
52
+
53
+ def __exit__(self, *exc_info: object) -> None:
54
+ self.close()
55
+
56
+ def close(self) -> None:
57
+ if self._owns_client:
58
+ self._http.close()
59
+
60
+ def list_models(self) -> TagsResponse:
61
+ """``GET /api/tags`` — models advertised by the mesh."""
62
+ response = self._http.get("api/tags")
63
+ raise_for_status(response)
64
+ return TagsResponse.from_json(response.json())
65
+
66
+ def chat(self, request: ChatRequest) -> ChatResponse:
67
+ """Blocking chat — ``POST /api/chat`` with ``stream:false``."""
68
+ request.stream = False
69
+ response = self._http.post("api/chat", json=request.to_json())
70
+ raise_for_status(response)
71
+ result = ChatResponse.from_json(response.json())
72
+ result.served_by = read_served_by(response)
73
+ result.source_ids = read_source_ids(response)
74
+ return result
75
+
76
+ def chat_stream(self, request: ChatRequest) -> Iterator[ChatResponse]:
77
+ """Streaming chat — ``POST /api/chat`` with ``stream:true``. Yields one
78
+ :class:`ChatResponse` per NDJSON line; a terminal error chunk raises
79
+ :class:`InferHubError` instead of the iterator hanging or ending quietly."""
80
+ request.stream = True
81
+ with self._http.stream("POST", "api/chat", json=request.to_json()) as response:
82
+ raise_for_status(response)
83
+ served_by = read_served_by(response)
84
+ source_ids = read_source_ids(response)
85
+ for line in response.iter_lines():
86
+ chunk = parse_ndjson_line(line)
87
+ if chunk is None:
88
+ continue
89
+ result = ChatResponse.from_json(chunk)
90
+ result.served_by = served_by
91
+ result.source_ids = source_ids
92
+ yield result
93
+ if result.done:
94
+ return
95
+
96
+ def generate(self, request: GenerateRequest) -> GenerateResponse:
97
+ """Blocking generate — ``POST /api/generate`` with ``stream:false``."""
98
+ request.stream = False
99
+ response = self._http.post("api/generate", json=request.to_json())
100
+ raise_for_status(response)
101
+ result = GenerateResponse.from_json(response.json())
102
+ result.served_by = read_served_by(response)
103
+ result.source_ids = read_source_ids(response)
104
+ return result
105
+
106
+ def generate_stream(self, request: GenerateRequest) -> Iterator[GenerateResponse]:
107
+ """Streaming generate — ``POST /api/generate`` with ``stream:true``."""
108
+ request.stream = True
109
+ with self._http.stream(
110
+ "POST", "api/generate", json=request.to_json()
111
+ ) as response:
112
+ raise_for_status(response)
113
+ served_by = read_served_by(response)
114
+ source_ids = read_source_ids(response)
115
+ for line in response.iter_lines():
116
+ chunk = parse_ndjson_line(line)
117
+ if chunk is None:
118
+ continue
119
+ result = GenerateResponse.from_json(chunk)
120
+ result.served_by = served_by
121
+ result.source_ids = source_ids
122
+ yield result
123
+ if result.done:
124
+ return
125
+
126
+ def embed(self, request: EmbedRequest) -> EmbedResponse:
127
+ """``POST /api/embed`` — batch embeddings. An empty vector list on a 200 is treated as a
128
+ malformed response and raised, never silently returned."""
129
+ response = self._http.post("api/embed", json=request.to_json())
130
+ raise_for_status(response)
131
+ result = EmbedResponse.from_json(response.json())
132
+ if not result.embeddings:
133
+ raise InferHubError(response.status_code, "embed response had no vectors")
134
+ return result
135
+
136
+ def embed_legacy(self, request: EmbeddingsRequest) -> EmbeddingsResponse:
137
+ """``POST /api/embeddings`` — the legacy single-input endpoint. Prefer :meth:`embed`."""
138
+ response = self._http.post("api/embeddings", json=request.to_json())
139
+ raise_for_status(response)
140
+ result = EmbeddingsResponse.from_json(response.json())
141
+ if not result.embedding:
142
+ raise InferHubError(
143
+ response.status_code, "embeddings response had no vector"
144
+ )
145
+ return result
146
+
147
+ def get_status(self) -> StatusResponse:
148
+ """``GET /api/status`` — coordinator/fleet snapshot."""
149
+ response = self._http.get("api/status")
150
+ raise_for_status(response)
151
+ return StatusResponse.from_json(response.json())
152
+
153
+ def ping(self) -> bool:
154
+ """``GET /health`` — ``True`` on 2xx, ``False`` otherwise. Never raises for a non-success
155
+ status; raises only on a transport error."""
156
+ response = self._http.get("health")
157
+ return response.is_success
@@ -0,0 +1,31 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import Optional
4
+
5
+
6
+ class InferHubError(Exception):
7
+ """Raised when the coordinator (or a solo node) answers a non-success HTTP status.
8
+
9
+ Carries the raw status code so a caller can distinguish 404 (model or collection missing),
10
+ 401/403 (auth), 501 (a backend that structurally cannot serve a capability) from 503 (a
11
+ capability an operator disabled, temporary, and worth the ``retry_after`` seconds).
12
+ """
13
+
14
+ def __init__(
15
+ self,
16
+ status_code: int,
17
+ message: str,
18
+ response_body: str = "",
19
+ *,
20
+ retry_after: Optional[float] = None,
21
+ ) -> None:
22
+ super().__init__(message)
23
+ self.status_code = status_code
24
+ self.message = message
25
+ self.response_body = response_body
26
+ self.retry_after = retry_after
27
+
28
+ def __repr__(self) -> str: # pragma: no cover - cosmetic
29
+ return (
30
+ f"InferHubError(status_code={self.status_code!r}, message={self.message!r})"
31
+ )
@@ -0,0 +1,334 @@
1
+ """Typed request/response shapes for the Ollama-dialect core surface.
2
+
3
+ Dataclasses, not pydantic (root ``CLAUDE.md`` rule 2 / roadmap-polyglot-clients D5): the wire is
4
+ small and stable enough that a validation library is somebody else's dependency war inherited by
5
+ every consumer. Every response type keeps an ``extra`` dict for fields the hub sends that this
6
+ version does not know about yet — the Python equivalent of the C# client's ``[JsonExtensionData]``
7
+ bag — so a caller reaching for a brand-new field is never blocked on a new release of this package.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from dataclasses import dataclass, field
13
+ from typing import Any, Dict, List, Optional, Union
14
+
15
+ JsonDict = Dict[str, Any]
16
+
17
+ _KNOWN_MESSAGE_FIELDS = {"role", "content", "images", "tool_calls"}
18
+
19
+
20
+ @dataclass
21
+ class ChatMessage:
22
+ """One message in a chat request or response."""
23
+
24
+ role: str
25
+ content: str = ""
26
+ images: Optional[List[str]] = None
27
+ tool_calls: Optional[List[JsonDict]] = None
28
+ extra: JsonDict = field(default_factory=dict)
29
+
30
+ def to_json(self) -> JsonDict:
31
+ body: JsonDict = {"role": self.role, "content": self.content}
32
+ if self.images is not None:
33
+ body["images"] = self.images
34
+ if self.tool_calls is not None:
35
+ body["tool_calls"] = self.tool_calls
36
+ body.update(self.extra)
37
+ return body
38
+
39
+ @classmethod
40
+ def from_json(cls, data: JsonDict) -> "ChatMessage":
41
+ return cls(
42
+ role=data.get("role", ""),
43
+ content=data.get("content", ""),
44
+ images=data.get("images"),
45
+ tool_calls=data.get("tool_calls"),
46
+ extra={k: v for k, v in data.items() if k not in _KNOWN_MESSAGE_FIELDS},
47
+ )
48
+
49
+
50
+ @dataclass
51
+ class ChatRequest:
52
+ """``POST /api/chat``. ``extra`` merges straight into the top-level body, untouched — the
53
+ passthrough that keeps every Ollama option (``options``, ``format``, ``keep_alive``, tool
54
+ definitions) reachable without this client typing each one (roadmap-polyglot-clients,
55
+ "typed request builders for every option" is a rejected non-goal, same as the C# client)."""
56
+
57
+ model: str
58
+ messages: List[ChatMessage] = field(default_factory=list)
59
+ stream: bool = False
60
+ options: Optional[JsonDict] = None
61
+ format: Optional[Union[str, JsonDict]] = None
62
+ keep_alive: Optional[str] = None
63
+ extra: JsonDict = field(default_factory=dict)
64
+
65
+ def to_json(self) -> JsonDict:
66
+ body: JsonDict = {
67
+ "model": self.model,
68
+ "messages": [m.to_json() for m in self.messages],
69
+ "stream": self.stream,
70
+ }
71
+ if self.options is not None:
72
+ body["options"] = self.options
73
+ if self.format is not None:
74
+ body["format"] = self.format
75
+ if self.keep_alive is not None:
76
+ body["keep_alive"] = self.keep_alive
77
+ body.update(self.extra)
78
+ return body
79
+
80
+
81
+ _KNOWN_CHAT_RESPONSE_FIELDS = {
82
+ "model",
83
+ "created_at",
84
+ "message",
85
+ "done",
86
+ "done_reason",
87
+ "total_duration",
88
+ "load_duration",
89
+ "prompt_eval_count",
90
+ "prompt_eval_duration",
91
+ "eval_count",
92
+ "eval_duration",
93
+ "error",
94
+ }
95
+
96
+
97
+ @dataclass
98
+ class ChatResponse:
99
+ """A blocking answer, or one NDJSON chunk of a streamed one."""
100
+
101
+ model: str = ""
102
+ created_at: Optional[str] = None
103
+ message: Optional[ChatMessage] = None
104
+ done: Optional[bool] = None
105
+ done_reason: Optional[str] = None
106
+ total_duration: Optional[int] = None
107
+ load_duration: Optional[int] = None
108
+ prompt_eval_count: Optional[int] = None
109
+ prompt_eval_duration: Optional[int] = None
110
+ eval_count: Optional[int] = None
111
+ eval_duration: Optional[int] = None
112
+ error: Optional[str] = None
113
+ extra: JsonDict = field(default_factory=dict)
114
+ # Set from response headers, not the body — never part of round-tripping the JSON.
115
+ served_by: Optional[str] = None
116
+ source_ids: Optional[List[str]] = None
117
+
118
+ @classmethod
119
+ def from_json(cls, data: JsonDict) -> "ChatResponse":
120
+ message = data.get("message")
121
+ return cls(
122
+ model=data.get("model", ""),
123
+ created_at=data.get("created_at"),
124
+ message=ChatMessage.from_json(message) if message is not None else None,
125
+ done=data.get("done"),
126
+ done_reason=data.get("done_reason"),
127
+ total_duration=data.get("total_duration"),
128
+ load_duration=data.get("load_duration"),
129
+ prompt_eval_count=data.get("prompt_eval_count"),
130
+ prompt_eval_duration=data.get("prompt_eval_duration"),
131
+ eval_count=data.get("eval_count"),
132
+ eval_duration=data.get("eval_duration"),
133
+ error=data.get("error"),
134
+ extra={
135
+ k: v for k, v in data.items() if k not in _KNOWN_CHAT_RESPONSE_FIELDS
136
+ },
137
+ )
138
+
139
+
140
+ @dataclass
141
+ class GenerateRequest:
142
+ """``POST /api/generate``. Same extension-bag contract as :class:`ChatRequest`."""
143
+
144
+ model: str
145
+ prompt: str = ""
146
+ stream: bool = False
147
+ options: Optional[JsonDict] = None
148
+ format: Optional[Union[str, JsonDict]] = None
149
+ keep_alive: Optional[str] = None
150
+ extra: JsonDict = field(default_factory=dict)
151
+
152
+ def to_json(self) -> JsonDict:
153
+ body: JsonDict = {
154
+ "model": self.model,
155
+ "prompt": self.prompt,
156
+ "stream": self.stream,
157
+ }
158
+ if self.options is not None:
159
+ body["options"] = self.options
160
+ if self.format is not None:
161
+ body["format"] = self.format
162
+ if self.keep_alive is not None:
163
+ body["keep_alive"] = self.keep_alive
164
+ body.update(self.extra)
165
+ return body
166
+
167
+
168
+ _KNOWN_GENERATE_RESPONSE_FIELDS = {
169
+ "model",
170
+ "created_at",
171
+ "response",
172
+ "done",
173
+ "done_reason",
174
+ "context",
175
+ "total_duration",
176
+ "load_duration",
177
+ "prompt_eval_count",
178
+ "prompt_eval_duration",
179
+ "eval_count",
180
+ "eval_duration",
181
+ "error",
182
+ }
183
+
184
+
185
+ @dataclass
186
+ class GenerateResponse:
187
+ model: str = ""
188
+ created_at: Optional[str] = None
189
+ response: str = ""
190
+ done: Optional[bool] = None
191
+ done_reason: Optional[str] = None
192
+ context: Optional[List[int]] = None
193
+ total_duration: Optional[int] = None
194
+ load_duration: Optional[int] = None
195
+ prompt_eval_count: Optional[int] = None
196
+ prompt_eval_duration: Optional[int] = None
197
+ eval_count: Optional[int] = None
198
+ eval_duration: Optional[int] = None
199
+ error: Optional[str] = None
200
+ extra: JsonDict = field(default_factory=dict)
201
+ served_by: Optional[str] = None
202
+ source_ids: Optional[List[str]] = None
203
+
204
+ @classmethod
205
+ def from_json(cls, data: JsonDict) -> "GenerateResponse":
206
+ return cls(
207
+ model=data.get("model", ""),
208
+ created_at=data.get("created_at"),
209
+ response=data.get("response", ""),
210
+ done=data.get("done"),
211
+ done_reason=data.get("done_reason"),
212
+ context=data.get("context"),
213
+ total_duration=data.get("total_duration"),
214
+ load_duration=data.get("load_duration"),
215
+ prompt_eval_count=data.get("prompt_eval_count"),
216
+ prompt_eval_duration=data.get("prompt_eval_duration"),
217
+ eval_count=data.get("eval_count"),
218
+ eval_duration=data.get("eval_duration"),
219
+ error=data.get("error"),
220
+ extra={
221
+ k: v
222
+ for k, v in data.items()
223
+ if k not in _KNOWN_GENERATE_RESPONSE_FIELDS
224
+ },
225
+ )
226
+
227
+
228
+ @dataclass
229
+ class EmbedRequest:
230
+ """``POST /api/embed`` — the modern batch endpoint. ``input`` is a single string or a list."""
231
+
232
+ model: str
233
+ input: Union[str, List[str]]
234
+
235
+ def to_json(self) -> JsonDict:
236
+ return {"model": self.model, "input": self.input}
237
+
238
+ @classmethod
239
+ def from_text(cls, model: str, text: str) -> "EmbedRequest":
240
+ return cls(model=model, input=text)
241
+
242
+ @classmethod
243
+ def from_texts(cls, model: str, texts: List[str]) -> "EmbedRequest":
244
+ return cls(model=model, input=list(texts))
245
+
246
+
247
+ @dataclass
248
+ class EmbedResponse:
249
+ model: str = ""
250
+ embeddings: List[List[float]] = field(default_factory=list)
251
+
252
+ @classmethod
253
+ def from_json(cls, data: JsonDict) -> "EmbedResponse":
254
+ return cls(model=data.get("model", ""), embeddings=data.get("embeddings") or [])
255
+
256
+
257
+ @dataclass
258
+ class EmbeddingsRequest:
259
+ """``POST /api/embeddings`` — the legacy single-input endpoint. Prefer :class:`EmbedRequest`."""
260
+
261
+ model: str
262
+ prompt: str
263
+
264
+ def to_json(self) -> JsonDict:
265
+ return {"model": self.model, "prompt": self.prompt}
266
+
267
+
268
+ @dataclass
269
+ class EmbeddingsResponse:
270
+ embedding: List[float] = field(default_factory=list)
271
+
272
+ @classmethod
273
+ def from_json(cls, data: JsonDict) -> "EmbeddingsResponse":
274
+ return cls(embedding=data.get("embedding") or [])
275
+
276
+
277
+ @dataclass
278
+ class ModelInfo:
279
+ name: str
280
+ digest: Optional[str] = None
281
+ size: Optional[int] = None
282
+
283
+ @classmethod
284
+ def from_json(cls, data: JsonDict) -> "ModelInfo":
285
+ return cls(
286
+ name=data.get("name", ""), digest=data.get("digest"), size=data.get("size")
287
+ )
288
+
289
+
290
+ @dataclass
291
+ class TagsResponse:
292
+ models: List[ModelInfo] = field(default_factory=list)
293
+
294
+ @classmethod
295
+ def from_json(cls, data: JsonDict) -> "TagsResponse":
296
+ return cls(models=[ModelInfo.from_json(m) for m in data.get("models") or []])
297
+
298
+
299
+ _KNOWN_STATUS_FIELDS = {
300
+ "coordinatorVersion",
301
+ "nowUtc",
302
+ "uptimeSeconds",
303
+ "nodes",
304
+ "models",
305
+ "metrics",
306
+ "vector",
307
+ }
308
+
309
+
310
+ @dataclass
311
+ class StatusResponse:
312
+ """``GET /api/status`` on a coordinator. See :mod:`inferhub_client.probe` for the solo-node shape
313
+ (added in a later phase) and how a caller tells the two apart."""
314
+
315
+ coordinator_version: Optional[str] = None
316
+ now_utc: Optional[str] = None
317
+ uptime_seconds: Optional[float] = None
318
+ nodes: Optional[List[JsonDict]] = None
319
+ models: Optional[List[ModelInfo]] = None
320
+ extra: JsonDict = field(default_factory=dict)
321
+
322
+ @classmethod
323
+ def from_json(cls, data: JsonDict) -> "StatusResponse":
324
+ models = data.get("models")
325
+ return cls(
326
+ coordinator_version=data.get("coordinatorVersion"),
327
+ now_utc=data.get("nowUtc"),
328
+ uptime_seconds=data.get("uptimeSeconds"),
329
+ nodes=data.get("nodes"),
330
+ models=[ModelInfo.from_json(m) for m in models]
331
+ if models is not None
332
+ else None,
333
+ extra={k: v for k, v in data.items() if k not in _KNOWN_STATUS_FIELDS},
334
+ )
@@ -0,0 +1 @@
1
+ __version__ = "0.1.0"