saia-python 0.10.0__tar.gz → 0.10.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. {saia_python-0.10.0/saia_python.egg-info → saia_python-0.10.1}/PKG-INFO +2 -2
  2. {saia_python-0.10.0 → saia_python-0.10.1}/README.md +1 -1
  3. {saia_python-0.10.0 → saia_python-0.10.1}/pyproject.toml +1 -1
  4. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/aio.py +1 -1
  5. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/client.py +1 -1
  6. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/openai_compat.py +1 -1
  7. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/tokenizer.py +60 -40
  8. {saia_python-0.10.0 → saia_python-0.10.1/saia_python.egg-info}/PKG-INFO +2 -2
  9. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_tokenizer.py +13 -2
  10. {saia_python-0.10.0 → saia_python-0.10.1}/LICENSE +0 -0
  11. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/__init__.py +0 -0
  12. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/_async_http.py +0 -0
  13. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/_async_streaming.py +0 -0
  14. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/_http.py +0 -0
  15. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/_payloads.py +0 -0
  16. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/_streaming.py +0 -0
  17. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/_util.py +0 -0
  18. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/arcana.py +0 -0
  19. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/arcana_references.py +0 -0
  20. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/auth.py +0 -0
  21. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/chat.py +0 -0
  22. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/documents.py +0 -0
  23. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/exceptions.py +0 -0
  24. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/models.py +0 -0
  25. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/py.typed +0 -0
  26. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/rate_limits.py +0 -0
  27. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/responses.py +0 -0
  28. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/structured.py +0 -0
  29. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python/voice.py +0 -0
  30. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python.egg-info/SOURCES.txt +0 -0
  31. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python.egg-info/dependency_links.txt +0 -0
  32. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python.egg-info/requires.txt +0 -0
  33. {saia_python-0.10.0 → saia_python-0.10.1}/saia_python.egg-info/top_level.txt +0 -0
  34. {saia_python-0.10.0 → saia_python-0.10.1}/setup.cfg +0 -0
  35. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_arcana.py +0 -0
  36. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_arcana_references.py +0 -0
  37. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_async_arcana.py +0 -0
  38. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_async_chat.py +0 -0
  39. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_async_client.py +0 -0
  40. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_async_httpx_integration.py +0 -0
  41. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_async_streaming.py +0 -0
  42. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_async_transport.py +0 -0
  43. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_auth.py +0 -0
  44. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_chat.py +0 -0
  45. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_client.py +0 -0
  46. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_documents.py +0 -0
  47. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_exceptions.py +0 -0
  48. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_health_check.py +0 -0
  49. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_live_responses_route.py +0 -0
  50. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_live_structured.py +0 -0
  51. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_models.py +0 -0
  52. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_openai_compat.py +0 -0
  53. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_payloads.py +0 -0
  54. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_rate_limit_message.py +0 -0
  55. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_rate_limits.py +0 -0
  56. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_responses.py +0 -0
  57. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_setup_from_directory.py +0 -0
  58. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_streaming.py +0 -0
  59. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_structured.py +0 -0
  60. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_transport_policy.py +0 -0
  61. {saia_python-0.10.0 → saia_python-0.10.1}/tests/test_voice.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: saia-python
3
- Version: 0.10.0
3
+ Version: 0.10.1
4
4
  Summary: Python wrapper for the GWDG SAIA platform REST API
5
5
  Author: Friedrich Schwarz
6
6
  License-Expression: AGPL-3.0-only
@@ -137,7 +137,7 @@ async def main():
137
137
  async with AsyncSAIAClient() as client:
138
138
  # Non-streaming RAG chat
139
139
  answer = await client.arcana.chat(
140
- model="openai-gpt-oss-120b",
140
+ model="deepseek-v4-flash-0731",
141
141
  messages=[{"role": "user", "content": "Summarise the DLBCL first line."}],
142
142
  arcana_id="owner/kb",
143
143
  )
@@ -76,7 +76,7 @@ async def main():
76
76
  async with AsyncSAIAClient() as client:
77
77
  # Non-streaming RAG chat
78
78
  answer = await client.arcana.chat(
79
- model="openai-gpt-oss-120b",
79
+ model="deepseek-v4-flash-0731",
80
80
  messages=[{"role": "user", "content": "Summarise the DLBCL first line."}],
81
81
  arcana_id="owner/kb",
82
82
  )
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "saia-python"
7
- version = "0.10.0"
7
+ version = "0.10.1"
8
8
  description = "Python wrapper for the GWDG SAIA platform REST API"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -12,7 +12,7 @@ calls and ``async for`` over a stream::
12
12
  async with AsyncSAIAClient() as client:
13
13
  # non-streaming RAG chat
14
14
  answer = await client.arcana.chat(
15
- model="openai-gpt-oss-120b",
15
+ model="deepseek-v4-flash-0731",
16
16
  messages=[{"role": "user", "content": "..."}],
17
17
  arcana_id="owner/kb",
18
18
  )
@@ -154,7 +154,7 @@ class SAIAClient:
154
154
  Example::
155
155
 
156
156
  response = client.openai.chat.completions.create(
157
- model="llama-3.3-70b-instruct",
157
+ model="deepseek-v4-flash-0731",
158
158
  messages=[{"role": "user", "content": "Hello!"}],
159
159
  )
160
160
  """
@@ -43,7 +43,7 @@ def create_openai_client(
43
43
 
44
44
  client = create_openai_client()
45
45
  response = client.chat.completions.create(
46
- model="llama-3.3-70b-instruct",
46
+ model="deepseek-v4-flash-0731",
47
47
  messages=[{"role": "user", "content": "Hello!"}],
48
48
  )
49
49
 
@@ -50,16 +50,64 @@ if TYPE_CHECKING: # pragma: no cover - typing only
50
50
  # ---------------------------------------------------------------------------
51
51
 
52
52
  # Single source of truth: ``(api_model_id, display_name, hf_repo)`` for every
53
- # open-weight model GWDG hosts. ``api_model_id`` is the string passed as
54
- # ``"model"`` to the API (and returned as ``id`` by ``GET /models``);
53
+ # open-weight model GWDG hosts or has hosted. ``api_model_id`` is the string
54
+ # passed as ``"model"`` to the API (and returned as ``id`` by ``GET /models``);
55
55
  # ``display_name`` is the catalogue's "Model" column (and the ``name`` field of
56
56
  # the ``/models`` payload); ``hf_repo`` is the ``org/name`` the catalogue links
57
57
  # to on https://huggingface.co . Sourced from the GWDG model catalogue and
58
- # cross-checked against the live ``/models`` listing on 2026-06-21. Only
59
- # open-weight models are listed — the externally hosted, proprietary models
60
- # (GPT-5.x, o3, Claude, ...) have no downloadable tokenizer; see
58
+ # cross-checked against the live ``/models`` listing on 2026-06-21; the models
59
+ # added on 2026-10-09 come from the catalogue page alone. Only open-weight
60
+ # models are listed — the externally hosted, proprietary models (GPT-5.x, o3,
61
+ # Claude, ...) have no downloadable tokenizer; see
61
62
  # :data:`OPENAI_TIKTOKEN_ENCODINGS` for their byte-pair encodings.
62
63
  _MODEL_TABLE: list[tuple[str, str, str]] = [
64
+ # Served by GWDG as of 2026-10-09.
65
+ # DeepSeek V4 ships no Jinja chat template (only a Python encoder), so
66
+ # chat_template_tokens falls back to a plain render with a warning for it.
67
+ (
68
+ "deepseek-v4-flash-0731",
69
+ "DeepSeek V4 Flash 0731",
70
+ "deepseek-ai/DeepSeek-V4-Flash-0731",
71
+ ),
72
+ ("gemma-4-31b-it", "Gemma 4 31B Instruct", "google/gemma-4-31B-it"),
73
+ # Its tokenizer_config names TokenizersBackend, a transformers 5 class.
74
+ ("glm-5.3-flash", "GLM 5.3 Flash", "zai-org/GLM-5.3-Flash"),
75
+ (
76
+ "meta-llama-3.1-8b-instruct",
77
+ "Llama 3.1 8B Instruct",
78
+ "nvidia/Llama-3.1-8B-Instruct-FP8",
79
+ ),
80
+ (
81
+ "qwen3-30b-a3b-instruct-2507",
82
+ "Qwen 3 30B A3B Instruct 2507",
83
+ "Qwen/Qwen3-30B-A3B-Instruct-2507-FP8",
84
+ ),
85
+ ("qwen3-coder-next", "Qwen 3 Coder Next", "Qwen/Qwen3-Coder-Next-FP8"),
86
+ (
87
+ "qwen3-omni-30b-a3b-instruct",
88
+ "Qwen 3 Omni 30B A3B Instruct",
89
+ "Qwen/Qwen3-Omni-30B-A3B-Instruct",
90
+ ),
91
+ ("qwen3.5-397b-a17b", "Qwen 3.5 397B A17B", "Qwen/Qwen3.5-397B-A17B-GPTQ-Int4"),
92
+ ("qwen3.6-35b-a3b", "Qwen 3.6 35B A3B", "Qwen/Qwen3.6-35B-A3B-FP8"),
93
+ ("qwen3.8-27b", "Qwen 3.8 27B", "Qwen/Qwen3.8-27B-FP8"),
94
+ # Embedding models — served via /embeddings rather than the chat /models
95
+ # listing, but their tokenizers are useful for sizing RAG chunks.
96
+ # ``qwen3-embedding-4b`` is the model ARCANA's RAG pipeline uses internally.
97
+ (
98
+ "qwen3-embedding-4b",
99
+ "Qwen3 Embedding 4B",
100
+ "Qwen/Qwen3-Embedding-4B",
101
+ ),
102
+ (
103
+ "e5-mistral-7b-instruct",
104
+ "E5 Mistral 7B Instruct",
105
+ "intfloat/e5-mistral-7b-instruct",
106
+ ),
107
+ # No longer served by GWDG (openai-gpt-oss-120b, devstral-2-123b-instruct-2512
108
+ # and apertus-70b-instruct-2509 were retired on 2026-10-08). Kept because
109
+ # their Hugging Face repos remain: the ids still resolve and the tokenizers
110
+ # still download.
63
111
  (
64
112
  "apertus-70b-instruct-2509",
65
113
  "Apertus 70B Instruct 2509",
@@ -75,57 +123,26 @@ _MODEL_TABLE: list[tuple[str, str, str]] = [
75
123
  "Devstral 2 123B Instruct 2512",
76
124
  "mistralai/Devstral-2-123B-Instruct-2512",
77
125
  ),
78
- ("gemma-4-31b-it", "Gemma 4 31B Instruct", "google/gemma-4-31B-it"),
79
126
  ("glm-4.7", "GLM-4.7", "zai-org/GLM-4.7-FP8"),
80
127
  ("internvl3.5-30b-a3b", "InternVL 3.5 30B A3B", "OpenGVLab/InternVL3_5-30B-A3B-HF"),
81
128
  ("medgemma-27b-it", "MedGemma 27B Instruct", "google/medgemma-27b-it"),
82
- (
83
- "meta-llama-3.1-8b-instruct",
84
- "Llama 3.1 8B Instruct",
85
- "nvidia/Llama-3.1-8B-Instruct-FP8",
86
- ),
87
129
  (
88
130
  "mistral-large-3-675b-instruct-2512",
89
131
  "Mistral Large 3 675B Instruct 2512",
90
132
  "mistralai/Mistral-Large-3-675B-Instruct-2512-NVFP4",
91
133
  ),
92
134
  ("openai-gpt-oss-120b", "GPT OSS 120B", "openai/gpt-oss-120b"),
93
- (
94
- "qwen3-30b-a3b-instruct-2507",
95
- "Qwen 3 30B A3B Instruct 2507",
96
- "Qwen/Qwen3-30B-A3B-Instruct-2507-FP8",
97
- ),
98
135
  (
99
136
  "qwen3-coder-30b-a3b-instruct",
100
137
  "Qwen 3 Coder 30B A3B Instruct",
101
138
  "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8",
102
139
  ),
103
- (
104
- "qwen3-omni-30b-a3b-instruct",
105
- "Qwen 3 Omni 30B A3B Instruct",
106
- "Qwen/Qwen3-Omni-30B-A3B-Instruct",
107
- ),
108
140
  ("qwen3.5-122b-a10b", "Qwen 3.5 122B A10B", "Qwen/Qwen3.5-122B-A10B-GPTQ-Int4"),
109
- ("qwen3.5-397b-a17b", "Qwen 3.5 397B A17B", "Qwen/Qwen3.5-397B-A17B-GPTQ-Int4"),
110
- ("qwen3.6-35b-a3b", "Qwen 3.6 35B A3B", "Qwen/Qwen3.6-35B-A3B-FP8"),
111
141
  (
112
142
  "teuken-7b-instruct-research",
113
143
  "Teuken 7B Instruct Research",
114
144
  "openGPT-X/Teuken-7B-instruct-research-v0.4",
115
145
  ),
116
- # Embedding models — served via /embeddings rather than the chat /models
117
- # listing, but their tokenizers are useful for sizing RAG chunks.
118
- # ``qwen3-embedding-4b`` is the model ARCANA's RAG pipeline uses internally.
119
- (
120
- "qwen3-embedding-4b",
121
- "Qwen3 Embedding 4B",
122
- "Qwen/Qwen3-Embedding-4B",
123
- ),
124
- (
125
- "e5-mistral-7b-instruct",
126
- "E5 Mistral 7B Instruct",
127
- "intfloat/e5-mistral-7b-instruct",
128
- ),
129
146
  ]
130
147
 
131
148
  #: Mapping of GWDG API model id → Hugging Face ``org/name`` repository.
@@ -279,7 +296,9 @@ def available_open_models() -> list[str]:
279
296
  """Return the GWDG open-weight model ids known to this module.
280
297
 
281
298
  These are the keys of :data:`GWDG_MODEL_REPOS` — the models for which a
282
- tokenizer repository is published and can be downloaded.
299
+ tokenizer repository is published and can be downloaded, including the ones
300
+ GWDG no longer serves. For the models available right now, annotate the live
301
+ listing instead (:meth:`TokenizerService.available_repos`).
283
302
  """
284
303
  return list(GWDG_MODEL_REPOS)
285
304
 
@@ -289,12 +308,13 @@ def resolve_repo(model: str) -> str:
289
308
 
290
309
  Accepts, in order of preference:
291
310
 
292
- 1. A GWDG API model id (e.g. ``"openai-gpt-oss-120b"``) — exactly as
311
+ 1. A GWDG API model id (e.g. ``"deepseek-v4-flash-0731"``) — exactly as
293
312
  returned by ``GET /models`` / passed as ``"model"`` in API calls.
294
313
  2. A full ``org/name`` Hugging Face repo (anything containing ``/``) — used
295
314
  verbatim, so callers can point at a model this module does not list yet.
296
- 3. A catalogue display name (e.g. ``"GPT OSS 120B"``) or a loose spelling of
297
- an id — matched after normalisation (case / punctuation insensitive).
315
+ 3. A catalogue display name (e.g. ``"DeepSeek V4 Flash 0731"``) or a loose
316
+ spelling of an id — matched after normalisation (case / punctuation
317
+ insensitive).
298
318
 
299
319
  Args:
300
320
  model: The model id, display name, or ``org/name`` repository.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: saia-python
3
- Version: 0.10.0
3
+ Version: 0.10.1
4
4
  Summary: Python wrapper for the GWDG SAIA platform REST API
5
5
  Author: Friedrich Schwarz
6
6
  License-Expression: AGPL-3.0-only
@@ -137,7 +137,7 @@ async def main():
137
137
  async with AsyncSAIAClient() as client:
138
138
  # Non-streaming RAG chat
139
139
  answer = await client.arcana.chat(
140
- model="openai-gpt-oss-120b",
140
+ model="deepseek-v4-flash-0731",
141
141
  messages=[{"role": "user", "content": "Summarise the DLBCL first line."}],
142
142
  arcana_id="owner/kb",
143
143
  )
@@ -98,7 +98,10 @@ class FakeTokenizer:
98
98
 
99
99
 
100
100
  def test_registry_is_well_formed():
101
- assert len(GWDG_MODEL_REPOS) >= 17
101
+ assert len(GWDG_MODEL_REPOS) >= 23
102
+ # no id or display name is listed twice (the dicts would silently collapse)
103
+ assert len(GWDG_MODEL_REPOS) == len(tk._MODEL_TABLE)
104
+ assert len(tk._DISPLAY_NORM_TO_REPO) == len(tk._MODEL_TABLE)
102
105
  # every value is a plausible org/name HF repo
103
106
  for mid, repo in GWDG_MODEL_REPOS.items():
104
107
  assert "/" in repo and not repo.startswith("/"), (mid, repo)
@@ -107,9 +110,14 @@ def test_registry_is_well_formed():
107
110
  @pytest.mark.parametrize(
108
111
  "model,expected",
109
112
  [
113
+ ("deepseek-v4-flash-0731", "deepseek-ai/DeepSeek-V4-Flash-0731"),
114
+ ("glm-5.3-flash", "zai-org/GLM-5.3-Flash"),
115
+ ("qwen3.8-27b", "Qwen/Qwen3.8-27B-FP8"),
116
+ ("qwen3-coder-next", "Qwen/Qwen3-Coder-Next-FP8"),
117
+ ("meta-llama-3.1-8b-instruct", "nvidia/Llama-3.1-8B-Instruct-FP8"),
118
+ # No longer served by GWDG, but kept in the catalogue so old ids resolve.
110
119
  ("openai-gpt-oss-120b", "openai/gpt-oss-120b"),
111
120
  ("qwen3-coder-30b-a3b-instruct", "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8"),
112
- ("meta-llama-3.1-8b-instruct", "nvidia/Llama-3.1-8B-Instruct-FP8"),
113
121
  ],
114
122
  )
115
123
  def test_resolve_repo_by_id(model, expected):
@@ -118,6 +126,9 @@ def test_resolve_repo_by_id(model, expected):
118
126
 
119
127
  def test_resolve_repo_by_display_name_and_passthrough():
120
128
  # Catalogue display name (also the live /models `name` field).
129
+ assert (
130
+ resolve_repo("DeepSeek V4 Flash 0731") == "deepseek-ai/DeepSeek-V4-Flash-0731"
131
+ )
121
132
  assert resolve_repo("GPT OSS 120B") == "openai/gpt-oss-120b"
122
133
  # Full org/name passes through untouched, even if unknown to the registry.
123
134
  assert resolve_repo("some-org/Custom-Tokenizer") == "some-org/Custom-Tokenizer"
File without changes
File without changes