infer-stack 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- infer_stack/__init__.py +2 -0
- infer_stack/backends/__init__.py +7 -0
- infer_stack/backends/compose_renderer.py +243 -0
- infer_stack/backends/kubeai_renderer.py +202 -0
- infer_stack/benchmark.py +38 -0
- infer_stack/catalog.py +438 -0
- infer_stack/cli/__init__.py +169 -0
- infer_stack/cli/__main__.py +4 -0
- infer_stack/cli/commands_profile.py +467 -0
- infer_stack/cli/commands_runtime.py +719 -0
- infer_stack/cli/commands_smoke.py +691 -0
- infer_stack/cli/compose.py +755 -0
- infer_stack/cli/context.py +471 -0
- infer_stack/cli/options.py +134 -0
- infer_stack/cli/probes.py +178 -0
- infer_stack/config.py +450 -0
- infer_stack/contracts.py +223 -0
- infer_stack/diff_prompt.py +117 -0
- infer_stack/docker_utils.py +230 -0
- infer_stack/env_utils.py +97 -0
- infer_stack/experimental/model_catalog_discover.py +1155 -0
- infer_stack/experimental/model_memory_estimator.py +1264 -0
- infer_stack/experimental/stress_test_long_context.py +397 -0
- infer_stack/hardware.py +70 -0
- infer_stack/kubeai_ops.py +76 -0
- infer_stack/paths.py +87 -0
- infer_stack/profile_runtime.py +46 -0
- infer_stack/renderer.py +19 -0
- infer_stack/resolver.py +1092 -0
- infer_stack/templates/default-models.yaml +674 -0
- infer_stack/templates/default-ollama-models.yaml +31 -0
- infer_stack/templates/default-profiles.yaml +1731 -0
- infer_stack/templates/default-vllm-models.yaml +714 -0
- infer_stack/templates/docker-compose.yml.j2 +430 -0
- infer_stack/templates/litellm_config.yaml.j2 +44 -0
- infer_stack/templates/nginx.conf.j2 +84 -0
- infer_stack/tuning.py +3 -0
- infer_stack/validator.py +314 -0
- infer_stack/verification.py +46 -0
- infer_stack-0.6.0.dist-info/METADATA +1034 -0
- infer_stack-0.6.0.dist-info/RECORD +44 -0
- infer_stack-0.6.0.dist-info/WHEEL +5 -0
- infer_stack-0.6.0.dist-info/entry_points.txt +2 -0
- infer_stack-0.6.0.dist-info/top_level.txt +1 -0
infer_stack/validator.py
ADDED
|
@@ -0,0 +1,314 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def _host_path_exists(raw_path: str, generated_dir: str | None) -> bool:
|
|
8
|
+
"""Mirror how Docker Compose resolves a bind-mount source path.
|
|
9
|
+
|
|
10
|
+
Relative paths in the rendered ``docker-compose.yml`` resolve against the
|
|
11
|
+
directory that holds the compose file (the generated dir), not the CWD.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
if not raw_path:
|
|
15
|
+
return False
|
|
16
|
+
p = Path(raw_path)
|
|
17
|
+
if not p.is_absolute() and generated_dir:
|
|
18
|
+
p = Path(generated_dir) / p
|
|
19
|
+
return p.exists()
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def validate_resolved(resolved: dict[str, Any]) -> dict[str, Any]:
|
|
23
|
+
inventory = resolved.get('inventory', {})
|
|
24
|
+
gpu_map = {g['index']: g for g in inventory.get('gpus', [])}
|
|
25
|
+
policy = resolved.get('policy', {})
|
|
26
|
+
backend = resolved.get('backend', 'compose')
|
|
27
|
+
errors: list[str] = []
|
|
28
|
+
warnings: list[str] = []
|
|
29
|
+
used_ports: set[int] = set()
|
|
30
|
+
|
|
31
|
+
providers = resolved.get('providers', {}) or {}
|
|
32
|
+
gateways = resolved.get('gateways', {}) or {}
|
|
33
|
+
frontends = resolved.get('frontends', {}) or {}
|
|
34
|
+
ollama = providers.get('ollama', {}) or {}
|
|
35
|
+
vllm = providers.get('vllm', {}) or {}
|
|
36
|
+
vllm_runtimes = vllm.get('runtimes', {}) or {}
|
|
37
|
+
litellm = gateways.get('litellm', {}) or {}
|
|
38
|
+
open_webui = frontends.get('open_webui', {}) or {}
|
|
39
|
+
reverse_proxy = frontends.get('reverse_proxy', {}) or {}
|
|
40
|
+
routes = (litellm.get('routes') or {}) if litellm.get('enabled') else {}
|
|
41
|
+
|
|
42
|
+
if backend == 'kubeai':
|
|
43
|
+
if ollama.get('enabled'):
|
|
44
|
+
errors.append(
|
|
45
|
+
'backend=kubeai does not support the Ollama provider yet'
|
|
46
|
+
)
|
|
47
|
+
if litellm.get('enabled'):
|
|
48
|
+
errors.append(
|
|
49
|
+
'backend=kubeai does not render the LiteLLM gateway yet'
|
|
50
|
+
)
|
|
51
|
+
if open_webui.get('enabled'):
|
|
52
|
+
errors.append(
|
|
53
|
+
'backend=kubeai does not render the Open WebUI frontend yet'
|
|
54
|
+
)
|
|
55
|
+
if reverse_proxy.get('enabled'):
|
|
56
|
+
errors.append(
|
|
57
|
+
'backend=kubeai does not render the Compose reverse proxy frontend'
|
|
58
|
+
)
|
|
59
|
+
profiles = resolved.get('resource_profiles', {})
|
|
60
|
+
if not profiles:
|
|
61
|
+
source = resolved.get(
|
|
62
|
+
'resource_profiles_source', 'kubeai-values.local.yaml'
|
|
63
|
+
)
|
|
64
|
+
errors.append(
|
|
65
|
+
'No local KubeAI resource profiles were loaded for validation. '
|
|
66
|
+
f'Expected them at {source}. '
|
|
67
|
+
'Run `python manage.py kubeai-sync-resource-profiles --from-file values-kubeai-local-gpu.yaml` first.'
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
if (
|
|
71
|
+
not ollama.get('enabled')
|
|
72
|
+
and not vllm_runtimes
|
|
73
|
+
and not open_webui.get('enabled')
|
|
74
|
+
and not litellm.get('enabled')
|
|
75
|
+
):
|
|
76
|
+
warnings.append(
|
|
77
|
+
'resolved profile has no enabled providers, gateways, or frontends'
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
if litellm.get('enabled'):
|
|
81
|
+
route_providers = {route.get('provider') for route in routes.values()}
|
|
82
|
+
if not routes:
|
|
83
|
+
warnings.append('LiteLLM is enabled but has no routes')
|
|
84
|
+
if 'ollama' in route_providers and not ollama.get('enabled'):
|
|
85
|
+
errors.append(
|
|
86
|
+
'LiteLLM has Ollama routes but the Ollama provider is disabled'
|
|
87
|
+
)
|
|
88
|
+
if 'vllm' in route_providers and not vllm_runtimes:
|
|
89
|
+
errors.append(
|
|
90
|
+
'LiteLLM has vLLM routes but no vLLM runtimes are enabled'
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
# Mixed direct backends with Open WebUI and no LiteLLM are intentionally not
|
|
94
|
+
# supported in this first graph rewrite.
|
|
95
|
+
if (
|
|
96
|
+
open_webui.get('enabled')
|
|
97
|
+
and not litellm.get('enabled')
|
|
98
|
+
and ollama.get('enabled')
|
|
99
|
+
and vllm_runtimes
|
|
100
|
+
):
|
|
101
|
+
if open_webui.get('provider') != 'ollama':
|
|
102
|
+
errors.append(
|
|
103
|
+
'Open WebUI with mixed Ollama+vLLM direct providers requires LiteLLM'
|
|
104
|
+
)
|
|
105
|
+
else:
|
|
106
|
+
warnings.append(
|
|
107
|
+
'Open WebUI is connected only to Ollama; vLLM runtimes are raw/direct endpoints'
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
if open_webui.get('enabled'):
|
|
111
|
+
provider = open_webui.get('provider')
|
|
112
|
+
if provider == 'litellm' and not litellm.get('enabled'):
|
|
113
|
+
errors.append('Open WebUI provider=litellm but LiteLLM is disabled')
|
|
114
|
+
if provider == 'ollama' and not ollama.get('enabled'):
|
|
115
|
+
errors.append('Open WebUI provider=ollama but Ollama is disabled')
|
|
116
|
+
|
|
117
|
+
if reverse_proxy.get('enabled'):
|
|
118
|
+
generated_dir = (resolved.get('output', {}) or {}).get('generated_dir')
|
|
119
|
+
target = reverse_proxy.get('target')
|
|
120
|
+
if target in {'open_webui', 'open-webui'} and not open_webui.get(
|
|
121
|
+
'enabled'
|
|
122
|
+
):
|
|
123
|
+
errors.append(
|
|
124
|
+
'reverse_proxy target=open_webui but Open WebUI is disabled'
|
|
125
|
+
)
|
|
126
|
+
if target == 'litellm' and not litellm.get('enabled'):
|
|
127
|
+
errors.append(
|
|
128
|
+
'reverse_proxy target=litellm but LiteLLM is disabled'
|
|
129
|
+
)
|
|
130
|
+
if target == 'ollama' and not ollama.get('enabled'):
|
|
131
|
+
errors.append('reverse_proxy target=ollama but Ollama is disabled')
|
|
132
|
+
|
|
133
|
+
ssl = reverse_proxy.get('ssl', {}) or {}
|
|
134
|
+
ssl_enabled = ssl.get('enabled', True)
|
|
135
|
+
config_path = reverse_proxy.get('config_path') or ''
|
|
136
|
+
|
|
137
|
+
# Publishing :443 with no TLS server block means nginx never listens on
|
|
138
|
+
# 443 and clients get connection refused. This only happens when a user
|
|
139
|
+
# explicitly overrides the publish_https default.
|
|
140
|
+
if reverse_proxy.get('publish_https', True) and not ssl_enabled:
|
|
141
|
+
errors.append(
|
|
142
|
+
'reverse_proxy publishes HTTPS (:443) but ssl.enabled is false; '
|
|
143
|
+
'nginx will not listen on 443. Set ssl.enabled: true or publish_https: false.'
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
if config_path:
|
|
147
|
+
# A manual nginx config bypasses the typed SSL renderer entirely, so
|
|
148
|
+
# certificate fields below do not apply; just sanity-check the file.
|
|
149
|
+
if not _host_path_exists(config_path, generated_dir):
|
|
150
|
+
warnings.append(
|
|
151
|
+
f'reverse_proxy.config_path {config_path!r} was not found '
|
|
152
|
+
'(Docker would create an empty directory mount and nginx would fail to start)'
|
|
153
|
+
)
|
|
154
|
+
elif ssl_enabled:
|
|
155
|
+
for field in ('certificate', 'certificate_key'):
|
|
156
|
+
value = ssl.get(field) or ''
|
|
157
|
+
if not value:
|
|
158
|
+
warnings.append(
|
|
159
|
+
f'reverse_proxy SSL is enabled but ssl.{field} is empty'
|
|
160
|
+
)
|
|
161
|
+
elif not _host_path_exists(value, generated_dir):
|
|
162
|
+
warnings.append(
|
|
163
|
+
f'reverse_proxy ssl.{field} {value!r} was not found '
|
|
164
|
+
'(relative paths resolve against the generated compose directory)'
|
|
165
|
+
)
|
|
166
|
+
dhparam = ssl.get('dhparam') or ''
|
|
167
|
+
if dhparam and not _host_path_exists(dhparam, generated_dir):
|
|
168
|
+
warnings.append(
|
|
169
|
+
f'reverse_proxy ssl.dhparam {dhparam!r} was not found '
|
|
170
|
+
'(relative paths resolve against the generated compose directory)'
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
if ollama.get('enabled'):
|
|
174
|
+
if ollama.get('placement_error'):
|
|
175
|
+
errors.append(
|
|
176
|
+
f'ollama placement failed: {ollama["placement_error"]}'
|
|
177
|
+
)
|
|
178
|
+
for idx in ollama.get('gpu_indices', []) or []:
|
|
179
|
+
if idx not in gpu_map:
|
|
180
|
+
errors.append(f'ollama references missing gpu index {idx}')
|
|
181
|
+
elif gpu_map[idx].get('display_active'):
|
|
182
|
+
warnings.append(f'ollama uses display-active GPU {idx}')
|
|
183
|
+
if ollama.get('publish_port'):
|
|
184
|
+
port = int(resolved.get('ports', {}).get('ollama', 11434))
|
|
185
|
+
if port in used_ports:
|
|
186
|
+
errors.append(f'duplicate host port assignment: {port}')
|
|
187
|
+
used_ports.add(port)
|
|
188
|
+
|
|
189
|
+
seen_service_names: set[str] = set()
|
|
190
|
+
for svc in vllm_runtimes.values():
|
|
191
|
+
if svc['service_name'] in seen_service_names:
|
|
192
|
+
errors.append(f'duplicate service name: {svc["service_name"]}')
|
|
193
|
+
seen_service_names.add(svc['service_name'])
|
|
194
|
+
|
|
195
|
+
if backend == 'kubeai':
|
|
196
|
+
resource_profile = str(svc.get('resource_profile', '')).strip()
|
|
197
|
+
if not resource_profile:
|
|
198
|
+
errors.append(
|
|
199
|
+
f'vLLM runtime {svc["runtime_name"]} is missing resource_profile for kubeai backend'
|
|
200
|
+
)
|
|
201
|
+
else:
|
|
202
|
+
profile_name = resource_profile.split(':', 1)[0]
|
|
203
|
+
if profile_name not in resolved.get('resource_profiles', {}):
|
|
204
|
+
errors.append(
|
|
205
|
+
f'vLLM runtime {svc["runtime_name"]} references unknown resource profile {profile_name!r}'
|
|
206
|
+
)
|
|
207
|
+
|
|
208
|
+
if not svc.get('hf_model_id'):
|
|
209
|
+
errors.append(
|
|
210
|
+
f'vLLM runtime {svc["runtime_name"]} is missing hf_model_id'
|
|
211
|
+
)
|
|
212
|
+
if not svc.get('served_model_name'):
|
|
213
|
+
errors.append(
|
|
214
|
+
f'vLLM runtime {svc["runtime_name"]} is missing served_model_name'
|
|
215
|
+
)
|
|
216
|
+
protocol_mode = svc.get('protocol_mode')
|
|
217
|
+
supported = list(svc.get('supported_protocols') or [])
|
|
218
|
+
if supported and protocol_mode not in supported:
|
|
219
|
+
errors.append(
|
|
220
|
+
f'vLLM runtime {svc["runtime_name"]} requests protocol_mode={protocol_mode}, '
|
|
221
|
+
f'but model {svc.get("model_ref")} supports only {supported}.'
|
|
222
|
+
)
|
|
223
|
+
if svc.get('placement_error'):
|
|
224
|
+
errors.append(
|
|
225
|
+
f'vLLM runtime {svc["runtime_name"]} placement failed: {svc["placement_error"]}'
|
|
226
|
+
)
|
|
227
|
+
gpu_indices = svc.get('gpu_indices', [])
|
|
228
|
+
if not gpu_indices:
|
|
229
|
+
warnings.append(
|
|
230
|
+
f'vLLM runtime {svc["runtime_name"]} has no concrete GPU assignment in the rendered plan'
|
|
231
|
+
)
|
|
232
|
+
continue
|
|
233
|
+
if svc.get('tensor_parallel_size', 1) > len(gpu_indices):
|
|
234
|
+
errors.append(
|
|
235
|
+
f'vLLM runtime {svc["runtime_name"]} has tensor_parallel_size larger than assigned GPU count'
|
|
236
|
+
)
|
|
237
|
+
tp = max(1, int(svc.get('tensor_parallel_size', 1)))
|
|
238
|
+
per_gpu_need = float(svc.get('min_vram_gib_per_replica', 0)) / tp
|
|
239
|
+
headroom = float(policy.get('minimum_vram_headroom_gib', 0))
|
|
240
|
+
for idx in gpu_indices:
|
|
241
|
+
if idx not in gpu_map:
|
|
242
|
+
errors.append(
|
|
243
|
+
f'vLLM runtime {svc["runtime_name"]} references missing gpu index {idx}'
|
|
244
|
+
)
|
|
245
|
+
continue
|
|
246
|
+
gpu = gpu_map[idx]
|
|
247
|
+
if (
|
|
248
|
+
policy.get('reserve_display_gpu') == 'auto'
|
|
249
|
+
and gpu.get('display_active')
|
|
250
|
+
and policy.get('forbid_reserved_gpu_use')
|
|
251
|
+
):
|
|
252
|
+
errors.append(
|
|
253
|
+
f'vLLM runtime {svc["runtime_name"]} uses display-active GPU {idx}'
|
|
254
|
+
)
|
|
255
|
+
elif gpu.get('display_active'):
|
|
256
|
+
warnings.append(
|
|
257
|
+
f'vLLM runtime {svc["runtime_name"]} uses display-active GPU {idx}'
|
|
258
|
+
)
|
|
259
|
+
if gpu.get('memory_gib', 0) < (per_gpu_need + headroom):
|
|
260
|
+
errors.append(
|
|
261
|
+
f'vLLM runtime {svc["runtime_name"]} estimates {per_gpu_need} GiB + {headroom} GiB headroom on GPU {idx}, '
|
|
262
|
+
f'but only {gpu.get("memory_gib")} GiB is available'
|
|
263
|
+
)
|
|
264
|
+
if len(gpu_indices) > 1 and policy.get(
|
|
265
|
+
'require_homogeneous_multi_gpu_groups'
|
|
266
|
+
):
|
|
267
|
+
names = {
|
|
268
|
+
gpu_map[idx]['name'] for idx in gpu_indices if idx in gpu_map
|
|
269
|
+
}
|
|
270
|
+
mems = {
|
|
271
|
+
gpu_map[idx]['memory_gib']
|
|
272
|
+
for idx in gpu_indices
|
|
273
|
+
if idx in gpu_map
|
|
274
|
+
}
|
|
275
|
+
if len(names) > 1 or len(mems) > 1:
|
|
276
|
+
errors.append(
|
|
277
|
+
f'vLLM runtime {svc["runtime_name"]} uses a heterogeneous multi-GPU group'
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
if backend == 'compose' and litellm.get('enabled'):
|
|
281
|
+
port = int(resolved.get('ports', {}).get('litellm', 14042))
|
|
282
|
+
if port in used_ports:
|
|
283
|
+
errors.append(f'duplicate host port assignment: {port}')
|
|
284
|
+
used_ports.add(port)
|
|
285
|
+
|
|
286
|
+
if (
|
|
287
|
+
backend == 'compose'
|
|
288
|
+
and open_webui.get('enabled')
|
|
289
|
+
and open_webui.get('publish_port', True)
|
|
290
|
+
):
|
|
291
|
+
port = int(resolved.get('ports', {}).get('open_webui', 13000))
|
|
292
|
+
if port in used_ports:
|
|
293
|
+
errors.append(f'duplicate host port assignment: {port}')
|
|
294
|
+
used_ports.add(port)
|
|
295
|
+
|
|
296
|
+
if backend == 'compose' and reverse_proxy.get('enabled'):
|
|
297
|
+
if reverse_proxy.get('publish_http', True):
|
|
298
|
+
port = int(
|
|
299
|
+
reverse_proxy.get('http_port')
|
|
300
|
+
or resolved.get('ports', {}).get('reverse_proxy_http', 80)
|
|
301
|
+
)
|
|
302
|
+
if port in used_ports:
|
|
303
|
+
errors.append(f'duplicate host port assignment: {port}')
|
|
304
|
+
used_ports.add(port)
|
|
305
|
+
if reverse_proxy.get('publish_https', True):
|
|
306
|
+
port = int(
|
|
307
|
+
reverse_proxy.get('https_port')
|
|
308
|
+
or resolved.get('ports', {}).get('reverse_proxy_https', 443)
|
|
309
|
+
)
|
|
310
|
+
if port in used_ports:
|
|
311
|
+
errors.append(f'duplicate host port assignment: {port}')
|
|
312
|
+
used_ports.add(port)
|
|
313
|
+
|
|
314
|
+
return {'ok': not errors, 'errors': errors, 'warnings': warnings}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from .config import KUBEAI_GENERATED_SUBDIR, normalized_output
|
|
7
|
+
from .profile_runtime import default_base_url
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def verify_profile(deployment: dict[str, Any]) -> dict[str, Any]:
|
|
11
|
+
service = (
|
|
12
|
+
deployment.get('services', [])[0] if deployment.get('services') else {}
|
|
13
|
+
)
|
|
14
|
+
output_root = Path(
|
|
15
|
+
normalized_output(deployment.get('output'))['generated_dir']
|
|
16
|
+
)
|
|
17
|
+
expected = {
|
|
18
|
+
'public_name': deployment['serving_profile']['public_name'],
|
|
19
|
+
'logical_model_name': service.get('logical_model_name', ''),
|
|
20
|
+
'protocol_mode': service.get(
|
|
21
|
+
'protocol_mode',
|
|
22
|
+
deployment['serving_profile'].get('protocol_mode', ''),
|
|
23
|
+
),
|
|
24
|
+
'served_model_name': service.get(
|
|
25
|
+
'served_model_name',
|
|
26
|
+
deployment['serving_profile'].get('served_model_name', ''),
|
|
27
|
+
),
|
|
28
|
+
'endpoint_base_url': default_base_url(deployment),
|
|
29
|
+
'generated_artifacts': {
|
|
30
|
+
'compose': str(output_root / 'docker-compose.yml'),
|
|
31
|
+
'kubeai_models': str(
|
|
32
|
+
output_root / KUBEAI_GENERATED_SUBDIR / 'models.yaml'
|
|
33
|
+
),
|
|
34
|
+
},
|
|
35
|
+
}
|
|
36
|
+
checks = {
|
|
37
|
+
'has_services': bool(deployment.get('services')),
|
|
38
|
+
'has_public_name': bool(expected['public_name']),
|
|
39
|
+
'has_logical_model_name': bool(expected['logical_model_name']),
|
|
40
|
+
'has_protocol_mode': bool(expected['protocol_mode']),
|
|
41
|
+
}
|
|
42
|
+
return {
|
|
43
|
+
'ok': all(checks.values()),
|
|
44
|
+
'profile': expected,
|
|
45
|
+
'checks': checks,
|
|
46
|
+
}
|