infer-stack 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- infer_stack/__init__.py +2 -0
- infer_stack/backends/__init__.py +7 -0
- infer_stack/backends/compose_renderer.py +243 -0
- infer_stack/backends/kubeai_renderer.py +202 -0
- infer_stack/benchmark.py +38 -0
- infer_stack/catalog.py +438 -0
- infer_stack/cli/__init__.py +169 -0
- infer_stack/cli/__main__.py +4 -0
- infer_stack/cli/commands_profile.py +467 -0
- infer_stack/cli/commands_runtime.py +719 -0
- infer_stack/cli/commands_smoke.py +691 -0
- infer_stack/cli/compose.py +755 -0
- infer_stack/cli/context.py +471 -0
- infer_stack/cli/options.py +134 -0
- infer_stack/cli/probes.py +178 -0
- infer_stack/config.py +450 -0
- infer_stack/contracts.py +223 -0
- infer_stack/diff_prompt.py +117 -0
- infer_stack/docker_utils.py +230 -0
- infer_stack/env_utils.py +97 -0
- infer_stack/experimental/model_catalog_discover.py +1155 -0
- infer_stack/experimental/model_memory_estimator.py +1264 -0
- infer_stack/experimental/stress_test_long_context.py +397 -0
- infer_stack/hardware.py +70 -0
- infer_stack/kubeai_ops.py +76 -0
- infer_stack/paths.py +87 -0
- infer_stack/profile_runtime.py +46 -0
- infer_stack/renderer.py +19 -0
- infer_stack/resolver.py +1092 -0
- infer_stack/templates/default-models.yaml +674 -0
- infer_stack/templates/default-ollama-models.yaml +31 -0
- infer_stack/templates/default-profiles.yaml +1731 -0
- infer_stack/templates/default-vllm-models.yaml +714 -0
- infer_stack/templates/docker-compose.yml.j2 +430 -0
- infer_stack/templates/litellm_config.yaml.j2 +44 -0
- infer_stack/templates/nginx.conf.j2 +84 -0
- infer_stack/tuning.py +3 -0
- infer_stack/validator.py +314 -0
- infer_stack/verification.py +46 -0
- infer_stack-0.6.0.dist-info/METADATA +1034 -0
- infer_stack-0.6.0.dist-info/RECORD +44 -0
- infer_stack-0.6.0.dist-info/WHEEL +5 -0
- infer_stack-0.6.0.dist-info/entry_points.txt +2 -0
- infer_stack-0.6.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,755 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from ..config import load_yaml
|
|
4
|
+
from ..docker_utils import PortInUseError
|
|
5
|
+
from ..docker_utils import check_ports_available
|
|
6
|
+
from ..docker_utils import compose_recreate_router
|
|
7
|
+
from ..docker_utils import compose_up
|
|
8
|
+
from ..docker_utils import our_published_ports
|
|
9
|
+
from ..env_utils import parse_env_file
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any
|
|
12
|
+
import json
|
|
13
|
+
import requests
|
|
14
|
+
import subprocess
|
|
15
|
+
|
|
16
|
+
from .context import (
|
|
17
|
+
generated_dir,
|
|
18
|
+
plan_path,
|
|
19
|
+
runtime_env_path,
|
|
20
|
+
runtime_litellm_config_path,
|
|
21
|
+
)
|
|
22
|
+
from .probes import (
|
|
23
|
+
_default_model_for_deployment,
|
|
24
|
+
_ready_openai_probe,
|
|
25
|
+
_resolve_smoke_protocol_from_deployment,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
# Backend-specific helpers
|
|
29
|
+
# ---------------------------------------------------------------------------
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _compose_base_cmd(cfg: dict[str, Any]) -> list[str]:
|
|
33
|
+
"""Build the shared ``docker compose -f ... --env-file ...`` prefix.
|
|
34
|
+
|
|
35
|
+
Used by every compose-wrapper subcommand so the user doesn't have to
|
|
36
|
+
cd into the rendered-artifacts directory just to run a one-shot
|
|
37
|
+
``ps`` / ``restart`` / ``pull`` / ``logs``.
|
|
38
|
+
"""
|
|
39
|
+
compose_file = generated_dir(cfg) / 'docker-compose.yml'
|
|
40
|
+
env_file = generated_dir(cfg) / '.env'
|
|
41
|
+
return cfg['runtime']['compose_cmd'].split() + [
|
|
42
|
+
'-f',
|
|
43
|
+
str(compose_file),
|
|
44
|
+
'--env-file',
|
|
45
|
+
str(env_file),
|
|
46
|
+
]
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _kubeai_stub(cmd_name: str) -> None:
|
|
50
|
+
"""Raise for a day-2-ops subcommand that has no kubeai implementation yet.
|
|
51
|
+
|
|
52
|
+
These wrappers (logs/ps/restart/pull/start/stop) compose docker-compose
|
|
53
|
+
invocations and have no kubectl equivalent in this CLI. Until somebody
|
|
54
|
+
writes one, surface the gap as ``NotImplementedError`` so callers can
|
|
55
|
+
distinguish "kubeai doesn't do this yet" from a real failure.
|
|
56
|
+
"""
|
|
57
|
+
raise NotImplementedError(
|
|
58
|
+
f'`{cmd_name}` is not implemented for the kubeai backend yet. '
|
|
59
|
+
f'Use the equivalent kubectl command in the meantime '
|
|
60
|
+
f'(e.g. `kubectl -n <namespace> ...`).'
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _compose_up_with_router_recreate(
|
|
65
|
+
cfg: dict[str, Any],
|
|
66
|
+
*,
|
|
67
|
+
detach: bool,
|
|
68
|
+
) -> None:
|
|
69
|
+
"""Run ``compose up`` and refresh LiteLLM's model list to match the new render.
|
|
70
|
+
|
|
71
|
+
A compose ``up`` against the rendered stack only restarts vLLM services
|
|
72
|
+
whose specs changed. LiteLLM and Open WebUI keep their existing
|
|
73
|
+
containers and would therefore serve stale model lists until something
|
|
74
|
+
else refreshed them. Two ways to do that:
|
|
75
|
+
|
|
76
|
+
1. **Live refresh** (preferred): talk to LiteLLM's admin API
|
|
77
|
+
(``/model/new`` / ``/model/delete``) to diff and apply alias changes
|
|
78
|
+
in-process. LiteLLM and Open WebUI stay up — users hitting unchanged
|
|
79
|
+
models see no disruption. Skipped automatically on cold start or if
|
|
80
|
+
the admin API is unreachable.
|
|
81
|
+
2. **Container recreate** (fallback): force-recreate only the LiteLLM
|
|
82
|
+
container so it reloads the rendered YAML on startup. Open WebUI stays
|
|
83
|
+
up; used only when live refresh fails and Compose did not already
|
|
84
|
+
reload LiteLLM during convergence.
|
|
85
|
+
"""
|
|
86
|
+
compose_file = generated_dir(cfg) / 'docker-compose.yml'
|
|
87
|
+
env_file = generated_dir(cfg) / '.env'
|
|
88
|
+
compose_cmd = cfg['runtime']['compose_cmd']
|
|
89
|
+
|
|
90
|
+
_preflight_check_ports(cfg)
|
|
91
|
+
|
|
92
|
+
litellm_in_render = _compose_has_service(compose_file, 'litellm')
|
|
93
|
+
litellm_before = (
|
|
94
|
+
_compose_service_state(compose_cmd, compose_file, env_file, 'litellm')
|
|
95
|
+
if litellm_in_render
|
|
96
|
+
else {}
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
compose_up(
|
|
100
|
+
compose_cmd,
|
|
101
|
+
compose_file,
|
|
102
|
+
env_file,
|
|
103
|
+
detach=detach,
|
|
104
|
+
remove_orphans=True,
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
# If `up` failed it would have already raised; only do the router refresh
|
|
108
|
+
# in detached mode (foreground `up` keeps the user attached to logs and
|
|
109
|
+
# leaves cycling decisions to compose).
|
|
110
|
+
if not detach:
|
|
111
|
+
return
|
|
112
|
+
|
|
113
|
+
if not runtime_litellm_config_path(cfg).exists():
|
|
114
|
+
return
|
|
115
|
+
|
|
116
|
+
if not litellm_in_render:
|
|
117
|
+
# Direct Ollama / raw-server profiles intentionally do not render a
|
|
118
|
+
# LiteLLM service. A stale litellm_config.yaml may still exist in the
|
|
119
|
+
# runtime directory from a previous gateway profile, but that must not
|
|
120
|
+
# trigger a router refresh/recreate against a service that is no longer
|
|
121
|
+
# present in the active compose file.
|
|
122
|
+
return
|
|
123
|
+
|
|
124
|
+
litellm_after = _compose_service_state(
|
|
125
|
+
compose_cmd, compose_file, env_file, 'litellm'
|
|
126
|
+
)
|
|
127
|
+
litellm_reloaded_by_compose = bool(litellm_after) and (
|
|
128
|
+
litellm_after.get('id') != litellm_before.get('id')
|
|
129
|
+
or litellm_after.get('started_at') != litellm_before.get('started_at')
|
|
130
|
+
or litellm_before.get('running') not in {'true', 'True'}
|
|
131
|
+
)
|
|
132
|
+
if litellm_reloaded_by_compose:
|
|
133
|
+
# Compose already created or restarted LiteLLM while converging the
|
|
134
|
+
# stack. The process has read the freshly rendered YAML, so a second
|
|
135
|
+
# live refresh or forced recreate is redundant churn.
|
|
136
|
+
print(
|
|
137
|
+
'LiteLLM was started/reloaded by compose; skipping extra router refresh.'
|
|
138
|
+
)
|
|
139
|
+
return
|
|
140
|
+
|
|
141
|
+
try:
|
|
142
|
+
_litellm_refresh_router_live(cfg)
|
|
143
|
+
return
|
|
144
|
+
except RouterRefreshError as ex:
|
|
145
|
+
print(
|
|
146
|
+
f'Live router refresh skipped ({ex}); '
|
|
147
|
+
'recreating litellm container to reload config from YAML '
|
|
148
|
+
'(open-webui stays up)...'
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
compose_recreate_router(
|
|
152
|
+
compose_cmd,
|
|
153
|
+
compose_file,
|
|
154
|
+
env_file,
|
|
155
|
+
detach=True,
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _compose_has_service(compose_file: Path, service_name: str) -> bool:
|
|
160
|
+
"""Return true when a rendered compose file contains ``service_name``.
|
|
161
|
+
|
|
162
|
+
This is intentionally based on the rendered compose file rather than the
|
|
163
|
+
presence of sidecar artifacts such as ``runtime/litellm_config.yaml``.
|
|
164
|
+
Runtime artifacts are persistent across profile switches, while the compose
|
|
165
|
+
service list is the active source of truth for what ``docker compose up``
|
|
166
|
+
can recreate.
|
|
167
|
+
"""
|
|
168
|
+
try:
|
|
169
|
+
doc = load_yaml(compose_file)
|
|
170
|
+
except FileNotFoundError:
|
|
171
|
+
return False
|
|
172
|
+
services = doc.get('services') or {}
|
|
173
|
+
return service_name in services
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _compose_service_container_id(
|
|
177
|
+
compose_cmd: str,
|
|
178
|
+
compose_file: Path,
|
|
179
|
+
env_file: Path,
|
|
180
|
+
service_name: str,
|
|
181
|
+
) -> str:
|
|
182
|
+
"""Return the current container id for a rendered compose service."""
|
|
183
|
+
return _compose_service_state(
|
|
184
|
+
compose_cmd, compose_file, env_file, service_name
|
|
185
|
+
).get('id', '')
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _compose_service_state(
|
|
189
|
+
compose_cmd: str,
|
|
190
|
+
compose_file: Path,
|
|
191
|
+
env_file: Path,
|
|
192
|
+
service_name: str,
|
|
193
|
+
) -> dict[str, str]:
|
|
194
|
+
"""Return a robust best-effort state snapshot for a compose service.
|
|
195
|
+
|
|
196
|
+
Use JSON ``docker inspect`` rather than a Go template. The template form is
|
|
197
|
+
brittle for containers without a healthcheck: missing ``State.Health`` can
|
|
198
|
+
make ``docker inspect --format`` fail, causing diagnostics to show only an
|
|
199
|
+
id and the convergence logic to misclassify a still-running LiteLLM
|
|
200
|
+
container as newly reloaded.
|
|
201
|
+
|
|
202
|
+
The returned fields are also used to diagnose ``exit code 137`` cases: when
|
|
203
|
+
a container disappears or restarts, ``oom_killed`` / ``exit_code`` make it
|
|
204
|
+
clear whether Docker killed it, Compose recreated it, or the process exited
|
|
205
|
+
normally.
|
|
206
|
+
"""
|
|
207
|
+
if not compose_file.exists():
|
|
208
|
+
return {}
|
|
209
|
+
ps_cmd = compose_cmd.split() + [
|
|
210
|
+
'-f',
|
|
211
|
+
str(compose_file),
|
|
212
|
+
'--env-file',
|
|
213
|
+
str(env_file),
|
|
214
|
+
'ps',
|
|
215
|
+
'-q',
|
|
216
|
+
service_name,
|
|
217
|
+
]
|
|
218
|
+
try:
|
|
219
|
+
proc = subprocess.run(
|
|
220
|
+
ps_cmd, capture_output=True, text=True, check=False, timeout=10
|
|
221
|
+
)
|
|
222
|
+
except (subprocess.SubprocessError, OSError):
|
|
223
|
+
return {}
|
|
224
|
+
if proc.returncode != 0 or not proc.stdout.strip():
|
|
225
|
+
return {}
|
|
226
|
+
container_id = proc.stdout.strip().splitlines()[0]
|
|
227
|
+
inspect_cmd = ['docker', 'inspect', container_id]
|
|
228
|
+
try:
|
|
229
|
+
proc = subprocess.run(
|
|
230
|
+
inspect_cmd, capture_output=True, text=True, check=False, timeout=10
|
|
231
|
+
)
|
|
232
|
+
except (subprocess.SubprocessError, OSError):
|
|
233
|
+
return {'id': container_id}
|
|
234
|
+
if proc.returncode != 0 or not proc.stdout.strip():
|
|
235
|
+
return {'id': container_id}
|
|
236
|
+
try:
|
|
237
|
+
payload = json.loads(proc.stdout)[0]
|
|
238
|
+
except (json.JSONDecodeError, IndexError, TypeError):
|
|
239
|
+
return {'id': container_id}
|
|
240
|
+
|
|
241
|
+
state = payload.get('State') or {}
|
|
242
|
+
health = state.get('Health') or {}
|
|
243
|
+
return {
|
|
244
|
+
'id': payload.get('Id') or container_id,
|
|
245
|
+
'name': str(payload.get('Name') or '').lstrip('/'),
|
|
246
|
+
'running': str(bool(state.get('Running'))).lower(),
|
|
247
|
+
'status': str(state.get('Status') or ''),
|
|
248
|
+
'health': str(health.get('Status') or 'none'),
|
|
249
|
+
'started_at': str(state.get('StartedAt') or ''),
|
|
250
|
+
'finished_at': str(state.get('FinishedAt') or ''),
|
|
251
|
+
'exit_code': str(
|
|
252
|
+
state.get('ExitCode') if state.get('ExitCode') is not None else ''
|
|
253
|
+
),
|
|
254
|
+
'oom_killed': str(bool(state.get('OOMKilled'))).lower(),
|
|
255
|
+
'restart_count': str(
|
|
256
|
+
payload.get('RestartCount')
|
|
257
|
+
if payload.get('RestartCount') is not None
|
|
258
|
+
else ''
|
|
259
|
+
),
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _compose_rendered_service_names(compose_file: Path) -> list[str]:
|
|
264
|
+
"""Return service names from the rendered compose file."""
|
|
265
|
+
try:
|
|
266
|
+
doc = load_yaml(compose_file)
|
|
267
|
+
except FileNotFoundError:
|
|
268
|
+
return []
|
|
269
|
+
return sorted((doc.get('services') or {}).keys())
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def _short_id(value: str) -> str:
|
|
273
|
+
"""Shorten a docker container id for human diagnostics."""
|
|
274
|
+
return value[:12] if value else '-'
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def _http_probe_summary(
|
|
278
|
+
method: str,
|
|
279
|
+
url: str,
|
|
280
|
+
*,
|
|
281
|
+
headers: dict[str, str] | None = None,
|
|
282
|
+
json_body: Any | None = None,
|
|
283
|
+
timeout: float = 8.0,
|
|
284
|
+
) -> str:
|
|
285
|
+
"""Return a concise one-line summary for a diagnostic HTTP probe."""
|
|
286
|
+
try:
|
|
287
|
+
resp = requests.request(
|
|
288
|
+
method, url, headers=headers, json=json_body, timeout=timeout
|
|
289
|
+
)
|
|
290
|
+
except requests.exceptions.RequestException as ex:
|
|
291
|
+
return f'ERR {type(ex).__name__}: {ex}'
|
|
292
|
+
body = (resp.text or '').strip().replace('\n', ' ')
|
|
293
|
+
if len(body) > 220:
|
|
294
|
+
body = body[:220] + '...'
|
|
295
|
+
return (
|
|
296
|
+
f'HTTP {resp.status_code}: {body}'
|
|
297
|
+
if resp.status_code >= 400
|
|
298
|
+
else f'HTTP {resp.status_code}'
|
|
299
|
+
)
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _print_gateway_diagnostics(
|
|
303
|
+
cfg: dict[str, Any],
|
|
304
|
+
deployment: dict[str, Any],
|
|
305
|
+
*,
|
|
306
|
+
model: str | None = None,
|
|
307
|
+
require_generation: bool = False,
|
|
308
|
+
) -> None:
|
|
309
|
+
"""Print targeted probes for the active gateway/provider graph."""
|
|
310
|
+
env = (
|
|
311
|
+
parse_env_file(runtime_env_path(cfg))
|
|
312
|
+
if runtime_env_path(cfg).exists()
|
|
313
|
+
else {}
|
|
314
|
+
)
|
|
315
|
+
ports = cfg.get('ports', {}) or {}
|
|
316
|
+
gateways = deployment.get('gateways', {}) or {}
|
|
317
|
+
providers = deployment.get('providers', {}) or {}
|
|
318
|
+
|
|
319
|
+
litellm = gateways.get('litellm') or {}
|
|
320
|
+
if litellm.get('enabled'):
|
|
321
|
+
litellm_port = ports.get('litellm')
|
|
322
|
+
base = f'http://127.0.0.1:{litellm_port}'
|
|
323
|
+
key = env.get('LITELLM_MASTER_KEY', '')
|
|
324
|
+
headers = {'Authorization': f'Bearer {key}'} if key else {}
|
|
325
|
+
print('\nLiteLLM probes:')
|
|
326
|
+
print(
|
|
327
|
+
f' GET {base}/model/info -> {_http_probe_summary("GET", base + "/model/info", headers=headers)}'
|
|
328
|
+
)
|
|
329
|
+
print(
|
|
330
|
+
f' GET {base}/v1/models -> {_http_probe_summary("GET", base + "/v1/models", headers=headers)}'
|
|
331
|
+
)
|
|
332
|
+
if require_generation:
|
|
333
|
+
probe_model = model or _default_model_for_deployment(deployment)
|
|
334
|
+
protocol = _resolve_smoke_protocol_from_deployment(
|
|
335
|
+
deployment, probe_model
|
|
336
|
+
)
|
|
337
|
+
ok, msg = _ready_openai_probe(
|
|
338
|
+
base_url=f'{base}/v1',
|
|
339
|
+
headers=headers,
|
|
340
|
+
model=probe_model,
|
|
341
|
+
protocol=protocol,
|
|
342
|
+
prompt='Reply with ready.',
|
|
343
|
+
max_tokens=1,
|
|
344
|
+
require_generation=True,
|
|
345
|
+
)
|
|
346
|
+
status = 'OK' if ok else 'WAIT'
|
|
347
|
+
print(
|
|
348
|
+
f' generation probe ({probe_model}, {protocol}) -> {status}: {msg}'
|
|
349
|
+
)
|
|
350
|
+
|
|
351
|
+
vllm = providers.get('vllm') or {}
|
|
352
|
+
runtimes = vllm.get('runtimes') or {}
|
|
353
|
+
if runtimes:
|
|
354
|
+
print('\nvLLM provider probes:')
|
|
355
|
+
for name, rt in runtimes.items():
|
|
356
|
+
service = rt.get('compose_service_name') or f'vllm-{name}'
|
|
357
|
+
host_port = rt.get('host_port') or ports.get('vllm') or 18000
|
|
358
|
+
print(
|
|
359
|
+
f' {service}: model={rt.get("served_model_name")} protocol={rt.get("protocol_mode")} gpu={rt.get("gpu_indices")}'
|
|
360
|
+
)
|
|
361
|
+
if host_port:
|
|
362
|
+
url = f'http://127.0.0.1:{host_port}/health'
|
|
363
|
+
print(f' GET {url} -> {_http_probe_summary("GET", url)}')
|
|
364
|
+
|
|
365
|
+
ollama = providers.get('ollama') or {}
|
|
366
|
+
if ollama.get('enabled') and ollama.get('publish_port'):
|
|
367
|
+
port = ports.get('ollama') or 11434
|
|
368
|
+
base = f'http://127.0.0.1:{port}'
|
|
369
|
+
print('\nOllama probes:')
|
|
370
|
+
print(
|
|
371
|
+
f' GET {base}/api/tags -> {_http_probe_summary("GET", base + "/api/tags")}'
|
|
372
|
+
)
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def _print_compose_diagnostics(cfg: dict[str, Any], *, tail: int = 0) -> None:
|
|
376
|
+
"""Print compose state and optionally recent logs for diagnostic purposes."""
|
|
377
|
+
compose_file = generated_dir(cfg) / 'docker-compose.yml'
|
|
378
|
+
env_file = runtime_env_path(cfg)
|
|
379
|
+
compose_cmd = cfg['runtime']['compose_cmd']
|
|
380
|
+
services = _compose_rendered_service_names(compose_file)
|
|
381
|
+
if not services:
|
|
382
|
+
print(f'No rendered compose services found at {compose_file}')
|
|
383
|
+
return
|
|
384
|
+
print('\nCompose services:')
|
|
385
|
+
for svc in services:
|
|
386
|
+
state = _compose_service_state(compose_cmd, compose_file, env_file, svc)
|
|
387
|
+
if not state:
|
|
388
|
+
print(f' {svc:22s} absent')
|
|
389
|
+
continue
|
|
390
|
+
print(
|
|
391
|
+
f' {svc:22s} id={_short_id(state.get("id", ""))} '
|
|
392
|
+
f'name={state.get("name", "-")} '
|
|
393
|
+
f'running={state.get("running", "-")} status={state.get("status", "-")} '
|
|
394
|
+
f'health={state.get("health", "-")} exit={state.get("exit_code", "-")} '
|
|
395
|
+
f'oom={state.get("oom_killed", "-")} restarts={state.get("restart_count", "-")} '
|
|
396
|
+
f'started_at={state.get("started_at", "-")}'
|
|
397
|
+
)
|
|
398
|
+
if tail:
|
|
399
|
+
log_services = [
|
|
400
|
+
svc
|
|
401
|
+
for svc in services
|
|
402
|
+
if svc == 'litellm'
|
|
403
|
+
or svc == 'open-webui'
|
|
404
|
+
or svc.startswith('vllm-')
|
|
405
|
+
or svc == 'ollama'
|
|
406
|
+
]
|
|
407
|
+
if log_services:
|
|
408
|
+
print(f'\nRecent logs (--tail {tail}):')
|
|
409
|
+
cmd = _compose_base_cmd(cfg) + [
|
|
410
|
+
'logs',
|
|
411
|
+
'--tail',
|
|
412
|
+
str(tail),
|
|
413
|
+
*log_services,
|
|
414
|
+
]
|
|
415
|
+
subprocess.run(cmd, check=False)
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _explain_readiness_message(msg: str) -> str:
|
|
419
|
+
"""Add operator-facing interpretation to common readiness failures."""
|
|
420
|
+
lower = msg.lower()
|
|
421
|
+
if (
|
|
422
|
+
'cannot connect to host litellm' in lower
|
|
423
|
+
or 'name or service not known' in lower
|
|
424
|
+
):
|
|
425
|
+
return (
|
|
426
|
+
msg
|
|
427
|
+
+ '\n hint: the frontend or caller cannot resolve/reach the LiteLLM service. '
|
|
428
|
+
'Run `infer-stack diagnose --logs --tail 80` to check whether the active profile renders LiteLLM and whether the container is running. '
|
|
429
|
+
'If diagnose shows `oom=true` or `exit=137`, Docker killed LiteLLM rather than merely waiting on a vLLM upstream.'
|
|
430
|
+
)
|
|
431
|
+
if (
|
|
432
|
+
'connection error' in lower
|
|
433
|
+
or 'connection refused' in lower
|
|
434
|
+
or 'cannot connect to host vllm' in lower
|
|
435
|
+
):
|
|
436
|
+
return (
|
|
437
|
+
msg
|
|
438
|
+
+ '\n hint: LiteLLM is responding, but the upstream vLLM process is not serving yet. '
|
|
439
|
+
'This is expected while a single vLLM runtime is being replaced; wait-ready will keep polling.'
|
|
440
|
+
)
|
|
441
|
+
return msg
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
class RouterRefreshError(RuntimeError):
|
|
445
|
+
"""Live LiteLLM router refresh did not complete; caller should fall back."""
|
|
446
|
+
|
|
447
|
+
|
|
448
|
+
def _resolve_env_refs(obj: Any, env: dict[str, str]) -> Any:
|
|
449
|
+
"""Recursively replace ``os.environ/VAR`` strings with the value from ``env``.
|
|
450
|
+
|
|
451
|
+
Mirrors LiteLLM's YAML-load substitution so we can feed the admin API
|
|
452
|
+
literal credentials. If a referenced variable isn't in ``env`` the original
|
|
453
|
+
string is left alone — caller will get the upstream error to debug.
|
|
454
|
+
"""
|
|
455
|
+
if isinstance(obj, str):
|
|
456
|
+
if obj.startswith('os.environ/'):
|
|
457
|
+
var = obj.removeprefix('os.environ/')
|
|
458
|
+
return env.get(var, obj)
|
|
459
|
+
return obj
|
|
460
|
+
if isinstance(obj, dict):
|
|
461
|
+
return {k: _resolve_env_refs(v, env) for k, v in obj.items()}
|
|
462
|
+
if isinstance(obj, list):
|
|
463
|
+
return [_resolve_env_refs(v, env) for v in obj]
|
|
464
|
+
return obj
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
def _litellm_delete_missed_config_model(resp: requests.Response) -> bool:
|
|
468
|
+
"""Return true for LiteLLM's config-model delete miss response.
|
|
469
|
+
|
|
470
|
+
``GET /model/info`` returns both models loaded from config.yaml and
|
|
471
|
+
models inserted into LiteLLM's DB via ``/model/new``. In current LiteLLM
|
|
472
|
+
releases, ``POST /model/delete`` only deletes DB-backed models. When the
|
|
473
|
+
reported id belongs to a config-backed model, the delete endpoint returns a
|
|
474
|
+
400/404 response whose body says the model id was not found in the DB.
|
|
475
|
+
That is not a proxy availability failure, so it should not trigger the
|
|
476
|
+
fallback path that restarts the LiteLLM container.
|
|
477
|
+
"""
|
|
478
|
+
try:
|
|
479
|
+
payload = resp.json()
|
|
480
|
+
except ValueError:
|
|
481
|
+
payload = resp.text
|
|
482
|
+
text = str(payload).lower()
|
|
483
|
+
return 'not found' in text and 'db' in text
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
def _litellm_refresh_router_live(cfg: dict[str, Any]) -> None:
|
|
487
|
+
"""Sync the running LiteLLM router's model list to match the rendered YAML.
|
|
488
|
+
|
|
489
|
+
Diffs ``GET /model/info`` (current state in the running container) against
|
|
490
|
+
the rendered ``litellm_config.yaml`` (desired state), then applies
|
|
491
|
+
``POST /model/delete`` and ``POST /model/new`` for the differences. Aliases
|
|
492
|
+
that didn't change keep serving traffic without interruption.
|
|
493
|
+
|
|
494
|
+
Raises ``RouterRefreshError`` on any failure (admin API unreachable,
|
|
495
|
+
auth missing, individual call fails); the caller falls back to the
|
|
496
|
+
full container-recreate path.
|
|
497
|
+
"""
|
|
498
|
+
import yaml as _yaml
|
|
499
|
+
|
|
500
|
+
litellm_port = cfg.get('ports', {}).get('litellm')
|
|
501
|
+
if not litellm_port:
|
|
502
|
+
raise RouterRefreshError('litellm port not configured in cfg')
|
|
503
|
+
base = f'http://127.0.0.1:{litellm_port}'
|
|
504
|
+
|
|
505
|
+
env = parse_env_file(runtime_env_path(cfg))
|
|
506
|
+
master_key = env.get('LITELLM_MASTER_KEY', '').strip()
|
|
507
|
+
if not master_key:
|
|
508
|
+
raise RouterRefreshError(
|
|
509
|
+
'LITELLM_MASTER_KEY missing from rendered .env'
|
|
510
|
+
)
|
|
511
|
+
headers = {
|
|
512
|
+
'Authorization': f'Bearer {master_key}',
|
|
513
|
+
'Content-Type': 'application/json',
|
|
514
|
+
}
|
|
515
|
+
|
|
516
|
+
config_path_ = runtime_litellm_config_path(cfg)
|
|
517
|
+
if not config_path_.exists():
|
|
518
|
+
raise RouterRefreshError(
|
|
519
|
+
f'rendered litellm config not found at {config_path_}'
|
|
520
|
+
)
|
|
521
|
+
try:
|
|
522
|
+
desired_doc = (
|
|
523
|
+
_yaml.safe_load(config_path_.read_text(encoding='utf-8')) or {}
|
|
524
|
+
)
|
|
525
|
+
except _yaml.YAMLError as ex:
|
|
526
|
+
raise RouterRefreshError(
|
|
527
|
+
f'could not parse {config_path_}: {ex}'
|
|
528
|
+
) from ex
|
|
529
|
+
desired_models = desired_doc.get('model_list') or []
|
|
530
|
+
|
|
531
|
+
# The rendered YAML keeps secrets as `os.environ/VAR` references so the
|
|
532
|
+
# file itself isn't sensitive. LiteLLM resolves these only at YAML-load
|
|
533
|
+
# time on container startup — the admin API takes literal values. Inline
|
|
534
|
+
# the actual env values now so /model/new gets a usable upstream.
|
|
535
|
+
desired_models = [_resolve_env_refs(m, env) for m in desired_models]
|
|
536
|
+
|
|
537
|
+
last_ex: requests.exceptions.RequestException | None = None
|
|
538
|
+
resp = None
|
|
539
|
+
max_attempts = 20
|
|
540
|
+
for attempt in range(1, max_attempts + 1):
|
|
541
|
+
try:
|
|
542
|
+
resp = requests.get(
|
|
543
|
+
f'{base}/model/info', headers=headers, timeout=5
|
|
544
|
+
)
|
|
545
|
+
resp.raise_for_status()
|
|
546
|
+
break
|
|
547
|
+
except requests.exceptions.RequestException as ex:
|
|
548
|
+
last_ex = ex
|
|
549
|
+
if attempt < max_attempts:
|
|
550
|
+
import time
|
|
551
|
+
|
|
552
|
+
time.sleep(1.5)
|
|
553
|
+
continue
|
|
554
|
+
raise RouterRefreshError(
|
|
555
|
+
f'GET /model/info failed after {attempt} attempts: {ex}'
|
|
556
|
+
) from ex
|
|
557
|
+
assert resp is not None
|
|
558
|
+
current_models = resp.json().get('data') or []
|
|
559
|
+
|
|
560
|
+
# Key by alias (model_name). Within an alias, the "upstream" identity is
|
|
561
|
+
# litellm_params.model (e.g. "openai/qwen3.5-9b"). If that changes, the
|
|
562
|
+
# alias points to a different service and must be re-added; if it matches,
|
|
563
|
+
# the alias is untouched and continues serving.
|
|
564
|
+
def upstream_of(entry):
|
|
565
|
+
return (entry.get('litellm_params') or {}).get('model')
|
|
566
|
+
|
|
567
|
+
current_by_alias = {m['model_name']: m for m in current_models}
|
|
568
|
+
desired_by_alias = {m['model_name']: m for m in desired_models}
|
|
569
|
+
|
|
570
|
+
to_delete: list[tuple[str, str]] = []
|
|
571
|
+
to_add: list[dict] = []
|
|
572
|
+
|
|
573
|
+
for alias, current in current_by_alias.items():
|
|
574
|
+
desired = desired_by_alias.get(alias)
|
|
575
|
+
if desired is None:
|
|
576
|
+
to_delete.append((alias, current['model_info']['id']))
|
|
577
|
+
elif upstream_of(current) != upstream_of(desired):
|
|
578
|
+
to_delete.append((alias, current['model_info']['id']))
|
|
579
|
+
to_add.append(desired)
|
|
580
|
+
|
|
581
|
+
for alias, desired in desired_by_alias.items():
|
|
582
|
+
if alias not in current_by_alias:
|
|
583
|
+
to_add.append(desired)
|
|
584
|
+
|
|
585
|
+
if not to_delete and not to_add:
|
|
586
|
+
return
|
|
587
|
+
|
|
588
|
+
# Delete-before-add so the same alias can transition to a new upstream
|
|
589
|
+
# without LiteLLM rejecting a duplicate model_name. LiteLLM distinguishes
|
|
590
|
+
# config-file models from DB-backed models: /model/delete only applies to
|
|
591
|
+
# DB-backed rows. When an existing model was loaded from config.yaml,
|
|
592
|
+
# LiteLLM may report it in /model/info but return "not found in db" from
|
|
593
|
+
# /model/delete. Treat that as a non-fatal stale-config alias instead of
|
|
594
|
+
# forcing a LiteLLM container restart; the desired new aliases can still be
|
|
595
|
+
# added live, and a later manual LiteLLM restart will clean up the stale
|
|
596
|
+
# config-backed aliases if the operator cares about /v1/models hygiene.
|
|
597
|
+
stale_config_aliases: set[str] = set()
|
|
598
|
+
deleted_count = 0
|
|
599
|
+
for alias, model_id in to_delete:
|
|
600
|
+
try:
|
|
601
|
+
resp = requests.post(
|
|
602
|
+
f'{base}/model/delete',
|
|
603
|
+
headers=headers,
|
|
604
|
+
json={'id': model_id},
|
|
605
|
+
timeout=5,
|
|
606
|
+
)
|
|
607
|
+
if resp.status_code in {
|
|
608
|
+
400,
|
|
609
|
+
404,
|
|
610
|
+
} and _litellm_delete_missed_config_model(resp):
|
|
611
|
+
stale_config_aliases.add(alias)
|
|
612
|
+
continue
|
|
613
|
+
resp.raise_for_status()
|
|
614
|
+
deleted_count += 1
|
|
615
|
+
except requests.exceptions.RequestException as ex:
|
|
616
|
+
raise RouterRefreshError(
|
|
617
|
+
f'DELETE alias={alias} id={model_id} failed: {ex}'
|
|
618
|
+
) from ex
|
|
619
|
+
|
|
620
|
+
skipped_add_aliases: set[str] = set()
|
|
621
|
+
for model in to_add:
|
|
622
|
+
alias = model.get('model_name', '<unknown>')
|
|
623
|
+
if alias in stale_config_aliases:
|
|
624
|
+
# Same alias, changed upstream, and the old alias is config-backed.
|
|
625
|
+
# Adding would collide and deleting would require a container restart.
|
|
626
|
+
skipped_add_aliases.add(alias)
|
|
627
|
+
continue
|
|
628
|
+
try:
|
|
629
|
+
resp = requests.post(
|
|
630
|
+
f'{base}/model/new',
|
|
631
|
+
headers=headers,
|
|
632
|
+
json=model,
|
|
633
|
+
timeout=5,
|
|
634
|
+
)
|
|
635
|
+
resp.raise_for_status()
|
|
636
|
+
except requests.exceptions.RequestException as ex:
|
|
637
|
+
raise RouterRefreshError(
|
|
638
|
+
f'POST /model/new alias={alias} failed: {ex}'
|
|
639
|
+
) from ex
|
|
640
|
+
|
|
641
|
+
summary_parts = []
|
|
642
|
+
if deleted_count:
|
|
643
|
+
summary_parts.append(f'removed {deleted_count} alias(es)')
|
|
644
|
+
added_count = len(to_add) - len(skipped_add_aliases)
|
|
645
|
+
if added_count:
|
|
646
|
+
summary_parts.append(f'added {added_count} alias(es)')
|
|
647
|
+
if stale_config_aliases:
|
|
648
|
+
summary_parts.append(
|
|
649
|
+
'left '
|
|
650
|
+
f'{len(stale_config_aliases)} stale config-backed alias(es) live '
|
|
651
|
+
'because LiteLLM would not delete them without a restart'
|
|
652
|
+
)
|
|
653
|
+
if skipped_add_aliases:
|
|
654
|
+
summary_parts.append(
|
|
655
|
+
'skipped '
|
|
656
|
+
f'{len(skipped_add_aliases)} same-name update(s); '
|
|
657
|
+
'restart LiteLLM to replace those aliases'
|
|
658
|
+
)
|
|
659
|
+
print(f'Live LiteLLM router refresh: {", ".join(summary_parts)}.')
|
|
660
|
+
|
|
661
|
+
|
|
662
|
+
def _preflight_check_ports(cfg: dict[str, Any]) -> None:
|
|
663
|
+
"""Verify only the host ports the current rendered stack will publish."""
|
|
664
|
+
ports = cfg.get('ports', {})
|
|
665
|
+
candidates: list[tuple[str, int, str]] = []
|
|
666
|
+
deployment: dict[str, Any] = {}
|
|
667
|
+
try:
|
|
668
|
+
if plan_path(cfg).exists():
|
|
669
|
+
deployment = load_yaml(plan_path(cfg)).get('deployment', {})
|
|
670
|
+
except Exception:
|
|
671
|
+
deployment = {}
|
|
672
|
+
|
|
673
|
+
frontends = deployment.get('frontends', {}) or {}
|
|
674
|
+
gateways = deployment.get('gateways', {}) or {}
|
|
675
|
+
providers = deployment.get('providers', {}) or {}
|
|
676
|
+
|
|
677
|
+
if not deployment:
|
|
678
|
+
# Fallback for very old rendered states; keep this conservative.
|
|
679
|
+
if ports.get('litellm'):
|
|
680
|
+
candidates.append(('litellm', int(ports['litellm']), '0.0.0.0'))
|
|
681
|
+
if ports.get('open_webui'):
|
|
682
|
+
candidates.append(
|
|
683
|
+
('open-webui', int(ports['open_webui']), '0.0.0.0')
|
|
684
|
+
)
|
|
685
|
+
else:
|
|
686
|
+
if (gateways.get('litellm') or {}).get('enabled') and ports.get(
|
|
687
|
+
'litellm'
|
|
688
|
+
):
|
|
689
|
+
candidates.append(('litellm', int(ports['litellm']), '0.0.0.0'))
|
|
690
|
+
if (
|
|
691
|
+
(frontends.get('open_webui') or {}).get('enabled')
|
|
692
|
+
and (frontends.get('open_webui') or {}).get('publish_port', True)
|
|
693
|
+
and ports.get('open_webui')
|
|
694
|
+
):
|
|
695
|
+
candidates.append(
|
|
696
|
+
('open-webui', int(ports['open_webui']), '0.0.0.0')
|
|
697
|
+
)
|
|
698
|
+
reverse_proxy = frontends.get('reverse_proxy') or {}
|
|
699
|
+
if reverse_proxy.get('enabled'):
|
|
700
|
+
if reverse_proxy.get('publish_http', True):
|
|
701
|
+
candidates.append(
|
|
702
|
+
(
|
|
703
|
+
'reverse-proxy-http',
|
|
704
|
+
int(
|
|
705
|
+
reverse_proxy.get('http_port')
|
|
706
|
+
or ports.get('reverse_proxy_http')
|
|
707
|
+
or 80
|
|
708
|
+
),
|
|
709
|
+
reverse_proxy.get('http_bind_host') or '0.0.0.0',
|
|
710
|
+
)
|
|
711
|
+
)
|
|
712
|
+
if reverse_proxy.get('publish_https', True):
|
|
713
|
+
candidates.append(
|
|
714
|
+
(
|
|
715
|
+
'reverse-proxy-https',
|
|
716
|
+
int(
|
|
717
|
+
reverse_proxy.get('https_port')
|
|
718
|
+
or ports.get('reverse_proxy_https')
|
|
719
|
+
or 443
|
|
720
|
+
),
|
|
721
|
+
reverse_proxy.get('https_bind_host') or '0.0.0.0',
|
|
722
|
+
)
|
|
723
|
+
)
|
|
724
|
+
ollama = providers.get('ollama') or {}
|
|
725
|
+
if (
|
|
726
|
+
ollama.get('enabled')
|
|
727
|
+
and ollama.get('publish_port')
|
|
728
|
+
and ports.get('ollama')
|
|
729
|
+
):
|
|
730
|
+
candidates.append(('ollama', int(ports['ollama']), '127.0.0.1'))
|
|
731
|
+
for name, rt in (
|
|
732
|
+
(providers.get('vllm') or {}).get('runtimes') or {}
|
|
733
|
+
).items():
|
|
734
|
+
if rt.get('publish_port'):
|
|
735
|
+
candidates.append(
|
|
736
|
+
(
|
|
737
|
+
f'vllm-{name}',
|
|
738
|
+
int(rt.get('host_port') or 18000),
|
|
739
|
+
'127.0.0.1',
|
|
740
|
+
)
|
|
741
|
+
)
|
|
742
|
+
|
|
743
|
+
owned = our_published_ports(
|
|
744
|
+
cfg['runtime']['compose_cmd'],
|
|
745
|
+
generated_dir(cfg) / 'docker-compose.yml',
|
|
746
|
+
runtime_env_path(cfg),
|
|
747
|
+
)
|
|
748
|
+
to_check = [
|
|
749
|
+
(svc, port, host) for svc, port, host in candidates if port not in owned
|
|
750
|
+
]
|
|
751
|
+
|
|
752
|
+
try:
|
|
753
|
+
check_ports_available(to_check)
|
|
754
|
+
except PortInUseError as ex:
|
|
755
|
+
raise SystemExit(str(ex)) from ex
|