infer-stack 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. infer_stack/__init__.py +2 -0
  2. infer_stack/backends/__init__.py +7 -0
  3. infer_stack/backends/compose_renderer.py +243 -0
  4. infer_stack/backends/kubeai_renderer.py +202 -0
  5. infer_stack/benchmark.py +38 -0
  6. infer_stack/catalog.py +438 -0
  7. infer_stack/cli/__init__.py +169 -0
  8. infer_stack/cli/__main__.py +4 -0
  9. infer_stack/cli/commands_profile.py +467 -0
  10. infer_stack/cli/commands_runtime.py +719 -0
  11. infer_stack/cli/commands_smoke.py +691 -0
  12. infer_stack/cli/compose.py +755 -0
  13. infer_stack/cli/context.py +471 -0
  14. infer_stack/cli/options.py +134 -0
  15. infer_stack/cli/probes.py +178 -0
  16. infer_stack/config.py +450 -0
  17. infer_stack/contracts.py +223 -0
  18. infer_stack/diff_prompt.py +117 -0
  19. infer_stack/docker_utils.py +230 -0
  20. infer_stack/env_utils.py +97 -0
  21. infer_stack/experimental/model_catalog_discover.py +1155 -0
  22. infer_stack/experimental/model_memory_estimator.py +1264 -0
  23. infer_stack/experimental/stress_test_long_context.py +397 -0
  24. infer_stack/hardware.py +70 -0
  25. infer_stack/kubeai_ops.py +76 -0
  26. infer_stack/paths.py +87 -0
  27. infer_stack/profile_runtime.py +46 -0
  28. infer_stack/renderer.py +19 -0
  29. infer_stack/resolver.py +1092 -0
  30. infer_stack/templates/default-models.yaml +674 -0
  31. infer_stack/templates/default-ollama-models.yaml +31 -0
  32. infer_stack/templates/default-profiles.yaml +1731 -0
  33. infer_stack/templates/default-vllm-models.yaml +714 -0
  34. infer_stack/templates/docker-compose.yml.j2 +430 -0
  35. infer_stack/templates/litellm_config.yaml.j2 +44 -0
  36. infer_stack/templates/nginx.conf.j2 +84 -0
  37. infer_stack/tuning.py +3 -0
  38. infer_stack/validator.py +314 -0
  39. infer_stack/verification.py +46 -0
  40. infer_stack-0.6.0.dist-info/METADATA +1034 -0
  41. infer_stack-0.6.0.dist-info/RECORD +44 -0
  42. infer_stack-0.6.0.dist-info/WHEEL +5 -0
  43. infer_stack-0.6.0.dist-info/entry_points.txt +2 -0
  44. infer_stack-0.6.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,755 @@
1
+ from __future__ import annotations
2
+
3
+ from ..config import load_yaml
4
+ from ..docker_utils import PortInUseError
5
+ from ..docker_utils import check_ports_available
6
+ from ..docker_utils import compose_recreate_router
7
+ from ..docker_utils import compose_up
8
+ from ..docker_utils import our_published_ports
9
+ from ..env_utils import parse_env_file
10
+ from pathlib import Path
11
+ from typing import Any
12
+ import json
13
+ import requests
14
+ import subprocess
15
+
16
+ from .context import (
17
+ generated_dir,
18
+ plan_path,
19
+ runtime_env_path,
20
+ runtime_litellm_config_path,
21
+ )
22
+ from .probes import (
23
+ _default_model_for_deployment,
24
+ _ready_openai_probe,
25
+ _resolve_smoke_protocol_from_deployment,
26
+ )
27
+
28
+ # Backend-specific helpers
29
+ # ---------------------------------------------------------------------------
30
+
31
+
32
+ def _compose_base_cmd(cfg: dict[str, Any]) -> list[str]:
33
+ """Build the shared ``docker compose -f ... --env-file ...`` prefix.
34
+
35
+ Used by every compose-wrapper subcommand so the user doesn't have to
36
+ cd into the rendered-artifacts directory just to run a one-shot
37
+ ``ps`` / ``restart`` / ``pull`` / ``logs``.
38
+ """
39
+ compose_file = generated_dir(cfg) / 'docker-compose.yml'
40
+ env_file = generated_dir(cfg) / '.env'
41
+ return cfg['runtime']['compose_cmd'].split() + [
42
+ '-f',
43
+ str(compose_file),
44
+ '--env-file',
45
+ str(env_file),
46
+ ]
47
+
48
+
49
+ def _kubeai_stub(cmd_name: str) -> None:
50
+ """Raise for a day-2-ops subcommand that has no kubeai implementation yet.
51
+
52
+ These wrappers (logs/ps/restart/pull/start/stop) compose docker-compose
53
+ invocations and have no kubectl equivalent in this CLI. Until somebody
54
+ writes one, surface the gap as ``NotImplementedError`` so callers can
55
+ distinguish "kubeai doesn't do this yet" from a real failure.
56
+ """
57
+ raise NotImplementedError(
58
+ f'`{cmd_name}` is not implemented for the kubeai backend yet. '
59
+ f'Use the equivalent kubectl command in the meantime '
60
+ f'(e.g. `kubectl -n <namespace> ...`).'
61
+ )
62
+
63
+
64
+ def _compose_up_with_router_recreate(
65
+ cfg: dict[str, Any],
66
+ *,
67
+ detach: bool,
68
+ ) -> None:
69
+ """Run ``compose up`` and refresh LiteLLM's model list to match the new render.
70
+
71
+ A compose ``up`` against the rendered stack only restarts vLLM services
72
+ whose specs changed. LiteLLM and Open WebUI keep their existing
73
+ containers and would therefore serve stale model lists until something
74
+ else refreshed them. Two ways to do that:
75
+
76
+ 1. **Live refresh** (preferred): talk to LiteLLM's admin API
77
+ (``/model/new`` / ``/model/delete``) to diff and apply alias changes
78
+ in-process. LiteLLM and Open WebUI stay up — users hitting unchanged
79
+ models see no disruption. Skipped automatically on cold start or if
80
+ the admin API is unreachable.
81
+ 2. **Container recreate** (fallback): force-recreate only the LiteLLM
82
+ container so it reloads the rendered YAML on startup. Open WebUI stays
83
+ up; used only when live refresh fails and Compose did not already
84
+ reload LiteLLM during convergence.
85
+ """
86
+ compose_file = generated_dir(cfg) / 'docker-compose.yml'
87
+ env_file = generated_dir(cfg) / '.env'
88
+ compose_cmd = cfg['runtime']['compose_cmd']
89
+
90
+ _preflight_check_ports(cfg)
91
+
92
+ litellm_in_render = _compose_has_service(compose_file, 'litellm')
93
+ litellm_before = (
94
+ _compose_service_state(compose_cmd, compose_file, env_file, 'litellm')
95
+ if litellm_in_render
96
+ else {}
97
+ )
98
+
99
+ compose_up(
100
+ compose_cmd,
101
+ compose_file,
102
+ env_file,
103
+ detach=detach,
104
+ remove_orphans=True,
105
+ )
106
+
107
+ # If `up` failed it would have already raised; only do the router refresh
108
+ # in detached mode (foreground `up` keeps the user attached to logs and
109
+ # leaves cycling decisions to compose).
110
+ if not detach:
111
+ return
112
+
113
+ if not runtime_litellm_config_path(cfg).exists():
114
+ return
115
+
116
+ if not litellm_in_render:
117
+ # Direct Ollama / raw-server profiles intentionally do not render a
118
+ # LiteLLM service. A stale litellm_config.yaml may still exist in the
119
+ # runtime directory from a previous gateway profile, but that must not
120
+ # trigger a router refresh/recreate against a service that is no longer
121
+ # present in the active compose file.
122
+ return
123
+
124
+ litellm_after = _compose_service_state(
125
+ compose_cmd, compose_file, env_file, 'litellm'
126
+ )
127
+ litellm_reloaded_by_compose = bool(litellm_after) and (
128
+ litellm_after.get('id') != litellm_before.get('id')
129
+ or litellm_after.get('started_at') != litellm_before.get('started_at')
130
+ or litellm_before.get('running') not in {'true', 'True'}
131
+ )
132
+ if litellm_reloaded_by_compose:
133
+ # Compose already created or restarted LiteLLM while converging the
134
+ # stack. The process has read the freshly rendered YAML, so a second
135
+ # live refresh or forced recreate is redundant churn.
136
+ print(
137
+ 'LiteLLM was started/reloaded by compose; skipping extra router refresh.'
138
+ )
139
+ return
140
+
141
+ try:
142
+ _litellm_refresh_router_live(cfg)
143
+ return
144
+ except RouterRefreshError as ex:
145
+ print(
146
+ f'Live router refresh skipped ({ex}); '
147
+ 'recreating litellm container to reload config from YAML '
148
+ '(open-webui stays up)...'
149
+ )
150
+
151
+ compose_recreate_router(
152
+ compose_cmd,
153
+ compose_file,
154
+ env_file,
155
+ detach=True,
156
+ )
157
+
158
+
159
+ def _compose_has_service(compose_file: Path, service_name: str) -> bool:
160
+ """Return true when a rendered compose file contains ``service_name``.
161
+
162
+ This is intentionally based on the rendered compose file rather than the
163
+ presence of sidecar artifacts such as ``runtime/litellm_config.yaml``.
164
+ Runtime artifacts are persistent across profile switches, while the compose
165
+ service list is the active source of truth for what ``docker compose up``
166
+ can recreate.
167
+ """
168
+ try:
169
+ doc = load_yaml(compose_file)
170
+ except FileNotFoundError:
171
+ return False
172
+ services = doc.get('services') or {}
173
+ return service_name in services
174
+
175
+
176
+ def _compose_service_container_id(
177
+ compose_cmd: str,
178
+ compose_file: Path,
179
+ env_file: Path,
180
+ service_name: str,
181
+ ) -> str:
182
+ """Return the current container id for a rendered compose service."""
183
+ return _compose_service_state(
184
+ compose_cmd, compose_file, env_file, service_name
185
+ ).get('id', '')
186
+
187
+
188
+ def _compose_service_state(
189
+ compose_cmd: str,
190
+ compose_file: Path,
191
+ env_file: Path,
192
+ service_name: str,
193
+ ) -> dict[str, str]:
194
+ """Return a robust best-effort state snapshot for a compose service.
195
+
196
+ Use JSON ``docker inspect`` rather than a Go template. The template form is
197
+ brittle for containers without a healthcheck: missing ``State.Health`` can
198
+ make ``docker inspect --format`` fail, causing diagnostics to show only an
199
+ id and the convergence logic to misclassify a still-running LiteLLM
200
+ container as newly reloaded.
201
+
202
+ The returned fields are also used to diagnose ``exit code 137`` cases: when
203
+ a container disappears or restarts, ``oom_killed`` / ``exit_code`` make it
204
+ clear whether Docker killed it, Compose recreated it, or the process exited
205
+ normally.
206
+ """
207
+ if not compose_file.exists():
208
+ return {}
209
+ ps_cmd = compose_cmd.split() + [
210
+ '-f',
211
+ str(compose_file),
212
+ '--env-file',
213
+ str(env_file),
214
+ 'ps',
215
+ '-q',
216
+ service_name,
217
+ ]
218
+ try:
219
+ proc = subprocess.run(
220
+ ps_cmd, capture_output=True, text=True, check=False, timeout=10
221
+ )
222
+ except (subprocess.SubprocessError, OSError):
223
+ return {}
224
+ if proc.returncode != 0 or not proc.stdout.strip():
225
+ return {}
226
+ container_id = proc.stdout.strip().splitlines()[0]
227
+ inspect_cmd = ['docker', 'inspect', container_id]
228
+ try:
229
+ proc = subprocess.run(
230
+ inspect_cmd, capture_output=True, text=True, check=False, timeout=10
231
+ )
232
+ except (subprocess.SubprocessError, OSError):
233
+ return {'id': container_id}
234
+ if proc.returncode != 0 or not proc.stdout.strip():
235
+ return {'id': container_id}
236
+ try:
237
+ payload = json.loads(proc.stdout)[0]
238
+ except (json.JSONDecodeError, IndexError, TypeError):
239
+ return {'id': container_id}
240
+
241
+ state = payload.get('State') or {}
242
+ health = state.get('Health') or {}
243
+ return {
244
+ 'id': payload.get('Id') or container_id,
245
+ 'name': str(payload.get('Name') or '').lstrip('/'),
246
+ 'running': str(bool(state.get('Running'))).lower(),
247
+ 'status': str(state.get('Status') or ''),
248
+ 'health': str(health.get('Status') or 'none'),
249
+ 'started_at': str(state.get('StartedAt') or ''),
250
+ 'finished_at': str(state.get('FinishedAt') or ''),
251
+ 'exit_code': str(
252
+ state.get('ExitCode') if state.get('ExitCode') is not None else ''
253
+ ),
254
+ 'oom_killed': str(bool(state.get('OOMKilled'))).lower(),
255
+ 'restart_count': str(
256
+ payload.get('RestartCount')
257
+ if payload.get('RestartCount') is not None
258
+ else ''
259
+ ),
260
+ }
261
+
262
+
263
+ def _compose_rendered_service_names(compose_file: Path) -> list[str]:
264
+ """Return service names from the rendered compose file."""
265
+ try:
266
+ doc = load_yaml(compose_file)
267
+ except FileNotFoundError:
268
+ return []
269
+ return sorted((doc.get('services') or {}).keys())
270
+
271
+
272
+ def _short_id(value: str) -> str:
273
+ """Shorten a docker container id for human diagnostics."""
274
+ return value[:12] if value else '-'
275
+
276
+
277
+ def _http_probe_summary(
278
+ method: str,
279
+ url: str,
280
+ *,
281
+ headers: dict[str, str] | None = None,
282
+ json_body: Any | None = None,
283
+ timeout: float = 8.0,
284
+ ) -> str:
285
+ """Return a concise one-line summary for a diagnostic HTTP probe."""
286
+ try:
287
+ resp = requests.request(
288
+ method, url, headers=headers, json=json_body, timeout=timeout
289
+ )
290
+ except requests.exceptions.RequestException as ex:
291
+ return f'ERR {type(ex).__name__}: {ex}'
292
+ body = (resp.text or '').strip().replace('\n', ' ')
293
+ if len(body) > 220:
294
+ body = body[:220] + '...'
295
+ return (
296
+ f'HTTP {resp.status_code}: {body}'
297
+ if resp.status_code >= 400
298
+ else f'HTTP {resp.status_code}'
299
+ )
300
+
301
+
302
+ def _print_gateway_diagnostics(
303
+ cfg: dict[str, Any],
304
+ deployment: dict[str, Any],
305
+ *,
306
+ model: str | None = None,
307
+ require_generation: bool = False,
308
+ ) -> None:
309
+ """Print targeted probes for the active gateway/provider graph."""
310
+ env = (
311
+ parse_env_file(runtime_env_path(cfg))
312
+ if runtime_env_path(cfg).exists()
313
+ else {}
314
+ )
315
+ ports = cfg.get('ports', {}) or {}
316
+ gateways = deployment.get('gateways', {}) or {}
317
+ providers = deployment.get('providers', {}) or {}
318
+
319
+ litellm = gateways.get('litellm') or {}
320
+ if litellm.get('enabled'):
321
+ litellm_port = ports.get('litellm')
322
+ base = f'http://127.0.0.1:{litellm_port}'
323
+ key = env.get('LITELLM_MASTER_KEY', '')
324
+ headers = {'Authorization': f'Bearer {key}'} if key else {}
325
+ print('\nLiteLLM probes:')
326
+ print(
327
+ f' GET {base}/model/info -> {_http_probe_summary("GET", base + "/model/info", headers=headers)}'
328
+ )
329
+ print(
330
+ f' GET {base}/v1/models -> {_http_probe_summary("GET", base + "/v1/models", headers=headers)}'
331
+ )
332
+ if require_generation:
333
+ probe_model = model or _default_model_for_deployment(deployment)
334
+ protocol = _resolve_smoke_protocol_from_deployment(
335
+ deployment, probe_model
336
+ )
337
+ ok, msg = _ready_openai_probe(
338
+ base_url=f'{base}/v1',
339
+ headers=headers,
340
+ model=probe_model,
341
+ protocol=protocol,
342
+ prompt='Reply with ready.',
343
+ max_tokens=1,
344
+ require_generation=True,
345
+ )
346
+ status = 'OK' if ok else 'WAIT'
347
+ print(
348
+ f' generation probe ({probe_model}, {protocol}) -> {status}: {msg}'
349
+ )
350
+
351
+ vllm = providers.get('vllm') or {}
352
+ runtimes = vllm.get('runtimes') or {}
353
+ if runtimes:
354
+ print('\nvLLM provider probes:')
355
+ for name, rt in runtimes.items():
356
+ service = rt.get('compose_service_name') or f'vllm-{name}'
357
+ host_port = rt.get('host_port') or ports.get('vllm') or 18000
358
+ print(
359
+ f' {service}: model={rt.get("served_model_name")} protocol={rt.get("protocol_mode")} gpu={rt.get("gpu_indices")}'
360
+ )
361
+ if host_port:
362
+ url = f'http://127.0.0.1:{host_port}/health'
363
+ print(f' GET {url} -> {_http_probe_summary("GET", url)}')
364
+
365
+ ollama = providers.get('ollama') or {}
366
+ if ollama.get('enabled') and ollama.get('publish_port'):
367
+ port = ports.get('ollama') or 11434
368
+ base = f'http://127.0.0.1:{port}'
369
+ print('\nOllama probes:')
370
+ print(
371
+ f' GET {base}/api/tags -> {_http_probe_summary("GET", base + "/api/tags")}'
372
+ )
373
+
374
+
375
+ def _print_compose_diagnostics(cfg: dict[str, Any], *, tail: int = 0) -> None:
376
+ """Print compose state and optionally recent logs for diagnostic purposes."""
377
+ compose_file = generated_dir(cfg) / 'docker-compose.yml'
378
+ env_file = runtime_env_path(cfg)
379
+ compose_cmd = cfg['runtime']['compose_cmd']
380
+ services = _compose_rendered_service_names(compose_file)
381
+ if not services:
382
+ print(f'No rendered compose services found at {compose_file}')
383
+ return
384
+ print('\nCompose services:')
385
+ for svc in services:
386
+ state = _compose_service_state(compose_cmd, compose_file, env_file, svc)
387
+ if not state:
388
+ print(f' {svc:22s} absent')
389
+ continue
390
+ print(
391
+ f' {svc:22s} id={_short_id(state.get("id", ""))} '
392
+ f'name={state.get("name", "-")} '
393
+ f'running={state.get("running", "-")} status={state.get("status", "-")} '
394
+ f'health={state.get("health", "-")} exit={state.get("exit_code", "-")} '
395
+ f'oom={state.get("oom_killed", "-")} restarts={state.get("restart_count", "-")} '
396
+ f'started_at={state.get("started_at", "-")}'
397
+ )
398
+ if tail:
399
+ log_services = [
400
+ svc
401
+ for svc in services
402
+ if svc == 'litellm'
403
+ or svc == 'open-webui'
404
+ or svc.startswith('vllm-')
405
+ or svc == 'ollama'
406
+ ]
407
+ if log_services:
408
+ print(f'\nRecent logs (--tail {tail}):')
409
+ cmd = _compose_base_cmd(cfg) + [
410
+ 'logs',
411
+ '--tail',
412
+ str(tail),
413
+ *log_services,
414
+ ]
415
+ subprocess.run(cmd, check=False)
416
+
417
+
418
+ def _explain_readiness_message(msg: str) -> str:
419
+ """Add operator-facing interpretation to common readiness failures."""
420
+ lower = msg.lower()
421
+ if (
422
+ 'cannot connect to host litellm' in lower
423
+ or 'name or service not known' in lower
424
+ ):
425
+ return (
426
+ msg
427
+ + '\n hint: the frontend or caller cannot resolve/reach the LiteLLM service. '
428
+ 'Run `infer-stack diagnose --logs --tail 80` to check whether the active profile renders LiteLLM and whether the container is running. '
429
+ 'If diagnose shows `oom=true` or `exit=137`, Docker killed LiteLLM rather than merely waiting on a vLLM upstream.'
430
+ )
431
+ if (
432
+ 'connection error' in lower
433
+ or 'connection refused' in lower
434
+ or 'cannot connect to host vllm' in lower
435
+ ):
436
+ return (
437
+ msg
438
+ + '\n hint: LiteLLM is responding, but the upstream vLLM process is not serving yet. '
439
+ 'This is expected while a single vLLM runtime is being replaced; wait-ready will keep polling.'
440
+ )
441
+ return msg
442
+
443
+
444
+ class RouterRefreshError(RuntimeError):
445
+ """Live LiteLLM router refresh did not complete; caller should fall back."""
446
+
447
+
448
+ def _resolve_env_refs(obj: Any, env: dict[str, str]) -> Any:
449
+ """Recursively replace ``os.environ/VAR`` strings with the value from ``env``.
450
+
451
+ Mirrors LiteLLM's YAML-load substitution so we can feed the admin API
452
+ literal credentials. If a referenced variable isn't in ``env`` the original
453
+ string is left alone — caller will get the upstream error to debug.
454
+ """
455
+ if isinstance(obj, str):
456
+ if obj.startswith('os.environ/'):
457
+ var = obj.removeprefix('os.environ/')
458
+ return env.get(var, obj)
459
+ return obj
460
+ if isinstance(obj, dict):
461
+ return {k: _resolve_env_refs(v, env) for k, v in obj.items()}
462
+ if isinstance(obj, list):
463
+ return [_resolve_env_refs(v, env) for v in obj]
464
+ return obj
465
+
466
+
467
+ def _litellm_delete_missed_config_model(resp: requests.Response) -> bool:
468
+ """Return true for LiteLLM's config-model delete miss response.
469
+
470
+ ``GET /model/info`` returns both models loaded from config.yaml and
471
+ models inserted into LiteLLM's DB via ``/model/new``. In current LiteLLM
472
+ releases, ``POST /model/delete`` only deletes DB-backed models. When the
473
+ reported id belongs to a config-backed model, the delete endpoint returns a
474
+ 400/404 response whose body says the model id was not found in the DB.
475
+ That is not a proxy availability failure, so it should not trigger the
476
+ fallback path that restarts the LiteLLM container.
477
+ """
478
+ try:
479
+ payload = resp.json()
480
+ except ValueError:
481
+ payload = resp.text
482
+ text = str(payload).lower()
483
+ return 'not found' in text and 'db' in text
484
+
485
+
486
+ def _litellm_refresh_router_live(cfg: dict[str, Any]) -> None:
487
+ """Sync the running LiteLLM router's model list to match the rendered YAML.
488
+
489
+ Diffs ``GET /model/info`` (current state in the running container) against
490
+ the rendered ``litellm_config.yaml`` (desired state), then applies
491
+ ``POST /model/delete`` and ``POST /model/new`` for the differences. Aliases
492
+ that didn't change keep serving traffic without interruption.
493
+
494
+ Raises ``RouterRefreshError`` on any failure (admin API unreachable,
495
+ auth missing, individual call fails); the caller falls back to the
496
+ full container-recreate path.
497
+ """
498
+ import yaml as _yaml
499
+
500
+ litellm_port = cfg.get('ports', {}).get('litellm')
501
+ if not litellm_port:
502
+ raise RouterRefreshError('litellm port not configured in cfg')
503
+ base = f'http://127.0.0.1:{litellm_port}'
504
+
505
+ env = parse_env_file(runtime_env_path(cfg))
506
+ master_key = env.get('LITELLM_MASTER_KEY', '').strip()
507
+ if not master_key:
508
+ raise RouterRefreshError(
509
+ 'LITELLM_MASTER_KEY missing from rendered .env'
510
+ )
511
+ headers = {
512
+ 'Authorization': f'Bearer {master_key}',
513
+ 'Content-Type': 'application/json',
514
+ }
515
+
516
+ config_path_ = runtime_litellm_config_path(cfg)
517
+ if not config_path_.exists():
518
+ raise RouterRefreshError(
519
+ f'rendered litellm config not found at {config_path_}'
520
+ )
521
+ try:
522
+ desired_doc = (
523
+ _yaml.safe_load(config_path_.read_text(encoding='utf-8')) or {}
524
+ )
525
+ except _yaml.YAMLError as ex:
526
+ raise RouterRefreshError(
527
+ f'could not parse {config_path_}: {ex}'
528
+ ) from ex
529
+ desired_models = desired_doc.get('model_list') or []
530
+
531
+ # The rendered YAML keeps secrets as `os.environ/VAR` references so the
532
+ # file itself isn't sensitive. LiteLLM resolves these only at YAML-load
533
+ # time on container startup — the admin API takes literal values. Inline
534
+ # the actual env values now so /model/new gets a usable upstream.
535
+ desired_models = [_resolve_env_refs(m, env) for m in desired_models]
536
+
537
+ last_ex: requests.exceptions.RequestException | None = None
538
+ resp = None
539
+ max_attempts = 20
540
+ for attempt in range(1, max_attempts + 1):
541
+ try:
542
+ resp = requests.get(
543
+ f'{base}/model/info', headers=headers, timeout=5
544
+ )
545
+ resp.raise_for_status()
546
+ break
547
+ except requests.exceptions.RequestException as ex:
548
+ last_ex = ex
549
+ if attempt < max_attempts:
550
+ import time
551
+
552
+ time.sleep(1.5)
553
+ continue
554
+ raise RouterRefreshError(
555
+ f'GET /model/info failed after {attempt} attempts: {ex}'
556
+ ) from ex
557
+ assert resp is not None
558
+ current_models = resp.json().get('data') or []
559
+
560
+ # Key by alias (model_name). Within an alias, the "upstream" identity is
561
+ # litellm_params.model (e.g. "openai/qwen3.5-9b"). If that changes, the
562
+ # alias points to a different service and must be re-added; if it matches,
563
+ # the alias is untouched and continues serving.
564
+ def upstream_of(entry):
565
+ return (entry.get('litellm_params') or {}).get('model')
566
+
567
+ current_by_alias = {m['model_name']: m for m in current_models}
568
+ desired_by_alias = {m['model_name']: m for m in desired_models}
569
+
570
+ to_delete: list[tuple[str, str]] = []
571
+ to_add: list[dict] = []
572
+
573
+ for alias, current in current_by_alias.items():
574
+ desired = desired_by_alias.get(alias)
575
+ if desired is None:
576
+ to_delete.append((alias, current['model_info']['id']))
577
+ elif upstream_of(current) != upstream_of(desired):
578
+ to_delete.append((alias, current['model_info']['id']))
579
+ to_add.append(desired)
580
+
581
+ for alias, desired in desired_by_alias.items():
582
+ if alias not in current_by_alias:
583
+ to_add.append(desired)
584
+
585
+ if not to_delete and not to_add:
586
+ return
587
+
588
+ # Delete-before-add so the same alias can transition to a new upstream
589
+ # without LiteLLM rejecting a duplicate model_name. LiteLLM distinguishes
590
+ # config-file models from DB-backed models: /model/delete only applies to
591
+ # DB-backed rows. When an existing model was loaded from config.yaml,
592
+ # LiteLLM may report it in /model/info but return "not found in db" from
593
+ # /model/delete. Treat that as a non-fatal stale-config alias instead of
594
+ # forcing a LiteLLM container restart; the desired new aliases can still be
595
+ # added live, and a later manual LiteLLM restart will clean up the stale
596
+ # config-backed aliases if the operator cares about /v1/models hygiene.
597
+ stale_config_aliases: set[str] = set()
598
+ deleted_count = 0
599
+ for alias, model_id in to_delete:
600
+ try:
601
+ resp = requests.post(
602
+ f'{base}/model/delete',
603
+ headers=headers,
604
+ json={'id': model_id},
605
+ timeout=5,
606
+ )
607
+ if resp.status_code in {
608
+ 400,
609
+ 404,
610
+ } and _litellm_delete_missed_config_model(resp):
611
+ stale_config_aliases.add(alias)
612
+ continue
613
+ resp.raise_for_status()
614
+ deleted_count += 1
615
+ except requests.exceptions.RequestException as ex:
616
+ raise RouterRefreshError(
617
+ f'DELETE alias={alias} id={model_id} failed: {ex}'
618
+ ) from ex
619
+
620
+ skipped_add_aliases: set[str] = set()
621
+ for model in to_add:
622
+ alias = model.get('model_name', '<unknown>')
623
+ if alias in stale_config_aliases:
624
+ # Same alias, changed upstream, and the old alias is config-backed.
625
+ # Adding would collide and deleting would require a container restart.
626
+ skipped_add_aliases.add(alias)
627
+ continue
628
+ try:
629
+ resp = requests.post(
630
+ f'{base}/model/new',
631
+ headers=headers,
632
+ json=model,
633
+ timeout=5,
634
+ )
635
+ resp.raise_for_status()
636
+ except requests.exceptions.RequestException as ex:
637
+ raise RouterRefreshError(
638
+ f'POST /model/new alias={alias} failed: {ex}'
639
+ ) from ex
640
+
641
+ summary_parts = []
642
+ if deleted_count:
643
+ summary_parts.append(f'removed {deleted_count} alias(es)')
644
+ added_count = len(to_add) - len(skipped_add_aliases)
645
+ if added_count:
646
+ summary_parts.append(f'added {added_count} alias(es)')
647
+ if stale_config_aliases:
648
+ summary_parts.append(
649
+ 'left '
650
+ f'{len(stale_config_aliases)} stale config-backed alias(es) live '
651
+ 'because LiteLLM would not delete them without a restart'
652
+ )
653
+ if skipped_add_aliases:
654
+ summary_parts.append(
655
+ 'skipped '
656
+ f'{len(skipped_add_aliases)} same-name update(s); '
657
+ 'restart LiteLLM to replace those aliases'
658
+ )
659
+ print(f'Live LiteLLM router refresh: {", ".join(summary_parts)}.')
660
+
661
+
662
+ def _preflight_check_ports(cfg: dict[str, Any]) -> None:
663
+ """Verify only the host ports the current rendered stack will publish."""
664
+ ports = cfg.get('ports', {})
665
+ candidates: list[tuple[str, int, str]] = []
666
+ deployment: dict[str, Any] = {}
667
+ try:
668
+ if plan_path(cfg).exists():
669
+ deployment = load_yaml(plan_path(cfg)).get('deployment', {})
670
+ except Exception:
671
+ deployment = {}
672
+
673
+ frontends = deployment.get('frontends', {}) or {}
674
+ gateways = deployment.get('gateways', {}) or {}
675
+ providers = deployment.get('providers', {}) or {}
676
+
677
+ if not deployment:
678
+ # Fallback for very old rendered states; keep this conservative.
679
+ if ports.get('litellm'):
680
+ candidates.append(('litellm', int(ports['litellm']), '0.0.0.0'))
681
+ if ports.get('open_webui'):
682
+ candidates.append(
683
+ ('open-webui', int(ports['open_webui']), '0.0.0.0')
684
+ )
685
+ else:
686
+ if (gateways.get('litellm') or {}).get('enabled') and ports.get(
687
+ 'litellm'
688
+ ):
689
+ candidates.append(('litellm', int(ports['litellm']), '0.0.0.0'))
690
+ if (
691
+ (frontends.get('open_webui') or {}).get('enabled')
692
+ and (frontends.get('open_webui') or {}).get('publish_port', True)
693
+ and ports.get('open_webui')
694
+ ):
695
+ candidates.append(
696
+ ('open-webui', int(ports['open_webui']), '0.0.0.0')
697
+ )
698
+ reverse_proxy = frontends.get('reverse_proxy') or {}
699
+ if reverse_proxy.get('enabled'):
700
+ if reverse_proxy.get('publish_http', True):
701
+ candidates.append(
702
+ (
703
+ 'reverse-proxy-http',
704
+ int(
705
+ reverse_proxy.get('http_port')
706
+ or ports.get('reverse_proxy_http')
707
+ or 80
708
+ ),
709
+ reverse_proxy.get('http_bind_host') or '0.0.0.0',
710
+ )
711
+ )
712
+ if reverse_proxy.get('publish_https', True):
713
+ candidates.append(
714
+ (
715
+ 'reverse-proxy-https',
716
+ int(
717
+ reverse_proxy.get('https_port')
718
+ or ports.get('reverse_proxy_https')
719
+ or 443
720
+ ),
721
+ reverse_proxy.get('https_bind_host') or '0.0.0.0',
722
+ )
723
+ )
724
+ ollama = providers.get('ollama') or {}
725
+ if (
726
+ ollama.get('enabled')
727
+ and ollama.get('publish_port')
728
+ and ports.get('ollama')
729
+ ):
730
+ candidates.append(('ollama', int(ports['ollama']), '127.0.0.1'))
731
+ for name, rt in (
732
+ (providers.get('vllm') or {}).get('runtimes') or {}
733
+ ).items():
734
+ if rt.get('publish_port'):
735
+ candidates.append(
736
+ (
737
+ f'vllm-{name}',
738
+ int(rt.get('host_port') or 18000),
739
+ '127.0.0.1',
740
+ )
741
+ )
742
+
743
+ owned = our_published_ports(
744
+ cfg['runtime']['compose_cmd'],
745
+ generated_dir(cfg) / 'docker-compose.yml',
746
+ runtime_env_path(cfg),
747
+ )
748
+ to_check = [
749
+ (svc, port, host) for svc, port, host in candidates if port not in owned
750
+ ]
751
+
752
+ try:
753
+ check_ports_available(to_check)
754
+ except PortInUseError as ex:
755
+ raise SystemExit(str(ex)) from ex