infer-stack 0.7.0__tar.gz → 0.7.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- infer_stack-0.7.2/PKG-INFO +780 -0
- infer_stack-0.7.2/README.md +744 -0
- infer_stack-0.7.2/infer_stack/__init__.py +2 -0
- infer_stack-0.7.2/infer_stack/backends/__init__.py +3 -0
- infer_stack-0.7.2/infer_stack/backends/kubeai.py +1464 -0
- infer_stack-0.7.2/infer_stack/backends/kubeai_gateway.py +380 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/cli/__init__.py +83 -7
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/cli/commands_catalog.py +315 -154
- infer_stack-0.7.2/infer_stack/cli/commands_kube.py +668 -0
- infer_stack-0.7.2/infer_stack/cli/commands_kube_worker.py +195 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/cli/commands_leasing.py +1372 -438
- infer_stack-0.7.2/infer_stack/cli/commands_ledger.py +80 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/cli/commands_meta.py +92 -17
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/cli/commands_mock.py +12 -12
- infer_stack-0.7.2/infer_stack/cli/commands_runtime.py +957 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/cli/context.py +1 -1
- infer_stack-0.7.2/infer_stack/cli/options.py +112 -0
- infer_stack-0.7.2/infer_stack/cli_equivalent.py +75 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/config.py +2 -1
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/env_utils.py +31 -2
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/experimental/model_catalog_discover.py +30 -22
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/gpu_doctor.py +83 -17
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/hardware.py +72 -31
- infer_stack-0.7.2/infer_stack/kube/__init__.py +13 -0
- infer_stack-0.7.2/infer_stack/kube/inspect.py +227 -0
- infer_stack-0.7.2/infer_stack/kube/k3s.py +343 -0
- infer_stack-0.7.2/infer_stack/kube/manage.py +817 -0
- infer_stack-0.7.2/infer_stack/kube/monitor.py +139 -0
- infer_stack-0.7.2/infer_stack/kube/operations.py +155 -0
- infer_stack-0.7.2/infer_stack/kube/worker.py +543 -0
- infer_stack-0.7.2/infer_stack/kubeai_ops.py +86 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/leasing/__init__.py +10 -1
- infer_stack-0.7.2/infer_stack/leasing/backend.py +916 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/leasing/catalog.py +233 -38
- infer_stack-0.7.2/infer_stack/leasing/catalog_edit.py +95 -0
- infer_stack-0.7.2/infer_stack/leasing/compose.py +2603 -0
- infer_stack-0.7.2/infer_stack/leasing/controller.py +2516 -0
- infer_stack-0.7.2/infer_stack/leasing/diagnosis.py +193 -0
- infer_stack-0.7.2/infer_stack/leasing/endpoints.py +277 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/leasing/envfile.py +37 -16
- infer_stack-0.7.2/infer_stack/leasing/gateway.py +1529 -0
- infer_stack-0.7.2/infer_stack/leasing/instances.py +321 -0
- infer_stack-0.7.2/infer_stack/leasing/launch.py +238 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/leasing/ledger.py +280 -59
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/leasing/models.py +46 -33
- infer_stack-0.7.2/infer_stack/leasing/naming.py +109 -0
- infer_stack-0.7.2/infer_stack/leasing/network.py +138 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/leasing/placement.py +88 -2
- infer_stack-0.7.2/infer_stack/leasing/profile.py +443 -0
- infer_stack-0.7.2/infer_stack/leasing/residency.py +502 -0
- infer_stack-0.7.2/infer_stack/leasing/routes.py +213 -0
- infer_stack-0.7.2/infer_stack/leasing/store.py +867 -0
- infer_stack-0.7.2/infer_stack/leasing/suggest.py +660 -0
- infer_stack-0.7.2/infer_stack/leasing/transition.py +76 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/leasing/vram.py +51 -13
- infer_stack-0.7.2/infer_stack/log_filter.py +399 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/mockserver/server.py +5 -2
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/profile_runtime.py +5 -3
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/templates/suggestion-pool.yaml +204 -4
- infer_stack-0.7.2/infer_stack/tui.py +3979 -0
- infer_stack-0.7.2/infer_stack.egg-info/PKG-INFO +780 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack.egg-info/SOURCES.txt +54 -3
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack.egg-info/requires.txt +1 -1
- {infer_stack-0.7.0 → infer_stack-0.7.2}/pyproject.toml +1 -4
- infer_stack-0.7.2/tests/test_backend_transition.py +201 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_cli_catalog.py +147 -0
- infer_stack-0.7.2/tests/test_cli_equivalent.py +85 -0
- infer_stack-0.7.2/tests/test_cli_kube.py +598 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_cli_leasing.py +418 -13
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_cli_meta.py +25 -4
- infer_stack-0.7.2/tests/test_day2.py +267 -0
- infer_stack-0.7.2/tests/test_external_endpoints.py +791 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_gpu_doctor.py +86 -2
- infer_stack-0.7.2/tests/test_hardware.py +42 -0
- infer_stack-0.7.2/tests/test_kube_readiness.py +470 -0
- infer_stack-0.7.2/tests/test_kube_review.py +336 -0
- infer_stack-0.7.2/tests/test_kube_worker.py +540 -0
- infer_stack-0.7.2/tests/test_leasing_admission.py +559 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_catalog.py +133 -16
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_compose.py +385 -58
- infer_stack-0.7.2/tests/test_leasing_context_metadata.py +450 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_controller.py +74 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_controller_lock.py +13 -4
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_controller_queue.py +23 -28
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_dynamic_routing.py +296 -34
- infer_stack-0.7.2/tests/test_leasing_image_pull.py +118 -0
- infer_stack-0.7.2/tests/test_leasing_kubeai.py +1240 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_ledger.py +20 -0
- infer_stack-0.7.2/tests/test_leasing_make_room.py +120 -0
- infer_stack-0.7.2/tests/test_leasing_network.py +229 -0
- infer_stack-0.7.2/tests/test_leasing_observability.py +72 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_placement.py +35 -3
- infer_stack-0.7.2/tests/test_leasing_placement_admission.py +81 -0
- infer_stack-0.7.2/tests/test_leasing_profile.py +681 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_reservation.py +3 -22
- infer_stack-0.7.2/tests/test_leasing_residency.py +268 -0
- infer_stack-0.7.2/tests/test_leasing_route_registry.py +360 -0
- infer_stack-0.7.2/tests/test_leasing_secrets.py +367 -0
- infer_stack-0.7.2/tests/test_leasing_selective_apply.py +212 -0
- infer_stack-0.7.2/tests/test_leasing_serialised_publication.py +822 -0
- infer_stack-0.7.2/tests/test_leasing_startup_failure.py +263 -0
- infer_stack-0.7.2/tests/test_leasing_store_threads.py +45 -0
- infer_stack-0.7.2/tests/test_leasing_suggest.py +399 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_vram.py +38 -0
- infer_stack-0.7.2/tests/test_log_filter.py +312 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_mock_vllm_serve.py +2 -2
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_mockserver.py +6 -0
- infer_stack-0.7.2/tests/test_parity.py +768 -0
- infer_stack-0.7.2/tests/test_publication_transaction.py +149 -0
- infer_stack-0.7.2/tests/test_shm_size.py +56 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_tui.py +936 -87
- infer_stack-0.7.2/tests/test_tui_kube.py +491 -0
- infer_stack-0.7.0/PKG-INFO +0 -1073
- infer_stack-0.7.0/README.md +0 -1037
- infer_stack-0.7.0/infer_stack/__init__.py +0 -2
- infer_stack-0.7.0/infer_stack/backends/__init__.py +0 -4
- infer_stack-0.7.0/infer_stack/backends/kubeai.py +0 -535
- infer_stack-0.7.0/infer_stack/backends/kubeai_renderer.py +0 -204
- infer_stack-0.7.0/infer_stack/cli/commands_runtime.py +0 -627
- infer_stack-0.7.0/infer_stack/cli/options.py +0 -148
- infer_stack-0.7.0/infer_stack/kubeai_ops.py +0 -76
- infer_stack-0.7.0/infer_stack/leasing/backend.py +0 -309
- infer_stack-0.7.0/infer_stack/leasing/compose.py +0 -2359
- infer_stack-0.7.0/infer_stack/leasing/controller.py +0 -840
- infer_stack-0.7.0/infer_stack/leasing/store.py +0 -484
- infer_stack-0.7.0/infer_stack/leasing/suggest.py +0 -274
- infer_stack-0.7.0/infer_stack/tui.py +0 -2574
- infer_stack-0.7.0/infer_stack.egg-info/PKG-INFO +0 -1073
- infer_stack-0.7.0/tests/test_leasing_coalesced_apply.py +0 -405
- infer_stack-0.7.0/tests/test_leasing_kubeai.py +0 -518
- infer_stack-0.7.0/tests/test_leasing_route_registry.py +0 -427
- infer_stack-0.7.0/tests/test_leasing_suggest.py +0 -132
- {infer_stack-0.7.0 → infer_stack-0.7.2}/LICENSE +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/__main__.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/_log.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/cli/__main__.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/diff_prompt.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/docker_utils.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/experimental/model_memory_estimator.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/experimental/stress_test_long_context.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/mockserver/__init__.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/mockserver/data/oracle_questions.yaml +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/mockserver/modes.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/mockserver/simulator.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/mockserver/vllm_serve.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/paths.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/probe.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack.egg-info/dependency_links.txt +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack.egg-info/entry_points.txt +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack.egg-info/top_level.txt +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/setup.cfg +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_cli_config.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_import.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_reservation_gpu_frame_e2e.py +0 -0
- {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_simulator_runtime.py +0 -0
|
@@ -0,0 +1,780 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: infer-stack
|
|
3
|
+
Version: 0.7.2
|
|
4
|
+
Summary: Profile-driven compose and KubeAI deployment compiler for inference stacks
|
|
5
|
+
Author-email: "jon.crall" <jon.crall@kitware.com>
|
|
6
|
+
License-Expression: Apache-2.0
|
|
7
|
+
Project-URL: Homepage, https://github.com/AIQ-Kitware/infer_stack
|
|
8
|
+
Classifier: Development Status :: 1 - Planning
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.15
|
|
16
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
17
|
+
Classifier: Topic :: Utilities
|
|
18
|
+
Requires-Python: >=3.10
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: jinja2>=3.1
|
|
22
|
+
Requires-Dist: kwconf>=0.11.0
|
|
23
|
+
Requires-Dist: loguru>=0.7
|
|
24
|
+
Requires-Dist: pyyaml>=6.0
|
|
25
|
+
Requires-Dist: requests>=2.31
|
|
26
|
+
Requires-Dist: rich>=13.0
|
|
27
|
+
Requires-Dist: ubelt>=1.3
|
|
28
|
+
Provides-Extra: tests
|
|
29
|
+
Requires-Dist: pytest>=7.0; extra == "tests"
|
|
30
|
+
Requires-Dist: pytest-codeblocks>=0.17; extra == "tests"
|
|
31
|
+
Requires-Dist: pytest-cov>=3.0; extra == "tests"
|
|
32
|
+
Requires-Dist: xdoctest>=1.1.5; extra == "tests"
|
|
33
|
+
Provides-Extra: tui
|
|
34
|
+
Requires-Dist: textual>=0.50; extra == "tui"
|
|
35
|
+
Dynamic: license-file
|
|
36
|
+
|
|
37
|
+
# Infer Stack
|
|
38
|
+
|
|
39
|
+
[](https://pypi.org/project/infer-stack/)
|
|
40
|
+
[](https://pypi.org/project/infer-stack/)
|
|
41
|
+
[](https://github.com/AIQ-Kitware/infer_stack/blob/main/LICENSE)
|
|
42
|
+
|
|
43
|
+
Declare models in a catalog (`infer-stack catalog …`) and `acquire` or `run`
|
|
44
|
+
endpoints on demand. `infer-stack help tree` prints the whole command surface;
|
|
45
|
+
[docs/source/manual/](docs/source/manual/) has the Ollama + Open WebUI
|
|
46
|
+
tutorial and the leasing demo.
|
|
47
|
+
|
|
48
|
+
## Primary leasing workflow
|
|
49
|
+
|
|
50
|
+
The normal user path has three steps and no separate publication phase:
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
infer-stack config init
|
|
54
|
+
infer-stack catalog suggest --apply
|
|
55
|
+
infer-stack acquire <endpoint>
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
No GPU here? `infer-stack catalog suggest --simulator --apply` adds
|
|
59
|
+
`mock-smol`, a simulator that answers like vLLM with random text, so the same
|
|
60
|
+
three steps run on any host with Docker ([docs/mock-endpoints.md](docs/mock-endpoints.md)).
|
|
61
|
+
|
|
62
|
+
`settings.yaml` (`infer-stack config …`) and `catalog.yaml` are the user
|
|
63
|
+
configuration. Leasing keeps an
|
|
64
|
+
internal frozen recovery snapshot so a crash cannot re-render committed state
|
|
65
|
+
with different settings, but ordinary `acquire` advances that snapshot
|
|
66
|
+
automatically. Compatible catalog additions can be acquired while other models
|
|
67
|
+
are live; conflicting redefinitions or global setting changes take effect once
|
|
68
|
+
the affected stack is quiescent. `infer-stack config publish` is an advanced
|
|
69
|
+
pre-seeding/preview tool for multi-catalog operators, not a required fourth
|
|
70
|
+
step. See [ADR 0001](docs/adr/0001-user-config-is-authoritative.md).
|
|
71
|
+
|
|
72
|
+
## Related work
|
|
73
|
+
|
|
74
|
+
- [HyperQwen](https://github.com/syv-ai/HyperQwen) is a specialized model-preparation
|
|
75
|
+
and serving stack for Qwen3.8-27B. infer-stack delegates requantization,
|
|
76
|
+
patched-vLLM launchers, speculative decoding, and GPU-level tuning to its
|
|
77
|
+
image, while hardware discovery and suggestion choose explicit catalog
|
|
78
|
+
profiles by GPU *class* (compute capability + VRAM) rather than product-name
|
|
79
|
+
strings. Memory-tight 24-32 GiB Ampere-or-newer cards get the reference
|
|
80
|
+
fast/long/huge choices. High-VRAM Turing cards (sm75-sm79, >=48 GiB) get the
|
|
81
|
+
measured full-262K FP16/Triton profiles, with the prepared `-fast` checkpoint
|
|
82
|
+
preferred. Roomy Ampere-or-newer cards (>=48 GiB) get the fast baseline plus
|
|
83
|
+
a conservative/provisional full-262K ordinary-KV prefab; use
|
|
84
|
+
`dev/profile_qwen38_hyperqwen.sh` to refine that class on new hardware such as
|
|
85
|
+
Blackwell. Exact GPU-name gates remain available for future exceptions backed
|
|
86
|
+
by model-specific measurements. The model identity is
|
|
87
|
+
`qwen3.8-27b-dbirks-hyperqwen` because it starts from the
|
|
88
|
+
`dbirks/Qwen3.8-27B-W4A16-AutoRound` derivative; the unsuffixed
|
|
89
|
+
`qwen3.8-27b` identity is left available for the official checkpoint.
|
|
90
|
+
|
|
91
|
+
## Supported platform
|
|
92
|
+
|
|
93
|
+
`infer-stack` supports **Linux hosts only**. Its process locking, container
|
|
94
|
+
runtime integration, GPU discovery, and deployment workflows are intentionally
|
|
95
|
+
Linux-oriented. Windows is not a supported or tested execution platform.
|
|
96
|
+
|
|
97
|
+
Operational and security constraints that are accepted during the current
|
|
98
|
+
planning-stage release are tracked in
|
|
99
|
+
[docs/planning/known-limitations.md](docs/planning/known-limitations.md).
|
|
100
|
+
|
|
101
|
+
`infer-stack` serves the endpoints declared in a catalog. `acquire` takes a
|
|
102
|
+
lease on an endpoint; the controller places its engine on free GPUs and
|
|
103
|
+
reconciles the backend to run it:
|
|
104
|
+
|
|
105
|
+
* **engines**: vLLM (one container per deployment) and Ollama (one daemon per
|
|
106
|
+
`runtime_hosts` entry, serving many tags);
|
|
107
|
+
* **LiteLLM gateway**: one OpenAI base URL, `http://127.0.0.1:14042/v1`, in
|
|
108
|
+
front of every endpoint alias. On by default; `config set litellm false`
|
|
109
|
+
drops it;
|
|
110
|
+
* **Open WebUI**: on by default at `http://127.0.0.1:13000`;
|
|
111
|
+
`config set ui false` or `acquire --no-ui` drops it;
|
|
112
|
+
* **reverse proxy**: an optional single-port nginx in front of both.
|
|
113
|
+
|
|
114
|
+
Two backends run this: **Compose** (single host; vLLM and Ollama) and
|
|
115
|
+
**KubeAI** (a Kubernetes cluster; vLLM only).
|
|
116
|
+
|
|
117
|
+
## Main commands
|
|
118
|
+
|
|
119
|
+
```bash
|
|
120
|
+
infer-stack config init # data dir + default backend -> settings.yaml
|
|
121
|
+
infer-stack catalog suggest --apply # seed catalog.yaml from this host's GPUs
|
|
122
|
+
infer-stack catalog show # what can be acquired
|
|
123
|
+
infer-stack kube inventory # Kubernetes/GPU/KubeAI facts, before or after install
|
|
124
|
+
infer-stack kube doctor # detailed readiness + actionable fixes
|
|
125
|
+
infer-stack kube node test NODE --expected-gpus=1 # plan worker GPU + real generation acceptance
|
|
126
|
+
infer-stack kube node detach NODE # preview handing a cluster GPU host to Compose
|
|
127
|
+
infer-stack acquire <endpoint> # lease, render, bring up, wait for a real generation
|
|
128
|
+
infer-stack access <endpoint> --env-file e.env # reach it, managed or external; leases only what runs here
|
|
129
|
+
infer-stack test <endpoint> # one generation through the gateway
|
|
130
|
+
infer-stack leases # desired vs running, per deployment
|
|
131
|
+
infer-stack status # paths, backend and a lease summary
|
|
132
|
+
infer-stack release --all # drop every lease
|
|
133
|
+
infer-stack paths # where settings, catalog, ledger and caches live
|
|
134
|
+
infer-stack version
|
|
135
|
+
infer-stack help tree # the whole command surface
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
The CLI is built on [`kwconf`](https://github.com/Erotemic/kwconf),
|
|
139
|
+
so every subcommand is also importable as a Python class:
|
|
140
|
+
|
|
141
|
+
```python
|
|
142
|
+
from infer_stack.cli import AcquireCLI, TestCLI
|
|
143
|
+
|
|
144
|
+
AcquireCLI.main(argv=False, names=['smol135-1'], yes=True)
|
|
145
|
+
TestCLI.main(argv=False, name='smol135-1')
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
`manage.py` and `infer-stack` are aliases for the same entry point;
|
|
149
|
+
shell examples below use `infer-stack`.
|
|
150
|
+
|
|
151
|
+
## Operating the rendered Compose stack
|
|
152
|
+
|
|
153
|
+
`ps` and `logs` read the backend directly; `stack` wraps `docker compose` on
|
|
154
|
+
the rendered project, so you never `cd` into it or repeat `-f`/`--env-file`.
|
|
155
|
+
|
|
156
|
+
```bash
|
|
157
|
+
infer-stack ps # engines, gateway, UI
|
|
158
|
+
infer-stack ps -a # include exited instances
|
|
159
|
+
infer-stack logs -f <endpoint> # follow whatever serves an endpoint
|
|
160
|
+
infer-stack logs --tail 200 litellm # the gateway's backlog
|
|
161
|
+
infer-stack logs -f --raw litellm # full LiteLLM tracebacks
|
|
162
|
+
infer-stack stack restart open-webui # docker compose restart
|
|
163
|
+
infer-stack stack stop # stop everything (no remove)
|
|
164
|
+
infer-stack stack start # start it back up
|
|
165
|
+
infer-stack stack pull # refresh images
|
|
166
|
+
infer-stack stack compose -- ps --format json # any other docker compose command
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
`logs` accepts a service or pod name, a container id prefix, a deployment id
|
|
170
|
+
or an endpoint alias. Interactive `infer-stack logs -f` compacts only
|
|
171
|
+
explicitly registered, known-noisy LiteLLM traceback shapes; unknown
|
|
172
|
+
tracebacks pass through unchanged. Redirected or piped output stays raw, and
|
|
173
|
+
`--raw` disables compaction in an interactive follow. `--no-color` drops the
|
|
174
|
+
name-prefix colors. The TUI uses the same compactor.
|
|
175
|
+
|
|
176
|
+
Ollama tags are pulled into the daemon on the first `acquire` of an endpoint
|
|
177
|
+
that serves them. `stack compose` reaches the daemon for anything else, e.g.
|
|
178
|
+
`infer-stack stack compose -- exec ollama-local-ollama ollama list`.
|
|
179
|
+
|
|
180
|
+
On the KubeAI backend `ps` and `logs` read pods, and the `stack` Compose verbs
|
|
181
|
+
act on the gateway's Compose project on this host.
|
|
182
|
+
|
|
183
|
+
## Inspect an endpoint before running it
|
|
184
|
+
|
|
185
|
+
```bash
|
|
186
|
+
infer-stack catalog show <endpoint>
|
|
187
|
+
infer-stack acquire <endpoint> --no-apply # declare + write the compose project; start nothing
|
|
188
|
+
infer-stack paths leasing # where docker-compose.yml landed
|
|
189
|
+
infer-stack apply # start it (or `release --all` to discard)
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
## Catalog model
|
|
193
|
+
|
|
194
|
+
`catalog.yaml` is the one user-edited description of what can run. Its
|
|
195
|
+
sections:
|
|
196
|
+
|
|
197
|
+
* `models`: weight sources (`hf://org/name`, with optional `revision`,
|
|
198
|
+
`quantization`, `dtype`);
|
|
199
|
+
* `endpoints`: served API names. Each picks an `engine` (`vllm` or `ollama`),
|
|
200
|
+
a `model` (a `models` key, or an Ollama tag), and optionally `runtime`
|
|
201
|
+
(vLLM settings), `protocol`, `placement`, `sharing` and `reclaim`;
|
|
202
|
+
* `runtime_hosts`: Ollama daemons, each with its GPUs and daemon settings;
|
|
203
|
+
* `bundles`: named lists of endpoints to acquire together.
|
|
204
|
+
|
|
205
|
+
An endpoint can instead name a server that already runs elsewhere, with
|
|
206
|
+
`external: {api_base, model, api_key_env}` in place of `engine`/`model`.
|
|
207
|
+
It has no lease: `infer-stack access` (or `run`) publishes its route on the
|
|
208
|
+
LiteLLM front door, and the client asks for the alias through the same base
|
|
209
|
+
URL as any other endpoint. Moving an alias between a managed runtime and an
|
|
210
|
+
external server changes neither the alias nor the workflow:
|
|
211
|
+
|
|
212
|
+
```bash
|
|
213
|
+
infer-stack env REMOTE_QWEN_KEY=sk-... # the upstream's key, by name
|
|
214
|
+
infer-stack access qwen-remote --env-file qwen.env
|
|
215
|
+
source qwen.env # OPENAI_BASE_URL, OPENAI_API_KEY, ...
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
`api_base` is resolved from inside the gateway's container or pod:
|
|
219
|
+
`localhost` there is the gateway itself, so a server on the same host is
|
|
220
|
+
reached through the Docker bridge address (e.g. `http://172.17.0.1:8000/v1`).
|
|
221
|
+
See [docs/planning/external-endpoints.md](docs/planning/external-endpoints.md).
|
|
222
|
+
|
|
223
|
+
```yaml
|
|
224
|
+
models:
|
|
225
|
+
smol135:
|
|
226
|
+
source: hf://HuggingFaceTB/SmolLM2-135M-Instruct
|
|
227
|
+
|
|
228
|
+
endpoints:
|
|
229
|
+
smol135-1:
|
|
230
|
+
engine: vllm
|
|
231
|
+
model: smol135
|
|
232
|
+
runtime: {max_model_len: 8192}
|
|
233
|
+
chat:
|
|
234
|
+
engine: ollama
|
|
235
|
+
host: local-ollama
|
|
236
|
+
model: qwen3.5:4b
|
|
237
|
+
|
|
238
|
+
runtime_hosts:
|
|
239
|
+
local-ollama:
|
|
240
|
+
engine: ollama
|
|
241
|
+
placement: {gpu_indices: [0]}
|
|
242
|
+
settings: {keep_alive: 30m}
|
|
243
|
+
|
|
244
|
+
bundles:
|
|
245
|
+
both: [smol135-1, chat]
|
|
246
|
+
```
|
|
247
|
+
|
|
248
|
+
Edit it with `infer-stack catalog model|endpoint|host|bundle add …`, or by
|
|
249
|
+
hand with `infer-stack catalog edit`; `infer-stack catalog validate` checks it.
|
|
250
|
+
The schema reference is the docstring of `infer_stack/leasing/catalog.py`.
|
|
251
|
+
|
|
252
|
+
Shapes the stack renders:
|
|
253
|
+
|
|
254
|
+
```text
|
|
255
|
+
Open WebUI -> LiteLLM -> vLLM / Ollama # the default
|
|
256
|
+
\-> an external OpenAI-compatible server (external:)
|
|
257
|
+
Open WebUI -> vLLM / Ollama # config set litellm false
|
|
258
|
+
```
|
|
259
|
+
|
|
260
|
+
The named stack profiles of earlier releases (`setup`, `switch`,
|
|
261
|
+
`--profile`) are gone; see
|
|
262
|
+
[docs/stack-graph-profiles.md](docs/stack-graph-profiles.md).
|
|
263
|
+
|
|
264
|
+
## Where config and rendered artifacts live
|
|
265
|
+
|
|
266
|
+
`infer-stack` follows XDG basedir conventions, so the directory you invoke it
|
|
267
|
+
from never changes which config it reads or where it writes. There are two
|
|
268
|
+
roots:
|
|
269
|
+
|
|
270
|
+
| What | Default location | How to relocate |
|
|
271
|
+
| --- | --- | --- |
|
|
272
|
+
| `settings.yaml`, `catalog.yaml` | `~/.config/infer_stack/` (resp. `$XDG_CONFIG_HOME`) | `--config-dir` or `INFER_STACK_CONFIG_DIR` |
|
|
273
|
+
| **Everything generated**: `leasing/` (the ledger, the compose project, its `.env`) and the bind-mounted state (`hf-cache/`, `vllm-cache/`, `open-webui/`, `ollama/`, …) | `~/.local/share/infer_stack/` (resp. `$XDG_DATA_HOME`) | `config set data_dir <path>`, `--data-dir` or `INFER_STACK_DATA_DIR` |
|
|
274
|
+
|
|
275
|
+
`infer-stack paths` prints every resolved path and whether it exists.
|
|
276
|
+
|
|
277
|
+
The data dir **relocates the one infer-stack installation controlling a
|
|
278
|
+
host/backend; it does not create an isolated second installation**. Do not run
|
|
279
|
+
controllers from multiple config/data roots against the same Docker host or
|
|
280
|
+
Kubernetes namespace. See the
|
|
281
|
+
[single-owner limitation](docs/planning/known-limitations.md#one-control-plane-per-host-or-backend-namespace).
|
|
282
|
+
|
|
283
|
+
```bash
|
|
284
|
+
# Persist it once; later commands read it from settings.yaml.
|
|
285
|
+
infer-stack config set data_dir /data/service/docker/infer-stack
|
|
286
|
+
|
|
287
|
+
# Or keep both roots in a checkout for an ad-hoc experiment.
|
|
288
|
+
export INFER_STACK_CONFIG_DIR=$PWD/cfg INFER_STACK_DATA_DIR=$PWD/stack
|
|
289
|
+
infer-stack paths
|
|
290
|
+
```
|
|
291
|
+
|
|
292
|
+
`--config-dir` / `--data-dir` are accepted by every subcommand, after the
|
|
293
|
+
subcommand name.
|
|
294
|
+
|
|
295
|
+
## Constraining placement to specific GPUs
|
|
296
|
+
|
|
297
|
+
```bash
|
|
298
|
+
# Only place onto GPU 1 (e.g. GPU 0 is running a display).
|
|
299
|
+
infer-stack acquire <endpoint> --allowed-gpus 1
|
|
300
|
+
|
|
301
|
+
# Or confine a TP=2 endpoint to physical GPUs 1 and 3.
|
|
302
|
+
infer-stack acquire <tp2-endpoint> --allowed-gpus 1,3
|
|
303
|
+
```
|
|
304
|
+
|
|
305
|
+
`--allowed-gpus` (or `INFER_STACK_ALLOWED_GPUS=1,3`) filters the detected
|
|
306
|
+
inventory before placement for that call only. Real indices are preserved, so
|
|
307
|
+
the rendered compose stack pins `device_ids` to those exact GPUs. The durable
|
|
308
|
+
forms live in data:
|
|
309
|
+
|
|
310
|
+
* `placement: {gpu_indices: [1]}` on a vLLM endpoint pins it exactly (the list
|
|
311
|
+
length must equal tp×pp×dp); Ollama daemons pin through their
|
|
312
|
+
`runtime_hosts` entry;
|
|
313
|
+
* `placement: {min_vram_gib: 24}` makes smaller GPUs ineligible;
|
|
314
|
+
* `config set skip_display_gpus true` (or `--skip-display-gpus`) leaves the
|
|
315
|
+
GPU driving a monitor free.
|
|
316
|
+
|
|
317
|
+
## Demos / integration recipes
|
|
318
|
+
|
|
319
|
+
The user manual under [docs/source/manual/](docs/source/manual/) has two
|
|
320
|
+
walkthroughs on the current CLI:
|
|
321
|
+
[the Ollama + Open WebUI tutorial](docs/source/manual/ollama-openwebui-tutorial.md)
|
|
322
|
+
and [the leasing demo](docs/source/manual/leasing-demo.md) (standing service,
|
|
323
|
+
Open WebUI, several models side by side).
|
|
324
|
+
|
|
325
|
+
---
|
|
326
|
+
|
|
327
|
+
## Backend 1: Compose
|
|
328
|
+
|
|
329
|
+
Use Compose for single-host serving. It runs vLLM and Ollama engines, mixed
|
|
330
|
+
freely, behind the optional gateway and UI.
|
|
331
|
+
|
|
332
|
+
### Getting started
|
|
333
|
+
|
|
334
|
+
Prerequisite: Docker and the `docker compose` plugin.
|
|
335
|
+
|
|
336
|
+
```bash
|
|
337
|
+
infer-stack config init --backend compose
|
|
338
|
+
infer-stack doctor --gpu
|
|
339
|
+
infer-stack catalog init
|
|
340
|
+
|
|
341
|
+
# A vLLM endpoint.
|
|
342
|
+
infer-stack catalog model add smol135 --source hf://HuggingFaceTB/SmolLM2-135M-Instruct
|
|
343
|
+
infer-stack catalog endpoint add --model smol135 # -> smol135-1
|
|
344
|
+
infer-stack acquire smol135-1
|
|
345
|
+
|
|
346
|
+
# An Ollama endpoint, on a daemon pinned to GPU 0.
|
|
347
|
+
infer-stack catalog host add local-ollama --engine ollama --gpu 0
|
|
348
|
+
infer-stack catalog endpoint add chat --engine ollama --host local-ollama --model qwen3.5:4b
|
|
349
|
+
infer-stack acquire chat
|
|
350
|
+
```
|
|
351
|
+
|
|
352
|
+
`infer-stack catalog suggest --apply` fills the catalog with endpoints sized
|
|
353
|
+
for the detected GPUs instead.
|
|
354
|
+
|
|
355
|
+
### Test that it is responding
|
|
356
|
+
|
|
357
|
+
With LiteLLM enabled, every endpoint is reachable by its alias at:
|
|
358
|
+
|
|
359
|
+
```text
|
|
360
|
+
http://127.0.0.1:14042/v1
|
|
361
|
+
```
|
|
362
|
+
|
|
363
|
+
`acquire` already blocks until the endpoint returns a real generation through
|
|
364
|
+
that front door, which is stronger than Docker's container health. After
|
|
365
|
+
`acquire --no-wait`, block separately; to check again later, send one request:
|
|
366
|
+
|
|
367
|
+
```bash
|
|
368
|
+
infer-stack wait smol135-1
|
|
369
|
+
infer-stack test smol135-1
|
|
370
|
+
infer-stack test chat --prompt "Name three colors." --max-tokens 32
|
|
371
|
+
```
|
|
372
|
+
|
|
373
|
+
Clients read the front door and the managed key from the env file:
|
|
374
|
+
|
|
375
|
+
```bash
|
|
376
|
+
export OPENAI_BASE_URL=$(infer-stack env OPENAI_BASE_URL)
|
|
377
|
+
export OPENAI_API_KEY=$(infer-stack env LITELLM_MASTER_KEY)
|
|
378
|
+
```
|
|
379
|
+
|
|
380
|
+
or get both, plus per-endpoint names, from
|
|
381
|
+
`infer-stack acquire <endpoint> --env-file lease.env`.
|
|
382
|
+
|
|
383
|
+
To replace the gateway's master key (refused while leases are active; the
|
|
384
|
+
gateway restarts, and clients must fetch the key again):
|
|
385
|
+
|
|
386
|
+
```bash
|
|
387
|
+
infer-stack secrets rotate
|
|
388
|
+
```
|
|
389
|
+
|
|
390
|
+
### Stop it
|
|
391
|
+
|
|
392
|
+
```bash
|
|
393
|
+
infer-stack release --all # drop every lease
|
|
394
|
+
infer-stack release --all --evict # ...and stop the engines now
|
|
395
|
+
infer-stack clean -f # no leases, nothing on a GPU; the gateway stays
|
|
396
|
+
infer-stack stack down # docker compose down, bypassing the ledger
|
|
397
|
+
```
|
|
398
|
+
|
|
399
|
+
After a plain `release`, `keep-warm` endpoints (the default `reclaim` policy)
|
|
400
|
+
stay loaded until another lease needs their GPUs or `infer-stack evict` stops
|
|
401
|
+
them. `stack down` releases no lease, so the next `apply` or `acquire` brings
|
|
402
|
+
leased models back.
|
|
403
|
+
|
|
404
|
+
All state is bind-mounted from the data dir, so none of these delete it,
|
|
405
|
+
including `stack down --volumes`. For a destructive reset, remove the
|
|
406
|
+
directories `infer-stack paths` lists.
|
|
407
|
+
|
|
408
|
+
### Open WebUI authentication
|
|
409
|
+
|
|
410
|
+
Open WebUI runs with `WEBUI_AUTH=False`: no login screen, and anyone who can
|
|
411
|
+
reach port 13000 gets the UI. No setting changes that. Keep the host on a
|
|
412
|
+
trusted network, or run without the UI (`config set ui false`).
|
|
413
|
+
|
|
414
|
+
### Reverse proxy
|
|
415
|
+
|
|
416
|
+
An optional nginx service publishes one HTTP port with the UI at `/` and the
|
|
417
|
+
API at `/v1`. It needs the LiteLLM gateway.
|
|
418
|
+
|
|
419
|
+
```bash
|
|
420
|
+
infer-stack config set reverse_proxy true # port 80
|
|
421
|
+
infer-stack config set reverse_proxy '{enabled: true, port: 8080}'
|
|
422
|
+
```
|
|
423
|
+
|
|
424
|
+
`acquire --reverse-proxy` turns it on for one call. Add `config_path:
|
|
425
|
+
/path/to/nginx.conf` to the block to mount your own config at
|
|
426
|
+
`/etc/nginx/conf.d/default.conf` instead of the generated one.
|
|
427
|
+
|
|
428
|
+
It does no TLS and no authentication. Only the one port is published and no
|
|
429
|
+
certificate is mounted, so terminate TLS in a proxy in front of it. The TLS
|
|
430
|
+
and LDAP settings of the pre-leasing profiles no longer exist.
|
|
431
|
+
|
|
432
|
+
### Persistent state and database layout
|
|
433
|
+
|
|
434
|
+
Everything lives under the data dir (`infer-stack paths`):
|
|
435
|
+
|
|
436
|
+
* `open-webui/`: Open WebUI's data directory (accounts, chats, settings),
|
|
437
|
+
mounted at `/app/backend/data`;
|
|
438
|
+
* `postgres-litellm/`: LiteLLM's route store, rendered only with
|
|
439
|
+
`config set dynamic_routing true`;
|
|
440
|
+
* `ollama/`: the Ollama model store, mounted at `/root/.ollama`;
|
|
441
|
+
* `hf-cache/`, `vllm-cache/`, `torch-cache/`, `triton-cache/`, `cuda-cache/`:
|
|
442
|
+
vLLM weights and compile caches (see
|
|
443
|
+
[docs/persistent-caches-and-warm-restarts.md](docs/persistent-caches-and-warm-restarts.md));
|
|
444
|
+
* `runtime/`: directories an endpoint's `runtime.mounts` asks for;
|
|
445
|
+
* `leasing/`: the ledger and the rendered compose project.
|
|
446
|
+
|
|
447
|
+
Open WebUI chat history is not tied to the models currently served, so old
|
|
448
|
+
chats may name aliases the gateway no longer advertises. That is expected.
|
|
449
|
+
|
|
450
|
+
### Custom .env values are preserved
|
|
451
|
+
|
|
452
|
+
The compose project's `.env` holds the managed secrets (`LITELLM_MASTER_KEY`,
|
|
453
|
+
`HF_TOKEN`, …). `infer-stack env KEY=VALUE` merges a value into it, and keys
|
|
454
|
+
you add are kept across renders. Compose uses the file for interpolation, so a
|
|
455
|
+
key reaches a container only when the rendered service references it (as
|
|
456
|
+
`HF_TOKEN` does for vLLM). Set `HF_TOKEN` before the first `acquire` of a
|
|
457
|
+
gated model.
|
|
458
|
+
|
|
459
|
+
### Switching models
|
|
460
|
+
|
|
461
|
+
There is no single active model to switch. `acquire` another endpoint and it
|
|
462
|
+
runs beside the first; release the first when you are done:
|
|
463
|
+
|
|
464
|
+
```bash
|
|
465
|
+
infer-stack acquire smol135-1
|
|
466
|
+
infer-stack acquire chat # now both are served
|
|
467
|
+
infer-stack leases # find smol135-1's lease id
|
|
468
|
+
infer-stack release <lease-id>
|
|
469
|
+
```
|
|
470
|
+
|
|
471
|
+
The gateway carries a route for every catalog endpoint, so adding or removing
|
|
472
|
+
a model does not recreate LiteLLM, and Open WebUI stays up. When GPUs are
|
|
473
|
+
short, `acquire` fails fast; `--queue` waits for a GPU instead, and idle
|
|
474
|
+
`keep-warm` deployments are evicted to make room. vLLM containers are named
|
|
475
|
+
after the served alias (`vllm-<alias>`), so `docker ps` and
|
|
476
|
+
`infer-stack logs vllm-<alias>` identify them.
|
|
477
|
+
|
|
478
|
+
### Protocol modes for base vs. instruct models
|
|
479
|
+
|
|
480
|
+
An endpoint's `protocol` is `chat` (the default) or `completions`. It decides
|
|
481
|
+
which surface the readiness probe and `infer-stack test` use, so a base model
|
|
482
|
+
without a chat template must declare `completions` or its `acquire` never sees
|
|
483
|
+
a ready generation:
|
|
484
|
+
|
|
485
|
+
```bash
|
|
486
|
+
infer-stack catalog model add pythia-160m --source hf://EleutherAI/pythia-160m
|
|
487
|
+
infer-stack catalog endpoint add --model pythia-160m --protocol completions
|
|
488
|
+
infer-stack acquire pythia-160m-1
|
|
489
|
+
infer-stack test pythia-160m-1 # hits /v1/completions
|
|
490
|
+
```
|
|
491
|
+
|
|
492
|
+
The gateway forwards `/v1/completions` unchanged, so evaluation clients that
|
|
493
|
+
need exact prompt control should call it directly. Open WebUI is a chat UI and
|
|
494
|
+
sends chat requests, which a base model cannot answer.
|
|
495
|
+
|
|
496
|
+
### Images with their own launcher
|
|
497
|
+
|
|
498
|
+
Some images wrap vLLM in their own launcher and are configured through
|
|
499
|
+
environment variables rather than `vllm serve` flags. Describe that in the
|
|
500
|
+
endpoint's `runtime`; infer-stack has no model-specific code for it:
|
|
501
|
+
|
|
502
|
+
```yaml
|
|
503
|
+
runtime:
|
|
504
|
+
image: example.org/my-vllm-launcher:1.0
|
|
505
|
+
max_model_len: 65536
|
|
506
|
+
gpu_memory_utilization: 0.93
|
|
507
|
+
command: [single] # replaces `vllm serve MODEL <flags>`
|
|
508
|
+
env: # container environment
|
|
509
|
+
PORT: '{port}'
|
|
510
|
+
MAX_LEN: '{max_model_len}' # filled from the field above
|
|
511
|
+
GPU_UTIL: '{gpu_memory_utilization}'
|
|
512
|
+
EXTRA_ARGS: '--served-model-name={served_model_name}'
|
|
513
|
+
mounts: # persisted under the runtime data dir
|
|
514
|
+
/app/models: my-launcher/models
|
|
515
|
+
```
|
|
516
|
+
|
|
517
|
+
Changing the launcher's mode is a data edit, in the catalog or the TUI's
|
|
518
|
+
endpoint editor. The HyperQwen suggestion (see "Related work") is a worked
|
|
519
|
+
example: `catalog suggest` uses compute capability and VRAM to emit the serving
|
|
520
|
+
profiles appropriate to the detected hardware class (and exact name gates only
|
|
521
|
+
for measured exceptions). The hardware check happens only while suggesting;
|
|
522
|
+
the resulting catalog holds ordinary explicit runtime data, so `apply`/`acquire`
|
|
523
|
+
never retunes an endpoint after the fact.
|
|
524
|
+
|
|
525
|
+
- `{max_model_len}`, `{gpu_memory_utilization}`, `{served_model_name}` and
|
|
526
|
+
`{port}` are filled in from the endpoint, so a launcher that takes them
|
|
527
|
+
through its own variables stays in step when the fields change.
|
|
528
|
+
- Env values are written as strings (`true`/`false` for booleans), and `$`
|
|
529
|
+
is literal. `HF_TOKEN`, `VLLM_ATTENTION_BACKEND`, `CUDA_VISIBLE_DEVICES` and
|
|
530
|
+
`NVIDIA_VISIBLE_DEVICES` are infer-stack's and are refused.
|
|
531
|
+
- `extra_args` stay what they were: flags appended to the stock `vllm serve`
|
|
532
|
+
command, after infer-stack's own, so vLLM keeps the extra value for a
|
|
533
|
+
repeated flag. Repeating a flag infer-stack acts on (served name, parallel
|
|
534
|
+
sizes, `--max-model-len`) is refused; with `command`, pass flags through the
|
|
535
|
+
launcher instead.
|
|
536
|
+
- All of these are deployment identity: endpoints that launch differently
|
|
537
|
+
never share a process.
|
|
538
|
+
- `env` works on both backends (on KubeAI it becomes the Model's `spec.env`).
|
|
539
|
+
`command` and `mounts` are Compose only; KubeAI refuses an endpoint with
|
|
540
|
+
either rather than serve stock vLLM in its place.
|
|
541
|
+
|
|
542
|
+
### Reasoning / thinking models
|
|
543
|
+
|
|
544
|
+
vLLM separates a reasoning trace from the answer when it is started with a
|
|
545
|
+
reasoning parser. Pass the flag through `runtime.extra_args`:
|
|
546
|
+
|
|
547
|
+
```yaml
|
|
548
|
+
endpoints:
|
|
549
|
+
qwen3-think:
|
|
550
|
+
engine: vllm
|
|
551
|
+
model: qwen3-0.6b
|
|
552
|
+
runtime:
|
|
553
|
+
extra_args: [--reasoning-parser=qwen3]
|
|
554
|
+
```
|
|
555
|
+
|
|
556
|
+
The parser name depends on the model family and the vLLM version (`vllm serve
|
|
557
|
+
--help` lists them). To test end to end:
|
|
558
|
+
|
|
559
|
+
```bash
|
|
560
|
+
infer-stack test qwen3-think --prompt "Think step by step: 17*23" --max-tokens 512
|
|
561
|
+
|
|
562
|
+
# Streaming, through the gateway:
|
|
563
|
+
curl -N "$(infer-stack env OPENAI_BASE_URL)/chat/completions" \
|
|
564
|
+
-H "Authorization: Bearer $(infer-stack env LITELLM_MASTER_KEY)" \
|
|
565
|
+
-H 'Content-Type: application/json' \
|
|
566
|
+
-d '{"model":"qwen3-think","stream":true,
|
|
567
|
+
"messages":[{"role":"user","content":"Think step by step: 17*23"}]}'
|
|
568
|
+
```
|
|
569
|
+
|
|
570
|
+
In Open WebUI, reasoning shows up best with streaming enabled in the
|
|
571
|
+
chat settings.
|
|
572
|
+
|
|
573
|
+
---
|
|
574
|
+
|
|
575
|
+
## Backend 2: KubeAI
|
|
576
|
+
|
|
577
|
+
`--backend kubeai` runs the same leasing verbs against a Kubernetes cluster
|
|
578
|
+
running [KubeAI](https://www.kubeai.org): the same catalog, ledger, TTLs,
|
|
579
|
+
env file and TUI, with `Model` custom resources in place of compose
|
|
580
|
+
services and the cluster scheduler in place of the local GPU planner. The
|
|
581
|
+
LiteLLM gateway still fronts everything, so a card sees one `OPENAI_BASE_URL`,
|
|
582
|
+
the managed key and the endpoint alias on either backend.
|
|
583
|
+
|
|
584
|
+
* Creating/joining a cluster and the distribution boundary:
|
|
585
|
+
[docs/cluster-setup.md](docs/cluster-setup.md).
|
|
586
|
+
* KubeAI setup, settings and semantics:
|
|
587
|
+
[docs/kubeai-backend.md](docs/kubeai-backend.md).
|
|
588
|
+
* What matches Compose, what does not yet, and what is deliberate:
|
|
589
|
+
[docs/backend-parity.md](docs/backend-parity.md); the plan to close the
|
|
590
|
+
rest: [docs/planning/backend-parity-roadmap.md](docs/planning/backend-parity-roadmap.md).
|
|
591
|
+
|
|
592
|
+
The short version:
|
|
593
|
+
|
|
594
|
+
```bash
|
|
595
|
+
infer-stack kube inventory
|
|
596
|
+
infer-stack kube doctor
|
|
597
|
+
# New local cluster / NVIDIA plugin + GFD prerequisites (review before applying):
|
|
598
|
+
infer-stack kube bootstrap --provider=k3s
|
|
599
|
+
sudo -v
|
|
600
|
+
infer-stack kube bootstrap --provider=k3s --apply
|
|
601
|
+
# Automatic profiles from Kubernetes node labels; no temporary values file needed:
|
|
602
|
+
infer-stack kube install
|
|
603
|
+
infer-stack kube install --apply
|
|
604
|
+
infer-stack kube doctor
|
|
605
|
+
infer-stack catalog suggest --backend kubeai
|
|
606
|
+
infer-stack config set backend kubeai
|
|
607
|
+
infer-stack config set kubeai_gateway cluster
|
|
608
|
+
infer-stack doctor --backend kubeai
|
|
609
|
+
infer-stack acquire <endpoint> --ttl 2h --env-file lease.env --yes
|
|
610
|
+
```
|
|
611
|
+
|
|
612
|
+
If the workstation already has a Compose ledger/configuration that you plan to
|
|
613
|
+
return to, keep that authority intact rather than switching backend kinds in
|
|
614
|
+
the same ledger. The cluster setup guide documents the temporary test pattern:
|
|
615
|
+
use `INFER_STACK_BACKEND=kubeai` with a separate KubeAI data root, and use
|
|
616
|
+
`infer-stack kube node detach/attach` to hand each physical GPU host between
|
|
617
|
+
Kubernetes scheduling and direct Compose use without uninstalling or rejoining
|
|
618
|
+
the node. See [docs/cluster-setup.md](docs/cluster-setup.md).
|
|
619
|
+
|
|
620
|
+
### Switching an existing recovery ledger to KubeAI
|
|
621
|
+
|
|
622
|
+
Stopping the Compose realization retains its recovery backend and historical
|
|
623
|
+
ledger. Backend kinds require separate recovery epochs. After releasing every
|
|
624
|
+
old lease and stopping the old runtime:
|
|
625
|
+
|
|
626
|
+
```bash
|
|
627
|
+
infer-stack release --all --backend compose
|
|
628
|
+
infer-stack stack down --backend compose
|
|
629
|
+
infer-stack config set backend kubeai
|
|
630
|
+
infer-stack status # configured kubeai / active recovery compose
|
|
631
|
+
infer-stack ledger rotate # verify quiescence and preview the archive
|
|
632
|
+
infer-stack ledger rotate --yes # archive old history; initialize KubeAI epoch
|
|
633
|
+
infer-stack ledger archives # archive paths + history inspection commands
|
|
634
|
+
infer-stack acquire <endpoint> --yes
|
|
635
|
+
```
|
|
636
|
+
|
|
637
|
+
Rotation refuses active leases, remaining old runtime objects, and unknown
|
|
638
|
+
runtime state. It does not change configuration/catalogs or silently tear down
|
|
639
|
+
workloads. Old leases, deployments and the recovery profile remain in SQLite
|
|
640
|
+
archives under `<ledger-directory>/archives/`. Rotation is transactional and
|
|
641
|
+
safe to retry; a matching recovery backend makes the command a no-op. `gc`
|
|
642
|
+
reports backend mismatches with transition instructions; `gc --forget` remains
|
|
643
|
+
a history-only operation.
|
|
644
|
+
|
|
645
|
+
### Kubernetes cluster setup and KubeAI integration
|
|
646
|
+
|
|
647
|
+
[docs/cluster-setup.md](docs/cluster-setup.md) describes server/worker setup.
|
|
648
|
+
`kube inventory` and `kube status` report structured facts; `--json` includes
|
|
649
|
+
probe errors without discarding other discoveries. Inventory and catalog
|
|
650
|
+
suggestions work before the KubeAI chart or namespace exists.
|
|
651
|
+
|
|
652
|
+
`kube doctor` checks Kubernetes, NVIDIA scheduling/discovery, and KubeAI in
|
|
653
|
+
dependency order. Top-level `doctor --backend kubeai` remains the operational
|
|
654
|
+
preflight for acquiring work. If Helm installation succeeds but the configured
|
|
655
|
+
KubeAI API URL is unavailable, doctor reports that remaining routing issue.
|
|
656
|
+
Set `kubeai_base_url` to a reachable OpenAI URL for the installed service.
|
|
657
|
+
|
|
658
|
+
`kube bootstrap` defaults to a read-only plan; `--apply` (or `--yes`) authorizes
|
|
659
|
+
host/cluster changes. K3s is the implemented provisioning provider; inventory,
|
|
660
|
+
installation and the backend work with any Kubernetes distribution. Bootstrap
|
|
661
|
+
always provisions/reconciles the local K3s server, preserves unrelated selected
|
|
662
|
+
contexts and kubeconfigs, installs Helm if missing,
|
|
663
|
+
and reconciles NVIDIA device plugin **0.17.1** with GPU Feature Discovery.
|
|
664
|
+
Install the host NVIDIA driver and container toolkit first. K3s discovers the
|
|
665
|
+
installed runtime on startup; bootstrap restarts local K3s only when runtime
|
|
666
|
+
discovery needs repair. It waits for Ready nodes, GPU allocation and GFD labels, and verifies unknown
|
|
667
|
+
GPU-node runtime handlers with small node-specific canary pods. Root admin
|
|
668
|
+
credentials stay `0600`; a private `0600` copy lives at
|
|
669
|
+
`~/.kube/infer-stack-k3s.yaml`. Existing default configs are preserved. Select the
|
|
670
|
+
local server explicitly with `export KUBECONFIG=~/.kube/infer-stack-k3s.yaml`
|
|
671
|
+
before inventory/install; rerun bootstrap to refresh copied certificates.
|
|
672
|
+
|
|
673
|
+
`kube install` shows inferred Helm values; `--apply` uses Helm upgrade/install
|
|
674
|
+
and checks readiness afterward. `--dry-run`/`--plan` forces read-only behavior.
|
|
675
|
+
`--namespace`, `--release`, `--chart`, `--version` and `--values` allow overrides.
|
|
676
|
+
Existing named profiles and custom values are preserved; `HF_TOKEN` overrides
|
|
677
|
+
the chart token using a temporary protected file. Generated public values live
|
|
678
|
+
at `<data>/generated/kube/kubeai-values.yaml`.
|
|
679
|
+
|
|
680
|
+
`kube setup` is a deprecated alias of `kube install`; it no longer owns a
|
|
681
|
+
separate prerequisite workflow. `kube k3s bootstrap` provisions only the local
|
|
682
|
+
server, `join` verifies requested worker membership, and `status` reports local
|
|
683
|
+
membership independently of any stale selected admin context. Setup scripts are compatibility wrappers around the
|
|
684
|
+
package commands.
|
|
685
|
+
|
|
686
|
+
### Debugging serving failures
|
|
687
|
+
|
|
688
|
+
`infer-stack kube status` summarizes live Kubernetes state. `infer-stack acquire`
|
|
689
|
+
reports pod failures such as `ImagePullBackOff`, `Unschedulable`, and engine
|
|
690
|
+
crashes. Use `infer-stack ps` and `infer-stack logs -f <endpoint>` to inspect
|
|
691
|
+
model serving. If `/models` works but completions return 404, check that clients
|
|
692
|
+
use the gateway alias; direct KubeAI requests use the Model name from the env
|
|
693
|
+
file (`INFER_STACK_ENDPOINT_*`). Missing `libcuda.so.1` usually indicates that
|
|
694
|
+
the model profile did not request GPUs or select the correct runtime.
|
|
695
|
+
|
|
696
|
+
---
|
|
697
|
+
|
|
698
|
+
## Which backend should I start with?
|
|
699
|
+
|
|
700
|
+
**Compose** when one workstation is enough: it is the fastest path to a
|
|
701
|
+
working server, everything it renders is a file you can read, and it needs
|
|
702
|
+
only Docker.
|
|
703
|
+
|
|
704
|
+
**KubeAI** when the models must run on more than one machine, or a cluster
|
|
705
|
+
already exists. It is Compose plus a scheduler: the same catalog, verbs and
|
|
706
|
+
env file, with a cluster and `resourceProfiles` supplied. Expect more
|
|
707
|
+
first-request overhead (pod creation, image pull, model load).
|
|
708
|
+
|
|
709
|
+
A catalog written for Compose runs on KubeAI, with two exceptions: ollama
|
|
710
|
+
endpoints and custom container launches (`runtime.command` / `mounts`),
|
|
711
|
+
which stay Compose-only. [docs/backend-parity.md](docs/backend-parity.md)
|
|
712
|
+
has the full matrix.
|
|
713
|
+
|
|
714
|
+
## vLLM startup caches
|
|
715
|
+
|
|
716
|
+
Generated Compose mounts persist Hugging Face, vLLM, PyTorch/TorchInductor,
|
|
717
|
+
Triton, and CUDA JIT caches. Warm starts avoid redownloading and redoing many
|
|
718
|
+
compile/JIT steps, but a vLLM model swap still creates a new engine process and
|
|
719
|
+
must reload weights into GPU memory.
|
|
720
|
+
|
|
721
|
+
### Diagnosing readiness
|
|
722
|
+
|
|
723
|
+
`docker compose` health only means a container-level healthcheck passed. It
|
|
724
|
+
is not "the routed model answers a request through the front door", which is
|
|
725
|
+
what `acquire` and `wait` check: a model swap starts a new engine process,
|
|
726
|
+
and LiteLLM stays up while returning upstream connection errors until vLLM
|
|
727
|
+
has loaded the weights.
|
|
728
|
+
|
|
729
|
+
```bash
|
|
730
|
+
infer-stack acquire <endpoint> --yes # waits for a real generation
|
|
731
|
+
infer-stack wait <endpoint> # after acquire --no-wait
|
|
732
|
+
infer-stack test <endpoint> # one generation through the gateway
|
|
733
|
+
infer-stack status # desired vs running, per deployment
|
|
734
|
+
infer-stack logs <service> --tail 80 # the engine or gateway log
|
|
735
|
+
```
|
|
736
|
+
|
|
737
|
+
A crash-looping engine fails the acquire at once with its error quoted, so
|
|
738
|
+
the timeout is only for a model that is loading. Reading Docker's own
|
|
739
|
+
signals: `litellm exited with code 137` is a SIGKILL (an OOM kill or a forced
|
|
740
|
+
replacement), whereas LiteLLM returning HTTP 500 with `Cannot connect to host
|
|
741
|
+
vllm-*` means LiteLLM is running and its upstream vLLM is not ready yet.
|
|
742
|
+
|
|
743
|
+
### Join and accept GPU workers
|
|
744
|
+
|
|
745
|
+
`infer-stack kube k3s onboard namek --server=... --token-file=... --kubeconfig=...`
|
|
746
|
+
plans a local GPU worker join and node-specific acceptance; add `--apply` to
|
|
747
|
+
execute it. Host NVIDIA drivers/toolkit are prerequisites. It infers the server
|
|
748
|
+
version, verifies every locally detected device, then checks actual one-GPU
|
|
749
|
+
KubeAI placement and generation on this worker. Admin credentials are explicit
|
|
750
|
+
private files, separate from any selected EKS context.
|
|
751
|
+
|
|
752
|
+
For an already joined worker, run `infer-stack kube node test namek
|
|
753
|
+
--expected-gpus=1 --namespace=default --apply` from the control plane. The same
|
|
754
|
+
surface accepts two-device `yardrat` and four-device `aiq-gpu2`; mixed products
|
|
755
|
+
are reported separately. Tests preserve leases and unrelated Models, remove
|
|
756
|
+
their temporary resources, and retain a reusable node profile. See the
|
|
757
|
+
[worker onboarding instructions](docs/cluster-setup.md#join-and-test-a-gpu-worker-in-one-reviewed-operation)
|
|
758
|
+
for plan/apply, credentials, interrupted cleanup and heterogeneous-node limits.
|
|
759
|
+
|
|
760
|
+
### Kubernetes dashboard
|
|
761
|
+
|
|
762
|
+
`infer-stack tui` provides a **Cluster** tab in the expanded runtime pane on the
|
|
763
|
+
KubeAI backend. It shows Ready/scheduling state, GPU allocation and scheduled
|
|
764
|
+
requests across namespaces, GFD labels, runtime evidence, managed Models and
|
|
765
|
+
KubeAI/NVIDIA pod startup/restart state. GPU requests are scheduling facts, not
|
|
766
|
+
live GPU utilization; the system pane explicitly describes the local host.
|
|
767
|
+
|
|
768
|
+
Cluster monitoring uses three batched API reads, cached for at least 15 seconds
|
|
769
|
+
and polled only while the tab is visible. Slow probes cannot overlap or block
|
|
770
|
+
keyboard input; **Refresh cluster** forces a sample, and **Doctor** runs the
|
|
771
|
+
shared detailed checks on demand. Pod samples also feed the Instances view.
|
|
772
|
+
|
|
773
|
+
Select a node for **Detach node** or **Attach node**. Each previews its workload
|
|
774
|
+
and scheduling state and requires confirmation; detach drains workloads and
|
|
775
|
+
emptyDir data, while attach confirms local Compose workloads have stopped.
|
|
776
|
+
These reuse `kube node detach/attach` and preserve cluster membership. The
|
|
777
|
+
Control tab provides **Apply** through the leasing controller and confirmed
|
|
778
|
+
**Down** for managed Models and the gateway. Leases remain after Down. The
|
|
779
|
+
KubeAI endpoint editor selects resource profiles; Kubernetes chooses node/GPU
|
|
780
|
+
placement.
|