infer-stack 0.7.0__tar.gz → 0.7.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (155) hide show
  1. infer_stack-0.7.2/PKG-INFO +780 -0
  2. infer_stack-0.7.2/README.md +744 -0
  3. infer_stack-0.7.2/infer_stack/__init__.py +2 -0
  4. infer_stack-0.7.2/infer_stack/backends/__init__.py +3 -0
  5. infer_stack-0.7.2/infer_stack/backends/kubeai.py +1464 -0
  6. infer_stack-0.7.2/infer_stack/backends/kubeai_gateway.py +380 -0
  7. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/cli/__init__.py +83 -7
  8. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/cli/commands_catalog.py +315 -154
  9. infer_stack-0.7.2/infer_stack/cli/commands_kube.py +668 -0
  10. infer_stack-0.7.2/infer_stack/cli/commands_kube_worker.py +195 -0
  11. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/cli/commands_leasing.py +1372 -438
  12. infer_stack-0.7.2/infer_stack/cli/commands_ledger.py +80 -0
  13. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/cli/commands_meta.py +92 -17
  14. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/cli/commands_mock.py +12 -12
  15. infer_stack-0.7.2/infer_stack/cli/commands_runtime.py +957 -0
  16. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/cli/context.py +1 -1
  17. infer_stack-0.7.2/infer_stack/cli/options.py +112 -0
  18. infer_stack-0.7.2/infer_stack/cli_equivalent.py +75 -0
  19. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/config.py +2 -1
  20. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/env_utils.py +31 -2
  21. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/experimental/model_catalog_discover.py +30 -22
  22. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/gpu_doctor.py +83 -17
  23. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/hardware.py +72 -31
  24. infer_stack-0.7.2/infer_stack/kube/__init__.py +13 -0
  25. infer_stack-0.7.2/infer_stack/kube/inspect.py +227 -0
  26. infer_stack-0.7.2/infer_stack/kube/k3s.py +343 -0
  27. infer_stack-0.7.2/infer_stack/kube/manage.py +817 -0
  28. infer_stack-0.7.2/infer_stack/kube/monitor.py +139 -0
  29. infer_stack-0.7.2/infer_stack/kube/operations.py +155 -0
  30. infer_stack-0.7.2/infer_stack/kube/worker.py +543 -0
  31. infer_stack-0.7.2/infer_stack/kubeai_ops.py +86 -0
  32. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/leasing/__init__.py +10 -1
  33. infer_stack-0.7.2/infer_stack/leasing/backend.py +916 -0
  34. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/leasing/catalog.py +233 -38
  35. infer_stack-0.7.2/infer_stack/leasing/catalog_edit.py +95 -0
  36. infer_stack-0.7.2/infer_stack/leasing/compose.py +2603 -0
  37. infer_stack-0.7.2/infer_stack/leasing/controller.py +2516 -0
  38. infer_stack-0.7.2/infer_stack/leasing/diagnosis.py +193 -0
  39. infer_stack-0.7.2/infer_stack/leasing/endpoints.py +277 -0
  40. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/leasing/envfile.py +37 -16
  41. infer_stack-0.7.2/infer_stack/leasing/gateway.py +1529 -0
  42. infer_stack-0.7.2/infer_stack/leasing/instances.py +321 -0
  43. infer_stack-0.7.2/infer_stack/leasing/launch.py +238 -0
  44. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/leasing/ledger.py +280 -59
  45. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/leasing/models.py +46 -33
  46. infer_stack-0.7.2/infer_stack/leasing/naming.py +109 -0
  47. infer_stack-0.7.2/infer_stack/leasing/network.py +138 -0
  48. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/leasing/placement.py +88 -2
  49. infer_stack-0.7.2/infer_stack/leasing/profile.py +443 -0
  50. infer_stack-0.7.2/infer_stack/leasing/residency.py +502 -0
  51. infer_stack-0.7.2/infer_stack/leasing/routes.py +213 -0
  52. infer_stack-0.7.2/infer_stack/leasing/store.py +867 -0
  53. infer_stack-0.7.2/infer_stack/leasing/suggest.py +660 -0
  54. infer_stack-0.7.2/infer_stack/leasing/transition.py +76 -0
  55. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/leasing/vram.py +51 -13
  56. infer_stack-0.7.2/infer_stack/log_filter.py +399 -0
  57. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/mockserver/server.py +5 -2
  58. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/profile_runtime.py +5 -3
  59. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/templates/suggestion-pool.yaml +204 -4
  60. infer_stack-0.7.2/infer_stack/tui.py +3979 -0
  61. infer_stack-0.7.2/infer_stack.egg-info/PKG-INFO +780 -0
  62. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack.egg-info/SOURCES.txt +54 -3
  63. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack.egg-info/requires.txt +1 -1
  64. {infer_stack-0.7.0 → infer_stack-0.7.2}/pyproject.toml +1 -4
  65. infer_stack-0.7.2/tests/test_backend_transition.py +201 -0
  66. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_cli_catalog.py +147 -0
  67. infer_stack-0.7.2/tests/test_cli_equivalent.py +85 -0
  68. infer_stack-0.7.2/tests/test_cli_kube.py +598 -0
  69. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_cli_leasing.py +418 -13
  70. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_cli_meta.py +25 -4
  71. infer_stack-0.7.2/tests/test_day2.py +267 -0
  72. infer_stack-0.7.2/tests/test_external_endpoints.py +791 -0
  73. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_gpu_doctor.py +86 -2
  74. infer_stack-0.7.2/tests/test_hardware.py +42 -0
  75. infer_stack-0.7.2/tests/test_kube_readiness.py +470 -0
  76. infer_stack-0.7.2/tests/test_kube_review.py +336 -0
  77. infer_stack-0.7.2/tests/test_kube_worker.py +540 -0
  78. infer_stack-0.7.2/tests/test_leasing_admission.py +559 -0
  79. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_catalog.py +133 -16
  80. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_compose.py +385 -58
  81. infer_stack-0.7.2/tests/test_leasing_context_metadata.py +450 -0
  82. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_controller.py +74 -0
  83. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_controller_lock.py +13 -4
  84. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_controller_queue.py +23 -28
  85. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_dynamic_routing.py +296 -34
  86. infer_stack-0.7.2/tests/test_leasing_image_pull.py +118 -0
  87. infer_stack-0.7.2/tests/test_leasing_kubeai.py +1240 -0
  88. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_ledger.py +20 -0
  89. infer_stack-0.7.2/tests/test_leasing_make_room.py +120 -0
  90. infer_stack-0.7.2/tests/test_leasing_network.py +229 -0
  91. infer_stack-0.7.2/tests/test_leasing_observability.py +72 -0
  92. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_placement.py +35 -3
  93. infer_stack-0.7.2/tests/test_leasing_placement_admission.py +81 -0
  94. infer_stack-0.7.2/tests/test_leasing_profile.py +681 -0
  95. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_reservation.py +3 -22
  96. infer_stack-0.7.2/tests/test_leasing_residency.py +268 -0
  97. infer_stack-0.7.2/tests/test_leasing_route_registry.py +360 -0
  98. infer_stack-0.7.2/tests/test_leasing_secrets.py +367 -0
  99. infer_stack-0.7.2/tests/test_leasing_selective_apply.py +212 -0
  100. infer_stack-0.7.2/tests/test_leasing_serialised_publication.py +822 -0
  101. infer_stack-0.7.2/tests/test_leasing_startup_failure.py +263 -0
  102. infer_stack-0.7.2/tests/test_leasing_store_threads.py +45 -0
  103. infer_stack-0.7.2/tests/test_leasing_suggest.py +399 -0
  104. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_leasing_vram.py +38 -0
  105. infer_stack-0.7.2/tests/test_log_filter.py +312 -0
  106. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_mock_vllm_serve.py +2 -2
  107. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_mockserver.py +6 -0
  108. infer_stack-0.7.2/tests/test_parity.py +768 -0
  109. infer_stack-0.7.2/tests/test_publication_transaction.py +149 -0
  110. infer_stack-0.7.2/tests/test_shm_size.py +56 -0
  111. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_tui.py +936 -87
  112. infer_stack-0.7.2/tests/test_tui_kube.py +491 -0
  113. infer_stack-0.7.0/PKG-INFO +0 -1073
  114. infer_stack-0.7.0/README.md +0 -1037
  115. infer_stack-0.7.0/infer_stack/__init__.py +0 -2
  116. infer_stack-0.7.0/infer_stack/backends/__init__.py +0 -4
  117. infer_stack-0.7.0/infer_stack/backends/kubeai.py +0 -535
  118. infer_stack-0.7.0/infer_stack/backends/kubeai_renderer.py +0 -204
  119. infer_stack-0.7.0/infer_stack/cli/commands_runtime.py +0 -627
  120. infer_stack-0.7.0/infer_stack/cli/options.py +0 -148
  121. infer_stack-0.7.0/infer_stack/kubeai_ops.py +0 -76
  122. infer_stack-0.7.0/infer_stack/leasing/backend.py +0 -309
  123. infer_stack-0.7.0/infer_stack/leasing/compose.py +0 -2359
  124. infer_stack-0.7.0/infer_stack/leasing/controller.py +0 -840
  125. infer_stack-0.7.0/infer_stack/leasing/store.py +0 -484
  126. infer_stack-0.7.0/infer_stack/leasing/suggest.py +0 -274
  127. infer_stack-0.7.0/infer_stack/tui.py +0 -2574
  128. infer_stack-0.7.0/infer_stack.egg-info/PKG-INFO +0 -1073
  129. infer_stack-0.7.0/tests/test_leasing_coalesced_apply.py +0 -405
  130. infer_stack-0.7.0/tests/test_leasing_kubeai.py +0 -518
  131. infer_stack-0.7.0/tests/test_leasing_route_registry.py +0 -427
  132. infer_stack-0.7.0/tests/test_leasing_suggest.py +0 -132
  133. {infer_stack-0.7.0 → infer_stack-0.7.2}/LICENSE +0 -0
  134. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/__main__.py +0 -0
  135. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/_log.py +0 -0
  136. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/cli/__main__.py +0 -0
  137. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/diff_prompt.py +0 -0
  138. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/docker_utils.py +0 -0
  139. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/experimental/model_memory_estimator.py +0 -0
  140. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/experimental/stress_test_long_context.py +0 -0
  141. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/mockserver/__init__.py +0 -0
  142. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/mockserver/data/oracle_questions.yaml +0 -0
  143. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/mockserver/modes.py +0 -0
  144. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/mockserver/simulator.py +0 -0
  145. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/mockserver/vllm_serve.py +0 -0
  146. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/paths.py +0 -0
  147. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack/probe.py +0 -0
  148. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack.egg-info/dependency_links.txt +0 -0
  149. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack.egg-info/entry_points.txt +0 -0
  150. {infer_stack-0.7.0 → infer_stack-0.7.2}/infer_stack.egg-info/top_level.txt +0 -0
  151. {infer_stack-0.7.0 → infer_stack-0.7.2}/setup.cfg +0 -0
  152. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_cli_config.py +0 -0
  153. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_import.py +0 -0
  154. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_reservation_gpu_frame_e2e.py +0 -0
  155. {infer_stack-0.7.0 → infer_stack-0.7.2}/tests/test_simulator_runtime.py +0 -0
@@ -0,0 +1,780 @@
1
+ Metadata-Version: 2.4
2
+ Name: infer-stack
3
+ Version: 0.7.2
4
+ Summary: Profile-driven compose and KubeAI deployment compiler for inference stacks
5
+ Author-email: "jon.crall" <jon.crall@kitware.com>
6
+ License-Expression: Apache-2.0
7
+ Project-URL: Homepage, https://github.com/AIQ-Kitware/infer_stack
8
+ Classifier: Development Status :: 1 - Planning
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: Programming Language :: Python :: 3.10
11
+ Classifier: Programming Language :: Python :: 3.11
12
+ Classifier: Programming Language :: Python :: 3.12
13
+ Classifier: Programming Language :: Python :: 3.13
14
+ Classifier: Programming Language :: Python :: 3.14
15
+ Classifier: Programming Language :: Python :: 3.15
16
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
17
+ Classifier: Topic :: Utilities
18
+ Requires-Python: >=3.10
19
+ Description-Content-Type: text/markdown
20
+ License-File: LICENSE
21
+ Requires-Dist: jinja2>=3.1
22
+ Requires-Dist: kwconf>=0.11.0
23
+ Requires-Dist: loguru>=0.7
24
+ Requires-Dist: pyyaml>=6.0
25
+ Requires-Dist: requests>=2.31
26
+ Requires-Dist: rich>=13.0
27
+ Requires-Dist: ubelt>=1.3
28
+ Provides-Extra: tests
29
+ Requires-Dist: pytest>=7.0; extra == "tests"
30
+ Requires-Dist: pytest-codeblocks>=0.17; extra == "tests"
31
+ Requires-Dist: pytest-cov>=3.0; extra == "tests"
32
+ Requires-Dist: xdoctest>=1.1.5; extra == "tests"
33
+ Provides-Extra: tui
34
+ Requires-Dist: textual>=0.50; extra == "tui"
35
+ Dynamic: license-file
36
+
37
+ # Infer Stack
38
+
39
+ [![PyPI version](https://img.shields.io/pypi/v/infer-stack.svg)](https://pypi.org/project/infer-stack/)
40
+ [![Python versions](https://img.shields.io/pypi/pyversions/infer-stack.svg)](https://pypi.org/project/infer-stack/)
41
+ [![License](https://img.shields.io/pypi/l/infer-stack.svg)](https://github.com/AIQ-Kitware/infer_stack/blob/main/LICENSE)
42
+
43
+ Declare models in a catalog (`infer-stack catalog …`) and `acquire` or `run`
44
+ endpoints on demand. `infer-stack help tree` prints the whole command surface;
45
+ [docs/source/manual/](docs/source/manual/) has the Ollama + Open WebUI
46
+ tutorial and the leasing demo.
47
+
48
+ ## Primary leasing workflow
49
+
50
+ The normal user path has three steps and no separate publication phase:
51
+
52
+ ```bash
53
+ infer-stack config init
54
+ infer-stack catalog suggest --apply
55
+ infer-stack acquire <endpoint>
56
+ ```
57
+
58
+ No GPU here? `infer-stack catalog suggest --simulator --apply` adds
59
+ `mock-smol`, a simulator that answers like vLLM with random text, so the same
60
+ three steps run on any host with Docker ([docs/mock-endpoints.md](docs/mock-endpoints.md)).
61
+
62
+ `settings.yaml` (`infer-stack config …`) and `catalog.yaml` are the user
63
+ configuration. Leasing keeps an
64
+ internal frozen recovery snapshot so a crash cannot re-render committed state
65
+ with different settings, but ordinary `acquire` advances that snapshot
66
+ automatically. Compatible catalog additions can be acquired while other models
67
+ are live; conflicting redefinitions or global setting changes take effect once
68
+ the affected stack is quiescent. `infer-stack config publish` is an advanced
69
+ pre-seeding/preview tool for multi-catalog operators, not a required fourth
70
+ step. See [ADR 0001](docs/adr/0001-user-config-is-authoritative.md).
71
+
72
+ ## Related work
73
+
74
+ - [HyperQwen](https://github.com/syv-ai/HyperQwen) is a specialized model-preparation
75
+ and serving stack for Qwen3.8-27B. infer-stack delegates requantization,
76
+ patched-vLLM launchers, speculative decoding, and GPU-level tuning to its
77
+ image, while hardware discovery and suggestion choose explicit catalog
78
+ profiles by GPU *class* (compute capability + VRAM) rather than product-name
79
+ strings. Memory-tight 24-32 GiB Ampere-or-newer cards get the reference
80
+ fast/long/huge choices. High-VRAM Turing cards (sm75-sm79, >=48 GiB) get the
81
+ measured full-262K FP16/Triton profiles, with the prepared `-fast` checkpoint
82
+ preferred. Roomy Ampere-or-newer cards (>=48 GiB) get the fast baseline plus
83
+ a conservative/provisional full-262K ordinary-KV prefab; use
84
+ `dev/profile_qwen38_hyperqwen.sh` to refine that class on new hardware such as
85
+ Blackwell. Exact GPU-name gates remain available for future exceptions backed
86
+ by model-specific measurements. The model identity is
87
+ `qwen3.8-27b-dbirks-hyperqwen` because it starts from the
88
+ `dbirks/Qwen3.8-27B-W4A16-AutoRound` derivative; the unsuffixed
89
+ `qwen3.8-27b` identity is left available for the official checkpoint.
90
+
91
+ ## Supported platform
92
+
93
+ `infer-stack` supports **Linux hosts only**. Its process locking, container
94
+ runtime integration, GPU discovery, and deployment workflows are intentionally
95
+ Linux-oriented. Windows is not a supported or tested execution platform.
96
+
97
+ Operational and security constraints that are accepted during the current
98
+ planning-stage release are tracked in
99
+ [docs/planning/known-limitations.md](docs/planning/known-limitations.md).
100
+
101
+ `infer-stack` serves the endpoints declared in a catalog. `acquire` takes a
102
+ lease on an endpoint; the controller places its engine on free GPUs and
103
+ reconciles the backend to run it:
104
+
105
+ * **engines**: vLLM (one container per deployment) and Ollama (one daemon per
106
+ `runtime_hosts` entry, serving many tags);
107
+ * **LiteLLM gateway**: one OpenAI base URL, `http://127.0.0.1:14042/v1`, in
108
+ front of every endpoint alias. On by default; `config set litellm false`
109
+ drops it;
110
+ * **Open WebUI**: on by default at `http://127.0.0.1:13000`;
111
+ `config set ui false` or `acquire --no-ui` drops it;
112
+ * **reverse proxy**: an optional single-port nginx in front of both.
113
+
114
+ Two backends run this: **Compose** (single host; vLLM and Ollama) and
115
+ **KubeAI** (a Kubernetes cluster; vLLM only).
116
+
117
+ ## Main commands
118
+
119
+ ```bash
120
+ infer-stack config init # data dir + default backend -> settings.yaml
121
+ infer-stack catalog suggest --apply # seed catalog.yaml from this host's GPUs
122
+ infer-stack catalog show # what can be acquired
123
+ infer-stack kube inventory # Kubernetes/GPU/KubeAI facts, before or after install
124
+ infer-stack kube doctor # detailed readiness + actionable fixes
125
+ infer-stack kube node test NODE --expected-gpus=1 # plan worker GPU + real generation acceptance
126
+ infer-stack kube node detach NODE # preview handing a cluster GPU host to Compose
127
+ infer-stack acquire <endpoint> # lease, render, bring up, wait for a real generation
128
+ infer-stack access <endpoint> --env-file e.env # reach it, managed or external; leases only what runs here
129
+ infer-stack test <endpoint> # one generation through the gateway
130
+ infer-stack leases # desired vs running, per deployment
131
+ infer-stack status # paths, backend and a lease summary
132
+ infer-stack release --all # drop every lease
133
+ infer-stack paths # where settings, catalog, ledger and caches live
134
+ infer-stack version
135
+ infer-stack help tree # the whole command surface
136
+ ```
137
+
138
+ The CLI is built on [`kwconf`](https://github.com/Erotemic/kwconf),
139
+ so every subcommand is also importable as a Python class:
140
+
141
+ ```python
142
+ from infer_stack.cli import AcquireCLI, TestCLI
143
+
144
+ AcquireCLI.main(argv=False, names=['smol135-1'], yes=True)
145
+ TestCLI.main(argv=False, name='smol135-1')
146
+ ```
147
+
148
+ `manage.py` and `infer-stack` are aliases for the same entry point;
149
+ shell examples below use `infer-stack`.
150
+
151
+ ## Operating the rendered Compose stack
152
+
153
+ `ps` and `logs` read the backend directly; `stack` wraps `docker compose` on
154
+ the rendered project, so you never `cd` into it or repeat `-f`/`--env-file`.
155
+
156
+ ```bash
157
+ infer-stack ps # engines, gateway, UI
158
+ infer-stack ps -a # include exited instances
159
+ infer-stack logs -f <endpoint> # follow whatever serves an endpoint
160
+ infer-stack logs --tail 200 litellm # the gateway's backlog
161
+ infer-stack logs -f --raw litellm # full LiteLLM tracebacks
162
+ infer-stack stack restart open-webui # docker compose restart
163
+ infer-stack stack stop # stop everything (no remove)
164
+ infer-stack stack start # start it back up
165
+ infer-stack stack pull # refresh images
166
+ infer-stack stack compose -- ps --format json # any other docker compose command
167
+ ```
168
+
169
+ `logs` accepts a service or pod name, a container id prefix, a deployment id
170
+ or an endpoint alias. Interactive `infer-stack logs -f` compacts only
171
+ explicitly registered, known-noisy LiteLLM traceback shapes; unknown
172
+ tracebacks pass through unchanged. Redirected or piped output stays raw, and
173
+ `--raw` disables compaction in an interactive follow. `--no-color` drops the
174
+ name-prefix colors. The TUI uses the same compactor.
175
+
176
+ Ollama tags are pulled into the daemon on the first `acquire` of an endpoint
177
+ that serves them. `stack compose` reaches the daemon for anything else, e.g.
178
+ `infer-stack stack compose -- exec ollama-local-ollama ollama list`.
179
+
180
+ On the KubeAI backend `ps` and `logs` read pods, and the `stack` Compose verbs
181
+ act on the gateway's Compose project on this host.
182
+
183
+ ## Inspect an endpoint before running it
184
+
185
+ ```bash
186
+ infer-stack catalog show <endpoint>
187
+ infer-stack acquire <endpoint> --no-apply # declare + write the compose project; start nothing
188
+ infer-stack paths leasing # where docker-compose.yml landed
189
+ infer-stack apply # start it (or `release --all` to discard)
190
+ ```
191
+
192
+ ## Catalog model
193
+
194
+ `catalog.yaml` is the one user-edited description of what can run. Its
195
+ sections:
196
+
197
+ * `models`: weight sources (`hf://org/name`, with optional `revision`,
198
+ `quantization`, `dtype`);
199
+ * `endpoints`: served API names. Each picks an `engine` (`vllm` or `ollama`),
200
+ a `model` (a `models` key, or an Ollama tag), and optionally `runtime`
201
+ (vLLM settings), `protocol`, `placement`, `sharing` and `reclaim`;
202
+ * `runtime_hosts`: Ollama daemons, each with its GPUs and daemon settings;
203
+ * `bundles`: named lists of endpoints to acquire together.
204
+
205
+ An endpoint can instead name a server that already runs elsewhere, with
206
+ `external: {api_base, model, api_key_env}` in place of `engine`/`model`.
207
+ It has no lease: `infer-stack access` (or `run`) publishes its route on the
208
+ LiteLLM front door, and the client asks for the alias through the same base
209
+ URL as any other endpoint. Moving an alias between a managed runtime and an
210
+ external server changes neither the alias nor the workflow:
211
+
212
+ ```bash
213
+ infer-stack env REMOTE_QWEN_KEY=sk-... # the upstream's key, by name
214
+ infer-stack access qwen-remote --env-file qwen.env
215
+ source qwen.env # OPENAI_BASE_URL, OPENAI_API_KEY, ...
216
+ ```
217
+
218
+ `api_base` is resolved from inside the gateway's container or pod:
219
+ `localhost` there is the gateway itself, so a server on the same host is
220
+ reached through the Docker bridge address (e.g. `http://172.17.0.1:8000/v1`).
221
+ See [docs/planning/external-endpoints.md](docs/planning/external-endpoints.md).
222
+
223
+ ```yaml
224
+ models:
225
+ smol135:
226
+ source: hf://HuggingFaceTB/SmolLM2-135M-Instruct
227
+
228
+ endpoints:
229
+ smol135-1:
230
+ engine: vllm
231
+ model: smol135
232
+ runtime: {max_model_len: 8192}
233
+ chat:
234
+ engine: ollama
235
+ host: local-ollama
236
+ model: qwen3.5:4b
237
+
238
+ runtime_hosts:
239
+ local-ollama:
240
+ engine: ollama
241
+ placement: {gpu_indices: [0]}
242
+ settings: {keep_alive: 30m}
243
+
244
+ bundles:
245
+ both: [smol135-1, chat]
246
+ ```
247
+
248
+ Edit it with `infer-stack catalog model|endpoint|host|bundle add …`, or by
249
+ hand with `infer-stack catalog edit`; `infer-stack catalog validate` checks it.
250
+ The schema reference is the docstring of `infer_stack/leasing/catalog.py`.
251
+
252
+ Shapes the stack renders:
253
+
254
+ ```text
255
+ Open WebUI -> LiteLLM -> vLLM / Ollama # the default
256
+ \-> an external OpenAI-compatible server (external:)
257
+ Open WebUI -> vLLM / Ollama # config set litellm false
258
+ ```
259
+
260
+ The named stack profiles of earlier releases (`setup`, `switch`,
261
+ `--profile`) are gone; see
262
+ [docs/stack-graph-profiles.md](docs/stack-graph-profiles.md).
263
+
264
+ ## Where config and rendered artifacts live
265
+
266
+ `infer-stack` follows XDG basedir conventions, so the directory you invoke it
267
+ from never changes which config it reads or where it writes. There are two
268
+ roots:
269
+
270
+ | What | Default location | How to relocate |
271
+ | --- | --- | --- |
272
+ | `settings.yaml`, `catalog.yaml` | `~/.config/infer_stack/` (resp. `$XDG_CONFIG_HOME`) | `--config-dir` or `INFER_STACK_CONFIG_DIR` |
273
+ | **Everything generated**: `leasing/` (the ledger, the compose project, its `.env`) and the bind-mounted state (`hf-cache/`, `vllm-cache/`, `open-webui/`, `ollama/`, …) | `~/.local/share/infer_stack/` (resp. `$XDG_DATA_HOME`) | `config set data_dir <path>`, `--data-dir` or `INFER_STACK_DATA_DIR` |
274
+
275
+ `infer-stack paths` prints every resolved path and whether it exists.
276
+
277
+ The data dir **relocates the one infer-stack installation controlling a
278
+ host/backend; it does not create an isolated second installation**. Do not run
279
+ controllers from multiple config/data roots against the same Docker host or
280
+ Kubernetes namespace. See the
281
+ [single-owner limitation](docs/planning/known-limitations.md#one-control-plane-per-host-or-backend-namespace).
282
+
283
+ ```bash
284
+ # Persist it once; later commands read it from settings.yaml.
285
+ infer-stack config set data_dir /data/service/docker/infer-stack
286
+
287
+ # Or keep both roots in a checkout for an ad-hoc experiment.
288
+ export INFER_STACK_CONFIG_DIR=$PWD/cfg INFER_STACK_DATA_DIR=$PWD/stack
289
+ infer-stack paths
290
+ ```
291
+
292
+ `--config-dir` / `--data-dir` are accepted by every subcommand, after the
293
+ subcommand name.
294
+
295
+ ## Constraining placement to specific GPUs
296
+
297
+ ```bash
298
+ # Only place onto GPU 1 (e.g. GPU 0 is running a display).
299
+ infer-stack acquire <endpoint> --allowed-gpus 1
300
+
301
+ # Or confine a TP=2 endpoint to physical GPUs 1 and 3.
302
+ infer-stack acquire <tp2-endpoint> --allowed-gpus 1,3
303
+ ```
304
+
305
+ `--allowed-gpus` (or `INFER_STACK_ALLOWED_GPUS=1,3`) filters the detected
306
+ inventory before placement for that call only. Real indices are preserved, so
307
+ the rendered compose stack pins `device_ids` to those exact GPUs. The durable
308
+ forms live in data:
309
+
310
+ * `placement: {gpu_indices: [1]}` on a vLLM endpoint pins it exactly (the list
311
+ length must equal tp×pp×dp); Ollama daemons pin through their
312
+ `runtime_hosts` entry;
313
+ * `placement: {min_vram_gib: 24}` makes smaller GPUs ineligible;
314
+ * `config set skip_display_gpus true` (or `--skip-display-gpus`) leaves the
315
+ GPU driving a monitor free.
316
+
317
+ ## Demos / integration recipes
318
+
319
+ The user manual under [docs/source/manual/](docs/source/manual/) has two
320
+ walkthroughs on the current CLI:
321
+ [the Ollama + Open WebUI tutorial](docs/source/manual/ollama-openwebui-tutorial.md)
322
+ and [the leasing demo](docs/source/manual/leasing-demo.md) (standing service,
323
+ Open WebUI, several models side by side).
324
+
325
+ ---
326
+
327
+ ## Backend 1: Compose
328
+
329
+ Use Compose for single-host serving. It runs vLLM and Ollama engines, mixed
330
+ freely, behind the optional gateway and UI.
331
+
332
+ ### Getting started
333
+
334
+ Prerequisite: Docker and the `docker compose` plugin.
335
+
336
+ ```bash
337
+ infer-stack config init --backend compose
338
+ infer-stack doctor --gpu
339
+ infer-stack catalog init
340
+
341
+ # A vLLM endpoint.
342
+ infer-stack catalog model add smol135 --source hf://HuggingFaceTB/SmolLM2-135M-Instruct
343
+ infer-stack catalog endpoint add --model smol135 # -> smol135-1
344
+ infer-stack acquire smol135-1
345
+
346
+ # An Ollama endpoint, on a daemon pinned to GPU 0.
347
+ infer-stack catalog host add local-ollama --engine ollama --gpu 0
348
+ infer-stack catalog endpoint add chat --engine ollama --host local-ollama --model qwen3.5:4b
349
+ infer-stack acquire chat
350
+ ```
351
+
352
+ `infer-stack catalog suggest --apply` fills the catalog with endpoints sized
353
+ for the detected GPUs instead.
354
+
355
+ ### Test that it is responding
356
+
357
+ With LiteLLM enabled, every endpoint is reachable by its alias at:
358
+
359
+ ```text
360
+ http://127.0.0.1:14042/v1
361
+ ```
362
+
363
+ `acquire` already blocks until the endpoint returns a real generation through
364
+ that front door, which is stronger than Docker's container health. After
365
+ `acquire --no-wait`, block separately; to check again later, send one request:
366
+
367
+ ```bash
368
+ infer-stack wait smol135-1
369
+ infer-stack test smol135-1
370
+ infer-stack test chat --prompt "Name three colors." --max-tokens 32
371
+ ```
372
+
373
+ Clients read the front door and the managed key from the env file:
374
+
375
+ ```bash
376
+ export OPENAI_BASE_URL=$(infer-stack env OPENAI_BASE_URL)
377
+ export OPENAI_API_KEY=$(infer-stack env LITELLM_MASTER_KEY)
378
+ ```
379
+
380
+ or get both, plus per-endpoint names, from
381
+ `infer-stack acquire <endpoint> --env-file lease.env`.
382
+
383
+ To replace the gateway's master key (refused while leases are active; the
384
+ gateway restarts, and clients must fetch the key again):
385
+
386
+ ```bash
387
+ infer-stack secrets rotate
388
+ ```
389
+
390
+ ### Stop it
391
+
392
+ ```bash
393
+ infer-stack release --all # drop every lease
394
+ infer-stack release --all --evict # ...and stop the engines now
395
+ infer-stack clean -f # no leases, nothing on a GPU; the gateway stays
396
+ infer-stack stack down # docker compose down, bypassing the ledger
397
+ ```
398
+
399
+ After a plain `release`, `keep-warm` endpoints (the default `reclaim` policy)
400
+ stay loaded until another lease needs their GPUs or `infer-stack evict` stops
401
+ them. `stack down` releases no lease, so the next `apply` or `acquire` brings
402
+ leased models back.
403
+
404
+ All state is bind-mounted from the data dir, so none of these delete it,
405
+ including `stack down --volumes`. For a destructive reset, remove the
406
+ directories `infer-stack paths` lists.
407
+
408
+ ### Open WebUI authentication
409
+
410
+ Open WebUI runs with `WEBUI_AUTH=False`: no login screen, and anyone who can
411
+ reach port 13000 gets the UI. No setting changes that. Keep the host on a
412
+ trusted network, or run without the UI (`config set ui false`).
413
+
414
+ ### Reverse proxy
415
+
416
+ An optional nginx service publishes one HTTP port with the UI at `/` and the
417
+ API at `/v1`. It needs the LiteLLM gateway.
418
+
419
+ ```bash
420
+ infer-stack config set reverse_proxy true # port 80
421
+ infer-stack config set reverse_proxy '{enabled: true, port: 8080}'
422
+ ```
423
+
424
+ `acquire --reverse-proxy` turns it on for one call. Add `config_path:
425
+ /path/to/nginx.conf` to the block to mount your own config at
426
+ `/etc/nginx/conf.d/default.conf` instead of the generated one.
427
+
428
+ It does no TLS and no authentication. Only the one port is published and no
429
+ certificate is mounted, so terminate TLS in a proxy in front of it. The TLS
430
+ and LDAP settings of the pre-leasing profiles no longer exist.
431
+
432
+ ### Persistent state and database layout
433
+
434
+ Everything lives under the data dir (`infer-stack paths`):
435
+
436
+ * `open-webui/`: Open WebUI's data directory (accounts, chats, settings),
437
+ mounted at `/app/backend/data`;
438
+ * `postgres-litellm/`: LiteLLM's route store, rendered only with
439
+ `config set dynamic_routing true`;
440
+ * `ollama/`: the Ollama model store, mounted at `/root/.ollama`;
441
+ * `hf-cache/`, `vllm-cache/`, `torch-cache/`, `triton-cache/`, `cuda-cache/`:
442
+ vLLM weights and compile caches (see
443
+ [docs/persistent-caches-and-warm-restarts.md](docs/persistent-caches-and-warm-restarts.md));
444
+ * `runtime/`: directories an endpoint's `runtime.mounts` asks for;
445
+ * `leasing/`: the ledger and the rendered compose project.
446
+
447
+ Open WebUI chat history is not tied to the models currently served, so old
448
+ chats may name aliases the gateway no longer advertises. That is expected.
449
+
450
+ ### Custom .env values are preserved
451
+
452
+ The compose project's `.env` holds the managed secrets (`LITELLM_MASTER_KEY`,
453
+ `HF_TOKEN`, …). `infer-stack env KEY=VALUE` merges a value into it, and keys
454
+ you add are kept across renders. Compose uses the file for interpolation, so a
455
+ key reaches a container only when the rendered service references it (as
456
+ `HF_TOKEN` does for vLLM). Set `HF_TOKEN` before the first `acquire` of a
457
+ gated model.
458
+
459
+ ### Switching models
460
+
461
+ There is no single active model to switch. `acquire` another endpoint and it
462
+ runs beside the first; release the first when you are done:
463
+
464
+ ```bash
465
+ infer-stack acquire smol135-1
466
+ infer-stack acquire chat # now both are served
467
+ infer-stack leases # find smol135-1's lease id
468
+ infer-stack release <lease-id>
469
+ ```
470
+
471
+ The gateway carries a route for every catalog endpoint, so adding or removing
472
+ a model does not recreate LiteLLM, and Open WebUI stays up. When GPUs are
473
+ short, `acquire` fails fast; `--queue` waits for a GPU instead, and idle
474
+ `keep-warm` deployments are evicted to make room. vLLM containers are named
475
+ after the served alias (`vllm-<alias>`), so `docker ps` and
476
+ `infer-stack logs vllm-<alias>` identify them.
477
+
478
+ ### Protocol modes for base vs. instruct models
479
+
480
+ An endpoint's `protocol` is `chat` (the default) or `completions`. It decides
481
+ which surface the readiness probe and `infer-stack test` use, so a base model
482
+ without a chat template must declare `completions` or its `acquire` never sees
483
+ a ready generation:
484
+
485
+ ```bash
486
+ infer-stack catalog model add pythia-160m --source hf://EleutherAI/pythia-160m
487
+ infer-stack catalog endpoint add --model pythia-160m --protocol completions
488
+ infer-stack acquire pythia-160m-1
489
+ infer-stack test pythia-160m-1 # hits /v1/completions
490
+ ```
491
+
492
+ The gateway forwards `/v1/completions` unchanged, so evaluation clients that
493
+ need exact prompt control should call it directly. Open WebUI is a chat UI and
494
+ sends chat requests, which a base model cannot answer.
495
+
496
+ ### Images with their own launcher
497
+
498
+ Some images wrap vLLM in their own launcher and are configured through
499
+ environment variables rather than `vllm serve` flags. Describe that in the
500
+ endpoint's `runtime`; infer-stack has no model-specific code for it:
501
+
502
+ ```yaml
503
+ runtime:
504
+ image: example.org/my-vllm-launcher:1.0
505
+ max_model_len: 65536
506
+ gpu_memory_utilization: 0.93
507
+ command: [single] # replaces `vllm serve MODEL <flags>`
508
+ env: # container environment
509
+ PORT: '{port}'
510
+ MAX_LEN: '{max_model_len}' # filled from the field above
511
+ GPU_UTIL: '{gpu_memory_utilization}'
512
+ EXTRA_ARGS: '--served-model-name={served_model_name}'
513
+ mounts: # persisted under the runtime data dir
514
+ /app/models: my-launcher/models
515
+ ```
516
+
517
+ Changing the launcher's mode is a data edit, in the catalog or the TUI's
518
+ endpoint editor. The HyperQwen suggestion (see "Related work") is a worked
519
+ example: `catalog suggest` uses compute capability and VRAM to emit the serving
520
+ profiles appropriate to the detected hardware class (and exact name gates only
521
+ for measured exceptions). The hardware check happens only while suggesting;
522
+ the resulting catalog holds ordinary explicit runtime data, so `apply`/`acquire`
523
+ never retunes an endpoint after the fact.
524
+
525
+ - `{max_model_len}`, `{gpu_memory_utilization}`, `{served_model_name}` and
526
+ `{port}` are filled in from the endpoint, so a launcher that takes them
527
+ through its own variables stays in step when the fields change.
528
+ - Env values are written as strings (`true`/`false` for booleans), and `$`
529
+ is literal. `HF_TOKEN`, `VLLM_ATTENTION_BACKEND`, `CUDA_VISIBLE_DEVICES` and
530
+ `NVIDIA_VISIBLE_DEVICES` are infer-stack's and are refused.
531
+ - `extra_args` stay what they were: flags appended to the stock `vllm serve`
532
+ command, after infer-stack's own, so vLLM keeps the extra value for a
533
+ repeated flag. Repeating a flag infer-stack acts on (served name, parallel
534
+ sizes, `--max-model-len`) is refused; with `command`, pass flags through the
535
+ launcher instead.
536
+ - All of these are deployment identity: endpoints that launch differently
537
+ never share a process.
538
+ - `env` works on both backends (on KubeAI it becomes the Model's `spec.env`).
539
+ `command` and `mounts` are Compose only; KubeAI refuses an endpoint with
540
+ either rather than serve stock vLLM in its place.
541
+
542
+ ### Reasoning / thinking models
543
+
544
+ vLLM separates a reasoning trace from the answer when it is started with a
545
+ reasoning parser. Pass the flag through `runtime.extra_args`:
546
+
547
+ ```yaml
548
+ endpoints:
549
+ qwen3-think:
550
+ engine: vllm
551
+ model: qwen3-0.6b
552
+ runtime:
553
+ extra_args: [--reasoning-parser=qwen3]
554
+ ```
555
+
556
+ The parser name depends on the model family and the vLLM version (`vllm serve
557
+ --help` lists them). To test end to end:
558
+
559
+ ```bash
560
+ infer-stack test qwen3-think --prompt "Think step by step: 17*23" --max-tokens 512
561
+
562
+ # Streaming, through the gateway:
563
+ curl -N "$(infer-stack env OPENAI_BASE_URL)/chat/completions" \
564
+ -H "Authorization: Bearer $(infer-stack env LITELLM_MASTER_KEY)" \
565
+ -H 'Content-Type: application/json' \
566
+ -d '{"model":"qwen3-think","stream":true,
567
+ "messages":[{"role":"user","content":"Think step by step: 17*23"}]}'
568
+ ```
569
+
570
+ In Open WebUI, reasoning shows up best with streaming enabled in the
571
+ chat settings.
572
+
573
+ ---
574
+
575
+ ## Backend 2: KubeAI
576
+
577
+ `--backend kubeai` runs the same leasing verbs against a Kubernetes cluster
578
+ running [KubeAI](https://www.kubeai.org): the same catalog, ledger, TTLs,
579
+ env file and TUI, with `Model` custom resources in place of compose
580
+ services and the cluster scheduler in place of the local GPU planner. The
581
+ LiteLLM gateway still fronts everything, so a card sees one `OPENAI_BASE_URL`,
582
+ the managed key and the endpoint alias on either backend.
583
+
584
+ * Creating/joining a cluster and the distribution boundary:
585
+ [docs/cluster-setup.md](docs/cluster-setup.md).
586
+ * KubeAI setup, settings and semantics:
587
+ [docs/kubeai-backend.md](docs/kubeai-backend.md).
588
+ * What matches Compose, what does not yet, and what is deliberate:
589
+ [docs/backend-parity.md](docs/backend-parity.md); the plan to close the
590
+ rest: [docs/planning/backend-parity-roadmap.md](docs/planning/backend-parity-roadmap.md).
591
+
592
+ The short version:
593
+
594
+ ```bash
595
+ infer-stack kube inventory
596
+ infer-stack kube doctor
597
+ # New local cluster / NVIDIA plugin + GFD prerequisites (review before applying):
598
+ infer-stack kube bootstrap --provider=k3s
599
+ sudo -v
600
+ infer-stack kube bootstrap --provider=k3s --apply
601
+ # Automatic profiles from Kubernetes node labels; no temporary values file needed:
602
+ infer-stack kube install
603
+ infer-stack kube install --apply
604
+ infer-stack kube doctor
605
+ infer-stack catalog suggest --backend kubeai
606
+ infer-stack config set backend kubeai
607
+ infer-stack config set kubeai_gateway cluster
608
+ infer-stack doctor --backend kubeai
609
+ infer-stack acquire <endpoint> --ttl 2h --env-file lease.env --yes
610
+ ```
611
+
612
+ If the workstation already has a Compose ledger/configuration that you plan to
613
+ return to, keep that authority intact rather than switching backend kinds in
614
+ the same ledger. The cluster setup guide documents the temporary test pattern:
615
+ use `INFER_STACK_BACKEND=kubeai` with a separate KubeAI data root, and use
616
+ `infer-stack kube node detach/attach` to hand each physical GPU host between
617
+ Kubernetes scheduling and direct Compose use without uninstalling or rejoining
618
+ the node. See [docs/cluster-setup.md](docs/cluster-setup.md).
619
+
620
+ ### Switching an existing recovery ledger to KubeAI
621
+
622
+ Stopping the Compose realization retains its recovery backend and historical
623
+ ledger. Backend kinds require separate recovery epochs. After releasing every
624
+ old lease and stopping the old runtime:
625
+
626
+ ```bash
627
+ infer-stack release --all --backend compose
628
+ infer-stack stack down --backend compose
629
+ infer-stack config set backend kubeai
630
+ infer-stack status # configured kubeai / active recovery compose
631
+ infer-stack ledger rotate # verify quiescence and preview the archive
632
+ infer-stack ledger rotate --yes # archive old history; initialize KubeAI epoch
633
+ infer-stack ledger archives # archive paths + history inspection commands
634
+ infer-stack acquire <endpoint> --yes
635
+ ```
636
+
637
+ Rotation refuses active leases, remaining old runtime objects, and unknown
638
+ runtime state. It does not change configuration/catalogs or silently tear down
639
+ workloads. Old leases, deployments and the recovery profile remain in SQLite
640
+ archives under `<ledger-directory>/archives/`. Rotation is transactional and
641
+ safe to retry; a matching recovery backend makes the command a no-op. `gc`
642
+ reports backend mismatches with transition instructions; `gc --forget` remains
643
+ a history-only operation.
644
+
645
+ ### Kubernetes cluster setup and KubeAI integration
646
+
647
+ [docs/cluster-setup.md](docs/cluster-setup.md) describes server/worker setup.
648
+ `kube inventory` and `kube status` report structured facts; `--json` includes
649
+ probe errors without discarding other discoveries. Inventory and catalog
650
+ suggestions work before the KubeAI chart or namespace exists.
651
+
652
+ `kube doctor` checks Kubernetes, NVIDIA scheduling/discovery, and KubeAI in
653
+ dependency order. Top-level `doctor --backend kubeai` remains the operational
654
+ preflight for acquiring work. If Helm installation succeeds but the configured
655
+ KubeAI API URL is unavailable, doctor reports that remaining routing issue.
656
+ Set `kubeai_base_url` to a reachable OpenAI URL for the installed service.
657
+
658
+ `kube bootstrap` defaults to a read-only plan; `--apply` (or `--yes`) authorizes
659
+ host/cluster changes. K3s is the implemented provisioning provider; inventory,
660
+ installation and the backend work with any Kubernetes distribution. Bootstrap
661
+ always provisions/reconciles the local K3s server, preserves unrelated selected
662
+ contexts and kubeconfigs, installs Helm if missing,
663
+ and reconciles NVIDIA device plugin **0.17.1** with GPU Feature Discovery.
664
+ Install the host NVIDIA driver and container toolkit first. K3s discovers the
665
+ installed runtime on startup; bootstrap restarts local K3s only when runtime
666
+ discovery needs repair. It waits for Ready nodes, GPU allocation and GFD labels, and verifies unknown
667
+ GPU-node runtime handlers with small node-specific canary pods. Root admin
668
+ credentials stay `0600`; a private `0600` copy lives at
669
+ `~/.kube/infer-stack-k3s.yaml`. Existing default configs are preserved. Select the
670
+ local server explicitly with `export KUBECONFIG=~/.kube/infer-stack-k3s.yaml`
671
+ before inventory/install; rerun bootstrap to refresh copied certificates.
672
+
673
+ `kube install` shows inferred Helm values; `--apply` uses Helm upgrade/install
674
+ and checks readiness afterward. `--dry-run`/`--plan` forces read-only behavior.
675
+ `--namespace`, `--release`, `--chart`, `--version` and `--values` allow overrides.
676
+ Existing named profiles and custom values are preserved; `HF_TOKEN` overrides
677
+ the chart token using a temporary protected file. Generated public values live
678
+ at `<data>/generated/kube/kubeai-values.yaml`.
679
+
680
+ `kube setup` is a deprecated alias of `kube install`; it no longer owns a
681
+ separate prerequisite workflow. `kube k3s bootstrap` provisions only the local
682
+ server, `join` verifies requested worker membership, and `status` reports local
683
+ membership independently of any stale selected admin context. Setup scripts are compatibility wrappers around the
684
+ package commands.
685
+
686
+ ### Debugging serving failures
687
+
688
+ `infer-stack kube status` summarizes live Kubernetes state. `infer-stack acquire`
689
+ reports pod failures such as `ImagePullBackOff`, `Unschedulable`, and engine
690
+ crashes. Use `infer-stack ps` and `infer-stack logs -f <endpoint>` to inspect
691
+ model serving. If `/models` works but completions return 404, check that clients
692
+ use the gateway alias; direct KubeAI requests use the Model name from the env
693
+ file (`INFER_STACK_ENDPOINT_*`). Missing `libcuda.so.1` usually indicates that
694
+ the model profile did not request GPUs or select the correct runtime.
695
+
696
+ ---
697
+
698
+ ## Which backend should I start with?
699
+
700
+ **Compose** when one workstation is enough: it is the fastest path to a
701
+ working server, everything it renders is a file you can read, and it needs
702
+ only Docker.
703
+
704
+ **KubeAI** when the models must run on more than one machine, or a cluster
705
+ already exists. It is Compose plus a scheduler: the same catalog, verbs and
706
+ env file, with a cluster and `resourceProfiles` supplied. Expect more
707
+ first-request overhead (pod creation, image pull, model load).
708
+
709
+ A catalog written for Compose runs on KubeAI, with two exceptions: ollama
710
+ endpoints and custom container launches (`runtime.command` / `mounts`),
711
+ which stay Compose-only. [docs/backend-parity.md](docs/backend-parity.md)
712
+ has the full matrix.
713
+
714
+ ## vLLM startup caches
715
+
716
+ Generated Compose mounts persist Hugging Face, vLLM, PyTorch/TorchInductor,
717
+ Triton, and CUDA JIT caches. Warm starts avoid redownloading and redoing many
718
+ compile/JIT steps, but a vLLM model swap still creates a new engine process and
719
+ must reload weights into GPU memory.
720
+
721
+ ### Diagnosing readiness
722
+
723
+ `docker compose` health only means a container-level healthcheck passed. It
724
+ is not "the routed model answers a request through the front door", which is
725
+ what `acquire` and `wait` check: a model swap starts a new engine process,
726
+ and LiteLLM stays up while returning upstream connection errors until vLLM
727
+ has loaded the weights.
728
+
729
+ ```bash
730
+ infer-stack acquire <endpoint> --yes # waits for a real generation
731
+ infer-stack wait <endpoint> # after acquire --no-wait
732
+ infer-stack test <endpoint> # one generation through the gateway
733
+ infer-stack status # desired vs running, per deployment
734
+ infer-stack logs <service> --tail 80 # the engine or gateway log
735
+ ```
736
+
737
+ A crash-looping engine fails the acquire at once with its error quoted, so
738
+ the timeout is only for a model that is loading. Reading Docker's own
739
+ signals: `litellm exited with code 137` is a SIGKILL (an OOM kill or a forced
740
+ replacement), whereas LiteLLM returning HTTP 500 with `Cannot connect to host
741
+ vllm-*` means LiteLLM is running and its upstream vLLM is not ready yet.
742
+
743
+ ### Join and accept GPU workers
744
+
745
+ `infer-stack kube k3s onboard namek --server=... --token-file=... --kubeconfig=...`
746
+ plans a local GPU worker join and node-specific acceptance; add `--apply` to
747
+ execute it. Host NVIDIA drivers/toolkit are prerequisites. It infers the server
748
+ version, verifies every locally detected device, then checks actual one-GPU
749
+ KubeAI placement and generation on this worker. Admin credentials are explicit
750
+ private files, separate from any selected EKS context.
751
+
752
+ For an already joined worker, run `infer-stack kube node test namek
753
+ --expected-gpus=1 --namespace=default --apply` from the control plane. The same
754
+ surface accepts two-device `yardrat` and four-device `aiq-gpu2`; mixed products
755
+ are reported separately. Tests preserve leases and unrelated Models, remove
756
+ their temporary resources, and retain a reusable node profile. See the
757
+ [worker onboarding instructions](docs/cluster-setup.md#join-and-test-a-gpu-worker-in-one-reviewed-operation)
758
+ for plan/apply, credentials, interrupted cleanup and heterogeneous-node limits.
759
+
760
+ ### Kubernetes dashboard
761
+
762
+ `infer-stack tui` provides a **Cluster** tab in the expanded runtime pane on the
763
+ KubeAI backend. It shows Ready/scheduling state, GPU allocation and scheduled
764
+ requests across namespaces, GFD labels, runtime evidence, managed Models and
765
+ KubeAI/NVIDIA pod startup/restart state. GPU requests are scheduling facts, not
766
+ live GPU utilization; the system pane explicitly describes the local host.
767
+
768
+ Cluster monitoring uses three batched API reads, cached for at least 15 seconds
769
+ and polled only while the tab is visible. Slow probes cannot overlap or block
770
+ keyboard input; **Refresh cluster** forces a sample, and **Doctor** runs the
771
+ shared detailed checks on demand. Pod samples also feed the Instances view.
772
+
773
+ Select a node for **Detach node** or **Attach node**. Each previews its workload
774
+ and scheduling state and requires confirmation; detach drains workloads and
775
+ emptyDir data, while attach confirms local Compose workloads have stopped.
776
+ These reuse `kube node detach/attach` and preserve cluster membership. The
777
+ Control tab provides **Apply** through the leasing controller and confirmed
778
+ **Down** for managed Models and the gateway. Leases remain after Down. The
779
+ KubeAI endpoint editor selects resource profiles; Kubernetes chooses node/GPU
780
+ placement.