infer-stack 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- infer_stack/__init__.py +2 -0
- infer_stack/backends/__init__.py +7 -0
- infer_stack/backends/compose_renderer.py +243 -0
- infer_stack/backends/kubeai_renderer.py +202 -0
- infer_stack/benchmark.py +38 -0
- infer_stack/catalog.py +438 -0
- infer_stack/cli/__init__.py +169 -0
- infer_stack/cli/__main__.py +4 -0
- infer_stack/cli/commands_profile.py +467 -0
- infer_stack/cli/commands_runtime.py +719 -0
- infer_stack/cli/commands_smoke.py +691 -0
- infer_stack/cli/compose.py +755 -0
- infer_stack/cli/context.py +471 -0
- infer_stack/cli/options.py +134 -0
- infer_stack/cli/probes.py +178 -0
- infer_stack/config.py +450 -0
- infer_stack/contracts.py +223 -0
- infer_stack/diff_prompt.py +117 -0
- infer_stack/docker_utils.py +230 -0
- infer_stack/env_utils.py +97 -0
- infer_stack/experimental/model_catalog_discover.py +1155 -0
- infer_stack/experimental/model_memory_estimator.py +1264 -0
- infer_stack/experimental/stress_test_long_context.py +397 -0
- infer_stack/hardware.py +70 -0
- infer_stack/kubeai_ops.py +76 -0
- infer_stack/paths.py +87 -0
- infer_stack/profile_runtime.py +46 -0
- infer_stack/renderer.py +19 -0
- infer_stack/resolver.py +1092 -0
- infer_stack/templates/default-models.yaml +674 -0
- infer_stack/templates/default-ollama-models.yaml +31 -0
- infer_stack/templates/default-profiles.yaml +1731 -0
- infer_stack/templates/default-vllm-models.yaml +714 -0
- infer_stack/templates/docker-compose.yml.j2 +430 -0
- infer_stack/templates/litellm_config.yaml.j2 +44 -0
- infer_stack/templates/nginx.conf.j2 +84 -0
- infer_stack/tuning.py +3 -0
- infer_stack/validator.py +314 -0
- infer_stack/verification.py +46 -0
- infer_stack-0.6.0.dist-info/METADATA +1034 -0
- infer_stack-0.6.0.dist-info/RECORD +44 -0
- infer_stack-0.6.0.dist-info/WHEEL +5 -0
- infer_stack-0.6.0.dist-info/entry_points.txt +2 -0
- infer_stack-0.6.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,430 @@
|
|
|
1
|
+
services:
|
|
2
|
+
{% if lock.deployment.frontends.open_webui.enabled %}
|
|
3
|
+
postgres-open-webui:
|
|
4
|
+
image: {{ lock.deployment.images.postgres }}
|
|
5
|
+
container_name: postgres-open-webui
|
|
6
|
+
restart: unless-stopped
|
|
7
|
+
environment:
|
|
8
|
+
POSTGRES_DB: ${OPENWEBUI_POSTGRES_DB}
|
|
9
|
+
POSTGRES_USER: ${OPENWEBUI_POSTGRES_USER}
|
|
10
|
+
POSTGRES_PASSWORD: ${OPENWEBUI_POSTGRES_PASSWORD}
|
|
11
|
+
volumes:
|
|
12
|
+
- {{ lock.deployment.state.postgres_open_webui }}:/var/lib/postgresql/data
|
|
13
|
+
healthcheck:
|
|
14
|
+
test: ["CMD-SHELL", "pg_isready -U $$POSTGRES_USER -d $$POSTGRES_DB"]
|
|
15
|
+
interval: 10s
|
|
16
|
+
timeout: 5s
|
|
17
|
+
retries: 10
|
|
18
|
+
|
|
19
|
+
{% endif %}
|
|
20
|
+
{% if lock.deployment.gateways.litellm.enabled %}
|
|
21
|
+
postgres-litellm:
|
|
22
|
+
image: {{ lock.deployment.images.postgres }}
|
|
23
|
+
container_name: postgres-litellm
|
|
24
|
+
restart: unless-stopped
|
|
25
|
+
environment:
|
|
26
|
+
POSTGRES_DB: ${LITELLM_POSTGRES_DB}
|
|
27
|
+
POSTGRES_USER: ${LITELLM_POSTGRES_USER}
|
|
28
|
+
POSTGRES_PASSWORD: ${LITELLM_POSTGRES_PASSWORD}
|
|
29
|
+
volumes:
|
|
30
|
+
- {{ lock.deployment.state.postgres_litellm }}:/var/lib/postgresql/data
|
|
31
|
+
healthcheck:
|
|
32
|
+
test: ["CMD-SHELL", "pg_isready -U $$POSTGRES_USER -d $$POSTGRES_DB"]
|
|
33
|
+
interval: 10s
|
|
34
|
+
timeout: 5s
|
|
35
|
+
retries: 10
|
|
36
|
+
|
|
37
|
+
{% endif %}
|
|
38
|
+
{% if lock.deployment.providers.ollama.enabled %}
|
|
39
|
+
ollama:
|
|
40
|
+
image: {{ lock.deployment.images.ollama }}
|
|
41
|
+
container_name: ollama
|
|
42
|
+
restart: unless-stopped
|
|
43
|
+
{% if lock.deployment.providers.ollama.labels %}
|
|
44
|
+
labels:
|
|
45
|
+
{% for key, value in lock.deployment.providers.ollama.labels.items() %}
|
|
46
|
+
{{ key }}: {{ value|compose_quote }}
|
|
47
|
+
{% endfor %}
|
|
48
|
+
{% endif %}
|
|
49
|
+
{% if lock.deployment.providers.ollama.env_file %}
|
|
50
|
+
env_file:
|
|
51
|
+
{% for env_file in lock.deployment.providers.ollama.env_file %}
|
|
52
|
+
- {{ env_file }}
|
|
53
|
+
{% endfor %}
|
|
54
|
+
{% endif %}
|
|
55
|
+
{% if lock.deployment.providers.ollama.extra_hosts %}
|
|
56
|
+
extra_hosts:
|
|
57
|
+
{% for extra_host in lock.deployment.providers.ollama.extra_hosts %}
|
|
58
|
+
- {{ extra_host|compose_quote }}
|
|
59
|
+
{% endfor %}
|
|
60
|
+
{% endif %}
|
|
61
|
+
environment:
|
|
62
|
+
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
63
|
+
CUDA_VISIBLE_DEVICES: {{ (lock.deployment.providers.ollama.gpu_indices | join(','))|compose_quote }}
|
|
64
|
+
OLLAMA_HOST: {{ lock.deployment.providers.ollama.host|compose_quote }}
|
|
65
|
+
OLLAMA_KEEP_ALIVE: {{ lock.deployment.providers.ollama.keep_alive|compose_quote }}
|
|
66
|
+
OLLAMA_CONTEXT_LENGTH: {{ lock.deployment.providers.ollama.context_length|compose_quote }}
|
|
67
|
+
OLLAMA_NUM_PARALLEL: {{ lock.deployment.providers.ollama.num_parallel|compose_quote }}
|
|
68
|
+
OLLAMA_MAX_LOADED_MODELS: {{ lock.deployment.providers.ollama.max_loaded_models|compose_quote }}
|
|
69
|
+
OLLAMA_MAX_QUEUE: {{ lock.deployment.providers.ollama.max_queue|compose_quote }}
|
|
70
|
+
{% for key, value in lock.deployment.providers.ollama.extra_env.items() %}
|
|
71
|
+
{{ key }}: {{ value|compose_quote }}
|
|
72
|
+
{% endfor %}
|
|
73
|
+
volumes:
|
|
74
|
+
- {{ lock.deployment.state.ollama }}:/root/.ollama
|
|
75
|
+
{% for volume in lock.deployment.providers.ollama.extra_volumes %}
|
|
76
|
+
- {{ volume }}
|
|
77
|
+
{% endfor %}
|
|
78
|
+
{% if lock.deployment.providers.ollama.publish_port or lock.deployment.providers.ollama.additional_ports %}
|
|
79
|
+
ports:
|
|
80
|
+
{% if lock.deployment.providers.ollama.publish_port %}
|
|
81
|
+
- "127.0.0.1:{{ lock.deployment.ports.ollama or 11434 }}:11434"
|
|
82
|
+
{% endif %}
|
|
83
|
+
{% for port in lock.deployment.providers.ollama.additional_ports %}
|
|
84
|
+
- {{ port|compose_quote }}
|
|
85
|
+
{% endfor %}
|
|
86
|
+
{% endif %}
|
|
87
|
+
healthcheck:
|
|
88
|
+
test: ["CMD-SHELL", "ollama list >/dev/null 2>&1 || exit 1"]
|
|
89
|
+
interval: 30s
|
|
90
|
+
timeout: 10s
|
|
91
|
+
retries: 30
|
|
92
|
+
start_period: 60s
|
|
93
|
+
{% if lock.deployment.providers.ollama.gpus is not none %}
|
|
94
|
+
{{ lock.deployment.providers.ollama.gpus|compose_gpus }}
|
|
95
|
+
{% elif lock.deployment.providers.ollama.gpu_indices %}
|
|
96
|
+
deploy:
|
|
97
|
+
resources:
|
|
98
|
+
reservations:
|
|
99
|
+
devices:
|
|
100
|
+
- driver: nvidia
|
|
101
|
+
device_ids: [{% for idx in lock.deployment.providers.ollama.gpu_indices %}"{{ idx }}"{% if not loop.last %}, {% endif %}{% endfor %}]
|
|
102
|
+
capabilities: [gpu]
|
|
103
|
+
{% endif %}
|
|
104
|
+
|
|
105
|
+
{% endif %}
|
|
106
|
+
{% for name, svc in lock.deployment.providers.vllm.runtimes.items() %}
|
|
107
|
+
{{ svc.compose_service_name }}:
|
|
108
|
+
image: {{ lock.deployment.images.vllm }}
|
|
109
|
+
container_name: {{ svc.container_name or svc.compose_service_name }}
|
|
110
|
+
restart: unless-stopped
|
|
111
|
+
ipc: host
|
|
112
|
+
labels:
|
|
113
|
+
infer_stack.profile_name: {{ svc.profile_name|compose_quote }}
|
|
114
|
+
infer_stack.public_name: {{ svc.profile_public_name|compose_quote }}
|
|
115
|
+
infer_stack.logical_model_name: {{ svc.logical_model_name|compose_quote }}
|
|
116
|
+
infer_stack.protocol_mode: {{ svc.protocol_mode|compose_quote }}
|
|
117
|
+
{% for key, value in svc.labels.items() %}
|
|
118
|
+
{{ key }}: {{ value|compose_quote }}
|
|
119
|
+
{% endfor %}
|
|
120
|
+
{% if svc.env_file %}
|
|
121
|
+
env_file:
|
|
122
|
+
{% for env_file in svc.env_file %}
|
|
123
|
+
- {{ env_file }}
|
|
124
|
+
{% endfor %}
|
|
125
|
+
{% endif %}
|
|
126
|
+
{% if svc.extra_hosts %}
|
|
127
|
+
extra_hosts:
|
|
128
|
+
{% for extra_host in svc.extra_hosts %}
|
|
129
|
+
- {{ extra_host|compose_quote }}
|
|
130
|
+
{% endfor %}
|
|
131
|
+
{% endif %}
|
|
132
|
+
environment:
|
|
133
|
+
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
134
|
+
HF_TOKEN: ${HF_TOKEN}
|
|
135
|
+
VLLM_API_KEY: ${VLLM_BACKEND_API_KEY}
|
|
136
|
+
HF_HOME: /root/.cache/huggingface
|
|
137
|
+
HF_HUB_CACHE: /root/.cache/huggingface/hub
|
|
138
|
+
HUGGINGFACE_HUB_CACHE: /root/.cache/huggingface/hub
|
|
139
|
+
VLLM_CACHE_ROOT: /root/.cache/vllm
|
|
140
|
+
TORCH_HOME: /root/.cache/torch
|
|
141
|
+
TORCHINDUCTOR_CACHE_DIR: /root/.cache/torch/inductor
|
|
142
|
+
TRITON_CACHE_DIR: /root/.cache/triton
|
|
143
|
+
CUDA_CACHE_PATH: /root/.cache/nvidia/ComputeCache
|
|
144
|
+
{% if lock.deployment.vllm.enable_responses_api_store %}
|
|
145
|
+
VLLM_ENABLE_RESPONSES_API_STORE: "1"
|
|
146
|
+
{% endif %}
|
|
147
|
+
VLLM_LOGGING_LEVEL: {{ lock.deployment.vllm.logging_level|compose_quote }}
|
|
148
|
+
{% for key, value in svc.extra_env.items() %}
|
|
149
|
+
{{ key }}: {{ value|compose_quote }}
|
|
150
|
+
{% endfor %}
|
|
151
|
+
command:
|
|
152
|
+
- {{ svc.hf_model_id|compose_quote }}
|
|
153
|
+
- --host
|
|
154
|
+
- 0.0.0.0
|
|
155
|
+
- --port
|
|
156
|
+
- "8000"
|
|
157
|
+
- --api-key
|
|
158
|
+
- ${VLLM_BACKEND_API_KEY}
|
|
159
|
+
- --served-model-name
|
|
160
|
+
- {{ svc.served_model_name|compose_quote }}
|
|
161
|
+
- --tensor-parallel-size
|
|
162
|
+
- {{ svc.tensor_parallel_size|compose_quote }}
|
|
163
|
+
- --data-parallel-size
|
|
164
|
+
- {{ svc.data_parallel_size|compose_quote }}
|
|
165
|
+
- --max-model-len
|
|
166
|
+
- {{ svc.max_model_len|compose_quote }}
|
|
167
|
+
- --gpu-memory-utilization
|
|
168
|
+
- {{ svc.gpu_memory_utilization|compose_quote }}
|
|
169
|
+
- --max-num-batched-tokens
|
|
170
|
+
- {{ svc.max_num_batched_tokens|compose_quote }}
|
|
171
|
+
- --max-num-seqs
|
|
172
|
+
- {{ svc.max_num_seqs|compose_quote }}
|
|
173
|
+
{% if svc.enable_prefix_caching %}
|
|
174
|
+
- --enable-prefix-caching
|
|
175
|
+
{% endif %}
|
|
176
|
+
{% if svc.enable_auto_tool_choice %}
|
|
177
|
+
- --enable-auto-tool-choice
|
|
178
|
+
- --tool-call-parser
|
|
179
|
+
- {{ svc.tool_call_parser|compose_quote }}
|
|
180
|
+
{% endif %}
|
|
181
|
+
{% if svc.reasoning_enabled and svc.reasoning_parser %}
|
|
182
|
+
- --reasoning-parser
|
|
183
|
+
- {{ svc.reasoning_parser|compose_quote }}
|
|
184
|
+
{% endif %}
|
|
185
|
+
{% for extra_arg in svc.extra_args %}
|
|
186
|
+
- {{ extra_arg|compose_quote }}
|
|
187
|
+
{% endfor %}
|
|
188
|
+
volumes:
|
|
189
|
+
- {{ lock.deployment.state.hf_cache }}:/root/.cache/huggingface
|
|
190
|
+
- {{ lock.deployment.state.vllm_cache }}:/root/.cache/vllm
|
|
191
|
+
- {{ lock.deployment.state.torch_cache }}:/root/.cache/torch
|
|
192
|
+
- {{ lock.deployment.state.triton_cache }}:/root/.cache/triton
|
|
193
|
+
- {{ lock.deployment.state.cuda_cache }}:/root/.cache/nvidia/ComputeCache
|
|
194
|
+
{% for volume in svc.extra_volumes %}
|
|
195
|
+
- {{ volume }}
|
|
196
|
+
{% endfor %}
|
|
197
|
+
{% if svc.publish_port or svc.additional_ports %}
|
|
198
|
+
ports:
|
|
199
|
+
{% if svc.publish_port %}
|
|
200
|
+
- "127.0.0.1:${INFER_STACK_VLLM_{{ loop.index0 }}_PORT:-{{ svc.host_port or (18000 + loop.index0) }}}:8000"
|
|
201
|
+
{% endif %}
|
|
202
|
+
{% for port in svc.additional_ports %}
|
|
203
|
+
- {{ port|compose_quote }}
|
|
204
|
+
{% endfor %}
|
|
205
|
+
{% endif %}
|
|
206
|
+
healthcheck:
|
|
207
|
+
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8000/health >/dev/null 2>&1 || exit 1"]
|
|
208
|
+
interval: 30s
|
|
209
|
+
timeout: 10s
|
|
210
|
+
retries: 30
|
|
211
|
+
start_period: 1800s
|
|
212
|
+
{% if svc.gpus is not none %}
|
|
213
|
+
{{ svc.gpus|compose_gpus }}
|
|
214
|
+
{% else %}
|
|
215
|
+
deploy:
|
|
216
|
+
resources:
|
|
217
|
+
reservations:
|
|
218
|
+
devices:
|
|
219
|
+
- driver: nvidia
|
|
220
|
+
device_ids: [{% for idx in svc.gpu_indices %}"{{ idx }}"{% if not loop.last %}, {% endif %}{% endfor %}]
|
|
221
|
+
capabilities: [gpu]
|
|
222
|
+
{% endif %}
|
|
223
|
+
|
|
224
|
+
{% endfor %}
|
|
225
|
+
{% if lock.deployment.gateways.litellm.enabled %}
|
|
226
|
+
litellm:
|
|
227
|
+
image: {{ lock.deployment.images.litellm }}
|
|
228
|
+
container_name: litellm
|
|
229
|
+
restart: unless-stopped
|
|
230
|
+
{% if lock.deployment.gateways.litellm.labels %}
|
|
231
|
+
labels:
|
|
232
|
+
{% for key, value in lock.deployment.gateways.litellm.labels.items() %}
|
|
233
|
+
{{ key }}: {{ value|compose_quote }}
|
|
234
|
+
{% endfor %}
|
|
235
|
+
{% endif %}
|
|
236
|
+
depends_on:
|
|
237
|
+
postgres-litellm:
|
|
238
|
+
condition: service_healthy
|
|
239
|
+
# Do not depend directly on provider health here. LiteLLM can load its
|
|
240
|
+
# route table before upstream providers are healthy, and avoiding provider
|
|
241
|
+
# dependency edges prevents Compose from restarting LiteLLM on every vLLM
|
|
242
|
+
# model swap. Smoke tests and clients should retry until the selected
|
|
243
|
+
# upstream model is healthy.
|
|
244
|
+
{% if lock.deployment.gateways.litellm.env_file %}
|
|
245
|
+
env_file:
|
|
246
|
+
{% for env_file in lock.deployment.gateways.litellm.env_file %}
|
|
247
|
+
- {{ env_file }}
|
|
248
|
+
{% endfor %}
|
|
249
|
+
{% endif %}
|
|
250
|
+
{% if lock.deployment.gateways.litellm.extra_hosts %}
|
|
251
|
+
extra_hosts:
|
|
252
|
+
{% for extra_host in lock.deployment.gateways.litellm.extra_hosts %}
|
|
253
|
+
- {{ extra_host|compose_quote }}
|
|
254
|
+
{% endfor %}
|
|
255
|
+
{% endif %}
|
|
256
|
+
environment:
|
|
257
|
+
LITELLM_MASTER_KEY: ${LITELLM_MASTER_KEY}
|
|
258
|
+
VLLM_BACKEND_API_KEY: ${VLLM_BACKEND_API_KEY}
|
|
259
|
+
DATABASE_URL: postgresql://${LITELLM_POSTGRES_USER}:${LITELLM_POSTGRES_PASSWORD}@postgres-litellm:5432/${LITELLM_POSTGRES_DB}
|
|
260
|
+
{% for key, value in lock.deployment.gateways.litellm.extra_env.items() %}
|
|
261
|
+
{{ key }}: {{ value|compose_quote }}
|
|
262
|
+
{% endfor %}
|
|
263
|
+
command:
|
|
264
|
+
- --config
|
|
265
|
+
- /app/config.yaml
|
|
266
|
+
- --port
|
|
267
|
+
- "4000"
|
|
268
|
+
- --num_workers
|
|
269
|
+
- "1"
|
|
270
|
+
volumes:
|
|
271
|
+
- {{ lock.deployment.state.runtime }}/litellm_config.yaml:/app/config.yaml:ro
|
|
272
|
+
{% for volume in lock.deployment.gateways.litellm.extra_volumes %}
|
|
273
|
+
- {{ volume }}
|
|
274
|
+
{% endfor %}
|
|
275
|
+
{% if lock.deployment.gateways.litellm.additional_ports or lock.deployment.ports.litellm %}
|
|
276
|
+
ports:
|
|
277
|
+
- "${INFER_STACK_LITELLM_PORT:-{{ lock.deployment.ports.litellm or 14042 }}}:4000"
|
|
278
|
+
{% for port in lock.deployment.gateways.litellm.additional_ports %}
|
|
279
|
+
- {{ port|compose_quote }}
|
|
280
|
+
{% endfor %}
|
|
281
|
+
{% endif %}
|
|
282
|
+
{% if lock.deployment.gateways.litellm.gpus is not none %}
|
|
283
|
+
{{ lock.deployment.gateways.litellm.gpus|compose_gpus }}
|
|
284
|
+
{% endif %}
|
|
285
|
+
|
|
286
|
+
{% endif %}
|
|
287
|
+
{% if lock.deployment.frontends.open_webui.enabled %}
|
|
288
|
+
open-webui:
|
|
289
|
+
image: {{ lock.deployment.images.open_webui }}
|
|
290
|
+
container_name: open-webui
|
|
291
|
+
restart: unless-stopped
|
|
292
|
+
{% if lock.deployment.frontends.open_webui.labels %}
|
|
293
|
+
labels:
|
|
294
|
+
{% for key, value in lock.deployment.frontends.open_webui.labels.items() %}
|
|
295
|
+
{{ key }}: {{ value|compose_quote }}
|
|
296
|
+
{% endfor %}
|
|
297
|
+
{% endif %}
|
|
298
|
+
depends_on:
|
|
299
|
+
postgres-open-webui:
|
|
300
|
+
condition: service_healthy
|
|
301
|
+
{% if lock.deployment.frontends.open_webui.provider == 'ollama' %}
|
|
302
|
+
ollama:
|
|
303
|
+
condition: service_started
|
|
304
|
+
{% elif lock.deployment.frontends.open_webui.provider == 'litellm' %}
|
|
305
|
+
litellm:
|
|
306
|
+
condition: service_started
|
|
307
|
+
{% endif %}
|
|
308
|
+
{% if lock.deployment.frontends.open_webui.env_file %}
|
|
309
|
+
env_file:
|
|
310
|
+
{% for env_file in lock.deployment.frontends.open_webui.env_file %}
|
|
311
|
+
- {{ env_file }}
|
|
312
|
+
{% endfor %}
|
|
313
|
+
{% endif %}
|
|
314
|
+
{% if lock.deployment.frontends.open_webui.extra_hosts %}
|
|
315
|
+
extra_hosts:
|
|
316
|
+
{% for extra_host in lock.deployment.frontends.open_webui.extra_hosts %}
|
|
317
|
+
- {{ extra_host|compose_quote }}
|
|
318
|
+
{% endfor %}
|
|
319
|
+
{% endif %}
|
|
320
|
+
environment:
|
|
321
|
+
WEBUI_AUTH: {{ ('True' if lock.deployment.frontends.open_webui.auth else 'False')|compose_quote }}
|
|
322
|
+
ENABLE_PERSISTENT_CONFIG: "False"
|
|
323
|
+
DATABASE_URL: postgresql://${OPENWEBUI_POSTGRES_USER}:${OPENWEBUI_POSTGRES_PASSWORD}@postgres-open-webui:5432/${OPENWEBUI_POSTGRES_DB}
|
|
324
|
+
{% if lock.deployment.frontends.open_webui.webui_url %}
|
|
325
|
+
WEBUI_URL: {{ lock.deployment.frontends.open_webui.webui_url|compose_quote }}
|
|
326
|
+
{% endif %}
|
|
327
|
+
{% if lock.deployment.frontends.open_webui.cors_allow_origin %}
|
|
328
|
+
CORS_ALLOW_ORIGIN: {{ lock.deployment.frontends.open_webui.cors_allow_origin|compose_quote }}
|
|
329
|
+
{% endif %}
|
|
330
|
+
{% if lock.deployment.frontends.open_webui.provider == 'ollama' %}
|
|
331
|
+
ENABLE_OLLAMA_API: "True"
|
|
332
|
+
OLLAMA_BASE_URL: http://ollama:11434
|
|
333
|
+
{% elif lock.deployment.frontends.open_webui.provider == 'litellm' %}
|
|
334
|
+
ENABLE_OLLAMA_API: "False"
|
|
335
|
+
OPENAI_API_BASE_URL: http://litellm:4000/v1
|
|
336
|
+
OPENAI_API_KEY: ${LITELLM_MASTER_KEY}
|
|
337
|
+
{% endif %}
|
|
338
|
+
WEBUI_SECRET_KEY: ${WEBUI_SECRET_KEY}
|
|
339
|
+
{% if lock.deployment.frontends.open_webui.ldap.enabled %}
|
|
340
|
+
{% for key, value in lock.deployment.frontends.open_webui.ldap.env.items() %}
|
|
341
|
+
{{ key }}: {{ value|compose_quote }}
|
|
342
|
+
{% endfor %}
|
|
343
|
+
{% endif %}
|
|
344
|
+
{% for key, value in lock.deployment.frontends.open_webui.extra_env.items() %}
|
|
345
|
+
{{ key }}: {{ value|compose_quote }}
|
|
346
|
+
{% endfor %}
|
|
347
|
+
volumes:
|
|
348
|
+
- {{ lock.deployment.state.open_webui }}:/app/backend/data
|
|
349
|
+
{% for volume in lock.deployment.frontends.open_webui.extra_volumes %}
|
|
350
|
+
- {{ volume }}
|
|
351
|
+
{% endfor %}
|
|
352
|
+
{% if lock.deployment.frontends.open_webui.publish_port or lock.deployment.frontends.open_webui.additional_ports %}
|
|
353
|
+
ports:
|
|
354
|
+
{% if lock.deployment.frontends.open_webui.publish_port %}
|
|
355
|
+
- "${INFER_STACK_OPEN_WEBUI_PORT:-{{ lock.deployment.ports.open_webui or 13000 }}}:8080"
|
|
356
|
+
{% endif %}
|
|
357
|
+
{% for port in lock.deployment.frontends.open_webui.additional_ports %}
|
|
358
|
+
- {{ port|compose_quote }}
|
|
359
|
+
{% endfor %}
|
|
360
|
+
{% endif %}
|
|
361
|
+
{% if lock.deployment.frontends.open_webui.gpus is not none %}
|
|
362
|
+
{{ lock.deployment.frontends.open_webui.gpus|compose_gpus }}
|
|
363
|
+
{% endif %}
|
|
364
|
+
|
|
365
|
+
{% endif %}
|
|
366
|
+
{% if lock.deployment.frontends.reverse_proxy.enabled %}
|
|
367
|
+
{{ lock.deployment.frontends.reverse_proxy.service_name }}:
|
|
368
|
+
image: {{ lock.deployment.frontends.reverse_proxy.image }}
|
|
369
|
+
container_name: {{ lock.deployment.frontends.reverse_proxy.container_name }}
|
|
370
|
+
restart: unless-stopped
|
|
371
|
+
{% if lock.deployment.frontends.reverse_proxy.labels %}
|
|
372
|
+
labels:
|
|
373
|
+
{% for key, value in lock.deployment.frontends.reverse_proxy.labels.items() %}
|
|
374
|
+
{{ key }}: {{ value|compose_quote }}
|
|
375
|
+
{% endfor %}
|
|
376
|
+
{% endif %}
|
|
377
|
+
{% if lock.deployment.frontends.reverse_proxy.depends_on %}
|
|
378
|
+
depends_on:
|
|
379
|
+
{% for service in lock.deployment.frontends.reverse_proxy.depends_on %}
|
|
380
|
+
- {{ service }}
|
|
381
|
+
{% endfor %}
|
|
382
|
+
{% endif %}
|
|
383
|
+
{% if lock.deployment.frontends.reverse_proxy.env_file %}
|
|
384
|
+
env_file:
|
|
385
|
+
{% for env_file in lock.deployment.frontends.reverse_proxy.env_file %}
|
|
386
|
+
- {{ env_file }}
|
|
387
|
+
{% endfor %}
|
|
388
|
+
{% endif %}
|
|
389
|
+
{% if lock.deployment.frontends.reverse_proxy.extra_env %}
|
|
390
|
+
environment:
|
|
391
|
+
{% for key, value in lock.deployment.frontends.reverse_proxy.extra_env.items() %}
|
|
392
|
+
{{ key }}: {{ value|compose_quote }}
|
|
393
|
+
{% endfor %}
|
|
394
|
+
{% endif %}
|
|
395
|
+
{% if lock.deployment.frontends.reverse_proxy.extra_hosts %}
|
|
396
|
+
extra_hosts:
|
|
397
|
+
{% for extra_host in lock.deployment.frontends.reverse_proxy.extra_hosts %}
|
|
398
|
+
- {{ extra_host|compose_quote }}
|
|
399
|
+
{% endfor %}
|
|
400
|
+
{% endif %}
|
|
401
|
+
volumes:
|
|
402
|
+
- {{ lock.deployment.frontends.reverse_proxy.nginx_config_path }}:/etc/nginx/conf.d/default.conf:ro
|
|
403
|
+
{% if lock.deployment.frontends.reverse_proxy.ssl.enabled and lock.deployment.frontends.reverse_proxy.ssl.certificate %}
|
|
404
|
+
- {{ lock.deployment.frontends.reverse_proxy.ssl.certificate }}:{{ lock.deployment.frontends.reverse_proxy.ssl.certificate_container_path }}:ro
|
|
405
|
+
{% endif %}
|
|
406
|
+
{% if lock.deployment.frontends.reverse_proxy.ssl.enabled and lock.deployment.frontends.reverse_proxy.ssl.certificate_key %}
|
|
407
|
+
- {{ lock.deployment.frontends.reverse_proxy.ssl.certificate_key }}:{{ lock.deployment.frontends.reverse_proxy.ssl.certificate_key_container_path }}:ro
|
|
408
|
+
{% endif %}
|
|
409
|
+
{% if lock.deployment.frontends.reverse_proxy.ssl.enabled and lock.deployment.frontends.reverse_proxy.ssl.dhparam %}
|
|
410
|
+
- {{ lock.deployment.frontends.reverse_proxy.ssl.dhparam }}:{{ lock.deployment.frontends.reverse_proxy.ssl.dhparam_container_path }}:ro
|
|
411
|
+
{% endif %}
|
|
412
|
+
{% for volume in lock.deployment.frontends.reverse_proxy.extra_volumes %}
|
|
413
|
+
- {{ volume }}
|
|
414
|
+
{% endfor %}
|
|
415
|
+
{% if lock.deployment.frontends.reverse_proxy.publish_http or lock.deployment.frontends.reverse_proxy.publish_https or lock.deployment.frontends.reverse_proxy.additional_ports %}
|
|
416
|
+
ports:
|
|
417
|
+
{% if lock.deployment.frontends.reverse_proxy.publish_http %}
|
|
418
|
+
- "{{ lock.deployment.frontends.reverse_proxy.http_bind_host }}{% if lock.deployment.frontends.reverse_proxy.http_bind_host %}:{% endif %}${INFER_STACK_REVERSE_PROXY_HTTP_PORT:-{{ lock.deployment.frontends.reverse_proxy.http_port }}}:80"
|
|
419
|
+
{% endif %}
|
|
420
|
+
{% if lock.deployment.frontends.reverse_proxy.publish_https %}
|
|
421
|
+
- "{{ lock.deployment.frontends.reverse_proxy.https_bind_host }}{% if lock.deployment.frontends.reverse_proxy.https_bind_host %}:{% endif %}${INFER_STACK_REVERSE_PROXY_HTTPS_PORT:-{{ lock.deployment.frontends.reverse_proxy.https_port }}}:443"
|
|
422
|
+
{% endif %}
|
|
423
|
+
{% for port in lock.deployment.frontends.reverse_proxy.additional_ports %}
|
|
424
|
+
- {{ port|compose_quote }}
|
|
425
|
+
{% endfor %}
|
|
426
|
+
{% endif %}
|
|
427
|
+
{% if lock.deployment.frontends.reverse_proxy.gpus is not none %}
|
|
428
|
+
{{ lock.deployment.frontends.reverse_proxy.gpus|compose_gpus }}
|
|
429
|
+
{% endif %}
|
|
430
|
+
{% endif %}
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
general_settings:
|
|
2
|
+
master_key: os.environ/LITELLM_MASTER_KEY
|
|
3
|
+
store_model_in_db: True
|
|
4
|
+
|
|
5
|
+
model_list:
|
|
6
|
+
{% for alias, route in lock.deployment.gateways.litellm.routes.items() %}
|
|
7
|
+
{% set max_model_len = route.max_model_len or 4096 %}
|
|
8
|
+
{% set max_output = max_model_len // 2 %}
|
|
9
|
+
{% set max_input = max_model_len - max_output %}
|
|
10
|
+
- model_name: {{ alias }}
|
|
11
|
+
litellm_params:
|
|
12
|
+
{% if route.provider == 'ollama' %}
|
|
13
|
+
model: ollama_chat/{{ route.upstream_model }}
|
|
14
|
+
api_base: http://ollama:11434
|
|
15
|
+
drop_params: true
|
|
16
|
+
{% elif route.protocol_mode == 'completions' %}
|
|
17
|
+
model: text-completion-openai/{{ route.served_model_name }}
|
|
18
|
+
api_base: http://{{ route.upstream_service_name or route.service_name }}:8000/v1
|
|
19
|
+
api_key: os.environ/VLLM_BACKEND_API_KEY
|
|
20
|
+
{% else %}
|
|
21
|
+
model: openai/{{ route.served_model_name }}
|
|
22
|
+
api_base: http://{{ route.upstream_service_name or route.service_name }}:8000/v1
|
|
23
|
+
api_key: os.environ/VLLM_BACKEND_API_KEY
|
|
24
|
+
merge_reasoning_content_in_choices: true
|
|
25
|
+
{% endif %}
|
|
26
|
+
{% if route.chat_compat_enabled and route.chat_compat_strategy == 'flat_messages' %}
|
|
27
|
+
initial_prompt_value: ""
|
|
28
|
+
roles:
|
|
29
|
+
system:
|
|
30
|
+
pre_message: ""
|
|
31
|
+
post_message: "\n"
|
|
32
|
+
user:
|
|
33
|
+
pre_message: ""
|
|
34
|
+
post_message: "\n"
|
|
35
|
+
assistant:
|
|
36
|
+
pre_message: ""
|
|
37
|
+
post_message: "\n"
|
|
38
|
+
final_prompt_value: ""
|
|
39
|
+
{% endif %}
|
|
40
|
+
model_info:
|
|
41
|
+
max_tokens: {{ max_model_len }}
|
|
42
|
+
max_input_tokens: {{ max_input }}
|
|
43
|
+
max_output_tokens: {{ max_output }}
|
|
44
|
+
{% endfor %}
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
# Rendered by infer-stack. To take full manual control, set
|
|
2
|
+
# frontends.reverse_proxy.config_path to an existing nginx config file.
|
|
3
|
+
upstream infer_stack_target {
|
|
4
|
+
server {{ rp.target_service }}:{{ rp.target_port }};
|
|
5
|
+
}
|
|
6
|
+
|
|
7
|
+
{% macro proxy_location(rp) %}
|
|
8
|
+
location / {
|
|
9
|
+
proxy_pass {{ rp.target_scheme }}://infer_stack_target;
|
|
10
|
+
proxy_redirect off;
|
|
11
|
+
proxy_http_version 1.1;
|
|
12
|
+
proxy_cache_bypass $http_upgrade;
|
|
13
|
+
proxy_set_header Upgrade $http_upgrade;
|
|
14
|
+
proxy_set_header Connection "upgrade";
|
|
15
|
+
proxy_set_header Host $host;
|
|
16
|
+
proxy_set_header X-Real-IP $remote_addr;
|
|
17
|
+
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
|
18
|
+
proxy_set_header X-Forwarded-Proto $scheme;
|
|
19
|
+
proxy_set_header X-Forwarded-Host $server_name;
|
|
20
|
+
proxy_buffer_size {{ rp.proxy_buffer_size }};
|
|
21
|
+
proxy_buffers {{ rp.proxy_buffers }};
|
|
22
|
+
proxy_busy_buffers_size {{ rp.proxy_busy_buffers_size }};
|
|
23
|
+
proxy_connect_timeout {{ rp.proxy_connect_timeout }};
|
|
24
|
+
proxy_read_timeout {{ rp.proxy_read_timeout }};
|
|
25
|
+
proxy_send_timeout {{ rp.proxy_send_timeout }};
|
|
26
|
+
{% if rp.proxy_buffering is not none %}
|
|
27
|
+
proxy_buffering {{ 'on' if rp.proxy_buffering else 'off' }};
|
|
28
|
+
{% endif %}
|
|
29
|
+
{% if rp.proxy_cache is not none %}
|
|
30
|
+
proxy_cache {{ rp.proxy_cache }};
|
|
31
|
+
{% endif %}
|
|
32
|
+
client_max_body_size {{ rp.client_max_body_size }};
|
|
33
|
+
}
|
|
34
|
+
{% endmacro %}
|
|
35
|
+
|
|
36
|
+
server {
|
|
37
|
+
listen 80;
|
|
38
|
+
listen [::]:80;
|
|
39
|
+
server_name {{ rp.server_name }};
|
|
40
|
+
{% if rp.ssl.enabled and rp.force_https %}
|
|
41
|
+
|
|
42
|
+
return 301 https://$host$request_uri;
|
|
43
|
+
{% else %}
|
|
44
|
+
|
|
45
|
+
{{ proxy_location(rp)|indent(4, true) }}
|
|
46
|
+
{% endif %}
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
{% if rp.ssl.enabled %}
|
|
50
|
+
server {
|
|
51
|
+
listen 443 ssl;
|
|
52
|
+
listen [::]:443 ssl;
|
|
53
|
+
server_name {{ rp.server_name }};
|
|
54
|
+
|
|
55
|
+
ssl_certificate {{ rp.ssl.certificate_container_path }};
|
|
56
|
+
ssl_certificate_key {{ rp.ssl.certificate_key_container_path }};
|
|
57
|
+
{% if rp.ssl.dhparam %}
|
|
58
|
+
ssl_dhparam {{ rp.ssl.dhparam_container_path }};
|
|
59
|
+
{% endif %}
|
|
60
|
+
ssl_protocols {{ rp.ssl.protocols }};
|
|
61
|
+
ssl_ciphers {{ rp.ssl.ciphers }};
|
|
62
|
+
ssl_prefer_server_ciphers {{ 'on' if rp.ssl.prefer_server_ciphers else 'off' }};
|
|
63
|
+
ssl_session_cache {{ rp.ssl.session_cache }};
|
|
64
|
+
ssl_ecdh_curve {{ rp.ssl.ecdh_curve }};
|
|
65
|
+
ssl_session_tickets {{ 'on' if rp.ssl.session_tickets else 'off' }};
|
|
66
|
+
ssl_stapling {{ 'on' if rp.ssl.stapling else 'off' }};
|
|
67
|
+
ssl_stapling_verify {{ 'on' if rp.ssl.stapling_verify else 'off' }};
|
|
68
|
+
{% if rp.resolver %}
|
|
69
|
+
resolver {{ rp.resolver | join(' ') }} valid=300s;
|
|
70
|
+
resolver_timeout {{ rp.resolver_timeout }};
|
|
71
|
+
{% endif %}
|
|
72
|
+
{% if rp.hsts.enabled %}
|
|
73
|
+
add_header Strict-Transport-Security "max-age={{ rp.hsts.max_age }}{% if rp.hsts.include_subdomains %}; includeSubdomains{% endif %}{% if rp.hsts.preload %}; preload{% endif %}" always;
|
|
74
|
+
{% endif %}
|
|
75
|
+
add_header X-Frame-Options DENY;
|
|
76
|
+
add_header X-Content-Type-Options nosniff;
|
|
77
|
+
|
|
78
|
+
{{ proxy_location(rp)|indent(4, true) }}
|
|
79
|
+
{% if rp.extra_config %}
|
|
80
|
+
|
|
81
|
+
{{ rp.extra_config }}
|
|
82
|
+
{% endif %}
|
|
83
|
+
}
|
|
84
|
+
{% endif %}
|
infer_stack/tuning.py
ADDED