infer-stack 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. infer_stack/__init__.py +2 -0
  2. infer_stack/backends/__init__.py +7 -0
  3. infer_stack/backends/compose_renderer.py +243 -0
  4. infer_stack/backends/kubeai_renderer.py +202 -0
  5. infer_stack/benchmark.py +38 -0
  6. infer_stack/catalog.py +438 -0
  7. infer_stack/cli/__init__.py +169 -0
  8. infer_stack/cli/__main__.py +4 -0
  9. infer_stack/cli/commands_profile.py +467 -0
  10. infer_stack/cli/commands_runtime.py +719 -0
  11. infer_stack/cli/commands_smoke.py +691 -0
  12. infer_stack/cli/compose.py +755 -0
  13. infer_stack/cli/context.py +471 -0
  14. infer_stack/cli/options.py +134 -0
  15. infer_stack/cli/probes.py +178 -0
  16. infer_stack/config.py +450 -0
  17. infer_stack/contracts.py +223 -0
  18. infer_stack/diff_prompt.py +117 -0
  19. infer_stack/docker_utils.py +230 -0
  20. infer_stack/env_utils.py +97 -0
  21. infer_stack/experimental/model_catalog_discover.py +1155 -0
  22. infer_stack/experimental/model_memory_estimator.py +1264 -0
  23. infer_stack/experimental/stress_test_long_context.py +397 -0
  24. infer_stack/hardware.py +70 -0
  25. infer_stack/kubeai_ops.py +76 -0
  26. infer_stack/paths.py +87 -0
  27. infer_stack/profile_runtime.py +46 -0
  28. infer_stack/renderer.py +19 -0
  29. infer_stack/resolver.py +1092 -0
  30. infer_stack/templates/default-models.yaml +674 -0
  31. infer_stack/templates/default-ollama-models.yaml +31 -0
  32. infer_stack/templates/default-profiles.yaml +1731 -0
  33. infer_stack/templates/default-vllm-models.yaml +714 -0
  34. infer_stack/templates/docker-compose.yml.j2 +430 -0
  35. infer_stack/templates/litellm_config.yaml.j2 +44 -0
  36. infer_stack/templates/nginx.conf.j2 +84 -0
  37. infer_stack/tuning.py +3 -0
  38. infer_stack/validator.py +314 -0
  39. infer_stack/verification.py +46 -0
  40. infer_stack-0.6.0.dist-info/METADATA +1034 -0
  41. infer_stack-0.6.0.dist-info/RECORD +44 -0
  42. infer_stack-0.6.0.dist-info/WHEEL +5 -0
  43. infer_stack-0.6.0.dist-info/entry_points.txt +2 -0
  44. infer_stack-0.6.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,430 @@
1
+ services:
2
+ {% if lock.deployment.frontends.open_webui.enabled %}
3
+ postgres-open-webui:
4
+ image: {{ lock.deployment.images.postgres }}
5
+ container_name: postgres-open-webui
6
+ restart: unless-stopped
7
+ environment:
8
+ POSTGRES_DB: ${OPENWEBUI_POSTGRES_DB}
9
+ POSTGRES_USER: ${OPENWEBUI_POSTGRES_USER}
10
+ POSTGRES_PASSWORD: ${OPENWEBUI_POSTGRES_PASSWORD}
11
+ volumes:
12
+ - {{ lock.deployment.state.postgres_open_webui }}:/var/lib/postgresql/data
13
+ healthcheck:
14
+ test: ["CMD-SHELL", "pg_isready -U $$POSTGRES_USER -d $$POSTGRES_DB"]
15
+ interval: 10s
16
+ timeout: 5s
17
+ retries: 10
18
+
19
+ {% endif %}
20
+ {% if lock.deployment.gateways.litellm.enabled %}
21
+ postgres-litellm:
22
+ image: {{ lock.deployment.images.postgres }}
23
+ container_name: postgres-litellm
24
+ restart: unless-stopped
25
+ environment:
26
+ POSTGRES_DB: ${LITELLM_POSTGRES_DB}
27
+ POSTGRES_USER: ${LITELLM_POSTGRES_USER}
28
+ POSTGRES_PASSWORD: ${LITELLM_POSTGRES_PASSWORD}
29
+ volumes:
30
+ - {{ lock.deployment.state.postgres_litellm }}:/var/lib/postgresql/data
31
+ healthcheck:
32
+ test: ["CMD-SHELL", "pg_isready -U $$POSTGRES_USER -d $$POSTGRES_DB"]
33
+ interval: 10s
34
+ timeout: 5s
35
+ retries: 10
36
+
37
+ {% endif %}
38
+ {% if lock.deployment.providers.ollama.enabled %}
39
+ ollama:
40
+ image: {{ lock.deployment.images.ollama }}
41
+ container_name: ollama
42
+ restart: unless-stopped
43
+ {% if lock.deployment.providers.ollama.labels %}
44
+ labels:
45
+ {% for key, value in lock.deployment.providers.ollama.labels.items() %}
46
+ {{ key }}: {{ value|compose_quote }}
47
+ {% endfor %}
48
+ {% endif %}
49
+ {% if lock.deployment.providers.ollama.env_file %}
50
+ env_file:
51
+ {% for env_file in lock.deployment.providers.ollama.env_file %}
52
+ - {{ env_file }}
53
+ {% endfor %}
54
+ {% endif %}
55
+ {% if lock.deployment.providers.ollama.extra_hosts %}
56
+ extra_hosts:
57
+ {% for extra_host in lock.deployment.providers.ollama.extra_hosts %}
58
+ - {{ extra_host|compose_quote }}
59
+ {% endfor %}
60
+ {% endif %}
61
+ environment:
62
+ NVIDIA_DRIVER_CAPABILITIES: compute,utility
63
+ CUDA_VISIBLE_DEVICES: {{ (lock.deployment.providers.ollama.gpu_indices | join(','))|compose_quote }}
64
+ OLLAMA_HOST: {{ lock.deployment.providers.ollama.host|compose_quote }}
65
+ OLLAMA_KEEP_ALIVE: {{ lock.deployment.providers.ollama.keep_alive|compose_quote }}
66
+ OLLAMA_CONTEXT_LENGTH: {{ lock.deployment.providers.ollama.context_length|compose_quote }}
67
+ OLLAMA_NUM_PARALLEL: {{ lock.deployment.providers.ollama.num_parallel|compose_quote }}
68
+ OLLAMA_MAX_LOADED_MODELS: {{ lock.deployment.providers.ollama.max_loaded_models|compose_quote }}
69
+ OLLAMA_MAX_QUEUE: {{ lock.deployment.providers.ollama.max_queue|compose_quote }}
70
+ {% for key, value in lock.deployment.providers.ollama.extra_env.items() %}
71
+ {{ key }}: {{ value|compose_quote }}
72
+ {% endfor %}
73
+ volumes:
74
+ - {{ lock.deployment.state.ollama }}:/root/.ollama
75
+ {% for volume in lock.deployment.providers.ollama.extra_volumes %}
76
+ - {{ volume }}
77
+ {% endfor %}
78
+ {% if lock.deployment.providers.ollama.publish_port or lock.deployment.providers.ollama.additional_ports %}
79
+ ports:
80
+ {% if lock.deployment.providers.ollama.publish_port %}
81
+ - "127.0.0.1:{{ lock.deployment.ports.ollama or 11434 }}:11434"
82
+ {% endif %}
83
+ {% for port in lock.deployment.providers.ollama.additional_ports %}
84
+ - {{ port|compose_quote }}
85
+ {% endfor %}
86
+ {% endif %}
87
+ healthcheck:
88
+ test: ["CMD-SHELL", "ollama list >/dev/null 2>&1 || exit 1"]
89
+ interval: 30s
90
+ timeout: 10s
91
+ retries: 30
92
+ start_period: 60s
93
+ {% if lock.deployment.providers.ollama.gpus is not none %}
94
+ {{ lock.deployment.providers.ollama.gpus|compose_gpus }}
95
+ {% elif lock.deployment.providers.ollama.gpu_indices %}
96
+ deploy:
97
+ resources:
98
+ reservations:
99
+ devices:
100
+ - driver: nvidia
101
+ device_ids: [{% for idx in lock.deployment.providers.ollama.gpu_indices %}"{{ idx }}"{% if not loop.last %}, {% endif %}{% endfor %}]
102
+ capabilities: [gpu]
103
+ {% endif %}
104
+
105
+ {% endif %}
106
+ {% for name, svc in lock.deployment.providers.vllm.runtimes.items() %}
107
+ {{ svc.compose_service_name }}:
108
+ image: {{ lock.deployment.images.vllm }}
109
+ container_name: {{ svc.container_name or svc.compose_service_name }}
110
+ restart: unless-stopped
111
+ ipc: host
112
+ labels:
113
+ infer_stack.profile_name: {{ svc.profile_name|compose_quote }}
114
+ infer_stack.public_name: {{ svc.profile_public_name|compose_quote }}
115
+ infer_stack.logical_model_name: {{ svc.logical_model_name|compose_quote }}
116
+ infer_stack.protocol_mode: {{ svc.protocol_mode|compose_quote }}
117
+ {% for key, value in svc.labels.items() %}
118
+ {{ key }}: {{ value|compose_quote }}
119
+ {% endfor %}
120
+ {% if svc.env_file %}
121
+ env_file:
122
+ {% for env_file in svc.env_file %}
123
+ - {{ env_file }}
124
+ {% endfor %}
125
+ {% endif %}
126
+ {% if svc.extra_hosts %}
127
+ extra_hosts:
128
+ {% for extra_host in svc.extra_hosts %}
129
+ - {{ extra_host|compose_quote }}
130
+ {% endfor %}
131
+ {% endif %}
132
+ environment:
133
+ NVIDIA_DRIVER_CAPABILITIES: compute,utility
134
+ HF_TOKEN: ${HF_TOKEN}
135
+ VLLM_API_KEY: ${VLLM_BACKEND_API_KEY}
136
+ HF_HOME: /root/.cache/huggingface
137
+ HF_HUB_CACHE: /root/.cache/huggingface/hub
138
+ HUGGINGFACE_HUB_CACHE: /root/.cache/huggingface/hub
139
+ VLLM_CACHE_ROOT: /root/.cache/vllm
140
+ TORCH_HOME: /root/.cache/torch
141
+ TORCHINDUCTOR_CACHE_DIR: /root/.cache/torch/inductor
142
+ TRITON_CACHE_DIR: /root/.cache/triton
143
+ CUDA_CACHE_PATH: /root/.cache/nvidia/ComputeCache
144
+ {% if lock.deployment.vllm.enable_responses_api_store %}
145
+ VLLM_ENABLE_RESPONSES_API_STORE: "1"
146
+ {% endif %}
147
+ VLLM_LOGGING_LEVEL: {{ lock.deployment.vllm.logging_level|compose_quote }}
148
+ {% for key, value in svc.extra_env.items() %}
149
+ {{ key }}: {{ value|compose_quote }}
150
+ {% endfor %}
151
+ command:
152
+ - {{ svc.hf_model_id|compose_quote }}
153
+ - --host
154
+ - 0.0.0.0
155
+ - --port
156
+ - "8000"
157
+ - --api-key
158
+ - ${VLLM_BACKEND_API_KEY}
159
+ - --served-model-name
160
+ - {{ svc.served_model_name|compose_quote }}
161
+ - --tensor-parallel-size
162
+ - {{ svc.tensor_parallel_size|compose_quote }}
163
+ - --data-parallel-size
164
+ - {{ svc.data_parallel_size|compose_quote }}
165
+ - --max-model-len
166
+ - {{ svc.max_model_len|compose_quote }}
167
+ - --gpu-memory-utilization
168
+ - {{ svc.gpu_memory_utilization|compose_quote }}
169
+ - --max-num-batched-tokens
170
+ - {{ svc.max_num_batched_tokens|compose_quote }}
171
+ - --max-num-seqs
172
+ - {{ svc.max_num_seqs|compose_quote }}
173
+ {% if svc.enable_prefix_caching %}
174
+ - --enable-prefix-caching
175
+ {% endif %}
176
+ {% if svc.enable_auto_tool_choice %}
177
+ - --enable-auto-tool-choice
178
+ - --tool-call-parser
179
+ - {{ svc.tool_call_parser|compose_quote }}
180
+ {% endif %}
181
+ {% if svc.reasoning_enabled and svc.reasoning_parser %}
182
+ - --reasoning-parser
183
+ - {{ svc.reasoning_parser|compose_quote }}
184
+ {% endif %}
185
+ {% for extra_arg in svc.extra_args %}
186
+ - {{ extra_arg|compose_quote }}
187
+ {% endfor %}
188
+ volumes:
189
+ - {{ lock.deployment.state.hf_cache }}:/root/.cache/huggingface
190
+ - {{ lock.deployment.state.vllm_cache }}:/root/.cache/vllm
191
+ - {{ lock.deployment.state.torch_cache }}:/root/.cache/torch
192
+ - {{ lock.deployment.state.triton_cache }}:/root/.cache/triton
193
+ - {{ lock.deployment.state.cuda_cache }}:/root/.cache/nvidia/ComputeCache
194
+ {% for volume in svc.extra_volumes %}
195
+ - {{ volume }}
196
+ {% endfor %}
197
+ {% if svc.publish_port or svc.additional_ports %}
198
+ ports:
199
+ {% if svc.publish_port %}
200
+ - "127.0.0.1:${INFER_STACK_VLLM_{{ loop.index0 }}_PORT:-{{ svc.host_port or (18000 + loop.index0) }}}:8000"
201
+ {% endif %}
202
+ {% for port in svc.additional_ports %}
203
+ - {{ port|compose_quote }}
204
+ {% endfor %}
205
+ {% endif %}
206
+ healthcheck:
207
+ test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8000/health >/dev/null 2>&1 || exit 1"]
208
+ interval: 30s
209
+ timeout: 10s
210
+ retries: 30
211
+ start_period: 1800s
212
+ {% if svc.gpus is not none %}
213
+ {{ svc.gpus|compose_gpus }}
214
+ {% else %}
215
+ deploy:
216
+ resources:
217
+ reservations:
218
+ devices:
219
+ - driver: nvidia
220
+ device_ids: [{% for idx in svc.gpu_indices %}"{{ idx }}"{% if not loop.last %}, {% endif %}{% endfor %}]
221
+ capabilities: [gpu]
222
+ {% endif %}
223
+
224
+ {% endfor %}
225
+ {% if lock.deployment.gateways.litellm.enabled %}
226
+ litellm:
227
+ image: {{ lock.deployment.images.litellm }}
228
+ container_name: litellm
229
+ restart: unless-stopped
230
+ {% if lock.deployment.gateways.litellm.labels %}
231
+ labels:
232
+ {% for key, value in lock.deployment.gateways.litellm.labels.items() %}
233
+ {{ key }}: {{ value|compose_quote }}
234
+ {% endfor %}
235
+ {% endif %}
236
+ depends_on:
237
+ postgres-litellm:
238
+ condition: service_healthy
239
+ # Do not depend directly on provider health here. LiteLLM can load its
240
+ # route table before upstream providers are healthy, and avoiding provider
241
+ # dependency edges prevents Compose from restarting LiteLLM on every vLLM
242
+ # model swap. Smoke tests and clients should retry until the selected
243
+ # upstream model is healthy.
244
+ {% if lock.deployment.gateways.litellm.env_file %}
245
+ env_file:
246
+ {% for env_file in lock.deployment.gateways.litellm.env_file %}
247
+ - {{ env_file }}
248
+ {% endfor %}
249
+ {% endif %}
250
+ {% if lock.deployment.gateways.litellm.extra_hosts %}
251
+ extra_hosts:
252
+ {% for extra_host in lock.deployment.gateways.litellm.extra_hosts %}
253
+ - {{ extra_host|compose_quote }}
254
+ {% endfor %}
255
+ {% endif %}
256
+ environment:
257
+ LITELLM_MASTER_KEY: ${LITELLM_MASTER_KEY}
258
+ VLLM_BACKEND_API_KEY: ${VLLM_BACKEND_API_KEY}
259
+ DATABASE_URL: postgresql://${LITELLM_POSTGRES_USER}:${LITELLM_POSTGRES_PASSWORD}@postgres-litellm:5432/${LITELLM_POSTGRES_DB}
260
+ {% for key, value in lock.deployment.gateways.litellm.extra_env.items() %}
261
+ {{ key }}: {{ value|compose_quote }}
262
+ {% endfor %}
263
+ command:
264
+ - --config
265
+ - /app/config.yaml
266
+ - --port
267
+ - "4000"
268
+ - --num_workers
269
+ - "1"
270
+ volumes:
271
+ - {{ lock.deployment.state.runtime }}/litellm_config.yaml:/app/config.yaml:ro
272
+ {% for volume in lock.deployment.gateways.litellm.extra_volumes %}
273
+ - {{ volume }}
274
+ {% endfor %}
275
+ {% if lock.deployment.gateways.litellm.additional_ports or lock.deployment.ports.litellm %}
276
+ ports:
277
+ - "${INFER_STACK_LITELLM_PORT:-{{ lock.deployment.ports.litellm or 14042 }}}:4000"
278
+ {% for port in lock.deployment.gateways.litellm.additional_ports %}
279
+ - {{ port|compose_quote }}
280
+ {% endfor %}
281
+ {% endif %}
282
+ {% if lock.deployment.gateways.litellm.gpus is not none %}
283
+ {{ lock.deployment.gateways.litellm.gpus|compose_gpus }}
284
+ {% endif %}
285
+
286
+ {% endif %}
287
+ {% if lock.deployment.frontends.open_webui.enabled %}
288
+ open-webui:
289
+ image: {{ lock.deployment.images.open_webui }}
290
+ container_name: open-webui
291
+ restart: unless-stopped
292
+ {% if lock.deployment.frontends.open_webui.labels %}
293
+ labels:
294
+ {% for key, value in lock.deployment.frontends.open_webui.labels.items() %}
295
+ {{ key }}: {{ value|compose_quote }}
296
+ {% endfor %}
297
+ {% endif %}
298
+ depends_on:
299
+ postgres-open-webui:
300
+ condition: service_healthy
301
+ {% if lock.deployment.frontends.open_webui.provider == 'ollama' %}
302
+ ollama:
303
+ condition: service_started
304
+ {% elif lock.deployment.frontends.open_webui.provider == 'litellm' %}
305
+ litellm:
306
+ condition: service_started
307
+ {% endif %}
308
+ {% if lock.deployment.frontends.open_webui.env_file %}
309
+ env_file:
310
+ {% for env_file in lock.deployment.frontends.open_webui.env_file %}
311
+ - {{ env_file }}
312
+ {% endfor %}
313
+ {% endif %}
314
+ {% if lock.deployment.frontends.open_webui.extra_hosts %}
315
+ extra_hosts:
316
+ {% for extra_host in lock.deployment.frontends.open_webui.extra_hosts %}
317
+ - {{ extra_host|compose_quote }}
318
+ {% endfor %}
319
+ {% endif %}
320
+ environment:
321
+ WEBUI_AUTH: {{ ('True' if lock.deployment.frontends.open_webui.auth else 'False')|compose_quote }}
322
+ ENABLE_PERSISTENT_CONFIG: "False"
323
+ DATABASE_URL: postgresql://${OPENWEBUI_POSTGRES_USER}:${OPENWEBUI_POSTGRES_PASSWORD}@postgres-open-webui:5432/${OPENWEBUI_POSTGRES_DB}
324
+ {% if lock.deployment.frontends.open_webui.webui_url %}
325
+ WEBUI_URL: {{ lock.deployment.frontends.open_webui.webui_url|compose_quote }}
326
+ {% endif %}
327
+ {% if lock.deployment.frontends.open_webui.cors_allow_origin %}
328
+ CORS_ALLOW_ORIGIN: {{ lock.deployment.frontends.open_webui.cors_allow_origin|compose_quote }}
329
+ {% endif %}
330
+ {% if lock.deployment.frontends.open_webui.provider == 'ollama' %}
331
+ ENABLE_OLLAMA_API: "True"
332
+ OLLAMA_BASE_URL: http://ollama:11434
333
+ {% elif lock.deployment.frontends.open_webui.provider == 'litellm' %}
334
+ ENABLE_OLLAMA_API: "False"
335
+ OPENAI_API_BASE_URL: http://litellm:4000/v1
336
+ OPENAI_API_KEY: ${LITELLM_MASTER_KEY}
337
+ {% endif %}
338
+ WEBUI_SECRET_KEY: ${WEBUI_SECRET_KEY}
339
+ {% if lock.deployment.frontends.open_webui.ldap.enabled %}
340
+ {% for key, value in lock.deployment.frontends.open_webui.ldap.env.items() %}
341
+ {{ key }}: {{ value|compose_quote }}
342
+ {% endfor %}
343
+ {% endif %}
344
+ {% for key, value in lock.deployment.frontends.open_webui.extra_env.items() %}
345
+ {{ key }}: {{ value|compose_quote }}
346
+ {% endfor %}
347
+ volumes:
348
+ - {{ lock.deployment.state.open_webui }}:/app/backend/data
349
+ {% for volume in lock.deployment.frontends.open_webui.extra_volumes %}
350
+ - {{ volume }}
351
+ {% endfor %}
352
+ {% if lock.deployment.frontends.open_webui.publish_port or lock.deployment.frontends.open_webui.additional_ports %}
353
+ ports:
354
+ {% if lock.deployment.frontends.open_webui.publish_port %}
355
+ - "${INFER_STACK_OPEN_WEBUI_PORT:-{{ lock.deployment.ports.open_webui or 13000 }}}:8080"
356
+ {% endif %}
357
+ {% for port in lock.deployment.frontends.open_webui.additional_ports %}
358
+ - {{ port|compose_quote }}
359
+ {% endfor %}
360
+ {% endif %}
361
+ {% if lock.deployment.frontends.open_webui.gpus is not none %}
362
+ {{ lock.deployment.frontends.open_webui.gpus|compose_gpus }}
363
+ {% endif %}
364
+
365
+ {% endif %}
366
+ {% if lock.deployment.frontends.reverse_proxy.enabled %}
367
+ {{ lock.deployment.frontends.reverse_proxy.service_name }}:
368
+ image: {{ lock.deployment.frontends.reverse_proxy.image }}
369
+ container_name: {{ lock.deployment.frontends.reverse_proxy.container_name }}
370
+ restart: unless-stopped
371
+ {% if lock.deployment.frontends.reverse_proxy.labels %}
372
+ labels:
373
+ {% for key, value in lock.deployment.frontends.reverse_proxy.labels.items() %}
374
+ {{ key }}: {{ value|compose_quote }}
375
+ {% endfor %}
376
+ {% endif %}
377
+ {% if lock.deployment.frontends.reverse_proxy.depends_on %}
378
+ depends_on:
379
+ {% for service in lock.deployment.frontends.reverse_proxy.depends_on %}
380
+ - {{ service }}
381
+ {% endfor %}
382
+ {% endif %}
383
+ {% if lock.deployment.frontends.reverse_proxy.env_file %}
384
+ env_file:
385
+ {% for env_file in lock.deployment.frontends.reverse_proxy.env_file %}
386
+ - {{ env_file }}
387
+ {% endfor %}
388
+ {% endif %}
389
+ {% if lock.deployment.frontends.reverse_proxy.extra_env %}
390
+ environment:
391
+ {% for key, value in lock.deployment.frontends.reverse_proxy.extra_env.items() %}
392
+ {{ key }}: {{ value|compose_quote }}
393
+ {% endfor %}
394
+ {% endif %}
395
+ {% if lock.deployment.frontends.reverse_proxy.extra_hosts %}
396
+ extra_hosts:
397
+ {% for extra_host in lock.deployment.frontends.reverse_proxy.extra_hosts %}
398
+ - {{ extra_host|compose_quote }}
399
+ {% endfor %}
400
+ {% endif %}
401
+ volumes:
402
+ - {{ lock.deployment.frontends.reverse_proxy.nginx_config_path }}:/etc/nginx/conf.d/default.conf:ro
403
+ {% if lock.deployment.frontends.reverse_proxy.ssl.enabled and lock.deployment.frontends.reverse_proxy.ssl.certificate %}
404
+ - {{ lock.deployment.frontends.reverse_proxy.ssl.certificate }}:{{ lock.deployment.frontends.reverse_proxy.ssl.certificate_container_path }}:ro
405
+ {% endif %}
406
+ {% if lock.deployment.frontends.reverse_proxy.ssl.enabled and lock.deployment.frontends.reverse_proxy.ssl.certificate_key %}
407
+ - {{ lock.deployment.frontends.reverse_proxy.ssl.certificate_key }}:{{ lock.deployment.frontends.reverse_proxy.ssl.certificate_key_container_path }}:ro
408
+ {% endif %}
409
+ {% if lock.deployment.frontends.reverse_proxy.ssl.enabled and lock.deployment.frontends.reverse_proxy.ssl.dhparam %}
410
+ - {{ lock.deployment.frontends.reverse_proxy.ssl.dhparam }}:{{ lock.deployment.frontends.reverse_proxy.ssl.dhparam_container_path }}:ro
411
+ {% endif %}
412
+ {% for volume in lock.deployment.frontends.reverse_proxy.extra_volumes %}
413
+ - {{ volume }}
414
+ {% endfor %}
415
+ {% if lock.deployment.frontends.reverse_proxy.publish_http or lock.deployment.frontends.reverse_proxy.publish_https or lock.deployment.frontends.reverse_proxy.additional_ports %}
416
+ ports:
417
+ {% if lock.deployment.frontends.reverse_proxy.publish_http %}
418
+ - "{{ lock.deployment.frontends.reverse_proxy.http_bind_host }}{% if lock.deployment.frontends.reverse_proxy.http_bind_host %}:{% endif %}${INFER_STACK_REVERSE_PROXY_HTTP_PORT:-{{ lock.deployment.frontends.reverse_proxy.http_port }}}:80"
419
+ {% endif %}
420
+ {% if lock.deployment.frontends.reverse_proxy.publish_https %}
421
+ - "{{ lock.deployment.frontends.reverse_proxy.https_bind_host }}{% if lock.deployment.frontends.reverse_proxy.https_bind_host %}:{% endif %}${INFER_STACK_REVERSE_PROXY_HTTPS_PORT:-{{ lock.deployment.frontends.reverse_proxy.https_port }}}:443"
422
+ {% endif %}
423
+ {% for port in lock.deployment.frontends.reverse_proxy.additional_ports %}
424
+ - {{ port|compose_quote }}
425
+ {% endfor %}
426
+ {% endif %}
427
+ {% if lock.deployment.frontends.reverse_proxy.gpus is not none %}
428
+ {{ lock.deployment.frontends.reverse_proxy.gpus|compose_gpus }}
429
+ {% endif %}
430
+ {% endif %}
@@ -0,0 +1,44 @@
1
+ general_settings:
2
+ master_key: os.environ/LITELLM_MASTER_KEY
3
+ store_model_in_db: True
4
+
5
+ model_list:
6
+ {% for alias, route in lock.deployment.gateways.litellm.routes.items() %}
7
+ {% set max_model_len = route.max_model_len or 4096 %}
8
+ {% set max_output = max_model_len // 2 %}
9
+ {% set max_input = max_model_len - max_output %}
10
+ - model_name: {{ alias }}
11
+ litellm_params:
12
+ {% if route.provider == 'ollama' %}
13
+ model: ollama_chat/{{ route.upstream_model }}
14
+ api_base: http://ollama:11434
15
+ drop_params: true
16
+ {% elif route.protocol_mode == 'completions' %}
17
+ model: text-completion-openai/{{ route.served_model_name }}
18
+ api_base: http://{{ route.upstream_service_name or route.service_name }}:8000/v1
19
+ api_key: os.environ/VLLM_BACKEND_API_KEY
20
+ {% else %}
21
+ model: openai/{{ route.served_model_name }}
22
+ api_base: http://{{ route.upstream_service_name or route.service_name }}:8000/v1
23
+ api_key: os.environ/VLLM_BACKEND_API_KEY
24
+ merge_reasoning_content_in_choices: true
25
+ {% endif %}
26
+ {% if route.chat_compat_enabled and route.chat_compat_strategy == 'flat_messages' %}
27
+ initial_prompt_value: ""
28
+ roles:
29
+ system:
30
+ pre_message: ""
31
+ post_message: "\n"
32
+ user:
33
+ pre_message: ""
34
+ post_message: "\n"
35
+ assistant:
36
+ pre_message: ""
37
+ post_message: "\n"
38
+ final_prompt_value: ""
39
+ {% endif %}
40
+ model_info:
41
+ max_tokens: {{ max_model_len }}
42
+ max_input_tokens: {{ max_input }}
43
+ max_output_tokens: {{ max_output }}
44
+ {% endfor %}
@@ -0,0 +1,84 @@
1
+ # Rendered by infer-stack. To take full manual control, set
2
+ # frontends.reverse_proxy.config_path to an existing nginx config file.
3
+ upstream infer_stack_target {
4
+ server {{ rp.target_service }}:{{ rp.target_port }};
5
+ }
6
+
7
+ {% macro proxy_location(rp) %}
8
+ location / {
9
+ proxy_pass {{ rp.target_scheme }}://infer_stack_target;
10
+ proxy_redirect off;
11
+ proxy_http_version 1.1;
12
+ proxy_cache_bypass $http_upgrade;
13
+ proxy_set_header Upgrade $http_upgrade;
14
+ proxy_set_header Connection "upgrade";
15
+ proxy_set_header Host $host;
16
+ proxy_set_header X-Real-IP $remote_addr;
17
+ proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
18
+ proxy_set_header X-Forwarded-Proto $scheme;
19
+ proxy_set_header X-Forwarded-Host $server_name;
20
+ proxy_buffer_size {{ rp.proxy_buffer_size }};
21
+ proxy_buffers {{ rp.proxy_buffers }};
22
+ proxy_busy_buffers_size {{ rp.proxy_busy_buffers_size }};
23
+ proxy_connect_timeout {{ rp.proxy_connect_timeout }};
24
+ proxy_read_timeout {{ rp.proxy_read_timeout }};
25
+ proxy_send_timeout {{ rp.proxy_send_timeout }};
26
+ {% if rp.proxy_buffering is not none %}
27
+ proxy_buffering {{ 'on' if rp.proxy_buffering else 'off' }};
28
+ {% endif %}
29
+ {% if rp.proxy_cache is not none %}
30
+ proxy_cache {{ rp.proxy_cache }};
31
+ {% endif %}
32
+ client_max_body_size {{ rp.client_max_body_size }};
33
+ }
34
+ {% endmacro %}
35
+
36
+ server {
37
+ listen 80;
38
+ listen [::]:80;
39
+ server_name {{ rp.server_name }};
40
+ {% if rp.ssl.enabled and rp.force_https %}
41
+
42
+ return 301 https://$host$request_uri;
43
+ {% else %}
44
+
45
+ {{ proxy_location(rp)|indent(4, true) }}
46
+ {% endif %}
47
+ }
48
+
49
+ {% if rp.ssl.enabled %}
50
+ server {
51
+ listen 443 ssl;
52
+ listen [::]:443 ssl;
53
+ server_name {{ rp.server_name }};
54
+
55
+ ssl_certificate {{ rp.ssl.certificate_container_path }};
56
+ ssl_certificate_key {{ rp.ssl.certificate_key_container_path }};
57
+ {% if rp.ssl.dhparam %}
58
+ ssl_dhparam {{ rp.ssl.dhparam_container_path }};
59
+ {% endif %}
60
+ ssl_protocols {{ rp.ssl.protocols }};
61
+ ssl_ciphers {{ rp.ssl.ciphers }};
62
+ ssl_prefer_server_ciphers {{ 'on' if rp.ssl.prefer_server_ciphers else 'off' }};
63
+ ssl_session_cache {{ rp.ssl.session_cache }};
64
+ ssl_ecdh_curve {{ rp.ssl.ecdh_curve }};
65
+ ssl_session_tickets {{ 'on' if rp.ssl.session_tickets else 'off' }};
66
+ ssl_stapling {{ 'on' if rp.ssl.stapling else 'off' }};
67
+ ssl_stapling_verify {{ 'on' if rp.ssl.stapling_verify else 'off' }};
68
+ {% if rp.resolver %}
69
+ resolver {{ rp.resolver | join(' ') }} valid=300s;
70
+ resolver_timeout {{ rp.resolver_timeout }};
71
+ {% endif %}
72
+ {% if rp.hsts.enabled %}
73
+ add_header Strict-Transport-Security "max-age={{ rp.hsts.max_age }}{% if rp.hsts.include_subdomains %}; includeSubdomains{% endif %}{% if rp.hsts.preload %}; preload{% endif %}" always;
74
+ {% endif %}
75
+ add_header X-Frame-Options DENY;
76
+ add_header X-Content-Type-Options nosniff;
77
+
78
+ {{ proxy_location(rp)|indent(4, true) }}
79
+ {% if rp.extra_config %}
80
+
81
+ {{ rp.extra_config }}
82
+ {% endif %}
83
+ }
84
+ {% endif %}
infer_stack/tuning.py ADDED
@@ -0,0 +1,3 @@
1
+ from __future__ import annotations
2
+
3
+ # Placeholder for future tuning expansion.