litserve 0.2.7.dev0__tar.gz → 0.2.8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {litserve-0.2.7.dev0/src/litserve.egg-info → litserve-0.2.8}/PKG-INFO +93 -96
- {litserve-0.2.7.dev0 → litserve-0.2.8}/README.md +86 -90
- {litserve-0.2.7.dev0 → litserve-0.2.8}/setup.py +1 -3
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/__about__.py +1 -1
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/api.py +8 -3
- litserve-0.2.8/src/litserve/cli.py +17 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/connector.py +7 -12
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/loops/base.py +12 -32
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/loops/continuous_batching_loop.py +6 -5
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/loops/loops.py +12 -20
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/loops/simple_loops.py +17 -16
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/loops/streaming_loops.py +19 -18
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/server.py +59 -49
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/specs/openai.py +12 -2
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/specs/openai_embedding.py +5 -2
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/test_examples/openai_spec_example.py +22 -3
- litserve-0.2.8/src/litserve/transport/__init__.py +4 -0
- litserve-0.2.8/src/litserve/transport/base.py +18 -0
- litserve-0.2.8/src/litserve/transport/factory.py +38 -0
- litserve-0.2.8/src/litserve/transport/process_transport.py +44 -0
- {litserve-0.2.7.dev0/src/litserve → litserve-0.2.8/src/litserve/transport}/zmq_queue.py +1 -1
- litserve-0.2.8/src/litserve/transport/zmq_transport.py +42 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/utils.py +2 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8/src/litserve.egg-info}/PKG-INFO +93 -96
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve.egg-info/SOURCES.txt +8 -2
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve.egg-info/entry_points.txt +1 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/LICENSE +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/MANIFEST.in +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/requirements.txt +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/setup.cfg +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/__init__.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/__main__.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/callbacks/__init__.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/callbacks/base.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/callbacks/defaults/__init__.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/callbacks/defaults/metric_callback.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/docker_builder.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/loggers.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/loops/__init__.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/middlewares.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/python_client.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/schema/__init__.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/schema/image.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/specs/__init__.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/specs/base.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/test_examples/__init__.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/test_examples/openai_embedding_spec_example.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/test_examples/simple_example.py +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve.egg-info/dependency_links.txt +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve.egg-info/not-zip-safe +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve.egg-info/requires.txt +0 -0
- {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: litserve
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.8
|
|
4
4
|
Summary: Lightweight AI server.
|
|
5
5
|
Home-page: https://github.com/Lightning-AI/litserve
|
|
6
6
|
Download-URL: https://github.com/Lightning-AI/litserve
|
|
@@ -30,10 +30,6 @@ License-File: LICENSE
|
|
|
30
30
|
Requires-Dist: fastapi>=0.100
|
|
31
31
|
Requires-Dist: uvicorn[standard]>=0.29.0
|
|
32
32
|
Requires-Dist: pyzmq>=22.0.0
|
|
33
|
-
Provides-Extra: perf
|
|
34
|
-
Requires-Dist: jsonargparse; extra == "perf"
|
|
35
|
-
Requires-Dist: tenacity; extra == "perf"
|
|
36
|
-
Requires-Dist: uvloop; extra == "perf"
|
|
37
33
|
Provides-Extra: test
|
|
38
34
|
Requires-Dist: asgi-lifespan; extra == "test"
|
|
39
35
|
Requires-Dist: coverage[toml]>=7.5.3; extra == "test"
|
|
@@ -52,6 +48,10 @@ Requires-Dist: python-multipart; extra == "test"
|
|
|
52
48
|
Requires-Dist: requests; extra == "test"
|
|
53
49
|
Requires-Dist: torch>2.0.0; extra == "test"
|
|
54
50
|
Requires-Dist: transformers; extra == "test"
|
|
51
|
+
Provides-Extra: perf
|
|
52
|
+
Requires-Dist: jsonargparse; extra == "perf"
|
|
53
|
+
Requires-Dist: tenacity; extra == "perf"
|
|
54
|
+
Requires-Dist: uvloop; extra == "perf"
|
|
55
55
|
Dynamic: author
|
|
56
56
|
Dynamic: author-email
|
|
57
57
|
Dynamic: classifier
|
|
@@ -61,6 +61,7 @@ Dynamic: download-url
|
|
|
61
61
|
Dynamic: home-page
|
|
62
62
|
Dynamic: keywords
|
|
63
63
|
Dynamic: license
|
|
64
|
+
Dynamic: license-file
|
|
64
65
|
Dynamic: project-url
|
|
65
66
|
Dynamic: provides-extra
|
|
66
67
|
Dynamic: requires-dist
|
|
@@ -69,33 +70,30 @@ Dynamic: summary
|
|
|
69
70
|
|
|
70
71
|
<div align='center'>
|
|
71
72
|
|
|
72
|
-
#
|
|
73
|
+
# Deploy AI models and inference pipelines - ⚡ fast
|
|
73
74
|
|
|
74
75
|
<img alt="Lightning" src="https://pl-bolts-doc-images.s3.us-east-2.amazonaws.com/app-2/ls_banner2.png" width="800px" style="max-width: 100%;">
|
|
75
76
|
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
<strong>Lightning-fast serving engine for AI models.</strong>
|
|
79
|
-
Easy. Flexible. Enterprise-scale.
|
|
77
|
+
|
|
80
78
|
</div>
|
|
81
79
|
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
**LitServe** is an easy-to-use, flexible serving engine for AI models built on FastAPI. It augments FastAPI with features like batching, streaming, and GPU autoscaling eliminate the need to rebuild a FastAPI server per model.
|
|
80
|
+
**LitServe** lets you build high-performance AI inference pipelines on top of FastAPI - no boilerplate. Define one or more models, connect vector DBs, stream responses, batch requests, and autoscale on GPUs out of the box.
|
|
85
81
|
|
|
86
82
|
LitServe is at least [2x faster](#performance) than plain FastAPI due to AI-specific multi-worker handling.
|
|
87
83
|
|
|
88
84
|
<div align='center'>
|
|
89
85
|
|
|
90
86
|
<pre>
|
|
91
|
-
✅ (2x)+ faster serving ✅ Easy to use
|
|
92
|
-
✅ Bring your own model ✅ PyTorch/JAX/TF/...
|
|
93
|
-
✅ GPU autoscaling ✅ Batching, Streaming
|
|
94
|
-
✅
|
|
87
|
+
✅ (2x)+ faster serving ✅ Easy to use ✅ LLMs, non LLMs and more
|
|
88
|
+
✅ Bring your own model ✅ PyTorch/JAX/TF/... ✅ Built on FastAPI
|
|
89
|
+
✅ GPU autoscaling ✅ Batching, Streaming ✅ Self-host or ⚡️ managed
|
|
90
|
+
✅ Inference pipeline ✅ Integrate with vLLM, etc ✅ Serverless
|
|
91
|
+
|
|
95
92
|
</pre>
|
|
96
93
|
|
|
97
94
|
<div align='center'>
|
|
98
95
|
|
|
96
|
+
[](https://pepy.tech/projects/litserve)
|
|
99
97
|
[](https://discord.gg/WajDThKAur)
|
|
100
98
|

|
|
101
99
|
[](https://codecov.io/gh/Lightning-AI/litserve)
|
|
@@ -133,16 +131,16 @@ pip install litserve
|
|
|
133
131
|
```
|
|
134
132
|
|
|
135
133
|
### Define a server
|
|
136
|
-
This toy example with 2 models (
|
|
134
|
+
This toy example with 2 models (inference pipeline) shows LitServe's flexibility ([see real examples](#examples)):
|
|
137
135
|
|
|
138
136
|
```python
|
|
139
137
|
# server.py
|
|
140
138
|
import litserve as ls
|
|
141
139
|
|
|
142
|
-
# (STEP 1) - DEFINE THE API (
|
|
140
|
+
# (STEP 1) - DEFINE THE API ("inference" pipeline)
|
|
143
141
|
class SimpleLitAPI(ls.LitAPI):
|
|
144
142
|
def setup(self, device):
|
|
145
|
-
# setup is called once at startup.
|
|
143
|
+
# setup is called once at startup. Defines elements of the pipeline: models, connect DBs, load data, etc...
|
|
146
144
|
self.model1 = lambda x: x**2
|
|
147
145
|
self.model2 = lambda x: x**3
|
|
148
146
|
|
|
@@ -151,11 +149,11 @@ class SimpleLitAPI(ls.LitAPI):
|
|
|
151
149
|
return request["input"]
|
|
152
150
|
|
|
153
151
|
def predict(self, x):
|
|
154
|
-
#
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
return {"output":
|
|
152
|
+
# Run the inference pipeline and return the output
|
|
153
|
+
a = self.model1(x)
|
|
154
|
+
b = self.model2(x)
|
|
155
|
+
c = a + b
|
|
156
|
+
return {"output": c}
|
|
159
157
|
|
|
160
158
|
def encode_response(self, output):
|
|
161
159
|
# Convert the model output to a response payload.
|
|
@@ -168,19 +166,25 @@ if __name__ == "__main__":
|
|
|
168
166
|
server.run(port=8000)
|
|
169
167
|
```
|
|
170
168
|
|
|
171
|
-
Now run the server via the command-line
|
|
169
|
+
Now run the server anywhere (local or cloud) via the command-line.
|
|
172
170
|
|
|
173
171
|
```bash
|
|
174
|
-
|
|
172
|
+
# Deploy to the cloud of your choice via Lightning AI (serverless, autoscaling, etc.)
|
|
173
|
+
lightning serve server.py
|
|
174
|
+
|
|
175
|
+
# Or run locally (self host anywhere)
|
|
176
|
+
lightning serve server.py --local
|
|
175
177
|
```
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
178
|
+
Learn more about managed hosting on [Lightning AI](#hosting-options).
|
|
179
|
+
|
|
180
|
+
You can also run the server manually:
|
|
181
|
+
|
|
182
|
+
```bash
|
|
183
|
+
python server.py
|
|
181
184
|
```
|
|
182
185
|
|
|
183
|
-
|
|
186
|
+
### Test the server
|
|
187
|
+
Simulate an http request (run this on any terminal):
|
|
184
188
|
```bash
|
|
185
189
|
curl -X POST http://127.0.0.1:8000/predict -H "Content-Type: application/json" -d '{"input": 4.0}'
|
|
186
190
|
```
|
|
@@ -197,22 +201,15 @@ litgpt serve microsoft/phi-2
|
|
|
197
201
|
- LitAPI lets you easily build complex AI systems with one or more models ([docs](https://lightning.ai/docs/litserve/api-reference/litapi)).
|
|
198
202
|
- Use the setup method for one-time tasks like connecting models, DBs, and loading data ([docs](https://lightning.ai/docs/litserve/api-reference/litapi#setup)).
|
|
199
203
|
- LitServer handles optimizations like batching, GPU autoscaling, streaming, etc... ([docs](https://lightning.ai/docs/litserve/api-reference/litserver)).
|
|
200
|
-
- Self host on your
|
|
204
|
+
- Self host on your machines or create a fully managed deployment with Lightning ([learn more](https://lightning.ai/docs/litserve/features/deploy-on-cloud)).
|
|
201
205
|
|
|
202
206
|
[Learn how to make this server 200x faster](https://lightning.ai/docs/litserve/home/speed-up-serving-by-200x).
|
|
203
207
|
|
|
204
208
|
|
|
205
209
|
|
|
206
210
|
# Featured examples
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
<div align='center'>
|
|
210
|
-
<div width='200px'>
|
|
211
|
-
<video src="https://github.com/user-attachments/assets/5e73549a-bc0f-47a9-9d9c-5b54389be5de" width='200px' controls></video>
|
|
212
|
-
</div>
|
|
213
|
-
</div>
|
|
214
|
-
|
|
215
|
-
## Examples
|
|
211
|
+
Here are examples of inference pipelines for common model types and use cases.
|
|
212
|
+
|
|
216
213
|
<pre>
|
|
217
214
|
<strong>Toy model:</strong> <a target="_blank" href="#define-a-server">Hello world</a>
|
|
218
215
|
<strong>LLMs:</strong> <a target="_blank" href="https://lightning.ai/lightning-ai/studios/deploy-llama-3-2-vision-with-litserve">Llama 3.2</a>, <a target="_blank" href="https://lightning.ai/lightning-ai/studios/openai-fault-tolerant-proxy-server">LLM Proxy server</a>, <a target="_blank" href="https://lightning.ai/lightning-ai/studios/deploy-ai-agent-with-tool-use">Agent with tool use</a>
|
|
@@ -232,31 +229,63 @@ Use LitServe to deploy any model or AI service: (Compound AI, Gen AI, classic ML
|
|
|
232
229
|
|
|
233
230
|
|
|
234
231
|
|
|
235
|
-
# Features
|
|
236
|
-
State-of-the-art features:
|
|
237
232
|
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
✅ [Build compound systems (1+ models)](https://lightning.ai/docs/litserve/home)
|
|
241
|
-
✅ [GPU autoscaling](https://lightning.ai/docs/litserve/features/gpu-inference)
|
|
242
|
-
✅ [Batching](https://lightning.ai/docs/litserve/features/batching)
|
|
243
|
-
✅ [Streaming](https://lightning.ai/docs/litserve/features/streaming)
|
|
244
|
-
✅ [Worker autoscaling](https://lightning.ai/docs/litserve/features/autoscaling)
|
|
245
|
-
✅ [Self-host on your machines](https://lightning.ai/docs/litserve/features/hosting-methods#host-on-your-own)
|
|
246
|
-
✅ [Host fully managed on Lightning AI](https://lightning.ai/docs/litserve/features/hosting-methods#host-on-lightning-studios)
|
|
247
|
-
✅ [Serve all models: (LLMs, vision, etc.)](https://lightning.ai/docs/litserve/examples)
|
|
248
|
-
✅ [Scale to zero (serverless)](https://lightning.ai/docs/litserve/features/streaming)
|
|
249
|
-
✅ [Supports PyTorch, JAX, TF, etc...](https://lightning.ai/docs/litserve/features/full-control)
|
|
250
|
-
✅ [OpenAPI compliant](https://www.openapis.org/)
|
|
251
|
-
✅ [Open AI compatibility](https://lightning.ai/docs/litserve/features/open-ai-spec)
|
|
252
|
-
✅ [Authentication](https://lightning.ai/docs/litserve/features/authentication)
|
|
253
|
-
✅ [Dockerization](https://lightning.ai/docs/litserve/features/dockerization-deployment)
|
|
233
|
+
# Hosting options
|
|
234
|
+
Self host LitServe anywhere or deploy to your favorite cloud via [Lightning AI](http://lightning.ai/deploy).
|
|
254
235
|
|
|
236
|
+
https://github.com/user-attachments/assets/ff83dab9-0c9f-4453-8dcb-fb9526726344
|
|
255
237
|
|
|
238
|
+
Self-hosting is ideal for hackers, students, and DIY developers while fully managed hosting is ideal for enterprise developers needing easy autoscaling, security, release management, and 99.995% uptime and observability.
|
|
256
239
|
|
|
257
|
-
|
|
240
|
+
*Note:* Lightning offers a generous free tier for developers.
|
|
258
241
|
|
|
259
|
-
|
|
242
|
+
To host on [Lightning AI](https://lightning.ai/deploy), simply run the command, login and choose the cloud of your choice.
|
|
243
|
+
```bash
|
|
244
|
+
lightning serve server.py
|
|
245
|
+
```
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
## Features
|
|
250
|
+
|
|
251
|
+
<div align='center'>
|
|
252
|
+
|
|
253
|
+
| [Feature](https://lightning.ai/docs/litserve/features) | Self Managed | [Fully Managed on Lightning](https://lightning.ai/deploy) |
|
|
254
|
+
|----------------------------------------------------------------------|-----------------------------------|------------------------------------|
|
|
255
|
+
| Docker-first deployment | ✅ DIY | ✅ One-click deploy |
|
|
256
|
+
| Cost | ✅ Free (DIY) | ✅ Generous [free tier](https://lightning.ai/pricing) with pay as you go |
|
|
257
|
+
| Full control | ✅ | ✅ |
|
|
258
|
+
| Use any engine (vLLM, etc.) | ✅ | ✅ vLLM, Ollama, LitServe, etc. |
|
|
259
|
+
| Own VPC | ✅ (manual setup) | ✅ Connect your own VPC |
|
|
260
|
+
| [(2x)+ faster than plain FastAPI](#performance) | ✅ | ✅ |
|
|
261
|
+
| [Bring your own model](https://lightning.ai/docs/litserve/features/full-control) | ✅ | ✅ |
|
|
262
|
+
| [Build compound systems (1+ models)](https://lightning.ai/docs/litserve/home) | ✅ | ✅ |
|
|
263
|
+
| [GPU autoscaling](https://lightning.ai/docs/litserve/features/gpu-inference) | ✅ | ✅ |
|
|
264
|
+
| [Batching](https://lightning.ai/docs/litserve/features/batching) | ✅ | ✅ |
|
|
265
|
+
| [Streaming](https://lightning.ai/docs/litserve/features/streaming) | ✅ | ✅ |
|
|
266
|
+
| [Worker autoscaling](https://lightning.ai/docs/litserve/features/autoscaling) | ✅ | ✅ |
|
|
267
|
+
| [Serve all models: (LLMs, vision, etc.)](https://lightning.ai/docs/litserve/examples) | ✅ | ✅ |
|
|
268
|
+
| [Supports PyTorch, JAX, TF, etc...](https://lightning.ai/docs/litserve/features/full-control) | ✅ | ✅ |
|
|
269
|
+
| [OpenAPI compliant](https://www.openapis.org/) | ✅ | ✅ |
|
|
270
|
+
| [Open AI compatibility](https://lightning.ai/docs/litserve/features/open-ai-spec) | ✅ | ✅ |
|
|
271
|
+
| [Authentication](https://lightning.ai/docs/litserve/features/authentication) | ❌ DIY | ✅ Token, password, custom |
|
|
272
|
+
| GPUs | ❌ DIY | ✅ 8+ GPU types, H100s from $1.75 |
|
|
273
|
+
| Load balancing | ❌ | ✅ Built-in |
|
|
274
|
+
| Scale to zero (serverless) | ❌ | ✅ No machine runs when idle |
|
|
275
|
+
| Autoscale up on demand | ❌ | ✅ Auto scale up/down |
|
|
276
|
+
| Multi-node inference | ❌ | ✅ Distribute across nodes |
|
|
277
|
+
| Use AWS/GCP credits | ❌ | ✅ Use existing cloud commits |
|
|
278
|
+
| Versioning | ❌ | ✅ Make and roll back releases |
|
|
279
|
+
| Enterprise-grade uptime (99.95%) | ❌ | ✅ SLA-backed |
|
|
280
|
+
| SOC2 / HIPAA compliance | ❌ | ✅ Certified & secure |
|
|
281
|
+
| Observability | ❌ | ✅ Built-in, connect 3rd party tools|
|
|
282
|
+
| CI/CD ready | ❌ | ✅ Lightning SDK |
|
|
283
|
+
| 24/7 enterprise support | ❌ | ✅ Dedicated support |
|
|
284
|
+
| Cost controls & audit logs | ❌ | ✅ Budgets, breakdowns, logs |
|
|
285
|
+
| Debug on GPUs | ❌ | ✅ Studio integration |
|
|
286
|
+
| [20+ features](https://lightning.ai/docs/litserve/features) | - | - |
|
|
287
|
+
|
|
288
|
+
</div>
|
|
260
289
|
|
|
261
290
|
|
|
262
291
|
|
|
@@ -275,40 +304,8 @@ These results are for image and text classification ML tasks. The performance re
|
|
|
275
304
|
|
|
276
305
|
***💡 Note on LLM serving:*** For high-performance LLM serving (like Ollama/vLLM), integrate [vLLM with LitServe](https://lightning.ai/lightning-ai/studios/deploy-a-private-llama-3-2-rag-api), use [LitGPT](https://github.com/Lightning-AI/litgpt?tab=readme-ov-file#deploy-an-llm), or build your custom vLLM-like server with LitServe. Optimizations like kv-caching, which can be done with LitServe, are needed to maximize LLM performance.
|
|
277
306
|
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
# Hosting options
|
|
281
|
-
LitServe can be hosted independently on your own machines or fully managed via Lightning Studios.
|
|
282
|
-
|
|
283
|
-
Self-hosting is ideal for hackers, students, and DIY developers, while fully managed hosting is ideal for enterprise developers needing easy autoscaling, security, release management, and 99.995% uptime and observability.
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
<div align="center">
|
|
288
|
-
<a target="_blank" href="https://lightning.ai/lightning-ai/studios/litserve-hello-world">
|
|
289
|
-
<img src="https://pl-bolts-doc-images.s3.us-east-2.amazonaws.com/app-2/host-on-lightning.svg" alt="Host on Lightning"/>
|
|
290
|
-
</a>
|
|
291
|
-
</div>
|
|
292
|
-
|
|
293
307
|
|
|
294
308
|
|
|
295
|
-
<div align='center'>
|
|
296
|
-
|
|
297
|
-
| Feature | Self Managed | Fully Managed on Studios |
|
|
298
|
-
|----------------------------------|-----------------------------------|-------------------------------------|
|
|
299
|
-
| Deployment | ✅ Do it yourself deployment | ✅ One-button cloud deploy |
|
|
300
|
-
| Load balancing | ❌ | ✅ |
|
|
301
|
-
| Autoscaling | ❌ | ✅ |
|
|
302
|
-
| Scale to zero | ❌ | ✅ |
|
|
303
|
-
| Multi-machine inference | ❌ | ✅ |
|
|
304
|
-
| Authentication | ❌ | ✅ |
|
|
305
|
-
| Own VPC | ❌ | ✅ |
|
|
306
|
-
| AWS, GCP | ❌ | ✅ |
|
|
307
|
-
| Use your own cloud commits | ❌ | ✅ |
|
|
308
|
-
|
|
309
|
-
</div>
|
|
310
|
-
|
|
311
|
-
|
|
312
309
|
|
|
313
310
|
# Community
|
|
314
311
|
LitServe is a [community project accepting contributions](https://lightning.ai/docs/litserve/community) - Let's make the world's most advanced AI inference engine.
|
|
@@ -1,32 +1,29 @@
|
|
|
1
1
|
<div align='center'>
|
|
2
2
|
|
|
3
|
-
#
|
|
3
|
+
# Deploy AI models and inference pipelines - ⚡ fast
|
|
4
4
|
|
|
5
5
|
<img alt="Lightning" src="https://pl-bolts-doc-images.s3.us-east-2.amazonaws.com/app-2/ls_banner2.png" width="800px" style="max-width: 100%;">
|
|
6
6
|
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
<strong>Lightning-fast serving engine for AI models.</strong>
|
|
10
|
-
Easy. Flexible. Enterprise-scale.
|
|
7
|
+
|
|
11
8
|
</div>
|
|
12
9
|
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
**LitServe** is an easy-to-use, flexible serving engine for AI models built on FastAPI. It augments FastAPI with features like batching, streaming, and GPU autoscaling eliminate the need to rebuild a FastAPI server per model.
|
|
10
|
+
**LitServe** lets you build high-performance AI inference pipelines on top of FastAPI - no boilerplate. Define one or more models, connect vector DBs, stream responses, batch requests, and autoscale on GPUs out of the box.
|
|
16
11
|
|
|
17
12
|
LitServe is at least [2x faster](#performance) than plain FastAPI due to AI-specific multi-worker handling.
|
|
18
13
|
|
|
19
14
|
<div align='center'>
|
|
20
15
|
|
|
21
16
|
<pre>
|
|
22
|
-
✅ (2x)+ faster serving ✅ Easy to use
|
|
23
|
-
✅ Bring your own model ✅ PyTorch/JAX/TF/...
|
|
24
|
-
✅ GPU autoscaling ✅ Batching, Streaming
|
|
25
|
-
✅
|
|
17
|
+
✅ (2x)+ faster serving ✅ Easy to use ✅ LLMs, non LLMs and more
|
|
18
|
+
✅ Bring your own model ✅ PyTorch/JAX/TF/... ✅ Built on FastAPI
|
|
19
|
+
✅ GPU autoscaling ✅ Batching, Streaming ✅ Self-host or ⚡️ managed
|
|
20
|
+
✅ Inference pipeline ✅ Integrate with vLLM, etc ✅ Serverless
|
|
21
|
+
|
|
26
22
|
</pre>
|
|
27
23
|
|
|
28
24
|
<div align='center'>
|
|
29
25
|
|
|
26
|
+
[](https://pepy.tech/projects/litserve)
|
|
30
27
|
[](https://discord.gg/WajDThKAur)
|
|
31
28
|

|
|
32
29
|
[](https://codecov.io/gh/Lightning-AI/litserve)
|
|
@@ -64,16 +61,16 @@ pip install litserve
|
|
|
64
61
|
```
|
|
65
62
|
|
|
66
63
|
### Define a server
|
|
67
|
-
This toy example with 2 models (
|
|
64
|
+
This toy example with 2 models (inference pipeline) shows LitServe's flexibility ([see real examples](#examples)):
|
|
68
65
|
|
|
69
66
|
```python
|
|
70
67
|
# server.py
|
|
71
68
|
import litserve as ls
|
|
72
69
|
|
|
73
|
-
# (STEP 1) - DEFINE THE API (
|
|
70
|
+
# (STEP 1) - DEFINE THE API ("inference" pipeline)
|
|
74
71
|
class SimpleLitAPI(ls.LitAPI):
|
|
75
72
|
def setup(self, device):
|
|
76
|
-
# setup is called once at startup.
|
|
73
|
+
# setup is called once at startup. Defines elements of the pipeline: models, connect DBs, load data, etc...
|
|
77
74
|
self.model1 = lambda x: x**2
|
|
78
75
|
self.model2 = lambda x: x**3
|
|
79
76
|
|
|
@@ -82,11 +79,11 @@ class SimpleLitAPI(ls.LitAPI):
|
|
|
82
79
|
return request["input"]
|
|
83
80
|
|
|
84
81
|
def predict(self, x):
|
|
85
|
-
#
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
return {"output":
|
|
82
|
+
# Run the inference pipeline and return the output
|
|
83
|
+
a = self.model1(x)
|
|
84
|
+
b = self.model2(x)
|
|
85
|
+
c = a + b
|
|
86
|
+
return {"output": c}
|
|
90
87
|
|
|
91
88
|
def encode_response(self, output):
|
|
92
89
|
# Convert the model output to a response payload.
|
|
@@ -99,19 +96,25 @@ if __name__ == "__main__":
|
|
|
99
96
|
server.run(port=8000)
|
|
100
97
|
```
|
|
101
98
|
|
|
102
|
-
Now run the server via the command-line
|
|
99
|
+
Now run the server anywhere (local or cloud) via the command-line.
|
|
103
100
|
|
|
104
101
|
```bash
|
|
105
|
-
|
|
102
|
+
# Deploy to the cloud of your choice via Lightning AI (serverless, autoscaling, etc.)
|
|
103
|
+
lightning serve server.py
|
|
104
|
+
|
|
105
|
+
# Or run locally (self host anywhere)
|
|
106
|
+
lightning serve server.py --local
|
|
106
107
|
```
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
108
|
+
Learn more about managed hosting on [Lightning AI](#hosting-options).
|
|
109
|
+
|
|
110
|
+
You can also run the server manually:
|
|
111
|
+
|
|
112
|
+
```bash
|
|
113
|
+
python server.py
|
|
112
114
|
```
|
|
113
115
|
|
|
114
|
-
|
|
116
|
+
### Test the server
|
|
117
|
+
Simulate an http request (run this on any terminal):
|
|
115
118
|
```bash
|
|
116
119
|
curl -X POST http://127.0.0.1:8000/predict -H "Content-Type: application/json" -d '{"input": 4.0}'
|
|
117
120
|
```
|
|
@@ -128,22 +131,15 @@ litgpt serve microsoft/phi-2
|
|
|
128
131
|
- LitAPI lets you easily build complex AI systems with one or more models ([docs](https://lightning.ai/docs/litserve/api-reference/litapi)).
|
|
129
132
|
- Use the setup method for one-time tasks like connecting models, DBs, and loading data ([docs](https://lightning.ai/docs/litserve/api-reference/litapi#setup)).
|
|
130
133
|
- LitServer handles optimizations like batching, GPU autoscaling, streaming, etc... ([docs](https://lightning.ai/docs/litserve/api-reference/litserver)).
|
|
131
|
-
- Self host on your
|
|
134
|
+
- Self host on your machines or create a fully managed deployment with Lightning ([learn more](https://lightning.ai/docs/litserve/features/deploy-on-cloud)).
|
|
132
135
|
|
|
133
136
|
[Learn how to make this server 200x faster](https://lightning.ai/docs/litserve/home/speed-up-serving-by-200x).
|
|
134
137
|
|
|
135
138
|
|
|
136
139
|
|
|
137
140
|
# Featured examples
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
<div align='center'>
|
|
141
|
-
<div width='200px'>
|
|
142
|
-
<video src="https://github.com/user-attachments/assets/5e73549a-bc0f-47a9-9d9c-5b54389be5de" width='200px' controls></video>
|
|
143
|
-
</div>
|
|
144
|
-
</div>
|
|
145
|
-
|
|
146
|
-
## Examples
|
|
141
|
+
Here are examples of inference pipelines for common model types and use cases.
|
|
142
|
+
|
|
147
143
|
<pre>
|
|
148
144
|
<strong>Toy model:</strong> <a target="_blank" href="#define-a-server">Hello world</a>
|
|
149
145
|
<strong>LLMs:</strong> <a target="_blank" href="https://lightning.ai/lightning-ai/studios/deploy-llama-3-2-vision-with-litserve">Llama 3.2</a>, <a target="_blank" href="https://lightning.ai/lightning-ai/studios/openai-fault-tolerant-proxy-server">LLM Proxy server</a>, <a target="_blank" href="https://lightning.ai/lightning-ai/studios/deploy-ai-agent-with-tool-use">Agent with tool use</a>
|
|
@@ -163,31 +159,63 @@ Use LitServe to deploy any model or AI service: (Compound AI, Gen AI, classic ML
|
|
|
163
159
|
|
|
164
160
|
|
|
165
161
|
|
|
166
|
-
# Features
|
|
167
|
-
State-of-the-art features:
|
|
168
162
|
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
✅ [Build compound systems (1+ models)](https://lightning.ai/docs/litserve/home)
|
|
172
|
-
✅ [GPU autoscaling](https://lightning.ai/docs/litserve/features/gpu-inference)
|
|
173
|
-
✅ [Batching](https://lightning.ai/docs/litserve/features/batching)
|
|
174
|
-
✅ [Streaming](https://lightning.ai/docs/litserve/features/streaming)
|
|
175
|
-
✅ [Worker autoscaling](https://lightning.ai/docs/litserve/features/autoscaling)
|
|
176
|
-
✅ [Self-host on your machines](https://lightning.ai/docs/litserve/features/hosting-methods#host-on-your-own)
|
|
177
|
-
✅ [Host fully managed on Lightning AI](https://lightning.ai/docs/litserve/features/hosting-methods#host-on-lightning-studios)
|
|
178
|
-
✅ [Serve all models: (LLMs, vision, etc.)](https://lightning.ai/docs/litserve/examples)
|
|
179
|
-
✅ [Scale to zero (serverless)](https://lightning.ai/docs/litserve/features/streaming)
|
|
180
|
-
✅ [Supports PyTorch, JAX, TF, etc...](https://lightning.ai/docs/litserve/features/full-control)
|
|
181
|
-
✅ [OpenAPI compliant](https://www.openapis.org/)
|
|
182
|
-
✅ [Open AI compatibility](https://lightning.ai/docs/litserve/features/open-ai-spec)
|
|
183
|
-
✅ [Authentication](https://lightning.ai/docs/litserve/features/authentication)
|
|
184
|
-
✅ [Dockerization](https://lightning.ai/docs/litserve/features/dockerization-deployment)
|
|
163
|
+
# Hosting options
|
|
164
|
+
Self host LitServe anywhere or deploy to your favorite cloud via [Lightning AI](http://lightning.ai/deploy).
|
|
185
165
|
|
|
166
|
+
https://github.com/user-attachments/assets/ff83dab9-0c9f-4453-8dcb-fb9526726344
|
|
186
167
|
|
|
168
|
+
Self-hosting is ideal for hackers, students, and DIY developers while fully managed hosting is ideal for enterprise developers needing easy autoscaling, security, release management, and 99.995% uptime and observability.
|
|
187
169
|
|
|
188
|
-
|
|
170
|
+
*Note:* Lightning offers a generous free tier for developers.
|
|
189
171
|
|
|
190
|
-
|
|
172
|
+
To host on [Lightning AI](https://lightning.ai/deploy), simply run the command, login and choose the cloud of your choice.
|
|
173
|
+
```bash
|
|
174
|
+
lightning serve server.py
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
## Features
|
|
180
|
+
|
|
181
|
+
<div align='center'>
|
|
182
|
+
|
|
183
|
+
| [Feature](https://lightning.ai/docs/litserve/features) | Self Managed | [Fully Managed on Lightning](https://lightning.ai/deploy) |
|
|
184
|
+
|----------------------------------------------------------------------|-----------------------------------|------------------------------------|
|
|
185
|
+
| Docker-first deployment | ✅ DIY | ✅ One-click deploy |
|
|
186
|
+
| Cost | ✅ Free (DIY) | ✅ Generous [free tier](https://lightning.ai/pricing) with pay as you go |
|
|
187
|
+
| Full control | ✅ | ✅ |
|
|
188
|
+
| Use any engine (vLLM, etc.) | ✅ | ✅ vLLM, Ollama, LitServe, etc. |
|
|
189
|
+
| Own VPC | ✅ (manual setup) | ✅ Connect your own VPC |
|
|
190
|
+
| [(2x)+ faster than plain FastAPI](#performance) | ✅ | ✅ |
|
|
191
|
+
| [Bring your own model](https://lightning.ai/docs/litserve/features/full-control) | ✅ | ✅ |
|
|
192
|
+
| [Build compound systems (1+ models)](https://lightning.ai/docs/litserve/home) | ✅ | ✅ |
|
|
193
|
+
| [GPU autoscaling](https://lightning.ai/docs/litserve/features/gpu-inference) | ✅ | ✅ |
|
|
194
|
+
| [Batching](https://lightning.ai/docs/litserve/features/batching) | ✅ | ✅ |
|
|
195
|
+
| [Streaming](https://lightning.ai/docs/litserve/features/streaming) | ✅ | ✅ |
|
|
196
|
+
| [Worker autoscaling](https://lightning.ai/docs/litserve/features/autoscaling) | ✅ | ✅ |
|
|
197
|
+
| [Serve all models: (LLMs, vision, etc.)](https://lightning.ai/docs/litserve/examples) | ✅ | ✅ |
|
|
198
|
+
| [Supports PyTorch, JAX, TF, etc...](https://lightning.ai/docs/litserve/features/full-control) | ✅ | ✅ |
|
|
199
|
+
| [OpenAPI compliant](https://www.openapis.org/) | ✅ | ✅ |
|
|
200
|
+
| [Open AI compatibility](https://lightning.ai/docs/litserve/features/open-ai-spec) | ✅ | ✅ |
|
|
201
|
+
| [Authentication](https://lightning.ai/docs/litserve/features/authentication) | ❌ DIY | ✅ Token, password, custom |
|
|
202
|
+
| GPUs | ❌ DIY | ✅ 8+ GPU types, H100s from $1.75 |
|
|
203
|
+
| Load balancing | ❌ | ✅ Built-in |
|
|
204
|
+
| Scale to zero (serverless) | ❌ | ✅ No machine runs when idle |
|
|
205
|
+
| Autoscale up on demand | ❌ | ✅ Auto scale up/down |
|
|
206
|
+
| Multi-node inference | ❌ | ✅ Distribute across nodes |
|
|
207
|
+
| Use AWS/GCP credits | ❌ | ✅ Use existing cloud commits |
|
|
208
|
+
| Versioning | ❌ | ✅ Make and roll back releases |
|
|
209
|
+
| Enterprise-grade uptime (99.95%) | ❌ | ✅ SLA-backed |
|
|
210
|
+
| SOC2 / HIPAA compliance | ❌ | ✅ Certified & secure |
|
|
211
|
+
| Observability | ❌ | ✅ Built-in, connect 3rd party tools|
|
|
212
|
+
| CI/CD ready | ❌ | ✅ Lightning SDK |
|
|
213
|
+
| 24/7 enterprise support | ❌ | ✅ Dedicated support |
|
|
214
|
+
| Cost controls & audit logs | ❌ | ✅ Budgets, breakdowns, logs |
|
|
215
|
+
| Debug on GPUs | ❌ | ✅ Studio integration |
|
|
216
|
+
| [20+ features](https://lightning.ai/docs/litserve/features) | - | - |
|
|
217
|
+
|
|
218
|
+
</div>
|
|
191
219
|
|
|
192
220
|
|
|
193
221
|
|
|
@@ -206,40 +234,8 @@ These results are for image and text classification ML tasks. The performance re
|
|
|
206
234
|
|
|
207
235
|
***💡 Note on LLM serving:*** For high-performance LLM serving (like Ollama/vLLM), integrate [vLLM with LitServe](https://lightning.ai/lightning-ai/studios/deploy-a-private-llama-3-2-rag-api), use [LitGPT](https://github.com/Lightning-AI/litgpt?tab=readme-ov-file#deploy-an-llm), or build your custom vLLM-like server with LitServe. Optimizations like kv-caching, which can be done with LitServe, are needed to maximize LLM performance.
|
|
208
236
|
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
# Hosting options
|
|
212
|
-
LitServe can be hosted independently on your own machines or fully managed via Lightning Studios.
|
|
213
|
-
|
|
214
|
-
Self-hosting is ideal for hackers, students, and DIY developers, while fully managed hosting is ideal for enterprise developers needing easy autoscaling, security, release management, and 99.995% uptime and observability.
|
|
215
|
-
|
|
216
237
|
|
|
217
238
|
|
|
218
|
-
<div align="center">
|
|
219
|
-
<a target="_blank" href="https://lightning.ai/lightning-ai/studios/litserve-hello-world">
|
|
220
|
-
<img src="https://pl-bolts-doc-images.s3.us-east-2.amazonaws.com/app-2/host-on-lightning.svg" alt="Host on Lightning"/>
|
|
221
|
-
</a>
|
|
222
|
-
</div>
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
<div align='center'>
|
|
227
|
-
|
|
228
|
-
| Feature | Self Managed | Fully Managed on Studios |
|
|
229
|
-
|----------------------------------|-----------------------------------|-------------------------------------|
|
|
230
|
-
| Deployment | ✅ Do it yourself deployment | ✅ One-button cloud deploy |
|
|
231
|
-
| Load balancing | ❌ | ✅ |
|
|
232
|
-
| Autoscaling | ❌ | ✅ |
|
|
233
|
-
| Scale to zero | ❌ | ✅ |
|
|
234
|
-
| Multi-machine inference | ❌ | ✅ |
|
|
235
|
-
| Authentication | ❌ | ✅ |
|
|
236
|
-
| Own VPC | ❌ | ✅ |
|
|
237
|
-
| AWS, GCP | ❌ | ✅ |
|
|
238
|
-
| Use your own cloud commits | ❌ | ✅ |
|
|
239
|
-
|
|
240
|
-
</div>
|
|
241
|
-
|
|
242
|
-
|
|
243
239
|
|
|
244
240
|
# Community
|
|
245
241
|
LitServe is a [community project accepting contributions](https://lightning.ai/docs/litserve/community) - Let's make the world's most advanced AI inference engine.
|
|
@@ -111,8 +111,6 @@ setup(
|
|
|
111
111
|
"Programming Language :: Python :: 3.11",
|
|
112
112
|
],
|
|
113
113
|
entry_points={
|
|
114
|
-
"console_scripts": [
|
|
115
|
-
"litserve=litserve.__main__:main",
|
|
116
|
-
],
|
|
114
|
+
"console_scripts": ["litserve=litserve.__main__:main", "lightning=litserve.cli:main"],
|
|
117
115
|
},
|
|
118
116
|
)
|
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
12
|
# See the License for the specific language governing permissions and
|
|
13
13
|
# limitations under the License.
|
|
14
|
-
__version__ = "0.2.
|
|
14
|
+
__version__ = "0.2.8"
|
|
15
15
|
__author__ = "Lightning-AI et al."
|
|
16
16
|
__author_email__ = "community@lightning.ai"
|
|
17
17
|
__license__ = "Apache-2.0"
|
|
@@ -15,7 +15,7 @@ import json
|
|
|
15
15
|
import warnings
|
|
16
16
|
from abc import ABC, abstractmethod
|
|
17
17
|
from queue import Queue
|
|
18
|
-
from typing import Optional
|
|
18
|
+
from typing import Callable, Optional
|
|
19
19
|
|
|
20
20
|
from pydantic import BaseModel
|
|
21
21
|
|
|
@@ -24,8 +24,8 @@ from litserve.specs.base import LitSpec
|
|
|
24
24
|
|
|
25
25
|
class LitAPI(ABC):
|
|
26
26
|
_stream: bool = False
|
|
27
|
-
_default_unbatch:
|
|
28
|
-
_spec: LitSpec = None
|
|
27
|
+
_default_unbatch: Optional[Callable] = None
|
|
28
|
+
_spec: Optional[LitSpec] = None
|
|
29
29
|
_device: Optional[str] = None
|
|
30
30
|
_logger_queue: Optional[Queue] = None
|
|
31
31
|
request_timeout: Optional[float] = None
|
|
@@ -76,6 +76,11 @@ class LitAPI(ABC):
|
|
|
76
76
|
|
|
77
77
|
def unbatch(self, output):
|
|
78
78
|
"""Convert a batched output to a list of outputs."""
|
|
79
|
+
if self._default_unbatch is None:
|
|
80
|
+
raise ValueError(
|
|
81
|
+
"Default implementation for `LitAPI.unbatch` method was not found. "
|
|
82
|
+
"Please implement the `LitAPI.unbatch` method."
|
|
83
|
+
)
|
|
79
84
|
return self._default_unbatch(output)
|
|
80
85
|
|
|
81
86
|
def encode_response(self, output, **kwargs):
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import importlib.util
|
|
2
|
+
import subprocess
|
|
3
|
+
import sys
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def _ensure_lightning_installed():
|
|
7
|
+
if not importlib.util.find_spec("lightning_sdk"):
|
|
8
|
+
print("Lightning CLI not found. Installing...")
|
|
9
|
+
subprocess.check_call([sys.executable, "-m", "pip", "install", "-U", "lightning-sdk"])
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def main():
|
|
13
|
+
_ensure_lightning_installed()
|
|
14
|
+
|
|
15
|
+
# Forward CLI arguments to the real lightning command
|
|
16
|
+
cli_args = sys.argv[1:]
|
|
17
|
+
subprocess.run(["lightning"] + cli_args)
|