litserve 0.2.7.dev0__tar.gz → 0.2.8__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. {litserve-0.2.7.dev0/src/litserve.egg-info → litserve-0.2.8}/PKG-INFO +93 -96
  2. {litserve-0.2.7.dev0 → litserve-0.2.8}/README.md +86 -90
  3. {litserve-0.2.7.dev0 → litserve-0.2.8}/setup.py +1 -3
  4. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/__about__.py +1 -1
  5. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/api.py +8 -3
  6. litserve-0.2.8/src/litserve/cli.py +17 -0
  7. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/connector.py +7 -12
  8. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/loops/base.py +12 -32
  9. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/loops/continuous_batching_loop.py +6 -5
  10. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/loops/loops.py +12 -20
  11. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/loops/simple_loops.py +17 -16
  12. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/loops/streaming_loops.py +19 -18
  13. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/server.py +59 -49
  14. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/specs/openai.py +12 -2
  15. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/specs/openai_embedding.py +5 -2
  16. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/test_examples/openai_spec_example.py +22 -3
  17. litserve-0.2.8/src/litserve/transport/__init__.py +4 -0
  18. litserve-0.2.8/src/litserve/transport/base.py +18 -0
  19. litserve-0.2.8/src/litserve/transport/factory.py +38 -0
  20. litserve-0.2.8/src/litserve/transport/process_transport.py +44 -0
  21. {litserve-0.2.7.dev0/src/litserve → litserve-0.2.8/src/litserve/transport}/zmq_queue.py +1 -1
  22. litserve-0.2.8/src/litserve/transport/zmq_transport.py +42 -0
  23. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/utils.py +2 -0
  24. {litserve-0.2.7.dev0 → litserve-0.2.8/src/litserve.egg-info}/PKG-INFO +93 -96
  25. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve.egg-info/SOURCES.txt +8 -2
  26. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve.egg-info/entry_points.txt +1 -0
  27. {litserve-0.2.7.dev0 → litserve-0.2.8}/LICENSE +0 -0
  28. {litserve-0.2.7.dev0 → litserve-0.2.8}/MANIFEST.in +0 -0
  29. {litserve-0.2.7.dev0 → litserve-0.2.8}/requirements.txt +0 -0
  30. {litserve-0.2.7.dev0 → litserve-0.2.8}/setup.cfg +0 -0
  31. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/__init__.py +0 -0
  32. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/__main__.py +0 -0
  33. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/callbacks/__init__.py +0 -0
  34. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/callbacks/base.py +0 -0
  35. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/callbacks/defaults/__init__.py +0 -0
  36. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/callbacks/defaults/metric_callback.py +0 -0
  37. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/docker_builder.py +0 -0
  38. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/loggers.py +0 -0
  39. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/loops/__init__.py +0 -0
  40. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/middlewares.py +0 -0
  41. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/python_client.py +0 -0
  42. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/schema/__init__.py +0 -0
  43. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/schema/image.py +0 -0
  44. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/specs/__init__.py +0 -0
  45. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/specs/base.py +0 -0
  46. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/test_examples/__init__.py +0 -0
  47. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/test_examples/openai_embedding_spec_example.py +0 -0
  48. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve/test_examples/simple_example.py +0 -0
  49. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve.egg-info/dependency_links.txt +0 -0
  50. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve.egg-info/not-zip-safe +0 -0
  51. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve.egg-info/requires.txt +0 -0
  52. {litserve-0.2.7.dev0 → litserve-0.2.8}/src/litserve.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.2
1
+ Metadata-Version: 2.4
2
2
  Name: litserve
3
- Version: 0.2.7.dev0
3
+ Version: 0.2.8
4
4
  Summary: Lightweight AI server.
5
5
  Home-page: https://github.com/Lightning-AI/litserve
6
6
  Download-URL: https://github.com/Lightning-AI/litserve
@@ -30,10 +30,6 @@ License-File: LICENSE
30
30
  Requires-Dist: fastapi>=0.100
31
31
  Requires-Dist: uvicorn[standard]>=0.29.0
32
32
  Requires-Dist: pyzmq>=22.0.0
33
- Provides-Extra: perf
34
- Requires-Dist: jsonargparse; extra == "perf"
35
- Requires-Dist: tenacity; extra == "perf"
36
- Requires-Dist: uvloop; extra == "perf"
37
33
  Provides-Extra: test
38
34
  Requires-Dist: asgi-lifespan; extra == "test"
39
35
  Requires-Dist: coverage[toml]>=7.5.3; extra == "test"
@@ -52,6 +48,10 @@ Requires-Dist: python-multipart; extra == "test"
52
48
  Requires-Dist: requests; extra == "test"
53
49
  Requires-Dist: torch>2.0.0; extra == "test"
54
50
  Requires-Dist: transformers; extra == "test"
51
+ Provides-Extra: perf
52
+ Requires-Dist: jsonargparse; extra == "perf"
53
+ Requires-Dist: tenacity; extra == "perf"
54
+ Requires-Dist: uvloop; extra == "perf"
55
55
  Dynamic: author
56
56
  Dynamic: author-email
57
57
  Dynamic: classifier
@@ -61,6 +61,7 @@ Dynamic: download-url
61
61
  Dynamic: home-page
62
62
  Dynamic: keywords
63
63
  Dynamic: license
64
+ Dynamic: license-file
64
65
  Dynamic: project-url
65
66
  Dynamic: provides-extra
66
67
  Dynamic: requires-dist
@@ -69,33 +70,30 @@ Dynamic: summary
69
70
 
70
71
  <div align='center'>
71
72
 
72
- # Easily serve AI models Lightning fast ⚡
73
+ # Deploy AI models and inference pipelines - ⚡ fast
73
74
 
74
75
  <img alt="Lightning" src="https://pl-bolts-doc-images.s3.us-east-2.amazonaws.com/app-2/ls_banner2.png" width="800px" style="max-width: 100%;">
75
76
 
76
- &nbsp;
77
-
78
- <strong>Lightning-fast serving engine for AI models.</strong>
79
- Easy. Flexible. Enterprise-scale.
77
+ &nbsp;
80
78
  </div>
81
79
 
82
- ----
83
-
84
- **LitServe** is an easy-to-use, flexible serving engine for AI models built on FastAPI. It augments FastAPI with features like batching, streaming, and GPU autoscaling eliminate the need to rebuild a FastAPI server per model.
80
+ **LitServe** lets you build high-performance AI inference pipelines on top of FastAPI - no boilerplate. Define one or more models, connect vector DBs, stream responses, batch requests, and autoscale on GPUs out of the box.
85
81
 
86
82
  LitServe is at least [2x faster](#performance) than plain FastAPI due to AI-specific multi-worker handling.
87
83
 
88
84
  <div align='center'>
89
85
 
90
86
  <pre>
91
- ✅ (2x)+ faster serving ✅ Easy to use ✅ LLMs, non LLMs and more
92
- ✅ Bring your own model ✅ PyTorch/JAX/TF/... ✅ Built on FastAPI
93
- ✅ GPU autoscaling ✅ Batching, Streaming ✅ Self-host or ⚡️ managed
94
- ✅ Compound AI ✅ Integrate with vLLM and more
87
+ ✅ (2x)+ faster serving ✅ Easy to use ✅ LLMs, non LLMs and more
88
+ ✅ Bring your own model ✅ PyTorch/JAX/TF/... ✅ Built on FastAPI
89
+ ✅ GPU autoscaling ✅ Batching, Streaming ✅ Self-host or ⚡️ managed
90
+ ✅ Inference pipeline ✅ Integrate with vLLM, etc ✅ Serverless  
91
+  
95
92
  </pre>
96
93
 
97
94
  <div align='center'>
98
95
 
96
+ [![PyPI Downloads](https://static.pepy.tech/badge/litserve)](https://pepy.tech/projects/litserve)
99
97
  [![Discord](https://img.shields.io/discord/1077906959069626439?label=Get%20help%20on%20Discord)](https://discord.gg/WajDThKAur)
100
98
  ![cpu-tests](https://github.com/Lightning-AI/litserve/actions/workflows/ci-testing.yml/badge.svg)
101
99
  [![codecov](https://codecov.io/gh/Lightning-AI/litserve/graph/badge.svg?token=SmzX8mnKlA)](https://codecov.io/gh/Lightning-AI/litserve)
@@ -133,16 +131,16 @@ pip install litserve
133
131
  ```
134
132
 
135
133
  ### Define a server
136
- This toy example with 2 models (AI compound system) shows LitServe's flexibility ([see real examples](#examples)):
134
+ This toy example with 2 models (inference pipeline) shows LitServe's flexibility ([see real examples](#examples)):
137
135
 
138
136
  ```python
139
137
  # server.py
140
138
  import litserve as ls
141
139
 
142
- # (STEP 1) - DEFINE THE API (compound AI system)
140
+ # (STEP 1) - DEFINE THE API ("inference" pipeline)
143
141
  class SimpleLitAPI(ls.LitAPI):
144
142
  def setup(self, device):
145
- # setup is called once at startup. Build a compound AI system (1+ models), connect DBs, load data, etc...
143
+ # setup is called once at startup. Defines elements of the pipeline: models, connect DBs, load data, etc...
146
144
  self.model1 = lambda x: x**2
147
145
  self.model2 = lambda x: x**3
148
146
 
@@ -151,11 +149,11 @@ class SimpleLitAPI(ls.LitAPI):
151
149
  return request["input"]
152
150
 
153
151
  def predict(self, x):
154
- # Easily build compound systems. Run inference and return the output.
155
- squared = self.model1(x)
156
- cubed = self.model2(x)
157
- output = squared + cubed
158
- return {"output": output}
152
+ # Run the inference pipeline and return the output
153
+ a = self.model1(x)
154
+ b = self.model2(x)
155
+ c = a + b
156
+ return {"output": c}
159
157
 
160
158
  def encode_response(self, output):
161
159
  # Convert the model output to a response payload.
@@ -168,19 +166,25 @@ if __name__ == "__main__":
168
166
  server.run(port=8000)
169
167
  ```
170
168
 
171
- Now run the server via the command-line
169
+ Now run the server anywhere (local or cloud) via the command-line.
172
170
 
173
171
  ```bash
174
- python server.py
172
+ # Deploy to the cloud of your choice via Lightning AI (serverless, autoscaling, etc.)
173
+ lightning serve server.py
174
+
175
+ # Or run locally (self host anywhere)
176
+ lightning serve server.py --local
175
177
  ```
176
-
177
- ### Test the server
178
- Run the auto-generated test client:
179
- ```bash
180
- python client.py
178
+ Learn more about managed hosting on [Lightning AI](#hosting-options).
179
+
180
+ You can also run the server manually:
181
+
182
+ ```bash
183
+ python server.py
181
184
  ```
182
185
 
183
- Or use this terminal command:
186
+ ### Test the server
187
+ Simulate an http request (run this on any terminal):
184
188
  ```bash
185
189
  curl -X POST http://127.0.0.1:8000/predict -H "Content-Type: application/json" -d '{"input": 4.0}'
186
190
  ```
@@ -197,22 +201,15 @@ litgpt serve microsoft/phi-2
197
201
  - LitAPI lets you easily build complex AI systems with one or more models ([docs](https://lightning.ai/docs/litserve/api-reference/litapi)).
198
202
  - Use the setup method for one-time tasks like connecting models, DBs, and loading data ([docs](https://lightning.ai/docs/litserve/api-reference/litapi#setup)).
199
203
  - LitServer handles optimizations like batching, GPU autoscaling, streaming, etc... ([docs](https://lightning.ai/docs/litserve/api-reference/litserver)).
200
- - Self host on your own machines or use Lightning Studios for a fully managed deployment ([learn more](#hosting-options)).
204
+ - Self host on your machines or create a fully managed deployment with Lightning ([learn more](https://lightning.ai/docs/litserve/features/deploy-on-cloud)).
201
205
 
202
206
  [Learn how to make this server 200x faster](https://lightning.ai/docs/litserve/home/speed-up-serving-by-200x).
203
207
 
204
208
  &nbsp;
205
209
 
206
210
  # Featured examples
207
- Use LitServe to deploy any model or AI service: (Compound AI, Gen AI, classic ML, embeddings, LLMs, vision, audio, etc...)
208
-
209
- <div align='center'>
210
- <div width='200px'>
211
- <video src="https://github.com/user-attachments/assets/5e73549a-bc0f-47a9-9d9c-5b54389be5de" width='200px' controls></video>
212
- </div>
213
- </div>
214
-
215
- ## Examples
211
+ Here are examples of inference pipelines for common model types and use cases.
212
+
216
213
  <pre>
217
214
  <strong>Toy model:</strong> <a target="_blank" href="#define-a-server">Hello world</a>
218
215
  <strong>LLMs:</strong> <a target="_blank" href="https://lightning.ai/lightning-ai/studios/deploy-llama-3-2-vision-with-litserve">Llama 3.2</a>, <a target="_blank" href="https://lightning.ai/lightning-ai/studios/openai-fault-tolerant-proxy-server">LLM Proxy server</a>, <a target="_blank" href="https://lightning.ai/lightning-ai/studios/deploy-ai-agent-with-tool-use">Agent with tool use</a>
@@ -232,31 +229,63 @@ Use LitServe to deploy any model or AI service: (Compound AI, Gen AI, classic ML
232
229
 
233
230
  &nbsp;
234
231
 
235
- # Features
236
- State-of-the-art features:
237
232
 
238
- ✅ [(2x)+ faster than plain FastAPI](#performance)
239
- ✅ [Bring your own model](https://lightning.ai/docs/litserve/features/full-control)
240
- ✅ [Build compound systems (1+ models)](https://lightning.ai/docs/litserve/home)
241
- ✅ [GPU autoscaling](https://lightning.ai/docs/litserve/features/gpu-inference)
242
- ✅ [Batching](https://lightning.ai/docs/litserve/features/batching)
243
- ✅ [Streaming](https://lightning.ai/docs/litserve/features/streaming)
244
- ✅ [Worker autoscaling](https://lightning.ai/docs/litserve/features/autoscaling)
245
- ✅ [Self-host on your machines](https://lightning.ai/docs/litserve/features/hosting-methods#host-on-your-own)
246
- ✅ [Host fully managed on Lightning AI](https://lightning.ai/docs/litserve/features/hosting-methods#host-on-lightning-studios)
247
- ✅ [Serve all models: (LLMs, vision, etc.)](https://lightning.ai/docs/litserve/examples)
248
- ✅ [Scale to zero (serverless)](https://lightning.ai/docs/litserve/features/streaming)
249
- ✅ [Supports PyTorch, JAX, TF, etc...](https://lightning.ai/docs/litserve/features/full-control)
250
- ✅ [OpenAPI compliant](https://www.openapis.org/)
251
- ✅ [Open AI compatibility](https://lightning.ai/docs/litserve/features/open-ai-spec)
252
- ✅ [Authentication](https://lightning.ai/docs/litserve/features/authentication)
253
- ✅ [Dockerization](https://lightning.ai/docs/litserve/features/dockerization-deployment)
233
+ # Hosting options
234
+ Self host LitServe anywhere or deploy to your favorite cloud via [Lightning AI](http://lightning.ai/deploy).
254
235
 
236
+ https://github.com/user-attachments/assets/ff83dab9-0c9f-4453-8dcb-fb9526726344
255
237
 
238
+ Self-hosting is ideal for hackers, students, and DIY developers while fully managed hosting is ideal for enterprise developers needing easy autoscaling, security, release management, and 99.995% uptime and observability.
256
239
 
257
- [10+ features...](https://lightning.ai/docs/litserve/features)
240
+ *Note:* Lightning offers a generous free tier for developers.
258
241
 
259
- **Note:** We prioritize scalable, enterprise-level features over hype.
242
+ To host on [Lightning AI](https://lightning.ai/deploy), simply run the command, login and choose the cloud of your choice.
243
+ ```bash
244
+ lightning serve server.py
245
+ ```
246
+
247
+ &nbsp;
248
+
249
+ ## Features
250
+
251
+ <div align='center'>
252
+
253
+ | [Feature](https://lightning.ai/docs/litserve/features) | Self Managed | [Fully Managed on Lightning](https://lightning.ai/deploy) |
254
+ |----------------------------------------------------------------------|-----------------------------------|------------------------------------|
255
+ | Docker-first deployment | ✅ DIY | ✅ One-click deploy |
256
+ | Cost | ✅ Free (DIY) | ✅ Generous [free tier](https://lightning.ai/pricing) with pay as you go |
257
+ | Full control | ✅ | ✅ |
258
+ | Use any engine (vLLM, etc.) | ✅ | ✅ vLLM, Ollama, LitServe, etc. |
259
+ | Own VPC | ✅ (manual setup) | ✅ Connect your own VPC |
260
+ | [(2x)+ faster than plain FastAPI](#performance) | ✅ | ✅ |
261
+ | [Bring your own model](https://lightning.ai/docs/litserve/features/full-control) | ✅ | ✅ |
262
+ | [Build compound systems (1+ models)](https://lightning.ai/docs/litserve/home) | ✅ | ✅ |
263
+ | [GPU autoscaling](https://lightning.ai/docs/litserve/features/gpu-inference) | ✅ | ✅ |
264
+ | [Batching](https://lightning.ai/docs/litserve/features/batching) | ✅ | ✅ |
265
+ | [Streaming](https://lightning.ai/docs/litserve/features/streaming) | ✅ | ✅ |
266
+ | [Worker autoscaling](https://lightning.ai/docs/litserve/features/autoscaling) | ✅ | ✅ |
267
+ | [Serve all models: (LLMs, vision, etc.)](https://lightning.ai/docs/litserve/examples) | ✅ | ✅ |
268
+ | [Supports PyTorch, JAX, TF, etc...](https://lightning.ai/docs/litserve/features/full-control) | ✅ | ✅ |
269
+ | [OpenAPI compliant](https://www.openapis.org/) | ✅ | ✅ |
270
+ | [Open AI compatibility](https://lightning.ai/docs/litserve/features/open-ai-spec) | ✅ | ✅ |
271
+ | [Authentication](https://lightning.ai/docs/litserve/features/authentication) | ❌ DIY | ✅ Token, password, custom |
272
+ | GPUs | ❌ DIY | ✅ 8+ GPU types, H100s from $1.75 |
273
+ | Load balancing | ❌ | ✅ Built-in |
274
+ | Scale to zero (serverless) | ❌ | ✅ No machine runs when idle |
275
+ | Autoscale up on demand | ❌ | ✅ Auto scale up/down |
276
+ | Multi-node inference | ❌ | ✅ Distribute across nodes |
277
+ | Use AWS/GCP credits | ❌ | ✅ Use existing cloud commits |
278
+ | Versioning | ❌ | ✅ Make and roll back releases |
279
+ | Enterprise-grade uptime (99.95%) | ❌ | ✅ SLA-backed |
280
+ | SOC2 / HIPAA compliance | ❌ | ✅ Certified & secure |
281
+ | Observability | ❌ | ✅ Built-in, connect 3rd party tools|
282
+ | CI/CD ready | ❌ | ✅ Lightning SDK |
283
+ | 24/7 enterprise support | ❌ | ✅ Dedicated support |
284
+ | Cost controls & audit logs | ❌ | ✅ Budgets, breakdowns, logs |
285
+ | Debug on GPUs | ❌ | ✅ Studio integration |
286
+ | [20+ features](https://lightning.ai/docs/litserve/features) | - | - |
287
+
288
+ </div>
260
289
 
261
290
  &nbsp;
262
291
 
@@ -275,40 +304,8 @@ These results are for image and text classification ML tasks. The performance re
275
304
 
276
305
  ***💡 Note on LLM serving:*** For high-performance LLM serving (like Ollama/vLLM), integrate [vLLM with LitServe](https://lightning.ai/lightning-ai/studios/deploy-a-private-llama-3-2-rag-api), use [LitGPT](https://github.com/Lightning-AI/litgpt?tab=readme-ov-file#deploy-an-llm), or build your custom vLLM-like server with LitServe. Optimizations like kv-caching, which can be done with LitServe, are needed to maximize LLM performance.
277
306
 
278
- &nbsp;
279
-
280
- # Hosting options
281
- LitServe can be hosted independently on your own machines or fully managed via Lightning Studios.
282
-
283
- Self-hosting is ideal for hackers, students, and DIY developers, while fully managed hosting is ideal for enterprise developers needing easy autoscaling, security, release management, and 99.995% uptime and observability.
284
-
285
- &nbsp;
286
-
287
- <div align="center">
288
- <a target="_blank" href="https://lightning.ai/lightning-ai/studios/litserve-hello-world">
289
- <img src="https://pl-bolts-doc-images.s3.us-east-2.amazonaws.com/app-2/host-on-lightning.svg" alt="Host on Lightning"/>
290
- </a>
291
- </div>
292
-
293
307
  &nbsp;
294
308
 
295
- <div align='center'>
296
-
297
- | Feature | Self Managed | Fully Managed on Studios |
298
- |----------------------------------|-----------------------------------|-------------------------------------|
299
- | Deployment | ✅ Do it yourself deployment | ✅ One-button cloud deploy |
300
- | Load balancing | ❌ | ✅ |
301
- | Autoscaling | ❌ | ✅ |
302
- | Scale to zero | ❌ | ✅ |
303
- | Multi-machine inference | ❌ | ✅ |
304
- | Authentication | ❌ | ✅ |
305
- | Own VPC | ❌ | ✅ |
306
- | AWS, GCP | ❌ | ✅ |
307
- | Use your own cloud commits | ❌ | ✅ |
308
-
309
- </div>
310
-
311
- &nbsp;
312
309
 
313
310
  # Community
314
311
  LitServe is a [community project accepting contributions](https://lightning.ai/docs/litserve/community) - Let's make the world's most advanced AI inference engine.
@@ -1,32 +1,29 @@
1
1
  <div align='center'>
2
2
 
3
- # Easily serve AI models Lightning fast ⚡
3
+ # Deploy AI models and inference pipelines - ⚡ fast
4
4
 
5
5
  <img alt="Lightning" src="https://pl-bolts-doc-images.s3.us-east-2.amazonaws.com/app-2/ls_banner2.png" width="800px" style="max-width: 100%;">
6
6
 
7
- &nbsp;
8
-
9
- <strong>Lightning-fast serving engine for AI models.</strong>
10
- Easy. Flexible. Enterprise-scale.
7
+ &nbsp;
11
8
  </div>
12
9
 
13
- ----
14
-
15
- **LitServe** is an easy-to-use, flexible serving engine for AI models built on FastAPI. It augments FastAPI with features like batching, streaming, and GPU autoscaling eliminate the need to rebuild a FastAPI server per model.
10
+ **LitServe** lets you build high-performance AI inference pipelines on top of FastAPI - no boilerplate. Define one or more models, connect vector DBs, stream responses, batch requests, and autoscale on GPUs out of the box.
16
11
 
17
12
  LitServe is at least [2x faster](#performance) than plain FastAPI due to AI-specific multi-worker handling.
18
13
 
19
14
  <div align='center'>
20
15
 
21
16
  <pre>
22
- ✅ (2x)+ faster serving ✅ Easy to use ✅ LLMs, non LLMs and more
23
- ✅ Bring your own model ✅ PyTorch/JAX/TF/... ✅ Built on FastAPI
24
- ✅ GPU autoscaling ✅ Batching, Streaming ✅ Self-host or ⚡️ managed
25
- ✅ Compound AI ✅ Integrate with vLLM and more
17
+ ✅ (2x)+ faster serving ✅ Easy to use ✅ LLMs, non LLMs and more
18
+ ✅ Bring your own model ✅ PyTorch/JAX/TF/... ✅ Built on FastAPI
19
+ ✅ GPU autoscaling ✅ Batching, Streaming ✅ Self-host or ⚡️ managed
20
+ ✅ Inference pipeline ✅ Integrate with vLLM, etc ✅ Serverless  
21
+  
26
22
  </pre>
27
23
 
28
24
  <div align='center'>
29
25
 
26
+ [![PyPI Downloads](https://static.pepy.tech/badge/litserve)](https://pepy.tech/projects/litserve)
30
27
  [![Discord](https://img.shields.io/discord/1077906959069626439?label=Get%20help%20on%20Discord)](https://discord.gg/WajDThKAur)
31
28
  ![cpu-tests](https://github.com/Lightning-AI/litserve/actions/workflows/ci-testing.yml/badge.svg)
32
29
  [![codecov](https://codecov.io/gh/Lightning-AI/litserve/graph/badge.svg?token=SmzX8mnKlA)](https://codecov.io/gh/Lightning-AI/litserve)
@@ -64,16 +61,16 @@ pip install litserve
64
61
  ```
65
62
 
66
63
  ### Define a server
67
- This toy example with 2 models (AI compound system) shows LitServe's flexibility ([see real examples](#examples)):
64
+ This toy example with 2 models (inference pipeline) shows LitServe's flexibility ([see real examples](#examples)):
68
65
 
69
66
  ```python
70
67
  # server.py
71
68
  import litserve as ls
72
69
 
73
- # (STEP 1) - DEFINE THE API (compound AI system)
70
+ # (STEP 1) - DEFINE THE API ("inference" pipeline)
74
71
  class SimpleLitAPI(ls.LitAPI):
75
72
  def setup(self, device):
76
- # setup is called once at startup. Build a compound AI system (1+ models), connect DBs, load data, etc...
73
+ # setup is called once at startup. Defines elements of the pipeline: models, connect DBs, load data, etc...
77
74
  self.model1 = lambda x: x**2
78
75
  self.model2 = lambda x: x**3
79
76
 
@@ -82,11 +79,11 @@ class SimpleLitAPI(ls.LitAPI):
82
79
  return request["input"]
83
80
 
84
81
  def predict(self, x):
85
- # Easily build compound systems. Run inference and return the output.
86
- squared = self.model1(x)
87
- cubed = self.model2(x)
88
- output = squared + cubed
89
- return {"output": output}
82
+ # Run the inference pipeline and return the output
83
+ a = self.model1(x)
84
+ b = self.model2(x)
85
+ c = a + b
86
+ return {"output": c}
90
87
 
91
88
  def encode_response(self, output):
92
89
  # Convert the model output to a response payload.
@@ -99,19 +96,25 @@ if __name__ == "__main__":
99
96
  server.run(port=8000)
100
97
  ```
101
98
 
102
- Now run the server via the command-line
99
+ Now run the server anywhere (local or cloud) via the command-line.
103
100
 
104
101
  ```bash
105
- python server.py
102
+ # Deploy to the cloud of your choice via Lightning AI (serverless, autoscaling, etc.)
103
+ lightning serve server.py
104
+
105
+ # Or run locally (self host anywhere)
106
+ lightning serve server.py --local
106
107
  ```
107
-
108
- ### Test the server
109
- Run the auto-generated test client:
110
- ```bash
111
- python client.py
108
+ Learn more about managed hosting on [Lightning AI](#hosting-options).
109
+
110
+ You can also run the server manually:
111
+
112
+ ```bash
113
+ python server.py
112
114
  ```
113
115
 
114
- Or use this terminal command:
116
+ ### Test the server
117
+ Simulate an http request (run this on any terminal):
115
118
  ```bash
116
119
  curl -X POST http://127.0.0.1:8000/predict -H "Content-Type: application/json" -d '{"input": 4.0}'
117
120
  ```
@@ -128,22 +131,15 @@ litgpt serve microsoft/phi-2
128
131
  - LitAPI lets you easily build complex AI systems with one or more models ([docs](https://lightning.ai/docs/litserve/api-reference/litapi)).
129
132
  - Use the setup method for one-time tasks like connecting models, DBs, and loading data ([docs](https://lightning.ai/docs/litserve/api-reference/litapi#setup)).
130
133
  - LitServer handles optimizations like batching, GPU autoscaling, streaming, etc... ([docs](https://lightning.ai/docs/litserve/api-reference/litserver)).
131
- - Self host on your own machines or use Lightning Studios for a fully managed deployment ([learn more](#hosting-options)).
134
+ - Self host on your machines or create a fully managed deployment with Lightning ([learn more](https://lightning.ai/docs/litserve/features/deploy-on-cloud)).
132
135
 
133
136
  [Learn how to make this server 200x faster](https://lightning.ai/docs/litserve/home/speed-up-serving-by-200x).
134
137
 
135
138
  &nbsp;
136
139
 
137
140
  # Featured examples
138
- Use LitServe to deploy any model or AI service: (Compound AI, Gen AI, classic ML, embeddings, LLMs, vision, audio, etc...)
139
-
140
- <div align='center'>
141
- <div width='200px'>
142
- <video src="https://github.com/user-attachments/assets/5e73549a-bc0f-47a9-9d9c-5b54389be5de" width='200px' controls></video>
143
- </div>
144
- </div>
145
-
146
- ## Examples
141
+ Here are examples of inference pipelines for common model types and use cases.
142
+
147
143
  <pre>
148
144
  <strong>Toy model:</strong> <a target="_blank" href="#define-a-server">Hello world</a>
149
145
  <strong>LLMs:</strong> <a target="_blank" href="https://lightning.ai/lightning-ai/studios/deploy-llama-3-2-vision-with-litserve">Llama 3.2</a>, <a target="_blank" href="https://lightning.ai/lightning-ai/studios/openai-fault-tolerant-proxy-server">LLM Proxy server</a>, <a target="_blank" href="https://lightning.ai/lightning-ai/studios/deploy-ai-agent-with-tool-use">Agent with tool use</a>
@@ -163,31 +159,63 @@ Use LitServe to deploy any model or AI service: (Compound AI, Gen AI, classic ML
163
159
 
164
160
  &nbsp;
165
161
 
166
- # Features
167
- State-of-the-art features:
168
162
 
169
- ✅ [(2x)+ faster than plain FastAPI](#performance)
170
- ✅ [Bring your own model](https://lightning.ai/docs/litserve/features/full-control)
171
- ✅ [Build compound systems (1+ models)](https://lightning.ai/docs/litserve/home)
172
- ✅ [GPU autoscaling](https://lightning.ai/docs/litserve/features/gpu-inference)
173
- ✅ [Batching](https://lightning.ai/docs/litserve/features/batching)
174
- ✅ [Streaming](https://lightning.ai/docs/litserve/features/streaming)
175
- ✅ [Worker autoscaling](https://lightning.ai/docs/litserve/features/autoscaling)
176
- ✅ [Self-host on your machines](https://lightning.ai/docs/litserve/features/hosting-methods#host-on-your-own)
177
- ✅ [Host fully managed on Lightning AI](https://lightning.ai/docs/litserve/features/hosting-methods#host-on-lightning-studios)
178
- ✅ [Serve all models: (LLMs, vision, etc.)](https://lightning.ai/docs/litserve/examples)
179
- ✅ [Scale to zero (serverless)](https://lightning.ai/docs/litserve/features/streaming)
180
- ✅ [Supports PyTorch, JAX, TF, etc...](https://lightning.ai/docs/litserve/features/full-control)
181
- ✅ [OpenAPI compliant](https://www.openapis.org/)
182
- ✅ [Open AI compatibility](https://lightning.ai/docs/litserve/features/open-ai-spec)
183
- ✅ [Authentication](https://lightning.ai/docs/litserve/features/authentication)
184
- ✅ [Dockerization](https://lightning.ai/docs/litserve/features/dockerization-deployment)
163
+ # Hosting options
164
+ Self host LitServe anywhere or deploy to your favorite cloud via [Lightning AI](http://lightning.ai/deploy).
185
165
 
166
+ https://github.com/user-attachments/assets/ff83dab9-0c9f-4453-8dcb-fb9526726344
186
167
 
168
+ Self-hosting is ideal for hackers, students, and DIY developers while fully managed hosting is ideal for enterprise developers needing easy autoscaling, security, release management, and 99.995% uptime and observability.
187
169
 
188
- [10+ features...](https://lightning.ai/docs/litserve/features)
170
+ *Note:* Lightning offers a generous free tier for developers.
189
171
 
190
- **Note:** We prioritize scalable, enterprise-level features over hype.
172
+ To host on [Lightning AI](https://lightning.ai/deploy), simply run the command, login and choose the cloud of your choice.
173
+ ```bash
174
+ lightning serve server.py
175
+ ```
176
+
177
+ &nbsp;
178
+
179
+ ## Features
180
+
181
+ <div align='center'>
182
+
183
+ | [Feature](https://lightning.ai/docs/litserve/features) | Self Managed | [Fully Managed on Lightning](https://lightning.ai/deploy) |
184
+ |----------------------------------------------------------------------|-----------------------------------|------------------------------------|
185
+ | Docker-first deployment | ✅ DIY | ✅ One-click deploy |
186
+ | Cost | ✅ Free (DIY) | ✅ Generous [free tier](https://lightning.ai/pricing) with pay as you go |
187
+ | Full control | ✅ | ✅ |
188
+ | Use any engine (vLLM, etc.) | ✅ | ✅ vLLM, Ollama, LitServe, etc. |
189
+ | Own VPC | ✅ (manual setup) | ✅ Connect your own VPC |
190
+ | [(2x)+ faster than plain FastAPI](#performance) | ✅ | ✅ |
191
+ | [Bring your own model](https://lightning.ai/docs/litserve/features/full-control) | ✅ | ✅ |
192
+ | [Build compound systems (1+ models)](https://lightning.ai/docs/litserve/home) | ✅ | ✅ |
193
+ | [GPU autoscaling](https://lightning.ai/docs/litserve/features/gpu-inference) | ✅ | ✅ |
194
+ | [Batching](https://lightning.ai/docs/litserve/features/batching) | ✅ | ✅ |
195
+ | [Streaming](https://lightning.ai/docs/litserve/features/streaming) | ✅ | ✅ |
196
+ | [Worker autoscaling](https://lightning.ai/docs/litserve/features/autoscaling) | ✅ | ✅ |
197
+ | [Serve all models: (LLMs, vision, etc.)](https://lightning.ai/docs/litserve/examples) | ✅ | ✅ |
198
+ | [Supports PyTorch, JAX, TF, etc...](https://lightning.ai/docs/litserve/features/full-control) | ✅ | ✅ |
199
+ | [OpenAPI compliant](https://www.openapis.org/) | ✅ | ✅ |
200
+ | [Open AI compatibility](https://lightning.ai/docs/litserve/features/open-ai-spec) | ✅ | ✅ |
201
+ | [Authentication](https://lightning.ai/docs/litserve/features/authentication) | ❌ DIY | ✅ Token, password, custom |
202
+ | GPUs | ❌ DIY | ✅ 8+ GPU types, H100s from $1.75 |
203
+ | Load balancing | ❌ | ✅ Built-in |
204
+ | Scale to zero (serverless) | ❌ | ✅ No machine runs when idle |
205
+ | Autoscale up on demand | ❌ | ✅ Auto scale up/down |
206
+ | Multi-node inference | ❌ | ✅ Distribute across nodes |
207
+ | Use AWS/GCP credits | ❌ | ✅ Use existing cloud commits |
208
+ | Versioning | ❌ | ✅ Make and roll back releases |
209
+ | Enterprise-grade uptime (99.95%) | ❌ | ✅ SLA-backed |
210
+ | SOC2 / HIPAA compliance | ❌ | ✅ Certified & secure |
211
+ | Observability | ❌ | ✅ Built-in, connect 3rd party tools|
212
+ | CI/CD ready | ❌ | ✅ Lightning SDK |
213
+ | 24/7 enterprise support | ❌ | ✅ Dedicated support |
214
+ | Cost controls & audit logs | ❌ | ✅ Budgets, breakdowns, logs |
215
+ | Debug on GPUs | ❌ | ✅ Studio integration |
216
+ | [20+ features](https://lightning.ai/docs/litserve/features) | - | - |
217
+
218
+ </div>
191
219
 
192
220
  &nbsp;
193
221
 
@@ -206,40 +234,8 @@ These results are for image and text classification ML tasks. The performance re
206
234
 
207
235
  ***💡 Note on LLM serving:*** For high-performance LLM serving (like Ollama/vLLM), integrate [vLLM with LitServe](https://lightning.ai/lightning-ai/studios/deploy-a-private-llama-3-2-rag-api), use [LitGPT](https://github.com/Lightning-AI/litgpt?tab=readme-ov-file#deploy-an-llm), or build your custom vLLM-like server with LitServe. Optimizations like kv-caching, which can be done with LitServe, are needed to maximize LLM performance.
208
236
 
209
- &nbsp;
210
-
211
- # Hosting options
212
- LitServe can be hosted independently on your own machines or fully managed via Lightning Studios.
213
-
214
- Self-hosting is ideal for hackers, students, and DIY developers, while fully managed hosting is ideal for enterprise developers needing easy autoscaling, security, release management, and 99.995% uptime and observability.
215
-
216
237
  &nbsp;
217
238
 
218
- <div align="center">
219
- <a target="_blank" href="https://lightning.ai/lightning-ai/studios/litserve-hello-world">
220
- <img src="https://pl-bolts-doc-images.s3.us-east-2.amazonaws.com/app-2/host-on-lightning.svg" alt="Host on Lightning"/>
221
- </a>
222
- </div>
223
-
224
- &nbsp;
225
-
226
- <div align='center'>
227
-
228
- | Feature | Self Managed | Fully Managed on Studios |
229
- |----------------------------------|-----------------------------------|-------------------------------------|
230
- | Deployment | ✅ Do it yourself deployment | ✅ One-button cloud deploy |
231
- | Load balancing | ❌ | ✅ |
232
- | Autoscaling | ❌ | ✅ |
233
- | Scale to zero | ❌ | ✅ |
234
- | Multi-machine inference | ❌ | ✅ |
235
- | Authentication | ❌ | ✅ |
236
- | Own VPC | ❌ | ✅ |
237
- | AWS, GCP | ❌ | ✅ |
238
- | Use your own cloud commits | ❌ | ✅ |
239
-
240
- </div>
241
-
242
- &nbsp;
243
239
 
244
240
  # Community
245
241
  LitServe is a [community project accepting contributions](https://lightning.ai/docs/litserve/community) - Let's make the world's most advanced AI inference engine.
@@ -111,8 +111,6 @@ setup(
111
111
  "Programming Language :: Python :: 3.11",
112
112
  ],
113
113
  entry_points={
114
- "console_scripts": [
115
- "litserve=litserve.__main__:main",
116
- ],
114
+ "console_scripts": ["litserve=litserve.__main__:main", "lightning=litserve.cli:main"],
117
115
  },
118
116
  )
@@ -11,7 +11,7 @@
11
11
  # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
12
  # See the License for the specific language governing permissions and
13
13
  # limitations under the License.
14
- __version__ = "0.2.7.dev0"
14
+ __version__ = "0.2.8"
15
15
  __author__ = "Lightning-AI et al."
16
16
  __author_email__ = "community@lightning.ai"
17
17
  __license__ = "Apache-2.0"
@@ -15,7 +15,7 @@ import json
15
15
  import warnings
16
16
  from abc import ABC, abstractmethod
17
17
  from queue import Queue
18
- from typing import Optional
18
+ from typing import Callable, Optional
19
19
 
20
20
  from pydantic import BaseModel
21
21
 
@@ -24,8 +24,8 @@ from litserve.specs.base import LitSpec
24
24
 
25
25
  class LitAPI(ABC):
26
26
  _stream: bool = False
27
- _default_unbatch: callable = None
28
- _spec: LitSpec = None
27
+ _default_unbatch: Optional[Callable] = None
28
+ _spec: Optional[LitSpec] = None
29
29
  _device: Optional[str] = None
30
30
  _logger_queue: Optional[Queue] = None
31
31
  request_timeout: Optional[float] = None
@@ -76,6 +76,11 @@ class LitAPI(ABC):
76
76
 
77
77
  def unbatch(self, output):
78
78
  """Convert a batched output to a list of outputs."""
79
+ if self._default_unbatch is None:
80
+ raise ValueError(
81
+ "Default implementation for `LitAPI.unbatch` method was not found. "
82
+ "Please implement the `LitAPI.unbatch` method."
83
+ )
79
84
  return self._default_unbatch(output)
80
85
 
81
86
  def encode_response(self, output, **kwargs):
@@ -0,0 +1,17 @@
1
+ import importlib.util
2
+ import subprocess
3
+ import sys
4
+
5
+
6
+ def _ensure_lightning_installed():
7
+ if not importlib.util.find_spec("lightning_sdk"):
8
+ print("Lightning CLI not found. Installing...")
9
+ subprocess.check_call([sys.executable, "-m", "pip", "install", "-U", "lightning-sdk"])
10
+
11
+
12
+ def main():
13
+ _ensure_lightning_installed()
14
+
15
+ # Forward CLI arguments to the real lightning command
16
+ cli_args = sys.argv[1:]
17
+ subprocess.run(["lightning"] + cli_args)