lmcache-cli 0.4.5.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (399) hide show
  1. lmcache/__init__.py +84 -0
  2. lmcache/_version.py +24 -0
  3. lmcache/cli/__init__.py +1 -0
  4. lmcache/cli/commands/__init__.py +34 -0
  5. lmcache/cli/commands/base.py +157 -0
  6. lmcache/cli/commands/bench/__init__.py +557 -0
  7. lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
  8. lmcache/cli/commands/bench/engine_bench/config.py +245 -0
  9. lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
  10. lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
  11. lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
  12. lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
  13. lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
  14. lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
  15. lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
  16. lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
  17. lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
  18. lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
  19. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
  20. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
  21. lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
  22. lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
  23. lmcache/cli/commands/describe.py +310 -0
  24. lmcache/cli/commands/kvcache.py +133 -0
  25. lmcache/cli/commands/mock.py +75 -0
  26. lmcache/cli/commands/ping.py +113 -0
  27. lmcache/cli/commands/query/__init__.py +155 -0
  28. lmcache/cli/commands/query/prompt.py +134 -0
  29. lmcache/cli/commands/query/request.py +357 -0
  30. lmcache/cli/commands/server.py +99 -0
  31. lmcache/cli/commands/tool/__init__.py +63 -0
  32. lmcache/cli/commands/tool/cache_simulator.py +113 -0
  33. lmcache/cli/commands/trace/__init__.py +505 -0
  34. lmcache/cli/commands/trace/dispatch.py +249 -0
  35. lmcache/cli/commands/trace/driver.py +372 -0
  36. lmcache/cli/commands/trace/stats.py +289 -0
  37. lmcache/cli/documents/lmcache.txt +11 -0
  38. lmcache/cli/main.py +42 -0
  39. lmcache/cli/metrics/__init__.py +29 -0
  40. lmcache/cli/metrics/formatter.py +171 -0
  41. lmcache/cli/metrics/handler.py +94 -0
  42. lmcache/cli/metrics/metrics.py +161 -0
  43. lmcache/cli/metrics/section.py +77 -0
  44. lmcache/connections.py +173 -0
  45. lmcache/integration/__init__.py +2 -0
  46. lmcache/integration/base_service_factory.py +165 -0
  47. lmcache/integration/request_telemetry/__init__.py +1 -0
  48. lmcache/integration/request_telemetry/base.py +51 -0
  49. lmcache/integration/request_telemetry/factory.py +113 -0
  50. lmcache/integration/request_telemetry/fastapi.py +109 -0
  51. lmcache/integration/request_telemetry/noop.py +35 -0
  52. lmcache/integration/sglang/__init__.py +2 -0
  53. lmcache/integration/sglang/sglang_adapter.py +326 -0
  54. lmcache/integration/sglang/utils.py +39 -0
  55. lmcache/integration/vllm/__init__.py +1 -0
  56. lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
  57. lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
  58. lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
  59. lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
  60. lmcache/integration/vllm/utils.py +433 -0
  61. lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
  62. lmcache/integration/vllm/vllm_service_factory.py +339 -0
  63. lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
  64. lmcache/logging.py +107 -0
  65. lmcache/native_storage_ops.pyi +230 -0
  66. lmcache/non_cuda_equivalents.py +1424 -0
  67. lmcache/observability.py +1958 -0
  68. lmcache/storage_backend/serde/__init__.py +1 -0
  69. lmcache/storage_backend/serde/cachegen_basics.py +210 -0
  70. lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
  71. lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
  72. lmcache/storage_backend/serde/serde.py +75 -0
  73. lmcache/tools/__init__.py +1 -0
  74. lmcache/tools/cache_simulator/README.md +392 -0
  75. lmcache/tools/cache_simulator/__init__.py +1 -0
  76. lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
  77. lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
  78. lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
  79. lmcache/tools/cache_simulator/lru_cache.py +124 -0
  80. lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
  81. lmcache/tools/cache_simulator/simulator.py +795 -0
  82. lmcache/tools/controller_benchmark/README.md +161 -0
  83. lmcache/tools/controller_benchmark/__init__.py +1 -0
  84. lmcache/tools/controller_benchmark/__main__.py +331 -0
  85. lmcache/tools/controller_benchmark/benchmark.py +660 -0
  86. lmcache/tools/controller_benchmark/config.py +44 -0
  87. lmcache/tools/controller_benchmark/constants.py +10 -0
  88. lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
  89. lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
  90. lmcache/tools/controller_benchmark/handlers/base.py +47 -0
  91. lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
  92. lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
  93. lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
  94. lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
  95. lmcache/tools/controller_benchmark/handlers/register.py +56 -0
  96. lmcache/tools/mp_status_viewer/__init__.py +1 -0
  97. lmcache/tools/mp_status_viewer/__main__.py +95 -0
  98. lmcache/usage_context.py +417 -0
  99. lmcache/utils.py +665 -0
  100. lmcache/v1/__init__.py +2 -0
  101. lmcache/v1/api_server/__init__.py +2 -0
  102. lmcache/v1/api_server/__main__.py +537 -0
  103. lmcache/v1/basic_check.py +112 -0
  104. lmcache/v1/cache_controller/__init__.py +9 -0
  105. lmcache/v1/cache_controller/commands/__init__.py +15 -0
  106. lmcache/v1/cache_controller/commands/base.py +35 -0
  107. lmcache/v1/cache_controller/commands/full_sync.py +49 -0
  108. lmcache/v1/cache_controller/config.py +176 -0
  109. lmcache/v1/cache_controller/controller_manager.py +535 -0
  110. lmcache/v1/cache_controller/controllers/__init__.py +11 -0
  111. lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
  112. lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
  113. lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
  114. lmcache/v1/cache_controller/executor.py +463 -0
  115. lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
  116. lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
  117. lmcache/v1/cache_controller/frontend/static/index.html +234 -0
  118. lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
  119. lmcache/v1/cache_controller/full_sync_sender.py +475 -0
  120. lmcache/v1/cache_controller/locks.py +149 -0
  121. lmcache/v1/cache_controller/message.py +828 -0
  122. lmcache/v1/cache_controller/observability.py +208 -0
  123. lmcache/v1/cache_controller/utils.py +679 -0
  124. lmcache/v1/cache_controller/worker.py +665 -0
  125. lmcache/v1/cache_engine.py +2058 -0
  126. lmcache/v1/cache_interface.py +19 -0
  127. lmcache/v1/check/__init__.py +74 -0
  128. lmcache/v1/check/check_mode_gen.py +86 -0
  129. lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
  130. lmcache/v1/check/check_mode_test_remote.py +155 -0
  131. lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
  132. lmcache/v1/check/utils.py +571 -0
  133. lmcache/v1/compute/__init__.py +2 -0
  134. lmcache/v1/compute/attention/__init__.py +0 -0
  135. lmcache/v1/compute/attention/abstract.py +39 -0
  136. lmcache/v1/compute/attention/flash_attn.py +129 -0
  137. lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
  138. lmcache/v1/compute/attention/metadata.py +85 -0
  139. lmcache/v1/compute/attention/utils.py +14 -0
  140. lmcache/v1/compute/blend/__init__.py +7 -0
  141. lmcache/v1/compute/blend/blender.py +168 -0
  142. lmcache/v1/compute/blend/metadata.py +34 -0
  143. lmcache/v1/compute/blend/utils.py +63 -0
  144. lmcache/v1/compute/models/__init__.py +0 -0
  145. lmcache/v1/compute/models/base.py +141 -0
  146. lmcache/v1/compute/models/llama.py +9 -0
  147. lmcache/v1/compute/models/qwen3.py +24 -0
  148. lmcache/v1/compute/models/utils.py +68 -0
  149. lmcache/v1/compute/positional_encoding.py +199 -0
  150. lmcache/v1/config.py +848 -0
  151. lmcache/v1/config_base.py +848 -0
  152. lmcache/v1/distributed/api.py +248 -0
  153. lmcache/v1/distributed/config.py +321 -0
  154. lmcache/v1/distributed/error.py +64 -0
  155. lmcache/v1/distributed/eviction.py +192 -0
  156. lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
  157. lmcache/v1/distributed/eviction_policy/factory.py +27 -0
  158. lmcache/v1/distributed/eviction_policy/lru.py +244 -0
  159. lmcache/v1/distributed/eviction_policy/noop.py +50 -0
  160. lmcache/v1/distributed/internal_api.py +170 -0
  161. lmcache/v1/distributed/l1_manager.py +835 -0
  162. lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
  163. lmcache/v1/distributed/l2_adapters/base.py +360 -0
  164. lmcache/v1/distributed/l2_adapters/config.py +385 -0
  165. lmcache/v1/distributed/l2_adapters/factory.py +205 -0
  166. lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
  167. lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
  168. lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
  169. lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
  170. lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
  171. lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
  172. lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
  173. lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
  174. lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
  175. lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
  176. lmcache/v1/distributed/memory_manager.py +179 -0
  177. lmcache/v1/distributed/storage_controller.py +39 -0
  178. lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
  179. lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
  180. lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
  181. lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
  182. lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
  183. lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
  184. lmcache/v1/distributed/storage_manager.py +532 -0
  185. lmcache/v1/event_manager.py +145 -0
  186. lmcache/v1/exceptions/__init__.py +16 -0
  187. lmcache/v1/gpu_connector/__init__.py +126 -0
  188. lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
  189. lmcache/v1/gpu_connector/gpu_ops.py +85 -0
  190. lmcache/v1/gpu_connector/hpu_connector.py +326 -0
  191. lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
  192. lmcache/v1/gpu_connector/utils.py +890 -0
  193. lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
  194. lmcache/v1/health_monitor/__init__.py +1 -0
  195. lmcache/v1/health_monitor/base.py +587 -0
  196. lmcache/v1/health_monitor/checks/__init__.py +1 -0
  197. lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
  198. lmcache/v1/health_monitor/constants.py +36 -0
  199. lmcache/v1/internal_api_server/__init__.py +0 -0
  200. lmcache/v1/internal_api_server/api_registry.py +59 -0
  201. lmcache/v1/internal_api_server/api_server.py +120 -0
  202. lmcache/v1/internal_api_server/common/__init__.py +1 -0
  203. lmcache/v1/internal_api_server/common/env_api.py +22 -0
  204. lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
  205. lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
  206. lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
  207. lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
  208. lmcache/v1/internal_api_server/common/thread_api.py +63 -0
  209. lmcache/v1/internal_api_server/controller/__init__.py +1 -0
  210. lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
  211. lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
  212. lmcache/v1/internal_api_server/utils.py +43 -0
  213. lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
  214. lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
  215. lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
  216. lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
  217. lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
  218. lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
  219. lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
  220. lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
  221. lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
  222. lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
  223. lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
  224. lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
  225. lmcache/v1/kv_layer_groups.py +267 -0
  226. lmcache/v1/lazy_memory_allocator.py +284 -0
  227. lmcache/v1/lookup_client/__init__.py +25 -0
  228. lmcache/v1/lookup_client/abstract_client.py +77 -0
  229. lmcache/v1/lookup_client/async_lookup_message.py +50 -0
  230. lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
  231. lmcache/v1/lookup_client/factory.py +251 -0
  232. lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
  233. lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
  234. lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
  235. lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
  236. lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
  237. lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
  238. lmcache/v1/lookup_client/record_strategies/base.py +327 -0
  239. lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
  240. lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
  241. lmcache/v1/manager.py +539 -0
  242. lmcache/v1/memory_management.py +2619 -0
  243. lmcache/v1/metadata.py +114 -0
  244. lmcache/v1/mp_observability/AGENTS.override.md +21 -0
  245. lmcache/v1/mp_observability/README.md +204 -0
  246. lmcache/v1/mp_observability/config.py +340 -0
  247. lmcache/v1/mp_observability/event.py +100 -0
  248. lmcache/v1/mp_observability/event_bus.py +313 -0
  249. lmcache/v1/mp_observability/otel_init.py +145 -0
  250. lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
  251. lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
  252. lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
  253. lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
  254. lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
  255. lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
  256. lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
  257. lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
  258. lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
  259. lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
  260. lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
  261. lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
  262. lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
  263. lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
  264. lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
  265. lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
  266. lmcache/v1/mp_observability/trace/__init__.py +50 -0
  267. lmcache/v1/mp_observability/trace/codecs.py +255 -0
  268. lmcache/v1/mp_observability/trace/decorator.py +147 -0
  269. lmcache/v1/mp_observability/trace/format.py +132 -0
  270. lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
  271. lmcache/v1/mp_observability/trace/reader.py +167 -0
  272. lmcache/v1/mp_observability/trace/recorder.py +300 -0
  273. lmcache/v1/multiprocess/__init__.py +0 -0
  274. lmcache/v1/multiprocess/affinity_pool.py +102 -0
  275. lmcache/v1/multiprocess/blend_server_v2.py +891 -0
  276. lmcache/v1/multiprocess/config.py +253 -0
  277. lmcache/v1/multiprocess/custom_types.py +281 -0
  278. lmcache/v1/multiprocess/futures.py +194 -0
  279. lmcache/v1/multiprocess/gpu_context.py +511 -0
  280. lmcache/v1/multiprocess/http_server.py +235 -0
  281. lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
  282. lmcache/v1/multiprocess/mq.py +732 -0
  283. lmcache/v1/multiprocess/protocol.py +86 -0
  284. lmcache/v1/multiprocess/protocols/README.md +213 -0
  285. lmcache/v1/multiprocess/protocols/__init__.py +127 -0
  286. lmcache/v1/multiprocess/protocols/base.py +89 -0
  287. lmcache/v1/multiprocess/protocols/blend.py +109 -0
  288. lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
  289. lmcache/v1/multiprocess/protocols/controller.py +53 -0
  290. lmcache/v1/multiprocess/protocols/debug.py +34 -0
  291. lmcache/v1/multiprocess/protocols/engine.py +146 -0
  292. lmcache/v1/multiprocess/protocols/observability.py +39 -0
  293. lmcache/v1/multiprocess/server.py +1134 -0
  294. lmcache/v1/multiprocess/session.py +190 -0
  295. lmcache/v1/multiprocess/token_hasher.py +441 -0
  296. lmcache/v1/offload_server/__init__.py +17 -0
  297. lmcache/v1/offload_server/abstract_server.py +37 -0
  298. lmcache/v1/offload_server/message.py +30 -0
  299. lmcache/v1/offload_server/zmq_server.py +122 -0
  300. lmcache/v1/periodic_thread.py +579 -0
  301. lmcache/v1/pin_monitor.py +246 -0
  302. lmcache/v1/plugin/__init__.py +0 -0
  303. lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
  304. lmcache/v1/protocol.py +317 -0
  305. lmcache/v1/rpc/__init__.py +17 -0
  306. lmcache/v1/rpc/transport.py +105 -0
  307. lmcache/v1/rpc/zmq_transport.py +213 -0
  308. lmcache/v1/rpc_utils.py +165 -0
  309. lmcache/v1/server/__init__.py +2 -0
  310. lmcache/v1/server/__main__.py +170 -0
  311. lmcache/v1/server/storage_backend/__init__.py +21 -0
  312. lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
  313. lmcache/v1/server/storage_backend/local_backend.py +75 -0
  314. lmcache/v1/server/utils.py +21 -0
  315. lmcache/v1/standalone/__init__.py +1 -0
  316. lmcache/v1/standalone/__main__.py +583 -0
  317. lmcache/v1/standalone/manager.py +80 -0
  318. lmcache/v1/standalone/standalone_service_factory.py +86 -0
  319. lmcache/v1/storage_backend/__init__.py +313 -0
  320. lmcache/v1/storage_backend/abstract_backend.py +445 -0
  321. lmcache/v1/storage_backend/audit_backend.py +233 -0
  322. lmcache/v1/storage_backend/batched_message_sender.py +222 -0
  323. lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
  324. lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
  325. lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
  326. lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
  327. lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
  328. lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
  329. lmcache/v1/storage_backend/connector/__init__.py +443 -0
  330. lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
  331. lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
  332. lmcache/v1/storage_backend/connector/base_connector.py +379 -0
  333. lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
  334. lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
  335. lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
  336. lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
  337. lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
  338. lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
  339. lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
  340. lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
  341. lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
  342. lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
  343. lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
  344. lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
  345. lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
  346. lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
  347. lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
  348. lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
  349. lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
  350. lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
  351. lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
  352. lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
  353. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
  354. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
  355. lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
  356. lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
  357. lmcache/v1/storage_backend/gds_backend.py +1199 -0
  358. lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
  359. lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
  360. lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
  361. lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
  362. lmcache/v1/storage_backend/local_disk_backend.py +656 -0
  363. lmcache/v1/storage_backend/maru_backend.py +734 -0
  364. lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
  365. lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
  366. lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
  367. lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
  368. lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
  369. lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
  370. lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
  371. lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
  372. lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
  373. lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
  374. lmcache/v1/storage_backend/p2p_backend.py +788 -0
  375. lmcache/v1/storage_backend/path_sharder.py +117 -0
  376. lmcache/v1/storage_backend/pd_backend.py +646 -0
  377. lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
  378. lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
  379. lmcache/v1/storage_backend/remote_backend.py +624 -0
  380. lmcache/v1/storage_backend/resp_client.py +227 -0
  381. lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
  382. lmcache/v1/storage_backend/storage_manager.py +1352 -0
  383. lmcache/v1/system_detection.py +110 -0
  384. lmcache/v1/token_database.py +551 -0
  385. lmcache/v1/transfer_channel/__init__.py +83 -0
  386. lmcache/v1/transfer_channel/abstract.py +285 -0
  387. lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
  388. lmcache/v1/transfer_channel/nixl_channel.py +639 -0
  389. lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
  390. lmcache/v1/transfer_channel/transfer_utils.py +63 -0
  391. lmcache/v1/utils/__init__.py +1 -0
  392. lmcache/v1/utils/bloom_filter.py +109 -0
  393. lmcache/v1/utils/cache_utils.py +125 -0
  394. lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
  395. lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
  396. lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
  397. lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
  398. lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
  399. lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,112 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Standard
3
+ import dataclasses
4
+
5
+ # Third Party
6
+ import torch
7
+
8
+ # First Party
9
+ from lmcache.integration.vllm.utils import (
10
+ apply_mm_hashes_to_token_ids,
11
+ hex_hash_to_int16,
12
+ )
13
+
14
+
15
+ @dataclasses.dataclass(frozen=True)
16
+ class DummyPlaceholderRange:
17
+ offset: int
18
+ length: int
19
+
20
+
21
+ def test_hex_hash_to_int16_accepts_hex_and_non_hex() -> None:
22
+ # Hex behavior preserved (with and without 0x prefix).
23
+ assert hex_hash_to_int16("0000") == 0
24
+ assert hex_hash_to_int16("ffff") == 0xFFFF
25
+ assert hex_hash_to_int16("0xFFFF") == 0xFFFF
26
+ assert hex_hash_to_int16("0x0001") == 1
27
+
28
+ # Non-hex identifiers must not raise and must be deterministic.
29
+ s = "chatcmpl-a2a48871c4aad192-image-0"
30
+ v1 = hex_hash_to_int16(s)
31
+ v2 = hex_hash_to_int16(s)
32
+ assert isinstance(v1, int)
33
+ assert 0 <= v1 <= 0xFFFF
34
+ assert v1 == v2
35
+
36
+
37
+ def test_hex_hash_to_int16_hex_variants_whitespace_and_truncation() -> None:
38
+ # Whitespace should be ignored and case should not matter.
39
+ assert hex_hash_to_int16(" FfFf ") == 0xFFFF
40
+ assert hex_hash_to_int16("\n0x00aB\t") == 0x00AB
41
+
42
+ # Long hex should be truncated to 16 bits via masking.
43
+ assert hex_hash_to_int16("123456") == 0x3456
44
+ assert hex_hash_to_int16("0x123456") == 0x3456
45
+
46
+
47
+ def test_hex_hash_to_int16_empty_and_invalid_hex_are_safe_and_deterministic() -> None:
48
+ # Empty (or effectively empty) values should not raise.
49
+ for s in ("", " ", "0x"):
50
+ v1 = hex_hash_to_int16(s)
51
+ v2 = hex_hash_to_int16(s)
52
+ assert isinstance(v1, int)
53
+ assert 0 <= v1 <= 0xFFFF
54
+ assert v1 == v2
55
+
56
+ # Invalid "hex-looking" strings must fall back to hashing.
57
+ for s in ("0xGG", "deadbeeg", "0x12xz"):
58
+ v1 = hex_hash_to_int16(s)
59
+ v2 = hex_hash_to_int16(s)
60
+ assert isinstance(v1, int)
61
+ assert 0 <= v1 <= 0xFFFF
62
+ assert v1 == v2
63
+
64
+
65
+ def test_hex_hash_to_int16_non_string_inputs_are_safe() -> None:
66
+ # Be defensive: callers may pass None or other non-string types.
67
+ for val in (None, 0, 12345, 3.14, b"deadbeef"):
68
+ v1 = hex_hash_to_int16(val) # type: ignore[arg-type]
69
+ v2 = hex_hash_to_int16(val) # type: ignore[arg-type]
70
+ assert isinstance(v1, int)
71
+ assert 0 <= v1 <= 0xFFFF
72
+ assert v1 == v2
73
+
74
+
75
+ def test_apply_mm_hashes_to_token_ids_handles_non_hex_mm_hash() -> None:
76
+ token_ids = torch.arange(0, 10, dtype=torch.long)
77
+ mm_hashes = ["chatcmpl-a2a48871c4aad192-image-0"]
78
+ mm_positions = [DummyPlaceholderRange(offset=2, length=4)]
79
+
80
+ out = apply_mm_hashes_to_token_ids(token_ids.clone(), mm_hashes, mm_positions)
81
+ expected_val = hex_hash_to_int16(mm_hashes[0])
82
+ assert out[2:6].tolist() == [expected_val] * 4
83
+
84
+
85
+ def test_apply_mm_hashes_to_token_ids_out_of_bounds_is_safe() -> None:
86
+ token_ids = torch.zeros(5, dtype=torch.long)
87
+ mm_hashes = ["deadbeef"]
88
+ mm_positions = [DummyPlaceholderRange(offset=999, length=10)]
89
+
90
+ out = apply_mm_hashes_to_token_ids(token_ids.clone(), mm_hashes, mm_positions)
91
+ assert out.tolist() == token_ids.tolist()
92
+
93
+
94
+ def test_apply_mm_hashes_to_token_ids_multiple_placeholders_and_length_mismatch() -> (
95
+ None
96
+ ):
97
+ token_ids = torch.zeros(12, dtype=torch.long)
98
+ mm_hashes = ["deadbeef", "chatcmpl-a2a48871c4aad192-image-0", "EXTRA_HASH_IGNORED"]
99
+ mm_positions = [
100
+ DummyPlaceholderRange(offset=0, length=3),
101
+ DummyPlaceholderRange(offset=5, length=4),
102
+ ]
103
+
104
+ out = apply_mm_hashes_to_token_ids(token_ids.clone(), mm_hashes, mm_positions)
105
+ v0 = hex_hash_to_int16(mm_hashes[0])
106
+ v1 = hex_hash_to_int16(mm_hashes[1])
107
+
108
+ assert out[0:3].tolist() == [v0] * 3
109
+ assert out[5:9].tolist() == [v1] * 4
110
+ # Other regions remain unchanged.
111
+ assert out[3:5].tolist() == [0, 0]
112
+ assert out[9:12].tolist() == [0, 0, 0]
@@ -0,0 +1,433 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Standard
3
+ from typing import TYPE_CHECKING, Literal, Optional, Tuple
4
+ import hashlib
5
+ import os
6
+ import string
7
+ import threading
8
+
9
+ if TYPE_CHECKING:
10
+ from vllm.config import ModelConfig, VllmConfig
11
+ from vllm.multimodal.inputs import PlaceholderRange
12
+ from vllm.v1.request import Request
13
+
14
+ # Third Party
15
+ import torch
16
+
17
+ # First Party
18
+ from lmcache.logging import init_logger
19
+ from lmcache.v1.config import LMCacheEngineConfig
20
+ from lmcache.v1.config_base import apply_remote_configs, fetch_remote_config
21
+
22
+ if TYPE_CHECKING:
23
+ # First Party
24
+ from lmcache.v1.gpu_connector.utils import LayoutHints
25
+
26
+ logger = init_logger(__name__)
27
+ ENGINE_NAME = "vllm-instance"
28
+
29
+ # Thread-safe singleton storage
30
+ _config_instance: Optional[LMCacheEngineConfig] = None
31
+ _config_lock = threading.Lock()
32
+
33
+
34
+ def is_false(value: str) -> bool:
35
+ """Check if the given string value is equivalent to 'false'."""
36
+ return value.lower() in ("false", "0", "no", "n", "off")
37
+
38
+
39
+ def vllm_layout_hints() -> "LayoutHints":
40
+ """Build layout_hints dict by querying vLLM at runtime."""
41
+ hints: dict[str, str] = {}
42
+ kv_layout = try_get_vllm_kv_cache_layout()
43
+ if kv_layout is not None:
44
+ hints["kv_layout"] = kv_layout
45
+ return hints # type: ignore[return-value]
46
+
47
+
48
+ def try_get_vllm_kv_cache_layout() -> Literal["NHD", "HND"] | None:
49
+ """Try to query the KV cache layout from vLLM at runtime.
50
+
51
+ Returns ``"NHD"`` or ``"HND"`` if vLLM is available and the layout
52
+ has been configured, otherwise ``None``.
53
+
54
+ Please only call this where vllm is available (i.e. not in the MP server)
55
+ We will print an error if we try to get vllm kv layout where vllm
56
+ is not available.
57
+ """
58
+
59
+ # Third Party
60
+ try:
61
+ # Third Party
62
+ from vllm.v1.attention.backends.utils import ( # type: ignore[import-untyped]
63
+ get_kv_cache_layout,
64
+ )
65
+
66
+ return get_kv_cache_layout()
67
+ except Exception:
68
+ logger.error(
69
+ "vLLM is not available but tried to query kv cache "
70
+ "layout information, cannot get KV cache layout"
71
+ )
72
+ return None
73
+
74
+
75
+ def lmcache_get_or_create_config() -> LMCacheEngineConfig:
76
+ """Get the LMCache configuration from the environment variable
77
+ `LMCACHE_CONFIG_FILE`. If the environment variable is not set, this
78
+ function will return the default configuration.
79
+
80
+ This function is thread-safe and implements singleton pattern,
81
+ ensuring the configuration is loaded only once.
82
+
83
+ After loading the configuration, if 'remote_config_url' is configured,
84
+ this function will attempt to fetch additional configuration from the
85
+ remote config service. The current config and LMCACHE environment
86
+ variables will be sent to the service, along with 'appid' if set.
87
+ """
88
+ global _config_instance
89
+
90
+ # Double-checked locking for thread-safe singleton
91
+ if _config_instance is None:
92
+ with _config_lock:
93
+ if _config_instance is None: # Check again within lock
94
+ if "LMCACHE_CONFIG_FILE" not in os.environ:
95
+ logger.warning(
96
+ "No LMCache configuration file is set. Trying to read"
97
+ " configurations from the environment variables."
98
+ )
99
+ logger.warning(
100
+ "You can set the configuration file through "
101
+ "the environment variable: LMCACHE_CONFIG_FILE"
102
+ )
103
+ _config_instance = LMCacheEngineConfig.from_env()
104
+ else:
105
+ config_file = os.environ["LMCACHE_CONFIG_FILE"]
106
+ logger.info(f"Loading LMCache config file {config_file}")
107
+ _config_instance = LMCacheEngineConfig.from_file(config_file)
108
+ # Update config from environment variables
109
+ _config_instance.update_config_from_env()
110
+
111
+ # Fetch and apply remote configuration if configured
112
+ remote_config_url = _config_instance.remote_config_url
113
+ if remote_config_url:
114
+ logger.info(
115
+ "Fetching remote configuration from %s", remote_config_url
116
+ )
117
+ app_id = _config_instance.app_id
118
+ remote_response = fetch_remote_config(
119
+ remote_config_url, app_id, _config_instance
120
+ )
121
+ if remote_response:
122
+ _config_instance = apply_remote_configs(
123
+ _config_instance, remote_response
124
+ )
125
+ else:
126
+ logger.warning(
127
+ "Failed to fetch remote configuration from %s. "
128
+ "Using local configuration only.",
129
+ remote_config_url,
130
+ )
131
+ return _config_instance
132
+
133
+
134
+ def hex_hash_to_int16(s: str) -> int:
135
+ """
136
+ Convert a hash identifier into a 16-bit integer.
137
+
138
+ Historically, LMCache expected multimodal identifiers to be hex strings.
139
+ In practice (e.g., OpenAI-style multimodal requests), identifiers may be
140
+ arbitrary strings like `chatcmpl-...-image-0`. This function therefore:
141
+ - Parses hex strings (optionally prefixed with `0x`) as before, or
142
+ - Falls back to a stable string hash (SHA-256) when the input is not hex.
143
+ """
144
+ # Be defensive: vLLM may pass non-string identifiers.
145
+ s = "" if s is None else str(s)
146
+ s_stripped = s.strip()
147
+
148
+ # Fast-path: pure hex (optionally 0x-prefixed).
149
+ hex_part = s_stripped[2:] if s_stripped.lower().startswith("0x") else s_stripped
150
+ if hex_part and all(c in string.hexdigits for c in hex_part):
151
+ try:
152
+ return int(hex_part, 16) & 0xFFFF
153
+ except ValueError:
154
+ # Extremely unlikely (e.g., oversized/odd formatting); fall back to hashing.
155
+ pass
156
+
157
+ # Fallback: stable 16-bit value derived from the full identifier string.
158
+ digest = hashlib.sha256(s_stripped.encode("utf-8")).digest()
159
+ return int.from_bytes(digest[:2], byteorder="big", signed=False)
160
+
161
+
162
+ def apply_mm_hashes_to_token_ids(
163
+ token_ids: torch.Tensor,
164
+ mm_hashes: list[str],
165
+ mm_positions: list["PlaceholderRange"],
166
+ ) -> torch.Tensor:
167
+ """
168
+ Overwrite token_ids in-place for multimodal placeholders using
169
+ efficient slice assignments.
170
+ """
171
+ n = token_ids.size(0)
172
+ for hash_str, placeholder in zip(mm_hashes, mm_positions, strict=False):
173
+ start, length = placeholder.offset, placeholder.length
174
+ if start >= n:
175
+ continue
176
+ end = min(start + length, n)
177
+ token_ids[start:end] = hex_hash_to_int16(hash_str)
178
+ return token_ids
179
+
180
+
181
+ def mla_enabled(model_config: "ModelConfig") -> bool:
182
+ return (
183
+ hasattr(model_config, "use_mla")
184
+ and isinstance(model_config.use_mla, bool)
185
+ and model_config.use_mla
186
+ )
187
+
188
+
189
+ def create_lmcache_metadata(
190
+ vllm_config=None,
191
+ model_config=None,
192
+ parallel_config=None,
193
+ cache_config=None,
194
+ role=None,
195
+ ):
196
+ """
197
+ Create LMCacheMetadata from vLLM configuration.
198
+
199
+ This function extracts common metadata creation logic that was duplicated
200
+ across multiple files.
201
+
202
+ Args:
203
+ vllm_config: vLLM configuration object containing model, parallel, and
204
+ cache configs (alternative to individual config parameters)
205
+ model_config: Model configuration (alternative to vllm_config)
206
+ parallel_config: Parallel configuration (alternative to vllm_config)
207
+ cache_config: Cache configuration (alternative to vllm_config)
208
+
209
+ Returns:
210
+ tuple: (LMCacheMetadata, LMCacheEngineConfig)
211
+ """
212
+ # Third Party
213
+ # Try to import from old location before merged https://github.com/vllm-project/vllm/pull/26908
214
+ try:
215
+ # Third Party
216
+ from vllm.utils.torch_utils import get_kv_cache_torch_dtype
217
+ except ImportError:
218
+ # Third Party
219
+ from vllm.utils import get_kv_cache_torch_dtype
220
+ # First Party
221
+ from lmcache.v1.metadata import LMCacheMetadata
222
+
223
+ config = lmcache_get_or_create_config()
224
+ # Support both vllm_config object and individual config parameters
225
+ if vllm_config is not None:
226
+ model_cfg = vllm_config.model_config
227
+ parallel_cfg = vllm_config.parallel_config
228
+ cache_cfg = vllm_config.cache_config
229
+ else:
230
+ model_cfg = model_config
231
+ parallel_cfg = parallel_config
232
+ cache_cfg = cache_config
233
+
234
+ # Get KV cache dtype
235
+ kv_dtype = get_kv_cache_torch_dtype(cache_cfg.cache_dtype, model_cfg.dtype)
236
+
237
+ # Check if MLA is enabled
238
+ use_mla = mla_enabled(model_cfg)
239
+
240
+ # Construct KV shape (for memory pool)
241
+ num_layer = model_cfg.get_num_layers(parallel_cfg)
242
+ chunk_size = config.chunk_size
243
+ num_kv_head = model_cfg.get_num_kv_heads(parallel_cfg)
244
+ head_size = model_cfg.get_head_size()
245
+ kv_shape = (num_layer, 1 if use_mla else 2, chunk_size, num_kv_head, head_size)
246
+
247
+ # Extract engine_id and kv_connector_extra_config from vllm_config if available
248
+ engine_id = None
249
+ kv_connector_extra_config = None
250
+ if vllm_config is not None and hasattr(vllm_config, "kv_transfer_config"):
251
+ kv_transfer_config = vllm_config.kv_transfer_config
252
+ if kv_transfer_config is not None:
253
+ engine_id = getattr(kv_transfer_config, "engine_id", None)
254
+ kv_connector_extra_config = getattr(
255
+ kv_transfer_config, "kv_connector_extra_config", None
256
+ )
257
+
258
+ # Create metadata
259
+ metadata = LMCacheMetadata(
260
+ model_name=model_cfg.model,
261
+ world_size=parallel_cfg.world_size,
262
+ local_world_size=parallel_cfg.world_size,
263
+ worker_id=parallel_cfg.rank,
264
+ local_worker_id=parallel_cfg.rank,
265
+ kv_dtype=kv_dtype,
266
+ kv_shape=kv_shape,
267
+ use_mla=use_mla,
268
+ role=role,
269
+ served_model_name=model_cfg.served_model_name,
270
+ engine_id=engine_id,
271
+ kv_connector_extra_config=kv_connector_extra_config,
272
+ )
273
+
274
+ return metadata, config
275
+
276
+
277
+ def extract_mm_features(
278
+ request: "Request", modify: bool = False
279
+ ) -> Tuple[list[str], list["PlaceholderRange"]]:
280
+ """
281
+ Normalize multimodal information from a Request into parallel lists.
282
+
283
+ This helper reads either:
284
+ 1) `request.mm_features` (objects each exposing `.identifier` and
285
+ `.mm_position`), or
286
+ 2) legacy fields `request.mm_hashes` and `request.mm_positions`.
287
+
288
+ It returns two equally sized lists: the multimodal hash identifiers and their
289
+ corresponding positions. If the request contains no multimodal info, it returns
290
+ `([], [])`.
291
+
292
+ Args:
293
+ request (Request): The source object.
294
+ modify (bool):
295
+ Controls copy semantics for the legacy-path return values.
296
+ - If True and legacy fields are used, shallow-copies are returned so
297
+ the caller can mutate the lists without affecting `request`.
298
+ - If False, the original legacy sequences are returned as-is
299
+ (zero-copy); treat them as read-only.
300
+
301
+ Returns:
302
+ Tuple[list[str], list[PlaceholderRange]]: (`mm_hashes`, `mm_positions`).
303
+ May be `([], [])` when no multimodal data is present.
304
+ """
305
+ if getattr(request, "mm_features", None):
306
+ mm_hashes, mm_positions = zip(
307
+ *((f.identifier, f.mm_position) for f in request.mm_features), strict=False
308
+ )
309
+ return (list(mm_hashes), list(mm_positions))
310
+ elif getattr(request, "mm_hashes", None):
311
+ if modify:
312
+ return (request.mm_hashes.copy(), request.mm_positions.copy())
313
+ else:
314
+ return (request.mm_hashes, request.mm_positions)
315
+ else:
316
+ return ([], [])
317
+
318
+
319
+ def get_size_bytes(shapes: list[torch.Size], kv_dtypes: list[torch.dtype]):
320
+ """
321
+ Calculate the size in bytes with the given shapes and dtypes.
322
+ """
323
+ assert len(shapes) == len(kv_dtypes), (
324
+ f"shapes and dtypes must have the same length, "
325
+ f"but got {len(shapes)} and {len(kv_dtypes)}"
326
+ )
327
+ return sum(
328
+ shape.numel() * kv_dtype.itemsize
329
+ for shape, kv_dtype in zip(shapes, kv_dtypes, strict=True)
330
+ )
331
+
332
+
333
+ def get_vllm_torch_dev():
334
+ """
335
+ Returns the torch device and device name for the vLLM engine.
336
+ e.g. (torch.cuda, "cuda") or (torch.xpu, "xpu")
337
+ """
338
+ # Third Party
339
+ from vllm.platforms import current_platform
340
+
341
+ if current_platform.is_cuda_alike():
342
+ logger.info("CUDA device is available. Using CUDA for LMCache engine.")
343
+ torch_dev = torch.cuda
344
+ dev_name = "cuda"
345
+ elif current_platform.is_xpu():
346
+ logger.info("XPU device is available. Using XPU for LMCache engine.")
347
+ torch_dev = torch.xpu
348
+ dev_name = "xpu"
349
+ elif hasattr(torch, "hpu") and torch.hpu.is_available():
350
+ logger.info("HPU device is available. Using HPU for LMCache engine.")
351
+ torch_dev = torch.hpu
352
+ dev_name = "hpu"
353
+ else:
354
+ raise RuntimeError("Unsupported device platform for LMCache engine.")
355
+ return torch_dev, dev_name
356
+
357
+
358
+ def calculate_local_rank_and_world_size(vllm_config: "VllmConfig") -> Tuple[int, int]:
359
+ """
360
+ Calculate the local worker id and local world size.
361
+
362
+ Current assumption (TODO: add custom logic in the future):
363
+ - Tensor Parallel is intra-node
364
+ - Pipeline Parallel is inter-node
365
+
366
+ Returns:
367
+ Tuple[int, int]: (local_worker_id, local_world_size)
368
+ """
369
+ parallel_config = vllm_config.parallel_config
370
+ global_rank = parallel_config.rank
371
+ global_world_size = parallel_config.world_size
372
+ torch_dev, dev_name = get_vllm_torch_dev()
373
+ num_gpus = torch_dev.device_count()
374
+ if global_world_size <= num_gpus:
375
+ # single node case
376
+ return parallel_config.rank, parallel_config.world_size
377
+ else:
378
+ tp_size = parallel_config.tensor_parallel_size
379
+ pp_size = parallel_config.pipeline_parallel_size
380
+ local_world_size = global_world_size // pp_size
381
+ assert local_world_size == tp_size, (
382
+ "LMCache is operating under the assumption that the "
383
+ "local world size is equal to the tensor parallel size "
384
+ "in multi-node deployment."
385
+ )
386
+ local_worker_id = global_rank % local_world_size
387
+ return local_worker_id, local_world_size
388
+
389
+
390
+ def validate_mla_config(config: LMCacheEngineConfig, use_mla: bool) -> None:
391
+ """Validate MLA-related configuration."""
392
+ if use_mla and (config.remote_serde != "naive" and config.remote_serde is not None):
393
+ raise ValueError("MLA only works with naive serde mode..")
394
+
395
+ if use_mla and config.use_layerwise and config.enable_blending:
396
+ raise ValueError(
397
+ "We haven't supported MLA with Cacheblend yet. Please disable blending."
398
+ )
399
+
400
+
401
+ def calculate_draft_layers(vllm_config: "VllmConfig") -> int:
402
+ """Calculate the number of draft layers for speculative decoding."""
403
+ assert vllm_config is not None, "vllm_config required for vLLM mode"
404
+
405
+ num_draft_layers = 0
406
+ model_config = vllm_config.model_config
407
+
408
+ if vllm_config.speculative_config is not None:
409
+ logger.info(
410
+ "vllm_config.speculative_config: %s", vllm_config.speculative_config
411
+ )
412
+ if vllm_config.speculative_config.method == "deepseek_mtp":
413
+ num_draft_layers = getattr(
414
+ model_config.hf_config, "num_nextn_predict_layers", 0
415
+ )
416
+ elif vllm_config.speculative_config.use_eagle():
417
+ try:
418
+ draft_model_config = vllm_config.speculative_config.draft_model_config
419
+ num_draft_layers = draft_model_config.get_num_layers(
420
+ vllm_config.parallel_config
421
+ )
422
+ logger.info("EAGLE detected %d extra layer(s)", num_draft_layers)
423
+ except Exception:
424
+ logger.info(
425
+ "EAGLE detected, but failed to get the number of extra layers"
426
+ "falling back to 1"
427
+ )
428
+ num_draft_layers = 1
429
+ return num_draft_layers
430
+
431
+
432
+ def is_dp_rank0(vllm_config: "VllmConfig") -> bool:
433
+ return vllm_config.parallel_config.data_parallel_rank_local == 0