lmcache-cli 0.4.5.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (399) hide show
  1. lmcache/__init__.py +84 -0
  2. lmcache/_version.py +24 -0
  3. lmcache/cli/__init__.py +1 -0
  4. lmcache/cli/commands/__init__.py +34 -0
  5. lmcache/cli/commands/base.py +157 -0
  6. lmcache/cli/commands/bench/__init__.py +557 -0
  7. lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
  8. lmcache/cli/commands/bench/engine_bench/config.py +245 -0
  9. lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
  10. lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
  11. lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
  12. lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
  13. lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
  14. lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
  15. lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
  16. lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
  17. lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
  18. lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
  19. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
  20. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
  21. lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
  22. lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
  23. lmcache/cli/commands/describe.py +310 -0
  24. lmcache/cli/commands/kvcache.py +133 -0
  25. lmcache/cli/commands/mock.py +75 -0
  26. lmcache/cli/commands/ping.py +113 -0
  27. lmcache/cli/commands/query/__init__.py +155 -0
  28. lmcache/cli/commands/query/prompt.py +134 -0
  29. lmcache/cli/commands/query/request.py +357 -0
  30. lmcache/cli/commands/server.py +99 -0
  31. lmcache/cli/commands/tool/__init__.py +63 -0
  32. lmcache/cli/commands/tool/cache_simulator.py +113 -0
  33. lmcache/cli/commands/trace/__init__.py +505 -0
  34. lmcache/cli/commands/trace/dispatch.py +249 -0
  35. lmcache/cli/commands/trace/driver.py +372 -0
  36. lmcache/cli/commands/trace/stats.py +289 -0
  37. lmcache/cli/documents/lmcache.txt +11 -0
  38. lmcache/cli/main.py +42 -0
  39. lmcache/cli/metrics/__init__.py +29 -0
  40. lmcache/cli/metrics/formatter.py +171 -0
  41. lmcache/cli/metrics/handler.py +94 -0
  42. lmcache/cli/metrics/metrics.py +161 -0
  43. lmcache/cli/metrics/section.py +77 -0
  44. lmcache/connections.py +173 -0
  45. lmcache/integration/__init__.py +2 -0
  46. lmcache/integration/base_service_factory.py +165 -0
  47. lmcache/integration/request_telemetry/__init__.py +1 -0
  48. lmcache/integration/request_telemetry/base.py +51 -0
  49. lmcache/integration/request_telemetry/factory.py +113 -0
  50. lmcache/integration/request_telemetry/fastapi.py +109 -0
  51. lmcache/integration/request_telemetry/noop.py +35 -0
  52. lmcache/integration/sglang/__init__.py +2 -0
  53. lmcache/integration/sglang/sglang_adapter.py +326 -0
  54. lmcache/integration/sglang/utils.py +39 -0
  55. lmcache/integration/vllm/__init__.py +1 -0
  56. lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
  57. lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
  58. lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
  59. lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
  60. lmcache/integration/vllm/utils.py +433 -0
  61. lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
  62. lmcache/integration/vllm/vllm_service_factory.py +339 -0
  63. lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
  64. lmcache/logging.py +107 -0
  65. lmcache/native_storage_ops.pyi +230 -0
  66. lmcache/non_cuda_equivalents.py +1424 -0
  67. lmcache/observability.py +1958 -0
  68. lmcache/storage_backend/serde/__init__.py +1 -0
  69. lmcache/storage_backend/serde/cachegen_basics.py +210 -0
  70. lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
  71. lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
  72. lmcache/storage_backend/serde/serde.py +75 -0
  73. lmcache/tools/__init__.py +1 -0
  74. lmcache/tools/cache_simulator/README.md +392 -0
  75. lmcache/tools/cache_simulator/__init__.py +1 -0
  76. lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
  77. lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
  78. lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
  79. lmcache/tools/cache_simulator/lru_cache.py +124 -0
  80. lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
  81. lmcache/tools/cache_simulator/simulator.py +795 -0
  82. lmcache/tools/controller_benchmark/README.md +161 -0
  83. lmcache/tools/controller_benchmark/__init__.py +1 -0
  84. lmcache/tools/controller_benchmark/__main__.py +331 -0
  85. lmcache/tools/controller_benchmark/benchmark.py +660 -0
  86. lmcache/tools/controller_benchmark/config.py +44 -0
  87. lmcache/tools/controller_benchmark/constants.py +10 -0
  88. lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
  89. lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
  90. lmcache/tools/controller_benchmark/handlers/base.py +47 -0
  91. lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
  92. lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
  93. lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
  94. lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
  95. lmcache/tools/controller_benchmark/handlers/register.py +56 -0
  96. lmcache/tools/mp_status_viewer/__init__.py +1 -0
  97. lmcache/tools/mp_status_viewer/__main__.py +95 -0
  98. lmcache/usage_context.py +417 -0
  99. lmcache/utils.py +665 -0
  100. lmcache/v1/__init__.py +2 -0
  101. lmcache/v1/api_server/__init__.py +2 -0
  102. lmcache/v1/api_server/__main__.py +537 -0
  103. lmcache/v1/basic_check.py +112 -0
  104. lmcache/v1/cache_controller/__init__.py +9 -0
  105. lmcache/v1/cache_controller/commands/__init__.py +15 -0
  106. lmcache/v1/cache_controller/commands/base.py +35 -0
  107. lmcache/v1/cache_controller/commands/full_sync.py +49 -0
  108. lmcache/v1/cache_controller/config.py +176 -0
  109. lmcache/v1/cache_controller/controller_manager.py +535 -0
  110. lmcache/v1/cache_controller/controllers/__init__.py +11 -0
  111. lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
  112. lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
  113. lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
  114. lmcache/v1/cache_controller/executor.py +463 -0
  115. lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
  116. lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
  117. lmcache/v1/cache_controller/frontend/static/index.html +234 -0
  118. lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
  119. lmcache/v1/cache_controller/full_sync_sender.py +475 -0
  120. lmcache/v1/cache_controller/locks.py +149 -0
  121. lmcache/v1/cache_controller/message.py +828 -0
  122. lmcache/v1/cache_controller/observability.py +208 -0
  123. lmcache/v1/cache_controller/utils.py +679 -0
  124. lmcache/v1/cache_controller/worker.py +665 -0
  125. lmcache/v1/cache_engine.py +2058 -0
  126. lmcache/v1/cache_interface.py +19 -0
  127. lmcache/v1/check/__init__.py +74 -0
  128. lmcache/v1/check/check_mode_gen.py +86 -0
  129. lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
  130. lmcache/v1/check/check_mode_test_remote.py +155 -0
  131. lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
  132. lmcache/v1/check/utils.py +571 -0
  133. lmcache/v1/compute/__init__.py +2 -0
  134. lmcache/v1/compute/attention/__init__.py +0 -0
  135. lmcache/v1/compute/attention/abstract.py +39 -0
  136. lmcache/v1/compute/attention/flash_attn.py +129 -0
  137. lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
  138. lmcache/v1/compute/attention/metadata.py +85 -0
  139. lmcache/v1/compute/attention/utils.py +14 -0
  140. lmcache/v1/compute/blend/__init__.py +7 -0
  141. lmcache/v1/compute/blend/blender.py +168 -0
  142. lmcache/v1/compute/blend/metadata.py +34 -0
  143. lmcache/v1/compute/blend/utils.py +63 -0
  144. lmcache/v1/compute/models/__init__.py +0 -0
  145. lmcache/v1/compute/models/base.py +141 -0
  146. lmcache/v1/compute/models/llama.py +9 -0
  147. lmcache/v1/compute/models/qwen3.py +24 -0
  148. lmcache/v1/compute/models/utils.py +68 -0
  149. lmcache/v1/compute/positional_encoding.py +199 -0
  150. lmcache/v1/config.py +848 -0
  151. lmcache/v1/config_base.py +848 -0
  152. lmcache/v1/distributed/api.py +248 -0
  153. lmcache/v1/distributed/config.py +321 -0
  154. lmcache/v1/distributed/error.py +64 -0
  155. lmcache/v1/distributed/eviction.py +192 -0
  156. lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
  157. lmcache/v1/distributed/eviction_policy/factory.py +27 -0
  158. lmcache/v1/distributed/eviction_policy/lru.py +244 -0
  159. lmcache/v1/distributed/eviction_policy/noop.py +50 -0
  160. lmcache/v1/distributed/internal_api.py +170 -0
  161. lmcache/v1/distributed/l1_manager.py +835 -0
  162. lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
  163. lmcache/v1/distributed/l2_adapters/base.py +360 -0
  164. lmcache/v1/distributed/l2_adapters/config.py +385 -0
  165. lmcache/v1/distributed/l2_adapters/factory.py +205 -0
  166. lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
  167. lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
  168. lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
  169. lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
  170. lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
  171. lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
  172. lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
  173. lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
  174. lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
  175. lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
  176. lmcache/v1/distributed/memory_manager.py +179 -0
  177. lmcache/v1/distributed/storage_controller.py +39 -0
  178. lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
  179. lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
  180. lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
  181. lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
  182. lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
  183. lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
  184. lmcache/v1/distributed/storage_manager.py +532 -0
  185. lmcache/v1/event_manager.py +145 -0
  186. lmcache/v1/exceptions/__init__.py +16 -0
  187. lmcache/v1/gpu_connector/__init__.py +126 -0
  188. lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
  189. lmcache/v1/gpu_connector/gpu_ops.py +85 -0
  190. lmcache/v1/gpu_connector/hpu_connector.py +326 -0
  191. lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
  192. lmcache/v1/gpu_connector/utils.py +890 -0
  193. lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
  194. lmcache/v1/health_monitor/__init__.py +1 -0
  195. lmcache/v1/health_monitor/base.py +587 -0
  196. lmcache/v1/health_monitor/checks/__init__.py +1 -0
  197. lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
  198. lmcache/v1/health_monitor/constants.py +36 -0
  199. lmcache/v1/internal_api_server/__init__.py +0 -0
  200. lmcache/v1/internal_api_server/api_registry.py +59 -0
  201. lmcache/v1/internal_api_server/api_server.py +120 -0
  202. lmcache/v1/internal_api_server/common/__init__.py +1 -0
  203. lmcache/v1/internal_api_server/common/env_api.py +22 -0
  204. lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
  205. lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
  206. lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
  207. lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
  208. lmcache/v1/internal_api_server/common/thread_api.py +63 -0
  209. lmcache/v1/internal_api_server/controller/__init__.py +1 -0
  210. lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
  211. lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
  212. lmcache/v1/internal_api_server/utils.py +43 -0
  213. lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
  214. lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
  215. lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
  216. lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
  217. lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
  218. lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
  219. lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
  220. lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
  221. lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
  222. lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
  223. lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
  224. lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
  225. lmcache/v1/kv_layer_groups.py +267 -0
  226. lmcache/v1/lazy_memory_allocator.py +284 -0
  227. lmcache/v1/lookup_client/__init__.py +25 -0
  228. lmcache/v1/lookup_client/abstract_client.py +77 -0
  229. lmcache/v1/lookup_client/async_lookup_message.py +50 -0
  230. lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
  231. lmcache/v1/lookup_client/factory.py +251 -0
  232. lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
  233. lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
  234. lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
  235. lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
  236. lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
  237. lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
  238. lmcache/v1/lookup_client/record_strategies/base.py +327 -0
  239. lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
  240. lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
  241. lmcache/v1/manager.py +539 -0
  242. lmcache/v1/memory_management.py +2619 -0
  243. lmcache/v1/metadata.py +114 -0
  244. lmcache/v1/mp_observability/AGENTS.override.md +21 -0
  245. lmcache/v1/mp_observability/README.md +204 -0
  246. lmcache/v1/mp_observability/config.py +340 -0
  247. lmcache/v1/mp_observability/event.py +100 -0
  248. lmcache/v1/mp_observability/event_bus.py +313 -0
  249. lmcache/v1/mp_observability/otel_init.py +145 -0
  250. lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
  251. lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
  252. lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
  253. lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
  254. lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
  255. lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
  256. lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
  257. lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
  258. lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
  259. lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
  260. lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
  261. lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
  262. lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
  263. lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
  264. lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
  265. lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
  266. lmcache/v1/mp_observability/trace/__init__.py +50 -0
  267. lmcache/v1/mp_observability/trace/codecs.py +255 -0
  268. lmcache/v1/mp_observability/trace/decorator.py +147 -0
  269. lmcache/v1/mp_observability/trace/format.py +132 -0
  270. lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
  271. lmcache/v1/mp_observability/trace/reader.py +167 -0
  272. lmcache/v1/mp_observability/trace/recorder.py +300 -0
  273. lmcache/v1/multiprocess/__init__.py +0 -0
  274. lmcache/v1/multiprocess/affinity_pool.py +102 -0
  275. lmcache/v1/multiprocess/blend_server_v2.py +891 -0
  276. lmcache/v1/multiprocess/config.py +253 -0
  277. lmcache/v1/multiprocess/custom_types.py +281 -0
  278. lmcache/v1/multiprocess/futures.py +194 -0
  279. lmcache/v1/multiprocess/gpu_context.py +511 -0
  280. lmcache/v1/multiprocess/http_server.py +235 -0
  281. lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
  282. lmcache/v1/multiprocess/mq.py +732 -0
  283. lmcache/v1/multiprocess/protocol.py +86 -0
  284. lmcache/v1/multiprocess/protocols/README.md +213 -0
  285. lmcache/v1/multiprocess/protocols/__init__.py +127 -0
  286. lmcache/v1/multiprocess/protocols/base.py +89 -0
  287. lmcache/v1/multiprocess/protocols/blend.py +109 -0
  288. lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
  289. lmcache/v1/multiprocess/protocols/controller.py +53 -0
  290. lmcache/v1/multiprocess/protocols/debug.py +34 -0
  291. lmcache/v1/multiprocess/protocols/engine.py +146 -0
  292. lmcache/v1/multiprocess/protocols/observability.py +39 -0
  293. lmcache/v1/multiprocess/server.py +1134 -0
  294. lmcache/v1/multiprocess/session.py +190 -0
  295. lmcache/v1/multiprocess/token_hasher.py +441 -0
  296. lmcache/v1/offload_server/__init__.py +17 -0
  297. lmcache/v1/offload_server/abstract_server.py +37 -0
  298. lmcache/v1/offload_server/message.py +30 -0
  299. lmcache/v1/offload_server/zmq_server.py +122 -0
  300. lmcache/v1/periodic_thread.py +579 -0
  301. lmcache/v1/pin_monitor.py +246 -0
  302. lmcache/v1/plugin/__init__.py +0 -0
  303. lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
  304. lmcache/v1/protocol.py +317 -0
  305. lmcache/v1/rpc/__init__.py +17 -0
  306. lmcache/v1/rpc/transport.py +105 -0
  307. lmcache/v1/rpc/zmq_transport.py +213 -0
  308. lmcache/v1/rpc_utils.py +165 -0
  309. lmcache/v1/server/__init__.py +2 -0
  310. lmcache/v1/server/__main__.py +170 -0
  311. lmcache/v1/server/storage_backend/__init__.py +21 -0
  312. lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
  313. lmcache/v1/server/storage_backend/local_backend.py +75 -0
  314. lmcache/v1/server/utils.py +21 -0
  315. lmcache/v1/standalone/__init__.py +1 -0
  316. lmcache/v1/standalone/__main__.py +583 -0
  317. lmcache/v1/standalone/manager.py +80 -0
  318. lmcache/v1/standalone/standalone_service_factory.py +86 -0
  319. lmcache/v1/storage_backend/__init__.py +313 -0
  320. lmcache/v1/storage_backend/abstract_backend.py +445 -0
  321. lmcache/v1/storage_backend/audit_backend.py +233 -0
  322. lmcache/v1/storage_backend/batched_message_sender.py +222 -0
  323. lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
  324. lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
  325. lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
  326. lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
  327. lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
  328. lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
  329. lmcache/v1/storage_backend/connector/__init__.py +443 -0
  330. lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
  331. lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
  332. lmcache/v1/storage_backend/connector/base_connector.py +379 -0
  333. lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
  334. lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
  335. lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
  336. lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
  337. lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
  338. lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
  339. lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
  340. lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
  341. lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
  342. lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
  343. lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
  344. lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
  345. lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
  346. lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
  347. lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
  348. lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
  349. lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
  350. lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
  351. lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
  352. lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
  353. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
  354. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
  355. lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
  356. lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
  357. lmcache/v1/storage_backend/gds_backend.py +1199 -0
  358. lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
  359. lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
  360. lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
  361. lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
  362. lmcache/v1/storage_backend/local_disk_backend.py +656 -0
  363. lmcache/v1/storage_backend/maru_backend.py +734 -0
  364. lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
  365. lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
  366. lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
  367. lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
  368. lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
  369. lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
  370. lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
  371. lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
  372. lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
  373. lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
  374. lmcache/v1/storage_backend/p2p_backend.py +788 -0
  375. lmcache/v1/storage_backend/path_sharder.py +117 -0
  376. lmcache/v1/storage_backend/pd_backend.py +646 -0
  377. lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
  378. lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
  379. lmcache/v1/storage_backend/remote_backend.py +624 -0
  380. lmcache/v1/storage_backend/resp_client.py +227 -0
  381. lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
  382. lmcache/v1/storage_backend/storage_manager.py +1352 -0
  383. lmcache/v1/system_detection.py +110 -0
  384. lmcache/v1/token_database.py +551 -0
  385. lmcache/v1/transfer_channel/__init__.py +83 -0
  386. lmcache/v1/transfer_channel/abstract.py +285 -0
  387. lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
  388. lmcache/v1/transfer_channel/nixl_channel.py +639 -0
  389. lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
  390. lmcache/v1/transfer_channel/transfer_utils.py +63 -0
  391. lmcache/v1/utils/__init__.py +1 -0
  392. lmcache/v1/utils/bloom_filter.py +109 -0
  393. lmcache/v1/utils/cache_utils.py +125 -0
  394. lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
  395. lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
  396. lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
  397. lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
  398. lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
  399. lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
@@ -0,0 +1 @@
1
+ # SPDX-License-Identifier: Apache-2.0
@@ -0,0 +1,210 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Standard
3
+ from dataclasses import dataclass
4
+ from typing import List
5
+ import io
6
+ import pickle
7
+
8
+ # Third Party
9
+ from transformers import AutoConfig
10
+ import torch
11
+
12
+ # First Party
13
+ from lmcache.logging import init_logger
14
+ from lmcache.utils import _lmcache_nvtx_annotate
15
+
16
+ logger = init_logger(__name__)
17
+
18
+ CACHEGEN_GPU_MAX_TOKENS_PER_CHUNK = 256
19
+
20
+
21
+ @dataclass
22
+ class QuantizationSpec:
23
+ start_layer: int
24
+ end_layer: int
25
+ bins: int
26
+
27
+ def __getitem__(self, key: str) -> int:
28
+ return getattr(self, key)
29
+
30
+
31
+ @dataclass
32
+ class CacheGenConfig:
33
+ # TODO: move this class to another file like "cachegen_basics.py"
34
+ nlayers: int
35
+ kspecs: List[QuantizationSpec]
36
+ vspecs: List[QuantizationSpec]
37
+
38
+ def __getitem__(self, key: str) -> int:
39
+ return getattr(self, key)
40
+
41
+ @staticmethod
42
+ def from_model_name(model_name: str) -> "CacheGenConfig":
43
+ family_7b = [
44
+ "mistralai/Mistral-7B-Instruct-v0.2",
45
+ "lmsys/longchat-7b-16k",
46
+ "Qwen/Qwen-7B",
47
+ ]
48
+ family_8b = ["meta-llama/Llama-3.1-8B-Instruct"]
49
+ family_9b = ["THUDM/glm-4-9b-chat"]
50
+ if model_name in family_7b:
51
+ return CacheGenConfig(
52
+ nlayers=32,
53
+ kspecs=[
54
+ QuantizationSpec(start_layer=0, end_layer=10, bins=32),
55
+ QuantizationSpec(start_layer=10, end_layer=32, bins=16),
56
+ ],
57
+ vspecs=[
58
+ QuantizationSpec(start_layer=0, end_layer=2, bins=32),
59
+ QuantizationSpec(start_layer=2, end_layer=32, bins=16),
60
+ ],
61
+ )
62
+ elif model_name in family_8b:
63
+ return CacheGenConfig(
64
+ nlayers=32,
65
+ kspecs=[
66
+ QuantizationSpec(start_layer=0, end_layer=10, bins=32),
67
+ QuantizationSpec(start_layer=10, end_layer=32, bins=16),
68
+ ],
69
+ vspecs=[
70
+ QuantizationSpec(start_layer=0, end_layer=2, bins=32),
71
+ QuantizationSpec(start_layer=2, end_layer=32, bins=16),
72
+ ],
73
+ )
74
+ # TODO(Jiayi): needs tuning for better quality
75
+ elif model_name in family_9b:
76
+ return CacheGenConfig(
77
+ nlayers=40,
78
+ kspecs=[
79
+ QuantizationSpec(start_layer=0, end_layer=10, bins=32),
80
+ QuantizationSpec(start_layer=10, end_layer=40, bins=16),
81
+ ],
82
+ vspecs=[
83
+ QuantizationSpec(start_layer=0, end_layer=2, bins=32),
84
+ QuantizationSpec(start_layer=2, end_layer=40, bins=16),
85
+ ],
86
+ )
87
+ else:
88
+ try:
89
+ config = AutoConfig.from_pretrained(model_name)
90
+ # Default name caught by num_hidden_layers
91
+ if config.num_hidden_layers is None:
92
+ raise ValueError(
93
+ f"num_hidden_layers is None for model {model_name}"
94
+ )
95
+ if config.num_hidden_layers < 10:
96
+ return CacheGenConfig(
97
+ nlayers=config.num_hidden_layers,
98
+ kspecs=[
99
+ QuantizationSpec(
100
+ start_layer=0,
101
+ end_layer=config.num_hidden_layers,
102
+ bins=32,
103
+ ),
104
+ ],
105
+ vspecs=[
106
+ QuantizationSpec(
107
+ start_layer=0,
108
+ end_layer=config.num_hidden_layers,
109
+ bins=32,
110
+ ),
111
+ ],
112
+ )
113
+ else:
114
+ return CacheGenConfig(
115
+ nlayers=config.num_hidden_layers,
116
+ kspecs=[
117
+ QuantizationSpec(start_layer=0, end_layer=10, bins=32),
118
+ QuantizationSpec(
119
+ start_layer=10,
120
+ end_layer=config.num_hidden_layers,
121
+ bins=16,
122
+ ),
123
+ ],
124
+ vspecs=[
125
+ QuantizationSpec(start_layer=0, end_layer=2, bins=32),
126
+ QuantizationSpec(
127
+ start_layer=2,
128
+ end_layer=config.num_hidden_layers,
129
+ bins=16,
130
+ ),
131
+ ],
132
+ )
133
+ except Exception as e:
134
+ raise ValueError(
135
+ f"Model {model_name} not supported by CacheGenConfig"
136
+ ) from e
137
+
138
+
139
+ @dataclass
140
+ class CacheGenEncoderOutput:
141
+ # TODO: maybe use numpy array so that we can directly tobytes() and
142
+ # frombuffer() to have a better performance
143
+ bytestream: bytes
144
+ start_indices: torch.Tensor
145
+ cdf: torch.Tensor
146
+ max_tensors_key: torch.Tensor
147
+ max_tensors_value: torch.Tensor
148
+ num_heads: int
149
+ head_size: int
150
+
151
+ def __getitem__(self, key: str) -> int:
152
+ return getattr(self, key)
153
+
154
+ def to_bytes(self) -> bytes:
155
+ """Save the output to a file"""
156
+ with io.BytesIO() as f:
157
+ # torch.save(self, f)
158
+ pickle.dump(self, f)
159
+ return f.getvalue()
160
+
161
+ @staticmethod
162
+ def from_bytes(bs: bytes) -> "CacheGenEncoderOutput":
163
+ with io.BytesIO(bs) as f:
164
+ return pickle.load(f)
165
+
166
+
167
+ @dataclass
168
+ class CacheGenGPUBytestream:
169
+ bytestream: torch.Tensor
170
+ bytestream_lengths: torch.Tensor # [nlayers, nchannels, bytestream_length]
171
+ ntokens: int
172
+
173
+ def __getitem__(self, key: str) -> int:
174
+ return getattr(self, key)
175
+
176
+
177
+ @dataclass
178
+ class CacheGenGPUEncoderOutput:
179
+ data_chunks: List[CacheGenGPUBytestream]
180
+ cdf: torch.Tensor
181
+ max_tensors_key: torch.Tensor
182
+ max_tensors_value: torch.Tensor
183
+ num_heads: int
184
+ head_size: int
185
+
186
+ def __getitem__(self, key: str) -> int:
187
+ return getattr(self, key)
188
+
189
+ @_lmcache_nvtx_annotate
190
+ def to_bytes(self) -> bytes:
191
+ """Save the output to a file"""
192
+ with io.BytesIO() as f:
193
+ pickle.dump(self, f)
194
+ return f.getvalue()
195
+
196
+ @staticmethod
197
+ @_lmcache_nvtx_annotate
198
+ def from_bytes(bs: bytes) -> "CacheGenGPUEncoderOutput":
199
+ with io.BytesIO(bs) as f:
200
+ return pickle.load(f)
201
+
202
+ def debug_print_device(self):
203
+ logger.debug(f"bytestream device: {self.data_chunks[0].bytestream.device}")
204
+ logger.debug(
205
+ f"bytestream_lengths device: "
206
+ f"{self.data_chunks[0].bytestream_lengths.device}"
207
+ )
208
+ logger.debug(f"cdf device: {self.cdf.device}")
209
+ logger.debug(f"max_tensors_key device: {self.max_tensors_key.device}")
210
+ logger.debug(f"max_tensors_value device: {self.max_tensors_value.device}")
@@ -0,0 +1,207 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Standard
3
+ from typing import List, Optional
4
+
5
+ # Third Party
6
+ import torch
7
+
8
+ # First Party
9
+ from lmcache.logging import init_logger
10
+ from lmcache.storage_backend.serde.cachegen_basics import (
11
+ CacheGenConfig,
12
+ CacheGenGPUBytestream,
13
+ CacheGenGPUEncoderOutput,
14
+ )
15
+ from lmcache.storage_backend.serde.serde import Deserializer
16
+ from lmcache.utils import _lmcache_nvtx_annotate
17
+ from lmcache.v1.config import LMCacheEngineConfig
18
+ from lmcache.v1.metadata import LMCacheMetadata
19
+ import lmcache.c_ops as lmc_ops
20
+ import lmcache.storage_backend.serde.cachegen_basics as CGBasics
21
+
22
+ logger = init_logger(__name__)
23
+
24
+
25
+ @_lmcache_nvtx_annotate
26
+ def quant(bins: int, xq: torch.Tensor, max1: float):
27
+ C = bins // 2 - 1
28
+ x = xq / C * max1
29
+ return x
30
+
31
+
32
+ def do_dequantize(t: torch.Tensor, bins: torch.Tensor, maxtensors: torch.Tensor):
33
+ """
34
+ t: [nlayers, ntokens, nchannels]
35
+ bins: [nlayers]
36
+ maxtensors: [nlayers, ntokens, 1]
37
+ """
38
+ C = (bins // 2 - 1)[:, None, None]
39
+ t = t - C
40
+ t = t / C
41
+ t = t * maxtensors
42
+ return t
43
+
44
+
45
+ @_lmcache_nvtx_annotate
46
+ def recombine_bytes(bytes_tensor, output_lengths) -> torch.Tensor:
47
+ output_buffer_size = CGBasics.CACHEGEN_GPU_MAX_TOKENS_PER_CHUNK
48
+ offsets = output_lengths.flatten().cumsum(0).roll(1).reshape(output_lengths.shape)
49
+ offsets[0][0] = 0
50
+ indexes = torch.arange(output_buffer_size, device=offsets.device).tile(
51
+ (output_lengths.shape[0], output_lengths.shape[1], 1)
52
+ )
53
+ final_indexes = (indexes + offsets[:, :, None]).clamp(max=len(bytes_tensor) - 1)
54
+ return bytes_tensor[final_indexes]
55
+
56
+
57
+ @_lmcache_nvtx_annotate
58
+ def decode_chunk(
59
+ cdf: torch.Tensor,
60
+ data_chunk: CacheGenGPUBytestream,
61
+ target_buffer: torch.Tensor,
62
+ ) -> None:
63
+ """
64
+ Write the decode output in target_buffer
65
+ Expected shape: [nlayers (kv in total), ntokens, nchannels]
66
+ """
67
+ bytes_tensor = data_chunk.bytestream
68
+ length_prefsum = (
69
+ data_chunk.bytestream_lengths.flatten()
70
+ .cumsum(0)
71
+ .reshape(data_chunk.bytestream_lengths.shape)
72
+ )
73
+ lmc_ops.decode_fast_prefsum(cdf, bytes_tensor, length_prefsum, target_buffer)
74
+
75
+
76
+ @_lmcache_nvtx_annotate
77
+ def decode_function_gpu(
78
+ cdf: torch.Tensor,
79
+ data_chunks: List[CacheGenGPUBytestream],
80
+ layers_in_key: int,
81
+ chunk_size: int,
82
+ output: torch.Tensor,
83
+ ):
84
+ # TODO: dtype and shape -- still have 128 and 8
85
+ """
86
+ Given the path to the encoded KV bytestream, decode the KV cache
87
+
88
+ Inputs:
89
+ cdf: the cdf tensor, in shape [2 * nlayers, nchannels, bins + 1]
90
+ data_chunks: the data_chunks in the encoder's output
91
+ layers_in_key: number of layers in K (or V)
92
+ (K/V should have the same number of layers)
93
+ chunk_size: the chunk_size
94
+ output: output buffer, in shape [ntokens, 2 * nlayers * nchannels]
95
+
96
+ Outputs:
97
+ key: the decoded key tensor in the shape of (layers, tokens, nchannels)
98
+ value: the decoded value tensor in the shape of
99
+ (layers, tokens, nchannels)
100
+ """
101
+ nlayers, nchannels, _ = cdf.shape
102
+ output = output.reshape((nlayers, chunk_size, nchannels))
103
+
104
+ start = 0
105
+ for data_chunk in data_chunks:
106
+ end = start + data_chunk.ntokens
107
+ decode_chunk(cdf, data_chunk, output[:, start:end, :])
108
+ start = end
109
+
110
+ out = output.reshape((2, layers_in_key, chunk_size, nchannels))
111
+ key, value = out.float()
112
+
113
+ return key, value
114
+
115
+
116
+ class CacheGenDeserializer(Deserializer):
117
+ def __init__(
118
+ self,
119
+ config: LMCacheEngineConfig,
120
+ metadata: LMCacheMetadata,
121
+ dtype,
122
+ ):
123
+ self.dtype = dtype
124
+ self.cachegen_config = CacheGenConfig.from_model_name(metadata.model_name)
125
+ self.chunk_size = config.chunk_size
126
+ self.output_buffer: Optional[torch.Tensor] = None
127
+ self.key_bins = self.make_key_bins(self.cachegen_config)
128
+ self.value_bins = self.make_value_bins(self.cachegen_config)
129
+
130
+ def make_key_bins(self, config: CacheGenConfig) -> torch.Tensor:
131
+ ret = torch.zeros(config.nlayers)
132
+ for spec in config.kspecs:
133
+ ret[spec.start_layer : spec.end_layer] = spec.bins
134
+ return ret.cuda()
135
+
136
+ def make_value_bins(self, config: CacheGenConfig) -> torch.Tensor:
137
+ ret = torch.zeros(config.nlayers)
138
+ for spec in config.vspecs:
139
+ ret[spec.start_layer : spec.end_layer] = spec.bins
140
+ return ret.cuda()
141
+
142
+ def get_output_buffer(self, nlayers: int, nchannels: int, ntokens: int):
143
+ if (
144
+ self.output_buffer is None
145
+ or self.output_buffer.shape[1] != 2 * nlayers * nchannels
146
+ ):
147
+ self.output_buffer = torch.zeros(
148
+ (self.chunk_size, 2 * nlayers * nchannels), dtype=torch.uint8
149
+ ).cuda()
150
+ return self.output_buffer[:ntokens, :]
151
+
152
+ @_lmcache_nvtx_annotate
153
+ def from_bytes(self, bs: bytes) -> torch.Tensor:
154
+ encoder_output = CacheGenGPUEncoderOutput.from_bytes(bs)
155
+ encoder_output.max_tensors_key = encoder_output.max_tensors_key.cuda()
156
+ encoder_output.max_tensors_value = encoder_output.max_tensors_value.cuda()
157
+
158
+ ntokens = encoder_output.max_tensors_key.shape[1]
159
+ layers_in_key = encoder_output.max_tensors_key.shape[0]
160
+ key, value = decode_function_gpu(
161
+ encoder_output.cdf,
162
+ encoder_output.data_chunks,
163
+ layers_in_key,
164
+ ntokens,
165
+ self.get_output_buffer(
166
+ encoder_output.cdf.shape[0] // 2,
167
+ encoder_output.cdf.shape[1],
168
+ ntokens,
169
+ ),
170
+ )
171
+
172
+ # Temporary fix for #83: change the device of key_bins and value_bins
173
+ # to the device of key and value
174
+ # This requires a long-term fix in the future. Currently,
175
+ # CacheGenGPUEncoderOutput has implicit device in itself.
176
+ # More specifically, if the encoder encodes the tensor on GPU0, the
177
+ # from_bytes will also return a tensor on GPU0
178
+ # We may want to dynamically configure the device based on config and
179
+ # metadata in the future
180
+ if self.key_bins.device != key.device:
181
+ self.key_bins = self.key_bins.to(key.device)
182
+
183
+ if self.value_bins.device != value.device:
184
+ self.value_bins = self.value_bins.cuda()
185
+
186
+ key = do_dequantize(key, self.key_bins, encoder_output.max_tensors_key)
187
+ value = do_dequantize(value, self.value_bins, encoder_output.max_tensors_value)
188
+ """ merge key and value back and reshape """
189
+ nlayers, ntokens, nchannels = key.shape
190
+ blob = torch.stack([key, value]) # [2, nlayers, ntokens, nchannels]
191
+ blob = blob.reshape(
192
+ (
193
+ 2,
194
+ nlayers,
195
+ ntokens,
196
+ encoder_output.num_heads,
197
+ encoder_output.head_size,
198
+ )
199
+ )
200
+
201
+ return blob.permute((1, 0, 2, 3, 4)).to(
202
+ self.dtype
203
+ ) # [nlayers, 2, ntokens, num_heads, head_size]
204
+ # huggingface
205
+ # return blob.permute((1, 0, 3, 2, 4)).to(
206
+ # self.dtype
207
+ # ) # [nlayers, 2, num_heads, ntokens, head_size]