lmcache-cli 0.4.5.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (399) hide show
  1. lmcache/__init__.py +84 -0
  2. lmcache/_version.py +24 -0
  3. lmcache/cli/__init__.py +1 -0
  4. lmcache/cli/commands/__init__.py +34 -0
  5. lmcache/cli/commands/base.py +157 -0
  6. lmcache/cli/commands/bench/__init__.py +557 -0
  7. lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
  8. lmcache/cli/commands/bench/engine_bench/config.py +245 -0
  9. lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
  10. lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
  11. lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
  12. lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
  13. lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
  14. lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
  15. lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
  16. lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
  17. lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
  18. lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
  19. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
  20. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
  21. lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
  22. lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
  23. lmcache/cli/commands/describe.py +310 -0
  24. lmcache/cli/commands/kvcache.py +133 -0
  25. lmcache/cli/commands/mock.py +75 -0
  26. lmcache/cli/commands/ping.py +113 -0
  27. lmcache/cli/commands/query/__init__.py +155 -0
  28. lmcache/cli/commands/query/prompt.py +134 -0
  29. lmcache/cli/commands/query/request.py +357 -0
  30. lmcache/cli/commands/server.py +99 -0
  31. lmcache/cli/commands/tool/__init__.py +63 -0
  32. lmcache/cli/commands/tool/cache_simulator.py +113 -0
  33. lmcache/cli/commands/trace/__init__.py +505 -0
  34. lmcache/cli/commands/trace/dispatch.py +249 -0
  35. lmcache/cli/commands/trace/driver.py +372 -0
  36. lmcache/cli/commands/trace/stats.py +289 -0
  37. lmcache/cli/documents/lmcache.txt +11 -0
  38. lmcache/cli/main.py +42 -0
  39. lmcache/cli/metrics/__init__.py +29 -0
  40. lmcache/cli/metrics/formatter.py +171 -0
  41. lmcache/cli/metrics/handler.py +94 -0
  42. lmcache/cli/metrics/metrics.py +161 -0
  43. lmcache/cli/metrics/section.py +77 -0
  44. lmcache/connections.py +173 -0
  45. lmcache/integration/__init__.py +2 -0
  46. lmcache/integration/base_service_factory.py +165 -0
  47. lmcache/integration/request_telemetry/__init__.py +1 -0
  48. lmcache/integration/request_telemetry/base.py +51 -0
  49. lmcache/integration/request_telemetry/factory.py +113 -0
  50. lmcache/integration/request_telemetry/fastapi.py +109 -0
  51. lmcache/integration/request_telemetry/noop.py +35 -0
  52. lmcache/integration/sglang/__init__.py +2 -0
  53. lmcache/integration/sglang/sglang_adapter.py +326 -0
  54. lmcache/integration/sglang/utils.py +39 -0
  55. lmcache/integration/vllm/__init__.py +1 -0
  56. lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
  57. lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
  58. lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
  59. lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
  60. lmcache/integration/vllm/utils.py +433 -0
  61. lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
  62. lmcache/integration/vllm/vllm_service_factory.py +339 -0
  63. lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
  64. lmcache/logging.py +107 -0
  65. lmcache/native_storage_ops.pyi +230 -0
  66. lmcache/non_cuda_equivalents.py +1424 -0
  67. lmcache/observability.py +1958 -0
  68. lmcache/storage_backend/serde/__init__.py +1 -0
  69. lmcache/storage_backend/serde/cachegen_basics.py +210 -0
  70. lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
  71. lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
  72. lmcache/storage_backend/serde/serde.py +75 -0
  73. lmcache/tools/__init__.py +1 -0
  74. lmcache/tools/cache_simulator/README.md +392 -0
  75. lmcache/tools/cache_simulator/__init__.py +1 -0
  76. lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
  77. lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
  78. lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
  79. lmcache/tools/cache_simulator/lru_cache.py +124 -0
  80. lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
  81. lmcache/tools/cache_simulator/simulator.py +795 -0
  82. lmcache/tools/controller_benchmark/README.md +161 -0
  83. lmcache/tools/controller_benchmark/__init__.py +1 -0
  84. lmcache/tools/controller_benchmark/__main__.py +331 -0
  85. lmcache/tools/controller_benchmark/benchmark.py +660 -0
  86. lmcache/tools/controller_benchmark/config.py +44 -0
  87. lmcache/tools/controller_benchmark/constants.py +10 -0
  88. lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
  89. lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
  90. lmcache/tools/controller_benchmark/handlers/base.py +47 -0
  91. lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
  92. lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
  93. lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
  94. lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
  95. lmcache/tools/controller_benchmark/handlers/register.py +56 -0
  96. lmcache/tools/mp_status_viewer/__init__.py +1 -0
  97. lmcache/tools/mp_status_viewer/__main__.py +95 -0
  98. lmcache/usage_context.py +417 -0
  99. lmcache/utils.py +665 -0
  100. lmcache/v1/__init__.py +2 -0
  101. lmcache/v1/api_server/__init__.py +2 -0
  102. lmcache/v1/api_server/__main__.py +537 -0
  103. lmcache/v1/basic_check.py +112 -0
  104. lmcache/v1/cache_controller/__init__.py +9 -0
  105. lmcache/v1/cache_controller/commands/__init__.py +15 -0
  106. lmcache/v1/cache_controller/commands/base.py +35 -0
  107. lmcache/v1/cache_controller/commands/full_sync.py +49 -0
  108. lmcache/v1/cache_controller/config.py +176 -0
  109. lmcache/v1/cache_controller/controller_manager.py +535 -0
  110. lmcache/v1/cache_controller/controllers/__init__.py +11 -0
  111. lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
  112. lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
  113. lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
  114. lmcache/v1/cache_controller/executor.py +463 -0
  115. lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
  116. lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
  117. lmcache/v1/cache_controller/frontend/static/index.html +234 -0
  118. lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
  119. lmcache/v1/cache_controller/full_sync_sender.py +475 -0
  120. lmcache/v1/cache_controller/locks.py +149 -0
  121. lmcache/v1/cache_controller/message.py +828 -0
  122. lmcache/v1/cache_controller/observability.py +208 -0
  123. lmcache/v1/cache_controller/utils.py +679 -0
  124. lmcache/v1/cache_controller/worker.py +665 -0
  125. lmcache/v1/cache_engine.py +2058 -0
  126. lmcache/v1/cache_interface.py +19 -0
  127. lmcache/v1/check/__init__.py +74 -0
  128. lmcache/v1/check/check_mode_gen.py +86 -0
  129. lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
  130. lmcache/v1/check/check_mode_test_remote.py +155 -0
  131. lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
  132. lmcache/v1/check/utils.py +571 -0
  133. lmcache/v1/compute/__init__.py +2 -0
  134. lmcache/v1/compute/attention/__init__.py +0 -0
  135. lmcache/v1/compute/attention/abstract.py +39 -0
  136. lmcache/v1/compute/attention/flash_attn.py +129 -0
  137. lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
  138. lmcache/v1/compute/attention/metadata.py +85 -0
  139. lmcache/v1/compute/attention/utils.py +14 -0
  140. lmcache/v1/compute/blend/__init__.py +7 -0
  141. lmcache/v1/compute/blend/blender.py +168 -0
  142. lmcache/v1/compute/blend/metadata.py +34 -0
  143. lmcache/v1/compute/blend/utils.py +63 -0
  144. lmcache/v1/compute/models/__init__.py +0 -0
  145. lmcache/v1/compute/models/base.py +141 -0
  146. lmcache/v1/compute/models/llama.py +9 -0
  147. lmcache/v1/compute/models/qwen3.py +24 -0
  148. lmcache/v1/compute/models/utils.py +68 -0
  149. lmcache/v1/compute/positional_encoding.py +199 -0
  150. lmcache/v1/config.py +848 -0
  151. lmcache/v1/config_base.py +848 -0
  152. lmcache/v1/distributed/api.py +248 -0
  153. lmcache/v1/distributed/config.py +321 -0
  154. lmcache/v1/distributed/error.py +64 -0
  155. lmcache/v1/distributed/eviction.py +192 -0
  156. lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
  157. lmcache/v1/distributed/eviction_policy/factory.py +27 -0
  158. lmcache/v1/distributed/eviction_policy/lru.py +244 -0
  159. lmcache/v1/distributed/eviction_policy/noop.py +50 -0
  160. lmcache/v1/distributed/internal_api.py +170 -0
  161. lmcache/v1/distributed/l1_manager.py +835 -0
  162. lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
  163. lmcache/v1/distributed/l2_adapters/base.py +360 -0
  164. lmcache/v1/distributed/l2_adapters/config.py +385 -0
  165. lmcache/v1/distributed/l2_adapters/factory.py +205 -0
  166. lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
  167. lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
  168. lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
  169. lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
  170. lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
  171. lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
  172. lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
  173. lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
  174. lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
  175. lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
  176. lmcache/v1/distributed/memory_manager.py +179 -0
  177. lmcache/v1/distributed/storage_controller.py +39 -0
  178. lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
  179. lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
  180. lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
  181. lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
  182. lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
  183. lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
  184. lmcache/v1/distributed/storage_manager.py +532 -0
  185. lmcache/v1/event_manager.py +145 -0
  186. lmcache/v1/exceptions/__init__.py +16 -0
  187. lmcache/v1/gpu_connector/__init__.py +126 -0
  188. lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
  189. lmcache/v1/gpu_connector/gpu_ops.py +85 -0
  190. lmcache/v1/gpu_connector/hpu_connector.py +326 -0
  191. lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
  192. lmcache/v1/gpu_connector/utils.py +890 -0
  193. lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
  194. lmcache/v1/health_monitor/__init__.py +1 -0
  195. lmcache/v1/health_monitor/base.py +587 -0
  196. lmcache/v1/health_monitor/checks/__init__.py +1 -0
  197. lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
  198. lmcache/v1/health_monitor/constants.py +36 -0
  199. lmcache/v1/internal_api_server/__init__.py +0 -0
  200. lmcache/v1/internal_api_server/api_registry.py +59 -0
  201. lmcache/v1/internal_api_server/api_server.py +120 -0
  202. lmcache/v1/internal_api_server/common/__init__.py +1 -0
  203. lmcache/v1/internal_api_server/common/env_api.py +22 -0
  204. lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
  205. lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
  206. lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
  207. lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
  208. lmcache/v1/internal_api_server/common/thread_api.py +63 -0
  209. lmcache/v1/internal_api_server/controller/__init__.py +1 -0
  210. lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
  211. lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
  212. lmcache/v1/internal_api_server/utils.py +43 -0
  213. lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
  214. lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
  215. lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
  216. lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
  217. lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
  218. lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
  219. lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
  220. lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
  221. lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
  222. lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
  223. lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
  224. lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
  225. lmcache/v1/kv_layer_groups.py +267 -0
  226. lmcache/v1/lazy_memory_allocator.py +284 -0
  227. lmcache/v1/lookup_client/__init__.py +25 -0
  228. lmcache/v1/lookup_client/abstract_client.py +77 -0
  229. lmcache/v1/lookup_client/async_lookup_message.py +50 -0
  230. lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
  231. lmcache/v1/lookup_client/factory.py +251 -0
  232. lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
  233. lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
  234. lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
  235. lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
  236. lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
  237. lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
  238. lmcache/v1/lookup_client/record_strategies/base.py +327 -0
  239. lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
  240. lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
  241. lmcache/v1/manager.py +539 -0
  242. lmcache/v1/memory_management.py +2619 -0
  243. lmcache/v1/metadata.py +114 -0
  244. lmcache/v1/mp_observability/AGENTS.override.md +21 -0
  245. lmcache/v1/mp_observability/README.md +204 -0
  246. lmcache/v1/mp_observability/config.py +340 -0
  247. lmcache/v1/mp_observability/event.py +100 -0
  248. lmcache/v1/mp_observability/event_bus.py +313 -0
  249. lmcache/v1/mp_observability/otel_init.py +145 -0
  250. lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
  251. lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
  252. lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
  253. lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
  254. lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
  255. lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
  256. lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
  257. lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
  258. lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
  259. lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
  260. lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
  261. lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
  262. lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
  263. lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
  264. lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
  265. lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
  266. lmcache/v1/mp_observability/trace/__init__.py +50 -0
  267. lmcache/v1/mp_observability/trace/codecs.py +255 -0
  268. lmcache/v1/mp_observability/trace/decorator.py +147 -0
  269. lmcache/v1/mp_observability/trace/format.py +132 -0
  270. lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
  271. lmcache/v1/mp_observability/trace/reader.py +167 -0
  272. lmcache/v1/mp_observability/trace/recorder.py +300 -0
  273. lmcache/v1/multiprocess/__init__.py +0 -0
  274. lmcache/v1/multiprocess/affinity_pool.py +102 -0
  275. lmcache/v1/multiprocess/blend_server_v2.py +891 -0
  276. lmcache/v1/multiprocess/config.py +253 -0
  277. lmcache/v1/multiprocess/custom_types.py +281 -0
  278. lmcache/v1/multiprocess/futures.py +194 -0
  279. lmcache/v1/multiprocess/gpu_context.py +511 -0
  280. lmcache/v1/multiprocess/http_server.py +235 -0
  281. lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
  282. lmcache/v1/multiprocess/mq.py +732 -0
  283. lmcache/v1/multiprocess/protocol.py +86 -0
  284. lmcache/v1/multiprocess/protocols/README.md +213 -0
  285. lmcache/v1/multiprocess/protocols/__init__.py +127 -0
  286. lmcache/v1/multiprocess/protocols/base.py +89 -0
  287. lmcache/v1/multiprocess/protocols/blend.py +109 -0
  288. lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
  289. lmcache/v1/multiprocess/protocols/controller.py +53 -0
  290. lmcache/v1/multiprocess/protocols/debug.py +34 -0
  291. lmcache/v1/multiprocess/protocols/engine.py +146 -0
  292. lmcache/v1/multiprocess/protocols/observability.py +39 -0
  293. lmcache/v1/multiprocess/server.py +1134 -0
  294. lmcache/v1/multiprocess/session.py +190 -0
  295. lmcache/v1/multiprocess/token_hasher.py +441 -0
  296. lmcache/v1/offload_server/__init__.py +17 -0
  297. lmcache/v1/offload_server/abstract_server.py +37 -0
  298. lmcache/v1/offload_server/message.py +30 -0
  299. lmcache/v1/offload_server/zmq_server.py +122 -0
  300. lmcache/v1/periodic_thread.py +579 -0
  301. lmcache/v1/pin_monitor.py +246 -0
  302. lmcache/v1/plugin/__init__.py +0 -0
  303. lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
  304. lmcache/v1/protocol.py +317 -0
  305. lmcache/v1/rpc/__init__.py +17 -0
  306. lmcache/v1/rpc/transport.py +105 -0
  307. lmcache/v1/rpc/zmq_transport.py +213 -0
  308. lmcache/v1/rpc_utils.py +165 -0
  309. lmcache/v1/server/__init__.py +2 -0
  310. lmcache/v1/server/__main__.py +170 -0
  311. lmcache/v1/server/storage_backend/__init__.py +21 -0
  312. lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
  313. lmcache/v1/server/storage_backend/local_backend.py +75 -0
  314. lmcache/v1/server/utils.py +21 -0
  315. lmcache/v1/standalone/__init__.py +1 -0
  316. lmcache/v1/standalone/__main__.py +583 -0
  317. lmcache/v1/standalone/manager.py +80 -0
  318. lmcache/v1/standalone/standalone_service_factory.py +86 -0
  319. lmcache/v1/storage_backend/__init__.py +313 -0
  320. lmcache/v1/storage_backend/abstract_backend.py +445 -0
  321. lmcache/v1/storage_backend/audit_backend.py +233 -0
  322. lmcache/v1/storage_backend/batched_message_sender.py +222 -0
  323. lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
  324. lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
  325. lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
  326. lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
  327. lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
  328. lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
  329. lmcache/v1/storage_backend/connector/__init__.py +443 -0
  330. lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
  331. lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
  332. lmcache/v1/storage_backend/connector/base_connector.py +379 -0
  333. lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
  334. lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
  335. lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
  336. lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
  337. lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
  338. lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
  339. lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
  340. lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
  341. lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
  342. lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
  343. lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
  344. lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
  345. lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
  346. lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
  347. lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
  348. lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
  349. lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
  350. lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
  351. lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
  352. lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
  353. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
  354. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
  355. lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
  356. lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
  357. lmcache/v1/storage_backend/gds_backend.py +1199 -0
  358. lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
  359. lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
  360. lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
  361. lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
  362. lmcache/v1/storage_backend/local_disk_backend.py +656 -0
  363. lmcache/v1/storage_backend/maru_backend.py +734 -0
  364. lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
  365. lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
  366. lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
  367. lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
  368. lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
  369. lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
  370. lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
  371. lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
  372. lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
  373. lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
  374. lmcache/v1/storage_backend/p2p_backend.py +788 -0
  375. lmcache/v1/storage_backend/path_sharder.py +117 -0
  376. lmcache/v1/storage_backend/pd_backend.py +646 -0
  377. lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
  378. lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
  379. lmcache/v1/storage_backend/remote_backend.py +624 -0
  380. lmcache/v1/storage_backend/resp_client.py +227 -0
  381. lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
  382. lmcache/v1/storage_backend/storage_manager.py +1352 -0
  383. lmcache/v1/system_detection.py +110 -0
  384. lmcache/v1/token_database.py +551 -0
  385. lmcache/v1/transfer_channel/__init__.py +83 -0
  386. lmcache/v1/transfer_channel/abstract.py +285 -0
  387. lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
  388. lmcache/v1/transfer_channel/nixl_channel.py +639 -0
  389. lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
  390. lmcache/v1/transfer_channel/transfer_utils.py +63 -0
  391. lmcache/v1/utils/__init__.py +1 -0
  392. lmcache/v1/utils/bloom_filter.py +109 -0
  393. lmcache/v1/utils/cache_utils.py +125 -0
  394. lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
  395. lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
  396. lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
  397. lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
  398. lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
  399. lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,392 @@
1
+ # Cache Simulator
2
+
3
+ The cache simulator replays recorded LMCache lookup events to measure **token cache hit rate** — the fraction of input tokens served from the KV cache rather than recomputed. It answers questions like:
4
+
5
+ - What hit rate can I expect for my workload at a given cache size?
6
+ - How much cache memory do I need to reach 80% / 90% hit rate?
7
+ - Which requests benefit most from caching?
8
+
9
+ ## Table of Contents
10
+
11
+ - [How it Works](#how-it-works)
12
+ - [Quick Start](#quick-start)
13
+ - [Step 1: Enable Lookup Hash Logging](#step-1-enable-lookup-hash-logging)
14
+ - [Step 2: Run the Simulator](#step-2-run-the-simulator)
15
+ - [Step 3: Plot Hit Rate vs Capacity](#step-3-plot-hit-rate-vs-capacity)
16
+ - [Understanding the Output](#understanding-the-output)
17
+ - [CLI Reference](#cli-reference)
18
+ - [For Developers](#for-developers)
19
+
20
+ ---
21
+
22
+ ## How it Works
23
+
24
+ LMCache splits each request's KV cache into fixed-size **chunks** and identifies them by hash. The simulator replays those hashes through a simulated LRU cache to predict hit rate without running the actual model.
25
+
26
+ ### Token hit rate
27
+
28
+ The primary metric is **token hit rate**, not chunk hit rate:
29
+
30
+ ```
31
+ token_hit_rate = total_hit_tokens / total_tokens
32
+ ```
33
+
34
+ Where:
35
+
36
+ - `total_tokens` is the sum of `seq_len` across all requests — this includes **tail tokens** (the partial chunk at the end of each sequence that does not fill a complete chunk). Tail tokens are always a miss because LMCache only caches complete chunks.
37
+ - `total_hit_tokens` is the number of tokens covered by a **continuous hit prefix** at the start of each request: if the first *k* chunks are all cache hits, that contributes `k × chunk_size` hit tokens.
38
+
39
+ This is the same definition used by the LMCache server itself.
40
+
41
+ ### Prefix caching semantics
42
+
43
+ The simulator enforces the same prefix rule as LMCache: a chunk is only counted as a hit if it **and every chunk before it** in the request are cached. The first miss breaks the prefix — subsequent chunks are not counted as hits even if they happen to be in cache.
44
+
45
+ ### Cache model
46
+
47
+ - **Eviction policy:** LRU (Least Recently Used)
48
+ - **Cache key:** chunk hash (hex string, opaque)
49
+ - **Capacity unit:** bytes of KV cache memory
50
+
51
+ ---
52
+
53
+ ## Quick Start
54
+
55
+ ```bash
56
+ # 1. Collect logs from a live server (see Step 1 below)
57
+ lmcache server --lookup-hash-log-dir /data/lmcache/lookup_hashes ...
58
+
59
+ # 2. Simulate at a fixed capacity — prints text report and saves a PNG chart
60
+ lmcache tool cache-simulator simulate \
61
+ -i /data/lmcache/lookup_hashes \
62
+ --cache-capacity-gib 64 \
63
+ -o stats.png
64
+
65
+ # 3. Sweep across capacities to find the right cache size
66
+ lmcache tool cache-simulator sweep \
67
+ -i /data/lmcache/lookup_hashes \
68
+ --min-capacity-gib 1 \
69
+ --max-capacity-gib 512 \
70
+ --points 30 \
71
+ -o sweep.png
72
+ ```
73
+
74
+ ---
75
+
76
+ ## Step 1: Enable Lookup Hash Logging
77
+
78
+ Start the LMCache server with `--lookup-hash-log-dir` pointing to a writable directory:
79
+
80
+ ```bash
81
+ lmcache server \
82
+ --host 0.0.0.0 \
83
+ --port 8080 \
84
+ --chunk-size 256 \
85
+ --lookup-hash-log-dir /data/lmcache/lookup_hashes \
86
+ --lookup-hash-log-rotation-interval 21600 \
87
+ --lookup-hash-log-rotation-max-size 104857600 \
88
+ --lookup-hash-log-max-files 100
89
+ ```
90
+
91
+ The server will write rotating JSONL files to that directory. Each line is one request:
92
+
93
+ ```json
94
+ {
95
+ "timestamp": 1711929600.123,
96
+ "request_id": "req-001",
97
+ "model_name": "DeepSeek-V3",
98
+ "chunk_size": 256,
99
+ "seq_len": 8192,
100
+ "dtypes": ["float8_e4m3fn"],
101
+ "shapes": [[32, 256, 128]],
102
+ "chunk_hashes": ["0xabcd1234...", "0xef567890...", ...]
103
+ }
104
+ ```
105
+
106
+ | Field | Description |
107
+ |---|---|
108
+ | `timestamp` | Unix timestamp of the lookup |
109
+ | `request_id` | Unique request identifier |
110
+ | `model_name` | Model being served |
111
+ | `chunk_size` | Tokens per chunk |
112
+ | `seq_len` | Total input tokens (including tail) |
113
+ | `dtypes` | KV tensor data types |
114
+ | `shapes` | KV tensor shapes (used to compute bytes/chunk) |
115
+ | `chunk_hashes` | Ordered list of full-chunk hashes (hex strings) |
116
+
117
+ Note: `chunk_hashes` only covers **complete** chunks. The tail tokens (`seq_len mod chunk_size`) are not represented — they are implicitly always a miss.
118
+
119
+ ---
120
+
121
+ ## Step 2: Run the Simulator
122
+
123
+ ```bash
124
+ lmcache tool cache-simulator simulate \
125
+ -i /data/lmcache/lookup_hashes \
126
+ --cache-capacity-gib 64 \
127
+ -o stats.png
128
+ ```
129
+
130
+ This prints a full text report to the terminal **and** saves a 7-panel statistics chart to `stats.png`. The `kv_bytes_per_chunk` value is auto-detected from the `shapes` and `dtypes` fields of the first event. You can override it:
131
+
132
+ ```bash
133
+ lmcache tool cache-simulator simulate \
134
+ -i /data/lmcache/lookup_hashes \
135
+ --cache-capacity-gib 64 \
136
+ --kv-bytes-per-chunk 20971520 \
137
+ -o stats.png
138
+ ```
139
+
140
+ To analyse only one model when the logs contain multiple:
141
+
142
+ ```bash
143
+ lmcache tool cache-simulator simulate \
144
+ -i /data/lmcache/lookup_hashes \
145
+ --cache-capacity-gib 64 \
146
+ --model DeepSeek-V3 \
147
+ -o stats.png
148
+ ```
149
+
150
+ ### Example text output
151
+
152
+ ```
153
+ ============================================================
154
+ Aggregate
155
+ ============================================================
156
+ Requests processed : 9,161
157
+ Total tokens : 449,114,449
158
+ Hit tokens : 268,330,752
159
+ Miss tokens : 180,783,697
160
+ Token hit rate : 59.75%
161
+ Cache capacity : 64.00 GiB (3,276 chunks × 20,971,520 bytes/chunk)
162
+ Cache occupancy : 3,276 / 3,276 chunks
163
+
164
+ ============================================================
165
+ Stat 1 — Per-request token hit rate distribution
166
+ ============================================================
167
+ Requests with 0% hit rate : 38 (0.4%)
168
+ Requests with 100% hit rate : 3 (0.0%)
169
+ Mean : 80.82%
170
+ p50 : 90.59%
171
+ p90 : 98.85%
172
+ p99 : 99.93%
173
+ ...
174
+ ```
175
+
176
+ ### Example statistics chart
177
+
178
+ ![Simulation statistics chart](docs/simulate_example.png)
179
+
180
+ ---
181
+
182
+ ## Step 3: Plot Hit Rate vs Capacity
183
+
184
+ The plot tool sweeps across a log-spaced range of cache sizes and shows how hit rate changes with capacity — the key curve for capacity planning.
185
+
186
+ ```bash
187
+ lmcache tool cache-simulator sweep \
188
+ -i /data/lmcache/lookup_hashes \
189
+ --min-capacity-gib 1 \
190
+ --max-capacity-gib 512 \
191
+ --points 30 \
192
+ -o sweep.png
193
+ ```
194
+
195
+ This prints a table and saves a PNG:
196
+
197
+ ```
198
+ Capacity (GiB) Hit rate
199
+ --------------------------------
200
+ 1.000 29.18%
201
+ 2.151 36.93%
202
+ 9.257 51.44%
203
+ 39.830 63.51%
204
+ 118.993 68.96%
205
+ 512.000 78.58%
206
+ ```
207
+
208
+ The x-axis is in GiB (log scale by default). Use `--linear` for a linear scale.
209
+
210
+ ### Example capacity sweep chart
211
+
212
+ ![Hit rate vs capacity chart](docs/sweep_example.png)
213
+
214
+ ---
215
+
216
+ ## Understanding the Output
217
+
218
+ ### Text report statistics
219
+
220
+ | Stat | What it measures |
221
+ |---|---|
222
+ | **Aggregate** | Overall token hit rate, capacity utilisation, eviction count |
223
+ | **Stat 1** | Per-request hit rate distribution (mean, percentiles, 0%/100% counts) |
224
+ | **Stat 2** | Hit prefix length per request in chunks (how far the prefix match extends) |
225
+ | **Stat 3** | Chunk reuse count distribution (how many times each unique chunk was hit) |
226
+ | **Stat 4** | Rolling cumulative hit rate over time (does the cache warm up quickly?) |
227
+ | **Stat 5** | Total evictions (non-zero means the cache was full and chunks were displaced) |
228
+ | **Stat 6** | Global span distribution — chunks processed between when a chunk was last stored and when it was hit again (measures temporal locality) |
229
+ | **Stat 7** | Cache position at hit time (0 = MRU, max = LRU; low values mean recently-used chunks are being hit, high values indicate stale hits that nearly got evicted) |
230
+
231
+ ### Chart panels (stats.png)
232
+
233
+ The saved PNG contains the same seven statistics as visual histograms:
234
+
235
+ | Panel | Chart |
236
+ |---|---|
237
+ | **1** | Per-request token hit rate histogram + hit/miss request pie inset |
238
+ | **1b** | Same zoomed into the 97–100% range |
239
+ | **2** | Hit prefix length per request + hit/miss token pie inset |
240
+ | **3** | Chunk reuse count histogram (x-axis capped at 100) |
241
+ | **4** | Rolling cumulative token hit rate over time |
242
+ | **5** | Input length per request (tokens) |
243
+ | **6** | Global span distribution |
244
+ | **7** | Cache position at hit time |
245
+
246
+ ---
247
+
248
+ ## CLI Reference
249
+
250
+ ### `simulate` — single-run report and chart
251
+
252
+ ```
253
+ lmcache tool cache-simulator simulate [OPTIONS]
254
+ ```
255
+
256
+ | Option | Default | Description |
257
+ |---|---|---|
258
+ | `-i / --input PATH [PATH ...]` | required | JSONL files or directories (directories are globbed for `lookup_hashes_*.jsonl`) |
259
+ | `--cache-capacity-gib GiB` | required | Cache size in gibibytes |
260
+ | `-o / --output FILE` | `cache_stats.png` | Output image path |
261
+ | `-n / --max-samples N` | all | Truncate to N events after sorting by timestamp |
262
+ | `--model NAME` | all | Filter by `model_name` (exact match) |
263
+ | `--kv-bytes-per-chunk BYTES` | auto | KV bytes per chunk; auto-computed from first event if omitted |
264
+
265
+ ### `sweep` — capacity sweep and plot
266
+
267
+ ```
268
+ lmcache tool cache-simulator sweep [OPTIONS]
269
+ ```
270
+
271
+ | Option | Default | Description |
272
+ |---|---|---|
273
+ | `-i / --input PATH [PATH ...]` | required | JSONL files or directories |
274
+ | `--min-capacity-gib GiB` | `0.5` | Lower bound of capacity sweep |
275
+ | `--max-capacity-gib GiB` | `500` | Upper bound of capacity sweep |
276
+ | `--points N` | `30` | Number of log-spaced capacity samples |
277
+ | `--linear` | off | Use linear x-axis instead of log scale |
278
+ | `-o / --output FILE` | `hit_rate_vs_capacity.png` | Output image path |
279
+ | `-n / --max-samples N` | all | Truncate events |
280
+ | `--model NAME` | all | Filter by model name |
281
+ | `--kv-bytes-per-chunk BYTES` | auto | KV bytes per chunk |
282
+
283
+ ### `gen-dataset` — generate vllm bench serve dataset
284
+
285
+ ```
286
+ lmcache tool cache-simulator gen-dataset [OPTIONS]
287
+ ```
288
+
289
+ | Option | Default | Description |
290
+ |---|---|---|
291
+ | `-i / --input PATH [PATH ...]` | required | JSONL files or directories |
292
+ | `--tokenizer PATH` | required | HuggingFace tokenizer path or name |
293
+ | `--output-len N` | `128` | `output_tokens` per request |
294
+ | `-o / --output FILE` | `bench_dataset.jsonl` | Output JSONL path |
295
+ | `-n / --max-samples N` | all | Truncate events |
296
+ | `--model NAME` | all | Filter by model name |
297
+
298
+ ### How token generation works
299
+
300
+ 1. A *safe vocabulary* is built from the tokenizer: token IDs that decode to printable text and round-trip stably through `encode(decode([id])) == [id]`. Tokens with a leading space are preferred to prevent BPE merges at chunk boundaries.
301
+ 2. Each unique chunk hash is mapped deterministically to `chunk_size` token IDs by seeding a PRNG with `SHA-256(hash)`. The same hash always produces the same tokens.
302
+ 3. Tail tokens (`seq_len mod chunk_size`) use a per-request seed so they are never accidentally shared across requests (matching LMCache's behaviour of never caching partial chunks).
303
+ 4. The full token list is decoded to text and written as the `"prompt"` field. `"output_tokens"` is set to `--output-len`.
304
+
305
+ ---
306
+
307
+ ## For Developers
308
+
309
+ ### Package layout
310
+
311
+ ```
312
+ lmcache/tools/cache_simulator/
313
+ __init__.py — package marker
314
+ lru_cache.py — LRUCache and LRUCacheFast implementations
315
+ simulator.py — event loading, simulation engine, text report, chart, CLI
316
+ plot_hit_rate.py — capacity sweep and matplotlib plot
317
+ gen_bench_dataset.py — lookup-hash → vllm bench serve dataset converter
318
+
319
+ lmcache/cli/commands/tool/
320
+ __init__.py — ToolCommand dispatcher (lmcache tool ...)
321
+ cache_simulator.py — wires cache-simulator into the lmcache CLI
322
+ ```
323
+
324
+ ### CLI integration
325
+
326
+ The same functionality is also accessible via the `lmcache` CLI (see
327
+ [Quick Start](#quick-start)). The CLI entry point lives in
328
+ `lmcache/cli/commands/tool/cache_simulator.py`, which calls
329
+ `add_simulate_arguments` / `run_simulate` from `simulator.py` and
330
+ `add_sweep_arguments` / `run_sweep` from `plot_hit_rate.py`.
331
+
332
+ **When adding or removing a CLI flag**, update only the relevant
333
+ `add_*_arguments` function in `simulator.py` or `plot_hit_rate.py` — the
334
+ `lmcache tool` command picks up the change automatically.
335
+
336
+ **When adding a new action** (e.g. `lmcache tool cache-simulator compare`),
337
+ register it in `lmcache/cli/commands/tool/cache_simulator.py` alongside the
338
+ existing `simulate` and `sweep` actions.
339
+
340
+ ### `lru_cache.py`
341
+
342
+ Two implementations are provided to trade off speed against feature richness:
343
+
344
+ **`LRUCacheFast`** — O(1) all operations, backed by a single `OrderedDict`. Supports `contains`, `access`, `insert`, and `eviction_count`. Used during capacity sweeps where only hit/miss counts are needed.
345
+
346
+ **`LRUCache`** — O(log n) operations, backed by a `dict` (for O(1) lookup) plus a `SortedList` (for O(log n) rank queries). Adds `position(key)` which returns the LRU rank of a key (0 = MRU, len−1 = LRU). Used in the single-run report for Stat 7.
347
+
348
+ Both take `capacity` in **number of chunks**. The byte-to-chunk conversion is done by the caller in `simulator.py`.
349
+
350
+ ### `simulator.py`
351
+
352
+ Key public functions:
353
+
354
+ **`compute_kv_bytes_per_chunk(event)`** — derives the byte footprint of one cached chunk from a record's `shapes` and `dtypes` fields. Uses a hard-coded `dtype → bytes` table; unknown dtypes warn and contribute 0 bytes.
355
+
356
+ **`load_lookup_events(paths, model, max_samples)`** — loads and merges events from one or more JSONL files or directories, sorts by `timestamp` ascending, applies optional model filter and sample cap.
357
+
358
+ **`simulate(events, cache_capacity_bytes, kv_bytes_per_chunk, fast)`** — the core replay loop. Converts byte capacity to chunk count (`capacity_bytes // kv_bytes_per_chunk`), then walks events in order. For each event:
359
+
360
+ 1. Walk `chunk_hashes` from the front; count consecutive hits as `hit_prefix`.
361
+ 2. Accumulate `hit_prefix × chunk_size` hit tokens and `seq_len` total tokens.
362
+ 3. Update the cache: `access` for hit chunks, `insert` for miss chunks.
363
+ 4. In `fast=False` mode, additionally track per-request rates, reuse counts, span distribution, and cache positions.
364
+
365
+ Returns a dict with all statistics. In `fast=True` mode the per-request and chunk-level lists are empty, making capacity sweeps significantly faster.
366
+
367
+ **`print_statistics(results)`** — formats and prints the text report to stdout.
368
+
369
+ **`plot_statistics(results, events, output)`** — renders the 7-panel chart and saves it to `output`.
370
+
371
+ ### Adding a new statistic
372
+
373
+ 1. Add accumulator variables in `simulate()` before the main loop.
374
+ 2. Populate them inside the `if not fast:` block.
375
+ 3. Include them in the returned dict.
376
+ 4. Add a print block in `print_statistics()`.
377
+ 5. Add a subplot in `plot_statistics()`.
378
+
379
+ ### Feeding custom data
380
+
381
+ `simulate()` accepts any list of dicts with the fields `chunk_hashes` (list of strings), `seq_len` (int), and `chunk_size` (int). The other fields (`timestamp`, `model_name`, etc.) are only used by `load_lookup_events`. You can construct events programmatically for unit tests or synthetic benchmarks:
382
+
383
+ ```python
384
+ from lmcache.tools.cache_simulator.simulator import simulate
385
+
386
+ events = [
387
+ {"chunk_hashes": ["0xaa", "0xbb"], "seq_len": 600, "chunk_size": 256},
388
+ {"chunk_hashes": ["0xaa", "0xbb"], "seq_len": 600, "chunk_size": 256},
389
+ ]
390
+ result = simulate(events, cache_capacity_bytes=10 * 1024**3, kv_bytes_per_chunk=20971520)
391
+ print(f"Token hit rate: {result['token_hit_rate']:.2%}")
392
+ ```
@@ -0,0 +1 @@
1
+ # SPDX-License-Identifier: Apache-2.0