lmcache-cli 0.4.5.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (399) hide show
  1. lmcache/__init__.py +84 -0
  2. lmcache/_version.py +24 -0
  3. lmcache/cli/__init__.py +1 -0
  4. lmcache/cli/commands/__init__.py +34 -0
  5. lmcache/cli/commands/base.py +157 -0
  6. lmcache/cli/commands/bench/__init__.py +557 -0
  7. lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
  8. lmcache/cli/commands/bench/engine_bench/config.py +245 -0
  9. lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
  10. lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
  11. lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
  12. lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
  13. lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
  14. lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
  15. lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
  16. lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
  17. lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
  18. lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
  19. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
  20. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
  21. lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
  22. lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
  23. lmcache/cli/commands/describe.py +310 -0
  24. lmcache/cli/commands/kvcache.py +133 -0
  25. lmcache/cli/commands/mock.py +75 -0
  26. lmcache/cli/commands/ping.py +113 -0
  27. lmcache/cli/commands/query/__init__.py +155 -0
  28. lmcache/cli/commands/query/prompt.py +134 -0
  29. lmcache/cli/commands/query/request.py +357 -0
  30. lmcache/cli/commands/server.py +99 -0
  31. lmcache/cli/commands/tool/__init__.py +63 -0
  32. lmcache/cli/commands/tool/cache_simulator.py +113 -0
  33. lmcache/cli/commands/trace/__init__.py +505 -0
  34. lmcache/cli/commands/trace/dispatch.py +249 -0
  35. lmcache/cli/commands/trace/driver.py +372 -0
  36. lmcache/cli/commands/trace/stats.py +289 -0
  37. lmcache/cli/documents/lmcache.txt +11 -0
  38. lmcache/cli/main.py +42 -0
  39. lmcache/cli/metrics/__init__.py +29 -0
  40. lmcache/cli/metrics/formatter.py +171 -0
  41. lmcache/cli/metrics/handler.py +94 -0
  42. lmcache/cli/metrics/metrics.py +161 -0
  43. lmcache/cli/metrics/section.py +77 -0
  44. lmcache/connections.py +173 -0
  45. lmcache/integration/__init__.py +2 -0
  46. lmcache/integration/base_service_factory.py +165 -0
  47. lmcache/integration/request_telemetry/__init__.py +1 -0
  48. lmcache/integration/request_telemetry/base.py +51 -0
  49. lmcache/integration/request_telemetry/factory.py +113 -0
  50. lmcache/integration/request_telemetry/fastapi.py +109 -0
  51. lmcache/integration/request_telemetry/noop.py +35 -0
  52. lmcache/integration/sglang/__init__.py +2 -0
  53. lmcache/integration/sglang/sglang_adapter.py +326 -0
  54. lmcache/integration/sglang/utils.py +39 -0
  55. lmcache/integration/vllm/__init__.py +1 -0
  56. lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
  57. lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
  58. lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
  59. lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
  60. lmcache/integration/vllm/utils.py +433 -0
  61. lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
  62. lmcache/integration/vllm/vllm_service_factory.py +339 -0
  63. lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
  64. lmcache/logging.py +107 -0
  65. lmcache/native_storage_ops.pyi +230 -0
  66. lmcache/non_cuda_equivalents.py +1424 -0
  67. lmcache/observability.py +1958 -0
  68. lmcache/storage_backend/serde/__init__.py +1 -0
  69. lmcache/storage_backend/serde/cachegen_basics.py +210 -0
  70. lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
  71. lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
  72. lmcache/storage_backend/serde/serde.py +75 -0
  73. lmcache/tools/__init__.py +1 -0
  74. lmcache/tools/cache_simulator/README.md +392 -0
  75. lmcache/tools/cache_simulator/__init__.py +1 -0
  76. lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
  77. lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
  78. lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
  79. lmcache/tools/cache_simulator/lru_cache.py +124 -0
  80. lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
  81. lmcache/tools/cache_simulator/simulator.py +795 -0
  82. lmcache/tools/controller_benchmark/README.md +161 -0
  83. lmcache/tools/controller_benchmark/__init__.py +1 -0
  84. lmcache/tools/controller_benchmark/__main__.py +331 -0
  85. lmcache/tools/controller_benchmark/benchmark.py +660 -0
  86. lmcache/tools/controller_benchmark/config.py +44 -0
  87. lmcache/tools/controller_benchmark/constants.py +10 -0
  88. lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
  89. lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
  90. lmcache/tools/controller_benchmark/handlers/base.py +47 -0
  91. lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
  92. lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
  93. lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
  94. lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
  95. lmcache/tools/controller_benchmark/handlers/register.py +56 -0
  96. lmcache/tools/mp_status_viewer/__init__.py +1 -0
  97. lmcache/tools/mp_status_viewer/__main__.py +95 -0
  98. lmcache/usage_context.py +417 -0
  99. lmcache/utils.py +665 -0
  100. lmcache/v1/__init__.py +2 -0
  101. lmcache/v1/api_server/__init__.py +2 -0
  102. lmcache/v1/api_server/__main__.py +537 -0
  103. lmcache/v1/basic_check.py +112 -0
  104. lmcache/v1/cache_controller/__init__.py +9 -0
  105. lmcache/v1/cache_controller/commands/__init__.py +15 -0
  106. lmcache/v1/cache_controller/commands/base.py +35 -0
  107. lmcache/v1/cache_controller/commands/full_sync.py +49 -0
  108. lmcache/v1/cache_controller/config.py +176 -0
  109. lmcache/v1/cache_controller/controller_manager.py +535 -0
  110. lmcache/v1/cache_controller/controllers/__init__.py +11 -0
  111. lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
  112. lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
  113. lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
  114. lmcache/v1/cache_controller/executor.py +463 -0
  115. lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
  116. lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
  117. lmcache/v1/cache_controller/frontend/static/index.html +234 -0
  118. lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
  119. lmcache/v1/cache_controller/full_sync_sender.py +475 -0
  120. lmcache/v1/cache_controller/locks.py +149 -0
  121. lmcache/v1/cache_controller/message.py +828 -0
  122. lmcache/v1/cache_controller/observability.py +208 -0
  123. lmcache/v1/cache_controller/utils.py +679 -0
  124. lmcache/v1/cache_controller/worker.py +665 -0
  125. lmcache/v1/cache_engine.py +2058 -0
  126. lmcache/v1/cache_interface.py +19 -0
  127. lmcache/v1/check/__init__.py +74 -0
  128. lmcache/v1/check/check_mode_gen.py +86 -0
  129. lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
  130. lmcache/v1/check/check_mode_test_remote.py +155 -0
  131. lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
  132. lmcache/v1/check/utils.py +571 -0
  133. lmcache/v1/compute/__init__.py +2 -0
  134. lmcache/v1/compute/attention/__init__.py +0 -0
  135. lmcache/v1/compute/attention/abstract.py +39 -0
  136. lmcache/v1/compute/attention/flash_attn.py +129 -0
  137. lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
  138. lmcache/v1/compute/attention/metadata.py +85 -0
  139. lmcache/v1/compute/attention/utils.py +14 -0
  140. lmcache/v1/compute/blend/__init__.py +7 -0
  141. lmcache/v1/compute/blend/blender.py +168 -0
  142. lmcache/v1/compute/blend/metadata.py +34 -0
  143. lmcache/v1/compute/blend/utils.py +63 -0
  144. lmcache/v1/compute/models/__init__.py +0 -0
  145. lmcache/v1/compute/models/base.py +141 -0
  146. lmcache/v1/compute/models/llama.py +9 -0
  147. lmcache/v1/compute/models/qwen3.py +24 -0
  148. lmcache/v1/compute/models/utils.py +68 -0
  149. lmcache/v1/compute/positional_encoding.py +199 -0
  150. lmcache/v1/config.py +848 -0
  151. lmcache/v1/config_base.py +848 -0
  152. lmcache/v1/distributed/api.py +248 -0
  153. lmcache/v1/distributed/config.py +321 -0
  154. lmcache/v1/distributed/error.py +64 -0
  155. lmcache/v1/distributed/eviction.py +192 -0
  156. lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
  157. lmcache/v1/distributed/eviction_policy/factory.py +27 -0
  158. lmcache/v1/distributed/eviction_policy/lru.py +244 -0
  159. lmcache/v1/distributed/eviction_policy/noop.py +50 -0
  160. lmcache/v1/distributed/internal_api.py +170 -0
  161. lmcache/v1/distributed/l1_manager.py +835 -0
  162. lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
  163. lmcache/v1/distributed/l2_adapters/base.py +360 -0
  164. lmcache/v1/distributed/l2_adapters/config.py +385 -0
  165. lmcache/v1/distributed/l2_adapters/factory.py +205 -0
  166. lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
  167. lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
  168. lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
  169. lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
  170. lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
  171. lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
  172. lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
  173. lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
  174. lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
  175. lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
  176. lmcache/v1/distributed/memory_manager.py +179 -0
  177. lmcache/v1/distributed/storage_controller.py +39 -0
  178. lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
  179. lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
  180. lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
  181. lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
  182. lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
  183. lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
  184. lmcache/v1/distributed/storage_manager.py +532 -0
  185. lmcache/v1/event_manager.py +145 -0
  186. lmcache/v1/exceptions/__init__.py +16 -0
  187. lmcache/v1/gpu_connector/__init__.py +126 -0
  188. lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
  189. lmcache/v1/gpu_connector/gpu_ops.py +85 -0
  190. lmcache/v1/gpu_connector/hpu_connector.py +326 -0
  191. lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
  192. lmcache/v1/gpu_connector/utils.py +890 -0
  193. lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
  194. lmcache/v1/health_monitor/__init__.py +1 -0
  195. lmcache/v1/health_monitor/base.py +587 -0
  196. lmcache/v1/health_monitor/checks/__init__.py +1 -0
  197. lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
  198. lmcache/v1/health_monitor/constants.py +36 -0
  199. lmcache/v1/internal_api_server/__init__.py +0 -0
  200. lmcache/v1/internal_api_server/api_registry.py +59 -0
  201. lmcache/v1/internal_api_server/api_server.py +120 -0
  202. lmcache/v1/internal_api_server/common/__init__.py +1 -0
  203. lmcache/v1/internal_api_server/common/env_api.py +22 -0
  204. lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
  205. lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
  206. lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
  207. lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
  208. lmcache/v1/internal_api_server/common/thread_api.py +63 -0
  209. lmcache/v1/internal_api_server/controller/__init__.py +1 -0
  210. lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
  211. lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
  212. lmcache/v1/internal_api_server/utils.py +43 -0
  213. lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
  214. lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
  215. lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
  216. lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
  217. lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
  218. lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
  219. lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
  220. lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
  221. lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
  222. lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
  223. lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
  224. lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
  225. lmcache/v1/kv_layer_groups.py +267 -0
  226. lmcache/v1/lazy_memory_allocator.py +284 -0
  227. lmcache/v1/lookup_client/__init__.py +25 -0
  228. lmcache/v1/lookup_client/abstract_client.py +77 -0
  229. lmcache/v1/lookup_client/async_lookup_message.py +50 -0
  230. lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
  231. lmcache/v1/lookup_client/factory.py +251 -0
  232. lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
  233. lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
  234. lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
  235. lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
  236. lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
  237. lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
  238. lmcache/v1/lookup_client/record_strategies/base.py +327 -0
  239. lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
  240. lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
  241. lmcache/v1/manager.py +539 -0
  242. lmcache/v1/memory_management.py +2619 -0
  243. lmcache/v1/metadata.py +114 -0
  244. lmcache/v1/mp_observability/AGENTS.override.md +21 -0
  245. lmcache/v1/mp_observability/README.md +204 -0
  246. lmcache/v1/mp_observability/config.py +340 -0
  247. lmcache/v1/mp_observability/event.py +100 -0
  248. lmcache/v1/mp_observability/event_bus.py +313 -0
  249. lmcache/v1/mp_observability/otel_init.py +145 -0
  250. lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
  251. lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
  252. lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
  253. lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
  254. lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
  255. lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
  256. lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
  257. lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
  258. lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
  259. lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
  260. lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
  261. lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
  262. lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
  263. lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
  264. lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
  265. lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
  266. lmcache/v1/mp_observability/trace/__init__.py +50 -0
  267. lmcache/v1/mp_observability/trace/codecs.py +255 -0
  268. lmcache/v1/mp_observability/trace/decorator.py +147 -0
  269. lmcache/v1/mp_observability/trace/format.py +132 -0
  270. lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
  271. lmcache/v1/mp_observability/trace/reader.py +167 -0
  272. lmcache/v1/mp_observability/trace/recorder.py +300 -0
  273. lmcache/v1/multiprocess/__init__.py +0 -0
  274. lmcache/v1/multiprocess/affinity_pool.py +102 -0
  275. lmcache/v1/multiprocess/blend_server_v2.py +891 -0
  276. lmcache/v1/multiprocess/config.py +253 -0
  277. lmcache/v1/multiprocess/custom_types.py +281 -0
  278. lmcache/v1/multiprocess/futures.py +194 -0
  279. lmcache/v1/multiprocess/gpu_context.py +511 -0
  280. lmcache/v1/multiprocess/http_server.py +235 -0
  281. lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
  282. lmcache/v1/multiprocess/mq.py +732 -0
  283. lmcache/v1/multiprocess/protocol.py +86 -0
  284. lmcache/v1/multiprocess/protocols/README.md +213 -0
  285. lmcache/v1/multiprocess/protocols/__init__.py +127 -0
  286. lmcache/v1/multiprocess/protocols/base.py +89 -0
  287. lmcache/v1/multiprocess/protocols/blend.py +109 -0
  288. lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
  289. lmcache/v1/multiprocess/protocols/controller.py +53 -0
  290. lmcache/v1/multiprocess/protocols/debug.py +34 -0
  291. lmcache/v1/multiprocess/protocols/engine.py +146 -0
  292. lmcache/v1/multiprocess/protocols/observability.py +39 -0
  293. lmcache/v1/multiprocess/server.py +1134 -0
  294. lmcache/v1/multiprocess/session.py +190 -0
  295. lmcache/v1/multiprocess/token_hasher.py +441 -0
  296. lmcache/v1/offload_server/__init__.py +17 -0
  297. lmcache/v1/offload_server/abstract_server.py +37 -0
  298. lmcache/v1/offload_server/message.py +30 -0
  299. lmcache/v1/offload_server/zmq_server.py +122 -0
  300. lmcache/v1/periodic_thread.py +579 -0
  301. lmcache/v1/pin_monitor.py +246 -0
  302. lmcache/v1/plugin/__init__.py +0 -0
  303. lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
  304. lmcache/v1/protocol.py +317 -0
  305. lmcache/v1/rpc/__init__.py +17 -0
  306. lmcache/v1/rpc/transport.py +105 -0
  307. lmcache/v1/rpc/zmq_transport.py +213 -0
  308. lmcache/v1/rpc_utils.py +165 -0
  309. lmcache/v1/server/__init__.py +2 -0
  310. lmcache/v1/server/__main__.py +170 -0
  311. lmcache/v1/server/storage_backend/__init__.py +21 -0
  312. lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
  313. lmcache/v1/server/storage_backend/local_backend.py +75 -0
  314. lmcache/v1/server/utils.py +21 -0
  315. lmcache/v1/standalone/__init__.py +1 -0
  316. lmcache/v1/standalone/__main__.py +583 -0
  317. lmcache/v1/standalone/manager.py +80 -0
  318. lmcache/v1/standalone/standalone_service_factory.py +86 -0
  319. lmcache/v1/storage_backend/__init__.py +313 -0
  320. lmcache/v1/storage_backend/abstract_backend.py +445 -0
  321. lmcache/v1/storage_backend/audit_backend.py +233 -0
  322. lmcache/v1/storage_backend/batched_message_sender.py +222 -0
  323. lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
  324. lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
  325. lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
  326. lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
  327. lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
  328. lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
  329. lmcache/v1/storage_backend/connector/__init__.py +443 -0
  330. lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
  331. lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
  332. lmcache/v1/storage_backend/connector/base_connector.py +379 -0
  333. lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
  334. lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
  335. lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
  336. lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
  337. lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
  338. lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
  339. lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
  340. lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
  341. lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
  342. lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
  343. lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
  344. lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
  345. lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
  346. lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
  347. lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
  348. lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
  349. lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
  350. lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
  351. lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
  352. lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
  353. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
  354. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
  355. lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
  356. lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
  357. lmcache/v1/storage_backend/gds_backend.py +1199 -0
  358. lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
  359. lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
  360. lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
  361. lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
  362. lmcache/v1/storage_backend/local_disk_backend.py +656 -0
  363. lmcache/v1/storage_backend/maru_backend.py +734 -0
  364. lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
  365. lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
  366. lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
  367. lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
  368. lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
  369. lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
  370. lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
  371. lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
  372. lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
  373. lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
  374. lmcache/v1/storage_backend/p2p_backend.py +788 -0
  375. lmcache/v1/storage_backend/path_sharder.py +117 -0
  376. lmcache/v1/storage_backend/pd_backend.py +646 -0
  377. lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
  378. lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
  379. lmcache/v1/storage_backend/remote_backend.py +624 -0
  380. lmcache/v1/storage_backend/resp_client.py +227 -0
  381. lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
  382. lmcache/v1/storage_backend/storage_manager.py +1352 -0
  383. lmcache/v1/system_detection.py +110 -0
  384. lmcache/v1/token_database.py +551 -0
  385. lmcache/v1/transfer_channel/__init__.py +83 -0
  386. lmcache/v1/transfer_channel/abstract.py +285 -0
  387. lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
  388. lmcache/v1/transfer_channel/nixl_channel.py +639 -0
  389. lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
  390. lmcache/v1/transfer_channel/transfer_utils.py +63 -0
  391. lmcache/v1/utils/__init__.py +1 -0
  392. lmcache/v1/utils/bloom_filter.py +109 -0
  393. lmcache/v1/utils/cache_utils.py +125 -0
  394. lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
  395. lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
  396. lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
  397. lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
  398. lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
  399. lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,535 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Standard
3
+ from typing import Optional, Union
4
+ import asyncio
5
+ import json
6
+ import threading
7
+ import time
8
+
9
+ # Third Party
10
+ import msgspec
11
+ import zmq
12
+
13
+ # First Party
14
+ from lmcache.logging import init_logger
15
+ from lmcache.v1.cache_controller.controllers import KVController, RegistrationController
16
+ from lmcache.v1.cache_controller.executor import LMCacheClusterExecutor
17
+ from lmcache.v1.cache_controller.observability import (
18
+ PrometheusLogger,
19
+ SocketMetricsContext,
20
+ SocketType,
21
+ )
22
+ from lmcache.v1.rpc_utils import (
23
+ get_ip,
24
+ get_zmq_context,
25
+ get_zmq_socket,
26
+ )
27
+
28
+ from lmcache.v1.cache_controller.message import ( # isort: skip
29
+ BatchedKVOperationMsg,
30
+ BatchedP2PLookupMsg,
31
+ CheckFinishMsg,
32
+ ClearMsg,
33
+ CompressMsg,
34
+ DecompressMsg,
35
+ DeRegisterMsg,
36
+ ErrorMsg,
37
+ FullSyncBatchMsg,
38
+ FullSyncEndMsg,
39
+ FullSyncStartMsg,
40
+ FullSyncStatusMsg,
41
+ HealthMsg,
42
+ HeartbeatMsg,
43
+ LookupMsg,
44
+ MoveMsg,
45
+ Msg,
46
+ MsgBase,
47
+ OrchMsg,
48
+ OrchRetMsg,
49
+ PinMsg,
50
+ QueryInstMsg,
51
+ QueryWorkerInfoMsg,
52
+ RegisterMsg,
53
+ WorkerMsg,
54
+ WorkerReqMsg,
55
+ WorkerReqRetMsg,
56
+ )
57
+
58
+ logger = init_logger(__name__)
59
+
60
+ # TODO(Jiayi): Need to align the message types. For example,
61
+ # a controller should take in an control message and return
62
+ # a control message.
63
+
64
+
65
+ class LMCacheControllerManager:
66
+ def __init__(
67
+ self,
68
+ controller_urls: dict[str, str],
69
+ health_check_interval: int,
70
+ lmcache_worker_timeout: int,
71
+ full_sync_completion_threshold: float = 0.8,
72
+ full_sync_timeout_s: float = 300.0,
73
+ ):
74
+ # Initialize stats logger
75
+ prometheus_labels = {
76
+ "role": "controller",
77
+ }
78
+ PrometheusLogger.GetOrCreate(prometheus_labels)
79
+ self.zmq_context = get_zmq_context()
80
+ self.controller_urls = controller_urls
81
+ # TODO(Jiayi): We might need multiple sockets if there are more
82
+ # controllers. For now, we use a single socket to receive messages
83
+ # for all controllers.
84
+ # Similarly we might need more sockets to handle different control
85
+ # messages. For now, we use one socket to handle all control messages.
86
+
87
+ # TODO(Jiayi): Another thing is that we might need to decoupe the
88
+ # interactions among `handle_worker_message`, `handle_control_message`
89
+ # and `handle_orchestration_message`. For example, in
90
+ # `handle_orchestration_message`, we might need to call
91
+ # `issue_control_message`. This will make the system less concurrent.
92
+
93
+ # Micro controllers
94
+ self.controller_pull_socket = get_zmq_socket(
95
+ self.zmq_context,
96
+ self.controller_urls["pull"],
97
+ protocol="tcp",
98
+ role=zmq.PULL, # type: ignore[attr-defined]
99
+ bind_or_connect="bind",
100
+ )
101
+
102
+ if self.controller_urls["reply"] is not None:
103
+ self.controller_reply_socket = get_zmq_socket(
104
+ self.zmq_context,
105
+ self.controller_urls["reply"],
106
+ protocol="tcp",
107
+ role=zmq.ROUTER, # type: ignore[attr-defined]
108
+ bind_or_connect="bind",
109
+ )
110
+
111
+ # Dedicated heartbeat socket to avoid blocking from other requests
112
+ if self.controller_urls.get("heartbeat") is not None:
113
+ self.controller_heartbeat_socket = get_zmq_socket(
114
+ self.zmq_context,
115
+ self.controller_urls["heartbeat"],
116
+ protocol="tcp",
117
+ role=zmq.ROUTER, # type: ignore[attr-defined]
118
+ bind_or_connect="bind",
119
+ )
120
+ else:
121
+ self.controller_heartbeat_socket = None
122
+ self.reg_controller = RegistrationController()
123
+ self.kv_controller = KVController(
124
+ registry=self.reg_controller.registry,
125
+ full_sync_completion_threshold=full_sync_completion_threshold,
126
+ full_sync_timeout_s=full_sync_timeout_s,
127
+ )
128
+
129
+ # Cluster executor
130
+ self.cluster_executor = LMCacheClusterExecutor(
131
+ reg_controller=self.reg_controller,
132
+ )
133
+
134
+ # post initialization of controllers
135
+ self.kv_controller.post_init(
136
+ reg_controller=self.reg_controller,
137
+ cluster_executor=self.cluster_executor,
138
+ )
139
+ self.reg_controller.post_init(
140
+ kv_controller=self.kv_controller,
141
+ cluster_executor=self.cluster_executor,
142
+ )
143
+ self.health_check_interval = health_check_interval
144
+ self.lmcache_worker_timeout = lmcache_worker_timeout
145
+
146
+ if self.health_check_interval > 0:
147
+ logger.info(
148
+ "Start health check thread, interval: %s", self.health_check_interval
149
+ )
150
+ self.loop = asyncio.new_event_loop()
151
+ self.thread = threading.Thread(
152
+ target=self.loop.run_forever,
153
+ daemon=True,
154
+ name="controller-health-thread",
155
+ )
156
+ self.thread.start()
157
+ asyncio.run_coroutine_threadsafe(self.health_check(), self.loop)
158
+
159
+ # Setup socket message count metrics
160
+ self._setup_socket_metrics()
161
+
162
+ async def handle_worker_message(self, msg: WorkerMsg) -> None:
163
+ if isinstance(msg, RegisterMsg):
164
+ await self.reg_controller.register(msg)
165
+ elif isinstance(msg, DeRegisterMsg):
166
+ await self.reg_controller.deregister(msg)
167
+ elif isinstance(msg, BatchedKVOperationMsg):
168
+ await self.kv_controller.handle_batched_kv_operations(msg)
169
+ elif isinstance(msg, FullSyncBatchMsg):
170
+ await self.kv_controller.handle_full_sync_batch(msg)
171
+ elif isinstance(msg, FullSyncEndMsg):
172
+ await self.kv_controller.handle_full_sync_end(msg)
173
+ else:
174
+ logger.error(f"Unknown worker message type: {msg}")
175
+
176
+ async def handle_worker_req_message(
177
+ self, msg: WorkerReqMsg
178
+ ) -> Union[WorkerReqRetMsg, ErrorMsg]:
179
+ ret_msg: Union[WorkerReqRetMsg, ErrorMsg]
180
+ if isinstance(msg, RegisterMsg):
181
+ # Build extra_config with heartbeat_url if available
182
+ extra_config: dict[str, str] = {}
183
+ if self.controller_urls.get("heartbeat") is not None:
184
+ # Convert bind address (e.g., "0.0.0.0:8082" or "*:8082")
185
+ # to a connectable address using actual controller IP
186
+ heartbeat_bind_url = self.controller_urls["heartbeat"]
187
+ heartbeat_url = self._convert_bind_to_connect_url(
188
+ heartbeat_bind_url, worker_ip=msg.ip
189
+ )
190
+ extra_config["heartbeat_url"] = heartbeat_url
191
+ logger.debug(
192
+ "Returning heartbeat_url to worker: %s (bind: %s)",
193
+ heartbeat_url,
194
+ heartbeat_bind_url,
195
+ )
196
+ ret_msg = await self.reg_controller.register(msg, extra_config)
197
+ elif isinstance(msg, BatchedP2PLookupMsg):
198
+ ret_msg = await self.kv_controller.batched_p2p_lookup(msg)
199
+ elif isinstance(msg, HeartbeatMsg):
200
+ ret_msg = await self.reg_controller.heartbeat(msg)
201
+ elif isinstance(msg, FullSyncStartMsg):
202
+ ret_msg = await self.kv_controller.handle_full_sync_start(msg)
203
+ elif isinstance(msg, FullSyncStatusMsg):
204
+ ret_msg = await self.kv_controller.handle_full_sync_status(msg)
205
+ else:
206
+ logger.error(f"Unknown worker request message type: {msg}")
207
+ ret_msg = ErrorMsg(error=f"Unknown message type: {type(msg)}")
208
+ return ret_msg
209
+
210
+ async def handle_orchestration_message(self, msg: OrchMsg) -> OrchRetMsg:
211
+ if isinstance(msg, LookupMsg):
212
+ return await self.kv_controller.lookup(msg)
213
+ elif isinstance(msg, HealthMsg):
214
+ return await self.reg_controller.health(msg)
215
+ elif isinstance(msg, QueryInstMsg):
216
+ return await self.reg_controller.get_instance_id(msg)
217
+ elif isinstance(msg, ClearMsg):
218
+ return await self.kv_controller.clear(msg)
219
+ elif isinstance(msg, PinMsg):
220
+ return await self.kv_controller.pin(msg)
221
+ elif isinstance(msg, CompressMsg):
222
+ return await self.kv_controller.compress(msg)
223
+ elif isinstance(msg, DecompressMsg):
224
+ return await self.kv_controller.decompress(msg)
225
+ elif isinstance(msg, MoveMsg):
226
+ return await self.kv_controller.move(msg)
227
+ elif isinstance(msg, CheckFinishMsg):
228
+ # FIXME(Jiayi): This `check_finish` thing
229
+ # shouldn't be implemented in kv_controller.
230
+ return await self.kv_controller.check_finish(msg)
231
+ elif isinstance(msg, QueryWorkerInfoMsg):
232
+ return await self.reg_controller.query_worker_info(msg)
233
+ else:
234
+ logger.error(f"Unknown orchestration message type: {msg}")
235
+ raise RuntimeError(f"Unknown orchestration message type: {msg}")
236
+
237
+ def _setup_socket_metrics(self):
238
+ """Setup metrics for socket message counts."""
239
+ # Initialize message counters for observability
240
+ self.pull_socket_message_count = 0
241
+ self.reply_socket_message_count = 0
242
+
243
+ # Initialize active request counters
244
+ self.pull_socket_active_requests = 0
245
+ self.reply_socket_active_requests = 0
246
+
247
+ prometheus_logger = PrometheusLogger.GetInstanceOrNone()
248
+ if prometheus_logger is not None:
249
+ prometheus_logger.pull_socket_message_count.set_function(
250
+ lambda: self.pull_socket_message_count
251
+ )
252
+ prometheus_logger.reply_socket_message_count.set_function(
253
+ lambda: self.reply_socket_message_count
254
+ )
255
+
256
+ # Socket queue/backlog metrics
257
+ prometheus_logger.pull_socket_has_pending.set_function(
258
+ lambda: self._check_socket_has_pending(self.controller_pull_socket)
259
+ )
260
+ if self.controller_urls["reply"] is not None:
261
+ prometheus_logger.reply_socket_has_pending.set_function(
262
+ lambda: self._check_socket_has_pending(self.controller_reply_socket)
263
+ )
264
+
265
+ # Active request metrics
266
+ prometheus_logger.pull_socket_active_requests.set_function(
267
+ lambda: self.pull_socket_active_requests
268
+ )
269
+ prometheus_logger.reply_socket_active_requests.set_function(
270
+ lambda: self.reply_socket_active_requests
271
+ )
272
+
273
+ def _convert_bind_to_connect_url(
274
+ self, bind_url: str, worker_ip: Optional[str] = None
275
+ ) -> str:
276
+ """Convert a bind address to a connectable address.
277
+
278
+ Bind addresses like "0.0.0.0:port" or "*:port" cannot be used
279
+ by workers to connect. We need to replace them with the actual
280
+ controller IP address.
281
+
282
+ If worker_ip is provided and matches the controller's IP, use
283
+ 127.0.0.1 for loopback connection (more reliable than external IP).
284
+
285
+ Args:
286
+ bind_url: The bind URL (e.g., "0.0.0.0:8082" or "*:8082")
287
+ worker_ip: The IP address of the requesting worker (optional)
288
+
289
+ Returns:
290
+ A connectable URL (e.g., "192.168.1.100:8082" or "127.0.0.1:8082")
291
+ """
292
+ if ":" not in bind_url:
293
+ return bind_url
294
+
295
+ host, port = bind_url.rsplit(":", 1)
296
+ # Replace bind-all addresses with actual IP
297
+ if host in ("0.0.0.0", "*", ""):
298
+ actual_ip = get_ip()
299
+ # If worker is on the same machine, use loopback for reliability
300
+ # This handles cases where external IP (e.g., VPN) doesn't support
301
+ # local loopback connections
302
+ if worker_ip is not None and worker_ip == actual_ip:
303
+ logger.debug(
304
+ "Worker IP %s matches controller IP, using 127.0.0.1",
305
+ worker_ip,
306
+ )
307
+ return f"127.0.0.1:{port}"
308
+ return f"{actual_ip}:{port}"
309
+ return bind_url
310
+
311
+ def _check_socket_has_pending(self, socket) -> int:
312
+ """Check if socket has pending messages.
313
+
314
+ Returns:
315
+ 1 if socket has pending messages, 0 otherwise
316
+ """
317
+ try:
318
+ events = socket.get(zmq.EVENTS) # type: ignore[attr-defined]
319
+ # Check if POLLIN flag is set (indicates readable/pending messages)
320
+ has_pending = 1 if (events & zmq.POLLIN) else 0 # type: ignore[attr-defined]
321
+ return has_pending
322
+ except Exception as e:
323
+ logger.error(f"Error checking socket pending status: {e}")
324
+ return 0
325
+
326
+ async def handle_batched_push_request(self, socket) -> Optional[MsgBase]:
327
+ while True:
328
+ parts = await socket.recv_multipart()
329
+ part_count = len(parts)
330
+ with SocketMetricsContext(self, SocketType.PULL, part_count):
331
+ for part in parts:
332
+ # Parse message based on format
333
+ if part.startswith(b"{"):
334
+ # JSON format - typically from external systems
335
+ # like Mooncake
336
+ msg_dict = json.loads(part)
337
+ msg = msgspec.convert(msg_dict, type=Msg)
338
+ else:
339
+ # MessagePack format - internal LMCache communication
340
+ msg = msgspec.msgpack.decode(part, type=Msg)
341
+ if isinstance(msg, WorkerMsg):
342
+ await self.handle_worker_message(msg)
343
+
344
+ # FIXME(Jiayi): The abstraction of control messages
345
+ # might not be necessary.
346
+ # elif isinstance(msg, ControlMsg):
347
+ # await self.issue_control_message(msg)
348
+ elif isinstance(msg, OrchMsg):
349
+ await self.handle_orchestration_message(msg)
350
+ else:
351
+ logger.error(f"Unknown message type: {type(msg)}")
352
+
353
+ async def handle_batched_req_request(self, socket) -> Optional[MsgBase]:
354
+ """Handle requests on ROUTER socket.
355
+
356
+ ROUTER socket receives multi-part messages:
357
+ [identity, empty_frame, payload]
358
+ and must reply with the same identity frame.
359
+ """
360
+ while True:
361
+ frames = await socket.recv_multipart()
362
+ with SocketMetricsContext(self, SocketType.REPLY):
363
+ identity = None
364
+ try:
365
+ # ROUTER socket: [identity, empty_frame, payload]
366
+ if len(frames) < 3:
367
+ logger.error(
368
+ "Invalid ROUTER message format, expected >= 3 frames, "
369
+ "got %d",
370
+ len(frames),
371
+ )
372
+ continue
373
+ identity = frames[0]
374
+ # frames[1] is empty delimiter
375
+ part = frames[2]
376
+
377
+ # Parse message based on format
378
+ if part.startswith(b"{"):
379
+ # JSON format - typically from external systems like Mooncake
380
+ msg_dict = json.loads(part)
381
+ msg = msgspec.convert(msg_dict, type=Msg)
382
+ else:
383
+ # MessagePack format - internal LMCache communication
384
+ msg = msgspec.msgpack.decode(part, type=Msg)
385
+
386
+ if isinstance(msg, WorkerReqMsg):
387
+ ret_msg = await self.handle_worker_req_message(msg)
388
+ # Reply with identity frame for ROUTER socket
389
+ await socket.send_multipart(
390
+ [identity, b"", msgspec.msgpack.encode(ret_msg)]
391
+ )
392
+ else:
393
+ logger.error("Unknown message type: %s", type(msg))
394
+ err_msg = ErrorMsg(error=f"Unknown message type: {type(msg)}")
395
+ await socket.send_multipart(
396
+ [identity, b"", msgspec.msgpack.encode(err_msg)]
397
+ )
398
+ except (
399
+ json.JSONDecodeError,
400
+ msgspec.DecodeError,
401
+ msgspec.ValidationError,
402
+ zmq.ZMQError,
403
+ ) as e:
404
+ logger.error("Error handling request message: %s", e, exc_info=True)
405
+ err_msg = ErrorMsg(error=str(e))
406
+ # Try to reply with error if we have identity
407
+ if identity is not None:
408
+ await socket.send_multipart(
409
+ [identity, b"", msgspec.msgpack.encode(err_msg)]
410
+ )
411
+
412
+ async def handle_heartbeat_request(self, socket) -> None:
413
+ """Handle heartbeat requests on dedicated ROUTER socket.
414
+
415
+ This runs on a separate socket to ensure heartbeats are processed
416
+ without being blocked by other requests.
417
+
418
+ ROUTER socket receives multi-part messages:
419
+ [identity, empty_frame, payload]
420
+ """
421
+ logger.info("Heartbeat handler task started, waiting for heartbeat requests...")
422
+ while True:
423
+ frames = await socket.recv_multipart()
424
+ logger.debug("Received heartbeat request with %d frames", len(frames))
425
+ with SocketMetricsContext(self, SocketType.REPLY):
426
+ identity = None
427
+ try:
428
+ # ROUTER socket: [identity, empty_frame, payload]
429
+ if len(frames) < 3:
430
+ logger.error(
431
+ "Invalid heartbeat ROUTER message format, "
432
+ "expected >= 3 frames, got %d",
433
+ len(frames),
434
+ )
435
+ continue
436
+ identity = frames[0]
437
+ part = frames[2]
438
+
439
+ if part.startswith(b"{"):
440
+ msg_dict = json.loads(part)
441
+ msg = msgspec.convert(msg_dict, type=Msg)
442
+ else:
443
+ msg = msgspec.msgpack.decode(part, type=Msg)
444
+
445
+ if isinstance(msg, HeartbeatMsg):
446
+ ret_msg = await self.reg_controller.heartbeat(msg)
447
+ await socket.send_multipart(
448
+ [identity, b"", msgspec.msgpack.encode(ret_msg)]
449
+ )
450
+ else:
451
+ logger.error(
452
+ "Unexpected message type on heartbeat socket: %s",
453
+ type(msg),
454
+ )
455
+ err_msg = ErrorMsg(
456
+ error=f"Expected HeartbeatMsg, got {type(msg)}"
457
+ )
458
+ await socket.send_multipart(
459
+ [identity, b"", msgspec.msgpack.encode(err_msg)]
460
+ )
461
+ except (
462
+ json.JSONDecodeError,
463
+ msgspec.DecodeError,
464
+ msgspec.ValidationError,
465
+ zmq.ZMQError,
466
+ ) as e:
467
+ logger.error(
468
+ "Error handling heartbeat request: %s", e, exc_info=True
469
+ )
470
+ err_msg = ErrorMsg(error=str(e))
471
+ # Try to reply with error if we have identity
472
+ if identity is not None:
473
+ await socket.send_multipart(
474
+ [identity, b"", msgspec.msgpack.encode(err_msg)]
475
+ )
476
+
477
+ async def health_check(self):
478
+ while True:
479
+ await asyncio.sleep(self.health_check_interval)
480
+ worker_infos = self.reg_controller.registry.get_all_worker_infos_cached(1)
481
+ for worker_info in worker_infos:
482
+ if (
483
+ time.time() - worker_info.last_heartbeat_time
484
+ > self.lmcache_worker_timeout
485
+ ):
486
+ logger.warning(
487
+ "Worker %s_%s last heartbeat time: %s, "
488
+ "current time: %s, more than %s seconds",
489
+ worker_info.instance_id,
490
+ worker_info.worker_id,
491
+ worker_info.last_heartbeat_time,
492
+ time.time(),
493
+ self.lmcache_worker_timeout,
494
+ )
495
+ # Perform a full deregister to clean up all associated resources.
496
+ deregister_msg = DeRegisterMsg(
497
+ instance_id=worker_info.instance_id,
498
+ worker_id=worker_info.worker_id,
499
+ ip=worker_info.ip,
500
+ port=worker_info.port,
501
+ )
502
+ await self.reg_controller.deregister(deregister_msg)
503
+
504
+ async def start_all(self):
505
+ tasks = []
506
+ if self.controller_urls["reply"] is not None:
507
+ logger.info(
508
+ "Starting reply request handler on %s", self.controller_urls["reply"]
509
+ )
510
+ tasks.append(self.handle_batched_req_request(self.controller_reply_socket))
511
+ if self.controller_heartbeat_socket is not None:
512
+ logger.info(
513
+ "Starting heartbeat request handler on %s",
514
+ self.controller_urls.get("heartbeat"),
515
+ )
516
+ tasks.append(
517
+ self.handle_heartbeat_request(self.controller_heartbeat_socket)
518
+ )
519
+ else:
520
+ logger.warning("Heartbeat socket is None, heartbeat handler not started!")
521
+ logger.info("Starting pull request handler on %s", self.controller_urls["pull"])
522
+ tasks.append(self.handle_batched_push_request(self.controller_pull_socket))
523
+ await asyncio.gather(
524
+ *tasks,
525
+ return_exceptions=True,
526
+ )
527
+
528
+ def close(self):
529
+ """Clean up all resources owned by the controller manager."""
530
+ if hasattr(self, "controller_pull_socket"):
531
+ self.controller_pull_socket.close()
532
+ if hasattr(self, "controller_reply_socket"):
533
+ self.controller_reply_socket.close()
534
+ if hasattr(self, "zmq_context"):
535
+ self.zmq_context.destroy()
@@ -0,0 +1,11 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # First Party
3
+ from lmcache.v1.cache_controller.controllers.kv_controller import KVController
4
+ from lmcache.v1.cache_controller.controllers.registration_controller import ( # noqa: E501
5
+ RegistrationController,
6
+ )
7
+
8
+ __all__ = [
9
+ "KVController",
10
+ "RegistrationController",
11
+ ]