lmcache-cli 0.4.5.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (399) hide show
  1. lmcache/__init__.py +84 -0
  2. lmcache/_version.py +24 -0
  3. lmcache/cli/__init__.py +1 -0
  4. lmcache/cli/commands/__init__.py +34 -0
  5. lmcache/cli/commands/base.py +157 -0
  6. lmcache/cli/commands/bench/__init__.py +557 -0
  7. lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
  8. lmcache/cli/commands/bench/engine_bench/config.py +245 -0
  9. lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
  10. lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
  11. lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
  12. lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
  13. lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
  14. lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
  15. lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
  16. lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
  17. lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
  18. lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
  19. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
  20. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
  21. lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
  22. lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
  23. lmcache/cli/commands/describe.py +310 -0
  24. lmcache/cli/commands/kvcache.py +133 -0
  25. lmcache/cli/commands/mock.py +75 -0
  26. lmcache/cli/commands/ping.py +113 -0
  27. lmcache/cli/commands/query/__init__.py +155 -0
  28. lmcache/cli/commands/query/prompt.py +134 -0
  29. lmcache/cli/commands/query/request.py +357 -0
  30. lmcache/cli/commands/server.py +99 -0
  31. lmcache/cli/commands/tool/__init__.py +63 -0
  32. lmcache/cli/commands/tool/cache_simulator.py +113 -0
  33. lmcache/cli/commands/trace/__init__.py +505 -0
  34. lmcache/cli/commands/trace/dispatch.py +249 -0
  35. lmcache/cli/commands/trace/driver.py +372 -0
  36. lmcache/cli/commands/trace/stats.py +289 -0
  37. lmcache/cli/documents/lmcache.txt +11 -0
  38. lmcache/cli/main.py +42 -0
  39. lmcache/cli/metrics/__init__.py +29 -0
  40. lmcache/cli/metrics/formatter.py +171 -0
  41. lmcache/cli/metrics/handler.py +94 -0
  42. lmcache/cli/metrics/metrics.py +161 -0
  43. lmcache/cli/metrics/section.py +77 -0
  44. lmcache/connections.py +173 -0
  45. lmcache/integration/__init__.py +2 -0
  46. lmcache/integration/base_service_factory.py +165 -0
  47. lmcache/integration/request_telemetry/__init__.py +1 -0
  48. lmcache/integration/request_telemetry/base.py +51 -0
  49. lmcache/integration/request_telemetry/factory.py +113 -0
  50. lmcache/integration/request_telemetry/fastapi.py +109 -0
  51. lmcache/integration/request_telemetry/noop.py +35 -0
  52. lmcache/integration/sglang/__init__.py +2 -0
  53. lmcache/integration/sglang/sglang_adapter.py +326 -0
  54. lmcache/integration/sglang/utils.py +39 -0
  55. lmcache/integration/vllm/__init__.py +1 -0
  56. lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
  57. lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
  58. lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
  59. lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
  60. lmcache/integration/vllm/utils.py +433 -0
  61. lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
  62. lmcache/integration/vllm/vllm_service_factory.py +339 -0
  63. lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
  64. lmcache/logging.py +107 -0
  65. lmcache/native_storage_ops.pyi +230 -0
  66. lmcache/non_cuda_equivalents.py +1424 -0
  67. lmcache/observability.py +1958 -0
  68. lmcache/storage_backend/serde/__init__.py +1 -0
  69. lmcache/storage_backend/serde/cachegen_basics.py +210 -0
  70. lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
  71. lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
  72. lmcache/storage_backend/serde/serde.py +75 -0
  73. lmcache/tools/__init__.py +1 -0
  74. lmcache/tools/cache_simulator/README.md +392 -0
  75. lmcache/tools/cache_simulator/__init__.py +1 -0
  76. lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
  77. lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
  78. lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
  79. lmcache/tools/cache_simulator/lru_cache.py +124 -0
  80. lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
  81. lmcache/tools/cache_simulator/simulator.py +795 -0
  82. lmcache/tools/controller_benchmark/README.md +161 -0
  83. lmcache/tools/controller_benchmark/__init__.py +1 -0
  84. lmcache/tools/controller_benchmark/__main__.py +331 -0
  85. lmcache/tools/controller_benchmark/benchmark.py +660 -0
  86. lmcache/tools/controller_benchmark/config.py +44 -0
  87. lmcache/tools/controller_benchmark/constants.py +10 -0
  88. lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
  89. lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
  90. lmcache/tools/controller_benchmark/handlers/base.py +47 -0
  91. lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
  92. lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
  93. lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
  94. lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
  95. lmcache/tools/controller_benchmark/handlers/register.py +56 -0
  96. lmcache/tools/mp_status_viewer/__init__.py +1 -0
  97. lmcache/tools/mp_status_viewer/__main__.py +95 -0
  98. lmcache/usage_context.py +417 -0
  99. lmcache/utils.py +665 -0
  100. lmcache/v1/__init__.py +2 -0
  101. lmcache/v1/api_server/__init__.py +2 -0
  102. lmcache/v1/api_server/__main__.py +537 -0
  103. lmcache/v1/basic_check.py +112 -0
  104. lmcache/v1/cache_controller/__init__.py +9 -0
  105. lmcache/v1/cache_controller/commands/__init__.py +15 -0
  106. lmcache/v1/cache_controller/commands/base.py +35 -0
  107. lmcache/v1/cache_controller/commands/full_sync.py +49 -0
  108. lmcache/v1/cache_controller/config.py +176 -0
  109. lmcache/v1/cache_controller/controller_manager.py +535 -0
  110. lmcache/v1/cache_controller/controllers/__init__.py +11 -0
  111. lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
  112. lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
  113. lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
  114. lmcache/v1/cache_controller/executor.py +463 -0
  115. lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
  116. lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
  117. lmcache/v1/cache_controller/frontend/static/index.html +234 -0
  118. lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
  119. lmcache/v1/cache_controller/full_sync_sender.py +475 -0
  120. lmcache/v1/cache_controller/locks.py +149 -0
  121. lmcache/v1/cache_controller/message.py +828 -0
  122. lmcache/v1/cache_controller/observability.py +208 -0
  123. lmcache/v1/cache_controller/utils.py +679 -0
  124. lmcache/v1/cache_controller/worker.py +665 -0
  125. lmcache/v1/cache_engine.py +2058 -0
  126. lmcache/v1/cache_interface.py +19 -0
  127. lmcache/v1/check/__init__.py +74 -0
  128. lmcache/v1/check/check_mode_gen.py +86 -0
  129. lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
  130. lmcache/v1/check/check_mode_test_remote.py +155 -0
  131. lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
  132. lmcache/v1/check/utils.py +571 -0
  133. lmcache/v1/compute/__init__.py +2 -0
  134. lmcache/v1/compute/attention/__init__.py +0 -0
  135. lmcache/v1/compute/attention/abstract.py +39 -0
  136. lmcache/v1/compute/attention/flash_attn.py +129 -0
  137. lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
  138. lmcache/v1/compute/attention/metadata.py +85 -0
  139. lmcache/v1/compute/attention/utils.py +14 -0
  140. lmcache/v1/compute/blend/__init__.py +7 -0
  141. lmcache/v1/compute/blend/blender.py +168 -0
  142. lmcache/v1/compute/blend/metadata.py +34 -0
  143. lmcache/v1/compute/blend/utils.py +63 -0
  144. lmcache/v1/compute/models/__init__.py +0 -0
  145. lmcache/v1/compute/models/base.py +141 -0
  146. lmcache/v1/compute/models/llama.py +9 -0
  147. lmcache/v1/compute/models/qwen3.py +24 -0
  148. lmcache/v1/compute/models/utils.py +68 -0
  149. lmcache/v1/compute/positional_encoding.py +199 -0
  150. lmcache/v1/config.py +848 -0
  151. lmcache/v1/config_base.py +848 -0
  152. lmcache/v1/distributed/api.py +248 -0
  153. lmcache/v1/distributed/config.py +321 -0
  154. lmcache/v1/distributed/error.py +64 -0
  155. lmcache/v1/distributed/eviction.py +192 -0
  156. lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
  157. lmcache/v1/distributed/eviction_policy/factory.py +27 -0
  158. lmcache/v1/distributed/eviction_policy/lru.py +244 -0
  159. lmcache/v1/distributed/eviction_policy/noop.py +50 -0
  160. lmcache/v1/distributed/internal_api.py +170 -0
  161. lmcache/v1/distributed/l1_manager.py +835 -0
  162. lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
  163. lmcache/v1/distributed/l2_adapters/base.py +360 -0
  164. lmcache/v1/distributed/l2_adapters/config.py +385 -0
  165. lmcache/v1/distributed/l2_adapters/factory.py +205 -0
  166. lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
  167. lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
  168. lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
  169. lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
  170. lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
  171. lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
  172. lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
  173. lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
  174. lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
  175. lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
  176. lmcache/v1/distributed/memory_manager.py +179 -0
  177. lmcache/v1/distributed/storage_controller.py +39 -0
  178. lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
  179. lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
  180. lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
  181. lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
  182. lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
  183. lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
  184. lmcache/v1/distributed/storage_manager.py +532 -0
  185. lmcache/v1/event_manager.py +145 -0
  186. lmcache/v1/exceptions/__init__.py +16 -0
  187. lmcache/v1/gpu_connector/__init__.py +126 -0
  188. lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
  189. lmcache/v1/gpu_connector/gpu_ops.py +85 -0
  190. lmcache/v1/gpu_connector/hpu_connector.py +326 -0
  191. lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
  192. lmcache/v1/gpu_connector/utils.py +890 -0
  193. lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
  194. lmcache/v1/health_monitor/__init__.py +1 -0
  195. lmcache/v1/health_monitor/base.py +587 -0
  196. lmcache/v1/health_monitor/checks/__init__.py +1 -0
  197. lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
  198. lmcache/v1/health_monitor/constants.py +36 -0
  199. lmcache/v1/internal_api_server/__init__.py +0 -0
  200. lmcache/v1/internal_api_server/api_registry.py +59 -0
  201. lmcache/v1/internal_api_server/api_server.py +120 -0
  202. lmcache/v1/internal_api_server/common/__init__.py +1 -0
  203. lmcache/v1/internal_api_server/common/env_api.py +22 -0
  204. lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
  205. lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
  206. lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
  207. lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
  208. lmcache/v1/internal_api_server/common/thread_api.py +63 -0
  209. lmcache/v1/internal_api_server/controller/__init__.py +1 -0
  210. lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
  211. lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
  212. lmcache/v1/internal_api_server/utils.py +43 -0
  213. lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
  214. lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
  215. lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
  216. lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
  217. lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
  218. lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
  219. lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
  220. lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
  221. lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
  222. lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
  223. lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
  224. lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
  225. lmcache/v1/kv_layer_groups.py +267 -0
  226. lmcache/v1/lazy_memory_allocator.py +284 -0
  227. lmcache/v1/lookup_client/__init__.py +25 -0
  228. lmcache/v1/lookup_client/abstract_client.py +77 -0
  229. lmcache/v1/lookup_client/async_lookup_message.py +50 -0
  230. lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
  231. lmcache/v1/lookup_client/factory.py +251 -0
  232. lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
  233. lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
  234. lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
  235. lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
  236. lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
  237. lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
  238. lmcache/v1/lookup_client/record_strategies/base.py +327 -0
  239. lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
  240. lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
  241. lmcache/v1/manager.py +539 -0
  242. lmcache/v1/memory_management.py +2619 -0
  243. lmcache/v1/metadata.py +114 -0
  244. lmcache/v1/mp_observability/AGENTS.override.md +21 -0
  245. lmcache/v1/mp_observability/README.md +204 -0
  246. lmcache/v1/mp_observability/config.py +340 -0
  247. lmcache/v1/mp_observability/event.py +100 -0
  248. lmcache/v1/mp_observability/event_bus.py +313 -0
  249. lmcache/v1/mp_observability/otel_init.py +145 -0
  250. lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
  251. lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
  252. lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
  253. lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
  254. lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
  255. lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
  256. lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
  257. lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
  258. lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
  259. lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
  260. lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
  261. lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
  262. lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
  263. lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
  264. lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
  265. lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
  266. lmcache/v1/mp_observability/trace/__init__.py +50 -0
  267. lmcache/v1/mp_observability/trace/codecs.py +255 -0
  268. lmcache/v1/mp_observability/trace/decorator.py +147 -0
  269. lmcache/v1/mp_observability/trace/format.py +132 -0
  270. lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
  271. lmcache/v1/mp_observability/trace/reader.py +167 -0
  272. lmcache/v1/mp_observability/trace/recorder.py +300 -0
  273. lmcache/v1/multiprocess/__init__.py +0 -0
  274. lmcache/v1/multiprocess/affinity_pool.py +102 -0
  275. lmcache/v1/multiprocess/blend_server_v2.py +891 -0
  276. lmcache/v1/multiprocess/config.py +253 -0
  277. lmcache/v1/multiprocess/custom_types.py +281 -0
  278. lmcache/v1/multiprocess/futures.py +194 -0
  279. lmcache/v1/multiprocess/gpu_context.py +511 -0
  280. lmcache/v1/multiprocess/http_server.py +235 -0
  281. lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
  282. lmcache/v1/multiprocess/mq.py +732 -0
  283. lmcache/v1/multiprocess/protocol.py +86 -0
  284. lmcache/v1/multiprocess/protocols/README.md +213 -0
  285. lmcache/v1/multiprocess/protocols/__init__.py +127 -0
  286. lmcache/v1/multiprocess/protocols/base.py +89 -0
  287. lmcache/v1/multiprocess/protocols/blend.py +109 -0
  288. lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
  289. lmcache/v1/multiprocess/protocols/controller.py +53 -0
  290. lmcache/v1/multiprocess/protocols/debug.py +34 -0
  291. lmcache/v1/multiprocess/protocols/engine.py +146 -0
  292. lmcache/v1/multiprocess/protocols/observability.py +39 -0
  293. lmcache/v1/multiprocess/server.py +1134 -0
  294. lmcache/v1/multiprocess/session.py +190 -0
  295. lmcache/v1/multiprocess/token_hasher.py +441 -0
  296. lmcache/v1/offload_server/__init__.py +17 -0
  297. lmcache/v1/offload_server/abstract_server.py +37 -0
  298. lmcache/v1/offload_server/message.py +30 -0
  299. lmcache/v1/offload_server/zmq_server.py +122 -0
  300. lmcache/v1/periodic_thread.py +579 -0
  301. lmcache/v1/pin_monitor.py +246 -0
  302. lmcache/v1/plugin/__init__.py +0 -0
  303. lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
  304. lmcache/v1/protocol.py +317 -0
  305. lmcache/v1/rpc/__init__.py +17 -0
  306. lmcache/v1/rpc/transport.py +105 -0
  307. lmcache/v1/rpc/zmq_transport.py +213 -0
  308. lmcache/v1/rpc_utils.py +165 -0
  309. lmcache/v1/server/__init__.py +2 -0
  310. lmcache/v1/server/__main__.py +170 -0
  311. lmcache/v1/server/storage_backend/__init__.py +21 -0
  312. lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
  313. lmcache/v1/server/storage_backend/local_backend.py +75 -0
  314. lmcache/v1/server/utils.py +21 -0
  315. lmcache/v1/standalone/__init__.py +1 -0
  316. lmcache/v1/standalone/__main__.py +583 -0
  317. lmcache/v1/standalone/manager.py +80 -0
  318. lmcache/v1/standalone/standalone_service_factory.py +86 -0
  319. lmcache/v1/storage_backend/__init__.py +313 -0
  320. lmcache/v1/storage_backend/abstract_backend.py +445 -0
  321. lmcache/v1/storage_backend/audit_backend.py +233 -0
  322. lmcache/v1/storage_backend/batched_message_sender.py +222 -0
  323. lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
  324. lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
  325. lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
  326. lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
  327. lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
  328. lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
  329. lmcache/v1/storage_backend/connector/__init__.py +443 -0
  330. lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
  331. lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
  332. lmcache/v1/storage_backend/connector/base_connector.py +379 -0
  333. lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
  334. lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
  335. lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
  336. lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
  337. lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
  338. lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
  339. lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
  340. lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
  341. lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
  342. lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
  343. lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
  344. lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
  345. lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
  346. lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
  347. lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
  348. lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
  349. lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
  350. lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
  351. lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
  352. lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
  353. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
  354. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
  355. lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
  356. lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
  357. lmcache/v1/storage_backend/gds_backend.py +1199 -0
  358. lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
  359. lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
  360. lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
  361. lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
  362. lmcache/v1/storage_backend/local_disk_backend.py +656 -0
  363. lmcache/v1/storage_backend/maru_backend.py +734 -0
  364. lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
  365. lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
  366. lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
  367. lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
  368. lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
  369. lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
  370. lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
  371. lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
  372. lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
  373. lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
  374. lmcache/v1/storage_backend/p2p_backend.py +788 -0
  375. lmcache/v1/storage_backend/path_sharder.py +117 -0
  376. lmcache/v1/storage_backend/pd_backend.py +646 -0
  377. lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
  378. lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
  379. lmcache/v1/storage_backend/remote_backend.py +624 -0
  380. lmcache/v1/storage_backend/resp_client.py +227 -0
  381. lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
  382. lmcache/v1/storage_backend/storage_manager.py +1352 -0
  383. lmcache/v1/system_detection.py +110 -0
  384. lmcache/v1/token_database.py +551 -0
  385. lmcache/v1/transfer_channel/__init__.py +83 -0
  386. lmcache/v1/transfer_channel/abstract.py +285 -0
  387. lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
  388. lmcache/v1/transfer_channel/nixl_channel.py +639 -0
  389. lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
  390. lmcache/v1/transfer_channel/transfer_utils.py +63 -0
  391. lmcache/v1/utils/__init__.py +1 -0
  392. lmcache/v1/utils/bloom_filter.py +109 -0
  393. lmcache/v1/utils/cache_utils.py +125 -0
  394. lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
  395. lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
  396. lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
  397. lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
  398. lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
  399. lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,660 @@
1
+ """Core benchmark class for LMCache Controller ZMQ testing"""
2
+
3
+ # SPDX-License-Identifier: Apache-2.0
4
+
5
+ # Standard
6
+ from dataclasses import dataclass, field
7
+ from typing import Any, Dict, List, Optional, Tuple
8
+ import asyncio
9
+ import random
10
+ import statistics
11
+ import time
12
+
13
+ # Third Party
14
+ import msgspec
15
+ import psutil
16
+ import zmq
17
+ import zmq.asyncio
18
+
19
+ # First Party
20
+ from lmcache.logging import init_logger
21
+ from lmcache.v1.cache_controller.message import (
22
+ DeRegisterMsg,
23
+ RegisterMsg,
24
+ RegisterRetMsg,
25
+ )
26
+ from lmcache.v1.cache_controller.utils import KVChunkInfo
27
+ from lmcache.v1.rpc_utils import (
28
+ close_zmq_socket,
29
+ get_zmq_context,
30
+ get_zmq_socket,
31
+ get_zmq_socket_with_timeout,
32
+ )
33
+
34
+ # Local
35
+ from .config import ZMQBenchmarkConfig
36
+ from .constants import (
37
+ DEFAULT_BATCH_SEND_SIZE,
38
+ DEFAULT_OP_DISTRIBUTION_BASE,
39
+ DEFAULT_RECV_TIMEOUT_MS,
40
+ DEFAULT_SEND_HWM,
41
+ DEFAULT_SEND_TIMEOUT_MS,
42
+ )
43
+ from .handlers import OPERATION_HANDLERS
44
+ from .handlers.base import SocketType
45
+
46
+ logger = init_logger(__name__)
47
+
48
+
49
+ @dataclass
50
+ class TestData:
51
+ """Test data for benchmark operations"""
52
+
53
+ instances: List[str]
54
+ workers: List[int]
55
+ locations: List[str]
56
+ keys: List[int]
57
+
58
+
59
+ @dataclass
60
+ class OperationStats:
61
+ """Statistics for a single operation type"""
62
+
63
+ qps: float = 0.0 # messages per second
64
+ rps: float = 0.0 # requests per second
65
+ avg_latency: float = 0.0
66
+ min_latency: float = 0.0
67
+ max_latency: float = 0.0
68
+ p95_latency: float = 0.0
69
+ errors: int = 0
70
+
71
+
72
+ @dataclass
73
+ class BenchmarkResults:
74
+ """Overall benchmark results"""
75
+
76
+ total_requests: int = 0
77
+ total_messages: int = 0
78
+ total_time: float = 0.0
79
+ overall_rps: float = 0.0 # requests per second
80
+ overall_qps: float = 0.0 # messages per second
81
+ operations: Dict[str, OperationStats] = field(default_factory=dict)
82
+ memory_usage: List[float] = field(default_factory=list)
83
+
84
+
85
+ class ZMQControllerBenchmark:
86
+ """Benchmark class for LMCache Controller via ZMQ"""
87
+
88
+ def __init__(self, config: ZMQBenchmarkConfig):
89
+ self.config = config
90
+ self.context: Optional[zmq.asyncio.Context] = None
91
+ self.push_socket: Optional[Any] = None
92
+ self.req_socket: Optional[Any] = None
93
+ self.heartbeat_socket: Optional[Any] = None
94
+ self.heartbeat_url: Optional[str] = None
95
+ self.results = BenchmarkResults()
96
+ self.running = False
97
+ # Track sequence numbers per KVChunkInfo (instance_id, worker_id, location)
98
+ self.sequence_numbers: Dict[KVChunkInfo, int] = {}
99
+
100
+ # Track registered workers for cleanup
101
+ self.registered_workers: List[Tuple[str, int, str, int]] = []
102
+
103
+ async def setup(self):
104
+ """Setup ZMQ sockets"""
105
+ self.context = get_zmq_context(use_asyncio=True)
106
+ self.push_socket = get_zmq_socket(
107
+ self.context,
108
+ self.config.controller_pull_url,
109
+ protocol="tcp",
110
+ role=zmq.PUSH,
111
+ bind_or_connect="connect",
112
+ )
113
+ # Set send timeout to avoid blocking indefinitely when controller is down
114
+ # SNDTIMEO: timeout in milliseconds, 0 means non-blocking
115
+ self.push_socket.setsockopt(zmq.SNDTIMEO, DEFAULT_SEND_TIMEOUT_MS)
116
+ # SNDHWM: high watermark for outbound messages
117
+ self.push_socket.setsockopt(zmq.SNDHWM, DEFAULT_SEND_HWM)
118
+ logger.info(
119
+ "Connected to controller PULL socket at %s",
120
+ self.config.controller_pull_url,
121
+ )
122
+
123
+ # Setup DEALER socket for request-reply operations (e.g., P2P lookup)
124
+ if self.config.controller_reply_url:
125
+ self.req_socket = get_zmq_socket_with_timeout(
126
+ self.context,
127
+ self.config.controller_reply_url,
128
+ protocol="tcp",
129
+ role=zmq.DEALER,
130
+ bind_or_connect="connect",
131
+ recv_timeout_ms=DEFAULT_RECV_TIMEOUT_MS,
132
+ send_timeout_ms=DEFAULT_SEND_TIMEOUT_MS,
133
+ )
134
+ logger.info(
135
+ "Connected to controller ROUTER socket at tcp://%s",
136
+ self.config.controller_reply_url,
137
+ )
138
+ logger.info(
139
+ "DEALER socket type: %d, last_endpoint: %s",
140
+ self.req_socket.get(zmq.TYPE),
141
+ self.req_socket.get_string(zmq.LAST_ENDPOINT, encoding="utf-8"),
142
+ )
143
+ # Give ZMQ time to establish the connection
144
+ time.sleep(0.1)
145
+
146
+ # Setup heartbeat DEALER socket if configured
147
+ if self.config.controller_heartbeat_url:
148
+ self.heartbeat_url = self.config.controller_heartbeat_url
149
+ self._setup_heartbeat_socket()
150
+ logger.info(
151
+ "Connected to heartbeat ROUTER socket at tcp://%s",
152
+ self.config.controller_heartbeat_url,
153
+ )
154
+
155
+ def cleanup(self):
156
+ """Cleanup ZMQ sockets"""
157
+ if self.push_socket:
158
+ close_zmq_socket(self.push_socket)
159
+ if self.req_socket:
160
+ close_zmq_socket(self.req_socket)
161
+ if self.heartbeat_socket:
162
+ close_zmq_socket(self.heartbeat_socket)
163
+ logger.info("ZMQ sockets closed")
164
+
165
+ def generate_test_data(self) -> TestData:
166
+ """Generate test data based on configuration
167
+
168
+ Each process gets a unique range of instance IDs to avoid conflicts.
169
+ Format: instance_p{process_id}_{instance_index}
170
+ """
171
+ process_id = self.config.process_id
172
+ return TestData(
173
+ instances=[
174
+ "instance_p%d_%d" % (process_id, i)
175
+ for i in range(self.config.num_instances)
176
+ ],
177
+ workers=list(range(self.config.num_workers)),
178
+ locations=["location_%d" % i for i in range(self.config.num_locations)],
179
+ keys=list(range(self.config.num_keys)),
180
+ )
181
+
182
+ def get_next_sequence_number(
183
+ self, instance_id: str, worker_id: int, location: str
184
+ ) -> int:
185
+ """
186
+ Get monotonically increasing sequence number for specific
187
+ instance-worker-location
188
+ """
189
+ key = KVChunkInfo(instance_id, worker_id, location)
190
+ if key not in self.sequence_numbers:
191
+ self.sequence_numbers[key] = 0
192
+ seq = self.sequence_numbers[key]
193
+ self.sequence_numbers[key] += 1
194
+ return seq
195
+
196
+ async def send_messages(self, messages: List[Any]) -> float:
197
+ """Send multiple messages via ZMQ PUSH socket
198
+
199
+ Args:
200
+ messages: List of messages to send
201
+
202
+ Returns:
203
+ Time taken to send all messages
204
+
205
+ Raises:
206
+ RuntimeError: If socket not initialized or send timeout
207
+ """
208
+ if self.push_socket is None:
209
+ raise RuntimeError("Socket not initialized. Call setup() first.")
210
+ start_time = time.time()
211
+ encoded_msgs = [msgspec.msgpack.encode(msg) for msg in messages]
212
+ try:
213
+ await self.push_socket.send_multipart(encoded_msgs)
214
+ except zmq.Again as e:
215
+ raise RuntimeError(
216
+ "Send timeout - Controller may not be running at %s"
217
+ % self.config.controller_pull_url
218
+ ) from e
219
+ return time.time() - start_time
220
+
221
+ async def send_request(self, message: Any) -> Tuple[float, Any]:
222
+ """Send a message via ZMQ DEALER socket and wait for reply
223
+
224
+ Args:
225
+ message: Message to send
226
+
227
+ Returns:
228
+ Tuple of (time taken, response)
229
+
230
+ Raises:
231
+ RuntimeError: If socket not initialized or timeout
232
+ zmq.ZMQError: If DEALER socket error occurs
233
+ """
234
+ if self.req_socket is None:
235
+ raise RuntimeError(
236
+ "DEALER socket not initialized. "
237
+ "Ensure controller_reply_url is configured."
238
+ )
239
+ start_time = time.time()
240
+ encoded_msg = msgspec.msgpack.encode(message)
241
+ logger.debug(
242
+ "Sending request to %s, message type: %s, size: %d bytes",
243
+ self.config.controller_reply_url,
244
+ type(message).__name__,
245
+ len(encoded_msg),
246
+ )
247
+
248
+ try:
249
+ # DEALER socket: send [empty_frame, payload]
250
+ await self.req_socket.send_multipart([b"", encoded_msg])
251
+ frames = await self.req_socket.recv_multipart()
252
+ # DEALER receives: [empty_frame, payload]
253
+ response = frames[-1]
254
+ logger.debug("Response received, size: %d bytes", len(response))
255
+ return time.time() - start_time, response
256
+ except zmq.Again as e:
257
+ logger.error("Request timeout after waiting for response")
258
+ raise RuntimeError(
259
+ "Request timeout - Controller may not be running at %s"
260
+ % self.config.controller_reply_url
261
+ ) from e
262
+ except zmq.ZMQError as e:
263
+ logger.error("ZMQ error: %s", e)
264
+ # Re-raise other ZMQ errors
265
+ raise
266
+
267
+ async def register_workers(self, test_data: TestData):
268
+ """Pre-register all workers before benchmark using DEALER-ROUTER mode"""
269
+ if not self.config.register_first:
270
+ return
271
+
272
+ if self.req_socket is None:
273
+ logger.warning(
274
+ "DEALER socket not initialized, skipping worker registration"
275
+ )
276
+ return
277
+
278
+ logger.info("Pre-registering workers via REQ-REP...")
279
+ for instance in test_data.instances:
280
+ for worker in test_data.workers:
281
+ ip = "192.168.1.%d" % (worker + 1)
282
+ port = 10000 + worker
283
+ peer_port = 20000 + worker
284
+ peer_init_url = "tcp://%s:%d" % (ip, peer_port)
285
+ msg = RegisterMsg(
286
+ instance_id=instance,
287
+ worker_id=worker,
288
+ ip=ip,
289
+ port=port,
290
+ peer_init_url=peer_init_url,
291
+ )
292
+ try:
293
+ _, response = await self.send_request(msg)
294
+ self.registered_workers.append((instance, worker, ip, port))
295
+ # Extract heartbeat_url from first successful registration
296
+ # (only if not already configured)
297
+ if self.heartbeat_url is None and response:
298
+ self._process_register_response(response)
299
+ except (RuntimeError, zmq.ZMQError) as e:
300
+ logger.error(
301
+ "Failed to register worker %s-%d: %s", instance, worker, e
302
+ )
303
+
304
+ logger.info("Registered %d workers", len(self.registered_workers))
305
+ # Setup heartbeat socket after getting heartbeat_url (if not already setup)
306
+ if self.heartbeat_url and self.heartbeat_socket is None:
307
+ self._setup_heartbeat_socket()
308
+ await asyncio.sleep(0.5)
309
+
310
+ def _process_register_response(self, response: bytes):
311
+ """Process RegisterRetMsg to extract heartbeat_url"""
312
+ try:
313
+ ret_msg = msgspec.msgpack.decode(response, type=RegisterRetMsg)
314
+ if ret_msg.extra_config and "heartbeat_url" in ret_msg.extra_config:
315
+ raw_url = ret_msg.extra_config["heartbeat_url"]
316
+ # Strip tcp:// prefix if present (get_zmq_socket adds it)
317
+ if raw_url.startswith("tcp://"):
318
+ raw_url = raw_url[6:]
319
+ # If benchmark connects to localhost but controller returns
320
+ # a different IP, use localhost for heartbeat as well
321
+ raw_url = self._normalize_heartbeat_url(raw_url)
322
+ self.heartbeat_url = raw_url
323
+ logger.info("Got heartbeat_url from register: %s", self.heartbeat_url)
324
+ except msgspec.DecodeError as e:
325
+ logger.warning("Failed to decode RegisterRetMsg: %s", e)
326
+
327
+ def _normalize_heartbeat_url(self, heartbeat_url: str) -> str:
328
+ """Normalize heartbeat URL based on controller connection.
329
+
330
+ If benchmark connects to controller via localhost (127.0.0.1),
331
+ but heartbeat_url contains a different IP (e.g., from get_ip()),
332
+ replace it with 127.0.0.1 to ensure connectivity.
333
+
334
+ Args:
335
+ heartbeat_url: The heartbeat URL from controller (e.g., "10.0.0.1:7557")
336
+
337
+ Returns:
338
+ Normalized URL (e.g., "127.0.0.1:7557" if connecting locally)
339
+ """
340
+ if ":" not in heartbeat_url:
341
+ return heartbeat_url
342
+
343
+ hb_host, hb_port = heartbeat_url.rsplit(":", 1)
344
+
345
+ # Check if we're connecting to controller via localhost
346
+ controller_host = self.config.controller_pull_url.split(":")[0]
347
+ if controller_host in ("127.0.0.1", "localhost"):
348
+ # If controller returns a non-localhost IP, use localhost instead
349
+ if hb_host not in ("127.0.0.1", "localhost"):
350
+ logger.info(
351
+ "Controller returned heartbeat IP %s, "
352
+ "but we're connecting locally. Using 127.0.0.1 instead.",
353
+ hb_host,
354
+ )
355
+ return "127.0.0.1:%s" % hb_port
356
+
357
+ return heartbeat_url
358
+
359
+ def _setup_heartbeat_socket(self):
360
+ """Setup heartbeat DEALER socket after getting heartbeat_url from register"""
361
+ if not self.heartbeat_url or not self.context:
362
+ logger.warning(
363
+ "Cannot setup heartbeat socket: heartbeat_url=%s, context=%s",
364
+ self.heartbeat_url,
365
+ self.context is not None,
366
+ )
367
+ return
368
+ if self.heartbeat_socket is not None:
369
+ return # Already setup
370
+ logger.info(
371
+ "Setting up heartbeat DEALER socket to %s, "
372
+ "recv_timeout=%dms, send_timeout=%dms",
373
+ self.heartbeat_url,
374
+ DEFAULT_RECV_TIMEOUT_MS,
375
+ DEFAULT_SEND_TIMEOUT_MS,
376
+ )
377
+ self.heartbeat_socket = get_zmq_socket_with_timeout(
378
+ self.context,
379
+ self.heartbeat_url,
380
+ protocol="tcp",
381
+ role=zmq.DEALER,
382
+ bind_or_connect="connect",
383
+ recv_timeout_ms=DEFAULT_RECV_TIMEOUT_MS,
384
+ send_timeout_ms=DEFAULT_SEND_TIMEOUT_MS,
385
+ )
386
+ logger.info("Heartbeat socket created successfully")
387
+
388
+ async def send_heartbeat(self, message: Any) -> Tuple[float, Any]:
389
+ """Send heartbeat via dedicated heartbeat DEALER socket
390
+
391
+ Args:
392
+ message: HeartbeatMsg to send
393
+
394
+ Returns:
395
+ Tuple of (time taken, response)
396
+ """
397
+ if self.heartbeat_socket is None:
398
+ raise RuntimeError(
399
+ "Heartbeat socket not initialized. "
400
+ "heartbeat_url=%s. Register first to get heartbeat_url."
401
+ % self.heartbeat_url
402
+ )
403
+ start_time = time.time()
404
+ encoded_msg = msgspec.msgpack.encode(message)
405
+ try:
406
+ # DEALER socket: send [empty_frame, payload]
407
+ await self.heartbeat_socket.send_multipart([b"", encoded_msg])
408
+ frames = await self.heartbeat_socket.recv_multipart()
409
+ response = frames[-1]
410
+ return time.time() - start_time, response
411
+ except zmq.Again as e:
412
+ raise RuntimeError(
413
+ "Heartbeat timeout waiting for response from %s" % self.heartbeat_url
414
+ ) from e
415
+
416
+ async def deregister_workers(self):
417
+ """Deregister all workers after benchmark"""
418
+ if not self.registered_workers:
419
+ return
420
+
421
+ logger.info("Deregistering workers...")
422
+ messages = []
423
+ for instance, worker, ip, port in self.registered_workers:
424
+ msg = DeRegisterMsg(
425
+ instance_id=instance,
426
+ worker_id=worker,
427
+ ip=ip,
428
+ port=port,
429
+ )
430
+ messages.append(msg)
431
+
432
+ # Send in batches
433
+ for i in range(0, len(messages), DEFAULT_BATCH_SEND_SIZE):
434
+ batch = messages[i : i + DEFAULT_BATCH_SEND_SIZE]
435
+ await self.send_messages(batch)
436
+
437
+ logger.info("Deregistered %d workers", len(messages))
438
+ self.registered_workers.clear()
439
+
440
+ def _build_operation_distribution(self) -> List[str]:
441
+ """Build operation distribution list based on percentages"""
442
+ operations = []
443
+ for op_name, percentage in self.config.operations.items():
444
+ count = int(DEFAULT_OP_DISTRIBUTION_BASE * percentage / 100)
445
+ operations.extend([op_name] * count)
446
+ random.shuffle(operations)
447
+ return operations
448
+
449
+ async def _execute_operation(
450
+ self, op_name: str, test_data: TestData
451
+ ) -> Tuple[int, int, float, Optional[Exception]]:
452
+ """Execute a single operation
453
+
454
+ Returns:
455
+ Tuple of (message_count, request_count, latency, error)
456
+ """
457
+ handler = OPERATION_HANDLERS.get(op_name)
458
+ if not handler:
459
+ logger.warning("Unknown operation: %s", op_name)
460
+ return 0, 0, 0.0, ValueError("Unknown operation")
461
+
462
+ try:
463
+ msg = handler.create_message(self, test_data)
464
+ socket_type = handler.socket_type
465
+
466
+ if socket_type == SocketType.HEARTBEAT:
467
+ latency, _ = await self.send_heartbeat(msg)
468
+ elif socket_type == SocketType.DEALER:
469
+ latency, _ = await self.send_request(msg)
470
+ else: # SocketType.PUSH
471
+ msg_start = time.time()
472
+ await self.send_messages([msg])
473
+ latency = time.time() - msg_start
474
+ return handler.get_message_count(self), 1, latency, None
475
+ except Exception as e:
476
+ logger.error("Error in %s: %s", op_name, e)
477
+ return 0, 0, 0.0, e
478
+
479
+ async def run_benchmark(self):
480
+ """Run the main benchmark"""
481
+ await self.setup()
482
+
483
+ try:
484
+ test_data = self.generate_test_data()
485
+
486
+ # Pre-register workers
487
+ await self.register_workers(test_data)
488
+
489
+ # Build operation distribution
490
+ operations = self._build_operation_distribution()
491
+
492
+ # Initialize tracking
493
+ latencies: Dict[str, List[float]] = {
494
+ op: [] for op in self.config.operations.keys()
495
+ }
496
+ errors: Dict[str, int] = {op: 0 for op in self.config.operations.keys()}
497
+ message_counts: Dict[str, int] = {
498
+ op: 0 for op in self.config.operations.keys()
499
+ }
500
+ request_counts: Dict[str, int] = {
501
+ op: 0 for op in self.config.operations.keys()
502
+ }
503
+ total_messages = 0
504
+ total_requests = 0
505
+
506
+ # Start monitoring
507
+ self.running = True
508
+ monitoring_task = asyncio.create_task(self.monitor_system())
509
+
510
+ start_time = time.time()
511
+ op_index = 0
512
+
513
+ logger.info("Starting benchmark for %d seconds...", self.config.duration)
514
+
515
+ while time.time() - start_time < self.config.duration:
516
+ # Get next operation
517
+ op_name = operations[op_index % len(operations)]
518
+ op_index += 1
519
+
520
+ msg_count, req_count, latency, error = await self._execute_operation(
521
+ op_name, test_data
522
+ )
523
+ total_messages += msg_count
524
+ total_requests += req_count
525
+ if error:
526
+ errors[op_name] += 1
527
+ else:
528
+ latencies[op_name].append(latency)
529
+ message_counts[op_name] += msg_count
530
+ request_counts[op_name] += req_count
531
+
532
+ # Small yield to prevent blocking
533
+ if op_index % 100 == 0:
534
+ await asyncio.sleep(0)
535
+
536
+ # Stop monitoring
537
+ self.running = False
538
+ monitoring_task.cancel()
539
+ try:
540
+ await monitoring_task
541
+ except asyncio.CancelledError:
542
+ pass
543
+
544
+ # Calculate results
545
+ total_time = time.time() - start_time
546
+ overall_qps = total_messages / total_time if total_time > 0 else 0
547
+ overall_rps = total_requests / total_time if total_time > 0 else 0
548
+
549
+ self.results.total_messages = total_messages
550
+ self.results.total_requests = total_requests
551
+ self.results.total_time = total_time
552
+ self.results.overall_qps = overall_qps
553
+ self.results.overall_rps = overall_rps
554
+
555
+ # Per-operation stats
556
+ for op_name in self.config.operations.keys():
557
+ if latencies[op_name]:
558
+ op_qps = (
559
+ message_counts[op_name] / total_time if total_time > 0 else 0
560
+ )
561
+ op_rps = (
562
+ request_counts[op_name] / total_time if total_time > 0 else 0
563
+ )
564
+ avg_latency = statistics.mean(latencies[op_name])
565
+
566
+ self.results.operations[op_name] = OperationStats(
567
+ qps=op_qps,
568
+ rps=op_rps,
569
+ avg_latency=avg_latency,
570
+ min_latency=min(latencies[op_name]),
571
+ max_latency=max(latencies[op_name]),
572
+ p95_latency=(
573
+ statistics.quantiles(latencies[op_name], n=20)[18]
574
+ if len(latencies[op_name]) >= 20
575
+ else max(latencies[op_name])
576
+ ),
577
+ errors=errors[op_name],
578
+ )
579
+
580
+ # Deregister workers
581
+ await self.deregister_workers()
582
+
583
+ finally:
584
+ self.cleanup()
585
+
586
+ async def monitor_system(self):
587
+ """Monitor system metrics during benchmark"""
588
+ while self.running:
589
+ try:
590
+ memory_usage = psutil.virtual_memory().percent
591
+ self.results.memory_usage.append(memory_usage)
592
+ except Exception as e:
593
+ logger.warning("Failed to get memory usage: %s", e)
594
+ await asyncio.sleep(1)
595
+
596
+ def print_results(self):
597
+ """Print benchmark results"""
598
+ print("\n" + "=" * 80)
599
+ if self.config.num_processes > 1:
600
+ print(
601
+ "LMCache Controller ZMQ Benchmark Results (Process %d/%d)"
602
+ % (self.config.process_id + 1, self.config.num_processes)
603
+ )
604
+ else:
605
+ print("LMCache Controller ZMQ Benchmark Results")
606
+ print("=" * 80)
607
+
608
+ print("\nConfiguration:")
609
+ print(" Controller URL: %s" % self.config.controller_pull_url)
610
+ print(" Duration: %d seconds" % self.config.duration)
611
+ print(" Batch Size: %d" % self.config.batch_size)
612
+ print(" Operations: %s" % self.config.operations)
613
+ print(
614
+ " Instances: %d, Workers: %d, Locations: %d, Keys: %d"
615
+ % (
616
+ self.config.num_instances,
617
+ self.config.num_workers,
618
+ self.config.num_locations,
619
+ self.config.num_keys,
620
+ )
621
+ )
622
+
623
+ print("\nOverall Performance:")
624
+ print(" Total Requests: %d" % self.results.total_requests)
625
+ print(" Total Messages: %d" % self.results.total_messages)
626
+ print(" Total Time: %.2fs" % self.results.total_time)
627
+ print(" Overall RPS (Requests/sec): %.2f" % self.results.overall_rps)
628
+ print(" Overall QPS (Messages/sec): %.2f" % self.results.overall_qps)
629
+
630
+ print("\nPer-Operation Performance:")
631
+ for op_name in self.config.operations.keys():
632
+ if op_name in self.results.operations:
633
+ stats = self.results.operations[op_name]
634
+ print(" %s:" % op_name)
635
+ print(" RPS (Requests/sec): %.2f" % stats.rps)
636
+ print(" QPS (Messages/sec): %.2f" % stats.qps)
637
+ print(
638
+ " Latency - Avg: %.3fms, Min: %.3fms, Max: %.3fms, P95: %.3fms"
639
+ % (
640
+ stats.avg_latency * 1000,
641
+ stats.min_latency * 1000,
642
+ stats.max_latency * 1000,
643
+ stats.p95_latency * 1000,
644
+ )
645
+ )
646
+ print(" Errors: %d" % stats.errors)
647
+
648
+ print("\nSystem Metrics:")
649
+ if self.results.memory_usage:
650
+ avg_memory = statistics.mean(self.results.memory_usage)
651
+ max_memory = max(self.results.memory_usage)
652
+ print(
653
+ " Memory Usage - Avg: %.1f%%, Max: %.1f%%" % (avg_memory, max_memory)
654
+ )
655
+
656
+ print("=" * 80)
657
+
658
+ def get_results(self) -> BenchmarkResults:
659
+ """Return benchmark results for aggregation"""
660
+ return self.results
@@ -0,0 +1,44 @@
1
+ """Configuration for LMCache Controller ZMQ Benchmark"""
2
+
3
+ # SPDX-License-Identifier: Apache-2.0
4
+
5
+ # Standard
6
+ from dataclasses import dataclass, field
7
+ from typing import Dict, Optional
8
+
9
+
10
+ @dataclass
11
+ class ZMQBenchmarkConfig:
12
+ """Configuration for ZMQ benchmark parameters"""
13
+
14
+ controller_pull_url: str
15
+ controller_reply_url: Optional[str]
16
+ duration: int
17
+ batch_size: int
18
+ num_instances: int
19
+ num_workers: int
20
+ num_locations: int
21
+ num_keys: int
22
+ controller_heartbeat_url: Optional[str] = None
23
+ num_hashes: int = 100
24
+ operations: Dict[str, float] = field(default_factory=dict)
25
+ heartbeat_interval: float = 1.0
26
+ register_first: bool = True
27
+ # Multi-process settings
28
+ num_processes: int = 1
29
+ process_id: int = 0
30
+
31
+ def __post_init__(self):
32
+ if not self.operations:
33
+ # Default: 70% admit, 25% evict, 5% heartbeat
34
+ self.operations = {
35
+ "admit": 70.0,
36
+ "evict": 25.0,
37
+ "heartbeat": 5.0,
38
+ }
39
+ # Validate operation percentages sum to 100
40
+ total_percentage = sum(self.operations.values())
41
+ if abs(total_percentage - 100.0) > 0.01:
42
+ raise ValueError(
43
+ "Operation percentages must sum to 100, got: %s" % total_percentage
44
+ )