lmcache-cli 0.4.5.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (399) hide show
  1. lmcache/__init__.py +84 -0
  2. lmcache/_version.py +24 -0
  3. lmcache/cli/__init__.py +1 -0
  4. lmcache/cli/commands/__init__.py +34 -0
  5. lmcache/cli/commands/base.py +157 -0
  6. lmcache/cli/commands/bench/__init__.py +557 -0
  7. lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
  8. lmcache/cli/commands/bench/engine_bench/config.py +245 -0
  9. lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
  10. lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
  11. lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
  12. lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
  13. lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
  14. lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
  15. lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
  16. lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
  17. lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
  18. lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
  19. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
  20. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
  21. lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
  22. lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
  23. lmcache/cli/commands/describe.py +310 -0
  24. lmcache/cli/commands/kvcache.py +133 -0
  25. lmcache/cli/commands/mock.py +75 -0
  26. lmcache/cli/commands/ping.py +113 -0
  27. lmcache/cli/commands/query/__init__.py +155 -0
  28. lmcache/cli/commands/query/prompt.py +134 -0
  29. lmcache/cli/commands/query/request.py +357 -0
  30. lmcache/cli/commands/server.py +99 -0
  31. lmcache/cli/commands/tool/__init__.py +63 -0
  32. lmcache/cli/commands/tool/cache_simulator.py +113 -0
  33. lmcache/cli/commands/trace/__init__.py +505 -0
  34. lmcache/cli/commands/trace/dispatch.py +249 -0
  35. lmcache/cli/commands/trace/driver.py +372 -0
  36. lmcache/cli/commands/trace/stats.py +289 -0
  37. lmcache/cli/documents/lmcache.txt +11 -0
  38. lmcache/cli/main.py +42 -0
  39. lmcache/cli/metrics/__init__.py +29 -0
  40. lmcache/cli/metrics/formatter.py +171 -0
  41. lmcache/cli/metrics/handler.py +94 -0
  42. lmcache/cli/metrics/metrics.py +161 -0
  43. lmcache/cli/metrics/section.py +77 -0
  44. lmcache/connections.py +173 -0
  45. lmcache/integration/__init__.py +2 -0
  46. lmcache/integration/base_service_factory.py +165 -0
  47. lmcache/integration/request_telemetry/__init__.py +1 -0
  48. lmcache/integration/request_telemetry/base.py +51 -0
  49. lmcache/integration/request_telemetry/factory.py +113 -0
  50. lmcache/integration/request_telemetry/fastapi.py +109 -0
  51. lmcache/integration/request_telemetry/noop.py +35 -0
  52. lmcache/integration/sglang/__init__.py +2 -0
  53. lmcache/integration/sglang/sglang_adapter.py +326 -0
  54. lmcache/integration/sglang/utils.py +39 -0
  55. lmcache/integration/vllm/__init__.py +1 -0
  56. lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
  57. lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
  58. lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
  59. lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
  60. lmcache/integration/vllm/utils.py +433 -0
  61. lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
  62. lmcache/integration/vllm/vllm_service_factory.py +339 -0
  63. lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
  64. lmcache/logging.py +107 -0
  65. lmcache/native_storage_ops.pyi +230 -0
  66. lmcache/non_cuda_equivalents.py +1424 -0
  67. lmcache/observability.py +1958 -0
  68. lmcache/storage_backend/serde/__init__.py +1 -0
  69. lmcache/storage_backend/serde/cachegen_basics.py +210 -0
  70. lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
  71. lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
  72. lmcache/storage_backend/serde/serde.py +75 -0
  73. lmcache/tools/__init__.py +1 -0
  74. lmcache/tools/cache_simulator/README.md +392 -0
  75. lmcache/tools/cache_simulator/__init__.py +1 -0
  76. lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
  77. lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
  78. lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
  79. lmcache/tools/cache_simulator/lru_cache.py +124 -0
  80. lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
  81. lmcache/tools/cache_simulator/simulator.py +795 -0
  82. lmcache/tools/controller_benchmark/README.md +161 -0
  83. lmcache/tools/controller_benchmark/__init__.py +1 -0
  84. lmcache/tools/controller_benchmark/__main__.py +331 -0
  85. lmcache/tools/controller_benchmark/benchmark.py +660 -0
  86. lmcache/tools/controller_benchmark/config.py +44 -0
  87. lmcache/tools/controller_benchmark/constants.py +10 -0
  88. lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
  89. lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
  90. lmcache/tools/controller_benchmark/handlers/base.py +47 -0
  91. lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
  92. lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
  93. lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
  94. lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
  95. lmcache/tools/controller_benchmark/handlers/register.py +56 -0
  96. lmcache/tools/mp_status_viewer/__init__.py +1 -0
  97. lmcache/tools/mp_status_viewer/__main__.py +95 -0
  98. lmcache/usage_context.py +417 -0
  99. lmcache/utils.py +665 -0
  100. lmcache/v1/__init__.py +2 -0
  101. lmcache/v1/api_server/__init__.py +2 -0
  102. lmcache/v1/api_server/__main__.py +537 -0
  103. lmcache/v1/basic_check.py +112 -0
  104. lmcache/v1/cache_controller/__init__.py +9 -0
  105. lmcache/v1/cache_controller/commands/__init__.py +15 -0
  106. lmcache/v1/cache_controller/commands/base.py +35 -0
  107. lmcache/v1/cache_controller/commands/full_sync.py +49 -0
  108. lmcache/v1/cache_controller/config.py +176 -0
  109. lmcache/v1/cache_controller/controller_manager.py +535 -0
  110. lmcache/v1/cache_controller/controllers/__init__.py +11 -0
  111. lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
  112. lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
  113. lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
  114. lmcache/v1/cache_controller/executor.py +463 -0
  115. lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
  116. lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
  117. lmcache/v1/cache_controller/frontend/static/index.html +234 -0
  118. lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
  119. lmcache/v1/cache_controller/full_sync_sender.py +475 -0
  120. lmcache/v1/cache_controller/locks.py +149 -0
  121. lmcache/v1/cache_controller/message.py +828 -0
  122. lmcache/v1/cache_controller/observability.py +208 -0
  123. lmcache/v1/cache_controller/utils.py +679 -0
  124. lmcache/v1/cache_controller/worker.py +665 -0
  125. lmcache/v1/cache_engine.py +2058 -0
  126. lmcache/v1/cache_interface.py +19 -0
  127. lmcache/v1/check/__init__.py +74 -0
  128. lmcache/v1/check/check_mode_gen.py +86 -0
  129. lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
  130. lmcache/v1/check/check_mode_test_remote.py +155 -0
  131. lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
  132. lmcache/v1/check/utils.py +571 -0
  133. lmcache/v1/compute/__init__.py +2 -0
  134. lmcache/v1/compute/attention/__init__.py +0 -0
  135. lmcache/v1/compute/attention/abstract.py +39 -0
  136. lmcache/v1/compute/attention/flash_attn.py +129 -0
  137. lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
  138. lmcache/v1/compute/attention/metadata.py +85 -0
  139. lmcache/v1/compute/attention/utils.py +14 -0
  140. lmcache/v1/compute/blend/__init__.py +7 -0
  141. lmcache/v1/compute/blend/blender.py +168 -0
  142. lmcache/v1/compute/blend/metadata.py +34 -0
  143. lmcache/v1/compute/blend/utils.py +63 -0
  144. lmcache/v1/compute/models/__init__.py +0 -0
  145. lmcache/v1/compute/models/base.py +141 -0
  146. lmcache/v1/compute/models/llama.py +9 -0
  147. lmcache/v1/compute/models/qwen3.py +24 -0
  148. lmcache/v1/compute/models/utils.py +68 -0
  149. lmcache/v1/compute/positional_encoding.py +199 -0
  150. lmcache/v1/config.py +848 -0
  151. lmcache/v1/config_base.py +848 -0
  152. lmcache/v1/distributed/api.py +248 -0
  153. lmcache/v1/distributed/config.py +321 -0
  154. lmcache/v1/distributed/error.py +64 -0
  155. lmcache/v1/distributed/eviction.py +192 -0
  156. lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
  157. lmcache/v1/distributed/eviction_policy/factory.py +27 -0
  158. lmcache/v1/distributed/eviction_policy/lru.py +244 -0
  159. lmcache/v1/distributed/eviction_policy/noop.py +50 -0
  160. lmcache/v1/distributed/internal_api.py +170 -0
  161. lmcache/v1/distributed/l1_manager.py +835 -0
  162. lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
  163. lmcache/v1/distributed/l2_adapters/base.py +360 -0
  164. lmcache/v1/distributed/l2_adapters/config.py +385 -0
  165. lmcache/v1/distributed/l2_adapters/factory.py +205 -0
  166. lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
  167. lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
  168. lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
  169. lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
  170. lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
  171. lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
  172. lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
  173. lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
  174. lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
  175. lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
  176. lmcache/v1/distributed/memory_manager.py +179 -0
  177. lmcache/v1/distributed/storage_controller.py +39 -0
  178. lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
  179. lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
  180. lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
  181. lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
  182. lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
  183. lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
  184. lmcache/v1/distributed/storage_manager.py +532 -0
  185. lmcache/v1/event_manager.py +145 -0
  186. lmcache/v1/exceptions/__init__.py +16 -0
  187. lmcache/v1/gpu_connector/__init__.py +126 -0
  188. lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
  189. lmcache/v1/gpu_connector/gpu_ops.py +85 -0
  190. lmcache/v1/gpu_connector/hpu_connector.py +326 -0
  191. lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
  192. lmcache/v1/gpu_connector/utils.py +890 -0
  193. lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
  194. lmcache/v1/health_monitor/__init__.py +1 -0
  195. lmcache/v1/health_monitor/base.py +587 -0
  196. lmcache/v1/health_monitor/checks/__init__.py +1 -0
  197. lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
  198. lmcache/v1/health_monitor/constants.py +36 -0
  199. lmcache/v1/internal_api_server/__init__.py +0 -0
  200. lmcache/v1/internal_api_server/api_registry.py +59 -0
  201. lmcache/v1/internal_api_server/api_server.py +120 -0
  202. lmcache/v1/internal_api_server/common/__init__.py +1 -0
  203. lmcache/v1/internal_api_server/common/env_api.py +22 -0
  204. lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
  205. lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
  206. lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
  207. lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
  208. lmcache/v1/internal_api_server/common/thread_api.py +63 -0
  209. lmcache/v1/internal_api_server/controller/__init__.py +1 -0
  210. lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
  211. lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
  212. lmcache/v1/internal_api_server/utils.py +43 -0
  213. lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
  214. lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
  215. lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
  216. lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
  217. lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
  218. lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
  219. lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
  220. lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
  221. lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
  222. lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
  223. lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
  224. lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
  225. lmcache/v1/kv_layer_groups.py +267 -0
  226. lmcache/v1/lazy_memory_allocator.py +284 -0
  227. lmcache/v1/lookup_client/__init__.py +25 -0
  228. lmcache/v1/lookup_client/abstract_client.py +77 -0
  229. lmcache/v1/lookup_client/async_lookup_message.py +50 -0
  230. lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
  231. lmcache/v1/lookup_client/factory.py +251 -0
  232. lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
  233. lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
  234. lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
  235. lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
  236. lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
  237. lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
  238. lmcache/v1/lookup_client/record_strategies/base.py +327 -0
  239. lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
  240. lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
  241. lmcache/v1/manager.py +539 -0
  242. lmcache/v1/memory_management.py +2619 -0
  243. lmcache/v1/metadata.py +114 -0
  244. lmcache/v1/mp_observability/AGENTS.override.md +21 -0
  245. lmcache/v1/mp_observability/README.md +204 -0
  246. lmcache/v1/mp_observability/config.py +340 -0
  247. lmcache/v1/mp_observability/event.py +100 -0
  248. lmcache/v1/mp_observability/event_bus.py +313 -0
  249. lmcache/v1/mp_observability/otel_init.py +145 -0
  250. lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
  251. lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
  252. lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
  253. lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
  254. lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
  255. lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
  256. lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
  257. lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
  258. lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
  259. lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
  260. lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
  261. lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
  262. lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
  263. lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
  264. lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
  265. lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
  266. lmcache/v1/mp_observability/trace/__init__.py +50 -0
  267. lmcache/v1/mp_observability/trace/codecs.py +255 -0
  268. lmcache/v1/mp_observability/trace/decorator.py +147 -0
  269. lmcache/v1/mp_observability/trace/format.py +132 -0
  270. lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
  271. lmcache/v1/mp_observability/trace/reader.py +167 -0
  272. lmcache/v1/mp_observability/trace/recorder.py +300 -0
  273. lmcache/v1/multiprocess/__init__.py +0 -0
  274. lmcache/v1/multiprocess/affinity_pool.py +102 -0
  275. lmcache/v1/multiprocess/blend_server_v2.py +891 -0
  276. lmcache/v1/multiprocess/config.py +253 -0
  277. lmcache/v1/multiprocess/custom_types.py +281 -0
  278. lmcache/v1/multiprocess/futures.py +194 -0
  279. lmcache/v1/multiprocess/gpu_context.py +511 -0
  280. lmcache/v1/multiprocess/http_server.py +235 -0
  281. lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
  282. lmcache/v1/multiprocess/mq.py +732 -0
  283. lmcache/v1/multiprocess/protocol.py +86 -0
  284. lmcache/v1/multiprocess/protocols/README.md +213 -0
  285. lmcache/v1/multiprocess/protocols/__init__.py +127 -0
  286. lmcache/v1/multiprocess/protocols/base.py +89 -0
  287. lmcache/v1/multiprocess/protocols/blend.py +109 -0
  288. lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
  289. lmcache/v1/multiprocess/protocols/controller.py +53 -0
  290. lmcache/v1/multiprocess/protocols/debug.py +34 -0
  291. lmcache/v1/multiprocess/protocols/engine.py +146 -0
  292. lmcache/v1/multiprocess/protocols/observability.py +39 -0
  293. lmcache/v1/multiprocess/server.py +1134 -0
  294. lmcache/v1/multiprocess/session.py +190 -0
  295. lmcache/v1/multiprocess/token_hasher.py +441 -0
  296. lmcache/v1/offload_server/__init__.py +17 -0
  297. lmcache/v1/offload_server/abstract_server.py +37 -0
  298. lmcache/v1/offload_server/message.py +30 -0
  299. lmcache/v1/offload_server/zmq_server.py +122 -0
  300. lmcache/v1/periodic_thread.py +579 -0
  301. lmcache/v1/pin_monitor.py +246 -0
  302. lmcache/v1/plugin/__init__.py +0 -0
  303. lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
  304. lmcache/v1/protocol.py +317 -0
  305. lmcache/v1/rpc/__init__.py +17 -0
  306. lmcache/v1/rpc/transport.py +105 -0
  307. lmcache/v1/rpc/zmq_transport.py +213 -0
  308. lmcache/v1/rpc_utils.py +165 -0
  309. lmcache/v1/server/__init__.py +2 -0
  310. lmcache/v1/server/__main__.py +170 -0
  311. lmcache/v1/server/storage_backend/__init__.py +21 -0
  312. lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
  313. lmcache/v1/server/storage_backend/local_backend.py +75 -0
  314. lmcache/v1/server/utils.py +21 -0
  315. lmcache/v1/standalone/__init__.py +1 -0
  316. lmcache/v1/standalone/__main__.py +583 -0
  317. lmcache/v1/standalone/manager.py +80 -0
  318. lmcache/v1/standalone/standalone_service_factory.py +86 -0
  319. lmcache/v1/storage_backend/__init__.py +313 -0
  320. lmcache/v1/storage_backend/abstract_backend.py +445 -0
  321. lmcache/v1/storage_backend/audit_backend.py +233 -0
  322. lmcache/v1/storage_backend/batched_message_sender.py +222 -0
  323. lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
  324. lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
  325. lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
  326. lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
  327. lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
  328. lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
  329. lmcache/v1/storage_backend/connector/__init__.py +443 -0
  330. lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
  331. lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
  332. lmcache/v1/storage_backend/connector/base_connector.py +379 -0
  333. lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
  334. lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
  335. lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
  336. lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
  337. lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
  338. lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
  339. lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
  340. lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
  341. lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
  342. lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
  343. lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
  344. lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
  345. lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
  346. lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
  347. lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
  348. lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
  349. lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
  350. lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
  351. lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
  352. lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
  353. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
  354. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
  355. lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
  356. lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
  357. lmcache/v1/storage_backend/gds_backend.py +1199 -0
  358. lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
  359. lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
  360. lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
  361. lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
  362. lmcache/v1/storage_backend/local_disk_backend.py +656 -0
  363. lmcache/v1/storage_backend/maru_backend.py +734 -0
  364. lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
  365. lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
  366. lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
  367. lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
  368. lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
  369. lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
  370. lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
  371. lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
  372. lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
  373. lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
  374. lmcache/v1/storage_backend/p2p_backend.py +788 -0
  375. lmcache/v1/storage_backend/path_sharder.py +117 -0
  376. lmcache/v1/storage_backend/pd_backend.py +646 -0
  377. lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
  378. lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
  379. lmcache/v1/storage_backend/remote_backend.py +624 -0
  380. lmcache/v1/storage_backend/resp_client.py +227 -0
  381. lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
  382. lmcache/v1/storage_backend/storage_manager.py +1352 -0
  383. lmcache/v1/system_detection.py +110 -0
  384. lmcache/v1/token_database.py +551 -0
  385. lmcache/v1/transfer_channel/__init__.py +83 -0
  386. lmcache/v1/transfer_channel/abstract.py +285 -0
  387. lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
  388. lmcache/v1/transfer_channel/nixl_channel.py +639 -0
  389. lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
  390. lmcache/v1/transfer_channel/transfer_utils.py +63 -0
  391. lmcache/v1/utils/__init__.py +1 -0
  392. lmcache/v1/utils/bloom_filter.py +109 -0
  393. lmcache/v1/utils/cache_utils.py +125 -0
  394. lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
  395. lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
  396. lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
  397. lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
  398. lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
  399. lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
lmcache/v1/metadata.py ADDED
@@ -0,0 +1,114 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Standard
3
+ from dataclasses import dataclass, field
4
+ from typing import Optional
5
+
6
+ # Third Party
7
+ import torch
8
+
9
+ # First Party
10
+ from lmcache.logging import init_logger
11
+ from lmcache.v1.kv_layer_groups import KVLayerGroupsManager
12
+
13
+ logger = init_logger(__name__)
14
+
15
+
16
+ @dataclass
17
+ class LMCacheMetadata:
18
+ """
19
+ LMCacheMetadata should be extracted from the northbound
20
+ serving engine configuration and wrap the extraction of
21
+ attributes (e.g. model name, tp rank, etc.)
22
+ """
23
+
24
+ """name of the LLM model"""
25
+ model_name: str
26
+ """ global world size when running under a distributed setting
27
+ (total number of workers)"""
28
+ world_size: int
29
+ """ host world size (workers on active localhost)
30
+ This information can be useful for multi-node
31
+ deployment. Will be the same as world_size
32
+ in single-node deployments.
33
+ """
34
+ local_world_size: int
35
+ """ worker id when running under a distributed setting """
36
+ worker_id: int
37
+ """ host worker id (a gpu bound worker id on active localhost)
38
+ This information can be useful for multi-node deployment.
39
+ Will be the same as worker_id in single-node deployments.
40
+ """
41
+ local_worker_id: int
42
+ """ the data type of kv tensors """
43
+ # (Deprecated) Will be replaced by kv_layer_groups_manager in the future
44
+ kv_dtype: torch.dtype
45
+ """ the shape of kv tensors """
46
+ # (Deprecated) Will be replaced by kv_layer_groups_manager in the future
47
+ """ (num_layer, 2, chunk_size, num_kv_head, head_size) """
48
+ kv_shape: tuple[int, int, int, int, int]
49
+ """ whether use MLA"""
50
+ use_mla: bool = False
51
+ """ the role of the current instance (e.g., 'scheduler', 'worker') """
52
+ role: Optional[str] = None
53
+ """ the first rank of the distributed setting """
54
+ # TODO(baoloongmao): first_rank should be configurable
55
+ first_rank = 0
56
+ served_model_name: Optional[str] = None
57
+ """chunk size"""
58
+ chunk_size: int = 256
59
+ """ Manager for groups of layers with identical KV cache structure """
60
+ kv_layer_groups_manager: KVLayerGroupsManager = field(
61
+ default_factory=KVLayerGroupsManager
62
+ )
63
+ """ engine_id for RPC path (used by lookup client/server) """
64
+ engine_id: Optional[str] = None
65
+ """ extra config from kv_connector (e.g., lmcache_rpc_port) """
66
+ kv_connector_extra_config: Optional[dict] = None
67
+
68
+ def is_first_rank(self) -> bool:
69
+ """Check if the current worker is the first rank"""
70
+ return self.worker_id == self.first_rank
71
+
72
+ # TODO(chunxiaozheng): some uts do not `build_kv_layer_groups`
73
+ def get_dtypes(self) -> list[torch.dtype]:
74
+ if self.kv_layer_groups_manager.kv_layer_groups:
75
+ return [
76
+ group.dtype for group in self.kv_layer_groups_manager.kv_layer_groups
77
+ ]
78
+ return [self.kv_dtype]
79
+
80
+ def get_shapes(self, num_tokens: Optional[int] = None) -> list[torch.Size]:
81
+ """Get the shapes of the KV cache in LMCache"""
82
+ if num_tokens is None:
83
+ num_tokens = self.chunk_size
84
+ if self.kv_layer_groups_manager.kv_layer_groups:
85
+ shapes = []
86
+ kv_size = 1 if self.use_mla else 2
87
+ for group in self.kv_layer_groups_manager.kv_layer_groups:
88
+ shapes.append(
89
+ torch.Size(
90
+ [
91
+ kv_size,
92
+ group.num_layers,
93
+ num_tokens,
94
+ group.hidden_dim_size,
95
+ ]
96
+ )
97
+ )
98
+ return shapes
99
+ else:
100
+ return [
101
+ torch.Size(
102
+ [
103
+ self.kv_shape[1],
104
+ self.kv_shape[0],
105
+ num_tokens,
106
+ self.kv_shape[3] * self.kv_shape[4],
107
+ ]
108
+ )
109
+ ]
110
+
111
+ def get_num_groups(self) -> int:
112
+ if self.kv_layer_groups_manager.kv_layer_groups:
113
+ return self.kv_layer_groups_manager.num_groups
114
+ return 1
@@ -0,0 +1,21 @@
1
+ # MP Observability — Agent Instructions
2
+
3
+ When working in `lmcache/v1/mp_observability/`:
4
+
5
+ Design docs live in `docs/design/v1/mp_observability/` — update them whenever
6
+ the contracts below change.
7
+
8
+ 1. **New `EventType`** — after adding an entry to `event.py`, update the
9
+ metadata contract table in `docs/design/v1/mp_observability/EVENTS.md`
10
+ with the new type's metadata keys and types.
11
+
12
+ 2. **New metrics subscriber** — after adding counters/histograms, update the
13
+ metrics table in `docs/design/v1/mp_observability/METRICS.md` with the
14
+ metric name, type, and description.
15
+
16
+ 3. **New subscriber class** — follow the step-by-step guide in `README.md`
17
+ (co-located with this file; "How to Add a New Event and Subscriber") and
18
+ the design rules in `docs/design/v1/mp_observability/event-bus.md`.
19
+
20
+ 4. **CLI args** — if you add or change observability CLI flags in `config.py`,
21
+ update `docs/source/mp/observability.rst` to match.
@@ -0,0 +1,204 @@
1
+ # MP Observability
2
+
3
+ Event-driven observability for LMCache's multiprocess (MP) mode, built on
4
+ [OpenTelemetry](https://opentelemetry.io/).
5
+
6
+ For metrics, see [METRICS.md](../../../docs/design/v1/mp_observability/METRICS.md).
7
+ For event metadata contracts, see
8
+ [EVENTS.md](../../../docs/design/v1/mp_observability/EVENTS.md).
9
+ For design rationale, see
10
+ [event-bus.md](../../../docs/design/v1/mp_observability/event-bus.md).
11
+ For the trace recording subsystem (`lmcache trace`), see
12
+ [trace.md](../../../docs/design/v1/mp_observability/trace.md).
13
+
14
+ ---
15
+
16
+ ## Architecture
17
+
18
+ ```
19
+ Producers (L1Manager, StorageManager, MPCacheEngine)
20
+
21
+ │ event_bus.publish(Event(...))
22
+
23
+ EventBus (async queue + drain thread)
24
+
25
+ ├──► L1MetricsSubscriber → OTel counter.add(...)
26
+ ├──► SMMetricsSubscriber → OTel counter.add(...)
27
+ ├──► L1LoggingSubscriber → logger.debug(...)
28
+ ├──► SMLoggingSubscriber → logger.debug(...)
29
+ ├──► MPServerLoggingSubscriber → logger.debug(...)
30
+ └──► MPServerTracingSubscriber → OTel span start/end
31
+
32
+ OTel SDK (configured at startup)
33
+
34
+ ├──► OTLP push (production) → OTel collector → Prometheus / Grafana / etc.
35
+ └──► Prometheus pull (dev/debug) → /metrics on configured port
36
+ ```
37
+
38
+ ---
39
+
40
+ ## Configuration
41
+
42
+ All observability behaviour is controlled by `ObservabilityConfig`
43
+ (defined in `config.py`). When running the LMCache MP mode server from the
44
+ CLI, pass the flags below; when embedding programmatically, construct an
45
+ `ObservabilityConfig` directly.
46
+
47
+ ### CLI flags
48
+
49
+ | Flag | Default | Description |
50
+ |---|---|---|
51
+ | `--disable-observability` | off | Disable the EventBus entirely. No events are published or consumed. |
52
+ | `--disable-metrics` | off | Skip registering metrics subscribers (OTel counters). |
53
+ | `--disable-logging` | off | Skip registering logging subscribers. |
54
+ | `--enable-tracing` | off | Register tracing subscribers (OTel spans). Disabled by default. **Requires `--otlp-endpoint`.** |
55
+ | `--event-bus-queue-size N` | `10000` | Maximum number of events in the EventBus queue before tail-drop. |
56
+ | `--otlp-endpoint URL` | *(none)* | OTLP gRPC endpoint (e.g. `http://localhost:4317`). When set, metrics and traces are pushed to an OTel collector. When unset, metrics fall back to Prometheus pull mode. |
57
+ | `--prometheus-port PORT` | `9090` | Port for the Prometheus `/metrics` endpoint. Only used when `--otlp-endpoint` is not set. |
58
+
59
+ ### `ObservabilityConfig` fields
60
+
61
+ | Field | Type | Default | Description |
62
+ |---|---|---|---|
63
+ | `enabled` | `bool` | `True` | Master switch for the EventBus. |
64
+ | `max_queue_size` | `int` | `10000` | Maximum events in the EventBus queue before tail-drop. |
65
+ | `metrics_enabled` | `bool` | `True` | Register metrics subscribers (OTel counters / histograms). |
66
+ | `logging_enabled` | `bool` | `True` | Register logging subscribers. |
67
+ | `tracing_enabled` | `bool` | `False` | Register tracing subscribers (OTel spans). |
68
+ | `otlp_endpoint` | `str \| None` | `None` | OTLP gRPC endpoint. When set, metrics and traces are pushed. When `None`, metrics use Prometheus pull fallback. |
69
+ | `prometheus_port` | `int` | `9090` | Port for the Prometheus `/metrics` endpoint (pull fallback only). |
70
+
71
+ ### Metrics export modes
72
+
73
+ | `otlp_endpoint` | Mode | How to query |
74
+ |---|---|---|
75
+ | `http://host:4317` | OTLP push | Query the OTel collector's Prometheus exporter |
76
+ | `None` | Prometheus pull fallback | `curl http://localhost:<prometheus-port>/metrics` |
77
+
78
+ > **Note:** OTel counters only appear on `/metrics` after the first increment.
79
+ > If you see only Python runtime metrics, trigger a store/retrieve first.
80
+
81
+ ### Tracing
82
+
83
+ Tracing is opt-in (`--enable-tracing`). When enabled, `MPServerTracingSubscriber`
84
+ creates OTel spans from MP server START/END event pairs (store, retrieve,
85
+ lookup/prefetch). Trace export requires an OTLP endpoint — there is no local
86
+ fallback. `--enable-tracing` **requires** `--otlp-endpoint`; the server will
87
+ raise a `ValueError` at startup if the endpoint is missing.
88
+
89
+ ---
90
+
91
+ ## How to Add a New Event and Subscriber
92
+
93
+ ### Step 1 — Define the event type
94
+
95
+ Add a new member to `EventType` in `event.py`:
96
+
97
+ ```python
98
+ class EventType(Enum):
99
+ # ... existing events ...
100
+
101
+ # My new component events
102
+ MY_COMPONENT_OPERATION = "my_component.operation"
103
+ ```
104
+
105
+ ### Step 2 — Publish the event from the producer
106
+
107
+ In your component (e.g., a manager class), publish to the EventBus:
108
+
109
+ ```python
110
+ from lmcache.v1.mp_observability.event import Event, EventType
111
+ from lmcache.v1.mp_observability.event_bus import get_event_bus
112
+
113
+ class MyComponent:
114
+ def __init__(self):
115
+ self._event_bus = get_event_bus()
116
+
117
+ def do_operation(self, keys):
118
+ # ... business logic ...
119
+
120
+ self._event_bus.publish(Event(
121
+ event_type=EventType.MY_COMPONENT_OPERATION,
122
+ metadata={"keys": keys},
123
+ ))
124
+ ```
125
+
126
+ ### Step 3 — Create a subscriber
127
+
128
+ Create a file under the appropriate `subscribers/` subdirectory:
129
+
130
+ - `subscribers/metrics/` for OTel counters / histograms
131
+ - `subscribers/logging/` for debug log output
132
+ - `subscribers/tracing/` for OTel spans
133
+
134
+ Example metrics subscriber (`subscribers/metrics/my_component.py`):
135
+
136
+ ```python
137
+ from opentelemetry import metrics
138
+ from lmcache.v1.mp_observability.event import Event, EventType
139
+ from lmcache.v1.mp_observability.event_bus import EventCallback, EventSubscriber
140
+
141
+
142
+ class MyComponentMetricsSubscriber(EventSubscriber):
143
+ def __init__(self):
144
+ meter = metrics.get_meter("lmcache.my_component")
145
+ self._op_counter = meter.create_counter(
146
+ "lmcache_mp.my_component_operations",
147
+ description="Total operations on my component",
148
+ )
149
+
150
+ def get_subscriptions(self) -> dict[EventType, EventCallback]:
151
+ return {
152
+ EventType.MY_COMPONENT_OPERATION: self._on_operation,
153
+ }
154
+
155
+ def _on_operation(self, event: Event) -> None:
156
+ self._op_counter.add(len(event.metadata["keys"]))
157
+ ```
158
+
159
+ ### Step 4 — Export from `__init__.py`
160
+
161
+ Add the subscriber to the corresponding `__init__.py` so it can be
162
+ imported from the package:
163
+
164
+ ```python
165
+ # subscribers/metrics/__init__.py
166
+ from lmcache.v1.mp_observability.subscribers.metrics.my_component import (
167
+ MyComponentMetricsSubscriber,
168
+ )
169
+ ```
170
+
171
+ ### Step 5 — Register the subscriber at startup
172
+
173
+ In the server startup function (e.g., `run_cache_server()` in `server.py`),
174
+ register conditionally based on `ObservabilityConfig`:
175
+
176
+ ```python
177
+ if obs_config.metrics_enabled:
178
+ from lmcache.v1.mp_observability.subscribers.metrics import (
179
+ MyComponentMetricsSubscriber,
180
+ )
181
+ bus.register_subscriber(MyComponentMetricsSubscriber())
182
+ ```
183
+
184
+ ### Step 6 — Document the metadata contract
185
+
186
+ Add a row to the metadata contracts table in
187
+ [EVENTS.md](../../../docs/design/v1/mp_observability/EVENTS.md) so
188
+ subscribers can rely on the schema:
189
+
190
+ ```markdown
191
+ | `MY_COMPONENT_OPERATION` | `keys` | `list[ObjectKey]` |
192
+ ```
193
+
194
+ ---
195
+
196
+ ## Design rules
197
+
198
+ | Rule | Reason |
199
+ |---|---|
200
+ | Create meters and counters in `__init__()`, not at module level | `MeterProvider` must be set before `get_meter()` is called. Module-level calls happen at import time, before setup. |
201
+ | Prefix OTel metric names with `lmcache_mp.` | Keeps the MP namespace separate from `lmcache.` (the single-process engine namespace). |
202
+ | Use `metadata: dict[str, Any]` for event payloads | Flexible, no coupling between producers and subscribers. See metadata contracts in [EVENTS.md](../../../docs/design/v1/mp_observability/EVENTS.md). |
203
+ | Separate metrics, logging, and tracing subscribers | Single responsibility. Can enable/disable independently via config. |
204
+ | Store `self._event_bus = get_event_bus()` in `__init__` | Avoids calling the singleton getter on every publish. |
@@ -0,0 +1,340 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+
3
+ """
4
+ Configuration for the MP-mode observability stack.
5
+ """
6
+
7
+ # Future
8
+ from __future__ import annotations
9
+
10
+ # Standard
11
+ from dataclasses import dataclass, field
12
+ from typing import TYPE_CHECKING
13
+ import argparse
14
+
15
+ if TYPE_CHECKING:
16
+ # First Party
17
+ from lmcache.v1.mp_observability.event_bus import EventBus
18
+
19
+ # First Party
20
+ from lmcache.v1.mp_observability.subscribers.logging.lookup_hash import (
21
+ LookupHashLogConfig,
22
+ )
23
+
24
+
25
+ @dataclass
26
+ class ObservabilityConfig:
27
+ """Unified configuration for the EventBus-based observability system.
28
+
29
+ Controls the EventBus, OTel metrics/tracing pipelines, and subscriber
30
+ registration.
31
+ """
32
+
33
+ enabled: bool = True
34
+ """Master switch for the EventBus."""
35
+
36
+ max_queue_size: int = 10_000
37
+ """Maximum events in the EventBus queue before tail-drop."""
38
+
39
+ metrics_enabled: bool = True
40
+ """Register metrics subscribers (OTel counters / histograms)."""
41
+
42
+ logging_enabled: bool = True
43
+ """Register logging subscribers."""
44
+
45
+ tracing_enabled: bool = False
46
+ """Register span subscribers (OTel traces)."""
47
+
48
+ otlp_endpoint: str | None = None
49
+ """OTLP gRPC endpoint (e.g. ``http://localhost:4317``). When set,
50
+ metrics and traces are pushed to an OTel collector. When ``None``,
51
+ metrics fall back to an in-process Prometheus ``/metrics`` endpoint."""
52
+
53
+ prometheus_port: int = 9090
54
+ """Port for the Prometheus /metrics endpoint. Only used when
55
+ ``otlp_endpoint`` is ``None`` (Prometheus pull fallback)."""
56
+
57
+ metrics_sample_rate: float = 0.01
58
+ """Fraction of chunks/blocks to track for lifecycle histograms (0, 1.0].
59
+ Counters always count all events regardless of this setting."""
60
+
61
+ lookup_hash_log: LookupHashLogConfig = field(default_factory=LookupHashLogConfig)
62
+ """Configuration for lookup hash file logging. Disabled by default
63
+ (empty ``output_dir``)."""
64
+
65
+ trace_level: str | None = None
66
+ """If set, enables trace recording at the given level. Currently
67
+ only ``"storage"`` is supported. See
68
+ :mod:`lmcache.v1.mp_observability.trace` for details."""
69
+
70
+ trace_output: str | None = None
71
+ """Path to write the trace file. When :attr:`trace_level` is set
72
+ but this is ``None``, a timestamped path under ``$TMPDIR`` is
73
+ minted and logged at INFO."""
74
+
75
+
76
+ DEFAULT_OBSERVABILITY_CONFIG = ObservabilityConfig(enabled=False)
77
+
78
+
79
+ def add_observability_args(
80
+ parser: argparse.ArgumentParser,
81
+ ) -> argparse.ArgumentParser:
82
+ """Add observability configuration arguments to an existing parser.
83
+
84
+ Args:
85
+ parser: The argument parser to add arguments to.
86
+
87
+ Returns:
88
+ The same parser with observability arguments added.
89
+ """
90
+ group = parser.add_argument_group(
91
+ "Observability", "Configuration for metrics, logging, and tracing"
92
+ )
93
+ group.add_argument(
94
+ "--disable-observability",
95
+ action="store_true",
96
+ default=False,
97
+ help="Disable the observability EventBus entirely.",
98
+ )
99
+ group.add_argument(
100
+ "--disable-metrics",
101
+ action="store_true",
102
+ default=False,
103
+ help="Disable metrics subscribers (OTel counters).",
104
+ )
105
+ group.add_argument(
106
+ "--disable-logging",
107
+ action="store_true",
108
+ default=False,
109
+ help="Disable logging subscribers.",
110
+ )
111
+ group.add_argument(
112
+ "--enable-tracing",
113
+ action="store_true",
114
+ default=False,
115
+ help="Enable span subscribers (OTel traces). Disabled by default.",
116
+ )
117
+ group.add_argument(
118
+ "--otlp-endpoint",
119
+ type=str,
120
+ default=None,
121
+ help=(
122
+ "OTLP gRPC endpoint (e.g. http://localhost:4317). "
123
+ "When set, metrics/traces are pushed to an OTel collector. "
124
+ "When unset, falls back to Prometheus pull mode."
125
+ ),
126
+ )
127
+ group.add_argument(
128
+ "--event-bus-queue-size",
129
+ type=int,
130
+ default=10_000,
131
+ help=(
132
+ "Maximum number of events in the EventBus queue before "
133
+ "tail-drop. Default is 10000."
134
+ ),
135
+ )
136
+ group.add_argument(
137
+ "--prometheus-port",
138
+ type=int,
139
+ default=9090,
140
+ help=(
141
+ "Port for the Prometheus /metrics endpoint. "
142
+ "Only used when --otlp-endpoint is not set. Default is 9090."
143
+ ),
144
+ )
145
+ group.add_argument(
146
+ "--metrics-sample-rate",
147
+ type=float,
148
+ default=0.01,
149
+ help=(
150
+ "Fraction of chunks/blocks to track for lifecycle histograms "
151
+ "(0, 1.0]. Counters always count all events. Default is 0.01 (1%%)."
152
+ ),
153
+ )
154
+
155
+ # Lookup hash logging config
156
+ log_group = parser.add_argument_group(
157
+ "Lookup Hash Logging",
158
+ "Configuration for lookup hash file logging (offline analysis)",
159
+ )
160
+ log_group.add_argument(
161
+ "--lookup-hash-log-dir",
162
+ type=str,
163
+ default="",
164
+ help="Directory to write lookup hash JSONL files for offline analysis. "
165
+ "Empty string (default) disables logging.",
166
+ )
167
+ log_group.add_argument(
168
+ "--lookup-hash-log-rotation-interval",
169
+ type=int,
170
+ default=6 * 3600,
171
+ help="Time interval in seconds before rotating to a new log file. "
172
+ "Default is 21600 (6 hours).",
173
+ )
174
+ log_group.add_argument(
175
+ "--lookup-hash-log-rotation-max-size",
176
+ type=int,
177
+ default=100 * 1024 * 1024,
178
+ help="Max file size in bytes before rotating even if the time "
179
+ "interval has not elapsed. Default is 100MB (104857600).",
180
+ )
181
+ log_group.add_argument(
182
+ "--lookup-hash-log-max-files",
183
+ type=int,
184
+ default=100,
185
+ help="Max number of lookup hash log files to keep. "
186
+ "Oldest files are deleted when this limit is exceeded. Default is 100.",
187
+ )
188
+
189
+ trace_group = parser.add_argument_group(
190
+ "Trace Recording",
191
+ "Capture LMCache operations to a binary trace file for replay "
192
+ "(see `lmcache trace`).",
193
+ )
194
+ trace_group.add_argument(
195
+ "--trace-level",
196
+ type=str,
197
+ choices=["storage"],
198
+ default=None,
199
+ help="Enable trace recording at the given level. Currently only "
200
+ "'storage' is supported (records StorageManager public-API calls).",
201
+ )
202
+ trace_group.add_argument(
203
+ "--trace-output",
204
+ type=str,
205
+ default=None,
206
+ help="Path to write the trace file. Defaults to a timestamped "
207
+ "file under $TMPDIR when --trace-level is set without an explicit "
208
+ "output path.",
209
+ )
210
+
211
+ return parser
212
+
213
+
214
+ def parse_args_to_observability_config(
215
+ args: argparse.Namespace,
216
+ ) -> ObservabilityConfig:
217
+ """Convert parsed command line arguments to an ObservabilityConfig.
218
+
219
+ Args:
220
+ args: Parsed arguments from the argument parser.
221
+
222
+ Returns:
223
+ The configuration object.
224
+ """
225
+ config = ObservabilityConfig(
226
+ enabled=not args.disable_observability,
227
+ max_queue_size=args.event_bus_queue_size,
228
+ metrics_enabled=not args.disable_metrics,
229
+ logging_enabled=not args.disable_logging,
230
+ tracing_enabled=args.enable_tracing,
231
+ otlp_endpoint=args.otlp_endpoint,
232
+ prometheus_port=args.prometheus_port,
233
+ metrics_sample_rate=args.metrics_sample_rate,
234
+ lookup_hash_log=LookupHashLogConfig(
235
+ output_dir=args.lookup_hash_log_dir,
236
+ rotation_interval_sec=args.lookup_hash_log_rotation_interval,
237
+ rotation_max_size=args.lookup_hash_log_rotation_max_size,
238
+ max_files=args.lookup_hash_log_max_files,
239
+ ),
240
+ trace_level=args.trace_level,
241
+ trace_output=args.trace_output,
242
+ )
243
+
244
+ if config.tracing_enabled and config.otlp_endpoint is None:
245
+ raise ValueError(
246
+ "--enable-tracing requires --otlp-endpoint to be set. "
247
+ "Tracing needs an OTLP gRPC endpoint to export spans."
248
+ )
249
+
250
+ return config
251
+
252
+
253
+ def init_observability(obs_config: ObservabilityConfig) -> EventBus:
254
+ """Initialize OTel providers, EventBus, and register subscribers.
255
+
256
+ This is the single entry-point that every MP server calls at startup.
257
+ Returns a **started** EventBus.
258
+ """
259
+ # First Party
260
+ from lmcache.v1.mp_observability.event_bus import (
261
+ EventBusConfig,
262
+ init_event_bus,
263
+ )
264
+
265
+ # Set up OTel providers BEFORE creating subscribers so that
266
+ # module-level get_meter()/get_tracer() calls bind to the real provider
267
+ if obs_config.enabled and obs_config.metrics_enabled:
268
+ # First Party
269
+ from lmcache.v1.mp_observability.otel_init import init_otel_metrics
270
+
271
+ init_otel_metrics(
272
+ otlp_endpoint=obs_config.otlp_endpoint,
273
+ prometheus_port=obs_config.prometheus_port,
274
+ )
275
+
276
+ if obs_config.enabled and obs_config.tracing_enabled:
277
+ # First Party
278
+ from lmcache.v1.mp_observability.otel_init import init_otel_tracing
279
+
280
+ init_otel_tracing(otlp_endpoint=obs_config.otlp_endpoint)
281
+
282
+ bus = init_event_bus(
283
+ EventBusConfig(
284
+ enabled=obs_config.enabled,
285
+ max_queue_size=obs_config.max_queue_size,
286
+ )
287
+ )
288
+
289
+ if obs_config.metrics_enabled:
290
+ # First Party
291
+ from lmcache.v1.mp_observability.subscribers.metrics import (
292
+ L0LifecycleSubscriber,
293
+ L1LifecycleSubscriber,
294
+ L1MetricsSubscriber,
295
+ L2MetricsSubscriber,
296
+ SMMetricsSubscriber,
297
+ )
298
+
299
+ sample_rate = obs_config.metrics_sample_rate
300
+ bus.register_subscriber(L0LifecycleSubscriber(sample_rate=sample_rate))
301
+ bus.register_subscriber(L1MetricsSubscriber())
302
+ bus.register_subscriber(L1LifecycleSubscriber(sample_rate=sample_rate))
303
+ bus.register_subscriber(L2MetricsSubscriber())
304
+ bus.register_subscriber(SMMetricsSubscriber())
305
+
306
+ if obs_config.logging_enabled:
307
+ # First Party
308
+ from lmcache.v1.mp_observability.subscribers.logging import (
309
+ L1LoggingSubscriber,
310
+ L2LoggingSubscriber,
311
+ MPServerLoggingSubscriber,
312
+ SMLoggingSubscriber,
313
+ )
314
+
315
+ bus.register_subscriber(MPServerLoggingSubscriber())
316
+ bus.register_subscriber(L1LoggingSubscriber())
317
+ bus.register_subscriber(L2LoggingSubscriber())
318
+ bus.register_subscriber(SMLoggingSubscriber())
319
+
320
+ if obs_config.tracing_enabled:
321
+ # First Party
322
+ from lmcache.v1.mp_observability.subscribers.tracing import (
323
+ MPServerTracingSubscriber,
324
+ get_span_registry,
325
+ )
326
+
327
+ bus.register_subscriber(MPServerTracingSubscriber(get_span_registry()))
328
+
329
+ # Lookup hash file logging (independent of the logging_enabled flag —
330
+ # it has its own enable gate via output_dir).
331
+ if obs_config.lookup_hash_log.enabled:
332
+ # First Party
333
+ from lmcache.v1.mp_observability.subscribers.logging.lookup_hash import (
334
+ LookupHashLoggingSubscriber,
335
+ )
336
+
337
+ bus.register_subscriber(LookupHashLoggingSubscriber(obs_config.lookup_hash_log))
338
+
339
+ bus.start()
340
+ return bus