lmcache-cli 0.4.5.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (399) hide show
  1. lmcache/__init__.py +84 -0
  2. lmcache/_version.py +24 -0
  3. lmcache/cli/__init__.py +1 -0
  4. lmcache/cli/commands/__init__.py +34 -0
  5. lmcache/cli/commands/base.py +157 -0
  6. lmcache/cli/commands/bench/__init__.py +557 -0
  7. lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
  8. lmcache/cli/commands/bench/engine_bench/config.py +245 -0
  9. lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
  10. lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
  11. lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
  12. lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
  13. lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
  14. lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
  15. lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
  16. lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
  17. lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
  18. lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
  19. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
  20. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
  21. lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
  22. lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
  23. lmcache/cli/commands/describe.py +310 -0
  24. lmcache/cli/commands/kvcache.py +133 -0
  25. lmcache/cli/commands/mock.py +75 -0
  26. lmcache/cli/commands/ping.py +113 -0
  27. lmcache/cli/commands/query/__init__.py +155 -0
  28. lmcache/cli/commands/query/prompt.py +134 -0
  29. lmcache/cli/commands/query/request.py +357 -0
  30. lmcache/cli/commands/server.py +99 -0
  31. lmcache/cli/commands/tool/__init__.py +63 -0
  32. lmcache/cli/commands/tool/cache_simulator.py +113 -0
  33. lmcache/cli/commands/trace/__init__.py +505 -0
  34. lmcache/cli/commands/trace/dispatch.py +249 -0
  35. lmcache/cli/commands/trace/driver.py +372 -0
  36. lmcache/cli/commands/trace/stats.py +289 -0
  37. lmcache/cli/documents/lmcache.txt +11 -0
  38. lmcache/cli/main.py +42 -0
  39. lmcache/cli/metrics/__init__.py +29 -0
  40. lmcache/cli/metrics/formatter.py +171 -0
  41. lmcache/cli/metrics/handler.py +94 -0
  42. lmcache/cli/metrics/metrics.py +161 -0
  43. lmcache/cli/metrics/section.py +77 -0
  44. lmcache/connections.py +173 -0
  45. lmcache/integration/__init__.py +2 -0
  46. lmcache/integration/base_service_factory.py +165 -0
  47. lmcache/integration/request_telemetry/__init__.py +1 -0
  48. lmcache/integration/request_telemetry/base.py +51 -0
  49. lmcache/integration/request_telemetry/factory.py +113 -0
  50. lmcache/integration/request_telemetry/fastapi.py +109 -0
  51. lmcache/integration/request_telemetry/noop.py +35 -0
  52. lmcache/integration/sglang/__init__.py +2 -0
  53. lmcache/integration/sglang/sglang_adapter.py +326 -0
  54. lmcache/integration/sglang/utils.py +39 -0
  55. lmcache/integration/vllm/__init__.py +1 -0
  56. lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
  57. lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
  58. lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
  59. lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
  60. lmcache/integration/vllm/utils.py +433 -0
  61. lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
  62. lmcache/integration/vllm/vllm_service_factory.py +339 -0
  63. lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
  64. lmcache/logging.py +107 -0
  65. lmcache/native_storage_ops.pyi +230 -0
  66. lmcache/non_cuda_equivalents.py +1424 -0
  67. lmcache/observability.py +1958 -0
  68. lmcache/storage_backend/serde/__init__.py +1 -0
  69. lmcache/storage_backend/serde/cachegen_basics.py +210 -0
  70. lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
  71. lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
  72. lmcache/storage_backend/serde/serde.py +75 -0
  73. lmcache/tools/__init__.py +1 -0
  74. lmcache/tools/cache_simulator/README.md +392 -0
  75. lmcache/tools/cache_simulator/__init__.py +1 -0
  76. lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
  77. lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
  78. lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
  79. lmcache/tools/cache_simulator/lru_cache.py +124 -0
  80. lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
  81. lmcache/tools/cache_simulator/simulator.py +795 -0
  82. lmcache/tools/controller_benchmark/README.md +161 -0
  83. lmcache/tools/controller_benchmark/__init__.py +1 -0
  84. lmcache/tools/controller_benchmark/__main__.py +331 -0
  85. lmcache/tools/controller_benchmark/benchmark.py +660 -0
  86. lmcache/tools/controller_benchmark/config.py +44 -0
  87. lmcache/tools/controller_benchmark/constants.py +10 -0
  88. lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
  89. lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
  90. lmcache/tools/controller_benchmark/handlers/base.py +47 -0
  91. lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
  92. lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
  93. lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
  94. lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
  95. lmcache/tools/controller_benchmark/handlers/register.py +56 -0
  96. lmcache/tools/mp_status_viewer/__init__.py +1 -0
  97. lmcache/tools/mp_status_viewer/__main__.py +95 -0
  98. lmcache/usage_context.py +417 -0
  99. lmcache/utils.py +665 -0
  100. lmcache/v1/__init__.py +2 -0
  101. lmcache/v1/api_server/__init__.py +2 -0
  102. lmcache/v1/api_server/__main__.py +537 -0
  103. lmcache/v1/basic_check.py +112 -0
  104. lmcache/v1/cache_controller/__init__.py +9 -0
  105. lmcache/v1/cache_controller/commands/__init__.py +15 -0
  106. lmcache/v1/cache_controller/commands/base.py +35 -0
  107. lmcache/v1/cache_controller/commands/full_sync.py +49 -0
  108. lmcache/v1/cache_controller/config.py +176 -0
  109. lmcache/v1/cache_controller/controller_manager.py +535 -0
  110. lmcache/v1/cache_controller/controllers/__init__.py +11 -0
  111. lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
  112. lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
  113. lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
  114. lmcache/v1/cache_controller/executor.py +463 -0
  115. lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
  116. lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
  117. lmcache/v1/cache_controller/frontend/static/index.html +234 -0
  118. lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
  119. lmcache/v1/cache_controller/full_sync_sender.py +475 -0
  120. lmcache/v1/cache_controller/locks.py +149 -0
  121. lmcache/v1/cache_controller/message.py +828 -0
  122. lmcache/v1/cache_controller/observability.py +208 -0
  123. lmcache/v1/cache_controller/utils.py +679 -0
  124. lmcache/v1/cache_controller/worker.py +665 -0
  125. lmcache/v1/cache_engine.py +2058 -0
  126. lmcache/v1/cache_interface.py +19 -0
  127. lmcache/v1/check/__init__.py +74 -0
  128. lmcache/v1/check/check_mode_gen.py +86 -0
  129. lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
  130. lmcache/v1/check/check_mode_test_remote.py +155 -0
  131. lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
  132. lmcache/v1/check/utils.py +571 -0
  133. lmcache/v1/compute/__init__.py +2 -0
  134. lmcache/v1/compute/attention/__init__.py +0 -0
  135. lmcache/v1/compute/attention/abstract.py +39 -0
  136. lmcache/v1/compute/attention/flash_attn.py +129 -0
  137. lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
  138. lmcache/v1/compute/attention/metadata.py +85 -0
  139. lmcache/v1/compute/attention/utils.py +14 -0
  140. lmcache/v1/compute/blend/__init__.py +7 -0
  141. lmcache/v1/compute/blend/blender.py +168 -0
  142. lmcache/v1/compute/blend/metadata.py +34 -0
  143. lmcache/v1/compute/blend/utils.py +63 -0
  144. lmcache/v1/compute/models/__init__.py +0 -0
  145. lmcache/v1/compute/models/base.py +141 -0
  146. lmcache/v1/compute/models/llama.py +9 -0
  147. lmcache/v1/compute/models/qwen3.py +24 -0
  148. lmcache/v1/compute/models/utils.py +68 -0
  149. lmcache/v1/compute/positional_encoding.py +199 -0
  150. lmcache/v1/config.py +848 -0
  151. lmcache/v1/config_base.py +848 -0
  152. lmcache/v1/distributed/api.py +248 -0
  153. lmcache/v1/distributed/config.py +321 -0
  154. lmcache/v1/distributed/error.py +64 -0
  155. lmcache/v1/distributed/eviction.py +192 -0
  156. lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
  157. lmcache/v1/distributed/eviction_policy/factory.py +27 -0
  158. lmcache/v1/distributed/eviction_policy/lru.py +244 -0
  159. lmcache/v1/distributed/eviction_policy/noop.py +50 -0
  160. lmcache/v1/distributed/internal_api.py +170 -0
  161. lmcache/v1/distributed/l1_manager.py +835 -0
  162. lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
  163. lmcache/v1/distributed/l2_adapters/base.py +360 -0
  164. lmcache/v1/distributed/l2_adapters/config.py +385 -0
  165. lmcache/v1/distributed/l2_adapters/factory.py +205 -0
  166. lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
  167. lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
  168. lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
  169. lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
  170. lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
  171. lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
  172. lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
  173. lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
  174. lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
  175. lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
  176. lmcache/v1/distributed/memory_manager.py +179 -0
  177. lmcache/v1/distributed/storage_controller.py +39 -0
  178. lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
  179. lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
  180. lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
  181. lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
  182. lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
  183. lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
  184. lmcache/v1/distributed/storage_manager.py +532 -0
  185. lmcache/v1/event_manager.py +145 -0
  186. lmcache/v1/exceptions/__init__.py +16 -0
  187. lmcache/v1/gpu_connector/__init__.py +126 -0
  188. lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
  189. lmcache/v1/gpu_connector/gpu_ops.py +85 -0
  190. lmcache/v1/gpu_connector/hpu_connector.py +326 -0
  191. lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
  192. lmcache/v1/gpu_connector/utils.py +890 -0
  193. lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
  194. lmcache/v1/health_monitor/__init__.py +1 -0
  195. lmcache/v1/health_monitor/base.py +587 -0
  196. lmcache/v1/health_monitor/checks/__init__.py +1 -0
  197. lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
  198. lmcache/v1/health_monitor/constants.py +36 -0
  199. lmcache/v1/internal_api_server/__init__.py +0 -0
  200. lmcache/v1/internal_api_server/api_registry.py +59 -0
  201. lmcache/v1/internal_api_server/api_server.py +120 -0
  202. lmcache/v1/internal_api_server/common/__init__.py +1 -0
  203. lmcache/v1/internal_api_server/common/env_api.py +22 -0
  204. lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
  205. lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
  206. lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
  207. lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
  208. lmcache/v1/internal_api_server/common/thread_api.py +63 -0
  209. lmcache/v1/internal_api_server/controller/__init__.py +1 -0
  210. lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
  211. lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
  212. lmcache/v1/internal_api_server/utils.py +43 -0
  213. lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
  214. lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
  215. lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
  216. lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
  217. lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
  218. lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
  219. lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
  220. lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
  221. lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
  222. lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
  223. lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
  224. lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
  225. lmcache/v1/kv_layer_groups.py +267 -0
  226. lmcache/v1/lazy_memory_allocator.py +284 -0
  227. lmcache/v1/lookup_client/__init__.py +25 -0
  228. lmcache/v1/lookup_client/abstract_client.py +77 -0
  229. lmcache/v1/lookup_client/async_lookup_message.py +50 -0
  230. lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
  231. lmcache/v1/lookup_client/factory.py +251 -0
  232. lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
  233. lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
  234. lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
  235. lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
  236. lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
  237. lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
  238. lmcache/v1/lookup_client/record_strategies/base.py +327 -0
  239. lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
  240. lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
  241. lmcache/v1/manager.py +539 -0
  242. lmcache/v1/memory_management.py +2619 -0
  243. lmcache/v1/metadata.py +114 -0
  244. lmcache/v1/mp_observability/AGENTS.override.md +21 -0
  245. lmcache/v1/mp_observability/README.md +204 -0
  246. lmcache/v1/mp_observability/config.py +340 -0
  247. lmcache/v1/mp_observability/event.py +100 -0
  248. lmcache/v1/mp_observability/event_bus.py +313 -0
  249. lmcache/v1/mp_observability/otel_init.py +145 -0
  250. lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
  251. lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
  252. lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
  253. lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
  254. lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
  255. lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
  256. lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
  257. lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
  258. lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
  259. lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
  260. lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
  261. lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
  262. lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
  263. lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
  264. lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
  265. lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
  266. lmcache/v1/mp_observability/trace/__init__.py +50 -0
  267. lmcache/v1/mp_observability/trace/codecs.py +255 -0
  268. lmcache/v1/mp_observability/trace/decorator.py +147 -0
  269. lmcache/v1/mp_observability/trace/format.py +132 -0
  270. lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
  271. lmcache/v1/mp_observability/trace/reader.py +167 -0
  272. lmcache/v1/mp_observability/trace/recorder.py +300 -0
  273. lmcache/v1/multiprocess/__init__.py +0 -0
  274. lmcache/v1/multiprocess/affinity_pool.py +102 -0
  275. lmcache/v1/multiprocess/blend_server_v2.py +891 -0
  276. lmcache/v1/multiprocess/config.py +253 -0
  277. lmcache/v1/multiprocess/custom_types.py +281 -0
  278. lmcache/v1/multiprocess/futures.py +194 -0
  279. lmcache/v1/multiprocess/gpu_context.py +511 -0
  280. lmcache/v1/multiprocess/http_server.py +235 -0
  281. lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
  282. lmcache/v1/multiprocess/mq.py +732 -0
  283. lmcache/v1/multiprocess/protocol.py +86 -0
  284. lmcache/v1/multiprocess/protocols/README.md +213 -0
  285. lmcache/v1/multiprocess/protocols/__init__.py +127 -0
  286. lmcache/v1/multiprocess/protocols/base.py +89 -0
  287. lmcache/v1/multiprocess/protocols/blend.py +109 -0
  288. lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
  289. lmcache/v1/multiprocess/protocols/controller.py +53 -0
  290. lmcache/v1/multiprocess/protocols/debug.py +34 -0
  291. lmcache/v1/multiprocess/protocols/engine.py +146 -0
  292. lmcache/v1/multiprocess/protocols/observability.py +39 -0
  293. lmcache/v1/multiprocess/server.py +1134 -0
  294. lmcache/v1/multiprocess/session.py +190 -0
  295. lmcache/v1/multiprocess/token_hasher.py +441 -0
  296. lmcache/v1/offload_server/__init__.py +17 -0
  297. lmcache/v1/offload_server/abstract_server.py +37 -0
  298. lmcache/v1/offload_server/message.py +30 -0
  299. lmcache/v1/offload_server/zmq_server.py +122 -0
  300. lmcache/v1/periodic_thread.py +579 -0
  301. lmcache/v1/pin_monitor.py +246 -0
  302. lmcache/v1/plugin/__init__.py +0 -0
  303. lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
  304. lmcache/v1/protocol.py +317 -0
  305. lmcache/v1/rpc/__init__.py +17 -0
  306. lmcache/v1/rpc/transport.py +105 -0
  307. lmcache/v1/rpc/zmq_transport.py +213 -0
  308. lmcache/v1/rpc_utils.py +165 -0
  309. lmcache/v1/server/__init__.py +2 -0
  310. lmcache/v1/server/__main__.py +170 -0
  311. lmcache/v1/server/storage_backend/__init__.py +21 -0
  312. lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
  313. lmcache/v1/server/storage_backend/local_backend.py +75 -0
  314. lmcache/v1/server/utils.py +21 -0
  315. lmcache/v1/standalone/__init__.py +1 -0
  316. lmcache/v1/standalone/__main__.py +583 -0
  317. lmcache/v1/standalone/manager.py +80 -0
  318. lmcache/v1/standalone/standalone_service_factory.py +86 -0
  319. lmcache/v1/storage_backend/__init__.py +313 -0
  320. lmcache/v1/storage_backend/abstract_backend.py +445 -0
  321. lmcache/v1/storage_backend/audit_backend.py +233 -0
  322. lmcache/v1/storage_backend/batched_message_sender.py +222 -0
  323. lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
  324. lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
  325. lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
  326. lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
  327. lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
  328. lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
  329. lmcache/v1/storage_backend/connector/__init__.py +443 -0
  330. lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
  331. lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
  332. lmcache/v1/storage_backend/connector/base_connector.py +379 -0
  333. lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
  334. lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
  335. lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
  336. lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
  337. lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
  338. lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
  339. lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
  340. lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
  341. lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
  342. lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
  343. lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
  344. lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
  345. lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
  346. lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
  347. lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
  348. lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
  349. lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
  350. lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
  351. lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
  352. lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
  353. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
  354. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
  355. lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
  356. lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
  357. lmcache/v1/storage_backend/gds_backend.py +1199 -0
  358. lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
  359. lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
  360. lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
  361. lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
  362. lmcache/v1/storage_backend/local_disk_backend.py +656 -0
  363. lmcache/v1/storage_backend/maru_backend.py +734 -0
  364. lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
  365. lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
  366. lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
  367. lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
  368. lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
  369. lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
  370. lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
  371. lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
  372. lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
  373. lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
  374. lmcache/v1/storage_backend/p2p_backend.py +788 -0
  375. lmcache/v1/storage_backend/path_sharder.py +117 -0
  376. lmcache/v1/storage_backend/pd_backend.py +646 -0
  377. lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
  378. lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
  379. lmcache/v1/storage_backend/remote_backend.py +624 -0
  380. lmcache/v1/storage_backend/resp_client.py +227 -0
  381. lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
  382. lmcache/v1/storage_backend/storage_manager.py +1352 -0
  383. lmcache/v1/system_detection.py +110 -0
  384. lmcache/v1/token_database.py +551 -0
  385. lmcache/v1/transfer_channel/__init__.py +83 -0
  386. lmcache/v1/transfer_channel/abstract.py +285 -0
  387. lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
  388. lmcache/v1/transfer_channel/nixl_channel.py +639 -0
  389. lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
  390. lmcache/v1/transfer_channel/transfer_utils.py +63 -0
  391. lmcache/v1/utils/__init__.py +1 -0
  392. lmcache/v1/utils/bloom_filter.py +109 -0
  393. lmcache/v1/utils/cache_utils.py +125 -0
  394. lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
  395. lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
  396. lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
  397. lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
  398. lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
  399. lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,835 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """
3
+ Managing objects and memory for L1 cache
4
+ """
5
+
6
+ # Standard
7
+ from dataclasses import dataclass
8
+ from typing import Literal
9
+ import threading
10
+
11
+ # First Party
12
+ from lmcache.logging import init_logger
13
+ from lmcache.native_storage_ops import TTLLock
14
+ from lmcache.v1.distributed.api import MemoryLayoutDesc, ObjectKey
15
+ from lmcache.v1.distributed.config import L1ManagerConfig
16
+ from lmcache.v1.distributed.error import L1Error
17
+ from lmcache.v1.distributed.internal_api import L1ManagerListener
18
+ from lmcache.v1.distributed.memory_manager import L1MemoryManager
19
+ from lmcache.v1.memory_management import MemoryObj
20
+ from lmcache.v1.mp_observability.event import Event, EventType
21
+ from lmcache.v1.mp_observability.event_bus import get_event_bus
22
+
23
+ logger = init_logger(__name__)
24
+
25
+
26
+ # Internal classes and helper functions
27
+ @dataclass
28
+ class L1ObjectState:
29
+ """
30
+ The internal state of an object in L1 cache
31
+ """
32
+
33
+ memory_obj: MemoryObj
34
+ """ The memory object stored in L1 cache. """
35
+
36
+ write_lock: TTLLock
37
+ """ Whether the object is write-locked. """
38
+
39
+ read_lock: TTLLock
40
+ """ The read lock with TTL for the object. """
41
+
42
+ is_temporary: bool
43
+ """ Whether the object is temporary (need to be deleted after read). """
44
+
45
+ def available_for_read(self) -> bool:
46
+ """Check if the object is available for read.
47
+
48
+ Returns:
49
+ True if the object is not write-locked, False otherwise.
50
+ """
51
+ return not self.write_lock.is_locked()
52
+
53
+ def available_for_write(self) -> bool:
54
+ """Check if the object is available for write.
55
+
56
+ Returns:
57
+ True if the object is not write-locked and has no read locks
58
+ and is not a temporary object, False otherwise.
59
+ """
60
+
61
+ return (
62
+ not self.write_lock.is_locked()
63
+ and not self.read_lock.is_locked()
64
+ and not self.is_temporary
65
+ )
66
+
67
+
68
+ def l1_mgr_synchronized(func):
69
+ """
70
+ Decorator to mark L1Manager methods as thread-safe
71
+ """
72
+
73
+ def wrapper(self: "L1Manager", *args, **kwargs):
74
+ with self._lock:
75
+ return func(self, *args, **kwargs)
76
+
77
+ return wrapper
78
+
79
+
80
+ L1OperationResult = tuple[L1Error, MemoryObj | None]
81
+
82
+ # Upper bound for the count parameter in reserve_read / finish_read
83
+ # to prevent a single call from holding the global lock for too long.
84
+ MAX_READ_LOCK_COUNT = 128
85
+
86
+
87
+ def _validate_extra_count(extra_count: int) -> int:
88
+ """Validate and clamp extra_count.
89
+
90
+ Args:
91
+ extra_count: Extra lock count on top of the
92
+ default 1 lock.
93
+
94
+ Returns:
95
+ Clamped value in [0, MAX_READ_LOCK_COUNT - 1].
96
+ """
97
+ if extra_count < 0:
98
+ logger.warning(
99
+ "L1Manager: extra_count=%d is invalid, clamping to 0",
100
+ extra_count,
101
+ )
102
+ return 0
103
+ upper = MAX_READ_LOCK_COUNT - 1
104
+ if extra_count > upper:
105
+ logger.warning(
106
+ "L1Manager: extra_count=%d exceeds limit=%d, clamping",
107
+ extra_count,
108
+ upper,
109
+ )
110
+ return upper
111
+ return extra_count
112
+
113
+
114
+ # Main classes
115
+
116
+
117
+ class L1Manager:
118
+ """
119
+ Object lifecycle state machine for L1 cache
120
+
121
+ +--------+
122
+ | None | <---------------------------------------+
123
+ +--------+ |
124
+ | ^ |
125
+ | | (write lock expired) | delete()
126
+ | | |
127
+ reserve | +----------------------+ |
128
+ write() | | |
129
+ v | |
130
+ +--------------+ +-----------+ |
131
+ | write_locked | | |---------------+
132
+ | |---------->| ready |
133
+ | | finish_ | |---------------+
134
+ +--------------+ write() +-----------+ |
135
+ ^ | |
136
+ | | reserve_read() | finish_read()
137
+ +--------------------------+ | (if count becomes 0)
138
+ reserve_write() | |
139
+ v |
140
+ +-----------------+ |
141
+ | read_locked |-----------+
142
+ | (count = 1) |
143
+ +-----------------+
144
+ | ^
145
+ reserve_read() | | finish_read()
146
+ v |
147
+ +-----------------+
148
+ | read_locked |
149
+ | (count = 2) |
150
+ +-----------------+
151
+ | ^
152
+ reserve_read() | | finish_read()
153
+ v |
154
+ (...) (...)
155
+ (Higher Counts)
156
+
157
+ For every operation on list of keys, the operation is atomic
158
+ """
159
+
160
+ def __init__(self, config: L1ManagerConfig):
161
+ self._lock = threading.Lock()
162
+
163
+ self._objects: dict[ObjectKey, L1ObjectState] = {}
164
+
165
+ self._memory_manager = L1MemoryManager(config.memory_config)
166
+
167
+ self._write_ttl_seconds = config.write_ttl_seconds
168
+ self._read_ttl_seconds = config.read_ttl_seconds
169
+
170
+ self._registered_listeners: list[L1ManagerListener] = []
171
+
172
+ self._event_bus = get_event_bus()
173
+
174
+ def register_listener(self, listener: L1ManagerListener) -> None:
175
+ """Register a listener for L1Manager events.
176
+
177
+ Args:
178
+ listener: The listener to register.
179
+ """
180
+ with self._lock:
181
+ self._registered_listeners.append(listener)
182
+
183
+ @l1_mgr_synchronized
184
+ def reserve_read(
185
+ self,
186
+ keys: list[ObjectKey],
187
+ extra_count: int = 0,
188
+ ) -> dict[ObjectKey, L1OperationResult]:
189
+ """Reserve read access for the given keys.
190
+
191
+ Args:
192
+ keys: The list of object keys to reserve
193
+ read access for.
194
+ extra_count: Extra read locks on top of the
195
+ default 1 lock. Total locks acquired per
196
+ key = 1 + extra_count. Useful when multiple
197
+ workers each consume one read lock for the
198
+ same key (e.g. MLA models with TP > 1).
199
+
200
+ Returns:
201
+ A dictionary mapping each object key to a tuple
202
+ of (L1Error, Optional[MemoryObj]).
203
+
204
+ Errors:
205
+ KEY_NOT_EXIST: The key does not exist.
206
+ KEY_NOT_READABLE: The key exists but is not
207
+ readable.
208
+ """
209
+ extra_count = _validate_extra_count(extra_count)
210
+ total = 1 + extra_count
211
+ ret: dict[ObjectKey, L1OperationResult] = {}
212
+ successful_keys: list[ObjectKey] = []
213
+ for key in keys:
214
+ entry = self._objects.get(key, None)
215
+ if entry is None:
216
+ ret[key] = (L1Error.KEY_NOT_EXIST, None)
217
+ continue
218
+
219
+ if not entry.available_for_read():
220
+ ret[key] = (L1Error.KEY_NOT_READABLE, None)
221
+ continue
222
+
223
+ # TODO(perf): support a count argument in
224
+ # TTLLock.lock() to avoid Python for-loop
225
+ # overhead (TTLLock is C++ std::atomic).
226
+ for _ in range(total):
227
+ entry.read_lock.lock()
228
+ ret[key] = (L1Error.SUCCESS, entry.memory_obj)
229
+ successful_keys.append(key)
230
+
231
+ for listener in self._registered_listeners:
232
+ listener.on_l1_keys_reserved_read(successful_keys)
233
+ self._event_bus.publish(
234
+ Event(
235
+ event_type=EventType.L1_READ_RESERVED,
236
+ metadata={"keys": successful_keys},
237
+ )
238
+ )
239
+ return ret
240
+
241
+ @l1_mgr_synchronized
242
+ def unsafe_read(
243
+ self,
244
+ keys: list[ObjectKey],
245
+ ) -> dict[ObjectKey, L1OperationResult]:
246
+ """Unsafe read the read-locked objects without adding new read locks.
247
+
248
+ This method does not acquire read locks. Therefore, the caller need
249
+ to make sure the `unsafe_read` is called between `reserve_read` and
250
+ `finish_read` calls.
251
+
252
+ Args:
253
+ keys: The list of object keys to read.
254
+
255
+ Returns:
256
+ A dictionary mapping each object key to a tuple of
257
+ (L1Error, Optional[MemoryObj]).
258
+
259
+ Errors:
260
+ KEY_NOT_EXIST: The key does not exist.
261
+ KEY_NOT_READABLE: The key is not readable (in this case, not read-locked).
262
+ """
263
+ ret: dict[ObjectKey, L1OperationResult] = {}
264
+
265
+ for key in keys:
266
+ entry = self._objects.get(key, None)
267
+ if entry is None:
268
+ ret[key] = (L1Error.KEY_NOT_EXIST, None)
269
+ continue
270
+
271
+ if not entry.read_lock.is_locked():
272
+ ret[key] = (L1Error.KEY_NOT_READABLE, None)
273
+ continue
274
+
275
+ ret[key] = (L1Error.SUCCESS, entry.memory_obj)
276
+
277
+ return ret
278
+
279
+ @l1_mgr_synchronized
280
+ def finish_read(
281
+ self,
282
+ keys: list[ObjectKey],
283
+ extra_count: int = 0,
284
+ ) -> dict[ObjectKey, L1Error]:
285
+ """Finish read access for the given keys.
286
+
287
+ Will delete the object if it is temporary and read
288
+ count reaches zero.
289
+
290
+ Args:
291
+ keys: The list of object keys to finish read
292
+ access for.
293
+ extra_count: Extra read locks to release on top
294
+ of the default 1. Must match the
295
+ ``extra_count`` used in the corresponding
296
+ ``reserve_read`` call.
297
+
298
+ Returns:
299
+ A dictionary mapping each object key to an
300
+ L1Error.
301
+
302
+ Errors:
303
+ KEY_NOT_EXIST: The key does not exist.
304
+ KEY_IN_WRONG_STATE: The key is write-locked or
305
+ non-read-locked, which means the reader may
306
+ read inconsistent data.
307
+ """
308
+ extra_count = _validate_extra_count(extra_count)
309
+ total = 1 + extra_count
310
+ need_to_free: list[MemoryObj] = []
311
+ need_to_free_keys: list[ObjectKey] = []
312
+ ret: dict[ObjectKey, L1Error] = {}
313
+ successful_keys: list[ObjectKey] = []
314
+
315
+ for key in keys:
316
+ entry = self._objects.get(key, None)
317
+ if entry is None:
318
+ logger.warning(
319
+ "L1Manager: finish read on non-existing key %s, "
320
+ "potential inconsistent data might be read",
321
+ key,
322
+ )
323
+ ret[key] = L1Error.KEY_NOT_EXIST
324
+ continue
325
+
326
+ if entry.write_lock.is_locked():
327
+ logger.warning(
328
+ "L1Manager: finish read on write-locked key %s, "
329
+ "potential inconsistent data might be read",
330
+ key,
331
+ )
332
+ ret[key] = L1Error.KEY_IN_WRONG_STATE
333
+ continue
334
+
335
+ if not entry.read_lock.is_locked():
336
+ logger.warning(
337
+ "L1Manager: finish read on non-read-locked key %s, "
338
+ "potential inconsistent data might be read",
339
+ key,
340
+ )
341
+ ret[key] = L1Error.KEY_IN_WRONG_STATE
342
+ continue
343
+
344
+ # TODO(perf): support a count argument in
345
+ # TTLLock.unlock() to avoid Python for-loop
346
+ # overhead (TTLLock is C++ std::atomic).
347
+ for _ in range(total):
348
+ entry.read_lock.unlock()
349
+ if entry.is_temporary and not entry.read_lock.is_locked():
350
+ # NOTE: temporary objects shouldn't have write-locks
351
+ need_to_free.append(entry.memory_obj)
352
+ need_to_free_keys.append(key)
353
+ del self._objects[key]
354
+
355
+ ret[key] = L1Error.SUCCESS
356
+ successful_keys.append(key)
357
+
358
+ self._memory_manager.free(need_to_free)
359
+
360
+ for listener in self._registered_listeners:
361
+ listener.on_l1_keys_read_finished(successful_keys)
362
+ listener.on_l1_keys_deleted_by_manager(need_to_free_keys)
363
+ self._event_bus.publish(
364
+ Event(
365
+ event_type=EventType.L1_READ_FINISHED,
366
+ metadata={"keys": successful_keys},
367
+ )
368
+ )
369
+ self._event_bus.publish(
370
+ Event(
371
+ event_type=EventType.L1_KEYS_EVICTED,
372
+ metadata={"keys": need_to_free_keys},
373
+ )
374
+ )
375
+
376
+ return ret
377
+
378
+ @l1_mgr_synchronized
379
+ def reserve_write(
380
+ self,
381
+ keys: list[ObjectKey],
382
+ is_temporary: list[bool],
383
+ layout_desc: MemoryLayoutDesc,
384
+ mode: Literal["new", "update", "all"] = "all",
385
+ ) -> dict[ObjectKey, L1OperationResult]:
386
+ """Reserve write access for the given keys.
387
+
388
+ Args:
389
+ keys: The list of object keys to reserve write access for.
390
+ is_temporary: The list of booleans indicating whether each key is
391
+ temporary.
392
+ shape_spec: The memory layout description for the objects to be
393
+ allocated.
394
+ mode (Literal["new", "update", "all"]): Reservation mode.
395
+ - "new": Reserve only new objects that do not exist.
396
+ - "update": Reserve only existing objects for update.
397
+ - "all": Reserve all writable objects regardless of existence.
398
+
399
+ Returns:
400
+ A dictionary mapping each object key to a tuple of
401
+ (L1Error, Optional[MemoryObj]).
402
+
403
+ Errors:
404
+ KEY_NOT_WRITABLE: The key exists but is not writable.
405
+ OUT_OF_MEMORY: Not enough memory to allocate for the object.
406
+ """
407
+ need_to_allocate: list[tuple[ObjectKey, bool]] = []
408
+ ret: dict[ObjectKey, L1OperationResult] = {}
409
+ successful_keys: list[ObjectKey] = []
410
+
411
+ for key, is_temp in zip(keys, is_temporary, strict=False):
412
+ entry = self._objects.get(key, None)
413
+ if entry is None:
414
+ need_to_allocate.append((key, is_temp))
415
+ continue
416
+
417
+ if mode == "new":
418
+ ret[key] = (L1Error.KEY_NOT_WRITABLE, None)
419
+ continue
420
+
421
+ if not entry.available_for_write():
422
+ ret[key] = (L1Error.KEY_NOT_WRITABLE, None)
423
+ continue
424
+
425
+ entry.write_lock.lock()
426
+ ret[key] = (L1Error.SUCCESS, entry.memory_obj)
427
+ successful_keys.append(key)
428
+
429
+ # Early return if no allocation is needed
430
+ if len(need_to_allocate) == 0:
431
+ return ret
432
+
433
+ # Don't allow allocation in "update" mode
434
+ if mode == "update":
435
+ for key, _ in need_to_allocate:
436
+ ret[key] = (L1Error.KEY_NOT_WRITABLE, None)
437
+ return ret
438
+
439
+ err, allocated_objs = self._memory_manager.allocate(
440
+ layout_desc, len(need_to_allocate)
441
+ )
442
+
443
+ if err != L1Error.SUCCESS:
444
+ for key, _ in need_to_allocate:
445
+ ret[key] = (L1Error.OUT_OF_MEMORY, None)
446
+
447
+ # Free the memory if partial allocation succeeded
448
+ if allocated_objs:
449
+ self._memory_manager.free(allocated_objs)
450
+
451
+ else:
452
+ for (key, is_temp), mem_obj in zip(
453
+ need_to_allocate, allocated_objs, strict=False
454
+ ):
455
+ self._objects[key] = L1ObjectState(
456
+ memory_obj=mem_obj,
457
+ write_lock=TTLLock(self._write_ttl_seconds),
458
+ read_lock=TTLLock(self._read_ttl_seconds),
459
+ is_temporary=is_temp,
460
+ )
461
+ self._objects[key].write_lock.lock()
462
+ ret[key] = (L1Error.SUCCESS, mem_obj)
463
+ successful_keys.append(key)
464
+
465
+ for listener in self._registered_listeners:
466
+ listener.on_l1_keys_reserved_write(successful_keys)
467
+ self._event_bus.publish(
468
+ Event(
469
+ event_type=EventType.L1_WRITE_RESERVED,
470
+ metadata={"keys": successful_keys},
471
+ )
472
+ )
473
+ return ret
474
+
475
+ @l1_mgr_synchronized
476
+ def finish_write(
477
+ self,
478
+ keys: list[ObjectKey],
479
+ ) -> dict[ObjectKey, L1Error]:
480
+ """Finish write access for the given keys.
481
+
482
+ Args:
483
+ keys: The list of object keys to finish write access for.
484
+
485
+ Returns:
486
+ A dictionary mapping each object key to an L1Error.
487
+
488
+ Errors:
489
+ KEY_NOT_EXIST: The key does not exist.
490
+ KEY_IN_WRONG_STATE: The key is not write-locked, or it's read-locked,
491
+ which means the writer may have caused inconsistent data.
492
+ """
493
+ ret: dict[ObjectKey, L1Error] = {}
494
+ successful_keys: list[ObjectKey] = []
495
+
496
+ for key in keys:
497
+ entry = self._objects.get(key, None)
498
+ if entry is None:
499
+ ret[key] = L1Error.KEY_NOT_EXIST
500
+ continue
501
+
502
+ if not entry.write_lock.is_locked():
503
+ logger.warning(
504
+ "L1Manager: finish write on non-write-locked key %s, "
505
+ "potential inconsistent data might be written",
506
+ key,
507
+ )
508
+ ret[key] = L1Error.KEY_IN_WRONG_STATE
509
+ continue
510
+
511
+ if entry.read_lock.is_locked():
512
+ logger.warning(
513
+ "L1Manager: finish write on read-locked key %s, "
514
+ "potential inconsistent data might be written",
515
+ key,
516
+ )
517
+ ret[key] = L1Error.KEY_IN_WRONG_STATE
518
+ continue
519
+
520
+ entry.write_lock.unlock()
521
+ ret[key] = L1Error.SUCCESS
522
+ successful_keys.append(key)
523
+
524
+ for listener in self._registered_listeners:
525
+ listener.on_l1_keys_write_finished(successful_keys)
526
+ self._event_bus.publish(
527
+ Event(
528
+ event_type=EventType.L1_WRITE_FINISHED,
529
+ metadata={"keys": successful_keys},
530
+ )
531
+ )
532
+ return ret
533
+
534
+ @l1_mgr_synchronized
535
+ def finish_write_and_reserve_read(
536
+ self,
537
+ keys: list[ObjectKey],
538
+ extra_count: int = 0,
539
+ ) -> dict[ObjectKey, L1OperationResult]:
540
+ """Atomically finish write and acquire read lock for the given keys.
541
+
542
+ This is used by the prefetch controller after successfully loading
543
+ data from L2 into write-reserved L1 buffers. It transitions the
544
+ object from write-locked to read-locked in a single atomic step,
545
+ preventing a race window where eviction could interfere.
546
+
547
+ Args:
548
+ keys: Keys to transition from write-locked to read-locked.
549
+ extra_count: Extra read locks on top of the default 1 lock.
550
+ Total locks acquired per key = 1 + extra_count. Useful
551
+ when multiple TP workers each consume one read lock for
552
+ the same key (e.g. MLA models with TP > 1).
553
+
554
+ Returns:
555
+ A dictionary mapping each object key to a tuple of
556
+ (L1Error, Optional[MemoryObj]).
557
+
558
+ Errors:
559
+ KEY_NOT_EXIST: The key does not exist.
560
+ KEY_IN_WRONG_STATE: The key is not write-locked, or it already
561
+ has read locks.
562
+ """
563
+ extra_count = _validate_extra_count(extra_count)
564
+ total = 1 + extra_count
565
+ ret: dict[ObjectKey, L1OperationResult] = {}
566
+ successful_keys: list[ObjectKey] = []
567
+
568
+ for key in keys:
569
+ entry = self._objects.get(key, None)
570
+ if entry is None:
571
+ ret[key] = (L1Error.KEY_NOT_EXIST, None)
572
+ continue
573
+
574
+ if not entry.write_lock.is_locked():
575
+ logger.warning(
576
+ "L1Manager: finish_write_and_reserve_read on "
577
+ "non-write-locked key %s",
578
+ key,
579
+ )
580
+ ret[key] = (L1Error.KEY_IN_WRONG_STATE, None)
581
+ continue
582
+
583
+ if entry.read_lock.is_locked():
584
+ logger.warning(
585
+ "L1Manager: finish_write_and_reserve_read on read-locked key %s",
586
+ key,
587
+ )
588
+ ret[key] = (L1Error.KEY_IN_WRONG_STATE, None)
589
+ continue
590
+
591
+ entry.write_lock.unlock()
592
+ for _ in range(total):
593
+ entry.read_lock.lock()
594
+ ret[key] = (L1Error.SUCCESS, entry.memory_obj)
595
+ successful_keys.append(key)
596
+
597
+ for listener in self._registered_listeners:
598
+ listener.on_l1_keys_finish_write_and_reserve_read(successful_keys)
599
+ self._event_bus.publish(
600
+ Event(
601
+ event_type=EventType.L1_WRITE_FINISHED_AND_READ_RESERVED,
602
+ metadata={"keys": successful_keys},
603
+ )
604
+ )
605
+ return ret
606
+
607
+ @l1_mgr_synchronized
608
+ def delete(self, keys: list[ObjectKey]) -> dict[ObjectKey, L1Error]:
609
+ """Delete the given keys from L1 cache.
610
+
611
+ Args:
612
+ keys: The list of object keys to delete.
613
+
614
+ Returns:
615
+ A dictionary mapping each object key to an L1Error.
616
+
617
+ Errors:
618
+ KEY_NOT_EXIST: The key does not exist.
619
+ KEY_IS_LOCKED: The key is locked (either write-locked or read-locked
620
+ and cannot be deleted).
621
+ """
622
+ need_to_free: list[MemoryObj] = []
623
+ ret: dict[ObjectKey, L1Error] = {}
624
+ successful_keys: list[ObjectKey] = []
625
+
626
+ for key in keys:
627
+ entry = self._objects.get(key, None)
628
+ if entry is None:
629
+ ret[key] = L1Error.KEY_NOT_EXIST
630
+ continue
631
+
632
+ if entry.read_lock.is_locked() or entry.write_lock.is_locked():
633
+ ret[key] = L1Error.KEY_IS_LOCKED
634
+ continue
635
+
636
+ need_to_free.append(entry.memory_obj)
637
+ del self._objects[key]
638
+ ret[key] = L1Error.SUCCESS
639
+ successful_keys.append(key)
640
+
641
+ self._memory_manager.free(need_to_free)
642
+
643
+ for listener in self._registered_listeners:
644
+ listener.on_l1_keys_deleted_by_manager(successful_keys)
645
+ self._event_bus.publish(
646
+ Event(
647
+ event_type=EventType.L1_KEYS_EVICTED,
648
+ metadata={"keys": successful_keys},
649
+ )
650
+ )
651
+ return ret
652
+
653
+ def touch_keys(self, keys: list[ObjectKey]):
654
+ """Touch the given keys, marking the keys as accessed(retrieved or stored).
655
+
656
+ Args:
657
+ keys: The list of object keys to touch.
658
+ """
659
+ for listener in self._registered_listeners:
660
+ listener.on_l1_keys_accessed(keys)
661
+
662
+ @l1_mgr_synchronized
663
+ def clear(self, force: bool = False) -> None:
664
+ """Clear objects from L1 cache.
665
+
666
+ Args:
667
+ force: If True, clear ALL objects including locked ones.
668
+ This may corrupt in-flight store/prefetch operations.
669
+ If False (default), only clear unlocked objects, keeping
670
+ write-locked and read-locked objects intact.
671
+ """
672
+ if force:
673
+ logger.warning(
674
+ "L1Manager: force-clearing all %d objects "
675
+ "(including locked ones). This may corrupt in-flight "
676
+ "store/prefetch operations — use with caution.",
677
+ len(self._objects),
678
+ )
679
+ all_keys = list(self._objects.keys())
680
+ all_memory_objs = [entry.memory_obj for entry in self._objects.values()]
681
+ self._memory_manager.free(all_memory_objs)
682
+ self._objects.clear()
683
+ for listener in self._registered_listeners:
684
+ listener.on_l1_keys_deleted_by_manager(all_keys)
685
+ self._event_bus.publish(
686
+ Event(
687
+ event_type=EventType.L1_KEYS_EVICTED,
688
+ metadata={"keys": all_keys},
689
+ )
690
+ )
691
+ logger.info(
692
+ "L1Manager: cleared %d objects, 0 remaining.",
693
+ len(all_keys),
694
+ )
695
+ return
696
+
697
+ keys_to_clear: list[ObjectKey] = []
698
+ objs_to_free: list[MemoryObj] = []
699
+ locked_count = 0
700
+
701
+ for key, entry in list(self._objects.items()):
702
+ if entry.write_lock.is_locked() or entry.read_lock.is_locked():
703
+ locked_count += 1
704
+ continue
705
+ keys_to_clear.append(key)
706
+ objs_to_free.append(entry.memory_obj)
707
+
708
+ for key in keys_to_clear:
709
+ del self._objects[key]
710
+
711
+ self._memory_manager.free(objs_to_free)
712
+
713
+ if keys_to_clear:
714
+ for listener in self._registered_listeners:
715
+ listener.on_l1_keys_deleted_by_manager(keys_to_clear)
716
+ self._event_bus.publish(
717
+ Event(
718
+ event_type=EventType.L1_KEYS_EVICTED,
719
+ metadata={"keys": keys_to_clear},
720
+ )
721
+ )
722
+
723
+ logger.info(
724
+ "L1Manager: cleared %d objects, %d locked objects remaining.",
725
+ len(keys_to_clear),
726
+ locked_count,
727
+ )
728
+
729
+ def is_key_evictable(self, key: ObjectKey) -> bool:
730
+ """Check if a key is eligible for eviction (not locked).
731
+
732
+ This method does NOT acquire the global L1Manager lock.
733
+ L1Manager.delete() will check again and safely reject a key
734
+ that became locked between the check and the actual deletion.
735
+
736
+ Args:
737
+ key: The object key to check.
738
+
739
+ Returns:
740
+ True if the key exists and is not locked (neither read-locked
741
+ nor write-locked), False otherwise.
742
+ """
743
+ entry = self._objects.get(key, None)
744
+ if entry is None:
745
+ return False
746
+ return not entry.read_lock.is_locked() and not entry.write_lock.is_locked()
747
+
748
+ def get_memory_usage(self) -> tuple[int, int]:
749
+ """Get the current memory usage of L1 cache.
750
+
751
+ Returns:
752
+ A tuple of (used_memory_bytes, total_memory_bytes).
753
+
754
+ Note:
755
+ In the future, we many want to make a "callback" based mechanism
756
+ via "L1ManagerListener" to notify the memory usage changes.
757
+ """
758
+ return self._memory_manager.get_memory_usage()
759
+
760
+ def get_l1_memory_desc(self):
761
+ """Return an L1MemoryDesc describing the underlying L1 memory buffer."""
762
+ return self._memory_manager.get_l1_memory_desc()
763
+
764
+ def close(self) -> None:
765
+ """Close the L1Manager and free all resources."""
766
+ with self._lock:
767
+ all_memory_objs = [entry.memory_obj for entry in self._objects.values()]
768
+ self._memory_manager.free(all_memory_objs)
769
+ self._objects.clear()
770
+
771
+ self._memory_manager.close()
772
+
773
+ # Status reporting
774
+ @l1_mgr_synchronized
775
+ def report_status(self) -> dict:
776
+ """Return a status dict describing L1 cache state."""
777
+ write_locked = 0
778
+ read_locked = 0
779
+ temporary = 0
780
+ for entry in self._objects.values():
781
+ if entry.write_lock.is_locked():
782
+ write_locked += 1
783
+ if entry.read_lock.is_locked():
784
+ read_locked += 1
785
+ if entry.is_temporary:
786
+ temporary += 1
787
+ used, total = self._memory_manager.get_memory_usage()
788
+ return {
789
+ "is_healthy": self._memory_manager.memcheck(),
790
+ "total_object_count": len(self._objects),
791
+ "write_locked_count": write_locked,
792
+ "read_locked_count": read_locked,
793
+ "temporary_count": temporary,
794
+ "memory_used_bytes": used,
795
+ "memory_total_bytes": total,
796
+ "memory_usage_ratio": used / total if total > 0 else 0.0,
797
+ "write_ttl_seconds": self._write_ttl_seconds,
798
+ "read_ttl_seconds": self._read_ttl_seconds,
799
+ }
800
+
801
+ # Debugging APIs
802
+ @l1_mgr_synchronized
803
+ def get_object_state(self, key: ObjectKey) -> L1ObjectState | None:
804
+ """Get the internal state of the object with the given key.
805
+
806
+ Args:
807
+ key: The object key.
808
+
809
+ Returns:
810
+ The L1ObjectState if the object exists, None otherwise.
811
+ """
812
+ return self._objects.get(key, None)
813
+
814
+ @l1_mgr_synchronized
815
+ def memcheck(self) -> bool:
816
+ """Perform memory check for L1 cache."""
817
+ mem_check_result = self._memory_manager.memcheck()
818
+
819
+ # Log the locked objects for debugging
820
+ num_write_locked = 0
821
+ num_read_locked = 0
822
+ for key, entry in self._objects.items():
823
+ if entry.write_lock.is_locked():
824
+ num_write_locked += 1
825
+ if entry.read_lock.is_locked():
826
+ num_read_locked += 1
827
+
828
+ logger.info(
829
+ "L1Manager memcheck: total objects = %d, write-locked = %d, "
830
+ "read-locked = %d",
831
+ len(self._objects),
832
+ num_write_locked,
833
+ num_read_locked,
834
+ )
835
+ return mem_check_result